"""Render block groups into chunk markdown strings with layout metadata.""" from __future__ import annotations from rag_cut.models import Block, BlockType, Chunk from rag_cut.parsers.pdf.tables import table_meta_summary def block_to_layout_dict(block: Block) -> dict: """Serialize a block for chunk metadata and frontend positional rendering.""" entry: dict = { "type": block.type.value, "order_index": block.meta.get("order_index"), "page": block.meta.get("page"), "pages": block.meta.get("pages"), "bbox": block.meta.get("bbox"), "bboxes": block.meta.get("bboxes"), "parent_heading": block.meta.get("parent_heading"), "nearest_heading": block.meta.get("nearest_heading"), "bound_heading": block.meta.get("bound_heading"), "preceding_text": block.meta.get("preceding_text"), "following_text": block.meta.get("following_text"), } if block.type == BlockType.HEADING: entry["text"] = block.text entry["level"] = block.level elif block.type == BlockType.IMAGE: entry["text"] = block.text entry["image_path"] = block.image_path entry["image_id"] = block.image_id entry["ocr_text"] = block.ocr_text elif block.type == BlockType.TABLE: entry["text"] = block.markdown or block.text entry["markdown"] = block.markdown or block.text entry["ocr_text"] = block.ocr_text entry["image_path"] = block.image_path entry["image_id"] = block.image_id entry["crop_path"] = block.meta.get("crop_path") entry.update(table_meta_summary(block.meta)) entry["embedding_text"] = block.meta.get("embedding_text") else: entry["text"] = block.text or block.markdown return entry def collect_chunk_layout_meta(blocks: list[Block]) -> dict: """Aggregate page/bbox/image/table metadata for a chunk group.""" pages = sorted({b.meta.get("page") for b in blocks if b.meta.get("page") is not None}) for b in blocks: for p in b.meta.get("pages") or []: if p is not None: pages.append(p) pages = sorted(set(pages)) bboxes = [ { "order_index": b.meta.get("order_index"), "page": b.meta.get("page"), "bbox": b.meta.get("bbox"), "type": b.type.value, } for b in blocks if b.meta.get("bbox") ] images = [ { "order_index": b.meta.get("order_index"), "image_id": b.image_id, "image_path": b.image_path, "page": b.meta.get("page"), "bbox": b.meta.get("bbox"), "ocr_text": b.ocr_text, "bound_heading": b.meta.get("bound_heading"), "preceding_text": b.meta.get("preceding_text"), "following_text": b.meta.get("following_text"), } for b in blocks if b.type == BlockType.IMAGE ] tables = [ { "order_index": b.meta.get("order_index"), "table_title": b.meta.get("table_title"), "image_path": b.image_path or b.meta.get("crop_path"), "crop_path": b.meta.get("crop_path"), "page": b.meta.get("page"), "pages": b.meta.get("pages"), "bbox": b.meta.get("bbox"), "bboxes": b.meta.get("bboxes"), "markdown": b.markdown, "ocr_text": b.ocr_text, "footnotes": b.meta.get("footnotes"), "keywords": b.meta.get("keywords"), "nearest_heading": b.meta.get("nearest_heading"), "chapter": b.meta.get("chapter"), "row_count": b.meta.get("row_count"), "col_count": b.meta.get("col_count"), "header_rows": b.meta.get("header_rows"), "table_source": b.meta.get("table_source"), "preceding_text": b.meta.get("preceding_text"), "following_text": b.meta.get("following_text"), "cross_page": b.meta.get("cross_page"), "embedding_text": b.meta.get("embedding_text"), } for b in blocks if b.type == BlockType.TABLE ] meta: dict = { "blocks": [block_to_layout_dict(b) for b in blocks], "bboxes": bboxes, "images": images, "tables": tables, } if pages: meta["pages"] = pages if len(pages) == 1: meta["page"] = pages[0] headings = [b.text for b in blocks if b.type == BlockType.HEADING] if headings: meta["heading"] = headings[0] meta["nearest_heading"] = headings[0] elif blocks and blocks[0].meta.get("nearest_heading"): meta["nearest_heading"] = blocks[0].meta.get("nearest_heading") table_blocks = [b for b in blocks if b.type == BlockType.TABLE] if table_blocks: primary = table_blocks[0] meta["table_title"] = primary.meta.get("table_title") or meta.get("nearest_heading") if not meta.get("chunk_strategy"): meta["chunk_strategy"] = "table_with_context" meta["retrieval"] = meta.get("retrieval", True) if primary.meta.get("embedding_text"): meta["embedding_text"] = primary.meta["embedding_text"] if primary.meta.get("keywords"): meta["keywords"] = primary.meta["keywords"] order_indices = [b.meta.get("order_index") for b in blocks if b.meta.get("order_index") is not None] if order_indices: meta["order_range"] = meta.get("order_range") or [min(order_indices), max(order_indices)] return meta def render_blocks(blocks: list[Block], index: int, meta: dict | None = None) -> Chunk: """Render blocks in reading order: heading → body → table/image → notes.""" parts: list[str] = [] for block in blocks: rendered = block.render().strip() if rendered: parts.append(rendered) content = "\n\n".join(parts) layout_meta = collect_chunk_layout_meta(blocks) chunk_meta = {**(meta or {}), **layout_meta} return Chunk( index=index, content=content, char_count=len(content), block_types=[b.type.value for b in blocks], meta=chunk_meta, )