Files
RAG-CUT/backend/rag_cut/renderer.py
T
2026-07-16 11:12:17 +08:00

165 lines
6.1 KiB
Python

"""Render block groups into chunk markdown strings with layout metadata."""
from __future__ import annotations
from rag_cut.models import Block, BlockType, Chunk
from rag_cut.parsers.pdf.tables import table_meta_summary
def block_to_layout_dict(block: Block) -> dict:
"""Serialize a block for chunk metadata and frontend positional rendering."""
entry: dict = {
"type": block.type.value,
"order_index": block.meta.get("order_index"),
"page": block.meta.get("page"),
"pages": block.meta.get("pages"),
"bbox": block.meta.get("bbox"),
"bboxes": block.meta.get("bboxes"),
"parent_heading": block.meta.get("parent_heading"),
"nearest_heading": block.meta.get("nearest_heading"),
"bound_heading": block.meta.get("bound_heading"),
"preceding_text": block.meta.get("preceding_text"),
"following_text": block.meta.get("following_text"),
}
if block.type == BlockType.HEADING:
entry["text"] = block.text
entry["level"] = block.level
elif block.type == BlockType.IMAGE:
entry["text"] = block.text
entry["image_path"] = block.image_path
entry["image_id"] = block.image_id
entry["ocr_text"] = block.ocr_text
elif block.type == BlockType.TABLE:
entry["text"] = block.markdown or block.text
entry["markdown"] = block.markdown or block.text
entry["ocr_text"] = block.ocr_text
entry["image_path"] = block.image_path
entry["image_id"] = block.image_id
entry["crop_path"] = block.meta.get("crop_path")
entry.update(table_meta_summary(block.meta))
entry["embedding_text"] = block.meta.get("embedding_text")
else:
entry["text"] = block.text or block.markdown
return entry
def collect_chunk_layout_meta(blocks: list[Block]) -> dict:
"""Aggregate page/bbox/image/table metadata for a chunk group."""
pages = sorted({b.meta.get("page") for b in blocks if b.meta.get("page") is not None})
for b in blocks:
for p in b.meta.get("pages") or []:
if p is not None:
pages.append(p)
pages = sorted(set(pages))
bboxes = [
{
"order_index": b.meta.get("order_index"),
"page": b.meta.get("page"),
"bbox": b.meta.get("bbox"),
"type": b.type.value,
}
for b in blocks
if b.meta.get("bbox")
]
images = [
{
"order_index": b.meta.get("order_index"),
"image_id": b.image_id,
"image_path": b.image_path,
"page": b.meta.get("page"),
"bbox": b.meta.get("bbox"),
"ocr_text": b.ocr_text,
"bound_heading": b.meta.get("bound_heading"),
"preceding_text": b.meta.get("preceding_text"),
"following_text": b.meta.get("following_text"),
}
for b in blocks
if b.type == BlockType.IMAGE
]
tables = [
{
"order_index": b.meta.get("order_index"),
"table_title": b.meta.get("table_title"),
"image_path": b.image_path or b.meta.get("crop_path"),
"crop_path": b.meta.get("crop_path"),
"page": b.meta.get("page"),
"pages": b.meta.get("pages"),
"bbox": b.meta.get("bbox"),
"bboxes": b.meta.get("bboxes"),
"markdown": b.markdown,
"ocr_text": b.ocr_text,
"footnotes": b.meta.get("footnotes"),
"keywords": b.meta.get("keywords"),
"nearest_heading": b.meta.get("nearest_heading"),
"chapter": b.meta.get("chapter"),
"row_count": b.meta.get("row_count"),
"col_count": b.meta.get("col_count"),
"header_rows": b.meta.get("header_rows"),
"table_source": b.meta.get("table_source"),
"preceding_text": b.meta.get("preceding_text"),
"following_text": b.meta.get("following_text"),
"cross_page": b.meta.get("cross_page"),
"embedding_text": b.meta.get("embedding_text"),
}
for b in blocks
if b.type == BlockType.TABLE
]
meta: dict = {
"blocks": [block_to_layout_dict(b) for b in blocks],
"bboxes": bboxes,
"images": images,
"tables": tables,
}
if pages:
meta["pages"] = pages
if len(pages) == 1:
meta["page"] = pages[0]
headings = [b.text for b in blocks if b.type == BlockType.HEADING]
if headings:
meta["heading"] = headings[0]
meta["nearest_heading"] = headings[0]
elif blocks and blocks[0].meta.get("nearest_heading"):
meta["nearest_heading"] = blocks[0].meta.get("nearest_heading")
table_blocks = [b for b in blocks if b.type == BlockType.TABLE]
if table_blocks:
primary = table_blocks[0]
meta["table_title"] = primary.meta.get("table_title") or meta.get("nearest_heading")
if not meta.get("chunk_strategy"):
meta["chunk_strategy"] = "table_with_context"
meta["retrieval"] = meta.get("retrieval", True)
if primary.meta.get("embedding_text"):
meta["embedding_text"] = primary.meta["embedding_text"]
if primary.meta.get("keywords"):
meta["keywords"] = primary.meta["keywords"]
order_indices = [b.meta.get("order_index") for b in blocks if b.meta.get("order_index") is not None]
if order_indices:
meta["order_range"] = meta.get("order_range") or [min(order_indices), max(order_indices)]
return meta
def render_blocks(blocks: list[Block], index: int, meta: dict | None = None) -> Chunk:
"""Render blocks in reading order: heading → body → table/image → notes."""
parts: list[str] = []
for block in blocks:
rendered = block.render().strip()
if rendered:
parts.append(rendered)
content = "\n\n".join(parts)
layout_meta = collect_chunk_layout_meta(blocks)
chunk_meta = {**(meta or {}), **layout_meta}
return Chunk(
index=index,
content=content,
char_count=len(content),
block_types=[b.type.value for b in blocks],
meta=chunk_meta,
)