165 lines
6.1 KiB
Python
165 lines
6.1 KiB
Python
"""Render block groups into chunk markdown strings with layout metadata."""
|
|||
|
|
|
||
|
|
from __future__ import annotations
|
||
|
|
|
||
|
|
from rag_cut.models import Block, BlockType, Chunk
|
||
|
|
from rag_cut.parsers.pdf.tables import table_meta_summary
|
||
|
|
|
||
|
|
|
||
|
|
def block_to_layout_dict(block: Block) -> dict:
|
||
|
|
"""Serialize a block for chunk metadata and frontend positional rendering."""
|
||
|
|
entry: dict = {
|
||
|
|
"type": block.type.value,
|
||
|
|
"order_index": block.meta.get("order_index"),
|
||
|
|
"page": block.meta.get("page"),
|
||
|
|
"pages": block.meta.get("pages"),
|
||
|
|
"bbox": block.meta.get("bbox"),
|
||
|
|
"bboxes": block.meta.get("bboxes"),
|
||
|
|
"parent_heading": block.meta.get("parent_heading"),
|
||
|
|
"nearest_heading": block.meta.get("nearest_heading"),
|
||
|
|
"bound_heading": block.meta.get("bound_heading"),
|
||
|
|
"preceding_text": block.meta.get("preceding_text"),
|
||
|
|
"following_text": block.meta.get("following_text"),
|
||
|
|
}
|
||
|
|
if block.type == BlockType.HEADING:
|
||
|
|
entry["text"] = block.text
|
||
|
|
entry["level"] = block.level
|
||
|
|
elif block.type == BlockType.IMAGE:
|
||
|
|
entry["text"] = block.text
|
||
|
|
entry["image_path"] = block.image_path
|
||
|
|
entry["image_id"] = block.image_id
|
||
|
|
entry["ocr_text"] = block.ocr_text
|
||
|
|
elif block.type == BlockType.TABLE:
|
||
|
|
entry["text"] = block.markdown or block.text
|
||
|
|
entry["markdown"] = block.markdown or block.text
|
||
|
|
entry["ocr_text"] = block.ocr_text
|
||
|
|
entry["image_path"] = block.image_path
|
||
|
|
entry["image_id"] = block.image_id
|
||
|
|
entry["crop_path"] = block.meta.get("crop_path")
|
||
|
|
entry.update(table_meta_summary(block.meta))
|
||
|
|
entry["embedding_text"] = block.meta.get("embedding_text")
|
||
|
|
else:
|
||
|
|
entry["text"] = block.text or block.markdown
|
||
|
|
return entry
|
||
|
|
|
||
|
|
|
||
|
|
def collect_chunk_layout_meta(blocks: list[Block]) -> dict:
|
||
|
|
"""Aggregate page/bbox/image/table metadata for a chunk group."""
|
||
|
|
pages = sorted({b.meta.get("page") for b in blocks if b.meta.get("page") is not None})
|
||
|
|
for b in blocks:
|
||
|
|
for p in b.meta.get("pages") or []:
|
||
|
|
if p is not None:
|
||
|
|
pages.append(p)
|
||
|
|
pages = sorted(set(pages))
|
||
|
|
|
||
|
|
bboxes = [
|
||
|
|
{
|
||
|
|
"order_index": b.meta.get("order_index"),
|
||
|
|
"page": b.meta.get("page"),
|
||
|
|
"bbox": b.meta.get("bbox"),
|
||
|
|
"type": b.type.value,
|
||
|
|
}
|
||
|
|
for b in blocks
|
||
|
|
if b.meta.get("bbox")
|
||
|
|
]
|
||
|
|
images = [
|
||
|
|
{
|
||
|
|
"order_index": b.meta.get("order_index"),
|
||
|
|
"image_id": b.image_id,
|
||
|
|
"image_path": b.image_path,
|
||
|
|
"page": b.meta.get("page"),
|
||
|
|
"bbox": b.meta.get("bbox"),
|
||
|
|
"ocr_text": b.ocr_text,
|
||
|
|
"bound_heading": b.meta.get("bound_heading"),
|
||
|
|
"preceding_text": b.meta.get("preceding_text"),
|
||
|
|
"following_text": b.meta.get("following_text"),
|
||
|
|
}
|
||
|
|
for b in blocks
|
||
|
|
if b.type == BlockType.IMAGE
|
||
|
|
]
|
||
|
|
tables = [
|
||
|
|
{
|
||
|
|
"order_index": b.meta.get("order_index"),
|
||
|
|
"table_title": b.meta.get("table_title"),
|
||
|
|
"image_path": b.image_path or b.meta.get("crop_path"),
|
||
|
|
"crop_path": b.meta.get("crop_path"),
|
||
|
|
"page": b.meta.get("page"),
|
||
|
|
"pages": b.meta.get("pages"),
|
||
|
|
"bbox": b.meta.get("bbox"),
|
||
|
|
"bboxes": b.meta.get("bboxes"),
|
||
|
|
"markdown": b.markdown,
|
||
|
|
"ocr_text": b.ocr_text,
|
||
|
|
"footnotes": b.meta.get("footnotes"),
|
||
|
|
"keywords": b.meta.get("keywords"),
|
||
|
|
"nearest_heading": b.meta.get("nearest_heading"),
|
||
|
|
"chapter": b.meta.get("chapter"),
|
||
|
|
"row_count": b.meta.get("row_count"),
|
||
|
|
"col_count": b.meta.get("col_count"),
|
||
|
|
"header_rows": b.meta.get("header_rows"),
|
||
|
|
"table_source": b.meta.get("table_source"),
|
||
|
|
"preceding_text": b.meta.get("preceding_text"),
|
||
|
|
"following_text": b.meta.get("following_text"),
|
||
|
|
"cross_page": b.meta.get("cross_page"),
|
||
|
|
"embedding_text": b.meta.get("embedding_text"),
|
||
|
|
}
|
||
|
|
for b in blocks
|
||
|
|
if b.type == BlockType.TABLE
|
||
|
|
]
|
||
|
|
|
||
|
|
meta: dict = {
|
||
|
|
"blocks": [block_to_layout_dict(b) for b in blocks],
|
||
|
|
"bboxes": bboxes,
|
||
|
|
"images": images,
|
||
|
|
"tables": tables,
|
||
|
|
}
|
||
|
|
if pages:
|
||
|
|
meta["pages"] = pages
|
||
|
|
if len(pages) == 1:
|
||
|
|
meta["page"] = pages[0]
|
||
|
|
|
||
|
|
headings = [b.text for b in blocks if b.type == BlockType.HEADING]
|
||
|
|
if headings:
|
||
|
|
meta["heading"] = headings[0]
|
||
|
|
meta["nearest_heading"] = headings[0]
|
||
|
|
elif blocks and blocks[0].meta.get("nearest_heading"):
|
||
|
|
meta["nearest_heading"] = blocks[0].meta.get("nearest_heading")
|
||
|
|
|
||
|
|
table_blocks = [b for b in blocks if b.type == BlockType.TABLE]
|
||
|
|
if table_blocks:
|
||
|
|
primary = table_blocks[0]
|
||
|
|
meta["table_title"] = primary.meta.get("table_title") or meta.get("nearest_heading")
|
||
|
|
if not meta.get("chunk_strategy"):
|
||
|
|
meta["chunk_strategy"] = "table_with_context"
|
||
|
|
meta["retrieval"] = meta.get("retrieval", True)
|
||
|
|
if primary.meta.get("embedding_text"):
|
||
|
|
meta["embedding_text"] = primary.meta["embedding_text"]
|
||
|
|
if primary.meta.get("keywords"):
|
||
|
|
meta["keywords"] = primary.meta["keywords"]
|
||
|
|
|
||
|
|
order_indices = [b.meta.get("order_index") for b in blocks if b.meta.get("order_index") is not None]
|
||
|
|
if order_indices:
|
||
|
|
meta["order_range"] = meta.get("order_range") or [min(order_indices), max(order_indices)]
|
||
|
|
|
||
|
|
return meta
|
||
|
|
|
||
|
|
|
||
|
|
def render_blocks(blocks: list[Block], index: int, meta: dict | None = None) -> Chunk:
|
||
|
|
"""Render blocks in reading order: heading → body → table/image → notes."""
|
||
|
|
parts: list[str] = []
|
||
|
|
for block in blocks:
|
||
|
|
rendered = block.render().strip()
|
||
|
|
if rendered:
|
||
|
|
parts.append(rendered)
|
||
|
|
content = "\n\n".join(parts)
|
||
|
|
|
||
|
|
layout_meta = collect_chunk_layout_meta(blocks)
|
||
|
|
chunk_meta = {**(meta or {}), **layout_meta}
|
||
|
|
|
||
|
|
return Chunk(
|
||
|
|
index=index,
|
||
|
|
content=content,
|
||
|
|
char_count=len(content),
|
||
|
|
block_types=[b.type.value for b in blocks],
|
||
|
|
meta=chunk_meta,
|
||
|
|
)
|