Files
陈辅元 8466ed2fbe Enhance table title handling and improve heading detection logic
- Added logic to set `table_title` from `caption` if it is under 160 characters and not already set.
- Updated `_blocks_from_content_list` to assign `table_title` based on `caption` length.
- Introduced new regex patterns for better detection of TOC entries and noise.
- Enhanced heading detection to differentiate between numbered instructions and actual headings.
- Added tests to verify that table captions are correctly assigned as titles and that numbered instructions are treated as body text.
2026-07-16 16:01:42 +08:00

249 lines
9.1 KiB
Python

"""Layout metadata: reading order, heading binding, image context."""
from __future__ import annotations
from rag_cut.models import Block, BlockType
from rag_cut.parsers.pdf.tables import build_table_embedding_text, extract_table_keywords, guess_table_title
from rag_cut.splitters.heading_splitter import is_heading_block, normalize_heading_block
def assign_order_index(blocks: list[Block], start: int = 0) -> list[Block]:
"""Assign a global document-order index to every block."""
result: list[Block] = []
for idx, block in enumerate(blocks):
meta = dict(block.meta)
meta["order_index"] = start + idx
result.append(block.model_copy(update={"meta": meta}))
return result
def sort_blocks_reading_order(blocks: list[Block]) -> list[Block]:
"""Sort by page then top-to-bottom / left-to-right; stable for missing bboxes."""
def sort_key(item: tuple[int, Block]) -> tuple[int, float, float, int]:
idx, block = item
page = block.meta.get("page")
try:
page_key = int(page) if page is not None else 10**9
except (TypeError, ValueError):
page_key = 10**9
bbox = block.meta.get("bbox")
if isinstance(bbox, (list, tuple)) and len(bbox) >= 4:
try:
return (page_key, float(bbox[1]), float(bbox[0]), idx)
except (TypeError, ValueError):
pass
return (page_key, float(idx), 0.0, idx)
return [block for _, block in sorted(enumerate(blocks), key=sort_key)]
def bind_heading_context(blocks: list[Block]) -> list[Block]:
"""
Propagate heading hierarchy and bind images to nearest heading and adjacent text.
Chunk content should follow: heading → body → image → OCR/caption → subsequent body.
"""
heading_stack: list[tuple[int, str]] = []
result: list[Block] = []
for i, block in enumerate(blocks):
meta = dict(block.meta)
candidate = normalize_heading_block(block)
if is_heading_block(candidate):
level = candidate.level or 1
while heading_stack and heading_stack[-1][0] >= level:
heading_stack.pop()
heading_stack.append((level, candidate.text))
meta["section_boundary"] = True
if block.type != BlockType.HEADING or candidate is not block:
block = candidate.model_copy(update={"meta": {**dict(candidate.meta), **meta}})
meta = dict(block.meta)
else:
meta["section_boundary"] = False
# Demote overlong HEADING blobs (title+body merge) back to paragraph.
if block.type == BlockType.HEADING and candidate.type != BlockType.HEADING:
block = candidate.model_copy(update={"meta": {**dict(candidate.meta), **meta}})
meta = dict(block.meta)
if heading_stack:
meta["parent_heading"] = heading_stack[-1][1]
meta["heading_path"] = [text for _, text in heading_stack]
meta["nearest_heading"] = heading_stack[-1][1]
if (block.level or 1) <= 2 and block.type == BlockType.HEADING:
meta["chapter"] = block.text
elif "chapter" not in meta and len(heading_stack) >= 1:
# Keep chapter as nearest level-1/2 ancestor
for lvl, text in reversed(heading_stack):
if lvl <= 2:
meta["chapter"] = text
break
if block.type == BlockType.IMAGE:
_bind_image_context(blocks, i, meta)
elif block.type == BlockType.TABLE:
_bind_table_context(blocks, i, meta)
result.append(block.model_copy(update={"meta": meta}))
return result
def _heading_text(block: Block) -> str | None:
candidate = normalize_heading_block(block)
if is_heading_block(candidate):
return (candidate.text or "").strip() or None
return None
def _spatial_heading_above(
blocks: list[Block],
index: int,
page: object,
bbox: list[float] | None,
) -> str | None:
"""Pick same-page heading whose bottom edge is nearest above the image top."""
if page is None or not bbox or len(bbox) < 4:
return None
try:
page_key = int(page)
img_y0 = float(bbox[1])
except (TypeError, ValueError):
return None
best_text: str | None = None
best_dist = float("inf")
for j, block in enumerate(blocks):
if j == index:
continue
text = _heading_text(block)
if not text:
continue
try:
if int(block.meta.get("page")) != page_key:
continue
except (TypeError, ValueError):
continue
hb = block.meta.get("bbox")
if not isinstance(hb, (list, tuple)) or len(hb) < 4:
continue
try:
heading_y1 = float(hb[3])
except (TypeError, ValueError):
continue
if heading_y1 > img_y0 + 2:
continue
dist = img_y0 - heading_y1
if dist < best_dist:
best_dist = dist
best_text = text
return best_text
def _list_heading_above(blocks: list[Block], index: int) -> str | None:
for j in range(index - 1, -1, -1):
text = _heading_text(blocks[j])
if text:
return text
return None
def _bind_image_context(blocks: list[Block], index: int, meta: dict) -> None:
"""Bind image to nearest heading above and adjacent body text on the same page."""
page = meta.get("page")
bbox = meta.get("bbox") if isinstance(meta.get("bbox"), list) else None
bound = _spatial_heading_above(blocks, index, page, bbox) or _list_heading_above(blocks, index)
if bound:
meta["bound_heading"] = bound
for j in range(index - 1, -1, -1):
prev = blocks[j]
if _heading_text(prev):
break
if prev.type == BlockType.PARAGRAPH and prev.meta.get("page") == page:
meta["preceding_text"] = (prev.text or "")[:300]
break
if not meta.get("bound_heading") and meta.get("nearest_heading"):
meta["bound_heading"] = meta["nearest_heading"]
for j in range(index + 1, len(blocks)):
nxt = blocks[j]
if nxt.type == BlockType.IMAGE:
break
if nxt.type == BlockType.PARAGRAPH and nxt.meta.get("page") == page:
meta["following_text"] = (nxt.text or "")[:300]
break
if _heading_text(nxt):
break
def _bind_table_context(blocks: list[Block], index: int, meta: dict) -> None:
"""Bind table title, surrounding text and retrieval fields."""
page = meta.get("page")
block = blocks[index]
caption = (meta.get("caption") or "").strip()
if caption and len(caption) <= 160 and not meta.get("table_title"):
meta["table_title"] = caption
for j in range(index - 1, -1, -1):
prev = blocks[j]
if prev.type == BlockType.TABLE:
break
heading = _heading_text(prev)
if heading:
if not meta.get("table_title"):
meta["table_title"] = heading
meta.setdefault("bound_heading", heading)
break
if prev.type == BlockType.PARAGRAPH and prev.meta.get("page") == page:
title = guess_table_title(prev.text or "")
if title:
meta["table_title"] = title
meta["preceding_text"] = (prev.text or "")[:400]
break
if not meta.get("table_title") and meta.get("nearest_heading"):
meta.setdefault("table_title", meta["nearest_heading"])
for j in range(index + 1, len(blocks)):
nxt = blocks[j]
if nxt.type == BlockType.TABLE:
break
if nxt.type == BlockType.PARAGRAPH and nxt.meta.get("page") == page:
text = (nxt.text or "").strip()
if text and len(text) <= 300:
meta["following_text"] = text
if not meta.get("footnotes") and any(k in text for k in ("注", "备注", "说明", "Note")):
meta["table_description"] = text
break
if _heading_text(nxt):
break
if not meta.get("keywords"):
meta["keywords"] = extract_table_keywords(
meta.get("table_title") or "",
block.markdown or "",
meta.get("table_description") or meta.get("footnotes") or "",
block.ocr_text or "",
)
meta["embedding_text"] = build_table_embedding_text(
chapter=meta.get("chapter") or meta.get("nearest_heading") or "",
table_title=meta.get("table_title") or "",
markdown=block.markdown or "",
description=meta.get("table_description") or meta.get("preceding_text") or "",
footnotes=meta.get("footnotes") or meta.get("following_text") or "",
keywords=meta.get("keywords") or [],
ocr_text=block.ocr_text or "",
)
def enrich_layout_metadata(blocks: list[Block]) -> list[Block]:
"""Full post-parse enrichment: spatial order + order index + heading/image context."""
blocks = sort_blocks_reading_order(blocks)
blocks = assign_order_index(blocks)
return bind_heading_context(blocks)