@@ -0,0 +1,104 @@
|
||||
"""Build rich table blocks from PDF layout regions."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import fitz
|
||||
|
||||
from rag_cut.models import Block, BlockType
|
||||
from rag_cut.parsers.pdf.layout import LayoutRegion, PageLayout
|
||||
from rag_cut.parsers.pdf.ocr import describe_visual, ocr_pixmap
|
||||
from rag_cut.parsers.pdf.tables import (
|
||||
detect_header_row_count,
|
||||
extract_table_keywords,
|
||||
header_signature,
|
||||
normalize_rows,
|
||||
rows_to_markdown,
|
||||
split_body_and_footnotes,
|
||||
)
|
||||
|
||||
|
||||
def _crop_region(page: fitz.Page, region: LayoutRegion, zoom: float = 2.0) -> fitz.Pixmap:
|
||||
clip = fitz.Rect(region.x0, region.y0, region.x1, region.y1)
|
||||
return page.get_pixmap(matrix=fitz.Matrix(zoom, zoom), clip=clip, alpha=False)
|
||||
|
||||
|
||||
def _save_pixmap(pix: fitz.Pixmap, path: Path) -> None:
|
||||
if pix.n - pix.alpha > 3:
|
||||
pix = fitz.Pixmap(fitz.csRGB, pix)
|
||||
pix.save(str(path))
|
||||
|
||||
|
||||
def build_table_block(
|
||||
page: fitz.Page,
|
||||
region: LayoutRegion,
|
||||
layout: PageLayout,
|
||||
assets_dir: Path,
|
||||
chapter_title: str | None,
|
||||
) -> Block | None:
|
||||
"""Extract a table block with rows, screenshot, OCR and structural metadata."""
|
||||
raw_rows = region.data.get("rows") or []
|
||||
normalized = normalize_rows(raw_rows)
|
||||
if not normalized:
|
||||
return None
|
||||
|
||||
header_rows = detect_header_row_count(normalized)
|
||||
body_rows, footnote_rows, footnotes = split_body_and_footnotes(normalized, header_rows)
|
||||
data_rows = normalized[:header_rows] + body_rows
|
||||
if not data_rows:
|
||||
return None
|
||||
|
||||
page_no = layout.page_index + 1
|
||||
table_index = int(region.data.get("table_index", 0)) + 1
|
||||
crop_id = f"page{page_no}_table{table_index}.png"
|
||||
crops_dir = assets_dir / "crops"
|
||||
crops_dir.mkdir(parents=True, exist_ok=True)
|
||||
crop_path = crops_dir / crop_id
|
||||
|
||||
ocr_text = ""
|
||||
try:
|
||||
crop_pix = _crop_region(page, region)
|
||||
_save_pixmap(crop_pix, crop_path)
|
||||
ocr_text = ocr_pixmap(crop_pix)
|
||||
except Exception:
|
||||
crop_path_str = ""
|
||||
else:
|
||||
crop_path_str = str(crop_path)
|
||||
|
||||
sig = header_signature(normalized, header_rows)
|
||||
md = rows_to_markdown(normalized, header_rows=header_rows, include_footnotes=footnotes)
|
||||
if not md:
|
||||
return None
|
||||
|
||||
header_text = " ".join(" ".join(r for r in row if r) for row in normalized[:header_rows])
|
||||
keywords = extract_table_keywords(header_text, md, footnotes, ocr_text)
|
||||
|
||||
meta: dict = {
|
||||
"page": page_no,
|
||||
"pages": [page_no],
|
||||
"bbox": [region.x0, region.y0, region.x1, region.y1],
|
||||
"bboxes": [{"page": page_no, "bbox": [region.x0, region.y0, region.x1, region.y1]}],
|
||||
"table_source": region.data.get("source", "pymupdf"),
|
||||
"crop_path": crop_path_str,
|
||||
"rows": normalized,
|
||||
"row_count": len(normalized),
|
||||
"col_count": max((len(r) for r in normalized), default=0),
|
||||
"header_rows": header_rows,
|
||||
"header_signature": [list(r) for r in sig],
|
||||
"footnotes": footnotes,
|
||||
"keywords": keywords,
|
||||
"cross_page": False,
|
||||
}
|
||||
if chapter_title:
|
||||
meta["chapter"] = chapter_title
|
||||
|
||||
return Block(
|
||||
type=BlockType.TABLE,
|
||||
markdown=md,
|
||||
ocr_text=ocr_text,
|
||||
text=describe_visual(ocr_text, "表格", crop_id) if ocr_text else "",
|
||||
image_id=crop_id,
|
||||
image_path=crop_path_str or None,
|
||||
meta=meta,
|
||||
)
|
||||
Reference in New Issue
Block a user