105 lines
3.3 KiB
Python
105 lines
3.3 KiB
Python
"""Build rich table blocks from PDF layout regions."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
|
|
import fitz
|
|
|
|
from rag_cut.models import Block, BlockType
|
|
from rag_cut.parsers.pdf.layout import LayoutRegion, PageLayout
|
|
from rag_cut.parsers.pdf.ocr import describe_visual, ocr_pixmap
|
|
from rag_cut.parsers.pdf.tables import (
|
|
detect_header_row_count,
|
|
extract_table_keywords,
|
|
header_signature,
|
|
normalize_rows,
|
|
rows_to_markdown,
|
|
split_body_and_footnotes,
|
|
)
|
|
|
|
|
|
def _crop_region(page: fitz.Page, region: LayoutRegion, zoom: float = 2.0) -> fitz.Pixmap:
|
|
clip = fitz.Rect(region.x0, region.y0, region.x1, region.y1)
|
|
return page.get_pixmap(matrix=fitz.Matrix(zoom, zoom), clip=clip, alpha=False)
|
|
|
|
|
|
def _save_pixmap(pix: fitz.Pixmap, path: Path) -> None:
|
|
if pix.n - pix.alpha > 3:
|
|
pix = fitz.Pixmap(fitz.csRGB, pix)
|
|
pix.save(str(path))
|
|
|
|
|
|
def build_table_block(
|
|
page: fitz.Page,
|
|
region: LayoutRegion,
|
|
layout: PageLayout,
|
|
assets_dir: Path,
|
|
chapter_title: str | None,
|
|
) -> Block | None:
|
|
"""Extract a table block with rows, screenshot, OCR and structural metadata."""
|
|
raw_rows = region.data.get("rows") or []
|
|
normalized = normalize_rows(raw_rows)
|
|
if not normalized:
|
|
return None
|
|
|
|
header_rows = detect_header_row_count(normalized)
|
|
body_rows, footnote_rows, footnotes = split_body_and_footnotes(normalized, header_rows)
|
|
data_rows = normalized[:header_rows] + body_rows
|
|
if not data_rows:
|
|
return None
|
|
|
|
page_no = layout.page_index + 1
|
|
table_index = int(region.data.get("table_index", 0)) + 1
|
|
crop_id = f"page{page_no}_table{table_index}.png"
|
|
crops_dir = assets_dir / "crops"
|
|
crops_dir.mkdir(parents=True, exist_ok=True)
|
|
crop_path = crops_dir / crop_id
|
|
|
|
ocr_text = ""
|
|
try:
|
|
crop_pix = _crop_region(page, region)
|
|
_save_pixmap(crop_pix, crop_path)
|
|
ocr_text = ocr_pixmap(crop_pix)
|
|
except Exception:
|
|
crop_path_str = ""
|
|
else:
|
|
crop_path_str = str(crop_path)
|
|
|
|
sig = header_signature(normalized, header_rows)
|
|
md = rows_to_markdown(normalized, header_rows=header_rows, include_footnotes=footnotes)
|
|
if not md:
|
|
return None
|
|
|
|
header_text = " ".join(" ".join(r for r in row if r) for row in normalized[:header_rows])
|
|
keywords = extract_table_keywords(header_text, md, footnotes, ocr_text)
|
|
|
|
meta: dict = {
|
|
"page": page_no,
|
|
"pages": [page_no],
|
|
"bbox": [region.x0, region.y0, region.x1, region.y1],
|
|
"bboxes": [{"page": page_no, "bbox": [region.x0, region.y0, region.x1, region.y1]}],
|
|
"table_source": region.data.get("source", "pymupdf"),
|
|
"crop_path": crop_path_str,
|
|
"rows": normalized,
|
|
"row_count": len(normalized),
|
|
"col_count": max((len(r) for r in normalized), default=0),
|
|
"header_rows": header_rows,
|
|
"header_signature": [list(r) for r in sig],
|
|
"footnotes": footnotes,
|
|
"keywords": keywords,
|
|
"cross_page": False,
|
|
}
|
|
if chapter_title:
|
|
meta["chapter"] = chapter_title
|
|
|
|
return Block(
|
|
type=BlockType.TABLE,
|
|
markdown=md,
|
|
ocr_text=ocr_text,
|
|
text=describe_visual(ocr_text, "表格", crop_id) if ocr_text else "",
|
|
image_id=crop_id,
|
|
image_path=crop_path_str or None,
|
|
meta=meta,
|
|
)
|