@@ -0,0 +1,213 @@
|
||||
"""End-to-end PyMuPDF PDF pipeline matching the architecture diagram."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
import pdfplumber
|
||||
|
||||
from rag_cut.models import Block
|
||||
from rag_cut.parsers.pdf.layout import (
|
||||
LayoutRegion,
|
||||
PageLayout,
|
||||
_rect_overlap,
|
||||
_sort_reading_order,
|
||||
analyze_page_layout,
|
||||
render_page,
|
||||
)
|
||||
from rag_cut.parsers.pdf.standardize import open_document
|
||||
from rag_cut.parsers.pdf.tables import rows_to_markdown
|
||||
from rag_cut.parsers.pdf.text_extract import extract_text_blocks
|
||||
from rag_cut.parsers.pdf.visual_extract import extract_visual_blocks
|
||||
|
||||
|
||||
@dataclass
|
||||
class _PageItem:
|
||||
y0: float
|
||||
x0: float
|
||||
kind: str
|
||||
block: Block
|
||||
|
||||
|
||||
def _pdfplumber_tables(path: Path) -> dict[int, list[dict]]:
|
||||
"""Extract tables per page with bbox from pdfplumber."""
|
||||
tables_by_page: dict[int, list[dict]] = {}
|
||||
try:
|
||||
with pdfplumber.open(path) as pdf:
|
||||
for i, page in enumerate(pdf.pages):
|
||||
found = []
|
||||
try:
|
||||
for idx, table in enumerate(page.find_tables()):
|
||||
rows = table.extract() or []
|
||||
if not rows:
|
||||
continue
|
||||
bbox = table.bbox
|
||||
found.append(
|
||||
{
|
||||
"rows": rows,
|
||||
"table_index": idx,
|
||||
"source": "pdfplumber",
|
||||
"bbox": list(bbox) if bbox else None,
|
||||
}
|
||||
)
|
||||
except Exception:
|
||||
rows_list = page.extract_tables() or []
|
||||
for idx, rows in enumerate(rows_list):
|
||||
found.append({"rows": rows, "table_index": idx, "source": "pdfplumber", "bbox": None})
|
||||
if found:
|
||||
tables_by_page[i] = found
|
||||
except Exception:
|
||||
pass
|
||||
return tables_by_page
|
||||
|
||||
|
||||
def _inject_pdfplumber_tables(layout: PageLayout, plumber_tables: list[dict]) -> PageLayout:
|
||||
if not plumber_tables:
|
||||
return layout
|
||||
if any(r.kind == "table" for r in layout.regions):
|
||||
return layout
|
||||
|
||||
for item in plumber_tables:
|
||||
rows = item.get("rows") or []
|
||||
md = rows_to_markdown(rows)
|
||||
if not md:
|
||||
continue
|
||||
bbox = item.get("bbox")
|
||||
if bbox and len(bbox) == 4:
|
||||
x0, y0, x1, y1 = bbox
|
||||
else:
|
||||
idx = int(item.get("table_index", 0))
|
||||
x0, y0 = 0, layout.page_height * (idx + 1) / (len(plumber_tables) + 1)
|
||||
x1, y1 = layout.page_width, layout.page_height * (idx + 2) / (len(plumber_tables) + 1)
|
||||
layout.regions.append(
|
||||
LayoutRegion(
|
||||
x0=x0,
|
||||
y0=y0,
|
||||
x1=x1,
|
||||
y1=y1,
|
||||
kind="table",
|
||||
data={
|
||||
"rows": rows,
|
||||
"table_index": item.get("table_index", 0),
|
||||
"source": item.get("source", "pdfplumber"),
|
||||
},
|
||||
)
|
||||
)
|
||||
ordered_boxes = _sort_reading_order(
|
||||
[(r.x0, r.y0, r.x1, r.y1) for r in layout.regions],
|
||||
layout.page_width,
|
||||
)
|
||||
order = {box: i for i, box in enumerate(ordered_boxes)}
|
||||
layout.regions.sort(key=lambda r: order.get((r.x0, r.y0, r.x1, r.y1), len(order)))
|
||||
return layout
|
||||
|
||||
|
||||
def _bbox_key(bbox: list[float] | tuple[float, ...]) -> tuple[float, ...]:
|
||||
return tuple(round(v, 1) for v in bbox)
|
||||
|
||||
|
||||
def _region_box(region: LayoutRegion) -> tuple[float, float, float, float]:
|
||||
return (region.x0, region.y0, region.x1, region.y1)
|
||||
|
||||
|
||||
def _match_block_to_region(
|
||||
region: LayoutRegion,
|
||||
candidates: list[Block],
|
||||
used: set[int],
|
||||
) -> Block | None:
|
||||
"""Match a layout region to a parsed block via exact bbox key or overlap."""
|
||||
region_box = _region_box(region)
|
||||
key = _bbox_key(region_box)
|
||||
|
||||
for block in candidates:
|
||||
if id(block) in used:
|
||||
continue
|
||||
bbox = block.meta.get("bbox")
|
||||
if bbox and _bbox_key(bbox) == key:
|
||||
return block
|
||||
|
||||
best: Block | None = None
|
||||
best_overlap = 0.35
|
||||
for block in candidates:
|
||||
if id(block) in used:
|
||||
continue
|
||||
bbox = block.meta.get("bbox")
|
||||
if not bbox:
|
||||
continue
|
||||
overlap = _rect_overlap(region_box, tuple(bbox))
|
||||
if overlap > best_overlap:
|
||||
best_overlap = overlap
|
||||
best = block
|
||||
return best
|
||||
|
||||
|
||||
def _merge_page_blocks(
|
||||
text_blocks: list[Block],
|
||||
visual_blocks: list[Block],
|
||||
layout: PageLayout,
|
||||
) -> list[Block]:
|
||||
"""Merge text and visual branches in page reading order."""
|
||||
used: set[int] = set()
|
||||
merged: list[_PageItem] = []
|
||||
|
||||
for region in layout.regions:
|
||||
if region.kind == "text":
|
||||
candidates = text_blocks
|
||||
elif region.kind in {"table", "image"}:
|
||||
candidates = visual_blocks
|
||||
else:
|
||||
continue
|
||||
|
||||
block = _match_block_to_region(region, candidates, used)
|
||||
if block is not None:
|
||||
merged.append(_PageItem(y0=region.y0, x0=region.x0, kind=region.kind, block=block))
|
||||
used.add(id(block))
|
||||
|
||||
leftovers: list[_PageItem] = []
|
||||
for block in text_blocks + visual_blocks:
|
||||
if id(block) not in used:
|
||||
bbox = block.meta.get("bbox", [0, 0, 0, 0])
|
||||
leftovers.append(_PageItem(y0=bbox[1], x0=bbox[0], kind=block.type.value, block=block))
|
||||
|
||||
merged.extend(sorted(leftovers, key=lambda it: (it.y0, it.x0)))
|
||||
return [it.block for it in merged]
|
||||
|
||||
|
||||
def parse_pdf(path: Path, assets_dir: Path) -> list[Block]:
|
||||
"""
|
||||
PDF pipeline:
|
||||
1. 文档标准化
|
||||
2. 页面渲染 + 版面分析
|
||||
3. 文本块提取 ∥ 图片/表格区域裁剪 + OCR
|
||||
4. 章节级语义切片(合并两路结果,标注章节上下文)
|
||||
"""
|
||||
assets_dir.mkdir(parents=True, exist_ok=True)
|
||||
renders_dir = assets_dir / "pages"
|
||||
renders_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
doc = open_document(path)
|
||||
plumber_tables = _pdfplumber_tables(path)
|
||||
all_blocks: list[Block] = []
|
||||
chapter_title: str | None = None
|
||||
img_counter = 0
|
||||
|
||||
try:
|
||||
for page_index, page in enumerate(doc):
|
||||
layout = analyze_page_layout(page, page_index)
|
||||
layout = _inject_pdfplumber_tables(layout, plumber_tables.get(page_index, []))
|
||||
|
||||
page_render = render_page(page)
|
||||
render_path = renders_dir / f"page{page_index + 1}.png"
|
||||
page_render.save(str(render_path))
|
||||
|
||||
text_blocks, chapter_title = extract_text_blocks(layout, chapter_title)
|
||||
visual_blocks, img_counter = extract_visual_blocks(
|
||||
page, layout, doc, assets_dir, chapter_title, img_counter
|
||||
)
|
||||
page_blocks = _merge_page_blocks(text_blocks, visual_blocks, layout)
|
||||
all_blocks.extend(page_blocks)
|
||||
finally:
|
||||
doc.close()
|
||||
|
||||
return all_blocks
|
||||
Reference in New Issue
Block a user