@@ -0,0 +1,5 @@
|
||||
"""PyMuPDF-based PDF parsing pipeline."""
|
||||
|
||||
from rag_cut.parsers.pdf.pipeline import parse_pdf
|
||||
|
||||
__all__ = ["parse_pdf"]
|
||||
@@ -0,0 +1,349 @@
|
||||
"""Page rendering and layout analysis."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any
|
||||
|
||||
import fitz
|
||||
|
||||
from rag_cut.parsers.pdf.tables import looks_like_table
|
||||
|
||||
|
||||
_NUMBERED_LINE_RE = re.compile(
|
||||
r"^\s*(\d+(?:\.\d+)*)(?:\.|.)?\s*([A-Za-z0-9\u4e00-\u9fff].{2,})$"
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class TextLine:
|
||||
x0: float
|
||||
y0: float
|
||||
x1: float
|
||||
y1: float
|
||||
text: str
|
||||
font_size: float
|
||||
|
||||
|
||||
@dataclass
|
||||
class LayoutRegion:
|
||||
x0: float
|
||||
y0: float
|
||||
x1: float
|
||||
y1: float
|
||||
kind: str # text | table | image
|
||||
data: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
|
||||
@dataclass
|
||||
class PageLayout:
|
||||
page_index: int
|
||||
page_width: float
|
||||
page_height: float
|
||||
body_font_size: float
|
||||
regions: list[LayoutRegion] = field(default_factory=list)
|
||||
|
||||
|
||||
def render_page(page: fitz.Page, zoom: float = 2.0) -> fitz.Pixmap:
|
||||
"""Render page to pixmap for region cropping and OCR."""
|
||||
return page.get_pixmap(matrix=fitz.Matrix(zoom, zoom), alpha=False)
|
||||
|
||||
|
||||
def _median_body_size(text_dict: dict) -> float:
|
||||
sizes: list[float] = []
|
||||
for block in text_dict.get("blocks", []):
|
||||
if block.get("type") != 0:
|
||||
continue
|
||||
for line in block.get("lines", []):
|
||||
for span in line.get("spans", []):
|
||||
sizes.append(span.get("size", 12))
|
||||
return sorted(sizes)[len(sizes) // 2] if sizes else 12
|
||||
|
||||
|
||||
def _collect_text_lines(page: fitz.Page, body_size: float) -> list[TextLine]:
|
||||
text_dict = page.get_text("dict")
|
||||
lines: list[TextLine] = []
|
||||
for block in text_dict.get("blocks", []):
|
||||
if block.get("type") != 0:
|
||||
continue
|
||||
for line in block.get("lines", []):
|
||||
text = "".join(span.get("text", "") for span in line.get("spans", [])).strip()
|
||||
if not text:
|
||||
continue
|
||||
x0, y0, x1, y1 = line["bbox"]
|
||||
max_size = body_size
|
||||
for span in line.get("spans", []):
|
||||
max_size = max(max_size, span.get("size", body_size))
|
||||
lines.append(TextLine(x0=x0, y0=y0, x1=x1, y1=y1, text=text, font_size=max_size))
|
||||
return lines
|
||||
|
||||
|
||||
def _detect_columns(
|
||||
items: list[tuple[float, float, float, float]],
|
||||
page_width: float,
|
||||
) -> list[tuple[float, float, float, float]]:
|
||||
if not items:
|
||||
return items
|
||||
mid = page_width / 2
|
||||
left = [b for b in items if (b[0] + b[2]) / 2 < mid]
|
||||
right = [b for b in items if (b[0] + b[2]) / 2 >= mid]
|
||||
if len(left) >= 2 and len(right) >= 2:
|
||||
left.sort(key=lambda b: (b[1], b[0]))
|
||||
right.sort(key=lambda b: (b[1], b[0]))
|
||||
return left + right
|
||||
return sorted(items, key=lambda b: (b[1], b[0]))
|
||||
|
||||
|
||||
def _sort_reading_order(
|
||||
items: list[tuple[float, float, float, float]],
|
||||
page_width: float,
|
||||
) -> list[tuple[float, float, float, float]]:
|
||||
"""Sort page boxes in a human reading order, preserving two-column flows."""
|
||||
if not items:
|
||||
return []
|
||||
|
||||
mid = page_width / 2
|
||||
full_width: list[tuple[float, float, float, float]] = []
|
||||
column_items: list[tuple[float, float, float, float]] = []
|
||||
for box in items:
|
||||
width = box[2] - box[0]
|
||||
spans_mid = box[0] < mid < box[2]
|
||||
if width >= page_width * 0.60 or (spans_mid and width >= page_width * 0.35):
|
||||
full_width.append(box)
|
||||
else:
|
||||
column_items.append(box)
|
||||
|
||||
left = [b for b in column_items if (b[0] + b[2]) / 2 < mid]
|
||||
right = [b for b in column_items if (b[0] + b[2]) / 2 >= mid]
|
||||
if len(left) < 2 or len(right) < 2:
|
||||
return sorted(items, key=lambda b: (b[1], b[0]))
|
||||
|
||||
full_width.sort(key=lambda b: (b[1], b[0]))
|
||||
ordered: list[tuple[float, float, float, float]] = []
|
||||
segment_top = float("-inf")
|
||||
|
||||
def add_columns_between(top: float, bottom: float) -> None:
|
||||
segment = [b for b in column_items if b[1] >= top and b[1] < bottom]
|
||||
segment_left = sorted([b for b in segment if (b[0] + b[2]) / 2 < mid], key=lambda b: (b[1], b[0]))
|
||||
segment_right = sorted([b for b in segment if (b[0] + b[2]) / 2 >= mid], key=lambda b: (b[1], b[0]))
|
||||
ordered.extend(segment_left)
|
||||
ordered.extend(segment_right)
|
||||
|
||||
for box in full_width:
|
||||
add_columns_between(segment_top, box[1])
|
||||
ordered.append(box)
|
||||
segment_top = box[3]
|
||||
add_columns_between(segment_top, float("inf"))
|
||||
|
||||
seen: set[tuple[float, float, float, float]] = set(ordered)
|
||||
ordered.extend(b for b in sorted(items, key=lambda b: (b[1], b[0])) if b not in seen)
|
||||
return ordered
|
||||
|
||||
|
||||
def _rect_overlap(a: tuple[float, float, float, float], b: tuple[float, float, float, float]) -> float:
|
||||
x0 = max(a[0], b[0])
|
||||
y0 = max(a[1], b[1])
|
||||
x1 = min(a[2], b[2])
|
||||
y1 = min(a[3], b[3])
|
||||
if x1 <= x0 or y1 <= y0:
|
||||
return 0.0
|
||||
inter = (x1 - x0) * (y1 - y0)
|
||||
area_a = max((a[2] - a[0]) * (a[3] - a[1]), 1e-6)
|
||||
return inter / area_a
|
||||
|
||||
|
||||
def _center_inside(inner: tuple[float, float, float, float], outer: tuple[float, float, float, float]) -> bool:
|
||||
cx = (inner[0] + inner[2]) / 2
|
||||
cy = (inner[1] + inner[3]) / 2
|
||||
return outer[0] <= cx <= outer[2] and outer[1] <= cy <= outer[3]
|
||||
|
||||
|
||||
def _horizontal_overlap_ratio(a: TextLine, b: TextLine) -> float:
|
||||
overlap = min(a.x1, b.x1) - max(a.x0, b.x0)
|
||||
if overlap <= 0:
|
||||
return 0.0
|
||||
return overlap / max(min(a.x1 - a.x0, b.x1 - b.x0), 1e-6)
|
||||
|
||||
|
||||
def _line_on_image(
|
||||
line: TextLine,
|
||||
image_boxes: list[tuple[float, float, float, float]],
|
||||
large_image_boxes: list[tuple[float, float, float, float]],
|
||||
) -> bool:
|
||||
# Never drop numbered section titles even if they sit near a screenshot.
|
||||
if _NUMBERED_LINE_RE.match(line.text.strip()):
|
||||
return False
|
||||
bbox = (line.x0, line.y0, line.x1, line.y1)
|
||||
# Large background/screenshot: require near-full coverage of the line, not
|
||||
# merely center-inside (which wiped text sitting in margins of wide figures).
|
||||
if large_image_boxes and any(
|
||||
_center_inside(bbox, box) and _rect_overlap(bbox, box) >= 0.85 for box in large_image_boxes
|
||||
):
|
||||
return True
|
||||
return any(_rect_overlap(bbox, box) >= 0.55 for box in image_boxes)
|
||||
|
||||
|
||||
def _area(box: tuple[float, float, float, float]) -> float:
|
||||
return max(box[2] - box[0], 0.0) * max(box[3] - box[1], 0.0)
|
||||
|
||||
|
||||
def _filter_nested_image_regions(regions: list[LayoutRegion]) -> list[LayoutRegion]:
|
||||
"""Drop image fragments that are already contained in a larger screenshot."""
|
||||
result: list[LayoutRegion] = []
|
||||
boxes = [(r.x0, r.y0, r.x1, r.y1) for r in regions]
|
||||
for region, box in zip(regions, boxes):
|
||||
box_area = _area(box)
|
||||
nested = False
|
||||
for other in boxes:
|
||||
other_area = _area(other)
|
||||
if other == box or other_area <= box_area * 1.5:
|
||||
continue
|
||||
if _center_inside(box, other) and _rect_overlap(box, other) >= 0.85:
|
||||
nested = True
|
||||
break
|
||||
if not nested:
|
||||
result.append(region)
|
||||
return result
|
||||
|
||||
|
||||
def _merge_lines_to_paragraphs(lines: list[TextLine], page_width: float) -> list[TextLine]:
|
||||
if not lines:
|
||||
return []
|
||||
|
||||
ordered_boxes = _sort_reading_order([(ln.x0, ln.y0, ln.x1, ln.y1) for ln in lines], page_width)
|
||||
order = {(b[0], b[1], b[2], b[3]): i for i, b in enumerate(ordered_boxes)}
|
||||
ordered = sorted(lines, key=lambda ln: order.get((ln.x0, ln.y0, ln.x1, ln.y1), (ln.y0, ln.x0)))
|
||||
|
||||
paragraphs: list[TextLine] = []
|
||||
current = ordered[0]
|
||||
for nxt in ordered[1:]:
|
||||
vgap = nxt.y0 - current.y1
|
||||
line_h = max(current.y1 - current.y0, nxt.y1 - nxt.y0, 8.0)
|
||||
size_gap = abs(current.font_size - nxt.font_size)
|
||||
size_ratio = max(current.font_size, nxt.font_size) / max(
|
||||
min(current.font_size, nxt.font_size), 1e-6
|
||||
)
|
||||
# Keep title vs body separate: small absolute gap is enough when ratio is large.
|
||||
same_style = size_gap <= 1.0 and size_ratio <= 1.15
|
||||
either_numbered = bool(
|
||||
_NUMBERED_LINE_RE.match(current.text.strip()) or _NUMBERED_LINE_RE.match(nxt.text.strip())
|
||||
)
|
||||
if (
|
||||
same_style
|
||||
and not either_numbered
|
||||
and vgap <= line_h * 2.2
|
||||
and _horizontal_overlap_ratio(current, nxt) >= 0.25
|
||||
):
|
||||
joiner = "" if current.text.endswith("-") or current.text.endswith(" ") else " "
|
||||
current = TextLine(
|
||||
x0=min(current.x0, nxt.x0),
|
||||
y0=current.y0,
|
||||
x1=max(current.x1, nxt.x1),
|
||||
y1=nxt.y1,
|
||||
text=f"{current.text}{joiner}{nxt.text}",
|
||||
font_size=max(current.font_size, nxt.font_size),
|
||||
)
|
||||
else:
|
||||
paragraphs.append(current)
|
||||
current = nxt
|
||||
paragraphs.append(current)
|
||||
return paragraphs
|
||||
|
||||
|
||||
def analyze_page_layout(page: fitz.Page, page_index: int) -> PageLayout:
|
||||
"""Layout analysis: text paragraphs, table regions, image regions."""
|
||||
text_dict = page.get_text("dict")
|
||||
body_size = _median_body_size(text_dict)
|
||||
page_width = page.rect.width
|
||||
page_height = page.rect.height
|
||||
page_area = page_width * page_height
|
||||
|
||||
table_regions: list[LayoutRegion] = []
|
||||
try:
|
||||
table_finder = page.find_tables()
|
||||
for idx, table in enumerate(table_finder.tables):
|
||||
bbox = table.bbox
|
||||
rows = table.extract() or []
|
||||
if not looks_like_table(rows):
|
||||
continue
|
||||
table_regions.append(
|
||||
LayoutRegion(
|
||||
x0=bbox[0],
|
||||
y0=bbox[1],
|
||||
x1=bbox[2],
|
||||
y1=bbox[3],
|
||||
kind="table",
|
||||
data={"rows": rows, "table_index": idx, "source": "pymupdf"},
|
||||
)
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
image_regions: list[LayoutRegion] = []
|
||||
for img_info in page.get_images(full=True):
|
||||
xref = img_info[0]
|
||||
try:
|
||||
rects = page.get_image_rects(xref)
|
||||
except Exception:
|
||||
rects = []
|
||||
if not rects:
|
||||
continue
|
||||
for rect_idx, rect in enumerate(rects):
|
||||
image_regions.append(
|
||||
LayoutRegion(
|
||||
x0=rect.x0,
|
||||
y0=rect.y0,
|
||||
x1=rect.x1,
|
||||
y1=rect.y1,
|
||||
kind="image",
|
||||
data={"xref": xref, "rect_index": rect_idx},
|
||||
)
|
||||
)
|
||||
image_regions = _filter_nested_image_regions(image_regions)
|
||||
|
||||
image_boxes = [(r.x0, r.y0, r.x1, r.y1) for r in image_regions]
|
||||
large_image_boxes = [
|
||||
box
|
||||
for box in image_boxes
|
||||
if (box[2] - box[0]) * (box[3] - box[1]) >= page_area * 0.12
|
||||
]
|
||||
table_boxes = [(r.x0, r.y0, r.x1, r.y1) for r in table_regions]
|
||||
|
||||
raw_lines = _collect_text_lines(page, body_size)
|
||||
filtered_lines = [
|
||||
ln
|
||||
for ln in raw_lines
|
||||
if not _line_on_image(ln, image_boxes, large_image_boxes)
|
||||
and not any(_rect_overlap((ln.x0, ln.y0, ln.x1, ln.y1), box) >= 0.55 for box in table_boxes)
|
||||
]
|
||||
paragraphs = _merge_lines_to_paragraphs(filtered_lines, page_width)
|
||||
|
||||
text_regions: list[LayoutRegion] = []
|
||||
for para in paragraphs:
|
||||
text_regions.append(
|
||||
LayoutRegion(
|
||||
x0=para.x0,
|
||||
y0=para.y0,
|
||||
x1=para.x1,
|
||||
y1=para.y1,
|
||||
kind="text",
|
||||
data={"text": para.text, "font_size": para.font_size},
|
||||
)
|
||||
)
|
||||
|
||||
regions = text_regions + table_regions + image_regions
|
||||
ordered_boxes = _sort_reading_order(
|
||||
[(r.x0, r.y0, r.x1, r.y1) for r in regions],
|
||||
page_width,
|
||||
)
|
||||
order = {box: i for i, box in enumerate(ordered_boxes)}
|
||||
regions.sort(key=lambda r: order.get((r.x0, r.y0, r.x1, r.y1), len(order)))
|
||||
|
||||
return PageLayout(
|
||||
page_index=page_index,
|
||||
page_width=page_width,
|
||||
page_height=page_height,
|
||||
body_font_size=body_size,
|
||||
regions=regions,
|
||||
)
|
||||
@@ -0,0 +1,446 @@
|
||||
"""Filter page headers, footers, page numbers and other margin noise from PDF blocks."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from collections import defaultdict
|
||||
|
||||
from rag_cut.models import Block, BlockType
|
||||
|
||||
HEADER_ZONE_RATIO = 0.12
|
||||
FOOTER_ZONE_RATIO = 0.10
|
||||
RUNNING_HEADER_MIN_PAGES = 3
|
||||
RUNNING_HEADER_PAGE_RATIO = 0.5
|
||||
MAX_RUNNING_HEADER_LEN = 48
|
||||
MIN_RUNNING_HEADER_LEN = 3
|
||||
# Drop decorative fragments (icons/logos) from layout detectors like MinerU.
|
||||
MIN_IMAGE_SIDE = 40.0
|
||||
MIN_IMAGE_AREA = 1600.0
|
||||
# Body paragraphs shorter than this can still exit a TOC zone when they
|
||||
# clearly are not directory entries (keeps in-section catalogs like 形態指標).
|
||||
TOC_BODY_EXIT_CHARS = 48
|
||||
TOC_PAGE_ENTRY_RATIO = 0.55
|
||||
TOC_PAGE_MIN_ENTRIES = 3
|
||||
|
||||
_PAGE_NUM_RE = re.compile(r"^\d{1,4}$")
|
||||
_NUMBERED_SECTION_RE = re.compile(
|
||||
r"^\s*(\d+(?:\.\d+)*)(?:\.|.)?\s*([A-Za-z0-9\u4e00-\u9fff][A-Za-z0-9\u4e00-\u9fff&/ \-_::]{2,})\s*$"
|
||||
)
|
||||
# Cover-page / front-matter directory headings only (exact-ish).
|
||||
_TOC_TITLE_RE = re.compile(
|
||||
r"^\s*(?:"
|
||||
r"contents|table\s+of\s+contents|toc|"
|
||||
r"\u76ee\u5f55|\u76ee\u9304|\u76ee\u6b21|" # 目录 / 目錄 / 目次
|
||||
r"\u7ae0\u8282\u76ee\u5f55|\u7ae0\u7bc0\u76ee\u9304|" # 章节目录 / 章節目錄
|
||||
r"list\s+of\s+(?:figures|tables|contents)"
|
||||
r")[\s.::·•…-]*$",
|
||||
re.I,
|
||||
)
|
||||
_DOT_LEADER_RE = re.compile(r"(?:\.{2,}|\u2026{2,}|\u00b7{2,}|\u2022{2,})")
|
||||
# Classic TOC line: title …… 12 / 1. Login ..... 3
|
||||
_TOC_ENTRY_LINE_RE = re.compile(
|
||||
r"^\s*.{1,120}?"
|
||||
r"(?:"
|
||||
r"(?:\.{2,}|\u2026{2,}|\u00b7{2,}|\s{2,})"
|
||||
r"\s*\d{1,4}"
|
||||
r"|"
|
||||
r"(?:\.{2,}|\u2026+)\s*\d{1,4}"
|
||||
r")"
|
||||
r"\s*$"
|
||||
)
|
||||
# Numbered entry with trailing page: "1.1 Account Status 12"
|
||||
_TOC_NUMBERED_PAGE_RE = re.compile(
|
||||
r"^\s*\d+(?:\.\d+)*(?:[\..]\s*|\s+)"
|
||||
r".{1,100}?"
|
||||
r"(?:\s{2,}|\s+)"
|
||||
r"\d{1,4}\s*$"
|
||||
)
|
||||
_MD_TOC_LINK_RE = re.compile(r"^\s*[-*+]\s+\[[^\]]+\]\([^)]+\)\s*$")
|
||||
|
||||
|
||||
def _looks_like_numbered_section(text: str) -> bool:
|
||||
match = _NUMBERED_SECTION_RE.match(text.strip())
|
||||
return bool(match and len(match.group(2).strip()) >= 3)
|
||||
|
||||
|
||||
def _norm_text(text: str) -> str:
|
||||
return " ".join((text or "").split())
|
||||
|
||||
|
||||
def is_toc_title_text(text: str) -> bool:
|
||||
"""True for standalone 目录 / Contents / TOC headings."""
|
||||
return bool(_TOC_TITLE_RE.match(_norm_text(text)))
|
||||
|
||||
|
||||
def is_toc_entry_line(text: str) -> bool:
|
||||
"""True for a single TOC row (leaders / trailing page number / md link)."""
|
||||
stripped = (text or "").strip()
|
||||
if not stripped or len(stripped) > 200:
|
||||
return False
|
||||
if is_toc_title_text(stripped):
|
||||
return False
|
||||
if _MD_TOC_LINK_RE.match(stripped):
|
||||
return True
|
||||
if _DOT_LEADER_RE.search(stripped) and re.search(r"\d\s*$", stripped):
|
||||
return True
|
||||
if _TOC_ENTRY_LINE_RE.match(stripped):
|
||||
return True
|
||||
if _TOC_NUMBERED_PAGE_RE.match(stripped):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def is_toc_noise_text(text: str) -> bool:
|
||||
"""True if the whole block text is a TOC title or TOC entries."""
|
||||
stripped = (text or "").strip()
|
||||
if not stripped:
|
||||
return False
|
||||
if is_toc_title_text(stripped):
|
||||
return True
|
||||
lines = [ln.strip() for ln in stripped.splitlines() if ln.strip()]
|
||||
if not lines:
|
||||
return False
|
||||
if len(lines) == 1:
|
||||
return is_toc_entry_line(lines[0])
|
||||
hits = sum(1 for ln in lines if is_toc_entry_line(ln) or is_toc_title_text(ln))
|
||||
return hits >= max(2, int(len(lines) * 0.6))
|
||||
|
||||
|
||||
def _is_toc_zone_exit_block(block: Block) -> bool:
|
||||
"""Substantial body / media ends a front-matter TOC stretch."""
|
||||
if block.type in {BlockType.IMAGE, BlockType.TABLE}:
|
||||
return True
|
||||
text = (block.text or block.markdown or "").strip()
|
||||
if not text or is_toc_noise_text(text):
|
||||
return False
|
||||
# Real section start right after TOC (no page-number trailer).
|
||||
if _looks_like_numbered_section(text) and not is_toc_entry_line(text):
|
||||
return True
|
||||
if len(text) >= TOC_BODY_EXIT_CHARS and not is_toc_entry_line(text):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _toc_heavy_pages(blocks: list[Block]) -> set[int]:
|
||||
"""Pages dominated by directory lines are dropped wholesale (text only)."""
|
||||
by_page: dict[int, list[Block]] = defaultdict(list)
|
||||
for block in blocks:
|
||||
if block.type in {BlockType.IMAGE, BlockType.TABLE}:
|
||||
continue
|
||||
page = block.meta.get("page")
|
||||
if page is None:
|
||||
continue
|
||||
try:
|
||||
page_key = int(page)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
by_page[page_key].append(block)
|
||||
|
||||
heavy: set[int] = set()
|
||||
for page_key, page_blocks in by_page.items():
|
||||
texts = [(b.text or b.markdown or "").strip() for b in page_blocks]
|
||||
texts = [t for t in texts if t]
|
||||
if not texts:
|
||||
continue
|
||||
entry_hits = sum(1 for t in texts if is_toc_noise_text(t))
|
||||
has_title = any(is_toc_title_text(t) for t in texts)
|
||||
if entry_hits >= TOC_PAGE_MIN_ENTRIES and entry_hits / len(texts) >= TOC_PAGE_ENTRY_RATIO:
|
||||
heavy.add(page_key)
|
||||
elif has_title and entry_hits >= 1 and entry_hits / len(texts) >= 0.4:
|
||||
heavy.add(page_key)
|
||||
return heavy
|
||||
|
||||
|
||||
def _toc_noise_bottoms(blocks: list[Block]) -> dict[int, float]:
|
||||
"""Lowest TOC text position per page, so real content below it survives."""
|
||||
bottoms: dict[int, float] = {}
|
||||
for block in blocks:
|
||||
if block.type in {BlockType.IMAGE, BlockType.TABLE}:
|
||||
continue
|
||||
text = (block.text or block.markdown or "").strip()
|
||||
if not is_toc_noise_text(text):
|
||||
continue
|
||||
page = block.meta.get("page")
|
||||
bbox = block.meta.get("bbox")
|
||||
if page is None or not isinstance(bbox, (list, tuple)) or len(bbox) < 4:
|
||||
continue
|
||||
try:
|
||||
page_key = int(page)
|
||||
bottom = float(bbox[3])
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
bottoms[page_key] = max(bottoms.get(page_key, 0.0), bottom)
|
||||
return bottoms
|
||||
|
||||
|
||||
def filter_toc_blocks(blocks: list[Block]) -> list[Block]:
|
||||
"""Drop document directories (目录 / Contents) — never chunk them.
|
||||
|
||||
Keeps in-section numbered catalogs without page leaders (e.g. 形態指標 list).
|
||||
"""
|
||||
if not blocks:
|
||||
return blocks
|
||||
|
||||
toc_pages = _toc_heavy_pages(blocks)
|
||||
toc_bottoms = _toc_noise_bottoms(blocks)
|
||||
filtered: list[Block] = []
|
||||
in_toc = False
|
||||
|
||||
for block in blocks:
|
||||
page = block.meta.get("page")
|
||||
try:
|
||||
page_key = int(page) if page is not None else None
|
||||
except (TypeError, ValueError):
|
||||
page_key = None
|
||||
|
||||
text = (block.text or block.markdown or "").strip()
|
||||
|
||||
if page_key is not None and page_key in toc_pages and block.type not in {
|
||||
BlockType.IMAGE,
|
||||
BlockType.TABLE,
|
||||
}:
|
||||
bbox = block.meta.get("bbox")
|
||||
toc_bottom = toc_bottoms.get(page_key)
|
||||
if (
|
||||
toc_bottom is None
|
||||
or not isinstance(bbox, (list, tuple))
|
||||
or len(bbox) < 4
|
||||
or float(bbox[1]) <= toc_bottom
|
||||
):
|
||||
continue
|
||||
|
||||
if is_toc_title_text(text):
|
||||
in_toc = True
|
||||
continue
|
||||
|
||||
if in_toc:
|
||||
if _is_toc_zone_exit_block(block):
|
||||
in_toc = False
|
||||
filtered.append(block)
|
||||
continue
|
||||
if is_toc_noise_text(text) or is_toc_entry_line(text):
|
||||
continue
|
||||
# Ambiguous short line right after TOC: treat as first real section.
|
||||
in_toc = False
|
||||
filtered.append(block)
|
||||
continue
|
||||
|
||||
if is_toc_noise_text(text):
|
||||
continue
|
||||
|
||||
filtered.append(block)
|
||||
|
||||
return filtered
|
||||
|
||||
|
||||
def _bbox4(block: Block) -> list[float] | None:
|
||||
bbox = block.meta.get("bbox")
|
||||
if not isinstance(bbox, (list, tuple)) or len(bbox) < 4:
|
||||
return None
|
||||
try:
|
||||
return [float(bbox[0]), float(bbox[1]), float(bbox[2]), float(bbox[3])]
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _bbox_area(bbox: list[float]) -> float:
|
||||
return max(bbox[2] - bbox[0], 0.0) * max(bbox[3] - bbox[1], 0.0)
|
||||
|
||||
|
||||
def _center_inside(inner: list[float], outer: list[float]) -> bool:
|
||||
cx = (inner[0] + inner[2]) / 2
|
||||
cy = (inner[1] + inner[3]) / 2
|
||||
return outer[0] <= cx <= outer[2] and outer[1] <= cy <= outer[3]
|
||||
|
||||
|
||||
def _overlap_ratio(inner: list[float], outer: list[float]) -> float:
|
||||
x0 = max(inner[0], outer[0])
|
||||
y0 = max(inner[1], outer[1])
|
||||
x1 = min(inner[2], outer[2])
|
||||
y1 = min(inner[3], outer[3])
|
||||
if x1 <= x0 or y1 <= y0:
|
||||
return 0.0
|
||||
inter = (x1 - x0) * (y1 - y0)
|
||||
return inter / max(_bbox_area(inner), 1e-6)
|
||||
|
||||
|
||||
def is_tiny_image_block(block: Block) -> bool:
|
||||
"""True for tiny image fragments that are usually logos/icons, not content figures."""
|
||||
if block.type != BlockType.IMAGE:
|
||||
return False
|
||||
bbox = _bbox4(block)
|
||||
if not bbox:
|
||||
return False
|
||||
width = bbox[2] - bbox[0]
|
||||
height = bbox[3] - bbox[1]
|
||||
if width <= 0 or height <= 0:
|
||||
return True
|
||||
if width < MIN_IMAGE_SIDE and height < MIN_IMAGE_SIDE:
|
||||
return True
|
||||
return width * height < MIN_IMAGE_AREA
|
||||
|
||||
|
||||
def filter_nested_image_blocks(blocks: list[Block]) -> list[Block]:
|
||||
"""Drop image fragments whose center lies inside a larger same-page image."""
|
||||
image_boxes: list[tuple[int, int, list[float], float]] = []
|
||||
for idx, block in enumerate(blocks):
|
||||
if block.type != BlockType.IMAGE:
|
||||
continue
|
||||
bbox = _bbox4(block)
|
||||
page = block.meta.get("page")
|
||||
if bbox is None or page is None:
|
||||
continue
|
||||
try:
|
||||
page_key = int(page)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
image_boxes.append((idx, page_key, bbox, _bbox_area(bbox)))
|
||||
|
||||
drop: set[int] = set()
|
||||
for idx, page_key, bbox, area in image_boxes:
|
||||
for other_idx, other_page, other_bbox, other_area in image_boxes:
|
||||
if idx == other_idx or page_key != other_page or other_area <= area * 1.5:
|
||||
continue
|
||||
if _center_inside(bbox, other_bbox) and _overlap_ratio(bbox, other_bbox) >= 0.85:
|
||||
drop.add(idx)
|
||||
break
|
||||
|
||||
if not drop:
|
||||
return blocks
|
||||
return [block for idx, block in enumerate(blocks) if idx not in drop]
|
||||
|
||||
|
||||
def _infer_page_heights(blocks: list[Block]) -> dict[int, float]:
|
||||
heights: dict[int, float] = {}
|
||||
for block in blocks:
|
||||
page = block.meta.get("page")
|
||||
if page is None:
|
||||
continue
|
||||
try:
|
||||
page_key = int(page)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
explicit = block.meta.get("page_height")
|
||||
if explicit:
|
||||
heights[page_key] = max(heights.get(page_key, 0), float(explicit))
|
||||
bbox = block.meta.get("bbox")
|
||||
if bbox and len(bbox) >= 4:
|
||||
heights[page_key] = max(heights.get(page_key, 0), float(bbox[3]))
|
||||
return {page: max(height, 1.0) for page, height in heights.items()}
|
||||
|
||||
|
||||
def _page_height(block: Block, heights: dict[int, float]) -> float | None:
|
||||
page = block.meta.get("page")
|
||||
if page is None:
|
||||
explicit = block.meta.get("page_height")
|
||||
return float(explicit) if explicit else None
|
||||
try:
|
||||
page_key = int(page)
|
||||
except (TypeError, ValueError):
|
||||
return block.meta.get("page_height")
|
||||
return heights.get(page_key) or block.meta.get("page_height")
|
||||
|
||||
|
||||
def _in_margin_zone(bbox: list[float], page_height: float, *, header: bool) -> bool:
|
||||
y_mid = (float(bbox[1]) + float(bbox[3])) / 2
|
||||
if header:
|
||||
return y_mid < page_height * HEADER_ZONE_RATIO
|
||||
return y_mid > page_height * (1 - FOOTER_ZONE_RATIO)
|
||||
|
||||
|
||||
def is_margin_noise_block(block: Block, page_height: float | None) -> bool:
|
||||
"""Heuristic margin noise filter for blocks missing explicit MinerU region types."""
|
||||
if block.type == BlockType.IMAGE:
|
||||
return is_tiny_image_block(block)
|
||||
if block.type == BlockType.TABLE:
|
||||
return False
|
||||
|
||||
text = (block.text or "").strip()
|
||||
if not text:
|
||||
return False
|
||||
|
||||
bbox = block.meta.get("bbox")
|
||||
if not page_height or not bbox or len(bbox) < 4:
|
||||
return block.type == BlockType.PARAGRAPH and bool(_PAGE_NUM_RE.match(text))
|
||||
|
||||
in_header = _in_margin_zone(bbox, page_height, header=True)
|
||||
in_footer = _in_margin_zone(bbox, page_height, header=False)
|
||||
if not in_header and not in_footer:
|
||||
return False
|
||||
|
||||
if _PAGE_NUM_RE.match(text):
|
||||
return True
|
||||
|
||||
if block.type != BlockType.PARAGRAPH:
|
||||
return False
|
||||
|
||||
# Numbered section titles often sit at the top of a continued page — keep them.
|
||||
if _looks_like_numbered_section(text):
|
||||
return False
|
||||
|
||||
return len(text) <= MAX_RUNNING_HEADER_LEN
|
||||
|
||||
|
||||
def detect_running_header_texts(blocks: list[Block]) -> set[str]:
|
||||
"""Texts that repeat across many pages are likely running headers/footers.
|
||||
|
||||
Only paragraphs in the header/footer margin bands are considered, so
|
||||
real section headings that happen to repeat are not wiped document-wide.
|
||||
"""
|
||||
pages_by_text: dict[str, set[int]] = defaultdict(set)
|
||||
all_pages: set[int] = set()
|
||||
heights = _infer_page_heights(blocks)
|
||||
|
||||
for block in blocks:
|
||||
if block.type != BlockType.PARAGRAPH:
|
||||
continue
|
||||
text = (block.text or "").strip()
|
||||
if not (MIN_RUNNING_HEADER_LEN <= len(text) <= MAX_RUNNING_HEADER_LEN):
|
||||
continue
|
||||
if _looks_like_numbered_section(text):
|
||||
continue
|
||||
page = block.meta.get("page")
|
||||
if page is None:
|
||||
continue
|
||||
try:
|
||||
page_key = int(page)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
page_height = _page_height(block, heights)
|
||||
bbox = _bbox4(block)
|
||||
if page_height and bbox:
|
||||
in_margin = _in_margin_zone(bbox, float(page_height), header=True) or _in_margin_zone(
|
||||
bbox, float(page_height), header=False
|
||||
)
|
||||
if not in_margin:
|
||||
continue
|
||||
all_pages.add(page_key)
|
||||
pages_by_text[text].add(page_key)
|
||||
|
||||
if len(all_pages) < RUNNING_HEADER_MIN_PAGES:
|
||||
return set()
|
||||
|
||||
threshold = max(RUNNING_HEADER_MIN_PAGES, int(len(all_pages) * RUNNING_HEADER_PAGE_RATIO))
|
||||
return {text for text, pages in pages_by_text.items() if len(pages) >= threshold}
|
||||
|
||||
|
||||
def filter_noise_blocks(blocks: list[Block]) -> list[Block]:
|
||||
"""Remove headers, footers, page numbers, TOC and other repeated margin noise."""
|
||||
if not blocks:
|
||||
return blocks
|
||||
|
||||
heights = _infer_page_heights(blocks)
|
||||
running_headers = detect_running_header_texts(blocks)
|
||||
filtered: list[Block] = []
|
||||
|
||||
for block in blocks:
|
||||
text = (block.text or "").strip()
|
||||
# Never wipe real section headings via running-header equality.
|
||||
if text and text in running_headers and block.type == BlockType.PARAGRAPH:
|
||||
continue
|
||||
page_height = _page_height(block, heights)
|
||||
if is_margin_noise_block(block, float(page_height) if page_height else None):
|
||||
continue
|
||||
filtered.append(block)
|
||||
|
||||
return filter_nested_image_blocks(filter_toc_blocks(filtered))
|
||||
@@ -0,0 +1,57 @@
|
||||
"""Optional OCR for cropped image/table regions."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
from pathlib import Path
|
||||
|
||||
import fitz
|
||||
|
||||
|
||||
def ocr_pixmap(pix: fitz.Pixmap) -> str:
|
||||
"""Run OCR on a pixmap; returns empty string when OCR is unavailable."""
|
||||
try:
|
||||
import pytesseract
|
||||
from PIL import Image
|
||||
except ImportError:
|
||||
return ""
|
||||
|
||||
try:
|
||||
image = Image.open(io.BytesIO(pix.tobytes("png")))
|
||||
return _ocr_pil(image)
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def ocr_image_file(path: Path) -> str:
|
||||
try:
|
||||
import pytesseract # noqa: F401
|
||||
from PIL import Image
|
||||
except ImportError:
|
||||
return ""
|
||||
|
||||
try:
|
||||
return _ocr_pil(Image.open(path))
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def _ocr_pil(image) -> str:
|
||||
import pytesseract
|
||||
|
||||
for lang in ("chi_tra+eng", "chi_sim+eng", "eng"):
|
||||
try:
|
||||
text = pytesseract.image_to_string(image, lang=lang)
|
||||
cleaned = " ".join(text.split())
|
||||
if cleaned:
|
||||
return cleaned
|
||||
except Exception:
|
||||
continue
|
||||
return ""
|
||||
|
||||
|
||||
def describe_visual(ocr_text: str, kind: str, label: str) -> str:
|
||||
"""Lightweight image/table description without an external vision model."""
|
||||
if ocr_text:
|
||||
return f"{kind}:{ocr_text[:300]}"
|
||||
return f"{kind}:{label}"
|
||||
@@ -0,0 +1,213 @@
|
||||
"""End-to-end PyMuPDF PDF pipeline matching the architecture diagram."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
import pdfplumber
|
||||
|
||||
from rag_cut.models import Block
|
||||
from rag_cut.parsers.pdf.layout import (
|
||||
LayoutRegion,
|
||||
PageLayout,
|
||||
_rect_overlap,
|
||||
_sort_reading_order,
|
||||
analyze_page_layout,
|
||||
render_page,
|
||||
)
|
||||
from rag_cut.parsers.pdf.standardize import open_document
|
||||
from rag_cut.parsers.pdf.tables import rows_to_markdown
|
||||
from rag_cut.parsers.pdf.text_extract import extract_text_blocks
|
||||
from rag_cut.parsers.pdf.visual_extract import extract_visual_blocks
|
||||
|
||||
|
||||
@dataclass
|
||||
class _PageItem:
|
||||
y0: float
|
||||
x0: float
|
||||
kind: str
|
||||
block: Block
|
||||
|
||||
|
||||
def _pdfplumber_tables(path: Path) -> dict[int, list[dict]]:
|
||||
"""Extract tables per page with bbox from pdfplumber."""
|
||||
tables_by_page: dict[int, list[dict]] = {}
|
||||
try:
|
||||
with pdfplumber.open(path) as pdf:
|
||||
for i, page in enumerate(pdf.pages):
|
||||
found = []
|
||||
try:
|
||||
for idx, table in enumerate(page.find_tables()):
|
||||
rows = table.extract() or []
|
||||
if not rows:
|
||||
continue
|
||||
bbox = table.bbox
|
||||
found.append(
|
||||
{
|
||||
"rows": rows,
|
||||
"table_index": idx,
|
||||
"source": "pdfplumber",
|
||||
"bbox": list(bbox) if bbox else None,
|
||||
}
|
||||
)
|
||||
except Exception:
|
||||
rows_list = page.extract_tables() or []
|
||||
for idx, rows in enumerate(rows_list):
|
||||
found.append({"rows": rows, "table_index": idx, "source": "pdfplumber", "bbox": None})
|
||||
if found:
|
||||
tables_by_page[i] = found
|
||||
except Exception:
|
||||
pass
|
||||
return tables_by_page
|
||||
|
||||
|
||||
def _inject_pdfplumber_tables(layout: PageLayout, plumber_tables: list[dict]) -> PageLayout:
|
||||
if not plumber_tables:
|
||||
return layout
|
||||
if any(r.kind == "table" for r in layout.regions):
|
||||
return layout
|
||||
|
||||
for item in plumber_tables:
|
||||
rows = item.get("rows") or []
|
||||
md = rows_to_markdown(rows)
|
||||
if not md:
|
||||
continue
|
||||
bbox = item.get("bbox")
|
||||
if bbox and len(bbox) == 4:
|
||||
x0, y0, x1, y1 = bbox
|
||||
else:
|
||||
idx = int(item.get("table_index", 0))
|
||||
x0, y0 = 0, layout.page_height * (idx + 1) / (len(plumber_tables) + 1)
|
||||
x1, y1 = layout.page_width, layout.page_height * (idx + 2) / (len(plumber_tables) + 1)
|
||||
layout.regions.append(
|
||||
LayoutRegion(
|
||||
x0=x0,
|
||||
y0=y0,
|
||||
x1=x1,
|
||||
y1=y1,
|
||||
kind="table",
|
||||
data={
|
||||
"rows": rows,
|
||||
"table_index": item.get("table_index", 0),
|
||||
"source": item.get("source", "pdfplumber"),
|
||||
},
|
||||
)
|
||||
)
|
||||
ordered_boxes = _sort_reading_order(
|
||||
[(r.x0, r.y0, r.x1, r.y1) for r in layout.regions],
|
||||
layout.page_width,
|
||||
)
|
||||
order = {box: i for i, box in enumerate(ordered_boxes)}
|
||||
layout.regions.sort(key=lambda r: order.get((r.x0, r.y0, r.x1, r.y1), len(order)))
|
||||
return layout
|
||||
|
||||
|
||||
def _bbox_key(bbox: list[float] | tuple[float, ...]) -> tuple[float, ...]:
|
||||
return tuple(round(v, 1) for v in bbox)
|
||||
|
||||
|
||||
def _region_box(region: LayoutRegion) -> tuple[float, float, float, float]:
|
||||
return (region.x0, region.y0, region.x1, region.y1)
|
||||
|
||||
|
||||
def _match_block_to_region(
|
||||
region: LayoutRegion,
|
||||
candidates: list[Block],
|
||||
used: set[int],
|
||||
) -> Block | None:
|
||||
"""Match a layout region to a parsed block via exact bbox key or overlap."""
|
||||
region_box = _region_box(region)
|
||||
key = _bbox_key(region_box)
|
||||
|
||||
for block in candidates:
|
||||
if id(block) in used:
|
||||
continue
|
||||
bbox = block.meta.get("bbox")
|
||||
if bbox and _bbox_key(bbox) == key:
|
||||
return block
|
||||
|
||||
best: Block | None = None
|
||||
best_overlap = 0.35
|
||||
for block in candidates:
|
||||
if id(block) in used:
|
||||
continue
|
||||
bbox = block.meta.get("bbox")
|
||||
if not bbox:
|
||||
continue
|
||||
overlap = _rect_overlap(region_box, tuple(bbox))
|
||||
if overlap > best_overlap:
|
||||
best_overlap = overlap
|
||||
best = block
|
||||
return best
|
||||
|
||||
|
||||
def _merge_page_blocks(
|
||||
text_blocks: list[Block],
|
||||
visual_blocks: list[Block],
|
||||
layout: PageLayout,
|
||||
) -> list[Block]:
|
||||
"""Merge text and visual branches in page reading order."""
|
||||
used: set[int] = set()
|
||||
merged: list[_PageItem] = []
|
||||
|
||||
for region in layout.regions:
|
||||
if region.kind == "text":
|
||||
candidates = text_blocks
|
||||
elif region.kind in {"table", "image"}:
|
||||
candidates = visual_blocks
|
||||
else:
|
||||
continue
|
||||
|
||||
block = _match_block_to_region(region, candidates, used)
|
||||
if block is not None:
|
||||
merged.append(_PageItem(y0=region.y0, x0=region.x0, kind=region.kind, block=block))
|
||||
used.add(id(block))
|
||||
|
||||
leftovers: list[_PageItem] = []
|
||||
for block in text_blocks + visual_blocks:
|
||||
if id(block) not in used:
|
||||
bbox = block.meta.get("bbox", [0, 0, 0, 0])
|
||||
leftovers.append(_PageItem(y0=bbox[1], x0=bbox[0], kind=block.type.value, block=block))
|
||||
|
||||
merged.extend(sorted(leftovers, key=lambda it: (it.y0, it.x0)))
|
||||
return [it.block for it in merged]
|
||||
|
||||
|
||||
def parse_pdf(path: Path, assets_dir: Path) -> list[Block]:
|
||||
"""
|
||||
PDF pipeline:
|
||||
1. 文档标准化
|
||||
2. 页面渲染 + 版面分析
|
||||
3. 文本块提取 ∥ 图片/表格区域裁剪 + OCR
|
||||
4. 章节级语义切片(合并两路结果,标注章节上下文)
|
||||
"""
|
||||
assets_dir.mkdir(parents=True, exist_ok=True)
|
||||
renders_dir = assets_dir / "pages"
|
||||
renders_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
doc = open_document(path)
|
||||
plumber_tables = _pdfplumber_tables(path)
|
||||
all_blocks: list[Block] = []
|
||||
chapter_title: str | None = None
|
||||
img_counter = 0
|
||||
|
||||
try:
|
||||
for page_index, page in enumerate(doc):
|
||||
layout = analyze_page_layout(page, page_index)
|
||||
layout = _inject_pdfplumber_tables(layout, plumber_tables.get(page_index, []))
|
||||
|
||||
page_render = render_page(page)
|
||||
render_path = renders_dir / f"page{page_index + 1}.png"
|
||||
page_render.save(str(render_path))
|
||||
|
||||
text_blocks, chapter_title = extract_text_blocks(layout, chapter_title)
|
||||
visual_blocks, img_counter = extract_visual_blocks(
|
||||
page, layout, doc, assets_dir, chapter_title, img_counter
|
||||
)
|
||||
page_blocks = _merge_page_blocks(text_blocks, visual_blocks, layout)
|
||||
all_blocks.extend(page_blocks)
|
||||
finally:
|
||||
doc.close()
|
||||
|
||||
return all_blocks
|
||||
@@ -0,0 +1,16 @@
|
||||
"""Document standardization for PDF input."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import fitz
|
||||
|
||||
|
||||
def open_document(path: Path) -> fitz.Document:
|
||||
"""Open and normalize a PDF for downstream layout processing."""
|
||||
doc = fitz.open(path)
|
||||
if doc.is_encrypted and not doc.authenticate(""):
|
||||
doc.close()
|
||||
raise ValueError(f"Encrypted PDF cannot be opened: {path.name}")
|
||||
return doc
|
||||
@@ -0,0 +1,104 @@
|
||||
"""Build rich table blocks from PDF layout regions."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import fitz
|
||||
|
||||
from rag_cut.models import Block, BlockType
|
||||
from rag_cut.parsers.pdf.layout import LayoutRegion, PageLayout
|
||||
from rag_cut.parsers.pdf.ocr import describe_visual, ocr_pixmap
|
||||
from rag_cut.parsers.pdf.tables import (
|
||||
detect_header_row_count,
|
||||
extract_table_keywords,
|
||||
header_signature,
|
||||
normalize_rows,
|
||||
rows_to_markdown,
|
||||
split_body_and_footnotes,
|
||||
)
|
||||
|
||||
|
||||
def _crop_region(page: fitz.Page, region: LayoutRegion, zoom: float = 2.0) -> fitz.Pixmap:
|
||||
clip = fitz.Rect(region.x0, region.y0, region.x1, region.y1)
|
||||
return page.get_pixmap(matrix=fitz.Matrix(zoom, zoom), clip=clip, alpha=False)
|
||||
|
||||
|
||||
def _save_pixmap(pix: fitz.Pixmap, path: Path) -> None:
|
||||
if pix.n - pix.alpha > 3:
|
||||
pix = fitz.Pixmap(fitz.csRGB, pix)
|
||||
pix.save(str(path))
|
||||
|
||||
|
||||
def build_table_block(
|
||||
page: fitz.Page,
|
||||
region: LayoutRegion,
|
||||
layout: PageLayout,
|
||||
assets_dir: Path,
|
||||
chapter_title: str | None,
|
||||
) -> Block | None:
|
||||
"""Extract a table block with rows, screenshot, OCR and structural metadata."""
|
||||
raw_rows = region.data.get("rows") or []
|
||||
normalized = normalize_rows(raw_rows)
|
||||
if not normalized:
|
||||
return None
|
||||
|
||||
header_rows = detect_header_row_count(normalized)
|
||||
body_rows, footnote_rows, footnotes = split_body_and_footnotes(normalized, header_rows)
|
||||
data_rows = normalized[:header_rows] + body_rows
|
||||
if not data_rows:
|
||||
return None
|
||||
|
||||
page_no = layout.page_index + 1
|
||||
table_index = int(region.data.get("table_index", 0)) + 1
|
||||
crop_id = f"page{page_no}_table{table_index}.png"
|
||||
crops_dir = assets_dir / "crops"
|
||||
crops_dir.mkdir(parents=True, exist_ok=True)
|
||||
crop_path = crops_dir / crop_id
|
||||
|
||||
ocr_text = ""
|
||||
try:
|
||||
crop_pix = _crop_region(page, region)
|
||||
_save_pixmap(crop_pix, crop_path)
|
||||
ocr_text = ocr_pixmap(crop_pix)
|
||||
except Exception:
|
||||
crop_path_str = ""
|
||||
else:
|
||||
crop_path_str = str(crop_path)
|
||||
|
||||
sig = header_signature(normalized, header_rows)
|
||||
md = rows_to_markdown(normalized, header_rows=header_rows, include_footnotes=footnotes)
|
||||
if not md:
|
||||
return None
|
||||
|
||||
header_text = " ".join(" ".join(r for r in row if r) for row in normalized[:header_rows])
|
||||
keywords = extract_table_keywords(header_text, md, footnotes, ocr_text)
|
||||
|
||||
meta: dict = {
|
||||
"page": page_no,
|
||||
"pages": [page_no],
|
||||
"bbox": [region.x0, region.y0, region.x1, region.y1],
|
||||
"bboxes": [{"page": page_no, "bbox": [region.x0, region.y0, region.x1, region.y1]}],
|
||||
"table_source": region.data.get("source", "pymupdf"),
|
||||
"crop_path": crop_path_str,
|
||||
"rows": normalized,
|
||||
"row_count": len(normalized),
|
||||
"col_count": max((len(r) for r in normalized), default=0),
|
||||
"header_rows": header_rows,
|
||||
"header_signature": [list(r) for r in sig],
|
||||
"footnotes": footnotes,
|
||||
"keywords": keywords,
|
||||
"cross_page": False,
|
||||
}
|
||||
if chapter_title:
|
||||
meta["chapter"] = chapter_title
|
||||
|
||||
return Block(
|
||||
type=BlockType.TABLE,
|
||||
markdown=md,
|
||||
ocr_text=ocr_text,
|
||||
text=describe_visual(ocr_text, "表格", crop_id) if ocr_text else "",
|
||||
image_id=crop_id,
|
||||
image_path=crop_path_str or None,
|
||||
meta=meta,
|
||||
)
|
||||
@@ -0,0 +1,177 @@
|
||||
"""Cross-page table detection and merging."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from rag_cut.models import Block, BlockType
|
||||
from rag_cut.parsers.pdf.tables import (
|
||||
detect_header_row_count,
|
||||
header_signature,
|
||||
normalize_rows,
|
||||
rows_to_markdown,
|
||||
split_body_and_footnotes,
|
||||
)
|
||||
|
||||
|
||||
def _page(block: Block) -> int | None:
|
||||
page = block.meta.get("page")
|
||||
return int(page) if page is not None else None
|
||||
|
||||
|
||||
def _header_rows(block: Block) -> int:
|
||||
rows = block.meta.get("rows") or []
|
||||
return int(block.meta.get("header_rows") or detect_header_row_count(normalize_rows(rows)) or 1)
|
||||
|
||||
|
||||
def _header_sig(block: Block) -> tuple[tuple[str, ...], ...]:
|
||||
if block.meta.get("header_signature"):
|
||||
raw = block.meta["header_signature"]
|
||||
return tuple(tuple(r) for r in raw)
|
||||
rows = normalize_rows(block.meta.get("rows") or [])
|
||||
return header_signature(rows, _header_rows(block))
|
||||
|
||||
|
||||
def _column_count(block: Block) -> int:
|
||||
rows = normalize_rows(block.meta.get("rows") or [])
|
||||
return max((len(r) for r in rows), default=0)
|
||||
|
||||
|
||||
def _similar_columns(a: Block, b: Block) -> bool:
|
||||
ca, cb = _column_count(a), _column_count(b)
|
||||
if ca == 0 or cb == 0:
|
||||
return False
|
||||
return ca == cb or abs(ca - cb) <= 1
|
||||
|
||||
|
||||
def _repeated_header(rows: list[list[str]], sig: tuple[tuple[str, ...], ...]) -> int:
|
||||
"""Return number of leading rows in `rows` that repeat the header signature."""
|
||||
if not sig:
|
||||
return 0
|
||||
n = len(sig)
|
||||
if len(rows) < n:
|
||||
return 0
|
||||
if header_signature(rows, n) == sig:
|
||||
return n
|
||||
if n == 1 and rows and tuple(rows[0]) == sig[0]:
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
def _can_merge_continuation(prev: Block, nxt: Block) -> bool:
|
||||
if prev.type != BlockType.TABLE or nxt.type != BlockType.TABLE:
|
||||
return False
|
||||
|
||||
prev_page = _page(prev)
|
||||
nxt_page = _page(nxt)
|
||||
if prev_page is None or nxt_page is None or nxt_page != prev_page + 1:
|
||||
return False
|
||||
if not _similar_columns(prev, nxt):
|
||||
return False
|
||||
|
||||
sig = _header_sig(prev)
|
||||
if not sig:
|
||||
return False
|
||||
|
||||
rows_b = normalize_rows(nxt.meta.get("rows") or [])
|
||||
if not rows_b:
|
||||
return False
|
||||
|
||||
if _repeated_header(rows_b, sig) > 0:
|
||||
return True
|
||||
|
||||
# Continuation without repeated header: similar width and no title on next table
|
||||
if nxt.meta.get("table_title"):
|
||||
return False
|
||||
|
||||
prev_bbox = prev.meta.get("bbox") or []
|
||||
nxt_bbox = nxt.meta.get("bbox") or []
|
||||
if len(prev_bbox) == 4 and len(nxt_bbox) == 4:
|
||||
prev_width = prev_bbox[2] - prev_bbox[0]
|
||||
nxt_width = nxt_bbox[2] - nxt_bbox[0]
|
||||
if prev_width > 0 and abs(prev_width - nxt_width) / prev_width <= 0.15:
|
||||
return True
|
||||
|
||||
return _column_count(prev) == _column_count(nxt)
|
||||
|
||||
|
||||
def _merge_two_tables(prev: Block, nxt: Block) -> Block:
|
||||
rows_a = normalize_rows(prev.meta.get("rows") or [])
|
||||
rows_b = normalize_rows(nxt.meta.get("rows") or [])
|
||||
header_rows = _header_rows(prev)
|
||||
sig = _header_sig(prev)
|
||||
|
||||
skip = _repeated_header(rows_b, sig)
|
||||
merged_rows = rows_a + rows_b[skip:]
|
||||
|
||||
body_rows, _, foot_a = split_body_and_footnotes(rows_a, header_rows)
|
||||
_, _, foot_b = split_body_and_footnotes(rows_b, skip or header_rows)
|
||||
footnotes = " ".join(x for x in (prev.meta.get("footnotes") or foot_a, foot_b) if x).strip()
|
||||
|
||||
pages = sorted(set((prev.meta.get("pages") or [_page(prev)]) + [_page(nxt)]))
|
||||
pages = [p for p in pages if p is not None]
|
||||
bboxes = list(prev.meta.get("bboxes") or [])
|
||||
if prev.meta.get("bbox"):
|
||||
bboxes.append({"page": _page(prev), "bbox": prev.meta["bbox"]})
|
||||
if nxt.meta.get("bbox"):
|
||||
bboxes.append({"page": _page(nxt), "bbox": nxt.meta["bbox"]})
|
||||
|
||||
prev_bbox = prev.meta.get("bbox") or [0, 0, 0, 0]
|
||||
nxt_bbox = nxt.meta.get("bbox") or prev_bbox
|
||||
merged_bbox = [
|
||||
min(prev_bbox[0], nxt_bbox[0]),
|
||||
min(prev_bbox[1], nxt_bbox[1]),
|
||||
max(prev_bbox[2], nxt_bbox[2]),
|
||||
max(prev_bbox[3], nxt_bbox[3]),
|
||||
]
|
||||
|
||||
md = rows_to_markdown(merged_rows, header_rows=header_rows, include_footnotes=footnotes)
|
||||
meta = dict(prev.meta)
|
||||
meta.update(
|
||||
{
|
||||
"rows": merged_rows,
|
||||
"row_count": len(merged_rows),
|
||||
"col_count": max((len(r) for r in merged_rows), default=0),
|
||||
"header_rows": header_rows,
|
||||
"header_signature": [list(r) for r in sig],
|
||||
"footnotes": footnotes,
|
||||
"pages": pages,
|
||||
"page": pages[0] if pages else prev.meta.get("page"),
|
||||
"bbox": merged_bbox,
|
||||
"bboxes": bboxes,
|
||||
"cross_page": len(pages) > 1,
|
||||
"merged_table_count": int(prev.meta.get("merged_table_count") or 1) + 1,
|
||||
}
|
||||
)
|
||||
if nxt.meta.get("following_text") and not meta.get("following_text"):
|
||||
meta["following_text"] = nxt.meta.get("following_text")
|
||||
|
||||
return prev.model_copy(update={"markdown": md, "meta": meta})
|
||||
|
||||
|
||||
def merge_cross_page_tables(blocks: list[Block]) -> list[Block]:
|
||||
"""Merge consecutive cross-page table blocks that share headers/structure."""
|
||||
if not blocks:
|
||||
return []
|
||||
|
||||
result: list[Block] = []
|
||||
i = 0
|
||||
while i < len(blocks):
|
||||
current = blocks[i]
|
||||
if current.type != BlockType.TABLE:
|
||||
result.append(current)
|
||||
i += 1
|
||||
continue
|
||||
|
||||
merged = current
|
||||
j = i + 1
|
||||
while j < len(blocks):
|
||||
nxt = blocks[j]
|
||||
if nxt.type == BlockType.TABLE and _can_merge_continuation(merged, nxt):
|
||||
merged = _merge_two_tables(merged, nxt)
|
||||
j += 1
|
||||
continue
|
||||
break
|
||||
|
||||
result.append(merged)
|
||||
i = j
|
||||
|
||||
return result
|
||||
@@ -0,0 +1,313 @@
|
||||
"""Table helpers: normalization, header detection, Markdown export, cross-page merge."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Any
|
||||
|
||||
FOOTNOTE_ROW_RE = re.compile(r"^[\s*※①②③④⑤]*(?:注[::]?|备注[::]?|说明[::]?|Note[::]?)", re.I)
|
||||
DATA_FIRST_CELL_RE = re.compile(r"^[a-z_][a-z0-9_.-]*$", re.I)
|
||||
TABLE_TITLE_RE = re.compile(
|
||||
r"^(?:表\s*\d+[::.]?|Table\s*\d+[::.]?|图\s*\d+[::.]?)?\s*.{2,80}$",
|
||||
re.I,
|
||||
)
|
||||
KEYWORD_TERMS = (
|
||||
"字段", "参数", "必填", "选填", "状态", "类型", "说明", "含义", "取值",
|
||||
"field", "parameter", "required", "optional", "status", "description",
|
||||
)
|
||||
|
||||
|
||||
def normalize_cell(value: str | None) -> str:
|
||||
"""Merge in-cell line breaks; escape pipe chars for Markdown tables."""
|
||||
if not value:
|
||||
return ""
|
||||
text = str(value).replace("\r\n", "\n").replace("\r", "\n")
|
||||
parts = [p.strip() for p in text.split("\n") if p.strip()]
|
||||
merged = " ".join(parts) if parts else ""
|
||||
return merged.replace("|", "\\|")
|
||||
|
||||
|
||||
def normalize_rows(rows: list[list[str | None]]) -> list[list[str]]:
|
||||
"""Pad columns and normalize every cell without losing row alignment."""
|
||||
if not rows:
|
||||
return []
|
||||
cleaned = [[normalize_cell(c) for c in row] for row in rows]
|
||||
col_count = max((len(r) for r in cleaned), default=0)
|
||||
return [row + [""] * (col_count - len(row)) for row in cleaned]
|
||||
|
||||
|
||||
def looks_like_table(rows: list[list[str | None]]) -> bool:
|
||||
"""Return True only for rows that have a real table-like grid."""
|
||||
if len(rows) < 2:
|
||||
return False
|
||||
|
||||
cleaned = normalize_rows(rows)
|
||||
col_count = max((len(r) for r in cleaned), default=0)
|
||||
if col_count < 2:
|
||||
return False
|
||||
|
||||
non_empty_cells = sum(1 for row in cleaned for cell in row if cell)
|
||||
rows_with_two_cells = sum(1 for row in cleaned if sum(1 for cell in row if cell) >= 2)
|
||||
return non_empty_cells >= 4 and rows_with_two_cells >= 2
|
||||
|
||||
|
||||
DESCRIPTION_HINTS = ("必填", "选填", "格式要求", "required", "optional", "格式", "用于标识")
|
||||
QA_HEADER_TERMS = ("query", "question", "用户输入", "reference_output", "answer", "标准答案", "session")
|
||||
|
||||
|
||||
def _row_fill(row: list[str]) -> int:
|
||||
return sum(1 for c in row if c)
|
||||
|
||||
|
||||
def looks_like_column_header_row(row: list[str]) -> bool:
|
||||
"""True when a row looks like short spreadsheet column names."""
|
||||
filled = [c.strip() for c in row if c and c.strip()]
|
||||
if len(filled) < 2:
|
||||
return False
|
||||
if any(len(c) > 40 for c in filled):
|
||||
return False
|
||||
identifier_like = sum(
|
||||
1
|
||||
for c in filled
|
||||
if DATA_FIRST_CELL_RE.match(c) or re.match(r"^[a-z][a-z0-9_]*$", c, re.I)
|
||||
)
|
||||
return identifier_like >= max(2, (len(filled) + 1) // 2)
|
||||
|
||||
|
||||
def looks_like_description_row(row: list[str]) -> bool:
|
||||
"""True when a row is a template field-description line (not data/header)."""
|
||||
filled = [c.strip() for c in row if c and c.strip()]
|
||||
if not filled:
|
||||
return False
|
||||
if max(len(c) for c in filled) >= 48:
|
||||
return True
|
||||
return sum(1 for c in filled if any(h in c for h in DESCRIPTION_HINTS)) >= 2
|
||||
|
||||
|
||||
def detect_header_row_count(rows: list[list[str]]) -> int:
|
||||
"""Detect 1-2 header rows from content patterns."""
|
||||
if len(rows) < 2:
|
||||
return 1 if rows else 0
|
||||
|
||||
first_fill = _row_fill(rows[0])
|
||||
second_fill = _row_fill(rows[1]) if len(rows) > 1 else 0
|
||||
if first_fill < 2:
|
||||
return 0
|
||||
|
||||
if len(rows) > 2 and second_fill >= 2:
|
||||
first_short = all(len(c) <= 24 for c in rows[0] if c)
|
||||
second_short = all(len(c) <= 24 for c in rows[1] if c)
|
||||
second_is_data = bool(rows[1][0]) and DATA_FIRST_CELL_RE.match(rows[1][0])
|
||||
third_data_like = _row_fill(rows[2]) >= max(1, first_fill - 1)
|
||||
if first_short and second_short and third_data_like and not second_is_data:
|
||||
return 2
|
||||
return 1
|
||||
|
||||
|
||||
def detect_spreadsheet_layout(rows: list[list[str]]) -> dict[str, int]:
|
||||
"""
|
||||
Detect spreadsheet preamble/header/data boundaries (1-based row numbers).
|
||||
|
||||
Common template: row 1 = field descriptions, row 2 = column names, row 3+ = data.
|
||||
"""
|
||||
normalized = normalize_rows(rows)
|
||||
if not normalized:
|
||||
return {
|
||||
"preamble_rows": 0,
|
||||
"header_rows": 1,
|
||||
"header_row_start": 1,
|
||||
"header_row_end": 1,
|
||||
"data_start_row": 2,
|
||||
}
|
||||
|
||||
preamble = 0
|
||||
header_index = 0
|
||||
|
||||
if (
|
||||
len(normalized) >= 3
|
||||
and looks_like_description_row(normalized[0])
|
||||
and looks_like_column_header_row(normalized[1])
|
||||
):
|
||||
preamble = 1
|
||||
header_index = 1
|
||||
header_rows = 1
|
||||
else:
|
||||
header_rows = detect_header_row_count(normalized)
|
||||
header_index = preamble
|
||||
|
||||
header_end_index = header_index + header_rows - 1
|
||||
data_start_index = header_end_index + 1
|
||||
|
||||
return {
|
||||
"preamble_rows": preamble,
|
||||
"header_rows": header_rows,
|
||||
"header_row_start": header_index + 1,
|
||||
"header_row_end": header_end_index + 1,
|
||||
"data_start_row": data_start_index + 1,
|
||||
}
|
||||
|
||||
|
||||
def is_qa_style_table(rows: list[list[str]], layout: dict[str, int] | None = None) -> bool:
|
||||
"""True for evaluation/Q&A sheets where each row should become one chunk."""
|
||||
if not rows:
|
||||
return False
|
||||
layout = layout or detect_spreadsheet_layout(rows)
|
||||
h_start = layout["header_row_start"] - 1
|
||||
h_end = layout["header_row_end"]
|
||||
header_text = " ".join(
|
||||
(cell or "").lower() for row in rows[h_start:h_end] for cell in row if cell
|
||||
)
|
||||
return sum(1 for term in QA_HEADER_TERMS if term in header_text) >= 2
|
||||
|
||||
|
||||
def split_body_and_footnotes(rows: list[list[str]], header_rows: int) -> tuple[list[list[str]], list[list[str]], str]:
|
||||
"""Separate data rows from trailing footnote rows."""
|
||||
if header_rows >= len(rows):
|
||||
return [], [], ""
|
||||
|
||||
body = rows[header_rows:]
|
||||
footnote_rows: list[list[str]] = []
|
||||
while body:
|
||||
first_cell = (body[-1][0] if body[-1] else "") or ""
|
||||
joined = " ".join(c for c in body[-1] if c)
|
||||
if FOOTNOTE_ROW_RE.match(first_cell) or FOOTNOTE_ROW_RE.match(joined):
|
||||
footnote_rows.insert(0, body.pop())
|
||||
elif len(joined) <= 80 and any(k in joined for k in ("注", "备注", "说明", "Note")):
|
||||
footnote_rows.insert(0, body.pop())
|
||||
else:
|
||||
break
|
||||
|
||||
footnotes = " ".join(" ".join(c for c in row if c) for row in footnote_rows).strip()
|
||||
return body, footnote_rows, footnotes
|
||||
|
||||
|
||||
def header_signature(rows: list[list[str]], header_rows: int) -> tuple[tuple[str, ...], ...]:
|
||||
if header_rows <= 0:
|
||||
return ()
|
||||
return tuple(tuple(row) for row in rows[:header_rows])
|
||||
|
||||
|
||||
def rows_to_markdown(
|
||||
rows: list[list[str | None]],
|
||||
header_rows: int | None = None,
|
||||
include_footnotes: str = "",
|
||||
) -> str:
|
||||
"""Render rows as a standard Markdown table."""
|
||||
if not looks_like_table(rows):
|
||||
return ""
|
||||
|
||||
normalized = normalize_rows(rows)
|
||||
if header_rows is None:
|
||||
header_rows = detect_header_row_count(normalized)
|
||||
|
||||
body_rows, _, inline_footnotes = split_body_and_footnotes(normalized, header_rows)
|
||||
data_rows = normalized[:header_rows] + body_rows
|
||||
if not data_rows:
|
||||
return ""
|
||||
|
||||
col_count = max(len(r) for r in data_rows)
|
||||
lines: list[str] = []
|
||||
for i, row in enumerate(data_rows):
|
||||
padded = row + [""] * (col_count - len(row))
|
||||
lines.append("| " + " | ".join(padded) + " |")
|
||||
if i == header_rows - 1:
|
||||
lines.append("| " + " | ".join(["---"] * col_count) + " |")
|
||||
|
||||
md = "\n".join(lines)
|
||||
footnotes = include_footnotes or inline_footnotes
|
||||
if footnotes:
|
||||
md += f"\n\n*{footnotes}*"
|
||||
return md
|
||||
|
||||
|
||||
def guess_table_title(text: str) -> str | None:
|
||||
"""Guess table title from a short preceding line."""
|
||||
cleaned = normalize_cell(text)
|
||||
if not cleaned or len(cleaned) > 120:
|
||||
return None
|
||||
if TABLE_TITLE_RE.match(cleaned):
|
||||
return cleaned
|
||||
if cleaned.endswith("表") or cleaned.endswith("列表") or cleaned.endswith("说明"):
|
||||
return cleaned
|
||||
if re.match(r"^表\s*\d+", cleaned):
|
||||
return cleaned
|
||||
return None
|
||||
|
||||
|
||||
def extract_table_keywords(*texts: str, limit: int = 20) -> list[str]:
|
||||
"""Extract retrieval keywords from table title, headers and body."""
|
||||
source = " ".join(t for t in texts if t)
|
||||
words = re.findall(r"[A-Za-z][A-Za-z0-9_-]{2,}|[\u4e00-\u9fff]{2,}", source)
|
||||
seen: set[str] = set()
|
||||
result: list[str] = []
|
||||
for word in words:
|
||||
if word in seen:
|
||||
continue
|
||||
seen.add(word)
|
||||
result.append(word)
|
||||
if len(result) >= limit:
|
||||
break
|
||||
for term in KEYWORD_TERMS:
|
||||
if term.lower() in source.lower() and term not in seen:
|
||||
result.append(term)
|
||||
seen.add(term)
|
||||
if len(result) >= limit:
|
||||
break
|
||||
return result[:limit]
|
||||
|
||||
|
||||
def build_table_embedding_text(
|
||||
*,
|
||||
chapter: str = "",
|
||||
table_title: str = "",
|
||||
markdown: str = "",
|
||||
description: str = "",
|
||||
footnotes: str = "",
|
||||
keywords: list[str] | None = None,
|
||||
ocr_text: str = "",
|
||||
) -> str:
|
||||
"""Compose embedding text: chapter + title + markdown + description + keywords."""
|
||||
parts: list[str] = []
|
||||
if chapter:
|
||||
parts.append(f"章节标题:{chapter}")
|
||||
if table_title:
|
||||
parts.append(f"表格标题:{table_title}")
|
||||
if markdown:
|
||||
parts.append(markdown)
|
||||
if description:
|
||||
parts.append(f"表格说明:{description}")
|
||||
if footnotes:
|
||||
parts.append(f"脚注说明:{footnotes}")
|
||||
if ocr_text:
|
||||
parts.append(f"表格 OCR:{ocr_text}")
|
||||
if keywords:
|
||||
parts.append(f"关键词:{','.join(keywords)}")
|
||||
return "\n".join(parts)
|
||||
|
||||
|
||||
def table_meta_summary(meta: dict[str, Any]) -> dict[str, Any]:
|
||||
"""Pick table-specific fields for chunk metadata."""
|
||||
keys = (
|
||||
"table_title",
|
||||
"table_description",
|
||||
"header_rows",
|
||||
"header_signature",
|
||||
"footnotes",
|
||||
"keywords",
|
||||
"chapter",
|
||||
"pages",
|
||||
"page",
|
||||
"bbox",
|
||||
"bboxes",
|
||||
"crop_path",
|
||||
"image_path",
|
||||
"row_count",
|
||||
"col_count",
|
||||
"cross_page",
|
||||
"table_source",
|
||||
"preceding_text",
|
||||
"following_text",
|
||||
"nearest_heading",
|
||||
"parent_heading",
|
||||
)
|
||||
return {k: meta[k] for k in keys if k in meta and meta[k] not in (None, "", [], {})}
|
||||
@@ -0,0 +1,104 @@
|
||||
"""Text block extraction from analyzed layout."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from rag_cut.models import Block, BlockType
|
||||
from rag_cut.parsers.pdf.layout import LayoutRegion, PageLayout
|
||||
|
||||
_UI_LABEL_RE = re.compile(r"^[\d\s\W]{0,6}[\u4e00-\u9fff]{1,6}$")
|
||||
# "10.上升三角形態" / "1.2 ACCOUNT STATUS" (space after number optional)
|
||||
_NUMBERED_HEADING_RE = re.compile(
|
||||
r"^\s*(\d+(?:\.\d+)*)(?:\.|.)?\s*([A-Za-z0-9\u4e00-\u9fff][A-Za-z0-9\u4e00-\u9fff&/ \-_::]{2,})\s*$"
|
||||
)
|
||||
|
||||
# Title+body merges must not become headings; real section titles stay shorter.
|
||||
MAX_HEADING_CHARS = 100
|
||||
|
||||
|
||||
def _numbered_heading_level(text: str) -> int | None:
|
||||
match = _NUMBERED_HEADING_RE.match(text.strip())
|
||||
if not match or len(match.group(2).strip()) < 3:
|
||||
return None
|
||||
return match.group(1).count(".") + 1
|
||||
|
||||
|
||||
def _looks_like_ui_label(text: str) -> bool:
|
||||
"""True for short UI chips; exclude numbered / CJK section titles."""
|
||||
stripped = text.strip()
|
||||
if _numbered_heading_level(stripped) is not None:
|
||||
return False
|
||||
if re.match(r"^\d+(?:\.\d+)*(?:\.|.)", stripped):
|
||||
return False
|
||||
# Short Chinese section banners like 「形態指標」are not toolbar labels.
|
||||
if re.fullmatch(r"[\u4e00-\u9fff]{2,12}", stripped):
|
||||
return False
|
||||
return bool(_UI_LABEL_RE.match(stripped))
|
||||
|
||||
|
||||
def _font_heading_level(size: float, body_size: float, text: str) -> int | None:
|
||||
stripped = text.strip()
|
||||
numbered = _numbered_heading_level(stripped)
|
||||
if numbered and len(stripped) <= MAX_HEADING_CHARS:
|
||||
return numbered
|
||||
if len(stripped) < 4 or len(stripped) > MAX_HEADING_CHARS:
|
||||
return None
|
||||
if _looks_like_ui_label(stripped):
|
||||
return None
|
||||
|
||||
# Compact CJK section titles such as 「形態指標」.
|
||||
if (
|
||||
size >= body_size + 3
|
||||
and re.fullmatch(r"[\u4e00-\u9fff]{2,12}", stripped)
|
||||
and not _numbered_heading_level(stripped)
|
||||
):
|
||||
return 1
|
||||
|
||||
if size >= body_size + 6:
|
||||
return 1 if len(stripped) >= 10 else 2
|
||||
if size >= body_size + 3:
|
||||
return 2 if len(stripped) >= 8 else 3
|
||||
if size >= body_size + 1.5:
|
||||
return 3
|
||||
return None
|
||||
|
||||
|
||||
def extract_text_blocks(layout: PageLayout, chapter_title: str | None = None) -> tuple[list[Block], str | None]:
|
||||
"""Extract heading/paragraph blocks; update chapter title when headings appear."""
|
||||
blocks: list[Block] = []
|
||||
current_chapter = chapter_title
|
||||
|
||||
for region in layout.regions:
|
||||
if region.kind != "text":
|
||||
continue
|
||||
|
||||
text = region.data.get("text", "").strip()
|
||||
if not text:
|
||||
continue
|
||||
|
||||
font_size = float(region.data.get("font_size", layout.body_font_size))
|
||||
level = _font_heading_level(font_size, layout.body_font_size, text)
|
||||
page_no = layout.page_index + 1
|
||||
meta = {
|
||||
"page": page_no,
|
||||
"bbox": [region.x0, region.y0, region.x1, region.y1],
|
||||
"font_size": font_size,
|
||||
"body_font_size": layout.body_font_size,
|
||||
"page_height": layout.page_height,
|
||||
}
|
||||
if current_chapter:
|
||||
meta["chapter"] = current_chapter
|
||||
|
||||
if level:
|
||||
if level <= 2:
|
||||
current_chapter = text
|
||||
if current_chapter:
|
||||
meta["chapter"] = current_chapter
|
||||
blk = Block(type=BlockType.HEADING, text=text, level=level, meta=meta)
|
||||
else:
|
||||
blk = Block(type=BlockType.PARAGRAPH, text=text, meta=meta)
|
||||
|
||||
blocks.append(blk)
|
||||
|
||||
return blocks, current_chapter
|
||||
@@ -0,0 +1,91 @@
|
||||
"""Image/table region cropping with OCR and descriptions."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import fitz
|
||||
|
||||
from rag_cut.models import Block, BlockType
|
||||
from rag_cut.parsers.pdf.layout import LayoutRegion, PageLayout
|
||||
from rag_cut.parsers.pdf.ocr import describe_visual, ocr_image_file, ocr_pixmap
|
||||
from rag_cut.parsers.pdf.table_extract import build_table_block
|
||||
|
||||
|
||||
def _crop_region(page: fitz.Page, region: LayoutRegion, zoom: float = 2.0) -> fitz.Pixmap:
|
||||
clip = fitz.Rect(region.x0, region.y0, region.x1, region.y1)
|
||||
return page.get_pixmap(matrix=fitz.Matrix(zoom, zoom), clip=clip, alpha=False)
|
||||
|
||||
|
||||
def _save_pixmap(pix: fitz.Pixmap, path: Path) -> None:
|
||||
if pix.n - pix.alpha > 3:
|
||||
pix = fitz.Pixmap(fitz.csRGB, pix)
|
||||
pix.save(str(path))
|
||||
|
||||
|
||||
def extract_visual_blocks(
|
||||
page: fitz.Page,
|
||||
layout: PageLayout,
|
||||
doc: fitz.Document,
|
||||
assets_dir: Path,
|
||||
chapter_title: str | None,
|
||||
img_counter: int,
|
||||
) -> tuple[list[Block], int]:
|
||||
blocks: list[Block] = []
|
||||
page_no = layout.page_index + 1
|
||||
|
||||
for region in layout.regions:
|
||||
if region.kind == "table":
|
||||
table_block = build_table_block(page, region, layout, assets_dir, chapter_title)
|
||||
if table_block:
|
||||
blocks.append(table_block)
|
||||
continue
|
||||
|
||||
if region.kind != "image":
|
||||
continue
|
||||
|
||||
img_counter += 1
|
||||
xref = region.data.get("xref")
|
||||
img_id = f"page{page_no}_img{img_counter}.png"
|
||||
img_path = assets_dir / img_id
|
||||
|
||||
try:
|
||||
# Prefer on-page crop so scaled/clipped placements OCR the visible region.
|
||||
crop_pix = _crop_region(page, region)
|
||||
_save_pixmap(crop_pix, img_path)
|
||||
except Exception:
|
||||
try:
|
||||
pix = fitz.Pixmap(doc, xref)
|
||||
if pix.n - pix.alpha > 3:
|
||||
pix = fitz.Pixmap(fitz.csRGB, pix)
|
||||
pix.save(str(img_path))
|
||||
except Exception:
|
||||
continue
|
||||
|
||||
ocr_text = ocr_image_file(img_path)
|
||||
if not ocr_text:
|
||||
try:
|
||||
ocr_text = ocr_pixmap(_crop_region(page, region))
|
||||
except Exception:
|
||||
ocr_text = ""
|
||||
|
||||
meta = {
|
||||
"page": page_no,
|
||||
"bbox": [region.x0, region.y0, region.x1, region.y1],
|
||||
"xref": xref,
|
||||
}
|
||||
if chapter_title:
|
||||
meta["chapter"] = chapter_title
|
||||
|
||||
blocks.append(
|
||||
Block(
|
||||
type=BlockType.IMAGE,
|
||||
image_id=img_id,
|
||||
image_path=str(img_path),
|
||||
ocr_text=ocr_text,
|
||||
text=describe_visual(ocr_text, "图片", img_id),
|
||||
meta=meta,
|
||||
)
|
||||
)
|
||||
|
||||
return blocks, img_counter
|
||||
Reference in New Issue
Block a user