92 lines
2.7 KiB
Python
92 lines
2.7 KiB
Python
"""Image/table region cropping with OCR and descriptions."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
|
|
import fitz
|
|
|
|
from rag_cut.models import Block, BlockType
|
|
from rag_cut.parsers.pdf.layout import LayoutRegion, PageLayout
|
|
from rag_cut.parsers.pdf.ocr import describe_visual, ocr_image_file, ocr_pixmap
|
|
from rag_cut.parsers.pdf.table_extract import build_table_block
|
|
|
|
|
|
def _crop_region(page: fitz.Page, region: LayoutRegion, zoom: float = 2.0) -> fitz.Pixmap:
|
|
clip = fitz.Rect(region.x0, region.y0, region.x1, region.y1)
|
|
return page.get_pixmap(matrix=fitz.Matrix(zoom, zoom), clip=clip, alpha=False)
|
|
|
|
|
|
def _save_pixmap(pix: fitz.Pixmap, path: Path) -> None:
|
|
if pix.n - pix.alpha > 3:
|
|
pix = fitz.Pixmap(fitz.csRGB, pix)
|
|
pix.save(str(path))
|
|
|
|
|
|
def extract_visual_blocks(
|
|
page: fitz.Page,
|
|
layout: PageLayout,
|
|
doc: fitz.Document,
|
|
assets_dir: Path,
|
|
chapter_title: str | None,
|
|
img_counter: int,
|
|
) -> tuple[list[Block], int]:
|
|
blocks: list[Block] = []
|
|
page_no = layout.page_index + 1
|
|
|
|
for region in layout.regions:
|
|
if region.kind == "table":
|
|
table_block = build_table_block(page, region, layout, assets_dir, chapter_title)
|
|
if table_block:
|
|
blocks.append(table_block)
|
|
continue
|
|
|
|
if region.kind != "image":
|
|
continue
|
|
|
|
img_counter += 1
|
|
xref = region.data.get("xref")
|
|
img_id = f"page{page_no}_img{img_counter}.png"
|
|
img_path = assets_dir / img_id
|
|
|
|
try:
|
|
# Prefer on-page crop so scaled/clipped placements OCR the visible region.
|
|
crop_pix = _crop_region(page, region)
|
|
_save_pixmap(crop_pix, img_path)
|
|
except Exception:
|
|
try:
|
|
pix = fitz.Pixmap(doc, xref)
|
|
if pix.n - pix.alpha > 3:
|
|
pix = fitz.Pixmap(fitz.csRGB, pix)
|
|
pix.save(str(img_path))
|
|
except Exception:
|
|
continue
|
|
|
|
ocr_text = ocr_image_file(img_path)
|
|
if not ocr_text:
|
|
try:
|
|
ocr_text = ocr_pixmap(_crop_region(page, region))
|
|
except Exception:
|
|
ocr_text = ""
|
|
|
|
meta = {
|
|
"page": page_no,
|
|
"bbox": [region.x0, region.y0, region.x1, region.y1],
|
|
"xref": xref,
|
|
}
|
|
if chapter_title:
|
|
meta["chapter"] = chapter_title
|
|
|
|
blocks.append(
|
|
Block(
|
|
type=BlockType.IMAGE,
|
|
image_id=img_id,
|
|
image_path=str(img_path),
|
|
ocr_text=ocr_text,
|
|
text=describe_visual(ocr_text, "图片", img_id),
|
|
meta=meta,
|
|
)
|
|
)
|
|
|
|
return blocks, img_counter
|