422 lines
13 KiB
Python
422 lines
13 KiB
Python
"""Optional MinerU parser adapter.
|
|
|
|
MinerU improves the parsing layer when installed, while RAG-cut keeps owning
|
|
chunking strategy and retrieval metadata.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import os
|
|
import signal
|
|
import shutil
|
|
import subprocess
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
from rag_cut.models import Block, BlockType
|
|
from rag_cut.parsers.pdf.ocr import describe_visual, ocr_image_file
|
|
|
|
|
|
def _mineru_command() -> str | None:
|
|
configured = os.getenv("RAG_CUT_MINERU_CMD")
|
|
if configured:
|
|
return configured
|
|
return shutil.which("mineru") or shutil.which("magic-pdf")
|
|
|
|
|
|
def _mineru_timeout_sec() -> float:
|
|
raw = os.getenv("RAG_CUT_MINERU_TIMEOUT", "540").strip()
|
|
try:
|
|
return max(30.0, float(raw))
|
|
except ValueError:
|
|
return 540.0
|
|
|
|
|
|
def _mineru_ocr_enabled() -> bool:
|
|
"""Fill empty OCR from image assets when pytesseract is available."""
|
|
raw = os.getenv("RAG_CUT_MINERU_OCR", "1").strip().lower()
|
|
return raw not in {"0", "false", "off", "no"}
|
|
|
|
|
|
def _terminate_process_tree(process: subprocess.Popen[str]) -> None:
|
|
"""Stop MinerU and any temporary API/model workers it started."""
|
|
try:
|
|
if os.name == "nt":
|
|
subprocess.run(
|
|
["taskkill", "/PID", str(process.pid), "/T", "/F"],
|
|
check=False,
|
|
capture_output=True,
|
|
timeout=10,
|
|
)
|
|
else:
|
|
os.killpg(process.pid, signal.SIGKILL)
|
|
except (OSError, subprocess.SubprocessError):
|
|
pass
|
|
|
|
if process.poll() is None:
|
|
process.kill()
|
|
try:
|
|
process.wait(timeout=5)
|
|
except (OSError, subprocess.SubprocessError):
|
|
pass
|
|
|
|
|
|
def _run_mineru(path: Path, output_dir: Path) -> bool:
|
|
cmd = _mineru_command()
|
|
if not cmd:
|
|
return False
|
|
|
|
output_dir.mkdir(parents=True, exist_ok=True)
|
|
if Path(cmd).name.lower() == "magic-pdf":
|
|
args = [cmd, "-p", str(path), "-o", str(output_dir)]
|
|
else:
|
|
args = [cmd, "-p", str(path), "-o", str(output_dir), "-b", "pipeline"]
|
|
api_url = os.getenv("RAG_CUT_MINERU_API_URL", "").strip()
|
|
if api_url:
|
|
args.extend(["--api-url", api_url])
|
|
|
|
try:
|
|
popen_kwargs: dict[str, Any] = {}
|
|
if os.name == "nt":
|
|
popen_kwargs["creationflags"] = getattr(subprocess, "CREATE_NEW_PROCESS_GROUP", 0)
|
|
else:
|
|
popen_kwargs["start_new_session"] = True
|
|
|
|
process = subprocess.Popen(
|
|
args,
|
|
stdout=subprocess.PIPE,
|
|
stderr=subprocess.PIPE,
|
|
text=True,
|
|
encoding="utf-8",
|
|
errors="replace",
|
|
**popen_kwargs,
|
|
)
|
|
process.communicate(timeout=_mineru_timeout_sec())
|
|
except subprocess.TimeoutExpired:
|
|
_terminate_process_tree(process)
|
|
return False
|
|
except (OSError, subprocess.SubprocessError):
|
|
return False
|
|
return process.returncode == 0
|
|
|
|
|
|
def _find_content_list(output_dir: Path) -> Path | None:
|
|
"""Prefer stable content_list.json over content_list_v2.json."""
|
|
candidates = list(output_dir.rglob("*content_list*.json"))
|
|
if not candidates:
|
|
return None
|
|
|
|
def rank(path: Path) -> tuple[int, int, str]:
|
|
name = path.name.lower()
|
|
is_v2 = 1 if "v2" in name else 0
|
|
return (is_v2, len(path.parts), str(path).lower())
|
|
|
|
return sorted(candidates, key=rank)[0]
|
|
|
|
|
|
def _read_json(path: Path) -> Any:
|
|
return json.loads(path.read_text(encoding="utf-8"))
|
|
|
|
|
|
def _resolve_asset(path_text: str, base_dir: Path) -> Path | None:
|
|
if not path_text:
|
|
return None
|
|
raw = Path(path_text)
|
|
if raw.is_absolute() and raw.exists():
|
|
return raw
|
|
candidate = base_dir / raw
|
|
if candidate.exists():
|
|
return candidate
|
|
matches = list(base_dir.rglob(raw.name))
|
|
return matches[0] if matches else None
|
|
|
|
|
|
def _copy_asset(src: Path | None, assets_dir: Path) -> tuple[str | None, str | None]:
|
|
if not src or not src.exists():
|
|
return None, None
|
|
assets_dir.mkdir(parents=True, exist_ok=True)
|
|
target = assets_dir / src.name
|
|
if src.resolve() != target.resolve():
|
|
shutil.copy2(src, target)
|
|
return target.name, str(target)
|
|
|
|
|
|
def _page(item: dict[str, Any]) -> int | None:
|
|
for key in ("page", "page_no", "page_num"):
|
|
if item.get(key) is not None:
|
|
try:
|
|
return int(item[key])
|
|
except (TypeError, ValueError):
|
|
return None
|
|
if item.get("page_idx") is not None:
|
|
try:
|
|
return int(item["page_idx"]) + 1
|
|
except (TypeError, ValueError):
|
|
return None
|
|
return None
|
|
|
|
|
|
def _bbox(item: dict[str, Any]) -> list[float]:
|
|
raw = item.get("bbox") or item.get("poly") or []
|
|
if isinstance(raw, list) and len(raw) >= 4:
|
|
try:
|
|
if all(isinstance(v, (int, float)) for v in raw[:4]):
|
|
return [float(v) for v in raw[:4]]
|
|
if all(isinstance(p, list) and len(p) >= 2 for p in raw):
|
|
xs = [float(p[0]) for p in raw]
|
|
ys = [float(p[1]) for p in raw]
|
|
return [min(xs), min(ys), max(xs), max(ys)]
|
|
except (TypeError, ValueError):
|
|
return []
|
|
return []
|
|
|
|
|
|
def _meta(item: dict[str, Any]) -> dict[str, Any]:
|
|
meta: dict[str, Any] = {"parser": "mineru"}
|
|
page = _page(item)
|
|
bbox = _bbox(item)
|
|
kind = _item_type(item)
|
|
if page:
|
|
meta["page"] = page
|
|
if bbox:
|
|
meta["bbox"] = bbox
|
|
if kind:
|
|
meta["mineru_type"] = kind
|
|
return meta
|
|
|
|
|
|
def _text_value(item: dict[str, Any]) -> str:
|
|
for key in (
|
|
"text",
|
|
"content",
|
|
"table_caption",
|
|
"image_caption",
|
|
"code_body",
|
|
"code",
|
|
"equation",
|
|
"latex",
|
|
):
|
|
value = item.get(key)
|
|
if isinstance(value, str) and value.strip():
|
|
return value.strip()
|
|
if isinstance(value, list):
|
|
joined = " ".join(str(v).strip() for v in value if str(v).strip())
|
|
if joined:
|
|
return joined
|
|
|
|
list_items = item.get("list_items")
|
|
if isinstance(list_items, list) and list_items:
|
|
parts: list[str] = []
|
|
for entry in list_items:
|
|
if isinstance(entry, str) and entry.strip():
|
|
parts.append(entry.strip())
|
|
elif isinstance(entry, dict):
|
|
piece = str(entry.get("text") or entry.get("content") or "").strip()
|
|
if piece:
|
|
parts.append(piece)
|
|
if parts:
|
|
return "\n".join(parts)
|
|
return ""
|
|
|
|
|
|
def _table_markdown(item: dict[str, Any]) -> str:
|
|
for key in ("table_body", "html", "text", "content"):
|
|
value = item.get(key)
|
|
if isinstance(value, str) and value.strip():
|
|
return value.strip()
|
|
return ""
|
|
|
|
|
|
def _image_path_text(item: dict[str, Any]) -> str:
|
|
for key in ("img_path", "image_path", "path"):
|
|
value = item.get(key)
|
|
if isinstance(value, str) and value.strip():
|
|
return value.strip()
|
|
return ""
|
|
|
|
|
|
def _item_type(item: dict[str, Any]) -> str:
|
|
return str(item.get("type") or item.get("category") or "").lower()
|
|
|
|
|
|
def _ocr_from_item(item: dict[str, Any], image_path: str | None) -> str:
|
|
ocr_text = str(item.get("ocr_text") or item.get("image_ocr") or item.get("img_caption") or "").strip()
|
|
if not ocr_text:
|
|
caption = item.get("image_caption")
|
|
if isinstance(caption, list):
|
|
ocr_text = " ".join(str(v).strip() for v in caption if str(v).strip())
|
|
elif isinstance(caption, str):
|
|
ocr_text = caption.strip()
|
|
if ocr_text or not image_path or not _mineru_ocr_enabled():
|
|
return ocr_text
|
|
return ocr_image_file(Path(image_path))
|
|
|
|
|
|
_SKIP_MINERU_TYPES = frozenset(
|
|
{
|
|
"header",
|
|
"page_header",
|
|
"footer",
|
|
"page_footer",
|
|
"page_number",
|
|
"page_num",
|
|
"header_image",
|
|
"footer_image",
|
|
"aside_text",
|
|
"toc",
|
|
"contents",
|
|
"table_of_contents",
|
|
}
|
|
)
|
|
|
|
# Visual regions MinerU may label separately from plain "image".
|
|
_IMAGE_KINDS = frozenset(
|
|
{
|
|
"image",
|
|
"figure",
|
|
"chart",
|
|
"diagram",
|
|
"graphic",
|
|
"photo",
|
|
"screenshot",
|
|
"equation",
|
|
"formula",
|
|
}
|
|
)
|
|
|
|
|
|
def _build_image_block(
|
|
item: dict[str, Any],
|
|
base_dir: Path,
|
|
assets_dir: Path,
|
|
*,
|
|
visual_kind: str = "图片",
|
|
) -> Block | None:
|
|
src = _resolve_asset(_image_path_text(item), base_dir)
|
|
image_id, image_path = _copy_asset(src, assets_dir)
|
|
if not image_id and not image_path:
|
|
return None
|
|
|
|
caption = _text_value(item)
|
|
ocr_text = _ocr_from_item(item, image_path)
|
|
meta = _meta(item)
|
|
meta.update(
|
|
{
|
|
"caption": caption,
|
|
"source_image_path": str(src) if src else None,
|
|
}
|
|
)
|
|
return Block(
|
|
type=BlockType.IMAGE,
|
|
text=caption or describe_visual(ocr_text, visual_kind, image_id or "image"),
|
|
image_id=image_id,
|
|
image_path=image_path,
|
|
ocr_text=ocr_text,
|
|
meta=meta,
|
|
)
|
|
|
|
|
|
def _blocks_from_content_list(content_list: list[dict[str, Any]], json_path: Path, assets_dir: Path) -> list[Block]:
|
|
blocks: list[Block] = []
|
|
base_dir = json_path.parent
|
|
|
|
for item in content_list:
|
|
kind = _item_type(item)
|
|
if kind in _SKIP_MINERU_TYPES:
|
|
continue
|
|
meta = _meta(item)
|
|
img_path = _image_path_text(item)
|
|
|
|
if kind != "table" and (
|
|
kind in _IMAGE_KINDS or (img_path and kind not in {"text", "title", "heading", "list"})
|
|
):
|
|
visual = "图表" if kind in {"chart", "diagram"} else "图片"
|
|
image_block = _build_image_block(item, base_dir, assets_dir, visual_kind=visual)
|
|
if image_block:
|
|
blocks.append(image_block)
|
|
continue
|
|
# Chart/image without a resolvable asset: fall through if there is caption text.
|
|
|
|
if kind == "table":
|
|
markdown = _table_markdown(item)
|
|
src = _resolve_asset(img_path, base_dir) if img_path else None
|
|
image_id, image_path = _copy_asset(src, assets_dir)
|
|
ocr_text = _ocr_from_item(item, image_path) if image_path else ""
|
|
if not markdown and not image_path:
|
|
continue
|
|
caption = _text_value(item)
|
|
if caption:
|
|
meta["caption"] = caption
|
|
if src:
|
|
meta["source_image_path"] = str(src)
|
|
if image_path:
|
|
meta["crop_path"] = image_path
|
|
blocks.append(
|
|
Block(
|
|
type=BlockType.TABLE,
|
|
markdown=markdown,
|
|
text=caption,
|
|
image_id=image_id,
|
|
image_path=image_path,
|
|
ocr_text=ocr_text,
|
|
meta=meta,
|
|
)
|
|
)
|
|
continue
|
|
|
|
if kind == "code":
|
|
image_block = _build_image_block(item, base_dir, assets_dir, visual_kind="代码")
|
|
if image_block:
|
|
blocks.append(image_block)
|
|
continue
|
|
code_text = _text_value(item)
|
|
if code_text:
|
|
blocks.append(Block(type=BlockType.PARAGRAPH, text=code_text, meta=meta))
|
|
continue
|
|
|
|
if kind in {"page_footnote", "footnote", "ref_text"}:
|
|
note = _text_value(item)
|
|
if note:
|
|
meta["is_footnote"] = True
|
|
blocks.append(Block(type=BlockType.PARAGRAPH, text=note, meta=meta))
|
|
continue
|
|
|
|
text = _text_value(item)
|
|
if not text:
|
|
# Last resort: unknown typed asset with an image should not be dropped.
|
|
if img_path:
|
|
image_block = _build_image_block(item, base_dir, assets_dir)
|
|
if image_block:
|
|
blocks.append(image_block)
|
|
continue
|
|
level = item.get("text_level") or item.get("level")
|
|
try:
|
|
level_int = int(level)
|
|
except (TypeError, ValueError):
|
|
level_int = 0
|
|
if kind in {"title", "heading"} or level_int > 0:
|
|
meta["source_heading"] = True
|
|
meta["source_heading_level"] = max(level_int, 1)
|
|
blocks.append(Block(type=BlockType.HEADING, text=text, level=max(level_int, 1), meta=meta))
|
|
else:
|
|
blocks.append(Block(type=BlockType.PARAGRAPH, text=text, meta=meta))
|
|
|
|
return blocks
|
|
|
|
|
|
def parse_pdf_with_mineru(path: Path, assets_dir: Path) -> list[Block] | None:
|
|
"""Return MinerU blocks when available; otherwise None for fallback."""
|
|
output_dir = assets_dir / "_mineru"
|
|
if not _run_mineru(path, output_dir):
|
|
return None
|
|
|
|
content_json = _find_content_list(output_dir)
|
|
if not content_json:
|
|
return None
|
|
data = _read_json(content_json)
|
|
if not isinstance(data, list):
|
|
return None
|
|
blocks = _blocks_from_content_list(data, content_json, assets_dir)
|
|
return blocks or None
|