@@ -0,0 +1,421 @@
|
||||
"""Optional MinerU parser adapter.
|
||||
|
||||
MinerU improves the parsing layer when installed, while RAG-cut keeps owning
|
||||
chunking strategy and retrieval metadata.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import signal
|
||||
import shutil
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from rag_cut.models import Block, BlockType
|
||||
from rag_cut.parsers.pdf.ocr import describe_visual, ocr_image_file
|
||||
|
||||
|
||||
def _mineru_command() -> str | None:
|
||||
configured = os.getenv("RAG_CUT_MINERU_CMD")
|
||||
if configured:
|
||||
return configured
|
||||
return shutil.which("mineru") or shutil.which("magic-pdf")
|
||||
|
||||
|
||||
def _mineru_timeout_sec() -> float:
|
||||
raw = os.getenv("RAG_CUT_MINERU_TIMEOUT", "540").strip()
|
||||
try:
|
||||
return max(30.0, float(raw))
|
||||
except ValueError:
|
||||
return 540.0
|
||||
|
||||
|
||||
def _mineru_ocr_enabled() -> bool:
|
||||
"""Fill empty OCR from image assets when pytesseract is available."""
|
||||
raw = os.getenv("RAG_CUT_MINERU_OCR", "1").strip().lower()
|
||||
return raw not in {"0", "false", "off", "no"}
|
||||
|
||||
|
||||
def _terminate_process_tree(process: subprocess.Popen[str]) -> None:
|
||||
"""Stop MinerU and any temporary API/model workers it started."""
|
||||
try:
|
||||
if os.name == "nt":
|
||||
subprocess.run(
|
||||
["taskkill", "/PID", str(process.pid), "/T", "/F"],
|
||||
check=False,
|
||||
capture_output=True,
|
||||
timeout=10,
|
||||
)
|
||||
else:
|
||||
os.killpg(process.pid, signal.SIGKILL)
|
||||
except (OSError, subprocess.SubprocessError):
|
||||
pass
|
||||
|
||||
if process.poll() is None:
|
||||
process.kill()
|
||||
try:
|
||||
process.wait(timeout=5)
|
||||
except (OSError, subprocess.SubprocessError):
|
||||
pass
|
||||
|
||||
|
||||
def _run_mineru(path: Path, output_dir: Path) -> bool:
|
||||
cmd = _mineru_command()
|
||||
if not cmd:
|
||||
return False
|
||||
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
if Path(cmd).name.lower() == "magic-pdf":
|
||||
args = [cmd, "-p", str(path), "-o", str(output_dir)]
|
||||
else:
|
||||
args = [cmd, "-p", str(path), "-o", str(output_dir), "-b", "pipeline"]
|
||||
api_url = os.getenv("RAG_CUT_MINERU_API_URL", "").strip()
|
||||
if api_url:
|
||||
args.extend(["--api-url", api_url])
|
||||
|
||||
try:
|
||||
popen_kwargs: dict[str, Any] = {}
|
||||
if os.name == "nt":
|
||||
popen_kwargs["creationflags"] = getattr(subprocess, "CREATE_NEW_PROCESS_GROUP", 0)
|
||||
else:
|
||||
popen_kwargs["start_new_session"] = True
|
||||
|
||||
process = subprocess.Popen(
|
||||
args,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
errors="replace",
|
||||
**popen_kwargs,
|
||||
)
|
||||
process.communicate(timeout=_mineru_timeout_sec())
|
||||
except subprocess.TimeoutExpired:
|
||||
_terminate_process_tree(process)
|
||||
return False
|
||||
except (OSError, subprocess.SubprocessError):
|
||||
return False
|
||||
return process.returncode == 0
|
||||
|
||||
|
||||
def _find_content_list(output_dir: Path) -> Path | None:
|
||||
"""Prefer stable content_list.json over content_list_v2.json."""
|
||||
candidates = list(output_dir.rglob("*content_list*.json"))
|
||||
if not candidates:
|
||||
return None
|
||||
|
||||
def rank(path: Path) -> tuple[int, int, str]:
|
||||
name = path.name.lower()
|
||||
is_v2 = 1 if "v2" in name else 0
|
||||
return (is_v2, len(path.parts), str(path).lower())
|
||||
|
||||
return sorted(candidates, key=rank)[0]
|
||||
|
||||
|
||||
def _read_json(path: Path) -> Any:
|
||||
return json.loads(path.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def _resolve_asset(path_text: str, base_dir: Path) -> Path | None:
|
||||
if not path_text:
|
||||
return None
|
||||
raw = Path(path_text)
|
||||
if raw.is_absolute() and raw.exists():
|
||||
return raw
|
||||
candidate = base_dir / raw
|
||||
if candidate.exists():
|
||||
return candidate
|
||||
matches = list(base_dir.rglob(raw.name))
|
||||
return matches[0] if matches else None
|
||||
|
||||
|
||||
def _copy_asset(src: Path | None, assets_dir: Path) -> tuple[str | None, str | None]:
|
||||
if not src or not src.exists():
|
||||
return None, None
|
||||
assets_dir.mkdir(parents=True, exist_ok=True)
|
||||
target = assets_dir / src.name
|
||||
if src.resolve() != target.resolve():
|
||||
shutil.copy2(src, target)
|
||||
return target.name, str(target)
|
||||
|
||||
|
||||
def _page(item: dict[str, Any]) -> int | None:
|
||||
for key in ("page", "page_no", "page_num"):
|
||||
if item.get(key) is not None:
|
||||
try:
|
||||
return int(item[key])
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
if item.get("page_idx") is not None:
|
||||
try:
|
||||
return int(item["page_idx"]) + 1
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def _bbox(item: dict[str, Any]) -> list[float]:
|
||||
raw = item.get("bbox") or item.get("poly") or []
|
||||
if isinstance(raw, list) and len(raw) >= 4:
|
||||
try:
|
||||
if all(isinstance(v, (int, float)) for v in raw[:4]):
|
||||
return [float(v) for v in raw[:4]]
|
||||
if all(isinstance(p, list) and len(p) >= 2 for p in raw):
|
||||
xs = [float(p[0]) for p in raw]
|
||||
ys = [float(p[1]) for p in raw]
|
||||
return [min(xs), min(ys), max(xs), max(ys)]
|
||||
except (TypeError, ValueError):
|
||||
return []
|
||||
return []
|
||||
|
||||
|
||||
def _meta(item: dict[str, Any]) -> dict[str, Any]:
|
||||
meta: dict[str, Any] = {"parser": "mineru"}
|
||||
page = _page(item)
|
||||
bbox = _bbox(item)
|
||||
kind = _item_type(item)
|
||||
if page:
|
||||
meta["page"] = page
|
||||
if bbox:
|
||||
meta["bbox"] = bbox
|
||||
if kind:
|
||||
meta["mineru_type"] = kind
|
||||
return meta
|
||||
|
||||
|
||||
def _text_value(item: dict[str, Any]) -> str:
|
||||
for key in (
|
||||
"text",
|
||||
"content",
|
||||
"table_caption",
|
||||
"image_caption",
|
||||
"code_body",
|
||||
"code",
|
||||
"equation",
|
||||
"latex",
|
||||
):
|
||||
value = item.get(key)
|
||||
if isinstance(value, str) and value.strip():
|
||||
return value.strip()
|
||||
if isinstance(value, list):
|
||||
joined = " ".join(str(v).strip() for v in value if str(v).strip())
|
||||
if joined:
|
||||
return joined
|
||||
|
||||
list_items = item.get("list_items")
|
||||
if isinstance(list_items, list) and list_items:
|
||||
parts: list[str] = []
|
||||
for entry in list_items:
|
||||
if isinstance(entry, str) and entry.strip():
|
||||
parts.append(entry.strip())
|
||||
elif isinstance(entry, dict):
|
||||
piece = str(entry.get("text") or entry.get("content") or "").strip()
|
||||
if piece:
|
||||
parts.append(piece)
|
||||
if parts:
|
||||
return "\n".join(parts)
|
||||
return ""
|
||||
|
||||
|
||||
def _table_markdown(item: dict[str, Any]) -> str:
|
||||
for key in ("table_body", "html", "text", "content"):
|
||||
value = item.get(key)
|
||||
if isinstance(value, str) and value.strip():
|
||||
return value.strip()
|
||||
return ""
|
||||
|
||||
|
||||
def _image_path_text(item: dict[str, Any]) -> str:
|
||||
for key in ("img_path", "image_path", "path"):
|
||||
value = item.get(key)
|
||||
if isinstance(value, str) and value.strip():
|
||||
return value.strip()
|
||||
return ""
|
||||
|
||||
|
||||
def _item_type(item: dict[str, Any]) -> str:
|
||||
return str(item.get("type") or item.get("category") or "").lower()
|
||||
|
||||
|
||||
def _ocr_from_item(item: dict[str, Any], image_path: str | None) -> str:
|
||||
ocr_text = str(item.get("ocr_text") or item.get("image_ocr") or item.get("img_caption") or "").strip()
|
||||
if not ocr_text:
|
||||
caption = item.get("image_caption")
|
||||
if isinstance(caption, list):
|
||||
ocr_text = " ".join(str(v).strip() for v in caption if str(v).strip())
|
||||
elif isinstance(caption, str):
|
||||
ocr_text = caption.strip()
|
||||
if ocr_text or not image_path or not _mineru_ocr_enabled():
|
||||
return ocr_text
|
||||
return ocr_image_file(Path(image_path))
|
||||
|
||||
|
||||
_SKIP_MINERU_TYPES = frozenset(
|
||||
{
|
||||
"header",
|
||||
"page_header",
|
||||
"footer",
|
||||
"page_footer",
|
||||
"page_number",
|
||||
"page_num",
|
||||
"header_image",
|
||||
"footer_image",
|
||||
"aside_text",
|
||||
"toc",
|
||||
"contents",
|
||||
"table_of_contents",
|
||||
}
|
||||
)
|
||||
|
||||
# Visual regions MinerU may label separately from plain "image".
|
||||
_IMAGE_KINDS = frozenset(
|
||||
{
|
||||
"image",
|
||||
"figure",
|
||||
"chart",
|
||||
"diagram",
|
||||
"graphic",
|
||||
"photo",
|
||||
"screenshot",
|
||||
"equation",
|
||||
"formula",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def _build_image_block(
|
||||
item: dict[str, Any],
|
||||
base_dir: Path,
|
||||
assets_dir: Path,
|
||||
*,
|
||||
visual_kind: str = "图片",
|
||||
) -> Block | None:
|
||||
src = _resolve_asset(_image_path_text(item), base_dir)
|
||||
image_id, image_path = _copy_asset(src, assets_dir)
|
||||
if not image_id and not image_path:
|
||||
return None
|
||||
|
||||
caption = _text_value(item)
|
||||
ocr_text = _ocr_from_item(item, image_path)
|
||||
meta = _meta(item)
|
||||
meta.update(
|
||||
{
|
||||
"caption": caption,
|
||||
"source_image_path": str(src) if src else None,
|
||||
}
|
||||
)
|
||||
return Block(
|
||||
type=BlockType.IMAGE,
|
||||
text=caption or describe_visual(ocr_text, visual_kind, image_id or "image"),
|
||||
image_id=image_id,
|
||||
image_path=image_path,
|
||||
ocr_text=ocr_text,
|
||||
meta=meta,
|
||||
)
|
||||
|
||||
|
||||
def _blocks_from_content_list(content_list: list[dict[str, Any]], json_path: Path, assets_dir: Path) -> list[Block]:
|
||||
blocks: list[Block] = []
|
||||
base_dir = json_path.parent
|
||||
|
||||
for item in content_list:
|
||||
kind = _item_type(item)
|
||||
if kind in _SKIP_MINERU_TYPES:
|
||||
continue
|
||||
meta = _meta(item)
|
||||
img_path = _image_path_text(item)
|
||||
|
||||
if kind != "table" and (
|
||||
kind in _IMAGE_KINDS or (img_path and kind not in {"text", "title", "heading", "list"})
|
||||
):
|
||||
visual = "图表" if kind in {"chart", "diagram"} else "图片"
|
||||
image_block = _build_image_block(item, base_dir, assets_dir, visual_kind=visual)
|
||||
if image_block:
|
||||
blocks.append(image_block)
|
||||
continue
|
||||
# Chart/image without a resolvable asset: fall through if there is caption text.
|
||||
|
||||
if kind == "table":
|
||||
markdown = _table_markdown(item)
|
||||
src = _resolve_asset(img_path, base_dir) if img_path else None
|
||||
image_id, image_path = _copy_asset(src, assets_dir)
|
||||
ocr_text = _ocr_from_item(item, image_path) if image_path else ""
|
||||
if not markdown and not image_path:
|
||||
continue
|
||||
caption = _text_value(item)
|
||||
if caption:
|
||||
meta["caption"] = caption
|
||||
if src:
|
||||
meta["source_image_path"] = str(src)
|
||||
if image_path:
|
||||
meta["crop_path"] = image_path
|
||||
blocks.append(
|
||||
Block(
|
||||
type=BlockType.TABLE,
|
||||
markdown=markdown,
|
||||
text=caption,
|
||||
image_id=image_id,
|
||||
image_path=image_path,
|
||||
ocr_text=ocr_text,
|
||||
meta=meta,
|
||||
)
|
||||
)
|
||||
continue
|
||||
|
||||
if kind == "code":
|
||||
image_block = _build_image_block(item, base_dir, assets_dir, visual_kind="代码")
|
||||
if image_block:
|
||||
blocks.append(image_block)
|
||||
continue
|
||||
code_text = _text_value(item)
|
||||
if code_text:
|
||||
blocks.append(Block(type=BlockType.PARAGRAPH, text=code_text, meta=meta))
|
||||
continue
|
||||
|
||||
if kind in {"page_footnote", "footnote", "ref_text"}:
|
||||
note = _text_value(item)
|
||||
if note:
|
||||
meta["is_footnote"] = True
|
||||
blocks.append(Block(type=BlockType.PARAGRAPH, text=note, meta=meta))
|
||||
continue
|
||||
|
||||
text = _text_value(item)
|
||||
if not text:
|
||||
# Last resort: unknown typed asset with an image should not be dropped.
|
||||
if img_path:
|
||||
image_block = _build_image_block(item, base_dir, assets_dir)
|
||||
if image_block:
|
||||
blocks.append(image_block)
|
||||
continue
|
||||
level = item.get("text_level") or item.get("level")
|
||||
try:
|
||||
level_int = int(level)
|
||||
except (TypeError, ValueError):
|
||||
level_int = 0
|
||||
if kind in {"title", "heading"} or level_int > 0:
|
||||
meta["source_heading"] = True
|
||||
meta["source_heading_level"] = max(level_int, 1)
|
||||
blocks.append(Block(type=BlockType.HEADING, text=text, level=max(level_int, 1), meta=meta))
|
||||
else:
|
||||
blocks.append(Block(type=BlockType.PARAGRAPH, text=text, meta=meta))
|
||||
|
||||
return blocks
|
||||
|
||||
|
||||
def parse_pdf_with_mineru(path: Path, assets_dir: Path) -> list[Block] | None:
|
||||
"""Return MinerU blocks when available; otherwise None for fallback."""
|
||||
output_dir = assets_dir / "_mineru"
|
||||
if not _run_mineru(path, output_dir):
|
||||
return None
|
||||
|
||||
content_json = _find_content_list(output_dir)
|
||||
if not content_json:
|
||||
return None
|
||||
data = _read_json(content_json)
|
||||
if not isinstance(data, list):
|
||||
return None
|
||||
blocks = _blocks_from_content_list(data, content_json, assets_dir)
|
||||
return blocks or None
|
||||
Reference in New Issue
Block a user