121 lines
4.5 KiB
Python
121 lines
4.5 KiB
Python
"""End-to-end document chunking pipeline."""
|
|||
|
|
|
||
|
|
from __future__ import annotations
|
||
|
|
|
||
|
|
import hashlib
|
||
|
|
import shutil
|
||
|
|
import uuid
|
||
|
|
from pathlib import Path
|
||
|
|
|
||
|
|
from rag_cut.layout_meta import enrich_layout_metadata
|
||
|
|
from rag_cut.models import Chunk, ChunkResult, SplitConfig, SplitMode
|
||
|
|
from rag_cut.parsers.pdf.noise_filter import filter_toc_blocks
|
||
|
|
from rag_cut.parsers.pdf.table_merge import merge_cross_page_tables
|
||
|
|
from rag_cut.parsers.registry import get_parser
|
||
|
|
from rag_cut.renderer import render_blocks
|
||
|
|
from rag_cut.split_policy import choose_split_config, split_config_summary
|
||
|
|
from rag_cut.splitters import split_by_delimiter, split_by_row
|
||
|
|
from rag_cut.splitters.default_splitter import split_default_with_meta
|
||
|
|
from rag_cut.splitters.heading_splitter import assign_parent_chunk_ids, chunk_groups_to_block_groups
|
||
|
|
from rag_cut.splitters.parent_child import CHUNK_STRATEGY as PARENT_CHILD_STRATEGY
|
||
|
|
from rag_cut.splitters.parent_child import split_by_parent_child
|
||
|
|
from rag_cut.splitters.pdf_semantic import CHUNK_STRATEGY, split_pdf_semantic
|
||
|
|
from rag_cut.splitters.pdf_strategy import choose_pdf_chunk_strategy
|
||
|
|
|
||
|
|
STORAGE_ROOT = Path(__file__).resolve().parent.parent.parent / "storage"
|
||
|
|
PDF_SEMANTIC_EXTS = {".pdf", ".doc", ".docx"}
|
||
|
|
|
||
|
|
|
||
|
|
def _doc_id(path: Path) -> str:
|
||
|
|
digest = hashlib.md5(f"{path.name}-{path.stat().st_mtime}".encode()).hexdigest()[:12]
|
||
|
|
return digest
|
||
|
|
|
||
|
|
|
||
|
|
def chunk_document(
|
||
|
|
path: Path | str,
|
||
|
|
config: SplitConfig | None = None,
|
||
|
|
storage_root: Path | None = None,
|
||
|
|
) -> ChunkResult:
|
||
|
|
path = Path(path)
|
||
|
|
if not path.exists():
|
||
|
|
raise FileNotFoundError(path)
|
||
|
|
|
||
|
|
root = storage_root or STORAGE_ROOT
|
||
|
|
doc_id = _doc_id(path)
|
||
|
|
assets_dir = root / "assets" / doc_id
|
||
|
|
uploads_dir = root / "uploads" / doc_id
|
||
|
|
uploads_dir.mkdir(parents=True, exist_ok=True)
|
||
|
|
|
||
|
|
stored = uploads_dir / path.name
|
||
|
|
if path.resolve() != stored.resolve():
|
||
|
|
shutil.copy2(path, stored)
|
||
|
|
|
||
|
|
parser = get_parser(path)
|
||
|
|
blocks = merge_cross_page_tables(parser.parse(stored, assets_dir))
|
||
|
|
blocks = enrich_layout_metadata(blocks)
|
||
|
|
# All formats: never parse/chunk document directories (目录 / Contents).
|
||
|
|
blocks = filter_toc_blocks(blocks)
|
||
|
|
config = config or choose_split_config(path, blocks)
|
||
|
|
|
||
|
|
for block in blocks:
|
||
|
|
if block.image_path:
|
||
|
|
rel = Path(block.image_path)
|
||
|
|
if "crops" in rel.parts:
|
||
|
|
block.image_path = f"assets/{doc_id}/crops/{rel.name}"
|
||
|
|
else:
|
||
|
|
block.image_path = f"assets/{doc_id}/{rel.name}"
|
||
|
|
crop = block.meta.get("crop_path")
|
||
|
|
if crop:
|
||
|
|
crop_name = Path(crop).name
|
||
|
|
url = f"assets/{doc_id}/crops/{crop_name}"
|
||
|
|
block.meta["crop_path"] = url
|
||
|
|
if block.type.value == "table" and not block.image_path:
|
||
|
|
block.image_path = url
|
||
|
|
|
||
|
|
ext = path.suffix.lower()
|
||
|
|
pdf_strategy = None
|
||
|
|
if ext in PDF_SEMANTIC_EXTS and config.mode == SplitMode.DEFAULT:
|
||
|
|
pdf_strategy = choose_pdf_chunk_strategy(path, blocks)
|
||
|
|
|
||
|
|
group_metas: list[dict] = []
|
||
|
|
if ext in PDF_SEMANTIC_EXTS and config.mode == SplitMode.DEFAULT:
|
||
|
|
chunks = split_pdf_semantic(blocks, config)
|
||
|
|
elif config.mode == SplitMode.BY_ROW:
|
||
|
|
groups = split_by_row(blocks, config)
|
||
|
|
chunks = []
|
||
|
|
elif config.mode == SplitMode.DELIMITER:
|
||
|
|
groups = split_by_delimiter(blocks, config)
|
||
|
|
chunks = []
|
||
|
|
elif config.mode == SplitMode.PARENT_CHILD:
|
||
|
|
parent_child_groups = split_by_parent_child(blocks, config)
|
||
|
|
groups, group_metas = chunk_groups_to_block_groups(parent_child_groups)
|
||
|
|
chunks = []
|
||
|
|
else:
|
||
|
|
groups, group_metas = split_default_with_meta(blocks, config)
|
||
|
|
chunks = []
|
||
|
|
|
||
|
|
if not chunks:
|
||
|
|
for i, group in enumerate(groups):
|
||
|
|
meta: dict = dict(group_metas[i]) if i < len(group_metas) else {}
|
||
|
|
chunks.append(render_blocks(group, index=i, meta=meta))
|
||
|
|
|
||
|
|
chunks = assign_parent_chunk_ids(chunks)
|
||
|
|
|
||
|
|
split_config = split_config_summary(config)
|
||
|
|
if pdf_strategy:
|
||
|
|
split_config["pdf_chunk_strategy"] = pdf_strategy
|
||
|
|
split_config["chunk_strategy"] = CHUNK_STRATEGY
|
||
|
|
elif config.mode == SplitMode.PARENT_CHILD:
|
||
|
|
split_config["chunk_strategy"] = PARENT_CHILD_STRATEGY
|
||
|
|
|
||
|
|
return ChunkResult(
|
||
|
|
filename=path.name,
|
||
|
|
doc_id=doc_id,
|
||
|
|
split_mode=config.mode,
|
||
|
|
split_config=split_config,
|
||
|
|
block_count=len(blocks),
|
||
|
|
chunk_count=len(chunks),
|
||
|
|
chunks=chunks,
|
||
|
|
assets_dir=str(assets_dir) if assets_dir.exists() else None,
|
||
|
|
)
|