"""Choose a PDF chunking strategy from parsed document signals.""" from __future__ import annotations import re from pathlib import Path from rag_cut.models import Block, BlockType FEATURE_TITLE_RE = re.compile(r"^\s*\d{1,2}[..、]\s*[A-Za-z0-9\u4e00-\u9fff&/ -]{2,24}") OPERATION_TERMS = ( "点击", "點擊", "选择", "選擇", "输入", "輸入", "打开", "打開", "登入", "用戶可", "用户可", "按<", "點撃", ) REPORT_TERMS = ( "年度报告", "年报", "財務報表", "财务报表", "公司治理", "董事会", "董事會", "审计报告", "審計報告", "合并资产负债表", "合併資產負債表", "经营情况", "經營情況", "营业收入", "營業收入", "现金流量", "現金流量", "股东", "股東", ) REPORT_NAME_TERMS = ("annual", "report", "年度", "年报", "年報", "研报", "研報") PDF_STRATEGY_FEATURE_STEPS = "pdf_feature_step_screenshot" PDF_STRATEGY_OUTLINE_REPORT = "pdf_outline_report" def _text(block: Block) -> str: return (block.text or block.markdown or "").strip() def _term_count(text: str, terms: tuple[str, ...]) -> int: return sum(text.count(term) for term in terms) def choose_pdf_chunk_strategy(path: Path, blocks: list[Block]) -> str: """Classify PDFs as operation manuals or report-like documents.""" text_blocks = [block for block in blocks if block.type != BlockType.IMAGE and _text(block)] image_count = sum(1 for block in blocks if block.type == BlockType.IMAGE) sample = "\n".join(_text(block) for block in text_blocks[:240]) filename = path.name.lower() report_score = _term_count(sample, REPORT_TERMS) if any(term in filename for term in REPORT_NAME_TERMS): report_score += 3 feature_title_count = sum(1 for block in text_blocks if FEATURE_TITLE_RE.match(_text(block))) operation_score = _term_count(sample, OPERATION_TERMS) screenshot_density = image_count / max(len(text_blocks), 1) if report_score >= 3 and operation_score < 18: return PDF_STRATEGY_OUTLINE_REPORT if feature_title_count >= 3 and operation_score >= 6 and image_count >= 3 and screenshot_density >= 0.12: return PDF_STRATEGY_FEATURE_STEPS return PDF_STRATEGY_OUTLINE_REPORT