Files
2026-07-16 11:12:17 +08:00

83 lines
2.3 KiB
Python

"""Choose a PDF chunking strategy from parsed document signals."""
from __future__ import annotations
import re
from pathlib import Path
from rag_cut.models import Block, BlockType
FEATURE_TITLE_RE = re.compile(r"^\s*\d{1,2}[..、]\s*[A-Za-z0-9\u4e00-\u9fff&/ -]{2,24}")
OPERATION_TERMS = (
"点击",
"點擊",
"选择",
"選擇",
"输入",
"輸入",
"打开",
"打開",
"登入",
"用戶可",
"用户可",
"按<",
"點撃",
)
REPORT_TERMS = (
"年度报告",
"年报",
"財務報表",
"财务报表",
"公司治理",
"董事会",
"董事會",
"审计报告",
"審計報告",
"合并资产负债表",
"合併資產負債表",
"经营情况",
"經營情況",
"营业收入",
"營業收入",
"现金流量",
"現金流量",
"股东",
"股東",
)
REPORT_NAME_TERMS = ("annual", "report", "年度", "年报", "年報", "研报", "研報")
PDF_STRATEGY_FEATURE_STEPS = "pdf_feature_step_screenshot"
PDF_STRATEGY_OUTLINE_REPORT = "pdf_outline_report"
def _text(block: Block) -> str:
return (block.text or block.markdown or "").strip()
def _term_count(text: str, terms: tuple[str, ...]) -> int:
return sum(text.count(term) for term in terms)
def choose_pdf_chunk_strategy(path: Path, blocks: list[Block]) -> str:
"""Classify PDFs as operation manuals or report-like documents."""
text_blocks = [block for block in blocks if block.type != BlockType.IMAGE and _text(block)]
image_count = sum(1 for block in blocks if block.type == BlockType.IMAGE)
sample = "\n".join(_text(block) for block in text_blocks[:240])
filename = path.name.lower()
report_score = _term_count(sample, REPORT_TERMS)
if any(term in filename for term in REPORT_NAME_TERMS):
report_score += 3
feature_title_count = sum(1 for block in text_blocks if FEATURE_TITLE_RE.match(_text(block)))
operation_score = _term_count(sample, OPERATION_TERMS)
screenshot_density = image_count / max(len(text_blocks), 1)
if report_score >= 3 and operation_score < 18:
return PDF_STRATEGY_OUTLINE_REPORT
if feature_title_count >= 3 and operation_score >= 6 and image_count >= 3 and screenshot_density >= 0.12:
return PDF_STRATEGY_FEATURE_STEPS
return PDF_STRATEGY_OUTLINE_REPORT