Enhance table title handling and improve heading detection logic

- Added logic to set `table_title` from `caption` if it is under 160 characters and not already set.
- Updated `_blocks_from_content_list` to assign `table_title` based on `caption` length.
- Introduced new regex patterns for better detection of TOC entries and noise.
- Enhanced heading detection to differentiate between numbered instructions and actual headings.
- Added tests to verify that table captions are correctly assigned as titles and that numbered instructions are treated as body text.
This commit is contained in:
陈辅元
2026-07-16 16:01:42 +08:00
parent 26461569c8
commit 8466ed2fbe
8 changed files with 275 additions and 22 deletions
+59 -5
View File
@@ -7,9 +7,27 @@ from dataclasses import dataclass, field
from rag_cut.models import Block, BlockType, SplitConfig
# e.g. "1.2 ACCOUNT STATUS CODE MASTER", "10.上升三角形態" (space after '.' optional)
# e.g. "1.2 ACCOUNT STATUS CODE MASTER", "10.上升三角形態".
# The title starts with a letter/CJK character so "1.1.1" cannot backtrack
# into number="1" + title="1.1".
NUMBERED_HEADING_RE = re.compile(
r"^\s*(\d+(?:\.\d+)*)(?:\.|.)?\s*([A-Za-z0-9\u4e00-\u9fff][A-Za-z0-9\u4e00-\u9fff&/ \-_::]{2,})\s*$"
r"^\s*(\d+(?:\.\d+)*)(?:[..、::]\s*|\s+)"
r"([A-Za-z\u4e00-\u9fff][^\n]{1,120})\s*$"
)
PURE_NUMBERED_HEADING_RE = re.compile(r"^\s*(\d+(?:\.\d+){1,5})\.?\s*$")
TIME_LIKE_RE = re.compile(r"^\s*\d{1,2}:\d{2}(?::\d{2})?\s*$")
NUMBERED_LIST_ITEM_RE = re.compile(r"^\s*\d+[.).、]\s+(?P<body>.+)$")
NUMBERED_OPTION_SENTENCE_RE = re.compile(
r"^\s*\d+(?:\.\d+)+\.?\s+.+\bthis\s+(?:option|function|feature)\s+"
r"(?:enables|allows)\b",
re.I,
)
INSTRUCTION_START_RE = re.compile(
r"^(?:select|click|choose|enter|input|open|close|press|perform|to\s+|on\s+|"
r"the\s+user|users?\s+|next\s+|then\s+|for\s+ease|option/tool|"
r"用户|点击|輸入|输入|選擇|选择|填写|當|当|在|首先|然后|然後|配置|系统|系統|"
r"若|如果|注|详细|詳細|列表|机构|機構|添加|删除|刪除)",
re.I,
)
FIGURE_TABLE_RE = re.compile(
r"^\s*(?:Figure|Fig\.|图|表|Table)\s*[\d.]+",
@@ -38,12 +56,39 @@ def _rendered_len(blocks: list[Block]) -> int:
def numbered_heading_level(text: str) -> int | None:
match = NUMBERED_HEADING_RE.match(text.strip())
stripped = text.strip()
match = PURE_NUMBERED_HEADING_RE.match(stripped)
if match:
return match.group(1).count(".") + 1
match = NUMBERED_HEADING_RE.match(stripped)
if not match:
return None
return match.group(1).count(".") + 1
def looks_like_numbered_instruction(text: str) -> bool:
"""True for numbered procedure/list sentences, not outline headings."""
stripped = text.strip()
if NUMBERED_OPTION_SENTENCE_RE.match(stripped):
return True
match = NUMBERED_LIST_ITEM_RE.match(stripped)
if not match:
return False
body = match.group("body").strip()
if INSTRUCTION_START_RE.match(body):
return True
return len(body) >= 45 and bool(re.search(r"[,.,。;;:]", body))
def looks_like_false_heading_text(text: str) -> bool:
stripped = text.strip()
if TIME_LIKE_RE.match(stripped) or looks_like_numbered_instruction(stripped):
return True
if numbered_heading_level(stripped):
return False
return len(stripped) >= 40 and bool(re.search(r"[.!?。!?;;]\s*$", stripped))
def infer_heading_level(block: Block) -> int:
if block.type == BlockType.HEADING and block.level:
numbered = numbered_heading_level(block.text)
@@ -60,6 +105,8 @@ def is_heading_block(block: Block) -> bool:
text = _text(block)
if not text or len(text) > MAX_HEADING_CHARS:
return False
if looks_like_false_heading_text(text):
return False
if block.type == BlockType.HEADING:
return True
if NUMBERED_HEADING_RE.match(text):
@@ -71,14 +118,21 @@ def is_heading_block(block: Block) -> bool:
def normalize_heading_block(block: Block) -> Block:
text = _text(block)
if block.type == BlockType.HEADING and len(text) > MAX_HEADING_CHARS:
false_heading = looks_like_false_heading_text(text)
if block.type == BlockType.HEADING and (
len(text) > MAX_HEADING_CHARS or false_heading
):
# Parser sometimes merges title+body then marks the blob as heading.
return Block(type=BlockType.PARAGRAPH, text=text, meta=dict(block.meta))
if false_heading:
return block
numbered = numbered_heading_level(text)
if block.type == BlockType.HEADING:
level = numbered or block.level or 1
return block.model_copy(update={"level": level})
if numbered and NUMBERED_HEADING_RE.match(text):
if numbered and (
NUMBERED_HEADING_RE.match(text) or PURE_NUMBERED_HEADING_RE.match(text)
):
return Block(
type=BlockType.HEADING,
text=text,