Enhance table title handling and improve heading detection logic
- Added logic to set `table_title` from `caption` if it is under 160 characters and not already set. - Updated `_blocks_from_content_list` to assign `table_title` based on `caption` length. - Introduced new regex patterns for better detection of TOC entries and noise. - Enhanced heading detection to differentiate between numbered instructions and actual headings. - Added tests to verify that table captions are correctly assigned as titles and that numbered instructions are treated as body text.
This commit is contained in:
@@ -7,9 +7,27 @@ from dataclasses import dataclass, field
|
||||
|
||||
from rag_cut.models import Block, BlockType, SplitConfig
|
||||
|
||||
# e.g. "1.2 ACCOUNT STATUS CODE MASTER", "10.上升三角形態" (space after '.' optional)
|
||||
# e.g. "1.2 ACCOUNT STATUS CODE MASTER", "10.上升三角形態".
|
||||
# The title starts with a letter/CJK character so "1.1.1" cannot backtrack
|
||||
# into number="1" + title="1.1".
|
||||
NUMBERED_HEADING_RE = re.compile(
|
||||
r"^\s*(\d+(?:\.\d+)*)(?:\.|.)?\s*([A-Za-z0-9\u4e00-\u9fff][A-Za-z0-9\u4e00-\u9fff&/ \-_::]{2,})\s*$"
|
||||
r"^\s*(\d+(?:\.\d+)*)(?:[..、::]\s*|\s+)"
|
||||
r"([A-Za-z\u4e00-\u9fff][^\n]{1,120})\s*$"
|
||||
)
|
||||
PURE_NUMBERED_HEADING_RE = re.compile(r"^\s*(\d+(?:\.\d+){1,5})\.?\s*$")
|
||||
TIME_LIKE_RE = re.compile(r"^\s*\d{1,2}:\d{2}(?::\d{2})?\s*$")
|
||||
NUMBERED_LIST_ITEM_RE = re.compile(r"^\s*\d+[.).、]\s+(?P<body>.+)$")
|
||||
NUMBERED_OPTION_SENTENCE_RE = re.compile(
|
||||
r"^\s*\d+(?:\.\d+)+\.?\s+.+\bthis\s+(?:option|function|feature)\s+"
|
||||
r"(?:enables|allows)\b",
|
||||
re.I,
|
||||
)
|
||||
INSTRUCTION_START_RE = re.compile(
|
||||
r"^(?:select|click|choose|enter|input|open|close|press|perform|to\s+|on\s+|"
|
||||
r"the\s+user|users?\s+|next\s+|then\s+|for\s+ease|option/tool|"
|
||||
r"用户|点击|輸入|输入|選擇|选择|填写|當|当|在|首先|然后|然後|配置|系统|系統|"
|
||||
r"若|如果|注|详细|詳細|列表|机构|機構|添加|删除|刪除)",
|
||||
re.I,
|
||||
)
|
||||
FIGURE_TABLE_RE = re.compile(
|
||||
r"^\s*(?:Figure|Fig\.|图|表|Table)\s*[\d.]+",
|
||||
@@ -38,12 +56,39 @@ def _rendered_len(blocks: list[Block]) -> int:
|
||||
|
||||
|
||||
def numbered_heading_level(text: str) -> int | None:
|
||||
match = NUMBERED_HEADING_RE.match(text.strip())
|
||||
stripped = text.strip()
|
||||
match = PURE_NUMBERED_HEADING_RE.match(stripped)
|
||||
if match:
|
||||
return match.group(1).count(".") + 1
|
||||
match = NUMBERED_HEADING_RE.match(stripped)
|
||||
if not match:
|
||||
return None
|
||||
return match.group(1).count(".") + 1
|
||||
|
||||
|
||||
def looks_like_numbered_instruction(text: str) -> bool:
|
||||
"""True for numbered procedure/list sentences, not outline headings."""
|
||||
stripped = text.strip()
|
||||
if NUMBERED_OPTION_SENTENCE_RE.match(stripped):
|
||||
return True
|
||||
match = NUMBERED_LIST_ITEM_RE.match(stripped)
|
||||
if not match:
|
||||
return False
|
||||
body = match.group("body").strip()
|
||||
if INSTRUCTION_START_RE.match(body):
|
||||
return True
|
||||
return len(body) >= 45 and bool(re.search(r"[,.,。;;:]", body))
|
||||
|
||||
|
||||
def looks_like_false_heading_text(text: str) -> bool:
|
||||
stripped = text.strip()
|
||||
if TIME_LIKE_RE.match(stripped) or looks_like_numbered_instruction(stripped):
|
||||
return True
|
||||
if numbered_heading_level(stripped):
|
||||
return False
|
||||
return len(stripped) >= 40 and bool(re.search(r"[.!?。!?;;]\s*$", stripped))
|
||||
|
||||
|
||||
def infer_heading_level(block: Block) -> int:
|
||||
if block.type == BlockType.HEADING and block.level:
|
||||
numbered = numbered_heading_level(block.text)
|
||||
@@ -60,6 +105,8 @@ def is_heading_block(block: Block) -> bool:
|
||||
text = _text(block)
|
||||
if not text or len(text) > MAX_HEADING_CHARS:
|
||||
return False
|
||||
if looks_like_false_heading_text(text):
|
||||
return False
|
||||
if block.type == BlockType.HEADING:
|
||||
return True
|
||||
if NUMBERED_HEADING_RE.match(text):
|
||||
@@ -71,14 +118,21 @@ def is_heading_block(block: Block) -> bool:
|
||||
|
||||
def normalize_heading_block(block: Block) -> Block:
|
||||
text = _text(block)
|
||||
if block.type == BlockType.HEADING and len(text) > MAX_HEADING_CHARS:
|
||||
false_heading = looks_like_false_heading_text(text)
|
||||
if block.type == BlockType.HEADING and (
|
||||
len(text) > MAX_HEADING_CHARS or false_heading
|
||||
):
|
||||
# Parser sometimes merges title+body then marks the blob as heading.
|
||||
return Block(type=BlockType.PARAGRAPH, text=text, meta=dict(block.meta))
|
||||
if false_heading:
|
||||
return block
|
||||
numbered = numbered_heading_level(text)
|
||||
if block.type == BlockType.HEADING:
|
||||
level = numbered or block.level or 1
|
||||
return block.model_copy(update={"level": level})
|
||||
if numbered and NUMBERED_HEADING_RE.match(text):
|
||||
if numbered and (
|
||||
NUMBERED_HEADING_RE.match(text) or PURE_NUMBERED_HEADING_RE.match(text)
|
||||
):
|
||||
return Block(
|
||||
type=BlockType.HEADING,
|
||||
text=text,
|
||||
|
||||
Reference in New Issue
Block a user