Enhance table title handling and improve heading detection logic

- Added logic to set `table_title` from `caption` if it is under 160 characters and not already set.
- Updated `_blocks_from_content_list` to assign `table_title` based on `caption` length.
- Introduced new regex patterns for better detection of TOC entries and noise.
- Enhanced heading detection to differentiate between numbered instructions and actual headings.
- Added tests to verify that table captions are correctly assigned as titles and that numbered instructions are treated as body text.
This commit is contained in:
陈辅元
2026-07-16 16:01:42 +08:00
parent 26461569c8
commit 8466ed2fbe
8 changed files with 275 additions and 22 deletions
+4
View File
@@ -184,6 +184,10 @@ def _bind_table_context(blocks: list[Block], index: int, meta: dict) -> None:
page = meta.get("page")
block = blocks[index]
caption = (meta.get("caption") or "").strip()
if caption and len(caption) <= 160 and not meta.get("table_title"):
meta["table_title"] = caption
for j in range(index - 1, -1, -1):
prev = blocks[j]
if prev.type == BlockType.TABLE:
@@ -348,6 +348,8 @@ def _blocks_from_content_list(content_list: list[dict[str, Any]], json_path: Pat
caption = _text_value(item)
if caption:
meta["caption"] = caption
if len(caption) <= 160:
meta["table_title"] = caption
if src:
meta["source_image_path"] = str(src)
if image_path:
+30 -2
View File
@@ -56,6 +56,23 @@ _TOC_NUMBERED_PAGE_RE = re.compile(
r"\d{1,4}\s*$"
)
_MD_TOC_LINK_RE = re.compile(r"^\s*[-*+]\s+\[[^\]]+\]\([^)]+\)\s*$")
_TOC_DASH_PAGE_RE = re.compile(
r"^\s*.{2,160}?(?:\.{1,}|\u2026+|\s{2,})\s*-\s*\d{1,4}\s*-\s*$"
)
_TOC_OUTLINE_PAGE_RE = re.compile(
r"^\s*(?:chapter\s+\d+|\d+(?:\.\d+)*\.?|"
r"\u7b2c[\u4e00-\u9fff\d]+[\u7ae0\u8282\u7bc0])\s+"
r".{1,140}?(?:\.{1,}|\u2026+|\s{2,})\s*\d{1,4}\s*$",
re.I,
)
_PAGE_LABEL_RE = re.compile(
r"^\s*(?:page\s*)?-?\s*\d{1,4}\s*-?\s*(?:of|/)\s*\d{1,4}\s*$",
re.I,
)
_COMPACT_TOC_PAGE_REF_RE = re.compile(
r"(?:\.{2,}|\u2026+)\s*-\s*\d{1,4}\s*-",
re.I,
)
def _looks_like_numbered_section(text: str) -> bool:
@@ -81,6 +98,10 @@ def is_toc_entry_line(text: str) -> bool:
return False
if _MD_TOC_LINK_RE.match(stripped):
return True
if _TOC_DASH_PAGE_RE.match(stripped):
return True
if _TOC_OUTLINE_PAGE_RE.match(stripped):
return True
if _DOT_LEADER_RE.search(stripped) and re.search(r"\d\s*$", stripped):
return True
if _TOC_ENTRY_LINE_RE.match(stripped):
@@ -97,6 +118,11 @@ def is_toc_noise_text(text: str) -> bool:
return False
if is_toc_title_text(stripped):
return True
# PyMuPDF may merge several TOC rows into one physical line. Repeated
# leader + "- page -" references are directory evidence even without
# newline boundaries.
if len(_COMPACT_TOC_PAGE_REF_RE.findall(stripped)) >= 2:
return True
lines = [ln.strip() for ln in stripped.splitlines() if ln.strip()]
if not lines:
return False
@@ -361,14 +387,16 @@ def is_margin_noise_block(block: Block, page_height: float | None) -> bool:
bbox = block.meta.get("bbox")
if not page_height or not bbox or len(bbox) < 4:
return block.type == BlockType.PARAGRAPH and bool(_PAGE_NUM_RE.match(text))
return block.type == BlockType.PARAGRAPH and bool(
_PAGE_NUM_RE.match(text) or _PAGE_LABEL_RE.match(text)
)
in_header = _in_margin_zone(bbox, page_height, header=True)
in_footer = _in_margin_zone(bbox, page_height, header=False)
if not in_header and not in_footer:
return False
if _PAGE_NUM_RE.match(text):
if _PAGE_NUM_RE.match(text) or _PAGE_LABEL_RE.match(text):
return True
if block.type != BlockType.PARAGRAPH:
+59 -5
View File
@@ -7,9 +7,27 @@ from dataclasses import dataclass, field
from rag_cut.models import Block, BlockType, SplitConfig
# e.g. "1.2 ACCOUNT STATUS CODE MASTER", "10.上升三角形態" (space after '.' optional)
# e.g. "1.2 ACCOUNT STATUS CODE MASTER", "10.上升三角形態".
# The title starts with a letter/CJK character so "1.1.1" cannot backtrack
# into number="1" + title="1.1".
NUMBERED_HEADING_RE = re.compile(
r"^\s*(\d+(?:\.\d+)*)(?:\.|.)?\s*([A-Za-z0-9\u4e00-\u9fff][A-Za-z0-9\u4e00-\u9fff&/ \-_::]{2,})\s*$"
r"^\s*(\d+(?:\.\d+)*)(?:[..、::]\s*|\s+)"
r"([A-Za-z\u4e00-\u9fff][^\n]{1,120})\s*$"
)
PURE_NUMBERED_HEADING_RE = re.compile(r"^\s*(\d+(?:\.\d+){1,5})\.?\s*$")
TIME_LIKE_RE = re.compile(r"^\s*\d{1,2}:\d{2}(?::\d{2})?\s*$")
NUMBERED_LIST_ITEM_RE = re.compile(r"^\s*\d+[.).、]\s+(?P<body>.+)$")
NUMBERED_OPTION_SENTENCE_RE = re.compile(
r"^\s*\d+(?:\.\d+)+\.?\s+.+\bthis\s+(?:option|function|feature)\s+"
r"(?:enables|allows)\b",
re.I,
)
INSTRUCTION_START_RE = re.compile(
r"^(?:select|click|choose|enter|input|open|close|press|perform|to\s+|on\s+|"
r"the\s+user|users?\s+|next\s+|then\s+|for\s+ease|option/tool|"
r"用户|点击|輸入|输入|選擇|选择|填写|當|当|在|首先|然后|然後|配置|系统|系統|"
r"若|如果|注|详细|詳細|列表|机构|機構|添加|删除|刪除)",
re.I,
)
FIGURE_TABLE_RE = re.compile(
r"^\s*(?:Figure|Fig\.|图|表|Table)\s*[\d.]+",
@@ -38,12 +56,39 @@ def _rendered_len(blocks: list[Block]) -> int:
def numbered_heading_level(text: str) -> int | None:
match = NUMBERED_HEADING_RE.match(text.strip())
stripped = text.strip()
match = PURE_NUMBERED_HEADING_RE.match(stripped)
if match:
return match.group(1).count(".") + 1
match = NUMBERED_HEADING_RE.match(stripped)
if not match:
return None
return match.group(1).count(".") + 1
def looks_like_numbered_instruction(text: str) -> bool:
"""True for numbered procedure/list sentences, not outline headings."""
stripped = text.strip()
if NUMBERED_OPTION_SENTENCE_RE.match(stripped):
return True
match = NUMBERED_LIST_ITEM_RE.match(stripped)
if not match:
return False
body = match.group("body").strip()
if INSTRUCTION_START_RE.match(body):
return True
return len(body) >= 45 and bool(re.search(r"[,.,。;;:]", body))
def looks_like_false_heading_text(text: str) -> bool:
stripped = text.strip()
if TIME_LIKE_RE.match(stripped) or looks_like_numbered_instruction(stripped):
return True
if numbered_heading_level(stripped):
return False
return len(stripped) >= 40 and bool(re.search(r"[.!?。!?;;]\s*$", stripped))
def infer_heading_level(block: Block) -> int:
if block.type == BlockType.HEADING and block.level:
numbered = numbered_heading_level(block.text)
@@ -60,6 +105,8 @@ def is_heading_block(block: Block) -> bool:
text = _text(block)
if not text or len(text) > MAX_HEADING_CHARS:
return False
if looks_like_false_heading_text(text):
return False
if block.type == BlockType.HEADING:
return True
if NUMBERED_HEADING_RE.match(text):
@@ -71,14 +118,21 @@ def is_heading_block(block: Block) -> bool:
def normalize_heading_block(block: Block) -> Block:
text = _text(block)
if block.type == BlockType.HEADING and len(text) > MAX_HEADING_CHARS:
false_heading = looks_like_false_heading_text(text)
if block.type == BlockType.HEADING and (
len(text) > MAX_HEADING_CHARS or false_heading
):
# Parser sometimes merges title+body then marks the blob as heading.
return Block(type=BlockType.PARAGRAPH, text=text, meta=dict(block.meta))
if false_heading:
return block
numbered = numbered_heading_level(text)
if block.type == BlockType.HEADING:
level = numbered or block.level or 1
return block.model_copy(update={"level": level})
if numbered and NUMBERED_HEADING_RE.match(text):
if numbered and (
NUMBERED_HEADING_RE.match(text) or PURE_NUMBERED_HEADING_RE.match(text)
):
return Block(
type=BlockType.HEADING,
text=text,
+39 -15
View File
@@ -8,12 +8,20 @@ from dataclasses import dataclass, field
from rag_cut.models import Block, BlockType, Chunk, SplitConfig
from rag_cut.parsers.pdf.noise_filter import is_toc_noise_text, is_toc_title_text
from rag_cut.renderer import collect_chunk_layout_meta, render_blocks
from rag_cut.splitters.heading_splitter import (
PURE_NUMBERED_HEADING_RE,
TIME_LIKE_RE,
looks_like_false_heading_text,
looks_like_numbered_instruction,
normalize_heading_block,
)
CHUNK_STRATEGY = "heading_layout_multimodal"
PAGE_NUMBER_RE = re.compile(r"^\s*(?:[-\u2013\u2014]?\s*)?\d{1,4}(?:\s*/\s*\d{1,4})?\s*$")
NUMBERED_HEADING_RE = re.compile(
r"^\s*(?P<num>\d+(?:\.\d+)*)(?:\.|.)?\s*(?P<title>[A-Za-z0-9\u4e00-\u9fff][^\n]{1,120})\s*$"
r"^\s*(?P<num>\d+(?:\.\d+)*)(?:[..、::]\s*|\s+)"
r"(?P<title>[A-Za-z\u4e00-\u9fff][^\n]{1,120})\s*$"
)
LETTER_HEADING_RE = re.compile(
r"^\s*(?P<letter>[A-Z])[\.)]\s+(?P<title>[A-Za-z0-9\u4e00-\u9fff][^\n]{1,100})\s*$"
@@ -132,7 +140,7 @@ def _is_noise(block: Block, running_headers: set[str]) -> bool:
text = " ".join(_text(block).split())
if block.type == BlockType.IMAGE:
return _is_decorative_image(block)
if PAGE_NUMBER_RE.match(text):
if PAGE_NUMBER_RE.match(text) or TIME_LIKE_RE.match(text):
return True
if _is_toc_text(text):
return True
@@ -146,19 +154,29 @@ def _heading_signal(block: Block, current_top_level: bool = False) -> _HeadingSi
if not text or len(text) > 180 or _is_toc_text(text):
return None
# Keep procedural steps inside the parent section; do not open a new group.
if STEP_INSTRUCTION_RE.match(text):
if (
STEP_INSTRUCTION_RE.match(text)
or looks_like_numbered_instruction(text)
or looks_like_false_heading_text(text)
):
return None
pure_numbered = PURE_NUMBERED_HEADING_RE.match(text)
if block.type == BlockType.HEADING:
level = block.level or 1
numbered = NUMBERED_HEADING_RE.match(text)
if numbered:
if pure_numbered:
level = pure_numbered.group(1).count(".") + 1
elif numbered:
level = numbered.group("num").count(".") + 1
elif LETTER_HEADING_RE.match(text):
level = 2 if current_top_level else max(2, level)
return _HeadingSignal(text=" ".join(text.split()), level=max(1, min(level, 6)))
numbered = NUMBERED_HEADING_RE.match(text)
if pure_numbered:
return _HeadingSignal(text=" ".join(text.split()), level=pure_numbered.group(1).count(".") + 1)
if numbered:
return _HeadingSignal(text=" ".join(text.split()), level=numbered.group("num").count(".") + 1)
if LETTER_HEADING_RE.match(text):
@@ -339,7 +357,7 @@ def _split_oversized_group(group: _Group, start_index: int, config: SplitConfig)
def flush() -> None:
nonlocal current, current_len, part_index
body = [b for b in current if b not in prefix]
body = [b for b in current if b.type != BlockType.HEADING]
if not body:
return
part_group = _Group(group.heading_path, list(current))
@@ -362,7 +380,17 @@ def _split_oversized_group(group: _Group, start_index: int, config: SplitConfig)
for block in group.blocks[len(prefix) :]:
block_len = len(block.render()) + 2
if current_len + block_len > config.max_chunk_size and len(current) > len(prefix):
flush()
context_len = sum(
len(item.render()) + 2
for item in current
if item.type != BlockType.HEADING
)
keep_atomic_context = (
block.type in {BlockType.IMAGE, BlockType.TABLE}
and context_len <= min(400, max(80, config.max_chunk_size // 2))
)
if not keep_atomic_context:
flush()
current.append(block)
current_len += block_len
flush()
@@ -386,9 +414,10 @@ def _build_groups(blocks: list[Block]) -> list[_Group]:
_attach_heading_only_to_previous(groups, current)
current = _Group()
for raw in blocks:
if _is_noise(raw, running_headers):
for source in blocks:
if _is_noise(source, running_headers):
continue
raw = normalize_heading_block(source)
signal = _heading_signal(raw, current_top_level=bool(heading_stack))
if signal:
@@ -399,16 +428,11 @@ def _build_groups(blocks: list[Block]) -> list[_Group]:
heading_stack.append((signal.level, signal.text, heading))
path = [item[1] for item in heading_stack]
current = _Group(heading_path=path)
for _, _, h_block in heading_stack:
current.blocks.append(_enrich_block(h_block, path, current.blocks))
current.blocks.append(_enrich_block(heading, path, current.blocks))
continue
path = [item[1] for item in heading_stack]
if not current.blocks and heading_stack:
current.heading_path = path
for _, _, h_block in heading_stack:
current.blocks.append(_enrich_block(h_block, path, current.blocks))
elif not current.heading_path:
if not current.heading_path:
current.heading_path = path
current.blocks.append(_enrich_block(raw, current.heading_path, current.blocks))