Enhance table title handling and improve heading detection logic
- Added logic to set `table_title` from `caption` if it is under 160 characters and not already set. - Updated `_blocks_from_content_list` to assign `table_title` based on `caption` length. - Introduced new regex patterns for better detection of TOC entries and noise. - Enhanced heading detection to differentiate between numbered instructions and actual headings. - Added tests to verify that table captions are correctly assigned as titles and that numbered instructions are treated as body text.
This commit is contained in:
@@ -8,12 +8,20 @@ from dataclasses import dataclass, field
|
||||
from rag_cut.models import Block, BlockType, Chunk, SplitConfig
|
||||
from rag_cut.parsers.pdf.noise_filter import is_toc_noise_text, is_toc_title_text
|
||||
from rag_cut.renderer import collect_chunk_layout_meta, render_blocks
|
||||
from rag_cut.splitters.heading_splitter import (
|
||||
PURE_NUMBERED_HEADING_RE,
|
||||
TIME_LIKE_RE,
|
||||
looks_like_false_heading_text,
|
||||
looks_like_numbered_instruction,
|
||||
normalize_heading_block,
|
||||
)
|
||||
|
||||
CHUNK_STRATEGY = "heading_layout_multimodal"
|
||||
|
||||
PAGE_NUMBER_RE = re.compile(r"^\s*(?:[-\u2013\u2014]?\s*)?\d{1,4}(?:\s*/\s*\d{1,4})?\s*$")
|
||||
NUMBERED_HEADING_RE = re.compile(
|
||||
r"^\s*(?P<num>\d+(?:\.\d+)*)(?:\.|.)?\s*(?P<title>[A-Za-z0-9\u4e00-\u9fff][^\n]{1,120})\s*$"
|
||||
r"^\s*(?P<num>\d+(?:\.\d+)*)(?:[..、::]\s*|\s+)"
|
||||
r"(?P<title>[A-Za-z\u4e00-\u9fff][^\n]{1,120})\s*$"
|
||||
)
|
||||
LETTER_HEADING_RE = re.compile(
|
||||
r"^\s*(?P<letter>[A-Z])[\.)]\s+(?P<title>[A-Za-z0-9\u4e00-\u9fff][^\n]{1,100})\s*$"
|
||||
@@ -132,7 +140,7 @@ def _is_noise(block: Block, running_headers: set[str]) -> bool:
|
||||
text = " ".join(_text(block).split())
|
||||
if block.type == BlockType.IMAGE:
|
||||
return _is_decorative_image(block)
|
||||
if PAGE_NUMBER_RE.match(text):
|
||||
if PAGE_NUMBER_RE.match(text) or TIME_LIKE_RE.match(text):
|
||||
return True
|
||||
if _is_toc_text(text):
|
||||
return True
|
||||
@@ -146,19 +154,29 @@ def _heading_signal(block: Block, current_top_level: bool = False) -> _HeadingSi
|
||||
if not text or len(text) > 180 or _is_toc_text(text):
|
||||
return None
|
||||
# Keep procedural steps inside the parent section; do not open a new group.
|
||||
if STEP_INSTRUCTION_RE.match(text):
|
||||
if (
|
||||
STEP_INSTRUCTION_RE.match(text)
|
||||
or looks_like_numbered_instruction(text)
|
||||
or looks_like_false_heading_text(text)
|
||||
):
|
||||
return None
|
||||
|
||||
pure_numbered = PURE_NUMBERED_HEADING_RE.match(text)
|
||||
|
||||
if block.type == BlockType.HEADING:
|
||||
level = block.level or 1
|
||||
numbered = NUMBERED_HEADING_RE.match(text)
|
||||
if numbered:
|
||||
if pure_numbered:
|
||||
level = pure_numbered.group(1).count(".") + 1
|
||||
elif numbered:
|
||||
level = numbered.group("num").count(".") + 1
|
||||
elif LETTER_HEADING_RE.match(text):
|
||||
level = 2 if current_top_level else max(2, level)
|
||||
return _HeadingSignal(text=" ".join(text.split()), level=max(1, min(level, 6)))
|
||||
|
||||
numbered = NUMBERED_HEADING_RE.match(text)
|
||||
if pure_numbered:
|
||||
return _HeadingSignal(text=" ".join(text.split()), level=pure_numbered.group(1).count(".") + 1)
|
||||
if numbered:
|
||||
return _HeadingSignal(text=" ".join(text.split()), level=numbered.group("num").count(".") + 1)
|
||||
if LETTER_HEADING_RE.match(text):
|
||||
@@ -339,7 +357,7 @@ def _split_oversized_group(group: _Group, start_index: int, config: SplitConfig)
|
||||
|
||||
def flush() -> None:
|
||||
nonlocal current, current_len, part_index
|
||||
body = [b for b in current if b not in prefix]
|
||||
body = [b for b in current if b.type != BlockType.HEADING]
|
||||
if not body:
|
||||
return
|
||||
part_group = _Group(group.heading_path, list(current))
|
||||
@@ -362,7 +380,17 @@ def _split_oversized_group(group: _Group, start_index: int, config: SplitConfig)
|
||||
for block in group.blocks[len(prefix) :]:
|
||||
block_len = len(block.render()) + 2
|
||||
if current_len + block_len > config.max_chunk_size and len(current) > len(prefix):
|
||||
flush()
|
||||
context_len = sum(
|
||||
len(item.render()) + 2
|
||||
for item in current
|
||||
if item.type != BlockType.HEADING
|
||||
)
|
||||
keep_atomic_context = (
|
||||
block.type in {BlockType.IMAGE, BlockType.TABLE}
|
||||
and context_len <= min(400, max(80, config.max_chunk_size // 2))
|
||||
)
|
||||
if not keep_atomic_context:
|
||||
flush()
|
||||
current.append(block)
|
||||
current_len += block_len
|
||||
flush()
|
||||
@@ -386,9 +414,10 @@ def _build_groups(blocks: list[Block]) -> list[_Group]:
|
||||
_attach_heading_only_to_previous(groups, current)
|
||||
current = _Group()
|
||||
|
||||
for raw in blocks:
|
||||
if _is_noise(raw, running_headers):
|
||||
for source in blocks:
|
||||
if _is_noise(source, running_headers):
|
||||
continue
|
||||
raw = normalize_heading_block(source)
|
||||
|
||||
signal = _heading_signal(raw, current_top_level=bool(heading_stack))
|
||||
if signal:
|
||||
@@ -399,16 +428,11 @@ def _build_groups(blocks: list[Block]) -> list[_Group]:
|
||||
heading_stack.append((signal.level, signal.text, heading))
|
||||
path = [item[1] for item in heading_stack]
|
||||
current = _Group(heading_path=path)
|
||||
for _, _, h_block in heading_stack:
|
||||
current.blocks.append(_enrich_block(h_block, path, current.blocks))
|
||||
current.blocks.append(_enrich_block(heading, path, current.blocks))
|
||||
continue
|
||||
|
||||
path = [item[1] for item in heading_stack]
|
||||
if not current.blocks and heading_stack:
|
||||
current.heading_path = path
|
||||
for _, _, h_block in heading_stack:
|
||||
current.blocks.append(_enrich_block(h_block, path, current.blocks))
|
||||
elif not current.heading_path:
|
||||
if not current.heading_path:
|
||||
current.heading_path = path
|
||||
|
||||
current.blocks.append(_enrich_block(raw, current.heading_path, current.blocks))
|
||||
|
||||
Reference in New Issue
Block a user