Enhance table title handling and improve heading detection logic
- Added logic to set `table_title` from `caption` if it is under 160 characters and not already set. - Updated `_blocks_from_content_list` to assign `table_title` based on `caption` length. - Introduced new regex patterns for better detection of TOC entries and noise. - Enhanced heading detection to differentiate between numbered instructions and actual headings. - Added tests to verify that table captions are correctly assigned as titles and that numbered instructions are treated as body text.
This commit is contained in:
@@ -184,6 +184,10 @@ def _bind_table_context(blocks: list[Block], index: int, meta: dict) -> None:
|
|||||||
page = meta.get("page")
|
page = meta.get("page")
|
||||||
block = blocks[index]
|
block = blocks[index]
|
||||||
|
|
||||||
|
caption = (meta.get("caption") or "").strip()
|
||||||
|
if caption and len(caption) <= 160 and not meta.get("table_title"):
|
||||||
|
meta["table_title"] = caption
|
||||||
|
|
||||||
for j in range(index - 1, -1, -1):
|
for j in range(index - 1, -1, -1):
|
||||||
prev = blocks[j]
|
prev = blocks[j]
|
||||||
if prev.type == BlockType.TABLE:
|
if prev.type == BlockType.TABLE:
|
||||||
|
|||||||
@@ -348,6 +348,8 @@ def _blocks_from_content_list(content_list: list[dict[str, Any]], json_path: Pat
|
|||||||
caption = _text_value(item)
|
caption = _text_value(item)
|
||||||
if caption:
|
if caption:
|
||||||
meta["caption"] = caption
|
meta["caption"] = caption
|
||||||
|
if len(caption) <= 160:
|
||||||
|
meta["table_title"] = caption
|
||||||
if src:
|
if src:
|
||||||
meta["source_image_path"] = str(src)
|
meta["source_image_path"] = str(src)
|
||||||
if image_path:
|
if image_path:
|
||||||
|
|||||||
@@ -56,6 +56,23 @@ _TOC_NUMBERED_PAGE_RE = re.compile(
|
|||||||
r"\d{1,4}\s*$"
|
r"\d{1,4}\s*$"
|
||||||
)
|
)
|
||||||
_MD_TOC_LINK_RE = re.compile(r"^\s*[-*+]\s+\[[^\]]+\]\([^)]+\)\s*$")
|
_MD_TOC_LINK_RE = re.compile(r"^\s*[-*+]\s+\[[^\]]+\]\([^)]+\)\s*$")
|
||||||
|
_TOC_DASH_PAGE_RE = re.compile(
|
||||||
|
r"^\s*.{2,160}?(?:\.{1,}|\u2026+|\s{2,})\s*-\s*\d{1,4}\s*-\s*$"
|
||||||
|
)
|
||||||
|
_TOC_OUTLINE_PAGE_RE = re.compile(
|
||||||
|
r"^\s*(?:chapter\s+\d+|\d+(?:\.\d+)*\.?|"
|
||||||
|
r"\u7b2c[\u4e00-\u9fff\d]+[\u7ae0\u8282\u7bc0])\s+"
|
||||||
|
r".{1,140}?(?:\.{1,}|\u2026+|\s{2,})\s*\d{1,4}\s*$",
|
||||||
|
re.I,
|
||||||
|
)
|
||||||
|
_PAGE_LABEL_RE = re.compile(
|
||||||
|
r"^\s*(?:page\s*)?-?\s*\d{1,4}\s*-?\s*(?:of|/)\s*\d{1,4}\s*$",
|
||||||
|
re.I,
|
||||||
|
)
|
||||||
|
_COMPACT_TOC_PAGE_REF_RE = re.compile(
|
||||||
|
r"(?:\.{2,}|\u2026+)\s*-\s*\d{1,4}\s*-",
|
||||||
|
re.I,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def _looks_like_numbered_section(text: str) -> bool:
|
def _looks_like_numbered_section(text: str) -> bool:
|
||||||
@@ -81,6 +98,10 @@ def is_toc_entry_line(text: str) -> bool:
|
|||||||
return False
|
return False
|
||||||
if _MD_TOC_LINK_RE.match(stripped):
|
if _MD_TOC_LINK_RE.match(stripped):
|
||||||
return True
|
return True
|
||||||
|
if _TOC_DASH_PAGE_RE.match(stripped):
|
||||||
|
return True
|
||||||
|
if _TOC_OUTLINE_PAGE_RE.match(stripped):
|
||||||
|
return True
|
||||||
if _DOT_LEADER_RE.search(stripped) and re.search(r"\d\s*$", stripped):
|
if _DOT_LEADER_RE.search(stripped) and re.search(r"\d\s*$", stripped):
|
||||||
return True
|
return True
|
||||||
if _TOC_ENTRY_LINE_RE.match(stripped):
|
if _TOC_ENTRY_LINE_RE.match(stripped):
|
||||||
@@ -97,6 +118,11 @@ def is_toc_noise_text(text: str) -> bool:
|
|||||||
return False
|
return False
|
||||||
if is_toc_title_text(stripped):
|
if is_toc_title_text(stripped):
|
||||||
return True
|
return True
|
||||||
|
# PyMuPDF may merge several TOC rows into one physical line. Repeated
|
||||||
|
# leader + "- page -" references are directory evidence even without
|
||||||
|
# newline boundaries.
|
||||||
|
if len(_COMPACT_TOC_PAGE_REF_RE.findall(stripped)) >= 2:
|
||||||
|
return True
|
||||||
lines = [ln.strip() for ln in stripped.splitlines() if ln.strip()]
|
lines = [ln.strip() for ln in stripped.splitlines() if ln.strip()]
|
||||||
if not lines:
|
if not lines:
|
||||||
return False
|
return False
|
||||||
@@ -361,14 +387,16 @@ def is_margin_noise_block(block: Block, page_height: float | None) -> bool:
|
|||||||
|
|
||||||
bbox = block.meta.get("bbox")
|
bbox = block.meta.get("bbox")
|
||||||
if not page_height or not bbox or len(bbox) < 4:
|
if not page_height or not bbox or len(bbox) < 4:
|
||||||
return block.type == BlockType.PARAGRAPH and bool(_PAGE_NUM_RE.match(text))
|
return block.type == BlockType.PARAGRAPH and bool(
|
||||||
|
_PAGE_NUM_RE.match(text) or _PAGE_LABEL_RE.match(text)
|
||||||
|
)
|
||||||
|
|
||||||
in_header = _in_margin_zone(bbox, page_height, header=True)
|
in_header = _in_margin_zone(bbox, page_height, header=True)
|
||||||
in_footer = _in_margin_zone(bbox, page_height, header=False)
|
in_footer = _in_margin_zone(bbox, page_height, header=False)
|
||||||
if not in_header and not in_footer:
|
if not in_header and not in_footer:
|
||||||
return False
|
return False
|
||||||
|
|
||||||
if _PAGE_NUM_RE.match(text):
|
if _PAGE_NUM_RE.match(text) or _PAGE_LABEL_RE.match(text):
|
||||||
return True
|
return True
|
||||||
|
|
||||||
if block.type != BlockType.PARAGRAPH:
|
if block.type != BlockType.PARAGRAPH:
|
||||||
|
|||||||
@@ -7,9 +7,27 @@ from dataclasses import dataclass, field
|
|||||||
|
|
||||||
from rag_cut.models import Block, BlockType, SplitConfig
|
from rag_cut.models import Block, BlockType, SplitConfig
|
||||||
|
|
||||||
# e.g. "1.2 ACCOUNT STATUS CODE MASTER", "10.上升三角形態" (space after '.' optional)
|
# e.g. "1.2 ACCOUNT STATUS CODE MASTER", "10.上升三角形態".
|
||||||
|
# The title starts with a letter/CJK character so "1.1.1" cannot backtrack
|
||||||
|
# into number="1" + title="1.1".
|
||||||
NUMBERED_HEADING_RE = re.compile(
|
NUMBERED_HEADING_RE = re.compile(
|
||||||
r"^\s*(\d+(?:\.\d+)*)(?:\.|.)?\s*([A-Za-z0-9\u4e00-\u9fff][A-Za-z0-9\u4e00-\u9fff&/ \-_::]{2,})\s*$"
|
r"^\s*(\d+(?:\.\d+)*)(?:[..、::]\s*|\s+)"
|
||||||
|
r"([A-Za-z\u4e00-\u9fff][^\n]{1,120})\s*$"
|
||||||
|
)
|
||||||
|
PURE_NUMBERED_HEADING_RE = re.compile(r"^\s*(\d+(?:\.\d+){1,5})\.?\s*$")
|
||||||
|
TIME_LIKE_RE = re.compile(r"^\s*\d{1,2}:\d{2}(?::\d{2})?\s*$")
|
||||||
|
NUMBERED_LIST_ITEM_RE = re.compile(r"^\s*\d+[.).、]\s+(?P<body>.+)$")
|
||||||
|
NUMBERED_OPTION_SENTENCE_RE = re.compile(
|
||||||
|
r"^\s*\d+(?:\.\d+)+\.?\s+.+\bthis\s+(?:option|function|feature)\s+"
|
||||||
|
r"(?:enables|allows)\b",
|
||||||
|
re.I,
|
||||||
|
)
|
||||||
|
INSTRUCTION_START_RE = re.compile(
|
||||||
|
r"^(?:select|click|choose|enter|input|open|close|press|perform|to\s+|on\s+|"
|
||||||
|
r"the\s+user|users?\s+|next\s+|then\s+|for\s+ease|option/tool|"
|
||||||
|
r"用户|点击|輸入|输入|選擇|选择|填写|當|当|在|首先|然后|然後|配置|系统|系統|"
|
||||||
|
r"若|如果|注|详细|詳細|列表|机构|機構|添加|删除|刪除)",
|
||||||
|
re.I,
|
||||||
)
|
)
|
||||||
FIGURE_TABLE_RE = re.compile(
|
FIGURE_TABLE_RE = re.compile(
|
||||||
r"^\s*(?:Figure|Fig\.|图|表|Table)\s*[\d.]+",
|
r"^\s*(?:Figure|Fig\.|图|表|Table)\s*[\d.]+",
|
||||||
@@ -38,12 +56,39 @@ def _rendered_len(blocks: list[Block]) -> int:
|
|||||||
|
|
||||||
|
|
||||||
def numbered_heading_level(text: str) -> int | None:
|
def numbered_heading_level(text: str) -> int | None:
|
||||||
match = NUMBERED_HEADING_RE.match(text.strip())
|
stripped = text.strip()
|
||||||
|
match = PURE_NUMBERED_HEADING_RE.match(stripped)
|
||||||
|
if match:
|
||||||
|
return match.group(1).count(".") + 1
|
||||||
|
match = NUMBERED_HEADING_RE.match(stripped)
|
||||||
if not match:
|
if not match:
|
||||||
return None
|
return None
|
||||||
return match.group(1).count(".") + 1
|
return match.group(1).count(".") + 1
|
||||||
|
|
||||||
|
|
||||||
|
def looks_like_numbered_instruction(text: str) -> bool:
|
||||||
|
"""True for numbered procedure/list sentences, not outline headings."""
|
||||||
|
stripped = text.strip()
|
||||||
|
if NUMBERED_OPTION_SENTENCE_RE.match(stripped):
|
||||||
|
return True
|
||||||
|
match = NUMBERED_LIST_ITEM_RE.match(stripped)
|
||||||
|
if not match:
|
||||||
|
return False
|
||||||
|
body = match.group("body").strip()
|
||||||
|
if INSTRUCTION_START_RE.match(body):
|
||||||
|
return True
|
||||||
|
return len(body) >= 45 and bool(re.search(r"[,.,。;;:]", body))
|
||||||
|
|
||||||
|
|
||||||
|
def looks_like_false_heading_text(text: str) -> bool:
|
||||||
|
stripped = text.strip()
|
||||||
|
if TIME_LIKE_RE.match(stripped) or looks_like_numbered_instruction(stripped):
|
||||||
|
return True
|
||||||
|
if numbered_heading_level(stripped):
|
||||||
|
return False
|
||||||
|
return len(stripped) >= 40 and bool(re.search(r"[.!?。!?;;]\s*$", stripped))
|
||||||
|
|
||||||
|
|
||||||
def infer_heading_level(block: Block) -> int:
|
def infer_heading_level(block: Block) -> int:
|
||||||
if block.type == BlockType.HEADING and block.level:
|
if block.type == BlockType.HEADING and block.level:
|
||||||
numbered = numbered_heading_level(block.text)
|
numbered = numbered_heading_level(block.text)
|
||||||
@@ -60,6 +105,8 @@ def is_heading_block(block: Block) -> bool:
|
|||||||
text = _text(block)
|
text = _text(block)
|
||||||
if not text or len(text) > MAX_HEADING_CHARS:
|
if not text or len(text) > MAX_HEADING_CHARS:
|
||||||
return False
|
return False
|
||||||
|
if looks_like_false_heading_text(text):
|
||||||
|
return False
|
||||||
if block.type == BlockType.HEADING:
|
if block.type == BlockType.HEADING:
|
||||||
return True
|
return True
|
||||||
if NUMBERED_HEADING_RE.match(text):
|
if NUMBERED_HEADING_RE.match(text):
|
||||||
@@ -71,14 +118,21 @@ def is_heading_block(block: Block) -> bool:
|
|||||||
|
|
||||||
def normalize_heading_block(block: Block) -> Block:
|
def normalize_heading_block(block: Block) -> Block:
|
||||||
text = _text(block)
|
text = _text(block)
|
||||||
if block.type == BlockType.HEADING and len(text) > MAX_HEADING_CHARS:
|
false_heading = looks_like_false_heading_text(text)
|
||||||
|
if block.type == BlockType.HEADING and (
|
||||||
|
len(text) > MAX_HEADING_CHARS or false_heading
|
||||||
|
):
|
||||||
# Parser sometimes merges title+body then marks the blob as heading.
|
# Parser sometimes merges title+body then marks the blob as heading.
|
||||||
return Block(type=BlockType.PARAGRAPH, text=text, meta=dict(block.meta))
|
return Block(type=BlockType.PARAGRAPH, text=text, meta=dict(block.meta))
|
||||||
|
if false_heading:
|
||||||
|
return block
|
||||||
numbered = numbered_heading_level(text)
|
numbered = numbered_heading_level(text)
|
||||||
if block.type == BlockType.HEADING:
|
if block.type == BlockType.HEADING:
|
||||||
level = numbered or block.level or 1
|
level = numbered or block.level or 1
|
||||||
return block.model_copy(update={"level": level})
|
return block.model_copy(update={"level": level})
|
||||||
if numbered and NUMBERED_HEADING_RE.match(text):
|
if numbered and (
|
||||||
|
NUMBERED_HEADING_RE.match(text) or PURE_NUMBERED_HEADING_RE.match(text)
|
||||||
|
):
|
||||||
return Block(
|
return Block(
|
||||||
type=BlockType.HEADING,
|
type=BlockType.HEADING,
|
||||||
text=text,
|
text=text,
|
||||||
|
|||||||
@@ -8,12 +8,20 @@ from dataclasses import dataclass, field
|
|||||||
from rag_cut.models import Block, BlockType, Chunk, SplitConfig
|
from rag_cut.models import Block, BlockType, Chunk, SplitConfig
|
||||||
from rag_cut.parsers.pdf.noise_filter import is_toc_noise_text, is_toc_title_text
|
from rag_cut.parsers.pdf.noise_filter import is_toc_noise_text, is_toc_title_text
|
||||||
from rag_cut.renderer import collect_chunk_layout_meta, render_blocks
|
from rag_cut.renderer import collect_chunk_layout_meta, render_blocks
|
||||||
|
from rag_cut.splitters.heading_splitter import (
|
||||||
|
PURE_NUMBERED_HEADING_RE,
|
||||||
|
TIME_LIKE_RE,
|
||||||
|
looks_like_false_heading_text,
|
||||||
|
looks_like_numbered_instruction,
|
||||||
|
normalize_heading_block,
|
||||||
|
)
|
||||||
|
|
||||||
CHUNK_STRATEGY = "heading_layout_multimodal"
|
CHUNK_STRATEGY = "heading_layout_multimodal"
|
||||||
|
|
||||||
PAGE_NUMBER_RE = re.compile(r"^\s*(?:[-\u2013\u2014]?\s*)?\d{1,4}(?:\s*/\s*\d{1,4})?\s*$")
|
PAGE_NUMBER_RE = re.compile(r"^\s*(?:[-\u2013\u2014]?\s*)?\d{1,4}(?:\s*/\s*\d{1,4})?\s*$")
|
||||||
NUMBERED_HEADING_RE = re.compile(
|
NUMBERED_HEADING_RE = re.compile(
|
||||||
r"^\s*(?P<num>\d+(?:\.\d+)*)(?:\.|.)?\s*(?P<title>[A-Za-z0-9\u4e00-\u9fff][^\n]{1,120})\s*$"
|
r"^\s*(?P<num>\d+(?:\.\d+)*)(?:[..、::]\s*|\s+)"
|
||||||
|
r"(?P<title>[A-Za-z\u4e00-\u9fff][^\n]{1,120})\s*$"
|
||||||
)
|
)
|
||||||
LETTER_HEADING_RE = re.compile(
|
LETTER_HEADING_RE = re.compile(
|
||||||
r"^\s*(?P<letter>[A-Z])[\.)]\s+(?P<title>[A-Za-z0-9\u4e00-\u9fff][^\n]{1,100})\s*$"
|
r"^\s*(?P<letter>[A-Z])[\.)]\s+(?P<title>[A-Za-z0-9\u4e00-\u9fff][^\n]{1,100})\s*$"
|
||||||
@@ -132,7 +140,7 @@ def _is_noise(block: Block, running_headers: set[str]) -> bool:
|
|||||||
text = " ".join(_text(block).split())
|
text = " ".join(_text(block).split())
|
||||||
if block.type == BlockType.IMAGE:
|
if block.type == BlockType.IMAGE:
|
||||||
return _is_decorative_image(block)
|
return _is_decorative_image(block)
|
||||||
if PAGE_NUMBER_RE.match(text):
|
if PAGE_NUMBER_RE.match(text) or TIME_LIKE_RE.match(text):
|
||||||
return True
|
return True
|
||||||
if _is_toc_text(text):
|
if _is_toc_text(text):
|
||||||
return True
|
return True
|
||||||
@@ -146,19 +154,29 @@ def _heading_signal(block: Block, current_top_level: bool = False) -> _HeadingSi
|
|||||||
if not text or len(text) > 180 or _is_toc_text(text):
|
if not text or len(text) > 180 or _is_toc_text(text):
|
||||||
return None
|
return None
|
||||||
# Keep procedural steps inside the parent section; do not open a new group.
|
# Keep procedural steps inside the parent section; do not open a new group.
|
||||||
if STEP_INSTRUCTION_RE.match(text):
|
if (
|
||||||
|
STEP_INSTRUCTION_RE.match(text)
|
||||||
|
or looks_like_numbered_instruction(text)
|
||||||
|
or looks_like_false_heading_text(text)
|
||||||
|
):
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
pure_numbered = PURE_NUMBERED_HEADING_RE.match(text)
|
||||||
|
|
||||||
if block.type == BlockType.HEADING:
|
if block.type == BlockType.HEADING:
|
||||||
level = block.level or 1
|
level = block.level or 1
|
||||||
numbered = NUMBERED_HEADING_RE.match(text)
|
numbered = NUMBERED_HEADING_RE.match(text)
|
||||||
if numbered:
|
if pure_numbered:
|
||||||
|
level = pure_numbered.group(1).count(".") + 1
|
||||||
|
elif numbered:
|
||||||
level = numbered.group("num").count(".") + 1
|
level = numbered.group("num").count(".") + 1
|
||||||
elif LETTER_HEADING_RE.match(text):
|
elif LETTER_HEADING_RE.match(text):
|
||||||
level = 2 if current_top_level else max(2, level)
|
level = 2 if current_top_level else max(2, level)
|
||||||
return _HeadingSignal(text=" ".join(text.split()), level=max(1, min(level, 6)))
|
return _HeadingSignal(text=" ".join(text.split()), level=max(1, min(level, 6)))
|
||||||
|
|
||||||
numbered = NUMBERED_HEADING_RE.match(text)
|
numbered = NUMBERED_HEADING_RE.match(text)
|
||||||
|
if pure_numbered:
|
||||||
|
return _HeadingSignal(text=" ".join(text.split()), level=pure_numbered.group(1).count(".") + 1)
|
||||||
if numbered:
|
if numbered:
|
||||||
return _HeadingSignal(text=" ".join(text.split()), level=numbered.group("num").count(".") + 1)
|
return _HeadingSignal(text=" ".join(text.split()), level=numbered.group("num").count(".") + 1)
|
||||||
if LETTER_HEADING_RE.match(text):
|
if LETTER_HEADING_RE.match(text):
|
||||||
@@ -339,7 +357,7 @@ def _split_oversized_group(group: _Group, start_index: int, config: SplitConfig)
|
|||||||
|
|
||||||
def flush() -> None:
|
def flush() -> None:
|
||||||
nonlocal current, current_len, part_index
|
nonlocal current, current_len, part_index
|
||||||
body = [b for b in current if b not in prefix]
|
body = [b for b in current if b.type != BlockType.HEADING]
|
||||||
if not body:
|
if not body:
|
||||||
return
|
return
|
||||||
part_group = _Group(group.heading_path, list(current))
|
part_group = _Group(group.heading_path, list(current))
|
||||||
@@ -362,7 +380,17 @@ def _split_oversized_group(group: _Group, start_index: int, config: SplitConfig)
|
|||||||
for block in group.blocks[len(prefix) :]:
|
for block in group.blocks[len(prefix) :]:
|
||||||
block_len = len(block.render()) + 2
|
block_len = len(block.render()) + 2
|
||||||
if current_len + block_len > config.max_chunk_size and len(current) > len(prefix):
|
if current_len + block_len > config.max_chunk_size and len(current) > len(prefix):
|
||||||
flush()
|
context_len = sum(
|
||||||
|
len(item.render()) + 2
|
||||||
|
for item in current
|
||||||
|
if item.type != BlockType.HEADING
|
||||||
|
)
|
||||||
|
keep_atomic_context = (
|
||||||
|
block.type in {BlockType.IMAGE, BlockType.TABLE}
|
||||||
|
and context_len <= min(400, max(80, config.max_chunk_size // 2))
|
||||||
|
)
|
||||||
|
if not keep_atomic_context:
|
||||||
|
flush()
|
||||||
current.append(block)
|
current.append(block)
|
||||||
current_len += block_len
|
current_len += block_len
|
||||||
flush()
|
flush()
|
||||||
@@ -386,9 +414,10 @@ def _build_groups(blocks: list[Block]) -> list[_Group]:
|
|||||||
_attach_heading_only_to_previous(groups, current)
|
_attach_heading_only_to_previous(groups, current)
|
||||||
current = _Group()
|
current = _Group()
|
||||||
|
|
||||||
for raw in blocks:
|
for source in blocks:
|
||||||
if _is_noise(raw, running_headers):
|
if _is_noise(source, running_headers):
|
||||||
continue
|
continue
|
||||||
|
raw = normalize_heading_block(source)
|
||||||
|
|
||||||
signal = _heading_signal(raw, current_top_level=bool(heading_stack))
|
signal = _heading_signal(raw, current_top_level=bool(heading_stack))
|
||||||
if signal:
|
if signal:
|
||||||
@@ -399,16 +428,11 @@ def _build_groups(blocks: list[Block]) -> list[_Group]:
|
|||||||
heading_stack.append((signal.level, signal.text, heading))
|
heading_stack.append((signal.level, signal.text, heading))
|
||||||
path = [item[1] for item in heading_stack]
|
path = [item[1] for item in heading_stack]
|
||||||
current = _Group(heading_path=path)
|
current = _Group(heading_path=path)
|
||||||
for _, _, h_block in heading_stack:
|
current.blocks.append(_enrich_block(heading, path, current.blocks))
|
||||||
current.blocks.append(_enrich_block(h_block, path, current.blocks))
|
|
||||||
continue
|
continue
|
||||||
|
|
||||||
path = [item[1] for item in heading_stack]
|
path = [item[1] for item in heading_stack]
|
||||||
if not current.blocks and heading_stack:
|
if not current.heading_path:
|
||||||
current.heading_path = path
|
|
||||||
for _, _, h_block in heading_stack:
|
|
||||||
current.blocks.append(_enrich_block(h_block, path, current.blocks))
|
|
||||||
elif not current.heading_path:
|
|
||||||
current.heading_path = path
|
current.heading_path = path
|
||||||
|
|
||||||
current.blocks.append(_enrich_block(raw, current.heading_path, current.blocks))
|
current.blocks.append(_enrich_block(raw, current.heading_path, current.blocks))
|
||||||
|
|||||||
@@ -59,6 +59,100 @@ def table(page: int = 1, y: int = 220) -> Block:
|
|||||||
|
|
||||||
|
|
||||||
class HeadingLayoutMultimodalTest(unittest.TestCase):
|
class HeadingLayoutMultimodalTest(unittest.TestCase):
|
||||||
|
def test_numbered_instructions_remain_body_text(self) -> None:
|
||||||
|
chunks = split_pdf_semantic(
|
||||||
|
[
|
||||||
|
h("1 Data Platform", level=1),
|
||||||
|
p("1. 用户点击发布则会显示对应的数据发布弹窗", y=140),
|
||||||
|
p("(1) 发布路径:下拉选择菜单,只可单选", y=170),
|
||||||
|
h("1.1 发布管理", level=2, y=220),
|
||||||
|
p("发布管理正文。", y=250),
|
||||||
|
],
|
||||||
|
SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=2000),
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual([chunk.meta["heading"] for chunk in chunks], ["1 Data Platform", "1.1 发布管理"])
|
||||||
|
self.assertIn("1. 用户点击发布", chunks[0].content)
|
||||||
|
self.assertNotIn("# 1. 用户点击发布", chunks[0].content)
|
||||||
|
|
||||||
|
def test_numbered_option_description_is_not_a_heading(self) -> None:
|
||||||
|
option = h(
|
||||||
|
"10.4 Default: This option enables you to set default Qty, Account and/or Best Price.",
|
||||||
|
level=2,
|
||||||
|
y=160,
|
||||||
|
)
|
||||||
|
option.meta["source_heading"] = True
|
||||||
|
chunks = split_pdf_semantic(
|
||||||
|
[h("10 Option settings", level=1), option, p("Following option details.", y=200)],
|
||||||
|
SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=2000),
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(len(chunks), 1)
|
||||||
|
self.assertEqual(chunks[0].meta["heading"], "10 Option settings")
|
||||||
|
self.assertIn("10.4 Default: This option enables", chunks[0].content)
|
||||||
|
self.assertNotIn("## 10.4 Default", chunks[0].content)
|
||||||
|
|
||||||
|
def test_pure_numeric_heading_keeps_its_real_depth(self) -> None:
|
||||||
|
numeric = h("1.1.1", level=2, page=2)
|
||||||
|
numeric.meta["source_heading"] = True
|
||||||
|
next_section = h("1.2 ACCOUNT STATUS CODE MASTER", level=2, page=5)
|
||||||
|
next_section.meta["source_heading"] = True
|
||||||
|
chunks = split_pdf_semantic(
|
||||||
|
[numeric, table(page=2), next_section, p("Status body.", page=5)],
|
||||||
|
SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=4000),
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual([chunk.meta["heading"] for chunk in chunks], ["1.1.1", "1.2 ACCOUNT STATUS CODE MASTER"])
|
||||||
|
self.assertNotIn("1.1.1", chunks[1].content)
|
||||||
|
self.assertEqual(chunks[1].meta["pages"], [5])
|
||||||
|
|
||||||
|
def test_clock_text_is_not_a_heading_or_ancestor(self) -> None:
|
||||||
|
clock = h("15:18:53", level=2, page=8)
|
||||||
|
clock.meta["source_heading"] = True
|
||||||
|
chunks = split_pdf_semantic(
|
||||||
|
[clock, h("第三节 买入/卖出序", level=2, page=9), p("Useful body.", page=9)],
|
||||||
|
SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=2000),
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(len(chunks), 1)
|
||||||
|
self.assertEqual(chunks[0].meta["heading"], "第三节 买入/卖出序")
|
||||||
|
self.assertNotIn("15:18:53", chunks[0].content)
|
||||||
|
self.assertEqual(chunks[0].meta["pages"], [9])
|
||||||
|
|
||||||
|
def test_parent_heading_is_metadata_not_repeated_page_block(self) -> None:
|
||||||
|
parent = h("1 Parent", level=1, page=1)
|
||||||
|
parent.meta["source_heading"] = True
|
||||||
|
child = h("1.1 Child", level=2, page=2)
|
||||||
|
child.meta["source_heading"] = True
|
||||||
|
chunks = split_pdf_semantic(
|
||||||
|
[parent, child, p("Child body.", page=2)],
|
||||||
|
SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=2000),
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(len(chunks), 1)
|
||||||
|
self.assertEqual(chunks[0].meta["heading_path"], ["1 Parent", "1.1 Child"])
|
||||||
|
self.assertNotIn("# 1 Parent", chunks[0].content)
|
||||||
|
self.assertEqual(chunks[0].meta["pages"], [2])
|
||||||
|
|
||||||
|
def test_oversized_table_keeps_nearest_heading_and_intro(self) -> None:
|
||||||
|
big_table = table(page=1, y=180)
|
||||||
|
big_table.markdown += "\n" + "| value | description |\n" * 80
|
||||||
|
chunks = split_pdf_semantic(
|
||||||
|
[
|
||||||
|
h("2 Settings", level=1),
|
||||||
|
p("The fields are listed below.", y=140),
|
||||||
|
big_table,
|
||||||
|
p("Following explanation " + "x" * 180, y=340),
|
||||||
|
p("Final paragraph " + "y" * 180, y=380),
|
||||||
|
],
|
||||||
|
SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=260),
|
||||||
|
)
|
||||||
|
|
||||||
|
table_chunks = [chunk for chunk in chunks if "table" in chunk.block_types]
|
||||||
|
self.assertEqual(len(table_chunks), 1)
|
||||||
|
self.assertIn("2 Settings", table_chunks[0].content)
|
||||||
|
self.assertIn("The fields are listed below.", table_chunks[0].content)
|
||||||
|
self.assertFalse(any(chunk.block_types and all(t == "heading" for t in chunk.block_types) for chunk in chunks))
|
||||||
def test_numbered_headings_create_same_level_boundaries(self) -> None:
|
def test_numbered_headings_create_same_level_boundaries(self) -> None:
|
||||||
chunks = split_pdf_semantic(
|
chunks = split_pdf_semantic(
|
||||||
[
|
[
|
||||||
|
|||||||
@@ -134,6 +134,21 @@ class MineruAdapterMappingTest(unittest.TestCase):
|
|||||||
self.assertEqual(blocks[0].type, BlockType.PARAGRAPH)
|
self.assertEqual(blocks[0].type, BlockType.PARAGRAPH)
|
||||||
self.assertTrue(blocks[0].meta.get("is_footnote"))
|
self.assertTrue(blocks[0].meta.get("is_footnote"))
|
||||||
|
|
||||||
|
def test_table_caption_becomes_table_title(self) -> None:
|
||||||
|
items = [
|
||||||
|
{
|
||||||
|
"type": "table",
|
||||||
|
"table_body": "| Field | Description |\n| --- | --- |\n| id | Identifier |",
|
||||||
|
"table_caption": ["第一节 综合资讯栏"],
|
||||||
|
"page_idx": 7,
|
||||||
|
}
|
||||||
|
]
|
||||||
|
|
||||||
|
blocks = _blocks_from_content_list(items, Path("content_list.json"), Path("assets"))
|
||||||
|
|
||||||
|
self.assertEqual(len(blocks), 1)
|
||||||
|
self.assertEqual(blocks[0].meta.get("table_title"), "第一节 综合资讯栏")
|
||||||
|
|
||||||
def test_list_items_become_paragraph(self) -> None:
|
def test_list_items_become_paragraph(self) -> None:
|
||||||
items = [
|
items = [
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -15,6 +15,7 @@ from rag_cut.parsers.pdf.noise_filter import (
|
|||||||
is_margin_noise_block,
|
is_margin_noise_block,
|
||||||
is_tiny_image_block,
|
is_tiny_image_block,
|
||||||
is_toc_entry_line,
|
is_toc_entry_line,
|
||||||
|
is_toc_noise_text,
|
||||||
is_toc_title_text,
|
is_toc_title_text,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -190,6 +191,37 @@ class NoiseFilterHeuristicTest(unittest.TestCase):
|
|||||||
self.assertTrue(is_toc_entry_line("5.1 AEOI ID SETTING.. 4"))
|
self.assertTrue(is_toc_entry_line("5.1 AEOI ID SETTING.. 4"))
|
||||||
self.assertEqual(filter_toc_blocks([contents]), [])
|
self.assertEqual(filter_toc_blocks([contents]), [])
|
||||||
|
|
||||||
|
def test_toc_entries_accept_real_manual_page_formats(self) -> None:
|
||||||
|
self.assertTrue(is_toc_entry_line("第一章 簡介 ........ - 5 -"))
|
||||||
|
self.assertTrue(is_toc_entry_line("4 格報價畫面(x4 報價畫面) ........ - 18 -"))
|
||||||
|
self.assertTrue(is_toc_entry_line("1.2. How to Trade an Odd Lot. .16"))
|
||||||
|
self.assertTrue(is_toc_entry_line("CHAPTER 1 LET'S START TRADING .8"))
|
||||||
|
|
||||||
|
def test_filter_toc_drops_long_compacted_manual_contents(self) -> None:
|
||||||
|
contents = Block(
|
||||||
|
type=BlockType.PARAGRAPH,
|
||||||
|
text=(
|
||||||
|
"CHAPTER 1 LET'S START TRADING .8\n"
|
||||||
|
"1.1. How to Trade A Normal Stock. . 9\n"
|
||||||
|
"1.2. How to Trade an Odd Lot. .16\n"
|
||||||
|
"1.3. How to Trade BuyIn, OddLot BuyIN and OFFmarket .18\n"
|
||||||
|
"CHAPTER 2 MANAGING THE ORDER BOOK. . 41"
|
||||||
|
),
|
||||||
|
meta={"page": 6, "bbox": [85, 103, 808, 907]},
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertTrue(is_toc_noise_text(contents.text))
|
||||||
|
self.assertEqual(filter_toc_blocks([contents]), [])
|
||||||
|
|
||||||
|
def test_toc_detects_multiple_entries_merged_on_one_line(self) -> None:
|
||||||
|
merged = (
|
||||||
|
"登入 忘记密码 ................................ - 6 -"
|
||||||
|
"画面介绍 ................................ - 7 -"
|
||||||
|
"综合画面 ................................ - 18 -"
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertTrue(is_toc_noise_text(merged))
|
||||||
|
|
||||||
def test_filter_toc_keeps_real_section_below_directory_on_same_page(self) -> None:
|
def test_filter_toc_keeps_real_section_below_directory_on_same_page(self) -> None:
|
||||||
blocks = [
|
blocks = [
|
||||||
Block(
|
Block(
|
||||||
|
|||||||
Reference in New Issue
Block a user