From 8466ed2fbeed7a6ef09b755103b57c3c1a470039 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E9=99=88=E8=BE=85=E5=85=83?= <2401926342@qq.com> Date: Thu, 16 Jul 2026 16:01:42 +0800 Subject: [PATCH] Enhance table title handling and improve heading detection logic - Added logic to set `table_title` from `caption` if it is under 160 characters and not already set. - Updated `_blocks_from_content_list` to assign `table_title` based on `caption` length. - Introduced new regex patterns for better detection of TOC entries and noise. - Enhanced heading detection to differentiate between numbered instructions and actual headings. - Added tests to verify that table captions are correctly assigned as titles and that numbered instructions are treated as body text. --- backend/rag_cut/layout_meta.py | 4 + backend/rag_cut/parsers/mineru_adapter.py | 2 + backend/rag_cut/parsers/pdf/noise_filter.py | 32 ++++++- backend/rag_cut/splitters/heading_splitter.py | 64 ++++++++++++- backend/rag_cut/splitters/pdf_semantic.py | 54 ++++++++--- .../tests/test_heading_layout_multimodal.py | 94 +++++++++++++++++++ backend/tests/test_mineru_adapter.py | 15 +++ backend/tests/test_pdf_noise_filter.py | 32 +++++++ 8 files changed, 275 insertions(+), 22 deletions(-) diff --git a/backend/rag_cut/layout_meta.py b/backend/rag_cut/layout_meta.py index 6ba9786..7339a8e 100644 --- a/backend/rag_cut/layout_meta.py +++ b/backend/rag_cut/layout_meta.py @@ -184,6 +184,10 @@ def _bind_table_context(blocks: list[Block], index: int, meta: dict) -> None: page = meta.get("page") block = blocks[index] + caption = (meta.get("caption") or "").strip() + if caption and len(caption) <= 160 and not meta.get("table_title"): + meta["table_title"] = caption + for j in range(index - 1, -1, -1): prev = blocks[j] if prev.type == BlockType.TABLE: diff --git a/backend/rag_cut/parsers/mineru_adapter.py b/backend/rag_cut/parsers/mineru_adapter.py index 2453054..0524cc8 100644 --- a/backend/rag_cut/parsers/mineru_adapter.py +++ b/backend/rag_cut/parsers/mineru_adapter.py @@ -348,6 +348,8 @@ def _blocks_from_content_list(content_list: list[dict[str, Any]], json_path: Pat caption = _text_value(item) if caption: meta["caption"] = caption + if len(caption) <= 160: + meta["table_title"] = caption if src: meta["source_image_path"] = str(src) if image_path: diff --git a/backend/rag_cut/parsers/pdf/noise_filter.py b/backend/rag_cut/parsers/pdf/noise_filter.py index 9dbbf2b..70eafd1 100644 --- a/backend/rag_cut/parsers/pdf/noise_filter.py +++ b/backend/rag_cut/parsers/pdf/noise_filter.py @@ -56,6 +56,23 @@ _TOC_NUMBERED_PAGE_RE = re.compile( r"\d{1,4}\s*$" ) _MD_TOC_LINK_RE = re.compile(r"^\s*[-*+]\s+\[[^\]]+\]\([^)]+\)\s*$") +_TOC_DASH_PAGE_RE = re.compile( + r"^\s*.{2,160}?(?:\.{1,}|\u2026+|\s{2,})\s*-\s*\d{1,4}\s*-\s*$" +) +_TOC_OUTLINE_PAGE_RE = re.compile( + r"^\s*(?:chapter\s+\d+|\d+(?:\.\d+)*\.?|" + r"\u7b2c[\u4e00-\u9fff\d]+[\u7ae0\u8282\u7bc0])\s+" + r".{1,140}?(?:\.{1,}|\u2026+|\s{2,})\s*\d{1,4}\s*$", + re.I, +) +_PAGE_LABEL_RE = re.compile( + r"^\s*(?:page\s*)?-?\s*\d{1,4}\s*-?\s*(?:of|/)\s*\d{1,4}\s*$", + re.I, +) +_COMPACT_TOC_PAGE_REF_RE = re.compile( + r"(?:\.{2,}|\u2026+)\s*-\s*\d{1,4}\s*-", + re.I, +) def _looks_like_numbered_section(text: str) -> bool: @@ -81,6 +98,10 @@ def is_toc_entry_line(text: str) -> bool: return False if _MD_TOC_LINK_RE.match(stripped): return True + if _TOC_DASH_PAGE_RE.match(stripped): + return True + if _TOC_OUTLINE_PAGE_RE.match(stripped): + return True if _DOT_LEADER_RE.search(stripped) and re.search(r"\d\s*$", stripped): return True if _TOC_ENTRY_LINE_RE.match(stripped): @@ -97,6 +118,11 @@ def is_toc_noise_text(text: str) -> bool: return False if is_toc_title_text(stripped): return True + # PyMuPDF may merge several TOC rows into one physical line. Repeated + # leader + "- page -" references are directory evidence even without + # newline boundaries. + if len(_COMPACT_TOC_PAGE_REF_RE.findall(stripped)) >= 2: + return True lines = [ln.strip() for ln in stripped.splitlines() if ln.strip()] if not lines: return False @@ -361,14 +387,16 @@ def is_margin_noise_block(block: Block, page_height: float | None) -> bool: bbox = block.meta.get("bbox") if not page_height or not bbox or len(bbox) < 4: - return block.type == BlockType.PARAGRAPH and bool(_PAGE_NUM_RE.match(text)) + return block.type == BlockType.PARAGRAPH and bool( + _PAGE_NUM_RE.match(text) or _PAGE_LABEL_RE.match(text) + ) in_header = _in_margin_zone(bbox, page_height, header=True) in_footer = _in_margin_zone(bbox, page_height, header=False) if not in_header and not in_footer: return False - if _PAGE_NUM_RE.match(text): + if _PAGE_NUM_RE.match(text) or _PAGE_LABEL_RE.match(text): return True if block.type != BlockType.PARAGRAPH: diff --git a/backend/rag_cut/splitters/heading_splitter.py b/backend/rag_cut/splitters/heading_splitter.py index a17ba2a..99af7fc 100644 --- a/backend/rag_cut/splitters/heading_splitter.py +++ b/backend/rag_cut/splitters/heading_splitter.py @@ -7,9 +7,27 @@ from dataclasses import dataclass, field from rag_cut.models import Block, BlockType, SplitConfig -# e.g. "1.2 ACCOUNT STATUS CODE MASTER", "10.上升三角形態" (space after '.' optional) +# e.g. "1.2 ACCOUNT STATUS CODE MASTER", "10.上升三角形態". +# The title starts with a letter/CJK character so "1.1.1" cannot backtrack +# into number="1" + title="1.1". NUMBERED_HEADING_RE = re.compile( - r"^\s*(\d+(?:\.\d+)*)(?:\.|.)?\s*([A-Za-z0-9\u4e00-\u9fff][A-Za-z0-9\u4e00-\u9fff&/ \-_::]{2,})\s*$" + r"^\s*(\d+(?:\.\d+)*)(?:[..、::]\s*|\s+)" + r"([A-Za-z\u4e00-\u9fff][^\n]{1,120})\s*$" +) +PURE_NUMBERED_HEADING_RE = re.compile(r"^\s*(\d+(?:\.\d+){1,5})\.?\s*$") +TIME_LIKE_RE = re.compile(r"^\s*\d{1,2}:\d{2}(?::\d{2})?\s*$") +NUMBERED_LIST_ITEM_RE = re.compile(r"^\s*\d+[.).、]\s+(?P.+)$") +NUMBERED_OPTION_SENTENCE_RE = re.compile( + r"^\s*\d+(?:\.\d+)+\.?\s+.+\bthis\s+(?:option|function|feature)\s+" + r"(?:enables|allows)\b", + re.I, +) +INSTRUCTION_START_RE = re.compile( + r"^(?:select|click|choose|enter|input|open|close|press|perform|to\s+|on\s+|" + r"the\s+user|users?\s+|next\s+|then\s+|for\s+ease|option/tool|" + r"用户|点击|輸入|输入|選擇|选择|填写|當|当|在|首先|然后|然後|配置|系统|系統|" + r"若|如果|注|详细|詳細|列表|机构|機構|添加|删除|刪除)", + re.I, ) FIGURE_TABLE_RE = re.compile( r"^\s*(?:Figure|Fig\.|图|表|Table)\s*[\d.]+", @@ -38,12 +56,39 @@ def _rendered_len(blocks: list[Block]) -> int: def numbered_heading_level(text: str) -> int | None: - match = NUMBERED_HEADING_RE.match(text.strip()) + stripped = text.strip() + match = PURE_NUMBERED_HEADING_RE.match(stripped) + if match: + return match.group(1).count(".") + 1 + match = NUMBERED_HEADING_RE.match(stripped) if not match: return None return match.group(1).count(".") + 1 +def looks_like_numbered_instruction(text: str) -> bool: + """True for numbered procedure/list sentences, not outline headings.""" + stripped = text.strip() + if NUMBERED_OPTION_SENTENCE_RE.match(stripped): + return True + match = NUMBERED_LIST_ITEM_RE.match(stripped) + if not match: + return False + body = match.group("body").strip() + if INSTRUCTION_START_RE.match(body): + return True + return len(body) >= 45 and bool(re.search(r"[,.,。;;:]", body)) + + +def looks_like_false_heading_text(text: str) -> bool: + stripped = text.strip() + if TIME_LIKE_RE.match(stripped) or looks_like_numbered_instruction(stripped): + return True + if numbered_heading_level(stripped): + return False + return len(stripped) >= 40 and bool(re.search(r"[.!?。!?;;]\s*$", stripped)) + + def infer_heading_level(block: Block) -> int: if block.type == BlockType.HEADING and block.level: numbered = numbered_heading_level(block.text) @@ -60,6 +105,8 @@ def is_heading_block(block: Block) -> bool: text = _text(block) if not text or len(text) > MAX_HEADING_CHARS: return False + if looks_like_false_heading_text(text): + return False if block.type == BlockType.HEADING: return True if NUMBERED_HEADING_RE.match(text): @@ -71,14 +118,21 @@ def is_heading_block(block: Block) -> bool: def normalize_heading_block(block: Block) -> Block: text = _text(block) - if block.type == BlockType.HEADING and len(text) > MAX_HEADING_CHARS: + false_heading = looks_like_false_heading_text(text) + if block.type == BlockType.HEADING and ( + len(text) > MAX_HEADING_CHARS or false_heading + ): # Parser sometimes merges title+body then marks the blob as heading. return Block(type=BlockType.PARAGRAPH, text=text, meta=dict(block.meta)) + if false_heading: + return block numbered = numbered_heading_level(text) if block.type == BlockType.HEADING: level = numbered or block.level or 1 return block.model_copy(update={"level": level}) - if numbered and NUMBERED_HEADING_RE.match(text): + if numbered and ( + NUMBERED_HEADING_RE.match(text) or PURE_NUMBERED_HEADING_RE.match(text) + ): return Block( type=BlockType.HEADING, text=text, diff --git a/backend/rag_cut/splitters/pdf_semantic.py b/backend/rag_cut/splitters/pdf_semantic.py index 1741cbf..3f87853 100644 --- a/backend/rag_cut/splitters/pdf_semantic.py +++ b/backend/rag_cut/splitters/pdf_semantic.py @@ -8,12 +8,20 @@ from dataclasses import dataclass, field from rag_cut.models import Block, BlockType, Chunk, SplitConfig from rag_cut.parsers.pdf.noise_filter import is_toc_noise_text, is_toc_title_text from rag_cut.renderer import collect_chunk_layout_meta, render_blocks +from rag_cut.splitters.heading_splitter import ( + PURE_NUMBERED_HEADING_RE, + TIME_LIKE_RE, + looks_like_false_heading_text, + looks_like_numbered_instruction, + normalize_heading_block, +) CHUNK_STRATEGY = "heading_layout_multimodal" PAGE_NUMBER_RE = re.compile(r"^\s*(?:[-\u2013\u2014]?\s*)?\d{1,4}(?:\s*/\s*\d{1,4})?\s*$") NUMBERED_HEADING_RE = re.compile( - r"^\s*(?P\d+(?:\.\d+)*)(?:\.|.)?\s*(?P[A-Za-z0-9\u4e00-\u9fff][^\n]{1,120})\s*$" + r"^\s*(?P<num>\d+(?:\.\d+)*)(?:[..、::]\s*|\s+)" + r"(?P<title>[A-Za-z\u4e00-\u9fff][^\n]{1,120})\s*$" ) LETTER_HEADING_RE = re.compile( r"^\s*(?P<letter>[A-Z])[\.)]\s+(?P<title>[A-Za-z0-9\u4e00-\u9fff][^\n]{1,100})\s*$" @@ -132,7 +140,7 @@ def _is_noise(block: Block, running_headers: set[str]) -> bool: text = " ".join(_text(block).split()) if block.type == BlockType.IMAGE: return _is_decorative_image(block) - if PAGE_NUMBER_RE.match(text): + if PAGE_NUMBER_RE.match(text) or TIME_LIKE_RE.match(text): return True if _is_toc_text(text): return True @@ -146,19 +154,29 @@ def _heading_signal(block: Block, current_top_level: bool = False) -> _HeadingSi if not text or len(text) > 180 or _is_toc_text(text): return None # Keep procedural steps inside the parent section; do not open a new group. - if STEP_INSTRUCTION_RE.match(text): + if ( + STEP_INSTRUCTION_RE.match(text) + or looks_like_numbered_instruction(text) + or looks_like_false_heading_text(text) + ): return None + pure_numbered = PURE_NUMBERED_HEADING_RE.match(text) + if block.type == BlockType.HEADING: level = block.level or 1 numbered = NUMBERED_HEADING_RE.match(text) - if numbered: + if pure_numbered: + level = pure_numbered.group(1).count(".") + 1 + elif numbered: level = numbered.group("num").count(".") + 1 elif LETTER_HEADING_RE.match(text): level = 2 if current_top_level else max(2, level) return _HeadingSignal(text=" ".join(text.split()), level=max(1, min(level, 6))) numbered = NUMBERED_HEADING_RE.match(text) + if pure_numbered: + return _HeadingSignal(text=" ".join(text.split()), level=pure_numbered.group(1).count(".") + 1) if numbered: return _HeadingSignal(text=" ".join(text.split()), level=numbered.group("num").count(".") + 1) if LETTER_HEADING_RE.match(text): @@ -339,7 +357,7 @@ def _split_oversized_group(group: _Group, start_index: int, config: SplitConfig) def flush() -> None: nonlocal current, current_len, part_index - body = [b for b in current if b not in prefix] + body = [b for b in current if b.type != BlockType.HEADING] if not body: return part_group = _Group(group.heading_path, list(current)) @@ -362,7 +380,17 @@ def _split_oversized_group(group: _Group, start_index: int, config: SplitConfig) for block in group.blocks[len(prefix) :]: block_len = len(block.render()) + 2 if current_len + block_len > config.max_chunk_size and len(current) > len(prefix): - flush() + context_len = sum( + len(item.render()) + 2 + for item in current + if item.type != BlockType.HEADING + ) + keep_atomic_context = ( + block.type in {BlockType.IMAGE, BlockType.TABLE} + and context_len <= min(400, max(80, config.max_chunk_size // 2)) + ) + if not keep_atomic_context: + flush() current.append(block) current_len += block_len flush() @@ -386,9 +414,10 @@ def _build_groups(blocks: list[Block]) -> list[_Group]: _attach_heading_only_to_previous(groups, current) current = _Group() - for raw in blocks: - if _is_noise(raw, running_headers): + for source in blocks: + if _is_noise(source, running_headers): continue + raw = normalize_heading_block(source) signal = _heading_signal(raw, current_top_level=bool(heading_stack)) if signal: @@ -399,16 +428,11 @@ def _build_groups(blocks: list[Block]) -> list[_Group]: heading_stack.append((signal.level, signal.text, heading)) path = [item[1] for item in heading_stack] current = _Group(heading_path=path) - for _, _, h_block in heading_stack: - current.blocks.append(_enrich_block(h_block, path, current.blocks)) + current.blocks.append(_enrich_block(heading, path, current.blocks)) continue path = [item[1] for item in heading_stack] - if not current.blocks and heading_stack: - current.heading_path = path - for _, _, h_block in heading_stack: - current.blocks.append(_enrich_block(h_block, path, current.blocks)) - elif not current.heading_path: + if not current.heading_path: current.heading_path = path current.blocks.append(_enrich_block(raw, current.heading_path, current.blocks)) diff --git a/backend/tests/test_heading_layout_multimodal.py b/backend/tests/test_heading_layout_multimodal.py index c67cf9a..708d814 100644 --- a/backend/tests/test_heading_layout_multimodal.py +++ b/backend/tests/test_heading_layout_multimodal.py @@ -59,6 +59,100 @@ def table(page: int = 1, y: int = 220) -> Block: class HeadingLayoutMultimodalTest(unittest.TestCase): + def test_numbered_instructions_remain_body_text(self) -> None: + chunks = split_pdf_semantic( + [ + h("1 Data Platform", level=1), + p("1. 用户点击发布则会显示对应的数据发布弹窗", y=140), + p("(1) 发布路径:下拉选择菜单,只可单选", y=170), + h("1.1 发布管理", level=2, y=220), + p("发布管理正文。", y=250), + ], + SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=2000), + ) + + self.assertEqual([chunk.meta["heading"] for chunk in chunks], ["1 Data Platform", "1.1 发布管理"]) + self.assertIn("1. 用户点击发布", chunks[0].content) + self.assertNotIn("# 1. 用户点击发布", chunks[0].content) + + def test_numbered_option_description_is_not_a_heading(self) -> None: + option = h( + "10.4 Default: This option enables you to set default Qty, Account and/or Best Price.", + level=2, + y=160, + ) + option.meta["source_heading"] = True + chunks = split_pdf_semantic( + [h("10 Option settings", level=1), option, p("Following option details.", y=200)], + SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=2000), + ) + + self.assertEqual(len(chunks), 1) + self.assertEqual(chunks[0].meta["heading"], "10 Option settings") + self.assertIn("10.4 Default: This option enables", chunks[0].content) + self.assertNotIn("## 10.4 Default", chunks[0].content) + + def test_pure_numeric_heading_keeps_its_real_depth(self) -> None: + numeric = h("1.1.1", level=2, page=2) + numeric.meta["source_heading"] = True + next_section = h("1.2 ACCOUNT STATUS CODE MASTER", level=2, page=5) + next_section.meta["source_heading"] = True + chunks = split_pdf_semantic( + [numeric, table(page=2), next_section, p("Status body.", page=5)], + SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=4000), + ) + + self.assertEqual([chunk.meta["heading"] for chunk in chunks], ["1.1.1", "1.2 ACCOUNT STATUS CODE MASTER"]) + self.assertNotIn("1.1.1", chunks[1].content) + self.assertEqual(chunks[1].meta["pages"], [5]) + + def test_clock_text_is_not_a_heading_or_ancestor(self) -> None: + clock = h("15:18:53", level=2, page=8) + clock.meta["source_heading"] = True + chunks = split_pdf_semantic( + [clock, h("第三节 买入/卖出序", level=2, page=9), p("Useful body.", page=9)], + SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=2000), + ) + + self.assertEqual(len(chunks), 1) + self.assertEqual(chunks[0].meta["heading"], "第三节 买入/卖出序") + self.assertNotIn("15:18:53", chunks[0].content) + self.assertEqual(chunks[0].meta["pages"], [9]) + + def test_parent_heading_is_metadata_not_repeated_page_block(self) -> None: + parent = h("1 Parent", level=1, page=1) + parent.meta["source_heading"] = True + child = h("1.1 Child", level=2, page=2) + child.meta["source_heading"] = True + chunks = split_pdf_semantic( + [parent, child, p("Child body.", page=2)], + SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=2000), + ) + + self.assertEqual(len(chunks), 1) + self.assertEqual(chunks[0].meta["heading_path"], ["1 Parent", "1.1 Child"]) + self.assertNotIn("# 1 Parent", chunks[0].content) + self.assertEqual(chunks[0].meta["pages"], [2]) + + def test_oversized_table_keeps_nearest_heading_and_intro(self) -> None: + big_table = table(page=1, y=180) + big_table.markdown += "\n" + "| value | description |\n" * 80 + chunks = split_pdf_semantic( + [ + h("2 Settings", level=1), + p("The fields are listed below.", y=140), + big_table, + p("Following explanation " + "x" * 180, y=340), + p("Final paragraph " + "y" * 180, y=380), + ], + SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=260), + ) + + table_chunks = [chunk for chunk in chunks if "table" in chunk.block_types] + self.assertEqual(len(table_chunks), 1) + self.assertIn("2 Settings", table_chunks[0].content) + self.assertIn("The fields are listed below.", table_chunks[0].content) + self.assertFalse(any(chunk.block_types and all(t == "heading" for t in chunk.block_types) for chunk in chunks)) def test_numbered_headings_create_same_level_boundaries(self) -> None: chunks = split_pdf_semantic( [ diff --git a/backend/tests/test_mineru_adapter.py b/backend/tests/test_mineru_adapter.py index fb3707f..511c7b4 100644 --- a/backend/tests/test_mineru_adapter.py +++ b/backend/tests/test_mineru_adapter.py @@ -134,6 +134,21 @@ class MineruAdapterMappingTest(unittest.TestCase): self.assertEqual(blocks[0].type, BlockType.PARAGRAPH) self.assertTrue(blocks[0].meta.get("is_footnote")) + def test_table_caption_becomes_table_title(self) -> None: + items = [ + { + "type": "table", + "table_body": "| Field | Description |\n| --- | --- |\n| id | Identifier |", + "table_caption": ["第一节 综合资讯栏"], + "page_idx": 7, + } + ] + + blocks = _blocks_from_content_list(items, Path("content_list.json"), Path("assets")) + + self.assertEqual(len(blocks), 1) + self.assertEqual(blocks[0].meta.get("table_title"), "第一节 综合资讯栏") + def test_list_items_become_paragraph(self) -> None: items = [ { diff --git a/backend/tests/test_pdf_noise_filter.py b/backend/tests/test_pdf_noise_filter.py index 817f90f..6676d4d 100644 --- a/backend/tests/test_pdf_noise_filter.py +++ b/backend/tests/test_pdf_noise_filter.py @@ -15,6 +15,7 @@ from rag_cut.parsers.pdf.noise_filter import ( is_margin_noise_block, is_tiny_image_block, is_toc_entry_line, + is_toc_noise_text, is_toc_title_text, ) @@ -190,6 +191,37 @@ class NoiseFilterHeuristicTest(unittest.TestCase): self.assertTrue(is_toc_entry_line("5.1 AEOI ID SETTING.. 4")) self.assertEqual(filter_toc_blocks([contents]), []) + def test_toc_entries_accept_real_manual_page_formats(self) -> None: + self.assertTrue(is_toc_entry_line("第一章 簡介 ........ - 5 -")) + self.assertTrue(is_toc_entry_line("4 格報價畫面(x4 報價畫面) ........ - 18 -")) + self.assertTrue(is_toc_entry_line("1.2. How to Trade an Odd Lot. .16")) + self.assertTrue(is_toc_entry_line("CHAPTER 1 LET'S START TRADING .8")) + + def test_filter_toc_drops_long_compacted_manual_contents(self) -> None: + contents = Block( + type=BlockType.PARAGRAPH, + text=( + "CHAPTER 1 LET'S START TRADING .8\n" + "1.1. How to Trade A Normal Stock. . 9\n" + "1.2. How to Trade an Odd Lot. .16\n" + "1.3. How to Trade BuyIn, OddLot BuyIN and OFFmarket .18\n" + "CHAPTER 2 MANAGING THE ORDER BOOK. . 41" + ), + meta={"page": 6, "bbox": [85, 103, 808, 907]}, + ) + + self.assertTrue(is_toc_noise_text(contents.text)) + self.assertEqual(filter_toc_blocks([contents]), []) + + def test_toc_detects_multiple_entries_merged_on_one_line(self) -> None: + merged = ( + "登入 忘记密码 ................................ - 6 -" + "画面介绍 ................................ - 7 -" + "综合画面 ................................ - 18 -" + ) + + self.assertTrue(is_toc_noise_text(merged)) + def test_filter_toc_keeps_real_section_below_directory_on_same_page(self) -> None: blocks = [ Block(