Enhance table title handling and improve heading detection logic
- Added logic to set `table_title` from `caption` if it is under 160 characters and not already set. - Updated `_blocks_from_content_list` to assign `table_title` based on `caption` length. - Introduced new regex patterns for better detection of TOC entries and noise. - Enhanced heading detection to differentiate between numbered instructions and actual headings. - Added tests to verify that table captions are correctly assigned as titles and that numbered instructions are treated as body text.
This commit is contained in:
@@ -59,6 +59,100 @@ def table(page: int = 1, y: int = 220) -> Block:
|
||||
|
||||
|
||||
class HeadingLayoutMultimodalTest(unittest.TestCase):
|
||||
def test_numbered_instructions_remain_body_text(self) -> None:
|
||||
chunks = split_pdf_semantic(
|
||||
[
|
||||
h("1 Data Platform", level=1),
|
||||
p("1. 用户点击发布则会显示对应的数据发布弹窗", y=140),
|
||||
p("(1) 发布路径:下拉选择菜单,只可单选", y=170),
|
||||
h("1.1 发布管理", level=2, y=220),
|
||||
p("发布管理正文。", y=250),
|
||||
],
|
||||
SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=2000),
|
||||
)
|
||||
|
||||
self.assertEqual([chunk.meta["heading"] for chunk in chunks], ["1 Data Platform", "1.1 发布管理"])
|
||||
self.assertIn("1. 用户点击发布", chunks[0].content)
|
||||
self.assertNotIn("# 1. 用户点击发布", chunks[0].content)
|
||||
|
||||
def test_numbered_option_description_is_not_a_heading(self) -> None:
|
||||
option = h(
|
||||
"10.4 Default: This option enables you to set default Qty, Account and/or Best Price.",
|
||||
level=2,
|
||||
y=160,
|
||||
)
|
||||
option.meta["source_heading"] = True
|
||||
chunks = split_pdf_semantic(
|
||||
[h("10 Option settings", level=1), option, p("Following option details.", y=200)],
|
||||
SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=2000),
|
||||
)
|
||||
|
||||
self.assertEqual(len(chunks), 1)
|
||||
self.assertEqual(chunks[0].meta["heading"], "10 Option settings")
|
||||
self.assertIn("10.4 Default: This option enables", chunks[0].content)
|
||||
self.assertNotIn("## 10.4 Default", chunks[0].content)
|
||||
|
||||
def test_pure_numeric_heading_keeps_its_real_depth(self) -> None:
|
||||
numeric = h("1.1.1", level=2, page=2)
|
||||
numeric.meta["source_heading"] = True
|
||||
next_section = h("1.2 ACCOUNT STATUS CODE MASTER", level=2, page=5)
|
||||
next_section.meta["source_heading"] = True
|
||||
chunks = split_pdf_semantic(
|
||||
[numeric, table(page=2), next_section, p("Status body.", page=5)],
|
||||
SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=4000),
|
||||
)
|
||||
|
||||
self.assertEqual([chunk.meta["heading"] for chunk in chunks], ["1.1.1", "1.2 ACCOUNT STATUS CODE MASTER"])
|
||||
self.assertNotIn("1.1.1", chunks[1].content)
|
||||
self.assertEqual(chunks[1].meta["pages"], [5])
|
||||
|
||||
def test_clock_text_is_not_a_heading_or_ancestor(self) -> None:
|
||||
clock = h("15:18:53", level=2, page=8)
|
||||
clock.meta["source_heading"] = True
|
||||
chunks = split_pdf_semantic(
|
||||
[clock, h("第三节 买入/卖出序", level=2, page=9), p("Useful body.", page=9)],
|
||||
SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=2000),
|
||||
)
|
||||
|
||||
self.assertEqual(len(chunks), 1)
|
||||
self.assertEqual(chunks[0].meta["heading"], "第三节 买入/卖出序")
|
||||
self.assertNotIn("15:18:53", chunks[0].content)
|
||||
self.assertEqual(chunks[0].meta["pages"], [9])
|
||||
|
||||
def test_parent_heading_is_metadata_not_repeated_page_block(self) -> None:
|
||||
parent = h("1 Parent", level=1, page=1)
|
||||
parent.meta["source_heading"] = True
|
||||
child = h("1.1 Child", level=2, page=2)
|
||||
child.meta["source_heading"] = True
|
||||
chunks = split_pdf_semantic(
|
||||
[parent, child, p("Child body.", page=2)],
|
||||
SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=2000),
|
||||
)
|
||||
|
||||
self.assertEqual(len(chunks), 1)
|
||||
self.assertEqual(chunks[0].meta["heading_path"], ["1 Parent", "1.1 Child"])
|
||||
self.assertNotIn("# 1 Parent", chunks[0].content)
|
||||
self.assertEqual(chunks[0].meta["pages"], [2])
|
||||
|
||||
def test_oversized_table_keeps_nearest_heading_and_intro(self) -> None:
|
||||
big_table = table(page=1, y=180)
|
||||
big_table.markdown += "\n" + "| value | description |\n" * 80
|
||||
chunks = split_pdf_semantic(
|
||||
[
|
||||
h("2 Settings", level=1),
|
||||
p("The fields are listed below.", y=140),
|
||||
big_table,
|
||||
p("Following explanation " + "x" * 180, y=340),
|
||||
p("Final paragraph " + "y" * 180, y=380),
|
||||
],
|
||||
SplitConfig(mode=SplitMode.DEFAULT, max_chunk_size=260),
|
||||
)
|
||||
|
||||
table_chunks = [chunk for chunk in chunks if "table" in chunk.block_types]
|
||||
self.assertEqual(len(table_chunks), 1)
|
||||
self.assertIn("2 Settings", table_chunks[0].content)
|
||||
self.assertIn("The fields are listed below.", table_chunks[0].content)
|
||||
self.assertFalse(any(chunk.block_types and all(t == "heading" for t in chunk.block_types) for chunk in chunks))
|
||||
def test_numbered_headings_create_same_level_boundaries(self) -> None:
|
||||
chunks = split_pdf_semantic(
|
||||
[
|
||||
|
||||
Reference in New Issue
Block a user