Enhance table title handling and improve heading detection logic
- Added logic to set `table_title` from `caption` if it is under 160 characters and not already set. - Updated `_blocks_from_content_list` to assign `table_title` based on `caption` length. - Introduced new regex patterns for better detection of TOC entries and noise. - Enhanced heading detection to differentiate between numbered instructions and actual headings. - Added tests to verify that table captions are correctly assigned as titles and that numbered instructions are treated as body text.
This commit is contained in:
@@ -56,6 +56,23 @@ _TOC_NUMBERED_PAGE_RE = re.compile(
|
||||
r"\d{1,4}\s*$"
|
||||
)
|
||||
_MD_TOC_LINK_RE = re.compile(r"^\s*[-*+]\s+\[[^\]]+\]\([^)]+\)\s*$")
|
||||
_TOC_DASH_PAGE_RE = re.compile(
|
||||
r"^\s*.{2,160}?(?:\.{1,}|\u2026+|\s{2,})\s*-\s*\d{1,4}\s*-\s*$"
|
||||
)
|
||||
_TOC_OUTLINE_PAGE_RE = re.compile(
|
||||
r"^\s*(?:chapter\s+\d+|\d+(?:\.\d+)*\.?|"
|
||||
r"\u7b2c[\u4e00-\u9fff\d]+[\u7ae0\u8282\u7bc0])\s+"
|
||||
r".{1,140}?(?:\.{1,}|\u2026+|\s{2,})\s*\d{1,4}\s*$",
|
||||
re.I,
|
||||
)
|
||||
_PAGE_LABEL_RE = re.compile(
|
||||
r"^\s*(?:page\s*)?-?\s*\d{1,4}\s*-?\s*(?:of|/)\s*\d{1,4}\s*$",
|
||||
re.I,
|
||||
)
|
||||
_COMPACT_TOC_PAGE_REF_RE = re.compile(
|
||||
r"(?:\.{2,}|\u2026+)\s*-\s*\d{1,4}\s*-",
|
||||
re.I,
|
||||
)
|
||||
|
||||
|
||||
def _looks_like_numbered_section(text: str) -> bool:
|
||||
@@ -81,6 +98,10 @@ def is_toc_entry_line(text: str) -> bool:
|
||||
return False
|
||||
if _MD_TOC_LINK_RE.match(stripped):
|
||||
return True
|
||||
if _TOC_DASH_PAGE_RE.match(stripped):
|
||||
return True
|
||||
if _TOC_OUTLINE_PAGE_RE.match(stripped):
|
||||
return True
|
||||
if _DOT_LEADER_RE.search(stripped) and re.search(r"\d\s*$", stripped):
|
||||
return True
|
||||
if _TOC_ENTRY_LINE_RE.match(stripped):
|
||||
@@ -97,6 +118,11 @@ def is_toc_noise_text(text: str) -> bool:
|
||||
return False
|
||||
if is_toc_title_text(stripped):
|
||||
return True
|
||||
# PyMuPDF may merge several TOC rows into one physical line. Repeated
|
||||
# leader + "- page -" references are directory evidence even without
|
||||
# newline boundaries.
|
||||
if len(_COMPACT_TOC_PAGE_REF_RE.findall(stripped)) >= 2:
|
||||
return True
|
||||
lines = [ln.strip() for ln in stripped.splitlines() if ln.strip()]
|
||||
if not lines:
|
||||
return False
|
||||
@@ -361,14 +387,16 @@ def is_margin_noise_block(block: Block, page_height: float | None) -> bool:
|
||||
|
||||
bbox = block.meta.get("bbox")
|
||||
if not page_height or not bbox or len(bbox) < 4:
|
||||
return block.type == BlockType.PARAGRAPH and bool(_PAGE_NUM_RE.match(text))
|
||||
return block.type == BlockType.PARAGRAPH and bool(
|
||||
_PAGE_NUM_RE.match(text) or _PAGE_LABEL_RE.match(text)
|
||||
)
|
||||
|
||||
in_header = _in_margin_zone(bbox, page_height, header=True)
|
||||
in_footer = _in_margin_zone(bbox, page_height, header=False)
|
||||
if not in_header and not in_footer:
|
||||
return False
|
||||
|
||||
if _PAGE_NUM_RE.match(text):
|
||||
if _PAGE_NUM_RE.match(text) or _PAGE_LABEL_RE.match(text):
|
||||
return True
|
||||
|
||||
if block.type != BlockType.PARAGRAPH:
|
||||
|
||||
Reference in New Issue
Block a user