124 lines
3.9 KiB
Python
124 lines
3.9 KiB
Python
"""Plain-text and markup parsers."""
|
|||
|
|
|
||
|
|
from __future__ import annotations
|
||
|
|
|
||
|
|
import json
|
||
|
|
import re
|
||
|
|
from html.parser import HTMLParser
|
||
|
|
from pathlib import Path
|
||
|
|
|
||
|
|
from rag_cut.models import Block, BlockType
|
||
|
|
from rag_cut.parsers.base import BaseParser
|
||
|
|
|
||
|
|
|
||
|
|
class _HTMLTextExtractor(HTMLParser):
|
||
|
|
def __init__(self) -> None:
|
||
|
|
super().__init__()
|
||
|
|
self._parts: list[str] = []
|
||
|
|
self._heading: tuple[int, str] | None = None
|
||
|
|
self._blocks: list[Block] = []
|
||
|
|
self._current: list[str] = []
|
||
|
|
self._in_heading = False
|
||
|
|
|
||
|
|
def handle_starttag(self, tag: str, attrs) -> None:
|
||
|
|
if tag in ("h1", "h2", "h3", "h4", "h5", "h6"):
|
||
|
|
self._flush_paragraph()
|
||
|
|
self._in_heading = True
|
||
|
|
self._heading_level = int(tag[1])
|
||
|
|
|
||
|
|
def handle_endtag(self, tag: str) -> None:
|
||
|
|
if tag in ("h1", "h2", "h3", "h4", "h5", "h6"):
|
||
|
|
text = "".join(self._current).strip()
|
||
|
|
self._current = []
|
||
|
|
self._in_heading = False
|
||
|
|
if text:
|
||
|
|
self._blocks.append(
|
||
|
|
Block(type=BlockType.HEADING, text=text, level=self._heading_level)
|
||
|
|
)
|
||
|
|
elif tag in ("p", "div", "br", "li"):
|
||
|
|
self._flush_paragraph()
|
||
|
|
|
||
|
|
def handle_data(self, data: str) -> None:
|
||
|
|
self._current.append(data)
|
||
|
|
|
||
|
|
def _flush_paragraph(self) -> None:
|
||
|
|
text = "".join(self._current).strip()
|
||
|
|
self._current = []
|
||
|
|
if text:
|
||
|
|
self._blocks.append(Block(type=BlockType.PARAGRAPH, text=text))
|
||
|
|
|
||
|
|
def get_blocks(self) -> list[Block]:
|
||
|
|
self._flush_paragraph()
|
||
|
|
return self._blocks
|
||
|
|
|
||
|
|
|
||
|
|
def _parse_markdown(text: str) -> list[Block]:
|
||
|
|
blocks: list[Block] = []
|
||
|
|
for line in text.splitlines():
|
||
|
|
stripped = line.strip()
|
||
|
|
if not stripped:
|
||
|
|
continue
|
||
|
|
m = re.match(r"^(#{1,6})\s+(.+)$", stripped)
|
||
|
|
if m:
|
||
|
|
blocks.append(
|
||
|
|
Block(type=BlockType.HEADING, text=m.group(2).strip(), level=len(m.group(1)))
|
||
|
|
)
|
||
|
|
else:
|
||
|
|
blocks.append(Block(type=BlockType.PARAGRAPH, text=stripped))
|
||
|
|
return blocks
|
||
|
|
|
||
|
|
|
||
|
|
def _parse_json(text: str) -> list[Block]:
|
||
|
|
data = json.loads(text)
|
||
|
|
blocks: list[Block] = []
|
||
|
|
if isinstance(data, list):
|
||
|
|
for i, item in enumerate(data):
|
||
|
|
blocks.append(
|
||
|
|
Block(
|
||
|
|
type=BlockType.CODE,
|
||
|
|
text=json.dumps(item, ensure_ascii=False, indent=2),
|
||
|
|
meta={"json_index": i},
|
||
|
|
)
|
||
|
|
)
|
||
|
|
elif isinstance(data, dict):
|
||
|
|
for key, value in data.items():
|
||
|
|
blocks.append(
|
||
|
|
Block(
|
||
|
|
type=BlockType.CODE,
|
||
|
|
text=json.dumps({key: value}, ensure_ascii=False, indent=2),
|
||
|
|
meta={"json_key": key},
|
||
|
|
)
|
||
|
|
)
|
||
|
|
else:
|
||
|
|
blocks.append(Block(type=BlockType.PARAGRAPH, text=str(data)))
|
||
|
|
return blocks
|
||
|
|
|
||
|
|
|
||
|
|
class TextParser(BaseParser):
|
||
|
|
def parse(self, path: Path, assets_dir: Path) -> list[Block]:
|
||
|
|
for encoding in ("utf-8-sig", "utf-8", "gbk", "latin-1"):
|
||
|
|
try:
|
||
|
|
text = path.read_text(encoding=encoding)
|
||
|
|
break
|
||
|
|
except UnicodeDecodeError:
|
||
|
|
continue
|
||
|
|
else:
|
||
|
|
raise ValueError(f"Cannot decode text file: {path}")
|
||
|
|
|
||
|
|
ext = path.suffix.lower()
|
||
|
|
if ext == ".md":
|
||
|
|
return _parse_markdown(text)
|
||
|
|
if ext in (".html", ".htm"):
|
||
|
|
parser = _HTMLTextExtractor()
|
||
|
|
parser.feed(text)
|
||
|
|
return parser.get_blocks()
|
||
|
|
if ext == ".json":
|
||
|
|
return _parse_json(text)
|
||
|
|
# txt, xml, log — paragraph split on blank lines
|
||
|
|
blocks: list[Block] = []
|
||
|
|
for para in re.split(r"\n\s*\n", text):
|
||
|
|
para = para.strip()
|
||
|
|
if para:
|
||
|
|
blocks.append(Block(type=BlockType.PARAGRAPH, text=para))
|
||
|
|
return blocks
|