90 lines
2.9 KiB
Python
90 lines
2.9 KiB
Python
"""Spreadsheet parser (xlsx/xls/csv) producing row-oriented table blocks."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import csv
|
|
from pathlib import Path
|
|
|
|
from rag_cut.models import Block, BlockType
|
|
from rag_cut.parsers.base import BaseParser
|
|
from rag_cut.parsers.pdf.tables import (
|
|
build_table_embedding_text,
|
|
detect_spreadsheet_layout,
|
|
extract_table_keywords,
|
|
normalize_rows,
|
|
rows_to_markdown,
|
|
)
|
|
|
|
|
|
def _read_csv(path: Path) -> list[list[str]]:
|
|
for encoding in ("utf-8-sig", "utf-8", "gbk", "latin-1"):
|
|
try:
|
|
with open(path, newline="", encoding=encoding) as f:
|
|
return [list(row) for row in csv.reader(f)]
|
|
except UnicodeDecodeError:
|
|
continue
|
|
raise ValueError(f"Cannot decode CSV: {path}")
|
|
|
|
|
|
def _read_xlsx(path: Path, sheet_name: str | None = None) -> tuple[str, list[list[str]]]:
|
|
import openpyxl
|
|
|
|
wb = openpyxl.load_workbook(path, read_only=True, data_only=True)
|
|
name = sheet_name or wb.sheetnames[0]
|
|
ws = wb[name]
|
|
rows: list[list[str]] = []
|
|
for row in ws.iter_rows(values_only=True):
|
|
rows.append(["" if v is None else str(v) for v in row])
|
|
wb.close()
|
|
# trim trailing empty rows/cols
|
|
while rows and all(not c for c in rows[-1]):
|
|
rows.pop()
|
|
return name, rows
|
|
|
|
|
|
class SpreadsheetParser(BaseParser):
|
|
def parse(self, path: Path, assets_dir: Path) -> list[Block]:
|
|
ext = path.suffix.lower()
|
|
if ext == ".csv":
|
|
rows = _read_csv(path)
|
|
sheet_name = path.stem
|
|
else:
|
|
sheet_name, rows = _read_xlsx(path)
|
|
|
|
if not rows:
|
|
return []
|
|
|
|
normalized = normalize_rows(rows)
|
|
layout = detect_spreadsheet_layout(normalized)
|
|
preamble = layout["preamble_rows"]
|
|
header_rows = layout["header_rows"]
|
|
display_rows = normalized[preamble:]
|
|
md = rows_to_markdown(display_rows, header_rows=header_rows)
|
|
description = ""
|
|
if preamble:
|
|
description = " ".join(cell for cell in normalized[0] if cell)[:500]
|
|
keywords = extract_table_keywords(sheet_name, md, description)
|
|
embedding_text = build_table_embedding_text(
|
|
table_title=sheet_name,
|
|
markdown=md,
|
|
description=description,
|
|
keywords=keywords,
|
|
)
|
|
return [
|
|
Block(
|
|
type=BlockType.TABLE,
|
|
markdown=md,
|
|
meta={
|
|
"sheet": sheet_name,
|
|
"table_title": sheet_name,
|
|
"row_count": len(normalized),
|
|
"col_count": max(len(r) for r in normalized),
|
|
"rows": normalized,
|
|
"keywords": keywords,
|
|
"embedding_text": embedding_text,
|
|
"table_description": description or None,
|
|
**layout,
|
|
},
|
|
)
|
|
]
|