"""Spreadsheet parser (xlsx/xls/csv) producing row-oriented table blocks.""" from __future__ import annotations import csv from pathlib import Path from rag_cut.models import Block, BlockType from rag_cut.parsers.base import BaseParser from rag_cut.parsers.pdf.tables import ( build_table_embedding_text, detect_spreadsheet_layout, extract_table_keywords, normalize_rows, rows_to_markdown, ) def _read_csv(path: Path) -> list[list[str]]: for encoding in ("utf-8-sig", "utf-8", "gbk", "latin-1"): try: with open(path, newline="", encoding=encoding) as f: return [list(row) for row in csv.reader(f)] except UnicodeDecodeError: continue raise ValueError(f"Cannot decode CSV: {path}") def _read_xlsx(path: Path, sheet_name: str | None = None) -> tuple[str, list[list[str]]]: import openpyxl wb = openpyxl.load_workbook(path, read_only=True, data_only=True) name = sheet_name or wb.sheetnames[0] ws = wb[name] rows: list[list[str]] = [] for row in ws.iter_rows(values_only=True): rows.append(["" if v is None else str(v) for v in row]) wb.close() # trim trailing empty rows/cols while rows and all(not c for c in rows[-1]): rows.pop() return name, rows class SpreadsheetParser(BaseParser): def parse(self, path: Path, assets_dir: Path) -> list[Block]: ext = path.suffix.lower() if ext == ".csv": rows = _read_csv(path) sheet_name = path.stem else: sheet_name, rows = _read_xlsx(path) if not rows: return [] normalized = normalize_rows(rows) layout = detect_spreadsheet_layout(normalized) preamble = layout["preamble_rows"] header_rows = layout["header_rows"] display_rows = normalized[preamble:] md = rows_to_markdown(display_rows, header_rows=header_rows) description = "" if preamble: description = " ".join(cell for cell in normalized[0] if cell)[:500] keywords = extract_table_keywords(sheet_name, md, description) embedding_text = build_table_embedding_text( table_title=sheet_name, markdown=md, description=description, keywords=keywords, ) return [ Block( type=BlockType.TABLE, markdown=md, meta={ "sheet": sheet_name, "table_title": sheet_name, "row_count": len(normalized), "col_count": max(len(r) for r in normalized), "rows": normalized, "keywords": keywords, "embedding_text": embedding_text, "table_description": description or None, **layout, }, ) ]