Files
RAG-CUT/backend/rag_cut/parsers/xlsx_parser.py
T
2026-07-16 11:12:17 +08:00

90 lines
2.9 KiB
Python

"""Spreadsheet parser (xlsx/xls/csv) producing row-oriented table blocks."""
from __future__ import annotations
import csv
from pathlib import Path
from rag_cut.models import Block, BlockType
from rag_cut.parsers.base import BaseParser
from rag_cut.parsers.pdf.tables import (
build_table_embedding_text,
detect_spreadsheet_layout,
extract_table_keywords,
normalize_rows,
rows_to_markdown,
)
def _read_csv(path: Path) -> list[list[str]]:
for encoding in ("utf-8-sig", "utf-8", "gbk", "latin-1"):
try:
with open(path, newline="", encoding=encoding) as f:
return [list(row) for row in csv.reader(f)]
except UnicodeDecodeError:
continue
raise ValueError(f"Cannot decode CSV: {path}")
def _read_xlsx(path: Path, sheet_name: str | None = None) -> tuple[str, list[list[str]]]:
import openpyxl
wb = openpyxl.load_workbook(path, read_only=True, data_only=True)
name = sheet_name or wb.sheetnames[0]
ws = wb[name]
rows: list[list[str]] = []
for row in ws.iter_rows(values_only=True):
rows.append(["" if v is None else str(v) for v in row])
wb.close()
# trim trailing empty rows/cols
while rows and all(not c for c in rows[-1]):
rows.pop()
return name, rows
class SpreadsheetParser(BaseParser):
def parse(self, path: Path, assets_dir: Path) -> list[Block]:
ext = path.suffix.lower()
if ext == ".csv":
rows = _read_csv(path)
sheet_name = path.stem
else:
sheet_name, rows = _read_xlsx(path)
if not rows:
return []
normalized = normalize_rows(rows)
layout = detect_spreadsheet_layout(normalized)
preamble = layout["preamble_rows"]
header_rows = layout["header_rows"]
display_rows = normalized[preamble:]
md = rows_to_markdown(display_rows, header_rows=header_rows)
description = ""
if preamble:
description = " ".join(cell for cell in normalized[0] if cell)[:500]
keywords = extract_table_keywords(sheet_name, md, description)
embedding_text = build_table_embedding_text(
table_title=sheet_name,
markdown=md,
description=description,
keywords=keywords,
)
return [
Block(
type=BlockType.TABLE,
markdown=md,
meta={
"sheet": sheet_name,
"table_title": sheet_name,
"row_count": len(normalized),
"col_count": max(len(r) for r in normalized),
"rows": normalized,
"keywords": keywords,
"embedding_text": embedding_text,
"table_description": description or None,
**layout,
},
)
]