@@ -0,0 +1,89 @@
|
||||
"""Spreadsheet parser (xlsx/xls/csv) producing row-oriented table blocks."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import csv
|
||||
from pathlib import Path
|
||||
|
||||
from rag_cut.models import Block, BlockType
|
||||
from rag_cut.parsers.base import BaseParser
|
||||
from rag_cut.parsers.pdf.tables import (
|
||||
build_table_embedding_text,
|
||||
detect_spreadsheet_layout,
|
||||
extract_table_keywords,
|
||||
normalize_rows,
|
||||
rows_to_markdown,
|
||||
)
|
||||
|
||||
|
||||
def _read_csv(path: Path) -> list[list[str]]:
|
||||
for encoding in ("utf-8-sig", "utf-8", "gbk", "latin-1"):
|
||||
try:
|
||||
with open(path, newline="", encoding=encoding) as f:
|
||||
return [list(row) for row in csv.reader(f)]
|
||||
except UnicodeDecodeError:
|
||||
continue
|
||||
raise ValueError(f"Cannot decode CSV: {path}")
|
||||
|
||||
|
||||
def _read_xlsx(path: Path, sheet_name: str | None = None) -> tuple[str, list[list[str]]]:
|
||||
import openpyxl
|
||||
|
||||
wb = openpyxl.load_workbook(path, read_only=True, data_only=True)
|
||||
name = sheet_name or wb.sheetnames[0]
|
||||
ws = wb[name]
|
||||
rows: list[list[str]] = []
|
||||
for row in ws.iter_rows(values_only=True):
|
||||
rows.append(["" if v is None else str(v) for v in row])
|
||||
wb.close()
|
||||
# trim trailing empty rows/cols
|
||||
while rows and all(not c for c in rows[-1]):
|
||||
rows.pop()
|
||||
return name, rows
|
||||
|
||||
|
||||
class SpreadsheetParser(BaseParser):
|
||||
def parse(self, path: Path, assets_dir: Path) -> list[Block]:
|
||||
ext = path.suffix.lower()
|
||||
if ext == ".csv":
|
||||
rows = _read_csv(path)
|
||||
sheet_name = path.stem
|
||||
else:
|
||||
sheet_name, rows = _read_xlsx(path)
|
||||
|
||||
if not rows:
|
||||
return []
|
||||
|
||||
normalized = normalize_rows(rows)
|
||||
layout = detect_spreadsheet_layout(normalized)
|
||||
preamble = layout["preamble_rows"]
|
||||
header_rows = layout["header_rows"]
|
||||
display_rows = normalized[preamble:]
|
||||
md = rows_to_markdown(display_rows, header_rows=header_rows)
|
||||
description = ""
|
||||
if preamble:
|
||||
description = " ".join(cell for cell in normalized[0] if cell)[:500]
|
||||
keywords = extract_table_keywords(sheet_name, md, description)
|
||||
embedding_text = build_table_embedding_text(
|
||||
table_title=sheet_name,
|
||||
markdown=md,
|
||||
description=description,
|
||||
keywords=keywords,
|
||||
)
|
||||
return [
|
||||
Block(
|
||||
type=BlockType.TABLE,
|
||||
markdown=md,
|
||||
meta={
|
||||
"sheet": sheet_name,
|
||||
"table_title": sheet_name,
|
||||
"row_count": len(normalized),
|
||||
"col_count": max(len(r) for r in normalized),
|
||||
"rows": normalized,
|
||||
"keywords": keywords,
|
||||
"embedding_text": embedding_text,
|
||||
"table_description": description or None,
|
||||
**layout,
|
||||
},
|
||||
)
|
||||
]
|
||||
Reference in New Issue
Block a user