@@ -0,0 +1,72 @@
|
||||
#!/usr/bin/env python3
|
||||
"""CLI to chunk documents and print/save results."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "backend"))
|
||||
|
||||
from rag_cut.models import SplitConfig, SplitMode
|
||||
from rag_cut.pipeline import chunk_document
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="RAG-cut document chunking CLI")
|
||||
parser.add_argument("file", type=Path, help="Path to document")
|
||||
parser.add_argument(
|
||||
"--mode",
|
||||
choices=[m.value for m in SplitMode],
|
||||
default="default",
|
||||
help="Split mode",
|
||||
)
|
||||
parser.add_argument("--delimiter", default=None, help="Delimiter for delimiter mode")
|
||||
parser.add_argument("--parent-delimiter", default=None, help="Parent delimiter for parent_child mode")
|
||||
parser.add_argument("--child-delimiter", default=None, help="Child delimiter for parent_child mode")
|
||||
parser.add_argument("--max-chunk-size", type=int, default=1500, help="Parent/primary max chunk size")
|
||||
parser.add_argument("--child-max-size", type=int, default=512, help="Child max size for parent_child mode")
|
||||
parser.add_argument("--overlap", type=int, default=150)
|
||||
parser.add_argument("--header-row-start", type=int, default=1)
|
||||
parser.add_argument("--header-row-end", type=int, default=1)
|
||||
parser.add_argument("--start-row", type=int, default=2)
|
||||
parser.add_argument("--rows-per-chunk", type=int, default=1)
|
||||
parser.add_argument("-o", "--output", type=Path, default=None, help="Save JSON result")
|
||||
parser.add_argument("--preview", type=int, default=3, help="Print first N chunks")
|
||||
args = parser.parse_args()
|
||||
|
||||
config = SplitConfig(
|
||||
mode=SplitMode(args.mode),
|
||||
delimiter=args.delimiter,
|
||||
parent_delimiter=args.parent_delimiter,
|
||||
child_delimiter=args.child_delimiter,
|
||||
max_chunk_size=args.max_chunk_size,
|
||||
child_max_size=args.child_max_size,
|
||||
overlap=args.overlap,
|
||||
header_row_start=args.header_row_start,
|
||||
header_row_end=args.header_row_end,
|
||||
start_row=args.start_row,
|
||||
rows_per_chunk=args.rows_per_chunk,
|
||||
)
|
||||
|
||||
result = chunk_document(args.file, config=config)
|
||||
print(f"File: {result.filename}")
|
||||
print(f"Doc ID: {result.doc_id}")
|
||||
print(f"Blocks: {result.block_count} -> Chunks: {result.chunk_count}")
|
||||
print(f"Mode: {result.split_mode.value}")
|
||||
print("-" * 60)
|
||||
|
||||
for chunk in result.chunks[: args.preview]:
|
||||
print(f"\n[Chunk {chunk.index}] ({chunk.char_count} chars) types={chunk.block_types}")
|
||||
preview = chunk.content[:500]
|
||||
print(preview + ("..." if len(chunk.content) > 500 else ""))
|
||||
|
||||
if args.output:
|
||||
args.output.write_text(result.model_dump_json(indent=2), encoding="utf-8")
|
||||
print(f"\nSaved full result to {args.output}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,38 @@
|
||||
"""Extract text and page summaries from requirements PDF."""
|
||||
from pathlib import Path
|
||||
|
||||
import fitz
|
||||
|
||||
PDF = Path(__file__).resolve().parent.parent / "docx" / "自研搭建AI助手知识库.pdf"
|
||||
OUT = Path(__file__).resolve().parent.parent / "docx" / "extracted_requirements.txt"
|
||||
IMG_DIR = Path(__file__).resolve().parent.parent / "docx" / "pdf_pages"
|
||||
|
||||
|
||||
def main() -> None:
|
||||
IMG_DIR.mkdir(parents=True, exist_ok=True)
|
||||
doc = fitz.open(PDF)
|
||||
lines = [f"Pages: {len(doc)}", f"Path: {PDF}", ""]
|
||||
|
||||
for i, page in enumerate(doc):
|
||||
text = page.get_text("text").strip()
|
||||
imgs = page.get_images()
|
||||
blocks = page.get_text("dict")["blocks"]
|
||||
text_blocks = sum(1 for b in blocks if b.get("type") == 0)
|
||||
img_path = IMG_DIR / f"page_{i + 1:02d}.png"
|
||||
pix = page.get_pixmap(matrix=fitz.Matrix(2, 2))
|
||||
pix.save(str(img_path))
|
||||
|
||||
lines.append(f"{'=' * 60}")
|
||||
lines.append(f"PAGE {i + 1} | images={len(imgs)} text_blocks={text_blocks}")
|
||||
lines.append(f"{'=' * 60}")
|
||||
lines.append(text if text else "[no extractable text - see pdf_pages/]")
|
||||
lines.append("")
|
||||
|
||||
doc.close()
|
||||
OUT.write_text("\n".join(lines), encoding="utf-8")
|
||||
print(f"Wrote {OUT}")
|
||||
print(f"Rendered {len(list(IMG_DIR.glob('*.png')))} page images to {IMG_DIR}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,14 @@
|
||||
$ErrorActionPreference = 'Stop'
|
||||
$out = 'C:\Users\24019\Desktop\RAG-cut\data\test_sample.pptx'
|
||||
$pp = New-Object -ComObject PowerPoint.Application
|
||||
try {
|
||||
$pres = $pp.Presentations.Add()
|
||||
$slide = $pres.Slides.Add(1, 1)
|
||||
$slide.Shapes.Title.TextFrame.TextRange.Text = 'RAG-cut Test Slide'
|
||||
$slide.Shapes.Item(2).TextFrame.TextRange.Text = 'Body text for chunking test.'
|
||||
$pres.SaveAs($out)
|
||||
$pres.Close()
|
||||
} finally {
|
||||
$pp.Quit()
|
||||
}
|
||||
Write-Output "Saved $out"
|
||||
@@ -0,0 +1,59 @@
|
||||
"""Quick probe of sample documents for chunking strategy planning."""
|
||||
import re
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
DATA = Path(__file__).resolve().parent.parent / "data"
|
||||
|
||||
|
||||
def probe_docx(p: Path) -> None:
|
||||
with zipfile.ZipFile(p) as z:
|
||||
media = [n for n in z.namelist() if n.startswith("word/media/")]
|
||||
xml = z.read("word/document.xml").decode("utf-8", errors="ignore")
|
||||
headings = len(re.findall(r'w:pStyle w:val="Heading', xml))
|
||||
tables = xml.count("<w:tbl")
|
||||
drawings = xml.count("<w:drawing") + xml.count("<w:pict")
|
||||
print(f" media: {len(media)}, tables: {tables}, drawings: {drawings}, heading_styles: {headings}")
|
||||
|
||||
|
||||
def probe_pdf(p: Path) -> None:
|
||||
import fitz
|
||||
|
||||
doc = fitz.open(p)
|
||||
imgs = sum(len(doc[i].get_images()) for i in range(len(doc)))
|
||||
text_len = sum(len(doc[i].get_text()) for i in range(len(doc)))
|
||||
print(f" pages: {len(doc)}, embedded_images: {imgs}, text_chars: {text_len}")
|
||||
doc.close()
|
||||
|
||||
|
||||
def probe_xlsx(p: Path) -> None:
|
||||
import openpyxl
|
||||
|
||||
wb = openpyxl.load_workbook(p, read_only=True, data_only=True)
|
||||
print(f" sheets: {len(wb.sheetnames)} -> {wb.sheetnames[:5]}")
|
||||
for sn in wb.sheetnames[:2]:
|
||||
ws = wb[sn]
|
||||
print(f" {sn}: rows={ws.max_row}, cols={ws.max_column}")
|
||||
wb.close()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
for p in sorted(DATA.iterdir()):
|
||||
if not p.is_file():
|
||||
continue
|
||||
print(f"=== {p.name} ({p.stat().st_size} bytes) ===")
|
||||
ext = p.suffix.lower()
|
||||
try:
|
||||
if ext == ".docx":
|
||||
probe_docx(p)
|
||||
elif ext == ".pdf":
|
||||
probe_pdf(p)
|
||||
elif ext in (".xlsx", ".xls"):
|
||||
probe_xlsx(p)
|
||||
except Exception as e:
|
||||
print(f" ERROR: {e}")
|
||||
print()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,42 @@
|
||||
"""Run chunking on all sample files in data/."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "backend"))
|
||||
|
||||
from rag_cut.models import SplitConfig, SplitMode
|
||||
from rag_cut.pipeline import chunk_document
|
||||
|
||||
DATA = Path(__file__).resolve().parent.parent / "data"
|
||||
OUT = Path(__file__).resolve().parent.parent / "storage"
|
||||
|
||||
|
||||
def main() -> None:
|
||||
for path in sorted(DATA.iterdir()):
|
||||
if not path.is_file():
|
||||
continue
|
||||
ext = path.suffix.lower()
|
||||
if ext == ".xlsx":
|
||||
config = SplitConfig(mode=SplitMode.BY_ROW, rows_per_chunk=5)
|
||||
else:
|
||||
config = SplitConfig(mode=SplitMode.DEFAULT)
|
||||
|
||||
print(f"\n{'=' * 60}\n{path.name} ({ext})")
|
||||
result = chunk_document(path, config=config)
|
||||
print(f"blocks={result.block_count} chunks={result.chunk_count} mode={result.split_mode.value}")
|
||||
|
||||
out_file = OUT / f"{path.stem}_chunks.json"
|
||||
out_file.write_text(result.model_dump_json(indent=2), encoding="utf-8")
|
||||
print(f"saved -> {out_file.name}")
|
||||
|
||||
if result.chunks:
|
||||
c0 = result.chunks[0]
|
||||
print(f"chunk[0] types={c0.block_types} chars={c0.char_count}")
|
||||
print(c0.content[:300].replace("\n", " ") + "...")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user