43 lines
1.3 KiB
Python
43 lines
1.3 KiB
Python
"""Run chunking on all sample files in data/."""
|
|||
|
|
|
||
|
|
from __future__ import annotations
|
||
|
|
|
||
|
|
import sys
|
||
|
|
from pathlib import Path
|
||
|
|
|
||
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "backend"))
|
||
|
|
|
||
|
|
from rag_cut.models import SplitConfig, SplitMode
|
||
|
|
from rag_cut.pipeline import chunk_document
|
||
|
|
|
||
|
|
DATA = Path(__file__).resolve().parent.parent / "data"
|
||
|
|
OUT = Path(__file__).resolve().parent.parent / "storage"
|
||
|
|
|
||
|
|
|
||
|
|
def main() -> None:
|
||
|
|
for path in sorted(DATA.iterdir()):
|
||
|
|
if not path.is_file():
|
||
|
|
continue
|
||
|
|
ext = path.suffix.lower()
|
||
|
|
if ext == ".xlsx":
|
||
|
|
config = SplitConfig(mode=SplitMode.BY_ROW, rows_per_chunk=5)
|
||
|
|
else:
|
||
|
|
config = SplitConfig(mode=SplitMode.DEFAULT)
|
||
|
|
|
||
|
|
print(f"\n{'=' * 60}\n{path.name} ({ext})")
|
||
|
|
result = chunk_document(path, config=config)
|
||
|
|
print(f"blocks={result.block_count} chunks={result.chunk_count} mode={result.split_mode.value}")
|
||
|
|
|
||
|
|
out_file = OUT / f"{path.stem}_chunks.json"
|
||
|
|
out_file.write_text(result.model_dump_json(indent=2), encoding="utf-8")
|
||
|
|
print(f"saved -> {out_file.name}")
|
||
|
|
|
||
|
|
if result.chunks:
|
||
|
|
c0 = result.chunks[0]
|
||
|
|
print(f"chunk[0] types={c0.block_types} chars={c0.char_count}")
|
||
|
|
print(c0.content[:300].replace("\n", " ") + "...")
|
||
|
|
|
||
|
|
|
||
|
|
if __name__ == "__main__":
|
||
|
|
main()
|