Files
RAG-CUT/scripts/extract_requirements_pdf.py
T

39 lines
1.3 KiB
Python
Raw Normal View History

2026-07-16 11:12:17 +08:00
"""Extract text and page summaries from requirements PDF."""
from pathlib import Path
import fitz
PDF = Path(__file__).resolve().parent.parent / "docx" / "自研搭建AI助手知识库.pdf"
OUT = Path(__file__).resolve().parent.parent / "docx" / "extracted_requirements.txt"
IMG_DIR = Path(__file__).resolve().parent.parent / "docx" / "pdf_pages"
def main() -> None:
IMG_DIR.mkdir(parents=True, exist_ok=True)
doc = fitz.open(PDF)
lines = [f"Pages: {len(doc)}", f"Path: {PDF}", ""]
for i, page in enumerate(doc):
text = page.get_text("text").strip()
imgs = page.get_images()
blocks = page.get_text("dict")["blocks"]
text_blocks = sum(1 for b in blocks if b.get("type") == 0)
img_path = IMG_DIR / f"page_{i + 1:02d}.png"
pix = page.get_pixmap(matrix=fitz.Matrix(2, 2))
pix.save(str(img_path))
lines.append(f"{'=' * 60}")
lines.append(f"PAGE {i + 1} | images={len(imgs)} text_blocks={text_blocks}")
lines.append(f"{'=' * 60}")
lines.append(text if text else "[no extractable text - see pdf_pages/]")
lines.append("")
doc.close()
OUT.write_text("\n".join(lines), encoding="utf-8")
print(f"Wrote {OUT}")
print(f"Rendered {len(list(IMG_DIR.glob('*.png')))} page images to {IMG_DIR}")
if __name__ == "__main__":
main()