"""Extract text and page summaries from requirements PDF.""" from pathlib import Path import fitz PDF = Path(__file__).resolve().parent.parent / "docx" / "自研搭建AI助手知识库.pdf" OUT = Path(__file__).resolve().parent.parent / "docx" / "extracted_requirements.txt" IMG_DIR = Path(__file__).resolve().parent.parent / "docx" / "pdf_pages" def main() -> None: IMG_DIR.mkdir(parents=True, exist_ok=True) doc = fitz.open(PDF) lines = [f"Pages: {len(doc)}", f"Path: {PDF}", ""] for i, page in enumerate(doc): text = page.get_text("text").strip() imgs = page.get_images() blocks = page.get_text("dict")["blocks"] text_blocks = sum(1 for b in blocks if b.get("type") == 0) img_path = IMG_DIR / f"page_{i + 1:02d}.png" pix = page.get_pixmap(matrix=fitz.Matrix(2, 2)) pix.save(str(img_path)) lines.append(f"{'=' * 60}") lines.append(f"PAGE {i + 1} | images={len(imgs)} text_blocks={text_blocks}") lines.append(f"{'=' * 60}") lines.append(text if text else "[no extractable text - see pdf_pages/]") lines.append("") doc.close() OUT.write_text("\n".join(lines), encoding="utf-8") print(f"Wrote {OUT}") print(f"Rendered {len(list(IMG_DIR.glob('*.png')))} page images to {IMG_DIR}") if __name__ == "__main__": main()