39 lines
1.3 KiB
Python
39 lines
1.3 KiB
Python
"""Extract text and page summaries from requirements PDF."""
|
|
from pathlib import Path
|
|
|
|
import fitz
|
|
|
|
PDF = Path(__file__).resolve().parent.parent / "docx" / "自研搭建AI助手知识库.pdf"
|
|
OUT = Path(__file__).resolve().parent.parent / "docx" / "extracted_requirements.txt"
|
|
IMG_DIR = Path(__file__).resolve().parent.parent / "docx" / "pdf_pages"
|
|
|
|
|
|
def main() -> None:
|
|
IMG_DIR.mkdir(parents=True, exist_ok=True)
|
|
doc = fitz.open(PDF)
|
|
lines = [f"Pages: {len(doc)}", f"Path: {PDF}", ""]
|
|
|
|
for i, page in enumerate(doc):
|
|
text = page.get_text("text").strip()
|
|
imgs = page.get_images()
|
|
blocks = page.get_text("dict")["blocks"]
|
|
text_blocks = sum(1 for b in blocks if b.get("type") == 0)
|
|
img_path = IMG_DIR / f"page_{i + 1:02d}.png"
|
|
pix = page.get_pixmap(matrix=fitz.Matrix(2, 2))
|
|
pix.save(str(img_path))
|
|
|
|
lines.append(f"{'=' * 60}")
|
|
lines.append(f"PAGE {i + 1} | images={len(imgs)} text_blocks={text_blocks}")
|
|
lines.append(f"{'=' * 60}")
|
|
lines.append(text if text else "[no extractable text - see pdf_pages/]")
|
|
lines.append("")
|
|
|
|
doc.close()
|
|
OUT.write_text("\n".join(lines), encoding="utf-8")
|
|
print(f"Wrote {OUT}")
|
|
print(f"Rendered {len(list(IMG_DIR.glob('*.png')))} page images to {IMG_DIR}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|