Files
RAG-CUT/backend/rag_cut/parsers/pdf/ocr.py
T
2026-07-16 11:12:17 +08:00

58 lines
1.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Optional OCR for cropped image/table regions."""
from __future__ import annotations
import io
from pathlib import Path
import fitz
def ocr_pixmap(pix: fitz.Pixmap) -> str:
"""Run OCR on a pixmap; returns empty string when OCR is unavailable."""
try:
import pytesseract
from PIL import Image
except ImportError:
return ""
try:
image = Image.open(io.BytesIO(pix.tobytes("png")))
return _ocr_pil(image)
except Exception:
return ""
def ocr_image_file(path: Path) -> str:
try:
import pytesseract # noqa: F401
from PIL import Image
except ImportError:
return ""
try:
return _ocr_pil(Image.open(path))
except Exception:
return ""
def _ocr_pil(image) -> str:
import pytesseract
for lang in ("chi_tra+eng", "chi_sim+eng", "eng"):
try:
text = pytesseract.image_to_string(image, lang=lang)
cleaned = " ".join(text.split())
if cleaned:
return cleaned
except Exception:
continue
return ""
def describe_visual(ocr_text: str, kind: str, label: str) -> str:
"""Lightweight image/table description without an external vision model."""
if ocr_text:
return f"{kind}:{ocr_text[:300]}"
return f"{kind}:{label}"