58 lines
1.3 KiB
Python
58 lines
1.3 KiB
Python
"""Optional OCR for cropped image/table regions."""
|
||
|
||
from __future__ import annotations
|
||
|
||
import io
|
||
from pathlib import Path
|
||
|
||
import fitz
|
||
|
||
|
||
def ocr_pixmap(pix: fitz.Pixmap) -> str:
|
||
"""Run OCR on a pixmap; returns empty string when OCR is unavailable."""
|
||
try:
|
||
import pytesseract
|
||
from PIL import Image
|
||
except ImportError:
|
||
return ""
|
||
|
||
try:
|
||
image = Image.open(io.BytesIO(pix.tobytes("png")))
|
||
return _ocr_pil(image)
|
||
except Exception:
|
||
return ""
|
||
|
||
|
||
def ocr_image_file(path: Path) -> str:
|
||
try:
|
||
import pytesseract # noqa: F401
|
||
from PIL import Image
|
||
except ImportError:
|
||
return ""
|
||
|
||
try:
|
||
return _ocr_pil(Image.open(path))
|
||
except Exception:
|
||
return ""
|
||
|
||
|
||
def _ocr_pil(image) -> str:
|
||
import pytesseract
|
||
|
||
for lang in ("chi_tra+eng", "chi_sim+eng", "eng"):
|
||
try:
|
||
text = pytesseract.image_to_string(image, lang=lang)
|
||
cleaned = " ".join(text.split())
|
||
if cleaned:
|
||
return cleaned
|
||
except Exception:
|
||
continue
|
||
return ""
|
||
|
||
|
||
def describe_visual(ocr_text: str, kind: str, label: str) -> str:
|
||
"""Lightweight image/table description without an external vision model."""
|
||
if ocr_text:
|
||
return f"{kind}:{ocr_text[:300]}"
|
||
return f"{kind}:{label}"
|