Files
SnapOtter/packages/ai/python/ocr_pdf.py
T

197 lines
6.5 KiB
Python
Raw Normal View History

"""OCR text extraction from PDF documents.
Rasterizes each page via PyMuPDF (fitz) and runs the same OCR engine
pipeline as ocr.py (tesseract / PaddleOCR PP-OCRv5 / PaddleOCR-VL).
Contract: argv[1] = pdf path, argv[2] = options JSON {"quality","language","pages"}.
Output: {"success": true, "engine": ..., "pages": n, "text": ...} to stdout.
Errors: {"error": ...} to stdout, exit 1.
Engine functions are imported from sibling ocr.py (the dispatcher adds the
scripts directory to sys.path, and `import ocr` loads ocr.py as a module
without triggering its __main__ guard).
"""
import sys
import json
import os
import tempfile
MAX_PAGES = 50
def emit_progress(percent, stage):
"""Emit structured progress to stderr for bridge.ts to capture."""
print(json.dumps({"progress": percent, "stage": stage}), file=sys.stderr, flush=True)
def parse_page_spec(spec, total_pages):
"""Parse a page spec like 'all', '1-3,5', '2,4-6' into a sorted list of 0-based indices.
Returns (pages, error) where error is a string if the spec is invalid.
"""
if spec.strip().lower() == "all":
return list(range(total_pages)), None
pages = set()
parts = spec.split(",")
for part in parts:
part = part.strip()
if not part:
continue
if "-" in part:
bounds = part.split("-", 1)
try:
start = int(bounds[0].strip())
end = int(bounds[1].strip())
except ValueError:
return None, f"Invalid page range: {part}"
if start < 1 or end < 1:
return None, f"Invalid page range: {part} (pages start at 1)"
if start > end:
return None, f"Invalid page range: {part} (start > end)"
if end > total_pages:
return None, f"Invalid page range: {part} (document has {total_pages} pages)"
for i in range(start, end + 1):
pages.add(i - 1) # 0-based
else:
try:
num = int(part)
except ValueError:
return None, f"Invalid page number: {part}"
if num < 1:
return None, f"Invalid page number: {num} (pages start at 1)"
if num > total_pages:
return None, f"Invalid page number: {num} (document has {total_pages} pages)"
pages.add(num - 1) # 0-based
if not pages:
return None, "No pages specified"
return sorted(pages), None
def main():
input_path = sys.argv[1]
settings = json.loads(sys.argv[2]) if len(sys.argv) > 2 else {}
quality = settings.get("quality", "balanced")
language = settings.get("language", "auto")
pages_spec = settings.get("pages", "all")
try:
emit_progress(2, "Opening PDF")
# Lazy import: fitz (PyMuPDF) is in the base venv but keep it lazy
# so py_compile succeeds without it.
import fitz
doc = fitz.open(input_path)
total_pages = len(doc)
if total_pages == 0:
print(json.dumps({"error": "PDF has no pages"}))
sys.exit(1)
# Parse page specification
page_indices, parse_error = parse_page_spec(pages_spec, total_pages)
if parse_error:
print(json.dumps({"error": parse_error}))
sys.exit(1)
if len(page_indices) > MAX_PAGES:
print(json.dumps({"error": f"Too many pages for OCR (max {MAX_PAGES})"}))
sys.exit(1)
emit_progress(5, "Preparing")
# Import engine functions from sibling ocr.py.
# The dispatcher's scripts directory is on sys.path, so this import
# loads ocr.py as a module (its __main__ guard prevents re-execution).
import ocr as ocr_module
# Language auto-detection on the first page's raster
was_auto = language == "auto"
# Create a temp directory for scratch PNGs next to the input
scratch_dir = tempfile.mkdtemp(
prefix="ocr_pdf_",
dir=os.path.dirname(input_path) or tempfile.gettempdir(),
)
page_texts = []
engine_used = quality
num_pages = len(page_indices)
for i, page_idx in enumerate(page_indices):
page_num = page_idx + 1 # 1-based for display
pct = 5 + int((i / num_pages) * 85)
emit_progress(pct, f"Page {page_num}/{total_pages}")
# Rasterize page to PNG at 200 DPI
page = doc[page_idx]
pix = page.get_pixmap(dpi=200)
png_path = os.path.join(scratch_dir, f"page_{page_idx}.png")
pix.save(png_path)
# Auto-detect language on the first page only
current_lang = language
if was_auto and i == 0:
current_lang = ocr_module.auto_detect_language(png_path)
language = current_lang # reuse for remaining pages
elif was_auto:
current_lang = language # already detected
# Always route PDF OCR through tesseract. PaddleOCR segfaults
# on arm64 CPU when processing rasterised PDF pages (SIGSEGV in
# the doc-orientation / structural-analysis stage). Tesseract is
# reliable for page-level images and is already installed in the
# container with multi-language packs. The image-OCR tool still
# offers PaddleOCR tiers for single images where it is stable.
if quality not in ("fast", "balanced", "best"):
print(json.dumps({"error": f"Unknown quality: {quality}"}))
sys.exit(1)
text = ocr_module.run_tesseract(png_path, current_lang, is_auto=was_auto)
engine_used = "tesseract"
page_texts.append((page_num, text))
# Clean up scratch PNG immediately
try:
os.remove(png_path)
except OSError:
pass
# Clean up scratch directory
try:
os.rmdir(scratch_dir)
except OSError:
pass
doc.close()
# Join page texts with page headers
parts = []
for page_num, text in page_texts:
parts.append(f"\n\n--- Page {page_num} ---\n\n{text}")
# Strip leading newlines from the first page
full_text = "".join(parts).strip()
emit_progress(95, "Done")
print(json.dumps({
"success": True,
"engine": engine_used,
"pages": num_pages,
"text": full_text,
}))
except Exception as e:
print(json.dumps({"error": str(e)}))
sys.exit(1)
if __name__ == "__main__":
main()