mirror of
https://github.com/snapotter-hq/SnapOtter.git
synced 2026-08-03 07:46:42 +02:00
* fix(docker): pin CUDA base to 12.6 so the GPU image starts on R560+ drivers The amd64 base nvidia/cuda:12.9.2-cudnn-runtime bakes a cuda>=12.9 driver gate enforced by nvidia-container-toolkit at container start, so the image fails to launch on common production drivers (e.g. 570.x / CUDA 12.8). The AI bundles are all cu126 wheels and the image installs libcublas-12-6, so 12.9 was misaligned with the workload. Pin to nvidia/cuda:12.6.3-cudnn-runtime-ubuntu24.04 to match the wheels and lower the driver floor to R560+. * fix(ai): broaden OOM detection so the rembg lighter-model fallback fires onnxruntime/CUDA allocation failures surface as 'Failed to allocate memory for requested buffer', CUBLAS_STATUS_ALLOC_FAILED, or bad_alloc, not just 'out of memory'. The background-removal and transparency-fixer fallback-to-lighter-model paths only matched the literal 'out of memory', so the fallback was dead code and transparency-fixer (default birefnet-hr-matting) always failed with an allocation error. Add isMemoryAllocError() and use it in both checks. * fix(ai): use bundled PaddleOCR models so OCR runs offline ocr.py passed no model dirs to PaddleOCR, so PaddleX resolved models from ~/.paddlex and downloaded them from HuggingFace at runtime (slow first use, broken air-gapped), ignoring the models the OCR bundle ships in MODELS_PATH; it also pulled doc-orientation/unwarping models that are not bundled. Pin detection, recognition and textline models to the bundled dirs in MODELS_PATH (per language) and disable use_doc_orientation_classify / use_doc_unwarping, with per-component fallback when a model is absent. Verified: OCR runs with zero HuggingFace requests. * fix(docker): add CAP_KILL so container shutdown is graceful cap_drop: ALL without re-adding KILL meant tini (PID 1, root) could not forward SIGTERM to the gosu-dropped snapotter process (root minus CAP_KILL cannot signal a different UID). docker stop logged '[FATAL tini] forwarding signal: Operation not permitted', never delivered the signal, and fell back to SIGKILL after the 10s timeout. Add KILL to cap_add in both compose files. Verified: docker stop completes in 0s with SIGTERM delivered (exit 143) and no FATAL tini. * fix(ai): serialize bundle installs against AI jobs to prevent sidecar segfault A feature bundle install rewrites the shared Python venv (pip + copytree of site-packages/*.so) as a background subprocess, with no coordination against AI tool jobs that dlopen native libs (torch / onnxruntime CUDA) from the same venv; a job loading a shared object while it is overwritten segfaults the sidecar. Add a process-wide async mutex (venv-lock.ts): bridge.run() acquires it before every AI script and the install route holds it across the installer subprocess. Both run in the same Node process so a module-level lock suffices. Verified: concurrent install + AI job produces zero segfaults and the job serializes behind the install. * fix(ai): make the venv lock read/write so concurrent AI jobs are not serialized The first cut used an exclusive mutex, which (a) deferred the dispatcher spawn by a microtask and broke unit tests that synchronously drive the mocked spawn, and (b) serialized AI jobs against each other, removing the dispatcher's by-id request multiplexing. Make it a writer-preferring read/write lock: AI jobs are shared readers (with a synchronous fast path so spawn still happens in-tick) and a bundle install is the exclusive writer. Verified: all 764 AI unit tests pass. * fix(ai): degrade OCR to Tesseract on CPU-only hosts instead of segfaulting The amd64 AI bundle ships paddlepaddle-gpu, whose native libs dlopen libcuda.so.1 at import and segfault on a host without a GPU (libcuda is the driver lib, injected only by nvidia-container-toolkit on GPU hosts). The segfault crashed the shared long-lived AI dispatcher and, after a few attempts, tripped the bridge crash-recovery permanent-disable, wedging all AI until a container restart. The standalone ocr tool defaults to quality=balanced (PaddleOCR), so it hit this on every CPU-only deployment; ocr-pdf already hardcoded Tesseract and was unaffected. ocr.py now gates the PaddleOCR tiers on gpu_available(): balanced/best transparently fall back to fast (Tesseract, CPU-capable) when no usable GPU is present, and run_paddleocr_v5/run_paddleocr_vl refuse before importing paddle so the GPU build is never dlopen'd on CPU. GPU hosts are unchanged. Verified on a CPU-only Windows/WSL2 box: ocr returns Tesseract text across repeated runs with the dispatcher staying healthy (no wedge).
429 lines
15 KiB
Python
429 lines
15 KiB
Python
"""Text extraction from images using Tesseract, PaddleOCR PP-OCRv5, or PaddleOCR-VL 1.5."""
|
|
import sys
|
|
import json
|
|
import os
|
|
|
|
# Prevent PaddlePaddle C++ runtime from probing for CUDA on CPU-only systems.
|
|
# Without these, paddlepaddle-gpu can segfault during import on machines without
|
|
# a GPU, because the C++ layer attempts GPU initialization before Python-level
|
|
# device routing takes effect. Must run before any PaddleOCR import.
|
|
from gpu import gpu_available
|
|
if not gpu_available():
|
|
if not os.environ.get("FLAGS_use_cuda"):
|
|
os.environ["FLAGS_use_cuda"] = "0"
|
|
if not os.environ.get("FLAGS_use_cudnn"):
|
|
os.environ["FLAGS_use_cudnn"] = "0"
|
|
|
|
# Lazy-loaded VLM instance (stays resident in dispatcher process)
|
|
_paddleocr_vl_instance = None
|
|
|
|
|
|
def emit_progress(percent, stage):
|
|
"""Emit structured progress to stderr for bridge.ts to capture."""
|
|
print(json.dumps({"progress": percent, "stage": stage}), file=sys.stderr, flush=True)
|
|
|
|
|
|
# OCR quality tiers backed by PaddleOCR. PaddleOCR ships as the GPU build
|
|
# (paddlepaddle-gpu) in the amd64 bundle; its native libs dlopen libcuda.so.1 at
|
|
# import and segfault on a CPU-only host, so these tiers need a usable GPU.
|
|
PADDLE_QUALITY_TIERS = ("balanced", "best")
|
|
|
|
|
|
def effective_quality(requested):
|
|
"""Return the OCR quality tier that can actually run on this host.
|
|
|
|
On a CPU-only host the PaddleOCR tiers (balanced/best) cannot load, so they
|
|
transparently fall back to "fast" (Tesseract), which runs on CPU. This keeps a
|
|
GPU-less host from importing paddlepaddle-gpu, whose import segfaults and wedges
|
|
the shared AI dispatcher.
|
|
"""
|
|
if requested in PADDLE_QUALITY_TIERS and not gpu_available():
|
|
return "fast"
|
|
return requested
|
|
|
|
|
|
TESSERACT_LANG_MAP = {
|
|
"en": "eng", "de": "deu", "fr": "fra", "es": "spa",
|
|
"zh": "chi_sim", "ja": "jpn", "ko": "kor",
|
|
}
|
|
|
|
PADDLE_LANG_MAP = {
|
|
"en": "en", "de": "latin", "fr": "latin", "es": "latin",
|
|
"zh": "ch", "ja": "japan", "ko": "korean",
|
|
}
|
|
|
|
# Bundled PaddleOCR models (shipped by the OCR feature bundle into MODELS_PATH).
|
|
# Pinning the constructor at these dirs keeps OCR fully offline / air-gapped and
|
|
# skips slow HuggingFace model resolution on first use. PP-OCRv5 server rec
|
|
# covers Chinese+English; latin covers en/de/fr/es; plus a dedicated Korean rec.
|
|
PADDLE_DET_MODEL = "PP-OCRv5_server_det"
|
|
PADDLE_TEXTLINE_MODEL = "PP-LCNet_x1_0_textline_ori"
|
|
PADDLE_REC_MODEL = {
|
|
"ch": "PP-OCRv5_server_rec",
|
|
"en": "latin_PP-OCRv5_mobile_rec",
|
|
"latin": "latin_PP-OCRv5_mobile_rec",
|
|
"korean": "korean_PP-OCRv5_mobile_rec",
|
|
}
|
|
|
|
|
|
def _bundled_paddle_kwargs(paddle_lang):
|
|
"""Build PaddleOCR kwargs that use the bundled models in MODELS_PATH.
|
|
|
|
Only pins a component when its model is actually present on disk, so a
|
|
partial bundle (or an unbundled language such as Japanese) falls back to
|
|
PaddleOCR's default resolution for that component. The doc-orientation and
|
|
doc-unwarping models are not bundled and not needed for plain OCR, so they
|
|
are disabled to avoid a runtime HuggingFace download.
|
|
"""
|
|
models_dir = os.environ.get("MODELS_PATH", "/data/ai/models")
|
|
|
|
def model_dir(name):
|
|
if not name:
|
|
return None
|
|
path = os.path.join(models_dir, name)
|
|
return path if os.path.isdir(path) else None
|
|
|
|
kwargs = {"use_doc_orientation_classify": False, "use_doc_unwarping": False}
|
|
|
|
det = model_dir(PADDLE_DET_MODEL)
|
|
if det:
|
|
kwargs["text_detection_model_name"] = PADDLE_DET_MODEL
|
|
kwargs["text_detection_model_dir"] = det
|
|
|
|
rec_name = PADDLE_REC_MODEL.get(paddle_lang)
|
|
rec = model_dir(rec_name)
|
|
if rec:
|
|
kwargs["text_recognition_model_name"] = rec_name
|
|
kwargs["text_recognition_model_dir"] = rec
|
|
|
|
textline = model_dir(PADDLE_TEXTLINE_MODEL)
|
|
if textline:
|
|
kwargs["textline_orientation_model_name"] = PADDLE_TEXTLINE_MODEL
|
|
kwargs["textline_orientation_model_dir"] = textline
|
|
kwargs["use_textline_orientation"] = True
|
|
else:
|
|
kwargs["use_textline_orientation"] = False
|
|
|
|
return kwargs
|
|
|
|
|
|
def auto_detect_language(input_path):
|
|
"""Detect the predominant script in the image using Tesseract multi-lang.
|
|
|
|
Runs a quick Tesseract pass with all installed language packs,
|
|
then analyzes the Unicode character ranges in the output to
|
|
determine which PaddleOCR language model to use.
|
|
"""
|
|
import subprocess
|
|
|
|
try:
|
|
result = subprocess.run(
|
|
["tesseract", input_path, "stdout", "-l", "eng+kor+chi_sim+jpn"],
|
|
capture_output=True, text=True, timeout=30,
|
|
)
|
|
text = result.stdout.strip()
|
|
if not text:
|
|
return "en"
|
|
|
|
hangul = sum(1 for c in text if "\uAC00" <= c <= "\uD7AF" or "\u1100" <= c <= "\u11FF")
|
|
cjk = sum(1 for c in text if "\u4E00" <= c <= "\u9FFF")
|
|
hiragana = sum(1 for c in text if "\u3040" <= c <= "\u309F")
|
|
katakana = sum(1 for c in text if "\u30A0" <= c <= "\u30FF")
|
|
latin = sum(1 for c in text if c.isascii() and c.isalpha())
|
|
|
|
total = hangul + cjk + hiragana + katakana + latin
|
|
if total == 0:
|
|
return "en"
|
|
|
|
if hangul / total > 0.3:
|
|
return "ko"
|
|
if (hiragana + katakana) / total > 0.2:
|
|
return "ja"
|
|
if cjk / total > 0.3:
|
|
return "zh"
|
|
return "en"
|
|
except Exception:
|
|
return "en"
|
|
|
|
|
|
def run_tesseract(input_path, language, is_auto=False):
|
|
"""Run Tesseract OCR (Fast tier)."""
|
|
import subprocess
|
|
|
|
# When auto-detected, use all installed language packs for best coverage
|
|
if is_auto:
|
|
tess_lang = "eng+kor+chi_sim+jpn+deu+fra+spa"
|
|
else:
|
|
tess_lang = TESSERACT_LANG_MAP.get(language, "eng")
|
|
|
|
emit_progress(30, "Scanning")
|
|
result = subprocess.run(
|
|
["tesseract", input_path, "stdout", "-l", tess_lang],
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=120,
|
|
)
|
|
emit_progress(70, "Extracting text")
|
|
text = result.stdout.strip()
|
|
if result.returncode != 0 and not text:
|
|
raise RuntimeError(result.stderr.strip() or "Tesseract failed")
|
|
return text
|
|
|
|
|
|
def _extract_ocr_texts(results):
|
|
"""Extract text from PaddleOCR 3.x result objects.
|
|
|
|
Handles multiple result formats across PaddleOCR versions:
|
|
- 3.4.x: OCRResult with .json["res"]["rec_texts"]
|
|
- Earlier: result objects with .res dict containing "text" list
|
|
"""
|
|
text_parts = []
|
|
for res in results:
|
|
# PaddleOCR 3.4.x format: OCRResult with .json dict
|
|
if hasattr(res, "json") and isinstance(res.json, dict):
|
|
inner = res.json.get("res", {})
|
|
rec_texts = inner.get("rec_texts", [])
|
|
if rec_texts:
|
|
text_parts.extend(rec_texts)
|
|
continue
|
|
# Older format: .res dict with "text" list
|
|
if hasattr(res, "res") and isinstance(res.res, dict):
|
|
text_parts.extend(res.res.get("text", []))
|
|
return "\n".join(text_parts)
|
|
|
|
|
|
def run_paddleocr_v5(input_path, language):
|
|
"""Run PaddleOCR PP-OCRv5 server models (Balanced tier)."""
|
|
# GPU-only: paddlepaddle-gpu segfaults at import on a CPU-only host (libcuda
|
|
# absent). Refuse before importing so the caller falls back to Tesseract.
|
|
if not gpu_available():
|
|
raise ImportError(
|
|
"PaddleOCR (paddlepaddle-gpu) requires a GPU; the amd64 bundle ships the "
|
|
"GPU build, which cannot load on a CPU-only host. Use quality=fast (Tesseract)."
|
|
)
|
|
os.environ["PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK"] = "True"
|
|
|
|
stdout_fd = os.dup(1)
|
|
os.dup2(2, 1)
|
|
|
|
try:
|
|
import logging
|
|
from paddleocr import PaddleOCR
|
|
|
|
# Suppress PaddleOCR internal logging (replaces removed show_log param)
|
|
for name in ("ppocr", "paddleocr", "paddle"):
|
|
logging.getLogger(name).setLevel(logging.ERROR)
|
|
|
|
paddle_lang = PADDLE_LANG_MAP.get(language, "en")
|
|
device = "gpu:0" if gpu_available() else "cpu"
|
|
|
|
emit_progress(20, "Loading")
|
|
mk = _bundled_paddle_kwargs(paddle_lang)
|
|
# When a bundled recognizer is pinned, the model selects the script, so
|
|
# we omit lang (this is the proven fully-offline path). Only fall back to
|
|
# lang-based (online) resolution when no bundled rec exists (e.g. ja).
|
|
if "text_recognition_model_dir" not in mk:
|
|
mk["lang"] = paddle_lang
|
|
ocr = PaddleOCR(
|
|
device=device,
|
|
ocr_version="PP-OCRv5",
|
|
enable_mkldnn=False,
|
|
**mk,
|
|
)
|
|
emit_progress(30, "Scanning")
|
|
results = ocr.predict(input=input_path)
|
|
emit_progress(70, "Extracting text")
|
|
|
|
text = _extract_ocr_texts(results)
|
|
finally:
|
|
os.dup2(stdout_fd, 1)
|
|
os.close(stdout_fd)
|
|
|
|
return text
|
|
|
|
|
|
def run_paddleocr_vl(input_path):
|
|
"""Run PaddleOCR-VL 1.5 vision-language model (Best tier).
|
|
|
|
The VLM is lazy-loaded on first call and stays resident in the
|
|
dispatcher process for subsequent requests.
|
|
Requires PaddlePaddle >= 3.2 for fused_rms_norm_ext.
|
|
"""
|
|
global _paddleocr_vl_instance
|
|
# GPU-only: see run_paddleocr_v5. Refuse before importing paddle on CPU.
|
|
if not gpu_available():
|
|
raise ImportError(
|
|
"PaddleOCR-VL (paddlepaddle-gpu) requires a GPU; the amd64 bundle ships the "
|
|
"GPU build, which cannot load on a CPU-only host. Use quality=balanced or fast."
|
|
)
|
|
os.environ["PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK"] = "True"
|
|
|
|
stdout_fd = os.dup(1)
|
|
os.dup2(2, 1)
|
|
|
|
try:
|
|
if _paddleocr_vl_instance is None:
|
|
emit_progress(15, "Loading model")
|
|
from paddleocr import PaddleOCRVL
|
|
|
|
device = "gpu" if gpu_available() else "cpu"
|
|
_paddleocr_vl_instance = PaddleOCRVL(device=device)
|
|
|
|
emit_progress(30, "Scanning")
|
|
output = _paddleocr_vl_instance.predict(input_path)
|
|
emit_progress(70, "Extracting text")
|
|
|
|
text_parts = []
|
|
for res in output:
|
|
# PaddleOCR-VL 1.5+: markdown_texts holds the extracted text
|
|
if hasattr(res, "markdown") and isinstance(res.markdown, dict):
|
|
md_text = res.markdown.get("markdown_texts", "")
|
|
if md_text:
|
|
text_parts.append(md_text)
|
|
continue
|
|
if hasattr(res, "parsing_res_list"):
|
|
for block in res.parsing_res_list:
|
|
content = block.get("block_content", "")
|
|
if content:
|
|
text_parts.append(content)
|
|
elif hasattr(res, "rec_text"):
|
|
text_parts.append(res.rec_text)
|
|
# Also try the json-based extraction as fallback
|
|
elif hasattr(res, "json") and isinstance(res.json, dict):
|
|
inner = res.json.get("res", {})
|
|
rec_texts = inner.get("rec_texts", [])
|
|
text_parts.extend(rec_texts)
|
|
|
|
text = "\n".join(text_parts)
|
|
finally:
|
|
os.dup2(stdout_fd, 1)
|
|
os.close(stdout_fd)
|
|
|
|
return text
|
|
|
|
|
|
def main():
|
|
input_path = sys.argv[1]
|
|
settings = json.loads(sys.argv[2]) if len(sys.argv) > 2 else {}
|
|
|
|
quality = settings.get("quality", None)
|
|
language = settings.get("language", "auto")
|
|
enhance = settings.get("enhance", True)
|
|
|
|
# Backward compat: old "engine" param maps to quality
|
|
if quality is None:
|
|
engine = settings.get("engine", "tesseract")
|
|
quality = "fast" if engine == "tesseract" else "balanced"
|
|
|
|
# On a CPU-only host, downgrade GPU-only tiers (PaddleOCR) to Tesseract so we
|
|
# never import paddlepaddle-gpu, whose import segfaults and wedges the dispatcher.
|
|
downgraded = effective_quality(quality)
|
|
if downgraded != quality:
|
|
print(
|
|
json.dumps({"info": f"{quality} OCR needs a GPU; using {downgraded} (Tesseract) on CPU"}),
|
|
file=sys.stderr,
|
|
flush=True,
|
|
)
|
|
quality = downgraded
|
|
|
|
preprocessed_path = None
|
|
try:
|
|
emit_progress(5, "Preparing")
|
|
|
|
# Preprocessing (if enabled)
|
|
if enhance:
|
|
emit_progress(8, "Enhancing image")
|
|
try:
|
|
from ocr_preprocess import preprocess
|
|
preprocessed_path = input_path + "_enhanced.png"
|
|
preprocess(input_path, preprocessed_path)
|
|
input_path = preprocessed_path
|
|
except Exception as e:
|
|
print(json.dumps({"warning": f"Enhancement skipped: {e}"}), file=sys.stderr, flush=True)
|
|
preprocessed_path = None
|
|
|
|
# Language auto-detection
|
|
was_auto = language == "auto"
|
|
if was_auto:
|
|
emit_progress(10, "Detecting language")
|
|
language = auto_detect_language(input_path)
|
|
|
|
engine_used = quality
|
|
|
|
# Route to engine based on quality tier
|
|
if quality == "fast":
|
|
try:
|
|
text = run_tesseract(input_path, language, is_auto=was_auto)
|
|
engine_used = "tesseract"
|
|
except FileNotFoundError:
|
|
print(json.dumps({"success": False, "error": "Tesseract is not installed"}))
|
|
sys.exit(1)
|
|
|
|
elif quality == "balanced":
|
|
try:
|
|
text = run_paddleocr_v5(input_path, language)
|
|
engine_used = "paddleocr-v5"
|
|
except ImportError as e:
|
|
print(json.dumps({
|
|
"success": False,
|
|
"error": (
|
|
f"PaddleOCR is not installed: {e}. "
|
|
"Install the OCR feature or use quality=fast for Tesseract."
|
|
),
|
|
}))
|
|
sys.exit(1)
|
|
except Exception as e:
|
|
print(json.dumps({
|
|
"success": False,
|
|
"error": (
|
|
f"PaddleOCR PP-OCRv5 failed: {type(e).__name__}: {e}. "
|
|
"Install the OCR feature or use quality=fast for Tesseract."
|
|
),
|
|
}))
|
|
sys.exit(1)
|
|
|
|
elif quality == "best":
|
|
try:
|
|
text = run_paddleocr_vl(input_path)
|
|
engine_used = "paddleocr-vl"
|
|
except ImportError as e:
|
|
print(json.dumps({
|
|
"success": False,
|
|
"error": (
|
|
f"PaddleOCR-VL is not available: {e}. "
|
|
"Install the OCR feature or use quality=balanced for PP-OCRv5."
|
|
),
|
|
}))
|
|
sys.exit(1)
|
|
except Exception as e:
|
|
print(json.dumps({
|
|
"success": False,
|
|
"error": (
|
|
f"PaddleOCR-VL failed: {type(e).__name__}: {e}. "
|
|
"Install the OCR feature or use quality=balanced for PP-OCRv5."
|
|
),
|
|
}))
|
|
sys.exit(1)
|
|
|
|
else:
|
|
print(json.dumps({"success": False, "error": f"Unknown quality: {quality}"}))
|
|
sys.exit(1)
|
|
|
|
emit_progress(95, "Done")
|
|
print(json.dumps({"success": True, "text": text, "engine": engine_used}))
|
|
|
|
except Exception as e:
|
|
print(json.dumps({"success": False, "error": str(e)}))
|
|
sys.exit(1)
|
|
finally:
|
|
# Clean up preprocessed temp file
|
|
if preprocessed_path:
|
|
try:
|
|
os.remove(preprocessed_path)
|
|
except OSError:
|
|
pass
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|