mirror of
https://github.com/snapotter-hq/SnapOtter.git
synced 2026-08-03 07:46:42 +02:00
fix: verbose error handling, batch processing, and multi-file support
- Replace [object Object] errors with readable messages across all 20+ API routes by normalizing Zod validation errors to strings (formatZodErrors) - Add parseApiError() on frontend to defensively handle any details type - Add global Fastify error handler with full stack traces in logs - Fix image-to-pdf auth: Object.entries(headers) → headers.forEach() - Fix passport-photo: safeParse + formatZodErrors, safe error extraction - Fix OCR silent fallbacks: log exception type/message when falling back, include actual engine used in API response and Docker logs - Fix split tool: process all uploaded images, combine into ZIP with subfolders per image - Fix batch support for blur-faces, strip-metadata, edit-metadata, vectorize: add processAllFiles branch for multi-file uploads - Docker: LOG_LEVEL=debug, PYTHONWARNINGS=default for visibility - Add Playwright e2e tests verifying all fixes against Docker container
This commit is contained in:
+39
-12
@@ -229,10 +229,13 @@ def main():
|
||||
emit_progress(10, "Detecting language")
|
||||
language = auto_detect_language(input_path)
|
||||
|
||||
engine_used = quality
|
||||
|
||||
# Route to engine based on quality tier
|
||||
if quality == "fast":
|
||||
try:
|
||||
text = run_tesseract(input_path, language, is_auto=was_auto)
|
||||
engine_used = "tesseract"
|
||||
except FileNotFoundError:
|
||||
print(json.dumps({"success": False, "error": "Tesseract is not installed"}))
|
||||
sys.exit(1)
|
||||
@@ -240,39 +243,63 @@ def main():
|
||||
elif quality == "balanced":
|
||||
try:
|
||||
text = run_paddleocr_v5(input_path, language)
|
||||
except ImportError:
|
||||
print(json.dumps({"success": False, "error": "PaddleOCR is not installed"}))
|
||||
engine_used = "paddleocr-v5"
|
||||
except ImportError as e:
|
||||
print(json.dumps({"success": False, "error": f"PaddleOCR is not installed: {e}"}))
|
||||
sys.exit(1)
|
||||
except Exception:
|
||||
emit_progress(25, "Falling back")
|
||||
except Exception as e:
|
||||
print(json.dumps({
|
||||
"warning": f"PaddleOCR PP-OCRv5 failed ({type(e).__name__}: {e}), falling back to Tesseract"
|
||||
}), file=sys.stderr, flush=True)
|
||||
emit_progress(25, "PaddleOCR failed, falling back to Tesseract")
|
||||
try:
|
||||
text = run_tesseract(input_path, language, is_auto=was_auto)
|
||||
engine_used = "tesseract (fallback from balanced)"
|
||||
except FileNotFoundError:
|
||||
print(json.dumps({"success": False, "error": "OCR engines unavailable"}))
|
||||
print(json.dumps({"success": False, "error": "OCR engines unavailable: PaddleOCR failed and Tesseract is not installed"}))
|
||||
sys.exit(1)
|
||||
|
||||
elif quality == "best":
|
||||
try:
|
||||
text = run_paddleocr_vl(input_path)
|
||||
except ImportError:
|
||||
emit_progress(20, "Falling back")
|
||||
engine_used = "paddleocr-vl"
|
||||
except ImportError as e:
|
||||
print(json.dumps({
|
||||
"warning": f"PaddleOCR-VL not available ({e}), trying PP-OCRv5"
|
||||
}), file=sys.stderr, flush=True)
|
||||
emit_progress(20, "VL model unavailable, trying PP-OCRv5")
|
||||
try:
|
||||
text = run_paddleocr_v5(input_path, language)
|
||||
except Exception:
|
||||
engine_used = "paddleocr-v5 (fallback from best)"
|
||||
except Exception as e2:
|
||||
print(json.dumps({
|
||||
"warning": f"PP-OCRv5 also failed ({type(e2).__name__}: {e2}), falling back to Tesseract"
|
||||
}), file=sys.stderr, flush=True)
|
||||
emit_progress(25, "PP-OCRv5 failed, falling back to Tesseract")
|
||||
text = run_tesseract(input_path, language, is_auto=was_auto)
|
||||
except Exception:
|
||||
emit_progress(20, "Falling back")
|
||||
engine_used = "tesseract (fallback from best)"
|
||||
except Exception as e:
|
||||
print(json.dumps({
|
||||
"warning": f"PaddleOCR-VL failed ({type(e).__name__}: {e}), trying PP-OCRv5"
|
||||
}), file=sys.stderr, flush=True)
|
||||
emit_progress(20, "VL model failed, trying PP-OCRv5")
|
||||
try:
|
||||
text = run_paddleocr_v5(input_path, language)
|
||||
except Exception:
|
||||
engine_used = "paddleocr-v5 (fallback from best)"
|
||||
except Exception as e2:
|
||||
print(json.dumps({
|
||||
"warning": f"PP-OCRv5 also failed ({type(e2).__name__}: {e2}), falling back to Tesseract"
|
||||
}), file=sys.stderr, flush=True)
|
||||
emit_progress(25, "PP-OCRv5 failed, falling back to Tesseract")
|
||||
text = run_tesseract(input_path, language, is_auto=was_auto)
|
||||
engine_used = "tesseract (fallback from best)"
|
||||
|
||||
else:
|
||||
print(json.dumps({"success": False, "error": f"Unknown quality: {quality}"}))
|
||||
sys.exit(1)
|
||||
|
||||
emit_progress(95, "Done")
|
||||
print(json.dumps({"success": True, "text": text}))
|
||||
print(json.dumps({"success": True, "text": text, "engine": engine_used}))
|
||||
|
||||
except Exception as e:
|
||||
print(json.dumps({"success": False, "error": str(e)}))
|
||||
|
||||
Reference in New Issue
Block a user