From c2c104e887f619cf3bbc82fa464bf555bb8314f8 Mon Sep 17 00:00:00 2001 From: stirling-image Date: Tue, 14 Apr 2026 16:01:35 +0800 Subject: [PATCH] fix(passport-photo): use bg-background for dropdown to match app theme --- .../tools/passport-photo-settings.tsx | 4 +- packages/ai/python/detect_faces.py | 153 ++++++++++++------ packages/ai/python/enhance_faces.py | 97 ++++++++--- packages/ai/python/red_eye_removal.py | 93 +++++++++-- packages/ai/python/restore.py | 93 ++++++++--- 5 files changed, 332 insertions(+), 108 deletions(-) diff --git a/apps/web/src/components/tools/passport-photo-settings.tsx b/apps/web/src/components/tools/passport-photo-settings.tsx index 112deac9..f273d8e1 100644 --- a/apps/web/src/components/tools/passport-photo-settings.tsx +++ b/apps/web/src/components/tools/passport-photo-settings.tsx @@ -570,7 +570,7 @@ export function PassportPhotoSettings() { {dropdownOpen && (
{/* Search input */} -
+
= 0.10.30).""" + import mediapipe as mp + + model_path = _ensure_face_detect_model() + options = mp.tasks.vision.FaceDetectorOptions( + base_options=mp.tasks.BaseOptions(model_asset_path=model_path), + running_mode=mp.tasks.vision.RunningMode.IMAGE, + min_detection_confidence=min_confidence, + ) + detector = mp.tasks.vision.FaceDetector.create_from_options(options) + mp_image = mp.Image(image_format=mp.ImageFormat.SRGB, data=img_array) + result = detector.detect(mp_image) + detector.close() + + faces = [] + for detection in result.detections: + bbox = detection.bounding_box + faces.append({ + "x": bbox.origin_x, + "y": bbox.origin_y, + "w": bbox.width, + "h": bbox.height, + }) + return faces + + +def _detect_faces(img_array, min_confidence): + """Detect faces, trying legacy API first then falling back to tasks API.""" + try: + return _detect_with_solutions(img_array, min_confidence) + except AttributeError: + return _detect_with_tasks(img_array, min_confidence) + + def main(): input_path = sys.argv[1] output_path = sys.argv[2] @@ -24,7 +111,6 @@ def main(): img = Image.open(input_path).convert("RGB") try: - import mediapipe as mp import numpy as np emit_progress(20, "Ready") @@ -34,56 +120,33 @@ def main(): min_confidence = max(0.1, 1.0 - sensitivity) img_array = np.array(img) - mp_face = mp.solutions.face_detection - # Try short-range model first (model_selection=0, best for faces - # within ~2m which covers most photos), then fall back to - # full-range model (model_selection=1) for distant/group shots. + # Try legacy mp.solutions API first, fall back to mp.tasks emit_progress(25, "Scanning for faces") - results = None - for model_sel in [0, 1]: - detector = mp_face.FaceDetection( - model_selection=model_sel, - min_detection_confidence=min_confidence, - ) - results = detector.process(img_array) - detector.close() - if results.detections: - break - - faces = [] - detections = results.detections or [] - num_faces = len(detections) + faces = _detect_faces(img_array, min_confidence) + num_faces = len(faces) emit_progress(50, f"Found {num_faces} face{'s' if num_faces != 1 else ''}") - if num_faces > 0: - ih, iw = img_array.shape[:2] - for i, detection in enumerate(detections): - bbox = detection.location_data.relative_bounding_box - x = int(bbox.xmin * iw) - y = int(bbox.ymin * ih) - w = int(bbox.width * iw) - h = int(bbox.height * ih) + if num_faces > 0 and not detect_only: + for i, face in enumerate(faces): + x, y, w, h = face["x"], face["y"], face["w"], face["h"] - if not detect_only: - # Add padding around the face - pad = int(max(w, h) * 0.1) - x1 = max(0, x - pad) - y1 = max(0, y - pad) - x2 = min(img.width, x + w + pad) - y2 = min(img.height, y + h + pad) + # Add padding around the face + pad = int(max(w, h) * 0.1) + x1 = max(0, x - pad) + y1 = max(0, y - pad) + x2 = min(img.width, x + w + pad) + y2 = min(img.height, y + h + pad) - face_region = img.crop((x1, y1, x2, y2)) - blurred = face_region.filter( - ImageFilter.GaussianBlur(blur_radius) - ) - img.paste(blurred, (x1, y1)) - emit_progress( - 50 + int((i + 1) / num_faces * 40), - f"Blurring face {i + 1} of {num_faces}", - ) - - faces.append({"x": x, "y": y, "w": w, "h": h}) + face_region = img.crop((x1, y1, x2, y2)) + blurred = face_region.filter( + ImageFilter.GaussianBlur(blur_radius) + ) + img.paste(blurred, (x1, y1)) + emit_progress( + 50 + int((i + 1) / num_faces * 40), + f"Blurring face {i + 1} of {num_faces}", + ) if not detect_only: emit_progress(95, "Saving result") diff --git a/packages/ai/python/enhance_faces.py b/packages/ai/python/enhance_faces.py index ddd11419..aa0de284 100644 --- a/packages/ai/python/enhance_faces.py +++ b/packages/ai/python/enhance_faces.py @@ -37,45 +37,90 @@ CODEFORMER_MODEL_PATH = os.environ.get( ) +# ── Model path for new mp.tasks API ───────────────────────────────── + +_FACE_DETECT_MODEL_URL = "https://storage.googleapis.com/mediapipe-models/face_detector/blaze_face_short_range/float16/latest/blaze_face_short_range.task" +_MODEL_DIR = os.path.join(os.path.dirname(__file__), "..", "..", "..", ".models") +_FACE_DETECT_MODEL_PATH = os.path.join(_MODEL_DIR, "blaze_face_short_range.task") + + +def _ensure_face_detect_model(): + """Download the face detector model if not present.""" + if os.path.exists(_FACE_DETECT_MODEL_PATH): + return _FACE_DETECT_MODEL_PATH + os.makedirs(_MODEL_DIR, exist_ok=True) + import urllib.request + emit_progress(15, "Downloading face detection model") + urllib.request.urlretrieve(_FACE_DETECT_MODEL_URL, _FACE_DETECT_MODEL_PATH) + return _FACE_DETECT_MODEL_PATH + + def detect_faces_mediapipe(img_array, sensitivity): """Detect faces using MediaPipe with dual-model approach. Returns a list of {x, y, w, h} dicts for each detected face. + Tries legacy mp.solutions API first, falls back to mp.tasks. """ import mediapipe as mp min_confidence = max(0.1, 1.0 - sensitivity) - mp_face = mp.solutions.face_detection - # Try short-range model first (model_selection=0, best for faces - # within ~2m which covers most photos), then fall back to - # full-range model (model_selection=1) for distant/group shots. - detections = [] - for model_sel in [0, 1]: - detector = mp_face.FaceDetection( - model_selection=model_sel, + try: + mp_face = mp.solutions.face_detection + + # Try short-range model first (model_selection=0, best for faces + # within ~2m which covers most photos), then fall back to + # full-range model (model_selection=1) for distant/group shots. + detections = [] + for model_sel in [0, 1]: + detector = mp_face.FaceDetection( + model_selection=model_sel, + min_detection_confidence=min_confidence, + ) + results = detector.process(img_array) + detector.close() + if results.detections: + detections = results.detections + break + + if not detections: + return [] + + ih, iw = img_array.shape[:2] + faces = [] + for detection in detections: + bbox = detection.location_data.relative_bounding_box + faces.append({ + "x": int(bbox.xmin * iw), + "y": int(bbox.ymin * ih), + "w": int(bbox.width * iw), + "h": int(bbox.height * ih), + }) + return faces + + except AttributeError: + # mediapipe >= 0.10.30 removed mp.solutions, use tasks API + model_path = _ensure_face_detect_model() + options = mp.tasks.vision.FaceDetectorOptions( + base_options=mp.tasks.BaseOptions(model_asset_path=model_path), + running_mode=mp.tasks.vision.RunningMode.IMAGE, min_detection_confidence=min_confidence, ) - results = detector.process(img_array) + detector = mp.tasks.vision.FaceDetector.create_from_options(options) + mp_image = mp.Image(image_format=mp.ImageFormat.SRGB, data=img_array) + result = detector.detect(mp_image) detector.close() - if results.detections: - detections = results.detections - break - if not detections: - return [] - - ih, iw = img_array.shape[:2] - faces = [] - for detection in detections: - bbox = detection.location_data.relative_bounding_box - x = int(bbox.xmin * iw) - y = int(bbox.ymin * ih) - w = int(bbox.width * iw) - h = int(bbox.height * ih) - faces.append({"x": x, "y": y, "w": w, "h": h}) - - return faces + faces = [] + for detection in result.detections: + bbox = detection.bounding_box + faces.append({ + "x": bbox.origin_x, + "y": bbox.origin_y, + "w": bbox.width, + "h": bbox.height, + }) + return faces def enhance_with_gfpgan(img_array, only_center_face): diff --git a/packages/ai/python/red_eye_removal.py b/packages/ai/python/red_eye_removal.py index 9746fb85..ab767166 100644 --- a/packages/ai/python/red_eye_removal.py +++ b/packages/ai/python/red_eye_removal.py @@ -9,6 +9,79 @@ def emit_progress(percent, stage): print(json.dumps({"progress": percent, "stage": stage}), file=sys.stderr, flush=True) +# ── Model path for new mp.tasks API ───────────────────────────────── + +_FACE_MESH_MODEL_URL = "https://storage.googleapis.com/mediapipe-models/face_landmarker/face_landmarker/float16/latest/face_landmarker.task" +_MODEL_DIR = os.path.join(os.path.dirname(__file__), "..", "..", "..", ".models") +_FACE_MESH_MODEL_PATH = os.path.join(_MODEL_DIR, "face_landmarker.task") + + +def _ensure_face_mesh_model(): + """Download the face landmarker model if not present.""" + if os.path.exists(_FACE_MESH_MODEL_PATH): + return _FACE_MESH_MODEL_PATH + os.makedirs(_MODEL_DIR, exist_ok=True) + import urllib.request + emit_progress(15, "Downloading face mesh model") + urllib.request.urlretrieve(_FACE_MESH_MODEL_URL, _FACE_MESH_MODEL_PATH) + return _FACE_MESH_MODEL_PATH + + +def _mesh_with_solutions(img_array, max_faces=10, min_confidence=0.5): + """FaceMesh using legacy mp.solutions API (mediapipe < 0.10.30). + + Returns list of landmark lists. Each landmark has .x, .y attributes. + """ + import mediapipe as mp + + mesh = mp.solutions.face_mesh.FaceMesh( + static_image_mode=True, + max_num_faces=max_faces, + refine_landmarks=True, + min_detection_confidence=min_confidence, + ) + results = mesh.process(img_array) + mesh.close() + + if not results.multi_face_landmarks: + return [] + + return [face.landmark for face in results.multi_face_landmarks] + + +def _mesh_with_tasks(img_array, max_faces=10, min_confidence=0.5): + """FaceMesh using new mp.tasks API (mediapipe >= 0.10.30). + + Returns list of landmark lists. Each landmark has .x, .y attributes. + """ + import mediapipe as mp + + model_path = _ensure_face_mesh_model() + options = mp.tasks.vision.FaceLandmarkerOptions( + base_options=mp.tasks.BaseOptions(model_asset_path=model_path), + running_mode=mp.tasks.vision.RunningMode.IMAGE, + num_faces=max_faces, + min_face_detection_confidence=min_confidence, + ) + landmarker = mp.tasks.vision.FaceLandmarker.create_from_options(options) + mp_image = mp.Image(image_format=mp.ImageFormat.SRGB, data=img_array) + result = landmarker.detect(mp_image) + landmarker.close() + + if not result.face_landmarks: + return [] + + return result.face_landmarks + + +def _detect_face_mesh(img_array, max_faces=10, min_confidence=0.5): + """Detect face mesh, trying legacy API first then falling back to tasks API.""" + try: + return _mesh_with_solutions(img_array, max_faces, min_confidence) + except AttributeError: + return _mesh_with_tasks(img_array, max_faces, min_confidence) + + def main(): input_path = sys.argv[1] output_path = sys.argv[2] @@ -65,28 +138,19 @@ def main(): format_label = "jpg" try: - import mediapipe as mp import numpy as np import cv2 emit_progress(25, "Detecting faces") img_array = np.array(img) - mesh = mp.solutions.face_mesh.FaceMesh( - static_image_mode=True, - max_num_faces=10, - refine_landmarks=True, - min_detection_confidence=0.5, - ) - results = mesh.process(img_array) - mesh.close() - faces_detected = 0 + # Try legacy mp.solutions API first, fall back to mp.tasks + all_face_landmarks = _detect_face_mesh(img_array) + + faces_detected = len(all_face_landmarks) eyes_corrected = 0 - if results.multi_face_landmarks: - faces_detected = len(results.multi_face_landmarks) - emit_progress(50, "Analyzing eyes") # Iris landmark indices @@ -95,8 +159,7 @@ def main(): if faces_detected > 0: all_eyes = [] - for face_landmarks in results.multi_face_landmarks: - landmarks = face_landmarks.landmark + for landmarks in all_face_landmarks: for iris_indices in [right_iris, left_iris]: center_idx = iris_indices[0] contour_indices = iris_indices[1:] diff --git a/packages/ai/python/restore.py b/packages/ai/python/restore.py index eb9d6f19..fda50fdc 100644 --- a/packages/ai/python/restore.py +++ b/packages/ai/python/restore.py @@ -217,6 +217,24 @@ def _get_codeformer_path(): return CODEFORMER_LOCAL_PATH +# ── Model path for new mp.tasks API ───────────────────────────────── + +_FACE_DETECT_MODEL_URL = "https://storage.googleapis.com/mediapipe-models/face_detector/blaze_face_short_range/float16/latest/blaze_face_short_range.task" +_FACE_DETECT_MODEL_DIR = os.path.join(os.path.dirname(__file__), "..", "..", "..", ".models") +_FACE_DETECT_MODEL_PATH = os.path.join(_FACE_DETECT_MODEL_DIR, "blaze_face_short_range.task") + + +def _ensure_face_detect_model(): + """Download the face detector model if not present.""" + if os.path.exists(_FACE_DETECT_MODEL_PATH): + return _FACE_DETECT_MODEL_PATH + os.makedirs(_FACE_DETECT_MODEL_DIR, exist_ok=True) + import urllib.request + emit_progress(15, "Downloading face detection model") + urllib.request.urlretrieve(_FACE_DETECT_MODEL_URL, _FACE_DETECT_MODEL_PATH) + return _FACE_DETECT_MODEL_PATH + + def enhance_faces(img_bgr, fidelity=0.7): """Enhance faces in the image using CodeFormer ONNX. @@ -239,20 +257,57 @@ def enhance_faces(img_bgr, fidelity=0.7): img_rgb = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB) ih, iw = img_bgr.shape[:2] - mp_face = mp.solutions.face_detection - detections = [] - for model_sel in [0, 1]: - detector = mp_face.FaceDetection( - model_selection=model_sel, min_detection_confidence=0.4 - ) - results = detector.process(img_rgb) - detector.close() - if results.detections: - detections = results.detections - break + try: + mp_face = mp.solutions.face_detection + detections = [] + for model_sel in [0, 1]: + detector = mp_face.FaceDetection( + model_selection=model_sel, min_detection_confidence=0.4 + ) + results = detector.process(img_rgb) + detector.close() + if results.detections: + detections = results.detections + break - if not detections: - return img_bgr, 0 + if not detections: + return img_bgr, 0 + + face_boxes = [] + for detection in detections: + bbox = detection.location_data.relative_bounding_box + face_boxes.append({ + "x": int(bbox.xmin * iw), + "y": int(bbox.ymin * ih), + "w": int(bbox.width * iw), + "h": int(bbox.height * ih), + }) + + except AttributeError: + # mediapipe >= 0.10.30 removed mp.solutions, use tasks API + model_path = _ensure_face_detect_model() + options = mp.tasks.vision.FaceDetectorOptions( + base_options=mp.tasks.BaseOptions(model_asset_path=model_path), + running_mode=mp.tasks.vision.RunningMode.IMAGE, + min_detection_confidence=0.4, + ) + fd = mp.tasks.vision.FaceDetector.create_from_options(options) + mp_image = mp.Image(image_format=mp.ImageFormat.SRGB, data=img_rgb) + result = fd.detect(mp_image) + fd.close() + + if not result.detections: + return img_bgr, 0 + + face_boxes = [] + for detection in result.detections: + bbox = detection.bounding_box + face_boxes.append({ + "x": bbox.origin_x, + "y": bbox.origin_y, + "w": bbox.width, + "h": bbox.height, + }) # Load CodeFormer model model_path = _get_codeformer_path() @@ -266,13 +321,11 @@ def enhance_faces(img_bgr, fidelity=0.7): result = img_bgr.copy() faces_enhanced = 0 - for detection in detections: - bbox = detection.location_data.relative_bounding_box - # Convert relative coords to absolute - x = int(bbox.xmin * iw) - y = int(bbox.ymin * ih) - w = int(bbox.width * iw) - h = int(bbox.height * ih) + for face_box in face_boxes: + x = face_box["x"] + y = face_box["y"] + w = face_box["w"] + h = face_box["h"] # Skip very small faces (under 48px) - enhancement won't help if w < 48 or h < 48: