= 0.10.30)."""
+ import mediapipe as mp
+
+ model_path = _ensure_face_detect_model()
+ options = mp.tasks.vision.FaceDetectorOptions(
+ base_options=mp.tasks.BaseOptions(model_asset_path=model_path),
+ running_mode=mp.tasks.vision.RunningMode.IMAGE,
+ min_detection_confidence=min_confidence,
+ )
+ detector = mp.tasks.vision.FaceDetector.create_from_options(options)
+ mp_image = mp.Image(image_format=mp.ImageFormat.SRGB, data=img_array)
+ result = detector.detect(mp_image)
+ detector.close()
+
+ faces = []
+ for detection in result.detections:
+ bbox = detection.bounding_box
+ faces.append({
+ "x": bbox.origin_x,
+ "y": bbox.origin_y,
+ "w": bbox.width,
+ "h": bbox.height,
+ })
+ return faces
+
+
+def _detect_faces(img_array, min_confidence):
+ """Detect faces, trying legacy API first then falling back to tasks API."""
+ try:
+ return _detect_with_solutions(img_array, min_confidence)
+ except AttributeError:
+ return _detect_with_tasks(img_array, min_confidence)
+
+
def main():
input_path = sys.argv[1]
output_path = sys.argv[2]
@@ -24,7 +111,6 @@ def main():
img = Image.open(input_path).convert("RGB")
try:
- import mediapipe as mp
import numpy as np
emit_progress(20, "Ready")
@@ -34,56 +120,33 @@ def main():
min_confidence = max(0.1, 1.0 - sensitivity)
img_array = np.array(img)
- mp_face = mp.solutions.face_detection
- # Try short-range model first (model_selection=0, best for faces
- # within ~2m which covers most photos), then fall back to
- # full-range model (model_selection=1) for distant/group shots.
+ # Try legacy mp.solutions API first, fall back to mp.tasks
emit_progress(25, "Scanning for faces")
- results = None
- for model_sel in [0, 1]:
- detector = mp_face.FaceDetection(
- model_selection=model_sel,
- min_detection_confidence=min_confidence,
- )
- results = detector.process(img_array)
- detector.close()
- if results.detections:
- break
-
- faces = []
- detections = results.detections or []
- num_faces = len(detections)
+ faces = _detect_faces(img_array, min_confidence)
+ num_faces = len(faces)
emit_progress(50, f"Found {num_faces} face{'s' if num_faces != 1 else ''}")
- if num_faces > 0:
- ih, iw = img_array.shape[:2]
- for i, detection in enumerate(detections):
- bbox = detection.location_data.relative_bounding_box
- x = int(bbox.xmin * iw)
- y = int(bbox.ymin * ih)
- w = int(bbox.width * iw)
- h = int(bbox.height * ih)
+ if num_faces > 0 and not detect_only:
+ for i, face in enumerate(faces):
+ x, y, w, h = face["x"], face["y"], face["w"], face["h"]
- if not detect_only:
- # Add padding around the face
- pad = int(max(w, h) * 0.1)
- x1 = max(0, x - pad)
- y1 = max(0, y - pad)
- x2 = min(img.width, x + w + pad)
- y2 = min(img.height, y + h + pad)
+ # Add padding around the face
+ pad = int(max(w, h) * 0.1)
+ x1 = max(0, x - pad)
+ y1 = max(0, y - pad)
+ x2 = min(img.width, x + w + pad)
+ y2 = min(img.height, y + h + pad)
- face_region = img.crop((x1, y1, x2, y2))
- blurred = face_region.filter(
- ImageFilter.GaussianBlur(blur_radius)
- )
- img.paste(blurred, (x1, y1))
- emit_progress(
- 50 + int((i + 1) / num_faces * 40),
- f"Blurring face {i + 1} of {num_faces}",
- )
-
- faces.append({"x": x, "y": y, "w": w, "h": h})
+ face_region = img.crop((x1, y1, x2, y2))
+ blurred = face_region.filter(
+ ImageFilter.GaussianBlur(blur_radius)
+ )
+ img.paste(blurred, (x1, y1))
+ emit_progress(
+ 50 + int((i + 1) / num_faces * 40),
+ f"Blurring face {i + 1} of {num_faces}",
+ )
if not detect_only:
emit_progress(95, "Saving result")
diff --git a/packages/ai/python/enhance_faces.py b/packages/ai/python/enhance_faces.py
index ddd11419..aa0de284 100644
--- a/packages/ai/python/enhance_faces.py
+++ b/packages/ai/python/enhance_faces.py
@@ -37,45 +37,90 @@ CODEFORMER_MODEL_PATH = os.environ.get(
)
+# ── Model path for new mp.tasks API ─────────────────────────────────
+
+_FACE_DETECT_MODEL_URL = "https://storage.googleapis.com/mediapipe-models/face_detector/blaze_face_short_range/float16/latest/blaze_face_short_range.task"
+_MODEL_DIR = os.path.join(os.path.dirname(__file__), "..", "..", "..", ".models")
+_FACE_DETECT_MODEL_PATH = os.path.join(_MODEL_DIR, "blaze_face_short_range.task")
+
+
+def _ensure_face_detect_model():
+ """Download the face detector model if not present."""
+ if os.path.exists(_FACE_DETECT_MODEL_PATH):
+ return _FACE_DETECT_MODEL_PATH
+ os.makedirs(_MODEL_DIR, exist_ok=True)
+ import urllib.request
+ emit_progress(15, "Downloading face detection model")
+ urllib.request.urlretrieve(_FACE_DETECT_MODEL_URL, _FACE_DETECT_MODEL_PATH)
+ return _FACE_DETECT_MODEL_PATH
+
+
def detect_faces_mediapipe(img_array, sensitivity):
"""Detect faces using MediaPipe with dual-model approach.
Returns a list of {x, y, w, h} dicts for each detected face.
+ Tries legacy mp.solutions API first, falls back to mp.tasks.
"""
import mediapipe as mp
min_confidence = max(0.1, 1.0 - sensitivity)
- mp_face = mp.solutions.face_detection
- # Try short-range model first (model_selection=0, best for faces
- # within ~2m which covers most photos), then fall back to
- # full-range model (model_selection=1) for distant/group shots.
- detections = []
- for model_sel in [0, 1]:
- detector = mp_face.FaceDetection(
- model_selection=model_sel,
+ try:
+ mp_face = mp.solutions.face_detection
+
+ # Try short-range model first (model_selection=0, best for faces
+ # within ~2m which covers most photos), then fall back to
+ # full-range model (model_selection=1) for distant/group shots.
+ detections = []
+ for model_sel in [0, 1]:
+ detector = mp_face.FaceDetection(
+ model_selection=model_sel,
+ min_detection_confidence=min_confidence,
+ )
+ results = detector.process(img_array)
+ detector.close()
+ if results.detections:
+ detections = results.detections
+ break
+
+ if not detections:
+ return []
+
+ ih, iw = img_array.shape[:2]
+ faces = []
+ for detection in detections:
+ bbox = detection.location_data.relative_bounding_box
+ faces.append({
+ "x": int(bbox.xmin * iw),
+ "y": int(bbox.ymin * ih),
+ "w": int(bbox.width * iw),
+ "h": int(bbox.height * ih),
+ })
+ return faces
+
+ except AttributeError:
+ # mediapipe >= 0.10.30 removed mp.solutions, use tasks API
+ model_path = _ensure_face_detect_model()
+ options = mp.tasks.vision.FaceDetectorOptions(
+ base_options=mp.tasks.BaseOptions(model_asset_path=model_path),
+ running_mode=mp.tasks.vision.RunningMode.IMAGE,
min_detection_confidence=min_confidence,
)
- results = detector.process(img_array)
+ detector = mp.tasks.vision.FaceDetector.create_from_options(options)
+ mp_image = mp.Image(image_format=mp.ImageFormat.SRGB, data=img_array)
+ result = detector.detect(mp_image)
detector.close()
- if results.detections:
- detections = results.detections
- break
- if not detections:
- return []
-
- ih, iw = img_array.shape[:2]
- faces = []
- for detection in detections:
- bbox = detection.location_data.relative_bounding_box
- x = int(bbox.xmin * iw)
- y = int(bbox.ymin * ih)
- w = int(bbox.width * iw)
- h = int(bbox.height * ih)
- faces.append({"x": x, "y": y, "w": w, "h": h})
-
- return faces
+ faces = []
+ for detection in result.detections:
+ bbox = detection.bounding_box
+ faces.append({
+ "x": bbox.origin_x,
+ "y": bbox.origin_y,
+ "w": bbox.width,
+ "h": bbox.height,
+ })
+ return faces
def enhance_with_gfpgan(img_array, only_center_face):
diff --git a/packages/ai/python/red_eye_removal.py b/packages/ai/python/red_eye_removal.py
index 9746fb85..ab767166 100644
--- a/packages/ai/python/red_eye_removal.py
+++ b/packages/ai/python/red_eye_removal.py
@@ -9,6 +9,79 @@ def emit_progress(percent, stage):
print(json.dumps({"progress": percent, "stage": stage}), file=sys.stderr, flush=True)
+# ── Model path for new mp.tasks API ─────────────────────────────────
+
+_FACE_MESH_MODEL_URL = "https://storage.googleapis.com/mediapipe-models/face_landmarker/face_landmarker/float16/latest/face_landmarker.task"
+_MODEL_DIR = os.path.join(os.path.dirname(__file__), "..", "..", "..", ".models")
+_FACE_MESH_MODEL_PATH = os.path.join(_MODEL_DIR, "face_landmarker.task")
+
+
+def _ensure_face_mesh_model():
+ """Download the face landmarker model if not present."""
+ if os.path.exists(_FACE_MESH_MODEL_PATH):
+ return _FACE_MESH_MODEL_PATH
+ os.makedirs(_MODEL_DIR, exist_ok=True)
+ import urllib.request
+ emit_progress(15, "Downloading face mesh model")
+ urllib.request.urlretrieve(_FACE_MESH_MODEL_URL, _FACE_MESH_MODEL_PATH)
+ return _FACE_MESH_MODEL_PATH
+
+
+def _mesh_with_solutions(img_array, max_faces=10, min_confidence=0.5):
+ """FaceMesh using legacy mp.solutions API (mediapipe < 0.10.30).
+
+ Returns list of landmark lists. Each landmark has .x, .y attributes.
+ """
+ import mediapipe as mp
+
+ mesh = mp.solutions.face_mesh.FaceMesh(
+ static_image_mode=True,
+ max_num_faces=max_faces,
+ refine_landmarks=True,
+ min_detection_confidence=min_confidence,
+ )
+ results = mesh.process(img_array)
+ mesh.close()
+
+ if not results.multi_face_landmarks:
+ return []
+
+ return [face.landmark for face in results.multi_face_landmarks]
+
+
+def _mesh_with_tasks(img_array, max_faces=10, min_confidence=0.5):
+ """FaceMesh using new mp.tasks API (mediapipe >= 0.10.30).
+
+ Returns list of landmark lists. Each landmark has .x, .y attributes.
+ """
+ import mediapipe as mp
+
+ model_path = _ensure_face_mesh_model()
+ options = mp.tasks.vision.FaceLandmarkerOptions(
+ base_options=mp.tasks.BaseOptions(model_asset_path=model_path),
+ running_mode=mp.tasks.vision.RunningMode.IMAGE,
+ num_faces=max_faces,
+ min_face_detection_confidence=min_confidence,
+ )
+ landmarker = mp.tasks.vision.FaceLandmarker.create_from_options(options)
+ mp_image = mp.Image(image_format=mp.ImageFormat.SRGB, data=img_array)
+ result = landmarker.detect(mp_image)
+ landmarker.close()
+
+ if not result.face_landmarks:
+ return []
+
+ return result.face_landmarks
+
+
+def _detect_face_mesh(img_array, max_faces=10, min_confidence=0.5):
+ """Detect face mesh, trying legacy API first then falling back to tasks API."""
+ try:
+ return _mesh_with_solutions(img_array, max_faces, min_confidence)
+ except AttributeError:
+ return _mesh_with_tasks(img_array, max_faces, min_confidence)
+
+
def main():
input_path = sys.argv[1]
output_path = sys.argv[2]
@@ -65,28 +138,19 @@ def main():
format_label = "jpg"
try:
- import mediapipe as mp
import numpy as np
import cv2
emit_progress(25, "Detecting faces")
img_array = np.array(img)
- mesh = mp.solutions.face_mesh.FaceMesh(
- static_image_mode=True,
- max_num_faces=10,
- refine_landmarks=True,
- min_detection_confidence=0.5,
- )
- results = mesh.process(img_array)
- mesh.close()
- faces_detected = 0
+ # Try legacy mp.solutions API first, fall back to mp.tasks
+ all_face_landmarks = _detect_face_mesh(img_array)
+
+ faces_detected = len(all_face_landmarks)
eyes_corrected = 0
- if results.multi_face_landmarks:
- faces_detected = len(results.multi_face_landmarks)
-
emit_progress(50, "Analyzing eyes")
# Iris landmark indices
@@ -95,8 +159,7 @@ def main():
if faces_detected > 0:
all_eyes = []
- for face_landmarks in results.multi_face_landmarks:
- landmarks = face_landmarks.landmark
+ for landmarks in all_face_landmarks:
for iris_indices in [right_iris, left_iris]:
center_idx = iris_indices[0]
contour_indices = iris_indices[1:]
diff --git a/packages/ai/python/restore.py b/packages/ai/python/restore.py
index eb9d6f19..fda50fdc 100644
--- a/packages/ai/python/restore.py
+++ b/packages/ai/python/restore.py
@@ -217,6 +217,24 @@ def _get_codeformer_path():
return CODEFORMER_LOCAL_PATH
+# ── Model path for new mp.tasks API ─────────────────────────────────
+
+_FACE_DETECT_MODEL_URL = "https://storage.googleapis.com/mediapipe-models/face_detector/blaze_face_short_range/float16/latest/blaze_face_short_range.task"
+_FACE_DETECT_MODEL_DIR = os.path.join(os.path.dirname(__file__), "..", "..", "..", ".models")
+_FACE_DETECT_MODEL_PATH = os.path.join(_FACE_DETECT_MODEL_DIR, "blaze_face_short_range.task")
+
+
+def _ensure_face_detect_model():
+ """Download the face detector model if not present."""
+ if os.path.exists(_FACE_DETECT_MODEL_PATH):
+ return _FACE_DETECT_MODEL_PATH
+ os.makedirs(_FACE_DETECT_MODEL_DIR, exist_ok=True)
+ import urllib.request
+ emit_progress(15, "Downloading face detection model")
+ urllib.request.urlretrieve(_FACE_DETECT_MODEL_URL, _FACE_DETECT_MODEL_PATH)
+ return _FACE_DETECT_MODEL_PATH
+
+
def enhance_faces(img_bgr, fidelity=0.7):
"""Enhance faces in the image using CodeFormer ONNX.
@@ -239,20 +257,57 @@ def enhance_faces(img_bgr, fidelity=0.7):
img_rgb = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)
ih, iw = img_bgr.shape[:2]
- mp_face = mp.solutions.face_detection
- detections = []
- for model_sel in [0, 1]:
- detector = mp_face.FaceDetection(
- model_selection=model_sel, min_detection_confidence=0.4
- )
- results = detector.process(img_rgb)
- detector.close()
- if results.detections:
- detections = results.detections
- break
+ try:
+ mp_face = mp.solutions.face_detection
+ detections = []
+ for model_sel in [0, 1]:
+ detector = mp_face.FaceDetection(
+ model_selection=model_sel, min_detection_confidence=0.4
+ )
+ results = detector.process(img_rgb)
+ detector.close()
+ if results.detections:
+ detections = results.detections
+ break
- if not detections:
- return img_bgr, 0
+ if not detections:
+ return img_bgr, 0
+
+ face_boxes = []
+ for detection in detections:
+ bbox = detection.location_data.relative_bounding_box
+ face_boxes.append({
+ "x": int(bbox.xmin * iw),
+ "y": int(bbox.ymin * ih),
+ "w": int(bbox.width * iw),
+ "h": int(bbox.height * ih),
+ })
+
+ except AttributeError:
+ # mediapipe >= 0.10.30 removed mp.solutions, use tasks API
+ model_path = _ensure_face_detect_model()
+ options = mp.tasks.vision.FaceDetectorOptions(
+ base_options=mp.tasks.BaseOptions(model_asset_path=model_path),
+ running_mode=mp.tasks.vision.RunningMode.IMAGE,
+ min_detection_confidence=0.4,
+ )
+ fd = mp.tasks.vision.FaceDetector.create_from_options(options)
+ mp_image = mp.Image(image_format=mp.ImageFormat.SRGB, data=img_rgb)
+ result = fd.detect(mp_image)
+ fd.close()
+
+ if not result.detections:
+ return img_bgr, 0
+
+ face_boxes = []
+ for detection in result.detections:
+ bbox = detection.bounding_box
+ face_boxes.append({
+ "x": bbox.origin_x,
+ "y": bbox.origin_y,
+ "w": bbox.width,
+ "h": bbox.height,
+ })
# Load CodeFormer model
model_path = _get_codeformer_path()
@@ -266,13 +321,11 @@ def enhance_faces(img_bgr, fidelity=0.7):
result = img_bgr.copy()
faces_enhanced = 0
- for detection in detections:
- bbox = detection.location_data.relative_bounding_box
- # Convert relative coords to absolute
- x = int(bbox.xmin * iw)
- y = int(bbox.ymin * ih)
- w = int(bbox.width * iw)
- h = int(bbox.height * ih)
+ for face_box in face_boxes:
+ x = face_box["x"]
+ y = face_box["y"]
+ w = face_box["w"]
+ h = face_box["h"]
# Skip very small faces (under 48px) - enhancement won't help
if w < 48 or h < 48: