From 341c6b0e85884762c84fca0b7f0d5336239cd975 Mon Sep 17 00:00:00 2001 From: Holden Date: Sat, 13 Jun 2026 23:13:50 +0000 Subject: [PATCH] Use InsightFace for landmark-based face crop alignment MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Immich's /api/faces endpoint only returns bounding boxes, not facial landmarks. This meant align_face() never fired and all crops fell back to a plain bbox rectangle — producing partial crops (forehead-only, off-angle faces) when Immich's detection was slightly off. Now, when ENABLE_FACE_ALIGNMENT is true and InsightFace is loaded, execute_jobs() passes the app to process_face_mode(). For each face, it expands the Immich bbox by 50%, crops that search region, runs InsightFace detection within it, and aligns the nearest face to the standard ArcFace 112×112 format using norm_crop(). Falls back to bbox crop if InsightFace finds no face in the search region. The InsightFace model is already in GPU memory from the diversity/ embedding phase, so the singleton lookup adds no load cost. --- winnow/executor.py | 14 +++++++++++- winnow/image_processing.py | 47 +++++++++++++++++++++++++++++++++----- 2 files changed, 54 insertions(+), 7 deletions(-) diff --git a/winnow/executor.py b/winnow/executor.py index e6279a9..4c673e8 100644 --- a/winnow/executor.py +++ b/winnow/executor.py @@ -155,6 +155,18 @@ def execute_jobs(jobs: list[dict]) -> None: use_full_res = Config.USE_FULL_RESOLUTION + # Load InsightFace app for landmark-based crop alignment (face mode only). + # The model is already resident from the diversity/embedding phase, so this + # is just a singleton lookup — no load cost. + insightface_app = None + if any(j["config"].get("mode", "face") == "face" for j in jobs) and Config.ENABLE_FACE_ALIGNMENT: + try: + from .embeddings import get_insightface_app + + insightface_app = get_insightface_app() + except Exception as e: + logger.debug(f"InsightFace unavailable for crop alignment: {e}") + with Progress( SpinnerColumn(), TextColumn("[progress.description]{task.description}"), @@ -216,7 +228,7 @@ def execute_jobs(jobs: list[dict]) -> None: progress.console.print(f"[red]Failed download {asset['id']}[/red]") else: saved = ( - process_face_mode(img, asset, person, person_dir, count) + process_face_mode(img, asset, person, person_dir, count, insightface_app=insightface_app) if mode == "face" else process_object_mode(img, config, person_dir, count) if mode == "object" diff --git a/winnow/image_processing.py b/winnow/image_processing.py index 08ef293..20549be 100644 --- a/winnow/image_processing.py +++ b/winnow/image_processing.py @@ -69,13 +69,15 @@ def process_face_mode( output_dir: str, count: int, min_width: int | None = None, + insightface_app=None, ) -> tuple[int, int] | None: """Crop face based on Immich metadata and save to output directory. Returns (width, height) of the saved crop, or None if no crop was saved. - If face alignment is enabled and landmarks are available, produces - an aligned 112x112 crop. Otherwise falls back to bounding box crop - with configurable margin. + When insightface_app is provided and ENABLE_FACE_ALIGNMENT is True, + re-detects the face in the Immich bbox region using InsightFace to get + precise landmarks for a proper 112x112 aligned crop. Falls back to + bounding box crop with configurable margin if alignment is unavailable. """ min_width = min_width or Config.MIN_FACE_WIDTH @@ -109,18 +111,51 @@ def process_face_mode( logger.debug(f"Face too small ({face_w:.1f}x{face_h:.1f})") return None - # Try face alignment if enabled and landmarks available + # Re-detect face with InsightFace for landmark-based alignment. + # Immich's /api/faces endpoint does not include landmarks, so the + # align_face fallback below never fires without this step. + if insightface_app is not None and Config.ENABLE_FACE_ALIGNMENT: + try: + # Expand the Immich bbox by 50% to give InsightFace enough context + # for detection and alignment, then search for the face nearest the + # centre of that region (handles group photos at the boundary). + pad_x, pad_y = face_w * 0.5, face_h * 0.5 + search_box = ( + max(0, x1 - pad_x), + max(0, y1 - pad_y), + min(img_w, x2 + pad_x), + min(img_h, y2 + pad_y), + ) + search_crop = img.crop(search_box) + detected = insightface_app.get(np.asarray(search_crop)) + if detected: + cx, cy = search_crop.width / 2, search_crop.height / 2 + best = min( + detected, + key=lambda f: abs((f.bbox[0] + f.bbox[2]) / 2 - cx) + + abs((f.bbox[1] + f.bbox[3]) / 2 - cy), + ) + kps = getattr(best, "kps", None) + if kps is not None and np.asarray(kps).shape == (5, 2): + aligned = align_face(search_crop, kps) + if aligned is not None: + _save_jpeg(aligned, os.path.join(output_dir, f"{count}.jpg")) + return aligned.size + except Exception as e: + logger.debug(f"InsightFace re-detection failed for {asset.get('id')}: {e}") + + # Landmark alignment from Immich metadata (Immich does not currently + # expose landmarks, so this path is a future-proofing fallback) if Config.ENABLE_FACE_ALIGNMENT: landmarks = face_info.get("landmarks") or face_info.get("landmark") if landmarks: - # Scale landmarks scaled_landmarks = [[lm[0] * scale_x, lm[1] * scale_y] for lm in landmarks] aligned = align_face(img, scaled_landmarks) if aligned is not None: _save_jpeg(aligned, os.path.join(output_dir, f"{count}.jpg")) return aligned.size - # Fall back to bounding box crop with configurable margin + # Final fallback: bounding box crop with configurable margin margin = Config.FACE_MARGIN margin_x, margin_y = face_w * margin, face_h * margin crop_box = (