Use InsightFace for landmark-based face crop alignment

Immich's /api/faces endpoint only returns bounding boxes, not facial
landmarks. This meant align_face() never fired and all crops fell back
to a plain bbox rectangle — producing partial crops (forehead-only,
off-angle faces) when Immich's detection was slightly off.

Now, when ENABLE_FACE_ALIGNMENT is true and InsightFace is loaded,
execute_jobs() passes the app to process_face_mode(). For each face,
it expands the Immich bbox by 50%, crops that search region, runs
InsightFace detection within it, and aligns the nearest face to the
standard ArcFace 112×112 format using norm_crop(). Falls back to bbox
crop if InsightFace finds no face in the search region.

The InsightFace model is already in GPU memory from the diversity/
embedding phase, so the singleton lookup adds no load cost.
This commit is contained in:
2026-06-13 23:13:50 +00:00
parent 6fb3d2f61d
commit 341c6b0e85
2 changed files with 54 additions and 7 deletions
+13 -1
View File
@@ -155,6 +155,18 @@ def execute_jobs(jobs: list[dict]) -> None:
use_full_res = Config.USE_FULL_RESOLUTION use_full_res = Config.USE_FULL_RESOLUTION
# Load InsightFace app for landmark-based crop alignment (face mode only).
# The model is already resident from the diversity/embedding phase, so this
# is just a singleton lookup — no load cost.
insightface_app = None
if any(j["config"].get("mode", "face") == "face" for j in jobs) and Config.ENABLE_FACE_ALIGNMENT:
try:
from .embeddings import get_insightface_app
insightface_app = get_insightface_app()
except Exception as e:
logger.debug(f"InsightFace unavailable for crop alignment: {e}")
with Progress( with Progress(
SpinnerColumn(), SpinnerColumn(),
TextColumn("[progress.description]{task.description}"), TextColumn("[progress.description]{task.description}"),
@@ -216,7 +228,7 @@ def execute_jobs(jobs: list[dict]) -> None:
progress.console.print(f"[red]Failed download {asset['id']}[/red]") progress.console.print(f"[red]Failed download {asset['id']}[/red]")
else: else:
saved = ( saved = (
process_face_mode(img, asset, person, person_dir, count) process_face_mode(img, asset, person, person_dir, count, insightface_app=insightface_app)
if mode == "face" if mode == "face"
else process_object_mode(img, config, person_dir, count) else process_object_mode(img, config, person_dir, count)
if mode == "object" if mode == "object"
+41 -6
View File
@@ -69,13 +69,15 @@ def process_face_mode(
output_dir: str, output_dir: str,
count: int, count: int,
min_width: int | None = None, min_width: int | None = None,
insightface_app=None,
) -> tuple[int, int] | None: ) -> tuple[int, int] | None:
"""Crop face based on Immich metadata and save to output directory. """Crop face based on Immich metadata and save to output directory.
Returns (width, height) of the saved crop, or None if no crop was saved. Returns (width, height) of the saved crop, or None if no crop was saved.
If face alignment is enabled and landmarks are available, produces When insightface_app is provided and ENABLE_FACE_ALIGNMENT is True,
an aligned 112x112 crop. Otherwise falls back to bounding box crop re-detects the face in the Immich bbox region using InsightFace to get
with configurable margin. precise landmarks for a proper 112x112 aligned crop. Falls back to
bounding box crop with configurable margin if alignment is unavailable.
""" """
min_width = min_width or Config.MIN_FACE_WIDTH min_width = min_width or Config.MIN_FACE_WIDTH
@@ -109,18 +111,51 @@ def process_face_mode(
logger.debug(f"Face too small ({face_w:.1f}x{face_h:.1f})") logger.debug(f"Face too small ({face_w:.1f}x{face_h:.1f})")
return None return None
# Try face alignment if enabled and landmarks available # Re-detect face with InsightFace for landmark-based alignment.
# Immich's /api/faces endpoint does not include landmarks, so the
# align_face fallback below never fires without this step.
if insightface_app is not None and Config.ENABLE_FACE_ALIGNMENT:
try:
# Expand the Immich bbox by 50% to give InsightFace enough context
# for detection and alignment, then search for the face nearest the
# centre of that region (handles group photos at the boundary).
pad_x, pad_y = face_w * 0.5, face_h * 0.5
search_box = (
max(0, x1 - pad_x),
max(0, y1 - pad_y),
min(img_w, x2 + pad_x),
min(img_h, y2 + pad_y),
)
search_crop = img.crop(search_box)
detected = insightface_app.get(np.asarray(search_crop))
if detected:
cx, cy = search_crop.width / 2, search_crop.height / 2
best = min(
detected,
key=lambda f: abs((f.bbox[0] + f.bbox[2]) / 2 - cx)
+ abs((f.bbox[1] + f.bbox[3]) / 2 - cy),
)
kps = getattr(best, "kps", None)
if kps is not None and np.asarray(kps).shape == (5, 2):
aligned = align_face(search_crop, kps)
if aligned is not None:
_save_jpeg(aligned, os.path.join(output_dir, f"{count}.jpg"))
return aligned.size
except Exception as e:
logger.debug(f"InsightFace re-detection failed for {asset.get('id')}: {e}")
# Landmark alignment from Immich metadata (Immich does not currently
# expose landmarks, so this path is a future-proofing fallback)
if Config.ENABLE_FACE_ALIGNMENT: if Config.ENABLE_FACE_ALIGNMENT:
landmarks = face_info.get("landmarks") or face_info.get("landmark") landmarks = face_info.get("landmarks") or face_info.get("landmark")
if landmarks: if landmarks:
# Scale landmarks
scaled_landmarks = [[lm[0] * scale_x, lm[1] * scale_y] for lm in landmarks] scaled_landmarks = [[lm[0] * scale_x, lm[1] * scale_y] for lm in landmarks]
aligned = align_face(img, scaled_landmarks) aligned = align_face(img, scaled_landmarks)
if aligned is not None: if aligned is not None:
_save_jpeg(aligned, os.path.join(output_dir, f"{count}.jpg")) _save_jpeg(aligned, os.path.join(output_dir, f"{count}.jpg"))
return aligned.size return aligned.size
# Fall back to bounding box crop with configurable margin # Final fallback: bounding box crop with configurable margin
margin = Config.FACE_MARGIN margin = Config.FACE_MARGIN
margin_x, margin_y = face_w * margin, face_h * margin margin_x, margin_y = face_w * margin, face_h * margin
crop_box = ( crop_box = (