Use InsightFace for landmark-based face crop alignment
Immich's /api/faces endpoint only returns bounding boxes, not facial landmarks. This meant align_face() never fired and all crops fell back to a plain bbox rectangle — producing partial crops (forehead-only, off-angle faces) when Immich's detection was slightly off. Now, when ENABLE_FACE_ALIGNMENT is true and InsightFace is loaded, execute_jobs() passes the app to process_face_mode(). For each face, it expands the Immich bbox by 50%, crops that search region, runs InsightFace detection within it, and aligns the nearest face to the standard ArcFace 112×112 format using norm_crop(). Falls back to bbox crop if InsightFace finds no face in the search region. The InsightFace model is already in GPU memory from the diversity/ embedding phase, so the singleton lookup adds no load cost.
This commit is contained in:
+13
-1
@@ -155,6 +155,18 @@ def execute_jobs(jobs: list[dict]) -> None:
|
|||||||
|
|
||||||
use_full_res = Config.USE_FULL_RESOLUTION
|
use_full_res = Config.USE_FULL_RESOLUTION
|
||||||
|
|
||||||
|
# Load InsightFace app for landmark-based crop alignment (face mode only).
|
||||||
|
# The model is already resident from the diversity/embedding phase, so this
|
||||||
|
# is just a singleton lookup — no load cost.
|
||||||
|
insightface_app = None
|
||||||
|
if any(j["config"].get("mode", "face") == "face" for j in jobs) and Config.ENABLE_FACE_ALIGNMENT:
|
||||||
|
try:
|
||||||
|
from .embeddings import get_insightface_app
|
||||||
|
|
||||||
|
insightface_app = get_insightface_app()
|
||||||
|
except Exception as e:
|
||||||
|
logger.debug(f"InsightFace unavailable for crop alignment: {e}")
|
||||||
|
|
||||||
with Progress(
|
with Progress(
|
||||||
SpinnerColumn(),
|
SpinnerColumn(),
|
||||||
TextColumn("[progress.description]{task.description}"),
|
TextColumn("[progress.description]{task.description}"),
|
||||||
@@ -216,7 +228,7 @@ def execute_jobs(jobs: list[dict]) -> None:
|
|||||||
progress.console.print(f"[red]Failed download {asset['id']}[/red]")
|
progress.console.print(f"[red]Failed download {asset['id']}[/red]")
|
||||||
else:
|
else:
|
||||||
saved = (
|
saved = (
|
||||||
process_face_mode(img, asset, person, person_dir, count)
|
process_face_mode(img, asset, person, person_dir, count, insightface_app=insightface_app)
|
||||||
if mode == "face"
|
if mode == "face"
|
||||||
else process_object_mode(img, config, person_dir, count)
|
else process_object_mode(img, config, person_dir, count)
|
||||||
if mode == "object"
|
if mode == "object"
|
||||||
|
|||||||
@@ -69,13 +69,15 @@ def process_face_mode(
|
|||||||
output_dir: str,
|
output_dir: str,
|
||||||
count: int,
|
count: int,
|
||||||
min_width: int | None = None,
|
min_width: int | None = None,
|
||||||
|
insightface_app=None,
|
||||||
) -> tuple[int, int] | None:
|
) -> tuple[int, int] | None:
|
||||||
"""Crop face based on Immich metadata and save to output directory.
|
"""Crop face based on Immich metadata and save to output directory.
|
||||||
|
|
||||||
Returns (width, height) of the saved crop, or None if no crop was saved.
|
Returns (width, height) of the saved crop, or None if no crop was saved.
|
||||||
If face alignment is enabled and landmarks are available, produces
|
When insightface_app is provided and ENABLE_FACE_ALIGNMENT is True,
|
||||||
an aligned 112x112 crop. Otherwise falls back to bounding box crop
|
re-detects the face in the Immich bbox region using InsightFace to get
|
||||||
with configurable margin.
|
precise landmarks for a proper 112x112 aligned crop. Falls back to
|
||||||
|
bounding box crop with configurable margin if alignment is unavailable.
|
||||||
"""
|
"""
|
||||||
min_width = min_width or Config.MIN_FACE_WIDTH
|
min_width = min_width or Config.MIN_FACE_WIDTH
|
||||||
|
|
||||||
@@ -109,18 +111,51 @@ def process_face_mode(
|
|||||||
logger.debug(f"Face too small ({face_w:.1f}x{face_h:.1f})")
|
logger.debug(f"Face too small ({face_w:.1f}x{face_h:.1f})")
|
||||||
return None
|
return None
|
||||||
|
|
||||||
# Try face alignment if enabled and landmarks available
|
# Re-detect face with InsightFace for landmark-based alignment.
|
||||||
|
# Immich's /api/faces endpoint does not include landmarks, so the
|
||||||
|
# align_face fallback below never fires without this step.
|
||||||
|
if insightface_app is not None and Config.ENABLE_FACE_ALIGNMENT:
|
||||||
|
try:
|
||||||
|
# Expand the Immich bbox by 50% to give InsightFace enough context
|
||||||
|
# for detection and alignment, then search for the face nearest the
|
||||||
|
# centre of that region (handles group photos at the boundary).
|
||||||
|
pad_x, pad_y = face_w * 0.5, face_h * 0.5
|
||||||
|
search_box = (
|
||||||
|
max(0, x1 - pad_x),
|
||||||
|
max(0, y1 - pad_y),
|
||||||
|
min(img_w, x2 + pad_x),
|
||||||
|
min(img_h, y2 + pad_y),
|
||||||
|
)
|
||||||
|
search_crop = img.crop(search_box)
|
||||||
|
detected = insightface_app.get(np.asarray(search_crop))
|
||||||
|
if detected:
|
||||||
|
cx, cy = search_crop.width / 2, search_crop.height / 2
|
||||||
|
best = min(
|
||||||
|
detected,
|
||||||
|
key=lambda f: abs((f.bbox[0] + f.bbox[2]) / 2 - cx)
|
||||||
|
+ abs((f.bbox[1] + f.bbox[3]) / 2 - cy),
|
||||||
|
)
|
||||||
|
kps = getattr(best, "kps", None)
|
||||||
|
if kps is not None and np.asarray(kps).shape == (5, 2):
|
||||||
|
aligned = align_face(search_crop, kps)
|
||||||
|
if aligned is not None:
|
||||||
|
_save_jpeg(aligned, os.path.join(output_dir, f"{count}.jpg"))
|
||||||
|
return aligned.size
|
||||||
|
except Exception as e:
|
||||||
|
logger.debug(f"InsightFace re-detection failed for {asset.get('id')}: {e}")
|
||||||
|
|
||||||
|
# Landmark alignment from Immich metadata (Immich does not currently
|
||||||
|
# expose landmarks, so this path is a future-proofing fallback)
|
||||||
if Config.ENABLE_FACE_ALIGNMENT:
|
if Config.ENABLE_FACE_ALIGNMENT:
|
||||||
landmarks = face_info.get("landmarks") or face_info.get("landmark")
|
landmarks = face_info.get("landmarks") or face_info.get("landmark")
|
||||||
if landmarks:
|
if landmarks:
|
||||||
# Scale landmarks
|
|
||||||
scaled_landmarks = [[lm[0] * scale_x, lm[1] * scale_y] for lm in landmarks]
|
scaled_landmarks = [[lm[0] * scale_x, lm[1] * scale_y] for lm in landmarks]
|
||||||
aligned = align_face(img, scaled_landmarks)
|
aligned = align_face(img, scaled_landmarks)
|
||||||
if aligned is not None:
|
if aligned is not None:
|
||||||
_save_jpeg(aligned, os.path.join(output_dir, f"{count}.jpg"))
|
_save_jpeg(aligned, os.path.join(output_dir, f"{count}.jpg"))
|
||||||
return aligned.size
|
return aligned.size
|
||||||
|
|
||||||
# Fall back to bounding box crop with configurable margin
|
# Final fallback: bounding box crop with configurable margin
|
||||||
margin = Config.FACE_MARGIN
|
margin = Config.FACE_MARGIN
|
||||||
margin_x, margin_y = face_w * margin, face_h * margin
|
margin_x, margin_y = face_w * margin, face_h * margin
|
||||||
crop_box = (
|
crop_box = (
|
||||||
|
|||||||
Reference in New Issue
Block a user