"""ArcFace embedding + face quality assessment. Preprocessing contract (this is where the old codebase broke recognition): aligned 112x112 BGR chip -> [RGB if the model wants it] -> (x - 127.5) / 127.5 -> NCHW float32. Exactly one colour conversion, the normalisation ArcFace was trained with, and L2-normalised output so cosine similarity is a plain dot product. Channel order is per-model (see color_order_for): ArcFace/InsightFace want RGB, AdaFace wants BGR. Same scaling, opposite channel order, and no error if you get it wrong — hence the explicit table. """ from __future__ import annotations import logging from pathlib import Path from typing import Optional import cv2 import numpy as np from .geometry import align_face log = logging.getLogger(__name__) EMBEDDING_DIM = 512 # Tried in order; first one that exists AND loads wins. The lightweight # MobileFaceNet (13 MB, same WebFace600K training data) sits before the # 260 MB r100 export because a model that loads on every boot beats a # marginally more accurate one that fails under memory pressure — and # embeddings from different models are incompatible, so boot-to-boot # consistency matters. Pin one explicitly via recognition config if needed. MODEL_CANDIDATES = [ "adaface_ir101.onnx", # best, ~250 MB - only loads on a roomy machine "adaface_ir50.onnx", # ~170 MB, quality-adaptive: best for blur/low light "w600k_r50.onnx", # ~166 MB, IJB-C 97.25 vs mbf's 95.02 "arcface_int8.onnx", "w600k_mbf.onnx", # 13 MB, always loads "arcface.onnx", # r100, 249 MB ] # Channel order each family was trained on. InsightFace/ArcFace exports expect # RGB; AdaFace expects BGR (mean=0.5/std=0.5, which is the same (x-127.5)/127.5 # scaling — ONLY the channel order differs). Getting it wrong raises nothing. # Measured on this camera with w600k_r50: the same face chip encoded RGB vs # BGR cross-matches at 0.945, so it is a mild perturbation rather than a # catastrophe (faces are low-saturation, so R and B correlate). Still declared # per model: it costs one lookup, it is the documented contract each model was # trained under, and it removes a needless source of drift near the 0.42 # decision boundary. BGR_MODELS = ("adaface",) DEFAULT_COLOR_ORDER = "RGB" def color_order_for(model_name: str) -> str: name = model_name.lower() return "BGR" if any(tag in name for tag in BGR_MODELS) else DEFAULT_COLOR_ORDER class ArcFaceEncoder: def __init__(self, models_dir: Path, model_file: str = "", color_order: str = ""): import onnxruntime as ort candidates = [model_file] if model_file else MODEL_CANDIDATES providers = ort.get_available_providers() self.session = None for name in candidates: model_path = Path(models_dir) / name if not model_path.exists(): continue try: self.session = ort.InferenceSession(str(model_path), providers=providers) except Exception: # Graph optimization of a large model needs a big transient # allocation; retry unoptimized before giving up on it. log.warning("%s: optimized load failed, retrying without " "graph optimization (low memory?)", name) try: so = ort.SessionOptions() so.graph_optimization_level = ( ort.GraphOptimizationLevel.ORT_DISABLE_ALL) so.enable_mem_pattern = False self.session = ort.InferenceSession( str(model_path), sess_options=so, providers=providers) except Exception: log.warning("%s: unusable on this machine, trying next " "candidate", name) continue self.model_name = model_path.stem break if self.session is None: raise FileNotFoundError( f"no usable recognition model in {models_dir} " f"(tried {', '.join(candidates)}) - " "run: python -m behavision setup-models") # Explicit config wins; otherwise infer from the model family. self.color_order = (color_order or color_order_for(self.model_name)).upper() if self.color_order not in ("RGB", "BGR"): raise ValueError(f"color_order must be RGB or BGR, got {color_order!r}") inp = self.session.get_inputs()[0] self.input_name = inp.name # Introspect instead of assuming: works for 112x112 r50/r100/mbf exports. self.size = inp.shape[-1] if isinstance(inp.shape[-1], int) else 112 self.output_name = self.session.get_outputs()[0].name log.info("recognition model '%s' loaded (input %sx%s, %s, providers=%s)", self.model_name, self.size, self.size, self.color_order, providers) def encode_chip(self, chip_bgr: np.ndarray) -> Optional[np.ndarray]: """Embed an already-aligned BGR chip. Returns unit-norm float32[512].""" if chip_bgr is None or chip_bgr.size == 0: return None if chip_bgr.shape[:2] != (self.size, self.size): chip_bgr = cv2.resize(chip_bgr, (self.size, self.size)) # Exactly one colour conversion, and only when the model wants RGB. chip = (cv2.cvtColor(chip_bgr, cv2.COLOR_BGR2RGB) if self.color_order == "RGB" else chip_bgr) blob = ((chip.astype(np.float32) - 127.5) / 127.5).transpose(2, 0, 1)[None] emb = self.session.run([self.output_name], {self.input_name: blob})[0][0] emb = np.asarray(emb, dtype=np.float32).ravel() norm = float(np.linalg.norm(emb)) if norm < 1e-6: # degenerate output — never store or match this return None return emb / norm def encode(self, frame_bgr: np.ndarray, kps: np.ndarray) -> Optional[np.ndarray]: """Align (full-frame landmarks) then embed.""" chip = align_face(frame_bgr, kps, size=self.size) return self.encode_chip(chip) def face_quality(frame: np.ndarray, box, kps: np.ndarray) -> float: """0..1 quality score used to gate enrollment. Every term is clamped so the weighted sum stays interpretable (the old code's size term made its own threshold unreachable).""" x1, y1, x2, y2 = box crop = frame[y1:y2, x1:x2] if crop.size == 0: return 0.0 gray = cv2.cvtColor(crop, cv2.COLOR_BGR2GRAY) sharpness = min(1.0, cv2.Laplacian(gray, cv2.CV_64F).var() / 250.0) size_score = min(1.0, min(x2 - x1, y2 - y1) / 112.0) mean_b = float(gray.mean()) if 60.0 <= mean_b <= 190.0: brightness = 1.0 elif mean_b < 60.0: brightness = max(0.0, mean_b / 60.0) else: brightness = max(0.0, (255.0 - mean_b) / 65.0) # Frontality: nose tip should sit near the horizontal midpoint of the # eyes; offset is normalised by inter-eye distance. eye_l, eye_r, nose = kps[0], kps[1], kps[2] eye_dist = float(np.linalg.norm(eye_r - eye_l)) if eye_dist < 1.0: frontality = 0.0 else: mid_x = (eye_l[0] + eye_r[0]) / 2.0 frontality = max(0.0, 1.0 - 2.0 * abs(nose[0] - mid_x) / eye_dist) score = (0.35 * sharpness + 0.25 * size_score + 0.15 * brightness + 0.25 * frontality) return float(max(0.0, min(1.0, score)))