Five components that ship as one product:
- behavision/ the recognition engine. RTSP ingest, YuNet detection, IoU
tracking, ArcFace embeddings, a FAISS/SQLite gallery, and a
FastAPI dashboard. Identity is decided once per TRACK from an
average of at least three embeddings, never per frame.
- agent/ the Go edge agent: supervises the engine, holds a durable
spool, and drains it to MQTT. Nothing is acked before the
broker confirms.
- desktop/ the shop PC application (Wails + React + tray).
- server/ the cloud API, MQTT consumer, reports and assistant.
- web/ platform.loyaly.ai, the head-office app, embedded in the
server binary.
The gallery stores 512-float embeddings and timestamps - no images unless
`app.store_faces` is switched on. Those embeddings are biometric personal
data under GDPR and India's DPDP: template inversion reconstructs a
recognisable face from an ArcFace vector, so data/behavision.db is treated
as a biometric database and DELETE /api/visitors/{id} is a real erasure.
CLAUDE.md carries the reasoning behind every non-obvious decision here,
including the ones that were measured and the ones that were wrong first.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HViLj9gYNRtSr7YVZmW5sn
168 lines
7.4 KiB
Python
168 lines
7.4 KiB
Python
"""ArcFace embedding + face quality assessment.
|
|
|
|
Preprocessing contract (this is where the old codebase broke recognition):
|
|
aligned 112x112 BGR chip -> [RGB if the model wants it] ->
|
|
(x - 127.5) / 127.5 -> NCHW float32.
|
|
Exactly one colour conversion, the normalisation ArcFace was trained with,
|
|
and L2-normalised output so cosine similarity is a plain dot product.
|
|
Channel order is per-model (see color_order_for): ArcFace/InsightFace want
|
|
RGB, AdaFace wants BGR. Same scaling, opposite channel order, and no error
|
|
if you get it wrong — hence the explicit table.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
from pathlib import Path
|
|
from typing import Optional
|
|
|
|
import cv2
|
|
import numpy as np
|
|
|
|
from .geometry import align_face
|
|
|
|
log = logging.getLogger(__name__)
|
|
|
|
EMBEDDING_DIM = 512
|
|
|
|
# Tried in order; first one that exists AND loads wins. The lightweight
|
|
# MobileFaceNet (13 MB, same WebFace600K training data) sits before the
|
|
# 260 MB r100 export because a model that loads on every boot beats a
|
|
# marginally more accurate one that fails under memory pressure — and
|
|
# embeddings from different models are incompatible, so boot-to-boot
|
|
# consistency matters. Pin one explicitly via recognition config if needed.
|
|
MODEL_CANDIDATES = [
|
|
"adaface_ir101.onnx", # best, ~250 MB - only loads on a roomy machine
|
|
"adaface_ir50.onnx", # ~170 MB, quality-adaptive: best for blur/low light
|
|
"w600k_r50.onnx", # ~166 MB, IJB-C 97.25 vs mbf's 95.02
|
|
"arcface_int8.onnx",
|
|
"w600k_mbf.onnx", # 13 MB, always loads
|
|
"arcface.onnx", # r100, 249 MB
|
|
]
|
|
|
|
# Channel order each family was trained on. InsightFace/ArcFace exports expect
|
|
# RGB; AdaFace expects BGR (mean=0.5/std=0.5, which is the same (x-127.5)/127.5
|
|
# scaling — ONLY the channel order differs). Getting it wrong raises nothing.
|
|
# Measured on this camera with w600k_r50: the same face chip encoded RGB vs
|
|
# BGR cross-matches at 0.945, so it is a mild perturbation rather than a
|
|
# catastrophe (faces are low-saturation, so R and B correlate). Still declared
|
|
# per model: it costs one lookup, it is the documented contract each model was
|
|
# trained under, and it removes a needless source of drift near the 0.42
|
|
# decision boundary.
|
|
BGR_MODELS = ("adaface",)
|
|
DEFAULT_COLOR_ORDER = "RGB"
|
|
|
|
|
|
def color_order_for(model_name: str) -> str:
|
|
name = model_name.lower()
|
|
return "BGR" if any(tag in name for tag in BGR_MODELS) else DEFAULT_COLOR_ORDER
|
|
|
|
|
|
class ArcFaceEncoder:
|
|
def __init__(self, models_dir: Path, model_file: str = "",
|
|
color_order: str = ""):
|
|
import onnxruntime as ort
|
|
|
|
candidates = [model_file] if model_file else MODEL_CANDIDATES
|
|
providers = ort.get_available_providers()
|
|
self.session = None
|
|
for name in candidates:
|
|
model_path = Path(models_dir) / name
|
|
if not model_path.exists():
|
|
continue
|
|
try:
|
|
self.session = ort.InferenceSession(str(model_path),
|
|
providers=providers)
|
|
except Exception:
|
|
# Graph optimization of a large model needs a big transient
|
|
# allocation; retry unoptimized before giving up on it.
|
|
log.warning("%s: optimized load failed, retrying without "
|
|
"graph optimization (low memory?)", name)
|
|
try:
|
|
so = ort.SessionOptions()
|
|
so.graph_optimization_level = (
|
|
ort.GraphOptimizationLevel.ORT_DISABLE_ALL)
|
|
so.enable_mem_pattern = False
|
|
self.session = ort.InferenceSession(
|
|
str(model_path), sess_options=so, providers=providers)
|
|
except Exception:
|
|
log.warning("%s: unusable on this machine, trying next "
|
|
"candidate", name)
|
|
continue
|
|
self.model_name = model_path.stem
|
|
break
|
|
if self.session is None:
|
|
raise FileNotFoundError(
|
|
f"no usable recognition model in {models_dir} "
|
|
f"(tried {', '.join(candidates)}) - "
|
|
"run: python -m behavision setup-models")
|
|
# Explicit config wins; otherwise infer from the model family.
|
|
self.color_order = (color_order or color_order_for(self.model_name)).upper()
|
|
if self.color_order not in ("RGB", "BGR"):
|
|
raise ValueError(f"color_order must be RGB or BGR, got {color_order!r}")
|
|
inp = self.session.get_inputs()[0]
|
|
self.input_name = inp.name
|
|
# Introspect instead of assuming: works for 112x112 r50/r100/mbf exports.
|
|
self.size = inp.shape[-1] if isinstance(inp.shape[-1], int) else 112
|
|
self.output_name = self.session.get_outputs()[0].name
|
|
log.info("recognition model '%s' loaded (input %sx%s, %s, providers=%s)",
|
|
self.model_name, self.size, self.size, self.color_order,
|
|
providers)
|
|
|
|
def encode_chip(self, chip_bgr: np.ndarray) -> Optional[np.ndarray]:
|
|
"""Embed an already-aligned BGR chip. Returns unit-norm float32[512]."""
|
|
if chip_bgr is None or chip_bgr.size == 0:
|
|
return None
|
|
if chip_bgr.shape[:2] != (self.size, self.size):
|
|
chip_bgr = cv2.resize(chip_bgr, (self.size, self.size))
|
|
# Exactly one colour conversion, and only when the model wants RGB.
|
|
chip = (cv2.cvtColor(chip_bgr, cv2.COLOR_BGR2RGB)
|
|
if self.color_order == "RGB" else chip_bgr)
|
|
blob = ((chip.astype(np.float32) - 127.5) / 127.5).transpose(2, 0, 1)[None]
|
|
emb = self.session.run([self.output_name], {self.input_name: blob})[0][0]
|
|
emb = np.asarray(emb, dtype=np.float32).ravel()
|
|
norm = float(np.linalg.norm(emb))
|
|
if norm < 1e-6: # degenerate output — never store or match this
|
|
return None
|
|
return emb / norm
|
|
|
|
def encode(self, frame_bgr: np.ndarray, kps: np.ndarray) -> Optional[np.ndarray]:
|
|
"""Align (full-frame landmarks) then embed."""
|
|
chip = align_face(frame_bgr, kps, size=self.size)
|
|
return self.encode_chip(chip)
|
|
|
|
|
|
def face_quality(frame: np.ndarray, box, kps: np.ndarray) -> float:
|
|
"""0..1 quality score used to gate enrollment. Every term is clamped so
|
|
the weighted sum stays interpretable (the old code's size term made its
|
|
own threshold unreachable)."""
|
|
x1, y1, x2, y2 = box
|
|
crop = frame[y1:y2, x1:x2]
|
|
if crop.size == 0:
|
|
return 0.0
|
|
gray = cv2.cvtColor(crop, cv2.COLOR_BGR2GRAY)
|
|
|
|
sharpness = min(1.0, cv2.Laplacian(gray, cv2.CV_64F).var() / 250.0)
|
|
size_score = min(1.0, min(x2 - x1, y2 - y1) / 112.0)
|
|
|
|
mean_b = float(gray.mean())
|
|
if 60.0 <= mean_b <= 190.0:
|
|
brightness = 1.0
|
|
elif mean_b < 60.0:
|
|
brightness = max(0.0, mean_b / 60.0)
|
|
else:
|
|
brightness = max(0.0, (255.0 - mean_b) / 65.0)
|
|
|
|
# Frontality: nose tip should sit near the horizontal midpoint of the
|
|
# eyes; offset is normalised by inter-eye distance.
|
|
eye_l, eye_r, nose = kps[0], kps[1], kps[2]
|
|
eye_dist = float(np.linalg.norm(eye_r - eye_l))
|
|
if eye_dist < 1.0:
|
|
frontality = 0.0
|
|
else:
|
|
mid_x = (eye_l[0] + eye_r[0]) / 2.0
|
|
frontality = max(0.0, 1.0 - 2.0 * abs(nose[0] - mid_x) / eye_dist)
|
|
|
|
score = (0.35 * sharpness + 0.25 * size_score
|
|
+ 0.15 * brightness + 0.25 * frontality)
|
|
return float(max(0.0, min(1.0, score)))
|