Behavision: face recognition for retail, edge to head office
Five components that ship as one product:
- behavision/ the recognition engine. RTSP ingest, YuNet detection, IoU
tracking, ArcFace embeddings, a FAISS/SQLite gallery, and a
FastAPI dashboard. Identity is decided once per TRACK from an
average of at least three embeddings, never per frame.
- agent/ the Go edge agent: supervises the engine, holds a durable
spool, and drains it to MQTT. Nothing is acked before the
broker confirms.
- desktop/ the shop PC application (Wails + React + tray).
- server/ the cloud API, MQTT consumer, reports and assistant.
- web/ platform.loyaly.ai, the head-office app, embedded in the
server binary.
The gallery stores 512-float embeddings and timestamps - no images unless
`app.store_faces` is switched on. Those embeddings are biometric personal
data under GDPR and India's DPDP: template inversion reconstructs a
recognisable face from an ArcFace vector, so data/behavision.db is treated
as a biometric database and DELETE /api/visitors/{id} is a real erasure.
CLAUDE.md carries the reasoning behind every non-obvious decision here,
including the ones that were measured and the ones that were wrong first.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HViLj9gYNRtSr7YVZmW5sn
This commit is contained in:
167
behavision/recognition.py
Normal file
167
behavision/recognition.py
Normal file
@@ -0,0 +1,167 @@
|
||||
"""ArcFace embedding + face quality assessment.
|
||||
|
||||
Preprocessing contract (this is where the old codebase broke recognition):
|
||||
aligned 112x112 BGR chip -> [RGB if the model wants it] ->
|
||||
(x - 127.5) / 127.5 -> NCHW float32.
|
||||
Exactly one colour conversion, the normalisation ArcFace was trained with,
|
||||
and L2-normalised output so cosine similarity is a plain dot product.
|
||||
Channel order is per-model (see color_order_for): ArcFace/InsightFace want
|
||||
RGB, AdaFace wants BGR. Same scaling, opposite channel order, and no error
|
||||
if you get it wrong — hence the explicit table.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import cv2
|
||||
import numpy as np
|
||||
|
||||
from .geometry import align_face
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
EMBEDDING_DIM = 512
|
||||
|
||||
# Tried in order; first one that exists AND loads wins. The lightweight
|
||||
# MobileFaceNet (13 MB, same WebFace600K training data) sits before the
|
||||
# 260 MB r100 export because a model that loads on every boot beats a
|
||||
# marginally more accurate one that fails under memory pressure — and
|
||||
# embeddings from different models are incompatible, so boot-to-boot
|
||||
# consistency matters. Pin one explicitly via recognition config if needed.
|
||||
MODEL_CANDIDATES = [
|
||||
"adaface_ir101.onnx", # best, ~250 MB - only loads on a roomy machine
|
||||
"adaface_ir50.onnx", # ~170 MB, quality-adaptive: best for blur/low light
|
||||
"w600k_r50.onnx", # ~166 MB, IJB-C 97.25 vs mbf's 95.02
|
||||
"arcface_int8.onnx",
|
||||
"w600k_mbf.onnx", # 13 MB, always loads
|
||||
"arcface.onnx", # r100, 249 MB
|
||||
]
|
||||
|
||||
# Channel order each family was trained on. InsightFace/ArcFace exports expect
|
||||
# RGB; AdaFace expects BGR (mean=0.5/std=0.5, which is the same (x-127.5)/127.5
|
||||
# scaling — ONLY the channel order differs). Getting it wrong raises nothing.
|
||||
# Measured on this camera with w600k_r50: the same face chip encoded RGB vs
|
||||
# BGR cross-matches at 0.945, so it is a mild perturbation rather than a
|
||||
# catastrophe (faces are low-saturation, so R and B correlate). Still declared
|
||||
# per model: it costs one lookup, it is the documented contract each model was
|
||||
# trained under, and it removes a needless source of drift near the 0.42
|
||||
# decision boundary.
|
||||
BGR_MODELS = ("adaface",)
|
||||
DEFAULT_COLOR_ORDER = "RGB"
|
||||
|
||||
|
||||
def color_order_for(model_name: str) -> str:
|
||||
name = model_name.lower()
|
||||
return "BGR" if any(tag in name for tag in BGR_MODELS) else DEFAULT_COLOR_ORDER
|
||||
|
||||
|
||||
class ArcFaceEncoder:
|
||||
def __init__(self, models_dir: Path, model_file: str = "",
|
||||
color_order: str = ""):
|
||||
import onnxruntime as ort
|
||||
|
||||
candidates = [model_file] if model_file else MODEL_CANDIDATES
|
||||
providers = ort.get_available_providers()
|
||||
self.session = None
|
||||
for name in candidates:
|
||||
model_path = Path(models_dir) / name
|
||||
if not model_path.exists():
|
||||
continue
|
||||
try:
|
||||
self.session = ort.InferenceSession(str(model_path),
|
||||
providers=providers)
|
||||
except Exception:
|
||||
# Graph optimization of a large model needs a big transient
|
||||
# allocation; retry unoptimized before giving up on it.
|
||||
log.warning("%s: optimized load failed, retrying without "
|
||||
"graph optimization (low memory?)", name)
|
||||
try:
|
||||
so = ort.SessionOptions()
|
||||
so.graph_optimization_level = (
|
||||
ort.GraphOptimizationLevel.ORT_DISABLE_ALL)
|
||||
so.enable_mem_pattern = False
|
||||
self.session = ort.InferenceSession(
|
||||
str(model_path), sess_options=so, providers=providers)
|
||||
except Exception:
|
||||
log.warning("%s: unusable on this machine, trying next "
|
||||
"candidate", name)
|
||||
continue
|
||||
self.model_name = model_path.stem
|
||||
break
|
||||
if self.session is None:
|
||||
raise FileNotFoundError(
|
||||
f"no usable recognition model in {models_dir} "
|
||||
f"(tried {', '.join(candidates)}) - "
|
||||
"run: python -m behavision setup-models")
|
||||
# Explicit config wins; otherwise infer from the model family.
|
||||
self.color_order = (color_order or color_order_for(self.model_name)).upper()
|
||||
if self.color_order not in ("RGB", "BGR"):
|
||||
raise ValueError(f"color_order must be RGB or BGR, got {color_order!r}")
|
||||
inp = self.session.get_inputs()[0]
|
||||
self.input_name = inp.name
|
||||
# Introspect instead of assuming: works for 112x112 r50/r100/mbf exports.
|
||||
self.size = inp.shape[-1] if isinstance(inp.shape[-1], int) else 112
|
||||
self.output_name = self.session.get_outputs()[0].name
|
||||
log.info("recognition model '%s' loaded (input %sx%s, %s, providers=%s)",
|
||||
self.model_name, self.size, self.size, self.color_order,
|
||||
providers)
|
||||
|
||||
def encode_chip(self, chip_bgr: np.ndarray) -> Optional[np.ndarray]:
|
||||
"""Embed an already-aligned BGR chip. Returns unit-norm float32[512]."""
|
||||
if chip_bgr is None or chip_bgr.size == 0:
|
||||
return None
|
||||
if chip_bgr.shape[:2] != (self.size, self.size):
|
||||
chip_bgr = cv2.resize(chip_bgr, (self.size, self.size))
|
||||
# Exactly one colour conversion, and only when the model wants RGB.
|
||||
chip = (cv2.cvtColor(chip_bgr, cv2.COLOR_BGR2RGB)
|
||||
if self.color_order == "RGB" else chip_bgr)
|
||||
blob = ((chip.astype(np.float32) - 127.5) / 127.5).transpose(2, 0, 1)[None]
|
||||
emb = self.session.run([self.output_name], {self.input_name: blob})[0][0]
|
||||
emb = np.asarray(emb, dtype=np.float32).ravel()
|
||||
norm = float(np.linalg.norm(emb))
|
||||
if norm < 1e-6: # degenerate output — never store or match this
|
||||
return None
|
||||
return emb / norm
|
||||
|
||||
def encode(self, frame_bgr: np.ndarray, kps: np.ndarray) -> Optional[np.ndarray]:
|
||||
"""Align (full-frame landmarks) then embed."""
|
||||
chip = align_face(frame_bgr, kps, size=self.size)
|
||||
return self.encode_chip(chip)
|
||||
|
||||
|
||||
def face_quality(frame: np.ndarray, box, kps: np.ndarray) -> float:
|
||||
"""0..1 quality score used to gate enrollment. Every term is clamped so
|
||||
the weighted sum stays interpretable (the old code's size term made its
|
||||
own threshold unreachable)."""
|
||||
x1, y1, x2, y2 = box
|
||||
crop = frame[y1:y2, x1:x2]
|
||||
if crop.size == 0:
|
||||
return 0.0
|
||||
gray = cv2.cvtColor(crop, cv2.COLOR_BGR2GRAY)
|
||||
|
||||
sharpness = min(1.0, cv2.Laplacian(gray, cv2.CV_64F).var() / 250.0)
|
||||
size_score = min(1.0, min(x2 - x1, y2 - y1) / 112.0)
|
||||
|
||||
mean_b = float(gray.mean())
|
||||
if 60.0 <= mean_b <= 190.0:
|
||||
brightness = 1.0
|
||||
elif mean_b < 60.0:
|
||||
brightness = max(0.0, mean_b / 60.0)
|
||||
else:
|
||||
brightness = max(0.0, (255.0 - mean_b) / 65.0)
|
||||
|
||||
# Frontality: nose tip should sit near the horizontal midpoint of the
|
||||
# eyes; offset is normalised by inter-eye distance.
|
||||
eye_l, eye_r, nose = kps[0], kps[1], kps[2]
|
||||
eye_dist = float(np.linalg.norm(eye_r - eye_l))
|
||||
if eye_dist < 1.0:
|
||||
frontality = 0.0
|
||||
else:
|
||||
mid_x = (eye_l[0] + eye_r[0]) / 2.0
|
||||
frontality = max(0.0, 1.0 - 2.0 * abs(nose[0] - mid_x) / eye_dist)
|
||||
|
||||
score = (0.35 * sharpness + 0.25 * size_score
|
||||
+ 0.15 * brightness + 0.25 * frontality)
|
||||
return float(max(0.0, min(1.0, score)))
|
||||
Reference in New Issue
Block a user