Five components that ship as one product:
- behavision/ the recognition engine. RTSP ingest, YuNet detection, IoU
tracking, ArcFace embeddings, a FAISS/SQLite gallery, and a
FastAPI dashboard. Identity is decided once per TRACK from an
average of at least three embeddings, never per frame.
- agent/ the Go edge agent: supervises the engine, holds a durable
spool, and drains it to MQTT. Nothing is acked before the
broker confirms.
- desktop/ the shop PC application (Wails + React + tray).
- server/ the cloud API, MQTT consumer, reports and assistant.
- web/ platform.loyaly.ai, the head-office app, embedded in the
server binary.
The gallery stores 512-float embeddings and timestamps - no images unless
`app.store_faces` is switched on. Those embeddings are biometric personal
data under GDPR and India's DPDP: template inversion reconstructs a
recognisable face from an ArcFace vector, so data/behavision.db is treated
as a biometric database and DELETE /api/visitors/{id} is a real erasure.
CLAUDE.md carries the reasoning behind every non-obvious decision here,
including the ones that were measured and the ones that were wrong first.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HViLj9gYNRtSr7YVZmW5sn
221 lines
9.1 KiB
Python
221 lines
9.1 KiB
Python
"""Optional age / gender / emotion estimation.
|
|
|
|
Primary gender+age model: InsightFace `genderage.onnx` (2021, CNN trained
|
|
jointly with the face-recognition stack; outputs age in YEARS). Fallback:
|
|
the 2015 Levi-Hassner Caffe nets. Emotion: FER+ ONNX.
|
|
|
|
Crop discipline — the part that made the old results absurd: gender/age
|
|
models are trained on LOOSE head crops (hair, chin, head shape included),
|
|
so they receive a 1.5x-expanded box from the full frame, never the tight
|
|
112x112 recognition chip. Only FER+ gets the aligned chip.
|
|
|
|
Everything is best-effort: any net missing or failing (e.g. out of memory)
|
|
is skipped or disabled without touching the recognition pipeline.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import threading
|
|
from pathlib import Path
|
|
|
|
import cv2
|
|
import numpy as np
|
|
|
|
log = logging.getLogger(__name__)
|
|
|
|
AGE_BUCKETS = ["0-2", "4-6", "8-12", "15-20", "25-32", "38-43", "48-53", "60+"]
|
|
GENDERS = ["Male", "Female"]
|
|
EMOTIONS = ["neutral", "happiness", "surprise", "sadness",
|
|
"anger", "disgust", "fear", "contempt"]
|
|
_CAFFE_MEAN = (78.4263377603, 87.7689143744, 114.895847746)
|
|
|
|
|
|
def _loose_head_crop(frame: np.ndarray, box, scale: float = 1.5) -> np.ndarray:
|
|
"""Square crop centered on the face box, expanded to include the whole
|
|
head; replicate-padded when it runs off-frame so aspect stays 1:1."""
|
|
x1, y1, x2, y2 = box
|
|
cx, cy = (x1 + x2) / 2.0, (y1 + y2) / 2.0
|
|
half = max(x2 - x1, y2 - y1) * scale / 2.0
|
|
fh, fw = frame.shape[:2]
|
|
gx1, gy1 = int(round(cx - half)), int(round(cy - half))
|
|
gx2, gy2 = int(round(cx + half)), int(round(cy + half))
|
|
pad_l, pad_t = max(0, -gx1), max(0, -gy1)
|
|
pad_r, pad_b = max(0, gx2 - fw), max(0, gy2 - fh)
|
|
crop = frame[max(0, gy1):min(fh, gy2), max(0, gx1):min(fw, gx2)]
|
|
if crop.size == 0:
|
|
return crop
|
|
if pad_l or pad_t or pad_r or pad_b:
|
|
crop = cv2.copyMakeBorder(crop, pad_t, pad_b, pad_l, pad_r,
|
|
cv2.BORDER_REPLICATE)
|
|
return crop
|
|
|
|
|
|
def aggregate(samples: "list[dict]") -> dict:
|
|
"""Combine per-frame estimates into one verdict for a track.
|
|
|
|
Age comes from a tiny CNN reading a single frame, so consecutive frames of
|
|
the same face can differ by a decade. Median over several frames (not mean)
|
|
keeps one wild frame from dragging the answer, and costs nothing but the
|
|
inferences already being run.
|
|
"""
|
|
samples = [s for s in samples if s]
|
|
if not samples:
|
|
return {}
|
|
out: dict = {}
|
|
ages = [s["age"] for s in samples if isinstance(s.get("age"), (int, float))]
|
|
if ages:
|
|
out["age"] = int(round(float(np.median(ages))))
|
|
out["age_spread"] = int(max(ages) - min(ages)) # honest uncertainty
|
|
for field, conf_field in (("gender", "gender_confidence"),
|
|
("emotion", "emotion_confidence"),
|
|
("age_range", None)):
|
|
votes: dict = {}
|
|
for s in samples:
|
|
v = s.get(field)
|
|
if v is None:
|
|
continue
|
|
votes.setdefault(v, []).append(s.get(conf_field, 1.0) if conf_field else 1.0)
|
|
if not votes:
|
|
continue
|
|
# most frames win; ties broken by mean confidence
|
|
best = max(votes, key=lambda k: (len(votes[k]), float(np.mean(votes[k]))))
|
|
out[field] = best
|
|
if conf_field:
|
|
out[conf_field] = round(float(np.mean(votes[best])), 3)
|
|
return out
|
|
|
|
|
|
class AttributeEstimator:
|
|
def __init__(self, models_dir: Path):
|
|
models_dir = Path(models_dir)
|
|
# cv2.dnn.Net (emotion + the Caffe fallbacks) is stateful across
|
|
# setInput/forward, so concurrent camera workers must not enter
|
|
# together. Attributes run once per TRACK, not per frame, so the
|
|
# contention this costs is negligible.
|
|
self._lock = threading.Lock()
|
|
self._genderage = None
|
|
self._ga_input = None
|
|
ga_path = models_dir / "genderage.onnx"
|
|
if ga_path.exists():
|
|
try:
|
|
import onnxruntime as ort
|
|
self._genderage = ort.InferenceSession(
|
|
str(ga_path), providers=["CPUExecutionProvider"])
|
|
inp = self._genderage.get_inputs()[0]
|
|
self._ga_input = inp.name
|
|
self._ga_size = (inp.shape[-1]
|
|
if isinstance(inp.shape[-1], int) else 96)
|
|
except Exception:
|
|
log.exception("genderage model failed to load")
|
|
self._genderage = None
|
|
|
|
# Legacy Caffe fallbacks, used only when genderage is unavailable.
|
|
self._gender = None
|
|
self._age = None
|
|
if self._genderage is None:
|
|
self._gender = self._load_caffe(models_dir, "gender")
|
|
self._age = self._load_caffe(models_dir, "age")
|
|
|
|
self._emotion = None
|
|
emo = models_dir / "emotion-ferplus-8.onnx"
|
|
if emo.exists():
|
|
try:
|
|
self._emotion = cv2.dnn.readNetFromONNX(str(emo))
|
|
except cv2.error:
|
|
log.exception("emotion model failed to load")
|
|
log.info("attributes: genderage=%s caffe(gender=%s age=%s) emotion=%s",
|
|
bool(self._genderage), bool(self._gender), bool(self._age),
|
|
bool(self._emotion))
|
|
|
|
@staticmethod
|
|
def _load_caffe(models_dir: Path, name: str):
|
|
proto = models_dir / f"{name}_deploy.prototxt"
|
|
weights = models_dir / f"{name}_net.caffemodel"
|
|
if not (proto.exists() and weights.exists()):
|
|
return None
|
|
try:
|
|
return cv2.dnn.readNetFromCaffe(str(proto), str(weights))
|
|
except cv2.error:
|
|
log.exception("%s model failed to load", name)
|
|
return None
|
|
|
|
@property
|
|
def has_genderage(self) -> bool:
|
|
return self._genderage is not None
|
|
|
|
@property
|
|
def any_loaded(self) -> bool:
|
|
return any([self._genderage, self._gender, self._age, self._emotion])
|
|
|
|
def estimate(self, frame_bgr: np.ndarray, box,
|
|
chip_bgr: np.ndarray) -> dict:
|
|
"""`frame_bgr` + `box` feed the gender/age nets (loose head crop);
|
|
`chip_bgr` (aligned 112x112) feeds FER+ emotion."""
|
|
out: dict = {}
|
|
head = _loose_head_crop(frame_bgr, box)
|
|
with self._lock:
|
|
if head.size:
|
|
if self._genderage is not None:
|
|
self._estimate_genderage(head, out)
|
|
elif self._gender is not None or self._age is not None:
|
|
self._estimate_caffe(head, out)
|
|
if (self._emotion is not None and chip_bgr is not None
|
|
and chip_bgr.size):
|
|
self._estimate_emotion(chip_bgr, out)
|
|
return out
|
|
|
|
# -- backends -------------------------------------------------------
|
|
def _estimate_genderage(self, head: np.ndarray, out: dict) -> None:
|
|
try:
|
|
size = self._ga_size
|
|
rgb = cv2.cvtColor(cv2.resize(head, (size, size)),
|
|
cv2.COLOR_BGR2RGB).astype(np.float32)
|
|
blob = rgb.transpose(2, 0, 1)[None]
|
|
pred = self._genderage.run(None, {self._ga_input: blob})[0][0]
|
|
# pred = [female_logit, male_logit, age/100]
|
|
g = np.array(pred[:2], dtype=np.float64)
|
|
probs = np.exp(g - g.max())
|
|
probs /= probs.sum()
|
|
out["gender"] = "Male" if pred[1] > pred[0] else "Female"
|
|
out["gender_confidence"] = round(float(probs.max()), 3)
|
|
out["age"] = int(round(float(pred[2]) * 100))
|
|
except Exception:
|
|
log.warning("genderage failed at inference - disabled")
|
|
self._genderage = None
|
|
|
|
def _estimate_caffe(self, head: np.ndarray, out: dict) -> None:
|
|
blob = cv2.dnn.blobFromImage(
|
|
cv2.resize(head, (227, 227)), 1.0, (227, 227),
|
|
_CAFFE_MEAN, swapRB=False)
|
|
if self._gender is not None:
|
|
try:
|
|
self._gender.setInput(blob)
|
|
probs = self._gender.forward().ravel()
|
|
out["gender"] = GENDERS[int(np.argmax(probs))]
|
|
out["gender_confidence"] = round(float(probs.max()), 3)
|
|
except cv2.error:
|
|
log.warning("gender net failed at inference - disabled")
|
|
self._gender = None
|
|
if self._age is not None:
|
|
try:
|
|
self._age.setInput(blob)
|
|
probs = self._age.forward().ravel()
|
|
out["age_range"] = AGE_BUCKETS[int(np.argmax(probs))]
|
|
except cv2.error:
|
|
log.warning("age net failed at inference - disabled")
|
|
self._age = None
|
|
|
|
def _estimate_emotion(self, chip_bgr: np.ndarray, out: dict) -> None:
|
|
try:
|
|
gray = cv2.cvtColor(chip_bgr, cv2.COLOR_BGR2GRAY)
|
|
blob = cv2.resize(gray, (64, 64)).astype(np.float32)[None, None]
|
|
self._emotion.setInput(blob)
|
|
logits = self._emotion.forward().ravel()
|
|
exp = np.exp(logits - logits.max())
|
|
probs = exp / exp.sum()
|
|
out["emotion"] = EMOTIONS[int(np.argmax(probs))]
|
|
out["emotion_confidence"] = round(float(probs.max()), 3)
|
|
except cv2.error:
|
|
log.warning("emotion net failed at inference - disabled")
|
|
self._emotion = None
|