"""Optional age / gender / emotion estimation. Primary gender+age model: InsightFace `genderage.onnx` (2021, CNN trained jointly with the face-recognition stack; outputs age in YEARS). Fallback: the 2015 Levi-Hassner Caffe nets. Emotion: FER+ ONNX. Crop discipline — the part that made the old results absurd: gender/age models are trained on LOOSE head crops (hair, chin, head shape included), so they receive a 1.5x-expanded box from the full frame, never the tight 112x112 recognition chip. Only FER+ gets the aligned chip. Everything is best-effort: any net missing or failing (e.g. out of memory) is skipped or disabled without touching the recognition pipeline. """ from __future__ import annotations import logging import threading from pathlib import Path import cv2 import numpy as np log = logging.getLogger(__name__) AGE_BUCKETS = ["0-2", "4-6", "8-12", "15-20", "25-32", "38-43", "48-53", "60+"] GENDERS = ["Male", "Female"] EMOTIONS = ["neutral", "happiness", "surprise", "sadness", "anger", "disgust", "fear", "contempt"] _CAFFE_MEAN = (78.4263377603, 87.7689143744, 114.895847746) def _loose_head_crop(frame: np.ndarray, box, scale: float = 1.5) -> np.ndarray: """Square crop centered on the face box, expanded to include the whole head; replicate-padded when it runs off-frame so aspect stays 1:1.""" x1, y1, x2, y2 = box cx, cy = (x1 + x2) / 2.0, (y1 + y2) / 2.0 half = max(x2 - x1, y2 - y1) * scale / 2.0 fh, fw = frame.shape[:2] gx1, gy1 = int(round(cx - half)), int(round(cy - half)) gx2, gy2 = int(round(cx + half)), int(round(cy + half)) pad_l, pad_t = max(0, -gx1), max(0, -gy1) pad_r, pad_b = max(0, gx2 - fw), max(0, gy2 - fh) crop = frame[max(0, gy1):min(fh, gy2), max(0, gx1):min(fw, gx2)] if crop.size == 0: return crop if pad_l or pad_t or pad_r or pad_b: crop = cv2.copyMakeBorder(crop, pad_t, pad_b, pad_l, pad_r, cv2.BORDER_REPLICATE) return crop def aggregate(samples: "list[dict]") -> dict: """Combine per-frame estimates into one verdict for a track. Age comes from a tiny CNN reading a single frame, so consecutive frames of the same face can differ by a decade. Median over several frames (not mean) keeps one wild frame from dragging the answer, and costs nothing but the inferences already being run. """ samples = [s for s in samples if s] if not samples: return {} out: dict = {} ages = [s["age"] for s in samples if isinstance(s.get("age"), (int, float))] if ages: out["age"] = int(round(float(np.median(ages)))) out["age_spread"] = int(max(ages) - min(ages)) # honest uncertainty for field, conf_field in (("gender", "gender_confidence"), ("emotion", "emotion_confidence"), ("age_range", None)): votes: dict = {} for s in samples: v = s.get(field) if v is None: continue votes.setdefault(v, []).append(s.get(conf_field, 1.0) if conf_field else 1.0) if not votes: continue # most frames win; ties broken by mean confidence best = max(votes, key=lambda k: (len(votes[k]), float(np.mean(votes[k])))) out[field] = best if conf_field: out[conf_field] = round(float(np.mean(votes[best])), 3) return out class AttributeEstimator: def __init__(self, models_dir: Path): models_dir = Path(models_dir) # cv2.dnn.Net (emotion + the Caffe fallbacks) is stateful across # setInput/forward, so concurrent camera workers must not enter # together. Attributes run once per TRACK, not per frame, so the # contention this costs is negligible. self._lock = threading.Lock() self._genderage = None self._ga_input = None ga_path = models_dir / "genderage.onnx" if ga_path.exists(): try: import onnxruntime as ort self._genderage = ort.InferenceSession( str(ga_path), providers=["CPUExecutionProvider"]) inp = self._genderage.get_inputs()[0] self._ga_input = inp.name self._ga_size = (inp.shape[-1] if isinstance(inp.shape[-1], int) else 96) except Exception: log.exception("genderage model failed to load") self._genderage = None # Legacy Caffe fallbacks, used only when genderage is unavailable. self._gender = None self._age = None if self._genderage is None: self._gender = self._load_caffe(models_dir, "gender") self._age = self._load_caffe(models_dir, "age") self._emotion = None emo = models_dir / "emotion-ferplus-8.onnx" if emo.exists(): try: self._emotion = cv2.dnn.readNetFromONNX(str(emo)) except cv2.error: log.exception("emotion model failed to load") log.info("attributes: genderage=%s caffe(gender=%s age=%s) emotion=%s", bool(self._genderage), bool(self._gender), bool(self._age), bool(self._emotion)) @staticmethod def _load_caffe(models_dir: Path, name: str): proto = models_dir / f"{name}_deploy.prototxt" weights = models_dir / f"{name}_net.caffemodel" if not (proto.exists() and weights.exists()): return None try: return cv2.dnn.readNetFromCaffe(str(proto), str(weights)) except cv2.error: log.exception("%s model failed to load", name) return None @property def has_genderage(self) -> bool: return self._genderage is not None @property def any_loaded(self) -> bool: return any([self._genderage, self._gender, self._age, self._emotion]) def estimate(self, frame_bgr: np.ndarray, box, chip_bgr: np.ndarray) -> dict: """`frame_bgr` + `box` feed the gender/age nets (loose head crop); `chip_bgr` (aligned 112x112) feeds FER+ emotion.""" out: dict = {} head = _loose_head_crop(frame_bgr, box) with self._lock: if head.size: if self._genderage is not None: self._estimate_genderage(head, out) elif self._gender is not None or self._age is not None: self._estimate_caffe(head, out) if (self._emotion is not None and chip_bgr is not None and chip_bgr.size): self._estimate_emotion(chip_bgr, out) return out # -- backends ------------------------------------------------------- def _estimate_genderage(self, head: np.ndarray, out: dict) -> None: try: size = self._ga_size rgb = cv2.cvtColor(cv2.resize(head, (size, size)), cv2.COLOR_BGR2RGB).astype(np.float32) blob = rgb.transpose(2, 0, 1)[None] pred = self._genderage.run(None, {self._ga_input: blob})[0][0] # pred = [female_logit, male_logit, age/100] g = np.array(pred[:2], dtype=np.float64) probs = np.exp(g - g.max()) probs /= probs.sum() out["gender"] = "Male" if pred[1] > pred[0] else "Female" out["gender_confidence"] = round(float(probs.max()), 3) out["age"] = int(round(float(pred[2]) * 100)) except Exception: log.warning("genderage failed at inference - disabled") self._genderage = None def _estimate_caffe(self, head: np.ndarray, out: dict) -> None: blob = cv2.dnn.blobFromImage( cv2.resize(head, (227, 227)), 1.0, (227, 227), _CAFFE_MEAN, swapRB=False) if self._gender is not None: try: self._gender.setInput(blob) probs = self._gender.forward().ravel() out["gender"] = GENDERS[int(np.argmax(probs))] out["gender_confidence"] = round(float(probs.max()), 3) except cv2.error: log.warning("gender net failed at inference - disabled") self._gender = None if self._age is not None: try: self._age.setInput(blob) probs = self._age.forward().ravel() out["age_range"] = AGE_BUCKETS[int(np.argmax(probs))] except cv2.error: log.warning("age net failed at inference - disabled") self._age = None def _estimate_emotion(self, chip_bgr: np.ndarray, out: dict) -> None: try: gray = cv2.cvtColor(chip_bgr, cv2.COLOR_BGR2GRAY) blob = cv2.resize(gray, (64, 64)).astype(np.float32)[None, None] self._emotion.setInput(blob) logits = self._emotion.forward().ravel() exp = np.exp(logits - logits.max()) probs = exp / exp.sum() out["emotion"] = EMOTIONS[int(np.argmax(probs))] out["emotion_confidence"] = round(float(probs.max()), 3) except cv2.error: log.warning("emotion net failed at inference - disabled") self._emotion = None