Files
Behavision/behavision/calibrate.py
Suriyakumarvijayanayagam dad04e8cda Behavision: face recognition for retail, edge to head office
Five components that ship as one product:

- behavision/  the recognition engine. RTSP ingest, YuNet detection, IoU
               tracking, ArcFace embeddings, a FAISS/SQLite gallery, and a
               FastAPI dashboard. Identity is decided once per TRACK from an
               average of at least three embeddings, never per frame.
- agent/       the Go edge agent: supervises the engine, holds a durable
               spool, and drains it to MQTT. Nothing is acked before the
               broker confirms.
- desktop/     the shop PC application (Wails + React + tray).
- server/      the cloud API, MQTT consumer, reports and assistant.
- web/         platform.loyaly.ai, the head-office app, embedded in the
               server binary.

The gallery stores 512-float embeddings and timestamps - no images unless
`app.store_faces` is switched on. Those embeddings are biometric personal
data under GDPR and India's DPDP: template inversion reconstructs a
recognisable face from an ArcFace vector, so data/behavision.db is treated
as a biometric database and DELETE /api/visitors/{id} is a real erasure.

CLAUDE.md carries the reasoning behind every non-obvious decision here,
including the ones that were measured and the ones that were wrong first.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HViLj9gYNRtSr7YVZmW5sn
2026-09-04 11:14:18 +05:30

536 lines
24 KiB
Python

"""Threshold calibration: derive match/enroll thresholds from measured data.
The three recognition thresholds are not universal constants — they describe a
particular *encoder* on a particular *camera*. Change either and the numbers
that were measured for the old pair silently stop describing the new one: the
gallery starts splitting one person into several (enroll_threshold too high) or
merging different people (match_threshold too low).
This module measures the two distributions that actually decide those numbers:
same-person similarity - how alike two views of ONE person look
cross-person similarity - how alike views of DIFFERENT people look
and reports thresholds that separate them, per model, so a model swap is a
measurement rather than a guess.
Two details make the measurement match runtime instead of merely resembling it:
- Identity is decided from the *mean* of `min_embeddings_for_id` embeddings,
never a single frame (see engine._identify). So samples are grouped and
averaged the same way before any similarity is computed. Measuring
single-frame similarity would report a much wider spread than the running
system ever sees.
- Only frames that pass the live quality gate are collected, because those are
the only frames the running system ever embeds.
Privacy: no images are written. Chips are embedded in memory and only the
resulting vectors are stored, matching the guarantee the rest of the system
makes.
"""
from __future__ import annotations
import logging
import time
from pathlib import Path
from typing import Optional
import numpy as np
log = logging.getLogger(__name__)
# Below this, a "recommendation" would be fitting noise.
MIN_GROUPS_PER_PERSON = 2
MIN_SAMPLES_PER_PERSON = 6
class CalibrationStore:
"""Embeddings per (model, person), persisted as a single .npz.
Keyed by model so one capture session can be replayed against several
encoders — that is what makes an A/B of two models fair: identical faces,
identical frames, only the encoder differs.
"""
def __init__(self, path: "Path | str"):
self.path = Path(path)
self.data: dict[str, np.ndarray] = {}
if self.path.exists():
with np.load(self.path) as npz:
self.data = {k: npz[k] for k in npz.files}
# Quality is stored under a parallel key rather than a second file, so a
# capture session stays one artefact. Suffixed (not prefixed) so the
# model/person parsing below keeps working on old archives.
_Q = "||__quality"
@staticmethod
def _key(model: str, person: str) -> str:
return f"{model}||{person}"
def add(self, model: str, person: str, embeddings: np.ndarray,
qualities: "np.ndarray | None" = None) -> int:
key = self._key(model, person)
if key in self.data and len(self.data[key]):
embeddings = np.vstack([self.data[key], embeddings])
self.data[key] = np.asarray(embeddings, dtype=np.float32)
if qualities is not None:
qkey = key + self._Q
q = np.asarray(qualities, dtype=np.float32).reshape(-1)
if qkey in self.data and len(self.data[qkey]):
q = np.concatenate([self.data[qkey], q])
self.data[qkey] = q
return len(self.data[key])
def models(self) -> "list[str]":
return sorted({k.split("||", 1)[0] for k in self.data
if not k.endswith(self._Q)})
def people(self, model: str) -> "list[str]":
return sorted(k.split("||", 1)[1] for k in self.data
if k.startswith(f"{model}||") and not k.endswith(self._Q))
def get(self, model: str, person: str) -> np.ndarray:
return self.data.get(self._key(model, person), np.empty((0, 512), np.float32))
def qualities(self, model: str, person: str) -> np.ndarray:
"""Per-embedding quality, or empty for an archive captured before
quality was recorded. Empty means 'unknown', never 'zero'."""
q = self.data.get(self._key(model, person) + self._Q,
np.empty(0, np.float32))
emb = self.get(model, person)
# A partially-upgraded archive would silently misalign the two arrays.
return q if len(q) == len(emb) else np.empty(0, np.float32)
def save(self) -> None:
self.path.parent.mkdir(parents=True, exist_ok=True)
np.savez_compressed(self.path, **self.data)
def _mean_unit(vectors: np.ndarray) -> Optional[np.ndarray]:
"""Normalised mean — the exact quantity the engine matches on."""
mean = vectors.mean(axis=0)
norm = float(np.linalg.norm(mean))
if norm < 1e-6:
return None
return (mean / norm).astype(np.float32)
def group_means(embeddings: np.ndarray, group_size: int) -> np.ndarray:
"""Chunk into groups of `group_size` and average each, mirroring the
multi-frame averaging in engine._identify. A trailing partial group is
kept only if it holds at least half a group, so one stray frame cannot
contribute a noisy 'identity' to the statistics."""
out = []
for start in range(0, len(embeddings), group_size):
chunk = embeddings[start:start + group_size]
if len(chunk) < max(2, (group_size + 1) // 2):
break
mean = _mean_unit(chunk)
if mean is not None:
out.append(mean)
return np.vstack(out) if out else np.empty((0, embeddings.shape[1]), np.float32)
def capture(source, label: str, seconds: float, cfg, detector, encoders: dict,
min_quality: Optional[float] = None
) -> "tuple[dict[str, np.ndarray], np.ndarray]":
"""Collect faces from `source`, embed with every encoder, keep the quality.
`min_quality` defaults to 0.0 — everything the detector finds is recorded,
with its score. It used to default to the live enrollment gate, which made
the gate impossible to calibrate: you cannot measure whether a threshold is
set correctly using only the data that threshold already admitted. The
filter now happens at analysis time (`distributions`), where it can be
varied, which keeps the runtime-matching property without the circularity.
`source` is anything cv2.VideoCapture accepts (webcam index, RTSP URL,
video file). Returns ({model_name: embeddings}, qualities) with the
quality array aligned to every model's rows. Raises if the source will not
open, since a silent empty capture is worse than a loud failure.
"""
import cv2
from .geometry import align_face
from .recognition import face_quality
if min_quality is None:
min_quality = 0.0
cap = cv2.VideoCapture(source)
if not cap.isOpened():
cap.release()
raise RuntimeError(f"cannot open capture source {source!r}")
per_model: dict[str, list] = {name: [] for name in encoders}
qualities: list = []
max_width = cfg.cameras[0].max_width if cfg.cameras else 1280
deadline = time.time() + seconds
seen = rejected = 0
try:
while time.time() < deadline:
ok, frame = cap.read()
if not ok or frame is None:
break
if max_width and frame.shape[1] > max_width:
scale = max_width / frame.shape[1]
frame = cv2.resize(frame, (max_width, int(frame.shape[0] * scale)),
interpolation=cv2.INTER_AREA)
detections = detector.detect(frame)
if len(detections) > 1:
# Two faces in frame makes the 'which person is this' label
# ambiguous, and a mislabelled sample poisons both curves.
rejected += 1
continue
for det in detections:
seen += 1
quality = face_quality(frame, det.box, det.kps)
if quality < min_quality:
rejected += 1
continue
# Commit a frame only if EVERY encoder embedded it. A partial
# row would desynchronise the models from each other and from
# the quality array, quietly breaking both the A/B comparison
# and the quality analysis.
row = {}
for name, enc in encoders.items():
chip = align_face(frame, det.kps, size=enc.size)
emb = enc.encode_chip(chip)
if emb is None:
break
row[name] = emb
if len(row) != len(encoders):
rejected += 1
continue
for name, emb in row.items():
per_model[name].append(emb)
qualities.append(quality)
finally:
cap.release()
kept = len(qualities)
log.info("[%s] %d faces seen, %d rejected (ambiguous/unencodable), %d kept "
"(quality p05 %.2f - p95 %.2f)", label, seen, rejected, kept,
float(np.percentile(qualities, 5)) if qualities else 0.0,
float(np.percentile(qualities, 95)) if qualities else 0.0)
return ({name: (np.vstack(v) if v else np.empty((0, 512), np.float32))
for name, v in per_model.items()},
np.asarray(qualities, dtype=np.float32))
def distributions(store: CalibrationStore, model: str, group_size: int,
min_quality: Optional[float] = None
) -> "tuple[np.ndarray, np.ndarray, dict]":
"""Same-person and cross-person similarity samples for one model.
`min_quality` filters to the frames the running system would actually
embed. Applied here rather than at capture time so the same archive can be
re-analysed against a different gate — that is what makes the gate itself
measurable instead of assumed.
"""
grouped, skipped = {}, {}
ungated, gated_out = [], {}
for person in store.people(model):
raw = store.get(model, person)
if min_quality is not None:
q = store.qualities(model, person)
if len(q):
kept = raw[q >= min_quality]
if len(kept) < len(raw):
gated_out[person] = (len(raw) - len(kept), len(raw))
raw = kept
else:
ungated.append(person)
means = group_means(raw, group_size)
if len(means) < MIN_GROUPS_PER_PERSON or len(raw) < MIN_SAMPLES_PER_PERSON:
skipped[person] = len(raw)
continue
grouped[person] = means
same, cross = [], []
people = sorted(grouped)
for i, person in enumerate(people):
m = grouped[person]
for a in range(len(m)):
for b in range(a + 1, len(m)):
same.append(float(m[a] @ m[b]))
for other in people[i + 1:]:
for va in m:
for vb in grouped[other]:
cross.append(float(va @ vb))
meta = {"people": people, "skipped": skipped,
"groups": {p: len(m) for p, m in grouped.items()}}
if ungated:
meta["ungated"] = ungated # captured before quality was recorded
if gated_out:
meta["gated_out"] = gated_out
return np.array(same), np.array(cross), meta
# -- quality gate -------------------------------------------------------
# Wide enough that a bucket holds real evidence, narrow enough to locate a
# knee; below this a bucket's median is one or two frames talking.
QUALITY_BUCKET = 0.05
MIN_BUCKET_SAMPLES = 5
# A bucket counts as "as good as this camera gets" within this fraction of the
# best bucket. Not an absolute target: what matters is whether a frame is
# materially worse than what this camera can produce, not how it compares to a
# number measured somewhere else.
KNEE_FRACTION = 0.90
def _self_similarity(embeddings: np.ndarray) -> np.ndarray:
"""Each embedding's similarity to its own person's mean, computed
leave-one-out so a sample is not compared against a mean it helped make."""
n = len(embeddings)
if n < 2:
return np.empty(0, np.float32)
total = embeddings.sum(axis=0)
others = (total - embeddings) / (n - 1)
norms = np.linalg.norm(others, axis=1, keepdims=True)
norms[norms < 1e-6] = 1.0
return np.einsum("ij,ij->i", embeddings, others / norms).astype(np.float32)
def quality_curve(store: CalibrationStore, model: str) -> dict:
"""Does face quality actually predict a usable embedding on this camera?
Pairs every captured frame's quality score with how much that frame looks
like its own person, then reports the relationship. This is the evidence
`min_enroll_quality` should be set from; it was previously the one
threshold in the system still chosen by hand.
"""
quals, sims = [], []
for person in store.people(model):
emb = store.get(model, person)
q = store.qualities(model, person)
if not len(q) or len(emb) < 2:
continue
sim = _self_similarity(emb)
if len(sim):
quals.append(q)
sims.append(sim)
if not quals:
return {"n": 0, "error": (
"no per-frame quality recorded - this archive predates quality "
"capture. Re-capture to calibrate the quality gate.")}
q = np.concatenate(quals)
sim = np.concatenate(sims)
out: dict = {"n": int(len(q)),
"quality": {"p05": round(float(np.percentile(q, 5)), 3),
"p50": round(float(np.percentile(q, 50)), 3),
"p95": round(float(np.percentile(q, 95)), 3)}}
# Whether the score means anything here at all. Undefined if every frame
# scored the same, which is itself the signature of a static artefact.
if q.std() > 1e-6 and sim.std() > 1e-6:
out["correlation"] = round(float(np.corrcoef(q, sim)[0, 1]), 3)
# Bin by integer index rather than by accumulating a float edge. Stepping
# `edge += 0.05` from 0.30 reaches 0.5000000000000001, so a quality of
# exactly 0.50 tests as *below* its own bucket and lands one step down —
# which shifts the recommended gate a whole bucket, and that number is
# copied straight into a config file.
idx = np.floor(q / QUALITY_BUCKET + 1e-9).astype(int)
buckets = []
for b in range(int(idx.min()), int(idx.max()) + 1):
sel = idx == b
if sel.sum() >= MIN_BUCKET_SAMPLES:
buckets.append({"lo": round(b * QUALITY_BUCKET, 2),
"hi": round((b + 1) * QUALITY_BUCKET, 2),
"n": int(sel.sum()),
"median_sim": round(float(np.median(sim[sel])), 3)})
out["buckets"] = buckets
if not buckets:
out["note"] = (f"fewer than {MIN_BUCKET_SAMPLES} frames in every "
"quality bucket - capture longer")
return out
best = max(b["median_sim"] for b in buckets)
target = best * KNEE_FRACTION
# Walk down from the top and stop at the first bucket that falls off, so a
# single noisy low bucket cannot drag the recommendation down with it.
gate = buckets[-1]["lo"]
for bucket in reversed(buckets):
if bucket["median_sim"] < target:
break
gate = bucket["lo"]
out["best_median_sim"] = round(float(best), 3)
out["min_enroll_quality"] = round(float(gate), 2)
out["retained_fraction"] = round(float((q >= gate).mean()), 3)
if gate <= buckets[0]["lo"]:
out["note"] = ("quality does not predict embedding stability on this "
"camera - every bucket is about as good as the best. "
"The gate is discarding frames for no measured benefit; "
"the limit here is the view, not the threshold.")
return out
def recommend(same: np.ndarray, cross: np.ndarray,
current_match: Optional[float] = None) -> dict:
"""Turn the two distributions into thresholds.
match_threshold - above the bulk of cross-person similarity, so a stranger
is not merged into an existing identity.
enroll_threshold - below the bulk of same-person similarity, so a returning
person is not minted as a duplicate.
Both are set from percentiles rather than raw min/max: one freak frame
should not move a production threshold. When the two curves overlap, no
pair of thresholds can separate them and that is reported as such rather
than papered over with a midpoint.
"""
out: dict = {"n_same": int(len(same)), "n_cross": int(len(cross))}
if len(same):
out["same"] = {"min": float(same.min()), "p01": float(np.percentile(same, 1)),
"p05": float(np.percentile(same, 5)),
"mean": float(same.mean()), "max": float(same.max())}
if len(cross):
out["cross"] = {"min": float(cross.min()), "mean": float(cross.mean()),
"p95": float(np.percentile(cross, 95)),
"p99": float(np.percentile(cross, 99)),
"max": float(cross.max())}
if not len(same):
out["error"] = ("no same-person pairs - capture more frames per person "
f"(need >={MIN_SAMPLES_PER_PERSON})")
return out
same_low = float(np.percentile(same, 5))
if len(cross):
cross_high = float(np.percentile(cross, 99))
out["separation"] = round(same_low - cross_high, 3)
if same_low <= cross_high:
out["error"] = (
"same-person and cross-person similarity OVERLAP - no threshold "
"pair separates them. Improve capture (pose, lighting, distance) "
"or use a stronger encoder before trusting any threshold.")
out["match_threshold"] = round(cross_high + 0.02, 2)
out["enroll_threshold"] = round(max(0.05, same_low - 0.02), 2)
return out
# BOTH thresholds are placed inside the gap between the curves, which
# keeps enroll < match however wide the separation turns out to be.
# Anchoring them to the distribution ends instead (same_p05 - margin)
# inverts the pair on well-separated data. match sits high in the gap
# because a false merge is unrecoverable — two people permanently share
# one identity — while a false split is a duplicate you can merge later.
gap = same_low - cross_high
out["match_threshold"] = round(cross_high + 0.55 * gap, 2)
out["enroll_threshold"] = round(cross_high + 0.15 * gap, 2)
if out["enroll_threshold"] >= out["match_threshold"]: # after rounding
out["enroll_threshold"] = round(out["match_threshold"] - 0.01, 2)
return out
out["note"] = ("only one person captured - cross-person similarity is "
"unmeasured, so match_threshold cannot be recommended. "
"Capture 2+ people to calibrate it.")
out["match_threshold"] = None
enroll = max(0.05, same_low - 0.05)
if current_match is not None and enroll > current_match - 0.01:
# config.py enforces enroll < match; never emit a value that would be
# rejected at load time against the match threshold still in force.
# Flag it, because a clamped value is the ceiling talking, not the
# data — without this, every model reports the same number and it
# reads like a measurement.
enroll = current_match - 0.01
out["clamped"] = (
f"same-person p05 is {same_low:.3f}, so the data supports an enroll "
f"threshold far above the current match threshold "
f"({current_match}). Clamped to sit just under it - calibrate "
f"match_threshold with 2+ people to lift both.")
out["enroll_threshold"] = round(enroll, 2)
return out
def format_report(store: CalibrationStore, cfg) -> str:
"""Human-readable report for every model in the store."""
group_size = cfg.tracking.min_embeddings_for_id
lines = [f"Calibration report ({store.path})",
f"grouping: mean of {group_size} embeddings (matches runtime)", ""]
for model in store.models():
same, cross, meta = distributions(store, model, group_size,
cfg.recognition.min_enroll_quality)
rec = recommend(same, cross, cfg.recognition.match_threshold)
qual = quality_curve(store, model)
lines.append(f"── {model} " + "─" * max(0, 56 - len(model)))
lines.append(f" people: {', '.join(meta['people']) or 'none'}")
if meta["skipped"]:
lines.append(" skipped (too few samples): " + ", ".join(
f"{p} ({n})" for p, n in meta["skipped"].items()))
if "same" in rec:
s = rec["same"]
lines.append(f" same-person n={rec['n_same']:<5} "
f"min {s['min']:.3f} p05 {s['p05']:.3f} mean {s['mean']:.3f}")
if "cross" in rec:
c = rec["cross"]
lines.append(f" cross-person n={rec['n_cross']:<5} "
f"mean {c['mean']:.3f} p99 {c['p99']:.3f} max {c['max']:.3f}")
if "separation" in rec:
lines.append(f" separation (same_p05 - cross_p99): {rec['separation']:+.3f}")
if "error" in rec:
lines.append(f" !! {rec['error']}")
if meta.get("gated_out"):
worst = ", ".join(f"{p} ({out}/{tot})"
for p, (out, tot) in meta["gated_out"].items())
lines.append(f" dropped by min_enroll_quality="
f"{cfg.recognition.min_enroll_quality}: {worst}")
if not meta["people"]:
# Otherwise the error above reads 'capture more frames' when
# plenty were captured and the gate discarded all of them —
# sending the operator to re-shoot instead of to the gate.
lines.append(" ^ every sample was captured, then filtered "
"out by the quality gate. The capture is fine; "
"the gate does not fit this camera.")
if "note" in rec:
lines.append(f" note: {rec['note']}")
if "clamped" in rec:
lines.append(f" clamped: {rec['clamped']}")
if meta.get("ungated"):
lines.append(" note: no per-frame quality for "
+ ", ".join(meta["ungated"])
+ " - analysed unfiltered (older capture)")
# -- quality gate ------------------------------------------------
if qual.get("error"):
lines.append(f" quality gate: {qual['error']}")
elif qual.get("buckets"):
q = qual["quality"]
lines.append("")
lines.append(f" face quality n={qual['n']:<5} "
f"p05 {q['p05']:.3f} p50 {q['p50']:.3f} "
f"p95 {q['p95']:.3f}")
if "correlation" in qual:
lines.append(" quality vs same-person similarity: "
f"r={qual['correlation']:+.3f}")
for b in qual["buckets"]:
bar = "#" * int(round(b["median_sim"] * 40))
lines.append(f" {b['lo']:.2f}-{b['hi']:.2f} "
f"n={b['n']:<4} med {b['median_sim']:.3f} {bar}")
if qual.get("note"):
lines.append(f" !! {qual['note']}")
elif qual.get("note"):
lines.append(f" quality gate: {qual['note']}")
if rec.get("match_threshold") is not None:
lines.append("")
lines.append(" recommended config/default.yaml:")
lines.append(" recognition:")
lines.append(f" match_threshold: {rec['match_threshold']}")
lines.append(f" enroll_threshold: {rec['enroll_threshold']}")
if qual.get("min_enroll_quality") is not None:
lines.append(f" min_enroll_quality: "
f"{qual['min_enroll_quality']}"
f" # keeps {qual['retained_fraction']:.0%} of faces")
elif rec.get("enroll_threshold") is not None:
lines.append("")
lines.append(" recommended (enroll only, match needs 2+ people):")
lines.append(f" enroll_threshold: {rec['enroll_threshold']}")
if qual.get("min_enroll_quality") is not None:
lines.append(f" min_enroll_quality: "
f"{qual['min_enroll_quality']}"
f" # keeps {qual['retained_fraction']:.0%} of faces")
lines.append("")
lines.append(f"current: match={cfg.recognition.match_threshold} "
f"enroll={cfg.recognition.enroll_threshold} "
f"min_enroll_quality={cfg.recognition.min_enroll_quality}")
return "\n".join(lines)