"""Threshold calibration: derive match/enroll thresholds from measured data. The three recognition thresholds are not universal constants — they describe a particular *encoder* on a particular *camera*. Change either and the numbers that were measured for the old pair silently stop describing the new one: the gallery starts splitting one person into several (enroll_threshold too high) or merging different people (match_threshold too low). This module measures the two distributions that actually decide those numbers: same-person similarity - how alike two views of ONE person look cross-person similarity - how alike views of DIFFERENT people look and reports thresholds that separate them, per model, so a model swap is a measurement rather than a guess. Two details make the measurement match runtime instead of merely resembling it: - Identity is decided from the *mean* of `min_embeddings_for_id` embeddings, never a single frame (see engine._identify). So samples are grouped and averaged the same way before any similarity is computed. Measuring single-frame similarity would report a much wider spread than the running system ever sees. - Only frames that pass the live quality gate are collected, because those are the only frames the running system ever embeds. Privacy: no images are written. Chips are embedded in memory and only the resulting vectors are stored, matching the guarantee the rest of the system makes. """ from __future__ import annotations import logging import time from pathlib import Path from typing import Optional import numpy as np log = logging.getLogger(__name__) # Below this, a "recommendation" would be fitting noise. MIN_GROUPS_PER_PERSON = 2 MIN_SAMPLES_PER_PERSON = 6 class CalibrationStore: """Embeddings per (model, person), persisted as a single .npz. Keyed by model so one capture session can be replayed against several encoders — that is what makes an A/B of two models fair: identical faces, identical frames, only the encoder differs. """ def __init__(self, path: "Path | str"): self.path = Path(path) self.data: dict[str, np.ndarray] = {} if self.path.exists(): with np.load(self.path) as npz: self.data = {k: npz[k] for k in npz.files} # Quality is stored under a parallel key rather than a second file, so a # capture session stays one artefact. Suffixed (not prefixed) so the # model/person parsing below keeps working on old archives. _Q = "||__quality" @staticmethod def _key(model: str, person: str) -> str: return f"{model}||{person}" def add(self, model: str, person: str, embeddings: np.ndarray, qualities: "np.ndarray | None" = None) -> int: key = self._key(model, person) if key in self.data and len(self.data[key]): embeddings = np.vstack([self.data[key], embeddings]) self.data[key] = np.asarray(embeddings, dtype=np.float32) if qualities is not None: qkey = key + self._Q q = np.asarray(qualities, dtype=np.float32).reshape(-1) if qkey in self.data and len(self.data[qkey]): q = np.concatenate([self.data[qkey], q]) self.data[qkey] = q return len(self.data[key]) def models(self) -> "list[str]": return sorted({k.split("||", 1)[0] for k in self.data if not k.endswith(self._Q)}) def people(self, model: str) -> "list[str]": return sorted(k.split("||", 1)[1] for k in self.data if k.startswith(f"{model}||") and not k.endswith(self._Q)) def get(self, model: str, person: str) -> np.ndarray: return self.data.get(self._key(model, person), np.empty((0, 512), np.float32)) def qualities(self, model: str, person: str) -> np.ndarray: """Per-embedding quality, or empty for an archive captured before quality was recorded. Empty means 'unknown', never 'zero'.""" q = self.data.get(self._key(model, person) + self._Q, np.empty(0, np.float32)) emb = self.get(model, person) # A partially-upgraded archive would silently misalign the two arrays. return q if len(q) == len(emb) else np.empty(0, np.float32) def save(self) -> None: self.path.parent.mkdir(parents=True, exist_ok=True) np.savez_compressed(self.path, **self.data) def _mean_unit(vectors: np.ndarray) -> Optional[np.ndarray]: """Normalised mean — the exact quantity the engine matches on.""" mean = vectors.mean(axis=0) norm = float(np.linalg.norm(mean)) if norm < 1e-6: return None return (mean / norm).astype(np.float32) def group_means(embeddings: np.ndarray, group_size: int) -> np.ndarray: """Chunk into groups of `group_size` and average each, mirroring the multi-frame averaging in engine._identify. A trailing partial group is kept only if it holds at least half a group, so one stray frame cannot contribute a noisy 'identity' to the statistics.""" out = [] for start in range(0, len(embeddings), group_size): chunk = embeddings[start:start + group_size] if len(chunk) < max(2, (group_size + 1) // 2): break mean = _mean_unit(chunk) if mean is not None: out.append(mean) return np.vstack(out) if out else np.empty((0, embeddings.shape[1]), np.float32) def capture(source, label: str, seconds: float, cfg, detector, encoders: dict, min_quality: Optional[float] = None ) -> "tuple[dict[str, np.ndarray], np.ndarray]": """Collect faces from `source`, embed with every encoder, keep the quality. `min_quality` defaults to 0.0 — everything the detector finds is recorded, with its score. It used to default to the live enrollment gate, which made the gate impossible to calibrate: you cannot measure whether a threshold is set correctly using only the data that threshold already admitted. The filter now happens at analysis time (`distributions`), where it can be varied, which keeps the runtime-matching property without the circularity. `source` is anything cv2.VideoCapture accepts (webcam index, RTSP URL, video file). Returns ({model_name: embeddings}, qualities) with the quality array aligned to every model's rows. Raises if the source will not open, since a silent empty capture is worse than a loud failure. """ import cv2 from .geometry import align_face from .recognition import face_quality if min_quality is None: min_quality = 0.0 cap = cv2.VideoCapture(source) if not cap.isOpened(): cap.release() raise RuntimeError(f"cannot open capture source {source!r}") per_model: dict[str, list] = {name: [] for name in encoders} qualities: list = [] max_width = cfg.cameras[0].max_width if cfg.cameras else 1280 deadline = time.time() + seconds seen = rejected = 0 try: while time.time() < deadline: ok, frame = cap.read() if not ok or frame is None: break if max_width and frame.shape[1] > max_width: scale = max_width / frame.shape[1] frame = cv2.resize(frame, (max_width, int(frame.shape[0] * scale)), interpolation=cv2.INTER_AREA) detections = detector.detect(frame) if len(detections) > 1: # Two faces in frame makes the 'which person is this' label # ambiguous, and a mislabelled sample poisons both curves. rejected += 1 continue for det in detections: seen += 1 quality = face_quality(frame, det.box, det.kps) if quality < min_quality: rejected += 1 continue # Commit a frame only if EVERY encoder embedded it. A partial # row would desynchronise the models from each other and from # the quality array, quietly breaking both the A/B comparison # and the quality analysis. row = {} for name, enc in encoders.items(): chip = align_face(frame, det.kps, size=enc.size) emb = enc.encode_chip(chip) if emb is None: break row[name] = emb if len(row) != len(encoders): rejected += 1 continue for name, emb in row.items(): per_model[name].append(emb) qualities.append(quality) finally: cap.release() kept = len(qualities) log.info("[%s] %d faces seen, %d rejected (ambiguous/unencodable), %d kept " "(quality p05 %.2f - p95 %.2f)", label, seen, rejected, kept, float(np.percentile(qualities, 5)) if qualities else 0.0, float(np.percentile(qualities, 95)) if qualities else 0.0) return ({name: (np.vstack(v) if v else np.empty((0, 512), np.float32)) for name, v in per_model.items()}, np.asarray(qualities, dtype=np.float32)) def distributions(store: CalibrationStore, model: str, group_size: int, min_quality: Optional[float] = None ) -> "tuple[np.ndarray, np.ndarray, dict]": """Same-person and cross-person similarity samples for one model. `min_quality` filters to the frames the running system would actually embed. Applied here rather than at capture time so the same archive can be re-analysed against a different gate — that is what makes the gate itself measurable instead of assumed. """ grouped, skipped = {}, {} ungated, gated_out = [], {} for person in store.people(model): raw = store.get(model, person) if min_quality is not None: q = store.qualities(model, person) if len(q): kept = raw[q >= min_quality] if len(kept) < len(raw): gated_out[person] = (len(raw) - len(kept), len(raw)) raw = kept else: ungated.append(person) means = group_means(raw, group_size) if len(means) < MIN_GROUPS_PER_PERSON or len(raw) < MIN_SAMPLES_PER_PERSON: skipped[person] = len(raw) continue grouped[person] = means same, cross = [], [] people = sorted(grouped) for i, person in enumerate(people): m = grouped[person] for a in range(len(m)): for b in range(a + 1, len(m)): same.append(float(m[a] @ m[b])) for other in people[i + 1:]: for va in m: for vb in grouped[other]: cross.append(float(va @ vb)) meta = {"people": people, "skipped": skipped, "groups": {p: len(m) for p, m in grouped.items()}} if ungated: meta["ungated"] = ungated # captured before quality was recorded if gated_out: meta["gated_out"] = gated_out return np.array(same), np.array(cross), meta # -- quality gate ------------------------------------------------------- # Wide enough that a bucket holds real evidence, narrow enough to locate a # knee; below this a bucket's median is one or two frames talking. QUALITY_BUCKET = 0.05 MIN_BUCKET_SAMPLES = 5 # A bucket counts as "as good as this camera gets" within this fraction of the # best bucket. Not an absolute target: what matters is whether a frame is # materially worse than what this camera can produce, not how it compares to a # number measured somewhere else. KNEE_FRACTION = 0.90 def _self_similarity(embeddings: np.ndarray) -> np.ndarray: """Each embedding's similarity to its own person's mean, computed leave-one-out so a sample is not compared against a mean it helped make.""" n = len(embeddings) if n < 2: return np.empty(0, np.float32) total = embeddings.sum(axis=0) others = (total - embeddings) / (n - 1) norms = np.linalg.norm(others, axis=1, keepdims=True) norms[norms < 1e-6] = 1.0 return np.einsum("ij,ij->i", embeddings, others / norms).astype(np.float32) def quality_curve(store: CalibrationStore, model: str) -> dict: """Does face quality actually predict a usable embedding on this camera? Pairs every captured frame's quality score with how much that frame looks like its own person, then reports the relationship. This is the evidence `min_enroll_quality` should be set from; it was previously the one threshold in the system still chosen by hand. """ quals, sims = [], [] for person in store.people(model): emb = store.get(model, person) q = store.qualities(model, person) if not len(q) or len(emb) < 2: continue sim = _self_similarity(emb) if len(sim): quals.append(q) sims.append(sim) if not quals: return {"n": 0, "error": ( "no per-frame quality recorded - this archive predates quality " "capture. Re-capture to calibrate the quality gate.")} q = np.concatenate(quals) sim = np.concatenate(sims) out: dict = {"n": int(len(q)), "quality": {"p05": round(float(np.percentile(q, 5)), 3), "p50": round(float(np.percentile(q, 50)), 3), "p95": round(float(np.percentile(q, 95)), 3)}} # Whether the score means anything here at all. Undefined if every frame # scored the same, which is itself the signature of a static artefact. if q.std() > 1e-6 and sim.std() > 1e-6: out["correlation"] = round(float(np.corrcoef(q, sim)[0, 1]), 3) # Bin by integer index rather than by accumulating a float edge. Stepping # `edge += 0.05` from 0.30 reaches 0.5000000000000001, so a quality of # exactly 0.50 tests as *below* its own bucket and lands one step down — # which shifts the recommended gate a whole bucket, and that number is # copied straight into a config file. idx = np.floor(q / QUALITY_BUCKET + 1e-9).astype(int) buckets = [] for b in range(int(idx.min()), int(idx.max()) + 1): sel = idx == b if sel.sum() >= MIN_BUCKET_SAMPLES: buckets.append({"lo": round(b * QUALITY_BUCKET, 2), "hi": round((b + 1) * QUALITY_BUCKET, 2), "n": int(sel.sum()), "median_sim": round(float(np.median(sim[sel])), 3)}) out["buckets"] = buckets if not buckets: out["note"] = (f"fewer than {MIN_BUCKET_SAMPLES} frames in every " "quality bucket - capture longer") return out best = max(b["median_sim"] for b in buckets) target = best * KNEE_FRACTION # Walk down from the top and stop at the first bucket that falls off, so a # single noisy low bucket cannot drag the recommendation down with it. gate = buckets[-1]["lo"] for bucket in reversed(buckets): if bucket["median_sim"] < target: break gate = bucket["lo"] out["best_median_sim"] = round(float(best), 3) out["min_enroll_quality"] = round(float(gate), 2) out["retained_fraction"] = round(float((q >= gate).mean()), 3) if gate <= buckets[0]["lo"]: out["note"] = ("quality does not predict embedding stability on this " "camera - every bucket is about as good as the best. " "The gate is discarding frames for no measured benefit; " "the limit here is the view, not the threshold.") return out def recommend(same: np.ndarray, cross: np.ndarray, current_match: Optional[float] = None) -> dict: """Turn the two distributions into thresholds. match_threshold - above the bulk of cross-person similarity, so a stranger is not merged into an existing identity. enroll_threshold - below the bulk of same-person similarity, so a returning person is not minted as a duplicate. Both are set from percentiles rather than raw min/max: one freak frame should not move a production threshold. When the two curves overlap, no pair of thresholds can separate them and that is reported as such rather than papered over with a midpoint. """ out: dict = {"n_same": int(len(same)), "n_cross": int(len(cross))} if len(same): out["same"] = {"min": float(same.min()), "p01": float(np.percentile(same, 1)), "p05": float(np.percentile(same, 5)), "mean": float(same.mean()), "max": float(same.max())} if len(cross): out["cross"] = {"min": float(cross.min()), "mean": float(cross.mean()), "p95": float(np.percentile(cross, 95)), "p99": float(np.percentile(cross, 99)), "max": float(cross.max())} if not len(same): out["error"] = ("no same-person pairs - capture more frames per person " f"(need >={MIN_SAMPLES_PER_PERSON})") return out same_low = float(np.percentile(same, 5)) if len(cross): cross_high = float(np.percentile(cross, 99)) out["separation"] = round(same_low - cross_high, 3) if same_low <= cross_high: out["error"] = ( "same-person and cross-person similarity OVERLAP - no threshold " "pair separates them. Improve capture (pose, lighting, distance) " "or use a stronger encoder before trusting any threshold.") out["match_threshold"] = round(cross_high + 0.02, 2) out["enroll_threshold"] = round(max(0.05, same_low - 0.02), 2) return out # BOTH thresholds are placed inside the gap between the curves, which # keeps enroll < match however wide the separation turns out to be. # Anchoring them to the distribution ends instead (same_p05 - margin) # inverts the pair on well-separated data. match sits high in the gap # because a false merge is unrecoverable — two people permanently share # one identity — while a false split is a duplicate you can merge later. gap = same_low - cross_high out["match_threshold"] = round(cross_high + 0.55 * gap, 2) out["enroll_threshold"] = round(cross_high + 0.15 * gap, 2) if out["enroll_threshold"] >= out["match_threshold"]: # after rounding out["enroll_threshold"] = round(out["match_threshold"] - 0.01, 2) return out out["note"] = ("only one person captured - cross-person similarity is " "unmeasured, so match_threshold cannot be recommended. " "Capture 2+ people to calibrate it.") out["match_threshold"] = None enroll = max(0.05, same_low - 0.05) if current_match is not None and enroll > current_match - 0.01: # config.py enforces enroll < match; never emit a value that would be # rejected at load time against the match threshold still in force. # Flag it, because a clamped value is the ceiling talking, not the # data — without this, every model reports the same number and it # reads like a measurement. enroll = current_match - 0.01 out["clamped"] = ( f"same-person p05 is {same_low:.3f}, so the data supports an enroll " f"threshold far above the current match threshold " f"({current_match}). Clamped to sit just under it - calibrate " f"match_threshold with 2+ people to lift both.") out["enroll_threshold"] = round(enroll, 2) return out def format_report(store: CalibrationStore, cfg) -> str: """Human-readable report for every model in the store.""" group_size = cfg.tracking.min_embeddings_for_id lines = [f"Calibration report ({store.path})", f"grouping: mean of {group_size} embeddings (matches runtime)", ""] for model in store.models(): same, cross, meta = distributions(store, model, group_size, cfg.recognition.min_enroll_quality) rec = recommend(same, cross, cfg.recognition.match_threshold) qual = quality_curve(store, model) lines.append(f"── {model} " + "─" * max(0, 56 - len(model))) lines.append(f" people: {', '.join(meta['people']) or 'none'}") if meta["skipped"]: lines.append(" skipped (too few samples): " + ", ".join( f"{p} ({n})" for p, n in meta["skipped"].items())) if "same" in rec: s = rec["same"] lines.append(f" same-person n={rec['n_same']:<5} " f"min {s['min']:.3f} p05 {s['p05']:.3f} mean {s['mean']:.3f}") if "cross" in rec: c = rec["cross"] lines.append(f" cross-person n={rec['n_cross']:<5} " f"mean {c['mean']:.3f} p99 {c['p99']:.3f} max {c['max']:.3f}") if "separation" in rec: lines.append(f" separation (same_p05 - cross_p99): {rec['separation']:+.3f}") if "error" in rec: lines.append(f" !! {rec['error']}") if meta.get("gated_out"): worst = ", ".join(f"{p} ({out}/{tot})" for p, (out, tot) in meta["gated_out"].items()) lines.append(f" dropped by min_enroll_quality=" f"{cfg.recognition.min_enroll_quality}: {worst}") if not meta["people"]: # Otherwise the error above reads 'capture more frames' when # plenty were captured and the gate discarded all of them — # sending the operator to re-shoot instead of to the gate. lines.append(" ^ every sample was captured, then filtered " "out by the quality gate. The capture is fine; " "the gate does not fit this camera.") if "note" in rec: lines.append(f" note: {rec['note']}") if "clamped" in rec: lines.append(f" clamped: {rec['clamped']}") if meta.get("ungated"): lines.append(" note: no per-frame quality for " + ", ".join(meta["ungated"]) + " - analysed unfiltered (older capture)") # -- quality gate ------------------------------------------------ if qual.get("error"): lines.append(f" quality gate: {qual['error']}") elif qual.get("buckets"): q = qual["quality"] lines.append("") lines.append(f" face quality n={qual['n']:<5} " f"p05 {q['p05']:.3f} p50 {q['p50']:.3f} " f"p95 {q['p95']:.3f}") if "correlation" in qual: lines.append(" quality vs same-person similarity: " f"r={qual['correlation']:+.3f}") for b in qual["buckets"]: bar = "#" * int(round(b["median_sim"] * 40)) lines.append(f" {b['lo']:.2f}-{b['hi']:.2f} " f"n={b['n']:<4} med {b['median_sim']:.3f} {bar}") if qual.get("note"): lines.append(f" !! {qual['note']}") elif qual.get("note"): lines.append(f" quality gate: {qual['note']}") if rec.get("match_threshold") is not None: lines.append("") lines.append(" recommended config/default.yaml:") lines.append(" recognition:") lines.append(f" match_threshold: {rec['match_threshold']}") lines.append(f" enroll_threshold: {rec['enroll_threshold']}") if qual.get("min_enroll_quality") is not None: lines.append(f" min_enroll_quality: " f"{qual['min_enroll_quality']}" f" # keeps {qual['retained_fraction']:.0%} of faces") elif rec.get("enroll_threshold") is not None: lines.append("") lines.append(" recommended (enroll only, match needs 2+ people):") lines.append(f" enroll_threshold: {rec['enroll_threshold']}") if qual.get("min_enroll_quality") is not None: lines.append(f" min_enroll_quality: " f"{qual['min_enroll_quality']}" f" # keeps {qual['retained_fraction']:.0%} of faces") lines.append("") lines.append(f"current: match={cfg.recognition.match_threshold} " f"enroll={cfg.recognition.enroll_threshold} " f"min_enroll_quality={cfg.recognition.min_enroll_quality}") return "\n".join(lines)