Files
Behavision/tests/test_quality_gate.py
Suriyakumarvijayanayagam dad04e8cda Behavision: face recognition for retail, edge to head office
Five components that ship as one product:

- behavision/  the recognition engine. RTSP ingest, YuNet detection, IoU
               tracking, ArcFace embeddings, a FAISS/SQLite gallery, and a
               FastAPI dashboard. Identity is decided once per TRACK from an
               average of at least three embeddings, never per frame.
- agent/       the Go edge agent: supervises the engine, holds a durable
               spool, and drains it to MQTT. Nothing is acked before the
               broker confirms.
- desktop/     the shop PC application (Wails + React + tray).
- server/      the cloud API, MQTT consumer, reports and assistant.
- web/         platform.loyaly.ai, the head-office app, embedded in the
               server binary.

The gallery stores 512-float embeddings and timestamps - no images unless
`app.store_faces` is switched on. Those embeddings are biometric personal
data under GDPR and India's DPDP: template inversion reconstructs a
recognisable face from an ArcFace vector, so data/behavision.db is treated
as a biometric database and DELETE /api/visitors/{id} is a real erasure.

CLAUDE.md carries the reasoning behind every non-obvious decision here,
including the ones that were measured and the ones that were wrong first.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HViLj9gYNRtSr7YVZmW5sn
2026-09-04 11:14:18 +05:30

162 lines
6.5 KiB
Python

"""Calibrating min_enroll_quality from measurement.
It was the last threshold in the system still chosen by hand, and it could not
have been anything else: capture() filtered by the gate before storing, so the
only data available to judge the gate was data the gate had already admitted.
"""
import numpy as np
import pytest
from behavision.calibrate import (CalibrationStore, _self_similarity,
distributions, quality_curve)
DIM = 32
def _unit(v):
return (v / np.linalg.norm(v)).astype(np.float32)
def _person(seed, n, noise):
"""n views of one person; `noise` controls how alike they are."""
rng = np.random.default_rng(seed)
base = _unit(rng.normal(size=DIM))
out = []
for scale in noise:
v = base + rng.normal(size=DIM) * scale
out.append(_unit(v))
assert len(out) == n
return np.vstack(out)
# -- leave-one-out ------------------------------------------------------
def test_self_similarity_excludes_the_sample_from_its_own_mean():
"""Including it inflates every score, and worst for the smallest sets."""
emb = _person(1, 4, [0.0, 0.0, 0.0, 4.0])
sims = _self_similarity(emb)
# The outlier is compared against the other three only, so it scores low.
assert sims[3] < 0.5
# Not ~1.0: sample 0's mean is built from two identical views AND the
# outlier, which is exactly the leave-one-out behaviour being asserted.
assert sims[0] > 0.8
def test_self_similarity_needs_two_samples():
assert len(_self_similarity(_person(1, 1, [0.0]))) == 0
# -- the curve ----------------------------------------------------------
def _store_with(tmp_path, quals, noise, seed=7):
st = CalibrationStore(tmp_path / "c.npz")
st.add("m", "alice", _person(seed, len(quals), noise),
np.array(quals, dtype=np.float32))
return st
def test_gate_lands_where_quality_stops_buying_stability(tmp_path):
"""Low-quality frames genuinely embed worse -> gate above them."""
quals = [0.30] * 6 + [0.35] * 6 + [0.70] * 6 + [0.75] * 6
noise = [2.5] * 12 + [0.05] * 12 # bad frames noisy, good ones tight
out = quality_curve(_store_with(tmp_path, quals, noise), "m")
assert out["min_enroll_quality"] >= 0.70
assert out["correlation"] > 0.5
assert 0 < out["retained_fraction"] < 1
def test_gate_drops_when_quality_predicts_nothing(tmp_path):
"""Every bucket equally good -> the gate is discarding data for free."""
quals = [0.30] * 6 + [0.35] * 6 + [0.70] * 6 + [0.75] * 6
noise = [0.05] * 24
out = quality_curve(_store_with(tmp_path, quals, noise), "m")
assert out["min_enroll_quality"] <= 0.30
assert out["retained_fraction"] == 1.0
assert "does not predict" in out["note"]
def test_a_single_noisy_low_bucket_cannot_drag_the_gate_down(tmp_path):
"""Walking down from the top stops at the first bucket that falls off."""
quals = [0.30] * 6 + [0.50] * 6 + [0.70] * 6 + [0.75] * 6 # on bin edges
noise = [3.0] * 6 + [3.0] * 6 + [0.05] * 12
out = quality_curve(_store_with(tmp_path, quals, noise), "m")
assert out["min_enroll_quality"] >= 0.70
def test_thin_buckets_are_ignored_not_averaged(tmp_path):
"""Two frames in a bucket is not a median, it is noise."""
out = quality_curve(_store_with(tmp_path, [0.3, 0.4], [0.1, 0.1]), "m")
assert out["buckets"] == []
assert "capture longer" in out["note"]
def test_archive_without_quality_says_so_instead_of_guessing(tmp_path):
st = CalibrationStore(tmp_path / "old.npz")
st.add("m", "alice", _person(1, 8, [0.1] * 8)) # no qualities passed
out = quality_curve(st, "m")
assert out["n"] == 0
assert "predates quality capture" in out["error"]
def test_misaligned_quality_array_is_treated_as_absent(tmp_path):
"""A half-upgraded archive must not pair frame i with someone else's
score; silently wrong numbers are worse than no numbers."""
st = CalibrationStore(tmp_path / "c.npz")
st.add("m", "alice", _person(1, 8, [0.1] * 8), np.arange(8, dtype=np.float32))
st.add("m", "alice", _person(2, 8, [0.1] * 8)) # embeddings only
assert len(st.get("m", "alice")) == 16
assert len(st.qualities("m", "alice")) == 0
# -- persistence and filtering -----------------------------------------
def test_quality_survives_save_and_reload(tmp_path):
st = _store_with(tmp_path, [0.3] * 8, [0.1] * 8)
st.save()
back = CalibrationStore(tmp_path / "c.npz")
assert back.models() == ["m"]
assert back.people("m") == ["alice"]
assert len(back.qualities("m", "alice")) == 8
def test_distributions_can_filter_at_analysis_time(tmp_path):
"""The whole point: re-analyse one archive against a different gate."""
st = CalibrationStore(tmp_path / "c.npz")
quals = np.array([0.2] * 10 + [0.8] * 10, dtype=np.float32)
st.add("m", "alice", _person(1, 20, [0.1] * 20), quals)
st.add("m", "bob", _person(2, 20, [0.1] * 20), quals)
wide, _, _ = distributions(st, "m", 3, min_quality=0.0)
narrow, _, _ = distributions(st, "m", 3, min_quality=0.5)
assert len(wide) > len(narrow) > 0
def test_older_archives_are_reported_not_silently_unfiltered(tmp_path):
st = CalibrationStore(tmp_path / "c.npz")
st.add("m", "alice", _person(1, 20, [0.1] * 20))
_, _, meta = distributions(st, "m", 3, min_quality=0.5)
assert meta["ungated"] == ["alice"]
def test_a_quality_on_a_bin_edge_lands_in_its_own_bin(tmp_path):
"""Accumulating a float edge (0.30 += 0.05 ...) reaches 0.5000000000000001,
so a quality of exactly 0.50 tested as below its own bucket and fell a
whole step down — moving the recommended gate, which is a number people
copy straight into a config file."""
st = _store_with(tmp_path, [0.50] * 8, [0.05] * 8)
lows = [b["lo"] for b in quality_curve(st, "m")["buckets"]]
assert lows == [0.50]
def test_report_blames_the_gate_not_the_capture(tmp_path):
"""When the gate filters out every sample, the threshold report otherwise
says 'capture more frames per person' — sending the operator back to
re-shoot a capture that was fine."""
from behavision.calibrate import format_report
from behavision.config import Config
st = CalibrationStore(tmp_path / "c.npz")
for seed, name in ((1, "alice"), (2, "bob")):
st.add("m", name, _person(seed, 20, [0.1] * 20),
np.full(20, 0.40, dtype=np.float32)) # all below the 0.65 gate
report = format_report(st, Config())
assert "dropped by min_enroll_quality=0.65" in report
assert "the gate does not fit this camera" in report