Files
Behavision/tests/test_calibrate.py
Suriyakumarvijayanayagam dad04e8cda Behavision: face recognition for retail, edge to head office
Five components that ship as one product:

- behavision/  the recognition engine. RTSP ingest, YuNet detection, IoU
               tracking, ArcFace embeddings, a FAISS/SQLite gallery, and a
               FastAPI dashboard. Identity is decided once per TRACK from an
               average of at least three embeddings, never per frame.
- agent/       the Go edge agent: supervises the engine, holds a durable
               spool, and drains it to MQTT. Nothing is acked before the
               broker confirms.
- desktop/     the shop PC application (Wails + React + tray).
- server/      the cloud API, MQTT consumer, reports and assistant.
- web/         platform.loyaly.ai, the head-office app, embedded in the
               server binary.

The gallery stores 512-float embeddings and timestamps - no images unless
`app.store_faces` is switched on. Those embeddings are biometric personal
data under GDPR and India's DPDP: template inversion reconstructs a
recognisable face from an ArcFace vector, so data/behavision.db is treated
as a biometric database and DELETE /api/visitors/{id} is a real erasure.

CLAUDE.md carries the reasoning behind every non-obvious decision here,
including the ones that were measured and the ones that were wrong first.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HViLj9gYNRtSr7YVZmW5sn
2026-09-04 11:14:18 +05:30

123 lines
4.6 KiB
Python

"""Calibration maths — no camera, no models (synthetic unit vectors only)."""
import numpy as np
import pytest
from behavision.calibrate import (CalibrationStore, distributions, group_means,
recommend)
from behavision.config import Config
DIM = 64
def _cluster(seed, n, tightness=0.97):
"""n unit vectors clustered around one direction; higher tightness =
more like the same person across frames."""
rng = np.random.default_rng(seed)
centre = rng.normal(size=DIM)
centre /= np.linalg.norm(centre)
out = []
for _ in range(n):
v = tightness * centre + (1 - tightness) * rng.normal(size=DIM)
out.append(v / np.linalg.norm(v))
return np.array(out, dtype=np.float32)
def test_group_means_mirrors_runtime_averaging():
embs = _cluster(1, 9)
means = group_means(embs, 3)
assert means.shape == (3, DIM)
assert np.allclose(np.linalg.norm(means, axis=1), 1.0, atol=1e-5)
def test_group_means_drops_a_tiny_trailing_group():
# 7 samples at group 3 -> two full groups; the leftover single frame is
# below half a group and must not become its own "identity".
assert len(group_means(_cluster(2, 7), 3)) == 2
def test_separable_people_yield_ordered_thresholds():
store = CalibrationStore("/nonexistent/never-written.npz")
for i, name in enumerate(["alice", "bob", "carol"]):
store.add("m", name, _cluster(10 + i, 12))
same, cross, meta = distributions(store, "m", 3)
assert len(meta["people"]) == 3
assert same.mean() > cross.mean()
rec = recommend(same, cross)
assert "error" not in rec
assert 0 < rec["enroll_threshold"] < rec["match_threshold"] < 1
# the invariant config.py enforces at load time
Config().recognition.model_copy(update={
"enroll_threshold": rec["enroll_threshold"],
"match_threshold": rec["match_threshold"]})
def test_overlapping_distributions_are_reported_not_smoothed_over():
"""Loose clusters that bleed into each other must fail loudly rather than
return a confident-looking midpoint."""
store = CalibrationStore("/nonexistent/never-written.npz")
rng = np.random.default_rng(0)
for name in ["alice", "bob"]:
v = rng.normal(size=(12, DIM)).astype(np.float32)
store.add("m", name, v / np.linalg.norm(v, axis=1, keepdims=True))
same, cross, _ = distributions(store, "m", 3)
rec = recommend(same, cross)
assert "error" in rec and "OVERLAP" in rec["error"]
def test_single_person_cannot_recommend_a_match_threshold():
store = CalibrationStore("/nonexistent/never-written.npz")
store.add("m", "alice", _cluster(5, 12))
same, cross, _ = distributions(store, "m", 3)
assert len(cross) == 0
rec = recommend(same, cross)
assert rec["match_threshold"] is None # honest about what it can't know
assert rec["enroll_threshold"] is not None # but this one it can
assert "note" in rec
def test_people_with_too_few_samples_are_skipped_not_averaged_in():
store = CalibrationStore("/nonexistent/never-written.npz")
store.add("m", "alice", _cluster(1, 12))
store.add("m", "flash", _cluster(2, 2)) # walked past, 2 frames
_, _, meta = distributions(store, "m", 3)
assert meta["people"] == ["alice"]
assert "flash" in meta["skipped"]
def test_store_roundtrips(tmp_path):
path = tmp_path / "cal.npz"
s = CalibrationStore(path)
s.add("model_a", "alice", _cluster(1, 5))
s.add("model_b", "alice", _cluster(2, 5))
s.save()
again = CalibrationStore(path)
assert again.models() == ["model_a", "model_b"]
assert again.get("model_a", "alice").shape == (5, DIM)
def test_store_appends_across_sessions(tmp_path):
path = tmp_path / "cal.npz"
s = CalibrationStore(path)
s.add("m", "alice", _cluster(1, 5))
assert s.add("m", "alice", _cluster(2, 4)) == 9 # second session accumulates
def test_clamped_enroll_recommendation_says_so():
"""A clamped value is the config ceiling talking, not the data - if it is
not flagged, every model reports the same number and it reads as a
measurement."""
store = CalibrationStore("/nonexistent/never-written.npz")
store.add("m", "alice", _cluster(5, 12, tightness=0.99)) # very tight
same, cross, _ = distributions(store, "m", 3)
rec = recommend(same, cross, current_match=0.42)
assert rec["enroll_threshold"] == 0.41
assert "clamped" in rec
def test_unclamped_recommendation_is_not_flagged():
store = CalibrationStore("/nonexistent/never-written.npz")
store.add("m", "alice", _cluster(5, 12, tightness=0.99))
same, cross, _ = distributions(store, "m", 3)
rec = recommend(same, cross, current_match=0.99)
assert "clamped" not in rec