Files
Behavision/tests/test_pipeline_stats.py
Suriyakumarvijayanayagam dad04e8cda Behavision: face recognition for retail, edge to head office
Five components that ship as one product:

- behavision/  the recognition engine. RTSP ingest, YuNet detection, IoU
               tracking, ArcFace embeddings, a FAISS/SQLite gallery, and a
               FastAPI dashboard. Identity is decided once per TRACK from an
               average of at least three embeddings, never per frame.
- agent/       the Go edge agent: supervises the engine, holds a durable
               spool, and drains it to MQTT. Nothing is acked before the
               broker confirms.
- desktop/     the shop PC application (Wails + React + tray).
- server/      the cloud API, MQTT consumer, reports and assistant.
- web/         platform.loyaly.ai, the head-office app, embedded in the
               server binary.

The gallery stores 512-float embeddings and timestamps - no images unless
`app.store_faces` is switched on. Those embeddings are biometric personal
data under GDPR and India's DPDP: template inversion reconstructs a
recognisable face from an ArcFace vector, so data/behavision.db is treated
as a biometric database and DELETE /api/visitors/{id} is a real erasure.

CLAUDE.md carries the reasoning behind every non-obvious decision here,
including the ones that were measured and the ones that were wrong first.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HViLj9gYNRtSr7YVZmW5sn
2026-09-04 11:14:18 +05:30

190 lines
7.0 KiB
Python

"""What happens to a track that never becomes an identity.
These cover the failure this pipeline was blind to: a visitor detected,
tracked and embedded, then dropped because their face was under the
enrollment gate — with no event, no counter and no log line, so a
mis-tuned gate was indistinguishable from an empty room.
"""
import threading
import numpy as np
import pytest
from behavision.config import Config
from behavision.engine import (CameraWorker, PipelineStats, _track_outcome,
_spread)
from behavision.events import Event
from behavision.gallery.service import Resolution
from behavision.faces import FaceOutbox
from behavision.tracking import Track
class RecordingBus:
def __init__(self):
self.events = []
def publish(self, event: Event):
self.events.append(event)
class StubGallery:
"""Returns one canned verdict, whatever it is handed."""
def __init__(self, resolution):
self.resolution = resolution
self.calls = 0
def resolve(self, *a, **kw):
self.calls += 1
return self.resolution
class StubEncoder:
size = 112
def encode_chip(self, chip):
v = np.ones(512, dtype=np.float32)
return v / np.linalg.norm(v)
def make_worker(resolution, monkeypatch, cfg=None):
"""A CameraWorker with no camera, no detector and no models."""
monkeypatch.setattr("behavision.engine.align_face",
lambda frame, kps, size: np.zeros((size, size, 3),
dtype=np.uint8))
w = object.__new__(CameraWorker)
w.cfg = cfg or Config()
w.rcfg = w.cfg.recognition.merged(None)
w.commission = None # no placement check running
w.cam_cfg = w.cfg.cameras[0] if w.cfg.cameras else None
w.encoder = StubEncoder()
w.gallery = StubGallery(resolution)
w.bus = RecordingBus()
w.attrs = None
w.pipeline = PipelineStats()
w._lock = threading.Lock()
# Images are off, the product default. A real FaceOutbox rather than a
# mock, so a worker built this way runs the same disabled path
# production does when store_faces is unset.
w.faces = FaceOutbox(w.cfg.app.data_dir, enabled=False)
class _Cam:
id = "cam1"
w.cam_cfg = _Cam()
return w
def ready_track(**kw):
"""A track that has already met every precondition for a decision."""
t = Track(id=1, box=(0, 0, 50, 50), kps=np.zeros((5, 2), dtype=np.float32),
score=0.9)
t.quality = t.best_quality = kw.pop("quality", 0.45)
t.hits = 10
t.emb_sum = np.ones(512, dtype=np.float32) * 3
t.emb_count = 3
for k, v in kw.items():
setattr(t, k, v)
return t
# -- the bug ------------------------------------------------------------
def test_skipped_verdict_is_recorded_not_swallowed(monkeypatch):
"""resolve() refusing on quality must leave evidence behind."""
w = make_worker(Resolution(kind="skipped", similarity=0.1), monkeypatch)
track = ready_track()
w._identify(track, np.zeros((100, 100, 3), np.uint8), 100.0)
assert w.gallery.calls == 1
assert track.quality_skips == 1, "the refusal left no trace on the track"
assert track.state == "ambiguous", "a skipped track must not stay pending"
def test_skipped_track_ends_as_a_quality_rejection(monkeypatch):
w = make_worker(Resolution(kind="skipped", similarity=0.1), monkeypatch)
track = ready_track()
w._identify(track, np.zeros((100, 100, 3), np.uint8), 100.0)
w._finish_track(track, 101.0)
assert w.pipeline.snapshot()["outcomes"] == {"rejected_quality": 1}
assert [e.type for e in w.bus.events] == ["person.missed"]
assert w.bus.events[0].data["reason"] == "rejected_quality"
def test_skipped_retries_are_throttled_not_burnt_in_one_burst(monkeypatch):
"""Marking it ambiguous buys the retry interval; without that the eight
attempts are spent on eight consecutive frames of the same instant."""
w = make_worker(Resolution(kind="skipped", similarity=0.1), monkeypatch)
track = ready_track()
for ts in (100.0, 100.03, 100.06): # three frames, ~30 ms apart
w._identify(track, np.zeros((100, 100, 3), np.uint8), ts)
assert w.gallery.calls == 1
assert track.id_attempts == 1
# -- outcome classification --------------------------------------------
def test_recognized_and_enrolled_are_distinguished():
assert _track_outcome(ready_track(state="resolved", is_new=True)) == "enrolled"
assert _track_outcome(ready_track(state="resolved")) == "recognized"
def test_quality_rejection_outranks_gave_up():
"""A track that exhausted its attempts on quality refusals is a quality
failure; calling it ambiguous sends whoever tunes the site to the match
threshold instead of to the camera mount."""
t = ready_track(state="gave_up", id_attempts=8, quality_skips=8)
assert _track_outcome(t) == "rejected_quality"
def test_track_that_never_encoded_is_not_a_recognition_failure():
t = ready_track(emb_sum=None, emb_count=0)
assert _track_outcome(t) == "no_embedding"
def test_track_that_left_before_deciding_is_too_brief():
assert _track_outcome(ready_track(id_attempts=0)) == "too_brief"
def test_brief_losses_do_not_raise_events(monkeypatch):
"""A face glimpsed for two frames is noise, not a lost visitor."""
w = make_worker(Resolution(kind="skipped"), monkeypatch)
t = ready_track(state="ambiguous", id_attempts=1, quality_skips=1,
emb_count=1)
w._finish_track(t, 100.0)
assert w.pipeline.snapshot()["outcomes"] == {"rejected_quality": 1}
assert w.bus.events == []
# -- distributions ------------------------------------------------------
def test_fraction_below_gate_names_the_real_problem():
"""The number that says the enrollment gate is wrong for this camera."""
stats = PipelineStats()
for q in (0.32, 0.38, 0.41, 0.45, 0.72): # measured overhead spread
stats.record(ready_track(quality=q, id_attempts=1), "rejected_quality")
snap = stats.snapshot(enroll_gate=0.65)
assert snap["best_quality"]["n"] == 5
assert snap["best_quality"]["fraction_below_gate"] == 0.8
assert snap["tracks_ended"] == 5
def test_similarity_only_counts_tracks_that_reached_a_decision():
"""Tracks that never called resolve() have similarity 0.0, and averaging
those in would drag every percentile toward zero."""
stats = PipelineStats()
stats.record(ready_track(id_attempts=1, similarity=0.5), "recognized")
stats.record(ready_track(id_attempts=0, similarity=0.0), "too_brief")
assert stats.snapshot()["similarity"]["n"] == 1
def test_spread_of_nothing_is_empty_not_zero():
assert _spread([]) == {"n": 0}
def test_distribution_window_is_bounded():
"""A camera running for weeks must not grow this without limit."""
stats = PipelineStats()
for _ in range(PipelineStats.WINDOW + 50):
stats.record(ready_track(id_attempts=1), "recognized")
snap = stats.snapshot()
assert snap["best_quality"]["n"] == PipelineStats.WINDOW
assert snap["tracks_ended"] == PipelineStats.WINDOW + 50