Files
Behavision/tests/test_merge.py
Suriyakumarvijayanayagam dad04e8cda Behavision: face recognition for retail, edge to head office
Five components that ship as one product:

- behavision/  the recognition engine. RTSP ingest, YuNet detection, IoU
               tracking, ArcFace embeddings, a FAISS/SQLite gallery, and a
               FastAPI dashboard. Identity is decided once per TRACK from an
               average of at least three embeddings, never per frame.
- agent/       the Go edge agent: supervises the engine, holds a durable
               spool, and drains it to MQTT. Nothing is acked before the
               broker confirms.
- desktop/     the shop PC application (Wails + React + tray).
- server/      the cloud API, MQTT consumer, reports and assistant.
- web/         platform.loyaly.ai, the head-office app, embedded in the
               server binary.

The gallery stores 512-float embeddings and timestamps - no images unless
`app.store_faces` is switched on. Those embeddings are biometric personal
data under GDPR and India's DPDP: template inversion reconstructs a
recognisable face from an ArcFace vector, so data/behavision.db is treated
as a biometric database and DELETE /api/visitors/{id} is a real erasure.

CLAUDE.md carries the reasoning behind every non-obvious decision here,
including the ones that were measured and the ones that were wrong first.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HViLj9gYNRtSr7YVZmW5sn
2026-09-04 11:14:18 +05:30

283 lines
10 KiB
Python

"""Identity merge: the repair path for one person enrolled twice.
Merging is the only destructive operation in the gallery that cannot be
undone — nothing records which embedding came from which identity — so most
of what is tested here is the refusal, not the merge.
"""
import numpy as np
import pytest
from behavision.config import RecognitionSection
from behavision.gallery import Gallery, IdentityStore, VectorIndex
DIM = 16
def _unit(seed):
rng = np.random.default_rng(seed)
v = rng.normal(size=DIM).astype(np.float32)
return v / np.linalg.norm(v)
def _at_similarity(base, target, seed=99):
"""A unit vector at exactly `target` cosine similarity to `base`.
Constructed, never assumed: 16-d random vectors are nowhere near
orthogonal (seed 1 and seed 2 sit 0.58 apart), so a test that picks two
seeds and calls them "different people" is testing the seeds.
"""
other = _unit(seed)
other -= (other @ base) * base
other /= np.linalg.norm(other)
v = target * base + np.sqrt(1 - target ** 2) * other
return (v / np.linalg.norm(v)).astype(np.float32)
@pytest.fixture
def gallery(tmp_path):
store = IdentityStore(tmp_path / "merge.db")
cfg = RecognitionSection(sighting_cooldown_seconds=0.0)
gal = Gallery(store, VectorIndex(DIM), cfg)
yield gal
store.close()
def _add_view(gal, identity_id, vec):
"""Store another embedding for an identity, as reinforcement does."""
emb_id = gal.store.add_embedding(identity_id, vec, 0.9, gal.model_name)
gal.index.add([emb_id], vec.reshape(1, -1))
return emb_id
def _two_identities(gal):
"""One person split in two, built the way it actually happens.
Not "two vectors 0.40 apart" — 0.40 is the ambiguous zone, where the
pipeline deliberately refuses to decide and creates nothing. A real split
needs the second view to be under enroll_threshold *at the moment it is
seen*; later views then fill both galleries out until the two identities
overlap. That is exactly the Office1 pair: max similarity 0.412 between
them, yet neither was ever close enough for the pipeline to join them.
"""
a = _unit(1)
b = _at_similarity(a, 0.20, seed=11)
first = gal.resolve(a, quality=0.9, camera_id="cam1")
second = gal.resolve(b, quality=0.9, camera_id="cam1")
assert first.kind == "new" and second.kind == "new", "fixture must split"
_add_view(gal, first.identity_id, _at_similarity(b, 0.40, seed=12))
return first.identity_id, second.identity_id
# -- the happy path ---------------------------------------------------------
def test_merge_moves_embeddings_and_sightings(gallery):
src, dst = _two_identities(gallery)
assert gallery.store.stats()["identities"] == 2
res = gallery.merge_identities(src, dst)
assert res["ok"] is True
assert res["embeddings_moved"] == 2
assert res["sightings_moved"] == 1
assert gallery.store.get_identity(src) is None
assert gallery.store.stats()["identities"] == 1
assert gallery.store.embedding_count(dst) == 3
def test_merged_person_is_recognised_from_either_view(gallery):
"""The point of the whole feature: after merging, the view that used to
mint a second identity resolves to the surviving one."""
src, dst = _two_identities(gallery)
a = _unit(1)
gallery.merge_identities(src, dst)
res = gallery.resolve(a, quality=0.9, camera_id="cam1")
assert res.kind == "known"
assert res.identity_id == dst
def test_index_needs_no_rebuild(gallery):
"""Embedding ids do not change on merge, so every vector stays valid in
the index. Regression guard: if merge ever starts copying rows instead of
re-pointing them, this count drifts."""
src, dst = _two_identities(gallery)
before = len(gallery.index)
gallery.merge_identities(src, dst)
assert len(gallery.index) == before
def test_sighting_count_is_recomputed_not_summed(gallery):
src, dst = _two_identities(gallery)
for _ in range(3):
gallery.resolve(_unit(1), quality=0.9, camera_id="cam1")
# Corrupt the stored counter the way a stale count would look.
gallery.store._db.execute(
"UPDATE identities SET sighting_count=999 WHERE id=?", (src,))
gallery.store._db.commit()
res = gallery.merge_identities(src, dst)
rows = gallery.store._db.execute(
"SELECT COUNT(*) AS n FROM sightings WHERE identity_id=?",
(dst,)).fetchone()["n"]
assert res["sighting_count"] == rows
assert gallery.store.get_identity(dst)["sighting_count"] == rows
# -- label and history policy ----------------------------------------------
def test_human_name_survives_merge_in_either_direction(gallery):
"""Merging Alice into "Visitor 4" must not leave the person called
Visitor 4 — that is silent data loss, and the operator cannot tell it
happened."""
second_view = _at_similarity(_unit(1), 0.20, seed=11)
named = gallery.enroll("Alice", [_unit(1)])
auto = gallery.resolve(second_view, quality=0.9,
camera_id="cam1").identity_id
# The bridging view has to resemble the *other* identity, not the first
# one - that is what brings the pair above enroll_threshold.
_add_view(gallery, named, _at_similarity(second_view, 0.40, seed=13))
res = gallery.merge_identities(named, auto) # named -> auto
assert res["label"] == "Alice"
assert gallery.store.get_identity(auto)["kind"] == "enrolled"
def test_target_label_kept_when_both_are_named(gallery):
a = gallery.enroll("Alice", [_unit(1)])
b = gallery.enroll("Alice Smith", [_at_similarity(_unit(1), 0.40)])
# 0.40 clears enroll_threshold, so the guard lets this through.
res = gallery.merge_identities(a, b)
assert res["label"] == "Alice Smith"
def test_created_at_takes_the_earlier(gallery):
src, dst = _two_identities(gallery)
before = min(gallery.store.get_identity(src)["created_at"],
gallery.store.get_identity(dst)["created_at"])
gallery.merge_identities(src, dst)
assert gallery.store.get_identity(dst)["created_at"] == pytest.approx(before)
def test_merge_trims_to_the_cap_and_drops_from_index(gallery):
"""Two identities at the cap would leave one holding double, quietly
overweighting that person in every search."""
cap = gallery.cfg.max_embeddings_per_identity
base = _unit(1)
a = gallery.enroll("A", [_at_similarity(base, 0.99, s + 20)
for s in range(cap)])
b = gallery.enroll("B", [_at_similarity(base, 0.98, s + 50)
for s in range(cap)])
assert len(gallery.index) == 2 * cap
res = gallery.merge_identities(a, b)
assert gallery.store.embedding_count(b) == cap
assert len(res["dropped_embeddings"]) == cap
assert len(gallery.index) == cap
# -- refusals ---------------------------------------------------------------
def test_refuses_self_merge(gallery):
src, _ = _two_identities(gallery)
res = gallery.merge_identities(src, src)
assert res["ok"] is False
assert "itself" in res["reason"]
@pytest.mark.parametrize("bad_side", ["source", "target"])
def test_refuses_missing_identity(gallery, bad_side):
src, dst = _two_identities(gallery)
args = (9999, dst) if bad_side == "source" else (src, 9999)
res = gallery.merge_identities(*args)
assert res["ok"] is False
assert "not found" in res["reason"]
def test_refuses_two_different_people(gallery):
"""Below enroll_threshold resolve() positively asserts these are
different people. Merging anyway would contradict the number driving
every other decision, so it needs an explicit override."""
a = gallery.enroll("A", [_unit(1)])
b = gallery.enroll("B", [_at_similarity(_unit(1), 0.10)])
res = gallery.merge_identities(a, b)
assert res["ok"] is False
assert res["similarity"] == pytest.approx(0.10, abs=0.01)
assert gallery.store.stats()["identities"] == 2
def test_force_overrides_the_similarity_guard(gallery):
a = gallery.enroll("A", [_unit(1)])
b = gallery.enroll("B", [_at_similarity(_unit(1), 0.10)])
res = gallery.merge_identities(a, b, force=True)
assert res["ok"] is True
assert res["forced"] is True
assert gallery.store.stats()["identities"] == 1
def test_refuses_when_models_differ(gallery):
"""Vectors from two encoders are not comparable, so the guard cannot run.
Refusing beats comparing numbers from different spaces and believing the
answer."""
a = gallery.store.create_identity("A")
gallery.store.add_embedding(a, _unit(1), 0.9, "other_model")
b = gallery.enroll("B", [_unit(1)])
res = gallery.merge_identities(a, b)
assert res["ok"] is False
assert res["similarity"] is None
assert "encoder model" in res["reason"]
def test_similarity_uses_the_best_pair_not_the_mean(gallery):
"""Two identities of one person exist precisely because their typical
views disagree; one agreeing pair is the evidence that matters."""
base = _unit(1)
a = gallery.enroll("A", [base])
b = gallery.enroll("B", [_at_similarity(base, 0.05, 7),
_at_similarity(base, 0.50, 8)])
sim, checkable = gallery._identity_similarity(a, b)
assert checkable
assert sim == pytest.approx(0.50, abs=0.01)
def test_sighting_cooldown_cache_forgets_the_source(gallery):
"""The cooldown is keyed by identity id; leaving the source's key behind
leaks an entry pointing at an identity that no longer exists."""
gal = gallery
gal.cfg = RecognitionSection(sighting_cooldown_seconds=30.0)
src, dst = _two_identities(gal)
assert any(k[0] == src for k in gal._last_sighting)
gal.merge_identities(src, dst)
assert not any(k[0] == src for k in gal._last_sighting)
# -- finding the duplicates in the first place ------------------------------
def test_duplicate_candidates_finds_the_split(gallery):
src, dst = _two_identities(gallery)
pairs = gallery.duplicate_candidates()
assert len(pairs) == 1
found = {pairs[0]["a"]["id"], pairs[0]["b"]["id"]}
assert found == {src, dst}
assert pairs[0]["similarity"] == pytest.approx(0.40, abs=0.01)
# 0.40 is under match_threshold, so it is a suggestion, not a verdict.
assert pairs[0]["confident"] is False
def test_duplicate_candidates_ignores_different_people(gallery):
gallery.enroll("A", [_unit(1)])
gallery.enroll("B", [_at_similarity(_unit(1), 0.10)])
assert gallery.duplicate_candidates() == []
def test_duplicate_candidates_empty_gallery(gallery):
assert gallery.duplicate_candidates() == []