Five components that ship as one product:
- behavision/ the recognition engine. RTSP ingest, YuNet detection, IoU
tracking, ArcFace embeddings, a FAISS/SQLite gallery, and a
FastAPI dashboard. Identity is decided once per TRACK from an
average of at least three embeddings, never per frame.
- agent/ the Go edge agent: supervises the engine, holds a durable
spool, and drains it to MQTT. Nothing is acked before the
broker confirms.
- desktop/ the shop PC application (Wails + React + tray).
- server/ the cloud API, MQTT consumer, reports and assistant.
- web/ platform.loyaly.ai, the head-office app, embedded in the
server binary.
The gallery stores 512-float embeddings and timestamps - no images unless
`app.store_faces` is switched on. Those embeddings are biometric personal
data under GDPR and India's DPDP: template inversion reconstructs a
recognisable face from an ArcFace vector, so data/behavision.db is treated
as a biometric database and DELETE /api/visitors/{id} is a real erasure.
CLAUDE.md carries the reasoning behind every non-obvious decision here,
including the ones that were measured and the ones that were wrong first.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HViLj9gYNRtSr7YVZmW5sn
283 lines
10 KiB
Python
283 lines
10 KiB
Python
"""Identity merge: the repair path for one person enrolled twice.
|
|
|
|
Merging is the only destructive operation in the gallery that cannot be
|
|
undone — nothing records which embedding came from which identity — so most
|
|
of what is tested here is the refusal, not the merge.
|
|
"""
|
|
import numpy as np
|
|
import pytest
|
|
|
|
from behavision.config import RecognitionSection
|
|
from behavision.gallery import Gallery, IdentityStore, VectorIndex
|
|
|
|
DIM = 16
|
|
|
|
|
|
def _unit(seed):
|
|
rng = np.random.default_rng(seed)
|
|
v = rng.normal(size=DIM).astype(np.float32)
|
|
return v / np.linalg.norm(v)
|
|
|
|
|
|
def _at_similarity(base, target, seed=99):
|
|
"""A unit vector at exactly `target` cosine similarity to `base`.
|
|
|
|
Constructed, never assumed: 16-d random vectors are nowhere near
|
|
orthogonal (seed 1 and seed 2 sit 0.58 apart), so a test that picks two
|
|
seeds and calls them "different people" is testing the seeds.
|
|
"""
|
|
other = _unit(seed)
|
|
other -= (other @ base) * base
|
|
other /= np.linalg.norm(other)
|
|
v = target * base + np.sqrt(1 - target ** 2) * other
|
|
return (v / np.linalg.norm(v)).astype(np.float32)
|
|
|
|
|
|
@pytest.fixture
|
|
def gallery(tmp_path):
|
|
store = IdentityStore(tmp_path / "merge.db")
|
|
cfg = RecognitionSection(sighting_cooldown_seconds=0.0)
|
|
gal = Gallery(store, VectorIndex(DIM), cfg)
|
|
yield gal
|
|
store.close()
|
|
|
|
|
|
def _add_view(gal, identity_id, vec):
|
|
"""Store another embedding for an identity, as reinforcement does."""
|
|
emb_id = gal.store.add_embedding(identity_id, vec, 0.9, gal.model_name)
|
|
gal.index.add([emb_id], vec.reshape(1, -1))
|
|
return emb_id
|
|
|
|
|
|
def _two_identities(gal):
|
|
"""One person split in two, built the way it actually happens.
|
|
|
|
Not "two vectors 0.40 apart" — 0.40 is the ambiguous zone, where the
|
|
pipeline deliberately refuses to decide and creates nothing. A real split
|
|
needs the second view to be under enroll_threshold *at the moment it is
|
|
seen*; later views then fill both galleries out until the two identities
|
|
overlap. That is exactly the Office1 pair: max similarity 0.412 between
|
|
them, yet neither was ever close enough for the pipeline to join them.
|
|
"""
|
|
a = _unit(1)
|
|
b = _at_similarity(a, 0.20, seed=11)
|
|
first = gal.resolve(a, quality=0.9, camera_id="cam1")
|
|
second = gal.resolve(b, quality=0.9, camera_id="cam1")
|
|
assert first.kind == "new" and second.kind == "new", "fixture must split"
|
|
_add_view(gal, first.identity_id, _at_similarity(b, 0.40, seed=12))
|
|
return first.identity_id, second.identity_id
|
|
|
|
|
|
# -- the happy path ---------------------------------------------------------
|
|
|
|
def test_merge_moves_embeddings_and_sightings(gallery):
|
|
src, dst = _two_identities(gallery)
|
|
assert gallery.store.stats()["identities"] == 2
|
|
|
|
res = gallery.merge_identities(src, dst)
|
|
|
|
assert res["ok"] is True
|
|
assert res["embeddings_moved"] == 2
|
|
assert res["sightings_moved"] == 1
|
|
assert gallery.store.get_identity(src) is None
|
|
assert gallery.store.stats()["identities"] == 1
|
|
assert gallery.store.embedding_count(dst) == 3
|
|
|
|
|
|
def test_merged_person_is_recognised_from_either_view(gallery):
|
|
"""The point of the whole feature: after merging, the view that used to
|
|
mint a second identity resolves to the surviving one."""
|
|
src, dst = _two_identities(gallery)
|
|
a = _unit(1)
|
|
gallery.merge_identities(src, dst)
|
|
|
|
res = gallery.resolve(a, quality=0.9, camera_id="cam1")
|
|
assert res.kind == "known"
|
|
assert res.identity_id == dst
|
|
|
|
|
|
def test_index_needs_no_rebuild(gallery):
|
|
"""Embedding ids do not change on merge, so every vector stays valid in
|
|
the index. Regression guard: if merge ever starts copying rows instead of
|
|
re-pointing them, this count drifts."""
|
|
src, dst = _two_identities(gallery)
|
|
before = len(gallery.index)
|
|
gallery.merge_identities(src, dst)
|
|
assert len(gallery.index) == before
|
|
|
|
|
|
def test_sighting_count_is_recomputed_not_summed(gallery):
|
|
src, dst = _two_identities(gallery)
|
|
for _ in range(3):
|
|
gallery.resolve(_unit(1), quality=0.9, camera_id="cam1")
|
|
# Corrupt the stored counter the way a stale count would look.
|
|
gallery.store._db.execute(
|
|
"UPDATE identities SET sighting_count=999 WHERE id=?", (src,))
|
|
gallery.store._db.commit()
|
|
|
|
res = gallery.merge_identities(src, dst)
|
|
|
|
rows = gallery.store._db.execute(
|
|
"SELECT COUNT(*) AS n FROM sightings WHERE identity_id=?",
|
|
(dst,)).fetchone()["n"]
|
|
assert res["sighting_count"] == rows
|
|
assert gallery.store.get_identity(dst)["sighting_count"] == rows
|
|
|
|
|
|
# -- label and history policy ----------------------------------------------
|
|
|
|
def test_human_name_survives_merge_in_either_direction(gallery):
|
|
"""Merging Alice into "Visitor 4" must not leave the person called
|
|
Visitor 4 — that is silent data loss, and the operator cannot tell it
|
|
happened."""
|
|
second_view = _at_similarity(_unit(1), 0.20, seed=11)
|
|
named = gallery.enroll("Alice", [_unit(1)])
|
|
auto = gallery.resolve(second_view, quality=0.9,
|
|
camera_id="cam1").identity_id
|
|
# The bridging view has to resemble the *other* identity, not the first
|
|
# one - that is what brings the pair above enroll_threshold.
|
|
_add_view(gallery, named, _at_similarity(second_view, 0.40, seed=13))
|
|
|
|
res = gallery.merge_identities(named, auto) # named -> auto
|
|
|
|
assert res["label"] == "Alice"
|
|
assert gallery.store.get_identity(auto)["kind"] == "enrolled"
|
|
|
|
|
|
def test_target_label_kept_when_both_are_named(gallery):
|
|
a = gallery.enroll("Alice", [_unit(1)])
|
|
b = gallery.enroll("Alice Smith", [_at_similarity(_unit(1), 0.40)])
|
|
# 0.40 clears enroll_threshold, so the guard lets this through.
|
|
res = gallery.merge_identities(a, b)
|
|
assert res["label"] == "Alice Smith"
|
|
|
|
|
|
def test_created_at_takes_the_earlier(gallery):
|
|
src, dst = _two_identities(gallery)
|
|
before = min(gallery.store.get_identity(src)["created_at"],
|
|
gallery.store.get_identity(dst)["created_at"])
|
|
gallery.merge_identities(src, dst)
|
|
assert gallery.store.get_identity(dst)["created_at"] == pytest.approx(before)
|
|
|
|
|
|
def test_merge_trims_to_the_cap_and_drops_from_index(gallery):
|
|
"""Two identities at the cap would leave one holding double, quietly
|
|
overweighting that person in every search."""
|
|
cap = gallery.cfg.max_embeddings_per_identity
|
|
base = _unit(1)
|
|
a = gallery.enroll("A", [_at_similarity(base, 0.99, s + 20)
|
|
for s in range(cap)])
|
|
b = gallery.enroll("B", [_at_similarity(base, 0.98, s + 50)
|
|
for s in range(cap)])
|
|
assert len(gallery.index) == 2 * cap
|
|
|
|
res = gallery.merge_identities(a, b)
|
|
|
|
assert gallery.store.embedding_count(b) == cap
|
|
assert len(res["dropped_embeddings"]) == cap
|
|
assert len(gallery.index) == cap
|
|
|
|
|
|
# -- refusals ---------------------------------------------------------------
|
|
|
|
def test_refuses_self_merge(gallery):
|
|
src, _ = _two_identities(gallery)
|
|
res = gallery.merge_identities(src, src)
|
|
assert res["ok"] is False
|
|
assert "itself" in res["reason"]
|
|
|
|
|
|
@pytest.mark.parametrize("bad_side", ["source", "target"])
|
|
def test_refuses_missing_identity(gallery, bad_side):
|
|
src, dst = _two_identities(gallery)
|
|
args = (9999, dst) if bad_side == "source" else (src, 9999)
|
|
res = gallery.merge_identities(*args)
|
|
assert res["ok"] is False
|
|
assert "not found" in res["reason"]
|
|
|
|
|
|
def test_refuses_two_different_people(gallery):
|
|
"""Below enroll_threshold resolve() positively asserts these are
|
|
different people. Merging anyway would contradict the number driving
|
|
every other decision, so it needs an explicit override."""
|
|
a = gallery.enroll("A", [_unit(1)])
|
|
b = gallery.enroll("B", [_at_similarity(_unit(1), 0.10)])
|
|
|
|
res = gallery.merge_identities(a, b)
|
|
|
|
assert res["ok"] is False
|
|
assert res["similarity"] == pytest.approx(0.10, abs=0.01)
|
|
assert gallery.store.stats()["identities"] == 2
|
|
|
|
|
|
def test_force_overrides_the_similarity_guard(gallery):
|
|
a = gallery.enroll("A", [_unit(1)])
|
|
b = gallery.enroll("B", [_at_similarity(_unit(1), 0.10)])
|
|
|
|
res = gallery.merge_identities(a, b, force=True)
|
|
|
|
assert res["ok"] is True
|
|
assert res["forced"] is True
|
|
assert gallery.store.stats()["identities"] == 1
|
|
|
|
|
|
def test_refuses_when_models_differ(gallery):
|
|
"""Vectors from two encoders are not comparable, so the guard cannot run.
|
|
Refusing beats comparing numbers from different spaces and believing the
|
|
answer."""
|
|
a = gallery.store.create_identity("A")
|
|
gallery.store.add_embedding(a, _unit(1), 0.9, "other_model")
|
|
b = gallery.enroll("B", [_unit(1)])
|
|
|
|
res = gallery.merge_identities(a, b)
|
|
|
|
assert res["ok"] is False
|
|
assert res["similarity"] is None
|
|
assert "encoder model" in res["reason"]
|
|
|
|
|
|
def test_similarity_uses_the_best_pair_not_the_mean(gallery):
|
|
"""Two identities of one person exist precisely because their typical
|
|
views disagree; one agreeing pair is the evidence that matters."""
|
|
base = _unit(1)
|
|
a = gallery.enroll("A", [base])
|
|
b = gallery.enroll("B", [_at_similarity(base, 0.05, 7),
|
|
_at_similarity(base, 0.50, 8)])
|
|
sim, checkable = gallery._identity_similarity(a, b)
|
|
assert checkable
|
|
assert sim == pytest.approx(0.50, abs=0.01)
|
|
|
|
|
|
def test_sighting_cooldown_cache_forgets_the_source(gallery):
|
|
"""The cooldown is keyed by identity id; leaving the source's key behind
|
|
leaks an entry pointing at an identity that no longer exists."""
|
|
gal = gallery
|
|
gal.cfg = RecognitionSection(sighting_cooldown_seconds=30.0)
|
|
src, dst = _two_identities(gal)
|
|
assert any(k[0] == src for k in gal._last_sighting)
|
|
gal.merge_identities(src, dst)
|
|
assert not any(k[0] == src for k in gal._last_sighting)
|
|
|
|
|
|
# -- finding the duplicates in the first place ------------------------------
|
|
|
|
def test_duplicate_candidates_finds_the_split(gallery):
|
|
src, dst = _two_identities(gallery)
|
|
pairs = gallery.duplicate_candidates()
|
|
assert len(pairs) == 1
|
|
found = {pairs[0]["a"]["id"], pairs[0]["b"]["id"]}
|
|
assert found == {src, dst}
|
|
assert pairs[0]["similarity"] == pytest.approx(0.40, abs=0.01)
|
|
# 0.40 is under match_threshold, so it is a suggestion, not a verdict.
|
|
assert pairs[0]["confident"] is False
|
|
|
|
|
|
def test_duplicate_candidates_ignores_different_people(gallery):
|
|
gallery.enroll("A", [_unit(1)])
|
|
gallery.enroll("B", [_at_similarity(_unit(1), 0.10)])
|
|
assert gallery.duplicate_candidates() == []
|
|
|
|
|
|
def test_duplicate_candidates_empty_gallery(gallery):
|
|
assert gallery.duplicate_candidates() == []
|