"""Identity merge: the repair path for one person enrolled twice. Merging is the only destructive operation in the gallery that cannot be undone — nothing records which embedding came from which identity — so most of what is tested here is the refusal, not the merge. """ import numpy as np import pytest from behavision.config import RecognitionSection from behavision.gallery import Gallery, IdentityStore, VectorIndex DIM = 16 def _unit(seed): rng = np.random.default_rng(seed) v = rng.normal(size=DIM).astype(np.float32) return v / np.linalg.norm(v) def _at_similarity(base, target, seed=99): """A unit vector at exactly `target` cosine similarity to `base`. Constructed, never assumed: 16-d random vectors are nowhere near orthogonal (seed 1 and seed 2 sit 0.58 apart), so a test that picks two seeds and calls them "different people" is testing the seeds. """ other = _unit(seed) other -= (other @ base) * base other /= np.linalg.norm(other) v = target * base + np.sqrt(1 - target ** 2) * other return (v / np.linalg.norm(v)).astype(np.float32) @pytest.fixture def gallery(tmp_path): store = IdentityStore(tmp_path / "merge.db") cfg = RecognitionSection(sighting_cooldown_seconds=0.0) gal = Gallery(store, VectorIndex(DIM), cfg) yield gal store.close() def _add_view(gal, identity_id, vec): """Store another embedding for an identity, as reinforcement does.""" emb_id = gal.store.add_embedding(identity_id, vec, 0.9, gal.model_name) gal.index.add([emb_id], vec.reshape(1, -1)) return emb_id def _two_identities(gal): """One person split in two, built the way it actually happens. Not "two vectors 0.40 apart" — 0.40 is the ambiguous zone, where the pipeline deliberately refuses to decide and creates nothing. A real split needs the second view to be under enroll_threshold *at the moment it is seen*; later views then fill both galleries out until the two identities overlap. That is exactly the Office1 pair: max similarity 0.412 between them, yet neither was ever close enough for the pipeline to join them. """ a = _unit(1) b = _at_similarity(a, 0.20, seed=11) first = gal.resolve(a, quality=0.9, camera_id="cam1") second = gal.resolve(b, quality=0.9, camera_id="cam1") assert first.kind == "new" and second.kind == "new", "fixture must split" _add_view(gal, first.identity_id, _at_similarity(b, 0.40, seed=12)) return first.identity_id, second.identity_id # -- the happy path --------------------------------------------------------- def test_merge_moves_embeddings_and_sightings(gallery): src, dst = _two_identities(gallery) assert gallery.store.stats()["identities"] == 2 res = gallery.merge_identities(src, dst) assert res["ok"] is True assert res["embeddings_moved"] == 2 assert res["sightings_moved"] == 1 assert gallery.store.get_identity(src) is None assert gallery.store.stats()["identities"] == 1 assert gallery.store.embedding_count(dst) == 3 def test_merged_person_is_recognised_from_either_view(gallery): """The point of the whole feature: after merging, the view that used to mint a second identity resolves to the surviving one.""" src, dst = _two_identities(gallery) a = _unit(1) gallery.merge_identities(src, dst) res = gallery.resolve(a, quality=0.9, camera_id="cam1") assert res.kind == "known" assert res.identity_id == dst def test_index_needs_no_rebuild(gallery): """Embedding ids do not change on merge, so every vector stays valid in the index. Regression guard: if merge ever starts copying rows instead of re-pointing them, this count drifts.""" src, dst = _two_identities(gallery) before = len(gallery.index) gallery.merge_identities(src, dst) assert len(gallery.index) == before def test_sighting_count_is_recomputed_not_summed(gallery): src, dst = _two_identities(gallery) for _ in range(3): gallery.resolve(_unit(1), quality=0.9, camera_id="cam1") # Corrupt the stored counter the way a stale count would look. gallery.store._db.execute( "UPDATE identities SET sighting_count=999 WHERE id=?", (src,)) gallery.store._db.commit() res = gallery.merge_identities(src, dst) rows = gallery.store._db.execute( "SELECT COUNT(*) AS n FROM sightings WHERE identity_id=?", (dst,)).fetchone()["n"] assert res["sighting_count"] == rows assert gallery.store.get_identity(dst)["sighting_count"] == rows # -- label and history policy ---------------------------------------------- def test_human_name_survives_merge_in_either_direction(gallery): """Merging Alice into "Visitor 4" must not leave the person called Visitor 4 — that is silent data loss, and the operator cannot tell it happened.""" second_view = _at_similarity(_unit(1), 0.20, seed=11) named = gallery.enroll("Alice", [_unit(1)]) auto = gallery.resolve(second_view, quality=0.9, camera_id="cam1").identity_id # The bridging view has to resemble the *other* identity, not the first # one - that is what brings the pair above enroll_threshold. _add_view(gallery, named, _at_similarity(second_view, 0.40, seed=13)) res = gallery.merge_identities(named, auto) # named -> auto assert res["label"] == "Alice" assert gallery.store.get_identity(auto)["kind"] == "enrolled" def test_target_label_kept_when_both_are_named(gallery): a = gallery.enroll("Alice", [_unit(1)]) b = gallery.enroll("Alice Smith", [_at_similarity(_unit(1), 0.40)]) # 0.40 clears enroll_threshold, so the guard lets this through. res = gallery.merge_identities(a, b) assert res["label"] == "Alice Smith" def test_created_at_takes_the_earlier(gallery): src, dst = _two_identities(gallery) before = min(gallery.store.get_identity(src)["created_at"], gallery.store.get_identity(dst)["created_at"]) gallery.merge_identities(src, dst) assert gallery.store.get_identity(dst)["created_at"] == pytest.approx(before) def test_merge_trims_to_the_cap_and_drops_from_index(gallery): """Two identities at the cap would leave one holding double, quietly overweighting that person in every search.""" cap = gallery.cfg.max_embeddings_per_identity base = _unit(1) a = gallery.enroll("A", [_at_similarity(base, 0.99, s + 20) for s in range(cap)]) b = gallery.enroll("B", [_at_similarity(base, 0.98, s + 50) for s in range(cap)]) assert len(gallery.index) == 2 * cap res = gallery.merge_identities(a, b) assert gallery.store.embedding_count(b) == cap assert len(res["dropped_embeddings"]) == cap assert len(gallery.index) == cap # -- refusals --------------------------------------------------------------- def test_refuses_self_merge(gallery): src, _ = _two_identities(gallery) res = gallery.merge_identities(src, src) assert res["ok"] is False assert "itself" in res["reason"] @pytest.mark.parametrize("bad_side", ["source", "target"]) def test_refuses_missing_identity(gallery, bad_side): src, dst = _two_identities(gallery) args = (9999, dst) if bad_side == "source" else (src, 9999) res = gallery.merge_identities(*args) assert res["ok"] is False assert "not found" in res["reason"] def test_refuses_two_different_people(gallery): """Below enroll_threshold resolve() positively asserts these are different people. Merging anyway would contradict the number driving every other decision, so it needs an explicit override.""" a = gallery.enroll("A", [_unit(1)]) b = gallery.enroll("B", [_at_similarity(_unit(1), 0.10)]) res = gallery.merge_identities(a, b) assert res["ok"] is False assert res["similarity"] == pytest.approx(0.10, abs=0.01) assert gallery.store.stats()["identities"] == 2 def test_force_overrides_the_similarity_guard(gallery): a = gallery.enroll("A", [_unit(1)]) b = gallery.enroll("B", [_at_similarity(_unit(1), 0.10)]) res = gallery.merge_identities(a, b, force=True) assert res["ok"] is True assert res["forced"] is True assert gallery.store.stats()["identities"] == 1 def test_refuses_when_models_differ(gallery): """Vectors from two encoders are not comparable, so the guard cannot run. Refusing beats comparing numbers from different spaces and believing the answer.""" a = gallery.store.create_identity("A") gallery.store.add_embedding(a, _unit(1), 0.9, "other_model") b = gallery.enroll("B", [_unit(1)]) res = gallery.merge_identities(a, b) assert res["ok"] is False assert res["similarity"] is None assert "encoder model" in res["reason"] def test_similarity_uses_the_best_pair_not_the_mean(gallery): """Two identities of one person exist precisely because their typical views disagree; one agreeing pair is the evidence that matters.""" base = _unit(1) a = gallery.enroll("A", [base]) b = gallery.enroll("B", [_at_similarity(base, 0.05, 7), _at_similarity(base, 0.50, 8)]) sim, checkable = gallery._identity_similarity(a, b) assert checkable assert sim == pytest.approx(0.50, abs=0.01) def test_sighting_cooldown_cache_forgets_the_source(gallery): """The cooldown is keyed by identity id; leaving the source's key behind leaks an entry pointing at an identity that no longer exists.""" gal = gallery gal.cfg = RecognitionSection(sighting_cooldown_seconds=30.0) src, dst = _two_identities(gal) assert any(k[0] == src for k in gal._last_sighting) gal.merge_identities(src, dst) assert not any(k[0] == src for k in gal._last_sighting) # -- finding the duplicates in the first place ------------------------------ def test_duplicate_candidates_finds_the_split(gallery): src, dst = _two_identities(gallery) pairs = gallery.duplicate_candidates() assert len(pairs) == 1 found = {pairs[0]["a"]["id"], pairs[0]["b"]["id"]} assert found == {src, dst} assert pairs[0]["similarity"] == pytest.approx(0.40, abs=0.01) # 0.40 is under match_threshold, so it is a suggestion, not a verdict. assert pairs[0]["confident"] is False def test_duplicate_candidates_ignores_different_people(gallery): gallery.enroll("A", [_unit(1)]) gallery.enroll("B", [_at_similarity(_unit(1), 0.10)]) assert gallery.duplicate_candidates() == [] def test_duplicate_candidates_empty_gallery(gallery): assert gallery.duplicate_candidates() == []