Files
Behavision/behavision/commission.py
Suriyakumarvijayanayagam dad04e8cda Behavision: face recognition for retail, edge to head office
Five components that ship as one product:

- behavision/  the recognition engine. RTSP ingest, YuNet detection, IoU
               tracking, ArcFace embeddings, a FAISS/SQLite gallery, and a
               FastAPI dashboard. Identity is decided once per TRACK from an
               average of at least three embeddings, never per frame.
- agent/       the Go edge agent: supervises the engine, holds a durable
               spool, and drains it to MQTT. Nothing is acked before the
               broker confirms.
- desktop/     the shop PC application (Wails + React + tray).
- server/      the cloud API, MQTT consumer, reports and assistant.
- web/         platform.loyaly.ai, the head-office app, embedded in the
               server binary.

The gallery stores 512-float embeddings and timestamps - no images unless
`app.store_faces` is switched on. Those embeddings are biometric personal
data under GDPR and India's DPDP: template inversion reconstructs a
recognisable face from an ArcFace vector, so data/behavision.db is treated
as a biometric database and DELETE /api/visitors/{id} is a real erasure.

CLAUDE.md carries the reasoning behind every non-obvious decision here,
including the ones that were measured and the ones that were wrong first.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HViLj9gYNRtSr7YVZmW5sn
2026-09-04 11:14:18 +05:30

243 lines
11 KiB
Python

"""Camera commissioning: is this camera placed well enough to recognise faces?
The Office1 camera was installed, ran for weeks, and recognised almost nobody.
Nothing was broken — the overhead angle tilted every face down and the frosted
glass backlit them, so ArcFace never received a view it could embed stably. It
took reading vectors out of SQLite by hand to find that out.
This turns that diagnosis into an install step. The person installing walks
past a few times and gets one of two answers: "this camera is good" or "move it
to head height facing the approach direction". A site cannot be signed off
broken and then discovered three weeks later from a footfall report that was
always zero.
It measures the *live pipeline*, not a separate probe: every finished track
reports the best face quality it managed. That is the right question — not
"were the frames sharp" but "did a person walking past produce at least one
view worth enrolling" — and it is the same number `fraction_below_gate` on the
dashboard is built from, so the wizard and the running system cannot disagree.
"""
from __future__ import annotations
import threading
import time
from typing import Optional
# Verdict boundaries, from measured data on real cameras (see CLAUDE.md):
# frontal faces at head height score 0.70-0.82, the overhead corridor scores
# 0.32-0.45 against a 0.65 gate. The fractions below are of faces that fall
# under whatever gate that camera is configured with.
GOOD_BELOW_GATE = 0.20
POOR_BELOW_GATE = 0.50
# A real walk-past varies; a static artifact does not. Frosted-glass tracks
# measured a flat 0.37 on every frame, and a constant score across many
# detections is the signature of a thing, not a person.
FLAT_SPREAD = 0.03
FLAT_MIN_SAMPLES = 6
# Below this many faces the numbers are anecdote, not measurement.
MIN_SAMPLES = 5
DEFAULT_SECONDS = 25.0
def quantile(ordered: "list[float]", frac: float) -> float:
if not ordered:
return 0.0
idx = min(len(ordered) - 1, int(round(frac * (len(ordered) - 1))))
return ordered[idx]
class CommissionRun:
"""One timed placement check on one camera.
Written by the worker thread as tracks end, read by API threads polling
for the result, hence the lock.
"""
def __init__(self, camera_id: str, gate: float,
seconds: float = DEFAULT_SECONDS,
now: "float | None" = None):
self.camera_id = camera_id
self.gate = gate
self.seconds = max(5.0, float(seconds))
self.started_at = now if now is not None else time.time()
self._lock = threading.Lock()
self._qualities: "list[float]" = []
# Frames on which at least one face was being tracked. A face in view
# and a face that completed a pass are different observations, and
# only the second one produces a quality sample.
self._live_frames = 0
self._cancelled = False
# -- written by the worker thread -----------------------------------
def record(self, best_quality: float, now: "float | None" = None) -> None:
"""One finished track's best view. Tracks that never held a face at
all are not evidence about placement — they are evidence about
detection — so they are dropped."""
if best_quality <= 0:
return
if not self.running(now):
return
with self._lock:
self._qualities.append(float(best_quality))
def observe(self, live_faces: int, now: "float | None" = None) -> None:
"""One frame's worth of live tracking, whether or not anything ended.
Without this the check cannot tell "the camera sees nobody" from
"somebody is standing in front of it right now", because both produce
zero finished tracks — and those two states need opposite advice.
"""
if live_faces <= 0 or not self.running(now):
return
with self._lock:
self._live_frames += 1
# -- read by API threads --------------------------------------------
def running(self, now: "float | None" = None) -> bool:
if self._cancelled:
return False
now = now if now is not None else time.time()
return now - self.started_at < self.seconds
def cancel(self) -> None:
self._cancelled = True
def report(self, now: "float | None" = None) -> dict:
now = now if now is not None else time.time()
with self._lock:
ordered = sorted(self._qualities)
live = self._live_frames
running = self.running(now)
out = {
"camera_id": self.camera_id,
"gate": round(self.gate, 3),
"seconds": self.seconds,
"elapsed": round(min(now - self.started_at, self.seconds), 1),
"running": running,
"cancelled": self._cancelled,
"faces": len(ordered),
"frames_with_a_face": live,
"quality": _spread(ordered, self.gate),
}
out.update(self._verdict(ordered, running, live))
return out
# -- internals ------------------------------------------------------
def _verdict(self, ordered: "list[float]", running: bool,
live: int = 0) -> dict:
n = len(ordered)
if running:
done = f"{n} pass{'' if n == 1 else 'es'} completed"
# Saying "0 faces" while a face is plainly on screen reads as a
# broken check, so report what is actually happening.
seen = " · face in view" if live else ""
return {"verdict": "running",
"headline": f"watching… {done}{seen}",
"advice": ["Walk past the camera the way a customer "
"would, and out of the frame."]}
if n == 0 and live:
# A face was tracked the whole time and never left. The camera is
# aimed correctly and the old advice ("check it is pointing at the
# walkway") would send an installer to move a camera looking
# straight at them — which is how a good camera gets made bad.
return {"verdict": "no_completed_passes",
"headline": "a face was in view, but nobody walked past",
"advice": [
"The camera is detecting a face, so it is pointed "
"correctly — but no one completed a pass.",
"This check scores the best view of each person as "
"they leave the frame, which is what recognition "
"actually uses, so standing still measures nothing.",
"Walk through the frame and out of it, a few times, "
"then run the check again."]}
if n == 0:
# Streaming but nothing detected. Distinguishing this from "placed
# badly" matters: the fix is completely different.
return {"verdict": "no_faces",
"headline": "no faces detected",
"advice": [
"The camera is streaming but saw no face at all.",
"Check it is pointing at the walkway, not the ceiling "
"or floor, and that someone walked through the frame.",
"If people did walk past, the view is too far, too "
"dark, or too steep for the detector."]}
below = sum(1 for q in ordered if q < self.gate) / n
p50 = quantile(ordered, 0.50)
spread = quantile(ordered, 0.95) - quantile(ordered, 0.05)
if n >= FLAT_MIN_SAMPLES and spread < FLAT_SPREAD:
# Every detection scoring the same is not a camera problem to
# solve by moving it - it is not seeing people at all.
return {"verdict": "artifact",
"headline": f"every detection scored {p50:.2f} — this is "
"probably not a face",
"advice": [
"A constant score across every detection is the "
"signature of a static object, not a person.",
"Glass, a poster, a reflection or a mannequin in view "
"will do this.",
"Point the camera away from it, or raise "
"detection.score_threshold for this camera."]}
if n < MIN_SAMPLES:
return {"verdict": "inconclusive",
"headline": f"only {n} face{'' if n == 1 else 's'} seen — "
"not enough to judge",
"advice": [
f"Median quality was {p50:.2f}, but {n} "
f"sample{'' if n == 1 else 's'} is anecdote, not "
"measurement.",
"Run the check again and walk past several times, "
"ideally with more than one person."]}
if below <= GOOD_BELOW_GATE:
return {"verdict": "good",
"headline": f"good placement — median quality {p50:.2f}",
"advice": [
f"{below:.0%} of faces fell below the {self.gate:.2f} "
"enrollment gate. This camera can enrol and recognise "
"people reliably."]}
if below <= POOR_BELOW_GATE:
return {"verdict": "marginal",
"headline": f"usable but weak — {below:.0%} of faces are "
"below the gate",
"advice": [
f"Median quality {p50:.2f} against a "
f"{self.gate:.2f} gate: roughly {below:.0%} of "
"visitors will be seen and then discarded.",
"Angling it to face the approach direction, or "
"lowering it toward head height, usually fixes this.",
"If the position cannot change, lower this camera's "
"min_enroll_quality — but only this camera's."]}
return {"verdict": "poor",
"headline": f"poor placement — {below:.0%} of faces are below "
"the gate",
"advice": [
f"Median quality {p50:.2f} against a {self.gate:.2f} "
"gate. Most people who walk past will not be enrolled or "
"recognised, and nothing will look broken.",
"Move the camera to roughly head height, facing the "
"direction people approach from.",
"An overhead camera tilts every face downward, which is "
"the single most common cause of this result.",
"Backlighting — a window or lit glass behind the "
"subject — is the second most common.",
"Re-run this check after moving it. Do not lower the "
"quality gate to make this message go away: it converts a "
"visible miss into an invisible wrong match."]}
def _spread(ordered: "list[float]", gate: float) -> dict:
out = {"n": len(ordered)}
if not ordered:
return out
out["p05"] = round(quantile(ordered, 0.05), 3)
out["p50"] = round(quantile(ordered, 0.50), 3)
out["p95"] = round(quantile(ordered, 0.95), 3)
out["fraction_below_gate"] = round(
sum(1 for v in ordered if v < gate) / len(ordered), 3)
return out