Behavision: face recognition for retail, edge to head office

Five components that ship as one product:

- behavision/  the recognition engine. RTSP ingest, YuNet detection, IoU
               tracking, ArcFace embeddings, a FAISS/SQLite gallery, and a
               FastAPI dashboard. Identity is decided once per TRACK from an
               average of at least three embeddings, never per frame.
- agent/       the Go edge agent: supervises the engine, holds a durable
               spool, and drains it to MQTT. Nothing is acked before the
               broker confirms.
- desktop/     the shop PC application (Wails + React + tray).
- server/      the cloud API, MQTT consumer, reports and assistant.
- web/         platform.loyaly.ai, the head-office app, embedded in the
               server binary.

The gallery stores 512-float embeddings and timestamps - no images unless
`app.store_faces` is switched on. Those embeddings are biometric personal
data under GDPR and India's DPDP: template inversion reconstructs a
recognisable face from an ArcFace vector, so data/behavision.db is treated
as a biometric database and DELETE /api/visitors/{id} is a real erasure.

CLAUDE.md carries the reasoning behind every non-obvious decision here,
including the ones that were measured and the ones that were wrong first.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HViLj9gYNRtSr7YVZmW5sn
This commit is contained in:
2026-09-04 11:14:18 +05:30
commit dad04e8cda
216 changed files with 40473 additions and 0 deletions

7
behavision/__init__.py Normal file
View File

@@ -0,0 +1,7 @@
"""Behavision — production face recognition over RTSP.
Pipeline: capture -> detect (YuNet) -> track (IoU) -> align + encode
(ArcFace ONNX) -> match / auto-enroll (FAISS + SQLite) -> events + API.
"""
__version__ = "1.0.0"

261
behavision/__main__.py Normal file
View File

@@ -0,0 +1,261 @@
"""CLI: python -m behavision {run | enroll | setup-models}"""
from __future__ import annotations
import argparse
import logging
import sys
from pathlib import Path
from .config import ensure_api_credentials, load_config
from .log import setup_logging
log = logging.getLogger("behavision")
def cmd_run(args: argparse.Namespace) -> int:
import uvicorn
from .api import create_app
from .engine import Engine
from .model_assets import setup_models
cfg = load_config(args.config)
setup_logging(cfg.app.log_level, cfg.app.data_dir)
missing = setup_models(cfg.app.models_dir)
if missing:
log.error("required models missing: %s", ", ".join(missing))
return 1
auth_on, generated = ensure_api_credentials(cfg)
if generated:
log.warning(
"no API credentials configured - generated one for %s:%s\n"
" username: %s\n password: %s\n"
" (saved to %s; set BEHAVISION_API_USER / "
"BEHAVISION_API_PASSWORD in .env to choose your own)",
cfg.api.host, cfg.api.port, cfg.api.username, cfg.api.password,
cfg.app.data_dir / "api_credentials.txt")
elif not auth_on:
log.info("API bound to %s - serving without authentication",
cfg.api.host)
engine = Engine(cfg)
if not engine.workers:
# A fresh install legitimately has no cameras - the user adds them
# from the dashboard. Refusing to boot here would mean they could
# never reach the UI that adds the first one.
log.info("no cameras yet - add one at http://%s:%s",
"localhost" if cfg.api.is_loopback else cfg.api.host,
cfg.api.port)
engine.start()
try:
uvicorn.run(create_app(engine), host=cfg.api.host, port=cfg.api.port,
log_level="warning")
finally:
engine.stop()
return 0
def cmd_enroll(args: argparse.Namespace) -> int:
import cv2
from .detection import FaceDetector
from .gallery import Gallery, IdentityStore, VectorIndex
from .recognition import EMBEDDING_DIM, ArcFaceEncoder, face_quality
cfg = load_config(args.config)
setup_logging(cfg.app.log_level)
detector = FaceDetector(cfg.app.models_dir,
cfg.detection.score_threshold,
cfg.detection.nms_threshold)
encoder = ArcFaceEncoder(cfg.app.models_dir, cfg.recognition.model_file)
store = IdentityStore(cfg.app.data_dir / "behavision.db")
gallery = Gallery(store, VectorIndex(EMBEDDING_DIM), cfg.recognition,
encoder.model_name)
paths: list[Path] = []
for p in args.images:
p = Path(p)
if p.is_dir():
paths += [f for f in sorted(p.iterdir())
if f.suffix.lower() in (".jpg", ".jpeg", ".png", ".bmp")]
else:
paths.append(p)
embeddings = []
for path in paths:
image = cv2.imread(str(path))
if image is None:
log.warning("unreadable image skipped: %s", path)
continue
detections = detector.detect(image)
if not detections:
log.warning("no face found in %s", path)
continue
best = max(detections, key=lambda d: (d.box[2] - d.box[0])
* (d.box[3] - d.box[1]))
emb = encoder.encode(image, best.kps)
if emb is None:
log.warning("could not embed face in %s", path)
continue
q = face_quality(image, best.box, best.kps)
embeddings.append(emb)
log.info("embedded %s (quality %.2f)", path.name, q)
if not embeddings:
log.error("no usable faces - nothing enrolled")
return 1
identity_id = gallery.enroll(args.name, embeddings)
log.info("enrolled '%s' as identity %d with %d embedding(s)",
args.name, identity_id, len(embeddings))
store.close()
return 0
def cmd_calibrate(args: argparse.Namespace) -> int:
"""Measure the similarity distributions this camera+encoder actually
produce, then report thresholds that separate them."""
from .calibrate import CalibrationStore, capture, format_report
from .detection import FaceDetector
from .recognition import MODEL_CANDIDATES, ArcFaceEncoder
cfg = load_config(args.config)
setup_logging(cfg.app.log_level)
store = CalibrationStore(cfg.app.data_dir / "calibration.npz")
if args.report:
if not store.models():
log.error("no samples yet - run: python -m behavision calibrate "
"--person NAME")
return 1
print(format_report(store, cfg))
return 0
if not args.person:
log.error("give --person NAME to capture, or --report to analyse")
return 1
# Every model that is present gets embedded from the SAME frames, so an
# A/B between encoders is a fair comparison rather than two sessions.
names = [args.model] if args.model else MODEL_CANDIDATES
encoders = {}
for name in names:
if not (cfg.app.models_dir / name).exists():
continue
try:
enc = ArcFaceEncoder(cfg.app.models_dir, name,
cfg.recognition.color_order)
encoders[enc.model_name] = enc
except Exception:
log.warning("%s did not load - skipping", name)
if not encoders:
log.error("no recognition model loaded from %s", cfg.app.models_dir)
return 1
log.info("calibrating with: %s", ", ".join(encoders))
detector = FaceDetector(cfg.app.models_dir, cfg.detection.score_threshold,
cfg.detection.nms_threshold, cfg.detection.max_faces,
cfg.detection.min_face_px)
source = args.source
if source is None:
cam = cfg.cameras[0] if cfg.cameras else None
if cam is None:
log.error("no cameras configured - pass --source")
return 1
source = cam.source()
log.info("capturing '%s' for %.0fs - vary pose, distance and expression",
args.person, args.seconds)
try:
# Deliberately ungated: the enrollment gate is one of the things being
# calibrated, and filtering by it here would make it unmeasurable.
samples, qualities = capture(source, args.person, args.seconds, cfg,
detector, encoders)
except RuntimeError:
log.exception("capture failed")
return 1
kept = 0
for model, embeddings in samples.items():
if len(embeddings):
kept = store.add(model, args.person, embeddings, qualities)
if not kept:
log.error("no usable faces captured for '%s' - nothing stored "
"(nobody in frame, two faces at once, or too far away?)",
args.person)
return 1
store.save()
log.info("stored %d embeddings for '%s' (total per model). Capture more "
"people, then: python -m behavision calibrate --report",
kept, args.person)
return 0
def cmd_setup_models(args: argparse.Namespace) -> int:
from .model_assets import setup_models
cfg = load_config(args.config)
setup_logging(cfg.app.log_level)
missing = setup_models(cfg.app.models_dir)
if missing:
log.error("still missing (place them in %s manually): %s",
cfg.app.models_dir, ", ".join(missing))
return 1
log.info("all required models present in %s", cfg.app.models_dir)
return 0
def cmd_paths(args) -> int:
"""Where everything lives. An installer and a support call both need this,
and installed it is not next to the code."""
from .paths import describe
# load_config first: it seeds the editable copy, and describing the config
# path before that would name the bundled file rather than the one the next
# run actually loads.
cfg = load_config(args.config)
info = describe()
info["data_dir"] = str(cfg.app.data_dir)
info["models_dir"] = str(cfg.app.models_dir)
width = max(len(k) for k in info)
for key, value in info.items():
print(f"{key.rjust(width)} : {value}")
return 0
def main() -> int:
parser = argparse.ArgumentParser(
prog="behavision", description="Face recognition over RTSP")
parser.add_argument("--config", default=None,
help="path to YAML config (default: config/default.yaml)")
sub = parser.add_subparsers(dest="command", required=True)
sub.add_parser("run", help="start the pipeline + API server")
enroll = sub.add_parser("enroll", help="enroll a person from images")
enroll.add_argument("--name", required=True)
enroll.add_argument("--images", nargs="+", required=True,
help="image files and/or directories")
sub.add_parser("setup-models", help="download/copy model files")
sub.add_parser("paths", help="show where config, data and models live")
cal = sub.add_parser(
"calibrate",
help="measure similarity distributions and recommend thresholds")
cal.add_argument("--person", help="label for this capture session")
cal.add_argument("--seconds", type=float, default=20.0)
cal.add_argument("--source", default=None,
help="capture source (default: first configured camera)")
cal.add_argument("--model", default=None,
help="only this model file (default: all present)")
cal.add_argument("--report", action="store_true",
help="analyse stored samples instead of capturing")
args = parser.parse_args()
handlers = {"run": cmd_run, "enroll": cmd_enroll,
"setup-models": cmd_setup_models, "calibrate": cmd_calibrate,
"paths": cmd_paths}
return handlers[args.command](args)
if __name__ == "__main__":
sys.exit(main())

341
behavision/api.py Normal file
View File

@@ -0,0 +1,341 @@
"""HTTP API + minimal live dashboard (FastAPI).
Every endpoint is guarded by engine readiness; the server can start before
models finish loading without a single unguarded None dereference.
"""
from __future__ import annotations
import asyncio
import logging
import secrets
from pathlib import Path
from typing import Optional
from fastapi import Depends, FastAPI, HTTPException
from fastapi.responses import HTMLResponse, Response, StreamingResponse
from fastapi.security import HTTPBasic, HTTPBasicCredentials
from pydantic import BaseModel, ValidationError
from .config import ApiSection, CameraConfig, CameraTuning
from .commission import CommissionRun
from .events import Event
from .engine import Engine
log = logging.getLogger(__name__)
_STATIC = Path(__file__).parent / "static"
class RenamePayload(BaseModel):
label: str
class CommissionPayload(BaseModel):
seconds: float = 25.0
class MergePayload(BaseModel):
"""`into` is the identity that survives. `force` overrides the
similarity guard and is never the default: a wrong merge cannot be
undone, because nothing records which embedding came from whom."""
into: int
force: bool = False
class CameraPayload(BaseModel):
"""Camera as the UI submits it. Mirrors CameraConfig but every field is
optional so PATCH can send a subset.
Optional[...] rather than `X | None`: pydantic evaluates field annotations
at runtime, and CameraConfig already uses this form.
"""
id: Optional[str] = None
url: Optional[str] = None
host: Optional[str] = None
port: Optional[int] = None
path: Optional[str] = None
username: Optional[str] = None
password: Optional[str] = None
webcam: Optional[int] = None
max_width: Optional[int] = None
# Per-camera gate overrides. Without this the store could hold them but
# nothing could set them, so the commissioning advice ("loosen this
# camera's quality gate") had no way to be acted on.
tuning: Optional[CameraTuning] = None
def camera_public(cam: CameraConfig, worker=None) -> dict:
"""Camera as the API returns it.
The password is NEVER included — not masked, not empty-string-if-set,
absent. `safe_url()` already exists for exactly this and masks credentials
inside the URL form too.
"""
out = {
"id": cam.id, "host": cam.host, "port": cam.port, "path": cam.path,
"username": cam.username, "webcam": cam.webcam,
"max_width": cam.max_width, "has_password": bool(cam.password),
"url": cam.safe_url(),
"tuning": cam.tuning.model_dump(),
}
if worker is not None:
out.update(worker.stats())
out["url"] = cam.safe_url() # worker.stats() also carries a url key
return out
def _auth_dependencies(api_cfg: ApiSection) -> list:
"""HTTP Basic over every route when credentials are configured.
Applied at app level rather than per-route so a future endpoint cannot be
added unprotected by omission. Basic (not a token) because the dashboard
is a browser page: the browser prompts once and then attaches the header
to the MJPEG <img> subresource too, which a bearer token cannot do.
"""
if not api_cfg.auth_enabled:
return []
scheme = HTTPBasic()
def check(credentials: HTTPBasicCredentials = Depends(scheme)) -> None:
# compare_digest on both halves: no early exit, no timing signal.
ok_user = secrets.compare_digest(
credentials.username.encode("utf-8"),
api_cfg.username.encode("utf-8"))
ok_pass = secrets.compare_digest(
credentials.password.encode("utf-8"),
api_cfg.password.encode("utf-8"))
if not (ok_user and ok_pass):
raise HTTPException(401, "invalid credentials",
headers={"WWW-Authenticate": "Basic"})
return [Depends(check)]
def create_app(engine: Engine) -> FastAPI:
app = FastAPI(title="Behavision", version="1.0.0",
dependencies=_auth_dependencies(engine.cfg.api))
def worker_or_404(camera_id: str):
worker = engine.workers.get(camera_id)
if worker is None:
raise HTTPException(404, f"unknown camera '{camera_id}'")
return worker
@app.get("/", response_class=HTMLResponse)
def dashboard() -> str:
return (_STATIC / "dashboard.html").read_text(encoding="utf-8")
@app.get("/api/health")
def health() -> dict:
from .paths import describe
return {"status": "ok" if engine.started_at else "starting",
"recognition_model": engine.encoder.model_name,
# "where is my database" must be answerable from the API: the
# tray, the installer and support all need it, and installed
# it is not next to the code.
"paths": {**describe(), "data_dir": str(engine.cfg.app.data_dir),
"models_dir": str(engine.cfg.app.models_dir)},
"cameras": {cid: w.source.connected
for cid, w in engine.workers.items()}}
@app.get("/api/stats")
def stats() -> dict:
return engine.stats()
@app.get("/api/events")
def events(limit: int = 50) -> list:
return list(engine.bus.recent)[:limit]
@app.get("/api/identities")
def identities(limit: int = 200) -> list:
return engine.store.list_identities(limit)
@app.get("/api/sightings")
def sightings(limit: int = 100) -> list:
return engine.store.recent_sightings(limit)
@app.patch("/api/identities/{identity_id}")
def rename_identity(identity_id: int, payload: RenamePayload) -> dict:
if not engine.store.rename_identity(identity_id, payload.label.strip()):
raise HTTPException(404, "identity not found")
return engine.store.get_identity(identity_id)
@app.get("/api/identities/{identity_id}/embedding")
def identity_embedding(identity_id: int) -> dict:
"""One identity's best stored vector, for forwarding to the server.
This returns biometric personal data. It is on the authenticated local
API and bound to loopback in the product, and it exists because the
event bus deliberately does not carry embeddings — putting a 512-float
template on the bus would send it to the log sink and the email sink
too.
"""
best = engine.store.best_embedding(identity_id, engine.gallery.model_name)
if best is None:
raise HTTPException(404, "no embedding for this identity "
"(or it was made by a different model)")
vector, quality = best
return {"identity_id": identity_id,
"model": engine.gallery.model_name,
"quality": round(quality, 3),
"embedding": [round(float(x), 6) for x in vector]}
@app.get("/api/identities/duplicates")
def duplicate_identities(limit: int = 20) -> list:
"""Identity pairs that look like one person enrolled twice."""
return engine.gallery.duplicate_candidates(limit)
@app.post("/api/identities/{identity_id}/merge")
def merge_identity(identity_id: int, payload: MergePayload) -> dict:
result = engine.gallery.merge_identities(
identity_id, payload.into, force=payload.force)
if not result.get("ok"):
reason = result.get("reason", "merge refused")
# 409, not 400: the request is well formed, it conflicts with what
# the gallery believes. The body carries the measured similarity so
# the UI can show the operator what it is asking them to override.
status = 404 if "not found" in reason else 409
raise HTTPException(status, detail=result)
engine.bus.publish(Event(
type="identity.merged", camera_id="",
data={k: result[k] for k in
("source", "target", "label", "similarity", "forced",
"embeddings_moved", "sightings_moved")}))
return result
@app.delete("/api/identities/{identity_id}")
def delete_identity(identity_id: int) -> dict:
if not engine.gallery.delete_identity(identity_id):
raise HTTPException(404, "identity not found")
return {"deleted": identity_id}
@app.get("/api/cameras")
def cameras() -> list:
out = []
for cam in engine.camera_store.list():
out.append(camera_public(cam, engine.workers.get(cam.id)))
return out
@app.post("/api/cameras", status_code=201)
def add_camera(payload: CameraPayload) -> dict:
data = payload.model_dump(exclude_none=True)
if not data.get("id"):
raise HTTPException(400, "id is required")
try:
cam = CameraConfig.model_validate(data)
cam.source() # reject "no url, no host, no webcam" before storing
# Resolve the per-camera gates here too. Without this an inverted
# enroll/match pair was only caught when the worker was built,
# which surfaced as a 500 "stored but failed to start" instead of
# telling the user what was wrong with what they typed.
engine.cfg.recognition.merged(cam.tuning)
except (ValidationError, ValueError) as exc:
raise HTTPException(400, str(exc))
try:
engine.camera_store.add(cam)
except ValueError as exc:
raise HTTPException(409, str(exc))
try:
engine.add_camera(cam)
except Exception as exc:
# Never leave the store describing a camera the engine refused —
# the two would disagree until the next restart.
engine.camera_store.delete(cam.id)
raise HTTPException(500, f"camera stored but failed to start: {exc}")
return camera_public(cam, engine.workers.get(cam.id))
@app.patch("/api/cameras/{camera_id}")
def edit_camera(camera_id: str, payload: CameraPayload) -> dict:
fields = payload.model_dump(exclude_none=True)
fields.pop("id", None)
try:
if "tuning" in fields:
engine.cfg.recognition.merged(
CameraTuning.model_validate(fields["tuning"]))
cam = engine.camera_store.update(camera_id, fields)
except (ValidationError, ValueError) as exc:
raise HTTPException(400, str(exc))
if cam is None:
raise HTTPException(404, f"unknown camera '{camera_id}'")
engine.restart_camera(cam) # a changed URL needs a fresh connection
return camera_public(cam, engine.workers.get(cam.id))
@app.delete("/api/cameras/{camera_id}")
def delete_camera(camera_id: str) -> dict:
if not engine.camera_store.delete(camera_id):
raise HTTPException(404, f"unknown camera '{camera_id}'")
engine.remove_camera(camera_id)
return {"deleted": camera_id}
@app.post("/api/cameras/{camera_id}/commission")
def start_commission(camera_id: str,
payload: CommissionPayload) -> dict:
"""Begin a placement check: watch this camera for N seconds and judge
whether faces here are good enough to enrol."""
worker = worker_or_404(camera_id)
# The camera's own gate, not the global one - the whole point is to
# judge this view against the threshold it will actually run under.
worker.commission = CommissionRun(
camera_id, worker.rcfg.min_enroll_quality, payload.seconds)
return worker.commission.report()
@app.get("/api/cameras/{camera_id}/commission")
def commission_result(camera_id: str) -> dict:
worker = worker_or_404(camera_id)
if worker.commission is None:
raise HTTPException(404, "no placement check has been run")
return worker.commission.report()
@app.delete("/api/cameras/{camera_id}/commission")
def cancel_commission(camera_id: str) -> dict:
worker = worker_or_404(camera_id)
if worker.commission is not None:
worker.commission.cancel()
return {"cancelled": camera_id}
@app.post("/api/cameras/test")
def test_camera(payload: CameraPayload) -> dict:
"""Try a camera WITHOUT saving it - the UI's Test button.
Deliberately a sync def so FastAPI runs it in the threadpool:
cv2.VideoCapture blocks hard and a wrong host can hang for the full
FFmpeg timeout, which would stall the whole event loop.
"""
from .capture import probe_source
data = payload.model_dump(exclude_none=True)
data.setdefault("id", "__test__")
try:
cam = CameraConfig.model_validate(data)
source = cam.source()
except (ValidationError, ValueError) as exc:
return {"ok": False, "error": str(exc)}
return probe_source(source, cam.max_width)
@app.get("/api/cameras/{camera_id}/frame.jpg")
def frame(camera_id: str) -> Response:
jpeg = worker_or_404(camera_id).latest_jpeg()
if jpeg is None:
raise HTTPException(503, "no frame yet")
return Response(jpeg, media_type="image/jpeg")
@app.get("/api/cameras/{camera_id}/stream.mjpeg")
async def stream(camera_id: str) -> StreamingResponse:
worker = worker_or_404(camera_id)
async def generate():
boundary = b"--frame\r\nContent-Type: image/jpeg\r\n\r\n"
# Stop when the camera is deleted or its worker dies - otherwise a
# removed camera leaves this generator running for the life of the
# process, holding a reference to a worker nothing else can see.
while engine.workers.get(camera_id) is worker and worker.is_alive():
jpeg = worker.latest_jpeg()
if jpeg is not None:
yield boundary + jpeg + b"\r\n"
await asyncio.sleep(0.1) # ~10 fps to the browser
return StreamingResponse(
generate(),
media_type="multipart/x-mixed-replace; boundary=frame")
return app

220
behavision/attributes.py Normal file
View File

@@ -0,0 +1,220 @@
"""Optional age / gender / emotion estimation.
Primary gender+age model: InsightFace `genderage.onnx` (2021, CNN trained
jointly with the face-recognition stack; outputs age in YEARS). Fallback:
the 2015 Levi-Hassner Caffe nets. Emotion: FER+ ONNX.
Crop discipline — the part that made the old results absurd: gender/age
models are trained on LOOSE head crops (hair, chin, head shape included),
so they receive a 1.5x-expanded box from the full frame, never the tight
112x112 recognition chip. Only FER+ gets the aligned chip.
Everything is best-effort: any net missing or failing (e.g. out of memory)
is skipped or disabled without touching the recognition pipeline.
"""
from __future__ import annotations
import logging
import threading
from pathlib import Path
import cv2
import numpy as np
log = logging.getLogger(__name__)
AGE_BUCKETS = ["0-2", "4-6", "8-12", "15-20", "25-32", "38-43", "48-53", "60+"]
GENDERS = ["Male", "Female"]
EMOTIONS = ["neutral", "happiness", "surprise", "sadness",
"anger", "disgust", "fear", "contempt"]
_CAFFE_MEAN = (78.4263377603, 87.7689143744, 114.895847746)
def _loose_head_crop(frame: np.ndarray, box, scale: float = 1.5) -> np.ndarray:
"""Square crop centered on the face box, expanded to include the whole
head; replicate-padded when it runs off-frame so aspect stays 1:1."""
x1, y1, x2, y2 = box
cx, cy = (x1 + x2) / 2.0, (y1 + y2) / 2.0
half = max(x2 - x1, y2 - y1) * scale / 2.0
fh, fw = frame.shape[:2]
gx1, gy1 = int(round(cx - half)), int(round(cy - half))
gx2, gy2 = int(round(cx + half)), int(round(cy + half))
pad_l, pad_t = max(0, -gx1), max(0, -gy1)
pad_r, pad_b = max(0, gx2 - fw), max(0, gy2 - fh)
crop = frame[max(0, gy1):min(fh, gy2), max(0, gx1):min(fw, gx2)]
if crop.size == 0:
return crop
if pad_l or pad_t or pad_r or pad_b:
crop = cv2.copyMakeBorder(crop, pad_t, pad_b, pad_l, pad_r,
cv2.BORDER_REPLICATE)
return crop
def aggregate(samples: "list[dict]") -> dict:
"""Combine per-frame estimates into one verdict for a track.
Age comes from a tiny CNN reading a single frame, so consecutive frames of
the same face can differ by a decade. Median over several frames (not mean)
keeps one wild frame from dragging the answer, and costs nothing but the
inferences already being run.
"""
samples = [s for s in samples if s]
if not samples:
return {}
out: dict = {}
ages = [s["age"] for s in samples if isinstance(s.get("age"), (int, float))]
if ages:
out["age"] = int(round(float(np.median(ages))))
out["age_spread"] = int(max(ages) - min(ages)) # honest uncertainty
for field, conf_field in (("gender", "gender_confidence"),
("emotion", "emotion_confidence"),
("age_range", None)):
votes: dict = {}
for s in samples:
v = s.get(field)
if v is None:
continue
votes.setdefault(v, []).append(s.get(conf_field, 1.0) if conf_field else 1.0)
if not votes:
continue
# most frames win; ties broken by mean confidence
best = max(votes, key=lambda k: (len(votes[k]), float(np.mean(votes[k]))))
out[field] = best
if conf_field:
out[conf_field] = round(float(np.mean(votes[best])), 3)
return out
class AttributeEstimator:
def __init__(self, models_dir: Path):
models_dir = Path(models_dir)
# cv2.dnn.Net (emotion + the Caffe fallbacks) is stateful across
# setInput/forward, so concurrent camera workers must not enter
# together. Attributes run once per TRACK, not per frame, so the
# contention this costs is negligible.
self._lock = threading.Lock()
self._genderage = None
self._ga_input = None
ga_path = models_dir / "genderage.onnx"
if ga_path.exists():
try:
import onnxruntime as ort
self._genderage = ort.InferenceSession(
str(ga_path), providers=["CPUExecutionProvider"])
inp = self._genderage.get_inputs()[0]
self._ga_input = inp.name
self._ga_size = (inp.shape[-1]
if isinstance(inp.shape[-1], int) else 96)
except Exception:
log.exception("genderage model failed to load")
self._genderage = None
# Legacy Caffe fallbacks, used only when genderage is unavailable.
self._gender = None
self._age = None
if self._genderage is None:
self._gender = self._load_caffe(models_dir, "gender")
self._age = self._load_caffe(models_dir, "age")
self._emotion = None
emo = models_dir / "emotion-ferplus-8.onnx"
if emo.exists():
try:
self._emotion = cv2.dnn.readNetFromONNX(str(emo))
except cv2.error:
log.exception("emotion model failed to load")
log.info("attributes: genderage=%s caffe(gender=%s age=%s) emotion=%s",
bool(self._genderage), bool(self._gender), bool(self._age),
bool(self._emotion))
@staticmethod
def _load_caffe(models_dir: Path, name: str):
proto = models_dir / f"{name}_deploy.prototxt"
weights = models_dir / f"{name}_net.caffemodel"
if not (proto.exists() and weights.exists()):
return None
try:
return cv2.dnn.readNetFromCaffe(str(proto), str(weights))
except cv2.error:
log.exception("%s model failed to load", name)
return None
@property
def has_genderage(self) -> bool:
return self._genderage is not None
@property
def any_loaded(self) -> bool:
return any([self._genderage, self._gender, self._age, self._emotion])
def estimate(self, frame_bgr: np.ndarray, box,
chip_bgr: np.ndarray) -> dict:
"""`frame_bgr` + `box` feed the gender/age nets (loose head crop);
`chip_bgr` (aligned 112x112) feeds FER+ emotion."""
out: dict = {}
head = _loose_head_crop(frame_bgr, box)
with self._lock:
if head.size:
if self._genderage is not None:
self._estimate_genderage(head, out)
elif self._gender is not None or self._age is not None:
self._estimate_caffe(head, out)
if (self._emotion is not None and chip_bgr is not None
and chip_bgr.size):
self._estimate_emotion(chip_bgr, out)
return out
# -- backends -------------------------------------------------------
def _estimate_genderage(self, head: np.ndarray, out: dict) -> None:
try:
size = self._ga_size
rgb = cv2.cvtColor(cv2.resize(head, (size, size)),
cv2.COLOR_BGR2RGB).astype(np.float32)
blob = rgb.transpose(2, 0, 1)[None]
pred = self._genderage.run(None, {self._ga_input: blob})[0][0]
# pred = [female_logit, male_logit, age/100]
g = np.array(pred[:2], dtype=np.float64)
probs = np.exp(g - g.max())
probs /= probs.sum()
out["gender"] = "Male" if pred[1] > pred[0] else "Female"
out["gender_confidence"] = round(float(probs.max()), 3)
out["age"] = int(round(float(pred[2]) * 100))
except Exception:
log.warning("genderage failed at inference - disabled")
self._genderage = None
def _estimate_caffe(self, head: np.ndarray, out: dict) -> None:
blob = cv2.dnn.blobFromImage(
cv2.resize(head, (227, 227)), 1.0, (227, 227),
_CAFFE_MEAN, swapRB=False)
if self._gender is not None:
try:
self._gender.setInput(blob)
probs = self._gender.forward().ravel()
out["gender"] = GENDERS[int(np.argmax(probs))]
out["gender_confidence"] = round(float(probs.max()), 3)
except cv2.error:
log.warning("gender net failed at inference - disabled")
self._gender = None
if self._age is not None:
try:
self._age.setInput(blob)
probs = self._age.forward().ravel()
out["age_range"] = AGE_BUCKETS[int(np.argmax(probs))]
except cv2.error:
log.warning("age net failed at inference - disabled")
self._age = None
def _estimate_emotion(self, chip_bgr: np.ndarray, out: dict) -> None:
try:
gray = cv2.cvtColor(chip_bgr, cv2.COLOR_BGR2GRAY)
blob = cv2.resize(gray, (64, 64)).astype(np.float32)[None, None]
self._emotion.setInput(blob)
logits = self._emotion.forward().ravel()
exp = np.exp(logits - logits.max())
probs = exp / exp.sum()
out["emotion"] = EMOTIONS[int(np.argmax(probs))]
out["emotion_confidence"] = round(float(probs.max()), 3)
except cv2.error:
log.warning("emotion net failed at inference - disabled")
self._emotion = None

535
behavision/calibrate.py Normal file
View File

@@ -0,0 +1,535 @@
"""Threshold calibration: derive match/enroll thresholds from measured data.
The three recognition thresholds are not universal constants — they describe a
particular *encoder* on a particular *camera*. Change either and the numbers
that were measured for the old pair silently stop describing the new one: the
gallery starts splitting one person into several (enroll_threshold too high) or
merging different people (match_threshold too low).
This module measures the two distributions that actually decide those numbers:
same-person similarity - how alike two views of ONE person look
cross-person similarity - how alike views of DIFFERENT people look
and reports thresholds that separate them, per model, so a model swap is a
measurement rather than a guess.
Two details make the measurement match runtime instead of merely resembling it:
- Identity is decided from the *mean* of `min_embeddings_for_id` embeddings,
never a single frame (see engine._identify). So samples are grouped and
averaged the same way before any similarity is computed. Measuring
single-frame similarity would report a much wider spread than the running
system ever sees.
- Only frames that pass the live quality gate are collected, because those are
the only frames the running system ever embeds.
Privacy: no images are written. Chips are embedded in memory and only the
resulting vectors are stored, matching the guarantee the rest of the system
makes.
"""
from __future__ import annotations
import logging
import time
from pathlib import Path
from typing import Optional
import numpy as np
log = logging.getLogger(__name__)
# Below this, a "recommendation" would be fitting noise.
MIN_GROUPS_PER_PERSON = 2
MIN_SAMPLES_PER_PERSON = 6
class CalibrationStore:
"""Embeddings per (model, person), persisted as a single .npz.
Keyed by model so one capture session can be replayed against several
encoders — that is what makes an A/B of two models fair: identical faces,
identical frames, only the encoder differs.
"""
def __init__(self, path: "Path | str"):
self.path = Path(path)
self.data: dict[str, np.ndarray] = {}
if self.path.exists():
with np.load(self.path) as npz:
self.data = {k: npz[k] for k in npz.files}
# Quality is stored under a parallel key rather than a second file, so a
# capture session stays one artefact. Suffixed (not prefixed) so the
# model/person parsing below keeps working on old archives.
_Q = "||__quality"
@staticmethod
def _key(model: str, person: str) -> str:
return f"{model}||{person}"
def add(self, model: str, person: str, embeddings: np.ndarray,
qualities: "np.ndarray | None" = None) -> int:
key = self._key(model, person)
if key in self.data and len(self.data[key]):
embeddings = np.vstack([self.data[key], embeddings])
self.data[key] = np.asarray(embeddings, dtype=np.float32)
if qualities is not None:
qkey = key + self._Q
q = np.asarray(qualities, dtype=np.float32).reshape(-1)
if qkey in self.data and len(self.data[qkey]):
q = np.concatenate([self.data[qkey], q])
self.data[qkey] = q
return len(self.data[key])
def models(self) -> "list[str]":
return sorted({k.split("||", 1)[0] for k in self.data
if not k.endswith(self._Q)})
def people(self, model: str) -> "list[str]":
return sorted(k.split("||", 1)[1] for k in self.data
if k.startswith(f"{model}||") and not k.endswith(self._Q))
def get(self, model: str, person: str) -> np.ndarray:
return self.data.get(self._key(model, person), np.empty((0, 512), np.float32))
def qualities(self, model: str, person: str) -> np.ndarray:
"""Per-embedding quality, or empty for an archive captured before
quality was recorded. Empty means 'unknown', never 'zero'."""
q = self.data.get(self._key(model, person) + self._Q,
np.empty(0, np.float32))
emb = self.get(model, person)
# A partially-upgraded archive would silently misalign the two arrays.
return q if len(q) == len(emb) else np.empty(0, np.float32)
def save(self) -> None:
self.path.parent.mkdir(parents=True, exist_ok=True)
np.savez_compressed(self.path, **self.data)
def _mean_unit(vectors: np.ndarray) -> Optional[np.ndarray]:
"""Normalised mean — the exact quantity the engine matches on."""
mean = vectors.mean(axis=0)
norm = float(np.linalg.norm(mean))
if norm < 1e-6:
return None
return (mean / norm).astype(np.float32)
def group_means(embeddings: np.ndarray, group_size: int) -> np.ndarray:
"""Chunk into groups of `group_size` and average each, mirroring the
multi-frame averaging in engine._identify. A trailing partial group is
kept only if it holds at least half a group, so one stray frame cannot
contribute a noisy 'identity' to the statistics."""
out = []
for start in range(0, len(embeddings), group_size):
chunk = embeddings[start:start + group_size]
if len(chunk) < max(2, (group_size + 1) // 2):
break
mean = _mean_unit(chunk)
if mean is not None:
out.append(mean)
return np.vstack(out) if out else np.empty((0, embeddings.shape[1]), np.float32)
def capture(source, label: str, seconds: float, cfg, detector, encoders: dict,
min_quality: Optional[float] = None
) -> "tuple[dict[str, np.ndarray], np.ndarray]":
"""Collect faces from `source`, embed with every encoder, keep the quality.
`min_quality` defaults to 0.0 — everything the detector finds is recorded,
with its score. It used to default to the live enrollment gate, which made
the gate impossible to calibrate: you cannot measure whether a threshold is
set correctly using only the data that threshold already admitted. The
filter now happens at analysis time (`distributions`), where it can be
varied, which keeps the runtime-matching property without the circularity.
`source` is anything cv2.VideoCapture accepts (webcam index, RTSP URL,
video file). Returns ({model_name: embeddings}, qualities) with the
quality array aligned to every model's rows. Raises if the source will not
open, since a silent empty capture is worse than a loud failure.
"""
import cv2
from .geometry import align_face
from .recognition import face_quality
if min_quality is None:
min_quality = 0.0
cap = cv2.VideoCapture(source)
if not cap.isOpened():
cap.release()
raise RuntimeError(f"cannot open capture source {source!r}")
per_model: dict[str, list] = {name: [] for name in encoders}
qualities: list = []
max_width = cfg.cameras[0].max_width if cfg.cameras else 1280
deadline = time.time() + seconds
seen = rejected = 0
try:
while time.time() < deadline:
ok, frame = cap.read()
if not ok or frame is None:
break
if max_width and frame.shape[1] > max_width:
scale = max_width / frame.shape[1]
frame = cv2.resize(frame, (max_width, int(frame.shape[0] * scale)),
interpolation=cv2.INTER_AREA)
detections = detector.detect(frame)
if len(detections) > 1:
# Two faces in frame makes the 'which person is this' label
# ambiguous, and a mislabelled sample poisons both curves.
rejected += 1
continue
for det in detections:
seen += 1
quality = face_quality(frame, det.box, det.kps)
if quality < min_quality:
rejected += 1
continue
# Commit a frame only if EVERY encoder embedded it. A partial
# row would desynchronise the models from each other and from
# the quality array, quietly breaking both the A/B comparison
# and the quality analysis.
row = {}
for name, enc in encoders.items():
chip = align_face(frame, det.kps, size=enc.size)
emb = enc.encode_chip(chip)
if emb is None:
break
row[name] = emb
if len(row) != len(encoders):
rejected += 1
continue
for name, emb in row.items():
per_model[name].append(emb)
qualities.append(quality)
finally:
cap.release()
kept = len(qualities)
log.info("[%s] %d faces seen, %d rejected (ambiguous/unencodable), %d kept "
"(quality p05 %.2f - p95 %.2f)", label, seen, rejected, kept,
float(np.percentile(qualities, 5)) if qualities else 0.0,
float(np.percentile(qualities, 95)) if qualities else 0.0)
return ({name: (np.vstack(v) if v else np.empty((0, 512), np.float32))
for name, v in per_model.items()},
np.asarray(qualities, dtype=np.float32))
def distributions(store: CalibrationStore, model: str, group_size: int,
min_quality: Optional[float] = None
) -> "tuple[np.ndarray, np.ndarray, dict]":
"""Same-person and cross-person similarity samples for one model.
`min_quality` filters to the frames the running system would actually
embed. Applied here rather than at capture time so the same archive can be
re-analysed against a different gate — that is what makes the gate itself
measurable instead of assumed.
"""
grouped, skipped = {}, {}
ungated, gated_out = [], {}
for person in store.people(model):
raw = store.get(model, person)
if min_quality is not None:
q = store.qualities(model, person)
if len(q):
kept = raw[q >= min_quality]
if len(kept) < len(raw):
gated_out[person] = (len(raw) - len(kept), len(raw))
raw = kept
else:
ungated.append(person)
means = group_means(raw, group_size)
if len(means) < MIN_GROUPS_PER_PERSON or len(raw) < MIN_SAMPLES_PER_PERSON:
skipped[person] = len(raw)
continue
grouped[person] = means
same, cross = [], []
people = sorted(grouped)
for i, person in enumerate(people):
m = grouped[person]
for a in range(len(m)):
for b in range(a + 1, len(m)):
same.append(float(m[a] @ m[b]))
for other in people[i + 1:]:
for va in m:
for vb in grouped[other]:
cross.append(float(va @ vb))
meta = {"people": people, "skipped": skipped,
"groups": {p: len(m) for p, m in grouped.items()}}
if ungated:
meta["ungated"] = ungated # captured before quality was recorded
if gated_out:
meta["gated_out"] = gated_out
return np.array(same), np.array(cross), meta
# -- quality gate -------------------------------------------------------
# Wide enough that a bucket holds real evidence, narrow enough to locate a
# knee; below this a bucket's median is one or two frames talking.
QUALITY_BUCKET = 0.05
MIN_BUCKET_SAMPLES = 5
# A bucket counts as "as good as this camera gets" within this fraction of the
# best bucket. Not an absolute target: what matters is whether a frame is
# materially worse than what this camera can produce, not how it compares to a
# number measured somewhere else.
KNEE_FRACTION = 0.90
def _self_similarity(embeddings: np.ndarray) -> np.ndarray:
"""Each embedding's similarity to its own person's mean, computed
leave-one-out so a sample is not compared against a mean it helped make."""
n = len(embeddings)
if n < 2:
return np.empty(0, np.float32)
total = embeddings.sum(axis=0)
others = (total - embeddings) / (n - 1)
norms = np.linalg.norm(others, axis=1, keepdims=True)
norms[norms < 1e-6] = 1.0
return np.einsum("ij,ij->i", embeddings, others / norms).astype(np.float32)
def quality_curve(store: CalibrationStore, model: str) -> dict:
"""Does face quality actually predict a usable embedding on this camera?
Pairs every captured frame's quality score with how much that frame looks
like its own person, then reports the relationship. This is the evidence
`min_enroll_quality` should be set from; it was previously the one
threshold in the system still chosen by hand.
"""
quals, sims = [], []
for person in store.people(model):
emb = store.get(model, person)
q = store.qualities(model, person)
if not len(q) or len(emb) < 2:
continue
sim = _self_similarity(emb)
if len(sim):
quals.append(q)
sims.append(sim)
if not quals:
return {"n": 0, "error": (
"no per-frame quality recorded - this archive predates quality "
"capture. Re-capture to calibrate the quality gate.")}
q = np.concatenate(quals)
sim = np.concatenate(sims)
out: dict = {"n": int(len(q)),
"quality": {"p05": round(float(np.percentile(q, 5)), 3),
"p50": round(float(np.percentile(q, 50)), 3),
"p95": round(float(np.percentile(q, 95)), 3)}}
# Whether the score means anything here at all. Undefined if every frame
# scored the same, which is itself the signature of a static artefact.
if q.std() > 1e-6 and sim.std() > 1e-6:
out["correlation"] = round(float(np.corrcoef(q, sim)[0, 1]), 3)
# Bin by integer index rather than by accumulating a float edge. Stepping
# `edge += 0.05` from 0.30 reaches 0.5000000000000001, so a quality of
# exactly 0.50 tests as *below* its own bucket and lands one step down —
# which shifts the recommended gate a whole bucket, and that number is
# copied straight into a config file.
idx = np.floor(q / QUALITY_BUCKET + 1e-9).astype(int)
buckets = []
for b in range(int(idx.min()), int(idx.max()) + 1):
sel = idx == b
if sel.sum() >= MIN_BUCKET_SAMPLES:
buckets.append({"lo": round(b * QUALITY_BUCKET, 2),
"hi": round((b + 1) * QUALITY_BUCKET, 2),
"n": int(sel.sum()),
"median_sim": round(float(np.median(sim[sel])), 3)})
out["buckets"] = buckets
if not buckets:
out["note"] = (f"fewer than {MIN_BUCKET_SAMPLES} frames in every "
"quality bucket - capture longer")
return out
best = max(b["median_sim"] for b in buckets)
target = best * KNEE_FRACTION
# Walk down from the top and stop at the first bucket that falls off, so a
# single noisy low bucket cannot drag the recommendation down with it.
gate = buckets[-1]["lo"]
for bucket in reversed(buckets):
if bucket["median_sim"] < target:
break
gate = bucket["lo"]
out["best_median_sim"] = round(float(best), 3)
out["min_enroll_quality"] = round(float(gate), 2)
out["retained_fraction"] = round(float((q >= gate).mean()), 3)
if gate <= buckets[0]["lo"]:
out["note"] = ("quality does not predict embedding stability on this "
"camera - every bucket is about as good as the best. "
"The gate is discarding frames for no measured benefit; "
"the limit here is the view, not the threshold.")
return out
def recommend(same: np.ndarray, cross: np.ndarray,
current_match: Optional[float] = None) -> dict:
"""Turn the two distributions into thresholds.
match_threshold - above the bulk of cross-person similarity, so a stranger
is not merged into an existing identity.
enroll_threshold - below the bulk of same-person similarity, so a returning
person is not minted as a duplicate.
Both are set from percentiles rather than raw min/max: one freak frame
should not move a production threshold. When the two curves overlap, no
pair of thresholds can separate them and that is reported as such rather
than papered over with a midpoint.
"""
out: dict = {"n_same": int(len(same)), "n_cross": int(len(cross))}
if len(same):
out["same"] = {"min": float(same.min()), "p01": float(np.percentile(same, 1)),
"p05": float(np.percentile(same, 5)),
"mean": float(same.mean()), "max": float(same.max())}
if len(cross):
out["cross"] = {"min": float(cross.min()), "mean": float(cross.mean()),
"p95": float(np.percentile(cross, 95)),
"p99": float(np.percentile(cross, 99)),
"max": float(cross.max())}
if not len(same):
out["error"] = ("no same-person pairs - capture more frames per person "
f"(need >={MIN_SAMPLES_PER_PERSON})")
return out
same_low = float(np.percentile(same, 5))
if len(cross):
cross_high = float(np.percentile(cross, 99))
out["separation"] = round(same_low - cross_high, 3)
if same_low <= cross_high:
out["error"] = (
"same-person and cross-person similarity OVERLAP - no threshold "
"pair separates them. Improve capture (pose, lighting, distance) "
"or use a stronger encoder before trusting any threshold.")
out["match_threshold"] = round(cross_high + 0.02, 2)
out["enroll_threshold"] = round(max(0.05, same_low - 0.02), 2)
return out
# BOTH thresholds are placed inside the gap between the curves, which
# keeps enroll < match however wide the separation turns out to be.
# Anchoring them to the distribution ends instead (same_p05 - margin)
# inverts the pair on well-separated data. match sits high in the gap
# because a false merge is unrecoverable — two people permanently share
# one identity — while a false split is a duplicate you can merge later.
gap = same_low - cross_high
out["match_threshold"] = round(cross_high + 0.55 * gap, 2)
out["enroll_threshold"] = round(cross_high + 0.15 * gap, 2)
if out["enroll_threshold"] >= out["match_threshold"]: # after rounding
out["enroll_threshold"] = round(out["match_threshold"] - 0.01, 2)
return out
out["note"] = ("only one person captured - cross-person similarity is "
"unmeasured, so match_threshold cannot be recommended. "
"Capture 2+ people to calibrate it.")
out["match_threshold"] = None
enroll = max(0.05, same_low - 0.05)
if current_match is not None and enroll > current_match - 0.01:
# config.py enforces enroll < match; never emit a value that would be
# rejected at load time against the match threshold still in force.
# Flag it, because a clamped value is the ceiling talking, not the
# data — without this, every model reports the same number and it
# reads like a measurement.
enroll = current_match - 0.01
out["clamped"] = (
f"same-person p05 is {same_low:.3f}, so the data supports an enroll "
f"threshold far above the current match threshold "
f"({current_match}). Clamped to sit just under it - calibrate "
f"match_threshold with 2+ people to lift both.")
out["enroll_threshold"] = round(enroll, 2)
return out
def format_report(store: CalibrationStore, cfg) -> str:
"""Human-readable report for every model in the store."""
group_size = cfg.tracking.min_embeddings_for_id
lines = [f"Calibration report ({store.path})",
f"grouping: mean of {group_size} embeddings (matches runtime)", ""]
for model in store.models():
same, cross, meta = distributions(store, model, group_size,
cfg.recognition.min_enroll_quality)
rec = recommend(same, cross, cfg.recognition.match_threshold)
qual = quality_curve(store, model)
lines.append(f"── {model} " + "─" * max(0, 56 - len(model)))
lines.append(f" people: {', '.join(meta['people']) or 'none'}")
if meta["skipped"]:
lines.append(" skipped (too few samples): " + ", ".join(
f"{p} ({n})" for p, n in meta["skipped"].items()))
if "same" in rec:
s = rec["same"]
lines.append(f" same-person n={rec['n_same']:<5} "
f"min {s['min']:.3f} p05 {s['p05']:.3f} mean {s['mean']:.3f}")
if "cross" in rec:
c = rec["cross"]
lines.append(f" cross-person n={rec['n_cross']:<5} "
f"mean {c['mean']:.3f} p99 {c['p99']:.3f} max {c['max']:.3f}")
if "separation" in rec:
lines.append(f" separation (same_p05 - cross_p99): {rec['separation']:+.3f}")
if "error" in rec:
lines.append(f" !! {rec['error']}")
if meta.get("gated_out"):
worst = ", ".join(f"{p} ({out}/{tot})"
for p, (out, tot) in meta["gated_out"].items())
lines.append(f" dropped by min_enroll_quality="
f"{cfg.recognition.min_enroll_quality}: {worst}")
if not meta["people"]:
# Otherwise the error above reads 'capture more frames' when
# plenty were captured and the gate discarded all of them —
# sending the operator to re-shoot instead of to the gate.
lines.append(" ^ every sample was captured, then filtered "
"out by the quality gate. The capture is fine; "
"the gate does not fit this camera.")
if "note" in rec:
lines.append(f" note: {rec['note']}")
if "clamped" in rec:
lines.append(f" clamped: {rec['clamped']}")
if meta.get("ungated"):
lines.append(" note: no per-frame quality for "
+ ", ".join(meta["ungated"])
+ " - analysed unfiltered (older capture)")
# -- quality gate ------------------------------------------------
if qual.get("error"):
lines.append(f" quality gate: {qual['error']}")
elif qual.get("buckets"):
q = qual["quality"]
lines.append("")
lines.append(f" face quality n={qual['n']:<5} "
f"p05 {q['p05']:.3f} p50 {q['p50']:.3f} "
f"p95 {q['p95']:.3f}")
if "correlation" in qual:
lines.append(" quality vs same-person similarity: "
f"r={qual['correlation']:+.3f}")
for b in qual["buckets"]:
bar = "#" * int(round(b["median_sim"] * 40))
lines.append(f" {b['lo']:.2f}-{b['hi']:.2f} "
f"n={b['n']:<4} med {b['median_sim']:.3f} {bar}")
if qual.get("note"):
lines.append(f" !! {qual['note']}")
elif qual.get("note"):
lines.append(f" quality gate: {qual['note']}")
if rec.get("match_threshold") is not None:
lines.append("")
lines.append(" recommended config/default.yaml:")
lines.append(" recognition:")
lines.append(f" match_threshold: {rec['match_threshold']}")
lines.append(f" enroll_threshold: {rec['enroll_threshold']}")
if qual.get("min_enroll_quality") is not None:
lines.append(f" min_enroll_quality: "
f"{qual['min_enroll_quality']}"
f" # keeps {qual['retained_fraction']:.0%} of faces")
elif rec.get("enroll_threshold") is not None:
lines.append("")
lines.append(" recommended (enroll only, match needs 2+ people):")
lines.append(f" enroll_threshold: {rec['enroll_threshold']}")
if qual.get("min_enroll_quality") is not None:
lines.append(f" min_enroll_quality: "
f"{qual['min_enroll_quality']}"
f" # keeps {qual['retained_fraction']:.0%} of faces")
lines.append("")
lines.append(f"current: match={cfg.recognition.match_threshold} "
f"enroll={cfg.recognition.enroll_threshold} "
f"min_enroll_quality={cfg.recognition.min_enroll_quality}")
return "\n".join(lines)

194
behavision/cameras.py Normal file
View File

@@ -0,0 +1,194 @@
"""Writable camera list — the store behind "user connects their camera".
Cameras used to live in `config/default.yaml` with credentials in `.env`, which
means adding one is an edit-and-restart. A product needs them added at runtime
from a UI, so they move here: a small JSON file the API can rewrite safely
while the engine is running.
Deliberately NOT stored in `behavision.db`. That file is a biometric database
with its own handling and erasure obligations; folding user-editable config
into it makes both harder to reason about, to back up, and to hand to support.
Passwords are protected at rest with Windows DPAPI. This is not theatre: the
file sits on the same disk as the face gallery, and an RTSP credential is a
live path into the camera itself.
"""
from __future__ import annotations
import base64
import json
import logging
import os
import threading
from pathlib import Path
from typing import Optional
from .config import CameraConfig
log = logging.getLogger(__name__)
_PLAIN = "plain:"
_DPAPI = "dpapi:"
# Machine scope, not user scope. The service (LocalSystem) and an admin running
# the CLI are different accounts, and a user-scoped blob written by one cannot
# be read by the other — a failure that only shows up after install, on the
# customer's machine. Machine scope still defends the actual threat here:
# someone copying cameras.json off the box.
_CRYPTPROTECT_LOCAL_MACHINE = 0x04
def _win32crypt():
try:
import win32crypt # type: ignore
return win32crypt
except ImportError:
return None
def protect(value: str) -> str:
"""Encrypt a secret for storage. Tagged so the format can change later."""
if not value:
return ""
crypt = _win32crypt()
if crypt is None:
return _PLAIN + value
try:
blob = crypt.CryptProtectData(value.encode("utf-8"), "behavision",
None, None, None,
_CRYPTPROTECT_LOCAL_MACHINE)
return _DPAPI + base64.b64encode(blob).decode("ascii")
except Exception:
log.warning("DPAPI unavailable - storing camera password unencrypted",
exc_info=True)
return _PLAIN + value
def unprotect(stored: str) -> str:
"""Inverse of `protect`. Never raises: a credential that cannot be read is
an empty credential, so one unreadable camera does not stop the engine."""
if not stored:
return ""
if stored.startswith(_PLAIN):
return stored[len(_PLAIN):]
if stored.startswith(_DPAPI):
crypt = _win32crypt()
if crypt is None:
log.error("camera password is DPAPI-encrypted but win32crypt is "
"unavailable - re-enter it on this machine")
return ""
try:
return crypt.CryptUnprotectData(
base64.b64decode(stored[len(_DPAPI):]),
None, None, None, 0)[1].decode("utf-8")
except Exception:
log.error("camera password could not be decrypted (config copied "
"from another machine?) - re-enter it", exc_info=True)
return ""
return stored # pre-tag file written before this module existed
class CameraStore:
"""Cameras as JSON, safe to rewrite while the engine is running."""
def __init__(self, path: "Path | str"):
self.path = Path(path)
self._lock = threading.RLock()
self._cameras: "dict[str, CameraConfig]" = {}
self._load()
# -- persistence ----------------------------------------------------
def _load(self) -> None:
if not self.path.exists():
return
try:
raw = json.loads(self.path.read_text(encoding="utf-8"))
except (json.JSONDecodeError, OSError):
log.exception("%s is unreadable - starting with no cameras "
"(the file is left in place, not overwritten)",
self.path)
return
for entry in raw.get("cameras", []):
try:
entry = dict(entry)
entry["password"] = unprotect(entry.get("password", ""))
cam = CameraConfig.model_validate(entry)
except Exception:
log.exception("skipping malformed camera entry %r", entry)
continue
self._cameras[cam.id] = cam
def _save(self) -> None:
"""Atomic: a crash mid-write must not leave a truncated camera list."""
payload = {"version": 1, "cameras": []}
for cam in self._cameras.values():
entry = cam.model_dump(mode="json")
entry["password"] = protect(cam.password)
payload["cameras"].append(entry)
self.path.parent.mkdir(parents=True, exist_ok=True)
tmp = self.path.with_suffix(".json.tmp")
tmp.write_text(json.dumps(payload, indent=2), encoding="utf-8")
try:
os.chmod(tmp, 0o600)
except OSError: # best effort (Windows)
pass
os.replace(tmp, self.path) # atomic on POSIX and NTFS
# -- CRUD -----------------------------------------------------------
def list(self) -> "list[CameraConfig]":
with self._lock:
return list(self._cameras.values())
def get(self, camera_id: str) -> Optional[CameraConfig]:
with self._lock:
return self._cameras.get(camera_id)
def add(self, camera: CameraConfig) -> CameraConfig:
with self._lock:
if camera.id in self._cameras:
raise ValueError(f"camera '{camera.id}' already exists")
camera.source() # validate now, not at connect time
self._cameras[camera.id] = camera
self._save()
return camera
def update(self, camera_id: str, fields: dict) -> Optional[CameraConfig]:
with self._lock:
existing = self._cameras.get(camera_id)
if existing is None:
return None
# id is the engine's key for the worker; renaming would orphan it
fields = {k: v for k, v in fields.items()
if k != "id" and v is not None}
# Re-validated rather than model_copy(update=...): copy does not
# coerce, so a nested `tuning` arriving as a plain dict from JSON
# would be stored as a dict and blow up the first time a camera
# asked it for its thresholds.
updated = CameraConfig.model_validate(
{**existing.model_dump(), **fields})
updated.source()
self._cameras[camera_id] = updated
self._save()
return updated
def delete(self, camera_id: str) -> bool:
with self._lock:
if self._cameras.pop(camera_id, None) is None:
return False
self._save()
return True
def seed(self, cameras: "list[CameraConfig]") -> bool:
"""Import YAML-declared cameras on first run only.
After that the store is authoritative — otherwise a camera the user
deleted in the UI would reappear on every restart.
"""
with self._lock:
if self.path.exists() or not cameras:
return False
for cam in cameras:
self._cameras[cam.id] = cam
self._save()
log.info("seeded %d camera(s) from YAML into %s",
len(cameras), self.path)
return True

237
behavision/capture.py Normal file
View File

@@ -0,0 +1,237 @@
"""Resilient video capture: RTSP (or webcam) reader thread with reconnect.
Design: one daemon thread per source holds the newest frame in a single
slot. Consumers always get the latest frame (never a backlog), and a lost
camera reconnects with exponential backoff instead of killing the pipeline.
"""
from __future__ import annotations
import logging
import os
import threading
import time
from typing import Optional
import cv2
import numpy as np
log = logging.getLogger(__name__)
# Force TCP transport and a 5s socket timeout for RTSP before OpenCV loads
# ffmpeg. UDP is the default and silently drops frames on lossy Wi-Fi.
os.environ.setdefault(
"OPENCV_FFMPEG_CAPTURE_OPTIONS", "rtsp_transport;tcp|stimeout;5000000"
)
def _tcp_reachable(source: "str | int", timeout: float
) -> "tuple[bool, str]":
"""Cheap pre-flight for an rtsp:// URL. Non-URL sources pass through."""
import socket
from urllib.parse import urlparse
if isinstance(source, int):
return True, ""
parsed = urlparse(source)
if not parsed.hostname:
return True, "" # not a form we can pre-check; let OpenCV try
port = parsed.port or (554 if parsed.scheme == "rtsp" else 80)
try:
with socket.create_connection((parsed.hostname, port), timeout):
return True, ""
except socket.timeout:
return False, (f"no response from {parsed.hostname}:{port} within "
f"{timeout:.0f}s - check the IP address and that the "
f"camera is on the same network")
except OSError as exc:
return False, f"cannot reach {parsed.hostname}:{port} - {exc.strerror or exc}"
def probe_source(source: "str | int", max_width: int = 1280,
timeout: float = 12.0, connect_timeout: float = 3.0) -> dict:
"""Open a candidate camera, grab one frame, and let go.
Backs the UI's Test button, so it must answer for a *wrong* URL as
reliably as a right one: no retries, no reconnect loop, and a hard deadline
because a bad host makes cv2.VideoCapture block until FFmpeg gives up.
Returns a JPEG snapshot so the user can confirm the camera is pointing
where they think it is.
"""
import base64
# cv2.VideoCapture blocks inside the constructor while FFmpeg completes a
# TCP connect, and against an unroutable host that is the OS connect
# timeout (~75s), not our deadline. A wrong IP or port is the single most
# likely thing a user types, so check reachability first — it turns the
# common failure into a sub-second answer instead of a frozen UI.
reachable, why = _tcp_reachable(source, connect_timeout)
if not reachable:
return {"ok": False, "error": why}
cap = None
try:
cap = (cv2.VideoCapture(source) if isinstance(source, int)
else cv2.VideoCapture(source, cv2.CAP_FFMPEG))
cap.set(cv2.CAP_PROP_BUFFERSIZE, 1)
if not cap.isOpened():
return {"ok": False, "error": "could not open stream - check the "
"host, port, path and credentials"}
deadline = time.time() + timeout
frame = None
while time.time() < deadline:
ok, candidate = cap.read()
if ok and candidate is not None and candidate.size:
frame = candidate
break
if frame is None:
return {"ok": False, "error": "connected but no frame arrived "
f"within {timeout:.0f}s"}
height, width = frame.shape[:2]
preview = frame
if max_width and width > max_width:
scale = max_width / width
preview = cv2.resize(frame, (max_width, int(height * scale)),
interpolation=cv2.INTER_AREA)
ok, buf = cv2.imencode(".jpg", preview,
[int(cv2.IMWRITE_JPEG_QUALITY), 70])
return {
"ok": True, "width": int(width), "height": int(height),
"downscaled_to": int(preview.shape[1]) if preview is not frame else None,
"snapshot": (base64.b64encode(buf.tobytes()).decode("ascii")
if ok else None),
}
except (cv2.error, MemoryError, OSError) as exc:
return {"ok": False, "error": f"{type(exc).__name__}: {exc}"}
finally:
if cap is not None:
cap.release()
class VideoSource(threading.Thread):
def __init__(self, camera_id: str, source: "str | int", display_url: str = "",
max_width: int = 1280):
super().__init__(daemon=True, name=f"capture-{camera_id}")
self.camera_id = camera_id
self._source = source
self._display_url = display_url or str(source)
# Downscale at ingest: 3MP+ streams waste memory and detector time,
# and on tight machines a full-res frame copy alone can OOM.
self.max_width = max_width
self._lock = threading.Lock()
self._frame: Optional[np.ndarray] = None
self._frame_ts: float = 0.0
# _stopping, NOT _stop. threading.Thread has its own private _stop(),
# and join() calls it: shadowing the name with an Event made every
# join() on a started worker raise "'Event' object is not callable".
# It only surfaces when a camera is removed or edited at runtime, so
# the engine answered 500 to every camera edit from head office while
# every test using a stubbed worker passed.
self._stopping = threading.Event()
self.connected = False
self.frames_total = 0
self.reconnects = 0
self._ever_connected = False
# -- public ---------------------------------------------------------
def latest(self) -> "tuple[Optional[np.ndarray], float]":
with self._lock:
if self._frame is None:
return None, 0.0
try:
return self._frame.copy(), self._frame_ts
except MemoryError:
return None, 0.0
def latest_since(self, known_ts: float) -> "tuple[Optional[np.ndarray], float]":
"""Latest frame, but only if it is newer than `known_ts`.
The staleness check happens under the lock so no frame is copied just
to be discarded — the worker polls far faster than the stream
delivers, and a discarded full-frame copy per poll is exactly the
allocation pattern that used to exhaust memory on small machines.
"""
with self._lock:
if self._frame is None or self._frame_ts == known_ts:
return None, self._frame_ts
try:
return self._frame.copy(), self._frame_ts
except MemoryError:
return None, 0.0
def stop(self) -> None:
self._stopping.set()
def stats(self) -> dict:
return {
"camera_id": self.camera_id,
"url": self._display_url,
"connected": self.connected,
"frames_total": self.frames_total,
"reconnects": self.reconnects,
"last_frame_age_s": round(time.time() - self._frame_ts, 1)
if self._frame_ts else None,
}
# -- thread ---------------------------------------------------------
def run(self) -> None:
backoff = 1.0
while not self._stopping.is_set():
cap = self._open()
if cap is None:
self.connected = False
log.warning("[%s] connect failed, retrying in %.0fs (%s)",
self.camera_id, backoff, self._display_url)
if self._stopping.wait(backoff):
break
backoff = min(backoff * 2, 30.0)
continue
self.connected = True
if self._ever_connected: # the first connect is not a reconnect
self.reconnects += 1
self._ever_connected = True
backoff = 1.0
log.info("[%s] connected (%s)", self.camera_id, self._display_url)
while not self._stopping.is_set():
try:
ok, frame = cap.read()
except (cv2.error, SystemError, MemoryError):
log.warning("[%s] read failed (low memory?), reconnecting",
self.camera_id)
break
if not ok or frame is None:
log.warning("[%s] stream dropped, reconnecting", self.camera_id)
break
try:
if self.max_width and frame.shape[1] > self.max_width:
scale = self.max_width / frame.shape[1]
frame = cv2.resize(
frame,
(self.max_width, int(frame.shape[0] * scale)),
interpolation=cv2.INTER_AREA)
except (cv2.error, MemoryError):
time.sleep(0.1) # transient allocation failure: drop frame
continue
with self._lock:
self._frame = frame
self._frame_ts = time.time()
self.frames_total += 1
cap.release()
self.connected = False
log.info("[%s] capture stopped", self.camera_id)
def _open(self) -> Optional[cv2.VideoCapture]:
try:
if isinstance(self._source, int):
cap = cv2.VideoCapture(self._source)
else:
cap = cv2.VideoCapture(self._source, cv2.CAP_FFMPEG)
cap.set(cv2.CAP_PROP_BUFFERSIZE, 1)
if not cap.isOpened():
cap.release()
return None
return cap
except cv2.error:
log.exception("[%s] VideoCapture error", self.camera_id)
return None

242
behavision/commission.py Normal file
View File

@@ -0,0 +1,242 @@
"""Camera commissioning: is this camera placed well enough to recognise faces?
The Office1 camera was installed, ran for weeks, and recognised almost nobody.
Nothing was broken — the overhead angle tilted every face down and the frosted
glass backlit them, so ArcFace never received a view it could embed stably. It
took reading vectors out of SQLite by hand to find that out.
This turns that diagnosis into an install step. The person installing walks
past a few times and gets one of two answers: "this camera is good" or "move it
to head height facing the approach direction". A site cannot be signed off
broken and then discovered three weeks later from a footfall report that was
always zero.
It measures the *live pipeline*, not a separate probe: every finished track
reports the best face quality it managed. That is the right question — not
"were the frames sharp" but "did a person walking past produce at least one
view worth enrolling" — and it is the same number `fraction_below_gate` on the
dashboard is built from, so the wizard and the running system cannot disagree.
"""
from __future__ import annotations
import threading
import time
from typing import Optional
# Verdict boundaries, from measured data on real cameras (see CLAUDE.md):
# frontal faces at head height score 0.70-0.82, the overhead corridor scores
# 0.32-0.45 against a 0.65 gate. The fractions below are of faces that fall
# under whatever gate that camera is configured with.
GOOD_BELOW_GATE = 0.20
POOR_BELOW_GATE = 0.50
# A real walk-past varies; a static artifact does not. Frosted-glass tracks
# measured a flat 0.37 on every frame, and a constant score across many
# detections is the signature of a thing, not a person.
FLAT_SPREAD = 0.03
FLAT_MIN_SAMPLES = 6
# Below this many faces the numbers are anecdote, not measurement.
MIN_SAMPLES = 5
DEFAULT_SECONDS = 25.0
def quantile(ordered: "list[float]", frac: float) -> float:
if not ordered:
return 0.0
idx = min(len(ordered) - 1, int(round(frac * (len(ordered) - 1))))
return ordered[idx]
class CommissionRun:
"""One timed placement check on one camera.
Written by the worker thread as tracks end, read by API threads polling
for the result, hence the lock.
"""
def __init__(self, camera_id: str, gate: float,
seconds: float = DEFAULT_SECONDS,
now: "float | None" = None):
self.camera_id = camera_id
self.gate = gate
self.seconds = max(5.0, float(seconds))
self.started_at = now if now is not None else time.time()
self._lock = threading.Lock()
self._qualities: "list[float]" = []
# Frames on which at least one face was being tracked. A face in view
# and a face that completed a pass are different observations, and
# only the second one produces a quality sample.
self._live_frames = 0
self._cancelled = False
# -- written by the worker thread -----------------------------------
def record(self, best_quality: float, now: "float | None" = None) -> None:
"""One finished track's best view. Tracks that never held a face at
all are not evidence about placement — they are evidence about
detection — so they are dropped."""
if best_quality <= 0:
return
if not self.running(now):
return
with self._lock:
self._qualities.append(float(best_quality))
def observe(self, live_faces: int, now: "float | None" = None) -> None:
"""One frame's worth of live tracking, whether or not anything ended.
Without this the check cannot tell "the camera sees nobody" from
"somebody is standing in front of it right now", because both produce
zero finished tracks — and those two states need opposite advice.
"""
if live_faces <= 0 or not self.running(now):
return
with self._lock:
self._live_frames += 1
# -- read by API threads --------------------------------------------
def running(self, now: "float | None" = None) -> bool:
if self._cancelled:
return False
now = now if now is not None else time.time()
return now - self.started_at < self.seconds
def cancel(self) -> None:
self._cancelled = True
def report(self, now: "float | None" = None) -> dict:
now = now if now is not None else time.time()
with self._lock:
ordered = sorted(self._qualities)
live = self._live_frames
running = self.running(now)
out = {
"camera_id": self.camera_id,
"gate": round(self.gate, 3),
"seconds": self.seconds,
"elapsed": round(min(now - self.started_at, self.seconds), 1),
"running": running,
"cancelled": self._cancelled,
"faces": len(ordered),
"frames_with_a_face": live,
"quality": _spread(ordered, self.gate),
}
out.update(self._verdict(ordered, running, live))
return out
# -- internals ------------------------------------------------------
def _verdict(self, ordered: "list[float]", running: bool,
live: int = 0) -> dict:
n = len(ordered)
if running:
done = f"{n} pass{'' if n == 1 else 'es'} completed"
# Saying "0 faces" while a face is plainly on screen reads as a
# broken check, so report what is actually happening.
seen = " · face in view" if live else ""
return {"verdict": "running",
"headline": f"watching… {done}{seen}",
"advice": ["Walk past the camera the way a customer "
"would, and out of the frame."]}
if n == 0 and live:
# A face was tracked the whole time and never left. The camera is
# aimed correctly and the old advice ("check it is pointing at the
# walkway") would send an installer to move a camera looking
# straight at them — which is how a good camera gets made bad.
return {"verdict": "no_completed_passes",
"headline": "a face was in view, but nobody walked past",
"advice": [
"The camera is detecting a face, so it is pointed "
"correctly — but no one completed a pass.",
"This check scores the best view of each person as "
"they leave the frame, which is what recognition "
"actually uses, so standing still measures nothing.",
"Walk through the frame and out of it, a few times, "
"then run the check again."]}
if n == 0:
# Streaming but nothing detected. Distinguishing this from "placed
# badly" matters: the fix is completely different.
return {"verdict": "no_faces",
"headline": "no faces detected",
"advice": [
"The camera is streaming but saw no face at all.",
"Check it is pointing at the walkway, not the ceiling "
"or floor, and that someone walked through the frame.",
"If people did walk past, the view is too far, too "
"dark, or too steep for the detector."]}
below = sum(1 for q in ordered if q < self.gate) / n
p50 = quantile(ordered, 0.50)
spread = quantile(ordered, 0.95) - quantile(ordered, 0.05)
if n >= FLAT_MIN_SAMPLES and spread < FLAT_SPREAD:
# Every detection scoring the same is not a camera problem to
# solve by moving it - it is not seeing people at all.
return {"verdict": "artifact",
"headline": f"every detection scored {p50:.2f} — this is "
"probably not a face",
"advice": [
"A constant score across every detection is the "
"signature of a static object, not a person.",
"Glass, a poster, a reflection or a mannequin in view "
"will do this.",
"Point the camera away from it, or raise "
"detection.score_threshold for this camera."]}
if n < MIN_SAMPLES:
return {"verdict": "inconclusive",
"headline": f"only {n} face{'' if n == 1 else 's'} seen — "
"not enough to judge",
"advice": [
f"Median quality was {p50:.2f}, but {n} "
f"sample{'' if n == 1 else 's'} is anecdote, not "
"measurement.",
"Run the check again and walk past several times, "
"ideally with more than one person."]}
if below <= GOOD_BELOW_GATE:
return {"verdict": "good",
"headline": f"good placement — median quality {p50:.2f}",
"advice": [
f"{below:.0%} of faces fell below the {self.gate:.2f} "
"enrollment gate. This camera can enrol and recognise "
"people reliably."]}
if below <= POOR_BELOW_GATE:
return {"verdict": "marginal",
"headline": f"usable but weak — {below:.0%} of faces are "
"below the gate",
"advice": [
f"Median quality {p50:.2f} against a "
f"{self.gate:.2f} gate: roughly {below:.0%} of "
"visitors will be seen and then discarded.",
"Angling it to face the approach direction, or "
"lowering it toward head height, usually fixes this.",
"If the position cannot change, lower this camera's "
"min_enroll_quality — but only this camera's."]}
return {"verdict": "poor",
"headline": f"poor placement — {below:.0%} of faces are below "
"the gate",
"advice": [
f"Median quality {p50:.2f} against a {self.gate:.2f} "
"gate. Most people who walk past will not be enrolled or "
"recognised, and nothing will look broken.",
"Move the camera to roughly head height, facing the "
"direction people approach from.",
"An overhead camera tilts every face downward, which is "
"the single most common cause of this result.",
"Backlighting — a window or lit glass behind the "
"subject — is the second most common.",
"Re-run this check after moving it. Do not lower the "
"quality gate to make this message go away: it converts a "
"visible miss into an invisible wrong match."]}
def _spread(ordered: "list[float]", gate: float) -> dict:
out = {"n": len(ordered)}
if not ordered:
return out
out["p05"] = round(quantile(ordered, 0.05), 3)
out["p50"] = round(quantile(ordered, 0.50), 3)
out["p95"] = round(quantile(ordered, 0.95), 3)
out["fraction_below_gate"] = round(
sum(1 for v in ordered if v < gate) / len(ordered), 3)
return out

316
behavision/config.py Normal file
View File

@@ -0,0 +1,316 @@
"""Typed configuration loaded from YAML with ${ENV} expansion.
Secrets never live in YAML: the YAML references environment variables
(populated from `.env`), so the config file is safe to commit.
"""
from __future__ import annotations
import os
import re
from pathlib import Path
from typing import Optional
from urllib.parse import quote
import yaml
from pydantic import BaseModel, Field, model_validator
_ENV_RE = re.compile(r"\$\{([A-Za-z_][A-Za-z0-9_]*)\}")
def _expand_env(text: str) -> str:
return _ENV_RE.sub(lambda m: os.environ.get(m.group(1), ""), text)
class CameraTuning(BaseModel):
"""Per-camera overrides for the recognition gates. None = use the global.
These are per-camera because they describe a *view*, not a preference: a
gate measured on an entrance camera at head height does not describe an
overhead corridor camera, and a real deployment has both at one site.
Measured on Office1: genuine faces score 0.32-0.45 there against a global
gate of 0.65, so every visitor was discarded — while the same gate is
correct for a frontal camera where real faces score 0.70-0.82.
Note the asymmetry before overriding the similarity thresholds. Quality is
purely local — it only asks whether THIS view is good enough to store.
match/enroll are not: every camera writes into one shared gallery, so a
camera set loose can merge two people into an identity that a stricter
camera then trusts. Loosen quality per camera freely; loosen match only
with measured cross-person data from that camera.
"""
min_enroll_quality: Optional[float] = None
match_threshold: Optional[float] = None
enroll_threshold: Optional[float] = None
class CameraConfig(BaseModel):
id: str
url: str = ""
host: str = ""
port: int = 554
path: str = "/"
username: str = ""
password: str = ""
webcam: Optional[int] = None
max_width: int = 1280 # frames wider than this are downscaled at ingest
tuning: CameraTuning = CameraTuning()
def source(self) -> "str | int":
"""Resolved capture source: webcam index, explicit URL, or a URL
built from parts with percent-encoded credentials."""
if self.webcam is not None:
return self.webcam
if self.url:
return self.url
if not self.host:
raise ValueError(f"camera '{self.id}': set url, host or webcam")
auth = ""
if self.username:
auth = quote(self.username, safe="")
if self.password:
auth += ":" + quote(self.password, safe="")
auth += "@"
path = self.path if self.path.startswith("/") else "/" + self.path
return f"rtsp://{auth}{self.host}:{self.port}{path}"
def safe_url(self) -> str:
"""Loggable form with the password masked."""
src = self.source()
if isinstance(src, int):
return f"webcam:{src}"
return re.sub(r"(rtsp://[^:/@]+:)[^@]*@", r"\1*****@", src)
class AppSection(BaseModel):
data_dir: Path = Path("data")
models_dir: Path = Path("models")
log_level: str = "INFO"
# Save every aligned chip the encoder sees to data/debug/ — diagnostic
# only, off in normal operation (it writes face images to disk).
debug_faces: bool = False
# Write one face image per resolved visit to data/outbox/ for the agent to
# upload. OFF by default, and that default is the product's privacy
# position rather than an oversight: with it off this machine holds
# templates and timestamps and nothing resembling a photograph. Turning it
# on changes what the system is under GDPR and India's DPDP, so it has to
# be a decision somebody makes rather than one they inherit.
store_faces: bool = False
class ApiSection(BaseModel):
host: str = "0.0.0.0"
port: int = 8010
# HTTP Basic credentials. Blank + loopback host = open (unreachable from
# off-box anyway); blank + routable host = generated, see
# ensure_api_credentials(). Never hardcode these — they come from .env.
username: str = ""
password: str = ""
@model_validator(mode="before")
@classmethod
def _normalize_blanks(cls, values):
# Unset ${ENV} placeholders parse as YAML null — treat as "".
if isinstance(values, dict):
values = {k: ("" if v is None else v) for k, v in values.items()}
if values.get("port") == "":
values["port"] = 8010
return values
@property
def auth_enabled(self) -> bool:
return bool(self.username and self.password)
@property
def is_loopback(self) -> bool:
return self.host in ("127.0.0.1", "::1", "localhost", "")
class DetectionSection(BaseModel):
# Measured on the deployment site: frosted-glass false positives pass
# 0.75, real faces score higher. Keep in step with config/default.yaml.
score_threshold: float = 0.82
nms_threshold: float = 0.3
min_face_px: int = 48
max_faces: int = 20
class RecognitionSection(BaseModel):
model_file: str = "" # pin a specific model filename; empty = auto
# Override the channel order the encoder feeds the model. Empty =
# inferred from the model family (ArcFace RGB, AdaFace BGR).
color_order: str = ""
match_threshold: float = 0.42
enroll_threshold: float = 0.32
reinforce_threshold: float = 0.55
max_embeddings_per_identity: int = 5
auto_enroll: bool = True
min_enroll_quality: float = 0.65 # real frontal faces 0.70-0.82, glass blurs <=0.54
sighting_cooldown_seconds: float = 30.0
@model_validator(mode="after")
def _sane(self) -> "RecognitionSection":
if not (0 < self.enroll_threshold < self.match_threshold < 1):
raise ValueError("need 0 < enroll_threshold < match_threshold < 1")
return self
def merged(self, tuning: "CameraTuning | None") -> "RecognitionSection":
"""This section with one camera's overrides applied.
Returns a validated copy, so a per-camera pair that inverts
enroll/match is rejected here rather than silently driving decisions
that contradict each other.
"""
if tuning is None:
return self
overrides = {k: v for k, v in tuning.model_dump().items()
if v is not None}
if not overrides:
return self
return RecognitionSection.model_validate(
{**self.model_dump(), **overrides})
class TrackingSection(BaseModel):
iou_threshold: float = 0.3
max_misses: int = 25
min_hits_for_id: int = 4
min_embeddings_for_id: int = 3
min_quality_to_encode: float = 0.35
max_id_attempts: int = 8
# Ambiguous tracks keep accumulating embeddings every frame but
# only re-decide this often, so max_id_attempts spans seconds of
# genuinely different frames rather than one burst.
id_retry_interval_seconds: float = 0.5
# A resolved track keeps contributing views for the rest of the
# visit, so an identity does not stay stuck on the single embedding
# it was born with. Sampled this often; each view is still subject
# to the reinforce/quality/cap gates in Gallery.
reinforce_during_track: bool = True
reinforce_interval_seconds: float = 1.0
class AttributesSection(BaseModel):
enabled: bool = True
# Gate for collecting a per-frame age/gender/emotion sample. Deliberately
# NOT recognition.min_enroll_quality, which it used to borrow: that gate
# is 0.65 and guards minting a permanent identity, while genuine faces on
# an overhead camera measure 0.32-0.45. Sharing it meant no track ever
# collected the multiple samples the median is computed from, so the
# aggregate silently collapsed to a single frame — the exact instability
# the median was added to remove.
min_quality: float = 0.35
class EmailSection(BaseModel):
smtp_host: str = ""
smtp_port: int = 587
username: str = ""
password: str = ""
to: str = ""
min_interval_seconds: float = 300.0
@model_validator(mode="before")
@classmethod
def _normalize_blanks(cls, values):
# Unset ${ENV} placeholders parse as YAML null — treat as "".
if isinstance(values, dict):
values = {k: ("" if v is None else v) for k, v in values.items()}
if values.get("smtp_port") == "":
values["smtp_port"] = 587
return values
@property
def enabled(self) -> bool:
return bool(self.smtp_host and self.username and self.to)
class EventsSection(BaseModel):
webhook_url: str = ""
email: EmailSection = Field(default_factory=EmailSection)
@model_validator(mode="before")
@classmethod
def _normalize_blanks(cls, values):
if isinstance(values, dict) and values.get("webhook_url") is None:
values["webhook_url"] = ""
return values
class Config(BaseModel):
app: AppSection = Field(default_factory=AppSection)
api: ApiSection = Field(default_factory=ApiSection)
cameras: list[CameraConfig] = Field(default_factory=list)
detection: DetectionSection = Field(default_factory=DetectionSection)
recognition: RecognitionSection = Field(default_factory=RecognitionSection)
tracking: TrackingSection = Field(default_factory=TrackingSection)
attributes: AttributesSection = Field(default_factory=AttributesSection)
events: EventsSection = Field(default_factory=EventsSection)
def ensure_api_credentials(cfg: "Config") -> "tuple[bool, bool]":
"""Make sure a routable API is never served unauthenticated.
A live face feed plus a biometric gallery must not be readable by anyone
who can reach the port. But failing to boot mid-deployment is its own
outage, so instead of refusing to start we mint a credential, persist it
0600 under data/, and log it. Returns (auth_enabled, was_generated).
"""
import os
import secrets
if cfg.api.auth_enabled:
return True, False
if cfg.api.is_loopback:
return False, False # not reachable off-box; leave it open
cred_file = cfg.app.data_dir / "api_credentials.txt"
if cred_file.exists():
parsed = dict(
line.split("=", 1) for line in
cred_file.read_text(encoding="utf-8").splitlines() if "=" in line)
cfg.api.username = parsed.get("username", "").strip()
cfg.api.password = parsed.get("password", "").strip()
if cfg.api.auth_enabled:
return True, False
cfg.api.username = "behavision"
cfg.api.password = secrets.token_urlsafe(16)
cred_file.write_text(
f"username={cfg.api.username}\npassword={cfg.api.password}\n",
encoding="utf-8")
try:
os.chmod(cred_file, 0o600)
except OSError: # best effort (Windows)
pass
return True, True
def load_config(path: "Path | str | None" = None) -> Config:
"""Load .env, then YAML with ${ENV} expansion, into a validated Config.
Relative `data_dir` / `models_dir` resolve against the **state root**, not
the code: installed, the code lives under `Program Files` where nothing may
write, and the database, logs and downloaded models still have to go
somewhere that survives an upgrade. In a checkout the two are the same
directory, so development is unaffected.
"""
from dotenv import load_dotenv
from .paths import ensure_config, env_file, state_root
env = env_file()
if env is not None:
load_dotenv(env)
cfg_path = Path(path) if path else ensure_config()
raw = yaml.safe_load(_expand_env(cfg_path.read_text(encoding="utf-8"))) or {}
cfg = Config.model_validate(raw)
root = state_root()
for key in ("data_dir", "models_dir"):
p = getattr(cfg.app, key)
if not p.is_absolute():
setattr(cfg.app, key, root / p)
cfg.app.data_dir.mkdir(parents=True, exist_ok=True)
cfg.app.models_dir.mkdir(parents=True, exist_ok=True)
return cfg

65
behavision/detection.py Normal file
View File

@@ -0,0 +1,65 @@
"""Face detection with YuNet (OpenCV FaceDetectorYN).
Why YuNet: modern CNN detector with 5-point landmarks built into OpenCV —
no compilation, no extra runtime, works on Windows out of the box, and its
landmarks feed ArcFace alignment directly. Accuracy on frontal surveillance
footage is on par with SCRFD-500M at a fraction of the operational cost.
"""
from __future__ import annotations
import logging
from dataclasses import dataclass, field
from pathlib import Path
import cv2
import numpy as np
from .geometry import clip_box
log = logging.getLogger(__name__)
YUNET_FILENAME = "face_detection_yunet_2023mar.onnx"
@dataclass
class Detection:
box: tuple # x1, y1, x2, y2 (int, clipped to frame)
kps: np.ndarray # (5, 2) float32, full-frame coordinates
score: float
quality: float = 0.0
attributes: dict = field(default_factory=dict)
class FaceDetector:
def __init__(self, models_dir: Path, score_threshold: float = 0.75,
nms_threshold: float = 0.3, max_faces: int = 20,
min_face_px: int = 48):
model_path = Path(models_dir) / YUNET_FILENAME
if not model_path.exists():
raise FileNotFoundError(
f"{model_path} missing - run: python -m behavision setup-models")
self._det = cv2.FaceDetectorYN_create(
str(model_path), "", (320, 320), score_threshold, nms_threshold,
max_faces)
self._input_size: "tuple[int, int] | None" = None
self.min_face_px = min_face_px
def detect(self, frame: np.ndarray) -> "list[Detection]":
h, w = frame.shape[:2]
if self._input_size != (w, h):
self._det.setInputSize((w, h))
self._input_size = (w, h)
_, faces = self._det.detect(frame)
if faces is None:
return []
out: list[Detection] = []
for f in faces:
x, y, bw, bh = f[:4]
if min(bw, bh) < self.min_face_px:
continue
box = clip_box((x, y, x + bw, y + bh), w, h)
if box is None:
continue
kps = f[4:14].reshape(5, 2).astype(np.float32)
out.append(Detection(box=box, kps=kps, score=float(f[14])))
return out

601
behavision/engine.py Normal file
View File

@@ -0,0 +1,601 @@
"""Pipeline engine: one worker thread per camera, shared models and gallery.
Per frame: detect -> score quality -> update tracker. Identity is resolved
per TRACK (once a track has enough hits and a good-enough frame), never per
frame. Ambiguous matches retry on later, better frames up to a bounded
number of attempts.
"""
from __future__ import annotations
import collections
import logging
import threading
import time
from typing import Optional
import cv2
import numpy as np
from .attributes import AttributeEstimator, aggregate as aggregate_attrs
from .cameras import CameraStore
from .capture import VideoSource
from .faces import FaceOutbox
from .commission import CommissionRun
from .config import CameraConfig, Config
from .detection import FaceDetector
from .events import EmailSink, Event, EventBus, LogSink, WebhookSink
from .gallery import Gallery, IdentityStore, VectorIndex
from .geometry import align_face
from .recognition import EMBEDDING_DIM, ArcFaceEncoder, face_quality
from .tracking import IouTracker, Track
log = logging.getLogger(__name__)
_COLORS = {"known": (80, 200, 80), "new": (60, 160, 255),
"pending": (160, 160, 160), "ambiguous": (60, 120, 200)}
# Terminal outcomes that mean "a person was on camera and we failed to place
# them", as opposed to "a person walked through too fast to try".
_LOST_OUTCOMES = ("rejected_quality", "gave_up_ambiguous", "ended_ambiguous")
def _track_outcome(track: Track) -> str:
"""Classify a finished track. Exactly one label per track, decided once.
Order matters: a track that exhausted its attempts because every one was
refused for quality is a *quality* failure, and reporting it as "ambiguous"
would send anyone tuning the site to the match threshold instead of to the
camera mount.
"""
if track.state == "resolved":
return "enrolled" if track.is_new else "recognized"
if track.emb_count == 0:
return "no_embedding" # never held a frame worth encoding
if track.id_attempts == 0:
return "too_brief" # left before enough evidence accumulated
if track.quality_skips:
return "rejected_quality" # face seen, too poor to mint an identity
if track.state == "gave_up":
return "gave_up_ambiguous"
return "ended_ambiguous"
def _quantile(ordered: "list[float]", frac: float) -> float:
if not ordered:
return 0.0
idx = min(len(ordered) - 1, int(round(frac * (len(ordered) - 1))))
return ordered[idx]
def _spread(values: "list[float]", gate: "float | None" = None) -> dict:
ordered = sorted(values)
out = {"n": len(ordered)}
if not ordered:
return out
out["p05"] = round(_quantile(ordered, 0.05), 3)
out["p50"] = round(_quantile(ordered, 0.50), 3)
out["p95"] = round(_quantile(ordered, 0.95), 3)
if gate is not None:
out["fraction_below_gate"] = round(
sum(1 for v in ordered if v < gate) / len(ordered), 3)
return out
class PipelineStats:
"""Per-camera tally of what became of each track.
The pipeline used to be unfalsifiable from outside: the only numbers were
frames and faces, so "nobody visited" and "every visitor was refused by the
quality gate" produced identical output, and every diagnosis meant reading
SQLite by hand. Deciding whether a site's camera placement works needs the
rejection reasons and the quality spread, not the frame count.
Written by the worker thread, read by API threads, hence the lock. The
distribution windows are bounded so a camera running for weeks cannot grow
this without limit.
"""
WINDOW = 500
def __init__(self) -> None:
self._lock = threading.Lock()
self._outcomes: "collections.Counter[str]" = collections.Counter()
self._qualities: "collections.deque[float]" = collections.deque(
maxlen=self.WINDOW)
self._similarities: "collections.deque[float]" = collections.deque(
maxlen=self.WINDOW)
self.tracks_ended = 0
def record(self, track: Track, outcome: str) -> None:
with self._lock:
self.tracks_ended += 1
self._outcomes[outcome] += 1
if track.best_quality > 0:
self._qualities.append(track.best_quality)
# Only tracks that actually reached resolve() have a similarity;
# zero from the others would drag every percentile down.
if track.id_attempts:
self._similarities.append(track.similarity)
def snapshot(self, enroll_gate: "float | None" = None) -> dict:
with self._lock:
outcomes = dict(self._outcomes)
qualities = list(self._qualities)
similarities = list(self._similarities)
ended = self.tracks_ended
return {
"tracks_ended": ended,
"outcomes": outcomes,
# fraction_below_gate is the number that says whether the
# enrollment gate is set wrong for this camera.
"best_quality": _spread(qualities, enroll_gate),
"similarity": _spread(similarities),
}
class CameraWorker(threading.Thread):
def __init__(self, cam_cfg: CameraConfig, cfg: Config, detector: FaceDetector,
encoder: ArcFaceEncoder, gallery: Gallery, bus: EventBus,
attrs: Optional[AttributeEstimator]):
super().__init__(daemon=True, name=f"worker-{cam_cfg.id}")
self.cam_cfg = cam_cfg
self.cfg = cfg
self.detector = detector
self.encoder = encoder
self.gallery = gallery
self.bus = bus
self.attrs = attrs
# Gates describe a view, so they are resolved per camera: an overhead
# corridor and an entrance camera at head height cannot share a
# quality gate, and a site has both.
self.rcfg = cfg.recognition.merged(cam_cfg.tuning)
self.source = VideoSource(cam_cfg.id, cam_cfg.source(),
cam_cfg.safe_url(), cam_cfg.max_width)
self.tracker = IouTracker(cfg.tracking.iou_threshold,
cfg.tracking.max_misses)
# _stopping, NOT _stop. threading.Thread has its own private _stop(),
# and join() calls it: shadowing the name with an Event made every
# join() on a started worker raise "'Event' object is not callable".
# It only surfaces when a camera is removed or edited at runtime, so
# the engine answered 500 to every camera edit from head office while
# every test using a stubbed worker passed.
self._stopping = threading.Event()
self._lock = threading.Lock()
self._annotated_jpeg: Optional[bytes] = None
self._last_frame_ts = 0.0
self._was_connected = False
self.frames_processed = 0
self.faces_seen = 0
self.pipeline = PipelineStats()
# One outbox per worker, all writing into the same directory. Files are
# uuid-named so two cameras resolving a visit in the same millisecond
# cannot collide.
self.faces = FaceOutbox(cfg.app.data_dir, cfg.app.store_faces)
# Set while a placement check is running. The check reads the live
# pipeline rather than a probe of its own, so what it measures is
# exactly what production will see.
self.commission: Optional[CommissionRun] = None
# -- public ---------------------------------------------------------
def start(self) -> None:
self.source.start()
super().start()
def stop(self) -> None:
self._stopping.set()
self.source.stop()
def latest_jpeg(self) -> Optional[bytes]:
with self._lock:
return self._annotated_jpeg
def stats(self) -> dict:
return {
**self.source.stats(),
"frames_processed": self.frames_processed,
"faces_seen": self.faces_seen,
"active_tracks": len(self.tracker.tracks),
"pipeline": self.pipeline.snapshot(self.rcfg.min_enroll_quality),
"gates": {"min_enroll_quality": self.rcfg.min_enroll_quality,
"match_threshold": self.rcfg.match_threshold,
"enroll_threshold": self.rcfg.enroll_threshold},
}
# -- thread ---------------------------------------------------------
def run(self) -> None:
tcfg = self.cfg.tracking
while not self._stopping.is_set():
try:
self._emit_connection_events()
frame, ts = self.source.latest_since(self._last_frame_ts)
if frame is None:
time.sleep(0.02)
continue
self._last_frame_ts = ts
detections = self.detector.detect(frame)
for det in detections:
det.quality = face_quality(frame, det.box, det.kps)
active, ended = self.tracker.update(detections, ts)
self.faces_seen += sum(1 for t in active if t.hits == 1)
# A placement check needs to know a face is in view even while
# its track is still open: someone standing in front of their
# own camera to test it produces no finished tracks at all.
run = self.commission
if run is not None:
run.observe(len(active), ts)
for track in active:
if self._should_identify(track, tcfg, ts):
self._identify(track, frame, ts)
# Every track ends exactly once, so this is the one place a
# per-visit outcome can be tallied without double counting.
for track in ended:
self._finish_track(track, ts)
self._publish_annotated(frame, active)
self.frames_processed += 1
except Exception:
log.exception("[%s] frame processing failed", self.cam_cfg.id)
time.sleep(0.5)
log.info("[%s] worker stopped", self.cam_cfg.id)
# -- internals ------------------------------------------------------
def _emit_connection_events(self) -> None:
connected = self.source.connected
if connected != self._was_connected:
self._was_connected = connected
self.bus.publish(Event(
type="camera.up" if connected else "camera.down",
camera_id=self.cam_cfg.id))
def _finish_track(self, track: Track, ts: float) -> None:
"""Record what became of a track, once, as it ends.
Tracks that never reached an identity previously vanished without a
trace. For a footfall product that is a headcount which is wrong in a
way nobody can detect, and it is why a mis-set quality gate was
indistinguishable from an empty corridor.
"""
outcome = _track_outcome(track)
self.pipeline.record(track, outcome)
run = self.commission
if run is not None:
run.record(track.best_quality, ts)
if outcome not in _LOST_OUTCOMES:
return
# Only worth an event once the track held enough evidence to have been
# a real decision; a face glimpsed for two frames is noise, not a loss.
if track.emb_count < self.cfg.tracking.min_embeddings_for_id:
return
self.bus.publish(Event(
type="person.missed", camera_id=self.cam_cfg.id, ts=ts,
data={"reason": outcome,
"quality": round(track.best_quality, 3),
"similarity": round(track.similarity, 3),
"attempts": track.id_attempts,
"embeddings": track.emb_count}))
def _should_identify(self, track: Track, tcfg, ts: float) -> bool:
if track.state == "resolved":
# Known person, still on screen: keep sampling their other angles
# so the identity does not stay frozen on the one embedding it was
# created with. Gated hard on quality — a blurred frame teaches
# the gallery nothing useful.
if not tcfg.reinforce_during_track:
return False
if track.identity_id is None:
return False
if track.quality < tcfg.min_quality_to_encode:
return False
return ts - track.last_reinforce_ts >= tcfg.reinforce_interval_seconds
if track.state not in ("pending", "ambiguous"):
return False
if track.id_attempts >= tcfg.max_id_attempts:
track.state = "gave_up"
return False
# Wait for a frame worth encoding, but don't wait forever: after
# twice the warmup period, take whatever the track has.
if (track.quality < tcfg.min_quality_to_encode
and track.hits < tcfg.min_hits_for_id * 2):
return False
return True
def _identify(self, track: Track, frame: np.ndarray, ts: float) -> None:
"""Accumulate an embedding for this frame; decide identity only from
the mean of several frames. Single-frame ArcFace embeddings under
steep camera angles / motion blur differ so much that one walk-by
can look like several people — the average is stable."""
if track.state == "resolved":
self._reinforce(track, frame, ts)
return
chip = align_face(frame, track.kps, size=self.encoder.size)
# Keep the best-looking view for the customer record. Quality is
# already computed for the enrolment gate, so choosing on it costs
# nothing and picks the frame a person would have picked.
if self.faces.enabled and track.quality > track.best_face_quality:
crop = self.faces.crop(frame, track.box)
if crop is not None:
track.best_face, track.best_face_quality = crop, track.quality
if self.cfg.app.debug_faces:
debug_dir = self.cfg.app.data_dir / "debug"
debug_dir.mkdir(parents=True, exist_ok=True)
cv2.imwrite(str(debug_dir / (
f"{ts:.1f}_track{track.id}_q{track.quality:.2f}.jpg")), chip)
embedding = self.encoder.encode_chip(chip)
if embedding is None:
return
if track.emb_sum is None:
track.emb_sum = embedding.copy()
else:
track.emb_sum += embedding
track.emb_count += 1
# One more attribute sample per accumulated frame, capped. A single
# frame's age estimate swings by a decade; a few frames median out.
tcfg = self.cfg.tracking
if (self.attrs is not None
and len(track.attr_samples) < tcfg.min_embeddings_for_id
and track.quality >= self.cfg.attributes.min_quality):
track.attr_samples.append(
self.attrs.estimate(frame, track.box, chip))
if (track.emb_count < tcfg.min_embeddings_for_id
or track.hits < tcfg.min_hits_for_id):
return # keep collecting evidence
# An already-ambiguous track keeps accumulating above (that is what
# improves the mean) but only re-decides after a real interval —
# otherwise max_id_attempts is spent on consecutive frames of the
# same instant instead of on the "later, better frame" it promises.
if (track.state == "ambiguous"
and ts - track.last_attempt_ts < tcfg.id_retry_interval_seconds):
return
mean = track.emb_sum / track.emb_count
norm = float(np.linalg.norm(mean))
if norm < 1e-6:
return
mean = (mean / norm).astype(np.float32)
# Aggregated before resolve() so the sighting row carries the settled
# verdict, not whichever frame happened to be first.
if self.attrs is not None and not track.attributes:
if not track.attr_samples:
track.attr_samples.append(
self.attrs.estimate(frame, track.box, chip))
track.attributes = aggregate_attrs(track.attr_samples)
track.id_attempts += 1
track.last_attempt_ts = ts
res = self.gallery.resolve(mean, track.best_quality,
self.cam_cfg.id, ts,
attributes=track.attributes or None,
rcfg=self.rcfg)
track.similarity = res.similarity
if res.kind in ("known", "new"):
track.state = "resolved"
track.is_new = res.kind == "new"
track.identity_id = res.identity_id
track.label = res.label
if res.new_sighting:
# Written once, at the moment the visit becomes real. Writing
# per frame would fill the outbox with images of visits that
# never resolved into anything.
image_path = self.faces.save(track.best_face)
track.best_face = None # let the array go; the file has it now
self.bus.publish(Event(
type="person.new" if res.kind == "new" else "person.seen",
camera_id=self.cam_cfg.id, ts=ts,
data={"identity_id": res.identity_id, "label": res.label,
"similarity": round(res.similarity, 3),
# A local file for the agent to upload and delete.
# The engine does not upload: a shop PC must never
# hold object-storage credentials.
**({"image_path": image_path} if image_path else {}),
# The gate ran on best_quality; reporting this
# frame's quality made events look like they had
# passed a threshold they were below.
"quality": round(track.best_quality, 3),
"frame_quality": round(track.quality, 3),
**track.attributes}))
elif res.kind == "ambiguous":
track.state = "ambiguous" # retried on a later, better frame
elif res.kind == "skipped":
# resolve() declined to mint an identity — in practice always
# because best_quality is under min_enroll_quality. This branch
# did not exist: the verdict fell through, the track stayed
# "pending", and the visitor was dropped with no event, no counter
# and no log line. Marking it ambiguous also buys the retry
# throttle, so the remaining attempts are spent on genuinely later
# frames instead of being burnt in one burst on the same instant.
track.quality_skips += 1
track.state = "ambiguous"
def _reinforce(self, track: Track, frame: np.ndarray, ts: float) -> None:
"""Feed one more view of an already-identified person to the gallery."""
track.last_reinforce_ts = ts
chip = align_face(frame, track.kps, size=self.encoder.size)
embedding = self.encoder.encode_chip(chip)
if embedding is None:
return
if self.gallery.reinforce_identity(track.identity_id, embedding,
track.quality, rcfg=self.rcfg):
track.reinforcements += 1
def _publish_annotated(self, frame: np.ndarray, tracks: "list[Track]") -> None:
canvas = frame.copy()
for t in tracks:
if t.misses > 0:
continue # only draw tracks matched in this frame
x1, y1, x2, y2 = t.box
if t.state == "resolved":
color = _COLORS["known"] if t.label and not str(t.label).startswith(
"Visitor") else _COLORS["new"]
text = f"{t.label} ({t.similarity:.2f})"
elif t.state == "ambiguous":
color, text = _COLORS["ambiguous"], "?"
else:
color, text = _COLORS["pending"], ""
cv2.rectangle(canvas, (x1, y1), (x2, y2), color, 2)
if text:
cv2.putText(canvas, text, (x1, max(20, y1 - 8)),
cv2.FONT_HERSHEY_SIMPLEX, 0.55, color, 2)
ok, buf = cv2.imencode(".jpg", canvas,
[int(cv2.IMWRITE_JPEG_QUALITY), 80])
if ok:
with self._lock:
self._annotated_jpeg = buf.tobytes()
class Engine:
"""Owns all shared components and one CameraWorker per camera."""
def __init__(self, cfg: Config):
self.cfg = cfg
self.bus = EventBus()
self.bus.add_sink(LogSink())
if cfg.events.webhook_url:
self.bus.add_sink(WebhookSink(cfg.events.webhook_url))
email = cfg.events.email
if email.enabled:
self.bus.add_sink(EmailSink(
email.smtp_host, email.smtp_port, email.username,
email.password, email.to, email.min_interval_seconds))
# One detector PER CAMERA. cv2.FaceDetectorYN carries mutable state
# (setInputSize + the cached input size) and is not thread-safe, so a
# shared instance races as soon as a second camera worker runs — and
# corrupts inference outright if the two streams differ in resolution.
# The YuNet model is ~230 KB, so per-worker copies are essentially free
# and avoid serialising the hottest per-frame call behind a lock.
self.detectors: "dict[str, FaceDetector]" = {}
# Shared deliberately: onnxruntime InferenceSession.run is thread-safe
# and the encoder weights are worth sharing (13-260 MB).
self.encoder = ArcFaceEncoder(cfg.app.models_dir,
cfg.recognition.model_file,
cfg.recognition.color_order)
self.store = IdentityStore(cfg.app.data_dir / "behavision.db")
self.gallery = Gallery(self.store, VectorIndex(EMBEDDING_DIM),
cfg.recognition, self.encoder.model_name)
self.attributes = None
if cfg.attributes.enabled:
est = AttributeEstimator(cfg.app.models_dir)
self.attributes = est if est.any_loaded else None
# Cameras are added and removed at runtime from the API, so this dict
# is mutated by request threads while the worker loop and stats() read
# it. RLock because add_camera/remove_camera call each other via
# restart_camera.
self._lock = threading.RLock()
self.workers: "dict[str, CameraWorker]" = {}
self.started_at: Optional[float] = None
self._running = False
# YAML seeds the store on first run; after that the store is
# authoritative, or a camera deleted in the UI would come back on the
# next restart.
self.camera_store = CameraStore(cfg.app.data_dir / "cameras.json")
self.camera_store.seed(cfg.cameras)
for cam in self.camera_store.list():
self._build_worker(cam)
# -- camera lifecycle -----------------------------------------------
def _build_worker(self, cam_cfg: CameraConfig) -> "CameraWorker":
"""Construct (but do not start) a worker and its own detector."""
det = self.cfg.detection
detector = FaceDetector(self.cfg.app.models_dir, det.score_threshold,
det.nms_threshold, det.max_faces,
det.min_face_px)
worker = CameraWorker(cam_cfg, self.cfg, detector, self.encoder,
self.gallery, self.bus, self.attributes)
self.detectors[cam_cfg.id] = detector
self.workers[cam_cfg.id] = worker
return worker
def add_camera(self, cam_cfg: CameraConfig) -> "CameraWorker":
"""Attach a camera to a live engine. Raises if the id is taken."""
with self._lock:
if cam_cfg.id in self.workers:
raise ValueError(f"camera '{cam_cfg.id}' is already running")
worker = self._build_worker(cam_cfg)
if self._running:
worker.start()
log.info("camera '%s' added (%s)", cam_cfg.id, cam_cfg.safe_url())
return worker
def remove_camera(self, camera_id: str) -> bool:
with self._lock:
worker = self.workers.pop(camera_id, None)
self.detectors.pop(camera_id, None)
if worker is None:
return False
# Outside the lock: join() can take seconds and must not block the
# frame loop's stats() calls or another camera being added.
worker.stop()
if worker.is_alive():
worker.join(timeout=5)
log.info("camera '%s' removed", camera_id)
return True
def restart_camera(self, cam_cfg: CameraConfig) -> "CameraWorker":
"""Apply an edited URL/credential. CameraWorker is a Thread, and a
stopped Thread cannot be restarted, so this must build a new one."""
with self._lock:
self.remove_camera(cam_cfg.id)
return self.add_camera(cam_cfg)
# -- lifecycle ------------------------------------------------------
def start(self) -> None:
self.bus.start()
with self._lock:
self._running = True
workers = list(self.workers.values())
for worker in workers:
worker.start()
self.started_at = time.time()
log.info("engine started with %d camera(s)", len(workers))
def stop(self) -> None:
with self._lock:
self._running = False
workers = list(self.workers.values())
for worker in workers:
worker.stop()
for worker in workers:
if worker.is_alive():
worker.join(timeout=5)
self.bus.stop()
self.store.close()
log.info("engine stopped")
def stats(self) -> dict:
return {
"uptime_s": round(time.time() - self.started_at, 1)
if self.started_at else 0,
# Which encoder actually won the fallback chain. On a
# memory-constrained box the big model can silently lose to the
# 13 MB one, and every stored embedding is tagged with whichever
# loaded — so this is the first thing to check after a deploy.
"recognition": {
"model": self.encoder.model_name,
"color_order": self.encoder.color_order,
"input_size": self.encoder.size,
},
"attributes": {
"enabled": self.attributes is not None,
"age_model": ("genderage" if self.attributes is not None
and self.attributes.has_genderage else "caffe/none"),
},
"gallery": self.store.stats(),
"cameras": [w.stats() for w in self.snapshot_workers()],
}
def snapshot_workers(self) -> "list[CameraWorker]":
"""Point-in-time copy — callers must never iterate self.workers
directly now that cameras come and go from request threads."""
with self._lock:
return list(self.workers.values())

122
behavision/events.py Normal file
View File

@@ -0,0 +1,122 @@
"""Async event bus with pluggable sinks (log, webhook, email).
Events are published from the pipeline thread and delivered on a dedicated
worker thread, so a slow webhook or SMTP server can never stall frame
processing. Sink failures are logged, never raised.
"""
from __future__ import annotations
import logging
import queue
import smtplib
import threading
import time
from collections import deque
from dataclasses import asdict, dataclass, field
from email.mime.text import MIMEText
log = logging.getLogger(__name__)
@dataclass
class Event:
type: str # person.new | person.seen | camera.up | camera.down | ...
camera_id: str
ts: float = field(default_factory=time.time)
data: dict = field(default_factory=dict)
def to_dict(self) -> dict:
return asdict(self)
class EventBus:
def __init__(self) -> None:
self._queue: "queue.Queue[Event | None]" = queue.Queue(maxsize=1000)
self._sinks: list = []
self.recent: deque = deque(maxlen=300)
self._worker = threading.Thread(
target=self._run, daemon=True, name="event-bus")
self._started = False
def add_sink(self, sink) -> None:
self._sinks.append(sink)
def start(self) -> None:
if not self._started:
self._started = True
self._worker.start()
def stop(self) -> None:
if self._started:
self._queue.put(None)
self._worker.join(timeout=5)
def publish(self, event: Event) -> None:
self.recent.appendleft(event.to_dict())
try:
self._queue.put_nowait(event)
except queue.Full:
log.warning("event queue full, dropping %s", event.type)
def _run(self) -> None:
while True:
event = self._queue.get()
if event is None:
return
for sink in self._sinks:
try:
sink.handle(event)
except Exception:
log.exception("sink %s failed for %s",
type(sink).__name__, event.type)
class LogSink:
def handle(self, event: Event) -> None:
log.info("event %s [%s] %s", event.type, event.camera_id, event.data)
class WebhookSink:
def __init__(self, url: str, timeout: float = 5.0):
self.url = url
self.timeout = timeout
def handle(self, event: Event) -> None:
import requests
requests.post(self.url, json=event.to_dict(), timeout=self.timeout)
class EmailSink:
"""Rate-limited email notifications for person events only."""
NOTIFY_TYPES = {"person.new", "person.seen"}
def __init__(self, smtp_host: str, smtp_port: int, username: str,
password: str, to: str, min_interval: float = 300.0):
self.smtp_host = smtp_host
self.smtp_port = smtp_port
self.username = username
self.password = password
self.to = to
self.min_interval = min_interval
self._last_sent = 0.0
def handle(self, event: Event) -> None:
if event.type not in self.NOTIFY_TYPES:
return
now = time.time()
if now - self._last_sent < self.min_interval:
return
self._last_sent = now
label = event.data.get("label", "someone")
body = (f"Behavision: {label} detected on camera {event.camera_id}\n"
f"Event: {event.type}\nDetails: {event.data}")
msg = MIMEText(body)
msg["Subject"] = f"Behavision: {label} on {event.camera_id}"
msg["From"] = self.username
msg["To"] = self.to
with smtplib.SMTP(self.smtp_host, self.smtp_port, timeout=10) as smtp:
smtp.starttls()
smtp.login(self.username, self.password)
smtp.send_message(msg)

142
behavision/faces.py Normal file
View File

@@ -0,0 +1,142 @@
"""Saving a face image for one visit — the outbox the agent uploads from.
This is the one place the engine writes a picture of a person to disk, and it
is off unless `app.store_faces` is set. That default is the product's original
privacy position, not an oversight: with images off, `data/behavision.db` holds
templates and timestamps and nothing that looks like a photograph. Turning them
on changes what the system is under GDPR and India's DPDP, so it is a decision
someone has to make on purpose.
The engine does NOT upload. It writes a file and names it on the event; the
agent uploads through a short-lived URL the server mints. A shop PC therefore
never holds object-storage credentials — the bucket is shared and a counter-top
machine is the least trustworthy thing in the estate.
"""
from __future__ import annotations
import logging
import time
import uuid
from pathlib import Path
import cv2
import numpy as np
log = logging.getLogger(__name__)
# A loose crop, not the aligned 112x112 chip.
#
# The chip is built for ArcFace: tight, warped to canonical landmarks, and
# nearly useless to a human trying to recognise a customer. This is the frame a
# person looks at, so it gets the same 1.5x head crop the attribute models use.
CROP_SCALE = 1.5
# Enough to see a face on a dashboard, small enough that a shop on ADSL can
# upload one per visitor without the queue backing up. ~15-25 KB at q80.
MAX_EDGE = 320
JPEG_QUALITY = 80
def _loose_crop(frame: np.ndarray, box, scale: float = CROP_SCALE) -> np.ndarray:
"""Square crop around the head, replicate-padded when it runs off-frame."""
x1, y1, x2, y2 = (float(v) for v in box)
cx, cy = (x1 + x2) / 2.0, (y1 + y2) / 2.0
half = max(x2 - x1, y2 - y1) * scale / 2.0
left, top = int(round(cx - half)), int(round(cy - half))
right, bottom = int(round(cx + half)), int(round(cy + half))
h, w = frame.shape[:2]
pad_l, pad_t = max(0, -left), max(0, -top)
pad_r, pad_b = max(0, right - w), max(0, bottom - h)
crop = frame[max(0, top):min(h, bottom), max(0, left):min(w, right)]
if crop.size == 0:
return frame
if pad_l or pad_t or pad_r or pad_b:
crop = cv2.copyMakeBorder(crop, pad_t, pad_b, pad_l, pad_r,
cv2.BORDER_REPLICATE)
return crop
class FaceOutbox:
"""Writes one JPEG per resolved visit for the agent to collect.
Files land in `data_dir/outbox`, which is deliberately NOT inside the
database directory: it is transient, the agent deletes each file after a
successful upload, and a backup of the database must not quietly start
including face images.
"""
def __init__(self, data_dir: Path, enabled: bool,
max_files: int = 500) -> None:
self.enabled = enabled
self.dir = Path(data_dir) / "outbox"
# Bounded. If the agent stops collecting — not running, no credentials,
# server unreachable for a week — this must not fill a shop's disk with
# pictures of its customers. Dropping the oldest is right: a stale
# photo of a visit already reported is the least valuable thing here.
self.max_files = max_files
if enabled:
self.dir.mkdir(parents=True, exist_ok=True)
log.warning(
"app.store_faces is ON: face images are being written to %s. "
"This changes what this machine holds under GDPR/DPDP.",
self.dir)
def crop(self, frame: np.ndarray, box) -> "np.ndarray | None":
"""The candidate image for one frame, downscaled and nothing else.
Kept as an array rather than encoded here because this runs on every
frame of every track: JPEG encoding per frame is milliseconds spent to
throw away all but the last one. At 320 px a crop is ~300 KB, so one
per live track is affordable even on the 16 GB box that already OOMs on
a 250 MB model — holding whole 1280x720 frames instead would not be.
"""
if not self.enabled:
return None
try:
crop = _loose_crop(frame, box)
h, w = crop.shape[:2]
if min(h, w) <= 0:
return None
if max(h, w) > MAX_EDGE:
s = MAX_EDGE / float(max(h, w))
crop = cv2.resize(crop, (max(1, int(w * s)), max(1, int(h * s))),
interpolation=cv2.INTER_AREA)
# A copy, because the slice from _loose_crop can be a view onto the
# capture buffer, which the capture thread overwrites in place.
return np.ascontiguousarray(crop)
except Exception:
log.exception("could not build a face crop")
return None
def save(self, crop: "np.ndarray | None") -> "str | None":
"""Write the crop and return its path, or None if images are off."""
if not self.enabled or crop is None:
return None
try:
path = self.dir / f"{time.time():.3f}_{uuid.uuid4().hex}.jpg"
ok, buf = cv2.imencode(".jpg", crop,
[int(cv2.IMWRITE_JPEG_QUALITY), JPEG_QUALITY])
if not ok:
return None
# Write-then-rename. The agent watches this directory, and a
# partially written JPEG that it picks up mid-write is an upload of
# a corrupt file that nothing will ever correct.
tmp = path.with_suffix(".part")
tmp.write_bytes(buf.tobytes())
tmp.replace(path)
self._trim()
return str(path)
except Exception:
# Never take the recognition loop down over a photo. A missing
# image is a cosmetic loss; a stalled worker is the product.
log.exception("could not write a face image")
return None
def _trim(self) -> None:
try:
files = sorted(self.dir.glob("*.jpg"), key=lambda p: p.stat().st_mtime)
for stale in files[:-self.max_files]:
stale.unlink(missing_ok=True)
except Exception:
log.debug("outbox trim failed", exc_info=True)

View File

@@ -0,0 +1,3 @@
from .service import Gallery, Resolution # noqa: F401
from .store import IdentityStore # noqa: F401
from .index import VectorIndex # noqa: F401

View File

@@ -0,0 +1,83 @@
"""Cosine-similarity vector index.
FAISS `IndexFlatIP` wrapped in `IndexIDMap2` when faiss is installed, plain
numpy otherwise — same interface, same results. Choices that fix the old
codebase's failure modes:
- Exact inner-product search (vectors are unit-norm, so IP == cosine).
No IVF: nothing to train, no wrong-metric trap, and exact search is
microseconds up to hundreds of thousands of vectors.
- `-1` ids from an empty index are filtered, never used as list indices.
- The index is rebuilt from SQLite at startup (SQLite is the source of
truth), so index and metadata can never drift apart.
"""
from __future__ import annotations
import logging
import numpy as np
log = logging.getLogger(__name__)
try:
import faiss # type: ignore
_HAVE_FAISS = True
except ImportError: # pragma: no cover - environment dependent
faiss = None
_HAVE_FAISS = False
class VectorIndex:
def __init__(self, dim: int):
self.dim = dim
if _HAVE_FAISS:
self._index = faiss.IndexIDMap2(faiss.IndexFlatIP(dim))
self._ids = None
self._vecs = None
else:
log.warning("faiss not installed - using exact numpy search "
"(identical results, slower at large scale)")
self._index = None
self._ids = np.empty((0,), dtype=np.int64)
self._vecs = np.empty((0, dim), dtype=np.float32)
def __len__(self) -> int:
if self._index is not None:
return self._index.ntotal
return len(self._ids)
def add(self, ids: "list[int]", vectors: np.ndarray) -> None:
if len(ids) == 0:
return
vectors = np.ascontiguousarray(vectors, dtype=np.float32).reshape(len(ids), self.dim)
id_arr = np.asarray(ids, dtype=np.int64)
if self._index is not None:
self._index.add_with_ids(vectors, id_arr)
else:
self._ids = np.concatenate([self._ids, id_arr])
self._vecs = np.vstack([self._vecs, vectors])
def remove(self, ids: "list[int]") -> None:
if len(ids) == 0:
return
id_arr = np.asarray(ids, dtype=np.int64)
if self._index is not None:
self._index.remove_ids(id_arr)
else:
keep = ~np.isin(self._ids, id_arr)
self._ids = self._ids[keep]
self._vecs = self._vecs[keep]
def search(self, vector: np.ndarray, k: int = 1) -> "list[tuple[int, float]]":
"""Top-k (embedding_id, cosine_similarity), best first."""
if len(self) == 0:
return []
q = np.ascontiguousarray(vector, dtype=np.float32).reshape(1, self.dim)
k = min(k, len(self))
if self._index is not None:
scores, ids = self._index.search(q, k)
return [(int(i), float(s))
for i, s in zip(ids[0], scores[0]) if i != -1]
sims = self._vecs @ q[0]
order = np.argsort(-sims)[:k]
return [(int(self._ids[i]), float(sims[i])) for i in order]

View File

@@ -0,0 +1,327 @@
"""Identity resolution: match, reinforce, or auto-enroll — with hysteresis.
Three-zone decision instead of one threshold:
similarity >= match_threshold -> same person
similarity < enroll_threshold -> genuinely new person
in between -> ambiguous: do NOTHING
The ambiguous zone is what prevents both duplicate identities and wrong
merges — the two failure modes the previous system had simultaneously.
"""
from __future__ import annotations
import logging
import threading
import time
from dataclasses import dataclass
from typing import Optional
import numpy as np
from ..config import RecognitionSection
from .index import VectorIndex
from .store import IdentityStore
log = logging.getLogger(__name__)
@dataclass
class Resolution:
kind: str # known | new | ambiguous | skipped
identity_id: Optional[int] = None
label: Optional[str] = None
similarity: float = 0.0
new_sighting: bool = False
class Gallery:
"""One gallery shared by every camera.
`cfg` here is the global recognition section — the default. Callers that
belong to a camera pass that camera's merged section as `rcfg`, because
the gates describe a view and cameras do not share one.
"""
def __init__(self, store: IdentityStore, index: VectorIndex,
cfg: RecognitionSection, model_name: str = "default"):
self.store = store
self.index = index
self.cfg = cfg
self.model_name = model_name
self._lock = threading.Lock()
self._last_sighting: dict[tuple[int, str], float] = {}
# Only embeddings produced by the active encoder enter the index;
# vectors from a different model are numerically incompatible.
ids, vecs = store.all_embeddings(index.dim, model=model_name)
index.add(ids, vecs)
log.info("gallery ready: %d embeddings (model '%s') across %d "
"identities", len(ids), model_name,
store.stats()["identities"])
def resolve(self, embedding: np.ndarray, quality: float, camera_id: str,
ts: "float | None" = None,
attributes: "dict | None" = None,
rcfg: "RecognitionSection | None" = None) -> Resolution:
"""`rcfg` is the calling camera's merged thresholds; the gallery is
shared across cameras but the gates that decide a view are not."""
ts = ts or time.time()
cfg = rcfg or self.cfg
with self._lock:
matches = self.index.search(embedding, k=1)
top_id, top_sim = matches[0] if matches else (None, -1.0)
if top_id is not None and top_sim >= cfg.match_threshold:
ident = self.store.identity_for_embedding(top_id)
if ident is None: # index/store race — treat as ambiguous
return Resolution(kind="ambiguous", similarity=top_sim)
self._maybe_reinforce(ident["id"], embedding, quality,
top_sim, cfg)
fresh = self._record_sighting(
ident["id"], camera_id, ts, top_sim, quality, attributes)
return Resolution(kind="known", identity_id=ident["id"],
label=ident["label"], similarity=top_sim,
new_sighting=fresh)
if top_id is None or top_sim < cfg.enroll_threshold:
if not cfg.auto_enroll:
return Resolution(kind="skipped", similarity=top_sim)
if quality < cfg.min_enroll_quality:
# Not confident enough in this face to mint an identity.
return Resolution(kind="skipped", similarity=top_sim)
identity_id, label = self.store.create_auto_identity()
emb_id = self.store.add_embedding(identity_id, embedding,
quality, self.model_name)
self.index.add([emb_id], embedding.reshape(1, -1))
self._record_sighting(identity_id, camera_id, ts, 1.0, quality,
attributes)
log.info("auto-enrolled %s (quality %.2f)", label, quality)
return Resolution(kind="new", identity_id=identity_id,
label=label, similarity=top_sim,
new_sighting=True)
return Resolution(kind="ambiguous", similarity=top_sim)
def enroll(self, label: str, embeddings: "list[np.ndarray]",
quality: float = 1.0) -> int:
"""Explicit enrollment (CLI / API) with a known name."""
with self._lock:
identity_id = self.store.create_identity(label, kind="enrolled")
for emb in embeddings[: self.cfg.max_embeddings_per_identity]:
emb_id = self.store.add_embedding(identity_id, emb, quality,
self.model_name)
self.index.add([emb_id], emb.reshape(1, -1))
return identity_id
def reinforce_identity(self, identity_id: int, embedding: np.ndarray,
quality: float,
rcfg: "RecognitionSection | None" = None) -> bool:
"""Add another view of an ALREADY-identified person.
A track is resolved once and then stops contributing, so an identity
was born holding a single embedding from the first second of a visit —
and the next encounter at a different angle had one reference vector to
beat. This lets the rest of the visit fill the gallery out.
Guarded three ways: the view must still map to *this* identity (a
track that drifted onto another face must not poison the gallery), it
must be similar enough that we actually believe it is this person
(>= enroll_threshold), and different enough to be worth storing
(< reinforce_threshold).
"""
cfg = rcfg or self.cfg
with self._lock:
if quality < cfg.min_enroll_quality:
return False
if (self.store.embedding_count(identity_id)
>= cfg.max_embeddings_per_identity):
return False
matches = self.index.search(embedding, k=1)
if not matches:
return False
top_id, top_sim = matches[0]
ident = self.store.identity_for_embedding(top_id)
if ident is None or ident["id"] != identity_id:
return False # looks more like someone else - do not store
if top_sim < cfg.enroll_threshold:
# Nearest neighbour is this identity, but only barely. Below
# enroll_threshold resolve() would call this a DIFFERENT
# person, so gluing it on here would contradict the decision
# the same numbers drive everywhere else. Measured on the
# overhead camera, unfloored reinforcement gave one identity
# two vectors 0.195 apart. The risk is asymmetric: a wrong
# face welded into an identity is unrecoverable, a missed
# hard angle is not.
return False
if top_sim >= cfg.reinforce_threshold:
return False # near-duplicate of what we already have
emb_id = self.store.add_embedding(identity_id, embedding, quality,
self.model_name)
self.index.add([emb_id], embedding.reshape(1, -1))
log.debug("reinforced identity %d (sim %.3f, quality %.2f)",
identity_id, top_sim, quality)
return True
def merge_identities(self, source_id: int, target_id: int,
force: bool = False) -> "dict":
"""Fold one identity into another — the repair for a person who was
enrolled twice.
Duplicates are not a hypothetical: two views of one face can score
below `match_threshold`, and when they do the system mints a second
identity and there is no way back. Deleting one loses that person's
history; leaving both means the same customer is greeted as new.
Merging is destructive and, unlike a duplicate, *unrecoverable* — two
different people welded together cannot be separated afterwards,
because nothing records which embedding came from whom. So the two
identities must look at least plausibly alike: below
`enroll_threshold` resolve() positively asserts they are different
people, and overriding that assertion requires `force`.
Returns a dict with `ok`; on refusal `reason` says why, so the UI can
offer the override instead of failing silently.
"""
cfg = self.cfg
with self._lock:
if source_id == target_id:
return {"ok": False, "reason": "cannot merge an identity "
"into itself"}
if self.store.get_identity(source_id) is None:
return {"ok": False, "reason": f"identity {source_id} not found"}
if self.store.get_identity(target_id) is None:
return {"ok": False, "reason": f"identity {target_id} not found"}
sim, checkable = self._identity_similarity(source_id, target_id)
if not force:
if not checkable:
return {"ok": False, "similarity": None,
"reason": "no comparable embeddings (different "
"encoder model) - cannot verify these "
"are the same person"}
if sim < cfg.enroll_threshold:
return {"ok": False, "similarity": round(sim, 3),
"threshold": cfg.enroll_threshold,
"reason": "these look like different people "
f"(best similarity {sim:.3f} < "
f"{cfg.enroll_threshold})"}
result = self.store.merge_identities(
source_id, target_id, cfg.max_embeddings_per_identity)
if result is None:
return {"ok": False, "reason": "identity not found"}
# Trimmed vectors must leave the index or it keeps answering with
# embedding ids that no longer exist in SQLite.
self.index.remove(result["dropped_embeddings"])
# The per-camera sighting cooldown is keyed by identity; the
# source's keys now point at an identity that is gone.
for key in [k for k in self._last_sighting if k[0] == source_id]:
self._last_sighting.pop(key, None)
log.warning("merged identity %d into %d (%s): %d embeddings, "
"%d sightings, similarity %s%s", source_id, target_id,
result["label"], result["embeddings_moved"],
result["sightings_moved"],
f"{sim:.3f}" if checkable else "n/a",
" [FORCED]" if force else "")
result.update(ok=True, forced=force,
similarity=round(sim, 3) if checkable else None)
return result
def duplicate_candidates(self, limit: int = 20, k: int = 6
) -> "list[dict]":
"""Identity pairs that look like the same person.
Found through the index rather than an all-pairs comparison: every
stored vector asks for its `k` nearest neighbours and any that belong
to a *different* identity is evidence those two are one person. That
is O(n*k) and needs no big matrix — an all-pairs float32 matrix over
10k embeddings is 400 MB, and this runs on a box that already OOMs on
a 250 MB model.
Only pairs at or above `enroll_threshold` are reported: below it the
gallery's own numbers say these are different people, and offering
that as a suggestion would invite exactly the merge that cannot be
undone.
"""
with self._lock:
owners = self.store.embedding_owners(self.model_name)
if not owners:
return []
ids, vecs = self.store.all_embeddings(self.index.dim,
model=self.model_name)
best: dict[tuple[int, int], float] = {}
for emb_id, vec in zip(ids, vecs):
mine = owners.get(emb_id)
if mine is None:
continue
for other_id, sim in self.index.search(vec, k=k):
theirs = owners.get(other_id)
if theirs is None or theirs == mine:
continue
if sim < self.cfg.enroll_threshold:
continue
pair = (min(mine, theirs), max(mine, theirs))
if sim > best.get(pair, -1.0):
best[pair] = float(sim)
out = []
for (a, b), sim in sorted(best.items(), key=lambda kv: -kv[1])[:limit]:
ia, ib = self.store.get_identity(a), self.store.get_identity(b)
if ia is None or ib is None:
continue
out.append({
"a": {"id": a, "label": ia["label"], "kind": ia["kind"],
"sighting_count": ia["sighting_count"]},
"b": {"id": b, "label": ib["label"], "kind": ib["kind"],
"sighting_count": ib["sighting_count"]},
"similarity": round(sim, 3),
"confident": sim >= self.cfg.match_threshold})
return out
def _identity_similarity(self, a: int, b: int) -> "tuple[float, bool]":
"""Best cosine similarity between any view of `a` and any view of `b`.
Best, not mean: two identities of one person exist precisely because
their *typical* views disagree. If any pair of views agrees, that is
the evidence they are the same person.
"""
_, va = self.store.identity_embeddings(a, self.index.dim,
self.model_name)
_, vb = self.store.identity_embeddings(b, self.index.dim,
self.model_name)
if len(va) == 0 or len(vb) == 0:
return 0.0, False
return float((va @ vb.T).max()), True
def delete_identity(self, identity_id: int) -> bool:
with self._lock:
removed = self.store.delete_identity(identity_id)
self.index.remove(removed)
return bool(removed)
# -- internals ------------------------------------------------------
def _maybe_reinforce(self, identity_id: int, embedding: np.ndarray,
quality: float, similarity: float,
cfg: RecognitionSection) -> None:
"""Add an extra embedding for a known person when this view is
confidently theirs but usefully different (pose/lighting), improving
recall over time without letting the identity drift."""
if similarity >= cfg.reinforce_threshold:
return # too similar to what we already have — adds nothing
if quality < cfg.min_enroll_quality:
return
if (self.store.embedding_count(identity_id)
>= cfg.max_embeddings_per_identity):
return
emb_id = self.store.add_embedding(identity_id, embedding, quality,
self.model_name)
self.index.add([emb_id], embedding.reshape(1, -1))
def _record_sighting(self, identity_id: int, camera_id: str, ts: float,
similarity: float, quality: float,
attributes: "dict | None" = None) -> bool:
key = (identity_id, camera_id)
last = self._last_sighting.get(key, 0.0)
if ts - last < self.cfg.sighting_cooldown_seconds:
return False
self._last_sighting[key] = ts
self.store.record_sighting(identity_id, camera_id, ts, similarity,
quality, attributes)
return True

338
behavision/gallery/store.py Normal file
View File

@@ -0,0 +1,338 @@
"""SQLite persistence for identities, embeddings and sightings.
Single writer class with an internal lock; WAL mode so the API can read
while the pipeline writes. Embeddings are stored as float32 BLOBs — SQLite
is the source of truth and the vector index is rebuilt from here at boot.
"""
from __future__ import annotations
import json
import sqlite3
import threading
import time
from pathlib import Path
import numpy as np
_SCHEMA = """
CREATE TABLE IF NOT EXISTS identities (
id INTEGER PRIMARY KEY AUTOINCREMENT,
label TEXT NOT NULL,
kind TEXT NOT NULL DEFAULT 'auto',
created_at REAL NOT NULL,
last_seen_at REAL,
sighting_count INTEGER NOT NULL DEFAULT 0
);
CREATE TABLE IF NOT EXISTS embeddings (
id INTEGER PRIMARY KEY AUTOINCREMENT,
identity_id INTEGER NOT NULL REFERENCES identities(id) ON DELETE CASCADE,
vector BLOB NOT NULL,
model TEXT NOT NULL DEFAULT '',
quality REAL NOT NULL DEFAULT 0,
created_at REAL NOT NULL
);
CREATE INDEX IF NOT EXISTS idx_embeddings_identity ON embeddings(identity_id);
CREATE TABLE IF NOT EXISTS sightings (
id INTEGER PRIMARY KEY AUTOINCREMENT,
identity_id INTEGER NOT NULL REFERENCES identities(id) ON DELETE CASCADE,
camera_id TEXT NOT NULL,
ts REAL NOT NULL,
similarity REAL NOT NULL DEFAULT 0,
quality REAL NOT NULL DEFAULT 0,
attributes TEXT
);
CREATE INDEX IF NOT EXISTS idx_sightings_identity ON sightings(identity_id);
CREATE INDEX IF NOT EXISTS idx_sightings_ts ON sightings(ts);
"""
class IdentityStore:
def __init__(self, db_path: "Path | str"):
Path(db_path).parent.mkdir(parents=True, exist_ok=True)
self._lock = threading.Lock()
self._db = sqlite3.connect(str(db_path), check_same_thread=False)
self._db.row_factory = sqlite3.Row
with self._lock:
self._db.execute("PRAGMA journal_mode=WAL")
self._db.execute("PRAGMA foreign_keys=ON")
self._db.executescript(_SCHEMA)
self._db.commit()
# -- identities -----------------------------------------------------
def create_identity(self, label: str, kind: str = "auto") -> int:
with self._lock:
cur = self._db.execute(
"INSERT INTO identities(label, kind, created_at) VALUES(?,?,?)",
(label, kind, time.time()))
self._db.commit()
return int(cur.lastrowid)
def create_auto_identity(self) -> "tuple[int, str]":
"""Create an auto-enrolled identity labelled 'Visitor <id>' in one
transaction; returns (id, label)."""
with self._lock:
cur = self._db.execute(
"INSERT INTO identities(label, kind, created_at) VALUES(?,?,?)",
("pending", "auto", time.time()))
identity_id = int(cur.lastrowid)
label = f"Visitor {identity_id}"
self._db.execute(
"UPDATE identities SET label=? WHERE id=?", (label, identity_id))
self._db.commit()
return identity_id, label
def rename_identity(self, identity_id: int, label: str) -> bool:
with self._lock:
cur = self._db.execute(
"UPDATE identities SET label=?, kind='enrolled' WHERE id=?",
(label, identity_id))
self._db.commit()
return cur.rowcount > 0
def delete_identity(self, identity_id: int) -> "list[int]":
"""Delete an identity; returns removed embedding ids (for the index)."""
with self._lock:
rows = self._db.execute(
"SELECT id FROM embeddings WHERE identity_id=?",
(identity_id,)).fetchall()
self._db.execute("DELETE FROM identities WHERE id=?", (identity_id,))
self._db.commit()
return [int(r["id"]) for r in rows]
def merge_identities(self, source_id: int, target_id: int,
max_embeddings: int = 5) -> "dict | None":
"""Fold `source_id` into `target_id`; returns a summary, or None if
either identity is missing.
Embeddings and sightings are re-pointed rather than copied, which is
what keeps this cheap AND keeps the vector index valid: the index maps
*embedding* id to vector, and those ids do not change here, so a merge
needs no reindex. Only trimmed embeddings have to be dropped from it,
which is why they are returned.
Everything happens in one transaction. A half-merge — sightings moved,
embeddings not — would leave two identities each holding part of one
person, which is strictly worse than the duplicate we started with.
"""
with self._lock:
src = self._db.execute("SELECT * FROM identities WHERE id=?",
(source_id,)).fetchone()
dst = self._db.execute("SELECT * FROM identities WHERE id=?",
(target_id,)).fetchone()
if src is None or dst is None or source_id == target_id:
return None
try:
emb = self._db.execute(
"UPDATE embeddings SET identity_id=? WHERE identity_id=?",
(target_id, source_id)).rowcount
sig = self._db.execute(
"UPDATE sightings SET identity_id=? WHERE identity_id=?",
(target_id, source_id)).rowcount
# A human-assigned name outranks an auto "Visitor N" whichever
# direction the operator merged in — silently turning "Alice"
# back into "Visitor 3" would be a data-loss bug, not a policy.
label, kind = dst["label"], dst["kind"]
if dst["kind"] == "auto" and src["kind"] != "auto":
label, kind = src["label"], src["kind"]
# The merged identity's history starts at the earlier of the
# two first-sightings; it is one person and always was.
created = min(float(src["created_at"]), float(dst["created_at"]))
# Trim to the highest-quality views. Merging two identities
# that each held the cap would otherwise leave one holding
# double, quietly overweighting that person in every search.
dropped = [int(r["id"]) for r in self._db.execute(
"SELECT id FROM embeddings WHERE identity_id=? "
"ORDER BY quality DESC, id ASC LIMIT -1 OFFSET ?",
(target_id, max_embeddings)).fetchall()]
if dropped:
self._db.execute(
"DELETE FROM embeddings WHERE id IN (%s)"
% ",".join("?" * len(dropped)), dropped)
# Recomputed, never summed: sighting_count on the source may
# itself be stale, and COUNT(*) is the only figure that cannot
# drift away from the rows actually present.
agg = self._db.execute(
"SELECT COUNT(*) AS n, MAX(ts) AS last FROM sightings "
"WHERE identity_id=?", (target_id,)).fetchone()
self._db.execute(
"UPDATE identities SET label=?, kind=?, created_at=?, "
"sighting_count=?, last_seen_at=? WHERE id=?",
(label, kind, created, int(agg["n"]), agg["last"],
target_id))
self._db.execute("DELETE FROM identities WHERE id=?",
(source_id,))
self._db.commit()
except Exception:
self._db.rollback()
raise
return {"source": source_id, "target": target_id, "label": label,
"embeddings_moved": int(emb), "sightings_moved": int(sig),
"dropped_embeddings": dropped,
"sighting_count": int(agg["n"])}
def identity_embeddings(self, identity_id: int, dim: int,
model: "str | None" = None
) -> "tuple[list[int], np.ndarray]":
"""One identity's stored vectors, for comparing two identities to each
other. Model-filtered for the same reason the index is."""
with self._lock:
if model is None:
rows = self._db.execute(
"SELECT id, vector FROM embeddings WHERE identity_id=? "
"ORDER BY id", (identity_id,)).fetchall()
else:
rows = self._db.execute(
"SELECT id, vector FROM embeddings WHERE identity_id=? "
"AND model=? ORDER BY id",
(identity_id, model)).fetchall()
ids = [int(r["id"]) for r in rows]
if not ids:
return [], np.empty((0, dim), dtype=np.float32)
return ids, np.vstack([
np.frombuffer(r["vector"], dtype=np.float32) for r in rows])
def get_identity(self, identity_id: int) -> "dict | None":
with self._lock:
row = self._db.execute(
"SELECT * FROM identities WHERE id=?", (identity_id,)).fetchone()
return dict(row) if row else None
def list_identities(self, limit: int = 200) -> "list[dict]":
with self._lock:
rows = self._db.execute(
"SELECT i.*, COUNT(e.id) AS embedding_count FROM identities i "
"LEFT JOIN embeddings e ON e.identity_id = i.id "
"GROUP BY i.id ORDER BY i.last_seen_at DESC LIMIT ?",
(limit,)).fetchall()
return [dict(r) for r in rows]
# -- embeddings -----------------------------------------------------
def add_embedding(self, identity_id: int, vector: np.ndarray,
quality: float, model: str = "") -> int:
blob = np.asarray(vector, dtype=np.float32).tobytes()
with self._lock:
cur = self._db.execute(
"INSERT INTO embeddings(identity_id, vector, model, quality,"
" created_at) VALUES(?,?,?,?,?)",
(identity_id, blob, model, quality, time.time()))
self._db.commit()
return int(cur.lastrowid)
def embedding_count(self, identity_id: int) -> int:
with self._lock:
row = self._db.execute(
"SELECT COUNT(*) AS n FROM embeddings WHERE identity_id=?",
(identity_id,)).fetchone()
return int(row["n"])
def identity_for_embedding(self, embedding_id: int) -> "dict | None":
with self._lock:
row = self._db.execute(
"SELECT i.* FROM identities i JOIN embeddings e "
"ON e.identity_id = i.id WHERE e.id=?",
(embedding_id,)).fetchone()
return dict(row) if row else None
def all_embeddings(self, dim: int, model: "str | None" = None
) -> "tuple[list[int], np.ndarray]":
"""Embeddings for the vector index. Filtering by `model` is what
keeps vectors from different encoders out of the same search space —
they are numerically incompatible."""
with self._lock:
if model is None:
rows = self._db.execute(
"SELECT id, vector FROM embeddings ORDER BY id").fetchall()
else:
rows = self._db.execute(
"SELECT id, vector FROM embeddings WHERE model=? "
"ORDER BY id", (model,)).fetchall()
ids = [int(r["id"]) for r in rows]
if not ids:
return [], np.empty((0, dim), dtype=np.float32)
vecs = np.vstack([
np.frombuffer(r["vector"], dtype=np.float32) for r in rows])
return ids, vecs
def best_embedding(self, identity_id: int, model: "str | None" = None
) -> "tuple[np.ndarray, float] | None":
"""The highest-quality stored view of one identity.
For handing an identity to the server: sending the best view rather
than the mean because a mean of two disagreeing views is a vector that
matches neither, which is precisely how one person becomes two
identities.
"""
with self._lock:
if model is None:
row = self._db.execute(
"SELECT vector, quality FROM embeddings WHERE identity_id=? "
"ORDER BY quality DESC, id ASC LIMIT 1",
(identity_id,)).fetchone()
else:
row = self._db.execute(
"SELECT vector, quality FROM embeddings WHERE identity_id=? "
"AND model=? ORDER BY quality DESC, id ASC LIMIT 1",
(identity_id, model)).fetchone()
if row is None:
return None
return np.frombuffer(row["vector"], dtype=np.float32), float(row["quality"])
def embedding_owners(self, model: "str | None" = None) -> "dict[int, int]":
"""embedding_id -> identity_id, for turning index hits into identity
pairs without a round trip to SQLite per hit."""
with self._lock:
if model is None:
rows = self._db.execute(
"SELECT id, identity_id FROM embeddings").fetchall()
else:
rows = self._db.execute(
"SELECT id, identity_id FROM embeddings WHERE model=?",
(model,)).fetchall()
return {int(r["id"]): int(r["identity_id"]) for r in rows}
# -- sightings ------------------------------------------------------
def record_sighting(self, identity_id: int, camera_id: str, ts: float,
similarity: float, quality: float,
attributes: "dict | None" = None) -> None:
with self._lock:
self._db.execute(
"INSERT INTO sightings(identity_id, camera_id, ts, similarity,"
" quality, attributes) VALUES(?,?,?,?,?,?)",
(identity_id, camera_id, ts, similarity, quality,
json.dumps(attributes) if attributes else None))
self._db.execute(
"UPDATE identities SET last_seen_at=?, "
"sighting_count=sighting_count+1 WHERE id=?", (ts, identity_id))
self._db.commit()
def recent_sightings(self, limit: int = 100) -> "list[dict]":
with self._lock:
rows = self._db.execute(
"SELECT s.*, i.label FROM sightings s JOIN identities i "
"ON i.id = s.identity_id ORDER BY s.ts DESC LIMIT ?",
(limit,)).fetchall()
out = []
for r in rows:
d = dict(r)
if d.get("attributes"):
d["attributes"] = json.loads(d["attributes"])
out.append(d)
return out
def stats(self) -> dict:
with self._lock:
n_id = self._db.execute(
"SELECT COUNT(*) AS n FROM identities").fetchone()["n"]
n_emb = self._db.execute(
"SELECT COUNT(*) AS n FROM embeddings").fetchone()["n"]
n_sight = self._db.execute(
"SELECT COUNT(*) AS n FROM sightings").fetchone()["n"]
return {"identities": n_id, "embeddings": n_emb, "sightings": n_sight}
def close(self) -> None:
with self._lock:
self._db.close()

81
behavision/geometry.py Normal file
View File

@@ -0,0 +1,81 @@
"""Box math and ArcFace 5-point alignment (Umeyama similarity transform)."""
from __future__ import annotations
import cv2
import numpy as np
# Canonical 5-point landmark template for a 112x112 ArcFace crop:
# left eye, right eye, nose tip, left mouth corner, right mouth corner.
ARCFACE_TEMPLATE = np.array(
[
[38.2946, 51.6963],
[73.5318, 51.5014],
[56.0252, 71.7366],
[41.5493, 92.3655],
[70.7299, 92.2041],
],
dtype=np.float32,
)
def clip_box(box, width: int, height: int):
"""Clamp an (x1, y1, x2, y2) box to image bounds.
Returns int coords, or None when nothing of the box remains inside the
frame. This is what prevents negative indices from silently wrapping
around in numpy slicing.
"""
x1, y1, x2, y2 = box
x1 = int(max(0, min(x1, width)))
y1 = int(max(0, min(y1, height)))
x2 = int(max(0, min(x2, width)))
y2 = int(max(0, min(y2, height)))
if x2 - x1 < 2 or y2 - y1 < 2:
return None
return x1, y1, x2, y2
def iou(a, b) -> float:
ax1, ay1, ax2, ay2 = a
bx1, by1, bx2, by2 = b
ix1, iy1 = max(ax1, bx1), max(ay1, by1)
ix2, iy2 = min(ax2, bx2), min(ay2, by2)
iw, ih = max(0.0, ix2 - ix1), max(0.0, iy2 - iy1)
inter = iw * ih
if inter <= 0:
return 0.0
union = (ax2 - ax1) * (ay2 - ay1) + (bx2 - bx1) * (by2 - by1) - inter
return float(inter / union) if union > 0 else 0.0
def umeyama(src: np.ndarray, dst: np.ndarray) -> np.ndarray:
"""Least-squares similarity transform (Umeyama 1991) mapping src -> dst.
Deterministic (no RANSAC), which keeps embeddings reproducible for the
same input frame. Returns a 2x3 affine matrix for cv2.warpAffine.
"""
src = np.asarray(src, dtype=np.float64)
dst = np.asarray(dst, dtype=np.float64)
n = src.shape[0]
src_mean, dst_mean = src.mean(0), dst.mean(0)
src_c, dst_c = src - src_mean, dst - dst_mean
cov = dst_c.T @ src_c / n
u, s, vt = np.linalg.svd(cov)
d = np.ones(2)
if np.linalg.det(u) * np.linalg.det(vt) < 0:
d[1] = -1.0
rot = u @ np.diag(d) @ vt
var_src = (src_c ** 2).sum() / n
scale = (s * d).sum() / var_src if var_src > 1e-12 else 1.0
t = dst_mean - scale * rot @ src_mean
return np.hstack([scale * rot, t.reshape(2, 1)]).astype(np.float32)
def align_face(image: np.ndarray, kps: np.ndarray, size: int = 112) -> np.ndarray:
"""Warp a full frame to a canonical `size`x`size` face chip using the
5 detected landmarks (full-frame coordinates — the whole point is that
landmarks and image are in the SAME coordinate space)."""
template = ARCFACE_TEMPLATE * (size / 112.0)
m = umeyama(np.asarray(kps, dtype=np.float32), template)
return cv2.warpAffine(image, m, (size, size), borderValue=0)

28
behavision/log.py Normal file
View File

@@ -0,0 +1,28 @@
"""Central logging setup: console + rotating file, no print() anywhere."""
from __future__ import annotations
import logging
import logging.handlers
from pathlib import Path
_FORMAT = "%(asctime)s %(levelname)-7s %(name)s: %(message)s"
def setup_logging(level: str = "INFO", data_dir: "Path | None" = None) -> None:
root = logging.getLogger()
if root.handlers: # already configured (tests, reload)
return
root.setLevel(getattr(logging, level.upper(), logging.INFO))
console = logging.StreamHandler()
console.setFormatter(logging.Formatter(_FORMAT))
root.addHandler(console)
if data_dir is not None:
log_dir = Path(data_dir) / "logs"
log_dir.mkdir(parents=True, exist_ok=True)
fileh = logging.handlers.RotatingFileHandler(
log_dir / "behavision.log", maxBytes=5_000_000, backupCount=3,
encoding="utf-8")
fileh.setFormatter(logging.Formatter(_FORMAT))
root.addHandler(fileh)

119
behavision/model_assets.py Normal file
View File

@@ -0,0 +1,119 @@
"""Model acquisition: download YuNet, copy reusable models from the old
projects on this machine when present. Idempotent — safe to re-run."""
from __future__ import annotations
import logging
import shutil
import urllib.request
from pathlib import Path
log = logging.getLogger(__name__)
YUNET_URL = ("https://github.com/opencv/opencv_zoo/raw/main/models/"
"face_detection_yunet/face_detection_yunet_2023mar.onnx")
BUFFALO_SC_URL = ("https://github.com/deepinsight/insightface/releases/"
"download/v0.7/buffalo_sc.zip")
RECOGNIZERS = ["adaface_ir101.onnx", "adaface_ir50.onnx", "w600k_r50.onnx",
"arcface_int8.onnx", "w600k_mbf.onnx", "arcface.onnx"]
# Known locations of reusable models from the previous projects.
_LEGACY_MODEL_DIRS = [
Path(r"D:\NEARLE\WOrking now\RTSP_16072025\pattern_reg\models"),
]
BUFFALO_L_URL = ("https://github.com/deepinsight/insightface/releases/"
"download/v0.7/buffalo_l.zip")
# target filename -> legacy filename
_COPY_MAP = {
"arcface.onnx": "arcface.onnx",
"age_deploy.prototxt": "age_deploy.prototxt",
"age_net.caffemodel": "age_net.caffemodel",
"gender_deploy.prototxt": "gender_deploy.prototxt",
"gender_net.caffemodel": "gender_net.caffemodel",
"emotion-ferplus-8.onnx": "emotion-ferplus-8.onnx",
}
def setup_models(models_dir: Path) -> "list[str]":
"""Ensure all model files exist in models_dir. Returns missing ones."""
models_dir = Path(models_dir)
models_dir.mkdir(parents=True, exist_ok=True)
yunet = models_dir / "face_detection_yunet_2023mar.onnx"
if not yunet.exists():
log.info("downloading YuNet face detector (~230 KB)...")
tmp = yunet.with_suffix(".part")
urllib.request.urlretrieve(YUNET_URL, tmp)
tmp.rename(yunet)
log.info("YuNet saved to %s", yunet)
for target_name, legacy_name in _COPY_MAP.items():
target = models_dir / target_name
if target.exists():
continue
for legacy_dir in _LEGACY_MODEL_DIRS:
src = legacy_dir / legacy_name
if src.exists():
log.info("copying %s from %s ...", legacy_name, legacy_dir)
shutil.copy2(src, target)
break
# Any one recognizer is enough; get the lightweight MobileFaceNet if
# none is present (13 MB, loads reliably on low-memory machines).
if not any((models_dir / n).exists() for n in RECOGNIZERS):
log.info("downloading MobileFaceNet recognizer (buffalo_sc, ~15 MB)...")
import io
import zipfile
with urllib.request.urlopen(BUFFALO_SC_URL) as resp:
payload = io.BytesIO(resp.read())
with zipfile.ZipFile(payload) as zf, \
zf.open("w600k_mbf.onnx") as src, \
open(models_dir / "w600k_mbf.onnx", "wb") as dst:
shutil.copyfileobj(src, dst)
log.info("w600k_mbf.onnx saved")
# buffalo_l carries both the modern gender+age net (1.3 MB) and the
# ResNet50 recognizer (~166 MB, IJB-C 97.25 vs MobileFaceNet's 95.02).
# One 275 MB download serves both, so fetch it once and take what is
# missing. Optional: failure here must never block the pipeline.
wanted = {name: models_dir / name
for name in ("genderage.onnx", "w600k_r50.onnx")
if not (models_dir / name).exists()}
if wanted:
try:
log.info("downloading %s from the buffalo_l bundle (~275 MB "
"one-time download)...", ", ".join(wanted))
import zipfile
tmp = models_dir / "buffalo_l.zip.part"
urllib.request.urlretrieve(BUFFALO_L_URL, tmp)
with zipfile.ZipFile(tmp) as zf:
for name, target in wanted.items():
member = next((n for n in zf.namelist()
if n.endswith(name)), None)
if member is None:
log.warning("%s not found in bundle", name)
continue
part = target.with_suffix(".part")
with zf.open(member) as src, open(part, "wb") as dst:
shutil.copyfileobj(src, dst)
part.rename(target) # never leave a half-written model
log.info("%s saved", name)
tmp.unlink()
except Exception:
log.warning("buffalo_l download failed - falling back to the "
"models already present", exc_info=True)
missing = []
if not (models_dir / "face_detection_yunet_2023mar.onnx").exists():
missing.append("face_detection_yunet_2023mar.onnx")
if not any((models_dir / n).exists() for n in RECOGNIZERS):
missing.append("a recognition model (any of: %s)" % ", ".join(RECOGNIZERS))
optional_missing = [n for n in _COPY_MAP
if not (models_dir / n).exists() and n not in missing]
if optional_missing:
log.warning("optional attribute models missing (age/gender/emotion "
"will be skipped): %s", ", ".join(optional_missing))
return missing

146
behavision/paths.py Normal file
View File

@@ -0,0 +1,146 @@
"""Where the code lives versus where the code may write.
In a checkout these are the same directory, which is why everything resolved
against the repo root until now. Installed, they are not: the code sits under
`Program Files`, which is read-only for a normal user and for a service running
as LocalSystem, while the database, logs, camera list and downloaded models all
have to be written somewhere that survives an upgrade.
Three roots, resolved in one place so nothing else has to know it is frozen:
- `install_root()` — the code and the bundled default config. Read-only.
- `state_root()` — everything we write. `%PROGRAMDATA%\\Behavision` when
frozen on Windows.
- `config_path()` — the YAML actually loaded.
Models live under `state_root()`, not next to the code: they are ~200 MB and
are downloaded on first run rather than bundled (a 300 MB installer that has to
be re-signed for a model change is a bad trade), so they must land somewhere
writable.
`BEHAVISION_DATA_DIR` and `BEHAVISION_CONFIG` override everything, which is
what makes the installed layout testable from a checkout and lets one machine
run two instances.
"""
from __future__ import annotations
import os
import sys
from pathlib import Path
APP_NAME = "Behavision"
def is_frozen() -> bool:
"""True inside a PyInstaller bundle."""
return bool(getattr(sys, "frozen", False))
def install_root() -> Path:
"""Directory holding the code and bundled data files.
Frozen, that is the folder containing the .exe — PyInstaller's one-folder
layout — not `_MEIPASS`, which for onefile is a temp dir that vanishes.
"""
if is_frozen():
return Path(sys.executable).resolve().parent
return Path(__file__).resolve().parent.parent
def _os_family() -> str:
"""Which install layout applies.
A seam, not decoration: a test cannot monkeypatch `os.name` to exercise the
Windows layout on another host, because `pathlib` dispatches on it and
every `Path()` in the process starts raising.
"""
if os.name == "nt":
return "windows"
if sys.platform == "darwin":
return "macos"
return "linux"
def state_root() -> Path:
"""Directory we may write to. Created by the caller, not here."""
override = os.environ.get("BEHAVISION_DATA_DIR", "").strip()
if override:
return Path(override).expanduser().resolve()
if not is_frozen():
# A checkout keeps everything together; that is the whole convenience
# of developing from one.
return install_root()
family = _os_family()
if family == "windows":
base = os.environ.get("PROGRAMDATA") or r"C:\ProgramData"
return Path(base) / APP_NAME
if family == "macos":
return Path.home() / "Library" / "Application Support" / APP_NAME
return Path(
os.environ.get("XDG_DATA_HOME") or Path.home() / ".local" / "share"
) / APP_NAME.lower()
def config_path() -> Path:
"""The YAML to load.
An installed system must be configurable without editing anything under
`Program Files`, so a copy in the state root wins over the bundled one.
`ensure_config()` puts it there on first run.
"""
override = os.environ.get("BEHAVISION_CONFIG", "").strip()
if override:
return Path(override).expanduser().resolve()
local = state_root() / "config" / "default.yaml"
if local.is_file():
return local
return install_root() / "config" / "default.yaml"
def env_file() -> "Path | None":
"""`.env`, preferring the writable copy. Returns None when there is none —
it is optional, and an installed system keeps its secrets in the config
and the camera store instead."""
for candidate in (state_root() / ".env", install_root() / ".env"):
if candidate.is_file():
return candidate
return None
def ensure_config() -> Path:
"""Seed an editable config in the state root on first run, and return the
path that will be loaded.
Copied, never symlinked, and never overwritten: an upgrade must not
silently revert an operator's thresholds.
"""
override = os.environ.get("BEHAVISION_CONFIG", "").strip()
if override:
return Path(override).expanduser().resolve()
local = state_root() / "config" / "default.yaml"
if local.is_file():
return local
bundled = install_root() / "config" / "default.yaml"
if not bundled.is_file():
# Nothing to seed. Return the bundled path so the caller's "no such
# file" names the place the file was supposed to be.
return bundled
if local.parent.exists() and local.resolve() == bundled.resolve():
# A checkout: install root and state root are the same directory, so
# the "copy" would be a file onto itself.
return bundled
local.parent.mkdir(parents=True, exist_ok=True)
local.write_text(bundled.read_text(encoding="utf-8"), encoding="utf-8")
return local
def describe() -> dict:
"""For /api/health and the tray - "where is my database" must be
answerable without reading the source."""
return {
"frozen": is_frozen(),
"install_root": str(install_root()),
"state_root": str(state_root()),
"config": str(config_path()),
}

167
behavision/recognition.py Normal file
View File

@@ -0,0 +1,167 @@
"""ArcFace embedding + face quality assessment.
Preprocessing contract (this is where the old codebase broke recognition):
aligned 112x112 BGR chip -> [RGB if the model wants it] ->
(x - 127.5) / 127.5 -> NCHW float32.
Exactly one colour conversion, the normalisation ArcFace was trained with,
and L2-normalised output so cosine similarity is a plain dot product.
Channel order is per-model (see color_order_for): ArcFace/InsightFace want
RGB, AdaFace wants BGR. Same scaling, opposite channel order, and no error
if you get it wrong — hence the explicit table.
"""
from __future__ import annotations
import logging
from pathlib import Path
from typing import Optional
import cv2
import numpy as np
from .geometry import align_face
log = logging.getLogger(__name__)
EMBEDDING_DIM = 512
# Tried in order; first one that exists AND loads wins. The lightweight
# MobileFaceNet (13 MB, same WebFace600K training data) sits before the
# 260 MB r100 export because a model that loads on every boot beats a
# marginally more accurate one that fails under memory pressure — and
# embeddings from different models are incompatible, so boot-to-boot
# consistency matters. Pin one explicitly via recognition config if needed.
MODEL_CANDIDATES = [
"adaface_ir101.onnx", # best, ~250 MB - only loads on a roomy machine
"adaface_ir50.onnx", # ~170 MB, quality-adaptive: best for blur/low light
"w600k_r50.onnx", # ~166 MB, IJB-C 97.25 vs mbf's 95.02
"arcface_int8.onnx",
"w600k_mbf.onnx", # 13 MB, always loads
"arcface.onnx", # r100, 249 MB
]
# Channel order each family was trained on. InsightFace/ArcFace exports expect
# RGB; AdaFace expects BGR (mean=0.5/std=0.5, which is the same (x-127.5)/127.5
# scaling — ONLY the channel order differs). Getting it wrong raises nothing.
# Measured on this camera with w600k_r50: the same face chip encoded RGB vs
# BGR cross-matches at 0.945, so it is a mild perturbation rather than a
# catastrophe (faces are low-saturation, so R and B correlate). Still declared
# per model: it costs one lookup, it is the documented contract each model was
# trained under, and it removes a needless source of drift near the 0.42
# decision boundary.
BGR_MODELS = ("adaface",)
DEFAULT_COLOR_ORDER = "RGB"
def color_order_for(model_name: str) -> str:
name = model_name.lower()
return "BGR" if any(tag in name for tag in BGR_MODELS) else DEFAULT_COLOR_ORDER
class ArcFaceEncoder:
def __init__(self, models_dir: Path, model_file: str = "",
color_order: str = ""):
import onnxruntime as ort
candidates = [model_file] if model_file else MODEL_CANDIDATES
providers = ort.get_available_providers()
self.session = None
for name in candidates:
model_path = Path(models_dir) / name
if not model_path.exists():
continue
try:
self.session = ort.InferenceSession(str(model_path),
providers=providers)
except Exception:
# Graph optimization of a large model needs a big transient
# allocation; retry unoptimized before giving up on it.
log.warning("%s: optimized load failed, retrying without "
"graph optimization (low memory?)", name)
try:
so = ort.SessionOptions()
so.graph_optimization_level = (
ort.GraphOptimizationLevel.ORT_DISABLE_ALL)
so.enable_mem_pattern = False
self.session = ort.InferenceSession(
str(model_path), sess_options=so, providers=providers)
except Exception:
log.warning("%s: unusable on this machine, trying next "
"candidate", name)
continue
self.model_name = model_path.stem
break
if self.session is None:
raise FileNotFoundError(
f"no usable recognition model in {models_dir} "
f"(tried {', '.join(candidates)}) - "
"run: python -m behavision setup-models")
# Explicit config wins; otherwise infer from the model family.
self.color_order = (color_order or color_order_for(self.model_name)).upper()
if self.color_order not in ("RGB", "BGR"):
raise ValueError(f"color_order must be RGB or BGR, got {color_order!r}")
inp = self.session.get_inputs()[0]
self.input_name = inp.name
# Introspect instead of assuming: works for 112x112 r50/r100/mbf exports.
self.size = inp.shape[-1] if isinstance(inp.shape[-1], int) else 112
self.output_name = self.session.get_outputs()[0].name
log.info("recognition model '%s' loaded (input %sx%s, %s, providers=%s)",
self.model_name, self.size, self.size, self.color_order,
providers)
def encode_chip(self, chip_bgr: np.ndarray) -> Optional[np.ndarray]:
"""Embed an already-aligned BGR chip. Returns unit-norm float32[512]."""
if chip_bgr is None or chip_bgr.size == 0:
return None
if chip_bgr.shape[:2] != (self.size, self.size):
chip_bgr = cv2.resize(chip_bgr, (self.size, self.size))
# Exactly one colour conversion, and only when the model wants RGB.
chip = (cv2.cvtColor(chip_bgr, cv2.COLOR_BGR2RGB)
if self.color_order == "RGB" else chip_bgr)
blob = ((chip.astype(np.float32) - 127.5) / 127.5).transpose(2, 0, 1)[None]
emb = self.session.run([self.output_name], {self.input_name: blob})[0][0]
emb = np.asarray(emb, dtype=np.float32).ravel()
norm = float(np.linalg.norm(emb))
if norm < 1e-6: # degenerate output — never store or match this
return None
return emb / norm
def encode(self, frame_bgr: np.ndarray, kps: np.ndarray) -> Optional[np.ndarray]:
"""Align (full-frame landmarks) then embed."""
chip = align_face(frame_bgr, kps, size=self.size)
return self.encode_chip(chip)
def face_quality(frame: np.ndarray, box, kps: np.ndarray) -> float:
"""0..1 quality score used to gate enrollment. Every term is clamped so
the weighted sum stays interpretable (the old code's size term made its
own threshold unreachable)."""
x1, y1, x2, y2 = box
crop = frame[y1:y2, x1:x2]
if crop.size == 0:
return 0.0
gray = cv2.cvtColor(crop, cv2.COLOR_BGR2GRAY)
sharpness = min(1.0, cv2.Laplacian(gray, cv2.CV_64F).var() / 250.0)
size_score = min(1.0, min(x2 - x1, y2 - y1) / 112.0)
mean_b = float(gray.mean())
if 60.0 <= mean_b <= 190.0:
brightness = 1.0
elif mean_b < 60.0:
brightness = max(0.0, mean_b / 60.0)
else:
brightness = max(0.0, (255.0 - mean_b) / 65.0)
# Frontality: nose tip should sit near the horizontal midpoint of the
# eyes; offset is normalised by inter-eye distance.
eye_l, eye_r, nose = kps[0], kps[1], kps[2]
eye_dist = float(np.linalg.norm(eye_r - eye_l))
if eye_dist < 1.0:
frontality = 0.0
else:
mid_x = (eye_l[0] + eye_r[0]) / 2.0
frontality = max(0.0, 1.0 - 2.0 * abs(nose[0] - mid_x) / eye_dist)
score = (0.35 * sharpness + 0.25 * size_score
+ 0.15 * brightness + 0.25 * frontality)
return float(max(0.0, min(1.0, score)))

View File

@@ -0,0 +1,552 @@
<!doctype html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>Behavision</title>
<style>
:root { color-scheme: dark; }
* { box-sizing: border-box; margin: 0; }
body { font: 14px/1.5 system-ui, sans-serif; background: #101418;
color: #dde3ea; padding: 1.25rem; }
h1 { font-size: 1.15rem; margin-bottom: 1rem; letter-spacing: .02em; }
h1 small { color: #7b8794; font-weight: 400; margin-left: .5rem; }
.grid { display: grid; grid-template-columns: 2fr 1fr; gap: 1rem; }
@media (max-width: 900px) { .grid { grid-template-columns: 1fr; } }
.card { background: #171d24; border: 1px solid #232c36;
border-radius: 10px; padding: 1rem; }
.card h2 { font-size: .8rem; text-transform: uppercase; color: #7b8794;
letter-spacing: .08em; margin-bottom: .75rem; }
img.feed { width: 100%; border-radius: 6px; background: #000;
min-height: 240px; }
table { width: 100%; border-collapse: collapse; }
td, th { padding: .35rem .5rem; text-align: left;
border-bottom: 1px solid #232c36; }
th { color: #7b8794; font-weight: 500; font-size: .78rem; }
.muted { color: #7b8794; }
ul#events { list-style: none; max-height: 380px; overflow-y: auto; }
ul#events li { padding: .4rem 0; border-bottom: 1px solid #232c36; }
.tag { display: inline-block; padding: .05rem .45rem; border-radius: 99px;
font-size: .72rem; margin-right: .4rem; }
.tag.new { background: #2b4e77; } .tag.seen { background: #2e5c3a; }
.tag.cam { background: #5a4a2a; }
.tag.miss { background: #6b3030; }
.tag.merge { background: #4a3a6b; }
.bar { display: flex; height: 8px; border-radius: 4px; overflow: hidden;
margin: .35rem 0 .5rem; background: #222; }
.bar i { display: block; }
.warn { color: #e0a33a; }
.pl-row { margin-bottom: .9rem; }
.pl-row b { font-weight: 600; }
.legend { font-size: .72rem; }
.legend span { margin-right: .7rem; white-space: nowrap; }
.btn { background: #2a3446; color: #cfd8e3; border: 1px solid #3a4658;
border-radius: 4px; padding: .25rem .6rem; cursor: pointer;
font: inherit; font-size: .78rem; }
.btn:hover { background: #35415a; }
.btn[disabled] { opacity: .5; cursor: default; }
.btn.primary { background: #2b4e77; border-color: #3a6291; }
.btn.danger { background: #4a2626; border-color: #6b3030; }
.card h2 .btn { float: right; margin-top: -.15rem; text-transform: none;
letter-spacing: 0; }
.cam-row { display: flex; align-items: center; gap: .5rem;
padding: .4rem 0; border-bottom: 1px solid #1e2430; }
.cam-row:last-child { border-bottom: 0; }
.cam-row .grow { flex: 1; min-width: 0; overflow: hidden;
text-overflow: ellipsis; white-space: nowrap; }
.pill { font-size: .7rem; padding: .05rem .45rem; border-radius: 99px; }
.pill.up { background: #2e5c3a; } .pill.down { background: #6b3030; }
.fields { display: grid; grid-template-columns: 1fr 1fr; gap: .5rem .75rem;
margin-bottom: .6rem; }
.fields label { display: block; font-size: .72rem; color: #7b8794;
margin-bottom: .15rem; }
.fields .wide { grid-column: 1 / -1; }
.fields input { width: 100%; background: #101418; color: #dde3ea;
border: 1px solid #2b3543; border-radius: 4px;
padding: .3rem .45rem; font: inherit; font-size: .82rem; }
.fields input:disabled { color: #7b8794; }
#cam-form { border-top: 1px solid #232c36; margin-top: .6rem;
padding-top: .75rem; }
#wizard { position: fixed; inset: 0; background: #000a;
display: flex; align-items: center; justify-content: center;
padding: 1rem; z-index: 20; }
/* An author `display` beats the UA stylesheet's `[hidden] { display: none }`,
so without this the placement wizard sits open over the dashboard on every
load - the modal is toggled by the `hidden` property, not by a class. */
#wizard[hidden] { display: none; }
#wizard .panel { background: #171d24; border: 1px solid #2b3543;
border-radius: 10px; padding: 1.25rem; max-width: 460px;
width: 100%; }
#wizard h3 { font-size: .95rem; margin-bottom: .5rem; }
#wizard .advice { margin: .6rem 0 .9rem; padding-left: 1.1rem;
font-size: .84rem; color: #b8c2ce; }
#wizard .advice li { margin-bottom: .3rem; }
.verdict { font-size: .95rem; font-weight: 600; margin: .4rem 0; }
.v-good { color: #4caf7d; } .v-marginal { color: #e0a33a; }
.v-poor, .v-artifact { color: #e05c5c; }
.v-no_faces, .v-inconclusive { color: #e0a33a; }
.progress { height: 6px; border-radius: 3px; background: #222;
overflow: hidden; margin: .6rem 0; }
.progress i { display: block; height: 100%; background: #3a6291; }
#cam-test { margin-top: .6rem; font-size: .82rem; }
#cam-test img { width: 100%; border-radius: 6px; margin-top: .4rem; }
.ok { color: #4caf7d; }
.err { color: #e05c5c; }
.dup { display: flex; align-items: center; gap: .5rem;
padding: .35rem 0; border-bottom: 1px solid #1e2430; }
.dup:last-child { border-bottom: 0; }
.dup .grow { flex: 1; min-width: 0; }
.dup button { background: #2a3446; color: #cfd8e3; border: 1px solid #3a4658;
border-radius: 4px; padding: .2rem .55rem; cursor: pointer;
font: inherit; font-size: .78rem; }
.dup button:hover { background: #35415a; }
.dup button[disabled] { opacity: .5; cursor: default; }
.sim { font-variant-numeric: tabular-nums; }
.dot { display: inline-block; width: .55rem; height: .55rem;
border-radius: 50%; margin-right: .25rem; vertical-align: middle; }
</style>
</head>
<body>
<h1>Behavision <small id="status">connecting…</small></h1>
<div class="grid">
<div>
<div class="card"><h2>Live</h2><div id="feeds"></div></div>
<div class="card" style="margin-top:1rem">
<h2>Cameras <button class="btn" id="cam-new">+ Add camera</button></h2>
<div id="cam-list" class="muted">none configured</div>
<div id="cam-form" hidden>
<div class="fields">
<div><label for="f-id">Camera id</label>
<input id="f-id" placeholder="entrance"></div>
<div><label for="f-host">Host / IP</label>
<input id="f-host" placeholder="192.168.0.138"></div>
<div><label for="f-port">Port</label>
<input id="f-port" type="number" value="554"></div>
<div><label for="f-path">Stream path</label>
<input id="f-path" placeholder="/ch0_0.264"></div>
<div><label for="f-username">Username</label>
<input id="f-username" placeholder="admin"></div>
<div><label for="f-password">Password</label>
<input id="f-password" type="password" autocomplete="new-password"></div>
<div><label for="f-max_width">Max width (px)</label>
<input id="f-max_width" type="number" value="1280"></div>
<div><label for="f-webcam">Webcam index (instead of RTSP)</label>
<input id="f-webcam" type="number" placeholder="0"></div>
<div class="wide"><label for="f-url">Full URL (overrides host/port/path)</label>
<input id="f-url" placeholder="rtsp://user:pass@host:554/stream"></div>
</div>
<button class="btn" id="cam-test-btn">Test connection</button>
<button class="btn primary" id="cam-save">Save</button>
<button class="btn" id="cam-cancel">Cancel</button>
<div id="cam-test"></div>
</div>
</div>
</div>
<div>
<div class="card"><h2>Recent events</h2><ul id="events"></ul></div>
<div class="card" style="margin-top:1rem"><h2>Recognition health</h2>
<div id="pipeline" class="muted">no tracks yet</div></div>
<div class="card" style="margin-top:1rem"><h2>Possible duplicates</h2>
<div id="dupes" class="muted">none found</div></div>
<div class="card" style="margin-top:1rem"><h2>People</h2>
<table><thead><tr><th>Label</th><th>Seen</th><th>Last</th></tr></thead>
<tbody id="people"></tbody></table>
</div>
</div>
</div>
<div id="wizard" hidden><div class="panel">
<h3 id="wz-title">Camera placement check</h3>
<div id="wz-body"></div>
<button class="btn" id="wz-close">Close</button>
<button class="btn primary" id="wz-again" hidden>Run again</button>
<button class="btn" id="wz-loosen" hidden>Use this camera's own gate</button>
</div></div>
<script>
const feeds = document.getElementById('feeds');
const fmtTime = ts => new Date(ts * 1000).toLocaleTimeString();
// Identity labels are user-supplied (PATCH /api/identities/{id}) and camera
// ids come from config, so every value interpolated into innerHTML below is
// escaped first. Without this a label like <img onerror=...> is stored XSS.
const esc = v => String(v ?? '').replace(/[&<>"']/g,
c => ({'&':'&amp;','<':'&lt;','>':'&gt;','"':'&quot;',"'":'&#39;'}[c]));
// `camera_id` only exists on cameras whose worker is running - it comes from
// worker.stats(). A stored camera that failed to start has only `id`, and
// using the wrong one put the string "undefined" in the stream URL.
let feedKey = null;
function renderFeeds(cams) {
const key = cams.map(c => c.id).join('|');
// Never re-create a live <img>: assigning src restarts the MJPEG stream, so
// rebuilding on every 3s refresh would make every feed flicker forever.
if (key === feedKey) return;
feedKey = key;
feeds.innerHTML = cams.map(cam => `<figure><img class="feed"
src="/api/cameras/${encodeURIComponent(cam.id)}/stream.mjpeg"
alt="${esc(cam.id)}"><figcaption class="muted">${esc(cam.id)}</figcaption>
</figure>`).join('') || '<span class="muted">no cameras configured</span>';
}
async function boot() {
refresh();
setInterval(refresh, 3000);
}
// Outcome of every finished track. This panel exists because the pipeline
// was previously unfalsifiable from the UI: a camera rejecting every visitor
// on quality looked exactly like a camera nobody walked past.
const OUTCOMES = [
['recognized', '#2e5c3a', 'returning'],
['enrolled', '#2b4e77', 'new'],
['rejected_quality', '#8a3b3b', 'face too poor to enroll'],
['gave_up_ambiguous','#8a6a2a', 'never settled'],
['ended_ambiguous', '#6a5a3a', 'left while unsure'],
['too_brief', '#444', 'gone too fast'],
['no_embedding', '#333', 'never encodable'],
];
function renderPipeline(c) {
const p = c.pipeline;
if (!p || !p.tracks_ended) {
return `<div class="pl-row"><b>${esc(c.camera_id)}</b>
<span class="muted"> — no finished tracks yet</span></div>`;
}
const total = p.tracks_ended;
const bar = OUTCOMES.map(([key, color]) => {
const n = p.outcomes[key] || 0;
return n ? `<i style="width:${(n / total * 100).toFixed(1)}%;
background:${color}" title="${key}: ${n}"></i>` : '';
}).join('');
const legend = OUTCOMES.filter(([k]) => p.outcomes[k]).map(([k, color, human]) =>
`<span><i class="dot" style="background:${color}"></i>${esc(human)}
${esc(p.outcomes[k])}</span>`).join('');
const q = p.best_quality || {};
// The number that says the enrollment gate is wrong for this camera, as
// opposed to the camera being pointed somewhere nobody walks.
const below = q.fraction_below_gate;
const gateWarn = below >= 0.5
? `<div class="warn">⚠ ${(below * 100).toFixed(0)}% of faces are below the
enrollment quality gate — these visitors are seen and discarded.
Fix camera placement before touching thresholds.</div>` : '';
const spread = q.n
? `<div class="muted legend">face quality p05 ${esc(q.p05)} ·
median ${esc(q.p50)} · p95 ${esc(q.p95)}${
below != null ? ` · ${(below * 100).toFixed(0)}% under gate` : ''}</div>`
: '';
return `<div class="pl-row"><b>${esc(c.camera_id)}</b>
<span class="muted"> — ${esc(total)} finished tracks</span>
<div class="bar">${bar}</div>
<div class="muted legend">${legend}</div>${spread}${gateWarn}</div>`;
}
// One person enrolled twice. Merging is irreversible - nothing records which
// embedding came from which identity - so this only ever *suggests*, and the
// operator confirms. Pairs below enroll_threshold are not offered at all.
function renderDupes(pairs) {
if (!pairs.length) return '<span class="muted">none found</span>';
return pairs.map(p => {
// Merge the sparser record into the richer one, and a "Visitor N" into a
// named person, so the surviving identity is the one with more history.
const named = x => x.kind !== 'auto';
let [from, into] = named(p.a) && !named(p.b) ? [p.b, p.a]
: named(p.b) && !named(p.a) ? [p.a, p.b]
: p.a.sighting_count <= p.b.sighting_count ? [p.a, p.b] : [p.b, p.a];
return `<div class="dup">
<span class="grow">${esc(from.label)} <span class="muted">&rarr;</span>
${esc(into.label)}</span>
<span class="muted sim">${esc(p.similarity)}${p.confident ? '' : ' ?'}</span>
<button data-from="${esc(from.id)}" data-into="${esc(into.id)}"
data-desc="${esc(from.label)} into ${esc(into.label)}">Merge</button>
</div>`;
}).join('');
}
document.getElementById('dupes').addEventListener('click', async ev => {
const btn = ev.target.closest('button[data-from]');
if (!btn) return;
const {from, into, desc} = btn.dataset;
if (!confirm(`Merge ${desc}?\n\nThis cannot be undone.`)) return;
btn.disabled = true;
const send = force => fetch(`/api/identities/${encodeURIComponent(from)}/merge`,
{method: 'POST', headers: {'Content-Type': 'application/json'},
body: JSON.stringify({into: Number(into), force})});
let r = await send(false);
if (r.status === 409) {
// The gallery's own numbers say these are different people. Show the
// measured similarity rather than a generic failure - overriding it is a
// decision, and the operator needs the number to make it.
const d = (await r.json()).detail || {};
if (!confirm(`${d.reason || 'Refused.'}\n\nMerge anyway?`)) {
btn.disabled = false; return;
}
r = await send(true);
}
if (!r.ok) alert('Merge failed.');
btn.disabled = false;
refresh();
});
// -- camera settings ------------------------------------------------------
// The API never returns a camera password - not masked, not empty-string-if-
// set, absent. So an edit sends `password` only when the user actually typed
// one; leaving it blank keeps whatever is stored.
const F = ['id', 'host', 'port', 'path', 'username', 'password', 'max_width',
'webcam', 'url'];
const fld = n => document.getElementById('f-' + n);
let editing = null; // camera id being edited, or null when adding
function renderCameras(cams) {
const el = document.getElementById('cam-list');
if (!cams.length) {
el.innerHTML = '<span class="muted">none configured</span>';
return;
}
el.innerHTML = cams.map(c => {
const live = c.connected === undefined ? null : !!c.connected;
const pill = live === null ? '<span class="pill muted">stopped</span>'
: `<span class="pill ${live ? 'up' : 'down'}">${live ? 'live' : 'offline'}</span>`;
return `<div class="cam-row">
<span class="grow"><b>${esc(c.id)}</b>
<span class="muted"> ${esc(c.url)}</span></span>
${pill}
<button class="btn" data-check="${esc(c.id)}">Check placement</button>
<button class="btn" data-edit="${esc(c.id)}">Edit</button>
<button class="btn danger" data-del="${esc(c.id)}">Delete</button>
</div>`;
}).join('');
}
function openForm(cam) {
editing = cam ? cam.id : null;
for (const n of F) fld(n).value = '';
fld('port').value = 554;
fld('max_width').value = 1280;
if (cam) {
for (const n of F) if (cam[n] !== null && cam[n] !== undefined) fld(n).value = cam[n];
fld('password').value = '';
fld('password').placeholder = cam.has_password ? '(unchanged)' : '';
} else {
fld('password').placeholder = '';
}
// The id is the store key and PATCH ignores it; showing it editable would
// imply a rename that silently does nothing.
fld('id').disabled = !!cam;
document.getElementById('cam-test').innerHTML = '';
document.getElementById('cam-form').hidden = false;
}
function closeForm() {
document.getElementById('cam-form').hidden = true;
editing = null;
}
function formBody() {
const body = {};
for (const n of F) {
const v = fld(n).value.trim();
if (v === '') continue; // blank = "leave alone", never "clear"
body[n] = (n === 'port' || n === 'max_width' || n === 'webcam')
? Number(v) : v;
}
return body;
}
async function testCamera() {
const btn = document.getElementById('cam-test-btn');
const out = document.getElementById('cam-test');
btn.disabled = true;
out.innerHTML = '<span class="muted">connecting…</span>';
try {
const r = await fetch('/api/cameras/test', {
method: 'POST', headers: {'Content-Type': 'application/json'},
body: JSON.stringify(formBody())});
const d = await r.json();
if (!d.ok) {
out.innerHTML = `<span class="err">${esc(d.error || 'failed')}</span>`;
} else {
const scaled = d.downscaled_to
? ` <span class="muted">(downscaled to ${esc(d.downscaled_to)}px)</span>` : '';
out.innerHTML = `<span class="ok">connected — ${esc(d.width)}×${esc(d.height)}</span>${scaled}`
+ (d.snapshot ? `<img src="data:image/jpeg;base64,${esc(d.snapshot)}" alt="snapshot">` : '');
}
} catch (e) {
out.innerHTML = '<span class="err">test request failed</span>';
}
btn.disabled = false;
}
async function saveCamera() {
const btn = document.getElementById('cam-save');
const out = document.getElementById('cam-test');
const body = formBody();
if (!editing && !body.id) {
out.innerHTML = '<span class="err">camera id is required</span>';
return;
}
btn.disabled = true;
const r = editing
? await fetch(`/api/cameras/${encodeURIComponent(editing)}`, {
method: 'PATCH', headers: {'Content-Type': 'application/json'},
body: JSON.stringify(body)})
: await fetch('/api/cameras', {
method: 'POST', headers: {'Content-Type': 'application/json'},
body: JSON.stringify(body)});
btn.disabled = false;
if (!r.ok) {
let msg = `save failed (${r.status})`;
try { const d = await r.json(); msg = typeof d.detail === 'string' ? d.detail : msg; } catch (e) {}
out.innerHTML = `<span class="err">${esc(msg)}</span>`;
return;
}
closeForm();
feedKey = null; // a new or edited camera needs its feed rebuilt
refresh();
}
document.getElementById('cam-new').onclick = () => openForm(null);
document.getElementById('cam-cancel').onclick = closeForm;
document.getElementById('cam-test-btn').onclick = testCamera;
document.getElementById('cam-save').onclick = saveCamera;
document.getElementById('cam-list').addEventListener('click', async ev => {
const check = ev.target.closest('button[data-check]');
if (check) { wzStart(check.dataset.check); return; }
const edit = ev.target.closest('button[data-edit]');
if (edit) {
const cams = await (await fetch('/api/cameras')).json();
const cam = cams.find(c => c.id === edit.dataset.edit);
if (cam) openForm(cam);
return;
}
const del = ev.target.closest('button[data-del]');
if (!del) return;
const id = del.dataset.del;
if (!confirm(`Delete camera "${id}"? Recognition from it stops immediately.`)) return;
del.disabled = true;
await fetch(`/api/cameras/${encodeURIComponent(id)}`, {method: 'DELETE'});
if (editing === id) closeForm();
feedKey = null;
refresh();
});
// -- placement wizard -----------------------------------------------------
// The Office1 camera ran for weeks recognising almost nobody, and nothing
// looked broken. This makes that discoverable at install time instead of from
// a footfall report that was always zero.
const wz = document.getElementById('wizard');
let wzCamera = null, wzTimer = null, wzLast = null;
function wzShow(html) { document.getElementById('wz-body').innerHTML = html; }
function wzRender(d) {
wzLast = d;
const pct = Math.round(100 * (d.elapsed / d.seconds));
const q = d.quality || {};
const bar = `<div class="progress"><i style="width:${d.running ? pct : 100}%"></i></div>`;
const advice = (d.advice || []).map(a => `<li>${esc(a)}</li>`).join('');
const numbers = q.n ? `<div class="muted legend">faces ${esc(q.n)} ·
quality p05 ${esc(q.p05)} · median ${esc(q.p50)} · p95 ${esc(q.p95)} ·
gate ${esc(d.gate)} · below gate ${Math.round(100 * q.fraction_below_gate)}%</div>` : '';
wzShow(`<div class="verdict v-${esc(d.verdict)}">${esc(d.headline)}</div>
${bar}<ul class="advice">${advice}</ul>${numbers}`);
document.getElementById('wz-again').hidden = d.running;
// Loosening the gate is only ever offered for a marginal camera. For a poor
// one the answer is to move the camera: dropping the gate there converts a
// visible miss into an invisible wrong match, which is strictly worse.
document.getElementById('wz-loosen').hidden =
!(d.verdict === 'marginal' && q.p05 !== undefined);
}
async function wzPoll() {
const r = await fetch(`/api/cameras/${encodeURIComponent(wzCamera)}/commission`);
if (!r.ok) return;
const d = await r.json();
wzRender(d);
if (!d.running) { clearInterval(wzTimer); wzTimer = null; }
}
async function wzStart(id) {
wzCamera = id;
wz.hidden = false;
document.getElementById('wz-title').textContent = `Placement check — ${id}`;
wzShow('<span class="muted">starting…</span>');
const r = await fetch(`/api/cameras/${encodeURIComponent(id)}/commission`,
{method: 'POST', headers: {'Content-Type': 'application/json'},
body: JSON.stringify({seconds: 25})});
if (!r.ok) { wzShow('<span class="err">could not start — is the camera running?</span>'); return; }
wzRender(await r.json());
if (wzTimer) clearInterval(wzTimer);
wzTimer = setInterval(wzPoll, 1000);
}
document.getElementById('wz-close').onclick = async () => {
if (wzTimer) { clearInterval(wzTimer); wzTimer = null; }
// Cancel rather than leave it running: a check still collecting after the
// installer walked away would keep changing the answer they just read.
if (wzCamera) await fetch(`/api/cameras/${encodeURIComponent(wzCamera)}/commission`,
{method: 'DELETE'});
wz.hidden = true; wzCamera = null;
};
document.getElementById('wz-again').onclick = () => wzStart(wzCamera);
document.getElementById('wz-loosen').onclick = async () => {
// Set this camera's gate just under the faces it actually sees. Per camera
// only - quality describes a view, and every camera writes into one gallery.
const gate = Math.max(0.1, Math.round((wzLast.quality.p05 - 0.02) * 100) / 100);
if (!confirm(`Set ${wzCamera} min_enroll_quality to ${gate}?\n\n` +
`Only this camera is affected.`)) return;
await fetch(`/api/cameras/${encodeURIComponent(wzCamera)}`, {
method: 'PATCH', headers: {'Content-Type': 'application/json'},
body: JSON.stringify({tuning: {min_enroll_quality: gate}})});
wzStart(wzCamera);
};
async function refresh() {
try {
const [stats, events, people, dupes, camList] = await Promise.all([
(await fetch('/api/stats')).json(),
(await fetch('/api/events?limit=30')).json(),
(await fetch('/api/identities?limit=30')).json(),
(await fetch('/api/identities/duplicates?limit=10')).json(),
(await fetch('/api/cameras')).json(),
]);
renderFeeds(camList);
renderCameras(camList);
const cams = stats.cameras.map(c =>
`${c.camera_id}: ${c.connected ? 'live' : 'offline'}`).join(' · ');
// textContent, not innerHTML — no escaping needed here.
document.getElementById('status').textContent =
`${cams} · ${stats.gallery.identities} people · ${stats.gallery.sightings} sightings`;
document.getElementById('events').innerHTML = events.map(e => {
const cls = e.type === 'person.new' ? 'new'
: e.type === 'person.seen' ? 'seen'
: e.type === 'person.missed' ? 'miss'
: e.type === 'identity.merged' ? 'merge' : 'cam';
const who = e.data.label ? ` ${esc(e.data.label)}` : '';
// genderage.onnx reports an integer `age`; the Caffe fallback reports
// a bucketed `age_range`. Show whichever this backend produced.
const age = e.data.age ?? e.data.age_range;
const extra = e.data.gender ? ` · ${esc(e.data.gender)}${age != null ? ', ' + esc(age) : ''}${e.data.emotion ? ', ' + esc(e.data.emotion) : ''}` : '';
return `<li><span class="tag ${cls}">${esc(e.type)}</span>${who}
<span class="muted">${extra} · ${esc(e.camera_id)} · ${fmtTime(e.ts)}</span></li>`;
}).join('');
document.getElementById('pipeline').innerHTML =
stats.cameras.map(renderPipeline).join('');
document.getElementById('dupes').innerHTML = renderDupes(dupes);
document.getElementById('people').innerHTML = people.map(p =>
`<tr><td>${esc(p.label)}</td><td>${esc(p.sighting_count)}</td>
<td class="muted">${p.last_seen_at ? fmtTime(p.last_seen_at) : '—'}</td></tr>`
).join('');
} catch (err) {
document.getElementById('status').textContent = 'api unreachable';
}
}
boot();
</script>
</body>
</html>

119
behavision/tracking.py Normal file
View File

@@ -0,0 +1,119 @@
"""IoU-based multi-face tracker.
Purpose: turn per-frame detections into per-person *tracks* so identity is
decided once per visit, not once per frame (the old backend registered a
new user for every frame). Greedy IoU association is deliberate: faces move
slowly relative to frame rate, and determinism beats a heavier Kalman/
ByteTrack stack for this workload.
"""
from __future__ import annotations
import itertools
import time
from dataclasses import dataclass, field
from typing import Optional
import numpy as np
from .detection import Detection
from .geometry import iou
@dataclass
class Track:
id: int
box: tuple
kps: np.ndarray
score: float
quality: float = 0.0
best_quality: float = 0.0
hits: int = 1
misses: int = 0
created_at: float = field(default_factory=time.time)
updated_at: float = field(default_factory=time.time)
# identity resolution state
state: str = "pending" # pending | resolved | ambiguous | gave_up
# Embeddings are accumulated over multiple frames and averaged before
# any identity decision: single-frame embeddings under extreme pose /
# motion blur are unstable, the mean is not.
emb_sum: Optional[np.ndarray] = None
emb_count: int = 0
id_attempts: int = 0
# Set when THIS track minted the identity, so a terminal tally can tell
# a first-time visitor from a returning one without re-querying the store.
is_new: bool = False
# resolve() refused to enroll this face (quality below the gate). Counted
# rather than ignored: a mis-set gate and an empty room used to look the
# same from outside.
quality_skips: int = 0
last_attempt_ts: float = 0.0
last_reinforce_ts: float = 0.0
reinforcements: int = 0
identity_id: Optional[int] = None
label: Optional[str] = None
similarity: float = 0.0
attributes: dict = field(default_factory=dict)
# The best-quality face crop seen on this track, kept only when
# app.store_faces is on. One small array per live track, replaced rather
# than accumulated; None when images are off, which is the default.
best_face: Optional[np.ndarray] = None
best_face_quality: float = 0.0
attr_samples: list = field(default_factory=list)
class IouTracker:
def __init__(self, iou_threshold: float = 0.3, max_misses: int = 15):
self.iou_threshold = iou_threshold
self.max_misses = max_misses
self.tracks: list[Track] = []
self._ids = itertools.count(1)
def update(self, detections: "list[Detection]", now: "float | None" = None
) -> "tuple[list[Track], list[Track]]":
"""Associate detections to tracks. Returns (active, ended)."""
now = now or time.time()
# Greedy matching on IoU, best pairs first.
pairs = []
for ti, track in enumerate(self.tracks):
for di, det in enumerate(detections):
overlap = iou(track.box, det.box)
if overlap >= self.iou_threshold:
pairs.append((overlap, ti, di))
pairs.sort(reverse=True)
matched_tracks: set[int] = set()
matched_dets: set[int] = set()
for overlap, ti, di in pairs:
if ti in matched_tracks or di in matched_dets:
continue
matched_tracks.add(ti)
matched_dets.add(di)
track, det = self.tracks[ti], detections[di]
track.box = det.box
track.kps = det.kps
track.score = det.score
track.quality = det.quality
track.best_quality = max(track.best_quality, det.quality)
track.hits += 1
track.misses = 0
track.updated_at = now
new_tracks = [
Track(id=next(self._ids), box=det.box, kps=det.kps,
score=det.score, quality=det.quality,
best_quality=det.quality, created_at=now, updated_at=now)
for di, det in enumerate(detections) if di not in matched_dets
]
ended: list[Track] = []
alive: list[Track] = []
for ti, track in enumerate(self.tracks):
if ti not in matched_tracks:
track.misses += 1
if track.misses > self.max_misses:
ended.append(track)
else:
alive.append(track)
self.tracks = alive + new_tracks
return self.tracks, ended