Five components that ship as one product:
- behavision/ the recognition engine. RTSP ingest, YuNet detection, IoU
tracking, ArcFace embeddings, a FAISS/SQLite gallery, and a
FastAPI dashboard. Identity is decided once per TRACK from an
average of at least three embeddings, never per frame.
- agent/ the Go edge agent: supervises the engine, holds a durable
spool, and drains it to MQTT. Nothing is acked before the
broker confirms.
- desktop/ the shop PC application (Wails + React + tray).
- server/ the cloud API, MQTT consumer, reports and assistant.
- web/ platform.loyaly.ai, the head-office app, embedded in the
server binary.
The gallery stores 512-float embeddings and timestamps - no images unless
`app.store_faces` is switched on. Those embeddings are biometric personal
data under GDPR and India's DPDP: template inversion reconstructs a
recognisable face from an ArcFace vector, so data/behavision.db is treated
as a biometric database and DELETE /api/visitors/{id} is a real erasure.
CLAUDE.md carries the reasoning behind every non-obvious decision here,
including the ones that were measured and the ones that were wrong first.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HViLj9gYNRtSr7YVZmW5sn
82 lines
2.7 KiB
Python
82 lines
2.7 KiB
Python
"""Box math and ArcFace 5-point alignment (Umeyama similarity transform)."""
|
|
from __future__ import annotations
|
|
|
|
import cv2
|
|
import numpy as np
|
|
|
|
# Canonical 5-point landmark template for a 112x112 ArcFace crop:
|
|
# left eye, right eye, nose tip, left mouth corner, right mouth corner.
|
|
ARCFACE_TEMPLATE = np.array(
|
|
[
|
|
[38.2946, 51.6963],
|
|
[73.5318, 51.5014],
|
|
[56.0252, 71.7366],
|
|
[41.5493, 92.3655],
|
|
[70.7299, 92.2041],
|
|
],
|
|
dtype=np.float32,
|
|
)
|
|
|
|
|
|
def clip_box(box, width: int, height: int):
|
|
"""Clamp an (x1, y1, x2, y2) box to image bounds.
|
|
|
|
Returns int coords, or None when nothing of the box remains inside the
|
|
frame. This is what prevents negative indices from silently wrapping
|
|
around in numpy slicing.
|
|
"""
|
|
x1, y1, x2, y2 = box
|
|
x1 = int(max(0, min(x1, width)))
|
|
y1 = int(max(0, min(y1, height)))
|
|
x2 = int(max(0, min(x2, width)))
|
|
y2 = int(max(0, min(y2, height)))
|
|
if x2 - x1 < 2 or y2 - y1 < 2:
|
|
return None
|
|
return x1, y1, x2, y2
|
|
|
|
|
|
def iou(a, b) -> float:
|
|
ax1, ay1, ax2, ay2 = a
|
|
bx1, by1, bx2, by2 = b
|
|
ix1, iy1 = max(ax1, bx1), max(ay1, by1)
|
|
ix2, iy2 = min(ax2, bx2), min(ay2, by2)
|
|
iw, ih = max(0.0, ix2 - ix1), max(0.0, iy2 - iy1)
|
|
inter = iw * ih
|
|
if inter <= 0:
|
|
return 0.0
|
|
union = (ax2 - ax1) * (ay2 - ay1) + (bx2 - bx1) * (by2 - by1) - inter
|
|
return float(inter / union) if union > 0 else 0.0
|
|
|
|
|
|
def umeyama(src: np.ndarray, dst: np.ndarray) -> np.ndarray:
|
|
"""Least-squares similarity transform (Umeyama 1991) mapping src -> dst.
|
|
|
|
Deterministic (no RANSAC), which keeps embeddings reproducible for the
|
|
same input frame. Returns a 2x3 affine matrix for cv2.warpAffine.
|
|
"""
|
|
src = np.asarray(src, dtype=np.float64)
|
|
dst = np.asarray(dst, dtype=np.float64)
|
|
n = src.shape[0]
|
|
src_mean, dst_mean = src.mean(0), dst.mean(0)
|
|
src_c, dst_c = src - src_mean, dst - dst_mean
|
|
|
|
cov = dst_c.T @ src_c / n
|
|
u, s, vt = np.linalg.svd(cov)
|
|
d = np.ones(2)
|
|
if np.linalg.det(u) * np.linalg.det(vt) < 0:
|
|
d[1] = -1.0
|
|
rot = u @ np.diag(d) @ vt
|
|
var_src = (src_c ** 2).sum() / n
|
|
scale = (s * d).sum() / var_src if var_src > 1e-12 else 1.0
|
|
t = dst_mean - scale * rot @ src_mean
|
|
return np.hstack([scale * rot, t.reshape(2, 1)]).astype(np.float32)
|
|
|
|
|
|
def align_face(image: np.ndarray, kps: np.ndarray, size: int = 112) -> np.ndarray:
|
|
"""Warp a full frame to a canonical `size`x`size` face chip using the
|
|
5 detected landmarks (full-frame coordinates — the whole point is that
|
|
landmarks and image are in the SAME coordinate space)."""
|
|
template = ARCFACE_TEMPLATE * (size / 112.0)
|
|
m = umeyama(np.asarray(kps, dtype=np.float32), template)
|
|
return cv2.warpAffine(image, m, (size, size), borderValue=0)
|