Behavision: face recognition for retail, edge to head office
Five components that ship as one product:
- behavision/ the recognition engine. RTSP ingest, YuNet detection, IoU
tracking, ArcFace embeddings, a FAISS/SQLite gallery, and a
FastAPI dashboard. Identity is decided once per TRACK from an
average of at least three embeddings, never per frame.
- agent/ the Go edge agent: supervises the engine, holds a durable
spool, and drains it to MQTT. Nothing is acked before the
broker confirms.
- desktop/ the shop PC application (Wails + React + tray).
- server/ the cloud API, MQTT consumer, reports and assistant.
- web/ platform.loyaly.ai, the head-office app, embedded in the
server binary.
The gallery stores 512-float embeddings and timestamps - no images unless
`app.store_faces` is switched on. Those embeddings are biometric personal
data under GDPR and India's DPDP: template inversion reconstructs a
recognisable face from an ArcFace vector, so data/behavision.db is treated
as a biometric database and DELETE /api/visitors/{id} is a real erasure.
CLAUDE.md carries the reasoning behind every non-obvious decision here,
including the ones that were measured and the ones that were wrong first.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HViLj9gYNRtSr7YVZmW5sn
This commit is contained in:
142
behavision/faces.py
Normal file
142
behavision/faces.py
Normal file
@@ -0,0 +1,142 @@
|
||||
"""Saving a face image for one visit — the outbox the agent uploads from.
|
||||
|
||||
This is the one place the engine writes a picture of a person to disk, and it
|
||||
is off unless `app.store_faces` is set. That default is the product's original
|
||||
privacy position, not an oversight: with images off, `data/behavision.db` holds
|
||||
templates and timestamps and nothing that looks like a photograph. Turning them
|
||||
on changes what the system is under GDPR and India's DPDP, so it is a decision
|
||||
someone has to make on purpose.
|
||||
|
||||
The engine does NOT upload. It writes a file and names it on the event; the
|
||||
agent uploads through a short-lived URL the server mints. A shop PC therefore
|
||||
never holds object-storage credentials — the bucket is shared and a counter-top
|
||||
machine is the least trustworthy thing in the estate.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import time
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
|
||||
import cv2
|
||||
import numpy as np
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
# A loose crop, not the aligned 112x112 chip.
|
||||
#
|
||||
# The chip is built for ArcFace: tight, warped to canonical landmarks, and
|
||||
# nearly useless to a human trying to recognise a customer. This is the frame a
|
||||
# person looks at, so it gets the same 1.5x head crop the attribute models use.
|
||||
CROP_SCALE = 1.5
|
||||
# Enough to see a face on a dashboard, small enough that a shop on ADSL can
|
||||
# upload one per visitor without the queue backing up. ~15-25 KB at q80.
|
||||
MAX_EDGE = 320
|
||||
JPEG_QUALITY = 80
|
||||
|
||||
|
||||
def _loose_crop(frame: np.ndarray, box, scale: float = CROP_SCALE) -> np.ndarray:
|
||||
"""Square crop around the head, replicate-padded when it runs off-frame."""
|
||||
x1, y1, x2, y2 = (float(v) for v in box)
|
||||
cx, cy = (x1 + x2) / 2.0, (y1 + y2) / 2.0
|
||||
half = max(x2 - x1, y2 - y1) * scale / 2.0
|
||||
left, top = int(round(cx - half)), int(round(cy - half))
|
||||
right, bottom = int(round(cx + half)), int(round(cy + half))
|
||||
|
||||
h, w = frame.shape[:2]
|
||||
pad_l, pad_t = max(0, -left), max(0, -top)
|
||||
pad_r, pad_b = max(0, right - w), max(0, bottom - h)
|
||||
crop = frame[max(0, top):min(h, bottom), max(0, left):min(w, right)]
|
||||
if crop.size == 0:
|
||||
return frame
|
||||
if pad_l or pad_t or pad_r or pad_b:
|
||||
crop = cv2.copyMakeBorder(crop, pad_t, pad_b, pad_l, pad_r,
|
||||
cv2.BORDER_REPLICATE)
|
||||
return crop
|
||||
|
||||
|
||||
class FaceOutbox:
|
||||
"""Writes one JPEG per resolved visit for the agent to collect.
|
||||
|
||||
Files land in `data_dir/outbox`, which is deliberately NOT inside the
|
||||
database directory: it is transient, the agent deletes each file after a
|
||||
successful upload, and a backup of the database must not quietly start
|
||||
including face images.
|
||||
"""
|
||||
|
||||
def __init__(self, data_dir: Path, enabled: bool,
|
||||
max_files: int = 500) -> None:
|
||||
self.enabled = enabled
|
||||
self.dir = Path(data_dir) / "outbox"
|
||||
# Bounded. If the agent stops collecting — not running, no credentials,
|
||||
# server unreachable for a week — this must not fill a shop's disk with
|
||||
# pictures of its customers. Dropping the oldest is right: a stale
|
||||
# photo of a visit already reported is the least valuable thing here.
|
||||
self.max_files = max_files
|
||||
if enabled:
|
||||
self.dir.mkdir(parents=True, exist_ok=True)
|
||||
log.warning(
|
||||
"app.store_faces is ON: face images are being written to %s. "
|
||||
"This changes what this machine holds under GDPR/DPDP.",
|
||||
self.dir)
|
||||
|
||||
def crop(self, frame: np.ndarray, box) -> "np.ndarray | None":
|
||||
"""The candidate image for one frame, downscaled and nothing else.
|
||||
|
||||
Kept as an array rather than encoded here because this runs on every
|
||||
frame of every track: JPEG encoding per frame is milliseconds spent to
|
||||
throw away all but the last one. At 320 px a crop is ~300 KB, so one
|
||||
per live track is affordable even on the 16 GB box that already OOMs on
|
||||
a 250 MB model — holding whole 1280x720 frames instead would not be.
|
||||
"""
|
||||
if not self.enabled:
|
||||
return None
|
||||
try:
|
||||
crop = _loose_crop(frame, box)
|
||||
h, w = crop.shape[:2]
|
||||
if min(h, w) <= 0:
|
||||
return None
|
||||
if max(h, w) > MAX_EDGE:
|
||||
s = MAX_EDGE / float(max(h, w))
|
||||
crop = cv2.resize(crop, (max(1, int(w * s)), max(1, int(h * s))),
|
||||
interpolation=cv2.INTER_AREA)
|
||||
# A copy, because the slice from _loose_crop can be a view onto the
|
||||
# capture buffer, which the capture thread overwrites in place.
|
||||
return np.ascontiguousarray(crop)
|
||||
except Exception:
|
||||
log.exception("could not build a face crop")
|
||||
return None
|
||||
|
||||
def save(self, crop: "np.ndarray | None") -> "str | None":
|
||||
"""Write the crop and return its path, or None if images are off."""
|
||||
if not self.enabled or crop is None:
|
||||
return None
|
||||
try:
|
||||
path = self.dir / f"{time.time():.3f}_{uuid.uuid4().hex}.jpg"
|
||||
ok, buf = cv2.imencode(".jpg", crop,
|
||||
[int(cv2.IMWRITE_JPEG_QUALITY), JPEG_QUALITY])
|
||||
if not ok:
|
||||
return None
|
||||
# Write-then-rename. The agent watches this directory, and a
|
||||
# partially written JPEG that it picks up mid-write is an upload of
|
||||
# a corrupt file that nothing will ever correct.
|
||||
tmp = path.with_suffix(".part")
|
||||
tmp.write_bytes(buf.tobytes())
|
||||
tmp.replace(path)
|
||||
self._trim()
|
||||
return str(path)
|
||||
except Exception:
|
||||
# Never take the recognition loop down over a photo. A missing
|
||||
# image is a cosmetic loss; a stalled worker is the product.
|
||||
log.exception("could not write a face image")
|
||||
return None
|
||||
|
||||
def _trim(self) -> None:
|
||||
try:
|
||||
files = sorted(self.dir.glob("*.jpg"), key=lambda p: p.stat().st_mtime)
|
||||
for stale in files[:-self.max_files]:
|
||||
stale.unlink(missing_ok=True)
|
||||
except Exception:
|
||||
log.debug("outbox trim failed", exc_info=True)
|
||||
Reference in New Issue
Block a user