Measured rather than guessed, and the first guess was wrong. Wall clock said H.265 decode cost 58 ms a frame; cap.read() blocks until the next frame arrives, so that was the frame interval, not work. As CPU time: decode 3.7 ms, detection 31.0 ms - and detection ran on every frame whether or not anything was in front of the camera, 6,649 of 8,634 frames with faces_seen 0 and active_tracks 0 throughout. detect_threads: OpenCV spreads a small repeated job over eight threads, costing 31.0 ms of CPU for 8.9 ms of wall. One thread costs 15.3 ms for 15.3 ms, against a 66 ms budget at 15 fps. Half the CPU for latency nothing can notice. motion_gate: a 160x90 greyscale absdiff, 0.1 ms against detection's 15. Consulted only while no track is open; forced to look every motion_max_skip frames; compared against the last frame SEARCHED so a slow drift cannot creep under the threshold; and a threshold above this camera's measured noise and far below a person, so anything ambiguous detects. tests/test_motion_gate.py pins each of those rather than the saving, including asserting the longest run of skips rather than the total - counting the total would pass a gate that slept forty frames and then looked forty times. Together 80% -> 16% of a core, detection skipped on 92% of frames. faces_seen is still 0 and the gate is not why: run directly over the same frames the detector finds nothing at threshold 0.50 either. The placement is the limit, as recorded; the CPU was being spent to rediscover that fifteen times a second. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01KGcjxF1cNLcuwc3DAPcnfj
273 lines
10 KiB
Python
273 lines
10 KiB
Python
"""CLI: python -m behavision {run | enroll | setup-models}"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import logging
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
from .config import ensure_api_credentials, load_config
|
|
from .log import setup_logging
|
|
|
|
log = logging.getLogger("behavision")
|
|
|
|
|
|
def cmd_run(args: argparse.Namespace) -> int:
|
|
import uvicorn
|
|
|
|
from .api import create_app
|
|
from .engine import Engine
|
|
from .model_assets import setup_models
|
|
|
|
cfg = load_config(args.config)
|
|
setup_logging(cfg.app.log_level, cfg.app.data_dir)
|
|
if cfg.app.detect_threads > 0:
|
|
# OpenCV sizes its pool for one big job on an idle machine. This is a
|
|
# small job repeated forever on a machine also running the recogniser,
|
|
# the tracker and possibly three other cameras, so the default costs
|
|
# twice the CPU for no useful latency. Measured: 31 ms CPU/frame at the
|
|
# default against 15 ms at one thread, for 6 ms more wall time against
|
|
# a 66 ms budget.
|
|
import cv2
|
|
cv2.setNumThreads(cfg.app.detect_threads)
|
|
log.info("detection threads: %d (OpenCV default was %d)",
|
|
cfg.app.detect_threads, cv2.getNumThreads())
|
|
missing = setup_models(cfg.app.models_dir)
|
|
if missing:
|
|
log.error("required models missing: %s", ", ".join(missing))
|
|
return 1
|
|
|
|
auth_on, generated = ensure_api_credentials(cfg)
|
|
if generated:
|
|
log.warning(
|
|
"no API credentials configured - generated one for %s:%s\n"
|
|
" username: %s\n password: %s\n"
|
|
" (saved to %s; set BEHAVISION_API_USER / "
|
|
"BEHAVISION_API_PASSWORD in .env to choose your own)",
|
|
cfg.api.host, cfg.api.port, cfg.api.username, cfg.api.password,
|
|
cfg.app.data_dir / "api_credentials.txt")
|
|
elif not auth_on:
|
|
log.info("API bound to %s - serving without authentication",
|
|
cfg.api.host)
|
|
|
|
engine = Engine(cfg)
|
|
if not engine.workers:
|
|
# A fresh install legitimately has no cameras - the user adds them
|
|
# from the dashboard. Refusing to boot here would mean they could
|
|
# never reach the UI that adds the first one.
|
|
log.info("no cameras yet - add one at http://%s:%s",
|
|
"localhost" if cfg.api.is_loopback else cfg.api.host,
|
|
cfg.api.port)
|
|
engine.start()
|
|
try:
|
|
uvicorn.run(create_app(engine), host=cfg.api.host, port=cfg.api.port,
|
|
log_level="warning")
|
|
finally:
|
|
engine.stop()
|
|
return 0
|
|
|
|
|
|
def cmd_enroll(args: argparse.Namespace) -> int:
|
|
import cv2
|
|
|
|
from .detection import FaceDetector
|
|
from .gallery import Gallery, IdentityStore, VectorIndex
|
|
from .recognition import EMBEDDING_DIM, ArcFaceEncoder, face_quality
|
|
|
|
cfg = load_config(args.config)
|
|
setup_logging(cfg.app.log_level)
|
|
detector = FaceDetector(cfg.app.models_dir,
|
|
cfg.detection.score_threshold,
|
|
cfg.detection.nms_threshold)
|
|
encoder = ArcFaceEncoder(cfg.app.models_dir, cfg.recognition.model_file)
|
|
store = IdentityStore(cfg.app.data_dir / "behavision.db")
|
|
gallery = Gallery(store, VectorIndex(EMBEDDING_DIM), cfg.recognition,
|
|
encoder.model_name)
|
|
|
|
paths: list[Path] = []
|
|
for p in args.images:
|
|
p = Path(p)
|
|
if p.is_dir():
|
|
paths += [f for f in sorted(p.iterdir())
|
|
if f.suffix.lower() in (".jpg", ".jpeg", ".png", ".bmp")]
|
|
else:
|
|
paths.append(p)
|
|
|
|
embeddings = []
|
|
for path in paths:
|
|
image = cv2.imread(str(path))
|
|
if image is None:
|
|
log.warning("unreadable image skipped: %s", path)
|
|
continue
|
|
detections = detector.detect(image)
|
|
if not detections:
|
|
log.warning("no face found in %s", path)
|
|
continue
|
|
best = max(detections, key=lambda d: (d.box[2] - d.box[0])
|
|
* (d.box[3] - d.box[1]))
|
|
emb = encoder.encode(image, best.kps)
|
|
if emb is None:
|
|
log.warning("could not embed face in %s", path)
|
|
continue
|
|
q = face_quality(image, best.box, best.kps)
|
|
embeddings.append(emb)
|
|
log.info("embedded %s (quality %.2f)", path.name, q)
|
|
|
|
if not embeddings:
|
|
log.error("no usable faces - nothing enrolled")
|
|
return 1
|
|
identity_id = gallery.enroll(args.name, embeddings)
|
|
log.info("enrolled '%s' as identity %d with %d embedding(s)",
|
|
args.name, identity_id, len(embeddings))
|
|
store.close()
|
|
return 0
|
|
|
|
|
|
def cmd_calibrate(args: argparse.Namespace) -> int:
|
|
"""Measure the similarity distributions this camera+encoder actually
|
|
produce, then report thresholds that separate them."""
|
|
from .calibrate import CalibrationStore, capture, format_report
|
|
from .detection import FaceDetector
|
|
from .recognition import MODEL_CANDIDATES, ArcFaceEncoder
|
|
|
|
cfg = load_config(args.config)
|
|
setup_logging(cfg.app.log_level)
|
|
store = CalibrationStore(cfg.app.data_dir / "calibration.npz")
|
|
|
|
if args.report:
|
|
if not store.models():
|
|
log.error("no samples yet - run: python -m behavision calibrate "
|
|
"--person NAME")
|
|
return 1
|
|
print(format_report(store, cfg))
|
|
return 0
|
|
|
|
if not args.person:
|
|
log.error("give --person NAME to capture, or --report to analyse")
|
|
return 1
|
|
|
|
# Every model that is present gets embedded from the SAME frames, so an
|
|
# A/B between encoders is a fair comparison rather than two sessions.
|
|
names = [args.model] if args.model else MODEL_CANDIDATES
|
|
encoders = {}
|
|
for name in names:
|
|
if not (cfg.app.models_dir / name).exists():
|
|
continue
|
|
try:
|
|
enc = ArcFaceEncoder(cfg.app.models_dir, name,
|
|
cfg.recognition.color_order)
|
|
encoders[enc.model_name] = enc
|
|
except Exception:
|
|
log.warning("%s did not load - skipping", name)
|
|
if not encoders:
|
|
log.error("no recognition model loaded from %s", cfg.app.models_dir)
|
|
return 1
|
|
log.info("calibrating with: %s", ", ".join(encoders))
|
|
|
|
detector = FaceDetector(cfg.app.models_dir, cfg.detection.score_threshold,
|
|
cfg.detection.nms_threshold, cfg.detection.max_faces,
|
|
cfg.detection.min_face_px)
|
|
|
|
source = args.source
|
|
if source is None:
|
|
cam = cfg.cameras[0] if cfg.cameras else None
|
|
if cam is None:
|
|
log.error("no cameras configured - pass --source")
|
|
return 1
|
|
source = cam.source()
|
|
log.info("capturing '%s' for %.0fs - vary pose, distance and expression",
|
|
args.person, args.seconds)
|
|
|
|
try:
|
|
# Deliberately ungated: the enrollment gate is one of the things being
|
|
# calibrated, and filtering by it here would make it unmeasurable.
|
|
samples, qualities = capture(source, args.person, args.seconds, cfg,
|
|
detector, encoders)
|
|
except RuntimeError:
|
|
log.exception("capture failed")
|
|
return 1
|
|
|
|
kept = 0
|
|
for model, embeddings in samples.items():
|
|
if len(embeddings):
|
|
kept = store.add(model, args.person, embeddings, qualities)
|
|
if not kept:
|
|
log.error("no usable faces captured for '%s' - nothing stored "
|
|
"(nobody in frame, two faces at once, or too far away?)",
|
|
args.person)
|
|
return 1
|
|
store.save()
|
|
log.info("stored %d embeddings for '%s' (total per model). Capture more "
|
|
"people, then: python -m behavision calibrate --report",
|
|
kept, args.person)
|
|
return 0
|
|
|
|
|
|
def cmd_setup_models(args: argparse.Namespace) -> int:
|
|
from .model_assets import setup_models
|
|
|
|
cfg = load_config(args.config)
|
|
setup_logging(cfg.app.log_level)
|
|
missing = setup_models(cfg.app.models_dir)
|
|
if missing:
|
|
log.error("still missing (place them in %s manually): %s",
|
|
cfg.app.models_dir, ", ".join(missing))
|
|
return 1
|
|
log.info("all required models present in %s", cfg.app.models_dir)
|
|
return 0
|
|
|
|
|
|
def cmd_paths(args) -> int:
|
|
"""Where everything lives. An installer and a support call both need this,
|
|
and installed it is not next to the code."""
|
|
from .paths import describe
|
|
|
|
# load_config first: it seeds the editable copy, and describing the config
|
|
# path before that would name the bundled file rather than the one the next
|
|
# run actually loads.
|
|
cfg = load_config(args.config)
|
|
info = describe()
|
|
info["data_dir"] = str(cfg.app.data_dir)
|
|
info["models_dir"] = str(cfg.app.models_dir)
|
|
width = max(len(k) for k in info)
|
|
for key, value in info.items():
|
|
print(f"{key.rjust(width)} : {value}")
|
|
return 0
|
|
|
|
|
|
def main() -> int:
|
|
parser = argparse.ArgumentParser(
|
|
prog="behavision", description="Face recognition over RTSP")
|
|
parser.add_argument("--config", default=None,
|
|
help="path to YAML config (default: config/default.yaml)")
|
|
sub = parser.add_subparsers(dest="command", required=True)
|
|
|
|
sub.add_parser("run", help="start the pipeline + API server")
|
|
enroll = sub.add_parser("enroll", help="enroll a person from images")
|
|
enroll.add_argument("--name", required=True)
|
|
enroll.add_argument("--images", nargs="+", required=True,
|
|
help="image files and/or directories")
|
|
sub.add_parser("setup-models", help="download/copy model files")
|
|
sub.add_parser("paths", help="show where config, data and models live")
|
|
cal = sub.add_parser(
|
|
"calibrate",
|
|
help="measure similarity distributions and recommend thresholds")
|
|
cal.add_argument("--person", help="label for this capture session")
|
|
cal.add_argument("--seconds", type=float, default=20.0)
|
|
cal.add_argument("--source", default=None,
|
|
help="capture source (default: first configured camera)")
|
|
cal.add_argument("--model", default=None,
|
|
help="only this model file (default: all present)")
|
|
cal.add_argument("--report", action="store_true",
|
|
help="analyse stored samples instead of capturing")
|
|
|
|
args = parser.parse_args()
|
|
handlers = {"run": cmd_run, "enroll": cmd_enroll,
|
|
"setup-models": cmd_setup_models, "calibrate": cmd_calibrate,
|
|
"paths": cmd_paths}
|
|
return handlers[args.command](args)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|