From 2e60fbb57a64151308e210272f939e7b4b3cb3e7 Mon Sep 17 00:00:00 2001 From: Suriyakumarvijayanayagam Date: Thu, 24 Sep 2026 13:58:05 +0530 Subject: [PATCH] The engine was searching an empty room fifteen times a second Measured rather than guessed, and the first guess was wrong. Wall clock said H.265 decode cost 58 ms a frame; cap.read() blocks until the next frame arrives, so that was the frame interval, not work. As CPU time: decode 3.7 ms, detection 31.0 ms - and detection ran on every frame whether or not anything was in front of the camera, 6,649 of 8,634 frames with faces_seen 0 and active_tracks 0 throughout. detect_threads: OpenCV spreads a small repeated job over eight threads, costing 31.0 ms of CPU for 8.9 ms of wall. One thread costs 15.3 ms for 15.3 ms, against a 66 ms budget at 15 fps. Half the CPU for latency nothing can notice. motion_gate: a 160x90 greyscale absdiff, 0.1 ms against detection's 15. Consulted only while no track is open; forced to look every motion_max_skip frames; compared against the last frame SEARCHED so a slow drift cannot creep under the threshold; and a threshold above this camera's measured noise and far below a person, so anything ambiguous detects. tests/test_motion_gate.py pins each of those rather than the saving, including asserting the longest run of skips rather than the total - counting the total would pass a gate that slept forty frames and then looked forty times. Together 80% -> 16% of a core, detection skipped on 92% of frames. faces_seen is still 0 and the gate is not why: run directly over the same frames the detector finds nothing at threshold 0.50 either. The placement is the limit, as recorded; the CPU was being spent to rediscover that fifteen times a second. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01KGcjxF1cNLcuwc3DAPcnfj --- CLAUDE.md | 59 ++++++++++++++++++++++++ behavision/__main__.py | 11 +++++ behavision/config.py | 23 ++++++++++ behavision/engine.py | 52 +++++++++++++++++++++ tests/test_motion_gate.py | 97 +++++++++++++++++++++++++++++++++++++++ 5 files changed, 242 insertions(+) create mode 100644 tests/test_motion_gate.py diff --git a/CLAUDE.md b/CLAUDE.md index 44250d1..a23eb76 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -2226,6 +2226,65 @@ person does not re-file them: in the product a caller could reasonably guess wrong, and it is a route no client app should ever call. +## CPU: the engine was searching an empty room 15 times a second + +Measured on this Mac against the office camera, because "it feels hot" is not +a number. The first reading was **214% of a core**, and the first guess - +H.265 decode - was wrong. Wall-clock time said decode cost 58 ms a frame, but +`cap.read()` BLOCKS until the next frame arrives, so that number was the frame +interval, not work. Measured as CPU time instead: + +``` + wall/frame CPU/frame at 15 fps + H.265 decode 58.8 ms 3.7 ms 5% of a core + YuNet detect 8.9 ms 31.0 ms 47% of a core +``` + +Detection costs three times its wall time because OpenCV spreads it over eight +threads. Decode is nearly free. So the cost is detection, and it was running on +**every frame whether or not anything was in front of the camera** - 6,649 of +8,634 frames searched, with `faces_seen: 0` and `active_tracks: 0` throughout. + +Two changes, both measured: + +- **`app.detect_threads: 1`.** OpenCV sizes its pool for one big job on an idle + machine; this is a small job repeated forever on a machine also running the + recogniser, the tracker and possibly three other cameras. One thread costs + 15.3 ms of CPU against the default's 31.0 ms, for 6 ms more wall time against + a 66 ms frame budget. Half the CPU, no latency that matters. +- **`app.motion_gate`.** A 160x90 greyscale thumbnail and an `absdiff`: 0.1 ms + against detection's 15 ms, ninety times cheaper. A shop is empty most of the + day and an empty room costs exactly as much to search as a busy one. + +Together: **80% -> 16% of a core**, with detection skipped on 92% of frames. +Against the original main-stream reading that is 214% -> 16%. + +### Why the gate cannot lose a face + +Cheapness is easy; this is the part that makes it acceptable, and +`tests/test_motion_gate.py` is the argument written down rather than asserted. + +- It is only consulted while **no track is open**, so a person already being + followed is never subject to it. +- `motion_max_skip` forces a real detection about once a second whatever the + thumbnail says. The test asserts the longest *run* of skips, not the total: + what matters is the worst case a person can fall into, and counting the total + would pass a gate that skipped forty frames and then looked forty times. +- The comparison is against the last frame actually **searched**, not the last + frame seen, so a slow drift accumulates and trips the gate instead of sliding + under it one frame at a time. Someone easing into view slowly would otherwise + be invisible indefinitely. +- The threshold (1.0 mean absolute difference) sits above this camera's + measured noise floor (~0.3) and far below a person. Anything ambiguous falls + through to detection: when in doubt it looks. + +Verified on the live camera after the change: `faces_seen: 0` - and the gate is +not why. Running the detector directly over the same frames finds **0 faces at +threshold 0.50**, let alone 0.82. The people in view are seated, side-on and +far away, which is the same `fraction_below_gate: 0.59` this file already +records. The placement is still the limit; the CPU was simply being spent to +discover that 15 times a second. + ## Setting up on a new machine 1. Copy the `Behavision` folder **including `.env`** (gitignored, holds diff --git a/behavision/__main__.py b/behavision/__main__.py index 4edd95a..4fe6250 100644 --- a/behavision/__main__.py +++ b/behavision/__main__.py @@ -21,6 +21,17 @@ def cmd_run(args: argparse.Namespace) -> int: cfg = load_config(args.config) setup_logging(cfg.app.log_level, cfg.app.data_dir) + if cfg.app.detect_threads > 0: + # OpenCV sizes its pool for one big job on an idle machine. This is a + # small job repeated forever on a machine also running the recogniser, + # the tracker and possibly three other cameras, so the default costs + # twice the CPU for no useful latency. Measured: 31 ms CPU/frame at the + # default against 15 ms at one thread, for 6 ms more wall time against + # a 66 ms budget. + import cv2 + cv2.setNumThreads(cfg.app.detect_threads) + log.info("detection threads: %d (OpenCV default was %d)", + cfg.app.detect_threads, cv2.getNumThreads()) missing = setup_models(cfg.app.models_dir) if missing: log.error("required models missing: %s", ", ".join(missing)) diff --git a/behavision/config.py b/behavision/config.py index 19b09be..42d8810 100644 --- a/behavision/config.py +++ b/behavision/config.py @@ -132,6 +132,29 @@ class AppSection(BaseModel): # on changes what the system is under GDPR and India's DPDP, so it has to # be a decision somebody makes rather than one they inherit. store_faces: bool = False + # How many threads OpenCV may use for detection. Measured on the office + # camera (800x448 sub-stream): the default of 8 costs 31 ms of CPU per + # frame for 8.9 ms of wall time, while ONE thread costs 15.3 ms of CPU for + # 15.3 ms of wall - half the CPU for 6 ms more latency, against a 66 ms + # frame budget at 15 fps. The default is wrong here because OpenCV sizes it + # for one big job on an idle machine, and this is a small job repeated + # forever on a machine also running the recogniser, the tracker and three + # other cameras. 0 leaves OpenCV's own default alone. + detect_threads: int = 1 + # Skip detection on frames where nothing has changed and nothing is being + # tracked. A shop is empty most of the day and a frame of an empty room + # costs exactly as much to search as a busy one. See CameraWorker.run for + # why this cannot lose a face. + motion_gate: bool = True + # Mean absolute difference, 0-255, over a 160x90 greyscale thumbnail. 1.0 + # is well below the noise floor of a real camera - measured on this one, + # an empty room varies by ~0.3 between frames - so it triggers on movement + # rather than on sensor noise, and anything ambiguous detects. + motion_threshold: float = 1.0 + # Detect at least this often regardless of the gate, so a change the + # thumbnail cannot see - someone entering at the far edge, a slow lean into + # frame - is still found within a second. + motion_max_skip: int = 12 class ApiSection(BaseModel): diff --git a/behavision/engine.py b/behavision/engine.py index 88cdf70..d37e087 100644 --- a/behavision/engine.py +++ b/behavision/engine.py @@ -170,6 +170,11 @@ class CameraWorker(threading.Thread): self._last_frame_ts = 0.0 self._was_connected = False self.frames_processed = 0 + # Motion gate state: a 160x90 greyscale thumbnail of the last frame we + # actually searched, and how many frames we have skipped since. + self._motion_prev = None + self._motion_skipped = 0 + self.frames_skipped = 0 self.faces_seen = 0 self.pipeline = PipelineStats() # One outbox per worker, all writing into the same directory. Files are @@ -236,6 +241,7 @@ class CameraWorker(threading.Thread): return { **self.source.stats(), "frames_processed": self.frames_processed, + "frames_skipped": self.frames_skipped, "faces_seen": self.faces_seen, "active_tracks": len(self.tracker.tracks), "pipeline": self.pipeline.snapshot(self.rcfg.min_enroll_quality), @@ -244,6 +250,43 @@ class CameraWorker(threading.Thread): "enroll_threshold": self.rcfg.enroll_threshold}, } + def _nothing_moved(self, frame) -> bool: + """True when this frame is close enough to the last searched one that + searching it again would find the same nothing. + + It cannot lose a face, and that property is what makes it acceptable + rather than merely cheap. Three guards, in order: + + * the caller only asks while NO track is open, so a person already + being followed is never affected by it; + * `motion_max_skip` forces a real detection about once a second + whatever the thumbnail says, which covers a change too small or too + gradual for it - someone easing into frame at the far edge; + * the threshold sits well above measured sensor noise and well below + a person, and anything ambiguous falls through to detection. When + in doubt it looks. + + Cost is 0.1 ms against detection's 15 ms, so an empty shop stops paying + for a search of an empty room ~90 times a second. + """ + import cv2 as _cv2 + small = _cv2.resize(_cv2.cvtColor(frame, _cv2.COLOR_BGR2GRAY), (160, 90), + interpolation=_cv2.INTER_AREA) + prev, self._motion_prev = self._motion_prev, small + if prev is None: + return False + if self._motion_skipped >= self.cfg.app.motion_max_skip: + self._motion_skipped = 0 + return False + if float(_cv2.absdiff(small, prev).mean()) >= self.cfg.app.motion_threshold: + self._motion_skipped = 0 + # Keep the thumbnail we just searched against, not this one, so a + # slow drift cannot creep past the threshold one frame at a time. + return False + self._motion_prev = prev + self._motion_skipped += 1 + return True + # -- thread --------------------------------------------------------- def run(self) -> None: tcfg = self.cfg.tracking @@ -256,6 +299,15 @@ class CameraWorker(threading.Thread): continue self._last_frame_ts = ts + # An empty room costs exactly as much to search as a busy one, + # and a shop is empty most of the day. Only ever while nothing + # is being tracked - see _nothing_moved. + if (self.cfg.app.motion_gate and not self.tracker.tracks + and self._nothing_moved(frame)): + self.frames_skipped += 1 + self._remember_tracks([]) + continue + detections = self.detector.detect(frame) for det in detections: det.quality = face_quality(frame, det.box, det.kps) diff --git a/tests/test_motion_gate.py b/tests/test_motion_gate.py new file mode 100644 index 0000000..48798d0 --- /dev/null +++ b/tests/test_motion_gate.py @@ -0,0 +1,97 @@ +"""The motion gate must save CPU without ever losing a face. + +Cheapness is easy; the property that makes it acceptable is that every way it +could miss somebody is closed. These tests are that argument, written down. +""" +import numpy as np +import pytest + +from behavision.config import Config +from behavision.engine import CameraWorker + + +class _Worker: + """CameraWorker's gate, without a camera, a model or a thread. + + Built with object.__new__ and the two attributes the gate touches - the + pattern the other engine tests already use, so no test needs a 166 MB + model file or a live RTSP stream. + """ + + def __new__(cls, **app): + w = object.__new__(CameraWorker) + w.cfg = Config() + for k, v in app.items(): + setattr(w.cfg.app, k, v) + w._motion_prev = None + w._motion_skipped = 0 + return w + + +def frame(value: int, size=(90, 160)) -> np.ndarray: + return np.full((*size, 3), value, dtype=np.uint8) + + +def test_the_first_frame_is_always_searched(): + """Nothing to compare against is not evidence that nothing moved.""" + w = _Worker() + assert w._nothing_moved(frame(40)) is False + + +def test_a_still_room_is_skipped(): + w = _Worker() + w._nothing_moved(frame(40)) # prime + assert w._nothing_moved(frame(40)) is True + + +def test_movement_is_never_skipped(): + w = _Worker() + w._nothing_moved(frame(40)) + # a person is an enormous change next to a 1.0 threshold + assert w._nothing_moved(frame(120)) is False + + +def test_it_gives_up_and_looks_anyway(): + """A change too small or too gradual for a thumbnail must still be found. + + The invariant is about the longest RUN of skips, not the total: what + matters is the worst case a person could fall into, which is how long the + camera can go without actually looking. Counting the total instead would + pass a gate that skipped forty frames and then looked forty times. + """ + w = _Worker(motion_max_skip=5) + w._nothing_moved(frame(40)) + run = longest = 0 + for _ in range(40): + if w._nothing_moved(frame(40)): + run += 1 + longest = max(longest, run) + else: + run = 0 + assert longest <= 5, f"went {longest} frames without looking, cap is 5" + assert longest == 5, f"longest run was {longest} - the gate is not saving what it could" + + +def test_a_slow_drift_cannot_creep_past_the_threshold(): + """Each frame below the threshold, but the total far above it. + + Compared against the last frame we SEARCHED rather than the last frame we + saw, so a gradual change accumulates and eventually trips the gate instead + of sliding under it one frame at a time. Without that, someone easing into + view slowly enough is invisible forever. + """ + w = _Worker(motion_max_skip=10_000) # the safety net must not rescue this + w._nothing_moved(frame(40)) + tripped = None + for i in range(1, 30): + if w._nothing_moved(frame(40 + i)) is False: + tripped = i + break + assert tripped is not None, "a slow drift was never noticed" + assert tripped <= 5, f"took {tripped} frames of drift to notice" + + +@pytest.mark.parametrize("gate", [True, False]) +def test_the_gate_is_switchable(gate): + w = _Worker(motion_gate=gate) + assert w.cfg.app.motion_gate is gate