The certificate fix shipped and failed on the machine it was written for, with
the exact traceback it was meant to prevent. The retry was written
except ssl.SSLCertVerificationError:
and urllib never raises that from urlopen. It catches it and re-raises
urllib.error.URLError(err), carrying the original on .reason. So the except
matched nothing, ever, and the fallback could not fire.
The unit test passed throughout, because the stub it used raised the bare SSL
error - a shape real urllib never produces. That is the lesson: a fake that
agrees with the author is worse than no test, because it converts an untested
path into a tested-looking one. This file already says that about
UPDATE ... RETURNING and about the in-memory API fake, and it got written
again anyway.
_is_cert_failure checks the exception and its .reason, and the tests now raise
URLError(SSLCertVerificationError(...)) - what the traceback actually shows. A
plain URLError is re-raised untouched, and a test asserts no second attempt is
made for one.
Beside the stubs there is now a real reproduction, opt-in behind
BEHAVISION_NETWORK_TESTS=1. Python's default context honours SSL_CERT_FILE, so
an empty file gives a context that trusts nobody - the python.org condition
exactly - while certifi is loaded by path and is unaffected. It skips rather
than passes where it cannot reproduce that, and the difference is measured:
macOS Command Line Tools LibreSSL 2.8.3 128 CAs with an empty CA file
python.org / pyenv build OpenSSL 3.5.8 0 CAs -> reproduces it
Checked for teeth by putting the shipped except back: both the corrected stub
test and the live one fail, and pass again when it is restored.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01KGcjxF1cNLcuwc3DAPcnfj
237 lines
10 KiB
Python
237 lines
10 KiB
Python
"""Model acquisition: download YuNet, copy reusable models from the old
|
|
projects on this machine when present. Idempotent — safe to re-run."""
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import shutil
|
|
import ssl
|
|
import urllib.error
|
|
import urllib.request
|
|
from pathlib import Path
|
|
|
|
log = logging.getLogger(__name__)
|
|
|
|
YUNET_URL = ("https://github.com/opencv/opencv_zoo/raw/main/models/"
|
|
"face_detection_yunet/face_detection_yunet_2023mar.onnx")
|
|
BUFFALO_SC_URL = ("https://github.com/deepinsight/insightface/releases/"
|
|
"download/v0.7/buffalo_sc.zip")
|
|
RECOGNIZERS = ["adaface_ir101.onnx", "adaface_ir50.onnx", "w600k_r50.onnx",
|
|
"arcface_int8.onnx", "w600k_mbf.onnx", "arcface.onnx"]
|
|
|
|
# Known locations of reusable models from the previous projects.
|
|
_LEGACY_MODEL_DIRS = [
|
|
Path(r"D:\NEARLE\WOrking now\RTSP_16072025\pattern_reg\models"),
|
|
]
|
|
|
|
BUFFALO_L_URL = ("https://github.com/deepinsight/insightface/releases/"
|
|
"download/v0.7/buffalo_l.zip")
|
|
|
|
# target filename -> legacy filename
|
|
_COPY_MAP = {
|
|
"arcface.onnx": "arcface.onnx",
|
|
"age_deploy.prototxt": "age_deploy.prototxt",
|
|
"age_net.caffemodel": "age_net.caffemodel",
|
|
"gender_deploy.prototxt": "gender_deploy.prototxt",
|
|
"gender_net.caffemodel": "gender_net.caffemodel",
|
|
"emotion-ferplus-8.onnx": "emotion-ferplus-8.onnx",
|
|
}
|
|
|
|
|
|
def _https_context() -> "ssl.SSLContext | None":
|
|
"""The CA store to trust, or None to use whatever Python defaults to.
|
|
|
|
Returning None first is deliberate. On Windows and on a Homebrew or
|
|
system Python, the default context reads the machine's own certificate
|
|
store - which is what makes a corporate proxy with its own root CA work.
|
|
Replacing that with certifi's bundle unconditionally would break every
|
|
site that has one, in order to fix a different platform.
|
|
|
|
The platform this fixes is a python.org macOS build. It ships its own
|
|
OpenSSL with NO trust store, and populates one only when somebody
|
|
double-clicks `Install Certificates.command` in the Python folder -
|
|
which nobody installing face-recognition software has any reason to know
|
|
about. Every HTTPS request from that interpreter fails with:
|
|
|
|
ssl.SSLCertVerificationError: [SSL: CERTIFICATE_VERIFY_FAILED]
|
|
certificate verify failed: unable to get local issuer certificate
|
|
|
|
Measured on a colleague's Mac: the engine installed perfectly and then
|
|
could not download a 230 KB model file, ending setup in forty lines of
|
|
traceback about `_ssl.c`.
|
|
"""
|
|
try:
|
|
import certifi
|
|
except ImportError: # pragma: no cover - certifi ships with requests
|
|
return None
|
|
return ssl.create_default_context(cafile=certifi.where())
|
|
|
|
|
|
def _is_cert_failure(err: BaseException) -> bool:
|
|
"""Is this a certificate-verification failure, however it is wrapped?
|
|
|
|
`urllib` does NOT let `ssl.SSLCertVerificationError` out. It catches it and
|
|
re-raises `urllib.error.URLError(err)`, carrying the original on `.reason`
|
|
- so `except ssl.SSLCertVerificationError` around `urlopen` matches
|
|
nothing, ever.
|
|
|
|
That is not a subtlety this file gets to record academically: the first
|
|
version of the fallback below was written exactly that way, shipped, and
|
|
failed on the machine it was written for with the very traceback it was
|
|
meant to prevent. The unit test passed throughout, because the fake it
|
|
used raised the bare SSL error - a shape real urllib never produces. A
|
|
stub that agrees with the author is worse than no test, and the test now
|
|
raises what urllib raises.
|
|
"""
|
|
reason = getattr(err, "reason", None)
|
|
return isinstance(err, ssl.SSLCertVerificationError) or \
|
|
isinstance(reason, ssl.SSLCertVerificationError)
|
|
|
|
|
|
def _urlopen(url: str, timeout: float = 60.0):
|
|
"""Open a URL, falling back to certifi's CA bundle on a verify failure.
|
|
|
|
Default first, certifi second, so the fix is additive: a machine whose
|
|
own store works keeps using it, and one with no store at all gets a
|
|
bundle rather than a traceback. certifi is already here - `requests` is a
|
|
hard dependency and brings it.
|
|
"""
|
|
try:
|
|
return urllib.request.urlopen(url, timeout=timeout)
|
|
except (urllib.error.URLError, ssl.SSLCertVerificationError) as err:
|
|
# Only a certificate problem is worth a second attempt. "No route to
|
|
# host" and "connection refused" arrive as URLError too, and retrying
|
|
# those with a different CA list changes nothing except how long the
|
|
# operator waits for the real message.
|
|
if not _is_cert_failure(err):
|
|
raise
|
|
ctx = _https_context()
|
|
if ctx is None:
|
|
raise
|
|
log.info("the system certificate store could not verify %s; "
|
|
"using the bundled CA list", url.split("/")[2])
|
|
return urllib.request.urlopen(url, timeout=timeout, context=ctx)
|
|
|
|
|
|
def _fetch(url: str, dest: Path, label: str) -> None:
|
|
"""Download with progress on stdout the supervisor can read.
|
|
|
|
On first run this is minutes of nothing: the API is not up yet, so the
|
|
app cannot ask the engine what it is doing, and a shop PC that shows a
|
|
stopped engine for five minutes after install looks broken. The
|
|
supervisor watches for `download: <label> <n>%` and puts the number in
|
|
the tray and the window. Logged every 5 points, not every chunk, so the
|
|
log file does not fill with a progress bar.
|
|
"""
|
|
last = -5
|
|
|
|
def hook(blocks: int, block_size: int, total: int) -> None:
|
|
nonlocal last
|
|
if total <= 0:
|
|
return
|
|
pct = min(100, blocks * block_size * 100 // total)
|
|
if pct >= last + 5:
|
|
last = pct
|
|
log.info("download: %s %d%%", label, pct)
|
|
|
|
# Streamed rather than urlretrieve, only because urlretrieve offers no way
|
|
# to pass an SSL context and the whole point here is choosing one. The
|
|
# `download: <label> <n>%` lines are a contract: the supervisor parses
|
|
# them (`progressRe`) to put first-run progress in the tray, and without
|
|
# them a shop PC shows a stopped engine for five minutes after install.
|
|
with _urlopen(url) as resp:
|
|
total = int(resp.headers.get("Content-Length") or 0)
|
|
blocks, block_size = 0, 64 * 1024
|
|
with open(dest, "wb") as out:
|
|
while True:
|
|
chunk = resp.read(block_size)
|
|
if not chunk:
|
|
break
|
|
out.write(chunk)
|
|
blocks += 1
|
|
hook(blocks, block_size, total)
|
|
log.info("download: %s 100%%", label)
|
|
|
|
|
|
def setup_models(models_dir: Path) -> "list[str]":
|
|
"""Ensure all model files exist in models_dir. Returns missing ones."""
|
|
models_dir = Path(models_dir)
|
|
models_dir.mkdir(parents=True, exist_ok=True)
|
|
|
|
yunet = models_dir / "face_detection_yunet_2023mar.onnx"
|
|
if not yunet.exists():
|
|
log.info("downloading YuNet face detector (~230 KB)...")
|
|
tmp = yunet.with_suffix(".part")
|
|
_fetch(YUNET_URL, tmp, "face detector")
|
|
tmp.rename(yunet)
|
|
log.info("YuNet saved to %s", yunet)
|
|
|
|
for target_name, legacy_name in _COPY_MAP.items():
|
|
target = models_dir / target_name
|
|
if target.exists():
|
|
continue
|
|
for legacy_dir in _LEGACY_MODEL_DIRS:
|
|
src = legacy_dir / legacy_name
|
|
if src.exists():
|
|
log.info("copying %s from %s ...", legacy_name, legacy_dir)
|
|
shutil.copy2(src, target)
|
|
break
|
|
|
|
# Any one recognizer is enough; get the lightweight MobileFaceNet if
|
|
# none is present (13 MB, loads reliably on low-memory machines).
|
|
if not any((models_dir / n).exists() for n in RECOGNIZERS):
|
|
log.info("downloading MobileFaceNet recognizer (buffalo_sc, ~15 MB)...")
|
|
import io
|
|
import zipfile
|
|
|
|
with _urlopen(BUFFALO_SC_URL) as resp:
|
|
payload = io.BytesIO(resp.read())
|
|
with zipfile.ZipFile(payload) as zf, \
|
|
zf.open("w600k_mbf.onnx") as src, \
|
|
open(models_dir / "w600k_mbf.onnx", "wb") as dst:
|
|
shutil.copyfileobj(src, dst)
|
|
log.info("w600k_mbf.onnx saved")
|
|
|
|
# buffalo_l carries both the modern gender+age net (1.3 MB) and the
|
|
# ResNet50 recognizer (~166 MB, IJB-C 97.25 vs MobileFaceNet's 95.02).
|
|
# One 275 MB download serves both, so fetch it once and take what is
|
|
# missing. Optional: failure here must never block the pipeline.
|
|
wanted = {name: models_dir / name
|
|
for name in ("genderage.onnx", "w600k_r50.onnx")
|
|
if not (models_dir / name).exists()}
|
|
if wanted:
|
|
try:
|
|
log.info("downloading %s from the buffalo_l bundle (~275 MB "
|
|
"one-time download)...", ", ".join(wanted))
|
|
import zipfile
|
|
|
|
tmp = models_dir / "buffalo_l.zip.part"
|
|
_fetch(BUFFALO_L_URL, tmp, "recognition models")
|
|
with zipfile.ZipFile(tmp) as zf:
|
|
for name, target in wanted.items():
|
|
member = next((n for n in zf.namelist()
|
|
if n.endswith(name)), None)
|
|
if member is None:
|
|
log.warning("%s not found in bundle", name)
|
|
continue
|
|
part = target.with_suffix(".part")
|
|
with zf.open(member) as src, open(part, "wb") as dst:
|
|
shutil.copyfileobj(src, dst)
|
|
part.rename(target) # never leave a half-written model
|
|
log.info("%s saved", name)
|
|
tmp.unlink()
|
|
except Exception:
|
|
log.warning("buffalo_l download failed - falling back to the "
|
|
"models already present", exc_info=True)
|
|
|
|
missing = []
|
|
if not (models_dir / "face_detection_yunet_2023mar.onnx").exists():
|
|
missing.append("face_detection_yunet_2023mar.onnx")
|
|
if not any((models_dir / n).exists() for n in RECOGNIZERS):
|
|
missing.append("a recognition model (any of: %s)" % ", ".join(RECOGNIZERS))
|
|
optional_missing = [n for n in _COPY_MAP
|
|
if not (models_dir / n).exists() and n not in missing]
|
|
if optional_missing:
|
|
log.warning("optional attribute models missing (age/gender/emotion "
|
|
"will be skipped): %s", ", ".join(optional_missing))
|
|
return missing
|