diff --git a/app/services/image_vector.py b/app/services/image_vector.py index d1977aa..fe02cc5 100644 --- a/app/services/image_vector.py +++ b/app/services/image_vector.py @@ -147,6 +147,20 @@ def to_pg(vector: Iterable[int]) -> str: # URL -> bytes # --------------------------------------------------------------------------- +_RATE_LIMIT_BACKOFF_SECONDS = 5.0 +_RATE_LIMIT_BACKOFF_CAP_SECONDS = 15.0 + + +def _retry_after_seconds(header: Optional[str]) -> float: + """Seconds to wait after a 429: the server's Retry-After if it is a plain + number, capped so one hostile header cannot stall the worker.""" + try: + seconds = float(header) if header else _RATE_LIMIT_BACKOFF_SECONDS + except (TypeError, ValueError): + seconds = _RATE_LIMIT_BACKOFF_SECONDS + return max(0.0, min(seconds, _RATE_LIMIT_BACKOFF_CAP_SECONDS)) + + def download_image_bytes(url: str, timeout: Optional[float] = None) -> Optional[bytes]: """Fetch `url` as image bytes, or None. @@ -165,13 +179,20 @@ def download_image_bytes(url: str, timeout: Optional[float] = None) -> Optional[ same_site = f"{parsed.scheme}://{parsed.netloc}/" if parsed.netloc else None timeout = timeout or IMAGE_VECTOR_TIMEOUT_SECONDS - for referer in (same_site, None): + # Third attempt exists for one reason: a 429. Wikimedia served ~100 of the + # Own Products images in a row and then throttled the rest; those URLs are + # fine, the pace was not. Honour Retry-After (capped) and try once more. + for referer in (same_site, None, same_site): headers = {"User-Agent": _BROWSER_UA, "Accept": "image/*,*/*;q=0.8"} if referer: headers["Referer"] = referer resp = None try: resp = requests.get(url, headers=headers, timeout=timeout, stream=True) + if resp.status_code == 429: + logger.debug("HTTP 429 for %s - backing off", url) + time.sleep(_retry_after_seconds(resp.headers.get("retry-after"))) + continue if resp.status_code != 200: logger.debug("HTTP %s for %s", resp.status_code, url) continue diff --git a/tests/test_image_vector.py b/tests/test_image_vector.py index cd592fe..546b888 100644 --- a/tests/test_image_vector.py +++ b/tests/test_image_vector.py @@ -244,6 +244,35 @@ def test_an_unlabelled_body_with_an_image_extension_is_accepted(monkeypatch): assert iv.download_image_bytes("https://cdn.example/p/1.jpg") == body +def test_a_429_backs_off_for_retry_after_and_tries_again(monkeypatch): + """Wikimedia throttled the Own Products backfill after ~100 images; the + URLs were fine. A 429 is a pause, not a failure.""" + body = b"x" * (iv.MIN_IMAGE_BYTES + 10) + throttled = _Resp(429, b"") + throttled.headers["retry-after"] = "2" + _patch_requests(monkeypatch, [throttled, _Resp(200, body)]) + slept: List[float] = [] + monkeypatch.setattr(iv.time, "sleep", slept.append) + + assert iv.download_image_bytes("https://upload.wikimedia.org/x.jpg") == body + assert slept == [2.0] + + +def test_a_persistent_429_gives_up_after_three_attempts(monkeypatch): + calls = _patch_requests(monkeypatch, [_Resp(429, b""), _Resp(429, b""), _Resp(429, b"")]) + monkeypatch.setattr(iv.time, "sleep", lambda s: None) + + assert iv.download_image_bytes("https://upload.wikimedia.org/x.jpg") is None + assert len(calls) == 3 + + +def test_retry_after_is_capped_and_tolerates_garbage(): + assert iv._retry_after_seconds("3") == 3.0 + assert iv._retry_after_seconds("9999") == iv._RATE_LIMIT_BACKOFF_CAP_SECONDS + assert iv._retry_after_seconds(None) == iv._RATE_LIMIT_BACKOFF_SECONDS + assert iv._retry_after_seconds("Wed, 21 Oct 2026 07:28:00 GMT") == iv._RATE_LIMIT_BACKOFF_SECONDS + + def test_a_raised_request_is_none_not_an_exception(monkeypatch): import requests