Image Status check

This commit is contained in:
sriram
2026-09-16 17:00:29 +05:30
parent 9c8dbf1759
commit deae694a1f
2 changed files with 51 additions and 1 deletions

View File

@@ -147,6 +147,20 @@ def to_pg(vector: Iterable[int]) -> str:
# URL -> bytes
# ---------------------------------------------------------------------------
_RATE_LIMIT_BACKOFF_SECONDS = 5.0
_RATE_LIMIT_BACKOFF_CAP_SECONDS = 15.0
def _retry_after_seconds(header: Optional[str]) -> float:
"""Seconds to wait after a 429: the server's Retry-After if it is a plain
number, capped so one hostile header cannot stall the worker."""
try:
seconds = float(header) if header else _RATE_LIMIT_BACKOFF_SECONDS
except (TypeError, ValueError):
seconds = _RATE_LIMIT_BACKOFF_SECONDS
return max(0.0, min(seconds, _RATE_LIMIT_BACKOFF_CAP_SECONDS))
def download_image_bytes(url: str, timeout: Optional[float] = None) -> Optional[bytes]:
"""Fetch `url` as image bytes, or None.
@@ -165,13 +179,20 @@ def download_image_bytes(url: str, timeout: Optional[float] = None) -> Optional[
same_site = f"{parsed.scheme}://{parsed.netloc}/" if parsed.netloc else None
timeout = timeout or IMAGE_VECTOR_TIMEOUT_SECONDS
for referer in (same_site, None):
# Third attempt exists for one reason: a 429. Wikimedia served ~100 of the
# Own Products images in a row and then throttled the rest; those URLs are
# fine, the pace was not. Honour Retry-After (capped) and try once more.
for referer in (same_site, None, same_site):
headers = {"User-Agent": _BROWSER_UA, "Accept": "image/*,*/*;q=0.8"}
if referer:
headers["Referer"] = referer
resp = None
try:
resp = requests.get(url, headers=headers, timeout=timeout, stream=True)
if resp.status_code == 429:
logger.debug("HTTP 429 for %s - backing off", url)
time.sleep(_retry_after_seconds(resp.headers.get("retry-after")))
continue
if resp.status_code != 200:
logger.debug("HTTP %s for %s", resp.status_code, url)
continue

View File

@@ -244,6 +244,35 @@ def test_an_unlabelled_body_with_an_image_extension_is_accepted(monkeypatch):
assert iv.download_image_bytes("https://cdn.example/p/1.jpg") == body
def test_a_429_backs_off_for_retry_after_and_tries_again(monkeypatch):
"""Wikimedia throttled the Own Products backfill after ~100 images; the
URLs were fine. A 429 is a pause, not a failure."""
body = b"x" * (iv.MIN_IMAGE_BYTES + 10)
throttled = _Resp(429, b"")
throttled.headers["retry-after"] = "2"
_patch_requests(monkeypatch, [throttled, _Resp(200, body)])
slept: List[float] = []
monkeypatch.setattr(iv.time, "sleep", slept.append)
assert iv.download_image_bytes("https://upload.wikimedia.org/x.jpg") == body
assert slept == [2.0]
def test_a_persistent_429_gives_up_after_three_attempts(monkeypatch):
calls = _patch_requests(monkeypatch, [_Resp(429, b""), _Resp(429, b""), _Resp(429, b"")])
monkeypatch.setattr(iv.time, "sleep", lambda s: None)
assert iv.download_image_bytes("https://upload.wikimedia.org/x.jpg") is None
assert len(calls) == 3
def test_retry_after_is_capped_and_tolerates_garbage():
assert iv._retry_after_seconds("3") == 3.0
assert iv._retry_after_seconds("9999") == iv._RATE_LIMIT_BACKOFF_CAP_SECONDS
assert iv._retry_after_seconds(None) == iv._RATE_LIMIT_BACKOFF_SECONDS
assert iv._retry_after_seconds("Wed, 21 Oct 2026 07:28:00 GMT") == iv._RATE_LIMIT_BACKOFF_SECONDS
def test_a_raised_request_is_none_not_an_exception(monkeypatch):
import requests