Brand image display cards

This commit is contained in:
sriram
2026-09-02 13:34:40 +05:30
parent 5399fea4cc
commit d5ec5755bf
8 changed files with 921 additions and 18 deletions

View File

@@ -14,6 +14,7 @@ from app.api.schemas import (
ProductListOut,
ProductOut,
)
from app.services.brand_registry import get_brand_logo
from app.services.s3_service import s3_service
from app.services.vector_store import (
get_brand_overview,
@@ -142,9 +143,22 @@ def get_brand_cards(
cards = []
for row in rows:
name = row["display_name"]
image_url = _clean_url(row.get("sample_image_url"))
# A curated logo wins over a sampled product photo. It has to be first,
# not a fallback: the brands that need one do not have an EMPTY
# sample_image_url, they have a non-empty string that only fails when
# the browser tries to fetch it. Ordering this second would leave them
# exactly as broken as they are now.
image_url = _clean_url(get_brand_logo(name)) or _clean_url(row.get("sample_image_url"))
if not image_url and s3_service.enabled and row.get("sample_image_id"):
image_url = s3_service.get_product_image_url(name, row["sample_image_id"])
# LIST, don't construct. get_product_image_url() builds
# .../{image_id}/image_000.jpg and returns it whether or not the
# object exists, which is how four brands ended up serving URLs
# that 404 on every request. get_product_image_urls() does a real
# list_objects_v2, so an empty result means genuinely no image -
# and it finds .jpeg where the constructor assumed .jpg.
listed = s3_service.get_product_image_urls(name, row["sample_image_id"])
image_url = _clean_url(listed[0]) if listed else None
cards.append(BrandCardOut(
name=name,

View File

@@ -386,3 +386,61 @@ def get_fssai_license(brand: str) -> Optional[str]:
"""
canonical = resolve_parent_brand(brand).lower().strip()
return FSSAI_LICENSES.get(canonical)
# ---------------------------------------------------------------------------
# Brand logos
# ---------------------------------------------------------------------------
# A curated logo for the brand card, used IN PREFERENCE to a sampled product
# photo. Same contract as FSSAI_LICENSES above: keyed on the canonical parent in
# lowercase, read through an accessor that resolves aliases first.
#
# Why a curated map rather than picking a product image: for some brands there
# is no usable product image at all. Everest, Haldirams, MDH and Naga each carry
# nothing but URLs into an S3 bucket that 404s on every one, so the card renders
# its initials monogram no matter which row is sampled.
#
# Deliberately sparse. A miss is the normal case and costs nothing - the card
# falls through to a product image, then to the monogram. Only add a brand here
# when the product-image path genuinely cannot serve it.
#
# Two rules for values, both load-bearing:
# * https ONLY. The site is served over https and an http:// image is blocked
# as mixed content - which is half of the bug this map was added to fix.
# * The URL must be hot-linkable and stable. Wikimedia is used here because it
# is both, and correctly licensed.
# test_brand_registry.py asserts both, and that every key is its own canonical
# parent - without that check a key the alias map rewrites is silently dead.
BRAND_LOGOS: Dict[str, str] = {
# Keyed on the PLURAL only. Unlike FSSAI_LICENSES, which keys both
# spellings, a "haldiram" key here would be unreachable: get_brand_logo
# resolves through resolve_parent_brand first, and that maps the singular
# to "haldirams" before the lookup ever happens. Both spellings still work
# for callers - the alias is what makes them work.
"haldirams": (
"https://upload.wikimedia.org/wikipedia/en/thumb/9/91/"
"Haldiram%27s_2024_Logo.svg/500px-Haldiram%27s_2024_Logo.svg.png"
),
"mdh": "https://upload.wikimedia.org/wikipedia/commons/5/5b/MDH_spices_logo.png",
# --- Slots, deliberately empty ------------------------------------------
# No free, stable logo source was found for these. Everest has a Wikipedia
# article but no page image; the others have no Wikipedia or Wikimedia
# Commons page at all. They are served by repaired product images instead
# (scripts/repair_brand_images.py). Fill a slot in if you source a logo.
#
# "everest": "",
# "naga": "",
# "kaleesuwari": "",
# "colin": "",
}
def get_brand_logo(brand: str) -> Optional[str]:
"""Return the curated logo URL for a brand, or None.
None is the normal answer for almost every brand and means "use a product
image instead", never an error - the same contract get_fssai_license has.
"""
canonical = resolve_parent_brand(brand).lower().strip()
return BRAND_LOGOS.get(canonical)

View File

@@ -583,10 +583,105 @@ _OVERVIEW_CACHE: Dict[str, Any] = {"at": 0.0, "data": None}
def invalidate_brand_overview_cache() -> None:
"""Drop the cached brand overview so the next read reflects a fresh write."""
"""Drop the cached brand overview so the next read reflects a fresh write.
PER-PROCESS. _OVERVIEW_CACHE is a module-level dict, so calling this from a
repair script clears that script's own cache and nothing else - the running
API keeps serving its copy for up to BRAND_OVERVIEW_TTL_SECONDS. After an
out-of-process write, hit GET /api/brands/overview?refresh=true instead.
"""
_OVERVIEW_CACHE.update(at=0.0, data=None)
# How many candidate rows to score when picking a brand's sample image. One row
# was not enough: the newest row can carry an unusable URL while a perfectly
# good one sits on the next. Eight rows of three narrow columns is noise next to
# the two COUNT scans this loop already runs per table, and it all sits behind
# the 30s cache above.
SAMPLE_IMAGE_CANDIDATE_ROWS = 8
# Hosts that are known not to serve our objects at all, so a URL pointing at one
# is dead on arrival and must never be handed to a brand card.
#
# `nearledaily.s3.ap-south-1.amazonaws.com` is the whole reason this exists:
# every image URL for Everest, Haldirams, MDH and Naga points there and every
# one 404s. The URLs were minted from a key convention by a constructor that
# never checked the object existed (see s3_service.get_product_image_url), so
# the rows look populated and render blank.
#
# Keep this list short and evidence-backed. It is a blunt instrument - blocking
# a whole host is only right when the host serves us nothing - and the honest
# fix for an individual rotten URL is to repair the row, not to blacklist its
# CDN. scripts/repair_brand_images.py --audit is what produces the evidence.
_DEAD_IMAGE_HOSTS = frozenset({
"nearledaily.s3.ap-south-1.amazonaws.com",
})
def _usable_image_url(url: Any) -> Optional[str]:
"""Normalise one candidate, or None if it could never render.
Rejects blanks, non-strings, anything without a usable scheme, and hosts in
_DEAD_IMAGE_HOSTS. A protocol-relative "//host/x.jpg" is promoted to https
rather than discarded.
"""
if not isinstance(url, str):
return None
candidate = url.strip()
if not candidate:
return None
if candidate.startswith("//"):
candidate = f"https:{candidate}"
if not candidate.startswith(("http://", "https://")):
return None
host = candidate.split("/", 3)[2].lower() if candidate.count("/") >= 2 else ""
if host in _DEAD_IMAGE_HOSTS:
return None
return candidate
def _pick_sample_image(
rows: List[Dict[str, Any]]
) -> Tuple[Optional[str], Optional[str]]:
"""Choose the brand card's image from candidate rows, newest first.
Returns (image_id, url). Ranked by (scheme, row position, position within
the row), which encodes three things:
HTTPS BEATS HTTP, EVEN ON AN OLDER ROW. The site is served over https, so an
http:// image is blocked as mixed content and renders blank. That single
rule is what repairs Balaji, Colin, Hindustan Unilever, Own Products and
PepsiCo - all of which already hold working https URLs and were simply being
handed the wrong one.
BUT HTTP IS STILL SELECTABLE. A brand whose only image is http:// keeps it;
returning None there would trade a card that might render for one that
certainly will not.
THE image_id COMES FROM THE ROW THAT WON. These used to be read off the same
single row so they agreed by construction; across several rows they can
diverge, and the router builds an S3 path out of the id (brands.py). Mixed
up, that path would point at a different product.
"""
best: Optional[Tuple[Tuple[int, int, int], str, Optional[str]]] = None
for row_index, row in enumerate(rows):
candidates: List[Any] = [row.get("image_url")]
candidates.extend(row.get("image_urls") or [])
for url_index, raw in enumerate(candidates):
url = _usable_image_url(raw)
if not url:
continue
rank = (0 if url.startswith("https://") else 1, row_index, url_index)
if best is None or rank < best[0]:
best = (rank, url, row.get("image_id"))
if best is not None:
return best[2], best[1]
# No usable URL anywhere. Still hand back the newest row's image_id so the
# router's S3 lookup stays reachable for tables that carry no URL columns.
return (rows[0].get("image_id") if rows else None), None
def get_brand_overview(force_refresh: bool = False) -> List[Dict[str, Any]]:
"""Per-brand summary rows backing the brand cards on the home page.
@@ -664,7 +759,7 @@ def get_brand_overview(force_refresh: bool = False) -> List[Dict[str, Any]]:
)
categories = [r[0] for r in cur.fetchall() if r[0]]
img = None
candidate_rows: List[Dict[str, Any]] = []
image_cols = [c for c in ("image_id", "image_url", "image_urls") if c in columns]
url_cols = [c for c in ("image_url", "image_urls") if c in columns]
# Prefer a row that carries a stored URL. When the table has
@@ -675,23 +770,30 @@ def get_brand_overview(force_refresh: bool = False) -> List[Dict[str, Any]]:
# both URL columns while their objects sit in the bucket.
filter_cols = url_cols or [c for c in image_cols if c == "image_id"]
if filter_cols:
where = " OR ".join(f"({c} IS NOT NULL)" for c in filter_cols)
order = " ORDER BY updated_at DESC" if "updated_at" in columns else ""
# An empty string and an empty array both satisfy
# IS NOT NULL, and both used to win the row and then
# yield no image at all. Exclude them in SQL so the
# LIMIT is spent on rows that can actually contribute.
clauses = []
for c in filter_cols:
if c == "image_urls":
clauses.append(
"(image_urls IS NOT NULL AND array_length(image_urls, 1) > 0)"
)
else:
clauses.append(f"({c} IS NOT NULL AND {c} <> '')")
where = " OR ".join(clauses)
order = " ORDER BY updated_at DESC NULLS LAST" if "updated_at" in columns else ""
cur.execute(
f"SELECT {', '.join(image_cols)} FROM {table_name} "
f"WHERE {where}{order} LIMIT 1"
f"WHERE {where}{order} LIMIT {SAMPLE_IMAGE_CANDIDATE_ROWS}"
)
row = cur.fetchone()
img = dict(zip(image_cols, row)) if row else None
candidate_rows = [dict(zip(image_cols, r)) for r in cur.fetchall()]
except Exception as e:
logger.warning("Brand overview skipped %s: %s", table_name, e)
continue
sample_image_id = (img or {}).get("image_id")
sample_image_url = (img or {}).get("image_url") or None
if not sample_image_url and img and img.get("image_urls"):
urls = [u for u in img["image_urls"] if u]
sample_image_url = urls[0] if urls else None
sample_image_id, sample_image_url = _pick_sample_image(candidate_rows)
overview.append({
"suffix": suffix,

View File

@@ -0,0 +1,27 @@
{
"created_at": "2026-09-02T13:19:22.064290",
"db_host": "31.97.228.132",
"db_name": "pgvector",
"tables": {
"brand_everest": [
{
"image_id": "everest_everest_garam_masala_100g",
"product_name": "Everest Garam Masala 100g",
"image_url": "https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/everest/everest_everest_garam_masala_100g/image_000.jpg",
"image_urls": [
"https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/everest/everest_everest_garam_masala_100g/image_000.jpg"
]
}
],
"brand_naga": [
{
"image_id": "naga_naga_maida_2kg",
"product_name": "Naga Maida 2kg",
"image_url": "https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/naga/naga_naga_maida_2kg/image_000.jpg",
"image_urls": [
"https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/naga/naga_naga_maida_2kg/image_000.jpg"
]
}
]
}
}

View File

@@ -57,11 +57,22 @@ def backfill_brand_images() -> None:
db_has_images = bool(image_url or (image_urls and any(image_urls)))
if not db_has_images:
# Fetch from S3 (or construct primary S3 URL)
# LIST ONLY - never construct.
#
# This used to fall back to
# s3_service.get_product_image_url(brand, image_id), which
# returns .../{image_id}/image_000.jpg whether or not the
# object exists. That is how Everest, Haldirams, MDH and
# Naga came to hold URLs that 404 on every request while
# looking perfectly populated in the database.
#
# Writing a URL we have not confirmed is worse than writing
# nothing: a row with no image renders the brand monogram,
# a row with a dead URL renders the same monogram but hides
# the fact that the image is missing. Leave it empty and let
# scripts/repair_brand_images.py find a real one.
s3_urls = s3_service.get_product_image_urls(brand, image_id)
primary_url = s3_urls[0] if s3_urls else s3_service.get_product_image_url(brand, image_id)
if not s3_urls and primary_url:
s3_urls = [primary_url]
primary_url = s3_urls[0] if s3_urls else None
if primary_url or s3_urls:
cur.execute(

View File

@@ -0,0 +1,516 @@
#!/usr/bin/env python3
"""
Find product rows whose stored image URLs no longer resolve, and repair them.
WHY THIS EXISTS
---------------
Ten of the fifty-five brand cards were rendering the initials monogram instead
of a photo, and not one of them was missing data. Every one held a non-empty
`image_url` that could not be fetched:
Balaji, Colin, Hindustan http://... - the site is https, so the
Unilever, Own Products, browser blocks it as mixed content
PepsiCo
Everest, Haldirams, https://nearledaily.s3.ap-south-1
MDH, Naga .amazonaws.com/... - every object 404s
India Gate asiancorner.pl - host is dead
The first group is already fixed in code: `_pick_sample_image` in vector_store
now prefers an https candidate, and those brands each held working https URLs
all along (Colin nine, PepsiCo fifty-five, Hindustan Unilever seventy-four).
This script is for the rest, where no usable URL exists in the row at all.
The S3 URLs were never real. `s3_service.get_product_image_url()` builds
`daily/brands/{brand}/{image_id}/image_000.jpg` and returns it whether or not
the object exists, so a row can look fully populated and render nothing. That
constructor is no longer called from the read path or from
`scripts/backfill_s3_urls.py`; if you reintroduce it, this damage comes back.
WHAT IT WILL NOT DO
-------------------
* It never rewrites a row that already has one working URL. Not re-ranked, not
re-searched, not touched.
* It never blanks a dead URL to NULL. A dead URL and no URL render identically,
so clearing gains nothing and destroys the only record of where the image was
meant to live. A row is written only when the search found a replacement.
* It never writes `image_id`. That is the UNIQUE upsert key and the S3 folder
name; changing it silently forks the product.
USAGE
-----
python -m scripts.repair_brand_images --audit
python -m scripts.repair_brand_images --audit --json
python -m scripts.repair_brand_images --brands Everest,Naga # dry run
python -m scripts.repair_brand_images --brands Everest --apply
python -m scripts.repair_brand_images --restore data/brand_image_repair_backup_X.json --apply
`--audit` is read-only by construction: it makes no image searches and opens no
write transaction. Everything else is a dry run until `--apply` is passed, and
`--apply` always writes a full backup of every targeted table first.
AFTERWARDS
----------
The running API caches the brand overview for BRAND_OVERVIEW_TTL_SECONDS in a
per-process dict, so this script cannot clear it. Either wait out the TTL or
hit `GET /api/brands/overview?refresh=true`. The script prints this reminder.
"""
from __future__ import annotations
import argparse
import json
import logging
import sys
import time
from concurrent.futures import ThreadPoolExecutor
from datetime import datetime
from decimal import Decimal
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.infrastructure.settings import DB_HOST, DB_NAME # noqa: E402
from app.services.brand_registry import resolve_parent_brand # noqa: E402
from app.services.vector_store import ( # noqa: E402
_connect,
_list_brand_table_suffixes,
_sanitize_name,
invalidate_brand_overview_cache,
)
logging.basicConfig(level=logging.INFO, format="%(message)s")
logger = logging.getLogger("repair_brand_images")
DATA_DIR = Path(__file__).resolve().parents[1] / "data"
HEALTHY = "healthy" # at least one URL resolves
BROKEN = "broken" # has URLs, none resolve
EMPTY = "empty" # no URLs at all
# ---------------------------------------------------------------------------
# URL validation
# ---------------------------------------------------------------------------
# Probing the same URL once per row would be pure waste: Kaleesuwari's eighty-six
# URLs share one host and one key prefix, and the four S3 brands share another.
# Memoised across the whole run, per process.
_URL_CACHE: Dict[str, bool] = {}
def _url_is_live(url: str, timeout: int) -> bool:
"""True when the URL serves something that looks like an image.
Uses the same probe the ingestion pipeline uses, so a URL this script
accepts is one stage 6 would also have accepted.
"""
if url in _URL_CACHE:
return _URL_CACHE[url]
ok = False
try:
from app.services.image_search import validate_image_url_live
ok = bool(validate_image_url_live(url, timeout=timeout))
except Exception: # noqa: BLE001 - an unreachable URL is the thing we are measuring
ok = False
_URL_CACHE[url] = ok
return ok
def _row_urls(row: Dict[str, Any]) -> List[str]:
"""Every candidate URL on a row, deduped, order preserved."""
out: List[str] = []
seen = set()
for value in [row.get("image_url"), *(row.get("image_urls") or [])]:
if not isinstance(value, str):
continue
url = value.strip()
if url and url not in seen:
seen.add(url)
out.append(url)
return out
def _classify(rows: List[Dict[str, Any]], timeout: int, workers: int) -> Dict[str, str]:
"""image_id -> HEALTHY | BROKEN | EMPTY, validating every distinct URL once."""
distinct: List[str] = []
seen = set()
for row in rows:
for url in _row_urls(row):
if url not in seen:
seen.add(url)
distinct.append(url)
if distinct:
with ThreadPoolExecutor(max_workers=workers) as pool:
list(pool.map(lambda u: _url_is_live(u, timeout), distinct))
verdict: Dict[str, str] = {}
for row in rows:
urls = _row_urls(row)
if not urls:
verdict[row["image_id"]] = EMPTY
elif any(_URL_CACHE.get(u) for u in urls):
verdict[row["image_id"]] = HEALTHY
else:
verdict[row["image_id"]] = BROKEN
return verdict
def _host(url: str) -> str:
return url.split("/", 3)[2].lower() if url.count("/") >= 2 else "?"
# ---------------------------------------------------------------------------
# Reading
# ---------------------------------------------------------------------------
def _brand_tables(cur, brands: Optional[List[str]]) -> List[Tuple[str, str]]:
"""(suffix, table) pairs. Named brands resolve through the alias map."""
if brands:
pairs = []
for raw in brands:
suffix = _sanitize_name(resolve_parent_brand(raw.strip()))
pairs.append((suffix, f"brand_{suffix}"))
return pairs
# include_inactive: an archived brand still has a table and still renders a
# card, so it still has to be auditable.
return [(s, f"brand_{s}") for s in sorted(set(_list_brand_table_suffixes(cur, include_inactive=True)))]
def _image_columns(cur, table: str) -> List[str]:
"""Which image columns this table actually has.
Brand tables written under an older DDL carry only `image_id` - brand_anil
and brand_kaleesuwari are both like this. A hardcoded SELECT raises on them,
and swallowing that error silently drops 67 products from the audit, so the
columns are discovered the same way get_brand_overview discovers them.
"""
cur.execute(
"SELECT column_name FROM information_schema.columns "
"WHERE table_name = %s AND column_name IN ('image_url', 'image_urls')",
(table,),
)
return [r[0] for r in cur.fetchall()]
def _read_rows(cur, table: str) -> List[Dict[str, Any]]:
"""Never SELECT * - `embedding` is large and must not be read or written."""
have = _image_columns(cur, table)
cols = ["image_id", "product_name"] + have
try:
cur.execute(f"SELECT {', '.join(cols)} FROM {table} WHERE image_id IS NOT NULL")
except Exception as exc: # noqa: BLE001 - table may not exist at all
logger.warning(" skipped %s: %s", table, exc)
return []
rows = []
for record in cur.fetchall():
row = dict(zip(cols, record))
# A table with no URL columns yields no URLs, which classifies every
# row as EMPTY. That is the honest answer: those products have nothing
# stored and depend entirely on the S3 listing fallback at read time.
rows.append({
"image_id": row["image_id"],
"product_name": row.get("product_name"),
"image_url": row.get("image_url"),
"image_urls": list(row.get("image_urls") or []),
})
return rows
# ---------------------------------------------------------------------------
# Audit - read-only
# ---------------------------------------------------------------------------
def audit(cur, brands: Optional[List[str]], timeout: int, workers: int,
as_json: bool) -> Dict[str, Any]:
report: List[Dict[str, Any]] = []
dead_hosts: Dict[str, int] = {}
for suffix, table in _brand_tables(cur, brands):
rows = _read_rows(cur, table)
if not rows:
continue
verdict = _classify(rows, timeout, workers)
broken = [r for r in rows if verdict[r["image_id"]] == BROKEN]
empty = [r for r in rows if verdict[r["image_id"]] == EMPTY]
for row in broken:
for url in _row_urls(row):
if not _URL_CACHE.get(url):
dead_hosts[_host(url)] = dead_hosts.get(_host(url), 0) + 1
report.append({
"brand": suffix,
"rows": len(rows),
"healthy": len(rows) - len(broken) - len(empty),
"broken": len(broken),
"empty": len(empty),
"unusable_pct": round(100.0 * (len(broken) + len(empty)) / len(rows), 1),
})
ranked = sorted(dead_hosts.items(), key=lambda kv: -kv[1])
result = {"brands": report, "dead_hosts": ranked}
if as_json:
print(json.dumps(result, indent=2))
return result
logger.info("")
logger.info("%-24s %6s %8s %7s %6s %7s", "BRAND", "ROWS", "HEALTHY", "BROKEN", "EMPTY", "UNUSABLE")
logger.info("%s", "-" * 64)
for entry in sorted(report, key=lambda e: -e["unusable_pct"]):
flag = " <-- every image unusable" if entry["unusable_pct"] == 100.0 else ""
logger.info(
"%-24s %6d %8d %7d %6d %6.1f%%%s",
entry["brand"], entry["rows"], entry["healthy"],
entry["broken"], entry["empty"], entry["unusable_pct"], flag,
)
total_rows = sum(e["rows"] for e in report)
total_bad = sum(e["broken"] + e["empty"] for e in report)
logger.info("%s", "-" * 64)
logger.info("%d brand(s), %d product(s), %d with no usable image (%.1f%%)",
len(report), total_rows, total_bad,
(100.0 * total_bad / total_rows) if total_rows else 0.0)
if ranked:
logger.info("")
logger.info("Dead hosts, most damaging first - this is what turns a set of")
logger.info("per-brand bugs into one systemic finding:")
for host, count in ranked[:12]:
logger.info(" %-52s %5d dead URL(s)", host, count)
return result
# ---------------------------------------------------------------------------
# Repair
# ---------------------------------------------------------------------------
def _plain(value: Any) -> Any:
"""JSON-safe. merge_haldiram hit both of these traps for real."""
if isinstance(value, Decimal):
return float(value)
if isinstance(value, datetime):
return value.isoformat()
return value
def _backup(tables: Dict[str, List[Dict[str, Any]]]) -> Path:
"""Every row of every targeted table, not only the ones about to change.
That is what makes --restore meaningful: restoring a partial snapshot would
leave the table in a third state that never existed.
"""
DATA_DIR.mkdir(parents=True, exist_ok=True)
path = DATA_DIR / f"brand_image_repair_backup_{datetime.now():%Y%m%d_%H%M%S}.json"
payload = {
"created_at": datetime.now().isoformat(),
"db_host": DB_HOST,
"db_name": DB_NAME,
"tables": {
table: [{k: _plain(v) for k, v in row.items()} for row in rows]
for table, rows in tables.items()
},
}
path.write_text(json.dumps(payload, indent=2), encoding="utf-8")
return path
def _search_replacement(product_name: str, brand: str, max_results: int) -> List[str]:
"""Validated image URLs for one product, best first.
find_all_image_urls validates every candidate internally, so what comes back
is live by construction. Ranking then reuses the same _select_best_images
the ingestion pipeline uses, so a repaired row is ranked by the rules
tests/test_image_selection.py already pins.
"""
from app.services.image_search import find_all_image_urls
urls = find_all_image_urls(product_name, brand=brand, validate=True, max_results=max_results)
if not urls:
return []
try:
from app.core.catalog_engine import CatalogEngine
urls = CatalogEngine._select_best_images(
CatalogEngine, urls, product_name, brand, max_images=10
)
except Exception: # noqa: BLE001 - ranking is a nicety; a live URL is the point
pass
# https first, mirroring _pick_sample_image, so the two layers cannot
# disagree about which of a row's URLs is the good one.
return sorted(urls, key=lambda u: 0 if u.startswith("https://") else 1)
def repair(cur, conn, brands: List[str], apply: bool, timeout: int, workers: int,
limit: Optional[int], max_results: int) -> Dict[str, int]:
totals = {"scanned": 0, "healthy": 0, "repaired": 0, "unrepairable": 0}
unrepairable: List[str] = []
snapshot: Dict[str, List[Dict[str, Any]]] = {}
backup_written = False
for suffix, table in _brand_tables(cur, brands):
rows = _read_rows(cur, table)
if not rows:
logger.info("%-20s no rows", suffix)
continue
snapshot[table] = rows
verdict = _classify(rows, timeout, workers)
needs = [r for r in rows if verdict[r["image_id"]] != HEALTHY]
totals["scanned"] += len(rows)
totals["healthy"] += len(rows) - len(needs)
logger.info("%-20s %3d row(s), %d need an image", suffix, len(rows), len(needs))
if not needs:
continue
# A legacy table with no URL columns has nowhere to put the answer.
# Adding columns is a migration, not a repair, so say so and move on
# rather than raising halfway through a multi-brand run.
if len(_image_columns(cur, table)) < 2:
logger.warning(
" %s has no image_url/image_urls column - skipping. "
"It needs _ensure_columns() run against it first.", table,
)
totals["unrepairable"] += len(needs)
unrepairable.append(f"{suffix} ({len(needs)} row(s): table lacks URL columns)")
continue
if apply and not backup_written:
# Snapshot every targeted table BEFORE the first write. Done lazily
# so an audit-shaped run that finds nothing writes no file.
for other_suffix, other_table in _brand_tables(cur, brands):
snapshot.setdefault(other_table, _read_rows(cur, other_table))
logger.info("Backup written: %s", _backup(snapshot))
backup_written = True
for row in needs[: limit or len(needs)]:
name = row.get("product_name") or ""
if not name:
unrepairable.append(f"{suffix}/{row['image_id']} (no product_name to search on)")
totals["unrepairable"] += 1
continue
found = _search_replacement(name, suffix, max_results)
if not found:
logger.info(" no replacement found %s", name[:58])
unrepairable.append(f"{suffix}/{row['image_id']} ({name})")
totals["unrepairable"] += 1
time.sleep(0.5)
continue
logger.info(" %s -> %s", name[:46], found[0][:70])
if apply:
cur.execute(
f"UPDATE {table} SET image_url = %s, image_urls = %s, "
f"updated_at = CURRENT_TIMESTAMP WHERE image_id = %s",
(found[0], found, row["image_id"]),
)
totals["repaired"] += 1
# find_all_image_urls fans out across five providers per product;
# polite at twenty rows, abusive at eight thousand.
time.sleep(0.5)
if apply:
conn.commit() # per brand, so a later failure keeps earlier work
if unrepairable:
logger.info("")
logger.info("Could not repair %d row(s) - curate these by hand:", len(unrepairable))
for item in unrepairable[:40]:
logger.info(" %s", item)
return totals
def restore(cur, conn, path: Path, apply: bool) -> int:
payload = json.loads(path.read_text(encoding="utf-8"))
if payload.get("db_host") != DB_HOST:
logger.error("Backup was taken from %s but DB_HOST is %s. Refusing.",
payload.get("db_host"), DB_HOST)
return 1
count = 0
for table, rows in payload["tables"].items():
for row in rows:
if apply:
cur.execute(
f"UPDATE {table} SET image_url = %s, image_urls = %s WHERE image_id = %s",
(row.get("image_url"), row.get("image_urls") or [], row["image_id"]),
)
count += 1
if apply:
conn.commit()
logger.info("%s %d row(s) from %s", "Restored" if apply else "Would restore", count, path.name)
return 0
# ---------------------------------------------------------------------------
def main() -> int:
parser = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
)
mode = parser.add_mutually_exclusive_group(required=True)
mode.add_argument("--audit", action="store_true",
help="read-only report across every brand; writes nothing")
mode.add_argument("--brands", help="comma-separated brands to repair")
mode.add_argument("--all", action="store_true",
help="repair every brand; requires --limit")
mode.add_argument("--restore", metavar="PATH", help="restore a backup file")
parser.add_argument("--apply", action="store_true",
help="commit the changes (default is a dry run)")
parser.add_argument("--dry-run", action="store_true",
help="explicit no-op; this is already the default")
parser.add_argument("--json", action="store_true", help="audit output as JSON")
parser.add_argument("--limit", type=int, help="max rows to repair per brand")
parser.add_argument("--timeout", type=int, default=8, help="per-URL probe seconds")
parser.add_argument("--workers", type=int, default=8, help="parallel URL probes")
parser.add_argument("--max-results", type=int, default=8,
help="candidates to request per product search")
args = parser.parse_args()
apply = args.apply and not args.dry_run
if args.audit and args.apply:
parser.error("--audit is read-only; it cannot be combined with --apply")
if args.all and apply and not args.limit:
parser.error("--all --apply requires --limit: a catalogue-wide network "
"sweep must be a deliberate decision")
logger.info("Target database: %s / %s", DB_HOST, DB_NAME)
if args.audit:
logger.info("Mode: AUDIT - read-only, no searches, nothing is written")
else:
logger.info("Mode: %s", "APPLY - this writes" if apply else "DRY RUN - nothing is written")
conn = _connect()
if conn is None:
logger.error("No database connection.")
return 1
try:
with conn.cursor() as cur:
if args.restore:
return restore(cur, conn, Path(args.restore), apply)
if args.audit:
audit(cur, None, args.timeout, args.workers, args.json)
return 0
brands = [b for b in (args.brands or "").split(",") if b.strip()] or None
if args.all:
brands = None
totals = repair(cur, conn, brands, apply, args.timeout,
args.workers, args.limit, args.max_results)
logger.info("")
logger.info("scanned %(scanned)d | already fine %(healthy)d | "
"repaired %(repaired)d | unrepairable %(unrepairable)d", totals)
if apply and totals["repaired"]:
invalidate_brand_overview_cache() # this process only
logger.info("")
logger.info("The running API still holds its own cached overview.")
logger.info("Refresh it with: GET /api/brands/overview?refresh=true")
finally:
conn.close()
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,133 @@
"""Which image a brand card ends up showing.
The brand grid renders one photo per brand, sampled from that brand's products.
Ten of the fifty-five cards were rendering the initials monogram instead, and
none of them were missing data - every one had a non-empty `image_url` that
simply could not be fetched:
* five (Balaji, Colin, Hindustan Unilever, Own Products, PepsiCo) were handed
an `http://` URL. The site is https, so the browser blocks it as mixed
content. Each of those brands already held working https URLs - Colin nine
of them, Hindustan Unilever seventy-four - on the very same rows.
* four (Everest, Haldirams, MDH, Naga) were handed a URL into an S3 bucket
that 404s on every object, minted by a constructor that never checked.
Both classes look identical to the frontend: `Boolean(image_url)` is true, so
the card renders an <img> and only finds out at fetch time. That is why the
choice has to be made here, where the alternatives are still visible.
These are pure-function tests on dicts. No database, no network - the module
is imported for two helpers and nothing else, which also keeps them honest
under `filterwarnings = error`.
"""
from __future__ import annotations
from app.services.vector_store import _pick_sample_image, _usable_image_url
DEAD_S3 = "https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/mdh/x/image_000.jpg"
def _row(image_id, image_url=None, image_urls=None):
return {"image_id": image_id, "image_url": image_url, "image_urls": image_urls or []}
# ---------------------------------------------------------------------------
# The bug that started this
# ---------------------------------------------------------------------------
def test_https_beats_http_even_on_an_older_row():
"""The Colin case, which is the whole reason the selection changed.
Colin's newest row leads with http://officio.in/... and carries working
Amazon https URLs further down. Picking positionally gave the blocked one.
"""
rows = [
_row("newest", "http://officio.in/Colin-Glass-Cleaner.jpg"),
_row("older", "https://m.media-amazon.com/images/I/61PAkjiijnL.jpg"),
]
image_id, url = _pick_sample_image(rows)
assert url == "https://m.media-amazon.com/images/I/61PAkjiijnL.jpg"
assert image_id == "older", "the id must follow the URL that won, not row 0"
def test_https_inside_the_array_beats_the_http_scalar_on_the_same_row():
rows = [_row(
"only",
"http://officio.in/x.jpg",
["http://officio.in/x.jpg", "https://m.media-amazon.com/i/61.jpg"],
)]
assert _pick_sample_image(rows)[1] == "https://m.media-amazon.com/i/61.jpg"
# ---------------------------------------------------------------------------
# The non-regression that matters most
# ---------------------------------------------------------------------------
def test_an_http_only_brand_keeps_its_image():
"""Preferring https must never mean discarding the only image there is.
A card that might render beats one that certainly will not, and plenty of
brands legitimately hold a single http URL.
"""
rows = [_row("b", "http://only-image-in-the-world/x.jpg")]
image_id, url = _pick_sample_image(rows)
assert url == "http://only-image-in-the-world/x.jpg"
assert image_id == "b"
# ---------------------------------------------------------------------------
# Values that pass IS NOT NULL but are not images
# ---------------------------------------------------------------------------
def test_empty_and_blank_strings_are_skipped():
rows = [_row("c", "", ["", " ", "https://good/y.jpg"])]
assert _pick_sample_image(rows)[1] == "https://good/y.jpg"
def test_non_http_values_are_rejected():
for junk in ("", " ", None, 42, "data:image/png;base64,AAAA", "daily/brands/x/i.jpg"):
assert _usable_image_url(junk) is None, junk
def test_protocol_relative_urls_are_promoted_to_https():
assert _usable_image_url("//cdn.example/x.jpg") == "https://cdn.example/x.jpg"
# ---------------------------------------------------------------------------
# The dead bucket
# ---------------------------------------------------------------------------
def test_a_known_dead_host_loses_to_a_live_one():
rows = [_row("newest", DEAD_S3), _row("older", "https://live.example/z.jpg")]
image_id, url = _pick_sample_image(rows)
assert url == "https://live.example/z.jpg"
assert image_id == "older"
def test_a_known_dead_host_is_never_returned_even_alone():
"""Everest, Haldirams, MDH and Naga hold nothing else.
Returning None here is correct: it lets the router try S3 and then the card
fall back to its monogram, rather than rendering a broken image.
"""
image_id, url = _pick_sample_image([_row("mdh_row", DEAD_S3)])
assert url is None
assert image_id == "mdh_row", "the id still has to reach the S3 lookup"
# ---------------------------------------------------------------------------
# Shape guarantees the router depends on
# ---------------------------------------------------------------------------
def test_no_usable_url_still_returns_the_newest_image_id():
"""brands.py resolves sample_image_id against S3 when there is no URL.
Some brand tables predate the URL columns entirely and have only image_id,
so dropping the id would regress them to a monogram.
"""
assert _pick_sample_image([_row("keep-me", None, [])]) == ("keep-me", None)
def test_no_rows_at_all():
assert _pick_sample_image([]) == (None, None)
def test_first_usable_row_wins_when_schemes_tie():
"""With nothing to separate them, newest-first ordering still decides."""
rows = [_row("new", "https://a/1.jpg"), _row("old", "https://b/2.jpg")]
assert _pick_sample_image(rows) == ("new", "https://a/1.jpg")

View File

@@ -18,6 +18,8 @@ import pytest
from app.services.brand_registry import (
BRAND_ALIASES,
BRAND_LOGOS,
get_brand_logo,
get_fssai_license,
resolve_parent_brand,
)
@@ -155,6 +157,46 @@ def test_the_haldiram_licence_survives_the_alias() -> None:
assert get_fssai_license("Haldirams") != get_fssai_license("lion dates")
def test_every_logo_key_is_its_own_canonical_parent() -> None:
"""A key the alias map rewrites is dead on arrival, and silently so.
get_brand_logo resolves to the parent before looking up, so a key like
"haldiram's" would never be reached - and a missing logo is a legitimate
outcome for 52 of 55 brands, so nothing else would ever flag it.
"""
for key in BRAND_LOGOS:
assert resolve_parent_brand(key).lower().strip() == key, (
f"{key!r} resolves to {resolve_parent_brand(key)!r} and can never be looked up"
)
def test_every_logo_is_https() -> None:
"""http:// is blocked as mixed content on the https site.
Serving a blocked image is half the bug this map was added to fix, so it
must not be reintroducible through the fix itself.
"""
for key, url in BRAND_LOGOS.items():
assert url.startswith("https://"), f"{key} logo is not https: {url}"
def test_the_haldiram_logo_survives_the_alias() -> None:
"""Same trap as the licence above: both spellings must resolve."""
assert get_brand_logo("Haldiram") == get_brand_logo("Haldirams")
assert get_brand_logo("HALDIRAMS") == get_brand_logo("haldiram's")
assert get_brand_logo("Haldiram") is not None
def test_a_brand_without_a_curated_logo_returns_none() -> None:
"""None is the normal answer and means "use a product image instead".
The card falls through to the sampled photo and then to its monogram, so
this path is the one almost every brand takes.
"""
assert get_brand_logo("Britannia") is None
assert get_brand_logo("no such brand at all") is None
def test_known_sub_brands_still_route_to_their_family() -> None:
"""Word-boundary matching must not break legitimate sub-brand routing."""
assert resolve_parent_brand("Dove") == "hindustan unilever"