Brand image display cards

This commit is contained in:
sriram
2026-09-02 13:34:40 +05:30
parent 5399fea4cc
commit d5ec5755bf
8 changed files with 921 additions and 18 deletions

View File

@@ -386,3 +386,61 @@ def get_fssai_license(brand: str) -> Optional[str]:
"""
canonical = resolve_parent_brand(brand).lower().strip()
return FSSAI_LICENSES.get(canonical)
# ---------------------------------------------------------------------------
# Brand logos
# ---------------------------------------------------------------------------
# A curated logo for the brand card, used IN PREFERENCE to a sampled product
# photo. Same contract as FSSAI_LICENSES above: keyed on the canonical parent in
# lowercase, read through an accessor that resolves aliases first.
#
# Why a curated map rather than picking a product image: for some brands there
# is no usable product image at all. Everest, Haldirams, MDH and Naga each carry
# nothing but URLs into an S3 bucket that 404s on every one, so the card renders
# its initials monogram no matter which row is sampled.
#
# Deliberately sparse. A miss is the normal case and costs nothing - the card
# falls through to a product image, then to the monogram. Only add a brand here
# when the product-image path genuinely cannot serve it.
#
# Two rules for values, both load-bearing:
# * https ONLY. The site is served over https and an http:// image is blocked
# as mixed content - which is half of the bug this map was added to fix.
# * The URL must be hot-linkable and stable. Wikimedia is used here because it
# is both, and correctly licensed.
# test_brand_registry.py asserts both, and that every key is its own canonical
# parent - without that check a key the alias map rewrites is silently dead.
BRAND_LOGOS: Dict[str, str] = {
# Keyed on the PLURAL only. Unlike FSSAI_LICENSES, which keys both
# spellings, a "haldiram" key here would be unreachable: get_brand_logo
# resolves through resolve_parent_brand first, and that maps the singular
# to "haldirams" before the lookup ever happens. Both spellings still work
# for callers - the alias is what makes them work.
"haldirams": (
"https://upload.wikimedia.org/wikipedia/en/thumb/9/91/"
"Haldiram%27s_2024_Logo.svg/500px-Haldiram%27s_2024_Logo.svg.png"
),
"mdh": "https://upload.wikimedia.org/wikipedia/commons/5/5b/MDH_spices_logo.png",
# --- Slots, deliberately empty ------------------------------------------
# No free, stable logo source was found for these. Everest has a Wikipedia
# article but no page image; the others have no Wikipedia or Wikimedia
# Commons page at all. They are served by repaired product images instead
# (scripts/repair_brand_images.py). Fill a slot in if you source a logo.
#
# "everest": "",
# "naga": "",
# "kaleesuwari": "",
# "colin": "",
}
def get_brand_logo(brand: str) -> Optional[str]:
"""Return the curated logo URL for a brand, or None.
None is the normal answer for almost every brand and means "use a product
image instead", never an error - the same contract get_fssai_license has.
"""
canonical = resolve_parent_brand(brand).lower().strip()
return BRAND_LOGOS.get(canonical)

View File

@@ -583,10 +583,105 @@ _OVERVIEW_CACHE: Dict[str, Any] = {"at": 0.0, "data": None}
def invalidate_brand_overview_cache() -> None:
"""Drop the cached brand overview so the next read reflects a fresh write."""
"""Drop the cached brand overview so the next read reflects a fresh write.
PER-PROCESS. _OVERVIEW_CACHE is a module-level dict, so calling this from a
repair script clears that script's own cache and nothing else - the running
API keeps serving its copy for up to BRAND_OVERVIEW_TTL_SECONDS. After an
out-of-process write, hit GET /api/brands/overview?refresh=true instead.
"""
_OVERVIEW_CACHE.update(at=0.0, data=None)
# How many candidate rows to score when picking a brand's sample image. One row
# was not enough: the newest row can carry an unusable URL while a perfectly
# good one sits on the next. Eight rows of three narrow columns is noise next to
# the two COUNT scans this loop already runs per table, and it all sits behind
# the 30s cache above.
SAMPLE_IMAGE_CANDIDATE_ROWS = 8
# Hosts that are known not to serve our objects at all, so a URL pointing at one
# is dead on arrival and must never be handed to a brand card.
#
# `nearledaily.s3.ap-south-1.amazonaws.com` is the whole reason this exists:
# every image URL for Everest, Haldirams, MDH and Naga points there and every
# one 404s. The URLs were minted from a key convention by a constructor that
# never checked the object existed (see s3_service.get_product_image_url), so
# the rows look populated and render blank.
#
# Keep this list short and evidence-backed. It is a blunt instrument - blocking
# a whole host is only right when the host serves us nothing - and the honest
# fix for an individual rotten URL is to repair the row, not to blacklist its
# CDN. scripts/repair_brand_images.py --audit is what produces the evidence.
_DEAD_IMAGE_HOSTS = frozenset({
"nearledaily.s3.ap-south-1.amazonaws.com",
})
def _usable_image_url(url: Any) -> Optional[str]:
"""Normalise one candidate, or None if it could never render.
Rejects blanks, non-strings, anything without a usable scheme, and hosts in
_DEAD_IMAGE_HOSTS. A protocol-relative "//host/x.jpg" is promoted to https
rather than discarded.
"""
if not isinstance(url, str):
return None
candidate = url.strip()
if not candidate:
return None
if candidate.startswith("//"):
candidate = f"https:{candidate}"
if not candidate.startswith(("http://", "https://")):
return None
host = candidate.split("/", 3)[2].lower() if candidate.count("/") >= 2 else ""
if host in _DEAD_IMAGE_HOSTS:
return None
return candidate
def _pick_sample_image(
rows: List[Dict[str, Any]]
) -> Tuple[Optional[str], Optional[str]]:
"""Choose the brand card's image from candidate rows, newest first.
Returns (image_id, url). Ranked by (scheme, row position, position within
the row), which encodes three things:
HTTPS BEATS HTTP, EVEN ON AN OLDER ROW. The site is served over https, so an
http:// image is blocked as mixed content and renders blank. That single
rule is what repairs Balaji, Colin, Hindustan Unilever, Own Products and
PepsiCo - all of which already hold working https URLs and were simply being
handed the wrong one.
BUT HTTP IS STILL SELECTABLE. A brand whose only image is http:// keeps it;
returning None there would trade a card that might render for one that
certainly will not.
THE image_id COMES FROM THE ROW THAT WON. These used to be read off the same
single row so they agreed by construction; across several rows they can
diverge, and the router builds an S3 path out of the id (brands.py). Mixed
up, that path would point at a different product.
"""
best: Optional[Tuple[Tuple[int, int, int], str, Optional[str]]] = None
for row_index, row in enumerate(rows):
candidates: List[Any] = [row.get("image_url")]
candidates.extend(row.get("image_urls") or [])
for url_index, raw in enumerate(candidates):
url = _usable_image_url(raw)
if not url:
continue
rank = (0 if url.startswith("https://") else 1, row_index, url_index)
if best is None or rank < best[0]:
best = (rank, url, row.get("image_id"))
if best is not None:
return best[2], best[1]
# No usable URL anywhere. Still hand back the newest row's image_id so the
# router's S3 lookup stays reachable for tables that carry no URL columns.
return (rows[0].get("image_id") if rows else None), None
def get_brand_overview(force_refresh: bool = False) -> List[Dict[str, Any]]:
"""Per-brand summary rows backing the brand cards on the home page.
@@ -664,7 +759,7 @@ def get_brand_overview(force_refresh: bool = False) -> List[Dict[str, Any]]:
)
categories = [r[0] for r in cur.fetchall() if r[0]]
img = None
candidate_rows: List[Dict[str, Any]] = []
image_cols = [c for c in ("image_id", "image_url", "image_urls") if c in columns]
url_cols = [c for c in ("image_url", "image_urls") if c in columns]
# Prefer a row that carries a stored URL. When the table has
@@ -675,23 +770,30 @@ def get_brand_overview(force_refresh: bool = False) -> List[Dict[str, Any]]:
# both URL columns while their objects sit in the bucket.
filter_cols = url_cols or [c for c in image_cols if c == "image_id"]
if filter_cols:
where = " OR ".join(f"({c} IS NOT NULL)" for c in filter_cols)
order = " ORDER BY updated_at DESC" if "updated_at" in columns else ""
# An empty string and an empty array both satisfy
# IS NOT NULL, and both used to win the row and then
# yield no image at all. Exclude them in SQL so the
# LIMIT is spent on rows that can actually contribute.
clauses = []
for c in filter_cols:
if c == "image_urls":
clauses.append(
"(image_urls IS NOT NULL AND array_length(image_urls, 1) > 0)"
)
else:
clauses.append(f"({c} IS NOT NULL AND {c} <> '')")
where = " OR ".join(clauses)
order = " ORDER BY updated_at DESC NULLS LAST" if "updated_at" in columns else ""
cur.execute(
f"SELECT {', '.join(image_cols)} FROM {table_name} "
f"WHERE {where}{order} LIMIT 1"
f"WHERE {where}{order} LIMIT {SAMPLE_IMAGE_CANDIDATE_ROWS}"
)
row = cur.fetchone()
img = dict(zip(image_cols, row)) if row else None
candidate_rows = [dict(zip(image_cols, r)) for r in cur.fetchall()]
except Exception as e:
logger.warning("Brand overview skipped %s: %s", table_name, e)
continue
sample_image_id = (img or {}).get("image_id")
sample_image_url = (img or {}).get("image_url") or None
if not sample_image_url and img and img.get("image_urls"):
urls = [u for u in img["image_urls"] if u]
sample_image_url = urls[0] if urls else None
sample_image_id, sample_image_url = _pick_sample_image(candidate_rows)
overview.append({
"suffix": suffix,