Brand Images Repairs

This commit is contained in:
sriram
2026-09-02 15:52:31 +05:30
parent d5ec5755bf
commit 1c20bfc408
3 changed files with 3140 additions and 10 deletions

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

View File

@@ -62,6 +62,7 @@ from __future__ import annotations
import argparse
import json
import logging
import re
import sys
import time
from concurrent.futures import ThreadPoolExecutor
@@ -86,6 +87,12 @@ logger = logging.getLogger("repair_brand_images")
DATA_DIR = Path(__file__).resolve().parents[1] / "data"
# How many ranked URLs to store per repaired row. Ingestion keeps up to 20;
# a repair keeps fewer on purpose. Ranking demotes a bad match rather than
# dropping it, so a long tail is where a contaminated image survives - and the
# only one that reaches the UI is the first.
MAX_STORED_IMAGES = 5
HEALTHY = "healthy" # at least one URL resolves
BROKEN = "broken" # has URLs, none resolve
EMPTY = "empty" # no URLs at all
@@ -315,6 +322,104 @@ def _backup(tables: Dict[str, List[Dict[str, Any]]]) -> Path:
return path
# One search per PRODUCT, not per pack size.
#
# These tables are mostly size variants of one item - Aachi alone has 175 rows
# that collapse to a few dozen distinct products ("Aachi Masala Powder" at 7g,
# 50g, 100g, ...). Searching each size separately means three to five times the
# network traffic for the same answer, and the providers throttle: DuckDuckGo
# already 403s and the pipeline falls through to Bing, which will not tolerate
# hundreds of near-identical queries in a row.
#
# Sharing one image across sizes is also correct rather than merely cheap: it is
# the same product in a different pack, and the ingestion pipeline's own
# size-variant explosion produces exactly these rows from a single source
# product.
_SEARCH_CACHE: Dict[Tuple[str, str], List[str]] = {}
_SIZE_TAIL = re.compile(
r"\s+\d+(?:\.\d+)?\s*(?:g|gm|gms|kg|ml|l|ltr|litre|liter|pcs|pc|n|no|nos)\b\.?\s*$",
re.IGNORECASE,
)
# --- Open Food Facts cross-check -------------------------------------------
# OFF is the first provider find_all_image_urls consults and it matches on NAME,
# not brand. Asked for "Aachi Pickles" it returned the front-of-pack photo for
# "Ducros Green Pitted Olives" - a French product, barcode 3275928006633 - and
# that image validated perfectly happily because it IS a real image. It was
# written to six Aachi rows before this check existed.
#
# The barcode is embedded in the image path, so the product is one cheap lookup
# away. If OFF's own record does not mention our brand, the image is somebody
# else's product and is dropped. Nothing else can catch this: the URL is opaque
# digits, so no amount of filename matching would help.
_OFF_CACHE: Dict[str, bool] = {}
_OFF_BARCODE = re.compile(r"/images/products/((?:\d+/)+)")
def _brand_tokens(brand: str) -> List[str]:
return [w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) if len(w) > 2]
def _off_product_matches_brand(url: str, brand: str) -> bool:
"""False only when OFF positively says this barcode is another brand."""
if "openfoodfacts.org" not in url:
return True
match = _OFF_BARCODE.search(url)
if not match:
return True
barcode = match.group(1).replace("/", "")
tokens = _brand_tokens(brand)
if not tokens:
return True
key = f"{barcode}:{brand.lower()}"
if key in _OFF_CACHE:
return _OFF_CACHE[key]
verdict = True
try:
import requests
resp = requests.get(
f"https://world.openfoodfacts.org/api/v2/product/{barcode}.json",
params={"fields": "brands,product_name"},
timeout=10,
headers={"User-Agent": "nearle-catalogue-repair/1.0"},
)
if resp.ok:
product = (resp.json() or {}).get("product") or {}
haystack = f"{product.get('brands') or ''} {product.get('product_name') or ''}".lower()
if haystack.strip():
verdict = any(t in haystack for t in tokens)
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
verdict = True
_OFF_CACHE[key] = verdict
return verdict
def _names_product(url: str, product_name: str, brand: str) -> bool:
"""True when the URL itself corroborates the match.
Not a requirement - Zepto and Flipkart serve opaque hashed paths for
perfectly correct images - but a strong signal, so corroborated URLs are
ranked ahead of uncorroborated ones.
"""
lowered = url.lower()
words = [
w for w in re.split(r"[^a-z0-9]+", (product_name or "").lower())
if len(w) > 2 and not re.fullmatch(r"\d+(?:g|kg|ml|l)?", w)
]
return any(w in lowered for w in words + _brand_tokens(brand))
def _search_key(product_name: str, brand: str) -> Tuple[str, str]:
"""Collapse trailing pack size so sizes of one product share a search."""
base = _SIZE_TAIL.sub("", product_name or "").strip()
return (brand.lower(), (base or product_name or "").lower())
def _search_replacement(product_name: str, brand: str, max_results: int) -> List[str]:
"""Validated image URLs for one product, best first.
@@ -323,20 +428,53 @@ def _search_replacement(product_name: str, brand: str, max_results: int) -> List
the ingestion pipeline uses, so a repaired row is ranked by the rules
tests/test_image_selection.py already pins.
"""
key = _search_key(product_name, brand)
if key in _SEARCH_CACHE:
return _SEARCH_CACHE[key]
from app.services.image_search import find_all_image_urls
urls = find_all_image_urls(product_name, brand=brand, validate=True, max_results=max_results)
if not urls:
_SEARCH_CACHE[key] = []
return []
try:
from app.core.catalog_engine import CatalogEngine
urls = CatalogEngine._select_best_images(
CatalogEngine, urls, product_name, brand, max_images=10
)
except Exception: # noqa: BLE001 - ranking is a nicety; a live URL is the point
pass
# https first, mirroring _pick_sample_image, so the two layers cannot
# disagree about which of a row's URLs is the good one.
return sorted(urls, key=lambda u: 0 if u.startswith("https://") else 1)
# Ranking is NOT optional, and this must not be wrapped in a bare except.
#
# find_all_image_urls only proves a URL serves an image - it does not prove
# the image is of THIS product. A search for "MTR Dosa Mix 50g" returned an
# anatomy-and-physiology textbook plate that validated perfectly happily.
# _select_best_images is the contamination filter that demotes it, scoring
# on the distinctive words of the title, and it is the same ranking stage 6
# applies during ingestion.
#
# An earlier version of this call named the class wrongly and swallowed the
# resulting ImportError, so ranking silently never ran and the textbook
# image would have been written as the primary image of four MTR products.
# Let an error here raise: unranked output is not safe to store.
from app.core.catalog_engine import ProductCatalogEngine
urls = ProductCatalogEngine._select_best_images(
ProductCatalogEngine, urls, product_name, brand, max_images=MAX_STORED_IMAGES
)
if not urls:
_SEARCH_CACHE[key] = []
return []
# Drop other brands' products before anything else looks at them.
urls = [u for u in urls if _off_product_matches_brand(u, brand)]
if not urls:
_SEARCH_CACHE[key] = []
return []
# Rank: corroborated first, then https, then the ranker's own order. A URL
# that names the product or the brand is the one to show; an opaque CDN path
# is kept but never preferred over a corroborated sibling.
ranked = sorted(
urls,
key=lambda u: (
0 if _names_product(u, product_name, brand) else 1,
0 if u.startswith("https://") else 1,
),
)
_SEARCH_CACHE[key] = ranked
return ranked
def repair(cur, conn, brands: List[str], apply: bool, timeout: int, workers: int,