Brand Images Repairs
This commit is contained in:
1486
data/brand_image_repair_backup_20260902_153945.json
Normal file
1486
data/brand_image_repair_backup_20260902_153945.json
Normal file
File diff suppressed because it is too large
Load Diff
1506
data/brand_image_repair_backup_20260902_154645.json
Normal file
1506
data/brand_image_repair_backup_20260902_154645.json
Normal file
File diff suppressed because it is too large
Load Diff
@@ -62,6 +62,7 @@ from __future__ import annotations
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
@@ -86,6 +87,12 @@ logger = logging.getLogger("repair_brand_images")
|
||||
|
||||
DATA_DIR = Path(__file__).resolve().parents[1] / "data"
|
||||
|
||||
# How many ranked URLs to store per repaired row. Ingestion keeps up to 20;
|
||||
# a repair keeps fewer on purpose. Ranking demotes a bad match rather than
|
||||
# dropping it, so a long tail is where a contaminated image survives - and the
|
||||
# only one that reaches the UI is the first.
|
||||
MAX_STORED_IMAGES = 5
|
||||
|
||||
HEALTHY = "healthy" # at least one URL resolves
|
||||
BROKEN = "broken" # has URLs, none resolve
|
||||
EMPTY = "empty" # no URLs at all
|
||||
@@ -315,6 +322,104 @@ def _backup(tables: Dict[str, List[Dict[str, Any]]]) -> Path:
|
||||
return path
|
||||
|
||||
|
||||
# One search per PRODUCT, not per pack size.
|
||||
#
|
||||
# These tables are mostly size variants of one item - Aachi alone has 175 rows
|
||||
# that collapse to a few dozen distinct products ("Aachi Masala Powder" at 7g,
|
||||
# 50g, 100g, ...). Searching each size separately means three to five times the
|
||||
# network traffic for the same answer, and the providers throttle: DuckDuckGo
|
||||
# already 403s and the pipeline falls through to Bing, which will not tolerate
|
||||
# hundreds of near-identical queries in a row.
|
||||
#
|
||||
# Sharing one image across sizes is also correct rather than merely cheap: it is
|
||||
# the same product in a different pack, and the ingestion pipeline's own
|
||||
# size-variant explosion produces exactly these rows from a single source
|
||||
# product.
|
||||
_SEARCH_CACHE: Dict[Tuple[str, str], List[str]] = {}
|
||||
|
||||
_SIZE_TAIL = re.compile(
|
||||
r"\s+\d+(?:\.\d+)?\s*(?:g|gm|gms|kg|ml|l|ltr|litre|liter|pcs|pc|n|no|nos)\b\.?\s*$",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
# --- Open Food Facts cross-check -------------------------------------------
|
||||
# OFF is the first provider find_all_image_urls consults and it matches on NAME,
|
||||
# not brand. Asked for "Aachi Pickles" it returned the front-of-pack photo for
|
||||
# "Ducros Green Pitted Olives" - a French product, barcode 3275928006633 - and
|
||||
# that image validated perfectly happily because it IS a real image. It was
|
||||
# written to six Aachi rows before this check existed.
|
||||
#
|
||||
# The barcode is embedded in the image path, so the product is one cheap lookup
|
||||
# away. If OFF's own record does not mention our brand, the image is somebody
|
||||
# else's product and is dropped. Nothing else can catch this: the URL is opaque
|
||||
# digits, so no amount of filename matching would help.
|
||||
_OFF_CACHE: Dict[str, bool] = {}
|
||||
_OFF_BARCODE = re.compile(r"/images/products/((?:\d+/)+)")
|
||||
|
||||
|
||||
def _brand_tokens(brand: str) -> List[str]:
|
||||
return [w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) if len(w) > 2]
|
||||
|
||||
|
||||
def _off_product_matches_brand(url: str, brand: str) -> bool:
|
||||
"""False only when OFF positively says this barcode is another brand."""
|
||||
if "openfoodfacts.org" not in url:
|
||||
return True
|
||||
match = _OFF_BARCODE.search(url)
|
||||
if not match:
|
||||
return True
|
||||
barcode = match.group(1).replace("/", "")
|
||||
tokens = _brand_tokens(brand)
|
||||
if not tokens:
|
||||
return True
|
||||
|
||||
key = f"{barcode}:{brand.lower()}"
|
||||
if key in _OFF_CACHE:
|
||||
return _OFF_CACHE[key]
|
||||
|
||||
verdict = True
|
||||
try:
|
||||
import requests
|
||||
resp = requests.get(
|
||||
f"https://world.openfoodfacts.org/api/v2/product/{barcode}.json",
|
||||
params={"fields": "brands,product_name"},
|
||||
timeout=10,
|
||||
headers={"User-Agent": "nearle-catalogue-repair/1.0"},
|
||||
)
|
||||
if resp.ok:
|
||||
product = (resp.json() or {}).get("product") or {}
|
||||
haystack = f"{product.get('brands') or ''} {product.get('product_name') or ''}".lower()
|
||||
if haystack.strip():
|
||||
verdict = any(t in haystack for t in tokens)
|
||||
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
|
||||
verdict = True
|
||||
|
||||
_OFF_CACHE[key] = verdict
|
||||
return verdict
|
||||
|
||||
|
||||
def _names_product(url: str, product_name: str, brand: str) -> bool:
|
||||
"""True when the URL itself corroborates the match.
|
||||
|
||||
Not a requirement - Zepto and Flipkart serve opaque hashed paths for
|
||||
perfectly correct images - but a strong signal, so corroborated URLs are
|
||||
ranked ahead of uncorroborated ones.
|
||||
"""
|
||||
lowered = url.lower()
|
||||
words = [
|
||||
w for w in re.split(r"[^a-z0-9]+", (product_name or "").lower())
|
||||
if len(w) > 2 and not re.fullmatch(r"\d+(?:g|kg|ml|l)?", w)
|
||||
]
|
||||
return any(w in lowered for w in words + _brand_tokens(brand))
|
||||
|
||||
|
||||
def _search_key(product_name: str, brand: str) -> Tuple[str, str]:
|
||||
"""Collapse trailing pack size so sizes of one product share a search."""
|
||||
base = _SIZE_TAIL.sub("", product_name or "").strip()
|
||||
return (brand.lower(), (base or product_name or "").lower())
|
||||
|
||||
|
||||
def _search_replacement(product_name: str, brand: str, max_results: int) -> List[str]:
|
||||
"""Validated image URLs for one product, best first.
|
||||
|
||||
@@ -323,20 +428,53 @@ def _search_replacement(product_name: str, brand: str, max_results: int) -> List
|
||||
the ingestion pipeline uses, so a repaired row is ranked by the rules
|
||||
tests/test_image_selection.py already pins.
|
||||
"""
|
||||
key = _search_key(product_name, brand)
|
||||
if key in _SEARCH_CACHE:
|
||||
return _SEARCH_CACHE[key]
|
||||
|
||||
from app.services.image_search import find_all_image_urls
|
||||
urls = find_all_image_urls(product_name, brand=brand, validate=True, max_results=max_results)
|
||||
if not urls:
|
||||
_SEARCH_CACHE[key] = []
|
||||
return []
|
||||
try:
|
||||
from app.core.catalog_engine import CatalogEngine
|
||||
urls = CatalogEngine._select_best_images(
|
||||
CatalogEngine, urls, product_name, brand, max_images=10
|
||||
)
|
||||
except Exception: # noqa: BLE001 - ranking is a nicety; a live URL is the point
|
||||
pass
|
||||
# https first, mirroring _pick_sample_image, so the two layers cannot
|
||||
# disagree about which of a row's URLs is the good one.
|
||||
return sorted(urls, key=lambda u: 0 if u.startswith("https://") else 1)
|
||||
# Ranking is NOT optional, and this must not be wrapped in a bare except.
|
||||
#
|
||||
# find_all_image_urls only proves a URL serves an image - it does not prove
|
||||
# the image is of THIS product. A search for "MTR Dosa Mix 50g" returned an
|
||||
# anatomy-and-physiology textbook plate that validated perfectly happily.
|
||||
# _select_best_images is the contamination filter that demotes it, scoring
|
||||
# on the distinctive words of the title, and it is the same ranking stage 6
|
||||
# applies during ingestion.
|
||||
#
|
||||
# An earlier version of this call named the class wrongly and swallowed the
|
||||
# resulting ImportError, so ranking silently never ran and the textbook
|
||||
# image would have been written as the primary image of four MTR products.
|
||||
# Let an error here raise: unranked output is not safe to store.
|
||||
from app.core.catalog_engine import ProductCatalogEngine
|
||||
urls = ProductCatalogEngine._select_best_images(
|
||||
ProductCatalogEngine, urls, product_name, brand, max_images=MAX_STORED_IMAGES
|
||||
)
|
||||
if not urls:
|
||||
_SEARCH_CACHE[key] = []
|
||||
return []
|
||||
# Drop other brands' products before anything else looks at them.
|
||||
urls = [u for u in urls if _off_product_matches_brand(u, brand)]
|
||||
if not urls:
|
||||
_SEARCH_CACHE[key] = []
|
||||
return []
|
||||
|
||||
# Rank: corroborated first, then https, then the ranker's own order. A URL
|
||||
# that names the product or the brand is the one to show; an opaque CDN path
|
||||
# is kept but never preferred over a corroborated sibling.
|
||||
ranked = sorted(
|
||||
urls,
|
||||
key=lambda u: (
|
||||
0 if _names_product(u, product_name, brand) else 1,
|
||||
0 if u.startswith("https://") else 1,
|
||||
),
|
||||
)
|
||||
_SEARCH_CACHE[key] = ranked
|
||||
return ranked
|
||||
|
||||
|
||||
def repair(cur, conn, brands: List[str], apply: bool, timeout: int, workers: int,
|
||||
|
||||
Reference in New Issue
Block a user