Brand valid image generation

This commit is contained in:
sriram
2026-09-11 15:47:56 +05:30
parent e1a5962f82
commit ce4fa70dee
31 changed files with 5586 additions and 1495 deletions

View File

@@ -107,6 +107,7 @@ from app.infrastructure.settings import (
)
from app.services import active_brands, brand_store, ollama_service, retail_presence
from app.services.brand_registry import (
get_brand_store_domain,
get_fssai_license,
get_known_sub_brands,
resolve_parent_brand,
@@ -449,19 +450,29 @@ def _existing_title_index(brand: str) -> Dict[str, str]:
# ---------------------------------------------------------------------------
# Source A - Open Food Facts (ground truth)
# ---------------------------------------------------------------------------
class OpenFactsUnavailable(RuntimeError):
"""Open Food Facts could not be consulted - as opposed to consulted and
found empty. `discover_brand_products` turns this into a warning that says
so, because "OFF has nothing for this brand" is a claim about the brand
and "OFF was down" is not."""
def _from_open_facts(brand: str, *, refresh: bool = False) -> List[Dict[str, Any]]:
"""Real products for `brand`, with a real GTIN and often a real pack size.
`off_bulk.fetch_brand_corpus` is already disk-cached, retry-wrapped and
paced at 2s between pages, and `scripts/backfill_barcodes_from_off.py`
already imports it from outside the enrichment package, so reaching for it
here is an established pattern rather than a new one.
`off_bulk.fetch_brand_corpus_result` is disk-cached, retry-wrapped and
paced at 6s between 100-row pages. Raises `OpenFactsUnavailable` when the fetch
did not complete and produced nothing; anything else is returned as-is,
an empty list meaning OFF genuinely has no products under this brand tag.
"""
try:
hits = off_bulk.fetch_brand_corpus(brand, refresh=refresh)
result = off_bulk.fetch_brand_corpus_result(brand, refresh=refresh)
except Exception as exc: # noqa: BLE001 - a dead OFF must not kill discovery
logger.warning("Open Food Facts lookup failed for %r: %s", brand, exc)
return []
raise OpenFactsUnavailable(str(exc) or exc.__class__.__name__) from exc
if result.error and not result.hits:
raise OpenFactsUnavailable(result.error)
hits = result.hits
out: List[Dict[str, Any]] = []
for hit in hits or ():
@@ -759,10 +770,10 @@ def _resolve_sizes(candidate: Dict[str, Any], title: str, category: str,
which is what makes `image_id` reproducible - the name is half of it.
The LLM's list is a guess and comes last. Only when all three are empty is
the cell left blank, which hands the decision to stage 4's
`default_size_variants` - and that invents 100g/250g/500g, which
`_sizes_for`'s own comment identifies as the mechanism behind observed
catalogue drift. Leaving it blank is the last resort, not the default.
the cell left blank; stage 4 then stores ONE unsized row (it no longer
invents 100g/250g/500g - see `_sizes_for`, which names that invention as
the mechanism behind observed catalogue drift). Blank is honest, but a
real size is what makes the row a distinct pack, so it is worth finding.
"""
in_title = _TITLE_SIZE_RE.search(title or "")
if in_title:
@@ -850,25 +861,21 @@ def _build_product(brand: str, candidate: Dict[str, Any], *, max_sizes: int,
# semantic search.
#
# A blank description is honest and already handled: `_to_storage_row`
# substitutes "<name> <size> from <brand>." exactly as it does for a store
# sheet that left the column empty.
# substitutes a short factual line (name, size, category, brand) exactly
# as it does for a store sheet that left the column empty.
description = (candidate.get("description") or "").strip()
# A BARCODE IDENTIFIES ONE PACK, SO IT ONLY SURVIVES A KNOWN PACK SIZE.
# With no real size, stage 4 falls back to `default_size_variants` and
# invents 100g/250g/500g - and because every column is copied into each
# exploded variant, one real GTIN would be stamped onto three packs, two of
# which do not exist. That is a worse error than a blank barcode: it is
# wrong data that looks authoritative, and `nutrition_by_barcode` would
# happily resolve all three to the same product.
# A BARCODE IDENTIFIES ONE PACK. That used to mean dropping it whenever no
# pack size was known, because stage 4 then invented 100g/250g/500g and
# every column is copied into each exploded variant - one real GTIN
# stamped onto three packs, two of which did not exist. Stage 4 now stores
# a single unsized row when it has no size, and one GTIN on one row is
# exactly what a GTIN means, so the barcode survives; the row is flagged
# so a reviewer knows the pack size is still unknown.
barcode = candidate.get("barcode")
notes: List[str] = []
if barcode and not sizes:
notes.append(
"barcode dropped: no real pack size is known, and stage 4 will "
"invent several - a GTIN names one pack, not three"
)
barcode = None
notes.append("pack size unknown: the barcode names one pack, but its size was not found")
shape = {"title": title, "category": category_hint, "size_variants": sizes,
"description": description}
@@ -929,12 +936,25 @@ def discover_brand_products(
registry_terms = _registry_terms(brand)
fssai = get_fssai_license(brand)
off_candidates = _from_open_facts(brand, refresh=refresh_corpus) if use_openfacts else []
if use_openfacts and not off_candidates:
warnings.append(
"Open Food Facts returned nothing for this brand. Every row below "
"rests on the language model alone - review them individually."
)
# Three distinct outcomes, three distinct warnings. "OFF has nothing under
# this brand tag" is a fact about the brand; "OFF could not be reached" is
# a fact about the network and must not be reported as the former - that
# is how Naga (2 real rows on OFF) was being shown as unknown to it.
off_candidates: List[Dict[str, Any]] = []
if use_openfacts:
try:
off_candidates = _from_open_facts(brand, refresh=refresh_corpus)
except OpenFactsUnavailable as exc:
warnings.append(
f"Open Food Facts could not be reached ({exc}). Nothing was "
"cached, so the next preview will try again; the rows below "
"come from the brand's shop and the language model only."
)
else:
if not off_candidates:
warnings.append(
"Open Food Facts has no products tagged with this brand."
)
off_keys = {}
for candidate in off_candidates:
key = _normalise_title(brand, candidate["title"])
@@ -951,12 +971,26 @@ def discover_brand_products(
if store_candidates:
logger.info("%s: %d products from the brand's own shop", brand, len(store_candidates))
elif use_store and not off_candidates:
# Say what was actually checked. An unregistered storefront is not a
# storefront that "has nothing" - nobody has looked.
store_domain = get_brand_store_domain(brand)
if store_domain:
warnings.append(
f"{store_domain} is registered as this brand's shop but its "
"catalogue has not been fetched - run "
f"scripts/backfill_brand_stores.py --brand \"{brand}\" to load it."
)
else:
warnings.append(
"No verified storefront is registered for this brand. If it "
"has its own online shop, add it to "
"brand_registry.BRAND_STORE_DOMAINS and run "
"scripts/backfill_brand_stores.py to populate a real catalogue."
)
if not off_candidates and not store_candidates:
warnings.append(
"Neither Open Food Facts nor a brand storefront has anything for "
"this brand. Every row below rests on the language model alone - "
"review them individually. If this brand has its own online shop, "
"adding it to brand_registry.BRAND_STORE_DOMAINS and running "
"scripts/backfill_brand_stores.py will populate a real catalogue."
"Every row below rests on the language model alone - review them "
"individually."
)

View File

@@ -11,9 +11,12 @@ never heard of there is nothing to verify, and the catalogue is whatever
`qwen2.5:1.5b` invents.
Measured 2026-09-10, Open Food Facts hits: Udhaiyam 0, Tenali Double Horse 0,
Gopuram 0, Double Horse 14, and Naga 2 - where both "Naga" rows are a UK /
Bangladeshi pickle brand, not the Tamil Nadu one. Regional South Indian brands
are simply not in that database.
Gopuram 0, Double Horse 14. (An earlier note here said Naga's two rows were a
UK / Bangladeshi pickle brand - that was the old free-text `brands:Naga`
Search-a-licious query matching "Mr Naga" and "Bombay Naga Jhal". Under the
exact `brands_tags=naga` filter the v2 API returns two real Naga Limited rows,
Sooji 500 g and Maida 500 g, both on the 890 GS1 prefix - see off_bulk.py.)
Regional South Indian brands are still thin in that database.
They are, however, on their own shop, and modern storefronts publish their
whole catalogue as structured JSON:

View File

@@ -10,34 +10,50 @@ resulting cascade "misses far more often than it hits". That module is wired
into the live enrichment pipeline and is deliberately NOT touched by this file.
This module inverts the problem: fetch a brand's **entire** OFF catalogue in one
or two requests from the Search-a-licious endpoint, cache it on disk, then match
every one of our products against that corpus offline. A whole-catalogue
backfill costs ~5 HTTP requests instead of ~600, and re-tuning the similarity
threshold costs zero network because the corpus is cached.
or two requests from the v2 search API, cache it on disk, then match every one
of our products against that corpus offline. A whole-catalogue backfill costs
~5 HTTP requests instead of ~600, and re-tuning the similarity threshold costs
zero network because the corpus is cached.
Nothing here is imported by the running app - the only consumer is
`scripts/backfill_barcodes_from_off.py`. Everything except `fetch_brand_corpus`
is a pure function so it can be unit-tested without network or database.
`fetch_brand_corpus` is the one network function here and is shared by the
backfill scripts, `product_grounding`, `post_ingest_barcodes` and
`brand_discovery`. Everything else is a pure function so it can be unit-tested
without network or database.
ENDPOINT NOTES (verified empirically, 2026-09)
GET https://search.openfoodfacts.org/search
?q=brands:amul AND countries_tags:"en:india"
ENDPOINT NOTES (verified empirically, 2026-09-11)
GET https://world.openfoodfacts.org/api/v2/search
?brands_tags=amul
&countries_tags=en:india
&fields=code,product_name,quantity,...
&page_size=250
* The double quotes around "en:india" are REQUIRED. Without them the query
parses as a bare term and silently returns count=0 rather than erroring.
* page_size up to 1000 is accepted; 250 keeps responses small.
&page_size=100&page=1
* `brands_tags` is an EXACT match on the brand's tag slug ("naga", not
"Naga"), which is what a brand catalogue needs. The earlier Search-a-licious
query `q=brands:Naga` was a free-text match: it returned "Mr Naga" and
"Bombay Naga Jhal" - other companies - and MISSED the real Naga rows,
because Search-a-licious reads a separate Elasticsearch index that was
stale for them (barcode 8906011830068 was still filed brandless under
Kuwait). The v2 API reads the live product database. Measured on the same
day: Amul 216 products here vs 140 there; Naga 2 vs 0.
* `page_count` in this API is the number of products ON THIS PAGE, not the
number of pages, and `page_size` in the RESPONSE is what the server
actually applied (a larger request is capped to 100 without complaint).
Paginate from `count` / that served `page_size`.
* OFF rate-limits all search endpoints to 10 requests/minute per IP.
PAUSE_SECONDS keeps a multi-page brand under that.
* `world.openfoodfacts.org` intermittently serves an HTML "Page temporarily
unavailable" page with a 200 status, so every response is content-type
checked before parsing.
unavailable" page with a 200 status, and plain 503s under load, so every
response is status- and content-type-checked before parsing, and a failed
fetch is never written to the cache as an empty corpus.
"""
from __future__ import annotations
import json
import logging
import math
import os
import re
import time
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple
@@ -63,21 +79,50 @@ from app.services.quantity_utils import quantities_match
logger = logging.getLogger(__name__)
SEARCH_URL = "https://search.openfoodfacts.org/search"
SEARCH_URL = "https://world.openfoodfacts.org/api/v2/search"
# OFF's usage policy requires a contactable custom User-Agent; requests sent
# with the default python-requests agent are treated as anonymous crawling.
USER_AGENT = "BrandCatalogRAG/1.0 (suriya@tenext.in)"
FIELDS = "code,product_name,product_name_en,brands,quantity,countries_tags"
PAGE_SIZE = 250
MAX_PAGES = 10 # 2500 products per brand is far beyond any real brand
PAUSE_SECONDS = 2.0 # search endpoints are the strictly-limited ones
PAGE_SIZE = 100 # the v2 API caps this at 100 and silently serves that
MAX_PAGES = 25 # 2500 products per brand is far beyond any real brand
PAUSE_SECONDS = 6.0 # 10 search requests/minute is OFF's published limit
# Bumped when the cache file's meaning changes. Files written before this key
# existed came from the Search-a-licious endpoint; an EMPTY one of those is
# more likely to be that endpoint's stale index (or a swallowed 503) than a
# real absence, so it is re-fetched. A non-empty one is real data and is kept.
CACHE_SCHEMA = 2
# backend/app/services/enrichment/barcode/sources/off_bulk.py -> backend/
_BACKEND_DIR = Path(__file__).resolve().parents[5]
CACHE_DIR = _BACKEND_DIR / "data" / "cache" / "off_brand_corpus"
class OffUnavailable(requests.exceptions.ConnectionError):
"""OFF answered but could not serve (429 / 5xx).
Subclasses ConnectionError so `retry.with_retry` treats it exactly like a
dropped socket - a few seconds later the same request usually succeeds -
without widening the retry set for every other barcode source.
"""
@dataclass
class CorpusFetch:
"""What `fetch_brand_corpus_result` learned.
`error` is None when OFF answered every page, even if it answered with
nothing - that is a real "this brand is not on Open Food Facts". When
`error` is set the fetch did not complete and NOTHING was cached, so a
caller can say "could not be reached" instead of "has nothing".
"""
hits: List[Dict[str, Any]] = field(default_factory=list)
error: Optional[str] = None
from_cache: bool = False
# Pack sizes embedded in a product name ("Marie Gold 250g", "Butter 1L").
# Mirrors the unit list in scripts/backfill_nutrition_from_barcodes.py, widened
# with the count-based units this catalog also uses.
@@ -122,84 +167,149 @@ def _cache_path(slug: str, cache_dir: Optional[Path] = None) -> Path:
return (cache_dir or CACHE_DIR) / f"{slug}.json"
def brand_tag_slug(brand: str) -> str:
"""The brand as Open Food Facts tags it: lowercase, runs of anything that
is not a letter or digit collapsed to one hyphen. "Hindustan Unilever" ->
"hindustan-unilever", "P&G" -> "p-g", "Naga" -> "naga". OFF matches
`brands_tags` case-insensitively, but sending the slug form is what its
own site does and avoids depending on that."""
return re.sub(r"[^a-z0-9]+", "-", (brand or "").lower()).strip("-")
@with_retry(max_attempts=3, min_wait=2.0, max_wait=10.0)
def _get_page(brand: str, country: Optional[str], page: int) -> Dict[str, Any]:
"""One page of OFF search results. Raises on transport errors (retried by
the decorator); returns an empty result dict for any response that is not
parseable JSON, which is how the HTML "temporarily unavailable" page and
any future error page are absorbed without killing the run."""
query = f"brands:{brand}"
def _get_page(brand: str, country: Optional[str], page: int) -> Optional[Dict[str, Any]]:
"""One page of OFF v2 search results.
Raises on transport errors and on 429 / 5xx (both retried by the
decorator); returns None for any other response that is not parseable
JSON - an HTML "temporarily unavailable" page, a 4xx, a truncated body.
None means "this page FAILED", which the caller must keep distinct from a
page that parsed fine and simply held no products.
"""
params: Dict[str, Any] = {
"brands_tags": brand_tag_slug(brand),
"fields": FIELDS,
"page_size": PAGE_SIZE,
"page": page,
}
if country:
# The quotes are load-bearing - see the module docstring.
query += f' AND countries_tags:"en:{country}"'
params["countries_tags"] = f"en:{country}"
resp = requests.get(
SEARCH_URL,
params={"q": query, "fields": FIELDS, "page_size": PAGE_SIZE, "page": page},
params=params,
headers={"User-Agent": USER_AGENT, "Accept": "application/json"},
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
)
if resp.status_code == 429 or resp.status_code >= 500:
raise OffUnavailable(f"HTTP {resp.status_code} from Open Food Facts")
if resp.status_code != 200:
logger.warning("OFF search returned HTTP %s for brand %r page %s",
resp.status_code, brand, page)
return {}
return None
if "json" not in (resp.headers.get("content-type") or "").lower():
logger.warning("OFF search returned non-JSON (%s) for brand %r - "
"the service is probably serving an error page",
resp.headers.get("content-type"), brand)
return {}
return None
try:
payload = resp.json()
except ValueError as e:
logger.warning("OFF search returned unparseable JSON for brand %r: %s", brand, e)
return {}
return payload if isinstance(payload, dict) else {}
return None
return payload if isinstance(payload, dict) else None
def fetch_brand_corpus(brand: str,
country: Optional[str] = None,
refresh: bool = False,
cache_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
"""Every OFF product for `brand`, from disk cache unless `refresh`.
def _read_cache(path: Path, brand: str) -> Optional[List[Dict[str, Any]]]:
"""Cached hits, or None when the cache must not be trusted: unreadable, or
an empty file from before CACHE_SCHEMA existed (see that constant)."""
try:
cached = json.loads(path.read_text(encoding="utf-8"))
except Exception as e: # noqa: BLE001 - a corrupt cache must not be fatal
logger.warning(" Ignoring unreadable OFF cache %s: %s", path.name, e)
return None
hits = cached.get("hits") or []
if not hits and cached.get("schema") != CACHE_SCHEMA:
logger.info(" Ignoring empty pre-v%s OFF cache for %r - re-fetching",
CACHE_SCHEMA, brand)
return None
logger.info(" OFF corpus for %r: %d product(s) (cached %s)",
brand, len(hits), cached.get("fetched_at_human", "?"))
return hits
def fetch_brand_corpus_result(brand: str,
country: Optional[str] = None,
refresh: bool = False,
cache_dir: Optional[Path] = None) -> CorpusFetch:
"""Every OFF product for `brand`, from disk cache unless `refresh`, plus
whether the fetch actually completed.
Hits with no usable product name are dropped here rather than at match time
(6 of 146 Amul hits, 1 of 233 Britannia hits) - a nameless hit can never
clear a name-similarity threshold, so carrying it forward only inflates the
corpus. Returns [] rather than raising when OFF is unreachable, so one bad
brand does not abort a multi-brand backfill.
corpus.
A fetch that fails part-way returns what it got with `error` set and
writes NO cache file. Caching a failure as `hits: []` is how "Open Food
Facts has nothing for this brand" was being asserted for brands OFF had
simply been too busy to answer about.
"""
country = BARCODE_COUNTRY_TAG if country is None else (country or None)
slug = re.sub(r"[^a-z0-9]+", "_", brand.lower()).strip("_")
path = _cache_path(slug, cache_dir)
if not refresh and path.exists():
try:
cached = json.loads(path.read_text(encoding="utf-8"))
hits = cached.get("hits") or []
logger.info(" OFF corpus for %r: %d product(s) (cached %s)",
brand, len(hits), cached.get("fetched_at_human", "?"))
return hits
except Exception as e: # noqa: BLE001 - a corrupt cache must not be fatal
logger.warning(" Ignoring unreadable OFF cache %s: %s", path.name, e)
cached = _read_cache(path, brand)
if cached is not None:
return CorpusFetch(hits=cached, from_cache=True)
hits: List[Dict[str, Any]] = []
error: Optional[str] = None
page = 1
while page <= MAX_PAGES:
total_pages = 1
while page <= min(total_pages, MAX_PAGES):
if page > 1:
time.sleep(PAUSE_SECONDS)
payload = _get_page(brand, country, page)
batch = payload.get("hits") or []
try:
payload = _get_page(brand, country, page)
except requests.exceptions.RequestException as e:
# Retries exhausted (OffUnavailable is a ConnectionError too).
error = str(e) or e.__class__.__name__
break
if payload is None:
error = "Open Food Facts returned an unusable response"
break
batch = payload.get("products") or []
hits.extend(h for h in batch if isinstance(h, dict) and (h.get("product_name") or "").strip())
page_count = payload.get("page_count") or 0
if page >= page_count or not batch:
if page == 1:
# `page_count` is products-on-this-page in the v2 API; the real
# page total is `count` over the page size the SERVER applied -
# it silently caps the requested size (250 asked, 100 served for
# Amul), so dividing by PAGE_SIZE under-pages.
count = payload.get("count") or 0
served = payload.get("page_size") or len(batch) or PAGE_SIZE
try:
total_pages = max(1, math.ceil(int(count) / int(served)))
except (TypeError, ValueError, ZeroDivisionError):
total_pages = 1
if not batch:
break
page += 1
if error:
logger.warning(" OFF corpus for %r: fetch failed after %d usable product(s) - %s "
"(nothing cached)", brand, len(hits), error)
return CorpusFetch(hits=hits, error=error)
path.parent.mkdir(parents=True, exist_ok=True)
tmp = path.with_suffix(".json.tmp")
tmp.write_text(json.dumps({
"schema": CACHE_SCHEMA,
"endpoint": SEARCH_URL,
"brand": brand,
"brand_tag": brand_tag_slug(brand),
"country": country,
"fetched_at": time.time(),
"fetched_at_human": time.strftime("%Y-%m-%d %H:%M:%S"),
@@ -208,7 +318,20 @@ def fetch_brand_corpus(brand: str,
os.replace(tmp, path)
logger.info(" OFF corpus for %r: %d usable product(s) fetched", brand, len(hits))
return hits
return CorpusFetch(hits=hits)
def fetch_brand_corpus(brand: str,
country: Optional[str] = None,
refresh: bool = False,
cache_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
"""`fetch_brand_corpus_result(...).hits` - the list-only form every
backfill and ingestion caller uses. Returns [] rather than raising when
OFF is unreachable, so one bad brand does not abort a multi-brand run;
callers that need to tell "unreachable" from "empty" use the result form.
"""
return fetch_brand_corpus_result(brand, country=country, refresh=refresh,
cache_dir=cache_dir).hits
# ---------------------------------------------------------------------------

View File

@@ -243,6 +243,16 @@ def _round2(value: Optional[float]) -> Optional[float]:
return round(value, 2)
def _held_number(value: object) -> Optional[float]:
"""A price the row already carries, or None for blank / unparseable."""
if value is None or (isinstance(value, str) and not value.strip()):
return None
try:
return float(value)
except (TypeError, ValueError):
return None
def enrich_pricing_fields(product: dict) -> Dict[str, object]:
"""Compute the pricing fields added by this stage for one catalog row.
@@ -253,8 +263,24 @@ def enrich_pricing_fields(product: dict) -> Dict[str, object]:
`cost_price`, `profit_before_tax` and `profit_after_tax` are not
derivable from the data this pipeline generates, so they stay None
(exactly as the existing enriched exports store them).
A PRICE THE ROW ALREADY HOLDS WINS. A store sheet that says "Retail Price
155" has stated a fact; the band derived from it is Rs143-167, and this
stage used to read the band's ceiling back as "selling_price = 167" and
hand that to `EnrichmentStage.apply`, which overwrites a held value with
a different one (it only refuses to BLANK one). Held prices are therefore
returned as None here so `apply` keeps them, the base for the tax is the
held selling price, then the held final price, and only then the band.
"""
selling_price = _extract_selling_price(product.get("price_range"))
held_selling = _held_number(product.get("selling_price"))
held_final = _held_number(product.get("final_selling_price"))
if held_selling is not None:
selling_price = held_selling
elif held_final is not None:
selling_price = held_final
else:
selling_price = _extract_selling_price(product.get("price_range"))
gst = product.get("gst_percent")
tax_amount = None
final_selling_price = None
@@ -263,10 +289,10 @@ def enrich_pricing_fields(product: dict) -> Dict[str, object]:
final_selling_price = _round2(selling_price + tax_amount)
return {
"selling_price": _round2(selling_price),
"selling_price": None if held_selling is not None else _round2(selling_price),
"cost_price": None,
"tax_amount": tax_amount,
"final_selling_price": final_selling_price,
"final_selling_price": None if held_final is not None else final_selling_price,
"profit_before_tax": None,
"profit_after_tax": None,
}

View File

@@ -76,7 +76,7 @@ logger = logging.getLogger(__name__)
# Matches scripts/backfill_barcodes_from_off.py, which measured it.
DEFAULT_MIN_SIMILARITY = 0.88
BARCODE_SOURCE = "openfoodfacts_bulk (search.openfoodfacts.org)"
BARCODE_SOURCE = "openfoodfacts_bulk (world.openfoodfacts.org/api/v2)"
def _rows_needing_a_barcode(cur, table: str) -> List[Dict[str, Any]]:

View File

@@ -63,7 +63,7 @@ import sqlite3
import threading
import time
from contextlib import closing
from dataclasses import dataclass
from dataclasses import dataclass, field
from pathlib import Path
from typing import Iterable, List, Optional, Sequence
from urllib.parse import urlparse
@@ -175,6 +175,23 @@ def distinctive_tokens(product_name: str, brand: str) -> List[str]:
return out
def _token_in(token: str, lowered_url: str) -> bool:
"""Does the URL carry this word, allowing for the plural on either side?
The catalogue names "Aachi Appalams" and "Aachi Pickles"; the photo is
`Aachi-Appalam-100-g-1.webp`. A whole-token substring test rejected the
correct image for the plural and then promoted an opaque Amazon URL over
it. The singular is tried as well - only for tokens long enough that
stripping the "s" leaves a real word ("gems" -> "gem" is fine; "kgs" never
gets here, size tokens are removed upstream).
"""
if token in lowered_url:
return True
if len(token) > 4 and token.endswith("s") and token[:-1] in lowered_url:
return True
return False
def search_key(product_name: str, brand: str) -> tuple:
"""Collapse a trailing pack size so sizes of one product share a lookup.
@@ -213,7 +230,7 @@ def corroborate(url: str, product_name: str, brand: str) -> Corroboration:
distinctive = distinctive_tokens(product_name, brand)
if distinctive:
hit = next((t for t in distinctive if t in lowered), None)
hit = next((t for t in distinctive if _token_in(t, lowered)), None)
if hit:
return Corroboration(True, f"url names {hit!r}")
return Corroboration(
@@ -272,6 +289,26 @@ _lock = threading.Lock()
_initialized = False
_CACHE_TTL_SECONDS = 30 * 24 * 3600
# Open*Facts allows 100 product reads a minute per IP. A brand ingestion or a
# re-gate asks about every distinct barcode it meets, and the Dabur run issued
# about a hundred in two minutes - the tail was throttled, and a throttled
# lookup fails open, which is how a Kellogg's honey got past the gate for a
# moment. Pacing the calls keeps the gate answering instead of guessing.
_OFF_MIN_INTERVAL_SECONDS = 0.65
_OFF_LOOKUP_ATTEMPTS = 2
_OFF_RETRY_SLEEP_SECONDS = 3.0
_off_last_call = 0.0
_off_pace_lock = threading.Lock()
def _pace_off_lookup() -> None:
global _off_last_call
with _off_pace_lock:
wait = _OFF_MIN_INTERVAL_SECONDS - (time.monotonic() - _off_last_call)
if wait > 0:
time.sleep(wait)
_off_last_call = time.monotonic()
def _connect() -> sqlite3.Connection:
_DB_PATH.parent.mkdir(parents=True, exist_ok=True)
@@ -366,29 +403,58 @@ def openfacts_product_matches_brand(url: str, brand: str, *, timeout: int = 10)
if not tokens:
return True
key = f"{api_host}:{barcode}:{(brand or '').lower()}"
# "v2:" - entries written before the fail-open verdicts stopped being
# cached are ignored rather than trusted; they age out with the TTL.
key = f"v2:{api_host}:{barcode}:{(brand or '').lower()}"
cached = _cache_get(key)
if cached is not None:
return cached
verdict = True
try:
resp = requests.get(
f"https://{api_host}/api/v2/product/{barcode}.json",
params={"fields": "brands,product_name"},
timeout=timeout,
headers={"User-Agent": _BROWSER_UA},
)
# ONLY A REAL ANSWER IS CACHED. The fail-open True for a lookup that did
# not happen - a 429 or 503 from Open*Facts, a non-JSON body - used to be
# written to the cache too, for thirty days. A Dabur ingestion issued a
# hundred lookups in two minutes, the tail of them were throttled, and
# Kellogg's "Miel Pops", a Toblerone and a Nature Valley bar were filed as
# Dabur products; "Dabur Honey 1kg" then showed the Kellogg's honey. A
# throttled lookup still fails open for THIS call, but the next call asks
# again.
payload = None
for attempt in range(_OFF_LOOKUP_ATTEMPTS):
_pace_off_lookup()
try:
resp = requests.get(
f"https://{api_host}/api/v2/product/{barcode}.json",
params={"fields": "brands,product_name"},
timeout=timeout,
headers={"User-Agent": _BROWSER_UA},
)
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
return True
if resp.ok:
product = (resp.json() or {}).get("product") or {}
haystack = (
f"{product.get('brands') or ''} {product.get('product_name') or ''}"
).lower()
if haystack.strip():
verdict = any(t in haystack for t in tokens)
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
try:
payload = resp.json() or {}
except ValueError:
return True
break
if resp.status_code == 429 or resp.status_code >= 500:
# Throttled or unwell. One paced retry is cheap and turns most of
# these into a real answer; a second failure fails open, uncached.
time.sleep(_OFF_RETRY_SLEEP_SECONDS * (attempt + 1))
continue
return True
if payload is None:
return True
product = payload.get("product") or {}
if not product and payload.get("status") == 0:
# Open*Facts positively says: no such barcode. Nothing to compare
# against, and asking again will not change that - cache the open verdict.
_cache_set(key, True)
return True
haystack = f"{product.get('brands') or ''} {product.get('product_name') or ''}".lower()
if not haystack.strip():
return True
verdict = any(t in haystack for t in tokens)
_cache_set(key, verdict)
return verdict
@@ -398,11 +464,25 @@ def openfacts_product_matches_brand(url: str, brand: str, *, timeout: int = 10)
# ---------------------------------------------------------------------------
@dataclass
class PrimaryChoice:
"""The outcome of choosing a product's `image_url` from its candidates."""
"""The outcome of choosing a product's `image_url` from its candidates.
`ordered` is every candidate, best first. `eligible` is the subset that
may be SHOWN - tiers 1 and 2 - and is what the store pipeline persists as
`image_urls`. The two differ for a reason that was learned the hard way:
tier 3 was kept in `image_urls` "for review, never promoted", but the
product card falls back to `image_urls[0]` whenever `image_url` is empty
and the product modal shows the whole list as a gallery. So for "Dabur
Honey 1kg" the gate correctly withheld the primary - every candidate was
another company's honey - and the UI displayed those very honeys anyway.
A candidate the gate would not promote must not be stored where the UI
will promote it.
"""
primary: Optional[str]
ordered: List[str]
reason: str
eligible: List[str] = field(default_factory=list)
rejected: List[str] = field(default_factory=list)
def choose_primary(
@@ -424,7 +504,10 @@ def choose_primary(
catalog_engine's comment protects, where dropping uncorroborated
candidates would leave real products with no image at all.
3. The URL could have named the product and did not - a human-authored
Commons filename, say. Kept in `image_urls`, never promoted.
Commons filename, say - or an Open*Facts photo whose barcode belongs
to another brand. Returned in `ordered` and `rejected`, never in
`eligible`, and the store pipeline does not persist it (see
PrimaryChoice for why "kept for review" was not safe).
Nothing eligible means `primary is None`. A blank image renders as the
brand monogram, which is honest; another company's product is not, and it
@@ -466,12 +549,16 @@ def choose_primary(
tier3.sort(key=looks_like_person_photo)
ordered = tier1 + tier2 + tier3
eligible = tier1 + tier2
if tier1:
return PrimaryChoice(tier1[0], ordered, "url names the product")
return PrimaryChoice(tier1[0], ordered, "url names the product",
eligible=eligible, rejected=tier3)
if tier2:
return PrimaryChoice(tier2[0], ordered, "opaque path on a known image host")
return PrimaryChoice(tier2[0], ordered, "opaque path on a known image host",
eligible=eligible, rejected=tier3)
return PrimaryChoice(
None,
ordered,
"no candidate names this product; primary withheld rather than guessed",
eligible=eligible, rejected=tier3,
)

View File

@@ -114,11 +114,44 @@ def _query_openfacts(query: str, max_results: int) -> list:
return []
def _openfacts_product_is_brand(product: dict, brand: Optional[str]) -> bool:
"""Does this Open*Facts record belong to `brand`?
Open*Facts' free-text search matches on the NAME, not the brand: asked for
"Dabur Honey 1kg" it answered with a UK, a French, a Swiss and a Spanish
honey, every one a real front-of-pack photo of somebody else's product.
The record says who made it - `brands` / `brands_tags` - and that field
used to be thrown away here, leaving a per-URL API round trip downstream
(`image_corroboration.openfacts_product_matches_brand`) as the only thing
standing between those photos and the catalogue. Check it at the source.
Fails OPEN on a record with no brand at all: an unlabelled record is not
evidence of another brand, and the downstream gate still runs.
"""
if not brand:
return True
from app.services.image_corroboration import brand_tokens
tokens = brand_tokens(brand)
if not tokens:
return True
tags = product.get("brands_tags") or []
haystack = " ".join([str(product.get("brands") or "")] + [str(t) for t in tags]).lower()
if not haystack.strip():
return True
return any(t in haystack for t in tokens)
def find_images_openfacts(title: str, brand: Optional[str] = None, max_results: int = 20) -> list:
"""Query the Open *Facts family of open product databases for real
product photos. No API key required. Falls back from a brand+title
query to a title-only query if the combined query is too specific to
match anything (small/regional brand name variants are a common case)."""
match anything (small/regional brand name variants are a common case).
Only records whose own `brands` field names our brand are used - see
`_openfacts_product_is_brand`. The title-only fallback makes this filter
load-bearing: "Honey 1kg" matches every honey on the site.
"""
if not USE_OPEN_FACTS:
return []
@@ -126,12 +159,12 @@ def find_images_openfacts(title: str, brand: Optional[str] = None, max_results:
if not query:
return []
products = _query_openfacts(query, max_results)
products = [p for p in _query_openfacts(query, max_results) if _openfacts_product_is_brand(p, brand)]
if not products and brand and title:
# Combined "brand + title" query found nothing - retry with just
# the title, since Open*Facts' free-text search is exact-ish and
# brand naming conventions vary (e.g. "Dettol" vs "Reckitt Dettol").
products = _query_openfacts(title, max_results)
products = [p for p in _query_openfacts(title, max_results) if _openfacts_product_is_brand(p, brand)]
urls: List[str] = []
for product in products:

View File

@@ -332,16 +332,29 @@ def fetch_brand_catalog_exhaustive(brand: str, max_products: int = 300) -> Dict[
return {"brand": brand, "products": products_out}
def fetch_product_details(brand: str, product_title: str) -> Dict[str, Any] | None:
def fetch_product_details(brand: str, product_title: str,
category: str | None = None,
size: str | None = None) -> Dict[str, Any] | None:
"""Get details for a single product: description, image_url, pricing fields.
Returns a dict with keys: description, image_url, size_variants, price_ranges, price_range, provider_examples.
`category` and `size` are optional context. A 1.5b model asked about
"Naga Maida" alone will happily describe a curry; told it is a 500g pack
in Flours & Grains it describes refined wheat flour. Both are facts the
caller already holds, so they cost nothing to pass.
"""
if not _ensure_client():
return None
context = f"Brand: {brand}\nProduct: {product_title}\n"
if category:
context += f"Category: {category}\n"
if size:
context += f"Pack size: {size}\n"
user_prompt = (
f"Brand: {brand}\nProduct: {product_title}\n"
"Return strictly JSON with keys: description, image_url?, size_variants?, price_ranges?, price_range?, provider_examples?.\n"
"description must be <=160 chars, concise and factual."
context
+ "Return strictly JSON with keys: description, image_url?, size_variants?, price_ranges?, price_range?, provider_examples?.\n"
"description: 1-2 sentences, <=220 chars, factual - what the product is, "
"its form and its typical use. No marketing adjectives, no claims you cannot know."
)
text = _generate(SYSTEM_PROMPT, user_prompt)
if not text:

View File

@@ -5,7 +5,7 @@ WHY THIS FILE EXISTS
--------------------
Open Food Facts answers "does this product exist" for food, and answers it
well. It does not answer it for anything else: `off_bulk` queries only
`search.openfoodfacts.org`, so a toothpaste or a detergent has no
Open Food Facts' food database, so a toothpaste or a detergent has no
product-discovery source at all and its entire catalogue is language-model
output. Measured 2026-09-10, India-tagged coverage on the sibling databases is
2-13 products per non-food brand against Britannia's 218 on OFF - real, but