Brand valid image generation
This commit is contained in:
@@ -107,6 +107,7 @@ from app.infrastructure.settings import (
|
||||
)
|
||||
from app.services import active_brands, brand_store, ollama_service, retail_presence
|
||||
from app.services.brand_registry import (
|
||||
get_brand_store_domain,
|
||||
get_fssai_license,
|
||||
get_known_sub_brands,
|
||||
resolve_parent_brand,
|
||||
@@ -449,19 +450,29 @@ def _existing_title_index(brand: str) -> Dict[str, str]:
|
||||
# ---------------------------------------------------------------------------
|
||||
# Source A - Open Food Facts (ground truth)
|
||||
# ---------------------------------------------------------------------------
|
||||
class OpenFactsUnavailable(RuntimeError):
|
||||
"""Open Food Facts could not be consulted - as opposed to consulted and
|
||||
found empty. `discover_brand_products` turns this into a warning that says
|
||||
so, because "OFF has nothing for this brand" is a claim about the brand
|
||||
and "OFF was down" is not."""
|
||||
|
||||
|
||||
def _from_open_facts(brand: str, *, refresh: bool = False) -> List[Dict[str, Any]]:
|
||||
"""Real products for `brand`, with a real GTIN and often a real pack size.
|
||||
|
||||
`off_bulk.fetch_brand_corpus` is already disk-cached, retry-wrapped and
|
||||
paced at 2s between pages, and `scripts/backfill_barcodes_from_off.py`
|
||||
already imports it from outside the enrichment package, so reaching for it
|
||||
here is an established pattern rather than a new one.
|
||||
`off_bulk.fetch_brand_corpus_result` is disk-cached, retry-wrapped and
|
||||
paced at 6s between 100-row pages. Raises `OpenFactsUnavailable` when the fetch
|
||||
did not complete and produced nothing; anything else is returned as-is,
|
||||
an empty list meaning OFF genuinely has no products under this brand tag.
|
||||
"""
|
||||
try:
|
||||
hits = off_bulk.fetch_brand_corpus(brand, refresh=refresh)
|
||||
result = off_bulk.fetch_brand_corpus_result(brand, refresh=refresh)
|
||||
except Exception as exc: # noqa: BLE001 - a dead OFF must not kill discovery
|
||||
logger.warning("Open Food Facts lookup failed for %r: %s", brand, exc)
|
||||
return []
|
||||
raise OpenFactsUnavailable(str(exc) or exc.__class__.__name__) from exc
|
||||
if result.error and not result.hits:
|
||||
raise OpenFactsUnavailable(result.error)
|
||||
hits = result.hits
|
||||
|
||||
out: List[Dict[str, Any]] = []
|
||||
for hit in hits or ():
|
||||
@@ -759,10 +770,10 @@ def _resolve_sizes(candidate: Dict[str, Any], title: str, category: str,
|
||||
which is what makes `image_id` reproducible - the name is half of it.
|
||||
|
||||
The LLM's list is a guess and comes last. Only when all three are empty is
|
||||
the cell left blank, which hands the decision to stage 4's
|
||||
`default_size_variants` - and that invents 100g/250g/500g, which
|
||||
`_sizes_for`'s own comment identifies as the mechanism behind observed
|
||||
catalogue drift. Leaving it blank is the last resort, not the default.
|
||||
the cell left blank; stage 4 then stores ONE unsized row (it no longer
|
||||
invents 100g/250g/500g - see `_sizes_for`, which names that invention as
|
||||
the mechanism behind observed catalogue drift). Blank is honest, but a
|
||||
real size is what makes the row a distinct pack, so it is worth finding.
|
||||
"""
|
||||
in_title = _TITLE_SIZE_RE.search(title or "")
|
||||
if in_title:
|
||||
@@ -850,25 +861,21 @@ def _build_product(brand: str, candidate: Dict[str, Any], *, max_sizes: int,
|
||||
# semantic search.
|
||||
#
|
||||
# A blank description is honest and already handled: `_to_storage_row`
|
||||
# substitutes "<name> <size> from <brand>." exactly as it does for a store
|
||||
# sheet that left the column empty.
|
||||
# substitutes a short factual line (name, size, category, brand) exactly
|
||||
# as it does for a store sheet that left the column empty.
|
||||
description = (candidate.get("description") or "").strip()
|
||||
|
||||
# A BARCODE IDENTIFIES ONE PACK, SO IT ONLY SURVIVES A KNOWN PACK SIZE.
|
||||
# With no real size, stage 4 falls back to `default_size_variants` and
|
||||
# invents 100g/250g/500g - and because every column is copied into each
|
||||
# exploded variant, one real GTIN would be stamped onto three packs, two of
|
||||
# which do not exist. That is a worse error than a blank barcode: it is
|
||||
# wrong data that looks authoritative, and `nutrition_by_barcode` would
|
||||
# happily resolve all three to the same product.
|
||||
# A BARCODE IDENTIFIES ONE PACK. That used to mean dropping it whenever no
|
||||
# pack size was known, because stage 4 then invented 100g/250g/500g and
|
||||
# every column is copied into each exploded variant - one real GTIN
|
||||
# stamped onto three packs, two of which did not exist. Stage 4 now stores
|
||||
# a single unsized row when it has no size, and one GTIN on one row is
|
||||
# exactly what a GTIN means, so the barcode survives; the row is flagged
|
||||
# so a reviewer knows the pack size is still unknown.
|
||||
barcode = candidate.get("barcode")
|
||||
notes: List[str] = []
|
||||
if barcode and not sizes:
|
||||
notes.append(
|
||||
"barcode dropped: no real pack size is known, and stage 4 will "
|
||||
"invent several - a GTIN names one pack, not three"
|
||||
)
|
||||
barcode = None
|
||||
notes.append("pack size unknown: the barcode names one pack, but its size was not found")
|
||||
|
||||
shape = {"title": title, "category": category_hint, "size_variants": sizes,
|
||||
"description": description}
|
||||
@@ -929,12 +936,25 @@ def discover_brand_products(
|
||||
registry_terms = _registry_terms(brand)
|
||||
fssai = get_fssai_license(brand)
|
||||
|
||||
off_candidates = _from_open_facts(brand, refresh=refresh_corpus) if use_openfacts else []
|
||||
if use_openfacts and not off_candidates:
|
||||
warnings.append(
|
||||
"Open Food Facts returned nothing for this brand. Every row below "
|
||||
"rests on the language model alone - review them individually."
|
||||
)
|
||||
# Three distinct outcomes, three distinct warnings. "OFF has nothing under
|
||||
# this brand tag" is a fact about the brand; "OFF could not be reached" is
|
||||
# a fact about the network and must not be reported as the former - that
|
||||
# is how Naga (2 real rows on OFF) was being shown as unknown to it.
|
||||
off_candidates: List[Dict[str, Any]] = []
|
||||
if use_openfacts:
|
||||
try:
|
||||
off_candidates = _from_open_facts(brand, refresh=refresh_corpus)
|
||||
except OpenFactsUnavailable as exc:
|
||||
warnings.append(
|
||||
f"Open Food Facts could not be reached ({exc}). Nothing was "
|
||||
"cached, so the next preview will try again; the rows below "
|
||||
"come from the brand's shop and the language model only."
|
||||
)
|
||||
else:
|
||||
if not off_candidates:
|
||||
warnings.append(
|
||||
"Open Food Facts has no products tagged with this brand."
|
||||
)
|
||||
off_keys = {}
|
||||
for candidate in off_candidates:
|
||||
key = _normalise_title(brand, candidate["title"])
|
||||
@@ -951,12 +971,26 @@ def discover_brand_products(
|
||||
if store_candidates:
|
||||
logger.info("%s: %d products from the brand's own shop", brand, len(store_candidates))
|
||||
elif use_store and not off_candidates:
|
||||
# Say what was actually checked. An unregistered storefront is not a
|
||||
# storefront that "has nothing" - nobody has looked.
|
||||
store_domain = get_brand_store_domain(brand)
|
||||
if store_domain:
|
||||
warnings.append(
|
||||
f"{store_domain} is registered as this brand's shop but its "
|
||||
"catalogue has not been fetched - run "
|
||||
f"scripts/backfill_brand_stores.py --brand \"{brand}\" to load it."
|
||||
)
|
||||
else:
|
||||
warnings.append(
|
||||
"No verified storefront is registered for this brand. If it "
|
||||
"has its own online shop, add it to "
|
||||
"brand_registry.BRAND_STORE_DOMAINS and run "
|
||||
"scripts/backfill_brand_stores.py to populate a real catalogue."
|
||||
)
|
||||
if not off_candidates and not store_candidates:
|
||||
warnings.append(
|
||||
"Neither Open Food Facts nor a brand storefront has anything for "
|
||||
"this brand. Every row below rests on the language model alone - "
|
||||
"review them individually. If this brand has its own online shop, "
|
||||
"adding it to brand_registry.BRAND_STORE_DOMAINS and running "
|
||||
"scripts/backfill_brand_stores.py will populate a real catalogue."
|
||||
"Every row below rests on the language model alone - review them "
|
||||
"individually."
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -11,9 +11,12 @@ never heard of there is nothing to verify, and the catalogue is whatever
|
||||
`qwen2.5:1.5b` invents.
|
||||
|
||||
Measured 2026-09-10, Open Food Facts hits: Udhaiyam 0, Tenali Double Horse 0,
|
||||
Gopuram 0, Double Horse 14, and Naga 2 - where both "Naga" rows are a UK /
|
||||
Bangladeshi pickle brand, not the Tamil Nadu one. Regional South Indian brands
|
||||
are simply not in that database.
|
||||
Gopuram 0, Double Horse 14. (An earlier note here said Naga's two rows were a
|
||||
UK / Bangladeshi pickle brand - that was the old free-text `brands:Naga`
|
||||
Search-a-licious query matching "Mr Naga" and "Bombay Naga Jhal". Under the
|
||||
exact `brands_tags=naga` filter the v2 API returns two real Naga Limited rows,
|
||||
Sooji 500 g and Maida 500 g, both on the 890 GS1 prefix - see off_bulk.py.)
|
||||
Regional South Indian brands are still thin in that database.
|
||||
|
||||
They are, however, on their own shop, and modern storefronts publish their
|
||||
whole catalogue as structured JSON:
|
||||
|
||||
@@ -10,34 +10,50 @@ resulting cascade "misses far more often than it hits". That module is wired
|
||||
into the live enrichment pipeline and is deliberately NOT touched by this file.
|
||||
|
||||
This module inverts the problem: fetch a brand's **entire** OFF catalogue in one
|
||||
or two requests from the Search-a-licious endpoint, cache it on disk, then match
|
||||
every one of our products against that corpus offline. A whole-catalogue
|
||||
backfill costs ~5 HTTP requests instead of ~600, and re-tuning the similarity
|
||||
threshold costs zero network because the corpus is cached.
|
||||
or two requests from the v2 search API, cache it on disk, then match every one
|
||||
of our products against that corpus offline. A whole-catalogue backfill costs
|
||||
~5 HTTP requests instead of ~600, and re-tuning the similarity threshold costs
|
||||
zero network because the corpus is cached.
|
||||
|
||||
Nothing here is imported by the running app - the only consumer is
|
||||
`scripts/backfill_barcodes_from_off.py`. Everything except `fetch_brand_corpus`
|
||||
is a pure function so it can be unit-tested without network or database.
|
||||
`fetch_brand_corpus` is the one network function here and is shared by the
|
||||
backfill scripts, `product_grounding`, `post_ingest_barcodes` and
|
||||
`brand_discovery`. Everything else is a pure function so it can be unit-tested
|
||||
without network or database.
|
||||
|
||||
ENDPOINT NOTES (verified empirically, 2026-09)
|
||||
GET https://search.openfoodfacts.org/search
|
||||
?q=brands:amul AND countries_tags:"en:india"
|
||||
ENDPOINT NOTES (verified empirically, 2026-09-11)
|
||||
GET https://world.openfoodfacts.org/api/v2/search
|
||||
?brands_tags=amul
|
||||
&countries_tags=en:india
|
||||
&fields=code,product_name,quantity,...
|
||||
&page_size=250
|
||||
* The double quotes around "en:india" are REQUIRED. Without them the query
|
||||
parses as a bare term and silently returns count=0 rather than erroring.
|
||||
* page_size up to 1000 is accepted; 250 keeps responses small.
|
||||
&page_size=100&page=1
|
||||
* `brands_tags` is an EXACT match on the brand's tag slug ("naga", not
|
||||
"Naga"), which is what a brand catalogue needs. The earlier Search-a-licious
|
||||
query `q=brands:Naga` was a free-text match: it returned "Mr Naga" and
|
||||
"Bombay Naga Jhal" - other companies - and MISSED the real Naga rows,
|
||||
because Search-a-licious reads a separate Elasticsearch index that was
|
||||
stale for them (barcode 8906011830068 was still filed brandless under
|
||||
Kuwait). The v2 API reads the live product database. Measured on the same
|
||||
day: Amul 216 products here vs 140 there; Naga 2 vs 0.
|
||||
* `page_count` in this API is the number of products ON THIS PAGE, not the
|
||||
number of pages, and `page_size` in the RESPONSE is what the server
|
||||
actually applied (a larger request is capped to 100 without complaint).
|
||||
Paginate from `count` / that served `page_size`.
|
||||
* OFF rate-limits all search endpoints to 10 requests/minute per IP.
|
||||
PAUSE_SECONDS keeps a multi-page brand under that.
|
||||
* `world.openfoodfacts.org` intermittently serves an HTML "Page temporarily
|
||||
unavailable" page with a 200 status, so every response is content-type
|
||||
checked before parsing.
|
||||
unavailable" page with a 200 status, and plain 503s under load, so every
|
||||
response is status- and content-type-checked before parsing, and a failed
|
||||
fetch is never written to the cache as an empty corpus.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import math
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple
|
||||
|
||||
@@ -63,21 +79,50 @@ from app.services.quantity_utils import quantities_match
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
SEARCH_URL = "https://search.openfoodfacts.org/search"
|
||||
SEARCH_URL = "https://world.openfoodfacts.org/api/v2/search"
|
||||
|
||||
# OFF's usage policy requires a contactable custom User-Agent; requests sent
|
||||
# with the default python-requests agent are treated as anonymous crawling.
|
||||
USER_AGENT = "BrandCatalogRAG/1.0 (suriya@tenext.in)"
|
||||
|
||||
FIELDS = "code,product_name,product_name_en,brands,quantity,countries_tags"
|
||||
PAGE_SIZE = 250
|
||||
MAX_PAGES = 10 # 2500 products per brand is far beyond any real brand
|
||||
PAUSE_SECONDS = 2.0 # search endpoints are the strictly-limited ones
|
||||
PAGE_SIZE = 100 # the v2 API caps this at 100 and silently serves that
|
||||
MAX_PAGES = 25 # 2500 products per brand is far beyond any real brand
|
||||
PAUSE_SECONDS = 6.0 # 10 search requests/minute is OFF's published limit
|
||||
|
||||
# Bumped when the cache file's meaning changes. Files written before this key
|
||||
# existed came from the Search-a-licious endpoint; an EMPTY one of those is
|
||||
# more likely to be that endpoint's stale index (or a swallowed 503) than a
|
||||
# real absence, so it is re-fetched. A non-empty one is real data and is kept.
|
||||
CACHE_SCHEMA = 2
|
||||
|
||||
# backend/app/services/enrichment/barcode/sources/off_bulk.py -> backend/
|
||||
_BACKEND_DIR = Path(__file__).resolve().parents[5]
|
||||
CACHE_DIR = _BACKEND_DIR / "data" / "cache" / "off_brand_corpus"
|
||||
|
||||
|
||||
class OffUnavailable(requests.exceptions.ConnectionError):
|
||||
"""OFF answered but could not serve (429 / 5xx).
|
||||
|
||||
Subclasses ConnectionError so `retry.with_retry` treats it exactly like a
|
||||
dropped socket - a few seconds later the same request usually succeeds -
|
||||
without widening the retry set for every other barcode source.
|
||||
"""
|
||||
|
||||
|
||||
@dataclass
|
||||
class CorpusFetch:
|
||||
"""What `fetch_brand_corpus_result` learned.
|
||||
|
||||
`error` is None when OFF answered every page, even if it answered with
|
||||
nothing - that is a real "this brand is not on Open Food Facts". When
|
||||
`error` is set the fetch did not complete and NOTHING was cached, so a
|
||||
caller can say "could not be reached" instead of "has nothing".
|
||||
"""
|
||||
hits: List[Dict[str, Any]] = field(default_factory=list)
|
||||
error: Optional[str] = None
|
||||
from_cache: bool = False
|
||||
|
||||
# Pack sizes embedded in a product name ("Marie Gold 250g", "Butter 1L").
|
||||
# Mirrors the unit list in scripts/backfill_nutrition_from_barcodes.py, widened
|
||||
# with the count-based units this catalog also uses.
|
||||
@@ -122,84 +167,149 @@ def _cache_path(slug: str, cache_dir: Optional[Path] = None) -> Path:
|
||||
return (cache_dir or CACHE_DIR) / f"{slug}.json"
|
||||
|
||||
|
||||
def brand_tag_slug(brand: str) -> str:
|
||||
"""The brand as Open Food Facts tags it: lowercase, runs of anything that
|
||||
is not a letter or digit collapsed to one hyphen. "Hindustan Unilever" ->
|
||||
"hindustan-unilever", "P&G" -> "p-g", "Naga" -> "naga". OFF matches
|
||||
`brands_tags` case-insensitively, but sending the slug form is what its
|
||||
own site does and avoids depending on that."""
|
||||
return re.sub(r"[^a-z0-9]+", "-", (brand or "").lower()).strip("-")
|
||||
|
||||
|
||||
@with_retry(max_attempts=3, min_wait=2.0, max_wait=10.0)
|
||||
def _get_page(brand: str, country: Optional[str], page: int) -> Dict[str, Any]:
|
||||
"""One page of OFF search results. Raises on transport errors (retried by
|
||||
the decorator); returns an empty result dict for any response that is not
|
||||
parseable JSON, which is how the HTML "temporarily unavailable" page and
|
||||
any future error page are absorbed without killing the run."""
|
||||
query = f"brands:{brand}"
|
||||
def _get_page(brand: str, country: Optional[str], page: int) -> Optional[Dict[str, Any]]:
|
||||
"""One page of OFF v2 search results.
|
||||
|
||||
Raises on transport errors and on 429 / 5xx (both retried by the
|
||||
decorator); returns None for any other response that is not parseable
|
||||
JSON - an HTML "temporarily unavailable" page, a 4xx, a truncated body.
|
||||
None means "this page FAILED", which the caller must keep distinct from a
|
||||
page that parsed fine and simply held no products.
|
||||
"""
|
||||
params: Dict[str, Any] = {
|
||||
"brands_tags": brand_tag_slug(brand),
|
||||
"fields": FIELDS,
|
||||
"page_size": PAGE_SIZE,
|
||||
"page": page,
|
||||
}
|
||||
if country:
|
||||
# The quotes are load-bearing - see the module docstring.
|
||||
query += f' AND countries_tags:"en:{country}"'
|
||||
params["countries_tags"] = f"en:{country}"
|
||||
|
||||
resp = requests.get(
|
||||
SEARCH_URL,
|
||||
params={"q": query, "fields": FIELDS, "page_size": PAGE_SIZE, "page": page},
|
||||
params=params,
|
||||
headers={"User-Agent": USER_AGENT, "Accept": "application/json"},
|
||||
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
|
||||
)
|
||||
|
||||
if resp.status_code == 429 or resp.status_code >= 500:
|
||||
raise OffUnavailable(f"HTTP {resp.status_code} from Open Food Facts")
|
||||
if resp.status_code != 200:
|
||||
logger.warning("OFF search returned HTTP %s for brand %r page %s",
|
||||
resp.status_code, brand, page)
|
||||
return {}
|
||||
return None
|
||||
if "json" not in (resp.headers.get("content-type") or "").lower():
|
||||
logger.warning("OFF search returned non-JSON (%s) for brand %r - "
|
||||
"the service is probably serving an error page",
|
||||
resp.headers.get("content-type"), brand)
|
||||
return {}
|
||||
return None
|
||||
try:
|
||||
payload = resp.json()
|
||||
except ValueError as e:
|
||||
logger.warning("OFF search returned unparseable JSON for brand %r: %s", brand, e)
|
||||
return {}
|
||||
return payload if isinstance(payload, dict) else {}
|
||||
return None
|
||||
return payload if isinstance(payload, dict) else None
|
||||
|
||||
|
||||
def fetch_brand_corpus(brand: str,
|
||||
country: Optional[str] = None,
|
||||
refresh: bool = False,
|
||||
cache_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
|
||||
"""Every OFF product for `brand`, from disk cache unless `refresh`.
|
||||
def _read_cache(path: Path, brand: str) -> Optional[List[Dict[str, Any]]]:
|
||||
"""Cached hits, or None when the cache must not be trusted: unreadable, or
|
||||
an empty file from before CACHE_SCHEMA existed (see that constant)."""
|
||||
try:
|
||||
cached = json.loads(path.read_text(encoding="utf-8"))
|
||||
except Exception as e: # noqa: BLE001 - a corrupt cache must not be fatal
|
||||
logger.warning(" Ignoring unreadable OFF cache %s: %s", path.name, e)
|
||||
return None
|
||||
hits = cached.get("hits") or []
|
||||
if not hits and cached.get("schema") != CACHE_SCHEMA:
|
||||
logger.info(" Ignoring empty pre-v%s OFF cache for %r - re-fetching",
|
||||
CACHE_SCHEMA, brand)
|
||||
return None
|
||||
logger.info(" OFF corpus for %r: %d product(s) (cached %s)",
|
||||
brand, len(hits), cached.get("fetched_at_human", "?"))
|
||||
return hits
|
||||
|
||||
|
||||
def fetch_brand_corpus_result(brand: str,
|
||||
country: Optional[str] = None,
|
||||
refresh: bool = False,
|
||||
cache_dir: Optional[Path] = None) -> CorpusFetch:
|
||||
"""Every OFF product for `brand`, from disk cache unless `refresh`, plus
|
||||
whether the fetch actually completed.
|
||||
|
||||
Hits with no usable product name are dropped here rather than at match time
|
||||
(6 of 146 Amul hits, 1 of 233 Britannia hits) - a nameless hit can never
|
||||
clear a name-similarity threshold, so carrying it forward only inflates the
|
||||
corpus. Returns [] rather than raising when OFF is unreachable, so one bad
|
||||
brand does not abort a multi-brand backfill.
|
||||
corpus.
|
||||
|
||||
A fetch that fails part-way returns what it got with `error` set and
|
||||
writes NO cache file. Caching a failure as `hits: []` is how "Open Food
|
||||
Facts has nothing for this brand" was being asserted for brands OFF had
|
||||
simply been too busy to answer about.
|
||||
"""
|
||||
country = BARCODE_COUNTRY_TAG if country is None else (country or None)
|
||||
slug = re.sub(r"[^a-z0-9]+", "_", brand.lower()).strip("_")
|
||||
path = _cache_path(slug, cache_dir)
|
||||
|
||||
if not refresh and path.exists():
|
||||
try:
|
||||
cached = json.loads(path.read_text(encoding="utf-8"))
|
||||
hits = cached.get("hits") or []
|
||||
logger.info(" OFF corpus for %r: %d product(s) (cached %s)",
|
||||
brand, len(hits), cached.get("fetched_at_human", "?"))
|
||||
return hits
|
||||
except Exception as e: # noqa: BLE001 - a corrupt cache must not be fatal
|
||||
logger.warning(" Ignoring unreadable OFF cache %s: %s", path.name, e)
|
||||
cached = _read_cache(path, brand)
|
||||
if cached is not None:
|
||||
return CorpusFetch(hits=cached, from_cache=True)
|
||||
|
||||
hits: List[Dict[str, Any]] = []
|
||||
error: Optional[str] = None
|
||||
page = 1
|
||||
while page <= MAX_PAGES:
|
||||
total_pages = 1
|
||||
while page <= min(total_pages, MAX_PAGES):
|
||||
if page > 1:
|
||||
time.sleep(PAUSE_SECONDS)
|
||||
payload = _get_page(brand, country, page)
|
||||
batch = payload.get("hits") or []
|
||||
try:
|
||||
payload = _get_page(brand, country, page)
|
||||
except requests.exceptions.RequestException as e:
|
||||
# Retries exhausted (OffUnavailable is a ConnectionError too).
|
||||
error = str(e) or e.__class__.__name__
|
||||
break
|
||||
if payload is None:
|
||||
error = "Open Food Facts returned an unusable response"
|
||||
break
|
||||
batch = payload.get("products") or []
|
||||
hits.extend(h for h in batch if isinstance(h, dict) and (h.get("product_name") or "").strip())
|
||||
page_count = payload.get("page_count") or 0
|
||||
if page >= page_count or not batch:
|
||||
if page == 1:
|
||||
# `page_count` is products-on-this-page in the v2 API; the real
|
||||
# page total is `count` over the page size the SERVER applied -
|
||||
# it silently caps the requested size (250 asked, 100 served for
|
||||
# Amul), so dividing by PAGE_SIZE under-pages.
|
||||
count = payload.get("count") or 0
|
||||
served = payload.get("page_size") or len(batch) or PAGE_SIZE
|
||||
try:
|
||||
total_pages = max(1, math.ceil(int(count) / int(served)))
|
||||
except (TypeError, ValueError, ZeroDivisionError):
|
||||
total_pages = 1
|
||||
if not batch:
|
||||
break
|
||||
page += 1
|
||||
|
||||
if error:
|
||||
logger.warning(" OFF corpus for %r: fetch failed after %d usable product(s) - %s "
|
||||
"(nothing cached)", brand, len(hits), error)
|
||||
return CorpusFetch(hits=hits, error=error)
|
||||
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = path.with_suffix(".json.tmp")
|
||||
tmp.write_text(json.dumps({
|
||||
"schema": CACHE_SCHEMA,
|
||||
"endpoint": SEARCH_URL,
|
||||
"brand": brand,
|
||||
"brand_tag": brand_tag_slug(brand),
|
||||
"country": country,
|
||||
"fetched_at": time.time(),
|
||||
"fetched_at_human": time.strftime("%Y-%m-%d %H:%M:%S"),
|
||||
@@ -208,7 +318,20 @@ def fetch_brand_corpus(brand: str,
|
||||
os.replace(tmp, path)
|
||||
|
||||
logger.info(" OFF corpus for %r: %d usable product(s) fetched", brand, len(hits))
|
||||
return hits
|
||||
return CorpusFetch(hits=hits)
|
||||
|
||||
|
||||
def fetch_brand_corpus(brand: str,
|
||||
country: Optional[str] = None,
|
||||
refresh: bool = False,
|
||||
cache_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
|
||||
"""`fetch_brand_corpus_result(...).hits` - the list-only form every
|
||||
backfill and ingestion caller uses. Returns [] rather than raising when
|
||||
OFF is unreachable, so one bad brand does not abort a multi-brand run;
|
||||
callers that need to tell "unreachable" from "empty" use the result form.
|
||||
"""
|
||||
return fetch_brand_corpus_result(brand, country=country, refresh=refresh,
|
||||
cache_dir=cache_dir).hits
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -243,6 +243,16 @@ def _round2(value: Optional[float]) -> Optional[float]:
|
||||
return round(value, 2)
|
||||
|
||||
|
||||
def _held_number(value: object) -> Optional[float]:
|
||||
"""A price the row already carries, or None for blank / unparseable."""
|
||||
if value is None or (isinstance(value, str) and not value.strip()):
|
||||
return None
|
||||
try:
|
||||
return float(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def enrich_pricing_fields(product: dict) -> Dict[str, object]:
|
||||
"""Compute the pricing fields added by this stage for one catalog row.
|
||||
|
||||
@@ -253,8 +263,24 @@ def enrich_pricing_fields(product: dict) -> Dict[str, object]:
|
||||
`cost_price`, `profit_before_tax` and `profit_after_tax` are not
|
||||
derivable from the data this pipeline generates, so they stay None
|
||||
(exactly as the existing enriched exports store them).
|
||||
|
||||
A PRICE THE ROW ALREADY HOLDS WINS. A store sheet that says "Retail Price
|
||||
155" has stated a fact; the band derived from it is Rs143-167, and this
|
||||
stage used to read the band's ceiling back as "selling_price = 167" and
|
||||
hand that to `EnrichmentStage.apply`, which overwrites a held value with
|
||||
a different one (it only refuses to BLANK one). Held prices are therefore
|
||||
returned as None here so `apply` keeps them, the base for the tax is the
|
||||
held selling price, then the held final price, and only then the band.
|
||||
"""
|
||||
selling_price = _extract_selling_price(product.get("price_range"))
|
||||
held_selling = _held_number(product.get("selling_price"))
|
||||
held_final = _held_number(product.get("final_selling_price"))
|
||||
if held_selling is not None:
|
||||
selling_price = held_selling
|
||||
elif held_final is not None:
|
||||
selling_price = held_final
|
||||
else:
|
||||
selling_price = _extract_selling_price(product.get("price_range"))
|
||||
|
||||
gst = product.get("gst_percent")
|
||||
tax_amount = None
|
||||
final_selling_price = None
|
||||
@@ -263,10 +289,10 @@ def enrich_pricing_fields(product: dict) -> Dict[str, object]:
|
||||
final_selling_price = _round2(selling_price + tax_amount)
|
||||
|
||||
return {
|
||||
"selling_price": _round2(selling_price),
|
||||
"selling_price": None if held_selling is not None else _round2(selling_price),
|
||||
"cost_price": None,
|
||||
"tax_amount": tax_amount,
|
||||
"final_selling_price": final_selling_price,
|
||||
"final_selling_price": None if held_final is not None else final_selling_price,
|
||||
"profit_before_tax": None,
|
||||
"profit_after_tax": None,
|
||||
}
|
||||
|
||||
@@ -76,7 +76,7 @@ logger = logging.getLogger(__name__)
|
||||
# Matches scripts/backfill_barcodes_from_off.py, which measured it.
|
||||
DEFAULT_MIN_SIMILARITY = 0.88
|
||||
|
||||
BARCODE_SOURCE = "openfoodfacts_bulk (search.openfoodfacts.org)"
|
||||
BARCODE_SOURCE = "openfoodfacts_bulk (world.openfoodfacts.org/api/v2)"
|
||||
|
||||
|
||||
def _rows_needing_a_barcode(cur, table: str) -> List[Dict[str, Any]]:
|
||||
|
||||
@@ -63,7 +63,7 @@ import sqlite3
|
||||
import threading
|
||||
import time
|
||||
from contextlib import closing
|
||||
from dataclasses import dataclass
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Iterable, List, Optional, Sequence
|
||||
from urllib.parse import urlparse
|
||||
@@ -175,6 +175,23 @@ def distinctive_tokens(product_name: str, brand: str) -> List[str]:
|
||||
return out
|
||||
|
||||
|
||||
def _token_in(token: str, lowered_url: str) -> bool:
|
||||
"""Does the URL carry this word, allowing for the plural on either side?
|
||||
|
||||
The catalogue names "Aachi Appalams" and "Aachi Pickles"; the photo is
|
||||
`Aachi-Appalam-100-g-1.webp`. A whole-token substring test rejected the
|
||||
correct image for the plural and then promoted an opaque Amazon URL over
|
||||
it. The singular is tried as well - only for tokens long enough that
|
||||
stripping the "s" leaves a real word ("gems" -> "gem" is fine; "kgs" never
|
||||
gets here, size tokens are removed upstream).
|
||||
"""
|
||||
if token in lowered_url:
|
||||
return True
|
||||
if len(token) > 4 and token.endswith("s") and token[:-1] in lowered_url:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def search_key(product_name: str, brand: str) -> tuple:
|
||||
"""Collapse a trailing pack size so sizes of one product share a lookup.
|
||||
|
||||
@@ -213,7 +230,7 @@ def corroborate(url: str, product_name: str, brand: str) -> Corroboration:
|
||||
|
||||
distinctive = distinctive_tokens(product_name, brand)
|
||||
if distinctive:
|
||||
hit = next((t for t in distinctive if t in lowered), None)
|
||||
hit = next((t for t in distinctive if _token_in(t, lowered)), None)
|
||||
if hit:
|
||||
return Corroboration(True, f"url names {hit!r}")
|
||||
return Corroboration(
|
||||
@@ -272,6 +289,26 @@ _lock = threading.Lock()
|
||||
_initialized = False
|
||||
_CACHE_TTL_SECONDS = 30 * 24 * 3600
|
||||
|
||||
# Open*Facts allows 100 product reads a minute per IP. A brand ingestion or a
|
||||
# re-gate asks about every distinct barcode it meets, and the Dabur run issued
|
||||
# about a hundred in two minutes - the tail was throttled, and a throttled
|
||||
# lookup fails open, which is how a Kellogg's honey got past the gate for a
|
||||
# moment. Pacing the calls keeps the gate answering instead of guessing.
|
||||
_OFF_MIN_INTERVAL_SECONDS = 0.65
|
||||
_OFF_LOOKUP_ATTEMPTS = 2
|
||||
_OFF_RETRY_SLEEP_SECONDS = 3.0
|
||||
_off_last_call = 0.0
|
||||
_off_pace_lock = threading.Lock()
|
||||
|
||||
|
||||
def _pace_off_lookup() -> None:
|
||||
global _off_last_call
|
||||
with _off_pace_lock:
|
||||
wait = _OFF_MIN_INTERVAL_SECONDS - (time.monotonic() - _off_last_call)
|
||||
if wait > 0:
|
||||
time.sleep(wait)
|
||||
_off_last_call = time.monotonic()
|
||||
|
||||
|
||||
def _connect() -> sqlite3.Connection:
|
||||
_DB_PATH.parent.mkdir(parents=True, exist_ok=True)
|
||||
@@ -366,29 +403,58 @@ def openfacts_product_matches_brand(url: str, brand: str, *, timeout: int = 10)
|
||||
if not tokens:
|
||||
return True
|
||||
|
||||
key = f"{api_host}:{barcode}:{(brand or '').lower()}"
|
||||
# "v2:" - entries written before the fail-open verdicts stopped being
|
||||
# cached are ignored rather than trusted; they age out with the TTL.
|
||||
key = f"v2:{api_host}:{barcode}:{(brand or '').lower()}"
|
||||
cached = _cache_get(key)
|
||||
if cached is not None:
|
||||
return cached
|
||||
|
||||
verdict = True
|
||||
try:
|
||||
resp = requests.get(
|
||||
f"https://{api_host}/api/v2/product/{barcode}.json",
|
||||
params={"fields": "brands,product_name"},
|
||||
timeout=timeout,
|
||||
headers={"User-Agent": _BROWSER_UA},
|
||||
)
|
||||
# ONLY A REAL ANSWER IS CACHED. The fail-open True for a lookup that did
|
||||
# not happen - a 429 or 503 from Open*Facts, a non-JSON body - used to be
|
||||
# written to the cache too, for thirty days. A Dabur ingestion issued a
|
||||
# hundred lookups in two minutes, the tail of them were throttled, and
|
||||
# Kellogg's "Miel Pops", a Toblerone and a Nature Valley bar were filed as
|
||||
# Dabur products; "Dabur Honey 1kg" then showed the Kellogg's honey. A
|
||||
# throttled lookup still fails open for THIS call, but the next call asks
|
||||
# again.
|
||||
payload = None
|
||||
for attempt in range(_OFF_LOOKUP_ATTEMPTS):
|
||||
_pace_off_lookup()
|
||||
try:
|
||||
resp = requests.get(
|
||||
f"https://{api_host}/api/v2/product/{barcode}.json",
|
||||
params={"fields": "brands,product_name"},
|
||||
timeout=timeout,
|
||||
headers={"User-Agent": _BROWSER_UA},
|
||||
)
|
||||
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
|
||||
return True
|
||||
if resp.ok:
|
||||
product = (resp.json() or {}).get("product") or {}
|
||||
haystack = (
|
||||
f"{product.get('brands') or ''} {product.get('product_name') or ''}"
|
||||
).lower()
|
||||
if haystack.strip():
|
||||
verdict = any(t in haystack for t in tokens)
|
||||
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
|
||||
try:
|
||||
payload = resp.json() or {}
|
||||
except ValueError:
|
||||
return True
|
||||
break
|
||||
if resp.status_code == 429 or resp.status_code >= 500:
|
||||
# Throttled or unwell. One paced retry is cheap and turns most of
|
||||
# these into a real answer; a second failure fails open, uncached.
|
||||
time.sleep(_OFF_RETRY_SLEEP_SECONDS * (attempt + 1))
|
||||
continue
|
||||
return True
|
||||
if payload is None:
|
||||
return True
|
||||
|
||||
product = payload.get("product") or {}
|
||||
if not product and payload.get("status") == 0:
|
||||
# Open*Facts positively says: no such barcode. Nothing to compare
|
||||
# against, and asking again will not change that - cache the open verdict.
|
||||
_cache_set(key, True)
|
||||
return True
|
||||
haystack = f"{product.get('brands') or ''} {product.get('product_name') or ''}".lower()
|
||||
if not haystack.strip():
|
||||
return True
|
||||
verdict = any(t in haystack for t in tokens)
|
||||
_cache_set(key, verdict)
|
||||
return verdict
|
||||
|
||||
@@ -398,11 +464,25 @@ def openfacts_product_matches_brand(url: str, brand: str, *, timeout: int = 10)
|
||||
# ---------------------------------------------------------------------------
|
||||
@dataclass
|
||||
class PrimaryChoice:
|
||||
"""The outcome of choosing a product's `image_url` from its candidates."""
|
||||
"""The outcome of choosing a product's `image_url` from its candidates.
|
||||
|
||||
`ordered` is every candidate, best first. `eligible` is the subset that
|
||||
may be SHOWN - tiers 1 and 2 - and is what the store pipeline persists as
|
||||
`image_urls`. The two differ for a reason that was learned the hard way:
|
||||
tier 3 was kept in `image_urls` "for review, never promoted", but the
|
||||
product card falls back to `image_urls[0]` whenever `image_url` is empty
|
||||
and the product modal shows the whole list as a gallery. So for "Dabur
|
||||
Honey 1kg" the gate correctly withheld the primary - every candidate was
|
||||
another company's honey - and the UI displayed those very honeys anyway.
|
||||
A candidate the gate would not promote must not be stored where the UI
|
||||
will promote it.
|
||||
"""
|
||||
|
||||
primary: Optional[str]
|
||||
ordered: List[str]
|
||||
reason: str
|
||||
eligible: List[str] = field(default_factory=list)
|
||||
rejected: List[str] = field(default_factory=list)
|
||||
|
||||
|
||||
def choose_primary(
|
||||
@@ -424,7 +504,10 @@ def choose_primary(
|
||||
catalog_engine's comment protects, where dropping uncorroborated
|
||||
candidates would leave real products with no image at all.
|
||||
3. The URL could have named the product and did not - a human-authored
|
||||
Commons filename, say. Kept in `image_urls`, never promoted.
|
||||
Commons filename, say - or an Open*Facts photo whose barcode belongs
|
||||
to another brand. Returned in `ordered` and `rejected`, never in
|
||||
`eligible`, and the store pipeline does not persist it (see
|
||||
PrimaryChoice for why "kept for review" was not safe).
|
||||
|
||||
Nothing eligible means `primary is None`. A blank image renders as the
|
||||
brand monogram, which is honest; another company's product is not, and it
|
||||
@@ -466,12 +549,16 @@ def choose_primary(
|
||||
tier3.sort(key=looks_like_person_photo)
|
||||
|
||||
ordered = tier1 + tier2 + tier3
|
||||
eligible = tier1 + tier2
|
||||
if tier1:
|
||||
return PrimaryChoice(tier1[0], ordered, "url names the product")
|
||||
return PrimaryChoice(tier1[0], ordered, "url names the product",
|
||||
eligible=eligible, rejected=tier3)
|
||||
if tier2:
|
||||
return PrimaryChoice(tier2[0], ordered, "opaque path on a known image host")
|
||||
return PrimaryChoice(tier2[0], ordered, "opaque path on a known image host",
|
||||
eligible=eligible, rejected=tier3)
|
||||
return PrimaryChoice(
|
||||
None,
|
||||
ordered,
|
||||
"no candidate names this product; primary withheld rather than guessed",
|
||||
eligible=eligible, rejected=tier3,
|
||||
)
|
||||
|
||||
@@ -114,11 +114,44 @@ def _query_openfacts(query: str, max_results: int) -> list:
|
||||
return []
|
||||
|
||||
|
||||
def _openfacts_product_is_brand(product: dict, brand: Optional[str]) -> bool:
|
||||
"""Does this Open*Facts record belong to `brand`?
|
||||
|
||||
Open*Facts' free-text search matches on the NAME, not the brand: asked for
|
||||
"Dabur Honey 1kg" it answered with a UK, a French, a Swiss and a Spanish
|
||||
honey, every one a real front-of-pack photo of somebody else's product.
|
||||
The record says who made it - `brands` / `brands_tags` - and that field
|
||||
used to be thrown away here, leaving a per-URL API round trip downstream
|
||||
(`image_corroboration.openfacts_product_matches_brand`) as the only thing
|
||||
standing between those photos and the catalogue. Check it at the source.
|
||||
|
||||
Fails OPEN on a record with no brand at all: an unlabelled record is not
|
||||
evidence of another brand, and the downstream gate still runs.
|
||||
"""
|
||||
if not brand:
|
||||
return True
|
||||
from app.services.image_corroboration import brand_tokens
|
||||
|
||||
tokens = brand_tokens(brand)
|
||||
if not tokens:
|
||||
return True
|
||||
tags = product.get("brands_tags") or []
|
||||
haystack = " ".join([str(product.get("brands") or "")] + [str(t) for t in tags]).lower()
|
||||
if not haystack.strip():
|
||||
return True
|
||||
return any(t in haystack for t in tokens)
|
||||
|
||||
|
||||
def find_images_openfacts(title: str, brand: Optional[str] = None, max_results: int = 20) -> list:
|
||||
"""Query the Open *Facts family of open product databases for real
|
||||
product photos. No API key required. Falls back from a brand+title
|
||||
query to a title-only query if the combined query is too specific to
|
||||
match anything (small/regional brand name variants are a common case)."""
|
||||
match anything (small/regional brand name variants are a common case).
|
||||
|
||||
Only records whose own `brands` field names our brand are used - see
|
||||
`_openfacts_product_is_brand`. The title-only fallback makes this filter
|
||||
load-bearing: "Honey 1kg" matches every honey on the site.
|
||||
"""
|
||||
if not USE_OPEN_FACTS:
|
||||
return []
|
||||
|
||||
@@ -126,12 +159,12 @@ def find_images_openfacts(title: str, brand: Optional[str] = None, max_results:
|
||||
if not query:
|
||||
return []
|
||||
|
||||
products = _query_openfacts(query, max_results)
|
||||
products = [p for p in _query_openfacts(query, max_results) if _openfacts_product_is_brand(p, brand)]
|
||||
if not products and brand and title:
|
||||
# Combined "brand + title" query found nothing - retry with just
|
||||
# the title, since Open*Facts' free-text search is exact-ish and
|
||||
# brand naming conventions vary (e.g. "Dettol" vs "Reckitt Dettol").
|
||||
products = _query_openfacts(title, max_results)
|
||||
products = [p for p in _query_openfacts(title, max_results) if _openfacts_product_is_brand(p, brand)]
|
||||
|
||||
urls: List[str] = []
|
||||
for product in products:
|
||||
|
||||
@@ -332,16 +332,29 @@ def fetch_brand_catalog_exhaustive(brand: str, max_products: int = 300) -> Dict[
|
||||
return {"brand": brand, "products": products_out}
|
||||
|
||||
|
||||
def fetch_product_details(brand: str, product_title: str) -> Dict[str, Any] | None:
|
||||
def fetch_product_details(brand: str, product_title: str,
|
||||
category: str | None = None,
|
||||
size: str | None = None) -> Dict[str, Any] | None:
|
||||
"""Get details for a single product: description, image_url, pricing fields.
|
||||
Returns a dict with keys: description, image_url, size_variants, price_ranges, price_range, provider_examples.
|
||||
|
||||
`category` and `size` are optional context. A 1.5b model asked about
|
||||
"Naga Maida" alone will happily describe a curry; told it is a 500g pack
|
||||
in Flours & Grains it describes refined wheat flour. Both are facts the
|
||||
caller already holds, so they cost nothing to pass.
|
||||
"""
|
||||
if not _ensure_client():
|
||||
return None
|
||||
context = f"Brand: {brand}\nProduct: {product_title}\n"
|
||||
if category:
|
||||
context += f"Category: {category}\n"
|
||||
if size:
|
||||
context += f"Pack size: {size}\n"
|
||||
user_prompt = (
|
||||
f"Brand: {brand}\nProduct: {product_title}\n"
|
||||
"Return strictly JSON with keys: description, image_url?, size_variants?, price_ranges?, price_range?, provider_examples?.\n"
|
||||
"description must be <=160 chars, concise and factual."
|
||||
context
|
||||
+ "Return strictly JSON with keys: description, image_url?, size_variants?, price_ranges?, price_range?, provider_examples?.\n"
|
||||
"description: 1-2 sentences, <=220 chars, factual - what the product is, "
|
||||
"its form and its typical use. No marketing adjectives, no claims you cannot know."
|
||||
)
|
||||
text = _generate(SYSTEM_PROMPT, user_prompt)
|
||||
if not text:
|
||||
|
||||
@@ -5,7 +5,7 @@ WHY THIS FILE EXISTS
|
||||
--------------------
|
||||
Open Food Facts answers "does this product exist" for food, and answers it
|
||||
well. It does not answer it for anything else: `off_bulk` queries only
|
||||
`search.openfoodfacts.org`, so a toothpaste or a detergent has no
|
||||
Open Food Facts' food database, so a toothpaste or a detergent has no
|
||||
product-discovery source at all and its entire catalogue is language-model
|
||||
output. Measured 2026-09-10, India-tagged coverage on the sibling databases is
|
||||
2-13 products per non-food brand against Britannia's 218 on OFF - real, but
|
||||
|
||||
Reference in New Issue
Block a user