Valid Barcode Generation

This commit is contained in:
sriram
2026-09-07 15:45:25 +05:30
parent c2d7f0ca1d
commit 75dd3eb3ce
40 changed files with 11893 additions and 557 deletions

View File

@@ -0,0 +1,423 @@
"""
Bulk Open Food Facts brand-corpus fetch + offline name matching.
WHY THIS EXISTS ALONGSIDE `open_food_facts.py`
----------------------------------------------
`sources/open_food_facts.py` searches OFF **once per product** against
`/cgi/search.pl`. OFF rate-limits that endpoint to 10 requests/minute, which
makes a 600-product backfill a multi-hour job, and `stage.py` records that the
resulting cascade "misses far more often than it hits". That module is wired
into the live enrichment pipeline and is deliberately NOT touched by this file.
This module inverts the problem: fetch a brand's **entire** OFF catalogue in one
or two requests from the Search-a-licious endpoint, cache it on disk, then match
every one of our products against that corpus offline. A whole-catalogue
backfill costs ~5 HTTP requests instead of ~600, and re-tuning the similarity
threshold costs zero network because the corpus is cached.
Nothing here is imported by the running app - the only consumer is
`scripts/backfill_barcodes_from_off.py`. Everything except `fetch_brand_corpus`
is a pure function so it can be unit-tested without network or database.
ENDPOINT NOTES (verified empirically, 2026-09)
GET https://search.openfoodfacts.org/search
?q=brands:amul AND countries_tags:"en:india"
&fields=code,product_name,quantity,...
&page_size=250
* The double quotes around "en:india" are REQUIRED. Without them the query
parses as a bare term and silently returns count=0 rather than erroring.
* page_size up to 1000 is accepted; 250 keeps responses small.
* `world.openfoodfacts.org` intermittently serves an HTML "Page temporarily
unavailable" page with a 200 status, so every response is content-type
checked before parsing.
"""
from __future__ import annotations
import json
import logging
import os
import re
import time
from pathlib import Path
from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple
import requests
from app.infrastructure.settings import (
BARCODE_COUNTRY_TAG,
BARCODE_LOOKUP_TIMEOUT_SECONDS,
)
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
from app.services.enrichment.barcode.matching import (
has_conflicting_variant_terms,
name_similarity,
)
from app.services.enrichment.barcode.models import BarcodeType
from app.services.enrichment.barcode.retry import with_retry
from app.services.enrichment.barcode.validators import (
classify_barcode_type,
to_ean13,
validate_barcode,
)
from app.services.quantity_utils import quantities_match
logger = logging.getLogger(__name__)
SEARCH_URL = "https://search.openfoodfacts.org/search"
# OFF's usage policy requires a contactable custom User-Agent; requests sent
# with the default python-requests agent are treated as anonymous crawling.
USER_AGENT = "BrandCatalogRAG/1.0 (suriya@tenext.in)"
FIELDS = "code,product_name,product_name_en,brands,quantity,countries_tags"
PAGE_SIZE = 250
MAX_PAGES = 10 # 2500 products per brand is far beyond any real brand
PAUSE_SECONDS = 2.0 # search endpoints are the strictly-limited ones
# backend/app/services/enrichment/barcode/sources/off_bulk.py -> backend/
_BACKEND_DIR = Path(__file__).resolve().parents[5]
CACHE_DIR = _BACKEND_DIR / "data" / "cache" / "off_brand_corpus"
# Pack sizes embedded in a product name ("Marie Gold 250g", "Butter 1L").
# Mirrors the unit list in scripts/backfill_nutrition_from_barcodes.py, widened
# with the count-based units this catalog also uses.
_SIZE_RE = re.compile(
r"\b\d+(?:[.,]\d+)?\s*"
r"(?:kgs|kg|gms|gm|grams|gram|mg|g|mls|ml|litres|litre|ltrs|ltr|l|"
r"pcs|pc|pieces|piece|nos|no)\b",
re.IGNORECASE,
)
# "(20)", "[6]" - pack-count noise seen in OFF titles such as
# "Brit 50-50 maska chaska 40g (20)".
_PAREN_RE = re.compile(r"[\(\[][^\)\]]*[\)\]]")
_NON_ALNUM_RE = re.compile(r"[^a-z0-9]+")
# GS1 prefix for barcodes issued in India. Every correct match observed in
# testing starts with it; the false ones were Mondelez EU codes (7622...) and a
# Japanese Maggi (4987...), i.e. the same product line sold in another market
# with a different pack and a different code.
INDIA_GS1_PREFIX = "890"
# Variant words that `matching.VARIANT_DISTINGUISHING_TERMS` does not carry but
# which mark a different retail SKU in this catalog. Kept local rather than
# added to matching.py, which the live enrichment pipeline shares.
EXTRA_VARIANT_TERMS = {
"minis", "mini", "plus", "max", "lite", "duo", "multipack", "sugarfree",
}
# A token shared by this fraction of a brand's non-prefixed aliases is treated
# as part of the brand's own name rather than a product name. For Hindustan
# Unilever every alias is "hul <subbrand>", so "hul" clears the bar and is
# stripped, while "lux"/"dove" appear once each and are preserved.
_BRAND_TOKEN_SHARE = 0.6
# ---------------------------------------------------------------------------
# Fetch
# ---------------------------------------------------------------------------
def _cache_path(slug: str, cache_dir: Optional[Path] = None) -> Path:
return (cache_dir or CACHE_DIR) / f"{slug}.json"
@with_retry(max_attempts=3, min_wait=2.0, max_wait=10.0)
def _get_page(brand: str, country: Optional[str], page: int) -> Dict[str, Any]:
"""One page of OFF search results. Raises on transport errors (retried by
the decorator); returns an empty result dict for any response that is not
parseable JSON, which is how the HTML "temporarily unavailable" page and
any future error page are absorbed without killing the run."""
query = f"brands:{brand}"
if country:
# The quotes are load-bearing - see the module docstring.
query += f' AND countries_tags:"en:{country}"'
resp = requests.get(
SEARCH_URL,
params={"q": query, "fields": FIELDS, "page_size": PAGE_SIZE, "page": page},
headers={"User-Agent": USER_AGENT, "Accept": "application/json"},
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
)
if resp.status_code != 200:
logger.warning("OFF search returned HTTP %s for brand %r page %s",
resp.status_code, brand, page)
return {}
if "json" not in (resp.headers.get("content-type") or "").lower():
logger.warning("OFF search returned non-JSON (%s) for brand %r - "
"the service is probably serving an error page",
resp.headers.get("content-type"), brand)
return {}
try:
payload = resp.json()
except ValueError as e:
logger.warning("OFF search returned unparseable JSON for brand %r: %s", brand, e)
return {}
return payload if isinstance(payload, dict) else {}
def fetch_brand_corpus(brand: str,
country: Optional[str] = None,
refresh: bool = False,
cache_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
"""Every OFF product for `brand`, from disk cache unless `refresh`.
Hits with no usable product name are dropped here rather than at match time
(6 of 146 Amul hits, 1 of 233 Britannia hits) - a nameless hit can never
clear a name-similarity threshold, so carrying it forward only inflates the
corpus. Returns [] rather than raising when OFF is unreachable, so one bad
brand does not abort a multi-brand backfill.
"""
country = BARCODE_COUNTRY_TAG if country is None else (country or None)
slug = re.sub(r"[^a-z0-9]+", "_", brand.lower()).strip("_")
path = _cache_path(slug, cache_dir)
if not refresh and path.exists():
try:
cached = json.loads(path.read_text(encoding="utf-8"))
hits = cached.get("hits") or []
logger.info(" OFF corpus for %r: %d product(s) (cached %s)",
brand, len(hits), cached.get("fetched_at_human", "?"))
return hits
except Exception as e: # noqa: BLE001 - a corrupt cache must not be fatal
logger.warning(" Ignoring unreadable OFF cache %s: %s", path.name, e)
hits: List[Dict[str, Any]] = []
page = 1
while page <= MAX_PAGES:
if page > 1:
time.sleep(PAUSE_SECONDS)
payload = _get_page(brand, country, page)
batch = payload.get("hits") or []
hits.extend(h for h in batch if isinstance(h, dict) and (h.get("product_name") or "").strip())
page_count = payload.get("page_count") or 0
if page >= page_count or not batch:
break
page += 1
path.parent.mkdir(parents=True, exist_ok=True)
tmp = path.with_suffix(".json.tmp")
tmp.write_text(json.dumps({
"brand": brand,
"country": country,
"fetched_at": time.time(),
"fetched_at_human": time.strftime("%Y-%m-%d %H:%M:%S"),
"hits": hits,
}, indent=2, ensure_ascii=False), encoding="utf-8")
os.replace(tmp, path)
logger.info(" OFF corpus for %r: %d usable product(s) fetched", brand, len(hits))
return hits
# ---------------------------------------------------------------------------
# Normalisation
# ---------------------------------------------------------------------------
def brand_tokens(brand: str) -> set:
"""Tokens that name the BRAND rather than the product, and so must be
stripped from both sides before names are compared.
Includes the brand as given, its canonical parent, and any token shared by
most of the brand's non-prefixed aliases. That last rule is what recovers
"hul": every Hindustan Unilever alias is "hul <subbrand>", so "hul" is
brand noise, while "lux" and "dove" each appear once and survive as real
product names. Without it our mangled "Hindustan Unilever Hul Lux" could
never match OFF's "Lux".
"""
canonical = resolve_parent_brand(brand)
canonical_l = canonical.lower().strip()
is_own_parent = canonical_l == brand.lower().strip()
# Only inherit anything from the canonical parent when the brand IS that
# parent. Several brands are filed under an unrelated parent for storage
# reasons - Tata's catalogue lives in brand_hindustan_unilever - and
# treating that parent's name as brand noise would strip real words out of
# Tata product names.
tokens = set(_tokenize(brand))
if is_own_parent:
tokens |= set(_tokenize(canonical))
non_prefixed = [
a for a, parent in BRAND_ALIASES.items()
if parent.lower().strip() == canonical_l and not a.startswith(canonical_l)
] if is_own_parent else []
if non_prefixed:
counts: Dict[str, int] = {}
for alias in non_prefixed:
for tok in set(_tokenize(alias)):
counts[tok] = counts.get(tok, 0) + 1
threshold = len(non_prefixed) * _BRAND_TOKEN_SHARE
tokens |= {tok for tok, n in counts.items() if n >= threshold}
return {t for t in tokens if t}
def _tokenize(text: Optional[str]) -> List[str]:
return [t for t in _NON_ALNUM_RE.sub(" ", (text or "").lower()).split() if t]
def strip_sizes(text: Optional[str]) -> str:
"""Remove pack sizes and pack-count parentheticals from a product name."""
cleaned = _PAREN_RE.sub(" ", (text or ""))
cleaned = _SIZE_RE.sub(" ", cleaned)
return re.sub(r"\s+", " ", cleaned).strip()
def normalize_for_match(text: Optional[str], tokens_to_drop: Optional[Iterable[str]] = None) -> str:
"""Lowercased, size-free, brand-free comparison key.
Returns "" when nothing survives, and callers MUST treat that as "no
product identity" rather than falling back to the brand-bearing form. The
catalog contains rows titled only by their brand and a size - "Amul 90g",
"Amul 1kg" - and OFF contains an equally anonymous entry named just "Amul".
With a fallback those two normalise to "amul" and match at 1.000, which
confidently stamps a real barcode onto six products that have no identity
in common beyond the brand. An empty key is the correct answer there.
"""
base = _tokenize(strip_sizes(text))
if not base:
return ""
drop = {t.lower() for t in (tokens_to_drop or ())}
return " ".join(t for t in base if t not in drop)
# ---------------------------------------------------------------------------
# Barcode acceptance
# ---------------------------------------------------------------------------
def accept_barcode(raw: Optional[str]) -> Optional[Tuple[str, str, Optional[str]]]:
"""Enforce the "valid 8 or 13 digit code" rule.
Returns (code, barcode_type, source_upc) or None. A checksum-valid 12-digit
UPC-A is widened to its EAN-13 form (a leading zero contributes nothing to
the GTIN checksum, so the check digit is unchanged and still valid), and the
original 12-digit form is handed back as `source_upc` so the catalog's `upc`
field can record where the code came from. GTIN-14 is a shipping-carton
code, not a retail one, and is rejected outright.
"""
code = validate_barcode(raw)
if not code:
return None
if len(code) == 12:
widened = to_ean13(code)
if not widened or not validate_barcode(widened):
return None
return widened, BarcodeType.EAN13.value, code
if len(code) in (8, 13):
return code, classify_barcode_type(code).value, None
return None
# ---------------------------------------------------------------------------
# Matching
# ---------------------------------------------------------------------------
class Candidate:
"""One scored OFF hit for one of our title groups."""
__slots__ = ("barcode", "barcode_type", "source_upc", "off_name", "off_qty",
"score", "size_bonus")
def __init__(self, barcode: str, barcode_type: str, source_upc: Optional[str],
off_name: str, off_qty: Optional[str], score: float, size_bonus: int):
self.barcode = barcode
self.barcode_type = barcode_type
self.source_upc = source_upc
self.off_name = off_name
self.off_qty = off_qty
self.score = score
self.size_bonus = size_bonus
@property
def rank(self) -> tuple:
"""Sort key, best first. Size agreement outranks a marginally better
name score, and the barcode breaks ties so a rerun over an unchanged
corpus always picks the same product."""
return (-self.size_bonus, -self.score, self.barcode)
def __repr__(self) -> str: # pragma: no cover - debugging aid
return f"<Candidate {self.barcode} {self.off_name!r} score={self.score}>"
def symmetric_similarity(clean_candidate: str, clean_target: str) -> float:
"""`name_similarity` in both directions, worst case wins.
`name_similarity` divides the token overlap by the TARGET's token count
only, so an OFF name that is a superset of ours scores near-perfectly. That
asymmetry is not theoretical - measured against the live corpus it accepted
"Butter milk amul" for our "Amul Butter" at 0.882, and "Dairy Milk Silk
Minis" for "Dairy Milk Silk" at 0.933. Scoring both directions and taking
the minimum drops those to 0.418 and 0.783 while leaving genuine matches
("Marie Gold" / "Marie Gold") at 1.000, because a true match is symmetric
by construction.
"""
return min(
name_similarity(clean_candidate, clean_target),
name_similarity(clean_target, clean_candidate),
)
def has_extra_variant_conflict(candidate_title: str, target_title: str) -> bool:
"""The `has_conflicting_variant_terms` rule over `EXTRA_VARIANT_TERMS`."""
cand = set(_tokenize(candidate_title))
target = set(_tokenize(target_title))
return bool((cand & EXTRA_VARIANT_TERMS) - target)
def score_candidates(off_hits: Sequence[Dict[str, Any]],
our_title: str,
our_sizes: Sequence[str],
tokens_to_drop: Iterable[str],
review_min: float,
require_india_prefix: bool = True) -> List[Candidate]:
"""Every acceptable OFF hit for one title group, best first.
Deliberately does NOT use `matching.is_match()`: its size gate returns False
whenever either side's size is blank (matching.py:85-86), and 57 of 146 Amul
hits carry `quantity: null`. That gate would reject the exact case this
backfill exists to handle - our "Britannia Marie Gold 250g" against OFF's
"Britannia Marie Gold". Size is used as a ranking bonus here instead of a
veto. `has_conflicting_variant_terms` IS still applied, because a
"Sugar Free" or "Family Pack" hit is a genuinely different retail product.
"""
drop = set(tokens_to_drop)
clean_target = normalize_for_match(our_title, drop)
if not clean_target:
return []
out: List[Candidate] = []
for hit in off_hits:
off_name = (hit.get("product_name") or hit.get("product_name_en") or "").strip()
if not off_name:
continue
if has_conflicting_variant_terms(off_name, our_title):
continue
if has_extra_variant_conflict(off_name, our_title):
continue
accepted = accept_barcode(hit.get("code"))
if not accepted:
continue
code, btype, source_upc = accepted
if require_india_prefix and not code.startswith(INDIA_GS1_PREFIX):
continue
clean_cand = normalize_for_match(off_name, drop)
if not clean_cand:
continue
score = symmetric_similarity(clean_cand, clean_target)
if score < review_min:
continue
off_qty = (hit.get("quantity") or "").strip() or None
size_bonus = 1 if off_qty and any(
quantities_match(off_qty, s, tolerance=0.03) for s in our_sizes if s
) else 0
out.append(Candidate(code, btype, source_upc, off_name, off_qty, score, size_bonus))
out.sort(key=lambda c: c.rank)
return out