Valid Barcode Generation
This commit is contained in:
423
app/services/enrichment/barcode/sources/off_bulk.py
Normal file
423
app/services/enrichment/barcode/sources/off_bulk.py
Normal file
@@ -0,0 +1,423 @@
|
||||
"""
|
||||
Bulk Open Food Facts brand-corpus fetch + offline name matching.
|
||||
|
||||
WHY THIS EXISTS ALONGSIDE `open_food_facts.py`
|
||||
----------------------------------------------
|
||||
`sources/open_food_facts.py` searches OFF **once per product** against
|
||||
`/cgi/search.pl`. OFF rate-limits that endpoint to 10 requests/minute, which
|
||||
makes a 600-product backfill a multi-hour job, and `stage.py` records that the
|
||||
resulting cascade "misses far more often than it hits". That module is wired
|
||||
into the live enrichment pipeline and is deliberately NOT touched by this file.
|
||||
|
||||
This module inverts the problem: fetch a brand's **entire** OFF catalogue in one
|
||||
or two requests from the Search-a-licious endpoint, cache it on disk, then match
|
||||
every one of our products against that corpus offline. A whole-catalogue
|
||||
backfill costs ~5 HTTP requests instead of ~600, and re-tuning the similarity
|
||||
threshold costs zero network because the corpus is cached.
|
||||
|
||||
Nothing here is imported by the running app - the only consumer is
|
||||
`scripts/backfill_barcodes_from_off.py`. Everything except `fetch_brand_corpus`
|
||||
is a pure function so it can be unit-tested without network or database.
|
||||
|
||||
ENDPOINT NOTES (verified empirically, 2026-09)
|
||||
GET https://search.openfoodfacts.org/search
|
||||
?q=brands:amul AND countries_tags:"en:india"
|
||||
&fields=code,product_name,quantity,...
|
||||
&page_size=250
|
||||
* The double quotes around "en:india" are REQUIRED. Without them the query
|
||||
parses as a bare term and silently returns count=0 rather than erroring.
|
||||
* page_size up to 1000 is accepted; 250 keeps responses small.
|
||||
* `world.openfoodfacts.org` intermittently serves an HTML "Page temporarily
|
||||
unavailable" page with a 200 status, so every response is content-type
|
||||
checked before parsing.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple
|
||||
|
||||
import requests
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
BARCODE_COUNTRY_TAG,
|
||||
BARCODE_LOOKUP_TIMEOUT_SECONDS,
|
||||
)
|
||||
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
|
||||
from app.services.enrichment.barcode.matching import (
|
||||
has_conflicting_variant_terms,
|
||||
name_similarity,
|
||||
)
|
||||
from app.services.enrichment.barcode.models import BarcodeType
|
||||
from app.services.enrichment.barcode.retry import with_retry
|
||||
from app.services.enrichment.barcode.validators import (
|
||||
classify_barcode_type,
|
||||
to_ean13,
|
||||
validate_barcode,
|
||||
)
|
||||
from app.services.quantity_utils import quantities_match
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
SEARCH_URL = "https://search.openfoodfacts.org/search"
|
||||
|
||||
# OFF's usage policy requires a contactable custom User-Agent; requests sent
|
||||
# with the default python-requests agent are treated as anonymous crawling.
|
||||
USER_AGENT = "BrandCatalogRAG/1.0 (suriya@tenext.in)"
|
||||
|
||||
FIELDS = "code,product_name,product_name_en,brands,quantity,countries_tags"
|
||||
PAGE_SIZE = 250
|
||||
MAX_PAGES = 10 # 2500 products per brand is far beyond any real brand
|
||||
PAUSE_SECONDS = 2.0 # search endpoints are the strictly-limited ones
|
||||
|
||||
# backend/app/services/enrichment/barcode/sources/off_bulk.py -> backend/
|
||||
_BACKEND_DIR = Path(__file__).resolve().parents[5]
|
||||
CACHE_DIR = _BACKEND_DIR / "data" / "cache" / "off_brand_corpus"
|
||||
|
||||
# Pack sizes embedded in a product name ("Marie Gold 250g", "Butter 1L").
|
||||
# Mirrors the unit list in scripts/backfill_nutrition_from_barcodes.py, widened
|
||||
# with the count-based units this catalog also uses.
|
||||
_SIZE_RE = re.compile(
|
||||
r"\b\d+(?:[.,]\d+)?\s*"
|
||||
r"(?:kgs|kg|gms|gm|grams|gram|mg|g|mls|ml|litres|litre|ltrs|ltr|l|"
|
||||
r"pcs|pc|pieces|piece|nos|no)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
# "(20)", "[6]" - pack-count noise seen in OFF titles such as
|
||||
# "Brit 50-50 maska chaska 40g (20)".
|
||||
_PAREN_RE = re.compile(r"[\(\[][^\)\]]*[\)\]]")
|
||||
|
||||
_NON_ALNUM_RE = re.compile(r"[^a-z0-9]+")
|
||||
|
||||
# GS1 prefix for barcodes issued in India. Every correct match observed in
|
||||
# testing starts with it; the false ones were Mondelez EU codes (7622...) and a
|
||||
# Japanese Maggi (4987...), i.e. the same product line sold in another market
|
||||
# with a different pack and a different code.
|
||||
INDIA_GS1_PREFIX = "890"
|
||||
|
||||
# Variant words that `matching.VARIANT_DISTINGUISHING_TERMS` does not carry but
|
||||
# which mark a different retail SKU in this catalog. Kept local rather than
|
||||
# added to matching.py, which the live enrichment pipeline shares.
|
||||
EXTRA_VARIANT_TERMS = {
|
||||
"minis", "mini", "plus", "max", "lite", "duo", "multipack", "sugarfree",
|
||||
}
|
||||
|
||||
# A token shared by this fraction of a brand's non-prefixed aliases is treated
|
||||
# as part of the brand's own name rather than a product name. For Hindustan
|
||||
# Unilever every alias is "hul <subbrand>", so "hul" clears the bar and is
|
||||
# stripped, while "lux"/"dove" appear once each and are preserved.
|
||||
_BRAND_TOKEN_SHARE = 0.6
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Fetch
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _cache_path(slug: str, cache_dir: Optional[Path] = None) -> Path:
|
||||
return (cache_dir or CACHE_DIR) / f"{slug}.json"
|
||||
|
||||
|
||||
@with_retry(max_attempts=3, min_wait=2.0, max_wait=10.0)
|
||||
def _get_page(brand: str, country: Optional[str], page: int) -> Dict[str, Any]:
|
||||
"""One page of OFF search results. Raises on transport errors (retried by
|
||||
the decorator); returns an empty result dict for any response that is not
|
||||
parseable JSON, which is how the HTML "temporarily unavailable" page and
|
||||
any future error page are absorbed without killing the run."""
|
||||
query = f"brands:{brand}"
|
||||
if country:
|
||||
# The quotes are load-bearing - see the module docstring.
|
||||
query += f' AND countries_tags:"en:{country}"'
|
||||
|
||||
resp = requests.get(
|
||||
SEARCH_URL,
|
||||
params={"q": query, "fields": FIELDS, "page_size": PAGE_SIZE, "page": page},
|
||||
headers={"User-Agent": USER_AGENT, "Accept": "application/json"},
|
||||
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
|
||||
)
|
||||
|
||||
if resp.status_code != 200:
|
||||
logger.warning("OFF search returned HTTP %s for brand %r page %s",
|
||||
resp.status_code, brand, page)
|
||||
return {}
|
||||
if "json" not in (resp.headers.get("content-type") or "").lower():
|
||||
logger.warning("OFF search returned non-JSON (%s) for brand %r - "
|
||||
"the service is probably serving an error page",
|
||||
resp.headers.get("content-type"), brand)
|
||||
return {}
|
||||
try:
|
||||
payload = resp.json()
|
||||
except ValueError as e:
|
||||
logger.warning("OFF search returned unparseable JSON for brand %r: %s", brand, e)
|
||||
return {}
|
||||
return payload if isinstance(payload, dict) else {}
|
||||
|
||||
|
||||
def fetch_brand_corpus(brand: str,
|
||||
country: Optional[str] = None,
|
||||
refresh: bool = False,
|
||||
cache_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
|
||||
"""Every OFF product for `brand`, from disk cache unless `refresh`.
|
||||
|
||||
Hits with no usable product name are dropped here rather than at match time
|
||||
(6 of 146 Amul hits, 1 of 233 Britannia hits) - a nameless hit can never
|
||||
clear a name-similarity threshold, so carrying it forward only inflates the
|
||||
corpus. Returns [] rather than raising when OFF is unreachable, so one bad
|
||||
brand does not abort a multi-brand backfill.
|
||||
"""
|
||||
country = BARCODE_COUNTRY_TAG if country is None else (country or None)
|
||||
slug = re.sub(r"[^a-z0-9]+", "_", brand.lower()).strip("_")
|
||||
path = _cache_path(slug, cache_dir)
|
||||
|
||||
if not refresh and path.exists():
|
||||
try:
|
||||
cached = json.loads(path.read_text(encoding="utf-8"))
|
||||
hits = cached.get("hits") or []
|
||||
logger.info(" OFF corpus for %r: %d product(s) (cached %s)",
|
||||
brand, len(hits), cached.get("fetched_at_human", "?"))
|
||||
return hits
|
||||
except Exception as e: # noqa: BLE001 - a corrupt cache must not be fatal
|
||||
logger.warning(" Ignoring unreadable OFF cache %s: %s", path.name, e)
|
||||
|
||||
hits: List[Dict[str, Any]] = []
|
||||
page = 1
|
||||
while page <= MAX_PAGES:
|
||||
if page > 1:
|
||||
time.sleep(PAUSE_SECONDS)
|
||||
payload = _get_page(brand, country, page)
|
||||
batch = payload.get("hits") or []
|
||||
hits.extend(h for h in batch if isinstance(h, dict) and (h.get("product_name") or "").strip())
|
||||
page_count = payload.get("page_count") or 0
|
||||
if page >= page_count or not batch:
|
||||
break
|
||||
page += 1
|
||||
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = path.with_suffix(".json.tmp")
|
||||
tmp.write_text(json.dumps({
|
||||
"brand": brand,
|
||||
"country": country,
|
||||
"fetched_at": time.time(),
|
||||
"fetched_at_human": time.strftime("%Y-%m-%d %H:%M:%S"),
|
||||
"hits": hits,
|
||||
}, indent=2, ensure_ascii=False), encoding="utf-8")
|
||||
os.replace(tmp, path)
|
||||
|
||||
logger.info(" OFF corpus for %r: %d usable product(s) fetched", brand, len(hits))
|
||||
return hits
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Normalisation
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def brand_tokens(brand: str) -> set:
|
||||
"""Tokens that name the BRAND rather than the product, and so must be
|
||||
stripped from both sides before names are compared.
|
||||
|
||||
Includes the brand as given, its canonical parent, and any token shared by
|
||||
most of the brand's non-prefixed aliases. That last rule is what recovers
|
||||
"hul": every Hindustan Unilever alias is "hul <subbrand>", so "hul" is
|
||||
brand noise, while "lux" and "dove" each appear once and survive as real
|
||||
product names. Without it our mangled "Hindustan Unilever Hul Lux" could
|
||||
never match OFF's "Lux".
|
||||
"""
|
||||
canonical = resolve_parent_brand(brand)
|
||||
canonical_l = canonical.lower().strip()
|
||||
is_own_parent = canonical_l == brand.lower().strip()
|
||||
|
||||
# Only inherit anything from the canonical parent when the brand IS that
|
||||
# parent. Several brands are filed under an unrelated parent for storage
|
||||
# reasons - Tata's catalogue lives in brand_hindustan_unilever - and
|
||||
# treating that parent's name as brand noise would strip real words out of
|
||||
# Tata product names.
|
||||
tokens = set(_tokenize(brand))
|
||||
if is_own_parent:
|
||||
tokens |= set(_tokenize(canonical))
|
||||
|
||||
non_prefixed = [
|
||||
a for a, parent in BRAND_ALIASES.items()
|
||||
if parent.lower().strip() == canonical_l and not a.startswith(canonical_l)
|
||||
] if is_own_parent else []
|
||||
if non_prefixed:
|
||||
counts: Dict[str, int] = {}
|
||||
for alias in non_prefixed:
|
||||
for tok in set(_tokenize(alias)):
|
||||
counts[tok] = counts.get(tok, 0) + 1
|
||||
threshold = len(non_prefixed) * _BRAND_TOKEN_SHARE
|
||||
tokens |= {tok for tok, n in counts.items() if n >= threshold}
|
||||
|
||||
return {t for t in tokens if t}
|
||||
|
||||
|
||||
def _tokenize(text: Optional[str]) -> List[str]:
|
||||
return [t for t in _NON_ALNUM_RE.sub(" ", (text or "").lower()).split() if t]
|
||||
|
||||
|
||||
def strip_sizes(text: Optional[str]) -> str:
|
||||
"""Remove pack sizes and pack-count parentheticals from a product name."""
|
||||
cleaned = _PAREN_RE.sub(" ", (text or ""))
|
||||
cleaned = _SIZE_RE.sub(" ", cleaned)
|
||||
return re.sub(r"\s+", " ", cleaned).strip()
|
||||
|
||||
|
||||
def normalize_for_match(text: Optional[str], tokens_to_drop: Optional[Iterable[str]] = None) -> str:
|
||||
"""Lowercased, size-free, brand-free comparison key.
|
||||
|
||||
Returns "" when nothing survives, and callers MUST treat that as "no
|
||||
product identity" rather than falling back to the brand-bearing form. The
|
||||
catalog contains rows titled only by their brand and a size - "Amul 90g",
|
||||
"Amul 1kg" - and OFF contains an equally anonymous entry named just "Amul".
|
||||
With a fallback those two normalise to "amul" and match at 1.000, which
|
||||
confidently stamps a real barcode onto six products that have no identity
|
||||
in common beyond the brand. An empty key is the correct answer there.
|
||||
"""
|
||||
base = _tokenize(strip_sizes(text))
|
||||
if not base:
|
||||
return ""
|
||||
drop = {t.lower() for t in (tokens_to_drop or ())}
|
||||
return " ".join(t for t in base if t not in drop)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Barcode acceptance
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def accept_barcode(raw: Optional[str]) -> Optional[Tuple[str, str, Optional[str]]]:
|
||||
"""Enforce the "valid 8 or 13 digit code" rule.
|
||||
|
||||
Returns (code, barcode_type, source_upc) or None. A checksum-valid 12-digit
|
||||
UPC-A is widened to its EAN-13 form (a leading zero contributes nothing to
|
||||
the GTIN checksum, so the check digit is unchanged and still valid), and the
|
||||
original 12-digit form is handed back as `source_upc` so the catalog's `upc`
|
||||
field can record where the code came from. GTIN-14 is a shipping-carton
|
||||
code, not a retail one, and is rejected outright.
|
||||
"""
|
||||
code = validate_barcode(raw)
|
||||
if not code:
|
||||
return None
|
||||
if len(code) == 12:
|
||||
widened = to_ean13(code)
|
||||
if not widened or not validate_barcode(widened):
|
||||
return None
|
||||
return widened, BarcodeType.EAN13.value, code
|
||||
if len(code) in (8, 13):
|
||||
return code, classify_barcode_type(code).value, None
|
||||
return None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Matching
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class Candidate:
|
||||
"""One scored OFF hit for one of our title groups."""
|
||||
|
||||
__slots__ = ("barcode", "barcode_type", "source_upc", "off_name", "off_qty",
|
||||
"score", "size_bonus")
|
||||
|
||||
def __init__(self, barcode: str, barcode_type: str, source_upc: Optional[str],
|
||||
off_name: str, off_qty: Optional[str], score: float, size_bonus: int):
|
||||
self.barcode = barcode
|
||||
self.barcode_type = barcode_type
|
||||
self.source_upc = source_upc
|
||||
self.off_name = off_name
|
||||
self.off_qty = off_qty
|
||||
self.score = score
|
||||
self.size_bonus = size_bonus
|
||||
|
||||
@property
|
||||
def rank(self) -> tuple:
|
||||
"""Sort key, best first. Size agreement outranks a marginally better
|
||||
name score, and the barcode breaks ties so a rerun over an unchanged
|
||||
corpus always picks the same product."""
|
||||
return (-self.size_bonus, -self.score, self.barcode)
|
||||
|
||||
def __repr__(self) -> str: # pragma: no cover - debugging aid
|
||||
return f"<Candidate {self.barcode} {self.off_name!r} score={self.score}>"
|
||||
|
||||
|
||||
def symmetric_similarity(clean_candidate: str, clean_target: str) -> float:
|
||||
"""`name_similarity` in both directions, worst case wins.
|
||||
|
||||
`name_similarity` divides the token overlap by the TARGET's token count
|
||||
only, so an OFF name that is a superset of ours scores near-perfectly. That
|
||||
asymmetry is not theoretical - measured against the live corpus it accepted
|
||||
"Butter milk amul" for our "Amul Butter" at 0.882, and "Dairy Milk Silk
|
||||
Minis" for "Dairy Milk Silk" at 0.933. Scoring both directions and taking
|
||||
the minimum drops those to 0.418 and 0.783 while leaving genuine matches
|
||||
("Marie Gold" / "Marie Gold") at 1.000, because a true match is symmetric
|
||||
by construction.
|
||||
"""
|
||||
return min(
|
||||
name_similarity(clean_candidate, clean_target),
|
||||
name_similarity(clean_target, clean_candidate),
|
||||
)
|
||||
|
||||
|
||||
def has_extra_variant_conflict(candidate_title: str, target_title: str) -> bool:
|
||||
"""The `has_conflicting_variant_terms` rule over `EXTRA_VARIANT_TERMS`."""
|
||||
cand = set(_tokenize(candidate_title))
|
||||
target = set(_tokenize(target_title))
|
||||
return bool((cand & EXTRA_VARIANT_TERMS) - target)
|
||||
|
||||
|
||||
def score_candidates(off_hits: Sequence[Dict[str, Any]],
|
||||
our_title: str,
|
||||
our_sizes: Sequence[str],
|
||||
tokens_to_drop: Iterable[str],
|
||||
review_min: float,
|
||||
require_india_prefix: bool = True) -> List[Candidate]:
|
||||
"""Every acceptable OFF hit for one title group, best first.
|
||||
|
||||
Deliberately does NOT use `matching.is_match()`: its size gate returns False
|
||||
whenever either side's size is blank (matching.py:85-86), and 57 of 146 Amul
|
||||
hits carry `quantity: null`. That gate would reject the exact case this
|
||||
backfill exists to handle - our "Britannia Marie Gold 250g" against OFF's
|
||||
"Britannia Marie Gold". Size is used as a ranking bonus here instead of a
|
||||
veto. `has_conflicting_variant_terms` IS still applied, because a
|
||||
"Sugar Free" or "Family Pack" hit is a genuinely different retail product.
|
||||
"""
|
||||
drop = set(tokens_to_drop)
|
||||
clean_target = normalize_for_match(our_title, drop)
|
||||
if not clean_target:
|
||||
return []
|
||||
|
||||
out: List[Candidate] = []
|
||||
for hit in off_hits:
|
||||
off_name = (hit.get("product_name") or hit.get("product_name_en") or "").strip()
|
||||
if not off_name:
|
||||
continue
|
||||
if has_conflicting_variant_terms(off_name, our_title):
|
||||
continue
|
||||
if has_extra_variant_conflict(off_name, our_title):
|
||||
continue
|
||||
|
||||
accepted = accept_barcode(hit.get("code"))
|
||||
if not accepted:
|
||||
continue
|
||||
code, btype, source_upc = accepted
|
||||
if require_india_prefix and not code.startswith(INDIA_GS1_PREFIX):
|
||||
continue
|
||||
|
||||
clean_cand = normalize_for_match(off_name, drop)
|
||||
if not clean_cand:
|
||||
continue
|
||||
|
||||
score = symmetric_similarity(clean_cand, clean_target)
|
||||
if score < review_min:
|
||||
continue
|
||||
|
||||
off_qty = (hit.get("quantity") or "").strip() or None
|
||||
size_bonus = 1 if off_qty and any(
|
||||
quantities_match(off_qty, s, tolerance=0.03) for s in our_sizes if s
|
||||
) else 0
|
||||
|
||||
out.append(Candidate(code, btype, source_upc, off_name, off_qty, score, size_bonus))
|
||||
|
||||
out.sort(key=lambda c: c.rank)
|
||||
return out
|
||||
Reference in New Issue
Block a user