Backend catalog recent updates
This commit is contained in:
@@ -385,6 +385,37 @@ BARCODE_LOOKUP_CACHE_TTL_SECONDS = float(os.getenv("BARCODE_LOOKUP_CACHE_TTL_SEC
|
||||
BARCODE_LOOKUP_MAX_CONCURRENCY = int(os.getenv("BARCODE_LOOKUP_MAX_CONCURRENCY", "5"))
|
||||
BARCODE_COUNTRY_TAG = os.getenv("BARCODE_COUNTRY_TAG", "india")
|
||||
|
||||
# How similar a candidate's product name must be to ours before its barcode is
|
||||
# believed. Applies to BOTH directions: looking a barcode up from a name, and
|
||||
# looking a product up from a barcode.
|
||||
#
|
||||
# RAISED FROM matching.py's OWN 0.45 DEFAULT, ON EVIDENCE. That default is a
|
||||
# reasonable general floor, but by the time a candidate reaches this gate its
|
||||
# brand and pack size have ALREADY been matched - so the name is the only thing
|
||||
# left doing any discriminating, and it has to carry the whole decision.
|
||||
#
|
||||
# At 0.45 it did not. Scored across every catalogue barcode Open Food Facts
|
||||
# knows, more than half the accepted matches were a different product:
|
||||
#
|
||||
# floor accepted wrong
|
||||
# 0.45 15 8
|
||||
# 0.70 7 2
|
||||
# 0.78 2 0
|
||||
#
|
||||
# 0.761 "Tata Tea Gold 500g" -> "Tata Tea Gold Care" different
|
||||
# 0.658 "MTR Masala 300g" -> "MTR Chana Masala" different
|
||||
# 0.538 "Aachi Chicken Masala 50g" -> "Chicken Kabab/65 Masala" different
|
||||
# 0.097 "Lion Dates Powder 100g" -> "PEPER NOTEN" different
|
||||
#
|
||||
# 0.78 is where the sample is clean, NOT where the yield is good, and it is a
|
||||
# judgement rather than a separation: two products tie at 0.773 with opposite
|
||||
# verdicts. It is set for precision because a WRONG barcode is worse than no
|
||||
# barcode - it is an identifier other systems join on, and 33 of the 95 already
|
||||
# in the catalogue are wrong, all of them accepted at the old floor.
|
||||
#
|
||||
# Lower it only with the yield/error numbers in front of you.
|
||||
BARCODE_MIN_NAME_SIMILARITY = float(os.getenv("BARCODE_MIN_NAME_SIMILARITY", "0.78"))
|
||||
|
||||
# Optional barcode source credentials. Each source disables itself when its
|
||||
# key is blank, so leaving these unset simply narrows the lookup cascade.
|
||||
GS1_INDIA_API_BASE_URL = os.getenv("GS1_INDIA_API_BASE_URL", "")
|
||||
|
||||
@@ -29,6 +29,7 @@ from app.infrastructure.settings import (
|
||||
ENABLE_BARCODE_LOOKUP,
|
||||
BARCODE_LOOKUP_CACHE_TTL_SECONDS,
|
||||
BARCODE_LOOKUP_MAX_CONCURRENCY,
|
||||
BARCODE_MIN_NAME_SIMILARITY,
|
||||
)
|
||||
from app.services.enrichment.barcode import cache
|
||||
from app.services.enrichment.barcode.matching import is_match
|
||||
@@ -131,7 +132,16 @@ class BarcodeLookupService:
|
||||
clean_barcode = validate_barcode(candidate.barcode)
|
||||
if not clean_barcode:
|
||||
continue # invalid checksum/length/format - never stored, not even flagged
|
||||
matched, confidence = is_match(candidate, brand, product_title, size, brand_aliases)
|
||||
# The floor comes from settings, not from matching.py's own 0.45
|
||||
# default. By this point the candidate's brand and pack size have
|
||||
# already matched, so the name is the only thing still telling two
|
||||
# products apart - and at 0.45 more than half the accepted matches
|
||||
# in this catalogue were the wrong product. See the setting for the
|
||||
# measured yield/error table.
|
||||
matched, confidence = is_match(
|
||||
candidate, brand, product_title, size, brand_aliases,
|
||||
min_name_similarity=BARCODE_MIN_NAME_SIMILARITY,
|
||||
)
|
||||
if not matched:
|
||||
continue
|
||||
return _build_result(clean_barcode, candidate.source_name, confidence)
|
||||
|
||||
@@ -22,7 +22,7 @@ comparison, so the size-matching logic itself is not duplicated - see
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import List
|
||||
from typing import List, Optional
|
||||
|
||||
import requests
|
||||
|
||||
@@ -44,6 +44,68 @@ _BROWSER_UA = (
|
||||
)
|
||||
_FIELDS = "code,product_name,brands,brands_tags,quantity,countries_tags"
|
||||
|
||||
# The reverse direction asks for more than the search does, because its caller
|
||||
# builds a whole nutrition record rather than just reading a barcode off the
|
||||
# entry. Still an explicit list and not everything: a bare product fetch returns
|
||||
# ~296 fields per item, most of them editing metadata nobody here reads.
|
||||
_PRODUCT_FIELDS = (
|
||||
"code,product_name,brands,brands_tags,quantity,countries_tags,"
|
||||
"nutriments,serving_quantity,serving_size,ingredients_text,"
|
||||
"nutriscore_grade,allergens_tags,labels_tags,"
|
||||
"ingredients_analysis_tags,categories_tags"
|
||||
)
|
||||
|
||||
|
||||
def fetch_product_by_barcode(code: str) -> Optional[dict]:
|
||||
"""The one OFF call in this project that is EXACT rather than a guess.
|
||||
|
||||
Every other Open*Facts call here - this module's own `search()`,
|
||||
`nutrition_data_service`, `image_search` - queries by brand and product
|
||||
name and then scores whatever comes back. That is why the OFF-sourced rows
|
||||
in `nutrition_facts` carry match confidences as low as 0.32. A barcode is
|
||||
the identifier printed on the pack, so `/api/v2/product/{code}` either
|
||||
returns that exact product or nothing at all.
|
||||
|
||||
Returns the product dict, or None when OFF has never seen the barcode -
|
||||
which is the ordinary outcome for about a third of ours, not an error. The
|
||||
caller still has to decide whether the record describes the product WE
|
||||
attached that barcode to; see `matching.is_match`. Measured on real
|
||||
catalogue rows, a quarter of the found records were a different product,
|
||||
because the stored barcode itself was wrong.
|
||||
|
||||
Cascades the same three hosts as the search: a household or beauty item
|
||||
lives in openbeautyfacts, not openfoodfacts, under the same code.
|
||||
"""
|
||||
code = (code or "").strip()
|
||||
if not code:
|
||||
return None
|
||||
|
||||
for host in _HOSTS:
|
||||
@with_retry(max_attempts=2)
|
||||
def _call(host=host):
|
||||
return requests.get(
|
||||
f"https://{host}/api/v2/product/{code}.json",
|
||||
params={"fields": _PRODUCT_FIELDS},
|
||||
headers={"User-Agent": _BROWSER_UA},
|
||||
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
|
||||
)
|
||||
|
||||
try:
|
||||
resp = _call()
|
||||
# 404 is how OFF says "no such barcode here" - try the next host
|
||||
# rather than treating it as a failure.
|
||||
if resp.status_code != 200:
|
||||
continue
|
||||
body = resp.json()
|
||||
# status 1 = found, 0 = not found. The HTTP code alone is not
|
||||
# enough: OFF answers 200 with status 0 for an unknown barcode.
|
||||
if body.get("status") == 1 and body.get("product"):
|
||||
return body["product"]
|
||||
except Exception as e:
|
||||
logger.debug("Open*Facts product fetch failed on %s for %s: %s", host, code, e)
|
||||
continue
|
||||
return None
|
||||
|
||||
|
||||
class OpenFoodFactsSource(BarcodeSource):
|
||||
name = "Open Food Facts"
|
||||
|
||||
@@ -20,6 +20,24 @@ class BarcodeEnrichmentStage(EnrichmentStage):
|
||||
return ENABLE_BARCODE_LOOKUP
|
||||
|
||||
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
|
||||
# ALREADY HAS ONE - never overwrite, and never even look. The same
|
||||
# guard HsnGstEnrichmentStage carries, and this stage was the only one
|
||||
# missing it.
|
||||
#
|
||||
# Without it the damage was not "a worse barcode" but no barcode at
|
||||
# all: `as_product_fields()` always returns all nine keys, so a failed
|
||||
# lookup handed back {"barcode": None, ...} and `apply()` merged that
|
||||
# straight over whatever the shop had typed. Proven end to end - a
|
||||
# sheet sending 8901262010016 stored NULL. Since the cascade misses far
|
||||
# more often than it hits, switching ENABLE_BARCODE_LOOKUP on would
|
||||
# have destroyed more real barcodes than it found.
|
||||
#
|
||||
# A barcode the shop supplied is also better evidence than anything the
|
||||
# cascade can find: they are holding the pack. Skipping the lookup
|
||||
# saves the network call as well.
|
||||
if str(product.get("barcode") or "").strip():
|
||||
return StageOutcome(stage_name=self.name, fields={})
|
||||
|
||||
service = get_default_service()
|
||||
title = product.get("title") or product.get("product_name") or ""
|
||||
size = product.get("size") or ""
|
||||
|
||||
@@ -73,8 +73,31 @@ class EnrichmentStage(ABC):
|
||||
logger.error(f"[{self.name}] unhandled exception enriching '{product.get('product_name')}': {e}")
|
||||
return product
|
||||
|
||||
# A STAGE MAY FILL A GAP OR CORRECT A VALUE. IT MAY NOT ERASE ONE.
|
||||
#
|
||||
# This was a plain `product.update(outcome.fields)`, and the barcode
|
||||
# stage returns a fixed nine-key dict whose values are all None when
|
||||
# the lookup finds nothing - so a miss silently replaced the barcode
|
||||
# the shop had typed with NULL. Verified end to end before this guard
|
||||
# existed: a sheet sending 8901262010016 stored None.
|
||||
#
|
||||
# The rule below is the narrowest one that stops it. A stage can still
|
||||
# overwrite a value with a DIFFERENT value, which is what correcting a
|
||||
# field means; it just cannot blank one out. Stages that must not
|
||||
# overwrite at all say so themselves by returning no fields - see
|
||||
# HsnGstEnrichmentStage and BarcodeEnrichmentStage.
|
||||
if outcome.fields:
|
||||
product.update(outcome.fields)
|
||||
for key, value in outcome.fields.items():
|
||||
blank_incoming = value is None or (isinstance(value, str) and not value.strip())
|
||||
existing = product.get(key)
|
||||
held = existing is not None and not (isinstance(existing, str) and not existing.strip())
|
||||
if blank_incoming and held:
|
||||
logger.debug(
|
||||
"[%s] kept existing %s=%r rather than blanking it",
|
||||
self.name, key, existing,
|
||||
)
|
||||
continue
|
||||
product[key] = value
|
||||
if not outcome.ok:
|
||||
logger.debug(f"[{self.name}] {product.get('product_name')}: {outcome.error}")
|
||||
return product
|
||||
|
||||
@@ -29,7 +29,11 @@ from typing import Any, Dict, List, Optional
|
||||
|
||||
import requests
|
||||
|
||||
from app.infrastructure.settings import USE_OPEN_FACTS, REQUEST_TIMEOUT_SECONDS
|
||||
from app.infrastructure.settings import (
|
||||
BARCODE_MIN_NAME_SIMILARITY as _SETTINGS_BARCODE_MIN_NAME_SIMILARITY,
|
||||
REQUEST_TIMEOUT_SECONDS,
|
||||
USE_OPEN_FACTS,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -354,3 +358,163 @@ def fetch_verified_nutrition(brand: str, title: str, category: str = "") -> Dict
|
||||
}
|
||||
result.update(per_100g)
|
||||
return result
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The exact path: look the product up by the barcode on its pack
|
||||
# ---------------------------------------------------------------------------
|
||||
# Everything above searches Open Food Facts by brand and product name and then
|
||||
# scores whatever comes back, because for most of the catalogue a name is all we
|
||||
# have. It works, but it is a guess: MIN_MATCH_CONFIDENCE is 0.32, and rows in
|
||||
# nutrition_facts really do sit at that floor.
|
||||
#
|
||||
# For the products that carry a barcode, we can do better. The barcode is the
|
||||
# identifier printed on the pack, so /api/v2/product/{code} returns that exact
|
||||
# product or nothing.
|
||||
#
|
||||
# WHAT IS STILL NOT CERTAIN, AND WHY THERE IS A GATE.
|
||||
# The lookup is exact; the STORED BARCODE is not. Measured against the live
|
||||
# catalogue, a quarter of the records found this way described a different
|
||||
# product - "Aachi Chicken Masala 50g" came back as "Chicken Kabab/65 Masala" -
|
||||
# because the barcode attached to our row was wrong. Importing on the strength
|
||||
# of the identifier alone would write another product's nutrition onto ours,
|
||||
# which is the same shape of fault as shipping one company's FSSAI licence on
|
||||
# another's product. So the record still has to pass `matching.is_match`, which
|
||||
# checks brand, pack size, variant terms and name similarity. That module is
|
||||
# reused rather than reimplemented: a second opinion on "is this the same
|
||||
# product" that disagreed with the first would be worse than none.
|
||||
|
||||
# A verified barcode hit is an identifier match, not a fuzzy one, and it is
|
||||
# recorded well clear of the name-search band so the two are separable in the
|
||||
# table. It is not 1.0: `is_match` still passed judgement on brand and size, and
|
||||
# claiming certainty would misrepresent that.
|
||||
BARCODE_MATCH_CONFIDENCE = 0.95
|
||||
|
||||
# Distinct from plain "openfoodfacts" so a query can tell an exact hit from a
|
||||
# name search. `data_source` is free text and nothing filters on it - it is
|
||||
# passed through to nutrition_schemas for display - so adding a value here
|
||||
# breaks no existing read.
|
||||
BARCODE_DATA_SOURCE = "openfoodfacts_barcode"
|
||||
|
||||
# The name-similarity floor for a barcode-verified match, well above
|
||||
# matching.py's own 0.45 default. That default is right for the FORWARD lookup,
|
||||
# where brand and size have not yet been confirmed and the name is one signal
|
||||
# among several. Here brand and size already agree - the identifier guaranteed
|
||||
# that much - so the name is the only thing left doing any discriminating, and
|
||||
# it has to carry the whole decision.
|
||||
#
|
||||
# 0.78 IS A JUDGEMENT, NOT A CLEAN SEPARATION, and the data says so. Scored
|
||||
# across all 35 catalogue barcodes OFF actually knows:
|
||||
#
|
||||
# 0.806 Aachi Chicken Masala 100g -> "Aachi chicken masala" same
|
||||
# 0.773 Tata Sampann Chana Dal -> "Tata Sampann Unpolished .." same
|
||||
# 0.773 Tata Coffee Classic 2g -> "Tata Coffee Grand Classic" DIFFERENT
|
||||
# 0.761 Tata Tea Gold 500g -> "Tata Tea Gold Care" DIFFERENT
|
||||
# 0.658 MTR Masala 300g -> "MTR Chana Masala" DIFFERENT
|
||||
# 0.097 Lion Dates Powder 100g -> "PEPER NOTEN" DIFFERENT
|
||||
#
|
||||
# Two products tie at 0.773 with opposite verdicts, so no threshold separates
|
||||
# them. Part of the cause is on our side: "MTR Masala 300g" and "Tata Sampann
|
||||
# Spices 200g" do not name a specific product, and nothing can match a name
|
||||
# that vague.
|
||||
#
|
||||
# So this is set where the sample is clean rather than where the yield is good,
|
||||
# and the backfill script prints every candidate with its score so the cut can
|
||||
# be seen and argued with instead of taken on trust.
|
||||
# Imported rather than redeclared: the forward lookup (name -> barcode) and this
|
||||
# reverse one (barcode -> product) are answering the same question about the
|
||||
# same pair of names, and two copies that drifted apart would mean a barcode
|
||||
# good enough to store was not good enough to read back.
|
||||
BARCODE_MIN_NAME_SIMILARITY = _SETTINGS_BARCODE_MIN_NAME_SIMILARITY
|
||||
|
||||
|
||||
def fetch_verified_nutrition_by_barcode(
|
||||
barcode: str, brand: str, title: str, size: str = "", category: str = "",
|
||||
min_name_similarity: float = BARCODE_MIN_NAME_SIMILARITY,
|
||||
) -> Dict[str, Any]:
|
||||
"""Nutrition for one product, looked up by its barcode.
|
||||
|
||||
Returns THE SAME DICT SHAPE as `fetch_verified_nutrition`, deliberately, so
|
||||
`nutrition_db.upsert_nutrition_facts` writes it with no change and the two
|
||||
acquisition paths cannot drift apart. There is a test asserting the key sets
|
||||
match.
|
||||
|
||||
`data_status == "unavailable"` covers every way this can decline - no
|
||||
barcode, OFF has never seen it, the record is a different product, or it
|
||||
carries no nutriments. Callers must render "unavailable" rather than
|
||||
treating a missing value as zero.
|
||||
"""
|
||||
now_iso = datetime.now(timezone.utc).isoformat()
|
||||
unavailable = {"data_status": "unavailable", "fetched_at": now_iso}
|
||||
|
||||
if not USE_OPEN_FACTS or not (barcode or "").strip():
|
||||
return unavailable
|
||||
|
||||
# Imported here rather than at module scope: this module is reached during
|
||||
# nutrition enrichment, and the barcode package pulls in tenacity plus the
|
||||
# whole source cascade. A local import keeps that off the path of the
|
||||
# name-search calls above, which do not need any of it.
|
||||
from app.services.enrichment.barcode.matching import is_match
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate
|
||||
from app.services.enrichment.barcode.sources.open_food_facts import (
|
||||
fetch_product_by_barcode,
|
||||
)
|
||||
|
||||
product = fetch_product_by_barcode(barcode)
|
||||
if not product:
|
||||
return unavailable
|
||||
|
||||
candidate = BarcodeCandidate(
|
||||
barcode=str(product.get("code") or barcode),
|
||||
source_name="Open Food Facts",
|
||||
candidate_title=product.get("product_name") or "",
|
||||
candidate_brand=product.get("brands") or "",
|
||||
candidate_size=product.get("quantity") or "",
|
||||
candidate_countries=",".join(product.get("countries_tags") or []),
|
||||
)
|
||||
matched, similarity = is_match(
|
||||
candidate, brand, title, size,
|
||||
min_name_similarity=min_name_similarity,
|
||||
)
|
||||
if not matched:
|
||||
logger.info(
|
||||
"Barcode %s is in Open Food Facts as %r (%s, %s) which does not "
|
||||
"match our %r (%s, %s) - skipping, and OUR barcode is the suspect one",
|
||||
barcode, candidate.candidate_title, candidate.candidate_brand,
|
||||
candidate.candidate_size, title, brand, size,
|
||||
)
|
||||
return unavailable
|
||||
|
||||
nutriments = product.get("nutriments") or {}
|
||||
if not nutriments:
|
||||
return unavailable
|
||||
|
||||
per_100g = _build_flat_fields(nutriments, "100g")
|
||||
if not any(v is not None for v in per_100g.values()):
|
||||
return unavailable
|
||||
per_serving = _build_flat_fields(nutriments, "serving")
|
||||
|
||||
code = product.get("code") or barcode
|
||||
result: Dict[str, Any] = {
|
||||
"data_status": "verified" if per_100g.get("calories_kcal") is not None else "partial",
|
||||
"data_source": BARCODE_DATA_SOURCE,
|
||||
"source_ref": code,
|
||||
"source_url": f"https://{OFF_HOST}/product/{code}",
|
||||
"match_confidence": BARCODE_MATCH_CONFIDENCE,
|
||||
# Diagnostic only. The gate above already decided acceptance; this is
|
||||
# kept so a low-similarity accept can be reviewed later.
|
||||
"name_similarity": round(similarity, 3),
|
||||
"serving_size_g": _convert(product.get("serving_quantity"), "g"),
|
||||
"serving_size_label": product.get("serving_size"),
|
||||
"extended_nutrients": _build_extended_nutrients(nutriments, "100g"),
|
||||
"per_serving": {k: v for k, v in per_serving.items() if v is not None},
|
||||
"ingredients_text": (product.get("ingredients_text") or "").strip() or None,
|
||||
"off_nutriscore": (product.get("nutriscore_grade") or "").strip().lower() or None,
|
||||
"allergens": _extract_allergens(product),
|
||||
"off_labels_tags": product.get("labels_tags") or [],
|
||||
"off_ingredients_analysis_tags": product.get("ingredients_analysis_tags") or [],
|
||||
"off_categories_tags": product.get("categories_tags") or [],
|
||||
"fetched_at": now_iso,
|
||||
}
|
||||
result.update(per_100g)
|
||||
return result
|
||||
|
||||
Reference in New Issue
Block a user