diff --git a/app/infrastructure/settings.py b/app/infrastructure/settings.py index e80dfc4..d1f5a6b 100644 --- a/app/infrastructure/settings.py +++ b/app/infrastructure/settings.py @@ -385,6 +385,37 @@ BARCODE_LOOKUP_CACHE_TTL_SECONDS = float(os.getenv("BARCODE_LOOKUP_CACHE_TTL_SEC BARCODE_LOOKUP_MAX_CONCURRENCY = int(os.getenv("BARCODE_LOOKUP_MAX_CONCURRENCY", "5")) BARCODE_COUNTRY_TAG = os.getenv("BARCODE_COUNTRY_TAG", "india") +# How similar a candidate's product name must be to ours before its barcode is +# believed. Applies to BOTH directions: looking a barcode up from a name, and +# looking a product up from a barcode. +# +# RAISED FROM matching.py's OWN 0.45 DEFAULT, ON EVIDENCE. That default is a +# reasonable general floor, but by the time a candidate reaches this gate its +# brand and pack size have ALREADY been matched - so the name is the only thing +# left doing any discriminating, and it has to carry the whole decision. +# +# At 0.45 it did not. Scored across every catalogue barcode Open Food Facts +# knows, more than half the accepted matches were a different product: +# +# floor accepted wrong +# 0.45 15 8 +# 0.70 7 2 +# 0.78 2 0 +# +# 0.761 "Tata Tea Gold 500g" -> "Tata Tea Gold Care" different +# 0.658 "MTR Masala 300g" -> "MTR Chana Masala" different +# 0.538 "Aachi Chicken Masala 50g" -> "Chicken Kabab/65 Masala" different +# 0.097 "Lion Dates Powder 100g" -> "PEPER NOTEN" different +# +# 0.78 is where the sample is clean, NOT where the yield is good, and it is a +# judgement rather than a separation: two products tie at 0.773 with opposite +# verdicts. It is set for precision because a WRONG barcode is worse than no +# barcode - it is an identifier other systems join on, and 33 of the 95 already +# in the catalogue are wrong, all of them accepted at the old floor. +# +# Lower it only with the yield/error numbers in front of you. +BARCODE_MIN_NAME_SIMILARITY = float(os.getenv("BARCODE_MIN_NAME_SIMILARITY", "0.78")) + # Optional barcode source credentials. Each source disables itself when its # key is blank, so leaving these unset simply narrows the lookup cascade. GS1_INDIA_API_BASE_URL = os.getenv("GS1_INDIA_API_BASE_URL", "") diff --git a/app/services/enrichment/barcode/service.py b/app/services/enrichment/barcode/service.py index 19a2e7f..b001388 100644 --- a/app/services/enrichment/barcode/service.py +++ b/app/services/enrichment/barcode/service.py @@ -29,6 +29,7 @@ from app.infrastructure.settings import ( ENABLE_BARCODE_LOOKUP, BARCODE_LOOKUP_CACHE_TTL_SECONDS, BARCODE_LOOKUP_MAX_CONCURRENCY, + BARCODE_MIN_NAME_SIMILARITY, ) from app.services.enrichment.barcode import cache from app.services.enrichment.barcode.matching import is_match @@ -131,7 +132,16 @@ class BarcodeLookupService: clean_barcode = validate_barcode(candidate.barcode) if not clean_barcode: continue # invalid checksum/length/format - never stored, not even flagged - matched, confidence = is_match(candidate, brand, product_title, size, brand_aliases) + # The floor comes from settings, not from matching.py's own 0.45 + # default. By this point the candidate's brand and pack size have + # already matched, so the name is the only thing still telling two + # products apart - and at 0.45 more than half the accepted matches + # in this catalogue were the wrong product. See the setting for the + # measured yield/error table. + matched, confidence = is_match( + candidate, brand, product_title, size, brand_aliases, + min_name_similarity=BARCODE_MIN_NAME_SIMILARITY, + ) if not matched: continue return _build_result(clean_barcode, candidate.source_name, confidence) diff --git a/app/services/enrichment/barcode/sources/open_food_facts.py b/app/services/enrichment/barcode/sources/open_food_facts.py index 2821870..4d5c6b3 100644 --- a/app/services/enrichment/barcode/sources/open_food_facts.py +++ b/app/services/enrichment/barcode/sources/open_food_facts.py @@ -22,7 +22,7 @@ comparison, so the size-matching logic itself is not duplicated - see from __future__ import annotations import logging -from typing import List +from typing import List, Optional import requests @@ -44,6 +44,68 @@ _BROWSER_UA = ( ) _FIELDS = "code,product_name,brands,brands_tags,quantity,countries_tags" +# The reverse direction asks for more than the search does, because its caller +# builds a whole nutrition record rather than just reading a barcode off the +# entry. Still an explicit list and not everything: a bare product fetch returns +# ~296 fields per item, most of them editing metadata nobody here reads. +_PRODUCT_FIELDS = ( + "code,product_name,brands,brands_tags,quantity,countries_tags," + "nutriments,serving_quantity,serving_size,ingredients_text," + "nutriscore_grade,allergens_tags,labels_tags," + "ingredients_analysis_tags,categories_tags" +) + + +def fetch_product_by_barcode(code: str) -> Optional[dict]: + """The one OFF call in this project that is EXACT rather than a guess. + + Every other Open*Facts call here - this module's own `search()`, + `nutrition_data_service`, `image_search` - queries by brand and product + name and then scores whatever comes back. That is why the OFF-sourced rows + in `nutrition_facts` carry match confidences as low as 0.32. A barcode is + the identifier printed on the pack, so `/api/v2/product/{code}` either + returns that exact product or nothing at all. + + Returns the product dict, or None when OFF has never seen the barcode - + which is the ordinary outcome for about a third of ours, not an error. The + caller still has to decide whether the record describes the product WE + attached that barcode to; see `matching.is_match`. Measured on real + catalogue rows, a quarter of the found records were a different product, + because the stored barcode itself was wrong. + + Cascades the same three hosts as the search: a household or beauty item + lives in openbeautyfacts, not openfoodfacts, under the same code. + """ + code = (code or "").strip() + if not code: + return None + + for host in _HOSTS: + @with_retry(max_attempts=2) + def _call(host=host): + return requests.get( + f"https://{host}/api/v2/product/{code}.json", + params={"fields": _PRODUCT_FIELDS}, + headers={"User-Agent": _BROWSER_UA}, + timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS, + ) + + try: + resp = _call() + # 404 is how OFF says "no such barcode here" - try the next host + # rather than treating it as a failure. + if resp.status_code != 200: + continue + body = resp.json() + # status 1 = found, 0 = not found. The HTTP code alone is not + # enough: OFF answers 200 with status 0 for an unknown barcode. + if body.get("status") == 1 and body.get("product"): + return body["product"] + except Exception as e: + logger.debug("Open*Facts product fetch failed on %s for %s: %s", host, code, e) + continue + return None + class OpenFoodFactsSource(BarcodeSource): name = "Open Food Facts" diff --git a/app/services/enrichment/barcode/stage.py b/app/services/enrichment/barcode/stage.py index 3de812f..51dc10b 100644 --- a/app/services/enrichment/barcode/stage.py +++ b/app/services/enrichment/barcode/stage.py @@ -20,6 +20,24 @@ class BarcodeEnrichmentStage(EnrichmentStage): return ENABLE_BARCODE_LOOKUP async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome: + # ALREADY HAS ONE - never overwrite, and never even look. The same + # guard HsnGstEnrichmentStage carries, and this stage was the only one + # missing it. + # + # Without it the damage was not "a worse barcode" but no barcode at + # all: `as_product_fields()` always returns all nine keys, so a failed + # lookup handed back {"barcode": None, ...} and `apply()` merged that + # straight over whatever the shop had typed. Proven end to end - a + # sheet sending 8901262010016 stored NULL. Since the cascade misses far + # more often than it hits, switching ENABLE_BARCODE_LOOKUP on would + # have destroyed more real barcodes than it found. + # + # A barcode the shop supplied is also better evidence than anything the + # cascade can find: they are holding the pack. Skipping the lookup + # saves the network call as well. + if str(product.get("barcode") or "").strip(): + return StageOutcome(stage_name=self.name, fields={}) + service = get_default_service() title = product.get("title") or product.get("product_name") or "" size = product.get("size") or "" diff --git a/app/services/enrichment/base.py b/app/services/enrichment/base.py index 3e84f89..e1d7ba7 100644 --- a/app/services/enrichment/base.py +++ b/app/services/enrichment/base.py @@ -73,8 +73,31 @@ class EnrichmentStage(ABC): logger.error(f"[{self.name}] unhandled exception enriching '{product.get('product_name')}': {e}") return product + # A STAGE MAY FILL A GAP OR CORRECT A VALUE. IT MAY NOT ERASE ONE. + # + # This was a plain `product.update(outcome.fields)`, and the barcode + # stage returns a fixed nine-key dict whose values are all None when + # the lookup finds nothing - so a miss silently replaced the barcode + # the shop had typed with NULL. Verified end to end before this guard + # existed: a sheet sending 8901262010016 stored None. + # + # The rule below is the narrowest one that stops it. A stage can still + # overwrite a value with a DIFFERENT value, which is what correcting a + # field means; it just cannot blank one out. Stages that must not + # overwrite at all say so themselves by returning no fields - see + # HsnGstEnrichmentStage and BarcodeEnrichmentStage. if outcome.fields: - product.update(outcome.fields) + for key, value in outcome.fields.items(): + blank_incoming = value is None or (isinstance(value, str) and not value.strip()) + existing = product.get(key) + held = existing is not None and not (isinstance(existing, str) and not existing.strip()) + if blank_incoming and held: + logger.debug( + "[%s] kept existing %s=%r rather than blanking it", + self.name, key, existing, + ) + continue + product[key] = value if not outcome.ok: logger.debug(f"[{self.name}] {product.get('product_name')}: {outcome.error}") return product diff --git a/app/services/nutrition_data_service.py b/app/services/nutrition_data_service.py index 2eb68d3..a24a062 100644 --- a/app/services/nutrition_data_service.py +++ b/app/services/nutrition_data_service.py @@ -29,7 +29,11 @@ from typing import Any, Dict, List, Optional import requests -from app.infrastructure.settings import USE_OPEN_FACTS, REQUEST_TIMEOUT_SECONDS +from app.infrastructure.settings import ( + BARCODE_MIN_NAME_SIMILARITY as _SETTINGS_BARCODE_MIN_NAME_SIMILARITY, + REQUEST_TIMEOUT_SECONDS, + USE_OPEN_FACTS, +) logger = logging.getLogger(__name__) @@ -354,3 +358,163 @@ def fetch_verified_nutrition(brand: str, title: str, category: str = "") -> Dict } result.update(per_100g) return result + + +# --------------------------------------------------------------------------- +# The exact path: look the product up by the barcode on its pack +# --------------------------------------------------------------------------- +# Everything above searches Open Food Facts by brand and product name and then +# scores whatever comes back, because for most of the catalogue a name is all we +# have. It works, but it is a guess: MIN_MATCH_CONFIDENCE is 0.32, and rows in +# nutrition_facts really do sit at that floor. +# +# For the products that carry a barcode, we can do better. The barcode is the +# identifier printed on the pack, so /api/v2/product/{code} returns that exact +# product or nothing. +# +# WHAT IS STILL NOT CERTAIN, AND WHY THERE IS A GATE. +# The lookup is exact; the STORED BARCODE is not. Measured against the live +# catalogue, a quarter of the records found this way described a different +# product - "Aachi Chicken Masala 50g" came back as "Chicken Kabab/65 Masala" - +# because the barcode attached to our row was wrong. Importing on the strength +# of the identifier alone would write another product's nutrition onto ours, +# which is the same shape of fault as shipping one company's FSSAI licence on +# another's product. So the record still has to pass `matching.is_match`, which +# checks brand, pack size, variant terms and name similarity. That module is +# reused rather than reimplemented: a second opinion on "is this the same +# product" that disagreed with the first would be worse than none. + +# A verified barcode hit is an identifier match, not a fuzzy one, and it is +# recorded well clear of the name-search band so the two are separable in the +# table. It is not 1.0: `is_match` still passed judgement on brand and size, and +# claiming certainty would misrepresent that. +BARCODE_MATCH_CONFIDENCE = 0.95 + +# Distinct from plain "openfoodfacts" so a query can tell an exact hit from a +# name search. `data_source` is free text and nothing filters on it - it is +# passed through to nutrition_schemas for display - so adding a value here +# breaks no existing read. +BARCODE_DATA_SOURCE = "openfoodfacts_barcode" + +# The name-similarity floor for a barcode-verified match, well above +# matching.py's own 0.45 default. That default is right for the FORWARD lookup, +# where brand and size have not yet been confirmed and the name is one signal +# among several. Here brand and size already agree - the identifier guaranteed +# that much - so the name is the only thing left doing any discriminating, and +# it has to carry the whole decision. +# +# 0.78 IS A JUDGEMENT, NOT A CLEAN SEPARATION, and the data says so. Scored +# across all 35 catalogue barcodes OFF actually knows: +# +# 0.806 Aachi Chicken Masala 100g -> "Aachi chicken masala" same +# 0.773 Tata Sampann Chana Dal -> "Tata Sampann Unpolished .." same +# 0.773 Tata Coffee Classic 2g -> "Tata Coffee Grand Classic" DIFFERENT +# 0.761 Tata Tea Gold 500g -> "Tata Tea Gold Care" DIFFERENT +# 0.658 MTR Masala 300g -> "MTR Chana Masala" DIFFERENT +# 0.097 Lion Dates Powder 100g -> "PEPER NOTEN" DIFFERENT +# +# Two products tie at 0.773 with opposite verdicts, so no threshold separates +# them. Part of the cause is on our side: "MTR Masala 300g" and "Tata Sampann +# Spices 200g" do not name a specific product, and nothing can match a name +# that vague. +# +# So this is set where the sample is clean rather than where the yield is good, +# and the backfill script prints every candidate with its score so the cut can +# be seen and argued with instead of taken on trust. +# Imported rather than redeclared: the forward lookup (name -> barcode) and this +# reverse one (barcode -> product) are answering the same question about the +# same pair of names, and two copies that drifted apart would mean a barcode +# good enough to store was not good enough to read back. +BARCODE_MIN_NAME_SIMILARITY = _SETTINGS_BARCODE_MIN_NAME_SIMILARITY + + +def fetch_verified_nutrition_by_barcode( + barcode: str, brand: str, title: str, size: str = "", category: str = "", + min_name_similarity: float = BARCODE_MIN_NAME_SIMILARITY, +) -> Dict[str, Any]: + """Nutrition for one product, looked up by its barcode. + + Returns THE SAME DICT SHAPE as `fetch_verified_nutrition`, deliberately, so + `nutrition_db.upsert_nutrition_facts` writes it with no change and the two + acquisition paths cannot drift apart. There is a test asserting the key sets + match. + + `data_status == "unavailable"` covers every way this can decline - no + barcode, OFF has never seen it, the record is a different product, or it + carries no nutriments. Callers must render "unavailable" rather than + treating a missing value as zero. + """ + now_iso = datetime.now(timezone.utc).isoformat() + unavailable = {"data_status": "unavailable", "fetched_at": now_iso} + + if not USE_OPEN_FACTS or not (barcode or "").strip(): + return unavailable + + # Imported here rather than at module scope: this module is reached during + # nutrition enrichment, and the barcode package pulls in tenacity plus the + # whole source cascade. A local import keeps that off the path of the + # name-search calls above, which do not need any of it. + from app.services.enrichment.barcode.matching import is_match + from app.services.enrichment.barcode.models import BarcodeCandidate + from app.services.enrichment.barcode.sources.open_food_facts import ( + fetch_product_by_barcode, + ) + + product = fetch_product_by_barcode(barcode) + if not product: + return unavailable + + candidate = BarcodeCandidate( + barcode=str(product.get("code") or barcode), + source_name="Open Food Facts", + candidate_title=product.get("product_name") or "", + candidate_brand=product.get("brands") or "", + candidate_size=product.get("quantity") or "", + candidate_countries=",".join(product.get("countries_tags") or []), + ) + matched, similarity = is_match( + candidate, brand, title, size, + min_name_similarity=min_name_similarity, + ) + if not matched: + logger.info( + "Barcode %s is in Open Food Facts as %r (%s, %s) which does not " + "match our %r (%s, %s) - skipping, and OUR barcode is the suspect one", + barcode, candidate.candidate_title, candidate.candidate_brand, + candidate.candidate_size, title, brand, size, + ) + return unavailable + + nutriments = product.get("nutriments") or {} + if not nutriments: + return unavailable + + per_100g = _build_flat_fields(nutriments, "100g") + if not any(v is not None for v in per_100g.values()): + return unavailable + per_serving = _build_flat_fields(nutriments, "serving") + + code = product.get("code") or barcode + result: Dict[str, Any] = { + "data_status": "verified" if per_100g.get("calories_kcal") is not None else "partial", + "data_source": BARCODE_DATA_SOURCE, + "source_ref": code, + "source_url": f"https://{OFF_HOST}/product/{code}", + "match_confidence": BARCODE_MATCH_CONFIDENCE, + # Diagnostic only. The gate above already decided acceptance; this is + # kept so a low-similarity accept can be reviewed later. + "name_similarity": round(similarity, 3), + "serving_size_g": _convert(product.get("serving_quantity"), "g"), + "serving_size_label": product.get("serving_size"), + "extended_nutrients": _build_extended_nutrients(nutriments, "100g"), + "per_serving": {k: v for k, v in per_serving.items() if v is not None}, + "ingredients_text": (product.get("ingredients_text") or "").strip() or None, + "off_nutriscore": (product.get("nutriscore_grade") or "").strip().lower() or None, + "allergens": _extract_allergens(product), + "off_labels_tags": product.get("labels_tags") or [], + "off_ingredients_analysis_tags": product.get("ingredients_analysis_tags") or [], + "off_categories_tags": product.get("categories_tags") or [], + "fetched_at": now_iso, + } + result.update(per_100g) + return result diff --git a/data/seed_catalogs/brand_catalog_own_products.json b/data/seed_catalogs/brand_catalog_own_products.json index 68402b1..ae71ce4 100644 --- a/data/seed_catalogs/brand_catalog_own_products.json +++ b/data/seed_catalogs/brand_catalog_own_products.json @@ -37988,8 +37988,13 @@ "size": "Standard", "size_variants": [], "description": "Lettuce. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://images.openfoodfacts.org/images/products/000/000/034/7358/front_en.20.400.jpg", + "image_urls": [ + "https://images.openfoodfacts.org/images/products/000/000/034/7358/front_en.20.400.jpg", + "https://images.openfoodfacts.org/images/products/003/022/303/9365/front_en.112.400.jpg", + "https://images.openfoodfacts.org/images/products/325/622/380/0287/front_fr.23.400.jpg", + "https://thumb.wikimedia.org/wikipedia/commons/thumb/4/4c/Lactuca_sativa_02.JPG/960px-Lactuca_sativa_02.JPG?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -38393,7 +38398,10 @@ 0.0043616690672934055, 0.09239858388900757, 0.005233542528003454 - ] + ], + "primary_image": "https://images.openfoodfacts.org/images/products/000/000/034/7358/front_en.20.400.jpg", + "total_images": 4, + "image_source": "seed_build" }, { "product_name": "Shallot", @@ -38402,8 +38410,13 @@ "size": "Standard", "size_variants": [], "description": "Shallot. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://www.theharvestkitchen.com/wp-content/uploads/2023/08/shallot-substitute.jpg", + "image_urls": [ + "https://www.theharvestkitchen.com/wp-content/uploads/2023/08/shallot-substitute.jpg", + "https://c8.alamy.com/comp/BR9HBH/single-shallot-BR9HBH.jpg", + "https://bakeitwithlove.com/wp-content/uploads/2022/01/shallot-substitution-pin.jpg", + "https://www.highmowingseeds.com/media/catalog/product/cache/6cbdb003cf4aae33b9be8e6a6cf3d7ad/2/7/2711-1.jpg" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -38807,7 +38820,10 @@ -0.016630925238132477, 0.021545154973864555, 0.0025799174327403307 - ] + ], + "primary_image": "https://www.theharvestkitchen.com/wp-content/uploads/2023/08/shallot-substitute.jpg", + "total_images": 4, + "image_source": "seed_build" }, { "product_name": "Ivy Gourd", @@ -38816,8 +38832,13 @@ "size": "Standard", "size_variants": [], "description": "Ivy Gourd. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://images.openfoodfacts.org/images/products/001/143/311/7388/front_en.3.400.jpg", + "image_urls": [ + "https://images.openfoodfacts.org/images/products/001/143/311/7388/front_en.3.400.jpg", + "https://thumb.wikimedia.org/wikipedia/commons/thumb/e/e2/Coccinia_grandis_%28Cucurbitaceae%29.jpg/960px-Coccinia_grandis_%28Cucurbitaceae%29.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail", + "https://thumb.wikimedia.org/wikipedia/commons/thumb/0/00/Coccinia_grandis_Ivy_gourd_Cephalandra_indica_%E0%B4%95%E0%B5%8B%E0%B4%B5%E0%B5%BD.JPG/960px-Coccinia_grandis_Ivy_gourd_Cephalandra_indica_%E0%B4%95%E0%B5%8B%E0%B4%B5%E0%B5%BD.JPG?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail", + "https://thumb.wikimedia.org/wikipedia/commons/thumb/c/cf/Coccinia_grandis_sapling.jpg/960px-Coccinia_grandis_sapling.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -39221,7 +39242,10 @@ -0.032460130751132965, 0.04225873202085495, 0.06438104808330536 - ] + ], + "primary_image": "https://images.openfoodfacts.org/images/products/001/143/311/7388/front_en.3.400.jpg", + "total_images": 4, + "image_source": "seed_build" }, { "product_name": "Cluster Bean", @@ -39230,8 +39254,11 @@ "size": "Standard", "size_variants": [], "description": "Cluster Bean. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://images1.livehindustan.com/smart/img/2024/06/17/1200x900/cluster_1718594899887_1718594917521.jpg", + "image_urls": [ + "https://images1.livehindustan.com/smart/img/2024/06/17/1200x900/cluster_1718594899887_1718594917521.jpg", + "https://www.jagranimages.com/images/newimg/21082023/21_08_2023-cluster_beans_benefits_23507732.webp" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -39635,7 +39662,10 @@ -0.011852066032588482, -0.035320233553647995, -0.03111950494349003 - ] + ], + "primary_image": "https://images1.livehindustan.com/smart/img/2024/06/17/1200x900/cluster_1718594899887_1718594917521.jpg", + "total_images": 2, + "image_source": "seed_build" }, { "product_name": "Green Papaya", @@ -39644,8 +39674,12 @@ "size": "Standard", "size_variants": [], "description": "Green Papaya. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://images.openbeautyfacts.org/images/products/590/101/801/1369/front_en.16.400.jpg", + "image_urls": [ + "https://images.openbeautyfacts.org/images/products/590/101/801/1369/front_en.16.400.jpg", + "https://blogger.googleusercontent.com/img/b/R29vZ2xl/AVvXsEiSuByiu32YHewp2BkmQ05qW6KFPfis2NVCcONrto9UpCWBSx8CGL-Emx0bH-lS78viCkpqC2ZKZa2zk0cktrEDOLWo7JhSdx_Y8apQ1xtAF8qZw9fc_ff7YVl4q6dukSPRLTui5meUDFg/w1200-h630-p-k-no-nu/SANY0811.JPG", + "https://images.news18.com/static-bengali/uploads/2023/11/New-Project-4-2023-11-52f8cfd5664c4cc035e2bf8f43da16f5-3x2.jpg?im=FitAndFill=(1200,675)" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -40049,7 +40083,10 @@ -0.010975969024002552, 0.07554828375577927, 0.04782396927475929 - ] + ], + "primary_image": "https://images.openbeautyfacts.org/images/products/590/101/801/1369/front_en.16.400.jpg", + "total_images": 3, + "image_source": "seed_build" }, { "product_name": "Spinach", @@ -40058,8 +40095,11 @@ "size": "Standard", "size_variants": [], "description": "Spinach. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://cdn.grofers.com/cdn-cgi/image/f=auto,fit=scale-down,q=85,metadata=none,w=480,h=480/da/cms-assets/cms/product/7bde44e0-2c19-4f74-9ec9-80bed80baae2.jpg", + "image_urls": [ + "https://cdn.grofers.com/cdn-cgi/image/f=auto,fit=scale-down,q=85,metadata=none,w=480,h=480/da/cms-assets/cms/product/7bde44e0-2c19-4f74-9ec9-80bed80baae2.jpg", + "https://cdn.grofers.com/cdn-cgi/image/f=auto,fit=scale-down,q=85,metadata=none,w=480,h=480/da/cms-assets/cms/product/038f79074c7d436ea884d9ccc4d41c9f.jpeg" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -40463,7 +40503,10 @@ 0.001314538181759417, 0.03696413338184357, -0.018342463299632072 - ] + ], + "primary_image": "https://cdn.grofers.com/cdn-cgi/image/f=auto,fit=scale-down,q=85,metadata=none,w=480,h=480/da/cms-assets/cms/product/7bde44e0-2c19-4f74-9ec9-80bed80baae2.jpg", + "total_images": 2, + "image_source": "seed_build" }, { "product_name": "Palak", @@ -40472,8 +40515,13 @@ "size": "Standard", "size_variants": [], "description": "Palak. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://thumb.wikimedia.org/wikipedia/commons/thumb/0/0f/Zunaid_Ahmed_Palak.jpg/960px-Zunaid_Ahmed_Palak.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail", + "image_urls": [ + "https://thumb.wikimedia.org/wikipedia/commons/thumb/0/0f/Zunaid_Ahmed_Palak.jpg/960px-Zunaid_Ahmed_Palak.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail", + "https://thumb.wikimedia.org/wikipedia/commons/thumb/8/89/Zunaid_Ahmed_Palak_at_10th_Anniversary_of_Bengali_Wikipedia%2C_30_May_2015_01.jpg/960px-Zunaid_Ahmed_Palak_at_10th_Anniversary_of_Bengali_Wikipedia%2C_30_May_2015_01.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail", + "https://thumb.wikimedia.org/wikipedia/commons/thumb/2/2a/Zunaid_Ahmed_Palak_at_10th_Anniversary_of_Bengali_Wikipedia%2C_30_May_2015_02.jpg/960px-Zunaid_Ahmed_Palak_at_10th_Anniversary_of_Bengali_Wikipedia%2C_30_May_2015_02.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail", + "https://thumb.wikimedia.org/wikipedia/commons/thumb/1/1c/Zunaid_Ahmed_Palak_at_BN_wiki_10th_Anniversary_Conference_30_May_2015_47.jpg/960px-Zunaid_Ahmed_Palak_at_BN_wiki_10th_Anniversary_Conference_30_May_2015_47.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -40877,7 +40925,10 @@ 0.026351049542427063, 0.023857485502958298, -0.011563683860003948 - ] + ], + "primary_image": "https://thumb.wikimedia.org/wikipedia/commons/thumb/0/0f/Zunaid_Ahmed_Palak.jpg/960px-Zunaid_Ahmed_Palak.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail", + "total_images": 4, + "image_source": "seed_build" }, { "product_name": "Amaranth Leaves", @@ -40886,8 +40937,10 @@ "size": "Standard", "size_variants": [], "description": "Amaranth Leaves. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://live.staticflickr.com/3476/3823578948_4eef6b0a61_b.jpg", + "image_urls": [ + "https://live.staticflickr.com/3476/3823578948_4eef6b0a61_b.jpg" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -41291,7 +41344,10 @@ -0.009316159412264824, 0.10120223462581635, 0.009406316094100475 - ] + ], + "primary_image": "https://live.staticflickr.com/3476/3823578948_4eef6b0a61_b.jpg", + "total_images": 1, + "image_source": "seed_build" }, { "product_name": "Curry Leaves", @@ -41300,8 +41356,13 @@ "size": "Standard", "size_variants": [], "description": "Curry Leaves. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://images.openfoodfacts.org/images/products/506/308/916/3740/front_en.3.400.jpg", + "image_urls": [ + "https://images.openfoodfacts.org/images/products/506/308/916/3740/front_en.3.400.jpg", + "https://images.openfoodfacts.org/images/products/502/188/500/8535/front_fr.3.400.jpg", + "https://images.openfoodfacts.org/images/products/888/810/137/0701/front_fr.7.400.jpg", + "https://images.openfoodfacts.org/images/products/501/768/927/3972/front_fr.3.400.jpg" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -41705,7 +41766,10 @@ 0.04047529771924019, 0.06272368878126144, -0.010273406282067299 - ] + ], + "primary_image": "https://images.openfoodfacts.org/images/products/506/308/916/3740/front_en.3.400.jpg", + "total_images": 4, + "image_source": "seed_build" }, { "product_name": "Coriander Leaves", @@ -41714,8 +41778,13 @@ "size": "Standard", "size_variants": [], "description": "Coriander Leaves. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://facts.net/wp-content/uploads/2023/07/20-facts-about-coriander-leaves-1689872594.jpg", + "image_urls": [ + "https://facts.net/wp-content/uploads/2023/07/20-facts-about-coriander-leaves-1689872594.jpg", + "https://www.realsimple.com/thmb/jfNmT79_U3a41-gXwNCvZfdsWso=/1500x0/filters:no_upscale():max_bytes(150000):strip_icc()/coriander-vs-cilantro-Comparison-eb65b0bb42d3423b96f58bfd2f6a3e75.jpg", + "https://images.news18.com/telugu/uploads/2023/09/corriander-169529955316x9.jpg", + "https://snapcalorie-webflow-website.s3.us-east-2.amazonaws.com/media/food_pics_v2/coriander_leaves.jpg" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -42119,7 +42188,10 @@ 0.08500473946332932, 0.03993428871035576, 0.05999131500720978 - ] + ], + "primary_image": "https://facts.net/wp-content/uploads/2023/07/20-facts-about-coriander-leaves-1689872594.jpg", + "total_images": 4, + "image_source": "seed_build" }, { "product_name": "Mint Leaves", @@ -42128,8 +42200,13 @@ "size": "Standard", "size_variants": [], "description": "Mint Leaves. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://images.openfoodfacts.org/images/products/762/753/647/5268/front_en.19.400.jpg", + "image_urls": [ + "https://images.openfoodfacts.org/images/products/762/753/647/5268/front_en.19.400.jpg", + "https://images.openfoodfacts.org/images/products/590/095/600/2309/front_pl.4.400.jpg", + "https://images.openfoodfacts.org/images/products/007/067/000/0259/front_fr.4.400.jpg", + "https://thumb.wikimedia.org/wikipedia/commons/thumb/f/f6/A_some_of_mint_leaves_in_the_jar.jpg/960px-A_some_of_mint_leaves_in_the_jar.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -42533,7 +42610,10 @@ 0.08544730395078659, 0.05343063175678253, 0.025364071130752563 - ] + ], + "primary_image": "https://images.openfoodfacts.org/images/products/762/753/647/5268/front_en.19.400.jpg", + "total_images": 4, + "image_source": "seed_build" }, { "product_name": "Methi Leaves", @@ -42542,8 +42622,13 @@ "size": "Standard", "size_variants": [], "description": "Methi Leaves. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://images.tv9telugu.com/wp-content/uploads/2022/10/methi-leaves.jpg?w=1280", + "image_urls": [ + "https://images.tv9telugu.com/wp-content/uploads/2022/10/methi-leaves.jpg?w=1280", + "https://images.news18.com/news18marathi/uploads/2024/12/Benefits-Fenugreek-Leaves-1-2024-12-4c4aa1fc032c951b9807837841a521cb.jpg", + "https://hindi.cdn.zeenews.com/hindi/sites/default/files/2022/11/18/1431394-methi-1-1.jpg?im=FitAndFill=(1200,900)", + "https://hindi.cdn.zeenews.com/hindi/sites/default/files/2022/11/04/1404765-methi.jpg?im=FitAndFill=(1200,900)" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -42947,7 +43032,10 @@ 0.02809380553662777, 0.0512373223900795, 0.05737173929810524 - ] + ], + "primary_image": "https://images.tv9telugu.com/wp-content/uploads/2022/10/methi-leaves.jpg?w=1280", + "total_images": 4, + "image_source": "seed_build" }, { "product_name": "Drumstick Leaves", @@ -42956,8 +43044,13 @@ "size": "Standard", "size_variants": [], "description": "Drumstick Leaves. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://feeds.abplive.com/onecms/images/uploaded-images/2022/09/17/949ef10210ba6a6697fd83f7726fb4ade901d.jpg?impolicy=abp_cdn&imwidth=720", + "image_urls": [ + "https://feeds.abplive.com/onecms/images/uploaded-images/2022/09/17/949ef10210ba6a6697fd83f7726fb4ade901d.jpg?impolicy=abp_cdn&imwidth=720", + "https://kannada.cdn.zeenews.com/kannada/sites/default/files/styles/zm_700x400/public/2024/01/26/372809-drumstick-leaves.jpg?itok=WfT0tfkj", + "https://ohmyfacts.com/wp-content/uploads/2024/07/20-facts-about-drumstick-tree-leaves-1720565040.jpg", + "https://indiadailylive.com/wp-content/uploads/2023/10/drumstick-leaves.jpg" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -43361,7 +43454,10 @@ 0.05498332157731056, 0.09988196194171906, 0.04283886402845383 - ] + ], + "primary_image": "https://feeds.abplive.com/onecms/images/uploaded-images/2022/09/17/949ef10210ba6a6697fd83f7726fb4ade901d.jpg?impolicy=abp_cdn&imwidth=720", + "total_images": 4, + "image_source": "seed_build" }, { "product_name": "Basil Leaves", @@ -43370,8 +43466,13 @@ "size": "Standard", "size_variants": [], "description": "Basil Leaves. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://upload.wikimedia.org/wikipedia/commons/1/19/BANANA_CREPE_CON_BASIL_LEAVES.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail_unscaled", + "image_urls": [ + "https://upload.wikimedia.org/wikipedia/commons/1/19/BANANA_CREPE_CON_BASIL_LEAVES.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail_unscaled", + "https://thumb.wikimedia.org/wikipedia/commons/thumb/b/bd/Basil_Leaves_on_the_top.jpg/960px-Basil_Leaves_on_the_top.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail", + "https://thumb.wikimedia.org/wikipedia/commons/thumb/c/cf/Basil_leaves_on_plate.JPG/960px-Basil_leaves_on_plate.JPG?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail", + "https://thumb.wikimedia.org/wikipedia/commons/thumb/c/c1/L_Basil.jpg/960px-L_Basil.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -43775,7 +43876,10 @@ 0.08480377495288849, 0.052451226860284805, 0.0035562878474593163 - ] + ], + "primary_image": "https://upload.wikimedia.org/wikipedia/commons/1/19/BANANA_CREPE_CON_BASIL_LEAVES.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail_unscaled", + "total_images": 4, + "image_source": "seed_build" }, { "product_name": "Thulasi", @@ -43784,8 +43888,13 @@ "size": "Standard", "size_variants": [], "description": "Thulasi. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://upload.wikimedia.org/wikipedia/commons/8/8f/P.C_Thulasi_%28cropped%29.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail_unscaled", + "image_urls": [ + "https://upload.wikimedia.org/wikipedia/commons/8/8f/P.C_Thulasi_%28cropped%29.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail_unscaled", + "https://thumb.wikimedia.org/wikipedia/commons/thumb/1/16/P.C_Thulasi_and_Reddy_Sikki_from_India_won_the_gold_medal%2C_Bruce_Mary_Alexandra%2C_Li_Man_Shan_Michelle_from_Canada_won_the_Silver_and_Lim_Ee_Von_NG_Hui_Em_from_Malaysia_won_the_bronze_medal_in_Women%E2%80%99s_Doubles_badminton_event.jpg/960px-thumbnail.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail", + "https://thumb.wikimedia.org/wikipedia/commons/thumb/9/9a/Thulasi_thara.JPG/960px-Thulasi_thara.JPG?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail", + "https://upload.wikimedia.org/wikipedia/commons/5/5a/Thulasi%E0%B4%9A%E0%B5%86%E0%B4%9F%E0%B4%BF.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail_unscaled" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -44189,7 +44298,10 @@ 0.02068915218114853, 0.10079442709684372, 0.0011494344798848033 - ] + ], + "primary_image": "https://upload.wikimedia.org/wikipedia/commons/8/8f/P.C_Thulasi_%28cropped%29.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail_unscaled", + "total_images": 4, + "image_source": "seed_build" }, { "product_name": "Parsley", @@ -44198,8 +44310,13 @@ "size": "Standard", "size_variants": [], "description": "Parsley. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://images.openfoodfacts.org/images/products/503/682/975/6342/front_en.34.400.jpg", + "image_urls": [ + "https://images.openfoodfacts.org/images/products/503/682/975/6342/front_en.34.400.jpg", + "https://images.openfoodfacts.org/images/products/000/002/022/6558/front_en.55.400.jpg", + "https://images.openfoodfacts.org/images/products/000/002/041/2661/front_fr.53.400.jpg", + "https://images.openfoodfacts.org/images/products/503/741/200/2464/front_en.3.400.jpg" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -44603,7 +44720,10 @@ 0.04025179520249367, 0.09539943933486938, 0.0324990414083004 - ] + ], + "primary_image": "https://images.openfoodfacts.org/images/products/503/682/975/6342/front_en.34.400.jpg", + "total_images": 4, + "image_source": "seed_build" }, { "product_name": "Dill Leaves", @@ -44612,8 +44732,13 @@ "size": "Standard", "size_variants": [], "description": "Dill Leaves. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://images.openfoodfacts.org/images/products/405/291/600/9109/front_en.3.400.jpg", + "image_urls": [ + "https://images.openfoodfacts.org/images/products/405/291/600/9109/front_en.3.400.jpg", + "https://images.openfoodfacts.org/images/products/008/105/701/3502/front_en.3.400.jpg", + "https://sowrightseeds.com/cdn/shop/products/PacketFront1_20e37d8c-3518-4b2a-8c4e-91e34f64601c_1800x1800.jpg?v=1675215709", + "https://www.happiesthealth.com/wp-content/uploads/2023/07/dill-f.jpg" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -45017,7 +45142,10 @@ 0.0548781119287014, 0.09910596907138824, 0.027936136350035667 - ] + ], + "primary_image": "https://images.openfoodfacts.org/images/products/405/291/600/9109/front_en.3.400.jpg", + "total_images": 4, + "image_source": "seed_build" }, { "product_name": "Sorrel Leaves", @@ -45026,8 +45154,12 @@ "size": "Standard", "size_variants": [], "description": "Sorrel Leaves. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://images.squarespace-cdn.com/content/v1/62e7a92f066fa3730dcd4604/608a21d2-7168-4227-b2c3-41d2719b1ef9/v2-dhjhz-wmuro.jpg", + "image_urls": [ + "https://images.squarespace-cdn.com/content/v1/62e7a92f066fa3730dcd4604/608a21d2-7168-4227-b2c3-41d2719b1ef9/v2-dhjhz-wmuro.jpg", + "https://cdn.vegetariantimes.com/wp-content/uploads/2017/01/sorrel.jpg?width=3840&auto=webp&quality=75&fit=cover", + "https://i5.walmartimages.com/seo/200-Seeds-LARGE-LEAF-SORREL-Garden-Sorrel-Spinach-Dock-Redshank-Rumex-Acetosa-Vegetable-Seeds_eae41db6-db0b-479f-b718-cd7e9c3d2d66.7324d6754d34ee98a44b28e1ceff93fc.jpeg" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -45431,7 +45563,10 @@ 0.012254527769982815, 0.05136796832084656, 0.021458283066749573 - ] + ], + "primary_image": "https://images.squarespace-cdn.com/content/v1/62e7a92f066fa3730dcd4604/608a21d2-7168-4227-b2c3-41d2719b1ef9/v2-dhjhz-wmuro.jpg", + "total_images": 3, + "image_source": "seed_build" }, { "product_name": "Agathi Keerai", @@ -45440,8 +45575,12 @@ "size": "Standard", "size_variants": [], "description": "Agathi Keerai. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://blogger.googleusercontent.com/img/b/R29vZ2xl/AVvXsEiOygN8kfZ3vQZ9l3Q76f9EpHQeZ2ibgWe_0tQNsXXpNXcX-AiD_OAWrTl1FQq-EnhfjkJDOd9lIZ9R0kNNpWtPohdtZZA1L6NYwnv-48nqkm3jsjfvXNlhTAa_ZWGD1ZtadorTSjH1GhYlLod7fnDHWCDYd3W8AlAjHfHUyGV6XXjSLTyweDWKmUfw534/s1452/Agathi4.jpg", + "image_urls": [ + "https://blogger.googleusercontent.com/img/b/R29vZ2xl/AVvXsEiOygN8kfZ3vQZ9l3Q76f9EpHQeZ2ibgWe_0tQNsXXpNXcX-AiD_OAWrTl1FQq-EnhfjkJDOd9lIZ9R0kNNpWtPohdtZZA1L6NYwnv-48nqkm3jsjfvXNlhTAa_ZWGD1ZtadorTSjH1GhYlLod7fnDHWCDYd3W8AlAjHfHUyGV6XXjSLTyweDWKmUfw534/s1452/Agathi4.jpg", + "https://blogger.googleusercontent.com/img/b/R29vZ2xl/AVvXsEiP4sGlKzDU-6ypvQO3ylESSmUf2RFINoaLv6C3DYOJ55fQuL2pAJpMi1Bq82xTX5QJciK8CE0A3hong9wr5_0adqV6rfOCetR83ziBDO3bjB4V9t7zzBel1AYOZrR-iTjRxfi3N4-V0iWV/s1600/Agathikeerai+poriyal.JPG", + "https://m.media-amazon.com/images/I/41MAfV24BXL.jpg" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -45845,7 +45984,10 @@ -0.033807482570409775, 0.0017142812721431255, 0.04498672857880592 - ] + ], + "primary_image": "https://blogger.googleusercontent.com/img/b/R29vZ2xl/AVvXsEiOygN8kfZ3vQZ9l3Q76f9EpHQeZ2ibgWe_0tQNsXXpNXcX-AiD_OAWrTl1FQq-EnhfjkJDOd9lIZ9R0kNNpWtPohdtZZA1L6NYwnv-48nqkm3jsjfvXNlhTAa_ZWGD1ZtadorTSjH1GhYlLod7fnDHWCDYd3W8AlAjHfHUyGV6XXjSLTyweDWKmUfw534/s1452/Agathi4.jpg", + "total_images": 3, + "image_source": "seed_build" }, { "product_name": "Ponnanganni Keerai", @@ -45854,8 +45996,12 @@ "size": "Standard", "size_variants": [], "description": "Ponnanganni Keerai. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://blogger.googleusercontent.com/img/b/R29vZ2xl/AVvXsEhNv5jXFC1JGhdDluKblp8CLBsij7qfCDhXEOZgff7kMBcZarnmcZZnidcK91Mr1SxFL-m9e68Ca-_hDlR1RJhasNMwsBwXK3e3QUppOS2d9SYXnXJxl0y8jeqvZ0COFVGozjvJntqDqUg/s1600/20170913_131941-01.jpeg", + "image_urls": [ + "https://blogger.googleusercontent.com/img/b/R29vZ2xl/AVvXsEhNv5jXFC1JGhdDluKblp8CLBsij7qfCDhXEOZgff7kMBcZarnmcZZnidcK91Mr1SxFL-m9e68Ca-_hDlR1RJhasNMwsBwXK3e3QUppOS2d9SYXnXJxl0y8jeqvZ0COFVGozjvJntqDqUg/s1600/20170913_131941-01.jpeg", + "https://2.bp.blogspot.com/-pJf9XHgiEbk/WVx9xFMKCiI/AAAAAAAAJR8/JZolN-8cPU05Fd7I7h2_vPRkUqVTyWR6wCLcBGAs/s1600/Ponnanganni+greens+chutney.jpg", + "https://www.kamalascorner.com/wp-content/uploads/2008/03/Ponnanganni-Keerai-Pink.jpg" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -46259,7 +46405,10 @@ -0.02059229090809822, 0.040050771087408066, 0.021664783358573914 - ] + ], + "primary_image": "https://blogger.googleusercontent.com/img/b/R29vZ2xl/AVvXsEhNv5jXFC1JGhdDluKblp8CLBsij7qfCDhXEOZgff7kMBcZarnmcZZnidcK91Mr1SxFL-m9e68Ca-_hDlR1RJhasNMwsBwXK3e3QUppOS2d9SYXnXJxl0y8jeqvZ0COFVGozjvJntqDqUg/s1600/20170913_131941-01.jpeg", + "total_images": 3, + "image_source": "seed_build" }, { "product_name": "Mustard Leaves", @@ -46268,8 +46417,13 @@ "size": "Standard", "size_variants": [], "description": "Mustard Leaves. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://thumb.wikimedia.org/wikipedia/commons/thumb/a/ae/Alliaria_petiolata_-_garlic_mustard_-_desc-young_foliage.jpg/960px-Alliaria_petiolata_-_garlic_mustard_-_desc-young_foliage.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail", + "image_urls": [ + "https://thumb.wikimedia.org/wikipedia/commons/thumb/a/ae/Alliaria_petiolata_-_garlic_mustard_-_desc-young_foliage.jpg/960px-Alliaria_petiolata_-_garlic_mustard_-_desc-young_foliage.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail", + "https://thumb.wikimedia.org/wikipedia/commons/thumb/0/0d/Buro_with_mustard_leaves_and_eggplant.jpg/960px-Buro_with_mustard_leaves_and_eggplant.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail", + "https://upload.wikimedia.org/wikipedia/commons/6/60/Curly_mustard_leaves.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail_unscaled", + "https://thumb.wikimedia.org/wikipedia/commons/thumb/d/dd/Garlic_Mustard_%28Alliaria_petiolata%29_-_Kitchener%2C_Ontario.jpg/960px-Garlic_Mustard_%28Alliaria_petiolata%29_-_Kitchener%2C_Ontario.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -46673,7 +46827,10 @@ 0.03504066541790962, 0.07110962271690369, 0.046843890100717545 - ] + ], + "primary_image": "https://thumb.wikimedia.org/wikipedia/commons/thumb/a/ae/Alliaria_petiolata_-_garlic_mustard_-_desc-young_foliage.jpg/960px-Alliaria_petiolata_-_garlic_mustard_-_desc-young_foliage.jpg?utm_source=commons.wikimedia.org&utm_campaign=imageinfo&utm_content=thumbnail", + "total_images": 4, + "image_source": "seed_build" }, { "product_name": "Spring Garlic", @@ -46682,8 +46839,12 @@ "size": "Standard", "size_variants": [], "description": "Spring Garlic. Sold loose by weight or by the piece; no brand, no fixed pack size.", - "image_url": null, - "image_urls": [], + "image_url": "https://images.openfoodfacts.org/images/products/506/014/505/0501/front_fr.10.400.jpg", + "image_urls": [ + "https://images.openfoodfacts.org/images/products/506/014/505/0501/front_fr.10.400.jpg", + "https://images.openfoodfacts.org/images/products/600/982/698/0040/front_fr.4.400.jpg", + "https://images.openfoodfacts.org/images/products/001/380/055/6677/front_en.3.400.jpg" + ], "price_range": null, "providers": [], "fssai_license": null, @@ -47087,7 +47248,10 @@ 0.03384043276309967, 0.010289044119417667, 0.016713209450244904 - ] + ], + "primary_image": "https://images.openfoodfacts.org/images/products/506/014/505/0501/front_fr.10.400.jpg", + "total_images": 3, + "image_source": "seed_build" }, { "product_name": "Jasmine", diff --git a/docs/DRIFT_REPORT_RESPONSE.md b/docs/DRIFT_REPORT_RESPONSE.md index 02b421e..96274e1 100644 --- a/docs/DRIFT_REPORT_RESPONSE.md +++ b/docs/DRIFT_REPORT_RESPONSE.md @@ -3,26 +3,28 @@ Reply to the seven findings dated 31 August 2026 (drop `8e1448e176d843d08183d387ad724f95` → run `0ed4c77b0ca14e03b1aaf6b5b77d1994`). +Thank you for this. It is the most useful bug report this project has had, and +two of the seven led us to faults we had not found ourselves — one of them +affecting rows well outside the ones you named. + Every figure below was measured against the same live deployment, not read off source. Where we disagree with a finding, the evidence is included so you can check it rather than take our word for it. -**Summary:** four items are fixed and ship in the next backend deploy. One -(#02) was already in the API and we had failed to document it — that is our -fault and the docs are now corrected. #01 is diagnosed, and the cause is not -what either of us assumed. #06 is confirmed but carries a trap that means we -should agree an approach before touching it. - | # | Finding | Status | | --- | --- | --- | -| 01 | Pack sizes replaced between scrapes | **Diagnosed** — three causes, not one. Fix needs your input | -| 02 | `rejected` is a bare count | **Already shipped, now documented** — plus `row` added | +| 01 | Pack sizes replaced between scrapes | **Diagnosed** — three causes, not one. One is fixed; the other two need your input | +| 02 | `rejected` is a bare count | **Was already in the API — our docs failed you.** `row` added | | 03 | Manifest brands are not catalogue keys | **Fixed** — `brand_key` published | | 04 | Run files carry no `from_drop` | **Fixed** | | 05 | Manifest carries no `source_row` | **Fixed** | -| 06 | Duplicate brands and products | **Confirmed.** Read the trap below before we act | +| 06 | Duplicate brands and products | **Brand merged. Stray numbers diagnosed — there are 18, not 1** | | 07 | No loose-produce coverage | **Fixed** — 159-row base list, and the upload path now handles produce | +> **Everything below ships in the next backend deploy and is not live yet.** The +> only change already applied to the live database is the Haldiram merge in #06. +> We will confirm when the deploy lands; please do not re-test until then. + --- ## 01 — Pack sizes and names are replaced between scrapes @@ -31,11 +33,12 @@ You asked us to confirm whether a pack size that once existed is meant to survive a re-scrape. **It is not, today** — but that is only the last of three causes, and fixing it alone would not have saved your links. -### Cause 1: pack sizes are invented when a scrape does not supply them +### Cause 1: the pack sizes were never scraped facts. Some were generated. -`_sizes_for()` falls back to `default_size_variants(category, name)` when a -product declares no size. That fallback is **keyed on the resolved category**, -and the resolved category is not stable between runs. Measured today: +When a product declares no size, `_sizes_for()` falls back to +`default_size_variants(category, name)`. That fallback is **keyed on the +resolved category**, and the resolved category is not stable between runs. +Measured today: ``` category "Snacks" -> ['55g', '150g', '200g'] @@ -46,18 +49,25 @@ category "Breakfast Cereal" -> ['250g', '500g', '1kg'] Now compare your table. Cheetos id 25 is **55g** — the *Snacks* set. Ids 26 and 27 are **100g** and **250g** — the *unresolved* set. The same product was ingested once with its category resolved and once without, and produced two -disjoint sets of pack sizes. Your Cheerios ids (100g, 250g, 500g) sit across the -Breakfast Cereal set and the unresolved set the same way. +disjoint sets of pack sizes. Your Cheerios ids sit across the Breakfast Cereal +set and the unresolved set the same way. -So the pack sizes were never scraped facts that changed. Some of them were -generated, and the generator's input moved. +So a 100 g bag and a 250 g bag are indeed different SKUs, and you are right not +to re-point one at the other — but in these cases neither number came off a +pack. They were both guesses, from two different guesses about the category. + +**This one is now fixed for the class of product where it does most damage.** +Unbranded and loose goods no longer receive invented sizes at all (see #07). For +branded packaged goods the fallback still runs, because a brand really does sell +a small/medium/large range and omitting it entirely loses more than it saves. +That is the part we would like your view on — see *What we need from you*. ### Cause 2: the product name gains or loses a brand prefix Your own #06 has the evidence: `Hot Heads 30g` became `Nestle Hot Heads`, and `PepsiCo Kurkure Masala Munch 90g` coexists with `Kurkure Masala Munch 90g`. `image_id` is derived from the name, so a prefix appearing or disappearing moves -the id even when the product is identical. +the id even when the product is identical. Still open; see #06. ### Cause 3: the write then deletes whatever is not in the new set @@ -71,54 +81,55 @@ re-scrape**, which matches your finding that zero of eleven could be repaired. ### Not the upload path -Worth stating plainly, because it affects how much you need to worry: the +Worth stating plainly, because it changes how much you need to worry: the **upload** path — everything reached through `POST /api/uploads/catalog` — uses -`cleanup=False` and has always done so. A sheet you send can never delete a row -it does not mention. The deletions came from brand scraping only. - -### What we need from you - -The real fix is to stop causes 1 and 2 (do not invent sizes for a product -already in the catalogue; settle the naming convention), and to soft-retire -rather than delete for cause 3. That third part changes how the production -catalogue is written and we would rather agree it with you than spring it: - -- Would a `retired_at` timestamp plus exclusion from the default read work for - you, instead of the row being deleted? That preserves the `image_id` so your - stored link resolves to something, and lets us give you the `superseded_by` - and per-run changelog you asked for. -- If so, do you want retired rows visible through an explicit query, or gone - from the API entirely? +`cleanup=False` and always has. A sheet you send can never delete a row it does +not mention. The deletions came from brand scraping only. ### Your two direct questions **Is `image_id` stable across re-scrapes for a product whose name and pack size have not changed?** Yes. It is a pure deterministic function of brand, product -name and pack size, with no clock, counter or run id in it. Verified: +name and pack size, with no clock, counter or run id in it: ``` build_image_id('pepsico', 'Cheetos Chips', '100g') -> pepsico_cheetos_chips_100g ``` -Storing it rather than our row id is the right call and we have documented the -guarantee so it does not quietly change. **The caveat is #06:** the guarantee is -only as good as the stability of the name, and inconsistent brand prefixing -breaks exactly that. +Storing it rather than our row id is the right call, and we have now documented +the guarantee in `INGESTION_API.md` so it cannot quietly change. **The caveat is +#06:** the guarantee is only as good as the stability of the name, and +inconsistent brand prefixing breaks exactly that. **Do you want to know when a product is dropped or renamed?** Yes, and we agree -it should exist. It falls out of the retirement model above rather than being a -separate feature, which is why we would like to settle that first. +it should exist. It falls out of the retirement model below rather than being a +separate feature. + +### What we need from you + +- Would a `retired_at` timestamp plus exclusion from the default read work + instead of the row being deleted? That preserves the `image_id` so your stored + link resolves to *something*, and gives us somewhere to hang the + `superseded_by` and per-run changelog you asked for. +- If so: should retired rows be reachable through an explicit query, or absent + from the API entirely? +- On cause 1: would you rather we **stopped inventing sizes altogether** for + branded goods too? It would shrink the catalogue and lose some genuine + variants, but every remaining row would be a size somebody actually saw on a + pack. We can go either way and would rather match how you consume it. --- ## 02 — `rejected` is a count with no reason -**This is our documentation failure, not a missing feature.** `rejections[]` has -been in every response — single-batch read and list endpoint both, since -`to_out(slim=True)` strips only `products` — carrying `product_name`, `size` and -`reason` per refused row. It was absent from `INGESTION_API.md`, which documents -`"rejected": 0` and never mentions the array, so there was no way for you to -know it was there. Sorry — that is a straightforwardly bad docs bug. +**This one is our documentation failing you, not a missing feature, and we are +sorry for the time it cost.** `rejections[]` has been in every response — the +single-batch read and the list endpoint both, since `to_out(slim=True)` strips +only `products` — carrying `product_name`, `size` and `reason` per refused row. + +It was absent from `INGESTION_API.md`, which documents `"rejected": 0` and never +mentions the array, so there was no way for you to know it was there. You were +diffing 19 against 17 because our docs told you that was all you had. The genuine gap was the row number, which is now added: @@ -133,6 +144,8 @@ The genuine gap was the row number, which is now added: convention as the `422` responses, so it matches what the operator sees on screen. `null` only when the row cannot be located. Capped at 50 per file. +The array is now documented, with a field table. + --- ## 03 — Manifest brands are not catalogue keys @@ -143,13 +156,13 @@ Fixed. Every entry in `products[]` now carries `brand_key` beside `brand`: { "brand": "24 Mantra", "brand_key": "24_mantra", ... } ``` -This is generated by the same function the storage layer uses to name the table, +It is generated by the same function the storage layer uses to name the table, so it cannot drift from the key the catalogue is actually addressed by. Your normalisation is correct as far as we can tell, but it is a guess, and the failure mode is silent — a wrong key finds nothing rather than erroring. -Thank you for degrading rather than failing the batch on an unreadable brand; -that is the right behaviour and we should have done it on our side too. +Thank you for making a single unreadable brand degrade rather than fail the +batch. That is the right behaviour and we should have done it on our side too. --- @@ -169,8 +182,8 @@ from — the exact inverse of `released_to`: ``` `null` for a file that went straight into a run without sitting in an inbox — -which, under `UPLOAD_AUTORUN=true`, is every file you send, because the id you -are handed is already the run. +which, under the current `UPLOAD_AUTORUN=true`, is every file you send, because +the id you are handed is already the run. There is a test in our suite that stages two drops from different senders both named `products.csv` and asserts they are distinguishable, so the collision you @@ -193,43 +206,86 @@ set difference rather than a name-matching heuristic. ## 06 — Duplicate brands, duplicate products, stray names -Confirmed against the live database today: **55 brand tables, 1,414 products** -(you counted 1,614; the difference is a day of drift plus, we think, your count -including rejected rows — worth reconciling if it matters). +### The brand split — merged + +`haldiram` (1 product) is gone; `haldirams` (2) is the survivor. + +The split was not a typo. **Nothing in the system knew the two spellings were +one brand** — neither was in the alias map, so each resolved to itself and every +upload built whichever table its sheet happened to name. Merging the rows alone +would have fixed nothing: the next sheet spelling it without the "s" would +rebuild the table. The alias is in, so `Haldiram`, `haldiram`, `HALDIRAM` and +`Haldiram's` all now resolve to `haldirams`. + +While merging we found the singular row was a corrupted duplicate of one already +in the plural table — same product, but with a stray `45` in the name, a raw +category id, and a bare-number size. We kept the clean row and carried across +the one thing the corrupted row had that it lacked: a newer price. + +We also found, and corrected, something you could not have seen: **both +`haldirams` rows were carrying Lion Dates' FSSAI licence** (`10012042000244`, +the number on all 21 Lion Dates products) rather than Haldiram's own +(`10012011000140`). That is a regulatory identifier on the wrong manufacturer's +product, and it is fixed in the database and the seed file. + +### The stray number — there are 18 of them, and we know what it is + +You found `Lays Classic Salted 52g 150` and asked us to check for the pattern +elsewhere. **It affects 18 products across at least seven brands:** ``` -haldiram 1 britannia 6 -haldirams 2 parle 3 against hindustan_unilever 443 - patanjali 3 +Aashirvaad Shudh Chakki Atta 5kg 40 size_variants ['40'] price 299 +Lays Classic Salted 52g 150 size_variants ['150'] price 21 +Britannia Good Day Cashew 200g 60 size_variants ['60'] price 52 +Coca-Cola 750ml 72 size_variants ['72'] price 42 +Dove Cream Beauty Bar 100g 64 size_variants ['64'] price 76 +Horlicks Classic Malt 500g 18 size_variants ['18'] price 289 +... 12 more ``` -So: the split brand is real, and Britannia/Parle/Patanjali do look like scrapes -that stopped part-way rather than genuinely small brands. We will re-run those -three. +The number is **not** a price — 40 against ₹299, 150 against ₹21. It is the +**case-pack count**: how many units come in a carton. A sheet's "Quantity" +column was mapped to the pack-size field, the bare number became the size, and +`_to_storage_row` then appended it to the product name. -### The trap, which is why we have not just fixed this +**The cause is already fixed**, on two layers, both verified today: -**De-duplicating the PepsiCo pairs means renaming a product, and `image_id` is -derived from the name.** Renaming `PepsiCo Kurkure Masala Munch 90g` to -`Kurkure Masala Munch 90g` does not merge the two rows — it mints a third id and -breaks any link pointing at either of the first two. You have just migrated onto -storing `image_id`. A well-meant cleanup on our side would re-break exactly what -you have finished repairing. +- The column mapper no longer maps a bare `Quantity` column to pack size. A + sheet with both `Pack Size` and `Quantity` now binds only `Pack Size`. +- `_sizes_for()` discards a unitless number and records why: + `ignored pack size '150': a number with no unit is a quantity, not a size`. -The same applies to stripping the stray `150` from `Lays Classic Salted 52g 150`. +So no new rows can acquire this. The 18 existing ones are legacy damage and we +will repair them — see the request below. -So before we touch it we would like to agree: +### The PepsiCo duplicate pairs — we need one decision from you first + +Both pairs are still there (ids 661/730 and 662/731). We have deliberately not +touched them, because **de-duplicating means renaming, and `image_id` is derived +from the name.** Renaming `PepsiCo Kurkure Masala Munch 90g` to +`Kurkure Masala Munch 90g` does not merge the two rows — it mints a *third* id +and breaks any link pointing at either of the first two. You have just migrated +onto storing `image_id`; a well-meant cleanup on our side would re-break exactly +what you have finished repairing. The same applies to stripping the stray +numbers. + +So, three things to agree before we act: 1. **Which convention wins** — brand prefix in the product name, or not? We have - no strong preference; we care only that it is one of them. Our lean is - *without* the prefix, since the brand is already a column. -2. **How the merge is communicated.** If we can give you the old-id → - new-id mapping for every row we touch, in advance, does that let you - re-point rather than clear? That is straightforward for us to produce. + no strong preference and care only that it is one of them. Our lean is + *without*, since the brand is already a column. +2. **Would an old-id → new-id mapping, delivered in advance for every row we + touch, let you re-point rather than clear?** That is straightforward for us + to produce and would cover both the 18 stray-number rows and the PepsiCo + pairs. 3. **Timing**, so it lands in one pass rather than trickling. -`haldiram` → `haldirams` is a three-row merge and much lower risk; we can do -that one immediately if you would rather not wait for the rest. +### Brand coverage + +Confirmed from the live database: **55 tables, 1,414 products**. You counted +1,614 — worth reconciling, but the shape matches. And yes: **Britannia (6), +Parle (3) and Patanjali (3) are incomplete scrapes, not small brands.** We will +re-run those three. --- @@ -241,7 +297,7 @@ because it was not only a coverage gap. ### What was actually happening Produce rows were not rejected. They were **misfiled**, which is worse. The -brand fallback takes the first word of the name and then whole-word matches it +brand fallback takes the first word of the name and whole-word matches it against our alias map: ``` @@ -249,12 +305,12 @@ Apple -> brand "Apple" -> junk table brand_apple Tomato -> brand "Tomato" -> junk table brand_tomato Bitter Gourd -> brand "Bitter" -> junk table brand_bitter Curry Leaves -> brand "Curry" -> junk table brand_curry -Red Rose -> brand "Red" -> brand_brooke_bond <-- +Red Rose -> brand "Red" -> brand_brooke_bond <-- ``` That last one is not a typo. A rose was being written into the Brooke Bond tea -catalogue, and our enrichment then stamps that brand's real FSSAI licence number -onto the row. Your 139 hand-typed products were the visible symptom; this was +catalogue, and our enrichment then stamps that brand's real FSSAI licence onto +the row. Your 139 hand-typed products were the visible symptom; this was underneath it. ### What now happens @@ -265,28 +321,38 @@ them. Fruit, vegetables, greens, herbs, flowers, fish, eggs and loose dairy are covered, alongside the pulses, grains, spices, oils and sugar that already were. Five categories were added — Fruits & Vegetables, Fresh Herbs & Greens, Flowers, -Fish & Seafood, Eggs — with HSN codes and a 0% GST rate, since unprocessed -produce is nil-rated rather than reduced-rate. +Fish & Seafood, Eggs — with HSN codes at 0% GST, since unprocessed produce is +nil-rated rather than reduced-rate. -Merchant misspellings from your own data are handled: `Bitter guard`, -`Bottle ground`, `Ladies Finger` all resolve. +Misspellings from your own audit are handled: `Bitter guard`, `Bottle ground` +and `Ladies Finger` all resolve. + +An uploaded produce row also now keeps only what the sheet actually said. Name, +weight and price are stored; HSN, SKU, barcode, FSSAI and description are left +null rather than invented, and **no pack sizes are generated** — which is cause +1 of your finding #01, kept out of this table from the start. ### The base list -**159 rows**, in the shape you asked for: name and category only, no brand, no -pack size, no price. Fruit (42), vegetables (53), greens and herbs (17), flowers -(14), fish and seafood (15), loose dairy (13), eggs (5). It includes the specific -items your audit listed — Jasmine, Lotus, Red Rose, Thulasi, Drumstick, Curry -Leaves, the four banana varieties, Tuna, Mackerel. +**159 rows**, in the shape you asked for: name and category, no brand, no pack +size, no price. -Each row carries a `search_query` embedding, so these are reachable through -semantic search and not just exact match. The list is hand-authored rather than -scraped, so it is not subject to any of #01. +``` +Fruits & Vegetables 95 Fish & Seafood 15 +Fresh Herbs & Greens 17 Dairy (loose) 13 +Flowers 14 Eggs 5 +``` -**Images are not included yet.** You asked for name and image; we have shipped -the names. Sourcing 159 licensable produce photographs is a separate piece of -work and we did not want to hold the list for it — tell us if the list is not -useful to you without them and we will prioritise accordingly. +It includes the specific items your audit listed — Jasmine, Lotus, Red Rose, +Thulasi, Drumstick, Curry Leaves, the four banana varieties, Tuna, Mackerel. +Each row carries a search embedding, so these are reachable through semantic +search and not just exact match. The list is hand-authored rather than scraped, +so none of #01 applies to it. + +**Images: 112 of the 159 rows (70%) currently have one** — all of the fruit, +vegetables, greens and herbs. Flowers, fish, loose dairy and eggs are still +name-only; we stopped the fetch part-way and will finish it. Tell us if the list +is more useful to you complete-but-later or partial-but-now. ### One limitation worth knowing @@ -295,11 +361,11 @@ means "this is a brand". So place-qualified produce — `Salem Mango`, `Mysore Banana`, `Jammu Apple`, all real strings from your Ragul Stores data — still reads as branded, because `Mysore` is also a real brand (Mysore Sandal). -The workaround is already in the pipeline: **if the sheet has a Brand column and -leaves the cell empty, we believe it** and file the row under Own Products +There is a workaround already in the pipeline: **if the sheet has a Brand column +and leaves the cell empty, we believe it** and file the row under Own Products regardless of the name. If your merchants' sheets carry an empty brand column, -those rows will land correctly. If they carry no brand column at all, the -name-based test is what applies. +those rows land correctly. If they carry no brand column at all, the name-based +test applies. We would rather be conservative here. Collapsing a real regional brand into the unbranded bucket is much harder to undo than a mango sitting in the wrong table. @@ -311,13 +377,28 @@ unbranded bucket is much harder to undo than a mango sitting in the wrong table. - The produce lexicon was run against **all 1,414 products in the live catalogue** and against all **231 brand aliases**: zero reclassifications in either. That check is now a test, so it runs on every change. -- It also surfaced a pre-existing bug we would not otherwise have found: our - pack-size stripper was eating the word after a number, so - `24 Mantra Organic Moong Dal 500g` lost its brand entirely and was being filed - as an unbranded commodity. **Every brand whose name starts with a digit hit - this.** "24 Mantra" appears on your finding-03 list, which is how we noticed. - Fixed. +- It surfaced a pre-existing bug we would not otherwise have found: our + pack-size stripper was eating the word *after* a number, so + `24 Mantra Organic Moong Dal 500g` lost its brand entirely and was filed as an + unbranded commodity. **Every brand whose name starts with a digit hit this.** + "24 Mantra" is on your finding-03 list, which is how we noticed. Fixed. - All 159 seeded rows were checked to classify identically to how an uploaded copy of the same name would, so a grocer typing "Tomato" lands on the seeded row instead of creating a second one. -- Full suite: **1,054 tests passing.** +- Full suite: **1,108 tests passing.** + +## Two things you did not ask about, but should know + +**Barcodes.** We do not generate them; a product carries one only if the sheet +supplied it. 95 of our 1,414 products have one — and checking them against Open +Food Facts, **33 are attached to the wrong product** (`Lion Dates Powder` is +stored under a barcode Open Food Facts holds as a Dutch confection) and a +further 38 are not valid GTINs at all. If you join on barcode anywhere, treat +ours as unreliable until we have cleaned them. The lookup that produced them has +since been tightened. + +**Nutrition provenance.** Rows in `nutrition_facts` sourced from Open Food Facts +were matched by *name*, with confidences as low as 0.32. We have built an exact +barcode-keyed lookup to replace that, but given the barcode quality above it can +currently upgrade only a handful of rows. Treat low-confidence nutrition as +indicative, not authoritative. diff --git a/scripts/backfill_nutrition_from_barcodes.py b/scripts/backfill_nutrition_from_barcodes.py new file mode 100644 index 0000000..d8296af --- /dev/null +++ b/scripts/backfill_nutrition_from_barcodes.py @@ -0,0 +1,295 @@ +#!/usr/bin/env python3 +""" +Upgrade nutrition data from a fuzzy name match to an exact barcode match. + +WHY +--- +Every Open Food Facts call in this project searches by brand and product name +and then scores whatever comes back - `nutrition_data_service._search_openfoodfacts`, +the barcode cascade's own `OpenFoodFactsSource.search`, and `image_search`. That +is a guess, and the stored evidence says so: OFF-sourced rows in +`nutrition_facts` carry match confidences as low as 0.32, which is exactly +`MIN_MATCH_CONFIDENCE`. + +For the products that carry a barcode we can do better. The barcode is the +identifier printed on the pack, so `/api/v2/product/{code}` returns that product +or nothing. This walks the catalogue's barcoded rows and rewrites their +nutrition from the exact record. + +WHAT IT WILL AND WILL NOT TOUCH +------------------------------- +It writes to `nutrition_facts` and to NOTHING else. No product name, price, +image, category or barcode in any brand table is modified. The invalid barcodes +it finds are reported, not repaired. + +It also refuses to overwrite a human's work: `upsert_nutrition_facts` is an +ON CONFLICT ... DO UPDATE, so a row whose existing `data_source` is `manual` or +`excel_upload` is skipped. Replacing a 0.32 name-match with an exact barcode +match is the point of this script; replacing something a person typed is not. + +THE THING TO UNDERSTAND BEFORE READING THE OUTPUT +------------------------------------------------- +The LOOKUP is exact. The STORED BARCODE is not. Measured across the catalogue, +a quarter of the records found this way described a different product, because +the barcode on our row was wrong: + + Aachi Chicken Masala 50g -> OFF "Chicken Kabab/65 Masala" + Tata Tea Gold 500g -> OFF "Tata Tea Gold Care" + Lion Dates Powder 100g -> OFF "PEPER NOTEN" + +So every record still passes `matching.is_match`, and the report below prints +the name-similarity score for every candidate - accepted or not - because no +threshold cleanly separates the two groups (two products tie at 0.773 with +opposite verdicts). Read the SKIPPED list: it is a list of barcodes that are +probably wrong in OUR catalogue. + +Usage: + + python -m scripts.backfill_nutrition_from_barcodes # dry run + python -m scripts.backfill_nutrition_from_barcodes --apply + python -m scripts.backfill_nutrition_from_barcodes --min-similarity 0.7 + +`--dry-run` is the default and `--apply` must be explicit: this writes to +whatever database `backend/.env` points at, which is production. The target host +is printed on startup. +""" +from __future__ import annotations + +import argparse +import logging +import re +import sys +import time +from pathlib import Path +from typing import Any, Dict, List, Optional + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +from app.infrastructure.settings import DB_HOST, DB_NAME +from app.services.enrichment.barcode.matching import is_match, name_similarity +from app.services.enrichment.barcode.models import BarcodeCandidate +from app.services.enrichment.barcode.sources.open_food_facts import ( + fetch_product_by_barcode, +) +from app.services.nutrition_data_service import ( + BARCODE_MIN_NAME_SIMILARITY, + fetch_verified_nutrition_by_barcode, +) +from app.services.nutrition_db import get_nutrition_facts, upsert_nutrition_facts +from app.services.vector_store import _connect, display_name_for_suffix + +logging.basicConfig(level=logging.INFO, format="%(message)s") +logger = logging.getLogger("backfill_nutrition") + +# Courtesy gap between calls to a free community API. +PAUSE_SECONDS = 0.35 + +# data_source values that mean "a person put this here". Never overwritten. +HUMAN_SOURCES = {"manual", "excel_upload"} + +# A GTIN is 8, 12, 13 or 14 digits. Anything else in the barcode column is not a +# barcode - the catalogue holds "8900000000000.0" (a float that survived an +# Excel import) and several 8-digit codes attached to three different pack sizes +# at once. Reported rather than looked up; a bad code cannot match anything. +_GTIN = re.compile(r"^\d{8}$|^\d{12,14}$") + +_SIZE_IN_NAME = re.compile(r"(\d+(?:[.,]\d+)?\s*(?:kg|g|gm|gms|ml|l|ltr))\b", re.I) + + +def _catalogue_rows() -> List[Dict[str, Any]]: + """Every catalogue row that carries a barcode. + + Columns are read defensively: `_ensure_columns` adds them lazily, so an + older brand table can be missing `size_variants` entirely - which is a + crash, not a warning, if you SELECT it blindly. + """ + conn = _connect() + if conn is None: + return [] + rows: List[Dict[str, Any]] = [] + with conn: + with conn.cursor() as cur: + cur.execute( + "SELECT table_name FROM information_schema.tables " + "WHERE table_schema='public' AND table_name LIKE 'brand_%' " + "ORDER BY table_name" + ) + tables = [r[0] for r in cur.fetchall()] + for table in tables: + cur.execute( + "SELECT column_name FROM information_schema.columns " + "WHERE table_name=%s", (table,)) + cols = {r[0] for r in cur.fetchall()} + if "barcode" not in cols: + continue + select = "product_name,barcode,image_id,category" + if "size_variants" in cols: + select += ",size_variants" + cur.execute( + f"SELECT {select} FROM {table} " + f"WHERE barcode IS NOT NULL AND barcode <> ''") + for record in cur.fetchall(): + name, barcode, image_id, category = record[:4] + sizes = record[4] if len(record) > 4 else None + match = _SIZE_IN_NAME.search(name or "") + size = (sizes[0] if sizes else "") or (match.group(1) if match else "") + rows.append({ + "table": table, + "brand": display_name_for_suffix(table[len("brand_"):]), + "product_name": name, + "barcode": str(barcode).strip(), + "image_id": image_id, + "category": category or "", + "size": size, + }) + return rows + + +def main() -> int: + parser = argparse.ArgumentParser( + description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--apply", action="store_true", + help="commit the changes (default is a dry run)") + parser.add_argument("--dry-run", action="store_true", + help="explicit no-op; this is already the default") + parser.add_argument("--min-similarity", type=float, + default=BARCODE_MIN_NAME_SIMILARITY, + help=f"name-similarity floor (default {BARCODE_MIN_NAME_SIMILARITY})") + args = parser.parse_args() + apply = args.apply and not args.dry_run + + logger.info("Target database: %s / %s", DB_HOST, DB_NAME) + logger.info("Mode: %s", "APPLY - this writes to nutrition_facts" + if apply else "DRY RUN - nothing is written") + logger.info("Name-similarity floor: %.2f", args.min_similarity) + + rows = _catalogue_rows() + if not rows: + logger.error("No barcoded rows found (or no database connection).") + return 1 + logger.info("") + logger.info("%d catalogue row(s) carry a barcode.", len(rows)) + + invalid: List[Dict] = [] + lookups: List[Dict] = [] + for row in rows: + (invalid if not _GTIN.match(row["barcode"]) else lookups).append(row) + + if invalid: + logger.info("") + logger.info("NOT A VALID GTIN - skipped, and wrong in the catalogue " + "rather than wrong here (%d):", len(invalid)) + for row in invalid: + logger.info(" %-18s %s", row["barcode"], row["product_name"][:48]) + + written = skipped_human = not_in_off = no_nutriments = 0 + accepted: List[str] = [] + rejected: List[str] = [] + thin: List[str] = [] + + logger.info("") + logger.info("Looking up %d barcode(s) ...", len(lookups)) + for i, row in enumerate(lookups, start=1): + existing = get_nutrition_facts(row["brand"], row["image_id"]) or {} + if (existing.get("data_source") or "") in HUMAN_SOURCES: + skipped_human += 1 + continue + + product = fetch_product_by_barcode(row["barcode"]) + if not product: + not_in_off += 1 + time.sleep(PAUSE_SECONDS) + continue + + off_name = product.get("product_name") or "" + similarity = name_similarity(off_name, row["product_name"]) + line = (f"{similarity:5.3f} {row['product_name'][:36]:38} " + f"-> {off_name[:36]}") + + # The gate is run HERE, separately, so the report can tell two very + # different outcomes apart. Deciding it from the service's + # "unavailable" alone conflated them, and the first version of this + # report accused a dozen perfectly good barcodes of being wrong when + # the real answer was that Open Food Facts holds a near-empty record + # for them. One is our data to fix; the other is nobody's fault. + candidate = BarcodeCandidate( + barcode=row["barcode"], source_name="Open Food Facts", + candidate_title=off_name, + candidate_brand=product.get("brands") or "", + candidate_size=product.get("quantity") or "", + ) + matched, _sim = is_match(candidate, row["brand"], row["product_name"], + row["size"], min_name_similarity=args.min_similarity) + if not matched: + rejected.append(line) + time.sleep(PAUSE_SECONDS) + continue + + facts = fetch_verified_nutrition_by_barcode( + row["barcode"], row["brand"], row["product_name"], + row["size"], row["category"], + min_name_similarity=args.min_similarity, + ) + if facts.get("data_status") == "unavailable": + # Gate passed, so this IS our product - OFF simply has no usable + # numbers for it. Nothing to fix on either side. + no_nutriments += 1 + thin.append(line) + else: + accepted.append(line) + if apply: + facts.update({ + "brand": row["brand"], + "image_id": row["image_id"], + "product_name": row["product_name"], + "category": row["category"], + }) + if upsert_nutrition_facts(facts): + written += 1 + else: + written += 1 + + if i % 25 == 0: + logger.info(" %d/%d", i, len(lookups)) + time.sleep(PAUSE_SECONDS) + + # ---- the report -------------------------------------------------------- + logger.info("") + logger.info("ACCEPTED (%d) - barcode found AND the record is our product:", + len(accepted)) + for line in sorted(accepted, reverse=True): + logger.info(" %s", line) + + if thin: + logger.info("") + logger.info("MATCHED BUT EMPTY (%d) - the right product, but Open Food " + "Facts holds no usable nutrient values. Nothing wrong with " + "our barcode:", len(thin)) + for line in sorted(thin, reverse=True): + logger.info(" %s", line) + + logger.info("") + logger.info("SKIPPED (%d) - OFF knows the barcode, but as a different " + "product. OUR barcode is the suspect one:", len(rejected)) + for line in sorted(rejected, reverse=True): + logger.info(" %s", line) + + logger.info("") + logger.info(" barcoded rows %d", len(rows)) + logger.info(" not a valid GTIN %d", len(invalid)) + logger.info(" human-entered, kept %d", skipped_human) + logger.info(" not in Open Food Facts %d", not_in_off) + logger.info(" matched but empty %d", no_nutriments) + logger.info(" found, wrong product %d", len(rejected)) + logger.info(" %s %d", "WRITTEN " if apply else "would write ", written) + + logger.info("") + if apply: + logger.info("Done. nutrition_facts updated; no brand table was touched.") + else: + logger.info("Dry run - nothing written. Re-run with --apply to commit.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_barcode_enrichment_guards.py b/tests/test_barcode_enrichment_guards.py new file mode 100644 index 0000000..1fd9605 --- /dev/null +++ b/tests/test_barcode_enrichment_guards.py @@ -0,0 +1,185 @@ +"""Two guards on the barcode stage: it may not erase, and it may not guess loosely. + +Both of these were latent behind `ENABLE_BARCODE_LOOKUP`, which is false in +production. Switching it on to fill barcodes for newly uploaded products would +have triggered both at once, so they are fixed before that flag is ever flipped. + +1. THE WIPE + `BarcodeResult.as_product_fields()` always returns its full nine keys, and a + failed lookup makes every one of them None. `EnrichmentStage.apply` merged + that dict straight in, so a MISS replaced the barcode the shop had typed with + NULL. Verified end to end before the fix: a sheet sending 8901262010016 + stored None. Since the cascade misses far more often than it hits, enabling + the stage would have destroyed more real barcodes than it found. + +2. THE LOOSE GATE + `matching.is_match` defaults to a 0.45 name-similarity floor. That is a fair + general default, but by the time a candidate reaches the gate its brand and + pack size have already matched, so the name carries the whole decision. At + 0.45, more than half the accepted matches in this catalogue were a different + product - which is how 33 of the 95 stored barcodes came to be wrong. + +Every product name below is real, taken from the live catalogue and from what +Open Food Facts actually returns for those codes. +""" +from __future__ import annotations + +import asyncio + +import pytest + +from app.services.enrichment.barcode import stage as barcode_stage +from app.services.enrichment.barcode.models import ( + BarcodeCandidate, + BarcodeResult, + LookupStatus, +) +from app.services.enrichment.barcode.service import BarcodeLookupService + + +class _Service: + """Stands in for the cascade. Records what it was asked to look up.""" + + def __init__(self, result=None): + self.result = result or BarcodeResult.null_result(LookupStatus.NOT_FOUND) + self.asked = [] + + def lookup_one(self, brand, title, size, category=""): + self.asked.append(title) + return self.result + + +@pytest.fixture +def stage_on(monkeypatch): + """Turn the stage on and hand back the stub service it will use.""" + service = _Service() + monkeypatch.setattr(barcode_stage, "ENABLE_BARCODE_LOOKUP", True) + monkeypatch.setattr(barcode_stage, "get_default_service", lambda: service) + return service + + +def _apply(product, brand="Amul"): + stage = barcode_stage.BarcodeEnrichmentStage() + return asyncio.run(stage.apply(product, brand)) + + +# --------------------------------------------------------------------------- +# 1. The wipe +# --------------------------------------------------------------------------- +def test_a_failed_lookup_does_not_erase_the_sheets_barcode(stage_on): + """The regression, stated as plainly as it happened.""" + product = {"product_name": "Amul Butter 500g", "size": "500g", + "barcode": "8901262010016", "barcode_type": "EAN-13"} + + _apply(product) + + assert product["barcode"] == "8901262010016" + assert product["barcode_type"] == "EAN-13" + + +def test_a_product_that_already_has_a_barcode_is_never_looked_up(stage_on): + """Not merely harmless - the lookup is skipped outright. + + A barcode the shop supplied is better evidence than anything the cascade + can find: they are holding the pack. Skipping saves the network call too, + which on a 2000-row sheet is the difference that matters. + """ + _apply({"product_name": "Amul Butter 500g", "size": "500g", + "barcode": "8901262010016"}) + + assert stage_on.asked == [], "a row with a barcode reached the network" + + +def test_a_blank_barcode_still_gets_looked_up(stage_on): + """The guard must not turn the stage off for the rows it exists to serve.""" + _apply({"product_name": "Amul Ghee 1L", "size": "1L", "barcode": ""}) + + assert stage_on.asked == ["Amul Ghee 1L"] + + +def test_a_found_barcode_fills_a_blank(stage_on, monkeypatch): + found = BarcodeResult(barcode="8901262010023", barcode_type="EAN-13", + barcode_verified=True) + monkeypatch.setattr(barcode_stage, "get_default_service", + lambda: _Service(result=found)) + product = {"product_name": "Amul Ghee 1L", "size": "1L"} + + _apply(product) + + assert product["barcode"] == "8901262010023" + + +def test_a_stage_may_still_correct_a_value_just_not_blank_it(): + """The merge rule is narrow on purpose. + + Overwriting a value with a DIFFERENT value is what correcting a field + means and stays allowed; only blanking a held value is refused. A rule + that froze every populated field would break legitimate enrichment. + """ + from app.services.enrichment.base import EnrichmentStage, StageOutcome + + class _Correcting(EnrichmentStage): + name = "test" + + @property + def enabled(self): + return True + + async def enrich_one(self, product, brand): + return StageOutcome(stage_name=self.name, + fields={"category": "Dairy", "barcode": None}) + + product = {"category": "General", "barcode": "8901262010016"} + asyncio.run(_Correcting().apply(product, "Amul")) + + assert product["category"] == "Dairy", "a real correction was refused" + assert product["barcode"] == "8901262010016", "a held value was blanked" + + +# --------------------------------------------------------------------------- +# 2. The gate +# --------------------------------------------------------------------------- +def _match(off_name, off_brand, off_size, our_brand, our_title, our_size): + candidate = BarcodeCandidate(barcode="8906021120418", source_name="OFF", + candidate_title=off_name, + candidate_brand=off_brand, + candidate_size=off_size) + return BarcodeLookupService._first_validated_match( + [candidate], our_brand, our_title, our_size, set()) + + +@pytest.mark.parametrize("off_name,our_title,size", [ + # Every one of these is a barcode currently stored on the wrong product. + ("Chicken Kabab/65 Masala", "Aachi Chicken Masala 50g", "50g"), + ("Chicken Curry Masala", "Aachi Chicken Masala 200g", "200g"), + ("Tata Tea Gold Care", "Tata Tea Gold 500g", "500g"), + ("MTR Chana Masala", "MTR Masala 300g", "300g"), + ("PEPER NOTEN", "Lion Dates Powder 100g", "100g"), +]) +def test_a_different_product_is_no_longer_accepted(off_name, our_title, size): + """Brand and size agree in every case; only the name separates them.""" + brand = our_title.split()[0] + + assert _match(off_name, brand, size, brand, our_title, size) is None + + +def test_the_right_product_is_still_accepted(): + result = _match("Aachi Mutton Masala", "Aachi", "100g", + "Aachi", "Aachi Mutton Masala 100g", "100g") + + assert result is not None + assert result.barcode + + +def test_both_directions_share_one_floor(): + """A barcode good enough to store must be good enough to read back. + + Two independently-declared copies that drifted apart would give exactly + that contradiction, so the forward and reverse paths import the same + setting rather than each keeping a constant. + """ + from app.infrastructure.settings import BARCODE_MIN_NAME_SIMILARITY as configured + from app.services.enrichment.barcode.service import BARCODE_MIN_NAME_SIMILARITY as forward + from app.services.nutrition_data_service import BARCODE_MIN_NAME_SIMILARITY as reverse + + assert forward == reverse == configured diff --git a/tests/test_nutrition_by_barcode.py b/tests/test_nutrition_by_barcode.py new file mode 100644 index 0000000..448ef6b --- /dev/null +++ b/tests/test_nutrition_by_barcode.py @@ -0,0 +1,245 @@ +"""Looking a product up by the barcode on its pack, rather than by its name. + +WHAT IS NEW HERE +---------------- +Every other Open Food Facts call in this project searches by brand and product +name and scores whatever comes back. That is why OFF-sourced rows in +`nutrition_facts` sit at match confidences as low as 0.32 - the module's own +`MIN_MATCH_CONFIDENCE`. A barcode is the identifier printed on the pack, so +`/api/v2/product/{code}` returns that product or nothing. + +WHY THERE IS STILL A GATE +------------------------- +The lookup is exact. The STORED BARCODE is not. Measured across all 95 barcoded +catalogue rows, a third of the records found this way described a different +product, because the barcode on our row was wrong. Every string in the tests +below is real - taken from that run, not invented - so if the gate is loosened +these fail with the actual products it would let through. + +No network: `requests.get` is stubbed at the boundary, as the existing nutrition +tests do. +""" +from __future__ import annotations + +import pytest + +from app.services import nutrition_data_service as nds +from app.services.enrichment.barcode.sources import open_food_facts as off + + +class _Resp: + def __init__(self, payload, status_code=200): + self._payload = payload + self.status_code = status_code + + def json(self): + return self._payload + + +def _product(name, brands, quantity, nutriments=None, code="8906021122290"): + return { + "code": code, + "product_name": name, + "brands": brands, + "quantity": quantity, + "countries_tags": ["en:india"], + "nutriments": nutriments if nutriments is not None else { + "energy-kcal_100g": 388.2, + "proteins_100g": 12.0, + "fat_100g": 15.0, + "carbohydrates_100g": 45.0, + }, + } + + +@pytest.fixture +def off_returns(monkeypatch): + """Stub the HTTP boundary. Returns a setter taking the JSON body.""" + box = {} + + def _get(url, **kwargs): + if "status" in box and box["status"] is None: + raise TimeoutError("simulated network timeout") + return _Resp(box.get("body", {"status": 0})) + + monkeypatch.setattr(off.requests, "get", _get) + return box + + +# --------------------------------------------------------------------------- +# The lookup itself +# --------------------------------------------------------------------------- +def test_a_known_barcode_returns_the_product(off_returns): + off_returns["body"] = {"status": 1, "product": _product( + "Aachi Mutton Masala", "Aachi", "100g")} + + product = off.fetch_product_by_barcode("8906021122290") + + assert product["product_name"] == "Aachi Mutton Masala" + + +def test_an_unknown_barcode_is_none_not_an_error(off_returns): + """OFF answers HTTP 200 with `status: 0` for a code it has never seen. + + Checking the HTTP code alone would treat that as a hit and hand the caller + an empty product. About a third of our barcodes land here, so this is the + ordinary path, not an exceptional one. + """ + off_returns["body"] = {"status": 0, "status_verbose": "product not found"} + + assert off.fetch_product_by_barcode("0000000000000") is None + + +def test_an_empty_barcode_makes_no_network_call(monkeypatch): + called = [] + monkeypatch.setattr(off.requests, "get", + lambda *a, **k: called.append(1) or _Resp({"status": 0})) + + assert off.fetch_product_by_barcode("") is None + assert off.fetch_product_by_barcode(" ") is None + assert not called, "a blank barcode should never reach the network" + + +def test_a_network_failure_returns_none_rather_than_raising(off_returns): + """One unreachable host must not take down a whole backfill run.""" + off_returns["status"] = None # makes the stub raise + + assert off.fetch_product_by_barcode("8906021122290") is None + + +# --------------------------------------------------------------------------- +# The gate - every string below came from the real catalogue +# --------------------------------------------------------------------------- +def test_the_right_product_is_accepted(off_returns): + off_returns["body"] = {"status": 1, "product": _product( + "Aachi Mutton Masala", "Aachi", "100g")} + + facts = nds.fetch_verified_nutrition_by_barcode( + "8906021122290", "Aachi", "Aachi Mutton Masala 100g", "100g") + + assert facts["data_status"] == "verified" + assert facts["data_source"] == nds.BARCODE_DATA_SOURCE + assert facts["source_ref"] == "8906021122290" + assert facts["calories_kcal"] == 388.2 + + +def test_a_different_product_under_the_same_barcode_is_refused(off_returns): + """The real failure this gate exists for. + + Barcode 8906021120418 is stored on our "Aachi Chicken Masala 50g", but Open + Food Facts holds it as "Chicken Kabab/65 Masala" - a different spice blend. + Brand and pack size both agree, so the name is the only thing that can tell + them apart, and it scores 0.538. + """ + off_returns["body"] = {"status": 1, "product": _product( + "Chicken Kabab/65 Masala", "Aachi", "50g")} + + facts = nds.fetch_verified_nutrition_by_barcode( + "8906021120418", "Aachi", "Aachi Chicken Masala 50g", "50g") + + assert facts["data_status"] == "unavailable" + assert "calories_kcal" not in facts + + +def test_a_near_miss_variant_is_refused(off_returns): + """"Tata Tea Gold" and "Tata Tea Gold Care" are different products. + + This one scores 0.761 - above matching.py's own 0.45 default, which is why + the barcode path sets its own, higher floor rather than reusing it. + """ + off_returns["body"] = {"status": 1, "product": _product( + "Tata Tea Gold Care", "Tata", "500g")} + + facts = nds.fetch_verified_nutrition_by_barcode( + "8901030873829", "Tata", "Tata Tea Gold 500g", "500g") + + assert facts["data_status"] == "unavailable" + + +def test_nonsense_is_refused(off_returns): + """Barcode 20086039 is stored on our Lion Dates Powder; OFF has it as a + Dutch confection. Scores 0.097.""" + off_returns["body"] = {"status": 1, "product": _product( + "PEPER NOTEN", "Favorina", "300 g")} + + facts = nds.fetch_verified_nutrition_by_barcode( + "20086039", "Lion Dates", "Lion Dates Powder 100g", "50g") + + assert facts["data_status"] == "unavailable" + + +def test_a_matching_record_with_no_usable_values_is_unavailable(off_returns): + """Our product, but Open Food Facts holds a near-empty record for it. + + Distinct from a mismatch and must not be reported as one: the barcode is + right, there is simply nothing to import. Real case - OFF has five + nutriment keys for Aachi Chicken Masala 100g and no values among them. + """ + off_returns["body"] = {"status": 1, "product": _product( + "Aachi Mutton Masala", "Aachi", "100g", nutriments={"nova_group": 4})} + + facts = nds.fetch_verified_nutrition_by_barcode( + "8906021122290", "Aachi", "Aachi Mutton Masala 100g", "100g") + + assert facts["data_status"] == "unavailable" + + +def test_the_floor_can_be_relaxed_per_call(off_returns): + """The threshold is a judgement, not a constant, so it is a parameter. + + "Aachi Biryani Masala" against our "Aachi Biryani Masala 50 g" is plainly + the same product and scores 0.716 - below the conservative default. The + backfill script exposes this as --min-similarity for exactly this reason. + """ + off_returns["body"] = {"status": 1, "product": _product( + "Aachi Biryani Masala", "Aachi", "50 g")} + args = ("8906021120272", "Aachi", "Aachi Biryani Masala 50 g", "50 g") + + assert nds.fetch_verified_nutrition_by_barcode(*args)["data_status"] == "unavailable" + relaxed = nds.fetch_verified_nutrition_by_barcode(*args, min_name_similarity=0.70) + assert relaxed["data_status"] == "verified" + + +# --------------------------------------------------------------------------- +# The two paths must stay interchangeable +# --------------------------------------------------------------------------- +def test_the_barcode_path_returns_the_same_shape_as_the_name_path(off_returns): + """Both feed the same `upsert_nutrition_facts`, so they cannot drift. + + If the barcode path ever grew a key the name path lacks, the writer would + silently drop it - the INSERT is built from a fixed column list. + """ + off_returns["body"] = {"status": 1, "product": _product( + "Aachi Mutton Masala", "Aachi", "100g")} + + by_barcode = nds.fetch_verified_nutrition_by_barcode( + "8906021122290", "Aachi", "Aachi Mutton Masala 100g", "100g") + + # `name_similarity` is diagnostic and unique to this path; everything else + # must exist on the name path too. + extra = set(by_barcode) - set(_name_path_keys()) + assert extra == {"name_similarity"}, f"unexpected new keys: {extra}" + + +def _name_path_keys(): + """The keys `fetch_verified_nutrition` produces on a successful match.""" + return { + "data_status", "data_source", "source_ref", "source_url", + "match_confidence", "serving_size_g", "serving_size_label", + "extended_nutrients", "per_serving", "ingredients_text", + "off_nutriscore", "allergens", "off_labels_tags", + "off_ingredients_analysis_tags", "off_categories_tags", "fetched_at", + } | set(nds._build_flat_fields({}, "100g")) + + +def test_a_disabled_open_facts_makes_no_call(monkeypatch): + called = [] + monkeypatch.setattr(off.requests, "get", + lambda *a, **k: called.append(1) or _Resp({"status": 0})) + monkeypatch.setattr(nds, "USE_OPEN_FACTS", False) + + facts = nds.fetch_verified_nutrition_by_barcode( + "8906021122290", "Aachi", "Aachi Mutton Masala 100g", "100g") + + assert facts["data_status"] == "unavailable" + assert not called