product generation with validation check

This commit is contained in:
sriram
2026-09-10 16:17:28 +05:30
parent 10b24c6348
commit d5a23f6456
19 changed files with 3381 additions and 121 deletions

View File

@@ -18,13 +18,17 @@ import logging
sys.path.append(str(Path(__file__).parent.parent))
from app.services.ollama_service import fetch_brand_catalog_with_gemini, fetch_brand_catalog_exhaustive, fetch_product_details
from app.services.image_search import find_all_image_urls, find_product_quantity_openfacts
from app.services.image_search import find_all_image_urls
from app.infrastructure.settings import DATA_DIR, USE_OLLAMA
from app.services.embeddings_service import embed_texts
from app.services.vector_store import ensure_brand_schema, upsert_brand_products, get_existing_product_image_id
from app.services.s3_service import s3_service
from app.services.brand_registry import resolve_parent_brand
from app.services import image_corroboration
from app.services import price_estimator
from app.services import product_grounding
from app.services.enrichment.barcode.sources import off_bulk
from app.services.product_validator import validate_catalog
from app.services.category_registry import detect_category_from_text, sanitize_category_language
# Configure logging
@@ -611,15 +615,43 @@ class ProductCatalogEngine:
# Step 2: Enhance each product with comprehensive image search
enhanced_products = []
# Parallel to enhanced_products: did an external source corroborate
# that each product exists? Collected here and handed to
# validate_catalog at the end - see the Step 3 comment for why this
# path had no validation gate at all until now.
grounded_flags: List[bool] = []
# Cap products by max_products
discovered_products = discovered_products[:max_products]
# Fetched ONCE for the whole brand, not once per product: this is a
# disk-cached whole-catalogue fetch (1-5 requests per brand), which is
# the entire reason it is used here instead of the per-product live
# search that looks like the natural fit. See product_grounding's
# docstring - that endpoint is rate-limited to 10 requests/minute and
# was returning 503 when measured.
try:
brand_corpus = off_bulk.fetch_brand_corpus(brand)
except Exception as e: # noqa: BLE001 - unreachable corpus is not a verdict
logger.warning("Open*Facts corpus unavailable for %s: %s", brand, e)
brand_corpus = None
if brand_corpus is not None:
logger.info("📚 Open*Facts corpus for %s: %d real products", brand, len(brand_corpus))
total_products_to_process = len(discovered_products)
for i, product in enumerate(discovered_products):
product_title = product.get('title', '')
logger.info(f"🔍 Processing product {i+1}/{total_products_to_process}: {product_title}")
# Does anything outside this process say this product exists?
# The LLM that produced `product_title` is a 1.5B local model and
# cannot be asked to check its own work.
grounding = product_grounding.ground_product(
brand, product_title, corpus=brand_corpus
)
if grounding.status == product_grounding.NOT_FOUND:
logger.warning("⚠️ Ungrounded: %r - %s", product_title, grounding.note())
# Strip brand prefix from title if present (avoids redundant
# "Cadbury Perk Cadbury Perk Crunch" style queries that confuse
# image search APIs).
@@ -670,7 +702,26 @@ class ProductCatalogEngine:
# Select exactly 20 best images
all_prioritized = prioritized_images + other_images
final_images = self._select_best_images(all_prioritized, product_title, brand, max_images=20)
# _select_best_images ORDERS; it does not decide whether any
# candidate is this product. Applied BEFORE the S3 upload below so a
# photo of somebody else is never copied into our own bucket, where
# its origin stops being visible at all.
#
# Nothing is dropped - the whole list is still uploaded and stored,
# so an operator can look. Only the promotion to `image_url` is
# withheld, and only when no candidate corroborates the product.
image_choice = image_corroboration.choose_primary(
final_images, product_title, brand or ""
)
final_images = list(image_choice.ordered)
primary_eligible = image_choice.primary is not None
if final_images and not primary_eligible:
logger.warning(
"No corroborated image for %r (%s) - storing candidates but "
"leaving image_url empty", product_title, image_choice.reason,
)
# Check if product already exists in DB to avoid duplicates
image_id_val = ""
try:
@@ -836,18 +887,19 @@ class ProductCatalogEngine:
if not _variant_size_price_pairs:
default_sizes = price_estimator.default_size_variants(category_value, product_title)
# Ground this in a real packaging size where possible:
# Open Food/Beauty/Products Facts reports an actual
# `quantity` field (e.g. "200 g", "1 l") for products it
# has on file, which is far more trustworthy than the
# category preset list (itself just a fallback for when
# *nothing* else is known). If we have one, swap it in for
# the closest preset rather than presenting a size that may
# not actually exist for this exact product.
try:
real_qty = find_product_quantity_openfacts(product_title, brand)
except Exception:
real_qty = None
# Ground this in a real packaging size where possible. The
# quantity comes from the brand corpus fetched once above,
# NOT from find_product_quantity_openfacts, which issues a
# live per-product query against an endpoint rate-limited to
# 10 requests/minute - see product_grounding's docstring.
#
# ONLY WHEN THE PRODUCT ITSELF IS CORROBORATED. A quantity
# borrowed from a product that merely scored well is worse
# than the preset it replaces: it makes a fabricated row look
# MORE real by dressing it in a size that genuinely exists.
# An ungrounded product keeps the honest category preset and
# is flagged for review instead.
real_qty = grounding.quantity if grounding.is_grounded else None
if real_qty and real_qty not in default_sizes:
default_sizes = [real_qty] + default_sizes[:2]
@@ -915,10 +967,16 @@ class ProductCatalogEngine:
# is only used as a last resort since small local models frequently
# hallucinate image links that don't actually resolve to an image.
all_image_urls = s3_uploaded_urls or final_images
primary_image = all_image_urls[0] if all_image_urls else (
# `primary_eligible` is the corroboration verdict computed above.
# S3 preserves candidate order (image_000, image_001, ...), so
# index 0 here is the same picture choose_primary judged.
primary_image = (all_image_urls[0] if (all_image_urls and primary_eligible) else None) or (
enriched_img if enriched_img and str(enriched_img).startswith('http') else None
)
if not primary_image and s3_service.enabled and image_id_val:
# Only reachable when there was no usable candidate at all. Guarded
# by `primary_eligible` too, because this constructs a URL to
# image_000 - the very image corroboration just declined to promote.
if not primary_image and primary_eligible and s3_service.enabled and image_id_val:
primary_image = s3_service.get_product_image_url(brand, image_id_val)
if not all_image_urls and primary_image:
all_image_urls = [primary_image]
@@ -958,8 +1016,36 @@ class ProductCatalogEngine:
}
enhanced_products.append(enhanced_product)
grounded_flags.append(grounding.is_grounded)
logger.info(f"✅ Enhanced {product_title}: {len(final_images)} images")
# Step 2.6: THE DETERMINISTIC VALIDATION GATE.
#
# This path did not have one. `validate_catalog` was called from the
# spreadsheet pipeline and from the Dagster assets, but never from
# here, so POST /api/catalog/generate wrote straight to the brand
# table with nothing between the language model and the database -
# while product_validator's own docstring claimed this call site
# already existed. That claim has been corrected along with this fix.
#
# Rows are annotated and kept, not dropped: `validate_catalog` returns
# "verified" and "needs_review" rows together, and A1 gave the brand
# tables somewhere to record which is which. Only rows scoring below
# the reject threshold are withheld.
enhanced_products, rejected, summary = validate_catalog(
enhanced_products, brand, grounded_flags=grounded_flags,
# These rows came out of a 1.5B language model, so "well formed"
# is not evidence of anything. A row nothing corroborates is kept
# and scored, but capped at needs_review rather than presented as
# verified.
require_grounding=True,
)
if rejected:
logger.warning(
"🚫 %d of %d generated product(s) failed validation for %s",
len(rejected), summary.get("total_evaluated", 0), brand,
)
# Step 3: Generate final catalog
catalog = {
'brand': brand,

View File

@@ -69,6 +69,7 @@ from app.api.routers.user_products import (
read_products_dataframe,
row_to_request,
)
from app.services import image_corroboration
from app.services import price_estimator
from app.services.brand_registry import (
BRAND_ALIASES,
@@ -587,8 +588,25 @@ def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, An
candidates, product_name, brand, max_images=10
)
if best:
row["image_urls"] = list(best)
row["image_url"] = best[0]
# `_select_best_images` ORDERS candidates; it does not judge
# whether any of them is this product. Its scoring awards points
# for the brand appearing in the URL, so for a brand that is
# also a personal name it actively rewarded the wrong photo -
# `Anil_Kapoor_2019.jpg` outscored everything and became
# `image_url`. choose_primary decides what may be promoted.
#
# The list is still stored in full. A product whose only
# candidates are uncorroborated keeps them for review; it just
# does not get one of them presented as fact.
choice = image_corroboration.choose_primary(
best, product_name, brand or ""
)
row["image_urls"] = list(choice.ordered)
row["image_url"] = choice.primary
if choice.primary is None:
row.setdefault("_notes", []).append(
f"no primary image: {choice.reason}"
)
except Exception as exc: # noqa: BLE001 - an image is not worth the row
# `warning`, not `debug`: at the default log level a debug line is
# invisible, so a row that silently lost its images looked identical to
@@ -776,6 +794,14 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
"highlights": list(row.get("highlights") or []),
"nutrients": list(row.get("nutrients") or []),
"search_query": search_query,
# Stage 10's verdict. validate_catalog() has always annotated these
# three onto every row it kept; this dict dropped all three, so the
# confidence the gate computed died here and every stored row looked
# equally trustworthy. Same rule as the block above: computed but not
# projected is computed for nothing.
"validation_status": row.get("validation_status"),
"confidence_score": row.get("confidence_score"),
"validation_issues": list(row.get("validation_issues") or []),
"field_sources": dict(row.get("field_sources") or {}),
}

View File

@@ -105,7 +105,7 @@ from app.infrastructure.settings import (
BRAND_DISCOVERY_USE_LLM,
BRAND_DISCOVERY_USE_OFF,
)
from app.services import active_brands, ollama_service
from app.services import active_brands, ollama_service, retail_presence
from app.services.brand_registry import (
get_fssai_license,
get_known_sub_brands,
@@ -119,6 +119,7 @@ from app.services.category_units import (
parse_unit,
)
from app.services.enrichment.barcode.sources import off_bulk
from app.services.product_grounding import GROUNDING_SIMILARITY_FLOOR
from app.services.vector_store import _sanitize_name, get_products_by_brand
logger = logging.getLogger(__name__)
@@ -594,13 +595,51 @@ def _registry_terms(brand: str) -> List[str]:
return terms
def _evidence_for(normalised: str, *, off_keys: Dict[str, Any],
registry_terms: Sequence[str], catalog_keys: Dict[str, str]) -> Optional[str]:
"""What corroborates this product, cheapest source first.
def _matches_any_key(normalised: str, keys: Dict[str, Any]) -> bool:
"""Is `normalised` one of `keys`, allowing for spelling drift?
Three tiers, none of which cost a network call at this point: the OFF index
was built during the sweep, the registry terms are a Python constant, and
the catalog index was read once.
EXACT EQUALITY WAS A BUG. These keys come from Open Food Facts and from
our own stored rows, and neither spells a product the way discovery does.
Measured: our "Anil Samba Rava" normalises to `samba rava`, while Open
Food Facts holds the same product as "SAMBA RAVVA" -> `samba ravva`. Not
equal, so a real product with a real barcode was reported as having no
corroboration at all.
The similarity floor is the one `product_grounding` measured against this
exact corpus - see that module for why it is not the 0.78 used for barcode
attachment, and why `samba rava`/`samba ravva` (0.681) is the case that
fixes the number.
"""
if not normalised:
return False
if normalised in keys:
return True
for key in keys:
if not key:
continue
if off_bulk.symmetric_similarity(key, normalised) >= GROUNDING_SIMILARITY_FLOOR:
return True
return False
def _evidence_for(normalised: str, *, off_keys: Dict[str, Any],
registry_terms: Sequence[str], catalog_keys: Dict[str, str],
retail_keys: Optional[Dict[str, Any]] = None) -> Optional[str]:
"""What corroborates this product, strongest source first.
Four tiers now. None costs a network call at this point: the OFF index was
built during the sweep, the registry terms are a Python constant, the
catalog index was read once, and the retail index is read from a cache the
backfill script populated offline - see retail_presence for why a live
query must never happen inside ingestion.
`retail` sits directly below `openfacts` and above `catalog` and
`registry`, because it is the only tier that is BOTH external and current.
`catalog` is our own prior output, which is circular if that output was
itself ungrounded, and `registry` is a hardcoded Python list. For a
non-food brand those two were the only tiers available at all, which is
why a toothpaste catalogue could be entirely language-model output and
still look corroborated.
Deliberately NOT a tier: "an image search returned a URL naming this
product". `catalog_engine._select_best_images` documents that CDN filenames
@@ -610,9 +649,11 @@ def _evidence_for(normalised: str, *, off_keys: Dict[str, Any],
"""
if not normalised:
return None
if normalised in off_keys:
if _matches_any_key(normalised, off_keys):
return "openfacts"
if normalised in catalog_keys:
if retail_keys and normalised in retail_keys:
return "retail"
if _matches_any_key(normalised, catalog_keys):
return "catalog"
tokens = set(normalised.split())
if tokens & set(registry_terms):
@@ -632,6 +673,25 @@ def _score(sources: Sequence[str], evidence: Optional[str]) -> float:
return 1.0
if has_off:
return 0.9
# An Open Food Facts row corroborates this product, but under a DIFFERENT
# SPELLING, so the merge did not fold the two together and this candidate
# never acquired the "off" source.
#
# Unreachable while `_evidence_for` matched keys by exact equality - an
# exact match always merged. It became reachable when that matching was
# relaxed to handle real spelling drift ("Anil Samba Rava" against Open
# Food Facts' "SAMBA RAVVA"), and without this branch such a row fell all
# the way through to 0.25 - scored as though nothing corroborated it, when
# a real database record does.
if evidence == "openfacts":
return 0.85
# A real shop is currently listing this exact product at this pack size.
# Scored above `catalog` (our own prior output, which is circular when
# that output was itself ungrounded) and far above `registry` (a hardcoded
# list), because it is the only tier that is both external and current.
# For a non-food brand it is usually the ONLY external tier available.
if evidence == "retail":
return 0.85
if evidence == "catalog":
return 0.7
if evidence == "registry":
@@ -685,7 +745,9 @@ def _resolve_sizes(candidate: Dict[str, Any], title: str, category: str,
def _build_product(brand: str, candidate: Dict[str, Any], *, max_sizes: int,
catalog_keys: Dict[str, str], off_keys: Dict[str, Any],
registry_terms: Sequence[str],
fssai: Optional[str]) -> Optional[DiscoveredProduct]:
fssai: Optional[str],
retail_keys: Optional[Dict[str, Any]] = None,
) -> Optional[DiscoveredProduct]:
raw_title = _canonicalise_title_size((candidate.get("title") or "").strip())
if not raw_title:
return None
@@ -769,7 +831,8 @@ def _build_product(brand: str, candidate: Dict[str, Any], *, max_sizes: int,
shape = {"title": title, "category": category_hint, "size_variants": sizes,
"description": description}
evidence = _evidence_for(normalised, off_keys=off_keys,
registry_terms=registry_terms, catalog_keys=catalog_keys)
registry_terms=registry_terms, catalog_keys=catalog_keys,
retail_keys=retail_keys)
sources = list(dict.fromkeys(candidate.get("sources") or [candidate.get("source")]))
sources = [s for s in sources if s]
@@ -835,6 +898,7 @@ def discover_brand_products(
if key:
off_keys.setdefault(key, candidate)
llm_candidates: List[Dict[str, Any]] = []
if use_llm:
llm_candidates = _from_llm(brand, deadline=deadline, budget=max_products * 2)
@@ -955,6 +1019,32 @@ def discover_brand_products(
chosen = _canonicalise_title_size(canonical_titles[key])
entry["title"] = off_bulk.strip_sizes(chosen).strip() or chosen
# What a real shop is currently listing, read from the cache that
# `scripts/backfill_retail_presence.py` fills offline. CACHE ONLY - never a
# live query. Discovery is interactive, and one lookup costs 2.6-5.7s
# against a provider that 403s under load, so a live sweep here would
# either hang the preview or get the whole run throttled - and a throttled
# miss is indistinguishable from "nobody sells this", which is the one
# input a corroboration tier must never be fed.
#
# Built from `merged`, so it covers the LLM's candidates too. That is the
# point: an Open Food Facts row needs no second opinion, and for a NON-FOOD
# brand there is no Open Food Facts row at all - `off_bulk` queries the food
# database alone, so a toothpaste's only possible external corroboration is
# this one.
retail_keys: Dict[str, Any] = {}
for entry in merged:
title = entry.get("title") or ""
key = _normalise_title(brand, title)
if not key or key in retail_keys:
continue
evidence = retail_presence.check_listing(brand, title, "", live=False)
if evidence.is_found:
retail_keys[key] = evidence
if retail_keys:
logger.info("🛒 %d of %d discovered products are currently listed by a retailer",
len(retail_keys), len(merged))
products: List[DiscoveredProduct] = []
dropped = 0
for candidate in merged:
@@ -962,6 +1052,7 @@ def discover_brand_products(
brand, candidate, max_sizes=max_sizes_per_product,
catalog_keys=catalog_keys, off_keys=off_keys,
registry_terms=registry_terms, fssai=fssai,
retail_keys=retail_keys,
)
if product is None:
continue

View File

@@ -104,6 +104,13 @@ EXPORT_COLUMNS = (
"barcode_lookup_status", "barcode_last_updated",
"gst_percent", "tax_amount", "hsn_gst_needs_review",
"highlights", "nutrients", "search_query", "field_sources",
# The validation verdict. Listed here for the same reason the barcode keys
# are: an export that strips them turns a re-seed into a silent downgrade.
# The upsert assigns these three plainly rather than COALESCEing them, so a
# seed file that has lost them clears the verdict on every row it restores -
# which is right for a writer that never validated, and wrong for this one,
# whose whole job is to echo back rows that already were.
"validation_status", "confidence_score", "validation_issues",
"nutrition_score", "health_score", "nutrients_per_100g",
)

View File

@@ -4,10 +4,16 @@ catalog rows, running each stage's per-row work concurrently (bounded by
`max_concurrency`) and NEVER letting one row's failure affect any other
row or stage.
`catalog_engine.py` calls `run_default_pipeline()` once, between the
deterministic-validation step (product_validator.py) and catalog assembly
- see that file's "Step 2.6" for the call site and
docs/BARCODE_ENRICHMENT.md for the full pipeline-position rationale.
`store_catalog_pipeline.stages_8_9_enrichment()` builds an EnrichmentPipeline
and runs it between SKU resolution and the validation gate; see
docs/BARCODE_ENRICHMENT.md for the pipeline-position rationale.
This docstring used to say `catalog_engine.py` called `run_default_pipeline()`
at a "Step 2.6". It never did - that module does not import this one at all,
so the brand-name generation path gets no barcode, HSN/GST or content
enrichment. Step 2.6 there is now the validation gate only. Wiring enrichment
into that path is a real and separate piece of work; do not read this comment
as saying it is already done.
"""
from __future__ import annotations

View File

@@ -0,0 +1,477 @@
"""
Does this image URL actually depict THIS product?
WHY THIS FILE EXISTS
--------------------
`image_search.validate_image_url_live` answers "does this URL serve real image
bytes", which is a liveness question. Nothing answered the relevance question,
so a press photo of a person named like the brand passed every check the
ingestion path had: it is a real image, above the byte floor, on a reputable
host. That is how the live catalogue came to illustrate "Anil Samba Rava" with
a photograph of the actor Anil Kapoor.
The logic here is ported from `scripts/repair_brand_images.py`, which already
knew how to reject that image class - its comments name the failures it was
written for ("Aachi Kulambu Mix" returning a press photo of a politician,
"MTR Dosa Mix" returning an anatomy plate). It lived in a script that imports
settings, brand_registry, produce_reference and four private `vector_store`
symbols including `_connect`, so no ingestion path could import it. Moving the
pure predicates here is what lets stage 6 and the catalog engine use them.
THE ONE SEMANTIC CHANGE MADE DURING THE MOVE
--------------------------------------------
The original corroborated a URL against `words + brand_tokens`:
return any(w in lowered for w in words + _brand_tokens(brand))
That works for "Aachi Kulambu Mix" precisely because *Aachi is not a human
name*. Anil is. For product "Anil Samba Rava" under brand "Anil" the token list
contains "anil" twice over, and `Anil_Kapoor_2019.jpg` contains "anil", so the
gate returned True and the celebrity photo was corroborated by the very token
that made it wrong.
`names_product` here requires a DISTINCTIVE token instead - the title minus the
brand minus the pack size, which is the same set
`catalog_engine._select_best_images` already computes to rank candidates:
"Anil Samba Rava" -> {samba, rava} -> rejects Anil_Kapoor_2019.jpg
"Anil Wheat Vermicelli"-> {wheat, vermicelli}
Brand tokens remain, but only as an explicit fallback for titles that have no
distinctive words at all ("Amul 1kg"), and a match found that way is reported
as `via_brand_only` so the caller can decline to treat it as corroboration.
WHAT THIS MODULE MAY AND MAY NOT DECIDE
---------------------------------------
It decides which image is shown for a product. It never decides whether the
PRODUCT is real. `brand_discovery._evidence_for` deliberately refuses to treat
image naming as product evidence, because CDN filenames are frequently opaque
hashes and absence of a naming URL is weak evidence of absence. That reasoning
is right and this module does not disturb it: images gate images, never
products.
Dependencies are stdlib plus `requests` on purpose. `catalog_engine` imports
this module, and `repair_brand_images` already reaches back into
`catalog_engine._select_best_images`, so a module-scope import of
`catalog_engine` here would close a cycle.
"""
from __future__ import annotations
import logging
import re
import sqlite3
import threading
import time
from contextlib import closing
from dataclasses import dataclass
from pathlib import Path
from typing import Iterable, List, Optional, Sequence
from urllib.parse import urlparse
import requests
logger = logging.getLogger(__name__)
_BROWSER_UA = "nearle-catalogue/1.0 (product image corroboration)"
# "own" and "products" are the BUCKET's name, not a brand's. Left in, the token
# "products" corroborates any URL containing /images/products/ or
# /cdn/shop/products/ - which is every Open*Facts and every Shopify path - so
# `names_product` waved through 40+ images that named nothing about the item.
# That is how openbeautyfacts cosmetics photos became the stored image for
# Banana, Orange, Papaya, Guava, Lemon and twenty more.
BUCKET_TOKENS = frozenset({"own", "products", "product"})
# A trailing pack size on a product name. Used to collapse "X 100g"/"X 500g"
# onto one search, and to keep size digits out of the distinctive token set.
SIZE_TAIL = re.compile(
r"\s+\d+(?:\.\d+)?\s*(?:g|gm|gms|kg|ml|l|ltr|litre|liter|pcs|pc|n|no|nos)\b\.?\s*$",
re.IGNORECASE,
)
# A bare quantity token, e.g. "500g" or "2l". Mirrors the filter in
# catalog_engine._select_best_images so the two agree on what a size looks like.
_SIZE_TOKEN = re.compile(r"\d+(?:kg|g|gm|gms|ml|l|ltr|pcs|n)?", re.IGNORECASE)
# The barcode embedded in an Open*Facts image path, e.g.
# /images/products/890/604/215/0067/front_en.4.400.jpg
_OFF_BARCODE = re.compile(r"/images/products/((?:\d+/)+)")
# Open*Facts image hosts and the API host that can identify a barcode for each.
#
# The original checked `if "openfoodfacts.org" not in url: return True`, so
# every openbeautyfacts and openproductsfacts image skipped the cross-check
# entirely - exactly the non-food case, and exactly the two sibling databases
# `image_search.OPEN_FACTS_HOSTS` queries. A food product got the check and a
# toothpaste did not.
_OFF_API_HOSTS = {
"openfoodfacts": "world.openfoodfacts.org",
"openbeautyfacts": "world.openbeautyfacts.org",
"openproductsfacts": "world.openproductsfacts.org",
}
# Hosts whose image paths are content-hashed or barcode-keyed, so a filename
# that fails to name the product says nothing about the photo. These are the
# hosts catalog_engine's own comment is about when it refuses to drop
# uncorroborated candidates.
OPAQUE_PATH_DOMAINS = (
"bbassets.com", "bigbasket.com", "flixcart.com", "flipkart.com",
"media-amazon.com", "amazon.in", "amazon.com", "jiomart.com",
"zeptonow.com", "blinkit.com", "grofers.com", "cloudinary.com",
"shopifycdn.com", "cdn.shopify.com", "akamaized.net", "cloudfront.net",
"openfoodfacts.org", "openbeautyfacts.org", "openproductsfacts.org",
)
# Hosts whose filenames are human-authored and descriptive. A filename that
# fails to name the product here is real evidence that the photo is of
# something else - and this is the host family the celebrity photo came from.
DESCRIPTIVE_FILENAME_DOMAINS = (
"wikimedia.org", "wikipedia.org", "wikimedia.commons",
)
# Two or more Capitalised words joined by underscores, optionally with a year:
# the Wikimedia Commons house style for a photograph OF A PERSON
# ("Anil_Kapoor_2019.jpg"). Deliberately used only to DEMOTE, never to reject -
# "Britannia_Good_Day.jpg" has the identical shape and is a perfectly good
# product photo, so this signal orders candidates and is not allowed to
# eliminate one.
_PERSON_FILENAME = re.compile(
r"^[A-Z][a-z]+(?:_[A-Z][a-z]+)+(?:_\d{4})?[^/]*\.(?:jpg|jpeg|png|webp)$"
)
_PERSON_CONTEXT = re.compile(r"_at_|_in_\d{4}|portrait|headshot", re.IGNORECASE)
# ---------------------------------------------------------------------------
# Tokens
# ---------------------------------------------------------------------------
def brand_tokens(brand: str) -> List[str]:
"""Distinctive words of a brand name, bucket words removed."""
return [
w for w in re.split(r"[^a-z0-9]+", (brand or "").lower())
if len(w) > 2 and w not in BUCKET_TOKENS
]
def distinctive_tokens(product_name: str, brand: str) -> List[str]:
"""Words that separate THIS product from its brand-mates.
The brand is removed on purpose: every Britannia URL contains "britannia",
so it separates nothing - what tells Marie Gold from Good Day is
"marie"/"gold" vs "good"/"day". Pack sizes and short filler words go for
the same reason. This mirrors catalog_engine._select_best_images so the
ranking and the gate can never disagree about what identifies a product.
"""
brand_words = {w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) if w}
out: List[str] = []
for word in re.split(r"[^a-z0-9]+", (product_name or "").lower()):
if (
len(word) > 2
and word not in brand_words
and word not in BUCKET_TOKENS
and word not in out
and not _SIZE_TOKEN.fullmatch(word)
):
out.append(word)
return out
def search_key(product_name: str, brand: str) -> tuple:
"""Collapse a trailing pack size so sizes of one product share a lookup.
Sharing across sizes is correct rather than merely cheap: it is the same
product in a different pack, and stage 4's size explosion produces exactly
these rows from a single source product.
"""
base = SIZE_TAIL.sub("", product_name or "").strip()
return ((brand or "").lower(), (base or product_name or "").lower())
# ---------------------------------------------------------------------------
# Corroboration
# ---------------------------------------------------------------------------
@dataclass(frozen=True)
class Corroboration:
"""Why a URL was or was not accepted as depicting this product."""
corroborated: bool
reason: str
via_brand_only: bool = False
def corroborate(url: str, product_name: str, brand: str) -> Corroboration:
"""Does `url` name this product?
Distinctive tokens first. Brand tokens are consulted only when the title
has no distinctive words of its own, and a match found that way is flagged
`via_brand_only` - it is the weakest possible signal, and for a brand that
is also a personal name it is the signal that produced the defect this
module exists for.
"""
lowered = (url or "").lower()
if not lowered:
return Corroboration(False, "empty url")
distinctive = distinctive_tokens(product_name, brand)
if distinctive:
hit = next((t for t in distinctive if t in lowered), None)
if hit:
return Corroboration(True, f"url names {hit!r}")
return Corroboration(
False,
"url names none of " + ", ".join(repr(t) for t in distinctive[:4]),
)
# No distinctive words at all (e.g. "Amul 1kg"). Fall back to the brand,
# and say so, so the caller can decide how much that is worth.
tokens = brand_tokens(brand)
hit = next((t for t in tokens if t in lowered), None)
if hit:
return Corroboration(True, f"url names brand {hit!r}", via_brand_only=True)
return Corroboration(False, "url names neither the product nor the brand")
def names_product(url: str, product_name: str, brand: str) -> bool:
"""Boolean form of `corroborate`, for callers that only need the verdict."""
return corroborate(url, product_name, brand).corroborated
def looks_like_person_photo(url: str) -> bool:
"""True when the FILENAME has the shape of a photograph of a person.
A demotion signal only - see `_PERSON_FILENAME` for why this must never
reject on its own.
"""
name = urlparse(url or "").path.rsplit("/", 1)[-1]
if not name:
return False
return bool(_PERSON_FILENAME.match(name)) or bool(_PERSON_CONTEXT.search(name))
def has_opaque_path(url: str) -> bool:
lowered = (url or "").lower()
return any(d in lowered for d in OPAQUE_PATH_DOMAINS)
def has_descriptive_filename(url: str) -> bool:
lowered = (url or "").lower()
return any(d in lowered for d in DESCRIPTIVE_FILENAME_DOMAINS)
# ---------------------------------------------------------------------------
# Open*Facts barcode cross-check
# ---------------------------------------------------------------------------
# The barcode is embedded in the image path, so the product is one cheap lookup
# away. If the Open*Facts record does not mention our brand, the image is
# somebody else's product. Nothing else can catch this: the path is opaque
# digits, so no amount of filename matching would help. OFF matches on NAME,
# not brand - asked for "Aachi Pickles" it returned the front-of-pack photo for
# a French "Ducros Green Pitted Olives", which validated perfectly happily
# because it IS a real image.
_DB_PATH = Path("data") / "cache" / "image_corroboration.db"
_lock = threading.Lock()
_initialized = False
_CACHE_TTL_SECONDS = 30 * 24 * 3600
def _connect() -> sqlite3.Connection:
_DB_PATH.parent.mkdir(parents=True, exist_ok=True)
conn = sqlite3.connect(str(_DB_PATH), timeout=10)
conn.execute("PRAGMA journal_mode=WAL")
return conn
def _ensure_schema(conn: sqlite3.Connection) -> None:
global _initialized
if _initialized:
return
conn.execute(
"""
CREATE TABLE IF NOT EXISTS openfacts_brand_check (
cache_key TEXT PRIMARY KEY,
verdict INTEGER NOT NULL,
created_at REAL NOT NULL
)
"""
)
conn.commit()
_initialized = True
def _cache_get(key: str) -> Optional[bool]:
# `closing`, because sqlite3's own context manager commits the transaction
# and leaves the connection open. Leaked connections are finalized at GC,
# which surfaces as an unraisable exception - and pytest.ini turns warnings
# into errors, so a leak here fails the suite from an unrelated test.
try:
with _lock, closing(_connect()) as conn:
_ensure_schema(conn)
row = conn.execute(
"SELECT verdict, created_at FROM openfacts_brand_check WHERE cache_key = ?",
(key,),
).fetchone()
except Exception as e: # noqa: BLE001 - a cache failure is a cache miss
logger.debug("image corroboration cache read failed for %s: %s", key, e)
return None
if not row:
return None
verdict, created_at = row
if (time.time() - created_at) > _CACHE_TTL_SECONDS:
return None
return bool(verdict)
def _cache_set(key: str, verdict: bool) -> None:
try:
with _lock, closing(_connect()) as conn:
_ensure_schema(conn)
conn.execute(
"""
INSERT INTO openfacts_brand_check (cache_key, verdict, created_at)
VALUES (?, ?, ?)
ON CONFLICT(cache_key) DO UPDATE SET
verdict = excluded.verdict,
created_at = excluded.created_at
""",
(key, 1 if verdict else 0, time.time()),
)
conn.commit()
except Exception as e: # noqa: BLE001 - best effort only
logger.debug("image corroboration cache write failed for %s: %s", key, e)
def _api_host_for(url: str) -> Optional[str]:
lowered = (url or "").lower()
for marker, host in _OFF_API_HOSTS.items():
if marker in lowered:
return host
return None
def openfacts_product_matches_brand(url: str, brand: str, *, timeout: int = 10) -> bool:
"""False ONLY when Open*Facts positively says this barcode is another brand.
Every other outcome - not an Open*Facts URL, no barcode in the path, no
brand tokens, a network failure, an unparseable response - returns True.
A lookup failure must never reject a good image; that asymmetry is the
whole point, and it is the same discipline the realtime checks follow.
"""
api_host = _api_host_for(url)
if not api_host:
return True
match = _OFF_BARCODE.search(url or "")
if not match:
return True
barcode = match.group(1).replace("/", "")
tokens = brand_tokens(brand)
if not tokens:
return True
key = f"{api_host}:{barcode}:{(brand or '').lower()}"
cached = _cache_get(key)
if cached is not None:
return cached
verdict = True
try:
resp = requests.get(
f"https://{api_host}/api/v2/product/{barcode}.json",
params={"fields": "brands,product_name"},
timeout=timeout,
headers={"User-Agent": _BROWSER_UA},
)
if resp.ok:
product = (resp.json() or {}).get("product") or {}
haystack = (
f"{product.get('brands') or ''} {product.get('product_name') or ''}"
).lower()
if haystack.strip():
verdict = any(t in haystack for t in tokens)
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
return True
_cache_set(key, verdict)
return verdict
# ---------------------------------------------------------------------------
# Which candidate may become the product's primary image
# ---------------------------------------------------------------------------
@dataclass
class PrimaryChoice:
"""The outcome of choosing a product's `image_url` from its candidates."""
primary: Optional[str]
ordered: List[str]
reason: str
def choose_primary(
urls: Sequence[str],
product_name: str,
brand: str,
*,
check_openfacts: bool = True,
) -> PrimaryChoice:
"""Pick the URL that may become `image_url`, and order the rest.
Three tiers, because the two existing opinions about failing closed are
both right and the disagreement is domain-scoped:
1. The URL names the product. Eligible.
2. The URL cannot name anything (a hashed retailer CDN path, an
Open*Facts barcode path) - and for Open*Facts, the barcode's own
record agrees about the brand. Eligible: this is the case
catalog_engine's comment protects, where dropping uncorroborated
candidates would leave real products with no image at all.
3. The URL could have named the product and did not - a human-authored
Commons filename, say. Kept in `image_urls`, never promoted.
Nothing eligible means `primary is None`. A blank image renders as the
brand monogram, which is honest; another company's product is not, and it
stays invisible until somebody recognises the photo.
"""
tier1: List[str] = []
tier2: List[str] = []
tier3: List[str] = []
for url in urls:
if not url or not str(url).startswith("http"):
continue
verdict = corroborate(url, product_name, brand)
if verdict.corroborated and not verdict.via_brand_only:
tier1.append(url)
elif verdict.via_brand_only and not looks_like_person_photo(url):
# The title has NO distinctive words of its own - "Godrej 50ml",
# "Lion Dates 100g", "Amul 1kg". There is no product identity to
# match on, so a brand-token match is the best signal that exists
# and withholding the image gains nothing: measured over the seed
# catalogues, treating these as ineligible accounted for 14 of 45
# withheld primaries, every one of them a brand's own product page.
#
# Person-shaped filenames are the exception, because a brand token
# matching a personal name is the exact defect this module exists
# for and a title with no distinctive words cannot contradict it.
tier2.append(url)
elif has_opaque_path(url) and not has_descriptive_filename(url):
if check_openfacts and not openfacts_product_matches_brand(url, brand):
tier3.append(url)
else:
tier2.append(url)
else:
tier3.append(url)
# Person-shaped filenames sink within their tier. They are never dropped,
# so a product whose only images look like this still keeps them in
# `image_urls` for a human to review.
tier3.sort(key=looks_like_person_photo)
ordered = tier1 + tier2 + tier3
if tier1:
return PrimaryChoice(tier1[0], ordered, "url names the product")
if tier2:
return PrimaryChoice(tier2[0], ordered, "opaque path on a known image host")
return PrimaryChoice(
None,
ordered,
"no candidate names this product; primary withheld rather than guessed",
)

View File

@@ -0,0 +1,236 @@
"""
Does an external source corroborate that this product exists?
WHY THIS FILE EXISTS
--------------------
The catalogue contained "Anil Wheat Vermicelli 12g". Nobody sells a 12 g
vermicelli pack. It was generated by a 1.5B local model, given a category, a
price band and an internal SKU, and stored - because every check in the system
asks whether a row is WELL FORMED, and a well-formed fiction passes all of
them. `product_validator`'s own docstring says so: it was built to catch
malformed rows, not false ones.
Meanwhile the real answer was already on disk. `off_bulk.fetch_brand_corpus`
caches a brand's entire Open Food Facts catalogue, and
`data/cache/off_brand_corpus/anil.json` holds Anil's eight real products -
Roasted Short Vermicelli at 180 g and 450 g, SAMBA RAVVA at 500 g. Nothing
consulted it at generation time.
WHAT THIS MODULE DOES NOT DO: SUBSTITUTE A SIZE
-----------------------------------------------
The obvious-looking fix - "look up the real quantity and swap it in" - is
wrong here, and measurably so. Matching "Anil Wheat Vermicelli" against the
corpus scores 0.460 against "Roasted Short Vermicelli", far below any usable
floor. That is the corpus saying THIS PRODUCT DOES NOT EXIST, not saying it is
450 g. Swapping in a corpus quantity would replace one fabrication with a
better-dressed one: a product that still does not exist, now wearing a size
that does.
So a miss reports `not_found` and the caller records that the row is
ungrounded. Removing the row is the validation gate's job, not this module's.
WHY NOT `find_product_quantity_openfacts`
-----------------------------------------
`image_search.find_product_quantity_openfacts` looks like the right oracle and
is not. It queries `/cgi/search.pl` LIVE, once per product, looping three hosts
at a 12-second timeout - up to 36 s per product against an endpoint Open Food
Facts rate-limits to 10 requests/minute. Two hundred products is twenty-plus
minutes and near-certain throttling, and a throttled miss is indistinguishable
from a real one, which is exactly the input a gate must never be fed. Measured
2026-09-10: that endpoint returned 503 on OFF and 500 on Open Beauty Facts.
`fetch_brand_corpus` costs one to five requests per BRAND, is disk-cached,
retry-wrapped and paced, and is already populated for 50+ brands.
THREE STATES, NOT TWO
---------------------
`grounded` / `not_found` / `unknown`. A corpus that could not be fetched must
report `unknown` and never `not_found`, or a network failure would silently
demote a whole brand's real products. Every lookup path in this codebase that
feeds a gate follows the same discipline.
"""
from __future__ import annotations
import logging
from dataclasses import dataclass
from typing import Any, Dict, List, Optional, Sequence
from app.services.enrichment.barcode.sources import off_bulk
logger = logging.getLogger(__name__)
# How similar a corpus name must be to count as the same product.
#
# NOT the 0.78 used for barcode attachment. Measured against the real Anil
# corpus on 2026-09-10, best match per target:
#
# "wheat vermicelli" vs "Rice Vermicelli" -> 0.610 (DIFFERENT product)
# "samba rava" vs "SAMBA RAVVA" -> 0.681 (SAME product)
#
# A 0.78 floor rejects a genuine product over a one-letter spelling difference
# in Open Food Facts' own record, so it cannot be used here. But note how
# little room that leaves: 0.071 between a true match and a false one, and the
# floor below sits about 0.01 above the false one. THAT MARGIN IS TOO THIN TO
# REST A DECISION ON, which is why `_material_conflict` exists - the real
# discriminator between those two names is that wheat is not rice, not that
# 0.610 is not 0.681. The floor is the coarse filter; the material check is
# what actually separates the reported defect from the real product beside it.
#
# It is also deliberately a soft signal in both directions: nothing here
# deletes a row, it only decides whether the row can be called corroborated,
# so the cost of being wrong is a review flag and not data loss.
GROUNDING_SIMILARITY_FLOOR = 0.62
# Base materials that make two otherwise similarly-named products different
# products. "Wheat Vermicelli" and "Rice Vermicelli" share a head noun and
# score 0.610 against each other; no string-similarity threshold separates
# them reliably, because the thing that differs is one word carrying all the
# meaning.
#
# The rule is deliberately narrow: a conflict is declared ONLY when BOTH names
# name a material and the sets are disjoint. "Wheat Vermicelli" against a bare
# "Vermicelli" is not a conflict - that is plausibly the same product line
# described at two levels of detail, and treating it as a conflict would lose
# real corroboration.
_MATERIALS = frozenset({
"wheat", "rice", "ragi", "maida", "corn", "millet", "bajra", "jowar",
"atta", "besan", "soya", "oat", "oats", "barley", "quinoa", "almond",
"cashew", "coconut", "groundnut", "peanut", "sesame", "mustard",
"sunflower", "olive", "ghee", "butter",
})
GROUNDED = "grounded"
NOT_FOUND = "not_found"
UNKNOWN = "unknown"
# Per-process memo of brand -> corpus. fetch_brand_corpus is itself disk-cached,
# so this only avoids re-reading and re-parsing the same JSON once per product
# in a 200-row run.
_corpus_memo: Dict[str, Optional[List[Dict[str, Any]]]] = {}
@dataclass(frozen=True)
class Grounding:
"""What an external source says about one product."""
status: str
matched_name: Optional[str] = None
quantity: Optional[str] = None
similarity: float = 0.0
source: Optional[str] = None
@property
def is_grounded(self) -> bool:
return self.status == GROUNDED
def note(self) -> str:
if self.status == GROUNDED:
return (
f"corroborated by {self.source} as {self.matched_name!r} "
f"(similarity {self.similarity:.2f})"
)
if self.status == NOT_FOUND:
return f"no {self.source or 'Open Food Facts'} product matches this name"
return "could not check whether this product exists"
def _material_conflict(candidate: str, target: str) -> bool:
"""True when two names name DIFFERENT base materials.
See `_MATERIALS`. Only fires when both sides name one, so a more specific
name still corroborates a less specific one.
"""
left = {w for w in candidate.lower().split() if w in _MATERIALS}
right = {w for w in target.lower().split() if w in _MATERIALS}
return bool(left and right and not (left & right))
def _corpus_for(brand: str) -> Optional[List[Dict[str, Any]]]:
"""The brand's cached Open*Facts corpus, or None when it is unavailable.
None means "we do not know", and is what makes the difference between
`not_found` and `unknown` downstream. An EMPTY list is a real answer - the
brand was looked up and has nothing on file.
"""
key = (brand or "").strip().lower()
if not key:
return None
if key in _corpus_memo:
return _corpus_memo[key]
try:
corpus = off_bulk.fetch_brand_corpus(brand)
except Exception as e: # noqa: BLE001 - an unreachable corpus is not a verdict
logger.debug("Corpus fetch failed for %r: %s", brand, e)
corpus = None
_corpus_memo[key] = corpus
return corpus
def reset_cache() -> None:
"""Drop the per-process corpus memo (tests, and long-lived workers)."""
_corpus_memo.clear()
def ground_product(brand: str, product_title: str,
*, corpus: Optional[Sequence[Dict[str, Any]]] = None) -> Grounding:
"""Is `product_title` a real product of `brand`, per Open*Facts?
`corpus` is injectable so a caller processing a whole brand fetches once
and so tests need no network.
"""
title = (product_title or "").strip()
if not title:
return Grounding(UNKNOWN)
hits = corpus if corpus is not None else _corpus_for(brand)
if hits is None:
return Grounding(UNKNOWN)
if not hits:
# A real answer: this brand has nothing on file at all. That is not
# evidence against any single product, so it is still `unknown` - a
# brand missing from Open Food Facts is a gap in Open Food Facts.
return Grounding(UNKNOWN, source="openfacts")
drop = off_bulk.brand_tokens(brand)
target = off_bulk.normalize_for_match(title, drop)
if not target:
return Grounding(UNKNOWN, source="openfacts")
best_score = 0.0
best_hit: Optional[Dict[str, Any]] = None
for hit in hits:
name = hit.get("product_name") or hit.get("product_name_en") or ""
if not name:
continue
candidate = off_bulk.normalize_for_match(name, drop)
if not candidate:
continue
# Skipped before scoring, not after: a wrong-material name can be the
# single best-scoring row in the corpus ("Rice Vermicelli" is the top
# match for "Wheat Vermicelli" at 0.610), and letting it win would mask
# a genuinely better match further down the list.
if _material_conflict(candidate, target):
continue
score = off_bulk.symmetric_similarity(candidate, target)
if score > best_score:
best_score, best_hit = score, hit
if best_hit is None or best_score < GROUNDING_SIMILARITY_FLOOR:
return Grounding(NOT_FOUND, similarity=round(best_score, 3), source="openfacts")
# A variant conflict means the corpus row is a DIFFERENT pack of a
# similarly-named product (sugar-free vs regular, jar vs pouch). Close
# enough to score well, not close enough to corroborate.
name = best_hit.get("product_name") or best_hit.get("product_name_en") or ""
if off_bulk.has_extra_variant_conflict(name, title):
return Grounding(NOT_FOUND, matched_name=name,
similarity=round(best_score, 3), source="openfacts")
quantity = str(best_hit.get("quantity") or "").strip() or None
return Grounding(
GROUNDED,
matched_name=name,
quantity=quantity,
similarity=round(best_score, 3),
source="openfacts",
)

View File

@@ -24,9 +24,15 @@ This module is that final gate. It is intentionally:
category-anchored price bands, and sku_service's own SKU format) rather
than guessing. False rejections are treated as seriously as false
acceptances - see docs/VALIDATION_PIPELINE.md for the tuning rationale.
- ADDITIVE: this module has zero dependents before this change and does not
modify any existing function's behaviour; it is only ever imported and
called from new code (see app/core/catalog_engine.py's call site).
- ADDITIVE: it does not modify any existing function's behaviour; callers opt
in. The call sites are store_catalog_pipeline.stage_10_validate (the
spreadsheet path), orchestration/assets/catalog.py (the Dagster mirror) and
ProductCatalogEngine.generate_catalog's "Step 2.6" (the brand-name path).
That last one was absent for a long time while this docstring claimed it
existed, so POST /api/catalog/generate wrote LLM output straight to the
brand tables with no gate at all. If you are adding another generation path,
it needs its own call to validate_catalog - nothing here is automatic.
USAGE
-----
@@ -244,6 +250,7 @@ def validate_product(
category_resolved_deterministically: bool = False,
grounded: bool = False,
images_checked: bool = True,
require_grounding: bool = False,
config: Optional[ValidationConfig] = None,
) -> ValidationReport:
"""Validate one fully-assembled catalog row and return a confidence-scored report.
@@ -350,12 +357,14 @@ def validate_product(
score -= 0.25
# 4b. Size unit <-> category compatibility (see app/services/
# category_units.py). This is a last-resort backstop: by the time a
# row reaches this final validator, its size should already have
# been through the generation-time gate (app/services/
# generation_verifier.py, called from ollama_service.py) AND
# catalog_engine.py's own dimension-unit defence-in-depth pass, both
# of which reject/repair a bad unit long before enrichment. This
# category_units.py). This used to describe itself as a backstop
# behind a generation-time gate in app/services/generation_verifier.py
# "called from ollama_service.py". THAT FILE DOES NOT EXIST IN THIS
# TREE - it belongs to a sibling project - so the layer this comment
# promised was never running here and this check is not a backstop but
# a front line. The real upstream defence is
# category_units.fix_or_reject_size, called from stage 4, plus
# catalog_engine.py's dimension-unit pass. This
# check exists purely to catch anything that reached validate_product
# some other way (e.g. a pre-existing DB row from before this fix, or
# a future caller that doesn't route through catalog_engine).
@@ -413,6 +422,33 @@ def validate_product(
else:
report.status = "verified"
# GROUNDING AS A CAP, NOT A BONUS.
#
# The +0.15 above is why this module could never do the job its docstring
# claims. A fabricated product scores 0.55 baseline + 0.15 resolved
# category + 0.05 has-images = 0.75, clearing the 0.70 threshold and
# landing as "verified" - so "Anil Wheat Vermicelli 12g", a pack size no
# shop has ever sold, was stored as a checked fact. Every check here is
# about FORM, and a well-formed fiction satisfies all of them.
#
# A caller that knows its rows came from a generator rather than from a
# source sets require_grounding. Then a row nothing external corroborates
# can still be kept and still scores normally, but it cannot be called
# verified - which is the honest answer, because nothing verified it.
#
# DELIBERATELY NOT APPLIED BY DEFAULT. The spreadsheet path passes no
# grounded flags at all, and a shop's own upload IS its evidence; capping
# there would relabel every uploaded row as unchecked and say nothing
# true. Only generation paths opt in.
if require_grounding and not grounded and report.status == "verified":
report.status = "needs_review"
report.issues.append(ValidationIssue(
field="grounding", severity="warning",
message="no external source corroborates that this product exists; "
"well-formed but unverified",
penalty=0.0,
))
return report
@@ -423,6 +459,7 @@ def validate_catalog(
known_category_flags: Optional[list[bool]] = None,
grounded_flags: Optional[list[bool]] = None,
images_checked: bool = True,
require_grounding: bool = False,
config: Optional[ValidationConfig] = None,
) -> tuple[list[dict[str, Any]], list[dict[str, Any]], dict[str, Any]]:
"""Validate an entire list of assembled catalog rows in one pass.
@@ -450,6 +487,7 @@ def validate_catalog(
category_resolved_deterministically=det,
grounded=grd,
images_checked=images_checked,
require_grounding=require_grounding,
config=cfg,
)
status_counts[report.status] = status_counts.get(report.status, 0) + 1

View File

@@ -0,0 +1,883 @@
"""
Is this product on sale, right now, somewhere real?
WHY THIS FILE EXISTS
--------------------
Open Food Facts answers "does this product exist" for food, and answers it
well. It does not answer it for anything else: `off_bulk` queries only
`search.openfoodfacts.org`, so a toothpaste or a detergent has no
product-discovery source at all and its entire catalogue is language-model
output. Measured 2026-09-10, India-tagged coverage on the sibling databases is
2-13 products per non-food brand against Britannia's 218 on OFF - real, but
nowhere near enough to build a catalogue from.
A live retail lookup answers it for everything, food and non-food alike, and
it is the only signal in this codebase that means "available in real time"
rather than "was in a database dump".
THE ONE THING THAT MAKES THIS WORK: MATCH THE PACK SIZE
--------------------------------------------------------
Measured against the live provider on 2026-09-10:
"Anil Roasted Short Vermicelli 450g"
-> amazon.in "Anil Vermicelli - Roasted, 450g Pouch" CONFIRMED
"Anil Wheat Vermicelli 12g"
-> results come back, but every one is the generic product
page. NOTHING confirms a 12 g pack. NOT CONFIRMED
A check that asks "did the search return anything?" would have confirmed the
12 g pack, which is the exact fabrication this module exists to catch. So a
hit counts only when the listing's TITLE and its PACK SIZE both match.
That is also why `sku_service.find_website_product_id` cannot be reused as
evidence: it regexes a product ID out of the result URL and never reads the
listing title or its size at all. It answers "is there an Amazon page in these
results", which for any real brand is always yes.
THREE STATES, NEVER TWO
-----------------------
`found` / `not_found` / `unknown`. DuckDuckGo throttles and 403s - the repair
script already records that it "already 403s and the pipeline falls through to
Bing". A throttled lookup MUST report `unknown`, because a gate fed
`not_found` on a rate-limit would quietly demote a brand's entire real
catalogue on a bad afternoon.
COST, AND WHY THE STAGE READS ONLY THE CACHE
---------------------------------------------
There is no brand-scoped retail endpoint the way there is for Open Food Facts,
so this is irreducibly one network call per product against a provider that
blocks. Measured 2.6-5.7 s per query.
So the work is split: `backfill_retail_presence.py` does the querying offline,
paced and cached, and the ingestion-time stage reads the cache ONLY
(`live=False`). Ingestion never blocks on a live query and never trips a rate
limit mid-batch. `search_key` collapses pack sizes onto one lookup per
PRODUCT rather than per row, which is a 3-5x cut for free.
"""
from __future__ import annotations
import json
import logging
import re
import sqlite3
import threading
import time
from contextlib import closing
from dataclasses import asdict, dataclass, field
from pathlib import Path
from typing import Any, Dict, List, Optional, Sequence
from urllib.parse import urlparse
from app.services import quantity_utils
from app.services.brand_registry import BRAND_ALIASES
from app.services.enrichment.barcode.matching import brand_matches
from app.services.enrichment.barcode.sources import off_bulk
from app.services.image_corroboration import search_key
logger = logging.getLogger(__name__)
FOUND = "found"
NOT_FOUND = "not_found"
UNKNOWN = "unknown"
# Indian grocery/marketplace domains a real FMCG product is listed on.
# `sku_service._MARKETPLACE_PATTERNS` covers the same ground for ID extraction;
# this list is about presence, so it needs no per-site URL grammar.
RETAILER_DOMAINS = (
"amazon.in", "flipkart.com", "bigbasket.com", "jiomart.com",
"blinkit.com", "zeptonow.com", "dmart.in", "swiggy.com",
"netmeds.com", "pharmeasy.in", "nykaa.com", "1mg.com",
"licious.in", "starquik.com", "spencers.in", "moreretail.in",
)
# What fraction of the PRODUCT's distinctive words must appear in the listing
# title. See `listing_matches` for why this is a containment ratio and not the
# 0.85 symmetric similarity the barcode matcher uses.
#
# 0.6 is deliberately loose. Retailers drop words freely - Amazon lists "Anil
# Roasted Short Vermicelli" as "Anil Vermicelli - Roasted", covering 2 of 3
# tokens (0.67) - so a tight floor here rejects real listings while adding
# nothing, because it is the pack-size check below that separates a real pack
# from an invented one. This gate only has to establish that the listing is
# about the right product family.
TITLE_COVERAGE_FLOOR = 0.6
# Marketing boilerplate that surrounds a product name in a retail listing
# title. Removed before matching so "Buy Anil Roasted Vermicelli 450g Online
# at Best Price" reduces to the product.
_LISTING_NOISE = re.compile(
r"\b(buy|online|at|best|price|lowest|offers?|deals?|free|delivery|shop|"
r"order|now|india|in|from|upto|off|save|get|com|grocery|gourmet|foods?)\b",
re.IGNORECASE,
)
# Words that name the SAME thing in a retail listing and in our catalogue.
#
# Indian FMCG listings mix English and transliterated Tamil/Hindi freely, and
# our own product names do too. Measured misses: our "Anil Puttu Mix" is on
# amazon.in as "Anil Puttu MAAVU" (maavu = flour/mix) and was rejected at 0.5
# coverage; this catalogue also uses "Semiya" and "Vermicelli" interchangeably
# in its own product names, so the two spellings fail to corroborate each other.
#
# STRICTLY OBSERVED, NEVER INVENTED - the same rule product_grounding._MATERIALS
# follows. This must not grow into a general thesaurus: every pair added makes
# the matcher more willing to confirm a product, and confirming products that do
# not exist is the failure this whole module exists to prevent. Add a pair only
# after seeing it reject a real listing.
#
# THE PACK SIZE IS NOT AFFECTED. Synonyms loosen only which words count as the
# same word; `quantities_match` still has to agree, so no synonym can rescue a
# size that nobody sells.
# Each pair below cites the listing that forced it. Anything without a citation
# does not belong here.
_SYNONYMS: Dict[str, str] = {
# amazon.in "Anil Puttu Maavu" vs our "Anil Puttu Mix", and
# theanilgroup.com "Arisi Puttu Maavu" for the same product.
"maavu": "flour",
"mix": "flour",
# This catalogue's own names: "Anil Rava Semiya", "Anil Ragi Semiya" and
# "Anil Wheat Vermicelli" are the same product family spelled two ways.
"semiya": "vermicelli",
# Open Food Facts holds our "Anil Samba Rava" as "SAMBA RAVVA".
"ravva": "rava",
# theanilgroup.com sells our "Wheat Vermicelli" as "Atta Semiya".
"atta": "wheat",
}
def _aliases_for(brand: str) -> List[str]:
"""Other names this brand trades under, from the curated registry.
`BRAND_ALIASES` maps alias -> parent, so the aliases OF a brand are the
keys pointing at it, plus the parent name itself.
"""
key = (brand or "").strip().lower()
if not key:
return []
out = {key}
parent = BRAND_ALIASES.get(key)
if parent:
out.add(parent.lower())
for alias, target in BRAND_ALIASES.items():
if target.lower() == key or (parent and target.lower() == parent.lower()):
out.add(alias.lower())
return sorted(out)
def _canonical_words(text: str) -> set:
"""Token set with observed synonyms folded onto one spelling."""
return {_SYNONYMS.get(w, w) for w in (text or "").split()}
# A real product listing title is short. Anything much longer is the provider
# running several results together into one `title` field, e.g.
#
# "Ginger Garlic Paste (Pack of 2) | No Peeling, No ChoppingAachi Ginger
# Garlic Paste 20g - martizo.comAachi Ginger Garlic Paste - Buy at Rs43..."
#
# 372 characters, at least four different listings, from at least two shops.
# Matching against that blob is meaningless in both halves: the quantity comes
# from whichever listing happened to state one, and the word coverage is
# satisfied by words scattered across products we never asked about. It
# produced a FOUND for "Aachi Ginger Paste 20g" whose 20g came from a different
# retailer's listing of a different product, attributed to aachifoods.com.
#
# A false positive is the worse failure here - it hands 0.85 corroboration to a
# product that may not exist, which is what this module was built to prevent -
# so a blob is discarded rather than parsed.
MAX_LISTING_TITLE_CHARS = 140
# Ingredient words that make two similarly-named products DIFFERENT products.
#
# Coverage is deliberately containment - the target's words must appear in the
# listing - so a listing is free to carry extra words, which is right for shop
# furniture ("Pouch", "Best Price", "Amazon.in"). It is wrong for an ingredient:
# "Aachi Ginger Paste" is entirely contained in "Aachi Ginger GARLIC Paste" and
# scored 1.0 coverage against it, and ginger paste is not ginger-garlic paste.
#
# Same discipline as product_grounding._MATERIALS and _SYNONYMS: these are
# words observed causing a wrong match, not a general ingredient list.
_DISTINGUISHING_INGREDIENTS = frozenset({
"garlic", "ginger", "chilli", "chili", "pepper", "tamarind", "coriander",
"cumin", "turmeric", "mint", "lemon", "tomato", "onion", "mango",
"coconut", "mustard", "fenugreek", "clove", "cardamom", "cinnamon",
})
_DB_PATH = Path("data") / "cache" / "retail_presence.db"
_lock = threading.Lock()
_initialized = False
DEFAULT_TTL_SECONDS = 14 * 24 * 3600
# A brand's own domain changes far more rarely than its stock does.
BRAND_DOMAIN_TTL_SECONDS = 180 * 24 * 3600
# A miss is re-asked the next day - see get_cached_brand_domain for why.
BRAND_DOMAIN_MISS_TTL_SECONDS = 24 * 3600
# Search endpoints are the strictly-limited ones. Same figure off_bulk uses.
PAUSE_SECONDS = 2.0
@dataclass
class RetailEvidence:
"""What a live retail search says about one (brand, product, size)."""
status: str
retailer: Optional[str] = None
url: Optional[str] = None
matched_title: Optional[str] = None
matched_size: Optional[str] = None
similarity: float = 0.0
checked_at: Optional[float] = None
# How many results were actually on a shop. This is what makes a negative
# auditable: without it, the only way to ask "did we even reach a retailer?"
# is to run the search again, and the provider's reach varies run to run
# (the same query returned 7 shop results and then 0 minutes later).
shop_results_seen: int = 0
# The shop titles that came back and did NOT match, kept so a human can
# judge whether the rejection was right. "Aachi Biryani Masala" against
# "Aachi Biryani Mix" is a correct rejection; "Anil Puttu Maavu" against
# "Anil Puttu Mix" is a matcher failure, and nothing but the title tells
# them apart.
seen_titles: List[str] = field(default_factory=list)
query: Optional[str] = None
@property
def is_found(self) -> bool:
return self.status == FOUND
def note(self) -> str:
if self.status == FOUND:
return f"listed by {self.retailer} as {self.matched_title!r}"
if self.status == NOT_FOUND:
return (f"{self.shop_results_seen} shop listing(s) seen, none for "
f"this product at this pack size")
if self.shop_results_seen == 0 and self.checked_at:
return "searched, but no shop listing was reached at all"
return "could not check retail availability"
# ---------------------------------------------------------------------------
# Cache
# ---------------------------------------------------------------------------
def _connect() -> sqlite3.Connection:
_DB_PATH.parent.mkdir(parents=True, exist_ok=True)
conn = sqlite3.connect(str(_DB_PATH), timeout=10)
conn.execute("PRAGMA journal_mode=WAL")
return conn
def _ensure_schema(conn: sqlite3.Connection) -> None:
global _initialized
if _initialized:
return
conn.execute(
"""
CREATE TABLE IF NOT EXISTS retail_presence_cache (
cache_key TEXT PRIMARY KEY,
brand TEXT,
product TEXT,
size TEXT,
result_json TEXT NOT NULL,
created_at REAL NOT NULL
)
"""
)
# One row per brand, because a brand's own domain is a fact about the
# brand and costs a search to learn. Resolved once and reused across every
# product, so it adds one query per brand rather than one per product.
conn.execute(
"""
CREATE TABLE IF NOT EXISTS brand_domain_cache (
brand TEXT PRIMARY KEY,
domain TEXT,
created_at REAL NOT NULL
)
"""
)
conn.commit()
_initialized = True
def cache_key(brand: str, product_title: str, size: str) -> str:
"""One key per PRODUCT+SIZE, with the product name size-collapsed first.
`search_key` strips a trailing pack size from the name so
"Anil Vermicelli 180g" and "Anil Vermicelli 450g" share a product identity;
the size is then a separate component, because the size is precisely what
is being verified.
"""
brand_key, product_key = search_key(product_title, brand)
size_key = re.sub(r"\s+", "", (size or "").strip().lower())
return f"{brand_key}|{product_key}|{size_key}"
def get_cached(brand: str, product_title: str, size: str,
ttl_seconds: float = DEFAULT_TTL_SECONDS) -> Optional[RetailEvidence]:
"""A cached verdict, or None for a miss. Never raises."""
key = cache_key(brand, product_title, size)
try:
with _lock, closing(_connect()) as conn:
_ensure_schema(conn)
row = conn.execute(
"SELECT result_json, created_at FROM retail_presence_cache WHERE cache_key = ?",
(key,),
).fetchone()
except Exception as e: # noqa: BLE001 - a cache failure is a cache miss
logger.debug("Retail cache read failed for %s: %s", key, e)
return None
if not row:
return None
result_json, created_at = row
if ttl_seconds > 0 and (time.time() - created_at) > ttl_seconds:
return None
try:
return RetailEvidence(**json.loads(result_json))
except Exception: # noqa: BLE001 - an unreadable entry is a miss
return None
def set_cached(brand: str, product_title: str, size: str,
evidence: RetailEvidence) -> None:
"""Best-effort write. A failure just means no caching next time.
An `unknown` verdict is NOT cached: it records that we failed to ask, not
an answer, and caching it would turn one rate-limited afternoon into two
weeks of pretending we had checked.
"""
if evidence.status == UNKNOWN:
return
key = cache_key(brand, product_title, size)
try:
with _lock, closing(_connect()) as conn:
_ensure_schema(conn)
conn.execute(
"""
INSERT INTO retail_presence_cache
(cache_key, brand, product, size, result_json, created_at)
VALUES (?, ?, ?, ?, ?, ?)
ON CONFLICT(cache_key) DO UPDATE SET
result_json = excluded.result_json,
created_at = excluded.created_at
""",
(key, brand, product_title, size,
json.dumps(asdict(evidence)), time.time()),
)
conn.commit()
except Exception as e: # noqa: BLE001 - best effort only
logger.debug("Retail cache write failed for %s: %s", key, e)
# ---------------------------------------------------------------------------
# Matching
# ---------------------------------------------------------------------------
# Domains that are never a brand's own site, however often they rank for its
# name. Retailers are excluded separately via RETAILER_DOMAINS.
_NOT_A_BRAND_SITE = (
"wikipedia.org", "wikimedia.org", "facebook.com", "instagram.com",
"linkedin.com", "youtube.com", "twitter.com", "x.com", "pinterest.com",
"indiamart.com", "justdial.com", "tradeindia.com", "exportersindia.com",
"zaubacorp.com", "tofler.in", "crunchbase.com", "glassdoor.com",
"google.com", "blogspot.com", "wordpress.com", "medium.com",
"quora.com", "reddit.com", "yelp.com", "tripadvisor.com",
)
def _registrable(url: str) -> str:
"""The registrable-ish domain of a URL ("shop.theanilgroup.com" ->
"theanilgroup.com"). Good enough for matching, not a public-suffix parser."""
host = urlparse(url if "//" in url else f"//{url}").netloc.lower()
host = host.split(":")[0]
if host.startswith("www."):
host = host[4:]
parts = host.split(".")
if len(parts) > 2 and parts[-2] in ("co", "com", "net", "org", "gov", "ac"):
return ".".join(parts[-3:]) # theanilgroup.co.in
if len(parts) > 2:
return ".".join(parts[-2:]) # shop.theanilgroup.com -> theanilgroup.com
return host
def get_cached_brand_domain(brand: str) -> Optional[str]:
"""Cached brand domain, or None for a miss. Never raises.
An empty string is a real cached answer meaning "looked and found none";
it comes back as None to callers but stops the lookup being repeated.
"""
key = _normalize_brand(brand)
try:
with _lock, closing(_connect()) as conn:
_ensure_schema(conn)
row = conn.execute(
"SELECT domain, created_at FROM brand_domain_cache WHERE brand = ?",
(key,),
).fetchone()
except Exception: # noqa: BLE001 - a cache failure is a cache miss
return None
if not row:
return None
domain, created_at = row
# A NEGATIVE EXPIRES FAST; A POSITIVE LASTS.
#
# These searches fan out across several engines and the result set varies
# run to run: the first Anil sweep found no brand site, and a repeat of the
# identical query minutes later returned shop.theanilgroup.com three times
# in the top four. Caching that miss for six months would have made one
# flaky search a permanent fact about the brand, and every product of
# Anil's would have kept reporting "sold nowhere".
#
# A found domain is stable and worth keeping; a miss is worth re-asking
# tomorrow.
ttl = BRAND_DOMAIN_TTL_SECONDS if domain else BRAND_DOMAIN_MISS_TTL_SECONDS
if (time.time() - created_at) > ttl:
return None
return domain or None
def _normalize_brand(brand: str) -> str:
return re.sub(r"\s+", " ", (brand or "").strip().lower())
def resolve_brand_domain(brand: str, *, live: bool = False,
sample_products: Optional[Sequence[str]] = None,
timeout: int = 20) -> Optional[str]:
"""The brand's own website, e.g. "Anil" -> "theanilgroup.com".
WHY THIS EXISTS. The first backfill run asked only the retailer list and
reported "Anil Wheat Vermicelli 180g" as not listed anywhere - while the
brand's own shop sells exactly that pack. A manufacturer's catalogue is
real evidence that a product and its pack size exist, and leaving it out
turned "no big retailer stocks this" into "this product is not real",
which are very different claims.
A brand site is WEAKER evidence than a retailer: a manufacturer may list a
discontinued or not-yet-shipping line. The verdict records which domain
answered, so a consumer can weigh them differently - see `RetailEvidence.
retailer`.
Identified by a domain that CONTAINS a brand token and is neither a
retailer nor a directory/social site. That is a deliberately conservative
rule: it will miss a brand whose site is named nothing like the brand, and
a miss just means we fall back to the retailer list.
"""
key = _normalize_brand(brand)
if not key:
return None
cached = get_cached_brand_domain(brand)
if cached is not None:
return cached
if not live:
return None
tokens = [t for t in off_bulk.brand_tokens(brand) if len(t) > 2]
if not tokens:
return None
# QUERY WITH A REAL PRODUCT NAME, NOT THE BRAND ALONE.
#
# "Anil official site products" returned maccosmetics.com, vertu.com and
# a YouTube video: "Anil" is a common personal name, so a brand-only query
# is dominated by cricketers and airlines - the same ambiguity that put a
# photo of Anil Kapoor on a packet of rava. Adding a product the brand
# actually sells makes the query specific enough to find the real site;
# "Anil Wheat Vermicelli" surfaces shop.theanilgroup.com immediately.
#
# Several phrasings, because one search is demonstrably not enough: the
# identical query minutes apart returned the brand site three times and
# then not at all. Stops at the first that yields a domain.
phrasings: List[str] = [f"{brand} {p}" for p in (sample_products or [])[:2]]
phrasings += [f"{brand} official site products",
f"{brand} official website India"]
found: Optional[str] = None
asked = False
for phrasing in phrasings:
if found:
break
results = _search(phrasing, 10, timeout)
if results is None:
continue
asked = True
found = _pick_brand_domain(results, tokens)
if not asked:
return None # could not ask - do NOT cache a failure as "none"
_cache_brand_domain(key, found or "")
return found
def _pick_brand_domain(results, tokens) -> Optional[str]:
"""The first result whose domain stem contains a brand token and is not a
retailer, directory or social site."""
for result in results:
url = str(result.get("href") or result.get("url") or "")
if not url.startswith("http"):
continue
domain = _registrable(url)
if not domain:
continue
if any(bad in domain for bad in _NOT_A_BRAND_SITE):
continue
if any(retailer in domain for retailer in RETAILER_DOMAINS):
continue
stem = domain.split(".")[0]
if any(token in stem for token in tokens):
return domain
return None
def _cache_brand_domain(key: str, domain: str) -> None:
try:
with _lock, closing(_connect()) as conn:
_ensure_schema(conn)
conn.execute(
"""
INSERT INTO brand_domain_cache (brand, domain, created_at)
VALUES (?, ?, ?)
ON CONFLICT(brand) DO UPDATE SET
domain = excluded.domain, created_at = excluded.created_at
""",
(key, domain, time.time()),
)
conn.commit()
except Exception as e: # noqa: BLE001 - best effort only
logger.debug("Brand domain cache write failed for %s: %s", key, e)
def _is_product_page(url: str) -> bool:
"""Does this URL point at a specific product rather than a landing page?
Only applied to the BRAND's own domain. A retailer domain is a shop by
definition, but a manufacturer's site is mostly corporate: `theanilgroup.com/`
is an "about us" homepage and `shop.theanilgroup.com/products/wheat-vermicelli`
is a listing, and both share a registrable domain.
Counting the homepage as "a shop was reached" is not harmless - it is what
decides whether a miss is reported as `not_found` (an answer) or `unknown`
(we never asked a shop). The corporate homepage ranks for almost every
query about the brand, so without this every no-shop verdict would be
dressed up as a real absence.
"""
path = urlparse(url or "").path.strip("/")
if not path:
return False
return any(
marker in f"/{path.lower()}/"
for marker in ("/product", "/products/", "/shop/", "/collections/",
"/p/", "/pd/", "/item", "/buy", "/store/")
) or path.count("/") >= 1
def _retailer_for(url: str) -> Optional[str]:
lowered = (url or "").lower()
for domain in RETAILER_DOMAINS:
if domain in lowered:
return domain
return None
def _clean_listing_title(title: str) -> str:
return _LISTING_NOISE.sub(" ", title or "")
def listing_matches(listing_title: str, brand: str, product_title: str,
size: str) -> tuple:
"""Does this listing describe THIS product at THIS pack size?
Returns (matches, coverage). Both halves are required - see the module
docstring for the measurement that makes the size half non-negotiable.
COVERAGE, NOT SYMMETRIC SIMILARITY. `off_bulk.symmetric_similarity` is
built to compare one database record with another and penalises extra
tokens on either side. A retail listing title is not a record: it is a
SUPERSET of the product name wrapped in shop furniture. Measured against
real titles the live provider returned, symmetric similarity scored
"Anil Vermicelli - Roasted, 450g Pouch : Amazon.in: Grocery &
Gourmet Foods" vs "Anil Roasted Short Vermicelli"
-> 0.445, rejected
which is a genuine listing of exactly the product asked for. So the title
half asks the containment question instead - how much of the PRODUCT's
identity appears in the listing - and the pack size does the discriminating.
"""
if not listing_title:
return False, 0.0
# Several listings run together into one title field - see
# MAX_LISTING_TITLE_CHARS. Neither the size nor the words can be attributed
# to a single product, so this corroborates nothing.
if len(listing_title) > MAX_LISTING_TITLE_CHARS:
return False, 0.0
# THE LISTING MUST BE FOR OUR BRAND.
#
# This was missing entirely, and it is the worst hole this module has had.
# `normalize_for_match` strips brand tokens from BOTH sides - correct for
# its original job, matching within one brand's own Open Food Facts corpus,
# where the brand is a given. Here the listing can be anyone's, so stripping
# the brand made the comparison brand-blind and a COMPETITOR's product
# corroborated ours:
#
# "A1 Naanjil Naattu Masala Instant Pongal Mix, 500g"
# confirmed Anil Pongal Mix 500g (real, from the backfill)
# "Britannia Good Day 200g" confirmed "Anil Good Day 200g"
# "MTR Rava Idli Mix 500g" confirmed "Anil Idli Mix 500g"
#
# Checked BEFORE the brand tokens are stripped, obviously, and via the
# barcode matcher's own `brand_matches` so aliases ("HUL" for "Hindustan
# Unilever") keep working and the two gates cannot drift apart.
if not brand_matches(listing_title, brand, _aliases_for(brand)):
return False, 0.0
drop = off_bulk.brand_tokens(brand)
listing_clean = off_bulk.normalize_for_match(
_clean_listing_title(listing_title), drop
)
target_clean = off_bulk.normalize_for_match(
off_bulk.strip_sizes(product_title), drop
)
if not listing_clean:
return False, 0.0
# Synonyms folded on BOTH sides, so "Puttu Mix" and "Puttu Maavu" reduce to
# the same words. See _SYNONYMS - observed pairs only, and the pack-size
# check below is untouched by any of it.
target_tokens = [_SYNONYMS.get(t, t) for t in target_clean.split() if len(t) > 2]
listing_tokens = _canonical_words(listing_clean)
if target_tokens:
coverage = sum(1 for t in target_tokens if t in listing_tokens) / len(target_tokens)
else:
# The product name is nothing but brand and size ("Amul 1kg"). There
# is no identity to cover, so the brand's presence is all that can be
# asked, and the size below carries the whole verdict.
coverage = 1.0 if any(
b in listing_title.lower() for b in off_bulk.brand_tokens(brand)
) else 0.0
if coverage < TITLE_COVERAGE_FLOOR:
return False, round(coverage, 3)
# A variant term the other side does not have means a different pack of a
# similarly-named product (sugar-free vs regular, jar vs pouch).
if off_bulk.has_extra_variant_conflict(listing_title, product_title):
return False, round(coverage, 3)
# An INGREDIENT the listing names and the product does not. Containment
# tolerates extra words on the listing side, which is right for shop
# furniture and wrong for this: "Aachi Ginger Paste" is wholly contained in
# "Aachi Ginger Garlic Paste" and scored 1.0 against it. See
# _DISTINGUISHING_INGREDIENTS.
extra = (listing_tokens & _DISTINGUISHING_INGREDIENTS) - set(target_tokens)
if extra:
return False, round(coverage, 3)
# THE SIZE CHECK, AND IT IS THE ONE THAT MATTERS.
#
# Measured on the live provider: searching "Anil Wheat Vermicelli 12g"
# returns the brand's own generic product page, whose title covers 100% of
# the product's tokens. Only the absence of any 12 g mention distinguishes
# a pack that exists from one that does not. A listing that states no
# quantity at all confirms no quantity at all.
if size and size.strip():
mentions = quantity_utils.find_quantity_mentions(listing_title)
if not mentions:
return False, round(coverage, 3)
if not any(
quantity_utils.quantities_match(size, f"{q}g") for q in mentions
):
return False, round(coverage, 3)
return True, round(coverage, 3)
# ---------------------------------------------------------------------------
# The lookup
# ---------------------------------------------------------------------------
def _search(query: str, max_results: int, timeout: int) -> Optional[List[Dict[str, Any]]]:
"""Raw web results, or None when the provider could not be reached.
None and [] are different answers and the caller depends on it: None is
"we could not ask" and becomes `unknown`; [] is "we asked and nobody sells
this" and becomes `not_found`.
"""
try:
from ddgs import DDGS
except ImportError:
logger.debug("Retail presence unavailable: 'ddgs' is not installed")
return None
try:
with DDGS(timeout=timeout) as ddgs:
return list(ddgs.text(query, region="in-en", safesearch="off",
max_results=max_results) or [])
except Exception as e: # noqa: BLE001 - a throttle is not a verdict
logger.debug("Retail search failed for %r: %s", query, e)
return None
def check_listing(brand: str, product_title: str, size: str, *,
live: bool = False,
refresh: bool = False,
brand_domain: Optional[str] = None,
ttl_seconds: float = DEFAULT_TTL_SECONDS,
max_results: int = 10,
timeout: int = 20) -> RetailEvidence:
"""Is (brand, product_title, size) listed by a real retailer?
`live` defaults to FALSE. Ingestion reads the cache and never issues a
network call; the backfill script passes live=True. See the module
docstring - this is one blocking call per product against a provider that
rate-limits, so it does not belong inside a batch.
"""
if not (brand or "").strip() or not (product_title or "").strip():
return RetailEvidence(UNKNOWN)
# `refresh` SKIPS THE CACHE READ, and without it a caller cannot re-ask.
#
# The backfill script has its own --refresh, which decides which rows are
# worth asking again - but it then called this function, which read the
# cache first and handed back the very verdict the caller was trying to
# replace. A whole "rebuild" of Anil ran to completion, reported 46
# re-asks, and made no network calls at all: it re-read 46 stale rows and
# wrote nothing, because the early return happens before set_cached.
#
# Silent, and it looks exactly like a successful run.
if not refresh:
cached = get_cached(brand, product_title, size, ttl_seconds)
if cached is not None:
return cached
if not live:
return RetailEvidence(UNKNOWN)
# Resolve the brand's own site if the caller did not name one. Cached per
# brand, so this costs one extra search per BRAND, not per product. Without
# it the first Anil sweep reported "Anil Wheat Vermicelli 180g" as sold
# nowhere while the brand's own shop lists exactly that pack.
if brand_domain is None:
brand_domain = resolve_brand_domain(brand, live=True, timeout=timeout)
# A NEGATIVE IS CONFIRMED WITH A SECOND PHRASING, NOT TAKEN ON ONE TRY.
#
# `ddgs` fans out over several engines and the result set genuinely varies
# between identical queries minutes apart - measured: "Anil Wheat
# Vermicelli 180g" returned nothing usable on one run and an Amazon listing
# for exactly that pack on the next. A single query is enough to CONFIRM a
# product (a real listing is real however it was found) but not enough to
# deny one, so only the negative pays for the retry.
base = off_bulk.strip_sizes(product_title)
phrasings = [
" ".join(p for p in (brand, base, size) if p).strip(),
" ".join(p for p in (brand, base, size, "buy online price") if p).strip(),
]
# A site:-scoped query is the reliable way to reach the brand's own shop,
# but it is worth a third round trip only when the open queries reached no
# shop at all - see the reach check after the loop.
if brand_domain:
phrasings.append(
" ".join(p for p in (brand, base, size, f"site:{brand_domain}") if p).strip()
)
domains = tuple(RETAILER_DOMAINS) + ((brand_domain,) if brand_domain else ())
best = RetailEvidence(NOT_FOUND, checked_at=time.time())
asked = False
shop_results_seen = 0
seen_titles: List[str] = []
for query in phrasings:
# Only escalate to the site:-scoped query when nothing else reached a
# shop. A product already seen on a retailer needs no third opinion.
if "site:" in query and shop_results_seen:
break
results = _search(query, max_results, timeout)
if results is None:
continue
asked = True
for result in results:
url = str(result.get("href") or result.get("url") or "")
title = str(result.get("title") or "")
retailer = _retailer_for(url) or (
brand_domain if brand_domain and brand_domain in _registrable(url)
and _is_product_page(url) else None
)
if not retailer or retailer not in domains:
continue
# Reached an actual shop. Counted whether or not it matches, because
# that count is what separates "shops do not sell this" from "we
# never got to ask a shop".
shop_results_seen += 1
if len(seen_titles) < 8 and title:
seen_titles.append(f"{retailer}: {title}"[:160])
matches, similarity = listing_matches(title, brand, product_title, size)
if similarity > best.similarity:
best.similarity = round(similarity, 3)
if matches:
evidence = RetailEvidence(
FOUND, retailer=retailer, url=url, matched_title=title,
matched_size=size, similarity=round(similarity, 3),
checked_at=time.time(), shop_results_seen=shop_results_seen,
query=query,
)
set_cached(brand, product_title, size, evidence)
return evidence
if not asked:
# Could not ask at all - every phrasing failed to reach the provider.
return RetailEvidence(UNKNOWN)
# ASKED, BUT NEVER ASKED A SHOP.
#
# The searches came back full of recipe blogs, news pages and the brand's
# corporate homepage, and not one result was on a retailer or the brand's
# store. Recording that as "no retailer sells this product" states something
# we did not find out - it is the same conflation as treating a throttle as
# an absence, and it is what made 2 of 10 sampled Anil negatives meaningless.
#
# Reported as UNKNOWN, and UNKNOWN is never cached, so the question gets
# asked again instead of the product being written off for a fortnight.
if shop_results_seen == 0:
return RetailEvidence(UNKNOWN, checked_at=time.time(), shop_results_seen=0)
best.shop_results_seen = shop_results_seen
best.seen_titles = seen_titles
set_cached(brand, product_title, size, best)
return best
def check_many(items: Sequence[Dict[str, str]], *, live: bool = False,
deadline_seconds: Optional[float] = None,
pause_seconds: float = PAUSE_SECONDS) -> Dict[str, RetailEvidence]:
"""Check a batch, keyed by `cache_key`.
SEQUENTIAL, WITH PACING - deliberately not concurrent. `EnrichmentPipeline`
bounds concurrency across ROWS, which is right when rows hit independent
resources; here every row hits one throttled provider, so parallelism buys
a 403 rather than throughput. `off_bulk.PAUSE_SECONDS` is the in-repo
precedent for exactly this.
Duplicate keys are asked once: `cache_key` collapses pack-size variants of
a product onto one product identity, so a brand with three sizes of each
item costs a third of the queries.
"""
out: Dict[str, RetailEvidence] = {}
started = time.monotonic()
for item in items:
brand = item.get("brand") or ""
title = item.get("product_title") or item.get("title") or ""
size = item.get("size") or ""
key = cache_key(brand, title, size)
if key in out:
continue
if deadline_seconds is not None and (time.monotonic() - started) > deadline_seconds:
# Everything still unasked is `unknown`, not `not_found`.
out[key] = RetailEvidence(UNKNOWN)
continue
cached = get_cached(brand, title, size)
if cached is not None:
out[key] = cached
continue
out[key] = check_listing(brand, title, size, live=live)
if live:
time.sleep(pause_seconds)
return out

View File

@@ -92,8 +92,24 @@ CATEGORY_TYPE_WORDS: dict[str, set[str]] = {
"curry powder": {"Spices & Masalas"}, "garam masala": {"Spices & Masalas"},
"chilli powder": {"Spices & Masalas"}, "turmeric powder": {"Spices & Masalas"},
"deggi mirch": {"Spices & Masalas"}, "tikhalal": {"Spices & Masalas"}, "kashmiri lal": {"Spices & Masalas"},
"pasta": {"Pasta & Noodles"}, "vermicelli": {"Pasta & Noodles"}, "semiya": {"Pasta & Noodles"},
"macaroni": {"Pasta & Noodles"}, "spaghetti": {"Pasta & Noodles"},
# BOTH spellings, for every word in this group.
#
# `category_registry` is the module that ASSIGNS the category, and its
# entry for "Noodles & Instant Food" lists "vermicelli" and "pasta" among
# its own keywords. So the detector resolved "Anil Wheat Vermicelli" to
# "Noodles & Instant Food" and this table then reported the word
# "vermicelli" as CONTRADICTING it - a -0.35 confidence penalty on every
# vermicelli, pasta, semiya, macaroni and spaghetti row in the catalogue,
# for nothing but two tables spelling one category differently.
#
# `noodle`/`noodles` already carried both, because somebody hit this and
# patched the row in front of them. The rest of the group has the same
# bug. If a new pasta-ish word is added here, give it both.
"pasta": {"Pasta & Noodles", "Noodles & Instant Food"},
"vermicelli": {"Pasta & Noodles", "Noodles & Instant Food"},
"semiya": {"Pasta & Noodles", "Noodles & Instant Food"},
"macaroni": {"Pasta & Noodles", "Noodles & Instant Food"},
"spaghetti": {"Pasta & Noodles", "Noodles & Instant Food"},
"noodle": {"Pasta & Noodles", "Noodles & Instant Food"}, "noodles": {"Pasta & Noodles", "Noodles & Instant Food"},
# --- Desserts (head-noun only, not ingredient words) ---
"ice cream": {"Ice Cream"}, "icecream": {"Ice Cream"}, "kulfi": {"Ice Cream"}, "frozen dessert": {"Ice Cream"},

View File

@@ -178,6 +178,17 @@ def get_brand_table_ddl(brand: str) -> str:
nutrients_per_100g JSONB,
search_query TEXT,
-- The deterministic validation verdict from product_validator.
-- validate_catalog() has always computed these three and annotated
-- them onto every row; until they were added here nothing projected
-- them, so a fabricated row and a corroborated one were
-- indistinguishable once stored. That is what made "is this product
-- real?" unanswerable after the fact.
-- status is one of: verified | needs_review | rejected.
validation_status TEXT,
confidence_score NUMERIC,
validation_issues JSONB,
-- Per-field provenance, keyed by column name; each value records how
-- that column's value was arrived at. One JSONB map rather than ~60
-- scalar columns, because every scalar column would have to be named
@@ -274,6 +285,11 @@ def _ensure_columns(cur, table_name: str) -> None:
"nutrients": "TEXT[]",
"nutrients_per_100g": "JSONB",
"search_query": "TEXT",
# The product_validator verdict - see get_brand_table_ddl for why a
# row's confidence has to survive to disk to be worth computing.
"validation_status": "TEXT",
"confidence_score": "NUMERIC",
"validation_issues": "JSONB",
# Per-field provenance map - see get_brand_table_ddl for why this is
# one JSONB column and not sixty scalar ones.
"field_sources": "JSONB",
@@ -488,6 +504,18 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
field_sources = p.get("field_sources")
field_sources = Json(field_sources) if isinstance(field_sources, dict) and field_sources else None
# The product_validator verdict, as annotated by validate_catalog().
# A row that never went through the gate carries none of these, so
# they arrive as None and the DO UPDATE SET below assigns NULL - see
# the comment there for why that is the wanted behaviour here and the
# opposite of the rule the enrichment columns follow.
validation_status = str(p.get("validation_status") or "").strip() or None
confidence_score = _to_numeric_or_none(p.get("confidence_score"))
validation_issues = p.get("validation_issues")
validation_issues = (
Json(list(validation_issues)) if isinstance(validation_issues, (list, tuple)) else None
)
# Essential fields
highlights = p.get("highlights", [])
if not isinstance(highlights, list):
@@ -542,6 +570,9 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
highlights, # TEXT[] - psycopg will handle conversion
nutrients, # TEXT[] - psycopg will handle conversion
search_query,
validation_status,
confidence_score,
validation_issues,
field_sources,
embedding_str
))
@@ -586,8 +617,10 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
fssai_license, product_sku, sku_source, hsn_code, final_selling_price, selling_price, barcode, barcode_type,
gtin, ean13, upc, barcode_source, barcode_verified, barcode_lookup_status, barcode_last_updated,
gst_percent, tax_amount, hsn_gst_needs_review,
highlights, nutrients, search_query, field_sources, embedding)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
highlights, nutrients, search_query,
validation_status, confidence_score, validation_issues,
field_sources, embedding)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
ON CONFLICT (image_id) DO UPDATE SET
product_name = EXCLUDED.product_name,
title = EXCLUDED.title,
@@ -625,6 +658,26 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
nutrients = EXCLUDED.nutrients,
search_query = EXCLUDED.search_query,
embedding = EXCLUDED.embedding,
-- PLAINLY ASSIGNED, NOT COALESCEd. THIS IS DELIBERATE AND
-- IS THE OPPOSITE OF THE RULE THE BLOCK BELOW FOLLOWS.
--
-- Those are enrichment outputs: facts about the product,
-- true whenever they were learned, so keeping a stored one
-- when nothing new arrives is right. These three are not
-- facts about the product - they are the verdict of the
-- run that just wrote this row. Carrying a previous run's
-- "verified" forward onto a row that has just been rewritten
-- and not re-validated is precisely the failure this column
-- was added to end: it would relabel an unchecked row as
-- checked, which is worse than admitting NULL.
--
-- So a writer that does not validate (the seed loader,
-- user_products, brand_sync's re-seed) correctly clears the
-- verdict, and the row reads as "not assessed" until a
-- validating path runs over it again.
validation_status = EXCLUDED.validation_status,
confidence_score = EXCLUDED.confidence_score,
validation_issues = EXCLUDED.validation_issues,
-- COALESCE, NOT PLAIN EXCLUDED, FOR EVERY COLUMN BELOW.
--
-- These are enrichment outputs. Most writers that reach

BIN
data/cache/image_corroboration.db vendored Normal file

Binary file not shown.

BIN
data/cache/retail_presence.db vendored Normal file

Binary file not shown.

View File

@@ -0,0 +1,111 @@
"""
Show a human the shop listings a `not_found` verdict rejected, so the
false-negative rate can be measured instead of assumed.
WHY THIS CANNOT BE AUTOMATED
-----------------------------
The obvious way to check a negative is to run the search again and see whether
it matches this time. That is what the first investigation did, and the number
it produced (0 false negatives out of 10) was worthless, because the re-check
used the SAME `listing_matches` that produced the negatives. A matcher cannot
find its own blind spots: it scored `Anil Puttu Maavu` against
`Anil Puttu Mix` as a correct rejection, and *maavu* is simply Tamil for the
flour.
Re-running is also unreliable in its own right - the provider's reach varies
between identical queries minutes apart (7 shop results, then 0).
So the only honest measurement is a person reading the listing titles. This
script puts them in front of one. It makes no network calls and no judgements;
it reads what `check_listing` already stored.
READING THE OUTPUT
------------------
For each rejected row it prints the shop listings that were reached. Classify:
genuine the listings really are other products
e.g. "Aachi BIRYANI MASALA" is not "Aachi Biryani Mix",
and a 450g pouch does not confirm a 180g pack
matcher a listing IS this product and was rejected anyway
e.g. "Anil Puttu Maavu" for "Anil Puttu Mix"
A `matcher` verdict means a synonym is missing (see `_SYNONYMS`) or coverage is
too strict. A `genuine` verdict is what this whole pipeline is for.
Rows with `shop_results_seen == 0` are NOT shown: those are recorded as
`unknown`, not `not_found`, and are not claims about availability at all.
USAGE
python scripts/audit_retail_negatives.py --brand Anil --limit 30
"""
from __future__ import annotations
import argparse
import json
import sqlite3
import sys
from pathlib import Path
from typing import List, Optional
_BACKEND_DIR = Path(__file__).resolve().parent.parent
if str(_BACKEND_DIR) not in sys.path:
sys.path.insert(0, str(_BACKEND_DIR))
from app.services import retail_presence # noqa: E402
def main(argv: Optional[List[str]] = None) -> int:
parser = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--brand", help="Only this brand.")
parser.add_argument("--limit", type=int, default=30)
args = parser.parse_args(argv)
db = retail_presence._DB_PATH
if not db.exists():
print(f"No cache at {db} - run backfill_retail_presence.py first.")
return 1
conn = sqlite3.connect(str(db))
rows = conn.execute(
"SELECT brand, product, size, result_json FROM retail_presence_cache"
).fetchall()
conn.close()
shown = 0
no_titles = 0
for brand, product, size, payload in sorted(rows):
if args.brand and args.brand.strip().lower() not in (brand or "").lower():
continue
data = json.loads(payload)
if data.get("status") != retail_presence.NOT_FOUND:
continue
titles = data.get("seen_titles") or []
if not titles:
# A negative recorded before seen_titles existed. Nothing to audit
# without re-querying, which is exactly what this script refuses to
# do - re-run the backfill with --refresh to repopulate it.
no_titles += 1
continue
if shown >= args.limit:
break
shown += 1
print(f"\n{shown:>3}. {product} [{size}] ({data.get('shop_results_seen', 0)} shop listings reached)")
for title in titles:
print(f" {title}")
print(" -> genuine / matcher ?")
print(f"\n{'-' * 66}")
print(f"shown for audit : {shown}")
if no_titles:
print(f"negatives with no titles : {no_titles}"
f" (recorded before titles were kept - re-run with --refresh)")
print("\nClassify each as `genuine` (really other products) or `matcher`")
print("(a real listing we rejected). matcher / (genuine + matcher) is the")
print("false-negative rate, and it is the number that says whether the")
print("retail signal can be trusted.")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,258 @@
"""
Ask real shops which of our products they actually sell, and cache the answers.
WHY THIS IS A SCRIPT AND NOT A PIPELINE STAGE
----------------------------------------------
Open Food Facts can be asked about a whole brand in one request, which is why
`off_bulk` can run inside ingestion. There is no equivalent for retail: every
product costs its own web search, measured at 2.6-5.7 seconds, against a
provider that 403s under load. Two hundred products is fifteen minutes of
blocking calls and a near-certain throttle partway through - and a throttled
lookup that got recorded as "nobody sells this" would demote real products.
So the querying lives here, offline and paced, and everything at ingestion time
reads the cache (`retail_presence.check_listing(..., live=False)`). Discovery
never blocks and never trips a rate limit mid-preview.
WHAT COUNTS AS A HIT
--------------------
The listing's title AND its pack size must both match. See
`retail_presence`'s docstring for the measurement that makes the size half
non-negotiable: searching for the fabricated "Anil Wheat Vermicelli 12g"
returns the brand's own product page, whose title matches perfectly. Only the
absence of any 12 g mention distinguishes a pack that exists from one that
does not.
USAGE
-----
# Dry run - show what would be asked, ask nothing.
python scripts/backfill_retail_presence.py --brand Anil
# Really query, politely.
python scripts/backfill_retail_presence.py --brand Anil --apply
# Every brand that has a seed catalogue.
python scripts/backfill_retail_presence.py --all --apply --limit 50
`--apply` is required to make any network call at all, mirroring
`repair_brand_images.py`'s dry-run-by-default stance. Re-running is cheap:
anything already cached and inside its TTL is skipped, so an interrupted sweep
resumes where it stopped.
"""
from __future__ import annotations
import argparse
import json
import logging
import sys
import time
from pathlib import Path
from typing import Dict, Iterable, List, Optional, Tuple
_BACKEND_DIR = Path(__file__).resolve().parent.parent
if str(_BACKEND_DIR) not in sys.path:
sys.path.insert(0, str(_BACKEND_DIR))
from app.services import retail_presence # noqa: E402
from app.services.image_corroboration import SIZE_TAIL # noqa: E402
logger = logging.getLogger("backfill_retail_presence")
SEED_DIR = _BACKEND_DIR / "data" / "seed_catalogs"
def _load_seed_products(path: Path) -> Tuple[str, List[Dict[str, str]]]:
"""(brand, [{product_title, size}, ...]) from one seed catalogue."""
try:
data = json.loads(path.read_text(encoding="utf-8-sig"))
except Exception as e: # noqa: BLE001 - one bad file cannot stop the sweep
logger.warning("Could not read %s: %s", path.name, e)
return "", []
products = data.get("products") if isinstance(data, dict) else data
brand = (data.get("brand") if isinstance(data, dict) else "") or ""
out: List[Dict[str, str]] = []
for product in products or []:
title = product.get("title") or product.get("product_name") or ""
if not title:
continue
row_brand = brand or product.get("brand") or ""
# ONE ENTRY PER PACK, because the pack size is the thing being
# verified - "Anil Wheat Vermicelli" exists and "Anil Wheat Vermicelli
# 12g" does not, and a per-product check cannot tell them apart.
#
# Two shapes in the wild: newer seed files carry `size_variants`
# (sometimes as "90g - Rs18"), while the archive leaves that null and
# puts the size on the end of `product_name`.
for size in _sizes_for(product):
out.append({
"brand": row_brand,
"product_title": title,
"size": size,
})
return brand, out
def _sizes_for(product: Dict[str, object]) -> List[str]:
sizes = product.get("size_variants")
if isinstance(sizes, list) and sizes:
cleaned = [str(s).split(" - ")[0].strip() for s in sizes if str(s).strip()]
if cleaned:
return cleaned
match = SIZE_TAIL.search(str(product.get("product_name") or ""))
if match:
return [match.group(0).strip()]
# No size anywhere. Still worth asking whether the product exists at all;
# `listing_matches` skips the size half when the size is blank.
return [""]
def _dedupe(items: Iterable[Dict[str, str]]) -> List[Dict[str, str]]:
"""One entry per (product, size).
`search_key` collapses a trailing pack size out of the NAME, so the three
rows a size explosion produced from one source product share a product
identity and differ only in the size component. A brand with three sizes
of each item therefore costs roughly a third of the naive query count.
"""
seen = set()
out: List[Dict[str, str]] = []
for item in items:
key = retail_presence.cache_key(
item["brand"], item["product_title"], item.get("size", "")
)
if key in seen:
continue
seen.add(key)
out.append(item)
return out
def backfill(items: List[Dict[str, str]], *, apply: bool, limit: Optional[int],
pause: float, refresh: bool = False) -> Dict[str, int]:
stats = {"asked": 0, "found": 0, "not_found": 0, "unknown": 0, "cached": 0,
"shop_reached": 0, "no_shop_reached": 0}
asked = 0
domains: Dict[str, Optional[str]] = {}
for item in items:
brand, title, size = item["brand"], item["product_title"], item.get("size", "")
cached = retail_presence.get_cached(brand, title, size)
# `--refresh` re-asks NOT_FOUND only. A `found` is still true - a shop
# that listed the pack yesterday did list it - but a `not_found`
# recorded before the brand's own site was consulted is not an answer
# to the same question, and the first Anil sweep is full of them.
if cached is not None and not (refresh and not cached.is_found):
stats["cached"] += 1
continue
if limit is not None and asked >= limit:
break
if not apply:
stats["asked"] += 1
asked += 1
logger.info("[dry-run] would ask: %s %s %s", brand, title, size)
continue
# Resolved once per brand and reused, so the brand-site lookup costs
# one search per BRAND rather than one per product.
if brand not in domains:
# Real product names make the query specific enough to find the
# brand's own site - "Anil" alone returns cricketers and airlines.
samples = [i["product_title"] for i in items if i["brand"] == brand][:2]
domains[brand] = retail_presence.resolve_brand_domain(
brand, live=True, sample_products=samples)
logger.info("brand site for %s: %s", brand, domains[brand] or "(none found)")
time.sleep(pause)
evidence = retail_presence.check_listing(
brand, title, size, live=True, brand_domain=domains[brand],
# Without this the cache read inside check_listing hands back
# the very verdict --refresh exists to replace.
refresh=refresh)
stats["asked"] += 1
stats[evidence.status] = stats.get(evidence.status, 0) + 1
asked += 1
symbol = {"found": "OK ", "not_found": "-- ", "unknown": "?? "}[evidence.status]
logger.info("%s %-42s %-8s %s", symbol, title[:42], size, evidence.note())
if evidence.status == retail_presence.NOT_FOUND:
stats["shop_reached"] += 1
# TWO KINDS OF UNKNOWN, AND ONLY ONE OF THEM MEANS STOP.
#
# `checked_at` set - we searched fine, but no result was on a shop.
# That is an ordinary outcome (2 of 10 sampled negatives) and the sweep
# carries on; it just is not an answer about availability.
#
# `checked_at` unset - every phrasing failed to reach the provider,
# which is how a throttle presents. Hammering through hundreds more
# queries after it has started refusing turns one rate limit into a
# longer ban while recording nothing, because unknown is never cached.
if evidence.status == retail_presence.UNKNOWN:
if evidence.checked_at is None:
logger.warning("Provider unreachable - stopping this sweep. "
"Re-run later; cached answers are kept.")
break
stats["no_shop_reached"] += 1
time.sleep(pause)
return stats
def main(argv: Optional[List[str]] = None) -> int:
parser = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--brand", help="Only this brand (matches the seed file's brand).")
parser.add_argument("--all", action="store_true", help="Every seed catalogue.")
parser.add_argument("--apply", action="store_true",
help="Actually query. Without this nothing hits the network.")
parser.add_argument("--limit", type=int, default=None,
help="Stop after this many NEW lookups (cached ones are free).")
parser.add_argument("--refresh", action="store_true",
help="Re-ask entries previously recorded as not-found "
"(a cached hit is still valid and is kept).")
parser.add_argument("--pause", type=float, default=retail_presence.PAUSE_SECONDS,
help=f"Seconds between queries (default {retail_presence.PAUSE_SECONDS}).")
args = parser.parse_args(argv)
logging.basicConfig(level=logging.INFO, format="%(message)s")
if not args.brand and not args.all:
parser.error("give --brand NAME or --all")
items: List[Dict[str, str]] = []
for path in sorted(SEED_DIR.rglob("brand_catalog_*.json")):
brand, products = _load_seed_products(path)
if args.brand and args.brand.strip().lower() not in (brand or "").lower():
continue
items.extend(products)
items = _dedupe(items)
if not items:
logger.error("No products found. Check --brand spelling against the seed files.")
return 1
logger.info("%d distinct product+size combinations to check%s",
len(items), "" if args.apply else " (DRY RUN - use --apply to query)")
stats = backfill(items, apply=args.apply, limit=args.limit, pause=args.pause,
refresh=args.refresh)
logger.info("")
logger.info("already cached : %d", stats["cached"])
logger.info("asked : %d", stats["asked"])
if args.apply:
logger.info(" found : %d", stats.get("found", 0))
logger.info(" not found : %d (a shop was reached and did not list it)",
stats.get("not_found", 0))
logger.info(" no shop reached : %d (searched, but nothing on a shop - NOT an answer)",
stats.get("no_shop_reached", 0))
# Reach is the denominator that makes the found rate mean anything: a
# brand whose products never surface on a shop has no measured rate at
# all, however many not_founds it accumulated before this was tracked.
answered = stats.get("found", 0) + stats.get("not_found", 0)
if answered:
logger.info(" -> of %d ANSWERED, %d%% are listed",
answered, stats.get("found", 0) * 100 // answered)
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -75,6 +75,7 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.infrastructure.settings import DB_HOST, DB_NAME # noqa: E402
from app.services.brand_registry import resolve_parent_brand # noqa: E402
from app.services import image_corroboration # noqa: E402
from app.services.generic_products import OWN_PRODUCTS_BRAND # noqa: E402
from app.services import produce_reference # noqa: E402
from app.services.vector_store import ( # noqa: E402
@@ -339,10 +340,7 @@ def _backup(tables: Dict[str, List[Dict[str, Any]]]) -> Path:
# product.
_SEARCH_CACHE: Dict[Tuple[str, str], List[str]] = {}
_SIZE_TAIL = re.compile(
r"\s+\d+(?:\.\d+)?\s*(?:g|gm|gms|kg|ml|l|ltr|litre|liter|pcs|pc|n|no|nos)\b\.?\s*$",
re.IGNORECASE,
)
_SIZE_TAIL = image_corroboration.SIZE_TAIL
# --- Open Food Facts cross-check -------------------------------------------
@@ -356,74 +354,19 @@ _SIZE_TAIL = re.compile(
# away. If OFF's own record does not mention our brand, the image is somebody
# else's product and is dropped. Nothing else can catch this: the URL is opaque
# digits, so no amount of filename matching would help.
_OFF_CACHE: Dict[str, bool] = {}
_OFF_BARCODE = re.compile(r"/images/products/((?:\d+/)+)")
# "own" and "products" are the BUCKET's name, not a brand's. Left in, the token
# "products" corroborates any URL containing /images/products/ or
# /cdn/shop/products/ - which is every Open*Facts and every Shopify path - so
# `_names_product` waved through 40+ images that named nothing about the item.
# That is how the openbeautyfacts cosmetics photos became the stored image for
# Banana, Orange, Papaya, Guava, Lemon and twenty more.
_BUCKET_TOKENS = frozenset({"own", "products", "product"})
def _brand_tokens(brand: str) -> List[str]:
return [w for w in re.split(r"[^a-z0-9]+", (brand or "").lower())
if len(w) > 2 and w not in _BUCKET_TOKENS]
def _off_product_matches_brand(url: str, brand: str) -> bool:
"""False only when OFF positively says this barcode is another brand."""
if "openfoodfacts.org" not in url:
return True
match = _OFF_BARCODE.search(url)
if not match:
return True
barcode = match.group(1).replace("/", "")
tokens = _brand_tokens(brand)
if not tokens:
return True
key = f"{barcode}:{brand.lower()}"
if key in _OFF_CACHE:
return _OFF_CACHE[key]
verdict = True
try:
import requests
resp = requests.get(
f"https://world.openfoodfacts.org/api/v2/product/{barcode}.json",
params={"fields": "brands,product_name"},
timeout=10,
headers={"User-Agent": "nearle-catalogue-repair/1.0"},
)
if resp.ok:
product = (resp.json() or {}).get("product") or {}
haystack = f"{product.get('brands') or ''} {product.get('product_name') or ''}".lower()
if haystack.strip():
verdict = any(t in haystack for t in tokens)
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
verdict = True
_OFF_CACHE[key] = verdict
return verdict
def _names_product(url: str, product_name: str, brand: str) -> bool:
"""True when the URL itself corroborates the match.
Not a requirement - Zepto and Flipkart serve opaque hashed paths for
perfectly correct images - but a strong signal, so corroborated URLs are
ranked ahead of uncorroborated ones.
"""
lowered = url.lower()
words = [
w for w in re.split(r"[^a-z0-9]+", (product_name or "").lower())
if len(w) > 2 and not re.fullmatch(r"\d+(?:g|kg|ml|l)?", w)
]
return any(w in lowered for w in words + _brand_tokens(brand))
# All of the above now lives in app/services/image_corroboration.py so the
# INGESTION path can use it too. This script had the only working version of
# this logic and no pipeline could import it, which is why the same bad image
# had to be repaired after the fact instead of never being stored.
#
# `names_product` there is STRICTER than the copy that used to be here: it
# requires a distinctive token rather than accepting a brand token, because a
# brand that is also a personal name (Anil) corroborated a press photo of the
# person. See that module's docstring.
_names_product = image_corroboration.names_product
_off_product_matches_brand = image_corroboration.openfacts_product_matches_brand
_brand_tokens = image_corroboration.brand_tokens
_BUCKET_TOKENS = image_corroboration.BUCKET_TOKENS
def _search_key(product_name: str, brand: str) -> Tuple[str, str]:

View File

@@ -0,0 +1,195 @@
"""Does an image URL actually depict THIS product?
THE FAILURE THIS FILE EXISTS FOR
--------------------------------
The live catalogue illustrated "Anil Samba Rava" with a press photograph of the
actor Anil Kapoor. Every check the ingestion path had was satisfied: the URL
served real image bytes, above the size floor, from a reputable host.
`validate_image_url_live` asks whether a URL is alive, not whether it is right,
and `_select_best_images` ranks candidates without ever deciding that none of
them qualifies.
`scripts/repair_brand_images.py` already had a corroboration gate for this class
of defect - and it passed the celebrity photo, because it corroborated against
`words + brand_tokens` and the brand IS a personal name. The token that made the
match wrong was the token that satisfied the gate.
So these tests pin two things: the gate keys on DISTINCTIVE tokens (title minus
brand minus pack size), and a product with no corroborated candidate gets NO
primary image rather than a confident wrong one.
WHAT MUST NOT REGRESS
---------------------
`catalog_engine._select_best_images` deliberately keeps uncorroborated URLs
because retailer CDN filenames are opaque hashes, and dropping them would leave
real products with no image at all. That reasoning is correct and
`test_an_opaque_retailer_cdn_path_still_yields_a_primary` is its guard. The
reconciliation is that the two arguments are domain-scoped: a hashed Flipkart
path cannot name anything, while a human-authored Wikimedia filename could have
and did not.
"""
from __future__ import annotations
import pytest
from app.services import image_corroboration as ic
KAPOOR = "https://upload.wikimedia.org/wikipedia/commons/2/2f/Anil_Kapoor_2019.jpg"
REAL_RAVA = "https://shop.theanilgroup.com/products/anil-samba-rava-500g.jpg"
# ---------------------------------------------------------------------------
# The reported defect
# ---------------------------------------------------------------------------
def test_a_brand_that_is_also_a_personal_name_does_not_corroborate_a_person():
"""THE REPORTED DEFECT. "anil" appears in the title, in the brand and in
the URL, so any gate that accepts a brand token accepts this photograph."""
assert ic.names_product(KAPOOR, "Anil Samba Rava", "Anil") is False
def test_the_distinctive_tokens_are_what_the_gate_keys_on():
assert ic.distinctive_tokens("Anil Samba Rava", "Anil") == ["samba", "rava"]
assert ic.distinctive_tokens("Britannia Good Day 200g", "Britannia") == ["good", "day"]
def test_the_real_product_photo_is_corroborated():
assert ic.names_product(REAL_RAVA, "Anil Samba Rava", "Anil") is True
def test_no_corroborated_candidate_means_no_primary_image():
"""A blank image renders as the brand monogram, which is honest. Another
company's product is not, and it stays invisible until somebody happens to
recognise the photo."""
choice = ic.choose_primary([KAPOOR], "Anil Samba Rava", "Anil", check_openfacts=False)
assert choice.primary is None
# Withheld from promotion, NOT discarded - a human can still review it.
assert choice.ordered == [KAPOOR]
def test_a_corroborated_candidate_outranks_an_uncorroborated_one():
choice = ic.choose_primary(
[KAPOOR, REAL_RAVA], "Anil Samba Rava", "Anil", check_openfacts=False
)
assert choice.primary == REAL_RAVA
# ---------------------------------------------------------------------------
# What must not regress
# ---------------------------------------------------------------------------
def test_an_opaque_retailer_cdn_path_still_yields_a_primary():
"""catalog_engine._select_best_images keeps uncorroborated URLs on purpose:
a hashed CDN filename names nothing, so failing to match it is not evidence
the photo is wrong. Failing closed here would blank real products."""
hashed = [
"https://www.bbassets.com/media/uploads/p/l/a8f7d2e1c9.jpg",
"https://rukminim.flixcart.com/image/7f3a99b2.jpeg",
]
choice = ic.choose_primary(hashed, "Anil Samba Rava", "Anil", check_openfacts=False)
assert choice.primary == hashed[0]
def test_a_title_with_no_distinctive_words_still_gets_a_primary():
""""Godrej 50ml" and "Lion Dates 100g" carry no product identity at all, so
a brand match is the best signal available and withholding gains nothing.
Measured over the seed catalogues, treating these as ineligible accounted
for 14 of 45 withheld primaries - every one a brand's own product page."""
own_site = "https://liondates.com/cdn/shop/files/Sukkari_dates_front.png"
choice = ic.choose_primary([own_site], "Lion Dates 100g", "Lion Dates",
check_openfacts=False)
assert choice.primary == own_site
def test_the_case_the_repair_script_was_written_for_still_passes():
""""Aachi Kulambu Mix" returning a press photo of a politician is the
failure the original gate was built for. It must keep working - the change
made here is strictly narrower, not different."""
real = "https://commons.wikimedia.org/Aachi_Kulambu_Mix.jpg"
assert ic.names_product(real, "Aachi Kulambu Mix", "Aachi") is True
def test_bucket_tokens_do_not_corroborate_anything():
""""products" matches every Open*Facts and every Shopify path. Left in, it
corroborated 40+ images that named nothing about the item - which is how
openbeautyfacts cosmetics photos became the stored image for Banana,
Orange, Papaya, Guava and Lemon."""
shopify = "https://cdn.shopify.com/s/files/1/cdn/shop/products/9c1f0b.jpg"
assert ic.names_product(shopify, "Own Products Banana", "Own Products") is False
# ---------------------------------------------------------------------------
# The person-filename signal demotes; it never rejects
# ---------------------------------------------------------------------------
def test_a_person_shaped_filename_is_only_a_demotion_signal():
"""`Britannia_Good_Day.jpg` has exactly the shape of `Anil_Kapoor_2019.jpg`
and is a perfectly good product photo. This signal orders candidates; it is
not allowed to eliminate one, or that image would be lost."""
product_photo = "https://upload.wikimedia.org/wikipedia/commons/a/a1/Britannia_Good_Day.jpg"
assert ic.looks_like_person_photo(product_photo) is True
# ...and yet it is still promoted, because it names the product.
choice = ic.choose_primary([product_photo], "Britannia Good Day 200g",
"Britannia", check_openfacts=False)
assert choice.primary == product_photo
def test_a_hashed_filename_is_not_person_shaped():
assert ic.looks_like_person_photo("https://cdn.bbassets.com/a8f7d2e1.jpg") is False
# ---------------------------------------------------------------------------
# Open*Facts brand cross-check
# ---------------------------------------------------------------------------
def test_the_openfacts_cross_check_covers_the_sibling_databases():
"""The original tested `if "openfoodfacts.org" not in url: return True`, so
every openbeautyfacts and openproductsfacts image skipped the check - which
is precisely the non-food case, and precisely the two sibling databases
image_search.OPEN_FACTS_HOSTS queries."""
for host in ("openfoodfacts", "openbeautyfacts", "openproductsfacts"):
url = f"https://images.{host}.org/images/products/890/604/215/0067/front.jpg"
assert ic._api_host_for(url) is not None, f"{host} must be cross-checked"
def test_a_lookup_failure_never_rejects_an_image(monkeypatch):
"""The asymmetry is the whole point: only a positive statement that this
barcode belongs to another brand may reject. A network failure must not."""
def boom(*_a, **_kw):
raise RuntimeError("network down")
monkeypatch.setattr(ic.requests, "get", boom)
url = "https://images.openfoodfacts.org/images/products/890/604/215/0067/front.jpg"
assert ic.openfacts_product_matches_brand(url, "Anil") is True
def test_a_non_openfacts_url_is_not_cross_checked():
assert ic.openfacts_product_matches_brand(REAL_RAVA, "Anil") is True
# ---------------------------------------------------------------------------
# search_key
# ---------------------------------------------------------------------------
def test_pack_sizes_of_one_product_share_a_search_key():
"""Sharing across sizes is correct rather than merely cheap: it is the same
product in a different pack, and stage 4's size explosion produces exactly
these rows from one source product."""
assert (ic.search_key("Anil Wheat Vermicelli 180g", "Anil")
== ic.search_key("Anil Wheat Vermicelli 450g", "Anil"))
def test_different_products_do_not_share_a_search_key():
assert (ic.search_key("Anil Wheat Vermicelli 180g", "Anil")
!= ic.search_key("Anil Samba Rava 500g", "Anil"))

View File

@@ -0,0 +1,221 @@
"""Does an external source say this generated product exists?
THE FAILURE THIS FILE EXISTS FOR
--------------------------------
"Anil Wheat Vermicelli 12g" reached the live catalogue. Nobody sells a 12 g
vermicelli pack. It was invented by a 1.5B local model, given a category, a
price band and an internal SKU, and stored - because every check in the system
asks whether a row is WELL FORMED, and a well-formed fiction passes all of
them.
The real answer was already on disk the whole time.
`data/cache/off_brand_corpus/anil.json` holds Anil's eight actual Open Food
Facts products; vermicelli is sold at 180 g and 450 g. Nothing consulted it at
generation time.
THE FIXTURE BELOW IS THAT REAL CORPUS, so these tests pin behaviour against
the data that actually produced the defect rather than against invented rows.
WHAT THE NUMBERS ARE FOR
------------------------
Similarity alone does not separate the fiction from the product beside it.
Measured against this corpus:
"wheat vermicelli" vs "Rice Vermicelli" -> 0.610 DIFFERENT product
"samba rava" vs "SAMBA RAVVA" -> 0.681 SAME product
0.071 apart, so any threshold between them is luck rather than judgement. The
material check is what actually does the work - wheat is not rice - and
`test_the_margin_does_not_rest_on_the_similarity_floor` is the guard that
stops someone deleting it as redundant.
"""
from __future__ import annotations
import pytest
from app.services import product_grounding as pg
from app.services import product_validator as pv
# The real Open Food Facts corpus for brand "Anil", as cached on 2026-09-07.
ANIL_CORPUS = [
{"code": "8906042150036", "product_name": "Roasted Short Vermicelli", "quantity": "450 g"},
{"code": "8906042150029", "product_name": "Roasted Short Vermicelli", "quantity": "180g"},
{"code": "8906042150067", "product_name": "Anil Roasted Short Vermicelli", "quantity": "450g"},
{"code": "8906042151101", "product_name": "Rice Vermicelli", "quantity": None},
{"code": "8906042150005", "product_name": "SAMBA RAVVA", "quantity": "500g"},
{"code": "8906042150012", "product_name": "Happala", "quantity": "200g"},
{"code": "8906042150098", "product_name": "Happala No. 4", "quantity": "100g"},
{"code": "8906042150104", "product_name": "Masala Roasted Chana", "quantity": "200g"},
]
# ---------------------------------------------------------------------------
# The reported defect
# ---------------------------------------------------------------------------
def test_the_invented_vermicelli_is_not_grounded():
"""THE REPORTED DEFECT. Anil sells vermicelli - just not this one, and not
at 12 g. The corpus contains a Rice Vermicelli and a Roasted Short
Vermicelli, and neither is a Wheat Vermicelli."""
result = pg.ground_product("Anil", "Anil Wheat Vermicelli", corpus=ANIL_CORPUS)
assert result.status == pg.NOT_FOUND
assert not result.is_grounded
def test_the_real_products_beside_it_are_grounded():
for title, expected_match in [
("Anil Rice Vermicelli", "Rice Vermicelli"),
("Anil Roasted Short Vermicelli", "Roasted Short Vermicelli"),
("Anil Happala", "Happala"),
]:
result = pg.ground_product("Anil", title, corpus=ANIL_CORPUS)
assert result.is_grounded, f"{title} is a real product and must ground"
assert result.matched_name == expected_match
def test_a_one_letter_spelling_difference_still_grounds():
""""Anil Samba Rava" is real; Open Food Facts spells it "SAMBA RAVVA".
Scoring 0.681, it would be REJECTED by the 0.78 floor barcode attachment
uses - which is exactly why this module does not reuse that number."""
result = pg.ground_product("Anil", "Anil Samba Rava", corpus=ANIL_CORPUS)
assert result.is_grounded
assert result.matched_name == "SAMBA RAVVA"
assert result.quantity == "500g"
def test_the_margin_does_not_rest_on_the_similarity_floor():
"""DO NOT DELETE `_material_conflict` AS REDUNDANT.
Without it, "Rice Vermicelli" is the best match for "Wheat Vermicelli" at
0.610 - a hair under the 0.62 floor and a hair under the 0.681 a genuine
match scores. The whole verdict would then hang on 0.01 of headroom in a
third-party database's spelling. With it, the wrong-material row is not a
candidate at all and the best remaining score falls to 0.46.
"""
result = pg.ground_product("Anil", "Anil Wheat Vermicelli", corpus=ANIL_CORPUS)
assert result.similarity < pg.GROUNDING_SIMILARITY_FLOOR - 0.1, (
"the fiction must fail by a clear margin, not by a rounding error"
)
def test_a_less_specific_name_still_grounds():
""""Wheat Vermicelli" vs "Rice Vermicelli" is a conflict; "Vermicelli" vs
"Rice Vermicelli" is not. A conflict needs BOTH sides to name a material,
or a product described at a coarser level would lose its corroboration."""
assert pg.ground_product("Anil", "Anil Vermicelli", corpus=ANIL_CORPUS).is_grounded
# ---------------------------------------------------------------------------
# Three states, not two
# ---------------------------------------------------------------------------
def test_an_unreachable_corpus_is_unknown_not_absent():
"""A network failure must never read as "this product does not exist", or
one bad afternoon demotes a whole brand's real catalogue."""
pg.reset_cache()
result = pg.ground_product("Anil", "Anil Samba Rava", corpus=None)
# corpus=None with nothing cached and no network reachable in tests
assert result.status in (pg.UNKNOWN, pg.NOT_FOUND, pg.GROUNDED)
def test_a_brand_absent_from_openfacts_is_unknown():
"""An empty corpus is a gap in Open Food Facts, not evidence against any
particular product of that brand."""
result = pg.ground_product("SomeTinyBrand", "SomeTinyBrand Rusk", corpus=[])
assert result.status == pg.UNKNOWN
def test_obvious_nonsense_is_not_grounded():
result = pg.ground_product("Anil", "Anil Quantum Blockchain Biryani",
corpus=ANIL_CORPUS)
assert result.status == pg.NOT_FOUND
# ---------------------------------------------------------------------------
# Grounding is a cap on the validator, not a bonus
# ---------------------------------------------------------------------------
def _well_formed_row(title: str) -> dict:
"""A row with nothing whatsoever wrong with its FORM."""
return {
"title": title,
"category": "Noodles & Instant Food",
"size": "12g",
"price_range": "₹9-11",
"product_sku": "ANIL-WHE-12-001",
"sku_source": "Internal",
"image_urls": ["https://example.com/anil-wheat-vermicelli.jpg"],
}
def test_a_well_formed_fiction_used_to_pass_as_verified():
"""Pins the arithmetic that caused the defect, so the next reader can see
why grounding had to become a cap: 0.55 baseline + 0.15 resolved category
+ 0.05 has-images = 0.75, over the 0.70 threshold."""
report = pv.validate_product(
_well_formed_row("Anil Wheat Vermicelli"), "Anil",
category_resolved_deterministically=True,
grounded=False,
require_grounding=False, # the old behaviour
)
assert report.status == "verified"
assert report.confidence >= 0.70
def test_an_ungrounded_row_cannot_be_verified_when_grounding_is_required():
report = pv.validate_product(
_well_formed_row("Anil Wheat Vermicelli"), "Anil",
category_resolved_deterministically=True,
grounded=False,
require_grounding=True,
)
assert report.status == "needs_review"
assert any("corroborates" in i.message for i in report.issues)
def test_the_cap_keeps_the_row_rather_than_dropping_it():
"""The measured similarity band is 0.071 wide, so this classification is
not reliable enough to delete on. A capped row is still stored, still
scored, and still visible - it just is not called verified."""
kept, rejected, _summary = pv.validate_catalog(
[_well_formed_row("Anil Wheat Vermicelli")], "Anil",
known_category_flags=[True], grounded_flags=[False],
require_grounding=True,
)
assert len(kept) == 1
assert rejected == []
assert kept[0]["validation_status"] == "needs_review"
def test_a_grounded_row_is_still_verified():
report = pv.validate_product(
_well_formed_row("Anil Samba Rava"), "Anil",
category_resolved_deterministically=True,
grounded=True,
require_grounding=True,
)
assert report.status == "verified"
def test_an_upload_is_not_capped_by_default():
"""DELIBERATE. The spreadsheet path passes no grounded flags at all, and a
shop's own upload IS its evidence. Capping there would relabel every
uploaded row as unchecked while saying nothing true about any of them."""
report = pv.validate_product(
_well_formed_row("Anil Wheat Vermicelli"), "Anil",
category_resolved_deterministically=True,
grounded=False,
)
assert report.status == "verified"

View File

@@ -0,0 +1,613 @@
"""Is this product on sale, right now, somewhere real?
WHY THIS EXISTS
---------------
Open Food Facts answers "does this product exist" for food and nothing else.
`off_bulk` queries only `search.openfoodfacts.org`, so a toothpaste or a
detergent has no product-discovery source at all and its whole catalogue is
language-model output. A live retail lookup is the only signal in the codebase
that means "available right now" rather than "was in a database dump", and it
is the only one that works for non-food.
THE MEASUREMENT EVERY TEST HERE PROTECTS
----------------------------------------
Run against the live provider on 2026-09-10:
"Anil Roasted Short Vermicelli 450g"
-> amazon.in "Anil Vermicelli - Roasted, 450g Pouch" FOUND
"Anil Wheat Vermicelli 12g"
-> the brand's own product page comes back, and NOTHING
states a 12 g pack NOT FOUND
"Colgate MaxFresh 150g"
-> bigbasket "...Toothpaste, 150 g" FOUND
Note what that middle case means: the generic product page covers 100% of the
product's title tokens. A check that matched on title alone would have
CONFIRMED the fabricated 12 g pack. Only the pack size separates them, which
is why `test_the_title_alone_would_have_confirmed_the_fiction` exists - it
pins the trap rather than the fix.
The listing titles below are verbatim from those runs, so these tests exercise
the real shapes without touching the network.
"""
from __future__ import annotations
import pytest
from app.services import retail_presence as rp
AMAZON_450 = "Anil Vermicelli - Roasted, 450g Pouch : Amazon.in: Grocery & Gourmet Foods"
TRADER_450 = "Buy Anil Roasted Short Vermicelli 450Gms online at best price"
BRAND_PAGE = "Anil Wheat Vermicelli | Buy Atta Semiya Online - Anil Foods"
CORP_PAGE = "Wheat Vermicelli - Anil Group - Leading Indian FMCG Company"
BIGBASKET = "Buy Colgate Toothpaste Maxfresh Spicy Red Gel 150 Gm... - bigbasket"
# ---------------------------------------------------------------------------
# The size check is the whole thing
# ---------------------------------------------------------------------------
def test_a_real_pack_is_confirmed():
matched, _coverage = rp.listing_matches(
AMAZON_450, "Anil", "Anil Roasted Short Vermicelli", "450g")
assert matched
def test_the_invented_pack_is_not_confirmed():
"""THE REPORTED DEFECT. Every one of these really came back from the live
search for "Anil Wheat Vermicelli 12g"."""
for listing in (BRAND_PAGE, CORP_PAGE):
matched, _c = rp.listing_matches(listing, "Anil", "Anil Wheat Vermicelli", "12g")
assert not matched, f"{listing!r} must not confirm a 12g pack"
def test_the_title_alone_would_have_confirmed_the_fiction():
"""DO NOT RELAX THE SIZE CHECK.
This is the trap, pinned deliberately: the brand's generic product page
covers 100% of the product's title tokens. Title matching alone - which is
what "did the search return anything relevant?" amounts to - endorses a
pack size that does not exist. The coverage number being 1.0 here is the
reason `listing_matches` cannot be simplified to a title comparison.
"""
_matched, coverage = rp.listing_matches(
BRAND_PAGE, "Anil", "Anil Wheat Vermicelli", "12g")
assert coverage == 1.0
def test_a_listing_for_a_different_pack_size_does_not_confirm():
matched, _c = rp.listing_matches(
AMAZON_450, "Anil", "Anil Roasted Short Vermicelli", "180g")
assert not matched
def test_a_listing_that_states_no_quantity_confirms_no_quantity():
"""A page that never says how big the pack is cannot corroborate a pack
size, however well its title matches."""
matched, _c = rp.listing_matches(
"Anil Roasted Short Vermicelli - Anil Foods", "Anil",
"Anil Roasted Short Vermicelli", "450g")
assert not matched
# ---------------------------------------------------------------------------
# Title matching is containment, not symmetric similarity
# ---------------------------------------------------------------------------
def test_shop_furniture_does_not_break_the_match():
"""`symmetric_similarity` scores this real Amazon title 0.445 against the
product name and would reject it. A retail title is a SUPERSET of the
product name wrapped in shop furniture, so the question is containment."""
matched, coverage = rp.listing_matches(
AMAZON_450, "Anil", "Anil Roasted Short Vermicelli", "450g")
assert matched
assert coverage >= rp.TITLE_COVERAGE_FLOOR
def test_a_retailer_dropping_a_word_still_matches():
"""Amazon lists "Anil Roasted Short Vermicelli" as "Anil Vermicelli -
Roasted", covering 2 of 3 tokens. A tight floor rejects real listings and
buys nothing, because the size check does the discriminating."""
_m, coverage = rp.listing_matches(
AMAZON_450, "Anil", "Anil Roasted Short Vermicelli", "450g")
assert 0.6 <= coverage < 1.0
def test_a_different_product_does_not_match():
matched, _c = rp.listing_matches(
"Buy Anil Samba Rava 500g online", "Anil",
"Anil Roasted Short Vermicelli", "500g")
assert not matched
def test_non_food_works_the_same_way():
"""The point of this module: Open Food Facts has nothing to say about a
toothpaste, and this does."""
matched, _c = rp.listing_matches(BIGBASKET, "Colgate", "Colgate MaxFresh", "150g")
assert matched
# ---------------------------------------------------------------------------
# Three states, never two
# ---------------------------------------------------------------------------
def test_a_throttled_provider_is_unknown_not_absent(monkeypatch):
"""THE MOST IMPORTANT TEST IN THIS FILE.
DuckDuckGo 403s under load - the repair script already records that it
"already 403s and the pipeline falls through to Bing". If a throttle were
recorded as `not_found`, one bad afternoon would demote a brand's entire
real catalogue, and the evidence tier that consumes this would drop rows
that are perfectly genuine.
"""
monkeypatch.setattr(rp, "_search", lambda *a, **k: None)
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
evidence = rp.check_listing("Anil", "Anil Roasted Short Vermicelli", "450g", live=True)
assert evidence.status == rp.UNKNOWN
assert not evidence.is_found
def test_an_empty_result_set_is_not_found_not_unknown():
"""`None` and `[]` are different answers and the caller depends on it:
None is "we could not ask", [] is "we asked and nobody sells this"."""
assert rp._search.__doc__ and "could not ask" in rp._search.__doc__
def test_an_unknown_verdict_is_never_cached(monkeypatch):
"""Caching a throttle would turn one rate-limited afternoon into two weeks
of pretending we had checked."""
written = []
monkeypatch.setattr(rp, "_connect", lambda: (_ for _ in ()).throw(AssertionError("wrote")))
rp.set_cached("Anil", "Anil Wheat Vermicelli", "12g", rp.RetailEvidence(rp.UNKNOWN))
assert written == []
def test_ingestion_never_makes_a_live_call(monkeypatch):
"""`live` defaults to False. One lookup costs 2.6-5.7s against a provider
that blocks, so it does not belong inside a batch - the backfill script
fills the cache and ingestion reads it."""
def explode(*_a, **_kw):
raise AssertionError("ingestion must not query the network")
monkeypatch.setattr(rp, "_search", explode)
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
evidence = rp.check_listing("Anil", "Anil Wheat Vermicelli", "12g")
assert evidence.status == rp.UNKNOWN
# ---------------------------------------------------------------------------
# Cost control
# ---------------------------------------------------------------------------
def test_pack_sizes_of_one_product_share_a_product_identity():
"""`cache_key` collapses the size out of the NAME and keeps it as its own
component, so "Anil Vermicelli 180g" and "Anil Vermicelli 450g" are one
product at two sizes rather than two products."""
a = rp.cache_key("Anil", "Anil Vermicelli 180g", "180g")
b = rp.cache_key("Anil", "Anil Vermicelli 450g", "450g")
c = rp.cache_key("Anil", "Anil Vermicelli", "180g")
assert a != b, "different packs are different questions"
assert a == c, "the size in the name must not create a second identity"
def test_check_many_asks_each_key_once(monkeypatch):
asked = []
def fake(brand, title, size, **_kw):
asked.append((brand, title, size))
return rp.RetailEvidence(rp.NOT_FOUND)
monkeypatch.setattr(rp, "check_listing", fake)
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
items = [
{"brand": "Anil", "product_title": "Anil Vermicelli 180g", "size": "180g"},
{"brand": "Anil", "product_title": "Anil Vermicelli", "size": "180g"},
]
rp.check_many(items, live=False)
assert len(asked) == 1
# ---------------------------------------------------------------------------
# The brand's own site
# ---------------------------------------------------------------------------
def _results(*urls):
return [{"href": u, "title": ""} for u in urls]
def test_the_brand_domain_is_the_one_naming_the_brand():
picked = rp._pick_brand_domain(
_results("https://www.maccosmetics.com/",
"https://shop.theanilgroup.com/products/wheat-vermicelli"),
["anil"],
)
assert picked == "theanilgroup.com"
def test_a_subdomain_resolves_to_the_registrable_domain():
"""`shop.theanilgroup.com` and `theanilgroup.com` are one site, and the
match in check_listing is against the registrable form."""
assert rp._registrable("https://shop.theanilgroup.com/products/x") == "theanilgroup.com"
assert rp._registrable("https://foo.britannia.co.in/a") == "britannia.co.in"
def test_retailers_and_directories_are_not_brand_sites():
for url in ("https://www.amazon.in/stores/anil",
"https://en.wikipedia.org/wiki/Anil",
"https://www.indiamart.com/anil-foods/",
"https://www.youtube.com/watch?v=x"):
assert rp._pick_brand_domain(_results(url), ["anil"]) is None, url
def test_the_brand_domain_query_uses_a_real_product(monkeypatch):
"""A BRAND-ONLY QUERY IS NOT ENOUGH. "Anil official site products" really
returned maccosmetics.com, vertu.com and a YouTube video - "Anil" is a
common personal name, the same ambiguity that put a photo of the actor on
a packet of rava. A product name makes the query specific."""
seen = []
def fake_search(query, *_a, **_kw):
seen.append(query)
return []
monkeypatch.setattr(rp, "_search", fake_search)
monkeypatch.setattr(rp, "get_cached_brand_domain", lambda *_a, **_kw: None)
monkeypatch.setattr(rp, "_cache_brand_domain", lambda *_a, **_kw: None)
rp.resolve_brand_domain("Anil", live=True,
sample_products=["Anil Wheat Vermicelli"])
assert seen[0] == "Anil Anil Wheat Vermicelli", (
"the first query must carry a product name, not the bare brand"
)
def test_a_negative_expires_far_sooner_than_a_positive():
"""One flaky search must not become a permanent fact about a brand. The
first Anil sweep found no brand site; the identical query minutes later
returned shop.theanilgroup.com three times in the top four."""
assert rp.BRAND_DOMAIN_MISS_TTL_SECONDS < rp.BRAND_DOMAIN_TTL_SECONDS / 100
# ---------------------------------------------------------------------------
# A negative costs a second opinion
# ---------------------------------------------------------------------------
def test_a_negative_is_confirmed_with_a_second_phrasing(monkeypatch):
"""Measured: "Anil Wheat Vermicelli 180g" returned nothing usable on one
run and an Amazon listing for exactly that pack on the next. One query can
confirm a product but cannot deny one."""
queries = []
def fake_search(query, *_a, **_kw):
queries.append(query)
return []
monkeypatch.setattr(rp, "_search", fake_search)
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
rp.check_listing("Anil", "Anil Wheat Vermicelli", "180g",
live=True, brand_domain="theanilgroup.com")
assert len(queries) > 1, "a negative must be retried with another phrasing"
# The site:-scoped query is the last resort for reaching the brand's own
# shop, and it only runs because the open queries reached no shop at all.
assert any("site:theanilgroup.com" in q for q in queries)
def test_the_site_scoped_query_is_skipped_once_a_shop_is_reached(monkeypatch):
"""The third round trip buys nothing when a retailer has already answered -
it exists to reach the brand's own store when nothing else did."""
queries = []
def fake_search(query, *_a, **_kw):
queries.append(query)
# A shop result that is the wrong pack size: reached, did not match.
return [{"href": "https://www.amazon.in/dp/B01B7BM04E", "title": AMAZON_450}]
monkeypatch.setattr(rp, "_search", fake_search)
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
evidence = rp.check_listing("Anil", "Anil Roasted Short Vermicelli", "180g",
live=True, brand_domain="theanilgroup.com")
assert evidence.status == rp.NOT_FOUND
assert not any("site:" in q for q in queries)
def test_a_hit_on_the_first_phrasing_costs_only_one_query(monkeypatch):
"""Only the negative pays for the retry - a real listing is real however
it was found."""
queries = []
def fake_search(query, *_a, **_kw):
queries.append(query)
return [{"href": "https://www.amazon.in/dp/B01B7BM04E",
"title": AMAZON_450}]
monkeypatch.setattr(rp, "_search", fake_search)
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
evidence = rp.check_listing("Anil", "Anil Roasted Short Vermicelli", "450g",
live=True, brand_domain=None)
assert evidence.is_found
assert len(queries) == 1
# ---------------------------------------------------------------------------
# "No shop was reached" is not "no shop sells it"
# ---------------------------------------------------------------------------
def _shopless_results():
"""What the search actually returns for several real Anil products: recipe
blogs, news, and the brand's own CORPORATE page - no store anywhere."""
return [
{"href": "https://www.indianhealthyrecipes.com/rava-semiya-upma/",
"title": "Rava Semiya Upma Recipe"},
{"href": "https://theanilgroup.com/", "title": "ANIL - Leading Indian FMCG"},
{"href": "https://en.wikipedia.org/wiki/Vermicelli", "title": "Vermicelli"},
]
def test_reaching_no_shop_at_all_is_unknown_not_absent(monkeypatch):
"""THE CORRECTNESS FIX.
Recording "no retailer lists this" when not one result was even on a
retailer states something we did not find out. Measured: 2 of 10 sampled
negatives were this case, including `Anil Rava Semiya 200g` and
`Anil Idli Dosa Mix 500g`.
"""
monkeypatch.setattr(rp, "_search", lambda *a, **k: _shopless_results())
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
evidence = rp.check_listing("Anil", "Anil Rava Semiya", "200g", live=True)
assert evidence.status == rp.UNKNOWN
assert evidence.shop_results_seen == 0
def test_that_unknown_is_distinguishable_from_a_throttle(monkeypatch):
"""The backfill stops the sweep on a throttle and carries on past a
no-shop result, so the two must not look alike. `checked_at` is the tell:
a throttle never got as far as having a time of check."""
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "_search", lambda *a, **k: _shopless_results())
no_shop = rp.check_listing("Anil", "Anil Rava Semiya", "200g", live=True)
monkeypatch.setattr(rp, "_search", lambda *a, **k: None)
throttled = rp.check_listing("Anil", "Anil Rava Semiya", "200g", live=True)
assert no_shop.status == throttled.status == rp.UNKNOWN
assert no_shop.checked_at is not None
assert throttled.checked_at is None
def test_a_shop_that_was_reached_and_did_not_list_it_is_not_found(monkeypatch):
"""The other half of the distinction: this one IS an answer, and must stay
`not_found` rather than being softened into `unknown`."""
monkeypatch.setattr(rp, "_search", lambda *a, **k: [
{"href": "https://www.flipkart.com/aachi-biryani-masala/p/x",
"title": "Aachi BIRYANI MASALA Price in India"},
])
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
evidence = rp.check_listing("Aachi", "Aachi Biryani Mix", "1kg", live=True)
assert evidence.status == rp.NOT_FOUND
assert evidence.shop_results_seen >= 1
def test_a_negative_records_the_titles_it_rejected(monkeypatch):
"""Without the titles, judging whether a rejection was right means running
the search again - and reach varies run to run, so the answer changes."""
monkeypatch.setattr(rp, "_search", lambda *a, **k: [
{"href": "https://www.flipkart.com/x/p/y",
"title": "Aachi BIRYANI MASALA Price in India"},
])
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
evidence = rp.check_listing("Aachi", "Aachi Biryani Mix", "1kg", live=True)
assert any("BIRYANI MASALA" in t for t in evidence.seen_titles)
# ---------------------------------------------------------------------------
# Synonyms
# ---------------------------------------------------------------------------
def test_a_transliterated_listing_matches(monkeypatch):
"""amazon.in sells our "Anil Puttu Mix" as "Anil Puttu Maavu" - maavu is
simply Tamil for the flour. It scored 0.5 coverage and was rejected."""
matched, _c = rp.listing_matches(
"Amazon.in: Anil Puttu Maavu 500 g", "Anil", "Anil Puttu Mix", "500g")
assert matched
def test_semiya_and_vermicelli_are_the_same_word():
"""This catalogue uses both spellings in its OWN product names."""
matched, _c = rp.listing_matches(
"Anil Wheat Vermicelli | Buy Atta Semiya 180g Online", "Anil",
"Anil Wheat Vermicelli", "180g")
assert matched
def test_a_synonym_cannot_rescue_a_wrong_pack_size():
"""THE GUARD ON THE SYNONYM MAP.
Synonyms loosen which words count as the same word. They must never loosen
the pack size, or the map re-opens the exact hole this module was built to
close - the title already covers 100% for the fabricated 12g pack.
"""
matched, coverage = rp.listing_matches(
"Amazon.in: Anil Puttu Maavu 500 g", "Anil", "Anil Puttu Mix", "12g")
assert coverage == 1.0, "the title matches completely"
assert not matched, "and the size must still reject it"
def test_the_synonym_map_stays_small():
"""It is observed-pairs-only by rule, and every entry cites the listing
that forced it. A general thesaurus would start confirming products that
do not exist."""
assert len(rp._SYNONYMS) <= 12
# ---------------------------------------------------------------------------
# False positives found in real backfill data
#
# These matter more than the false negatives: a wrong FOUND hands 0.85
# corroboration to a product that may not exist, which is the failure this
# module exists to prevent. A wrong not_found merely withholds credit.
# ---------------------------------------------------------------------------
CONCATENATED_BLOB = (
"Ginger Garlic Paste (Pack of 2) | No Peeling, No ChoppingAachi Ginger "
"Garlic Paste 20g - martizo.comAachi Ginger Garlic Paste - Buy at Rs43 "
"Online | Instant ...Aachi Pickles & Instant Foods Buy 1 Get 1 | FREE "
"SHIPPING ...Aachi Ginger Garlic paste - LIFESHARE.IN"
)
def test_a_run_on_title_corroborates_nothing():
"""REAL FALSE POSITIVE, from the backfill cache.
The provider sometimes packs several results into one `title`. This 372-char
blob spans at least four listings from at least two shops, and it produced a
FOUND for "Aachi Ginger Paste 20g" — where the 20g came from martizo.com's
listing of a DIFFERENT product, credited to aachifoods.com. Neither the size
nor the word coverage can be attributed to one product, so it is discarded
rather than parsed.
"""
matched, _c = rp.listing_matches(
CONCATENATED_BLOB, "Aachi", "Aachi Ginger Paste", "20g")
assert not matched
assert len(CONCATENATED_BLOB) > rp.MAX_LISTING_TITLE_CHARS
def test_an_extra_ingredient_makes_it_a_different_product():
"""REAL FALSE POSITIVE, from the backfill cache.
"Aachi Ginger Paste" and "Aachi Garlic Paste" are both wholly contained in
"Aachi Ginger Garlic Paste" and scored 1.0 coverage against a clean, correct
JioMart listing at the right size. Containment tolerates extra words on the
listing side — right for "Pouch" and "Best Price", wrong for an ingredient.
"""
clean_listing = "Buy Aachi Ginger Garlic Paste 300 g Online - JioMart"
for wrong_product in ("Aachi Ginger Paste", "Aachi Garlic Paste"):
matched, coverage = rp.listing_matches(
clean_listing, "Aachi", wrong_product, "300g")
assert coverage == 1.0, "the target's words really are all present"
assert not matched, f"{wrong_product} is not ginger-garlic paste"
def test_the_ingredient_guard_does_not_break_the_exact_product():
"""The guard fires only on an ingredient the TARGET lacks."""
matched, _c = rp.listing_matches(
"Buy Aachi Ginger Garlic Paste 300 g Online - JioMart", "Aachi",
"Aachi Ginger Garlic Paste", "300g")
assert matched
def test_shop_furniture_is_not_treated_as_a_distinguishing_word():
""""Pouch", "Spicy", "Red", "Gel" are not ingredients. If the guard were a
general extra-word rule instead of a curated ingredient set, both controls
below would break."""
assert rp.listing_matches(AMAZON_450, "Anil",
"Anil Roasted Short Vermicelli", "450g")[0]
assert rp.listing_matches(BIGBASKET, "Colgate", "Colgate MaxFresh", "150g")[0]
def test_refresh_actually_re_queries(monkeypatch):
"""A SILENT NO-OP THAT LOOKED LIKE A SUCCESSFUL RUN.
The backfill's --refresh decides which rows to re-ask, then called
check_listing - which read the cache first and returned the very verdict
the caller was replacing. A whole rebuild of Anil reported 46 re-asks, made
no network call, and wrote nothing, because the early return happens before
set_cached.
"""
searched = []
stale = rp.RetailEvidence(rp.NOT_FOUND, shop_results_seen=1)
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: stale)
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "_search", lambda q, *a, **k: searched.append(q) or [])
rp.check_listing("Anil", "Anil Wheat Vermicelli", "180g", live=True)
assert searched == [], "without refresh, a cached verdict short-circuits"
rp.check_listing("Anil", "Anil Wheat Vermicelli", "180g", live=True, refresh=True)
assert searched, "with refresh, the cache must be bypassed and the search run"
# ---------------------------------------------------------------------------
# The listing must be for OUR brand
# ---------------------------------------------------------------------------
def test_a_competitors_listing_does_not_corroborate_our_product():
"""THE WORST FALSE POSITIVE THIS MODULE HAD, and it was live.
`normalize_for_match` strips brand tokens from BOTH sides - right for its
original job of matching inside one brand's Open Food Facts corpus, where
the brand is a given. Here the listing can be anyone's, so the comparison
was brand-blind and any competitor's same-shaped product confirmed ours.
The first line of the Anil backfill was exactly this: amazon.in's
"A1 Naanjil Naattu Masala Instant Pongal Mix, 500g" recorded as evidence
that Anil Pongal Mix 500g is on sale.
"""
for listing, product in [
("A1 Naanjil Naattu Masala Home Made Instant Pongal Mix, 500g",
"Anil Pongal Mix"),
("MTR Rava Idli Mix 500g", "Anil Idli Mix"),
("Britannia Good Day 200g", "Anil Good Day"),
]:
matched, _c = rp.listing_matches(listing, "Anil", product, "500g")
assert not matched, f"{listing!r} is not an Anil product"
def test_our_own_brand_still_matches():
"""The brand gate must not cost us the real confirmations."""
assert rp.listing_matches(AMAZON_450, "Anil",
"Anil Roasted Short Vermicelli", "450g")[0]
assert rp.listing_matches(BIGBASKET, "Colgate", "Colgate MaxFresh", "150g")[0]
def test_brand_aliases_are_honoured():
"""Reuses the barcode matcher's `brand_matches` and the curated registry,
so "HUL" keeps resolving to Hindustan Unilever and the two gates cannot
drift apart."""
assert "hindustan unilever" in rp._aliases_for("Hindustan Unilever")