product generation with validation check

This commit is contained in:
sriram
2026-09-10 16:17:28 +05:30
parent 10b24c6348
commit d5a23f6456
19 changed files with 3381 additions and 121 deletions

View File

@@ -18,13 +18,17 @@ import logging
sys.path.append(str(Path(__file__).parent.parent))
from app.services.ollama_service import fetch_brand_catalog_with_gemini, fetch_brand_catalog_exhaustive, fetch_product_details
from app.services.image_search import find_all_image_urls, find_product_quantity_openfacts
from app.services.image_search import find_all_image_urls
from app.infrastructure.settings import DATA_DIR, USE_OLLAMA
from app.services.embeddings_service import embed_texts
from app.services.vector_store import ensure_brand_schema, upsert_brand_products, get_existing_product_image_id
from app.services.s3_service import s3_service
from app.services.brand_registry import resolve_parent_brand
from app.services import image_corroboration
from app.services import price_estimator
from app.services import product_grounding
from app.services.enrichment.barcode.sources import off_bulk
from app.services.product_validator import validate_catalog
from app.services.category_registry import detect_category_from_text, sanitize_category_language
# Configure logging
@@ -611,15 +615,43 @@ class ProductCatalogEngine:
# Step 2: Enhance each product with comprehensive image search
enhanced_products = []
# Parallel to enhanced_products: did an external source corroborate
# that each product exists? Collected here and handed to
# validate_catalog at the end - see the Step 3 comment for why this
# path had no validation gate at all until now.
grounded_flags: List[bool] = []
# Cap products by max_products
discovered_products = discovered_products[:max_products]
# Fetched ONCE for the whole brand, not once per product: this is a
# disk-cached whole-catalogue fetch (1-5 requests per brand), which is
# the entire reason it is used here instead of the per-product live
# search that looks like the natural fit. See product_grounding's
# docstring - that endpoint is rate-limited to 10 requests/minute and
# was returning 503 when measured.
try:
brand_corpus = off_bulk.fetch_brand_corpus(brand)
except Exception as e: # noqa: BLE001 - unreachable corpus is not a verdict
logger.warning("Open*Facts corpus unavailable for %s: %s", brand, e)
brand_corpus = None
if brand_corpus is not None:
logger.info("📚 Open*Facts corpus for %s: %d real products", brand, len(brand_corpus))
total_products_to_process = len(discovered_products)
for i, product in enumerate(discovered_products):
product_title = product.get('title', '')
logger.info(f"🔍 Processing product {i+1}/{total_products_to_process}: {product_title}")
# Does anything outside this process say this product exists?
# The LLM that produced `product_title` is a 1.5B local model and
# cannot be asked to check its own work.
grounding = product_grounding.ground_product(
brand, product_title, corpus=brand_corpus
)
if grounding.status == product_grounding.NOT_FOUND:
logger.warning("⚠️ Ungrounded: %r - %s", product_title, grounding.note())
# Strip brand prefix from title if present (avoids redundant
# "Cadbury Perk Cadbury Perk Crunch" style queries that confuse
# image search APIs).
@@ -670,7 +702,26 @@ class ProductCatalogEngine:
# Select exactly 20 best images
all_prioritized = prioritized_images + other_images
final_images = self._select_best_images(all_prioritized, product_title, brand, max_images=20)
# _select_best_images ORDERS; it does not decide whether any
# candidate is this product. Applied BEFORE the S3 upload below so a
# photo of somebody else is never copied into our own bucket, where
# its origin stops being visible at all.
#
# Nothing is dropped - the whole list is still uploaded and stored,
# so an operator can look. Only the promotion to `image_url` is
# withheld, and only when no candidate corroborates the product.
image_choice = image_corroboration.choose_primary(
final_images, product_title, brand or ""
)
final_images = list(image_choice.ordered)
primary_eligible = image_choice.primary is not None
if final_images and not primary_eligible:
logger.warning(
"No corroborated image for %r (%s) - storing candidates but "
"leaving image_url empty", product_title, image_choice.reason,
)
# Check if product already exists in DB to avoid duplicates
image_id_val = ""
try:
@@ -836,18 +887,19 @@ class ProductCatalogEngine:
if not _variant_size_price_pairs:
default_sizes = price_estimator.default_size_variants(category_value, product_title)
# Ground this in a real packaging size where possible:
# Open Food/Beauty/Products Facts reports an actual
# `quantity` field (e.g. "200 g", "1 l") for products it
# has on file, which is far more trustworthy than the
# category preset list (itself just a fallback for when
# *nothing* else is known). If we have one, swap it in for
# the closest preset rather than presenting a size that may
# not actually exist for this exact product.
try:
real_qty = find_product_quantity_openfacts(product_title, brand)
except Exception:
real_qty = None
# Ground this in a real packaging size where possible. The
# quantity comes from the brand corpus fetched once above,
# NOT from find_product_quantity_openfacts, which issues a
# live per-product query against an endpoint rate-limited to
# 10 requests/minute - see product_grounding's docstring.
#
# ONLY WHEN THE PRODUCT ITSELF IS CORROBORATED. A quantity
# borrowed from a product that merely scored well is worse
# than the preset it replaces: it makes a fabricated row look
# MORE real by dressing it in a size that genuinely exists.
# An ungrounded product keeps the honest category preset and
# is flagged for review instead.
real_qty = grounding.quantity if grounding.is_grounded else None
if real_qty and real_qty not in default_sizes:
default_sizes = [real_qty] + default_sizes[:2]
@@ -915,10 +967,16 @@ class ProductCatalogEngine:
# is only used as a last resort since small local models frequently
# hallucinate image links that don't actually resolve to an image.
all_image_urls = s3_uploaded_urls or final_images
primary_image = all_image_urls[0] if all_image_urls else (
# `primary_eligible` is the corroboration verdict computed above.
# S3 preserves candidate order (image_000, image_001, ...), so
# index 0 here is the same picture choose_primary judged.
primary_image = (all_image_urls[0] if (all_image_urls and primary_eligible) else None) or (
enriched_img if enriched_img and str(enriched_img).startswith('http') else None
)
if not primary_image and s3_service.enabled and image_id_val:
# Only reachable when there was no usable candidate at all. Guarded
# by `primary_eligible` too, because this constructs a URL to
# image_000 - the very image corroboration just declined to promote.
if not primary_image and primary_eligible and s3_service.enabled and image_id_val:
primary_image = s3_service.get_product_image_url(brand, image_id_val)
if not all_image_urls and primary_image:
all_image_urls = [primary_image]
@@ -958,8 +1016,36 @@ class ProductCatalogEngine:
}
enhanced_products.append(enhanced_product)
grounded_flags.append(grounding.is_grounded)
logger.info(f"✅ Enhanced {product_title}: {len(final_images)} images")
# Step 2.6: THE DETERMINISTIC VALIDATION GATE.
#
# This path did not have one. `validate_catalog` was called from the
# spreadsheet pipeline and from the Dagster assets, but never from
# here, so POST /api/catalog/generate wrote straight to the brand
# table with nothing between the language model and the database -
# while product_validator's own docstring claimed this call site
# already existed. That claim has been corrected along with this fix.
#
# Rows are annotated and kept, not dropped: `validate_catalog` returns
# "verified" and "needs_review" rows together, and A1 gave the brand
# tables somewhere to record which is which. Only rows scoring below
# the reject threshold are withheld.
enhanced_products, rejected, summary = validate_catalog(
enhanced_products, brand, grounded_flags=grounded_flags,
# These rows came out of a 1.5B language model, so "well formed"
# is not evidence of anything. A row nothing corroborates is kept
# and scored, but capped at needs_review rather than presented as
# verified.
require_grounding=True,
)
if rejected:
logger.warning(
"🚫 %d of %d generated product(s) failed validation for %s",
len(rejected), summary.get("total_evaluated", 0), brand,
)
# Step 3: Generate final catalog
catalog = {
'brand': brand,

View File

@@ -69,6 +69,7 @@ from app.api.routers.user_products import (
read_products_dataframe,
row_to_request,
)
from app.services import image_corroboration
from app.services import price_estimator
from app.services.brand_registry import (
BRAND_ALIASES,
@@ -587,8 +588,25 @@ def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, An
candidates, product_name, brand, max_images=10
)
if best:
row["image_urls"] = list(best)
row["image_url"] = best[0]
# `_select_best_images` ORDERS candidates; it does not judge
# whether any of them is this product. Its scoring awards points
# for the brand appearing in the URL, so for a brand that is
# also a personal name it actively rewarded the wrong photo -
# `Anil_Kapoor_2019.jpg` outscored everything and became
# `image_url`. choose_primary decides what may be promoted.
#
# The list is still stored in full. A product whose only
# candidates are uncorroborated keeps them for review; it just
# does not get one of them presented as fact.
choice = image_corroboration.choose_primary(
best, product_name, brand or ""
)
row["image_urls"] = list(choice.ordered)
row["image_url"] = choice.primary
if choice.primary is None:
row.setdefault("_notes", []).append(
f"no primary image: {choice.reason}"
)
except Exception as exc: # noqa: BLE001 - an image is not worth the row
# `warning`, not `debug`: at the default log level a debug line is
# invisible, so a row that silently lost its images looked identical to
@@ -776,6 +794,14 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
"highlights": list(row.get("highlights") or []),
"nutrients": list(row.get("nutrients") or []),
"search_query": search_query,
# Stage 10's verdict. validate_catalog() has always annotated these
# three onto every row it kept; this dict dropped all three, so the
# confidence the gate computed died here and every stored row looked
# equally trustworthy. Same rule as the block above: computed but not
# projected is computed for nothing.
"validation_status": row.get("validation_status"),
"confidence_score": row.get("confidence_score"),
"validation_issues": list(row.get("validation_issues") or []),
"field_sources": dict(row.get("field_sources") or {}),
}

View File

@@ -105,7 +105,7 @@ from app.infrastructure.settings import (
BRAND_DISCOVERY_USE_LLM,
BRAND_DISCOVERY_USE_OFF,
)
from app.services import active_brands, ollama_service
from app.services import active_brands, ollama_service, retail_presence
from app.services.brand_registry import (
get_fssai_license,
get_known_sub_brands,
@@ -119,6 +119,7 @@ from app.services.category_units import (
parse_unit,
)
from app.services.enrichment.barcode.sources import off_bulk
from app.services.product_grounding import GROUNDING_SIMILARITY_FLOOR
from app.services.vector_store import _sanitize_name, get_products_by_brand
logger = logging.getLogger(__name__)
@@ -594,13 +595,51 @@ def _registry_terms(brand: str) -> List[str]:
return terms
def _evidence_for(normalised: str, *, off_keys: Dict[str, Any],
registry_terms: Sequence[str], catalog_keys: Dict[str, str]) -> Optional[str]:
"""What corroborates this product, cheapest source first.
def _matches_any_key(normalised: str, keys: Dict[str, Any]) -> bool:
"""Is `normalised` one of `keys`, allowing for spelling drift?
Three tiers, none of which cost a network call at this point: the OFF index
was built during the sweep, the registry terms are a Python constant, and
the catalog index was read once.
EXACT EQUALITY WAS A BUG. These keys come from Open Food Facts and from
our own stored rows, and neither spells a product the way discovery does.
Measured: our "Anil Samba Rava" normalises to `samba rava`, while Open
Food Facts holds the same product as "SAMBA RAVVA" -> `samba ravva`. Not
equal, so a real product with a real barcode was reported as having no
corroboration at all.
The similarity floor is the one `product_grounding` measured against this
exact corpus - see that module for why it is not the 0.78 used for barcode
attachment, and why `samba rava`/`samba ravva` (0.681) is the case that
fixes the number.
"""
if not normalised:
return False
if normalised in keys:
return True
for key in keys:
if not key:
continue
if off_bulk.symmetric_similarity(key, normalised) >= GROUNDING_SIMILARITY_FLOOR:
return True
return False
def _evidence_for(normalised: str, *, off_keys: Dict[str, Any],
registry_terms: Sequence[str], catalog_keys: Dict[str, str],
retail_keys: Optional[Dict[str, Any]] = None) -> Optional[str]:
"""What corroborates this product, strongest source first.
Four tiers now. None costs a network call at this point: the OFF index was
built during the sweep, the registry terms are a Python constant, the
catalog index was read once, and the retail index is read from a cache the
backfill script populated offline - see retail_presence for why a live
query must never happen inside ingestion.
`retail` sits directly below `openfacts` and above `catalog` and
`registry`, because it is the only tier that is BOTH external and current.
`catalog` is our own prior output, which is circular if that output was
itself ungrounded, and `registry` is a hardcoded Python list. For a
non-food brand those two were the only tiers available at all, which is
why a toothpaste catalogue could be entirely language-model output and
still look corroborated.
Deliberately NOT a tier: "an image search returned a URL naming this
product". `catalog_engine._select_best_images` documents that CDN filenames
@@ -610,9 +649,11 @@ def _evidence_for(normalised: str, *, off_keys: Dict[str, Any],
"""
if not normalised:
return None
if normalised in off_keys:
if _matches_any_key(normalised, off_keys):
return "openfacts"
if normalised in catalog_keys:
if retail_keys and normalised in retail_keys:
return "retail"
if _matches_any_key(normalised, catalog_keys):
return "catalog"
tokens = set(normalised.split())
if tokens & set(registry_terms):
@@ -632,6 +673,25 @@ def _score(sources: Sequence[str], evidence: Optional[str]) -> float:
return 1.0
if has_off:
return 0.9
# An Open Food Facts row corroborates this product, but under a DIFFERENT
# SPELLING, so the merge did not fold the two together and this candidate
# never acquired the "off" source.
#
# Unreachable while `_evidence_for` matched keys by exact equality - an
# exact match always merged. It became reachable when that matching was
# relaxed to handle real spelling drift ("Anil Samba Rava" against Open
# Food Facts' "SAMBA RAVVA"), and without this branch such a row fell all
# the way through to 0.25 - scored as though nothing corroborated it, when
# a real database record does.
if evidence == "openfacts":
return 0.85
# A real shop is currently listing this exact product at this pack size.
# Scored above `catalog` (our own prior output, which is circular when
# that output was itself ungrounded) and far above `registry` (a hardcoded
# list), because it is the only tier that is both external and current.
# For a non-food brand it is usually the ONLY external tier available.
if evidence == "retail":
return 0.85
if evidence == "catalog":
return 0.7
if evidence == "registry":
@@ -685,7 +745,9 @@ def _resolve_sizes(candidate: Dict[str, Any], title: str, category: str,
def _build_product(brand: str, candidate: Dict[str, Any], *, max_sizes: int,
catalog_keys: Dict[str, str], off_keys: Dict[str, Any],
registry_terms: Sequence[str],
fssai: Optional[str]) -> Optional[DiscoveredProduct]:
fssai: Optional[str],
retail_keys: Optional[Dict[str, Any]] = None,
) -> Optional[DiscoveredProduct]:
raw_title = _canonicalise_title_size((candidate.get("title") or "").strip())
if not raw_title:
return None
@@ -769,7 +831,8 @@ def _build_product(brand: str, candidate: Dict[str, Any], *, max_sizes: int,
shape = {"title": title, "category": category_hint, "size_variants": sizes,
"description": description}
evidence = _evidence_for(normalised, off_keys=off_keys,
registry_terms=registry_terms, catalog_keys=catalog_keys)
registry_terms=registry_terms, catalog_keys=catalog_keys,
retail_keys=retail_keys)
sources = list(dict.fromkeys(candidate.get("sources") or [candidate.get("source")]))
sources = [s for s in sources if s]
@@ -835,6 +898,7 @@ def discover_brand_products(
if key:
off_keys.setdefault(key, candidate)
llm_candidates: List[Dict[str, Any]] = []
if use_llm:
llm_candidates = _from_llm(brand, deadline=deadline, budget=max_products * 2)
@@ -955,6 +1019,32 @@ def discover_brand_products(
chosen = _canonicalise_title_size(canonical_titles[key])
entry["title"] = off_bulk.strip_sizes(chosen).strip() or chosen
# What a real shop is currently listing, read from the cache that
# `scripts/backfill_retail_presence.py` fills offline. CACHE ONLY - never a
# live query. Discovery is interactive, and one lookup costs 2.6-5.7s
# against a provider that 403s under load, so a live sweep here would
# either hang the preview or get the whole run throttled - and a throttled
# miss is indistinguishable from "nobody sells this", which is the one
# input a corroboration tier must never be fed.
#
# Built from `merged`, so it covers the LLM's candidates too. That is the
# point: an Open Food Facts row needs no second opinion, and for a NON-FOOD
# brand there is no Open Food Facts row at all - `off_bulk` queries the food
# database alone, so a toothpaste's only possible external corroboration is
# this one.
retail_keys: Dict[str, Any] = {}
for entry in merged:
title = entry.get("title") or ""
key = _normalise_title(brand, title)
if not key or key in retail_keys:
continue
evidence = retail_presence.check_listing(brand, title, "", live=False)
if evidence.is_found:
retail_keys[key] = evidence
if retail_keys:
logger.info("🛒 %d of %d discovered products are currently listed by a retailer",
len(retail_keys), len(merged))
products: List[DiscoveredProduct] = []
dropped = 0
for candidate in merged:
@@ -962,6 +1052,7 @@ def discover_brand_products(
brand, candidate, max_sizes=max_sizes_per_product,
catalog_keys=catalog_keys, off_keys=off_keys,
registry_terms=registry_terms, fssai=fssai,
retail_keys=retail_keys,
)
if product is None:
continue

View File

@@ -104,6 +104,13 @@ EXPORT_COLUMNS = (
"barcode_lookup_status", "barcode_last_updated",
"gst_percent", "tax_amount", "hsn_gst_needs_review",
"highlights", "nutrients", "search_query", "field_sources",
# The validation verdict. Listed here for the same reason the barcode keys
# are: an export that strips them turns a re-seed into a silent downgrade.
# The upsert assigns these three plainly rather than COALESCEing them, so a
# seed file that has lost them clears the verdict on every row it restores -
# which is right for a writer that never validated, and wrong for this one,
# whose whole job is to echo back rows that already were.
"validation_status", "confidence_score", "validation_issues",
"nutrition_score", "health_score", "nutrients_per_100g",
)

View File

@@ -4,10 +4,16 @@ catalog rows, running each stage's per-row work concurrently (bounded by
`max_concurrency`) and NEVER letting one row's failure affect any other
row or stage.
`catalog_engine.py` calls `run_default_pipeline()` once, between the
deterministic-validation step (product_validator.py) and catalog assembly
- see that file's "Step 2.6" for the call site and
docs/BARCODE_ENRICHMENT.md for the full pipeline-position rationale.
`store_catalog_pipeline.stages_8_9_enrichment()` builds an EnrichmentPipeline
and runs it between SKU resolution and the validation gate; see
docs/BARCODE_ENRICHMENT.md for the pipeline-position rationale.
This docstring used to say `catalog_engine.py` called `run_default_pipeline()`
at a "Step 2.6". It never did - that module does not import this one at all,
so the brand-name generation path gets no barcode, HSN/GST or content
enrichment. Step 2.6 there is now the validation gate only. Wiring enrichment
into that path is a real and separate piece of work; do not read this comment
as saying it is already done.
"""
from __future__ import annotations

View File

@@ -0,0 +1,477 @@
"""
Does this image URL actually depict THIS product?
WHY THIS FILE EXISTS
--------------------
`image_search.validate_image_url_live` answers "does this URL serve real image
bytes", which is a liveness question. Nothing answered the relevance question,
so a press photo of a person named like the brand passed every check the
ingestion path had: it is a real image, above the byte floor, on a reputable
host. That is how the live catalogue came to illustrate "Anil Samba Rava" with
a photograph of the actor Anil Kapoor.
The logic here is ported from `scripts/repair_brand_images.py`, which already
knew how to reject that image class - its comments name the failures it was
written for ("Aachi Kulambu Mix" returning a press photo of a politician,
"MTR Dosa Mix" returning an anatomy plate). It lived in a script that imports
settings, brand_registry, produce_reference and four private `vector_store`
symbols including `_connect`, so no ingestion path could import it. Moving the
pure predicates here is what lets stage 6 and the catalog engine use them.
THE ONE SEMANTIC CHANGE MADE DURING THE MOVE
--------------------------------------------
The original corroborated a URL against `words + brand_tokens`:
return any(w in lowered for w in words + _brand_tokens(brand))
That works for "Aachi Kulambu Mix" precisely because *Aachi is not a human
name*. Anil is. For product "Anil Samba Rava" under brand "Anil" the token list
contains "anil" twice over, and `Anil_Kapoor_2019.jpg` contains "anil", so the
gate returned True and the celebrity photo was corroborated by the very token
that made it wrong.
`names_product` here requires a DISTINCTIVE token instead - the title minus the
brand minus the pack size, which is the same set
`catalog_engine._select_best_images` already computes to rank candidates:
"Anil Samba Rava" -> {samba, rava} -> rejects Anil_Kapoor_2019.jpg
"Anil Wheat Vermicelli"-> {wheat, vermicelli}
Brand tokens remain, but only as an explicit fallback for titles that have no
distinctive words at all ("Amul 1kg"), and a match found that way is reported
as `via_brand_only` so the caller can decline to treat it as corroboration.
WHAT THIS MODULE MAY AND MAY NOT DECIDE
---------------------------------------
It decides which image is shown for a product. It never decides whether the
PRODUCT is real. `brand_discovery._evidence_for` deliberately refuses to treat
image naming as product evidence, because CDN filenames are frequently opaque
hashes and absence of a naming URL is weak evidence of absence. That reasoning
is right and this module does not disturb it: images gate images, never
products.
Dependencies are stdlib plus `requests` on purpose. `catalog_engine` imports
this module, and `repair_brand_images` already reaches back into
`catalog_engine._select_best_images`, so a module-scope import of
`catalog_engine` here would close a cycle.
"""
from __future__ import annotations
import logging
import re
import sqlite3
import threading
import time
from contextlib import closing
from dataclasses import dataclass
from pathlib import Path
from typing import Iterable, List, Optional, Sequence
from urllib.parse import urlparse
import requests
logger = logging.getLogger(__name__)
_BROWSER_UA = "nearle-catalogue/1.0 (product image corroboration)"
# "own" and "products" are the BUCKET's name, not a brand's. Left in, the token
# "products" corroborates any URL containing /images/products/ or
# /cdn/shop/products/ - which is every Open*Facts and every Shopify path - so
# `names_product` waved through 40+ images that named nothing about the item.
# That is how openbeautyfacts cosmetics photos became the stored image for
# Banana, Orange, Papaya, Guava, Lemon and twenty more.
BUCKET_TOKENS = frozenset({"own", "products", "product"})
# A trailing pack size on a product name. Used to collapse "X 100g"/"X 500g"
# onto one search, and to keep size digits out of the distinctive token set.
SIZE_TAIL = re.compile(
r"\s+\d+(?:\.\d+)?\s*(?:g|gm|gms|kg|ml|l|ltr|litre|liter|pcs|pc|n|no|nos)\b\.?\s*$",
re.IGNORECASE,
)
# A bare quantity token, e.g. "500g" or "2l". Mirrors the filter in
# catalog_engine._select_best_images so the two agree on what a size looks like.
_SIZE_TOKEN = re.compile(r"\d+(?:kg|g|gm|gms|ml|l|ltr|pcs|n)?", re.IGNORECASE)
# The barcode embedded in an Open*Facts image path, e.g.
# /images/products/890/604/215/0067/front_en.4.400.jpg
_OFF_BARCODE = re.compile(r"/images/products/((?:\d+/)+)")
# Open*Facts image hosts and the API host that can identify a barcode for each.
#
# The original checked `if "openfoodfacts.org" not in url: return True`, so
# every openbeautyfacts and openproductsfacts image skipped the cross-check
# entirely - exactly the non-food case, and exactly the two sibling databases
# `image_search.OPEN_FACTS_HOSTS` queries. A food product got the check and a
# toothpaste did not.
_OFF_API_HOSTS = {
"openfoodfacts": "world.openfoodfacts.org",
"openbeautyfacts": "world.openbeautyfacts.org",
"openproductsfacts": "world.openproductsfacts.org",
}
# Hosts whose image paths are content-hashed or barcode-keyed, so a filename
# that fails to name the product says nothing about the photo. These are the
# hosts catalog_engine's own comment is about when it refuses to drop
# uncorroborated candidates.
OPAQUE_PATH_DOMAINS = (
"bbassets.com", "bigbasket.com", "flixcart.com", "flipkart.com",
"media-amazon.com", "amazon.in", "amazon.com", "jiomart.com",
"zeptonow.com", "blinkit.com", "grofers.com", "cloudinary.com",
"shopifycdn.com", "cdn.shopify.com", "akamaized.net", "cloudfront.net",
"openfoodfacts.org", "openbeautyfacts.org", "openproductsfacts.org",
)
# Hosts whose filenames are human-authored and descriptive. A filename that
# fails to name the product here is real evidence that the photo is of
# something else - and this is the host family the celebrity photo came from.
DESCRIPTIVE_FILENAME_DOMAINS = (
"wikimedia.org", "wikipedia.org", "wikimedia.commons",
)
# Two or more Capitalised words joined by underscores, optionally with a year:
# the Wikimedia Commons house style for a photograph OF A PERSON
# ("Anil_Kapoor_2019.jpg"). Deliberately used only to DEMOTE, never to reject -
# "Britannia_Good_Day.jpg" has the identical shape and is a perfectly good
# product photo, so this signal orders candidates and is not allowed to
# eliminate one.
_PERSON_FILENAME = re.compile(
r"^[A-Z][a-z]+(?:_[A-Z][a-z]+)+(?:_\d{4})?[^/]*\.(?:jpg|jpeg|png|webp)$"
)
_PERSON_CONTEXT = re.compile(r"_at_|_in_\d{4}|portrait|headshot", re.IGNORECASE)
# ---------------------------------------------------------------------------
# Tokens
# ---------------------------------------------------------------------------
def brand_tokens(brand: str) -> List[str]:
"""Distinctive words of a brand name, bucket words removed."""
return [
w for w in re.split(r"[^a-z0-9]+", (brand or "").lower())
if len(w) > 2 and w not in BUCKET_TOKENS
]
def distinctive_tokens(product_name: str, brand: str) -> List[str]:
"""Words that separate THIS product from its brand-mates.
The brand is removed on purpose: every Britannia URL contains "britannia",
so it separates nothing - what tells Marie Gold from Good Day is
"marie"/"gold" vs "good"/"day". Pack sizes and short filler words go for
the same reason. This mirrors catalog_engine._select_best_images so the
ranking and the gate can never disagree about what identifies a product.
"""
brand_words = {w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) if w}
out: List[str] = []
for word in re.split(r"[^a-z0-9]+", (product_name or "").lower()):
if (
len(word) > 2
and word not in brand_words
and word not in BUCKET_TOKENS
and word not in out
and not _SIZE_TOKEN.fullmatch(word)
):
out.append(word)
return out
def search_key(product_name: str, brand: str) -> tuple:
"""Collapse a trailing pack size so sizes of one product share a lookup.
Sharing across sizes is correct rather than merely cheap: it is the same
product in a different pack, and stage 4's size explosion produces exactly
these rows from a single source product.
"""
base = SIZE_TAIL.sub("", product_name or "").strip()
return ((brand or "").lower(), (base or product_name or "").lower())
# ---------------------------------------------------------------------------
# Corroboration
# ---------------------------------------------------------------------------
@dataclass(frozen=True)
class Corroboration:
"""Why a URL was or was not accepted as depicting this product."""
corroborated: bool
reason: str
via_brand_only: bool = False
def corroborate(url: str, product_name: str, brand: str) -> Corroboration:
"""Does `url` name this product?
Distinctive tokens first. Brand tokens are consulted only when the title
has no distinctive words of its own, and a match found that way is flagged
`via_brand_only` - it is the weakest possible signal, and for a brand that
is also a personal name it is the signal that produced the defect this
module exists for.
"""
lowered = (url or "").lower()
if not lowered:
return Corroboration(False, "empty url")
distinctive = distinctive_tokens(product_name, brand)
if distinctive:
hit = next((t for t in distinctive if t in lowered), None)
if hit:
return Corroboration(True, f"url names {hit!r}")
return Corroboration(
False,
"url names none of " + ", ".join(repr(t) for t in distinctive[:4]),
)
# No distinctive words at all (e.g. "Amul 1kg"). Fall back to the brand,
# and say so, so the caller can decide how much that is worth.
tokens = brand_tokens(brand)
hit = next((t for t in tokens if t in lowered), None)
if hit:
return Corroboration(True, f"url names brand {hit!r}", via_brand_only=True)
return Corroboration(False, "url names neither the product nor the brand")
def names_product(url: str, product_name: str, brand: str) -> bool:
"""Boolean form of `corroborate`, for callers that only need the verdict."""
return corroborate(url, product_name, brand).corroborated
def looks_like_person_photo(url: str) -> bool:
"""True when the FILENAME has the shape of a photograph of a person.
A demotion signal only - see `_PERSON_FILENAME` for why this must never
reject on its own.
"""
name = urlparse(url or "").path.rsplit("/", 1)[-1]
if not name:
return False
return bool(_PERSON_FILENAME.match(name)) or bool(_PERSON_CONTEXT.search(name))
def has_opaque_path(url: str) -> bool:
lowered = (url or "").lower()
return any(d in lowered for d in OPAQUE_PATH_DOMAINS)
def has_descriptive_filename(url: str) -> bool:
lowered = (url or "").lower()
return any(d in lowered for d in DESCRIPTIVE_FILENAME_DOMAINS)
# ---------------------------------------------------------------------------
# Open*Facts barcode cross-check
# ---------------------------------------------------------------------------
# The barcode is embedded in the image path, so the product is one cheap lookup
# away. If the Open*Facts record does not mention our brand, the image is
# somebody else's product. Nothing else can catch this: the path is opaque
# digits, so no amount of filename matching would help. OFF matches on NAME,
# not brand - asked for "Aachi Pickles" it returned the front-of-pack photo for
# a French "Ducros Green Pitted Olives", which validated perfectly happily
# because it IS a real image.
_DB_PATH = Path("data") / "cache" / "image_corroboration.db"
_lock = threading.Lock()
_initialized = False
_CACHE_TTL_SECONDS = 30 * 24 * 3600
def _connect() -> sqlite3.Connection:
_DB_PATH.parent.mkdir(parents=True, exist_ok=True)
conn = sqlite3.connect(str(_DB_PATH), timeout=10)
conn.execute("PRAGMA journal_mode=WAL")
return conn
def _ensure_schema(conn: sqlite3.Connection) -> None:
global _initialized
if _initialized:
return
conn.execute(
"""
CREATE TABLE IF NOT EXISTS openfacts_brand_check (
cache_key TEXT PRIMARY KEY,
verdict INTEGER NOT NULL,
created_at REAL NOT NULL
)
"""
)
conn.commit()
_initialized = True
def _cache_get(key: str) -> Optional[bool]:
# `closing`, because sqlite3's own context manager commits the transaction
# and leaves the connection open. Leaked connections are finalized at GC,
# which surfaces as an unraisable exception - and pytest.ini turns warnings
# into errors, so a leak here fails the suite from an unrelated test.
try:
with _lock, closing(_connect()) as conn:
_ensure_schema(conn)
row = conn.execute(
"SELECT verdict, created_at FROM openfacts_brand_check WHERE cache_key = ?",
(key,),
).fetchone()
except Exception as e: # noqa: BLE001 - a cache failure is a cache miss
logger.debug("image corroboration cache read failed for %s: %s", key, e)
return None
if not row:
return None
verdict, created_at = row
if (time.time() - created_at) > _CACHE_TTL_SECONDS:
return None
return bool(verdict)
def _cache_set(key: str, verdict: bool) -> None:
try:
with _lock, closing(_connect()) as conn:
_ensure_schema(conn)
conn.execute(
"""
INSERT INTO openfacts_brand_check (cache_key, verdict, created_at)
VALUES (?, ?, ?)
ON CONFLICT(cache_key) DO UPDATE SET
verdict = excluded.verdict,
created_at = excluded.created_at
""",
(key, 1 if verdict else 0, time.time()),
)
conn.commit()
except Exception as e: # noqa: BLE001 - best effort only
logger.debug("image corroboration cache write failed for %s: %s", key, e)
def _api_host_for(url: str) -> Optional[str]:
lowered = (url or "").lower()
for marker, host in _OFF_API_HOSTS.items():
if marker in lowered:
return host
return None
def openfacts_product_matches_brand(url: str, brand: str, *, timeout: int = 10) -> bool:
"""False ONLY when Open*Facts positively says this barcode is another brand.
Every other outcome - not an Open*Facts URL, no barcode in the path, no
brand tokens, a network failure, an unparseable response - returns True.
A lookup failure must never reject a good image; that asymmetry is the
whole point, and it is the same discipline the realtime checks follow.
"""
api_host = _api_host_for(url)
if not api_host:
return True
match = _OFF_BARCODE.search(url or "")
if not match:
return True
barcode = match.group(1).replace("/", "")
tokens = brand_tokens(brand)
if not tokens:
return True
key = f"{api_host}:{barcode}:{(brand or '').lower()}"
cached = _cache_get(key)
if cached is not None:
return cached
verdict = True
try:
resp = requests.get(
f"https://{api_host}/api/v2/product/{barcode}.json",
params={"fields": "brands,product_name"},
timeout=timeout,
headers={"User-Agent": _BROWSER_UA},
)
if resp.ok:
product = (resp.json() or {}).get("product") or {}
haystack = (
f"{product.get('brands') or ''} {product.get('product_name') or ''}"
).lower()
if haystack.strip():
verdict = any(t in haystack for t in tokens)
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
return True
_cache_set(key, verdict)
return verdict
# ---------------------------------------------------------------------------
# Which candidate may become the product's primary image
# ---------------------------------------------------------------------------
@dataclass
class PrimaryChoice:
"""The outcome of choosing a product's `image_url` from its candidates."""
primary: Optional[str]
ordered: List[str]
reason: str
def choose_primary(
urls: Sequence[str],
product_name: str,
brand: str,
*,
check_openfacts: bool = True,
) -> PrimaryChoice:
"""Pick the URL that may become `image_url`, and order the rest.
Three tiers, because the two existing opinions about failing closed are
both right and the disagreement is domain-scoped:
1. The URL names the product. Eligible.
2. The URL cannot name anything (a hashed retailer CDN path, an
Open*Facts barcode path) - and for Open*Facts, the barcode's own
record agrees about the brand. Eligible: this is the case
catalog_engine's comment protects, where dropping uncorroborated
candidates would leave real products with no image at all.
3. The URL could have named the product and did not - a human-authored
Commons filename, say. Kept in `image_urls`, never promoted.
Nothing eligible means `primary is None`. A blank image renders as the
brand monogram, which is honest; another company's product is not, and it
stays invisible until somebody recognises the photo.
"""
tier1: List[str] = []
tier2: List[str] = []
tier3: List[str] = []
for url in urls:
if not url or not str(url).startswith("http"):
continue
verdict = corroborate(url, product_name, brand)
if verdict.corroborated and not verdict.via_brand_only:
tier1.append(url)
elif verdict.via_brand_only and not looks_like_person_photo(url):
# The title has NO distinctive words of its own - "Godrej 50ml",
# "Lion Dates 100g", "Amul 1kg". There is no product identity to
# match on, so a brand-token match is the best signal that exists
# and withholding the image gains nothing: measured over the seed
# catalogues, treating these as ineligible accounted for 14 of 45
# withheld primaries, every one of them a brand's own product page.
#
# Person-shaped filenames are the exception, because a brand token
# matching a personal name is the exact defect this module exists
# for and a title with no distinctive words cannot contradict it.
tier2.append(url)
elif has_opaque_path(url) and not has_descriptive_filename(url):
if check_openfacts and not openfacts_product_matches_brand(url, brand):
tier3.append(url)
else:
tier2.append(url)
else:
tier3.append(url)
# Person-shaped filenames sink within their tier. They are never dropped,
# so a product whose only images look like this still keeps them in
# `image_urls` for a human to review.
tier3.sort(key=looks_like_person_photo)
ordered = tier1 + tier2 + tier3
if tier1:
return PrimaryChoice(tier1[0], ordered, "url names the product")
if tier2:
return PrimaryChoice(tier2[0], ordered, "opaque path on a known image host")
return PrimaryChoice(
None,
ordered,
"no candidate names this product; primary withheld rather than guessed",
)

View File

@@ -0,0 +1,236 @@
"""
Does an external source corroborate that this product exists?
WHY THIS FILE EXISTS
--------------------
The catalogue contained "Anil Wheat Vermicelli 12g". Nobody sells a 12 g
vermicelli pack. It was generated by a 1.5B local model, given a category, a
price band and an internal SKU, and stored - because every check in the system
asks whether a row is WELL FORMED, and a well-formed fiction passes all of
them. `product_validator`'s own docstring says so: it was built to catch
malformed rows, not false ones.
Meanwhile the real answer was already on disk. `off_bulk.fetch_brand_corpus`
caches a brand's entire Open Food Facts catalogue, and
`data/cache/off_brand_corpus/anil.json` holds Anil's eight real products -
Roasted Short Vermicelli at 180 g and 450 g, SAMBA RAVVA at 500 g. Nothing
consulted it at generation time.
WHAT THIS MODULE DOES NOT DO: SUBSTITUTE A SIZE
-----------------------------------------------
The obvious-looking fix - "look up the real quantity and swap it in" - is
wrong here, and measurably so. Matching "Anil Wheat Vermicelli" against the
corpus scores 0.460 against "Roasted Short Vermicelli", far below any usable
floor. That is the corpus saying THIS PRODUCT DOES NOT EXIST, not saying it is
450 g. Swapping in a corpus quantity would replace one fabrication with a
better-dressed one: a product that still does not exist, now wearing a size
that does.
So a miss reports `not_found` and the caller records that the row is
ungrounded. Removing the row is the validation gate's job, not this module's.
WHY NOT `find_product_quantity_openfacts`
-----------------------------------------
`image_search.find_product_quantity_openfacts` looks like the right oracle and
is not. It queries `/cgi/search.pl` LIVE, once per product, looping three hosts
at a 12-second timeout - up to 36 s per product against an endpoint Open Food
Facts rate-limits to 10 requests/minute. Two hundred products is twenty-plus
minutes and near-certain throttling, and a throttled miss is indistinguishable
from a real one, which is exactly the input a gate must never be fed. Measured
2026-09-10: that endpoint returned 503 on OFF and 500 on Open Beauty Facts.
`fetch_brand_corpus` costs one to five requests per BRAND, is disk-cached,
retry-wrapped and paced, and is already populated for 50+ brands.
THREE STATES, NOT TWO
---------------------
`grounded` / `not_found` / `unknown`. A corpus that could not be fetched must
report `unknown` and never `not_found`, or a network failure would silently
demote a whole brand's real products. Every lookup path in this codebase that
feeds a gate follows the same discipline.
"""
from __future__ import annotations
import logging
from dataclasses import dataclass
from typing import Any, Dict, List, Optional, Sequence
from app.services.enrichment.barcode.sources import off_bulk
logger = logging.getLogger(__name__)
# How similar a corpus name must be to count as the same product.
#
# NOT the 0.78 used for barcode attachment. Measured against the real Anil
# corpus on 2026-09-10, best match per target:
#
# "wheat vermicelli" vs "Rice Vermicelli" -> 0.610 (DIFFERENT product)
# "samba rava" vs "SAMBA RAVVA" -> 0.681 (SAME product)
#
# A 0.78 floor rejects a genuine product over a one-letter spelling difference
# in Open Food Facts' own record, so it cannot be used here. But note how
# little room that leaves: 0.071 between a true match and a false one, and the
# floor below sits about 0.01 above the false one. THAT MARGIN IS TOO THIN TO
# REST A DECISION ON, which is why `_material_conflict` exists - the real
# discriminator between those two names is that wheat is not rice, not that
# 0.610 is not 0.681. The floor is the coarse filter; the material check is
# what actually separates the reported defect from the real product beside it.
#
# It is also deliberately a soft signal in both directions: nothing here
# deletes a row, it only decides whether the row can be called corroborated,
# so the cost of being wrong is a review flag and not data loss.
GROUNDING_SIMILARITY_FLOOR = 0.62
# Base materials that make two otherwise similarly-named products different
# products. "Wheat Vermicelli" and "Rice Vermicelli" share a head noun and
# score 0.610 against each other; no string-similarity threshold separates
# them reliably, because the thing that differs is one word carrying all the
# meaning.
#
# The rule is deliberately narrow: a conflict is declared ONLY when BOTH names
# name a material and the sets are disjoint. "Wheat Vermicelli" against a bare
# "Vermicelli" is not a conflict - that is plausibly the same product line
# described at two levels of detail, and treating it as a conflict would lose
# real corroboration.
_MATERIALS = frozenset({
"wheat", "rice", "ragi", "maida", "corn", "millet", "bajra", "jowar",
"atta", "besan", "soya", "oat", "oats", "barley", "quinoa", "almond",
"cashew", "coconut", "groundnut", "peanut", "sesame", "mustard",
"sunflower", "olive", "ghee", "butter",
})
GROUNDED = "grounded"
NOT_FOUND = "not_found"
UNKNOWN = "unknown"
# Per-process memo of brand -> corpus. fetch_brand_corpus is itself disk-cached,
# so this only avoids re-reading and re-parsing the same JSON once per product
# in a 200-row run.
_corpus_memo: Dict[str, Optional[List[Dict[str, Any]]]] = {}
@dataclass(frozen=True)
class Grounding:
"""What an external source says about one product."""
status: str
matched_name: Optional[str] = None
quantity: Optional[str] = None
similarity: float = 0.0
source: Optional[str] = None
@property
def is_grounded(self) -> bool:
return self.status == GROUNDED
def note(self) -> str:
if self.status == GROUNDED:
return (
f"corroborated by {self.source} as {self.matched_name!r} "
f"(similarity {self.similarity:.2f})"
)
if self.status == NOT_FOUND:
return f"no {self.source or 'Open Food Facts'} product matches this name"
return "could not check whether this product exists"
def _material_conflict(candidate: str, target: str) -> bool:
"""True when two names name DIFFERENT base materials.
See `_MATERIALS`. Only fires when both sides name one, so a more specific
name still corroborates a less specific one.
"""
left = {w for w in candidate.lower().split() if w in _MATERIALS}
right = {w for w in target.lower().split() if w in _MATERIALS}
return bool(left and right and not (left & right))
def _corpus_for(brand: str) -> Optional[List[Dict[str, Any]]]:
"""The brand's cached Open*Facts corpus, or None when it is unavailable.
None means "we do not know", and is what makes the difference between
`not_found` and `unknown` downstream. An EMPTY list is a real answer - the
brand was looked up and has nothing on file.
"""
key = (brand or "").strip().lower()
if not key:
return None
if key in _corpus_memo:
return _corpus_memo[key]
try:
corpus = off_bulk.fetch_brand_corpus(brand)
except Exception as e: # noqa: BLE001 - an unreachable corpus is not a verdict
logger.debug("Corpus fetch failed for %r: %s", brand, e)
corpus = None
_corpus_memo[key] = corpus
return corpus
def reset_cache() -> None:
"""Drop the per-process corpus memo (tests, and long-lived workers)."""
_corpus_memo.clear()
def ground_product(brand: str, product_title: str,
*, corpus: Optional[Sequence[Dict[str, Any]]] = None) -> Grounding:
"""Is `product_title` a real product of `brand`, per Open*Facts?
`corpus` is injectable so a caller processing a whole brand fetches once
and so tests need no network.
"""
title = (product_title or "").strip()
if not title:
return Grounding(UNKNOWN)
hits = corpus if corpus is not None else _corpus_for(brand)
if hits is None:
return Grounding(UNKNOWN)
if not hits:
# A real answer: this brand has nothing on file at all. That is not
# evidence against any single product, so it is still `unknown` - a
# brand missing from Open Food Facts is a gap in Open Food Facts.
return Grounding(UNKNOWN, source="openfacts")
drop = off_bulk.brand_tokens(brand)
target = off_bulk.normalize_for_match(title, drop)
if not target:
return Grounding(UNKNOWN, source="openfacts")
best_score = 0.0
best_hit: Optional[Dict[str, Any]] = None
for hit in hits:
name = hit.get("product_name") or hit.get("product_name_en") or ""
if not name:
continue
candidate = off_bulk.normalize_for_match(name, drop)
if not candidate:
continue
# Skipped before scoring, not after: a wrong-material name can be the
# single best-scoring row in the corpus ("Rice Vermicelli" is the top
# match for "Wheat Vermicelli" at 0.610), and letting it win would mask
# a genuinely better match further down the list.
if _material_conflict(candidate, target):
continue
score = off_bulk.symmetric_similarity(candidate, target)
if score > best_score:
best_score, best_hit = score, hit
if best_hit is None or best_score < GROUNDING_SIMILARITY_FLOOR:
return Grounding(NOT_FOUND, similarity=round(best_score, 3), source="openfacts")
# A variant conflict means the corpus row is a DIFFERENT pack of a
# similarly-named product (sugar-free vs regular, jar vs pouch). Close
# enough to score well, not close enough to corroborate.
name = best_hit.get("product_name") or best_hit.get("product_name_en") or ""
if off_bulk.has_extra_variant_conflict(name, title):
return Grounding(NOT_FOUND, matched_name=name,
similarity=round(best_score, 3), source="openfacts")
quantity = str(best_hit.get("quantity") or "").strip() or None
return Grounding(
GROUNDED,
matched_name=name,
quantity=quantity,
similarity=round(best_score, 3),
source="openfacts",
)

View File

@@ -24,9 +24,15 @@ This module is that final gate. It is intentionally:
category-anchored price bands, and sku_service's own SKU format) rather
than guessing. False rejections are treated as seriously as false
acceptances - see docs/VALIDATION_PIPELINE.md for the tuning rationale.
- ADDITIVE: this module has zero dependents before this change and does not
modify any existing function's behaviour; it is only ever imported and
called from new code (see app/core/catalog_engine.py's call site).
- ADDITIVE: it does not modify any existing function's behaviour; callers opt
in. The call sites are store_catalog_pipeline.stage_10_validate (the
spreadsheet path), orchestration/assets/catalog.py (the Dagster mirror) and
ProductCatalogEngine.generate_catalog's "Step 2.6" (the brand-name path).
That last one was absent for a long time while this docstring claimed it
existed, so POST /api/catalog/generate wrote LLM output straight to the
brand tables with no gate at all. If you are adding another generation path,
it needs its own call to validate_catalog - nothing here is automatic.
USAGE
-----
@@ -244,6 +250,7 @@ def validate_product(
category_resolved_deterministically: bool = False,
grounded: bool = False,
images_checked: bool = True,
require_grounding: bool = False,
config: Optional[ValidationConfig] = None,
) -> ValidationReport:
"""Validate one fully-assembled catalog row and return a confidence-scored report.
@@ -350,12 +357,14 @@ def validate_product(
score -= 0.25
# 4b. Size unit <-> category compatibility (see app/services/
# category_units.py). This is a last-resort backstop: by the time a
# row reaches this final validator, its size should already have
# been through the generation-time gate (app/services/
# generation_verifier.py, called from ollama_service.py) AND
# catalog_engine.py's own dimension-unit defence-in-depth pass, both
# of which reject/repair a bad unit long before enrichment. This
# category_units.py). This used to describe itself as a backstop
# behind a generation-time gate in app/services/generation_verifier.py
# "called from ollama_service.py". THAT FILE DOES NOT EXIST IN THIS
# TREE - it belongs to a sibling project - so the layer this comment
# promised was never running here and this check is not a backstop but
# a front line. The real upstream defence is
# category_units.fix_or_reject_size, called from stage 4, plus
# catalog_engine.py's dimension-unit pass. This
# check exists purely to catch anything that reached validate_product
# some other way (e.g. a pre-existing DB row from before this fix, or
# a future caller that doesn't route through catalog_engine).
@@ -413,6 +422,33 @@ def validate_product(
else:
report.status = "verified"
# GROUNDING AS A CAP, NOT A BONUS.
#
# The +0.15 above is why this module could never do the job its docstring
# claims. A fabricated product scores 0.55 baseline + 0.15 resolved
# category + 0.05 has-images = 0.75, clearing the 0.70 threshold and
# landing as "verified" - so "Anil Wheat Vermicelli 12g", a pack size no
# shop has ever sold, was stored as a checked fact. Every check here is
# about FORM, and a well-formed fiction satisfies all of them.
#
# A caller that knows its rows came from a generator rather than from a
# source sets require_grounding. Then a row nothing external corroborates
# can still be kept and still scores normally, but it cannot be called
# verified - which is the honest answer, because nothing verified it.
#
# DELIBERATELY NOT APPLIED BY DEFAULT. The spreadsheet path passes no
# grounded flags at all, and a shop's own upload IS its evidence; capping
# there would relabel every uploaded row as unchecked and say nothing
# true. Only generation paths opt in.
if require_grounding and not grounded and report.status == "verified":
report.status = "needs_review"
report.issues.append(ValidationIssue(
field="grounding", severity="warning",
message="no external source corroborates that this product exists; "
"well-formed but unverified",
penalty=0.0,
))
return report
@@ -423,6 +459,7 @@ def validate_catalog(
known_category_flags: Optional[list[bool]] = None,
grounded_flags: Optional[list[bool]] = None,
images_checked: bool = True,
require_grounding: bool = False,
config: Optional[ValidationConfig] = None,
) -> tuple[list[dict[str, Any]], list[dict[str, Any]], dict[str, Any]]:
"""Validate an entire list of assembled catalog rows in one pass.
@@ -450,6 +487,7 @@ def validate_catalog(
category_resolved_deterministically=det,
grounded=grd,
images_checked=images_checked,
require_grounding=require_grounding,
config=cfg,
)
status_counts[report.status] = status_counts.get(report.status, 0) + 1

View File

@@ -0,0 +1,883 @@
"""
Is this product on sale, right now, somewhere real?
WHY THIS FILE EXISTS
--------------------
Open Food Facts answers "does this product exist" for food, and answers it
well. It does not answer it for anything else: `off_bulk` queries only
`search.openfoodfacts.org`, so a toothpaste or a detergent has no
product-discovery source at all and its entire catalogue is language-model
output. Measured 2026-09-10, India-tagged coverage on the sibling databases is
2-13 products per non-food brand against Britannia's 218 on OFF - real, but
nowhere near enough to build a catalogue from.
A live retail lookup answers it for everything, food and non-food alike, and
it is the only signal in this codebase that means "available in real time"
rather than "was in a database dump".
THE ONE THING THAT MAKES THIS WORK: MATCH THE PACK SIZE
--------------------------------------------------------
Measured against the live provider on 2026-09-10:
"Anil Roasted Short Vermicelli 450g"
-> amazon.in "Anil Vermicelli - Roasted, 450g Pouch" CONFIRMED
"Anil Wheat Vermicelli 12g"
-> results come back, but every one is the generic product
page. NOTHING confirms a 12 g pack. NOT CONFIRMED
A check that asks "did the search return anything?" would have confirmed the
12 g pack, which is the exact fabrication this module exists to catch. So a
hit counts only when the listing's TITLE and its PACK SIZE both match.
That is also why `sku_service.find_website_product_id` cannot be reused as
evidence: it regexes a product ID out of the result URL and never reads the
listing title or its size at all. It answers "is there an Amazon page in these
results", which for any real brand is always yes.
THREE STATES, NEVER TWO
-----------------------
`found` / `not_found` / `unknown`. DuckDuckGo throttles and 403s - the repair
script already records that it "already 403s and the pipeline falls through to
Bing". A throttled lookup MUST report `unknown`, because a gate fed
`not_found` on a rate-limit would quietly demote a brand's entire real
catalogue on a bad afternoon.
COST, AND WHY THE STAGE READS ONLY THE CACHE
---------------------------------------------
There is no brand-scoped retail endpoint the way there is for Open Food Facts,
so this is irreducibly one network call per product against a provider that
blocks. Measured 2.6-5.7 s per query.
So the work is split: `backfill_retail_presence.py` does the querying offline,
paced and cached, and the ingestion-time stage reads the cache ONLY
(`live=False`). Ingestion never blocks on a live query and never trips a rate
limit mid-batch. `search_key` collapses pack sizes onto one lookup per
PRODUCT rather than per row, which is a 3-5x cut for free.
"""
from __future__ import annotations
import json
import logging
import re
import sqlite3
import threading
import time
from contextlib import closing
from dataclasses import asdict, dataclass, field
from pathlib import Path
from typing import Any, Dict, List, Optional, Sequence
from urllib.parse import urlparse
from app.services import quantity_utils
from app.services.brand_registry import BRAND_ALIASES
from app.services.enrichment.barcode.matching import brand_matches
from app.services.enrichment.barcode.sources import off_bulk
from app.services.image_corroboration import search_key
logger = logging.getLogger(__name__)
FOUND = "found"
NOT_FOUND = "not_found"
UNKNOWN = "unknown"
# Indian grocery/marketplace domains a real FMCG product is listed on.
# `sku_service._MARKETPLACE_PATTERNS` covers the same ground for ID extraction;
# this list is about presence, so it needs no per-site URL grammar.
RETAILER_DOMAINS = (
"amazon.in", "flipkart.com", "bigbasket.com", "jiomart.com",
"blinkit.com", "zeptonow.com", "dmart.in", "swiggy.com",
"netmeds.com", "pharmeasy.in", "nykaa.com", "1mg.com",
"licious.in", "starquik.com", "spencers.in", "moreretail.in",
)
# What fraction of the PRODUCT's distinctive words must appear in the listing
# title. See `listing_matches` for why this is a containment ratio and not the
# 0.85 symmetric similarity the barcode matcher uses.
#
# 0.6 is deliberately loose. Retailers drop words freely - Amazon lists "Anil
# Roasted Short Vermicelli" as "Anil Vermicelli - Roasted", covering 2 of 3
# tokens (0.67) - so a tight floor here rejects real listings while adding
# nothing, because it is the pack-size check below that separates a real pack
# from an invented one. This gate only has to establish that the listing is
# about the right product family.
TITLE_COVERAGE_FLOOR = 0.6
# Marketing boilerplate that surrounds a product name in a retail listing
# title. Removed before matching so "Buy Anil Roasted Vermicelli 450g Online
# at Best Price" reduces to the product.
_LISTING_NOISE = re.compile(
r"\b(buy|online|at|best|price|lowest|offers?|deals?|free|delivery|shop|"
r"order|now|india|in|from|upto|off|save|get|com|grocery|gourmet|foods?)\b",
re.IGNORECASE,
)
# Words that name the SAME thing in a retail listing and in our catalogue.
#
# Indian FMCG listings mix English and transliterated Tamil/Hindi freely, and
# our own product names do too. Measured misses: our "Anil Puttu Mix" is on
# amazon.in as "Anil Puttu MAAVU" (maavu = flour/mix) and was rejected at 0.5
# coverage; this catalogue also uses "Semiya" and "Vermicelli" interchangeably
# in its own product names, so the two spellings fail to corroborate each other.
#
# STRICTLY OBSERVED, NEVER INVENTED - the same rule product_grounding._MATERIALS
# follows. This must not grow into a general thesaurus: every pair added makes
# the matcher more willing to confirm a product, and confirming products that do
# not exist is the failure this whole module exists to prevent. Add a pair only
# after seeing it reject a real listing.
#
# THE PACK SIZE IS NOT AFFECTED. Synonyms loosen only which words count as the
# same word; `quantities_match` still has to agree, so no synonym can rescue a
# size that nobody sells.
# Each pair below cites the listing that forced it. Anything without a citation
# does not belong here.
_SYNONYMS: Dict[str, str] = {
# amazon.in "Anil Puttu Maavu" vs our "Anil Puttu Mix", and
# theanilgroup.com "Arisi Puttu Maavu" for the same product.
"maavu": "flour",
"mix": "flour",
# This catalogue's own names: "Anil Rava Semiya", "Anil Ragi Semiya" and
# "Anil Wheat Vermicelli" are the same product family spelled two ways.
"semiya": "vermicelli",
# Open Food Facts holds our "Anil Samba Rava" as "SAMBA RAVVA".
"ravva": "rava",
# theanilgroup.com sells our "Wheat Vermicelli" as "Atta Semiya".
"atta": "wheat",
}
def _aliases_for(brand: str) -> List[str]:
"""Other names this brand trades under, from the curated registry.
`BRAND_ALIASES` maps alias -> parent, so the aliases OF a brand are the
keys pointing at it, plus the parent name itself.
"""
key = (brand or "").strip().lower()
if not key:
return []
out = {key}
parent = BRAND_ALIASES.get(key)
if parent:
out.add(parent.lower())
for alias, target in BRAND_ALIASES.items():
if target.lower() == key or (parent and target.lower() == parent.lower()):
out.add(alias.lower())
return sorted(out)
def _canonical_words(text: str) -> set:
"""Token set with observed synonyms folded onto one spelling."""
return {_SYNONYMS.get(w, w) for w in (text or "").split()}
# A real product listing title is short. Anything much longer is the provider
# running several results together into one `title` field, e.g.
#
# "Ginger Garlic Paste (Pack of 2) | No Peeling, No ChoppingAachi Ginger
# Garlic Paste 20g - martizo.comAachi Ginger Garlic Paste - Buy at Rs43..."
#
# 372 characters, at least four different listings, from at least two shops.
# Matching against that blob is meaningless in both halves: the quantity comes
# from whichever listing happened to state one, and the word coverage is
# satisfied by words scattered across products we never asked about. It
# produced a FOUND for "Aachi Ginger Paste 20g" whose 20g came from a different
# retailer's listing of a different product, attributed to aachifoods.com.
#
# A false positive is the worse failure here - it hands 0.85 corroboration to a
# product that may not exist, which is what this module was built to prevent -
# so a blob is discarded rather than parsed.
MAX_LISTING_TITLE_CHARS = 140
# Ingredient words that make two similarly-named products DIFFERENT products.
#
# Coverage is deliberately containment - the target's words must appear in the
# listing - so a listing is free to carry extra words, which is right for shop
# furniture ("Pouch", "Best Price", "Amazon.in"). It is wrong for an ingredient:
# "Aachi Ginger Paste" is entirely contained in "Aachi Ginger GARLIC Paste" and
# scored 1.0 coverage against it, and ginger paste is not ginger-garlic paste.
#
# Same discipline as product_grounding._MATERIALS and _SYNONYMS: these are
# words observed causing a wrong match, not a general ingredient list.
_DISTINGUISHING_INGREDIENTS = frozenset({
"garlic", "ginger", "chilli", "chili", "pepper", "tamarind", "coriander",
"cumin", "turmeric", "mint", "lemon", "tomato", "onion", "mango",
"coconut", "mustard", "fenugreek", "clove", "cardamom", "cinnamon",
})
_DB_PATH = Path("data") / "cache" / "retail_presence.db"
_lock = threading.Lock()
_initialized = False
DEFAULT_TTL_SECONDS = 14 * 24 * 3600
# A brand's own domain changes far more rarely than its stock does.
BRAND_DOMAIN_TTL_SECONDS = 180 * 24 * 3600
# A miss is re-asked the next day - see get_cached_brand_domain for why.
BRAND_DOMAIN_MISS_TTL_SECONDS = 24 * 3600
# Search endpoints are the strictly-limited ones. Same figure off_bulk uses.
PAUSE_SECONDS = 2.0
@dataclass
class RetailEvidence:
"""What a live retail search says about one (brand, product, size)."""
status: str
retailer: Optional[str] = None
url: Optional[str] = None
matched_title: Optional[str] = None
matched_size: Optional[str] = None
similarity: float = 0.0
checked_at: Optional[float] = None
# How many results were actually on a shop. This is what makes a negative
# auditable: without it, the only way to ask "did we even reach a retailer?"
# is to run the search again, and the provider's reach varies run to run
# (the same query returned 7 shop results and then 0 minutes later).
shop_results_seen: int = 0
# The shop titles that came back and did NOT match, kept so a human can
# judge whether the rejection was right. "Aachi Biryani Masala" against
# "Aachi Biryani Mix" is a correct rejection; "Anil Puttu Maavu" against
# "Anil Puttu Mix" is a matcher failure, and nothing but the title tells
# them apart.
seen_titles: List[str] = field(default_factory=list)
query: Optional[str] = None
@property
def is_found(self) -> bool:
return self.status == FOUND
def note(self) -> str:
if self.status == FOUND:
return f"listed by {self.retailer} as {self.matched_title!r}"
if self.status == NOT_FOUND:
return (f"{self.shop_results_seen} shop listing(s) seen, none for "
f"this product at this pack size")
if self.shop_results_seen == 0 and self.checked_at:
return "searched, but no shop listing was reached at all"
return "could not check retail availability"
# ---------------------------------------------------------------------------
# Cache
# ---------------------------------------------------------------------------
def _connect() -> sqlite3.Connection:
_DB_PATH.parent.mkdir(parents=True, exist_ok=True)
conn = sqlite3.connect(str(_DB_PATH), timeout=10)
conn.execute("PRAGMA journal_mode=WAL")
return conn
def _ensure_schema(conn: sqlite3.Connection) -> None:
global _initialized
if _initialized:
return
conn.execute(
"""
CREATE TABLE IF NOT EXISTS retail_presence_cache (
cache_key TEXT PRIMARY KEY,
brand TEXT,
product TEXT,
size TEXT,
result_json TEXT NOT NULL,
created_at REAL NOT NULL
)
"""
)
# One row per brand, because a brand's own domain is a fact about the
# brand and costs a search to learn. Resolved once and reused across every
# product, so it adds one query per brand rather than one per product.
conn.execute(
"""
CREATE TABLE IF NOT EXISTS brand_domain_cache (
brand TEXT PRIMARY KEY,
domain TEXT,
created_at REAL NOT NULL
)
"""
)
conn.commit()
_initialized = True
def cache_key(brand: str, product_title: str, size: str) -> str:
"""One key per PRODUCT+SIZE, with the product name size-collapsed first.
`search_key` strips a trailing pack size from the name so
"Anil Vermicelli 180g" and "Anil Vermicelli 450g" share a product identity;
the size is then a separate component, because the size is precisely what
is being verified.
"""
brand_key, product_key = search_key(product_title, brand)
size_key = re.sub(r"\s+", "", (size or "").strip().lower())
return f"{brand_key}|{product_key}|{size_key}"
def get_cached(brand: str, product_title: str, size: str,
ttl_seconds: float = DEFAULT_TTL_SECONDS) -> Optional[RetailEvidence]:
"""A cached verdict, or None for a miss. Never raises."""
key = cache_key(brand, product_title, size)
try:
with _lock, closing(_connect()) as conn:
_ensure_schema(conn)
row = conn.execute(
"SELECT result_json, created_at FROM retail_presence_cache WHERE cache_key = ?",
(key,),
).fetchone()
except Exception as e: # noqa: BLE001 - a cache failure is a cache miss
logger.debug("Retail cache read failed for %s: %s", key, e)
return None
if not row:
return None
result_json, created_at = row
if ttl_seconds > 0 and (time.time() - created_at) > ttl_seconds:
return None
try:
return RetailEvidence(**json.loads(result_json))
except Exception: # noqa: BLE001 - an unreadable entry is a miss
return None
def set_cached(brand: str, product_title: str, size: str,
evidence: RetailEvidence) -> None:
"""Best-effort write. A failure just means no caching next time.
An `unknown` verdict is NOT cached: it records that we failed to ask, not
an answer, and caching it would turn one rate-limited afternoon into two
weeks of pretending we had checked.
"""
if evidence.status == UNKNOWN:
return
key = cache_key(brand, product_title, size)
try:
with _lock, closing(_connect()) as conn:
_ensure_schema(conn)
conn.execute(
"""
INSERT INTO retail_presence_cache
(cache_key, brand, product, size, result_json, created_at)
VALUES (?, ?, ?, ?, ?, ?)
ON CONFLICT(cache_key) DO UPDATE SET
result_json = excluded.result_json,
created_at = excluded.created_at
""",
(key, brand, product_title, size,
json.dumps(asdict(evidence)), time.time()),
)
conn.commit()
except Exception as e: # noqa: BLE001 - best effort only
logger.debug("Retail cache write failed for %s: %s", key, e)
# ---------------------------------------------------------------------------
# Matching
# ---------------------------------------------------------------------------
# Domains that are never a brand's own site, however often they rank for its
# name. Retailers are excluded separately via RETAILER_DOMAINS.
_NOT_A_BRAND_SITE = (
"wikipedia.org", "wikimedia.org", "facebook.com", "instagram.com",
"linkedin.com", "youtube.com", "twitter.com", "x.com", "pinterest.com",
"indiamart.com", "justdial.com", "tradeindia.com", "exportersindia.com",
"zaubacorp.com", "tofler.in", "crunchbase.com", "glassdoor.com",
"google.com", "blogspot.com", "wordpress.com", "medium.com",
"quora.com", "reddit.com", "yelp.com", "tripadvisor.com",
)
def _registrable(url: str) -> str:
"""The registrable-ish domain of a URL ("shop.theanilgroup.com" ->
"theanilgroup.com"). Good enough for matching, not a public-suffix parser."""
host = urlparse(url if "//" in url else f"//{url}").netloc.lower()
host = host.split(":")[0]
if host.startswith("www."):
host = host[4:]
parts = host.split(".")
if len(parts) > 2 and parts[-2] in ("co", "com", "net", "org", "gov", "ac"):
return ".".join(parts[-3:]) # theanilgroup.co.in
if len(parts) > 2:
return ".".join(parts[-2:]) # shop.theanilgroup.com -> theanilgroup.com
return host
def get_cached_brand_domain(brand: str) -> Optional[str]:
"""Cached brand domain, or None for a miss. Never raises.
An empty string is a real cached answer meaning "looked and found none";
it comes back as None to callers but stops the lookup being repeated.
"""
key = _normalize_brand(brand)
try:
with _lock, closing(_connect()) as conn:
_ensure_schema(conn)
row = conn.execute(
"SELECT domain, created_at FROM brand_domain_cache WHERE brand = ?",
(key,),
).fetchone()
except Exception: # noqa: BLE001 - a cache failure is a cache miss
return None
if not row:
return None
domain, created_at = row
# A NEGATIVE EXPIRES FAST; A POSITIVE LASTS.
#
# These searches fan out across several engines and the result set varies
# run to run: the first Anil sweep found no brand site, and a repeat of the
# identical query minutes later returned shop.theanilgroup.com three times
# in the top four. Caching that miss for six months would have made one
# flaky search a permanent fact about the brand, and every product of
# Anil's would have kept reporting "sold nowhere".
#
# A found domain is stable and worth keeping; a miss is worth re-asking
# tomorrow.
ttl = BRAND_DOMAIN_TTL_SECONDS if domain else BRAND_DOMAIN_MISS_TTL_SECONDS
if (time.time() - created_at) > ttl:
return None
return domain or None
def _normalize_brand(brand: str) -> str:
return re.sub(r"\s+", " ", (brand or "").strip().lower())
def resolve_brand_domain(brand: str, *, live: bool = False,
sample_products: Optional[Sequence[str]] = None,
timeout: int = 20) -> Optional[str]:
"""The brand's own website, e.g. "Anil" -> "theanilgroup.com".
WHY THIS EXISTS. The first backfill run asked only the retailer list and
reported "Anil Wheat Vermicelli 180g" as not listed anywhere - while the
brand's own shop sells exactly that pack. A manufacturer's catalogue is
real evidence that a product and its pack size exist, and leaving it out
turned "no big retailer stocks this" into "this product is not real",
which are very different claims.
A brand site is WEAKER evidence than a retailer: a manufacturer may list a
discontinued or not-yet-shipping line. The verdict records which domain
answered, so a consumer can weigh them differently - see `RetailEvidence.
retailer`.
Identified by a domain that CONTAINS a brand token and is neither a
retailer nor a directory/social site. That is a deliberately conservative
rule: it will miss a brand whose site is named nothing like the brand, and
a miss just means we fall back to the retailer list.
"""
key = _normalize_brand(brand)
if not key:
return None
cached = get_cached_brand_domain(brand)
if cached is not None:
return cached
if not live:
return None
tokens = [t for t in off_bulk.brand_tokens(brand) if len(t) > 2]
if not tokens:
return None
# QUERY WITH A REAL PRODUCT NAME, NOT THE BRAND ALONE.
#
# "Anil official site products" returned maccosmetics.com, vertu.com and
# a YouTube video: "Anil" is a common personal name, so a brand-only query
# is dominated by cricketers and airlines - the same ambiguity that put a
# photo of Anil Kapoor on a packet of rava. Adding a product the brand
# actually sells makes the query specific enough to find the real site;
# "Anil Wheat Vermicelli" surfaces shop.theanilgroup.com immediately.
#
# Several phrasings, because one search is demonstrably not enough: the
# identical query minutes apart returned the brand site three times and
# then not at all. Stops at the first that yields a domain.
phrasings: List[str] = [f"{brand} {p}" for p in (sample_products or [])[:2]]
phrasings += [f"{brand} official site products",
f"{brand} official website India"]
found: Optional[str] = None
asked = False
for phrasing in phrasings:
if found:
break
results = _search(phrasing, 10, timeout)
if results is None:
continue
asked = True
found = _pick_brand_domain(results, tokens)
if not asked:
return None # could not ask - do NOT cache a failure as "none"
_cache_brand_domain(key, found or "")
return found
def _pick_brand_domain(results, tokens) -> Optional[str]:
"""The first result whose domain stem contains a brand token and is not a
retailer, directory or social site."""
for result in results:
url = str(result.get("href") or result.get("url") or "")
if not url.startswith("http"):
continue
domain = _registrable(url)
if not domain:
continue
if any(bad in domain for bad in _NOT_A_BRAND_SITE):
continue
if any(retailer in domain for retailer in RETAILER_DOMAINS):
continue
stem = domain.split(".")[0]
if any(token in stem for token in tokens):
return domain
return None
def _cache_brand_domain(key: str, domain: str) -> None:
try:
with _lock, closing(_connect()) as conn:
_ensure_schema(conn)
conn.execute(
"""
INSERT INTO brand_domain_cache (brand, domain, created_at)
VALUES (?, ?, ?)
ON CONFLICT(brand) DO UPDATE SET
domain = excluded.domain, created_at = excluded.created_at
""",
(key, domain, time.time()),
)
conn.commit()
except Exception as e: # noqa: BLE001 - best effort only
logger.debug("Brand domain cache write failed for %s: %s", key, e)
def _is_product_page(url: str) -> bool:
"""Does this URL point at a specific product rather than a landing page?
Only applied to the BRAND's own domain. A retailer domain is a shop by
definition, but a manufacturer's site is mostly corporate: `theanilgroup.com/`
is an "about us" homepage and `shop.theanilgroup.com/products/wheat-vermicelli`
is a listing, and both share a registrable domain.
Counting the homepage as "a shop was reached" is not harmless - it is what
decides whether a miss is reported as `not_found` (an answer) or `unknown`
(we never asked a shop). The corporate homepage ranks for almost every
query about the brand, so without this every no-shop verdict would be
dressed up as a real absence.
"""
path = urlparse(url or "").path.strip("/")
if not path:
return False
return any(
marker in f"/{path.lower()}/"
for marker in ("/product", "/products/", "/shop/", "/collections/",
"/p/", "/pd/", "/item", "/buy", "/store/")
) or path.count("/") >= 1
def _retailer_for(url: str) -> Optional[str]:
lowered = (url or "").lower()
for domain in RETAILER_DOMAINS:
if domain in lowered:
return domain
return None
def _clean_listing_title(title: str) -> str:
return _LISTING_NOISE.sub(" ", title or "")
def listing_matches(listing_title: str, brand: str, product_title: str,
size: str) -> tuple:
"""Does this listing describe THIS product at THIS pack size?
Returns (matches, coverage). Both halves are required - see the module
docstring for the measurement that makes the size half non-negotiable.
COVERAGE, NOT SYMMETRIC SIMILARITY. `off_bulk.symmetric_similarity` is
built to compare one database record with another and penalises extra
tokens on either side. A retail listing title is not a record: it is a
SUPERSET of the product name wrapped in shop furniture. Measured against
real titles the live provider returned, symmetric similarity scored
"Anil Vermicelli - Roasted, 450g Pouch : Amazon.in: Grocery &
Gourmet Foods" vs "Anil Roasted Short Vermicelli"
-> 0.445, rejected
which is a genuine listing of exactly the product asked for. So the title
half asks the containment question instead - how much of the PRODUCT's
identity appears in the listing - and the pack size does the discriminating.
"""
if not listing_title:
return False, 0.0
# Several listings run together into one title field - see
# MAX_LISTING_TITLE_CHARS. Neither the size nor the words can be attributed
# to a single product, so this corroborates nothing.
if len(listing_title) > MAX_LISTING_TITLE_CHARS:
return False, 0.0
# THE LISTING MUST BE FOR OUR BRAND.
#
# This was missing entirely, and it is the worst hole this module has had.
# `normalize_for_match` strips brand tokens from BOTH sides - correct for
# its original job, matching within one brand's own Open Food Facts corpus,
# where the brand is a given. Here the listing can be anyone's, so stripping
# the brand made the comparison brand-blind and a COMPETITOR's product
# corroborated ours:
#
# "A1 Naanjil Naattu Masala Instant Pongal Mix, 500g"
# confirmed Anil Pongal Mix 500g (real, from the backfill)
# "Britannia Good Day 200g" confirmed "Anil Good Day 200g"
# "MTR Rava Idli Mix 500g" confirmed "Anil Idli Mix 500g"
#
# Checked BEFORE the brand tokens are stripped, obviously, and via the
# barcode matcher's own `brand_matches` so aliases ("HUL" for "Hindustan
# Unilever") keep working and the two gates cannot drift apart.
if not brand_matches(listing_title, brand, _aliases_for(brand)):
return False, 0.0
drop = off_bulk.brand_tokens(brand)
listing_clean = off_bulk.normalize_for_match(
_clean_listing_title(listing_title), drop
)
target_clean = off_bulk.normalize_for_match(
off_bulk.strip_sizes(product_title), drop
)
if not listing_clean:
return False, 0.0
# Synonyms folded on BOTH sides, so "Puttu Mix" and "Puttu Maavu" reduce to
# the same words. See _SYNONYMS - observed pairs only, and the pack-size
# check below is untouched by any of it.
target_tokens = [_SYNONYMS.get(t, t) for t in target_clean.split() if len(t) > 2]
listing_tokens = _canonical_words(listing_clean)
if target_tokens:
coverage = sum(1 for t in target_tokens if t in listing_tokens) / len(target_tokens)
else:
# The product name is nothing but brand and size ("Amul 1kg"). There
# is no identity to cover, so the brand's presence is all that can be
# asked, and the size below carries the whole verdict.
coverage = 1.0 if any(
b in listing_title.lower() for b in off_bulk.brand_tokens(brand)
) else 0.0
if coverage < TITLE_COVERAGE_FLOOR:
return False, round(coverage, 3)
# A variant term the other side does not have means a different pack of a
# similarly-named product (sugar-free vs regular, jar vs pouch).
if off_bulk.has_extra_variant_conflict(listing_title, product_title):
return False, round(coverage, 3)
# An INGREDIENT the listing names and the product does not. Containment
# tolerates extra words on the listing side, which is right for shop
# furniture and wrong for this: "Aachi Ginger Paste" is wholly contained in
# "Aachi Ginger Garlic Paste" and scored 1.0 against it. See
# _DISTINGUISHING_INGREDIENTS.
extra = (listing_tokens & _DISTINGUISHING_INGREDIENTS) - set(target_tokens)
if extra:
return False, round(coverage, 3)
# THE SIZE CHECK, AND IT IS THE ONE THAT MATTERS.
#
# Measured on the live provider: searching "Anil Wheat Vermicelli 12g"
# returns the brand's own generic product page, whose title covers 100% of
# the product's tokens. Only the absence of any 12 g mention distinguishes
# a pack that exists from one that does not. A listing that states no
# quantity at all confirms no quantity at all.
if size and size.strip():
mentions = quantity_utils.find_quantity_mentions(listing_title)
if not mentions:
return False, round(coverage, 3)
if not any(
quantity_utils.quantities_match(size, f"{q}g") for q in mentions
):
return False, round(coverage, 3)
return True, round(coverage, 3)
# ---------------------------------------------------------------------------
# The lookup
# ---------------------------------------------------------------------------
def _search(query: str, max_results: int, timeout: int) -> Optional[List[Dict[str, Any]]]:
"""Raw web results, or None when the provider could not be reached.
None and [] are different answers and the caller depends on it: None is
"we could not ask" and becomes `unknown`; [] is "we asked and nobody sells
this" and becomes `not_found`.
"""
try:
from ddgs import DDGS
except ImportError:
logger.debug("Retail presence unavailable: 'ddgs' is not installed")
return None
try:
with DDGS(timeout=timeout) as ddgs:
return list(ddgs.text(query, region="in-en", safesearch="off",
max_results=max_results) or [])
except Exception as e: # noqa: BLE001 - a throttle is not a verdict
logger.debug("Retail search failed for %r: %s", query, e)
return None
def check_listing(brand: str, product_title: str, size: str, *,
live: bool = False,
refresh: bool = False,
brand_domain: Optional[str] = None,
ttl_seconds: float = DEFAULT_TTL_SECONDS,
max_results: int = 10,
timeout: int = 20) -> RetailEvidence:
"""Is (brand, product_title, size) listed by a real retailer?
`live` defaults to FALSE. Ingestion reads the cache and never issues a
network call; the backfill script passes live=True. See the module
docstring - this is one blocking call per product against a provider that
rate-limits, so it does not belong inside a batch.
"""
if not (brand or "").strip() or not (product_title or "").strip():
return RetailEvidence(UNKNOWN)
# `refresh` SKIPS THE CACHE READ, and without it a caller cannot re-ask.
#
# The backfill script has its own --refresh, which decides which rows are
# worth asking again - but it then called this function, which read the
# cache first and handed back the very verdict the caller was trying to
# replace. A whole "rebuild" of Anil ran to completion, reported 46
# re-asks, and made no network calls at all: it re-read 46 stale rows and
# wrote nothing, because the early return happens before set_cached.
#
# Silent, and it looks exactly like a successful run.
if not refresh:
cached = get_cached(brand, product_title, size, ttl_seconds)
if cached is not None:
return cached
if not live:
return RetailEvidence(UNKNOWN)
# Resolve the brand's own site if the caller did not name one. Cached per
# brand, so this costs one extra search per BRAND, not per product. Without
# it the first Anil sweep reported "Anil Wheat Vermicelli 180g" as sold
# nowhere while the brand's own shop lists exactly that pack.
if brand_domain is None:
brand_domain = resolve_brand_domain(brand, live=True, timeout=timeout)
# A NEGATIVE IS CONFIRMED WITH A SECOND PHRASING, NOT TAKEN ON ONE TRY.
#
# `ddgs` fans out over several engines and the result set genuinely varies
# between identical queries minutes apart - measured: "Anil Wheat
# Vermicelli 180g" returned nothing usable on one run and an Amazon listing
# for exactly that pack on the next. A single query is enough to CONFIRM a
# product (a real listing is real however it was found) but not enough to
# deny one, so only the negative pays for the retry.
base = off_bulk.strip_sizes(product_title)
phrasings = [
" ".join(p for p in (brand, base, size) if p).strip(),
" ".join(p for p in (brand, base, size, "buy online price") if p).strip(),
]
# A site:-scoped query is the reliable way to reach the brand's own shop,
# but it is worth a third round trip only when the open queries reached no
# shop at all - see the reach check after the loop.
if brand_domain:
phrasings.append(
" ".join(p for p in (brand, base, size, f"site:{brand_domain}") if p).strip()
)
domains = tuple(RETAILER_DOMAINS) + ((brand_domain,) if brand_domain else ())
best = RetailEvidence(NOT_FOUND, checked_at=time.time())
asked = False
shop_results_seen = 0
seen_titles: List[str] = []
for query in phrasings:
# Only escalate to the site:-scoped query when nothing else reached a
# shop. A product already seen on a retailer needs no third opinion.
if "site:" in query and shop_results_seen:
break
results = _search(query, max_results, timeout)
if results is None:
continue
asked = True
for result in results:
url = str(result.get("href") or result.get("url") or "")
title = str(result.get("title") or "")
retailer = _retailer_for(url) or (
brand_domain if brand_domain and brand_domain in _registrable(url)
and _is_product_page(url) else None
)
if not retailer or retailer not in domains:
continue
# Reached an actual shop. Counted whether or not it matches, because
# that count is what separates "shops do not sell this" from "we
# never got to ask a shop".
shop_results_seen += 1
if len(seen_titles) < 8 and title:
seen_titles.append(f"{retailer}: {title}"[:160])
matches, similarity = listing_matches(title, brand, product_title, size)
if similarity > best.similarity:
best.similarity = round(similarity, 3)
if matches:
evidence = RetailEvidence(
FOUND, retailer=retailer, url=url, matched_title=title,
matched_size=size, similarity=round(similarity, 3),
checked_at=time.time(), shop_results_seen=shop_results_seen,
query=query,
)
set_cached(brand, product_title, size, evidence)
return evidence
if not asked:
# Could not ask at all - every phrasing failed to reach the provider.
return RetailEvidence(UNKNOWN)
# ASKED, BUT NEVER ASKED A SHOP.
#
# The searches came back full of recipe blogs, news pages and the brand's
# corporate homepage, and not one result was on a retailer or the brand's
# store. Recording that as "no retailer sells this product" states something
# we did not find out - it is the same conflation as treating a throttle as
# an absence, and it is what made 2 of 10 sampled Anil negatives meaningless.
#
# Reported as UNKNOWN, and UNKNOWN is never cached, so the question gets
# asked again instead of the product being written off for a fortnight.
if shop_results_seen == 0:
return RetailEvidence(UNKNOWN, checked_at=time.time(), shop_results_seen=0)
best.shop_results_seen = shop_results_seen
best.seen_titles = seen_titles
set_cached(brand, product_title, size, best)
return best
def check_many(items: Sequence[Dict[str, str]], *, live: bool = False,
deadline_seconds: Optional[float] = None,
pause_seconds: float = PAUSE_SECONDS) -> Dict[str, RetailEvidence]:
"""Check a batch, keyed by `cache_key`.
SEQUENTIAL, WITH PACING - deliberately not concurrent. `EnrichmentPipeline`
bounds concurrency across ROWS, which is right when rows hit independent
resources; here every row hits one throttled provider, so parallelism buys
a 403 rather than throughput. `off_bulk.PAUSE_SECONDS` is the in-repo
precedent for exactly this.
Duplicate keys are asked once: `cache_key` collapses pack-size variants of
a product onto one product identity, so a brand with three sizes of each
item costs a third of the queries.
"""
out: Dict[str, RetailEvidence] = {}
started = time.monotonic()
for item in items:
brand = item.get("brand") or ""
title = item.get("product_title") or item.get("title") or ""
size = item.get("size") or ""
key = cache_key(brand, title, size)
if key in out:
continue
if deadline_seconds is not None and (time.monotonic() - started) > deadline_seconds:
# Everything still unasked is `unknown`, not `not_found`.
out[key] = RetailEvidence(UNKNOWN)
continue
cached = get_cached(brand, title, size)
if cached is not None:
out[key] = cached
continue
out[key] = check_listing(brand, title, size, live=live)
if live:
time.sleep(pause_seconds)
return out

View File

@@ -92,8 +92,24 @@ CATEGORY_TYPE_WORDS: dict[str, set[str]] = {
"curry powder": {"Spices & Masalas"}, "garam masala": {"Spices & Masalas"},
"chilli powder": {"Spices & Masalas"}, "turmeric powder": {"Spices & Masalas"},
"deggi mirch": {"Spices & Masalas"}, "tikhalal": {"Spices & Masalas"}, "kashmiri lal": {"Spices & Masalas"},
"pasta": {"Pasta & Noodles"}, "vermicelli": {"Pasta & Noodles"}, "semiya": {"Pasta & Noodles"},
"macaroni": {"Pasta & Noodles"}, "spaghetti": {"Pasta & Noodles"},
# BOTH spellings, for every word in this group.
#
# `category_registry` is the module that ASSIGNS the category, and its
# entry for "Noodles & Instant Food" lists "vermicelli" and "pasta" among
# its own keywords. So the detector resolved "Anil Wheat Vermicelli" to
# "Noodles & Instant Food" and this table then reported the word
# "vermicelli" as CONTRADICTING it - a -0.35 confidence penalty on every
# vermicelli, pasta, semiya, macaroni and spaghetti row in the catalogue,
# for nothing but two tables spelling one category differently.
#
# `noodle`/`noodles` already carried both, because somebody hit this and
# patched the row in front of them. The rest of the group has the same
# bug. If a new pasta-ish word is added here, give it both.
"pasta": {"Pasta & Noodles", "Noodles & Instant Food"},
"vermicelli": {"Pasta & Noodles", "Noodles & Instant Food"},
"semiya": {"Pasta & Noodles", "Noodles & Instant Food"},
"macaroni": {"Pasta & Noodles", "Noodles & Instant Food"},
"spaghetti": {"Pasta & Noodles", "Noodles & Instant Food"},
"noodle": {"Pasta & Noodles", "Noodles & Instant Food"}, "noodles": {"Pasta & Noodles", "Noodles & Instant Food"},
# --- Desserts (head-noun only, not ingredient words) ---
"ice cream": {"Ice Cream"}, "icecream": {"Ice Cream"}, "kulfi": {"Ice Cream"}, "frozen dessert": {"Ice Cream"},

View File

@@ -178,6 +178,17 @@ def get_brand_table_ddl(brand: str) -> str:
nutrients_per_100g JSONB,
search_query TEXT,
-- The deterministic validation verdict from product_validator.
-- validate_catalog() has always computed these three and annotated
-- them onto every row; until they were added here nothing projected
-- them, so a fabricated row and a corroborated one were
-- indistinguishable once stored. That is what made "is this product
-- real?" unanswerable after the fact.
-- status is one of: verified | needs_review | rejected.
validation_status TEXT,
confidence_score NUMERIC,
validation_issues JSONB,
-- Per-field provenance, keyed by column name; each value records how
-- that column's value was arrived at. One JSONB map rather than ~60
-- scalar columns, because every scalar column would have to be named
@@ -274,6 +285,11 @@ def _ensure_columns(cur, table_name: str) -> None:
"nutrients": "TEXT[]",
"nutrients_per_100g": "JSONB",
"search_query": "TEXT",
# The product_validator verdict - see get_brand_table_ddl for why a
# row's confidence has to survive to disk to be worth computing.
"validation_status": "TEXT",
"confidence_score": "NUMERIC",
"validation_issues": "JSONB",
# Per-field provenance map - see get_brand_table_ddl for why this is
# one JSONB column and not sixty scalar ones.
"field_sources": "JSONB",
@@ -488,6 +504,18 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
field_sources = p.get("field_sources")
field_sources = Json(field_sources) if isinstance(field_sources, dict) and field_sources else None
# The product_validator verdict, as annotated by validate_catalog().
# A row that never went through the gate carries none of these, so
# they arrive as None and the DO UPDATE SET below assigns NULL - see
# the comment there for why that is the wanted behaviour here and the
# opposite of the rule the enrichment columns follow.
validation_status = str(p.get("validation_status") or "").strip() or None
confidence_score = _to_numeric_or_none(p.get("confidence_score"))
validation_issues = p.get("validation_issues")
validation_issues = (
Json(list(validation_issues)) if isinstance(validation_issues, (list, tuple)) else None
)
# Essential fields
highlights = p.get("highlights", [])
if not isinstance(highlights, list):
@@ -542,6 +570,9 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
highlights, # TEXT[] - psycopg will handle conversion
nutrients, # TEXT[] - psycopg will handle conversion
search_query,
validation_status,
confidence_score,
validation_issues,
field_sources,
embedding_str
))
@@ -586,8 +617,10 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
fssai_license, product_sku, sku_source, hsn_code, final_selling_price, selling_price, barcode, barcode_type,
gtin, ean13, upc, barcode_source, barcode_verified, barcode_lookup_status, barcode_last_updated,
gst_percent, tax_amount, hsn_gst_needs_review,
highlights, nutrients, search_query, field_sources, embedding)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
highlights, nutrients, search_query,
validation_status, confidence_score, validation_issues,
field_sources, embedding)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
ON CONFLICT (image_id) DO UPDATE SET
product_name = EXCLUDED.product_name,
title = EXCLUDED.title,
@@ -625,6 +658,26 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
nutrients = EXCLUDED.nutrients,
search_query = EXCLUDED.search_query,
embedding = EXCLUDED.embedding,
-- PLAINLY ASSIGNED, NOT COALESCEd. THIS IS DELIBERATE AND
-- IS THE OPPOSITE OF THE RULE THE BLOCK BELOW FOLLOWS.
--
-- Those are enrichment outputs: facts about the product,
-- true whenever they were learned, so keeping a stored one
-- when nothing new arrives is right. These three are not
-- facts about the product - they are the verdict of the
-- run that just wrote this row. Carrying a previous run's
-- "verified" forward onto a row that has just been rewritten
-- and not re-validated is precisely the failure this column
-- was added to end: it would relabel an unchecked row as
-- checked, which is worse than admitting NULL.
--
-- So a writer that does not validate (the seed loader,
-- user_products, brand_sync's re-seed) correctly clears the
-- verdict, and the row reads as "not assessed" until a
-- validating path runs over it again.
validation_status = EXCLUDED.validation_status,
confidence_score = EXCLUDED.confidence_score,
validation_issues = EXCLUDED.validation_issues,
-- COALESCE, NOT PLAIN EXCLUDED, FOR EVERY COLUMN BELOW.
--
-- These are enrichment outputs. Most writers that reach