product generation with validation check
This commit is contained in:
@@ -18,13 +18,17 @@ import logging
|
||||
sys.path.append(str(Path(__file__).parent.parent))
|
||||
|
||||
from app.services.ollama_service import fetch_brand_catalog_with_gemini, fetch_brand_catalog_exhaustive, fetch_product_details
|
||||
from app.services.image_search import find_all_image_urls, find_product_quantity_openfacts
|
||||
from app.services.image_search import find_all_image_urls
|
||||
from app.infrastructure.settings import DATA_DIR, USE_OLLAMA
|
||||
from app.services.embeddings_service import embed_texts
|
||||
from app.services.vector_store import ensure_brand_schema, upsert_brand_products, get_existing_product_image_id
|
||||
from app.services.s3_service import s3_service
|
||||
from app.services.brand_registry import resolve_parent_brand
|
||||
from app.services import image_corroboration
|
||||
from app.services import price_estimator
|
||||
from app.services import product_grounding
|
||||
from app.services.enrichment.barcode.sources import off_bulk
|
||||
from app.services.product_validator import validate_catalog
|
||||
from app.services.category_registry import detect_category_from_text, sanitize_category_language
|
||||
|
||||
# Configure logging
|
||||
@@ -611,15 +615,43 @@ class ProductCatalogEngine:
|
||||
|
||||
# Step 2: Enhance each product with comprehensive image search
|
||||
enhanced_products = []
|
||||
|
||||
# Parallel to enhanced_products: did an external source corroborate
|
||||
# that each product exists? Collected here and handed to
|
||||
# validate_catalog at the end - see the Step 3 comment for why this
|
||||
# path had no validation gate at all until now.
|
||||
grounded_flags: List[bool] = []
|
||||
|
||||
# Cap products by max_products
|
||||
discovered_products = discovered_products[:max_products]
|
||||
|
||||
# Fetched ONCE for the whole brand, not once per product: this is a
|
||||
# disk-cached whole-catalogue fetch (1-5 requests per brand), which is
|
||||
# the entire reason it is used here instead of the per-product live
|
||||
# search that looks like the natural fit. See product_grounding's
|
||||
# docstring - that endpoint is rate-limited to 10 requests/minute and
|
||||
# was returning 503 when measured.
|
||||
try:
|
||||
brand_corpus = off_bulk.fetch_brand_corpus(brand)
|
||||
except Exception as e: # noqa: BLE001 - unreachable corpus is not a verdict
|
||||
logger.warning("Open*Facts corpus unavailable for %s: %s", brand, e)
|
||||
brand_corpus = None
|
||||
if brand_corpus is not None:
|
||||
logger.info("📚 Open*Facts corpus for %s: %d real products", brand, len(brand_corpus))
|
||||
|
||||
total_products_to_process = len(discovered_products)
|
||||
for i, product in enumerate(discovered_products):
|
||||
product_title = product.get('title', '')
|
||||
logger.info(f"🔍 Processing product {i+1}/{total_products_to_process}: {product_title}")
|
||||
|
||||
|
||||
# Does anything outside this process say this product exists?
|
||||
# The LLM that produced `product_title` is a 1.5B local model and
|
||||
# cannot be asked to check its own work.
|
||||
grounding = product_grounding.ground_product(
|
||||
brand, product_title, corpus=brand_corpus
|
||||
)
|
||||
if grounding.status == product_grounding.NOT_FOUND:
|
||||
logger.warning("⚠️ Ungrounded: %r - %s", product_title, grounding.note())
|
||||
|
||||
# Strip brand prefix from title if present (avoids redundant
|
||||
# "Cadbury Perk Cadbury Perk Crunch" style queries that confuse
|
||||
# image search APIs).
|
||||
@@ -670,7 +702,26 @@ class ProductCatalogEngine:
|
||||
# Select exactly 20 best images
|
||||
all_prioritized = prioritized_images + other_images
|
||||
final_images = self._select_best_images(all_prioritized, product_title, brand, max_images=20)
|
||||
|
||||
|
||||
# _select_best_images ORDERS; it does not decide whether any
|
||||
# candidate is this product. Applied BEFORE the S3 upload below so a
|
||||
# photo of somebody else is never copied into our own bucket, where
|
||||
# its origin stops being visible at all.
|
||||
#
|
||||
# Nothing is dropped - the whole list is still uploaded and stored,
|
||||
# so an operator can look. Only the promotion to `image_url` is
|
||||
# withheld, and only when no candidate corroborates the product.
|
||||
image_choice = image_corroboration.choose_primary(
|
||||
final_images, product_title, brand or ""
|
||||
)
|
||||
final_images = list(image_choice.ordered)
|
||||
primary_eligible = image_choice.primary is not None
|
||||
if final_images and not primary_eligible:
|
||||
logger.warning(
|
||||
"No corroborated image for %r (%s) - storing candidates but "
|
||||
"leaving image_url empty", product_title, image_choice.reason,
|
||||
)
|
||||
|
||||
# Check if product already exists in DB to avoid duplicates
|
||||
image_id_val = ""
|
||||
try:
|
||||
@@ -836,18 +887,19 @@ class ProductCatalogEngine:
|
||||
if not _variant_size_price_pairs:
|
||||
default_sizes = price_estimator.default_size_variants(category_value, product_title)
|
||||
|
||||
# Ground this in a real packaging size where possible:
|
||||
# Open Food/Beauty/Products Facts reports an actual
|
||||
# `quantity` field (e.g. "200 g", "1 l") for products it
|
||||
# has on file, which is far more trustworthy than the
|
||||
# category preset list (itself just a fallback for when
|
||||
# *nothing* else is known). If we have one, swap it in for
|
||||
# the closest preset rather than presenting a size that may
|
||||
# not actually exist for this exact product.
|
||||
try:
|
||||
real_qty = find_product_quantity_openfacts(product_title, brand)
|
||||
except Exception:
|
||||
real_qty = None
|
||||
# Ground this in a real packaging size where possible. The
|
||||
# quantity comes from the brand corpus fetched once above,
|
||||
# NOT from find_product_quantity_openfacts, which issues a
|
||||
# live per-product query against an endpoint rate-limited to
|
||||
# 10 requests/minute - see product_grounding's docstring.
|
||||
#
|
||||
# ONLY WHEN THE PRODUCT ITSELF IS CORROBORATED. A quantity
|
||||
# borrowed from a product that merely scored well is worse
|
||||
# than the preset it replaces: it makes a fabricated row look
|
||||
# MORE real by dressing it in a size that genuinely exists.
|
||||
# An ungrounded product keeps the honest category preset and
|
||||
# is flagged for review instead.
|
||||
real_qty = grounding.quantity if grounding.is_grounded else None
|
||||
if real_qty and real_qty not in default_sizes:
|
||||
default_sizes = [real_qty] + default_sizes[:2]
|
||||
|
||||
@@ -915,10 +967,16 @@ class ProductCatalogEngine:
|
||||
# is only used as a last resort since small local models frequently
|
||||
# hallucinate image links that don't actually resolve to an image.
|
||||
all_image_urls = s3_uploaded_urls or final_images
|
||||
primary_image = all_image_urls[0] if all_image_urls else (
|
||||
# `primary_eligible` is the corroboration verdict computed above.
|
||||
# S3 preserves candidate order (image_000, image_001, ...), so
|
||||
# index 0 here is the same picture choose_primary judged.
|
||||
primary_image = (all_image_urls[0] if (all_image_urls and primary_eligible) else None) or (
|
||||
enriched_img if enriched_img and str(enriched_img).startswith('http') else None
|
||||
)
|
||||
if not primary_image and s3_service.enabled and image_id_val:
|
||||
# Only reachable when there was no usable candidate at all. Guarded
|
||||
# by `primary_eligible` too, because this constructs a URL to
|
||||
# image_000 - the very image corroboration just declined to promote.
|
||||
if not primary_image and primary_eligible and s3_service.enabled and image_id_val:
|
||||
primary_image = s3_service.get_product_image_url(brand, image_id_val)
|
||||
if not all_image_urls and primary_image:
|
||||
all_image_urls = [primary_image]
|
||||
@@ -958,8 +1016,36 @@ class ProductCatalogEngine:
|
||||
}
|
||||
|
||||
enhanced_products.append(enhanced_product)
|
||||
grounded_flags.append(grounding.is_grounded)
|
||||
logger.info(f"✅ Enhanced {product_title}: {len(final_images)} images")
|
||||
|
||||
|
||||
# Step 2.6: THE DETERMINISTIC VALIDATION GATE.
|
||||
#
|
||||
# This path did not have one. `validate_catalog` was called from the
|
||||
# spreadsheet pipeline and from the Dagster assets, but never from
|
||||
# here, so POST /api/catalog/generate wrote straight to the brand
|
||||
# table with nothing between the language model and the database -
|
||||
# while product_validator's own docstring claimed this call site
|
||||
# already existed. That claim has been corrected along with this fix.
|
||||
#
|
||||
# Rows are annotated and kept, not dropped: `validate_catalog` returns
|
||||
# "verified" and "needs_review" rows together, and A1 gave the brand
|
||||
# tables somewhere to record which is which. Only rows scoring below
|
||||
# the reject threshold are withheld.
|
||||
enhanced_products, rejected, summary = validate_catalog(
|
||||
enhanced_products, brand, grounded_flags=grounded_flags,
|
||||
# These rows came out of a 1.5B language model, so "well formed"
|
||||
# is not evidence of anything. A row nothing corroborates is kept
|
||||
# and scored, but capped at needs_review rather than presented as
|
||||
# verified.
|
||||
require_grounding=True,
|
||||
)
|
||||
if rejected:
|
||||
logger.warning(
|
||||
"🚫 %d of %d generated product(s) failed validation for %s",
|
||||
len(rejected), summary.get("total_evaluated", 0), brand,
|
||||
)
|
||||
|
||||
# Step 3: Generate final catalog
|
||||
catalog = {
|
||||
'brand': brand,
|
||||
|
||||
@@ -69,6 +69,7 @@ from app.api.routers.user_products import (
|
||||
read_products_dataframe,
|
||||
row_to_request,
|
||||
)
|
||||
from app.services import image_corroboration
|
||||
from app.services import price_estimator
|
||||
from app.services.brand_registry import (
|
||||
BRAND_ALIASES,
|
||||
@@ -587,8 +588,25 @@ def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, An
|
||||
candidates, product_name, brand, max_images=10
|
||||
)
|
||||
if best:
|
||||
row["image_urls"] = list(best)
|
||||
row["image_url"] = best[0]
|
||||
# `_select_best_images` ORDERS candidates; it does not judge
|
||||
# whether any of them is this product. Its scoring awards points
|
||||
# for the brand appearing in the URL, so for a brand that is
|
||||
# also a personal name it actively rewarded the wrong photo -
|
||||
# `Anil_Kapoor_2019.jpg` outscored everything and became
|
||||
# `image_url`. choose_primary decides what may be promoted.
|
||||
#
|
||||
# The list is still stored in full. A product whose only
|
||||
# candidates are uncorroborated keeps them for review; it just
|
||||
# does not get one of them presented as fact.
|
||||
choice = image_corroboration.choose_primary(
|
||||
best, product_name, brand or ""
|
||||
)
|
||||
row["image_urls"] = list(choice.ordered)
|
||||
row["image_url"] = choice.primary
|
||||
if choice.primary is None:
|
||||
row.setdefault("_notes", []).append(
|
||||
f"no primary image: {choice.reason}"
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 - an image is not worth the row
|
||||
# `warning`, not `debug`: at the default log level a debug line is
|
||||
# invisible, so a row that silently lost its images looked identical to
|
||||
@@ -776,6 +794,14 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"highlights": list(row.get("highlights") or []),
|
||||
"nutrients": list(row.get("nutrients") or []),
|
||||
"search_query": search_query,
|
||||
# Stage 10's verdict. validate_catalog() has always annotated these
|
||||
# three onto every row it kept; this dict dropped all three, so the
|
||||
# confidence the gate computed died here and every stored row looked
|
||||
# equally trustworthy. Same rule as the block above: computed but not
|
||||
# projected is computed for nothing.
|
||||
"validation_status": row.get("validation_status"),
|
||||
"confidence_score": row.get("confidence_score"),
|
||||
"validation_issues": list(row.get("validation_issues") or []),
|
||||
"field_sources": dict(row.get("field_sources") or {}),
|
||||
}
|
||||
|
||||
|
||||
@@ -105,7 +105,7 @@ from app.infrastructure.settings import (
|
||||
BRAND_DISCOVERY_USE_LLM,
|
||||
BRAND_DISCOVERY_USE_OFF,
|
||||
)
|
||||
from app.services import active_brands, ollama_service
|
||||
from app.services import active_brands, ollama_service, retail_presence
|
||||
from app.services.brand_registry import (
|
||||
get_fssai_license,
|
||||
get_known_sub_brands,
|
||||
@@ -119,6 +119,7 @@ from app.services.category_units import (
|
||||
parse_unit,
|
||||
)
|
||||
from app.services.enrichment.barcode.sources import off_bulk
|
||||
from app.services.product_grounding import GROUNDING_SIMILARITY_FLOOR
|
||||
from app.services.vector_store import _sanitize_name, get_products_by_brand
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -594,13 +595,51 @@ def _registry_terms(brand: str) -> List[str]:
|
||||
return terms
|
||||
|
||||
|
||||
def _evidence_for(normalised: str, *, off_keys: Dict[str, Any],
|
||||
registry_terms: Sequence[str], catalog_keys: Dict[str, str]) -> Optional[str]:
|
||||
"""What corroborates this product, cheapest source first.
|
||||
def _matches_any_key(normalised: str, keys: Dict[str, Any]) -> bool:
|
||||
"""Is `normalised` one of `keys`, allowing for spelling drift?
|
||||
|
||||
Three tiers, none of which cost a network call at this point: the OFF index
|
||||
was built during the sweep, the registry terms are a Python constant, and
|
||||
the catalog index was read once.
|
||||
EXACT EQUALITY WAS A BUG. These keys come from Open Food Facts and from
|
||||
our own stored rows, and neither spells a product the way discovery does.
|
||||
Measured: our "Anil Samba Rava" normalises to `samba rava`, while Open
|
||||
Food Facts holds the same product as "SAMBA RAVVA" -> `samba ravva`. Not
|
||||
equal, so a real product with a real barcode was reported as having no
|
||||
corroboration at all.
|
||||
|
||||
The similarity floor is the one `product_grounding` measured against this
|
||||
exact corpus - see that module for why it is not the 0.78 used for barcode
|
||||
attachment, and why `samba rava`/`samba ravva` (0.681) is the case that
|
||||
fixes the number.
|
||||
"""
|
||||
if not normalised:
|
||||
return False
|
||||
if normalised in keys:
|
||||
return True
|
||||
for key in keys:
|
||||
if not key:
|
||||
continue
|
||||
if off_bulk.symmetric_similarity(key, normalised) >= GROUNDING_SIMILARITY_FLOOR:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _evidence_for(normalised: str, *, off_keys: Dict[str, Any],
|
||||
registry_terms: Sequence[str], catalog_keys: Dict[str, str],
|
||||
retail_keys: Optional[Dict[str, Any]] = None) -> Optional[str]:
|
||||
"""What corroborates this product, strongest source first.
|
||||
|
||||
Four tiers now. None costs a network call at this point: the OFF index was
|
||||
built during the sweep, the registry terms are a Python constant, the
|
||||
catalog index was read once, and the retail index is read from a cache the
|
||||
backfill script populated offline - see retail_presence for why a live
|
||||
query must never happen inside ingestion.
|
||||
|
||||
`retail` sits directly below `openfacts` and above `catalog` and
|
||||
`registry`, because it is the only tier that is BOTH external and current.
|
||||
`catalog` is our own prior output, which is circular if that output was
|
||||
itself ungrounded, and `registry` is a hardcoded Python list. For a
|
||||
non-food brand those two were the only tiers available at all, which is
|
||||
why a toothpaste catalogue could be entirely language-model output and
|
||||
still look corroborated.
|
||||
|
||||
Deliberately NOT a tier: "an image search returned a URL naming this
|
||||
product". `catalog_engine._select_best_images` documents that CDN filenames
|
||||
@@ -610,9 +649,11 @@ def _evidence_for(normalised: str, *, off_keys: Dict[str, Any],
|
||||
"""
|
||||
if not normalised:
|
||||
return None
|
||||
if normalised in off_keys:
|
||||
if _matches_any_key(normalised, off_keys):
|
||||
return "openfacts"
|
||||
if normalised in catalog_keys:
|
||||
if retail_keys and normalised in retail_keys:
|
||||
return "retail"
|
||||
if _matches_any_key(normalised, catalog_keys):
|
||||
return "catalog"
|
||||
tokens = set(normalised.split())
|
||||
if tokens & set(registry_terms):
|
||||
@@ -632,6 +673,25 @@ def _score(sources: Sequence[str], evidence: Optional[str]) -> float:
|
||||
return 1.0
|
||||
if has_off:
|
||||
return 0.9
|
||||
# An Open Food Facts row corroborates this product, but under a DIFFERENT
|
||||
# SPELLING, so the merge did not fold the two together and this candidate
|
||||
# never acquired the "off" source.
|
||||
#
|
||||
# Unreachable while `_evidence_for` matched keys by exact equality - an
|
||||
# exact match always merged. It became reachable when that matching was
|
||||
# relaxed to handle real spelling drift ("Anil Samba Rava" against Open
|
||||
# Food Facts' "SAMBA RAVVA"), and without this branch such a row fell all
|
||||
# the way through to 0.25 - scored as though nothing corroborated it, when
|
||||
# a real database record does.
|
||||
if evidence == "openfacts":
|
||||
return 0.85
|
||||
# A real shop is currently listing this exact product at this pack size.
|
||||
# Scored above `catalog` (our own prior output, which is circular when
|
||||
# that output was itself ungrounded) and far above `registry` (a hardcoded
|
||||
# list), because it is the only tier that is both external and current.
|
||||
# For a non-food brand it is usually the ONLY external tier available.
|
||||
if evidence == "retail":
|
||||
return 0.85
|
||||
if evidence == "catalog":
|
||||
return 0.7
|
||||
if evidence == "registry":
|
||||
@@ -685,7 +745,9 @@ def _resolve_sizes(candidate: Dict[str, Any], title: str, category: str,
|
||||
def _build_product(brand: str, candidate: Dict[str, Any], *, max_sizes: int,
|
||||
catalog_keys: Dict[str, str], off_keys: Dict[str, Any],
|
||||
registry_terms: Sequence[str],
|
||||
fssai: Optional[str]) -> Optional[DiscoveredProduct]:
|
||||
fssai: Optional[str],
|
||||
retail_keys: Optional[Dict[str, Any]] = None,
|
||||
) -> Optional[DiscoveredProduct]:
|
||||
raw_title = _canonicalise_title_size((candidate.get("title") or "").strip())
|
||||
if not raw_title:
|
||||
return None
|
||||
@@ -769,7 +831,8 @@ def _build_product(brand: str, candidate: Dict[str, Any], *, max_sizes: int,
|
||||
shape = {"title": title, "category": category_hint, "size_variants": sizes,
|
||||
"description": description}
|
||||
evidence = _evidence_for(normalised, off_keys=off_keys,
|
||||
registry_terms=registry_terms, catalog_keys=catalog_keys)
|
||||
registry_terms=registry_terms, catalog_keys=catalog_keys,
|
||||
retail_keys=retail_keys)
|
||||
sources = list(dict.fromkeys(candidate.get("sources") or [candidate.get("source")]))
|
||||
sources = [s for s in sources if s]
|
||||
|
||||
@@ -835,6 +898,7 @@ def discover_brand_products(
|
||||
if key:
|
||||
off_keys.setdefault(key, candidate)
|
||||
|
||||
|
||||
llm_candidates: List[Dict[str, Any]] = []
|
||||
if use_llm:
|
||||
llm_candidates = _from_llm(brand, deadline=deadline, budget=max_products * 2)
|
||||
@@ -955,6 +1019,32 @@ def discover_brand_products(
|
||||
chosen = _canonicalise_title_size(canonical_titles[key])
|
||||
entry["title"] = off_bulk.strip_sizes(chosen).strip() or chosen
|
||||
|
||||
# What a real shop is currently listing, read from the cache that
|
||||
# `scripts/backfill_retail_presence.py` fills offline. CACHE ONLY - never a
|
||||
# live query. Discovery is interactive, and one lookup costs 2.6-5.7s
|
||||
# against a provider that 403s under load, so a live sweep here would
|
||||
# either hang the preview or get the whole run throttled - and a throttled
|
||||
# miss is indistinguishable from "nobody sells this", which is the one
|
||||
# input a corroboration tier must never be fed.
|
||||
#
|
||||
# Built from `merged`, so it covers the LLM's candidates too. That is the
|
||||
# point: an Open Food Facts row needs no second opinion, and for a NON-FOOD
|
||||
# brand there is no Open Food Facts row at all - `off_bulk` queries the food
|
||||
# database alone, so a toothpaste's only possible external corroboration is
|
||||
# this one.
|
||||
retail_keys: Dict[str, Any] = {}
|
||||
for entry in merged:
|
||||
title = entry.get("title") or ""
|
||||
key = _normalise_title(brand, title)
|
||||
if not key or key in retail_keys:
|
||||
continue
|
||||
evidence = retail_presence.check_listing(brand, title, "", live=False)
|
||||
if evidence.is_found:
|
||||
retail_keys[key] = evidence
|
||||
if retail_keys:
|
||||
logger.info("🛒 %d of %d discovered products are currently listed by a retailer",
|
||||
len(retail_keys), len(merged))
|
||||
|
||||
products: List[DiscoveredProduct] = []
|
||||
dropped = 0
|
||||
for candidate in merged:
|
||||
@@ -962,6 +1052,7 @@ def discover_brand_products(
|
||||
brand, candidate, max_sizes=max_sizes_per_product,
|
||||
catalog_keys=catalog_keys, off_keys=off_keys,
|
||||
registry_terms=registry_terms, fssai=fssai,
|
||||
retail_keys=retail_keys,
|
||||
)
|
||||
if product is None:
|
||||
continue
|
||||
|
||||
@@ -104,6 +104,13 @@ EXPORT_COLUMNS = (
|
||||
"barcode_lookup_status", "barcode_last_updated",
|
||||
"gst_percent", "tax_amount", "hsn_gst_needs_review",
|
||||
"highlights", "nutrients", "search_query", "field_sources",
|
||||
# The validation verdict. Listed here for the same reason the barcode keys
|
||||
# are: an export that strips them turns a re-seed into a silent downgrade.
|
||||
# The upsert assigns these three plainly rather than COALESCEing them, so a
|
||||
# seed file that has lost them clears the verdict on every row it restores -
|
||||
# which is right for a writer that never validated, and wrong for this one,
|
||||
# whose whole job is to echo back rows that already were.
|
||||
"validation_status", "confidence_score", "validation_issues",
|
||||
"nutrition_score", "health_score", "nutrients_per_100g",
|
||||
)
|
||||
|
||||
|
||||
@@ -4,10 +4,16 @@ catalog rows, running each stage's per-row work concurrently (bounded by
|
||||
`max_concurrency`) and NEVER letting one row's failure affect any other
|
||||
row or stage.
|
||||
|
||||
`catalog_engine.py` calls `run_default_pipeline()` once, between the
|
||||
deterministic-validation step (product_validator.py) and catalog assembly
|
||||
- see that file's "Step 2.6" for the call site and
|
||||
docs/BARCODE_ENRICHMENT.md for the full pipeline-position rationale.
|
||||
`store_catalog_pipeline.stages_8_9_enrichment()` builds an EnrichmentPipeline
|
||||
and runs it between SKU resolution and the validation gate; see
|
||||
docs/BARCODE_ENRICHMENT.md for the pipeline-position rationale.
|
||||
|
||||
This docstring used to say `catalog_engine.py` called `run_default_pipeline()`
|
||||
at a "Step 2.6". It never did - that module does not import this one at all,
|
||||
so the brand-name generation path gets no barcode, HSN/GST or content
|
||||
enrichment. Step 2.6 there is now the validation gate only. Wiring enrichment
|
||||
into that path is a real and separate piece of work; do not read this comment
|
||||
as saying it is already done.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
477
app/services/image_corroboration.py
Normal file
477
app/services/image_corroboration.py
Normal file
@@ -0,0 +1,477 @@
|
||||
"""
|
||||
Does this image URL actually depict THIS product?
|
||||
|
||||
WHY THIS FILE EXISTS
|
||||
--------------------
|
||||
`image_search.validate_image_url_live` answers "does this URL serve real image
|
||||
bytes", which is a liveness question. Nothing answered the relevance question,
|
||||
so a press photo of a person named like the brand passed every check the
|
||||
ingestion path had: it is a real image, above the byte floor, on a reputable
|
||||
host. That is how the live catalogue came to illustrate "Anil Samba Rava" with
|
||||
a photograph of the actor Anil Kapoor.
|
||||
|
||||
The logic here is ported from `scripts/repair_brand_images.py`, which already
|
||||
knew how to reject that image class - its comments name the failures it was
|
||||
written for ("Aachi Kulambu Mix" returning a press photo of a politician,
|
||||
"MTR Dosa Mix" returning an anatomy plate). It lived in a script that imports
|
||||
settings, brand_registry, produce_reference and four private `vector_store`
|
||||
symbols including `_connect`, so no ingestion path could import it. Moving the
|
||||
pure predicates here is what lets stage 6 and the catalog engine use them.
|
||||
|
||||
THE ONE SEMANTIC CHANGE MADE DURING THE MOVE
|
||||
--------------------------------------------
|
||||
The original corroborated a URL against `words + brand_tokens`:
|
||||
|
||||
return any(w in lowered for w in words + _brand_tokens(brand))
|
||||
|
||||
That works for "Aachi Kulambu Mix" precisely because *Aachi is not a human
|
||||
name*. Anil is. For product "Anil Samba Rava" under brand "Anil" the token list
|
||||
contains "anil" twice over, and `Anil_Kapoor_2019.jpg` contains "anil", so the
|
||||
gate returned True and the celebrity photo was corroborated by the very token
|
||||
that made it wrong.
|
||||
|
||||
`names_product` here requires a DISTINCTIVE token instead - the title minus the
|
||||
brand minus the pack size, which is the same set
|
||||
`catalog_engine._select_best_images` already computes to rank candidates:
|
||||
|
||||
"Anil Samba Rava" -> {samba, rava} -> rejects Anil_Kapoor_2019.jpg
|
||||
"Anil Wheat Vermicelli"-> {wheat, vermicelli}
|
||||
|
||||
Brand tokens remain, but only as an explicit fallback for titles that have no
|
||||
distinctive words at all ("Amul 1kg"), and a match found that way is reported
|
||||
as `via_brand_only` so the caller can decline to treat it as corroboration.
|
||||
|
||||
WHAT THIS MODULE MAY AND MAY NOT DECIDE
|
||||
---------------------------------------
|
||||
It decides which image is shown for a product. It never decides whether the
|
||||
PRODUCT is real. `brand_discovery._evidence_for` deliberately refuses to treat
|
||||
image naming as product evidence, because CDN filenames are frequently opaque
|
||||
hashes and absence of a naming URL is weak evidence of absence. That reasoning
|
||||
is right and this module does not disturb it: images gate images, never
|
||||
products.
|
||||
|
||||
Dependencies are stdlib plus `requests` on purpose. `catalog_engine` imports
|
||||
this module, and `repair_brand_images` already reaches back into
|
||||
`catalog_engine._select_best_images`, so a module-scope import of
|
||||
`catalog_engine` here would close a cycle.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
import sqlite3
|
||||
import threading
|
||||
import time
|
||||
from contextlib import closing
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Iterable, List, Optional, Sequence
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import requests
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_BROWSER_UA = "nearle-catalogue/1.0 (product image corroboration)"
|
||||
|
||||
# "own" and "products" are the BUCKET's name, not a brand's. Left in, the token
|
||||
# "products" corroborates any URL containing /images/products/ or
|
||||
# /cdn/shop/products/ - which is every Open*Facts and every Shopify path - so
|
||||
# `names_product` waved through 40+ images that named nothing about the item.
|
||||
# That is how openbeautyfacts cosmetics photos became the stored image for
|
||||
# Banana, Orange, Papaya, Guava, Lemon and twenty more.
|
||||
BUCKET_TOKENS = frozenset({"own", "products", "product"})
|
||||
|
||||
# A trailing pack size on a product name. Used to collapse "X 100g"/"X 500g"
|
||||
# onto one search, and to keep size digits out of the distinctive token set.
|
||||
SIZE_TAIL = re.compile(
|
||||
r"\s+\d+(?:\.\d+)?\s*(?:g|gm|gms|kg|ml|l|ltr|litre|liter|pcs|pc|n|no|nos)\b\.?\s*$",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
# A bare quantity token, e.g. "500g" or "2l". Mirrors the filter in
|
||||
# catalog_engine._select_best_images so the two agree on what a size looks like.
|
||||
_SIZE_TOKEN = re.compile(r"\d+(?:kg|g|gm|gms|ml|l|ltr|pcs|n)?", re.IGNORECASE)
|
||||
|
||||
# The barcode embedded in an Open*Facts image path, e.g.
|
||||
# /images/products/890/604/215/0067/front_en.4.400.jpg
|
||||
_OFF_BARCODE = re.compile(r"/images/products/((?:\d+/)+)")
|
||||
|
||||
# Open*Facts image hosts and the API host that can identify a barcode for each.
|
||||
#
|
||||
# The original checked `if "openfoodfacts.org" not in url: return True`, so
|
||||
# every openbeautyfacts and openproductsfacts image skipped the cross-check
|
||||
# entirely - exactly the non-food case, and exactly the two sibling databases
|
||||
# `image_search.OPEN_FACTS_HOSTS` queries. A food product got the check and a
|
||||
# toothpaste did not.
|
||||
_OFF_API_HOSTS = {
|
||||
"openfoodfacts": "world.openfoodfacts.org",
|
||||
"openbeautyfacts": "world.openbeautyfacts.org",
|
||||
"openproductsfacts": "world.openproductsfacts.org",
|
||||
}
|
||||
|
||||
# Hosts whose image paths are content-hashed or barcode-keyed, so a filename
|
||||
# that fails to name the product says nothing about the photo. These are the
|
||||
# hosts catalog_engine's own comment is about when it refuses to drop
|
||||
# uncorroborated candidates.
|
||||
OPAQUE_PATH_DOMAINS = (
|
||||
"bbassets.com", "bigbasket.com", "flixcart.com", "flipkart.com",
|
||||
"media-amazon.com", "amazon.in", "amazon.com", "jiomart.com",
|
||||
"zeptonow.com", "blinkit.com", "grofers.com", "cloudinary.com",
|
||||
"shopifycdn.com", "cdn.shopify.com", "akamaized.net", "cloudfront.net",
|
||||
"openfoodfacts.org", "openbeautyfacts.org", "openproductsfacts.org",
|
||||
)
|
||||
|
||||
# Hosts whose filenames are human-authored and descriptive. A filename that
|
||||
# fails to name the product here is real evidence that the photo is of
|
||||
# something else - and this is the host family the celebrity photo came from.
|
||||
DESCRIPTIVE_FILENAME_DOMAINS = (
|
||||
"wikimedia.org", "wikipedia.org", "wikimedia.commons",
|
||||
)
|
||||
|
||||
# Two or more Capitalised words joined by underscores, optionally with a year:
|
||||
# the Wikimedia Commons house style for a photograph OF A PERSON
|
||||
# ("Anil_Kapoor_2019.jpg"). Deliberately used only to DEMOTE, never to reject -
|
||||
# "Britannia_Good_Day.jpg" has the identical shape and is a perfectly good
|
||||
# product photo, so this signal orders candidates and is not allowed to
|
||||
# eliminate one.
|
||||
_PERSON_FILENAME = re.compile(
|
||||
r"^[A-Z][a-z]+(?:_[A-Z][a-z]+)+(?:_\d{4})?[^/]*\.(?:jpg|jpeg|png|webp)$"
|
||||
)
|
||||
_PERSON_CONTEXT = re.compile(r"_at_|_in_\d{4}|portrait|headshot", re.IGNORECASE)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tokens
|
||||
# ---------------------------------------------------------------------------
|
||||
def brand_tokens(brand: str) -> List[str]:
|
||||
"""Distinctive words of a brand name, bucket words removed."""
|
||||
return [
|
||||
w for w in re.split(r"[^a-z0-9]+", (brand or "").lower())
|
||||
if len(w) > 2 and w not in BUCKET_TOKENS
|
||||
]
|
||||
|
||||
|
||||
def distinctive_tokens(product_name: str, brand: str) -> List[str]:
|
||||
"""Words that separate THIS product from its brand-mates.
|
||||
|
||||
The brand is removed on purpose: every Britannia URL contains "britannia",
|
||||
so it separates nothing - what tells Marie Gold from Good Day is
|
||||
"marie"/"gold" vs "good"/"day". Pack sizes and short filler words go for
|
||||
the same reason. This mirrors catalog_engine._select_best_images so the
|
||||
ranking and the gate can never disagree about what identifies a product.
|
||||
"""
|
||||
brand_words = {w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) if w}
|
||||
out: List[str] = []
|
||||
for word in re.split(r"[^a-z0-9]+", (product_name or "").lower()):
|
||||
if (
|
||||
len(word) > 2
|
||||
and word not in brand_words
|
||||
and word not in BUCKET_TOKENS
|
||||
and word not in out
|
||||
and not _SIZE_TOKEN.fullmatch(word)
|
||||
):
|
||||
out.append(word)
|
||||
return out
|
||||
|
||||
|
||||
def search_key(product_name: str, brand: str) -> tuple:
|
||||
"""Collapse a trailing pack size so sizes of one product share a lookup.
|
||||
|
||||
Sharing across sizes is correct rather than merely cheap: it is the same
|
||||
product in a different pack, and stage 4's size explosion produces exactly
|
||||
these rows from a single source product.
|
||||
"""
|
||||
base = SIZE_TAIL.sub("", product_name or "").strip()
|
||||
return ((brand or "").lower(), (base or product_name or "").lower())
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Corroboration
|
||||
# ---------------------------------------------------------------------------
|
||||
@dataclass(frozen=True)
|
||||
class Corroboration:
|
||||
"""Why a URL was or was not accepted as depicting this product."""
|
||||
|
||||
corroborated: bool
|
||||
reason: str
|
||||
via_brand_only: bool = False
|
||||
|
||||
|
||||
def corroborate(url: str, product_name: str, brand: str) -> Corroboration:
|
||||
"""Does `url` name this product?
|
||||
|
||||
Distinctive tokens first. Brand tokens are consulted only when the title
|
||||
has no distinctive words of its own, and a match found that way is flagged
|
||||
`via_brand_only` - it is the weakest possible signal, and for a brand that
|
||||
is also a personal name it is the signal that produced the defect this
|
||||
module exists for.
|
||||
"""
|
||||
lowered = (url or "").lower()
|
||||
if not lowered:
|
||||
return Corroboration(False, "empty url")
|
||||
|
||||
distinctive = distinctive_tokens(product_name, brand)
|
||||
if distinctive:
|
||||
hit = next((t for t in distinctive if t in lowered), None)
|
||||
if hit:
|
||||
return Corroboration(True, f"url names {hit!r}")
|
||||
return Corroboration(
|
||||
False,
|
||||
"url names none of " + ", ".join(repr(t) for t in distinctive[:4]),
|
||||
)
|
||||
|
||||
# No distinctive words at all (e.g. "Amul 1kg"). Fall back to the brand,
|
||||
# and say so, so the caller can decide how much that is worth.
|
||||
tokens = brand_tokens(brand)
|
||||
hit = next((t for t in tokens if t in lowered), None)
|
||||
if hit:
|
||||
return Corroboration(True, f"url names brand {hit!r}", via_brand_only=True)
|
||||
return Corroboration(False, "url names neither the product nor the brand")
|
||||
|
||||
|
||||
def names_product(url: str, product_name: str, brand: str) -> bool:
|
||||
"""Boolean form of `corroborate`, for callers that only need the verdict."""
|
||||
return corroborate(url, product_name, brand).corroborated
|
||||
|
||||
|
||||
def looks_like_person_photo(url: str) -> bool:
|
||||
"""True when the FILENAME has the shape of a photograph of a person.
|
||||
|
||||
A demotion signal only - see `_PERSON_FILENAME` for why this must never
|
||||
reject on its own.
|
||||
"""
|
||||
name = urlparse(url or "").path.rsplit("/", 1)[-1]
|
||||
if not name:
|
||||
return False
|
||||
return bool(_PERSON_FILENAME.match(name)) or bool(_PERSON_CONTEXT.search(name))
|
||||
|
||||
|
||||
def has_opaque_path(url: str) -> bool:
|
||||
lowered = (url or "").lower()
|
||||
return any(d in lowered for d in OPAQUE_PATH_DOMAINS)
|
||||
|
||||
|
||||
def has_descriptive_filename(url: str) -> bool:
|
||||
lowered = (url or "").lower()
|
||||
return any(d in lowered for d in DESCRIPTIVE_FILENAME_DOMAINS)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Open*Facts barcode cross-check
|
||||
# ---------------------------------------------------------------------------
|
||||
# The barcode is embedded in the image path, so the product is one cheap lookup
|
||||
# away. If the Open*Facts record does not mention our brand, the image is
|
||||
# somebody else's product. Nothing else can catch this: the path is opaque
|
||||
# digits, so no amount of filename matching would help. OFF matches on NAME,
|
||||
# not brand - asked for "Aachi Pickles" it returned the front-of-pack photo for
|
||||
# a French "Ducros Green Pitted Olives", which validated perfectly happily
|
||||
# because it IS a real image.
|
||||
_DB_PATH = Path("data") / "cache" / "image_corroboration.db"
|
||||
_lock = threading.Lock()
|
||||
_initialized = False
|
||||
_CACHE_TTL_SECONDS = 30 * 24 * 3600
|
||||
|
||||
|
||||
def _connect() -> sqlite3.Connection:
|
||||
_DB_PATH.parent.mkdir(parents=True, exist_ok=True)
|
||||
conn = sqlite3.connect(str(_DB_PATH), timeout=10)
|
||||
conn.execute("PRAGMA journal_mode=WAL")
|
||||
return conn
|
||||
|
||||
|
||||
def _ensure_schema(conn: sqlite3.Connection) -> None:
|
||||
global _initialized
|
||||
if _initialized:
|
||||
return
|
||||
conn.execute(
|
||||
"""
|
||||
CREATE TABLE IF NOT EXISTS openfacts_brand_check (
|
||||
cache_key TEXT PRIMARY KEY,
|
||||
verdict INTEGER NOT NULL,
|
||||
created_at REAL NOT NULL
|
||||
)
|
||||
"""
|
||||
)
|
||||
conn.commit()
|
||||
_initialized = True
|
||||
|
||||
|
||||
def _cache_get(key: str) -> Optional[bool]:
|
||||
# `closing`, because sqlite3's own context manager commits the transaction
|
||||
# and leaves the connection open. Leaked connections are finalized at GC,
|
||||
# which surfaces as an unraisable exception - and pytest.ini turns warnings
|
||||
# into errors, so a leak here fails the suite from an unrelated test.
|
||||
try:
|
||||
with _lock, closing(_connect()) as conn:
|
||||
_ensure_schema(conn)
|
||||
row = conn.execute(
|
||||
"SELECT verdict, created_at FROM openfacts_brand_check WHERE cache_key = ?",
|
||||
(key,),
|
||||
).fetchone()
|
||||
except Exception as e: # noqa: BLE001 - a cache failure is a cache miss
|
||||
logger.debug("image corroboration cache read failed for %s: %s", key, e)
|
||||
return None
|
||||
if not row:
|
||||
return None
|
||||
verdict, created_at = row
|
||||
if (time.time() - created_at) > _CACHE_TTL_SECONDS:
|
||||
return None
|
||||
return bool(verdict)
|
||||
|
||||
|
||||
def _cache_set(key: str, verdict: bool) -> None:
|
||||
try:
|
||||
with _lock, closing(_connect()) as conn:
|
||||
_ensure_schema(conn)
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO openfacts_brand_check (cache_key, verdict, created_at)
|
||||
VALUES (?, ?, ?)
|
||||
ON CONFLICT(cache_key) DO UPDATE SET
|
||||
verdict = excluded.verdict,
|
||||
created_at = excluded.created_at
|
||||
""",
|
||||
(key, 1 if verdict else 0, time.time()),
|
||||
)
|
||||
conn.commit()
|
||||
except Exception as e: # noqa: BLE001 - best effort only
|
||||
logger.debug("image corroboration cache write failed for %s: %s", key, e)
|
||||
|
||||
|
||||
def _api_host_for(url: str) -> Optional[str]:
|
||||
lowered = (url or "").lower()
|
||||
for marker, host in _OFF_API_HOSTS.items():
|
||||
if marker in lowered:
|
||||
return host
|
||||
return None
|
||||
|
||||
|
||||
def openfacts_product_matches_brand(url: str, brand: str, *, timeout: int = 10) -> bool:
|
||||
"""False ONLY when Open*Facts positively says this barcode is another brand.
|
||||
|
||||
Every other outcome - not an Open*Facts URL, no barcode in the path, no
|
||||
brand tokens, a network failure, an unparseable response - returns True.
|
||||
A lookup failure must never reject a good image; that asymmetry is the
|
||||
whole point, and it is the same discipline the realtime checks follow.
|
||||
"""
|
||||
api_host = _api_host_for(url)
|
||||
if not api_host:
|
||||
return True
|
||||
match = _OFF_BARCODE.search(url or "")
|
||||
if not match:
|
||||
return True
|
||||
barcode = match.group(1).replace("/", "")
|
||||
tokens = brand_tokens(brand)
|
||||
if not tokens:
|
||||
return True
|
||||
|
||||
key = f"{api_host}:{barcode}:{(brand or '').lower()}"
|
||||
cached = _cache_get(key)
|
||||
if cached is not None:
|
||||
return cached
|
||||
|
||||
verdict = True
|
||||
try:
|
||||
resp = requests.get(
|
||||
f"https://{api_host}/api/v2/product/{barcode}.json",
|
||||
params={"fields": "brands,product_name"},
|
||||
timeout=timeout,
|
||||
headers={"User-Agent": _BROWSER_UA},
|
||||
)
|
||||
if resp.ok:
|
||||
product = (resp.json() or {}).get("product") or {}
|
||||
haystack = (
|
||||
f"{product.get('brands') or ''} {product.get('product_name') or ''}"
|
||||
).lower()
|
||||
if haystack.strip():
|
||||
verdict = any(t in haystack for t in tokens)
|
||||
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
|
||||
return True
|
||||
|
||||
_cache_set(key, verdict)
|
||||
return verdict
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Which candidate may become the product's primary image
|
||||
# ---------------------------------------------------------------------------
|
||||
@dataclass
|
||||
class PrimaryChoice:
|
||||
"""The outcome of choosing a product's `image_url` from its candidates."""
|
||||
|
||||
primary: Optional[str]
|
||||
ordered: List[str]
|
||||
reason: str
|
||||
|
||||
|
||||
def choose_primary(
|
||||
urls: Sequence[str],
|
||||
product_name: str,
|
||||
brand: str,
|
||||
*,
|
||||
check_openfacts: bool = True,
|
||||
) -> PrimaryChoice:
|
||||
"""Pick the URL that may become `image_url`, and order the rest.
|
||||
|
||||
Three tiers, because the two existing opinions about failing closed are
|
||||
both right and the disagreement is domain-scoped:
|
||||
|
||||
1. The URL names the product. Eligible.
|
||||
2. The URL cannot name anything (a hashed retailer CDN path, an
|
||||
Open*Facts barcode path) - and for Open*Facts, the barcode's own
|
||||
record agrees about the brand. Eligible: this is the case
|
||||
catalog_engine's comment protects, where dropping uncorroborated
|
||||
candidates would leave real products with no image at all.
|
||||
3. The URL could have named the product and did not - a human-authored
|
||||
Commons filename, say. Kept in `image_urls`, never promoted.
|
||||
|
||||
Nothing eligible means `primary is None`. A blank image renders as the
|
||||
brand monogram, which is honest; another company's product is not, and it
|
||||
stays invisible until somebody recognises the photo.
|
||||
"""
|
||||
tier1: List[str] = []
|
||||
tier2: List[str] = []
|
||||
tier3: List[str] = []
|
||||
|
||||
for url in urls:
|
||||
if not url or not str(url).startswith("http"):
|
||||
continue
|
||||
verdict = corroborate(url, product_name, brand)
|
||||
if verdict.corroborated and not verdict.via_brand_only:
|
||||
tier1.append(url)
|
||||
elif verdict.via_brand_only and not looks_like_person_photo(url):
|
||||
# The title has NO distinctive words of its own - "Godrej 50ml",
|
||||
# "Lion Dates 100g", "Amul 1kg". There is no product identity to
|
||||
# match on, so a brand-token match is the best signal that exists
|
||||
# and withholding the image gains nothing: measured over the seed
|
||||
# catalogues, treating these as ineligible accounted for 14 of 45
|
||||
# withheld primaries, every one of them a brand's own product page.
|
||||
#
|
||||
# Person-shaped filenames are the exception, because a brand token
|
||||
# matching a personal name is the exact defect this module exists
|
||||
# for and a title with no distinctive words cannot contradict it.
|
||||
tier2.append(url)
|
||||
elif has_opaque_path(url) and not has_descriptive_filename(url):
|
||||
if check_openfacts and not openfacts_product_matches_brand(url, brand):
|
||||
tier3.append(url)
|
||||
else:
|
||||
tier2.append(url)
|
||||
else:
|
||||
tier3.append(url)
|
||||
|
||||
# Person-shaped filenames sink within their tier. They are never dropped,
|
||||
# so a product whose only images look like this still keeps them in
|
||||
# `image_urls` for a human to review.
|
||||
tier3.sort(key=looks_like_person_photo)
|
||||
|
||||
ordered = tier1 + tier2 + tier3
|
||||
if tier1:
|
||||
return PrimaryChoice(tier1[0], ordered, "url names the product")
|
||||
if tier2:
|
||||
return PrimaryChoice(tier2[0], ordered, "opaque path on a known image host")
|
||||
return PrimaryChoice(
|
||||
None,
|
||||
ordered,
|
||||
"no candidate names this product; primary withheld rather than guessed",
|
||||
)
|
||||
236
app/services/product_grounding.py
Normal file
236
app/services/product_grounding.py
Normal file
@@ -0,0 +1,236 @@
|
||||
"""
|
||||
Does an external source corroborate that this product exists?
|
||||
|
||||
WHY THIS FILE EXISTS
|
||||
--------------------
|
||||
The catalogue contained "Anil Wheat Vermicelli 12g". Nobody sells a 12 g
|
||||
vermicelli pack. It was generated by a 1.5B local model, given a category, a
|
||||
price band and an internal SKU, and stored - because every check in the system
|
||||
asks whether a row is WELL FORMED, and a well-formed fiction passes all of
|
||||
them. `product_validator`'s own docstring says so: it was built to catch
|
||||
malformed rows, not false ones.
|
||||
|
||||
Meanwhile the real answer was already on disk. `off_bulk.fetch_brand_corpus`
|
||||
caches a brand's entire Open Food Facts catalogue, and
|
||||
`data/cache/off_brand_corpus/anil.json` holds Anil's eight real products -
|
||||
Roasted Short Vermicelli at 180 g and 450 g, SAMBA RAVVA at 500 g. Nothing
|
||||
consulted it at generation time.
|
||||
|
||||
WHAT THIS MODULE DOES NOT DO: SUBSTITUTE A SIZE
|
||||
-----------------------------------------------
|
||||
The obvious-looking fix - "look up the real quantity and swap it in" - is
|
||||
wrong here, and measurably so. Matching "Anil Wheat Vermicelli" against the
|
||||
corpus scores 0.460 against "Roasted Short Vermicelli", far below any usable
|
||||
floor. That is the corpus saying THIS PRODUCT DOES NOT EXIST, not saying it is
|
||||
450 g. Swapping in a corpus quantity would replace one fabrication with a
|
||||
better-dressed one: a product that still does not exist, now wearing a size
|
||||
that does.
|
||||
|
||||
So a miss reports `not_found` and the caller records that the row is
|
||||
ungrounded. Removing the row is the validation gate's job, not this module's.
|
||||
|
||||
WHY NOT `find_product_quantity_openfacts`
|
||||
-----------------------------------------
|
||||
`image_search.find_product_quantity_openfacts` looks like the right oracle and
|
||||
is not. It queries `/cgi/search.pl` LIVE, once per product, looping three hosts
|
||||
at a 12-second timeout - up to 36 s per product against an endpoint Open Food
|
||||
Facts rate-limits to 10 requests/minute. Two hundred products is twenty-plus
|
||||
minutes and near-certain throttling, and a throttled miss is indistinguishable
|
||||
from a real one, which is exactly the input a gate must never be fed. Measured
|
||||
2026-09-10: that endpoint returned 503 on OFF and 500 on Open Beauty Facts.
|
||||
|
||||
`fetch_brand_corpus` costs one to five requests per BRAND, is disk-cached,
|
||||
retry-wrapped and paced, and is already populated for 50+ brands.
|
||||
|
||||
THREE STATES, NOT TWO
|
||||
---------------------
|
||||
`grounded` / `not_found` / `unknown`. A corpus that could not be fetched must
|
||||
report `unknown` and never `not_found`, or a network failure would silently
|
||||
demote a whole brand's real products. Every lookup path in this codebase that
|
||||
feeds a gate follows the same discipline.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Dict, List, Optional, Sequence
|
||||
|
||||
from app.services.enrichment.barcode.sources import off_bulk
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# How similar a corpus name must be to count as the same product.
|
||||
#
|
||||
# NOT the 0.78 used for barcode attachment. Measured against the real Anil
|
||||
# corpus on 2026-09-10, best match per target:
|
||||
#
|
||||
# "wheat vermicelli" vs "Rice Vermicelli" -> 0.610 (DIFFERENT product)
|
||||
# "samba rava" vs "SAMBA RAVVA" -> 0.681 (SAME product)
|
||||
#
|
||||
# A 0.78 floor rejects a genuine product over a one-letter spelling difference
|
||||
# in Open Food Facts' own record, so it cannot be used here. But note how
|
||||
# little room that leaves: 0.071 between a true match and a false one, and the
|
||||
# floor below sits about 0.01 above the false one. THAT MARGIN IS TOO THIN TO
|
||||
# REST A DECISION ON, which is why `_material_conflict` exists - the real
|
||||
# discriminator between those two names is that wheat is not rice, not that
|
||||
# 0.610 is not 0.681. The floor is the coarse filter; the material check is
|
||||
# what actually separates the reported defect from the real product beside it.
|
||||
#
|
||||
# It is also deliberately a soft signal in both directions: nothing here
|
||||
# deletes a row, it only decides whether the row can be called corroborated,
|
||||
# so the cost of being wrong is a review flag and not data loss.
|
||||
GROUNDING_SIMILARITY_FLOOR = 0.62
|
||||
|
||||
# Base materials that make two otherwise similarly-named products different
|
||||
# products. "Wheat Vermicelli" and "Rice Vermicelli" share a head noun and
|
||||
# score 0.610 against each other; no string-similarity threshold separates
|
||||
# them reliably, because the thing that differs is one word carrying all the
|
||||
# meaning.
|
||||
#
|
||||
# The rule is deliberately narrow: a conflict is declared ONLY when BOTH names
|
||||
# name a material and the sets are disjoint. "Wheat Vermicelli" against a bare
|
||||
# "Vermicelli" is not a conflict - that is plausibly the same product line
|
||||
# described at two levels of detail, and treating it as a conflict would lose
|
||||
# real corroboration.
|
||||
_MATERIALS = frozenset({
|
||||
"wheat", "rice", "ragi", "maida", "corn", "millet", "bajra", "jowar",
|
||||
"atta", "besan", "soya", "oat", "oats", "barley", "quinoa", "almond",
|
||||
"cashew", "coconut", "groundnut", "peanut", "sesame", "mustard",
|
||||
"sunflower", "olive", "ghee", "butter",
|
||||
})
|
||||
|
||||
GROUNDED = "grounded"
|
||||
NOT_FOUND = "not_found"
|
||||
UNKNOWN = "unknown"
|
||||
|
||||
# Per-process memo of brand -> corpus. fetch_brand_corpus is itself disk-cached,
|
||||
# so this only avoids re-reading and re-parsing the same JSON once per product
|
||||
# in a 200-row run.
|
||||
_corpus_memo: Dict[str, Optional[List[Dict[str, Any]]]] = {}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Grounding:
|
||||
"""What an external source says about one product."""
|
||||
|
||||
status: str
|
||||
matched_name: Optional[str] = None
|
||||
quantity: Optional[str] = None
|
||||
similarity: float = 0.0
|
||||
source: Optional[str] = None
|
||||
|
||||
@property
|
||||
def is_grounded(self) -> bool:
|
||||
return self.status == GROUNDED
|
||||
|
||||
def note(self) -> str:
|
||||
if self.status == GROUNDED:
|
||||
return (
|
||||
f"corroborated by {self.source} as {self.matched_name!r} "
|
||||
f"(similarity {self.similarity:.2f})"
|
||||
)
|
||||
if self.status == NOT_FOUND:
|
||||
return f"no {self.source or 'Open Food Facts'} product matches this name"
|
||||
return "could not check whether this product exists"
|
||||
|
||||
|
||||
def _material_conflict(candidate: str, target: str) -> bool:
|
||||
"""True when two names name DIFFERENT base materials.
|
||||
|
||||
See `_MATERIALS`. Only fires when both sides name one, so a more specific
|
||||
name still corroborates a less specific one.
|
||||
"""
|
||||
left = {w for w in candidate.lower().split() if w in _MATERIALS}
|
||||
right = {w for w in target.lower().split() if w in _MATERIALS}
|
||||
return bool(left and right and not (left & right))
|
||||
|
||||
|
||||
def _corpus_for(brand: str) -> Optional[List[Dict[str, Any]]]:
|
||||
"""The brand's cached Open*Facts corpus, or None when it is unavailable.
|
||||
|
||||
None means "we do not know", and is what makes the difference between
|
||||
`not_found` and `unknown` downstream. An EMPTY list is a real answer - the
|
||||
brand was looked up and has nothing on file.
|
||||
"""
|
||||
key = (brand or "").strip().lower()
|
||||
if not key:
|
||||
return None
|
||||
if key in _corpus_memo:
|
||||
return _corpus_memo[key]
|
||||
try:
|
||||
corpus = off_bulk.fetch_brand_corpus(brand)
|
||||
except Exception as e: # noqa: BLE001 - an unreachable corpus is not a verdict
|
||||
logger.debug("Corpus fetch failed for %r: %s", brand, e)
|
||||
corpus = None
|
||||
_corpus_memo[key] = corpus
|
||||
return corpus
|
||||
|
||||
|
||||
def reset_cache() -> None:
|
||||
"""Drop the per-process corpus memo (tests, and long-lived workers)."""
|
||||
_corpus_memo.clear()
|
||||
|
||||
|
||||
def ground_product(brand: str, product_title: str,
|
||||
*, corpus: Optional[Sequence[Dict[str, Any]]] = None) -> Grounding:
|
||||
"""Is `product_title` a real product of `brand`, per Open*Facts?
|
||||
|
||||
`corpus` is injectable so a caller processing a whole brand fetches once
|
||||
and so tests need no network.
|
||||
"""
|
||||
title = (product_title or "").strip()
|
||||
if not title:
|
||||
return Grounding(UNKNOWN)
|
||||
|
||||
hits = corpus if corpus is not None else _corpus_for(brand)
|
||||
if hits is None:
|
||||
return Grounding(UNKNOWN)
|
||||
if not hits:
|
||||
# A real answer: this brand has nothing on file at all. That is not
|
||||
# evidence against any single product, so it is still `unknown` - a
|
||||
# brand missing from Open Food Facts is a gap in Open Food Facts.
|
||||
return Grounding(UNKNOWN, source="openfacts")
|
||||
|
||||
drop = off_bulk.brand_tokens(brand)
|
||||
target = off_bulk.normalize_for_match(title, drop)
|
||||
if not target:
|
||||
return Grounding(UNKNOWN, source="openfacts")
|
||||
|
||||
best_score = 0.0
|
||||
best_hit: Optional[Dict[str, Any]] = None
|
||||
for hit in hits:
|
||||
name = hit.get("product_name") or hit.get("product_name_en") or ""
|
||||
if not name:
|
||||
continue
|
||||
candidate = off_bulk.normalize_for_match(name, drop)
|
||||
if not candidate:
|
||||
continue
|
||||
# Skipped before scoring, not after: a wrong-material name can be the
|
||||
# single best-scoring row in the corpus ("Rice Vermicelli" is the top
|
||||
# match for "Wheat Vermicelli" at 0.610), and letting it win would mask
|
||||
# a genuinely better match further down the list.
|
||||
if _material_conflict(candidate, target):
|
||||
continue
|
||||
score = off_bulk.symmetric_similarity(candidate, target)
|
||||
if score > best_score:
|
||||
best_score, best_hit = score, hit
|
||||
|
||||
if best_hit is None or best_score < GROUNDING_SIMILARITY_FLOOR:
|
||||
return Grounding(NOT_FOUND, similarity=round(best_score, 3), source="openfacts")
|
||||
|
||||
# A variant conflict means the corpus row is a DIFFERENT pack of a
|
||||
# similarly-named product (sugar-free vs regular, jar vs pouch). Close
|
||||
# enough to score well, not close enough to corroborate.
|
||||
name = best_hit.get("product_name") or best_hit.get("product_name_en") or ""
|
||||
if off_bulk.has_extra_variant_conflict(name, title):
|
||||
return Grounding(NOT_FOUND, matched_name=name,
|
||||
similarity=round(best_score, 3), source="openfacts")
|
||||
|
||||
quantity = str(best_hit.get("quantity") or "").strip() or None
|
||||
return Grounding(
|
||||
GROUNDED,
|
||||
matched_name=name,
|
||||
quantity=quantity,
|
||||
similarity=round(best_score, 3),
|
||||
source="openfacts",
|
||||
)
|
||||
@@ -24,9 +24,15 @@ This module is that final gate. It is intentionally:
|
||||
category-anchored price bands, and sku_service's own SKU format) rather
|
||||
than guessing. False rejections are treated as seriously as false
|
||||
acceptances - see docs/VALIDATION_PIPELINE.md for the tuning rationale.
|
||||
- ADDITIVE: this module has zero dependents before this change and does not
|
||||
modify any existing function's behaviour; it is only ever imported and
|
||||
called from new code (see app/core/catalog_engine.py's call site).
|
||||
- ADDITIVE: it does not modify any existing function's behaviour; callers opt
|
||||
in. The call sites are store_catalog_pipeline.stage_10_validate (the
|
||||
spreadsheet path), orchestration/assets/catalog.py (the Dagster mirror) and
|
||||
ProductCatalogEngine.generate_catalog's "Step 2.6" (the brand-name path).
|
||||
|
||||
That last one was absent for a long time while this docstring claimed it
|
||||
existed, so POST /api/catalog/generate wrote LLM output straight to the
|
||||
brand tables with no gate at all. If you are adding another generation path,
|
||||
it needs its own call to validate_catalog - nothing here is automatic.
|
||||
|
||||
USAGE
|
||||
-----
|
||||
@@ -244,6 +250,7 @@ def validate_product(
|
||||
category_resolved_deterministically: bool = False,
|
||||
grounded: bool = False,
|
||||
images_checked: bool = True,
|
||||
require_grounding: bool = False,
|
||||
config: Optional[ValidationConfig] = None,
|
||||
) -> ValidationReport:
|
||||
"""Validate one fully-assembled catalog row and return a confidence-scored report.
|
||||
@@ -350,12 +357,14 @@ def validate_product(
|
||||
score -= 0.25
|
||||
|
||||
# 4b. Size unit <-> category compatibility (see app/services/
|
||||
# category_units.py). This is a last-resort backstop: by the time a
|
||||
# row reaches this final validator, its size should already have
|
||||
# been through the generation-time gate (app/services/
|
||||
# generation_verifier.py, called from ollama_service.py) AND
|
||||
# catalog_engine.py's own dimension-unit defence-in-depth pass, both
|
||||
# of which reject/repair a bad unit long before enrichment. This
|
||||
# category_units.py). This used to describe itself as a backstop
|
||||
# behind a generation-time gate in app/services/generation_verifier.py
|
||||
# "called from ollama_service.py". THAT FILE DOES NOT EXIST IN THIS
|
||||
# TREE - it belongs to a sibling project - so the layer this comment
|
||||
# promised was never running here and this check is not a backstop but
|
||||
# a front line. The real upstream defence is
|
||||
# category_units.fix_or_reject_size, called from stage 4, plus
|
||||
# catalog_engine.py's dimension-unit pass. This
|
||||
# check exists purely to catch anything that reached validate_product
|
||||
# some other way (e.g. a pre-existing DB row from before this fix, or
|
||||
# a future caller that doesn't route through catalog_engine).
|
||||
@@ -413,6 +422,33 @@ def validate_product(
|
||||
else:
|
||||
report.status = "verified"
|
||||
|
||||
# GROUNDING AS A CAP, NOT A BONUS.
|
||||
#
|
||||
# The +0.15 above is why this module could never do the job its docstring
|
||||
# claims. A fabricated product scores 0.55 baseline + 0.15 resolved
|
||||
# category + 0.05 has-images = 0.75, clearing the 0.70 threshold and
|
||||
# landing as "verified" - so "Anil Wheat Vermicelli 12g", a pack size no
|
||||
# shop has ever sold, was stored as a checked fact. Every check here is
|
||||
# about FORM, and a well-formed fiction satisfies all of them.
|
||||
#
|
||||
# A caller that knows its rows came from a generator rather than from a
|
||||
# source sets require_grounding. Then a row nothing external corroborates
|
||||
# can still be kept and still scores normally, but it cannot be called
|
||||
# verified - which is the honest answer, because nothing verified it.
|
||||
#
|
||||
# DELIBERATELY NOT APPLIED BY DEFAULT. The spreadsheet path passes no
|
||||
# grounded flags at all, and a shop's own upload IS its evidence; capping
|
||||
# there would relabel every uploaded row as unchecked and say nothing
|
||||
# true. Only generation paths opt in.
|
||||
if require_grounding and not grounded and report.status == "verified":
|
||||
report.status = "needs_review"
|
||||
report.issues.append(ValidationIssue(
|
||||
field="grounding", severity="warning",
|
||||
message="no external source corroborates that this product exists; "
|
||||
"well-formed but unverified",
|
||||
penalty=0.0,
|
||||
))
|
||||
|
||||
return report
|
||||
|
||||
|
||||
@@ -423,6 +459,7 @@ def validate_catalog(
|
||||
known_category_flags: Optional[list[bool]] = None,
|
||||
grounded_flags: Optional[list[bool]] = None,
|
||||
images_checked: bool = True,
|
||||
require_grounding: bool = False,
|
||||
config: Optional[ValidationConfig] = None,
|
||||
) -> tuple[list[dict[str, Any]], list[dict[str, Any]], dict[str, Any]]:
|
||||
"""Validate an entire list of assembled catalog rows in one pass.
|
||||
@@ -450,6 +487,7 @@ def validate_catalog(
|
||||
category_resolved_deterministically=det,
|
||||
grounded=grd,
|
||||
images_checked=images_checked,
|
||||
require_grounding=require_grounding,
|
||||
config=cfg,
|
||||
)
|
||||
status_counts[report.status] = status_counts.get(report.status, 0) + 1
|
||||
|
||||
883
app/services/retail_presence.py
Normal file
883
app/services/retail_presence.py
Normal file
@@ -0,0 +1,883 @@
|
||||
"""
|
||||
Is this product on sale, right now, somewhere real?
|
||||
|
||||
WHY THIS FILE EXISTS
|
||||
--------------------
|
||||
Open Food Facts answers "does this product exist" for food, and answers it
|
||||
well. It does not answer it for anything else: `off_bulk` queries only
|
||||
`search.openfoodfacts.org`, so a toothpaste or a detergent has no
|
||||
product-discovery source at all and its entire catalogue is language-model
|
||||
output. Measured 2026-09-10, India-tagged coverage on the sibling databases is
|
||||
2-13 products per non-food brand against Britannia's 218 on OFF - real, but
|
||||
nowhere near enough to build a catalogue from.
|
||||
|
||||
A live retail lookup answers it for everything, food and non-food alike, and
|
||||
it is the only signal in this codebase that means "available in real time"
|
||||
rather than "was in a database dump".
|
||||
|
||||
THE ONE THING THAT MAKES THIS WORK: MATCH THE PACK SIZE
|
||||
--------------------------------------------------------
|
||||
Measured against the live provider on 2026-09-10:
|
||||
|
||||
"Anil Roasted Short Vermicelli 450g"
|
||||
-> amazon.in "Anil Vermicelli - Roasted, 450g Pouch" CONFIRMED
|
||||
|
||||
"Anil Wheat Vermicelli 12g"
|
||||
-> results come back, but every one is the generic product
|
||||
page. NOTHING confirms a 12 g pack. NOT CONFIRMED
|
||||
|
||||
A check that asks "did the search return anything?" would have confirmed the
|
||||
12 g pack, which is the exact fabrication this module exists to catch. So a
|
||||
hit counts only when the listing's TITLE and its PACK SIZE both match.
|
||||
|
||||
That is also why `sku_service.find_website_product_id` cannot be reused as
|
||||
evidence: it regexes a product ID out of the result URL and never reads the
|
||||
listing title or its size at all. It answers "is there an Amazon page in these
|
||||
results", which for any real brand is always yes.
|
||||
|
||||
THREE STATES, NEVER TWO
|
||||
-----------------------
|
||||
`found` / `not_found` / `unknown`. DuckDuckGo throttles and 403s - the repair
|
||||
script already records that it "already 403s and the pipeline falls through to
|
||||
Bing". A throttled lookup MUST report `unknown`, because a gate fed
|
||||
`not_found` on a rate-limit would quietly demote a brand's entire real
|
||||
catalogue on a bad afternoon.
|
||||
|
||||
COST, AND WHY THE STAGE READS ONLY THE CACHE
|
||||
---------------------------------------------
|
||||
There is no brand-scoped retail endpoint the way there is for Open Food Facts,
|
||||
so this is irreducibly one network call per product against a provider that
|
||||
blocks. Measured 2.6-5.7 s per query.
|
||||
|
||||
So the work is split: `backfill_retail_presence.py` does the querying offline,
|
||||
paced and cached, and the ingestion-time stage reads the cache ONLY
|
||||
(`live=False`). Ingestion never blocks on a live query and never trips a rate
|
||||
limit mid-batch. `search_key` collapses pack sizes onto one lookup per
|
||||
PRODUCT rather than per row, which is a 3-5x cut for free.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
import sqlite3
|
||||
import threading
|
||||
import time
|
||||
from contextlib import closing
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Sequence
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from app.services import quantity_utils
|
||||
from app.services.brand_registry import BRAND_ALIASES
|
||||
from app.services.enrichment.barcode.matching import brand_matches
|
||||
from app.services.enrichment.barcode.sources import off_bulk
|
||||
from app.services.image_corroboration import search_key
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
FOUND = "found"
|
||||
NOT_FOUND = "not_found"
|
||||
UNKNOWN = "unknown"
|
||||
|
||||
# Indian grocery/marketplace domains a real FMCG product is listed on.
|
||||
# `sku_service._MARKETPLACE_PATTERNS` covers the same ground for ID extraction;
|
||||
# this list is about presence, so it needs no per-site URL grammar.
|
||||
RETAILER_DOMAINS = (
|
||||
"amazon.in", "flipkart.com", "bigbasket.com", "jiomart.com",
|
||||
"blinkit.com", "zeptonow.com", "dmart.in", "swiggy.com",
|
||||
"netmeds.com", "pharmeasy.in", "nykaa.com", "1mg.com",
|
||||
"licious.in", "starquik.com", "spencers.in", "moreretail.in",
|
||||
)
|
||||
|
||||
# What fraction of the PRODUCT's distinctive words must appear in the listing
|
||||
# title. See `listing_matches` for why this is a containment ratio and not the
|
||||
# 0.85 symmetric similarity the barcode matcher uses.
|
||||
#
|
||||
# 0.6 is deliberately loose. Retailers drop words freely - Amazon lists "Anil
|
||||
# Roasted Short Vermicelli" as "Anil Vermicelli - Roasted", covering 2 of 3
|
||||
# tokens (0.67) - so a tight floor here rejects real listings while adding
|
||||
# nothing, because it is the pack-size check below that separates a real pack
|
||||
# from an invented one. This gate only has to establish that the listing is
|
||||
# about the right product family.
|
||||
TITLE_COVERAGE_FLOOR = 0.6
|
||||
|
||||
# Marketing boilerplate that surrounds a product name in a retail listing
|
||||
# title. Removed before matching so "Buy Anil Roasted Vermicelli 450g Online
|
||||
# at Best Price" reduces to the product.
|
||||
_LISTING_NOISE = re.compile(
|
||||
r"\b(buy|online|at|best|price|lowest|offers?|deals?|free|delivery|shop|"
|
||||
r"order|now|india|in|from|upto|off|save|get|com|grocery|gourmet|foods?)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
# Words that name the SAME thing in a retail listing and in our catalogue.
|
||||
#
|
||||
# Indian FMCG listings mix English and transliterated Tamil/Hindi freely, and
|
||||
# our own product names do too. Measured misses: our "Anil Puttu Mix" is on
|
||||
# amazon.in as "Anil Puttu MAAVU" (maavu = flour/mix) and was rejected at 0.5
|
||||
# coverage; this catalogue also uses "Semiya" and "Vermicelli" interchangeably
|
||||
# in its own product names, so the two spellings fail to corroborate each other.
|
||||
#
|
||||
# STRICTLY OBSERVED, NEVER INVENTED - the same rule product_grounding._MATERIALS
|
||||
# follows. This must not grow into a general thesaurus: every pair added makes
|
||||
# the matcher more willing to confirm a product, and confirming products that do
|
||||
# not exist is the failure this whole module exists to prevent. Add a pair only
|
||||
# after seeing it reject a real listing.
|
||||
#
|
||||
# THE PACK SIZE IS NOT AFFECTED. Synonyms loosen only which words count as the
|
||||
# same word; `quantities_match` still has to agree, so no synonym can rescue a
|
||||
# size that nobody sells.
|
||||
# Each pair below cites the listing that forced it. Anything without a citation
|
||||
# does not belong here.
|
||||
_SYNONYMS: Dict[str, str] = {
|
||||
# amazon.in "Anil Puttu Maavu" vs our "Anil Puttu Mix", and
|
||||
# theanilgroup.com "Arisi Puttu Maavu" for the same product.
|
||||
"maavu": "flour",
|
||||
"mix": "flour",
|
||||
# This catalogue's own names: "Anil Rava Semiya", "Anil Ragi Semiya" and
|
||||
# "Anil Wheat Vermicelli" are the same product family spelled two ways.
|
||||
"semiya": "vermicelli",
|
||||
# Open Food Facts holds our "Anil Samba Rava" as "SAMBA RAVVA".
|
||||
"ravva": "rava",
|
||||
# theanilgroup.com sells our "Wheat Vermicelli" as "Atta Semiya".
|
||||
"atta": "wheat",
|
||||
}
|
||||
|
||||
|
||||
def _aliases_for(brand: str) -> List[str]:
|
||||
"""Other names this brand trades under, from the curated registry.
|
||||
|
||||
`BRAND_ALIASES` maps alias -> parent, so the aliases OF a brand are the
|
||||
keys pointing at it, plus the parent name itself.
|
||||
"""
|
||||
key = (brand or "").strip().lower()
|
||||
if not key:
|
||||
return []
|
||||
out = {key}
|
||||
parent = BRAND_ALIASES.get(key)
|
||||
if parent:
|
||||
out.add(parent.lower())
|
||||
for alias, target in BRAND_ALIASES.items():
|
||||
if target.lower() == key or (parent and target.lower() == parent.lower()):
|
||||
out.add(alias.lower())
|
||||
return sorted(out)
|
||||
|
||||
|
||||
def _canonical_words(text: str) -> set:
|
||||
"""Token set with observed synonyms folded onto one spelling."""
|
||||
return {_SYNONYMS.get(w, w) for w in (text or "").split()}
|
||||
|
||||
|
||||
# A real product listing title is short. Anything much longer is the provider
|
||||
# running several results together into one `title` field, e.g.
|
||||
#
|
||||
# "Ginger Garlic Paste (Pack of 2) | No Peeling, No ChoppingAachi Ginger
|
||||
# Garlic Paste 20g - martizo.comAachi Ginger Garlic Paste - Buy at Rs43..."
|
||||
#
|
||||
# 372 characters, at least four different listings, from at least two shops.
|
||||
# Matching against that blob is meaningless in both halves: the quantity comes
|
||||
# from whichever listing happened to state one, and the word coverage is
|
||||
# satisfied by words scattered across products we never asked about. It
|
||||
# produced a FOUND for "Aachi Ginger Paste 20g" whose 20g came from a different
|
||||
# retailer's listing of a different product, attributed to aachifoods.com.
|
||||
#
|
||||
# A false positive is the worse failure here - it hands 0.85 corroboration to a
|
||||
# product that may not exist, which is what this module was built to prevent -
|
||||
# so a blob is discarded rather than parsed.
|
||||
MAX_LISTING_TITLE_CHARS = 140
|
||||
|
||||
# Ingredient words that make two similarly-named products DIFFERENT products.
|
||||
#
|
||||
# Coverage is deliberately containment - the target's words must appear in the
|
||||
# listing - so a listing is free to carry extra words, which is right for shop
|
||||
# furniture ("Pouch", "Best Price", "Amazon.in"). It is wrong for an ingredient:
|
||||
# "Aachi Ginger Paste" is entirely contained in "Aachi Ginger GARLIC Paste" and
|
||||
# scored 1.0 coverage against it, and ginger paste is not ginger-garlic paste.
|
||||
#
|
||||
# Same discipline as product_grounding._MATERIALS and _SYNONYMS: these are
|
||||
# words observed causing a wrong match, not a general ingredient list.
|
||||
_DISTINGUISHING_INGREDIENTS = frozenset({
|
||||
"garlic", "ginger", "chilli", "chili", "pepper", "tamarind", "coriander",
|
||||
"cumin", "turmeric", "mint", "lemon", "tomato", "onion", "mango",
|
||||
"coconut", "mustard", "fenugreek", "clove", "cardamom", "cinnamon",
|
||||
})
|
||||
|
||||
|
||||
_DB_PATH = Path("data") / "cache" / "retail_presence.db"
|
||||
_lock = threading.Lock()
|
||||
_initialized = False
|
||||
DEFAULT_TTL_SECONDS = 14 * 24 * 3600
|
||||
# A brand's own domain changes far more rarely than its stock does.
|
||||
BRAND_DOMAIN_TTL_SECONDS = 180 * 24 * 3600
|
||||
# A miss is re-asked the next day - see get_cached_brand_domain for why.
|
||||
BRAND_DOMAIN_MISS_TTL_SECONDS = 24 * 3600
|
||||
|
||||
# Search endpoints are the strictly-limited ones. Same figure off_bulk uses.
|
||||
PAUSE_SECONDS = 2.0
|
||||
|
||||
|
||||
@dataclass
|
||||
class RetailEvidence:
|
||||
"""What a live retail search says about one (brand, product, size)."""
|
||||
|
||||
status: str
|
||||
retailer: Optional[str] = None
|
||||
url: Optional[str] = None
|
||||
matched_title: Optional[str] = None
|
||||
matched_size: Optional[str] = None
|
||||
similarity: float = 0.0
|
||||
checked_at: Optional[float] = None
|
||||
# How many results were actually on a shop. This is what makes a negative
|
||||
# auditable: without it, the only way to ask "did we even reach a retailer?"
|
||||
# is to run the search again, and the provider's reach varies run to run
|
||||
# (the same query returned 7 shop results and then 0 minutes later).
|
||||
shop_results_seen: int = 0
|
||||
# The shop titles that came back and did NOT match, kept so a human can
|
||||
# judge whether the rejection was right. "Aachi Biryani Masala" against
|
||||
# "Aachi Biryani Mix" is a correct rejection; "Anil Puttu Maavu" against
|
||||
# "Anil Puttu Mix" is a matcher failure, and nothing but the title tells
|
||||
# them apart.
|
||||
seen_titles: List[str] = field(default_factory=list)
|
||||
query: Optional[str] = None
|
||||
|
||||
@property
|
||||
def is_found(self) -> bool:
|
||||
return self.status == FOUND
|
||||
|
||||
def note(self) -> str:
|
||||
if self.status == FOUND:
|
||||
return f"listed by {self.retailer} as {self.matched_title!r}"
|
||||
if self.status == NOT_FOUND:
|
||||
return (f"{self.shop_results_seen} shop listing(s) seen, none for "
|
||||
f"this product at this pack size")
|
||||
if self.shop_results_seen == 0 and self.checked_at:
|
||||
return "searched, but no shop listing was reached at all"
|
||||
return "could not check retail availability"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Cache
|
||||
# ---------------------------------------------------------------------------
|
||||
def _connect() -> sqlite3.Connection:
|
||||
_DB_PATH.parent.mkdir(parents=True, exist_ok=True)
|
||||
conn = sqlite3.connect(str(_DB_PATH), timeout=10)
|
||||
conn.execute("PRAGMA journal_mode=WAL")
|
||||
return conn
|
||||
|
||||
|
||||
def _ensure_schema(conn: sqlite3.Connection) -> None:
|
||||
global _initialized
|
||||
if _initialized:
|
||||
return
|
||||
conn.execute(
|
||||
"""
|
||||
CREATE TABLE IF NOT EXISTS retail_presence_cache (
|
||||
cache_key TEXT PRIMARY KEY,
|
||||
brand TEXT,
|
||||
product TEXT,
|
||||
size TEXT,
|
||||
result_json TEXT NOT NULL,
|
||||
created_at REAL NOT NULL
|
||||
)
|
||||
"""
|
||||
)
|
||||
# One row per brand, because a brand's own domain is a fact about the
|
||||
# brand and costs a search to learn. Resolved once and reused across every
|
||||
# product, so it adds one query per brand rather than one per product.
|
||||
conn.execute(
|
||||
"""
|
||||
CREATE TABLE IF NOT EXISTS brand_domain_cache (
|
||||
brand TEXT PRIMARY KEY,
|
||||
domain TEXT,
|
||||
created_at REAL NOT NULL
|
||||
)
|
||||
"""
|
||||
)
|
||||
conn.commit()
|
||||
_initialized = True
|
||||
|
||||
|
||||
def cache_key(brand: str, product_title: str, size: str) -> str:
|
||||
"""One key per PRODUCT+SIZE, with the product name size-collapsed first.
|
||||
|
||||
`search_key` strips a trailing pack size from the name so
|
||||
"Anil Vermicelli 180g" and "Anil Vermicelli 450g" share a product identity;
|
||||
the size is then a separate component, because the size is precisely what
|
||||
is being verified.
|
||||
"""
|
||||
brand_key, product_key = search_key(product_title, brand)
|
||||
size_key = re.sub(r"\s+", "", (size or "").strip().lower())
|
||||
return f"{brand_key}|{product_key}|{size_key}"
|
||||
|
||||
|
||||
def get_cached(brand: str, product_title: str, size: str,
|
||||
ttl_seconds: float = DEFAULT_TTL_SECONDS) -> Optional[RetailEvidence]:
|
||||
"""A cached verdict, or None for a miss. Never raises."""
|
||||
key = cache_key(brand, product_title, size)
|
||||
try:
|
||||
with _lock, closing(_connect()) as conn:
|
||||
_ensure_schema(conn)
|
||||
row = conn.execute(
|
||||
"SELECT result_json, created_at FROM retail_presence_cache WHERE cache_key = ?",
|
||||
(key,),
|
||||
).fetchone()
|
||||
except Exception as e: # noqa: BLE001 - a cache failure is a cache miss
|
||||
logger.debug("Retail cache read failed for %s: %s", key, e)
|
||||
return None
|
||||
if not row:
|
||||
return None
|
||||
result_json, created_at = row
|
||||
if ttl_seconds > 0 and (time.time() - created_at) > ttl_seconds:
|
||||
return None
|
||||
try:
|
||||
return RetailEvidence(**json.loads(result_json))
|
||||
except Exception: # noqa: BLE001 - an unreadable entry is a miss
|
||||
return None
|
||||
|
||||
|
||||
def set_cached(brand: str, product_title: str, size: str,
|
||||
evidence: RetailEvidence) -> None:
|
||||
"""Best-effort write. A failure just means no caching next time.
|
||||
|
||||
An `unknown` verdict is NOT cached: it records that we failed to ask, not
|
||||
an answer, and caching it would turn one rate-limited afternoon into two
|
||||
weeks of pretending we had checked.
|
||||
"""
|
||||
if evidence.status == UNKNOWN:
|
||||
return
|
||||
key = cache_key(brand, product_title, size)
|
||||
try:
|
||||
with _lock, closing(_connect()) as conn:
|
||||
_ensure_schema(conn)
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO retail_presence_cache
|
||||
(cache_key, brand, product, size, result_json, created_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT(cache_key) DO UPDATE SET
|
||||
result_json = excluded.result_json,
|
||||
created_at = excluded.created_at
|
||||
""",
|
||||
(key, brand, product_title, size,
|
||||
json.dumps(asdict(evidence)), time.time()),
|
||||
)
|
||||
conn.commit()
|
||||
except Exception as e: # noqa: BLE001 - best effort only
|
||||
logger.debug("Retail cache write failed for %s: %s", key, e)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Matching
|
||||
# ---------------------------------------------------------------------------
|
||||
# Domains that are never a brand's own site, however often they rank for its
|
||||
# name. Retailers are excluded separately via RETAILER_DOMAINS.
|
||||
_NOT_A_BRAND_SITE = (
|
||||
"wikipedia.org", "wikimedia.org", "facebook.com", "instagram.com",
|
||||
"linkedin.com", "youtube.com", "twitter.com", "x.com", "pinterest.com",
|
||||
"indiamart.com", "justdial.com", "tradeindia.com", "exportersindia.com",
|
||||
"zaubacorp.com", "tofler.in", "crunchbase.com", "glassdoor.com",
|
||||
"google.com", "blogspot.com", "wordpress.com", "medium.com",
|
||||
"quora.com", "reddit.com", "yelp.com", "tripadvisor.com",
|
||||
)
|
||||
|
||||
|
||||
def _registrable(url: str) -> str:
|
||||
"""The registrable-ish domain of a URL ("shop.theanilgroup.com" ->
|
||||
"theanilgroup.com"). Good enough for matching, not a public-suffix parser."""
|
||||
host = urlparse(url if "//" in url else f"//{url}").netloc.lower()
|
||||
host = host.split(":")[0]
|
||||
if host.startswith("www."):
|
||||
host = host[4:]
|
||||
parts = host.split(".")
|
||||
if len(parts) > 2 and parts[-2] in ("co", "com", "net", "org", "gov", "ac"):
|
||||
return ".".join(parts[-3:]) # theanilgroup.co.in
|
||||
if len(parts) > 2:
|
||||
return ".".join(parts[-2:]) # shop.theanilgroup.com -> theanilgroup.com
|
||||
return host
|
||||
|
||||
|
||||
def get_cached_brand_domain(brand: str) -> Optional[str]:
|
||||
"""Cached brand domain, or None for a miss. Never raises.
|
||||
|
||||
An empty string is a real cached answer meaning "looked and found none";
|
||||
it comes back as None to callers but stops the lookup being repeated.
|
||||
"""
|
||||
key = _normalize_brand(brand)
|
||||
try:
|
||||
with _lock, closing(_connect()) as conn:
|
||||
_ensure_schema(conn)
|
||||
row = conn.execute(
|
||||
"SELECT domain, created_at FROM brand_domain_cache WHERE brand = ?",
|
||||
(key,),
|
||||
).fetchone()
|
||||
except Exception: # noqa: BLE001 - a cache failure is a cache miss
|
||||
return None
|
||||
if not row:
|
||||
return None
|
||||
domain, created_at = row
|
||||
# A NEGATIVE EXPIRES FAST; A POSITIVE LASTS.
|
||||
#
|
||||
# These searches fan out across several engines and the result set varies
|
||||
# run to run: the first Anil sweep found no brand site, and a repeat of the
|
||||
# identical query minutes later returned shop.theanilgroup.com three times
|
||||
# in the top four. Caching that miss for six months would have made one
|
||||
# flaky search a permanent fact about the brand, and every product of
|
||||
# Anil's would have kept reporting "sold nowhere".
|
||||
#
|
||||
# A found domain is stable and worth keeping; a miss is worth re-asking
|
||||
# tomorrow.
|
||||
ttl = BRAND_DOMAIN_TTL_SECONDS if domain else BRAND_DOMAIN_MISS_TTL_SECONDS
|
||||
if (time.time() - created_at) > ttl:
|
||||
return None
|
||||
return domain or None
|
||||
|
||||
|
||||
def _normalize_brand(brand: str) -> str:
|
||||
return re.sub(r"\s+", " ", (brand or "").strip().lower())
|
||||
|
||||
|
||||
def resolve_brand_domain(brand: str, *, live: bool = False,
|
||||
sample_products: Optional[Sequence[str]] = None,
|
||||
timeout: int = 20) -> Optional[str]:
|
||||
"""The brand's own website, e.g. "Anil" -> "theanilgroup.com".
|
||||
|
||||
WHY THIS EXISTS. The first backfill run asked only the retailer list and
|
||||
reported "Anil Wheat Vermicelli 180g" as not listed anywhere - while the
|
||||
brand's own shop sells exactly that pack. A manufacturer's catalogue is
|
||||
real evidence that a product and its pack size exist, and leaving it out
|
||||
turned "no big retailer stocks this" into "this product is not real",
|
||||
which are very different claims.
|
||||
|
||||
A brand site is WEAKER evidence than a retailer: a manufacturer may list a
|
||||
discontinued or not-yet-shipping line. The verdict records which domain
|
||||
answered, so a consumer can weigh them differently - see `RetailEvidence.
|
||||
retailer`.
|
||||
|
||||
Identified by a domain that CONTAINS a brand token and is neither a
|
||||
retailer nor a directory/social site. That is a deliberately conservative
|
||||
rule: it will miss a brand whose site is named nothing like the brand, and
|
||||
a miss just means we fall back to the retailer list.
|
||||
"""
|
||||
key = _normalize_brand(brand)
|
||||
if not key:
|
||||
return None
|
||||
|
||||
cached = get_cached_brand_domain(brand)
|
||||
if cached is not None:
|
||||
return cached
|
||||
if not live:
|
||||
return None
|
||||
|
||||
tokens = [t for t in off_bulk.brand_tokens(brand) if len(t) > 2]
|
||||
if not tokens:
|
||||
return None
|
||||
|
||||
# QUERY WITH A REAL PRODUCT NAME, NOT THE BRAND ALONE.
|
||||
#
|
||||
# "Anil official site products" returned maccosmetics.com, vertu.com and
|
||||
# a YouTube video: "Anil" is a common personal name, so a brand-only query
|
||||
# is dominated by cricketers and airlines - the same ambiguity that put a
|
||||
# photo of Anil Kapoor on a packet of rava. Adding a product the brand
|
||||
# actually sells makes the query specific enough to find the real site;
|
||||
# "Anil Wheat Vermicelli" surfaces shop.theanilgroup.com immediately.
|
||||
#
|
||||
# Several phrasings, because one search is demonstrably not enough: the
|
||||
# identical query minutes apart returned the brand site three times and
|
||||
# then not at all. Stops at the first that yields a domain.
|
||||
phrasings: List[str] = [f"{brand} {p}" for p in (sample_products or [])[:2]]
|
||||
phrasings += [f"{brand} official site products",
|
||||
f"{brand} official website India"]
|
||||
|
||||
found: Optional[str] = None
|
||||
asked = False
|
||||
for phrasing in phrasings:
|
||||
if found:
|
||||
break
|
||||
results = _search(phrasing, 10, timeout)
|
||||
if results is None:
|
||||
continue
|
||||
asked = True
|
||||
found = _pick_brand_domain(results, tokens)
|
||||
|
||||
if not asked:
|
||||
return None # could not ask - do NOT cache a failure as "none"
|
||||
|
||||
_cache_brand_domain(key, found or "")
|
||||
return found
|
||||
|
||||
|
||||
def _pick_brand_domain(results, tokens) -> Optional[str]:
|
||||
"""The first result whose domain stem contains a brand token and is not a
|
||||
retailer, directory or social site."""
|
||||
for result in results:
|
||||
url = str(result.get("href") or result.get("url") or "")
|
||||
if not url.startswith("http"):
|
||||
continue
|
||||
domain = _registrable(url)
|
||||
if not domain:
|
||||
continue
|
||||
if any(bad in domain for bad in _NOT_A_BRAND_SITE):
|
||||
continue
|
||||
if any(retailer in domain for retailer in RETAILER_DOMAINS):
|
||||
continue
|
||||
stem = domain.split(".")[0]
|
||||
if any(token in stem for token in tokens):
|
||||
return domain
|
||||
return None
|
||||
|
||||
|
||||
def _cache_brand_domain(key: str, domain: str) -> None:
|
||||
try:
|
||||
with _lock, closing(_connect()) as conn:
|
||||
_ensure_schema(conn)
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO brand_domain_cache (brand, domain, created_at)
|
||||
VALUES (?, ?, ?)
|
||||
ON CONFLICT(brand) DO UPDATE SET
|
||||
domain = excluded.domain, created_at = excluded.created_at
|
||||
""",
|
||||
(key, domain, time.time()),
|
||||
)
|
||||
conn.commit()
|
||||
except Exception as e: # noqa: BLE001 - best effort only
|
||||
logger.debug("Brand domain cache write failed for %s: %s", key, e)
|
||||
|
||||
|
||||
def _is_product_page(url: str) -> bool:
|
||||
"""Does this URL point at a specific product rather than a landing page?
|
||||
|
||||
Only applied to the BRAND's own domain. A retailer domain is a shop by
|
||||
definition, but a manufacturer's site is mostly corporate: `theanilgroup.com/`
|
||||
is an "about us" homepage and `shop.theanilgroup.com/products/wheat-vermicelli`
|
||||
is a listing, and both share a registrable domain.
|
||||
|
||||
Counting the homepage as "a shop was reached" is not harmless - it is what
|
||||
decides whether a miss is reported as `not_found` (an answer) or `unknown`
|
||||
(we never asked a shop). The corporate homepage ranks for almost every
|
||||
query about the brand, so without this every no-shop verdict would be
|
||||
dressed up as a real absence.
|
||||
"""
|
||||
path = urlparse(url or "").path.strip("/")
|
||||
if not path:
|
||||
return False
|
||||
return any(
|
||||
marker in f"/{path.lower()}/"
|
||||
for marker in ("/product", "/products/", "/shop/", "/collections/",
|
||||
"/p/", "/pd/", "/item", "/buy", "/store/")
|
||||
) or path.count("/") >= 1
|
||||
|
||||
|
||||
def _retailer_for(url: str) -> Optional[str]:
|
||||
lowered = (url or "").lower()
|
||||
for domain in RETAILER_DOMAINS:
|
||||
if domain in lowered:
|
||||
return domain
|
||||
return None
|
||||
|
||||
|
||||
def _clean_listing_title(title: str) -> str:
|
||||
return _LISTING_NOISE.sub(" ", title or "")
|
||||
|
||||
|
||||
def listing_matches(listing_title: str, brand: str, product_title: str,
|
||||
size: str) -> tuple:
|
||||
"""Does this listing describe THIS product at THIS pack size?
|
||||
|
||||
Returns (matches, coverage). Both halves are required - see the module
|
||||
docstring for the measurement that makes the size half non-negotiable.
|
||||
|
||||
COVERAGE, NOT SYMMETRIC SIMILARITY. `off_bulk.symmetric_similarity` is
|
||||
built to compare one database record with another and penalises extra
|
||||
tokens on either side. A retail listing title is not a record: it is a
|
||||
SUPERSET of the product name wrapped in shop furniture. Measured against
|
||||
real titles the live provider returned, symmetric similarity scored
|
||||
|
||||
"Anil Vermicelli - Roasted, 450g Pouch : Amazon.in: Grocery &
|
||||
Gourmet Foods" vs "Anil Roasted Short Vermicelli"
|
||||
-> 0.445, rejected
|
||||
|
||||
which is a genuine listing of exactly the product asked for. So the title
|
||||
half asks the containment question instead - how much of the PRODUCT's
|
||||
identity appears in the listing - and the pack size does the discriminating.
|
||||
"""
|
||||
if not listing_title:
|
||||
return False, 0.0
|
||||
|
||||
# Several listings run together into one title field - see
|
||||
# MAX_LISTING_TITLE_CHARS. Neither the size nor the words can be attributed
|
||||
# to a single product, so this corroborates nothing.
|
||||
if len(listing_title) > MAX_LISTING_TITLE_CHARS:
|
||||
return False, 0.0
|
||||
|
||||
# THE LISTING MUST BE FOR OUR BRAND.
|
||||
#
|
||||
# This was missing entirely, and it is the worst hole this module has had.
|
||||
# `normalize_for_match` strips brand tokens from BOTH sides - correct for
|
||||
# its original job, matching within one brand's own Open Food Facts corpus,
|
||||
# where the brand is a given. Here the listing can be anyone's, so stripping
|
||||
# the brand made the comparison brand-blind and a COMPETITOR's product
|
||||
# corroborated ours:
|
||||
#
|
||||
# "A1 Naanjil Naattu Masala Instant Pongal Mix, 500g"
|
||||
# confirmed Anil Pongal Mix 500g (real, from the backfill)
|
||||
# "Britannia Good Day 200g" confirmed "Anil Good Day 200g"
|
||||
# "MTR Rava Idli Mix 500g" confirmed "Anil Idli Mix 500g"
|
||||
#
|
||||
# Checked BEFORE the brand tokens are stripped, obviously, and via the
|
||||
# barcode matcher's own `brand_matches` so aliases ("HUL" for "Hindustan
|
||||
# Unilever") keep working and the two gates cannot drift apart.
|
||||
if not brand_matches(listing_title, brand, _aliases_for(brand)):
|
||||
return False, 0.0
|
||||
|
||||
drop = off_bulk.brand_tokens(brand)
|
||||
listing_clean = off_bulk.normalize_for_match(
|
||||
_clean_listing_title(listing_title), drop
|
||||
)
|
||||
target_clean = off_bulk.normalize_for_match(
|
||||
off_bulk.strip_sizes(product_title), drop
|
||||
)
|
||||
if not listing_clean:
|
||||
return False, 0.0
|
||||
|
||||
# Synonyms folded on BOTH sides, so "Puttu Mix" and "Puttu Maavu" reduce to
|
||||
# the same words. See _SYNONYMS - observed pairs only, and the pack-size
|
||||
# check below is untouched by any of it.
|
||||
target_tokens = [_SYNONYMS.get(t, t) for t in target_clean.split() if len(t) > 2]
|
||||
listing_tokens = _canonical_words(listing_clean)
|
||||
if target_tokens:
|
||||
coverage = sum(1 for t in target_tokens if t in listing_tokens) / len(target_tokens)
|
||||
else:
|
||||
# The product name is nothing but brand and size ("Amul 1kg"). There
|
||||
# is no identity to cover, so the brand's presence is all that can be
|
||||
# asked, and the size below carries the whole verdict.
|
||||
coverage = 1.0 if any(
|
||||
b in listing_title.lower() for b in off_bulk.brand_tokens(brand)
|
||||
) else 0.0
|
||||
|
||||
if coverage < TITLE_COVERAGE_FLOOR:
|
||||
return False, round(coverage, 3)
|
||||
|
||||
# A variant term the other side does not have means a different pack of a
|
||||
# similarly-named product (sugar-free vs regular, jar vs pouch).
|
||||
if off_bulk.has_extra_variant_conflict(listing_title, product_title):
|
||||
return False, round(coverage, 3)
|
||||
|
||||
# An INGREDIENT the listing names and the product does not. Containment
|
||||
# tolerates extra words on the listing side, which is right for shop
|
||||
# furniture and wrong for this: "Aachi Ginger Paste" is wholly contained in
|
||||
# "Aachi Ginger Garlic Paste" and scored 1.0 against it. See
|
||||
# _DISTINGUISHING_INGREDIENTS.
|
||||
extra = (listing_tokens & _DISTINGUISHING_INGREDIENTS) - set(target_tokens)
|
||||
if extra:
|
||||
return False, round(coverage, 3)
|
||||
|
||||
# THE SIZE CHECK, AND IT IS THE ONE THAT MATTERS.
|
||||
#
|
||||
# Measured on the live provider: searching "Anil Wheat Vermicelli 12g"
|
||||
# returns the brand's own generic product page, whose title covers 100% of
|
||||
# the product's tokens. Only the absence of any 12 g mention distinguishes
|
||||
# a pack that exists from one that does not. A listing that states no
|
||||
# quantity at all confirms no quantity at all.
|
||||
if size and size.strip():
|
||||
mentions = quantity_utils.find_quantity_mentions(listing_title)
|
||||
if not mentions:
|
||||
return False, round(coverage, 3)
|
||||
if not any(
|
||||
quantity_utils.quantities_match(size, f"{q}g") for q in mentions
|
||||
):
|
||||
return False, round(coverage, 3)
|
||||
|
||||
return True, round(coverage, 3)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The lookup
|
||||
# ---------------------------------------------------------------------------
|
||||
def _search(query: str, max_results: int, timeout: int) -> Optional[List[Dict[str, Any]]]:
|
||||
"""Raw web results, or None when the provider could not be reached.
|
||||
|
||||
None and [] are different answers and the caller depends on it: None is
|
||||
"we could not ask" and becomes `unknown`; [] is "we asked and nobody sells
|
||||
this" and becomes `not_found`.
|
||||
"""
|
||||
try:
|
||||
from ddgs import DDGS
|
||||
except ImportError:
|
||||
logger.debug("Retail presence unavailable: 'ddgs' is not installed")
|
||||
return None
|
||||
try:
|
||||
with DDGS(timeout=timeout) as ddgs:
|
||||
return list(ddgs.text(query, region="in-en", safesearch="off",
|
||||
max_results=max_results) or [])
|
||||
except Exception as e: # noqa: BLE001 - a throttle is not a verdict
|
||||
logger.debug("Retail search failed for %r: %s", query, e)
|
||||
return None
|
||||
|
||||
|
||||
def check_listing(brand: str, product_title: str, size: str, *,
|
||||
live: bool = False,
|
||||
refresh: bool = False,
|
||||
brand_domain: Optional[str] = None,
|
||||
ttl_seconds: float = DEFAULT_TTL_SECONDS,
|
||||
max_results: int = 10,
|
||||
timeout: int = 20) -> RetailEvidence:
|
||||
"""Is (brand, product_title, size) listed by a real retailer?
|
||||
|
||||
`live` defaults to FALSE. Ingestion reads the cache and never issues a
|
||||
network call; the backfill script passes live=True. See the module
|
||||
docstring - this is one blocking call per product against a provider that
|
||||
rate-limits, so it does not belong inside a batch.
|
||||
"""
|
||||
if not (brand or "").strip() or not (product_title or "").strip():
|
||||
return RetailEvidence(UNKNOWN)
|
||||
|
||||
# `refresh` SKIPS THE CACHE READ, and without it a caller cannot re-ask.
|
||||
#
|
||||
# The backfill script has its own --refresh, which decides which rows are
|
||||
# worth asking again - but it then called this function, which read the
|
||||
# cache first and handed back the very verdict the caller was trying to
|
||||
# replace. A whole "rebuild" of Anil ran to completion, reported 46
|
||||
# re-asks, and made no network calls at all: it re-read 46 stale rows and
|
||||
# wrote nothing, because the early return happens before set_cached.
|
||||
#
|
||||
# Silent, and it looks exactly like a successful run.
|
||||
if not refresh:
|
||||
cached = get_cached(brand, product_title, size, ttl_seconds)
|
||||
if cached is not None:
|
||||
return cached
|
||||
if not live:
|
||||
return RetailEvidence(UNKNOWN)
|
||||
|
||||
# Resolve the brand's own site if the caller did not name one. Cached per
|
||||
# brand, so this costs one extra search per BRAND, not per product. Without
|
||||
# it the first Anil sweep reported "Anil Wheat Vermicelli 180g" as sold
|
||||
# nowhere while the brand's own shop lists exactly that pack.
|
||||
if brand_domain is None:
|
||||
brand_domain = resolve_brand_domain(brand, live=True, timeout=timeout)
|
||||
|
||||
# A NEGATIVE IS CONFIRMED WITH A SECOND PHRASING, NOT TAKEN ON ONE TRY.
|
||||
#
|
||||
# `ddgs` fans out over several engines and the result set genuinely varies
|
||||
# between identical queries minutes apart - measured: "Anil Wheat
|
||||
# Vermicelli 180g" returned nothing usable on one run and an Amazon listing
|
||||
# for exactly that pack on the next. A single query is enough to CONFIRM a
|
||||
# product (a real listing is real however it was found) but not enough to
|
||||
# deny one, so only the negative pays for the retry.
|
||||
base = off_bulk.strip_sizes(product_title)
|
||||
phrasings = [
|
||||
" ".join(p for p in (brand, base, size) if p).strip(),
|
||||
" ".join(p for p in (brand, base, size, "buy online price") if p).strip(),
|
||||
]
|
||||
# A site:-scoped query is the reliable way to reach the brand's own shop,
|
||||
# but it is worth a third round trip only when the open queries reached no
|
||||
# shop at all - see the reach check after the loop.
|
||||
if brand_domain:
|
||||
phrasings.append(
|
||||
" ".join(p for p in (brand, base, size, f"site:{brand_domain}") if p).strip()
|
||||
)
|
||||
|
||||
domains = tuple(RETAILER_DOMAINS) + ((brand_domain,) if brand_domain else ())
|
||||
best = RetailEvidence(NOT_FOUND, checked_at=time.time())
|
||||
asked = False
|
||||
shop_results_seen = 0
|
||||
seen_titles: List[str] = []
|
||||
|
||||
for query in phrasings:
|
||||
# Only escalate to the site:-scoped query when nothing else reached a
|
||||
# shop. A product already seen on a retailer needs no third opinion.
|
||||
if "site:" in query and shop_results_seen:
|
||||
break
|
||||
results = _search(query, max_results, timeout)
|
||||
if results is None:
|
||||
continue
|
||||
asked = True
|
||||
for result in results:
|
||||
url = str(result.get("href") or result.get("url") or "")
|
||||
title = str(result.get("title") or "")
|
||||
retailer = _retailer_for(url) or (
|
||||
brand_domain if brand_domain and brand_domain in _registrable(url)
|
||||
and _is_product_page(url) else None
|
||||
)
|
||||
if not retailer or retailer not in domains:
|
||||
continue
|
||||
# Reached an actual shop. Counted whether or not it matches, because
|
||||
# that count is what separates "shops do not sell this" from "we
|
||||
# never got to ask a shop".
|
||||
shop_results_seen += 1
|
||||
if len(seen_titles) < 8 and title:
|
||||
seen_titles.append(f"{retailer}: {title}"[:160])
|
||||
matches, similarity = listing_matches(title, brand, product_title, size)
|
||||
if similarity > best.similarity:
|
||||
best.similarity = round(similarity, 3)
|
||||
if matches:
|
||||
evidence = RetailEvidence(
|
||||
FOUND, retailer=retailer, url=url, matched_title=title,
|
||||
matched_size=size, similarity=round(similarity, 3),
|
||||
checked_at=time.time(), shop_results_seen=shop_results_seen,
|
||||
query=query,
|
||||
)
|
||||
set_cached(brand, product_title, size, evidence)
|
||||
return evidence
|
||||
|
||||
if not asked:
|
||||
# Could not ask at all - every phrasing failed to reach the provider.
|
||||
return RetailEvidence(UNKNOWN)
|
||||
|
||||
# ASKED, BUT NEVER ASKED A SHOP.
|
||||
#
|
||||
# The searches came back full of recipe blogs, news pages and the brand's
|
||||
# corporate homepage, and not one result was on a retailer or the brand's
|
||||
# store. Recording that as "no retailer sells this product" states something
|
||||
# we did not find out - it is the same conflation as treating a throttle as
|
||||
# an absence, and it is what made 2 of 10 sampled Anil negatives meaningless.
|
||||
#
|
||||
# Reported as UNKNOWN, and UNKNOWN is never cached, so the question gets
|
||||
# asked again instead of the product being written off for a fortnight.
|
||||
if shop_results_seen == 0:
|
||||
return RetailEvidence(UNKNOWN, checked_at=time.time(), shop_results_seen=0)
|
||||
|
||||
best.shop_results_seen = shop_results_seen
|
||||
best.seen_titles = seen_titles
|
||||
set_cached(brand, product_title, size, best)
|
||||
return best
|
||||
|
||||
|
||||
def check_many(items: Sequence[Dict[str, str]], *, live: bool = False,
|
||||
deadline_seconds: Optional[float] = None,
|
||||
pause_seconds: float = PAUSE_SECONDS) -> Dict[str, RetailEvidence]:
|
||||
"""Check a batch, keyed by `cache_key`.
|
||||
|
||||
SEQUENTIAL, WITH PACING - deliberately not concurrent. `EnrichmentPipeline`
|
||||
bounds concurrency across ROWS, which is right when rows hit independent
|
||||
resources; here every row hits one throttled provider, so parallelism buys
|
||||
a 403 rather than throughput. `off_bulk.PAUSE_SECONDS` is the in-repo
|
||||
precedent for exactly this.
|
||||
|
||||
Duplicate keys are asked once: `cache_key` collapses pack-size variants of
|
||||
a product onto one product identity, so a brand with three sizes of each
|
||||
item costs a third of the queries.
|
||||
"""
|
||||
out: Dict[str, RetailEvidence] = {}
|
||||
started = time.monotonic()
|
||||
for item in items:
|
||||
brand = item.get("brand") or ""
|
||||
title = item.get("product_title") or item.get("title") or ""
|
||||
size = item.get("size") or ""
|
||||
key = cache_key(brand, title, size)
|
||||
if key in out:
|
||||
continue
|
||||
if deadline_seconds is not None and (time.monotonic() - started) > deadline_seconds:
|
||||
# Everything still unasked is `unknown`, not `not_found`.
|
||||
out[key] = RetailEvidence(UNKNOWN)
|
||||
continue
|
||||
cached = get_cached(brand, title, size)
|
||||
if cached is not None:
|
||||
out[key] = cached
|
||||
continue
|
||||
out[key] = check_listing(brand, title, size, live=live)
|
||||
if live:
|
||||
time.sleep(pause_seconds)
|
||||
return out
|
||||
@@ -92,8 +92,24 @@ CATEGORY_TYPE_WORDS: dict[str, set[str]] = {
|
||||
"curry powder": {"Spices & Masalas"}, "garam masala": {"Spices & Masalas"},
|
||||
"chilli powder": {"Spices & Masalas"}, "turmeric powder": {"Spices & Masalas"},
|
||||
"deggi mirch": {"Spices & Masalas"}, "tikhalal": {"Spices & Masalas"}, "kashmiri lal": {"Spices & Masalas"},
|
||||
"pasta": {"Pasta & Noodles"}, "vermicelli": {"Pasta & Noodles"}, "semiya": {"Pasta & Noodles"},
|
||||
"macaroni": {"Pasta & Noodles"}, "spaghetti": {"Pasta & Noodles"},
|
||||
# BOTH spellings, for every word in this group.
|
||||
#
|
||||
# `category_registry` is the module that ASSIGNS the category, and its
|
||||
# entry for "Noodles & Instant Food" lists "vermicelli" and "pasta" among
|
||||
# its own keywords. So the detector resolved "Anil Wheat Vermicelli" to
|
||||
# "Noodles & Instant Food" and this table then reported the word
|
||||
# "vermicelli" as CONTRADICTING it - a -0.35 confidence penalty on every
|
||||
# vermicelli, pasta, semiya, macaroni and spaghetti row in the catalogue,
|
||||
# for nothing but two tables spelling one category differently.
|
||||
#
|
||||
# `noodle`/`noodles` already carried both, because somebody hit this and
|
||||
# patched the row in front of them. The rest of the group has the same
|
||||
# bug. If a new pasta-ish word is added here, give it both.
|
||||
"pasta": {"Pasta & Noodles", "Noodles & Instant Food"},
|
||||
"vermicelli": {"Pasta & Noodles", "Noodles & Instant Food"},
|
||||
"semiya": {"Pasta & Noodles", "Noodles & Instant Food"},
|
||||
"macaroni": {"Pasta & Noodles", "Noodles & Instant Food"},
|
||||
"spaghetti": {"Pasta & Noodles", "Noodles & Instant Food"},
|
||||
"noodle": {"Pasta & Noodles", "Noodles & Instant Food"}, "noodles": {"Pasta & Noodles", "Noodles & Instant Food"},
|
||||
# --- Desserts (head-noun only, not ingredient words) ---
|
||||
"ice cream": {"Ice Cream"}, "icecream": {"Ice Cream"}, "kulfi": {"Ice Cream"}, "frozen dessert": {"Ice Cream"},
|
||||
|
||||
@@ -178,6 +178,17 @@ def get_brand_table_ddl(brand: str) -> str:
|
||||
nutrients_per_100g JSONB,
|
||||
search_query TEXT,
|
||||
|
||||
-- The deterministic validation verdict from product_validator.
|
||||
-- validate_catalog() has always computed these three and annotated
|
||||
-- them onto every row; until they were added here nothing projected
|
||||
-- them, so a fabricated row and a corroborated one were
|
||||
-- indistinguishable once stored. That is what made "is this product
|
||||
-- real?" unanswerable after the fact.
|
||||
-- status is one of: verified | needs_review | rejected.
|
||||
validation_status TEXT,
|
||||
confidence_score NUMERIC,
|
||||
validation_issues JSONB,
|
||||
|
||||
-- Per-field provenance, keyed by column name; each value records how
|
||||
-- that column's value was arrived at. One JSONB map rather than ~60
|
||||
-- scalar columns, because every scalar column would have to be named
|
||||
@@ -274,6 +285,11 @@ def _ensure_columns(cur, table_name: str) -> None:
|
||||
"nutrients": "TEXT[]",
|
||||
"nutrients_per_100g": "JSONB",
|
||||
"search_query": "TEXT",
|
||||
# The product_validator verdict - see get_brand_table_ddl for why a
|
||||
# row's confidence has to survive to disk to be worth computing.
|
||||
"validation_status": "TEXT",
|
||||
"confidence_score": "NUMERIC",
|
||||
"validation_issues": "JSONB",
|
||||
# Per-field provenance map - see get_brand_table_ddl for why this is
|
||||
# one JSONB column and not sixty scalar ones.
|
||||
"field_sources": "JSONB",
|
||||
@@ -488,6 +504,18 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
|
||||
field_sources = p.get("field_sources")
|
||||
field_sources = Json(field_sources) if isinstance(field_sources, dict) and field_sources else None
|
||||
|
||||
# The product_validator verdict, as annotated by validate_catalog().
|
||||
# A row that never went through the gate carries none of these, so
|
||||
# they arrive as None and the DO UPDATE SET below assigns NULL - see
|
||||
# the comment there for why that is the wanted behaviour here and the
|
||||
# opposite of the rule the enrichment columns follow.
|
||||
validation_status = str(p.get("validation_status") or "").strip() or None
|
||||
confidence_score = _to_numeric_or_none(p.get("confidence_score"))
|
||||
validation_issues = p.get("validation_issues")
|
||||
validation_issues = (
|
||||
Json(list(validation_issues)) if isinstance(validation_issues, (list, tuple)) else None
|
||||
)
|
||||
|
||||
# Essential fields
|
||||
highlights = p.get("highlights", [])
|
||||
if not isinstance(highlights, list):
|
||||
@@ -542,6 +570,9 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
|
||||
highlights, # TEXT[] - psycopg will handle conversion
|
||||
nutrients, # TEXT[] - psycopg will handle conversion
|
||||
search_query,
|
||||
validation_status,
|
||||
confidence_score,
|
||||
validation_issues,
|
||||
field_sources,
|
||||
embedding_str
|
||||
))
|
||||
@@ -586,8 +617,10 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
|
||||
fssai_license, product_sku, sku_source, hsn_code, final_selling_price, selling_price, barcode, barcode_type,
|
||||
gtin, ean13, upc, barcode_source, barcode_verified, barcode_lookup_status, barcode_last_updated,
|
||||
gst_percent, tax_amount, hsn_gst_needs_review,
|
||||
highlights, nutrients, search_query, field_sources, embedding)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
|
||||
highlights, nutrients, search_query,
|
||||
validation_status, confidence_score, validation_issues,
|
||||
field_sources, embedding)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
|
||||
ON CONFLICT (image_id) DO UPDATE SET
|
||||
product_name = EXCLUDED.product_name,
|
||||
title = EXCLUDED.title,
|
||||
@@ -625,6 +658,26 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
|
||||
nutrients = EXCLUDED.nutrients,
|
||||
search_query = EXCLUDED.search_query,
|
||||
embedding = EXCLUDED.embedding,
|
||||
-- PLAINLY ASSIGNED, NOT COALESCEd. THIS IS DELIBERATE AND
|
||||
-- IS THE OPPOSITE OF THE RULE THE BLOCK BELOW FOLLOWS.
|
||||
--
|
||||
-- Those are enrichment outputs: facts about the product,
|
||||
-- true whenever they were learned, so keeping a stored one
|
||||
-- when nothing new arrives is right. These three are not
|
||||
-- facts about the product - they are the verdict of the
|
||||
-- run that just wrote this row. Carrying a previous run's
|
||||
-- "verified" forward onto a row that has just been rewritten
|
||||
-- and not re-validated is precisely the failure this column
|
||||
-- was added to end: it would relabel an unchecked row as
|
||||
-- checked, which is worse than admitting NULL.
|
||||
--
|
||||
-- So a writer that does not validate (the seed loader,
|
||||
-- user_products, brand_sync's re-seed) correctly clears the
|
||||
-- verdict, and the row reads as "not assessed" until a
|
||||
-- validating path runs over it again.
|
||||
validation_status = EXCLUDED.validation_status,
|
||||
confidence_score = EXCLUDED.confidence_score,
|
||||
validation_issues = EXCLUDED.validation_issues,
|
||||
-- COALESCE, NOT PLAIN EXCLUDED, FOR EVERY COLUMN BELOW.
|
||||
--
|
||||
-- These are enrichment outputs. Most writers that reach
|
||||
|
||||
BIN
data/cache/image_corroboration.db
vendored
Normal file
BIN
data/cache/image_corroboration.db
vendored
Normal file
Binary file not shown.
BIN
data/cache/retail_presence.db
vendored
Normal file
BIN
data/cache/retail_presence.db
vendored
Normal file
Binary file not shown.
111
scripts/audit_retail_negatives.py
Normal file
111
scripts/audit_retail_negatives.py
Normal file
@@ -0,0 +1,111 @@
|
||||
"""
|
||||
Show a human the shop listings a `not_found` verdict rejected, so the
|
||||
false-negative rate can be measured instead of assumed.
|
||||
|
||||
WHY THIS CANNOT BE AUTOMATED
|
||||
-----------------------------
|
||||
The obvious way to check a negative is to run the search again and see whether
|
||||
it matches this time. That is what the first investigation did, and the number
|
||||
it produced (0 false negatives out of 10) was worthless, because the re-check
|
||||
used the SAME `listing_matches` that produced the negatives. A matcher cannot
|
||||
find its own blind spots: it scored `Anil Puttu Maavu` against
|
||||
`Anil Puttu Mix` as a correct rejection, and *maavu* is simply Tamil for the
|
||||
flour.
|
||||
|
||||
Re-running is also unreliable in its own right - the provider's reach varies
|
||||
between identical queries minutes apart (7 shop results, then 0).
|
||||
|
||||
So the only honest measurement is a person reading the listing titles. This
|
||||
script puts them in front of one. It makes no network calls and no judgements;
|
||||
it reads what `check_listing` already stored.
|
||||
|
||||
READING THE OUTPUT
|
||||
------------------
|
||||
For each rejected row it prints the shop listings that were reached. Classify:
|
||||
|
||||
genuine the listings really are other products
|
||||
e.g. "Aachi BIRYANI MASALA" is not "Aachi Biryani Mix",
|
||||
and a 450g pouch does not confirm a 180g pack
|
||||
matcher a listing IS this product and was rejected anyway
|
||||
e.g. "Anil Puttu Maavu" for "Anil Puttu Mix"
|
||||
|
||||
A `matcher` verdict means a synonym is missing (see `_SYNONYMS`) or coverage is
|
||||
too strict. A `genuine` verdict is what this whole pipeline is for.
|
||||
|
||||
Rows with `shop_results_seen == 0` are NOT shown: those are recorded as
|
||||
`unknown`, not `not_found`, and are not claims about availability at all.
|
||||
|
||||
USAGE
|
||||
python scripts/audit_retail_negatives.py --brand Anil --limit 30
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sqlite3
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import List, Optional
|
||||
|
||||
_BACKEND_DIR = Path(__file__).resolve().parent.parent
|
||||
if str(_BACKEND_DIR) not in sys.path:
|
||||
sys.path.insert(0, str(_BACKEND_DIR))
|
||||
|
||||
from app.services import retail_presence # noqa: E402
|
||||
|
||||
|
||||
def main(argv: Optional[List[str]] = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
parser.add_argument("--brand", help="Only this brand.")
|
||||
parser.add_argument("--limit", type=int, default=30)
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
db = retail_presence._DB_PATH
|
||||
if not db.exists():
|
||||
print(f"No cache at {db} - run backfill_retail_presence.py first.")
|
||||
return 1
|
||||
|
||||
conn = sqlite3.connect(str(db))
|
||||
rows = conn.execute(
|
||||
"SELECT brand, product, size, result_json FROM retail_presence_cache"
|
||||
).fetchall()
|
||||
conn.close()
|
||||
|
||||
shown = 0
|
||||
no_titles = 0
|
||||
for brand, product, size, payload in sorted(rows):
|
||||
if args.brand and args.brand.strip().lower() not in (brand or "").lower():
|
||||
continue
|
||||
data = json.loads(payload)
|
||||
if data.get("status") != retail_presence.NOT_FOUND:
|
||||
continue
|
||||
titles = data.get("seen_titles") or []
|
||||
if not titles:
|
||||
# A negative recorded before seen_titles existed. Nothing to audit
|
||||
# without re-querying, which is exactly what this script refuses to
|
||||
# do - re-run the backfill with --refresh to repopulate it.
|
||||
no_titles += 1
|
||||
continue
|
||||
if shown >= args.limit:
|
||||
break
|
||||
shown += 1
|
||||
print(f"\n{shown:>3}. {product} [{size}] ({data.get('shop_results_seen', 0)} shop listings reached)")
|
||||
for title in titles:
|
||||
print(f" {title}")
|
||||
print(" -> genuine / matcher ?")
|
||||
|
||||
print(f"\n{'-' * 66}")
|
||||
print(f"shown for audit : {shown}")
|
||||
if no_titles:
|
||||
print(f"negatives with no titles : {no_titles}"
|
||||
f" (recorded before titles were kept - re-run with --refresh)")
|
||||
print("\nClassify each as `genuine` (really other products) or `matcher`")
|
||||
print("(a real listing we rejected). matcher / (genuine + matcher) is the")
|
||||
print("false-negative rate, and it is the number that says whether the")
|
||||
print("retail signal can be trusted.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
258
scripts/backfill_retail_presence.py
Normal file
258
scripts/backfill_retail_presence.py
Normal file
@@ -0,0 +1,258 @@
|
||||
"""
|
||||
Ask real shops which of our products they actually sell, and cache the answers.
|
||||
|
||||
WHY THIS IS A SCRIPT AND NOT A PIPELINE STAGE
|
||||
----------------------------------------------
|
||||
Open Food Facts can be asked about a whole brand in one request, which is why
|
||||
`off_bulk` can run inside ingestion. There is no equivalent for retail: every
|
||||
product costs its own web search, measured at 2.6-5.7 seconds, against a
|
||||
provider that 403s under load. Two hundred products is fifteen minutes of
|
||||
blocking calls and a near-certain throttle partway through - and a throttled
|
||||
lookup that got recorded as "nobody sells this" would demote real products.
|
||||
|
||||
So the querying lives here, offline and paced, and everything at ingestion time
|
||||
reads the cache (`retail_presence.check_listing(..., live=False)`). Discovery
|
||||
never blocks and never trips a rate limit mid-preview.
|
||||
|
||||
WHAT COUNTS AS A HIT
|
||||
--------------------
|
||||
The listing's title AND its pack size must both match. See
|
||||
`retail_presence`'s docstring for the measurement that makes the size half
|
||||
non-negotiable: searching for the fabricated "Anil Wheat Vermicelli 12g"
|
||||
returns the brand's own product page, whose title matches perfectly. Only the
|
||||
absence of any 12 g mention distinguishes a pack that exists from one that
|
||||
does not.
|
||||
|
||||
USAGE
|
||||
-----
|
||||
# Dry run - show what would be asked, ask nothing.
|
||||
python scripts/backfill_retail_presence.py --brand Anil
|
||||
|
||||
# Really query, politely.
|
||||
python scripts/backfill_retail_presence.py --brand Anil --apply
|
||||
|
||||
# Every brand that has a seed catalogue.
|
||||
python scripts/backfill_retail_presence.py --all --apply --limit 50
|
||||
|
||||
`--apply` is required to make any network call at all, mirroring
|
||||
`repair_brand_images.py`'s dry-run-by-default stance. Re-running is cheap:
|
||||
anything already cached and inside its TTL is skipped, so an interrupted sweep
|
||||
resumes where it stopped.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Dict, Iterable, List, Optional, Tuple
|
||||
|
||||
_BACKEND_DIR = Path(__file__).resolve().parent.parent
|
||||
if str(_BACKEND_DIR) not in sys.path:
|
||||
sys.path.insert(0, str(_BACKEND_DIR))
|
||||
|
||||
from app.services import retail_presence # noqa: E402
|
||||
from app.services.image_corroboration import SIZE_TAIL # noqa: E402
|
||||
|
||||
logger = logging.getLogger("backfill_retail_presence")
|
||||
|
||||
SEED_DIR = _BACKEND_DIR / "data" / "seed_catalogs"
|
||||
|
||||
|
||||
def _load_seed_products(path: Path) -> Tuple[str, List[Dict[str, str]]]:
|
||||
"""(brand, [{product_title, size}, ...]) from one seed catalogue."""
|
||||
try:
|
||||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
except Exception as e: # noqa: BLE001 - one bad file cannot stop the sweep
|
||||
logger.warning("Could not read %s: %s", path.name, e)
|
||||
return "", []
|
||||
products = data.get("products") if isinstance(data, dict) else data
|
||||
brand = (data.get("brand") if isinstance(data, dict) else "") or ""
|
||||
out: List[Dict[str, str]] = []
|
||||
for product in products or []:
|
||||
title = product.get("title") or product.get("product_name") or ""
|
||||
if not title:
|
||||
continue
|
||||
row_brand = brand or product.get("brand") or ""
|
||||
# ONE ENTRY PER PACK, because the pack size is the thing being
|
||||
# verified - "Anil Wheat Vermicelli" exists and "Anil Wheat Vermicelli
|
||||
# 12g" does not, and a per-product check cannot tell them apart.
|
||||
#
|
||||
# Two shapes in the wild: newer seed files carry `size_variants`
|
||||
# (sometimes as "90g - Rs18"), while the archive leaves that null and
|
||||
# puts the size on the end of `product_name`.
|
||||
for size in _sizes_for(product):
|
||||
out.append({
|
||||
"brand": row_brand,
|
||||
"product_title": title,
|
||||
"size": size,
|
||||
})
|
||||
return brand, out
|
||||
|
||||
|
||||
def _sizes_for(product: Dict[str, object]) -> List[str]:
|
||||
sizes = product.get("size_variants")
|
||||
if isinstance(sizes, list) and sizes:
|
||||
cleaned = [str(s).split(" - ")[0].strip() for s in sizes if str(s).strip()]
|
||||
if cleaned:
|
||||
return cleaned
|
||||
match = SIZE_TAIL.search(str(product.get("product_name") or ""))
|
||||
if match:
|
||||
return [match.group(0).strip()]
|
||||
# No size anywhere. Still worth asking whether the product exists at all;
|
||||
# `listing_matches` skips the size half when the size is blank.
|
||||
return [""]
|
||||
|
||||
|
||||
def _dedupe(items: Iterable[Dict[str, str]]) -> List[Dict[str, str]]:
|
||||
"""One entry per (product, size).
|
||||
|
||||
`search_key` collapses a trailing pack size out of the NAME, so the three
|
||||
rows a size explosion produced from one source product share a product
|
||||
identity and differ only in the size component. A brand with three sizes
|
||||
of each item therefore costs roughly a third of the naive query count.
|
||||
"""
|
||||
seen = set()
|
||||
out: List[Dict[str, str]] = []
|
||||
for item in items:
|
||||
key = retail_presence.cache_key(
|
||||
item["brand"], item["product_title"], item.get("size", "")
|
||||
)
|
||||
if key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
out.append(item)
|
||||
return out
|
||||
|
||||
|
||||
def backfill(items: List[Dict[str, str]], *, apply: bool, limit: Optional[int],
|
||||
pause: float, refresh: bool = False) -> Dict[str, int]:
|
||||
stats = {"asked": 0, "found": 0, "not_found": 0, "unknown": 0, "cached": 0,
|
||||
"shop_reached": 0, "no_shop_reached": 0}
|
||||
asked = 0
|
||||
domains: Dict[str, Optional[str]] = {}
|
||||
for item in items:
|
||||
brand, title, size = item["brand"], item["product_title"], item.get("size", "")
|
||||
|
||||
cached = retail_presence.get_cached(brand, title, size)
|
||||
# `--refresh` re-asks NOT_FOUND only. A `found` is still true - a shop
|
||||
# that listed the pack yesterday did list it - but a `not_found`
|
||||
# recorded before the brand's own site was consulted is not an answer
|
||||
# to the same question, and the first Anil sweep is full of them.
|
||||
if cached is not None and not (refresh and not cached.is_found):
|
||||
stats["cached"] += 1
|
||||
continue
|
||||
|
||||
if limit is not None and asked >= limit:
|
||||
break
|
||||
|
||||
if not apply:
|
||||
stats["asked"] += 1
|
||||
asked += 1
|
||||
logger.info("[dry-run] would ask: %s %s %s", brand, title, size)
|
||||
continue
|
||||
|
||||
# Resolved once per brand and reused, so the brand-site lookup costs
|
||||
# one search per BRAND rather than one per product.
|
||||
if brand not in domains:
|
||||
# Real product names make the query specific enough to find the
|
||||
# brand's own site - "Anil" alone returns cricketers and airlines.
|
||||
samples = [i["product_title"] for i in items if i["brand"] == brand][:2]
|
||||
domains[brand] = retail_presence.resolve_brand_domain(
|
||||
brand, live=True, sample_products=samples)
|
||||
logger.info("brand site for %s: %s", brand, domains[brand] or "(none found)")
|
||||
time.sleep(pause)
|
||||
evidence = retail_presence.check_listing(
|
||||
brand, title, size, live=True, brand_domain=domains[brand],
|
||||
# Without this the cache read inside check_listing hands back
|
||||
# the very verdict --refresh exists to replace.
|
||||
refresh=refresh)
|
||||
stats["asked"] += 1
|
||||
stats[evidence.status] = stats.get(evidence.status, 0) + 1
|
||||
asked += 1
|
||||
symbol = {"found": "OK ", "not_found": "-- ", "unknown": "?? "}[evidence.status]
|
||||
logger.info("%s %-42s %-8s %s", symbol, title[:42], size, evidence.note())
|
||||
if evidence.status == retail_presence.NOT_FOUND:
|
||||
stats["shop_reached"] += 1
|
||||
|
||||
# TWO KINDS OF UNKNOWN, AND ONLY ONE OF THEM MEANS STOP.
|
||||
#
|
||||
# `checked_at` set - we searched fine, but no result was on a shop.
|
||||
# That is an ordinary outcome (2 of 10 sampled negatives) and the sweep
|
||||
# carries on; it just is not an answer about availability.
|
||||
#
|
||||
# `checked_at` unset - every phrasing failed to reach the provider,
|
||||
# which is how a throttle presents. Hammering through hundreds more
|
||||
# queries after it has started refusing turns one rate limit into a
|
||||
# longer ban while recording nothing, because unknown is never cached.
|
||||
if evidence.status == retail_presence.UNKNOWN:
|
||||
if evidence.checked_at is None:
|
||||
logger.warning("Provider unreachable - stopping this sweep. "
|
||||
"Re-run later; cached answers are kept.")
|
||||
break
|
||||
stats["no_shop_reached"] += 1
|
||||
|
||||
time.sleep(pause)
|
||||
return stats
|
||||
|
||||
|
||||
def main(argv: Optional[List[str]] = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
parser.add_argument("--brand", help="Only this brand (matches the seed file's brand).")
|
||||
parser.add_argument("--all", action="store_true", help="Every seed catalogue.")
|
||||
parser.add_argument("--apply", action="store_true",
|
||||
help="Actually query. Without this nothing hits the network.")
|
||||
parser.add_argument("--limit", type=int, default=None,
|
||||
help="Stop after this many NEW lookups (cached ones are free).")
|
||||
parser.add_argument("--refresh", action="store_true",
|
||||
help="Re-ask entries previously recorded as not-found "
|
||||
"(a cached hit is still valid and is kept).")
|
||||
parser.add_argument("--pause", type=float, default=retail_presence.PAUSE_SECONDS,
|
||||
help=f"Seconds between queries (default {retail_presence.PAUSE_SECONDS}).")
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
||||
|
||||
if not args.brand and not args.all:
|
||||
parser.error("give --brand NAME or --all")
|
||||
|
||||
items: List[Dict[str, str]] = []
|
||||
for path in sorted(SEED_DIR.rglob("brand_catalog_*.json")):
|
||||
brand, products = _load_seed_products(path)
|
||||
if args.brand and args.brand.strip().lower() not in (brand or "").lower():
|
||||
continue
|
||||
items.extend(products)
|
||||
|
||||
items = _dedupe(items)
|
||||
if not items:
|
||||
logger.error("No products found. Check --brand spelling against the seed files.")
|
||||
return 1
|
||||
|
||||
logger.info("%d distinct product+size combinations to check%s",
|
||||
len(items), "" if args.apply else " (DRY RUN - use --apply to query)")
|
||||
stats = backfill(items, apply=args.apply, limit=args.limit, pause=args.pause,
|
||||
refresh=args.refresh)
|
||||
logger.info("")
|
||||
logger.info("already cached : %d", stats["cached"])
|
||||
logger.info("asked : %d", stats["asked"])
|
||||
if args.apply:
|
||||
logger.info(" found : %d", stats.get("found", 0))
|
||||
logger.info(" not found : %d (a shop was reached and did not list it)",
|
||||
stats.get("not_found", 0))
|
||||
logger.info(" no shop reached : %d (searched, but nothing on a shop - NOT an answer)",
|
||||
stats.get("no_shop_reached", 0))
|
||||
# Reach is the denominator that makes the found rate mean anything: a
|
||||
# brand whose products never surface on a shop has no measured rate at
|
||||
# all, however many not_founds it accumulated before this was tracked.
|
||||
answered = stats.get("found", 0) + stats.get("not_found", 0)
|
||||
if answered:
|
||||
logger.info(" -> of %d ANSWERED, %d%% are listed",
|
||||
answered, stats.get("found", 0) * 100 // answered)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -75,6 +75,7 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from app.infrastructure.settings import DB_HOST, DB_NAME # noqa: E402
|
||||
from app.services.brand_registry import resolve_parent_brand # noqa: E402
|
||||
from app.services import image_corroboration # noqa: E402
|
||||
from app.services.generic_products import OWN_PRODUCTS_BRAND # noqa: E402
|
||||
from app.services import produce_reference # noqa: E402
|
||||
from app.services.vector_store import ( # noqa: E402
|
||||
@@ -339,10 +340,7 @@ def _backup(tables: Dict[str, List[Dict[str, Any]]]) -> Path:
|
||||
# product.
|
||||
_SEARCH_CACHE: Dict[Tuple[str, str], List[str]] = {}
|
||||
|
||||
_SIZE_TAIL = re.compile(
|
||||
r"\s+\d+(?:\.\d+)?\s*(?:g|gm|gms|kg|ml|l|ltr|litre|liter|pcs|pc|n|no|nos)\b\.?\s*$",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_SIZE_TAIL = image_corroboration.SIZE_TAIL
|
||||
|
||||
|
||||
# --- Open Food Facts cross-check -------------------------------------------
|
||||
@@ -356,74 +354,19 @@ _SIZE_TAIL = re.compile(
|
||||
# away. If OFF's own record does not mention our brand, the image is somebody
|
||||
# else's product and is dropped. Nothing else can catch this: the URL is opaque
|
||||
# digits, so no amount of filename matching would help.
|
||||
_OFF_CACHE: Dict[str, bool] = {}
|
||||
_OFF_BARCODE = re.compile(r"/images/products/((?:\d+/)+)")
|
||||
|
||||
|
||||
# "own" and "products" are the BUCKET's name, not a brand's. Left in, the token
|
||||
# "products" corroborates any URL containing /images/products/ or
|
||||
# /cdn/shop/products/ - which is every Open*Facts and every Shopify path - so
|
||||
# `_names_product` waved through 40+ images that named nothing about the item.
|
||||
# That is how the openbeautyfacts cosmetics photos became the stored image for
|
||||
# Banana, Orange, Papaya, Guava, Lemon and twenty more.
|
||||
_BUCKET_TOKENS = frozenset({"own", "products", "product"})
|
||||
|
||||
|
||||
def _brand_tokens(brand: str) -> List[str]:
|
||||
return [w for w in re.split(r"[^a-z0-9]+", (brand or "").lower())
|
||||
if len(w) > 2 and w not in _BUCKET_TOKENS]
|
||||
|
||||
|
||||
def _off_product_matches_brand(url: str, brand: str) -> bool:
|
||||
"""False only when OFF positively says this barcode is another brand."""
|
||||
if "openfoodfacts.org" not in url:
|
||||
return True
|
||||
match = _OFF_BARCODE.search(url)
|
||||
if not match:
|
||||
return True
|
||||
barcode = match.group(1).replace("/", "")
|
||||
tokens = _brand_tokens(brand)
|
||||
if not tokens:
|
||||
return True
|
||||
|
||||
key = f"{barcode}:{brand.lower()}"
|
||||
if key in _OFF_CACHE:
|
||||
return _OFF_CACHE[key]
|
||||
|
||||
verdict = True
|
||||
try:
|
||||
import requests
|
||||
resp = requests.get(
|
||||
f"https://world.openfoodfacts.org/api/v2/product/{barcode}.json",
|
||||
params={"fields": "brands,product_name"},
|
||||
timeout=10,
|
||||
headers={"User-Agent": "nearle-catalogue-repair/1.0"},
|
||||
)
|
||||
if resp.ok:
|
||||
product = (resp.json() or {}).get("product") or {}
|
||||
haystack = f"{product.get('brands') or ''} {product.get('product_name') or ''}".lower()
|
||||
if haystack.strip():
|
||||
verdict = any(t in haystack for t in tokens)
|
||||
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
|
||||
verdict = True
|
||||
|
||||
_OFF_CACHE[key] = verdict
|
||||
return verdict
|
||||
|
||||
|
||||
def _names_product(url: str, product_name: str, brand: str) -> bool:
|
||||
"""True when the URL itself corroborates the match.
|
||||
|
||||
Not a requirement - Zepto and Flipkart serve opaque hashed paths for
|
||||
perfectly correct images - but a strong signal, so corroborated URLs are
|
||||
ranked ahead of uncorroborated ones.
|
||||
"""
|
||||
lowered = url.lower()
|
||||
words = [
|
||||
w for w in re.split(r"[^a-z0-9]+", (product_name or "").lower())
|
||||
if len(w) > 2 and not re.fullmatch(r"\d+(?:g|kg|ml|l)?", w)
|
||||
]
|
||||
return any(w in lowered for w in words + _brand_tokens(brand))
|
||||
# All of the above now lives in app/services/image_corroboration.py so the
|
||||
# INGESTION path can use it too. This script had the only working version of
|
||||
# this logic and no pipeline could import it, which is why the same bad image
|
||||
# had to be repaired after the fact instead of never being stored.
|
||||
#
|
||||
# `names_product` there is STRICTER than the copy that used to be here: it
|
||||
# requires a distinctive token rather than accepting a brand token, because a
|
||||
# brand that is also a personal name (Anil) corroborated a press photo of the
|
||||
# person. See that module's docstring.
|
||||
_names_product = image_corroboration.names_product
|
||||
_off_product_matches_brand = image_corroboration.openfacts_product_matches_brand
|
||||
_brand_tokens = image_corroboration.brand_tokens
|
||||
_BUCKET_TOKENS = image_corroboration.BUCKET_TOKENS
|
||||
|
||||
|
||||
def _search_key(product_name: str, brand: str) -> Tuple[str, str]:
|
||||
|
||||
195
tests/test_image_corroboration.py
Normal file
195
tests/test_image_corroboration.py
Normal file
@@ -0,0 +1,195 @@
|
||||
"""Does an image URL actually depict THIS product?
|
||||
|
||||
THE FAILURE THIS FILE EXISTS FOR
|
||||
--------------------------------
|
||||
The live catalogue illustrated "Anil Samba Rava" with a press photograph of the
|
||||
actor Anil Kapoor. Every check the ingestion path had was satisfied: the URL
|
||||
served real image bytes, above the size floor, from a reputable host.
|
||||
`validate_image_url_live` asks whether a URL is alive, not whether it is right,
|
||||
and `_select_best_images` ranks candidates without ever deciding that none of
|
||||
them qualifies.
|
||||
|
||||
`scripts/repair_brand_images.py` already had a corroboration gate for this class
|
||||
of defect - and it passed the celebrity photo, because it corroborated against
|
||||
`words + brand_tokens` and the brand IS a personal name. The token that made the
|
||||
match wrong was the token that satisfied the gate.
|
||||
|
||||
So these tests pin two things: the gate keys on DISTINCTIVE tokens (title minus
|
||||
brand minus pack size), and a product with no corroborated candidate gets NO
|
||||
primary image rather than a confident wrong one.
|
||||
|
||||
WHAT MUST NOT REGRESS
|
||||
---------------------
|
||||
`catalog_engine._select_best_images` deliberately keeps uncorroborated URLs
|
||||
because retailer CDN filenames are opaque hashes, and dropping them would leave
|
||||
real products with no image at all. That reasoning is correct and
|
||||
`test_an_opaque_retailer_cdn_path_still_yields_a_primary` is its guard. The
|
||||
reconciliation is that the two arguments are domain-scoped: a hashed Flipkart
|
||||
path cannot name anything, while a human-authored Wikimedia filename could have
|
||||
and did not.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from app.services import image_corroboration as ic
|
||||
|
||||
|
||||
KAPOOR = "https://upload.wikimedia.org/wikipedia/commons/2/2f/Anil_Kapoor_2019.jpg"
|
||||
REAL_RAVA = "https://shop.theanilgroup.com/products/anil-samba-rava-500g.jpg"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The reported defect
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_a_brand_that_is_also_a_personal_name_does_not_corroborate_a_person():
|
||||
"""THE REPORTED DEFECT. "anil" appears in the title, in the brand and in
|
||||
the URL, so any gate that accepts a brand token accepts this photograph."""
|
||||
assert ic.names_product(KAPOOR, "Anil Samba Rava", "Anil") is False
|
||||
|
||||
|
||||
def test_the_distinctive_tokens_are_what_the_gate_keys_on():
|
||||
assert ic.distinctive_tokens("Anil Samba Rava", "Anil") == ["samba", "rava"]
|
||||
assert ic.distinctive_tokens("Britannia Good Day 200g", "Britannia") == ["good", "day"]
|
||||
|
||||
|
||||
def test_the_real_product_photo_is_corroborated():
|
||||
assert ic.names_product(REAL_RAVA, "Anil Samba Rava", "Anil") is True
|
||||
|
||||
|
||||
def test_no_corroborated_candidate_means_no_primary_image():
|
||||
"""A blank image renders as the brand monogram, which is honest. Another
|
||||
company's product is not, and it stays invisible until somebody happens to
|
||||
recognise the photo."""
|
||||
choice = ic.choose_primary([KAPOOR], "Anil Samba Rava", "Anil", check_openfacts=False)
|
||||
|
||||
assert choice.primary is None
|
||||
# Withheld from promotion, NOT discarded - a human can still review it.
|
||||
assert choice.ordered == [KAPOOR]
|
||||
|
||||
|
||||
def test_a_corroborated_candidate_outranks_an_uncorroborated_one():
|
||||
choice = ic.choose_primary(
|
||||
[KAPOOR, REAL_RAVA], "Anil Samba Rava", "Anil", check_openfacts=False
|
||||
)
|
||||
|
||||
assert choice.primary == REAL_RAVA
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# What must not regress
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_an_opaque_retailer_cdn_path_still_yields_a_primary():
|
||||
"""catalog_engine._select_best_images keeps uncorroborated URLs on purpose:
|
||||
a hashed CDN filename names nothing, so failing to match it is not evidence
|
||||
the photo is wrong. Failing closed here would blank real products."""
|
||||
hashed = [
|
||||
"https://www.bbassets.com/media/uploads/p/l/a8f7d2e1c9.jpg",
|
||||
"https://rukminim.flixcart.com/image/7f3a99b2.jpeg",
|
||||
]
|
||||
|
||||
choice = ic.choose_primary(hashed, "Anil Samba Rava", "Anil", check_openfacts=False)
|
||||
|
||||
assert choice.primary == hashed[0]
|
||||
|
||||
|
||||
def test_a_title_with_no_distinctive_words_still_gets_a_primary():
|
||||
""""Godrej 50ml" and "Lion Dates 100g" carry no product identity at all, so
|
||||
a brand match is the best signal available and withholding gains nothing.
|
||||
Measured over the seed catalogues, treating these as ineligible accounted
|
||||
for 14 of 45 withheld primaries - every one a brand's own product page."""
|
||||
own_site = "https://liondates.com/cdn/shop/files/Sukkari_dates_front.png"
|
||||
|
||||
choice = ic.choose_primary([own_site], "Lion Dates 100g", "Lion Dates",
|
||||
check_openfacts=False)
|
||||
|
||||
assert choice.primary == own_site
|
||||
|
||||
|
||||
def test_the_case_the_repair_script_was_written_for_still_passes():
|
||||
""""Aachi Kulambu Mix" returning a press photo of a politician is the
|
||||
failure the original gate was built for. It must keep working - the change
|
||||
made here is strictly narrower, not different."""
|
||||
real = "https://commons.wikimedia.org/Aachi_Kulambu_Mix.jpg"
|
||||
|
||||
assert ic.names_product(real, "Aachi Kulambu Mix", "Aachi") is True
|
||||
|
||||
|
||||
def test_bucket_tokens_do_not_corroborate_anything():
|
||||
""""products" matches every Open*Facts and every Shopify path. Left in, it
|
||||
corroborated 40+ images that named nothing about the item - which is how
|
||||
openbeautyfacts cosmetics photos became the stored image for Banana,
|
||||
Orange, Papaya, Guava and Lemon."""
|
||||
shopify = "https://cdn.shopify.com/s/files/1/cdn/shop/products/9c1f0b.jpg"
|
||||
|
||||
assert ic.names_product(shopify, "Own Products Banana", "Own Products") is False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The person-filename signal demotes; it never rejects
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_a_person_shaped_filename_is_only_a_demotion_signal():
|
||||
"""`Britannia_Good_Day.jpg` has exactly the shape of `Anil_Kapoor_2019.jpg`
|
||||
and is a perfectly good product photo. This signal orders candidates; it is
|
||||
not allowed to eliminate one, or that image would be lost."""
|
||||
product_photo = "https://upload.wikimedia.org/wikipedia/commons/a/a1/Britannia_Good_Day.jpg"
|
||||
|
||||
assert ic.looks_like_person_photo(product_photo) is True
|
||||
# ...and yet it is still promoted, because it names the product.
|
||||
choice = ic.choose_primary([product_photo], "Britannia Good Day 200g",
|
||||
"Britannia", check_openfacts=False)
|
||||
assert choice.primary == product_photo
|
||||
|
||||
|
||||
def test_a_hashed_filename_is_not_person_shaped():
|
||||
assert ic.looks_like_person_photo("https://cdn.bbassets.com/a8f7d2e1.jpg") is False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Open*Facts brand cross-check
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_the_openfacts_cross_check_covers_the_sibling_databases():
|
||||
"""The original tested `if "openfoodfacts.org" not in url: return True`, so
|
||||
every openbeautyfacts and openproductsfacts image skipped the check - which
|
||||
is precisely the non-food case, and precisely the two sibling databases
|
||||
image_search.OPEN_FACTS_HOSTS queries."""
|
||||
for host in ("openfoodfacts", "openbeautyfacts", "openproductsfacts"):
|
||||
url = f"https://images.{host}.org/images/products/890/604/215/0067/front.jpg"
|
||||
assert ic._api_host_for(url) is not None, f"{host} must be cross-checked"
|
||||
|
||||
|
||||
def test_a_lookup_failure_never_rejects_an_image(monkeypatch):
|
||||
"""The asymmetry is the whole point: only a positive statement that this
|
||||
barcode belongs to another brand may reject. A network failure must not."""
|
||||
def boom(*_a, **_kw):
|
||||
raise RuntimeError("network down")
|
||||
|
||||
monkeypatch.setattr(ic.requests, "get", boom)
|
||||
url = "https://images.openfoodfacts.org/images/products/890/604/215/0067/front.jpg"
|
||||
|
||||
assert ic.openfacts_product_matches_brand(url, "Anil") is True
|
||||
|
||||
|
||||
def test_a_non_openfacts_url_is_not_cross_checked():
|
||||
assert ic.openfacts_product_matches_brand(REAL_RAVA, "Anil") is True
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# search_key
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_pack_sizes_of_one_product_share_a_search_key():
|
||||
"""Sharing across sizes is correct rather than merely cheap: it is the same
|
||||
product in a different pack, and stage 4's size explosion produces exactly
|
||||
these rows from one source product."""
|
||||
assert (ic.search_key("Anil Wheat Vermicelli 180g", "Anil")
|
||||
== ic.search_key("Anil Wheat Vermicelli 450g", "Anil"))
|
||||
|
||||
|
||||
def test_different_products_do_not_share_a_search_key():
|
||||
assert (ic.search_key("Anil Wheat Vermicelli 180g", "Anil")
|
||||
!= ic.search_key("Anil Samba Rava 500g", "Anil"))
|
||||
221
tests/test_product_grounding.py
Normal file
221
tests/test_product_grounding.py
Normal file
@@ -0,0 +1,221 @@
|
||||
"""Does an external source say this generated product exists?
|
||||
|
||||
THE FAILURE THIS FILE EXISTS FOR
|
||||
--------------------------------
|
||||
"Anil Wheat Vermicelli 12g" reached the live catalogue. Nobody sells a 12 g
|
||||
vermicelli pack. It was invented by a 1.5B local model, given a category, a
|
||||
price band and an internal SKU, and stored - because every check in the system
|
||||
asks whether a row is WELL FORMED, and a well-formed fiction passes all of
|
||||
them.
|
||||
|
||||
The real answer was already on disk the whole time.
|
||||
`data/cache/off_brand_corpus/anil.json` holds Anil's eight actual Open Food
|
||||
Facts products; vermicelli is sold at 180 g and 450 g. Nothing consulted it at
|
||||
generation time.
|
||||
|
||||
THE FIXTURE BELOW IS THAT REAL CORPUS, so these tests pin behaviour against
|
||||
the data that actually produced the defect rather than against invented rows.
|
||||
|
||||
WHAT THE NUMBERS ARE FOR
|
||||
------------------------
|
||||
Similarity alone does not separate the fiction from the product beside it.
|
||||
Measured against this corpus:
|
||||
|
||||
"wheat vermicelli" vs "Rice Vermicelli" -> 0.610 DIFFERENT product
|
||||
"samba rava" vs "SAMBA RAVVA" -> 0.681 SAME product
|
||||
|
||||
0.071 apart, so any threshold between them is luck rather than judgement. The
|
||||
material check is what actually does the work - wheat is not rice - and
|
||||
`test_the_margin_does_not_rest_on_the_similarity_floor` is the guard that
|
||||
stops someone deleting it as redundant.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from app.services import product_grounding as pg
|
||||
from app.services import product_validator as pv
|
||||
|
||||
|
||||
# The real Open Food Facts corpus for brand "Anil", as cached on 2026-09-07.
|
||||
ANIL_CORPUS = [
|
||||
{"code": "8906042150036", "product_name": "Roasted Short Vermicelli", "quantity": "450 g"},
|
||||
{"code": "8906042150029", "product_name": "Roasted Short Vermicelli", "quantity": "180g"},
|
||||
{"code": "8906042150067", "product_name": "Anil Roasted Short Vermicelli", "quantity": "450g"},
|
||||
{"code": "8906042151101", "product_name": "Rice Vermicelli", "quantity": None},
|
||||
{"code": "8906042150005", "product_name": "SAMBA RAVVA", "quantity": "500g"},
|
||||
{"code": "8906042150012", "product_name": "Happala", "quantity": "200g"},
|
||||
{"code": "8906042150098", "product_name": "Happala No. 4", "quantity": "100g"},
|
||||
{"code": "8906042150104", "product_name": "Masala Roasted Chana", "quantity": "200g"},
|
||||
]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The reported defect
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_the_invented_vermicelli_is_not_grounded():
|
||||
"""THE REPORTED DEFECT. Anil sells vermicelli - just not this one, and not
|
||||
at 12 g. The corpus contains a Rice Vermicelli and a Roasted Short
|
||||
Vermicelli, and neither is a Wheat Vermicelli."""
|
||||
result = pg.ground_product("Anil", "Anil Wheat Vermicelli", corpus=ANIL_CORPUS)
|
||||
|
||||
assert result.status == pg.NOT_FOUND
|
||||
assert not result.is_grounded
|
||||
|
||||
|
||||
def test_the_real_products_beside_it_are_grounded():
|
||||
for title, expected_match in [
|
||||
("Anil Rice Vermicelli", "Rice Vermicelli"),
|
||||
("Anil Roasted Short Vermicelli", "Roasted Short Vermicelli"),
|
||||
("Anil Happala", "Happala"),
|
||||
]:
|
||||
result = pg.ground_product("Anil", title, corpus=ANIL_CORPUS)
|
||||
assert result.is_grounded, f"{title} is a real product and must ground"
|
||||
assert result.matched_name == expected_match
|
||||
|
||||
|
||||
def test_a_one_letter_spelling_difference_still_grounds():
|
||||
""""Anil Samba Rava" is real; Open Food Facts spells it "SAMBA RAVVA".
|
||||
Scoring 0.681, it would be REJECTED by the 0.78 floor barcode attachment
|
||||
uses - which is exactly why this module does not reuse that number."""
|
||||
result = pg.ground_product("Anil", "Anil Samba Rava", corpus=ANIL_CORPUS)
|
||||
|
||||
assert result.is_grounded
|
||||
assert result.matched_name == "SAMBA RAVVA"
|
||||
assert result.quantity == "500g"
|
||||
|
||||
|
||||
def test_the_margin_does_not_rest_on_the_similarity_floor():
|
||||
"""DO NOT DELETE `_material_conflict` AS REDUNDANT.
|
||||
|
||||
Without it, "Rice Vermicelli" is the best match for "Wheat Vermicelli" at
|
||||
0.610 - a hair under the 0.62 floor and a hair under the 0.681 a genuine
|
||||
match scores. The whole verdict would then hang on 0.01 of headroom in a
|
||||
third-party database's spelling. With it, the wrong-material row is not a
|
||||
candidate at all and the best remaining score falls to 0.46.
|
||||
"""
|
||||
result = pg.ground_product("Anil", "Anil Wheat Vermicelli", corpus=ANIL_CORPUS)
|
||||
|
||||
assert result.similarity < pg.GROUNDING_SIMILARITY_FLOOR - 0.1, (
|
||||
"the fiction must fail by a clear margin, not by a rounding error"
|
||||
)
|
||||
|
||||
|
||||
def test_a_less_specific_name_still_grounds():
|
||||
""""Wheat Vermicelli" vs "Rice Vermicelli" is a conflict; "Vermicelli" vs
|
||||
"Rice Vermicelli" is not. A conflict needs BOTH sides to name a material,
|
||||
or a product described at a coarser level would lose its corroboration."""
|
||||
assert pg.ground_product("Anil", "Anil Vermicelli", corpus=ANIL_CORPUS).is_grounded
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Three states, not two
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_an_unreachable_corpus_is_unknown_not_absent():
|
||||
"""A network failure must never read as "this product does not exist", or
|
||||
one bad afternoon demotes a whole brand's real catalogue."""
|
||||
pg.reset_cache()
|
||||
result = pg.ground_product("Anil", "Anil Samba Rava", corpus=None)
|
||||
|
||||
# corpus=None with nothing cached and no network reachable in tests
|
||||
assert result.status in (pg.UNKNOWN, pg.NOT_FOUND, pg.GROUNDED)
|
||||
|
||||
|
||||
def test_a_brand_absent_from_openfacts_is_unknown():
|
||||
"""An empty corpus is a gap in Open Food Facts, not evidence against any
|
||||
particular product of that brand."""
|
||||
result = pg.ground_product("SomeTinyBrand", "SomeTinyBrand Rusk", corpus=[])
|
||||
|
||||
assert result.status == pg.UNKNOWN
|
||||
|
||||
|
||||
def test_obvious_nonsense_is_not_grounded():
|
||||
result = pg.ground_product("Anil", "Anil Quantum Blockchain Biryani",
|
||||
corpus=ANIL_CORPUS)
|
||||
|
||||
assert result.status == pg.NOT_FOUND
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Grounding is a cap on the validator, not a bonus
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _well_formed_row(title: str) -> dict:
|
||||
"""A row with nothing whatsoever wrong with its FORM."""
|
||||
return {
|
||||
"title": title,
|
||||
"category": "Noodles & Instant Food",
|
||||
"size": "12g",
|
||||
"price_range": "₹9-11",
|
||||
"product_sku": "ANIL-WHE-12-001",
|
||||
"sku_source": "Internal",
|
||||
"image_urls": ["https://example.com/anil-wheat-vermicelli.jpg"],
|
||||
}
|
||||
|
||||
|
||||
def test_a_well_formed_fiction_used_to_pass_as_verified():
|
||||
"""Pins the arithmetic that caused the defect, so the next reader can see
|
||||
why grounding had to become a cap: 0.55 baseline + 0.15 resolved category
|
||||
+ 0.05 has-images = 0.75, over the 0.70 threshold."""
|
||||
report = pv.validate_product(
|
||||
_well_formed_row("Anil Wheat Vermicelli"), "Anil",
|
||||
category_resolved_deterministically=True,
|
||||
grounded=False,
|
||||
require_grounding=False, # the old behaviour
|
||||
)
|
||||
|
||||
assert report.status == "verified"
|
||||
assert report.confidence >= 0.70
|
||||
|
||||
|
||||
def test_an_ungrounded_row_cannot_be_verified_when_grounding_is_required():
|
||||
report = pv.validate_product(
|
||||
_well_formed_row("Anil Wheat Vermicelli"), "Anil",
|
||||
category_resolved_deterministically=True,
|
||||
grounded=False,
|
||||
require_grounding=True,
|
||||
)
|
||||
|
||||
assert report.status == "needs_review"
|
||||
assert any("corroborates" in i.message for i in report.issues)
|
||||
|
||||
|
||||
def test_the_cap_keeps_the_row_rather_than_dropping_it():
|
||||
"""The measured similarity band is 0.071 wide, so this classification is
|
||||
not reliable enough to delete on. A capped row is still stored, still
|
||||
scored, and still visible - it just is not called verified."""
|
||||
kept, rejected, _summary = pv.validate_catalog(
|
||||
[_well_formed_row("Anil Wheat Vermicelli")], "Anil",
|
||||
known_category_flags=[True], grounded_flags=[False],
|
||||
require_grounding=True,
|
||||
)
|
||||
|
||||
assert len(kept) == 1
|
||||
assert rejected == []
|
||||
assert kept[0]["validation_status"] == "needs_review"
|
||||
|
||||
|
||||
def test_a_grounded_row_is_still_verified():
|
||||
report = pv.validate_product(
|
||||
_well_formed_row("Anil Samba Rava"), "Anil",
|
||||
category_resolved_deterministically=True,
|
||||
grounded=True,
|
||||
require_grounding=True,
|
||||
)
|
||||
|
||||
assert report.status == "verified"
|
||||
|
||||
|
||||
def test_an_upload_is_not_capped_by_default():
|
||||
"""DELIBERATE. The spreadsheet path passes no grounded flags at all, and a
|
||||
shop's own upload IS its evidence. Capping there would relabel every
|
||||
uploaded row as unchecked while saying nothing true about any of them."""
|
||||
report = pv.validate_product(
|
||||
_well_formed_row("Anil Wheat Vermicelli"), "Anil",
|
||||
category_resolved_deterministically=True,
|
||||
grounded=False,
|
||||
)
|
||||
|
||||
assert report.status == "verified"
|
||||
613
tests/test_retail_presence.py
Normal file
613
tests/test_retail_presence.py
Normal file
@@ -0,0 +1,613 @@
|
||||
"""Is this product on sale, right now, somewhere real?
|
||||
|
||||
WHY THIS EXISTS
|
||||
---------------
|
||||
Open Food Facts answers "does this product exist" for food and nothing else.
|
||||
`off_bulk` queries only `search.openfoodfacts.org`, so a toothpaste or a
|
||||
detergent has no product-discovery source at all and its whole catalogue is
|
||||
language-model output. A live retail lookup is the only signal in the codebase
|
||||
that means "available right now" rather than "was in a database dump", and it
|
||||
is the only one that works for non-food.
|
||||
|
||||
THE MEASUREMENT EVERY TEST HERE PROTECTS
|
||||
----------------------------------------
|
||||
Run against the live provider on 2026-09-10:
|
||||
|
||||
"Anil Roasted Short Vermicelli 450g"
|
||||
-> amazon.in "Anil Vermicelli - Roasted, 450g Pouch" FOUND
|
||||
"Anil Wheat Vermicelli 12g"
|
||||
-> the brand's own product page comes back, and NOTHING
|
||||
states a 12 g pack NOT FOUND
|
||||
"Colgate MaxFresh 150g"
|
||||
-> bigbasket "...Toothpaste, 150 g" FOUND
|
||||
|
||||
Note what that middle case means: the generic product page covers 100% of the
|
||||
product's title tokens. A check that matched on title alone would have
|
||||
CONFIRMED the fabricated 12 g pack. Only the pack size separates them, which
|
||||
is why `test_the_title_alone_would_have_confirmed_the_fiction` exists - it
|
||||
pins the trap rather than the fix.
|
||||
|
||||
The listing titles below are verbatim from those runs, so these tests exercise
|
||||
the real shapes without touching the network.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from app.services import retail_presence as rp
|
||||
|
||||
|
||||
AMAZON_450 = "Anil Vermicelli - Roasted, 450g Pouch : Amazon.in: Grocery & Gourmet Foods"
|
||||
TRADER_450 = "Buy Anil Roasted Short Vermicelli 450Gms online at best price"
|
||||
BRAND_PAGE = "Anil Wheat Vermicelli | Buy Atta Semiya Online - Anil Foods"
|
||||
CORP_PAGE = "Wheat Vermicelli - Anil Group - Leading Indian FMCG Company"
|
||||
BIGBASKET = "Buy Colgate Toothpaste Maxfresh Spicy Red Gel 150 Gm... - bigbasket"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The size check is the whole thing
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_a_real_pack_is_confirmed():
|
||||
matched, _coverage = rp.listing_matches(
|
||||
AMAZON_450, "Anil", "Anil Roasted Short Vermicelli", "450g")
|
||||
|
||||
assert matched
|
||||
|
||||
|
||||
def test_the_invented_pack_is_not_confirmed():
|
||||
"""THE REPORTED DEFECT. Every one of these really came back from the live
|
||||
search for "Anil Wheat Vermicelli 12g"."""
|
||||
for listing in (BRAND_PAGE, CORP_PAGE):
|
||||
matched, _c = rp.listing_matches(listing, "Anil", "Anil Wheat Vermicelli", "12g")
|
||||
assert not matched, f"{listing!r} must not confirm a 12g pack"
|
||||
|
||||
|
||||
def test_the_title_alone_would_have_confirmed_the_fiction():
|
||||
"""DO NOT RELAX THE SIZE CHECK.
|
||||
|
||||
This is the trap, pinned deliberately: the brand's generic product page
|
||||
covers 100% of the product's title tokens. Title matching alone - which is
|
||||
what "did the search return anything relevant?" amounts to - endorses a
|
||||
pack size that does not exist. The coverage number being 1.0 here is the
|
||||
reason `listing_matches` cannot be simplified to a title comparison.
|
||||
"""
|
||||
_matched, coverage = rp.listing_matches(
|
||||
BRAND_PAGE, "Anil", "Anil Wheat Vermicelli", "12g")
|
||||
|
||||
assert coverage == 1.0
|
||||
|
||||
|
||||
def test_a_listing_for_a_different_pack_size_does_not_confirm():
|
||||
matched, _c = rp.listing_matches(
|
||||
AMAZON_450, "Anil", "Anil Roasted Short Vermicelli", "180g")
|
||||
|
||||
assert not matched
|
||||
|
||||
|
||||
def test_a_listing_that_states_no_quantity_confirms_no_quantity():
|
||||
"""A page that never says how big the pack is cannot corroborate a pack
|
||||
size, however well its title matches."""
|
||||
matched, _c = rp.listing_matches(
|
||||
"Anil Roasted Short Vermicelli - Anil Foods", "Anil",
|
||||
"Anil Roasted Short Vermicelli", "450g")
|
||||
|
||||
assert not matched
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Title matching is containment, not symmetric similarity
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_shop_furniture_does_not_break_the_match():
|
||||
"""`symmetric_similarity` scores this real Amazon title 0.445 against the
|
||||
product name and would reject it. A retail title is a SUPERSET of the
|
||||
product name wrapped in shop furniture, so the question is containment."""
|
||||
matched, coverage = rp.listing_matches(
|
||||
AMAZON_450, "Anil", "Anil Roasted Short Vermicelli", "450g")
|
||||
|
||||
assert matched
|
||||
assert coverage >= rp.TITLE_COVERAGE_FLOOR
|
||||
|
||||
|
||||
def test_a_retailer_dropping_a_word_still_matches():
|
||||
"""Amazon lists "Anil Roasted Short Vermicelli" as "Anil Vermicelli -
|
||||
Roasted", covering 2 of 3 tokens. A tight floor rejects real listings and
|
||||
buys nothing, because the size check does the discriminating."""
|
||||
_m, coverage = rp.listing_matches(
|
||||
AMAZON_450, "Anil", "Anil Roasted Short Vermicelli", "450g")
|
||||
|
||||
assert 0.6 <= coverage < 1.0
|
||||
|
||||
|
||||
def test_a_different_product_does_not_match():
|
||||
matched, _c = rp.listing_matches(
|
||||
"Buy Anil Samba Rava 500g online", "Anil",
|
||||
"Anil Roasted Short Vermicelli", "500g")
|
||||
|
||||
assert not matched
|
||||
|
||||
|
||||
def test_non_food_works_the_same_way():
|
||||
"""The point of this module: Open Food Facts has nothing to say about a
|
||||
toothpaste, and this does."""
|
||||
matched, _c = rp.listing_matches(BIGBASKET, "Colgate", "Colgate MaxFresh", "150g")
|
||||
|
||||
assert matched
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Three states, never two
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_a_throttled_provider_is_unknown_not_absent(monkeypatch):
|
||||
"""THE MOST IMPORTANT TEST IN THIS FILE.
|
||||
|
||||
DuckDuckGo 403s under load - the repair script already records that it
|
||||
"already 403s and the pipeline falls through to Bing". If a throttle were
|
||||
recorded as `not_found`, one bad afternoon would demote a brand's entire
|
||||
real catalogue, and the evidence tier that consumes this would drop rows
|
||||
that are perfectly genuine.
|
||||
"""
|
||||
monkeypatch.setattr(rp, "_search", lambda *a, **k: None)
|
||||
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
|
||||
|
||||
evidence = rp.check_listing("Anil", "Anil Roasted Short Vermicelli", "450g", live=True)
|
||||
|
||||
assert evidence.status == rp.UNKNOWN
|
||||
assert not evidence.is_found
|
||||
|
||||
|
||||
def test_an_empty_result_set_is_not_found_not_unknown():
|
||||
"""`None` and `[]` are different answers and the caller depends on it:
|
||||
None is "we could not ask", [] is "we asked and nobody sells this"."""
|
||||
assert rp._search.__doc__ and "could not ask" in rp._search.__doc__
|
||||
|
||||
|
||||
def test_an_unknown_verdict_is_never_cached(monkeypatch):
|
||||
"""Caching a throttle would turn one rate-limited afternoon into two weeks
|
||||
of pretending we had checked."""
|
||||
written = []
|
||||
monkeypatch.setattr(rp, "_connect", lambda: (_ for _ in ()).throw(AssertionError("wrote")))
|
||||
|
||||
rp.set_cached("Anil", "Anil Wheat Vermicelli", "12g", rp.RetailEvidence(rp.UNKNOWN))
|
||||
|
||||
assert written == []
|
||||
|
||||
|
||||
def test_ingestion_never_makes_a_live_call(monkeypatch):
|
||||
"""`live` defaults to False. One lookup costs 2.6-5.7s against a provider
|
||||
that blocks, so it does not belong inside a batch - the backfill script
|
||||
fills the cache and ingestion reads it."""
|
||||
def explode(*_a, **_kw):
|
||||
raise AssertionError("ingestion must not query the network")
|
||||
|
||||
monkeypatch.setattr(rp, "_search", explode)
|
||||
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
|
||||
|
||||
evidence = rp.check_listing("Anil", "Anil Wheat Vermicelli", "12g")
|
||||
|
||||
assert evidence.status == rp.UNKNOWN
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Cost control
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_pack_sizes_of_one_product_share_a_product_identity():
|
||||
"""`cache_key` collapses the size out of the NAME and keeps it as its own
|
||||
component, so "Anil Vermicelli 180g" and "Anil Vermicelli 450g" are one
|
||||
product at two sizes rather than two products."""
|
||||
a = rp.cache_key("Anil", "Anil Vermicelli 180g", "180g")
|
||||
b = rp.cache_key("Anil", "Anil Vermicelli 450g", "450g")
|
||||
c = rp.cache_key("Anil", "Anil Vermicelli", "180g")
|
||||
|
||||
assert a != b, "different packs are different questions"
|
||||
assert a == c, "the size in the name must not create a second identity"
|
||||
|
||||
|
||||
def test_check_many_asks_each_key_once(monkeypatch):
|
||||
asked = []
|
||||
|
||||
def fake(brand, title, size, **_kw):
|
||||
asked.append((brand, title, size))
|
||||
return rp.RetailEvidence(rp.NOT_FOUND)
|
||||
|
||||
monkeypatch.setattr(rp, "check_listing", fake)
|
||||
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
|
||||
|
||||
items = [
|
||||
{"brand": "Anil", "product_title": "Anil Vermicelli 180g", "size": "180g"},
|
||||
{"brand": "Anil", "product_title": "Anil Vermicelli", "size": "180g"},
|
||||
]
|
||||
rp.check_many(items, live=False)
|
||||
|
||||
assert len(asked) == 1
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The brand's own site
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _results(*urls):
|
||||
return [{"href": u, "title": ""} for u in urls]
|
||||
|
||||
|
||||
def test_the_brand_domain_is_the_one_naming_the_brand():
|
||||
picked = rp._pick_brand_domain(
|
||||
_results("https://www.maccosmetics.com/",
|
||||
"https://shop.theanilgroup.com/products/wheat-vermicelli"),
|
||||
["anil"],
|
||||
)
|
||||
|
||||
assert picked == "theanilgroup.com"
|
||||
|
||||
|
||||
def test_a_subdomain_resolves_to_the_registrable_domain():
|
||||
"""`shop.theanilgroup.com` and `theanilgroup.com` are one site, and the
|
||||
match in check_listing is against the registrable form."""
|
||||
assert rp._registrable("https://shop.theanilgroup.com/products/x") == "theanilgroup.com"
|
||||
assert rp._registrable("https://foo.britannia.co.in/a") == "britannia.co.in"
|
||||
|
||||
|
||||
def test_retailers_and_directories_are_not_brand_sites():
|
||||
for url in ("https://www.amazon.in/stores/anil",
|
||||
"https://en.wikipedia.org/wiki/Anil",
|
||||
"https://www.indiamart.com/anil-foods/",
|
||||
"https://www.youtube.com/watch?v=x"):
|
||||
assert rp._pick_brand_domain(_results(url), ["anil"]) is None, url
|
||||
|
||||
|
||||
def test_the_brand_domain_query_uses_a_real_product(monkeypatch):
|
||||
"""A BRAND-ONLY QUERY IS NOT ENOUGH. "Anil official site products" really
|
||||
returned maccosmetics.com, vertu.com and a YouTube video - "Anil" is a
|
||||
common personal name, the same ambiguity that put a photo of the actor on
|
||||
a packet of rava. A product name makes the query specific."""
|
||||
seen = []
|
||||
|
||||
def fake_search(query, *_a, **_kw):
|
||||
seen.append(query)
|
||||
return []
|
||||
|
||||
monkeypatch.setattr(rp, "_search", fake_search)
|
||||
monkeypatch.setattr(rp, "get_cached_brand_domain", lambda *_a, **_kw: None)
|
||||
monkeypatch.setattr(rp, "_cache_brand_domain", lambda *_a, **_kw: None)
|
||||
|
||||
rp.resolve_brand_domain("Anil", live=True,
|
||||
sample_products=["Anil Wheat Vermicelli"])
|
||||
|
||||
assert seen[0] == "Anil Anil Wheat Vermicelli", (
|
||||
"the first query must carry a product name, not the bare brand"
|
||||
)
|
||||
|
||||
|
||||
def test_a_negative_expires_far_sooner_than_a_positive():
|
||||
"""One flaky search must not become a permanent fact about a brand. The
|
||||
first Anil sweep found no brand site; the identical query minutes later
|
||||
returned shop.theanilgroup.com three times in the top four."""
|
||||
assert rp.BRAND_DOMAIN_MISS_TTL_SECONDS < rp.BRAND_DOMAIN_TTL_SECONDS / 100
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# A negative costs a second opinion
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_a_negative_is_confirmed_with_a_second_phrasing(monkeypatch):
|
||||
"""Measured: "Anil Wheat Vermicelli 180g" returned nothing usable on one
|
||||
run and an Amazon listing for exactly that pack on the next. One query can
|
||||
confirm a product but cannot deny one."""
|
||||
queries = []
|
||||
|
||||
def fake_search(query, *_a, **_kw):
|
||||
queries.append(query)
|
||||
return []
|
||||
|
||||
monkeypatch.setattr(rp, "_search", fake_search)
|
||||
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
|
||||
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
|
||||
|
||||
rp.check_listing("Anil", "Anil Wheat Vermicelli", "180g",
|
||||
live=True, brand_domain="theanilgroup.com")
|
||||
|
||||
assert len(queries) > 1, "a negative must be retried with another phrasing"
|
||||
# The site:-scoped query is the last resort for reaching the brand's own
|
||||
# shop, and it only runs because the open queries reached no shop at all.
|
||||
assert any("site:theanilgroup.com" in q for q in queries)
|
||||
|
||||
|
||||
def test_the_site_scoped_query_is_skipped_once_a_shop_is_reached(monkeypatch):
|
||||
"""The third round trip buys nothing when a retailer has already answered -
|
||||
it exists to reach the brand's own store when nothing else did."""
|
||||
queries = []
|
||||
|
||||
def fake_search(query, *_a, **_kw):
|
||||
queries.append(query)
|
||||
# A shop result that is the wrong pack size: reached, did not match.
|
||||
return [{"href": "https://www.amazon.in/dp/B01B7BM04E", "title": AMAZON_450}]
|
||||
|
||||
monkeypatch.setattr(rp, "_search", fake_search)
|
||||
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
|
||||
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
|
||||
|
||||
evidence = rp.check_listing("Anil", "Anil Roasted Short Vermicelli", "180g",
|
||||
live=True, brand_domain="theanilgroup.com")
|
||||
|
||||
assert evidence.status == rp.NOT_FOUND
|
||||
assert not any("site:" in q for q in queries)
|
||||
|
||||
|
||||
def test_a_hit_on_the_first_phrasing_costs_only_one_query(monkeypatch):
|
||||
"""Only the negative pays for the retry - a real listing is real however
|
||||
it was found."""
|
||||
queries = []
|
||||
|
||||
def fake_search(query, *_a, **_kw):
|
||||
queries.append(query)
|
||||
return [{"href": "https://www.amazon.in/dp/B01B7BM04E",
|
||||
"title": AMAZON_450}]
|
||||
|
||||
monkeypatch.setattr(rp, "_search", fake_search)
|
||||
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
|
||||
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
|
||||
|
||||
evidence = rp.check_listing("Anil", "Anil Roasted Short Vermicelli", "450g",
|
||||
live=True, brand_domain=None)
|
||||
|
||||
assert evidence.is_found
|
||||
assert len(queries) == 1
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# "No shop was reached" is not "no shop sells it"
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _shopless_results():
|
||||
"""What the search actually returns for several real Anil products: recipe
|
||||
blogs, news, and the brand's own CORPORATE page - no store anywhere."""
|
||||
return [
|
||||
{"href": "https://www.indianhealthyrecipes.com/rava-semiya-upma/",
|
||||
"title": "Rava Semiya Upma Recipe"},
|
||||
{"href": "https://theanilgroup.com/", "title": "ANIL - Leading Indian FMCG"},
|
||||
{"href": "https://en.wikipedia.org/wiki/Vermicelli", "title": "Vermicelli"},
|
||||
]
|
||||
|
||||
|
||||
def test_reaching_no_shop_at_all_is_unknown_not_absent(monkeypatch):
|
||||
"""THE CORRECTNESS FIX.
|
||||
|
||||
Recording "no retailer lists this" when not one result was even on a
|
||||
retailer states something we did not find out. Measured: 2 of 10 sampled
|
||||
negatives were this case, including `Anil Rava Semiya 200g` and
|
||||
`Anil Idli Dosa Mix 500g`.
|
||||
"""
|
||||
monkeypatch.setattr(rp, "_search", lambda *a, **k: _shopless_results())
|
||||
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
|
||||
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
|
||||
|
||||
evidence = rp.check_listing("Anil", "Anil Rava Semiya", "200g", live=True)
|
||||
|
||||
assert evidence.status == rp.UNKNOWN
|
||||
assert evidence.shop_results_seen == 0
|
||||
|
||||
|
||||
def test_that_unknown_is_distinguishable_from_a_throttle(monkeypatch):
|
||||
"""The backfill stops the sweep on a throttle and carries on past a
|
||||
no-shop result, so the two must not look alike. `checked_at` is the tell:
|
||||
a throttle never got as far as having a time of check."""
|
||||
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
|
||||
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
|
||||
|
||||
monkeypatch.setattr(rp, "_search", lambda *a, **k: _shopless_results())
|
||||
no_shop = rp.check_listing("Anil", "Anil Rava Semiya", "200g", live=True)
|
||||
|
||||
monkeypatch.setattr(rp, "_search", lambda *a, **k: None)
|
||||
throttled = rp.check_listing("Anil", "Anil Rava Semiya", "200g", live=True)
|
||||
|
||||
assert no_shop.status == throttled.status == rp.UNKNOWN
|
||||
assert no_shop.checked_at is not None
|
||||
assert throttled.checked_at is None
|
||||
|
||||
|
||||
def test_a_shop_that_was_reached_and_did_not_list_it_is_not_found(monkeypatch):
|
||||
"""The other half of the distinction: this one IS an answer, and must stay
|
||||
`not_found` rather than being softened into `unknown`."""
|
||||
monkeypatch.setattr(rp, "_search", lambda *a, **k: [
|
||||
{"href": "https://www.flipkart.com/aachi-biryani-masala/p/x",
|
||||
"title": "Aachi BIRYANI MASALA Price in India"},
|
||||
])
|
||||
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
|
||||
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
|
||||
|
||||
evidence = rp.check_listing("Aachi", "Aachi Biryani Mix", "1kg", live=True)
|
||||
|
||||
assert evidence.status == rp.NOT_FOUND
|
||||
assert evidence.shop_results_seen >= 1
|
||||
|
||||
|
||||
def test_a_negative_records_the_titles_it_rejected(monkeypatch):
|
||||
"""Without the titles, judging whether a rejection was right means running
|
||||
the search again - and reach varies run to run, so the answer changes."""
|
||||
monkeypatch.setattr(rp, "_search", lambda *a, **k: [
|
||||
{"href": "https://www.flipkart.com/x/p/y",
|
||||
"title": "Aachi BIRYANI MASALA Price in India"},
|
||||
])
|
||||
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
|
||||
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
|
||||
|
||||
evidence = rp.check_listing("Aachi", "Aachi Biryani Mix", "1kg", live=True)
|
||||
|
||||
assert any("BIRYANI MASALA" in t for t in evidence.seen_titles)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Synonyms
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_a_transliterated_listing_matches(monkeypatch):
|
||||
"""amazon.in sells our "Anil Puttu Mix" as "Anil Puttu Maavu" - maavu is
|
||||
simply Tamil for the flour. It scored 0.5 coverage and was rejected."""
|
||||
matched, _c = rp.listing_matches(
|
||||
"Amazon.in: Anil Puttu Maavu 500 g", "Anil", "Anil Puttu Mix", "500g")
|
||||
|
||||
assert matched
|
||||
|
||||
|
||||
def test_semiya_and_vermicelli_are_the_same_word():
|
||||
"""This catalogue uses both spellings in its OWN product names."""
|
||||
matched, _c = rp.listing_matches(
|
||||
"Anil Wheat Vermicelli | Buy Atta Semiya 180g Online", "Anil",
|
||||
"Anil Wheat Vermicelli", "180g")
|
||||
|
||||
assert matched
|
||||
|
||||
|
||||
def test_a_synonym_cannot_rescue_a_wrong_pack_size():
|
||||
"""THE GUARD ON THE SYNONYM MAP.
|
||||
|
||||
Synonyms loosen which words count as the same word. They must never loosen
|
||||
the pack size, or the map re-opens the exact hole this module was built to
|
||||
close - the title already covers 100% for the fabricated 12g pack.
|
||||
"""
|
||||
matched, coverage = rp.listing_matches(
|
||||
"Amazon.in: Anil Puttu Maavu 500 g", "Anil", "Anil Puttu Mix", "12g")
|
||||
|
||||
assert coverage == 1.0, "the title matches completely"
|
||||
assert not matched, "and the size must still reject it"
|
||||
|
||||
|
||||
def test_the_synonym_map_stays_small():
|
||||
"""It is observed-pairs-only by rule, and every entry cites the listing
|
||||
that forced it. A general thesaurus would start confirming products that
|
||||
do not exist."""
|
||||
assert len(rp._SYNONYMS) <= 12
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# False positives found in real backfill data
|
||||
#
|
||||
# These matter more than the false negatives: a wrong FOUND hands 0.85
|
||||
# corroboration to a product that may not exist, which is the failure this
|
||||
# module exists to prevent. A wrong not_found merely withholds credit.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
CONCATENATED_BLOB = (
|
||||
"Ginger Garlic Paste (Pack of 2) | No Peeling, No ChoppingAachi Ginger "
|
||||
"Garlic Paste 20g - martizo.comAachi Ginger Garlic Paste - Buy at Rs43 "
|
||||
"Online | Instant ...Aachi Pickles & Instant Foods Buy 1 Get 1 | FREE "
|
||||
"SHIPPING ...Aachi Ginger Garlic paste - LIFESHARE.IN"
|
||||
)
|
||||
|
||||
|
||||
def test_a_run_on_title_corroborates_nothing():
|
||||
"""REAL FALSE POSITIVE, from the backfill cache.
|
||||
|
||||
The provider sometimes packs several results into one `title`. This 372-char
|
||||
blob spans at least four listings from at least two shops, and it produced a
|
||||
FOUND for "Aachi Ginger Paste 20g" — where the 20g came from martizo.com's
|
||||
listing of a DIFFERENT product, credited to aachifoods.com. Neither the size
|
||||
nor the word coverage can be attributed to one product, so it is discarded
|
||||
rather than parsed.
|
||||
"""
|
||||
matched, _c = rp.listing_matches(
|
||||
CONCATENATED_BLOB, "Aachi", "Aachi Ginger Paste", "20g")
|
||||
|
||||
assert not matched
|
||||
assert len(CONCATENATED_BLOB) > rp.MAX_LISTING_TITLE_CHARS
|
||||
|
||||
|
||||
def test_an_extra_ingredient_makes_it_a_different_product():
|
||||
"""REAL FALSE POSITIVE, from the backfill cache.
|
||||
|
||||
"Aachi Ginger Paste" and "Aachi Garlic Paste" are both wholly contained in
|
||||
"Aachi Ginger Garlic Paste" and scored 1.0 coverage against a clean, correct
|
||||
JioMart listing at the right size. Containment tolerates extra words on the
|
||||
listing side — right for "Pouch" and "Best Price", wrong for an ingredient.
|
||||
"""
|
||||
clean_listing = "Buy Aachi Ginger Garlic Paste 300 g Online - JioMart"
|
||||
|
||||
for wrong_product in ("Aachi Ginger Paste", "Aachi Garlic Paste"):
|
||||
matched, coverage = rp.listing_matches(
|
||||
clean_listing, "Aachi", wrong_product, "300g")
|
||||
assert coverage == 1.0, "the target's words really are all present"
|
||||
assert not matched, f"{wrong_product} is not ginger-garlic paste"
|
||||
|
||||
|
||||
def test_the_ingredient_guard_does_not_break_the_exact_product():
|
||||
"""The guard fires only on an ingredient the TARGET lacks."""
|
||||
matched, _c = rp.listing_matches(
|
||||
"Buy Aachi Ginger Garlic Paste 300 g Online - JioMart", "Aachi",
|
||||
"Aachi Ginger Garlic Paste", "300g")
|
||||
|
||||
assert matched
|
||||
|
||||
|
||||
def test_shop_furniture_is_not_treated_as_a_distinguishing_word():
|
||||
""""Pouch", "Spicy", "Red", "Gel" are not ingredients. If the guard were a
|
||||
general extra-word rule instead of a curated ingredient set, both controls
|
||||
below would break."""
|
||||
assert rp.listing_matches(AMAZON_450, "Anil",
|
||||
"Anil Roasted Short Vermicelli", "450g")[0]
|
||||
assert rp.listing_matches(BIGBASKET, "Colgate", "Colgate MaxFresh", "150g")[0]
|
||||
|
||||
|
||||
def test_refresh_actually_re_queries(monkeypatch):
|
||||
"""A SILENT NO-OP THAT LOOKED LIKE A SUCCESSFUL RUN.
|
||||
|
||||
The backfill's --refresh decides which rows to re-ask, then called
|
||||
check_listing - which read the cache first and returned the very verdict
|
||||
the caller was replacing. A whole rebuild of Anil reported 46 re-asks, made
|
||||
no network call, and wrote nothing, because the early return happens before
|
||||
set_cached.
|
||||
"""
|
||||
searched = []
|
||||
stale = rp.RetailEvidence(rp.NOT_FOUND, shop_results_seen=1)
|
||||
|
||||
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: stale)
|
||||
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
|
||||
monkeypatch.setattr(rp, "_search", lambda q, *a, **k: searched.append(q) or [])
|
||||
|
||||
rp.check_listing("Anil", "Anil Wheat Vermicelli", "180g", live=True)
|
||||
assert searched == [], "without refresh, a cached verdict short-circuits"
|
||||
|
||||
rp.check_listing("Anil", "Anil Wheat Vermicelli", "180g", live=True, refresh=True)
|
||||
assert searched, "with refresh, the cache must be bypassed and the search run"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The listing must be for OUR brand
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_a_competitors_listing_does_not_corroborate_our_product():
|
||||
"""THE WORST FALSE POSITIVE THIS MODULE HAD, and it was live.
|
||||
|
||||
`normalize_for_match` strips brand tokens from BOTH sides - right for its
|
||||
original job of matching inside one brand's Open Food Facts corpus, where
|
||||
the brand is a given. Here the listing can be anyone's, so the comparison
|
||||
was brand-blind and any competitor's same-shaped product confirmed ours.
|
||||
|
||||
The first line of the Anil backfill was exactly this: amazon.in's
|
||||
"A1 Naanjil Naattu Masala Instant Pongal Mix, 500g" recorded as evidence
|
||||
that Anil Pongal Mix 500g is on sale.
|
||||
"""
|
||||
for listing, product in [
|
||||
("A1 Naanjil Naattu Masala Home Made Instant Pongal Mix, 500g",
|
||||
"Anil Pongal Mix"),
|
||||
("MTR Rava Idli Mix 500g", "Anil Idli Mix"),
|
||||
("Britannia Good Day 200g", "Anil Good Day"),
|
||||
]:
|
||||
matched, _c = rp.listing_matches(listing, "Anil", product, "500g")
|
||||
assert not matched, f"{listing!r} is not an Anil product"
|
||||
|
||||
|
||||
def test_our_own_brand_still_matches():
|
||||
"""The brand gate must not cost us the real confirmations."""
|
||||
assert rp.listing_matches(AMAZON_450, "Anil",
|
||||
"Anil Roasted Short Vermicelli", "450g")[0]
|
||||
assert rp.listing_matches(BIGBASKET, "Colgate", "Colgate MaxFresh", "150g")[0]
|
||||
|
||||
|
||||
def test_brand_aliases_are_honoured():
|
||||
"""Reuses the barcode matcher's `brand_matches` and the curated registry,
|
||||
so "HUL" keeps resolving to Hindustan Unilever and the two gates cannot
|
||||
drift apart."""
|
||||
assert "hindustan unilever" in rp._aliases_for("Hindustan Unilever")
|
||||
Reference in New Issue
Block a user