From d5a23f645659be6320bf2307461e0458312dfc60 Mon Sep 17 00:00:00 2001 From: sriram Date: Thu, 10 Sep 2026 16:17:28 +0530 Subject: [PATCH] product generation with validation check --- app/core/catalog_engine.py | 124 +++- app/core/store_catalog_pipeline.py | 30 +- app/services/brand_discovery.py | 113 +++- app/services/brand_sync.py | 7 + app/services/enrichment/pipeline.py | 14 +- app/services/image_corroboration.py | 477 +++++++++++++++ app/services/product_grounding.py | 236 ++++++++ app/services/product_validator.py | 56 +- app/services/retail_presence.py | 883 ++++++++++++++++++++++++++++ app/services/title_validator.py | 20 +- app/services/vector_store.py | 57 +- data/cache/image_corroboration.db | Bin 0 -> 12288 bytes data/cache/retail_presence.db | Bin 0 -> 114688 bytes scripts/audit_retail_negatives.py | 111 ++++ scripts/backfill_retail_presence.py | 258 ++++++++ scripts/repair_brand_images.py | 87 +-- tests/test_image_corroboration.py | 195 ++++++ tests/test_product_grounding.py | 221 +++++++ tests/test_retail_presence.py | 613 +++++++++++++++++++ 19 files changed, 3381 insertions(+), 121 deletions(-) create mode 100644 app/services/image_corroboration.py create mode 100644 app/services/product_grounding.py create mode 100644 app/services/retail_presence.py create mode 100644 data/cache/image_corroboration.db create mode 100644 data/cache/retail_presence.db create mode 100644 scripts/audit_retail_negatives.py create mode 100644 scripts/backfill_retail_presence.py create mode 100644 tests/test_image_corroboration.py create mode 100644 tests/test_product_grounding.py create mode 100644 tests/test_retail_presence.py diff --git a/app/core/catalog_engine.py b/app/core/catalog_engine.py index f188b45..1aa3cfb 100644 --- a/app/core/catalog_engine.py +++ b/app/core/catalog_engine.py @@ -18,13 +18,17 @@ import logging sys.path.append(str(Path(__file__).parent.parent)) from app.services.ollama_service import fetch_brand_catalog_with_gemini, fetch_brand_catalog_exhaustive, fetch_product_details -from app.services.image_search import find_all_image_urls, find_product_quantity_openfacts +from app.services.image_search import find_all_image_urls from app.infrastructure.settings import DATA_DIR, USE_OLLAMA from app.services.embeddings_service import embed_texts from app.services.vector_store import ensure_brand_schema, upsert_brand_products, get_existing_product_image_id from app.services.s3_service import s3_service from app.services.brand_registry import resolve_parent_brand +from app.services import image_corroboration from app.services import price_estimator +from app.services import product_grounding +from app.services.enrichment.barcode.sources import off_bulk +from app.services.product_validator import validate_catalog from app.services.category_registry import detect_category_from_text, sanitize_category_language # Configure logging @@ -611,15 +615,43 @@ class ProductCatalogEngine: # Step 2: Enhance each product with comprehensive image search enhanced_products = [] - + # Parallel to enhanced_products: did an external source corroborate + # that each product exists? Collected here and handed to + # validate_catalog at the end - see the Step 3 comment for why this + # path had no validation gate at all until now. + grounded_flags: List[bool] = [] + # Cap products by max_products discovered_products = discovered_products[:max_products] + # Fetched ONCE for the whole brand, not once per product: this is a + # disk-cached whole-catalogue fetch (1-5 requests per brand), which is + # the entire reason it is used here instead of the per-product live + # search that looks like the natural fit. See product_grounding's + # docstring - that endpoint is rate-limited to 10 requests/minute and + # was returning 503 when measured. + try: + brand_corpus = off_bulk.fetch_brand_corpus(brand) + except Exception as e: # noqa: BLE001 - unreachable corpus is not a verdict + logger.warning("Open*Facts corpus unavailable for %s: %s", brand, e) + brand_corpus = None + if brand_corpus is not None: + logger.info("📚 Open*Facts corpus for %s: %d real products", brand, len(brand_corpus)) + total_products_to_process = len(discovered_products) for i, product in enumerate(discovered_products): product_title = product.get('title', '') logger.info(f"🔍 Processing product {i+1}/{total_products_to_process}: {product_title}") - + + # Does anything outside this process say this product exists? + # The LLM that produced `product_title` is a 1.5B local model and + # cannot be asked to check its own work. + grounding = product_grounding.ground_product( + brand, product_title, corpus=brand_corpus + ) + if grounding.status == product_grounding.NOT_FOUND: + logger.warning("⚠️ Ungrounded: %r - %s", product_title, grounding.note()) + # Strip brand prefix from title if present (avoids redundant # "Cadbury Perk Cadbury Perk Crunch" style queries that confuse # image search APIs). @@ -670,7 +702,26 @@ class ProductCatalogEngine: # Select exactly 20 best images all_prioritized = prioritized_images + other_images final_images = self._select_best_images(all_prioritized, product_title, brand, max_images=20) - + + # _select_best_images ORDERS; it does not decide whether any + # candidate is this product. Applied BEFORE the S3 upload below so a + # photo of somebody else is never copied into our own bucket, where + # its origin stops being visible at all. + # + # Nothing is dropped - the whole list is still uploaded and stored, + # so an operator can look. Only the promotion to `image_url` is + # withheld, and only when no candidate corroborates the product. + image_choice = image_corroboration.choose_primary( + final_images, product_title, brand or "" + ) + final_images = list(image_choice.ordered) + primary_eligible = image_choice.primary is not None + if final_images and not primary_eligible: + logger.warning( + "No corroborated image for %r (%s) - storing candidates but " + "leaving image_url empty", product_title, image_choice.reason, + ) + # Check if product already exists in DB to avoid duplicates image_id_val = "" try: @@ -836,18 +887,19 @@ class ProductCatalogEngine: if not _variant_size_price_pairs: default_sizes = price_estimator.default_size_variants(category_value, product_title) - # Ground this in a real packaging size where possible: - # Open Food/Beauty/Products Facts reports an actual - # `quantity` field (e.g. "200 g", "1 l") for products it - # has on file, which is far more trustworthy than the - # category preset list (itself just a fallback for when - # *nothing* else is known). If we have one, swap it in for - # the closest preset rather than presenting a size that may - # not actually exist for this exact product. - try: - real_qty = find_product_quantity_openfacts(product_title, brand) - except Exception: - real_qty = None + # Ground this in a real packaging size where possible. The + # quantity comes from the brand corpus fetched once above, + # NOT from find_product_quantity_openfacts, which issues a + # live per-product query against an endpoint rate-limited to + # 10 requests/minute - see product_grounding's docstring. + # + # ONLY WHEN THE PRODUCT ITSELF IS CORROBORATED. A quantity + # borrowed from a product that merely scored well is worse + # than the preset it replaces: it makes a fabricated row look + # MORE real by dressing it in a size that genuinely exists. + # An ungrounded product keeps the honest category preset and + # is flagged for review instead. + real_qty = grounding.quantity if grounding.is_grounded else None if real_qty and real_qty not in default_sizes: default_sizes = [real_qty] + default_sizes[:2] @@ -915,10 +967,16 @@ class ProductCatalogEngine: # is only used as a last resort since small local models frequently # hallucinate image links that don't actually resolve to an image. all_image_urls = s3_uploaded_urls or final_images - primary_image = all_image_urls[0] if all_image_urls else ( + # `primary_eligible` is the corroboration verdict computed above. + # S3 preserves candidate order (image_000, image_001, ...), so + # index 0 here is the same picture choose_primary judged. + primary_image = (all_image_urls[0] if (all_image_urls and primary_eligible) else None) or ( enriched_img if enriched_img and str(enriched_img).startswith('http') else None ) - if not primary_image and s3_service.enabled and image_id_val: + # Only reachable when there was no usable candidate at all. Guarded + # by `primary_eligible` too, because this constructs a URL to + # image_000 - the very image corroboration just declined to promote. + if not primary_image and primary_eligible and s3_service.enabled and image_id_val: primary_image = s3_service.get_product_image_url(brand, image_id_val) if not all_image_urls and primary_image: all_image_urls = [primary_image] @@ -958,8 +1016,36 @@ class ProductCatalogEngine: } enhanced_products.append(enhanced_product) + grounded_flags.append(grounding.is_grounded) logger.info(f"✅ Enhanced {product_title}: {len(final_images)} images") - + + # Step 2.6: THE DETERMINISTIC VALIDATION GATE. + # + # This path did not have one. `validate_catalog` was called from the + # spreadsheet pipeline and from the Dagster assets, but never from + # here, so POST /api/catalog/generate wrote straight to the brand + # table with nothing between the language model and the database - + # while product_validator's own docstring claimed this call site + # already existed. That claim has been corrected along with this fix. + # + # Rows are annotated and kept, not dropped: `validate_catalog` returns + # "verified" and "needs_review" rows together, and A1 gave the brand + # tables somewhere to record which is which. Only rows scoring below + # the reject threshold are withheld. + enhanced_products, rejected, summary = validate_catalog( + enhanced_products, brand, grounded_flags=grounded_flags, + # These rows came out of a 1.5B language model, so "well formed" + # is not evidence of anything. A row nothing corroborates is kept + # and scored, but capped at needs_review rather than presented as + # verified. + require_grounding=True, + ) + if rejected: + logger.warning( + "🚫 %d of %d generated product(s) failed validation for %s", + len(rejected), summary.get("total_evaluated", 0), brand, + ) + # Step 3: Generate final catalog catalog = { 'brand': brand, diff --git a/app/core/store_catalog_pipeline.py b/app/core/store_catalog_pipeline.py index 51264e6..eb63382 100644 --- a/app/core/store_catalog_pipeline.py +++ b/app/core/store_catalog_pipeline.py @@ -69,6 +69,7 @@ from app.api.routers.user_products import ( read_products_dataframe, row_to_request, ) +from app.services import image_corroboration from app.services import price_estimator from app.services.brand_registry import ( BRAND_ALIASES, @@ -587,8 +588,25 @@ def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, An candidates, product_name, brand, max_images=10 ) if best: - row["image_urls"] = list(best) - row["image_url"] = best[0] + # `_select_best_images` ORDERS candidates; it does not judge + # whether any of them is this product. Its scoring awards points + # for the brand appearing in the URL, so for a brand that is + # also a personal name it actively rewarded the wrong photo - + # `Anil_Kapoor_2019.jpg` outscored everything and became + # `image_url`. choose_primary decides what may be promoted. + # + # The list is still stored in full. A product whose only + # candidates are uncorroborated keeps them for review; it just + # does not get one of them presented as fact. + choice = image_corroboration.choose_primary( + best, product_name, brand or "" + ) + row["image_urls"] = list(choice.ordered) + row["image_url"] = choice.primary + if choice.primary is None: + row.setdefault("_notes", []).append( + f"no primary image: {choice.reason}" + ) except Exception as exc: # noqa: BLE001 - an image is not worth the row # `warning`, not `debug`: at the default log level a debug line is # invisible, so a row that silently lost its images looked identical to @@ -776,6 +794,14 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]: "highlights": list(row.get("highlights") or []), "nutrients": list(row.get("nutrients") or []), "search_query": search_query, + # Stage 10's verdict. validate_catalog() has always annotated these + # three onto every row it kept; this dict dropped all three, so the + # confidence the gate computed died here and every stored row looked + # equally trustworthy. Same rule as the block above: computed but not + # projected is computed for nothing. + "validation_status": row.get("validation_status"), + "confidence_score": row.get("confidence_score"), + "validation_issues": list(row.get("validation_issues") or []), "field_sources": dict(row.get("field_sources") or {}), } diff --git a/app/services/brand_discovery.py b/app/services/brand_discovery.py index 053f135..fcc39ac 100644 --- a/app/services/brand_discovery.py +++ b/app/services/brand_discovery.py @@ -105,7 +105,7 @@ from app.infrastructure.settings import ( BRAND_DISCOVERY_USE_LLM, BRAND_DISCOVERY_USE_OFF, ) -from app.services import active_brands, ollama_service +from app.services import active_brands, ollama_service, retail_presence from app.services.brand_registry import ( get_fssai_license, get_known_sub_brands, @@ -119,6 +119,7 @@ from app.services.category_units import ( parse_unit, ) from app.services.enrichment.barcode.sources import off_bulk +from app.services.product_grounding import GROUNDING_SIMILARITY_FLOOR from app.services.vector_store import _sanitize_name, get_products_by_brand logger = logging.getLogger(__name__) @@ -594,13 +595,51 @@ def _registry_terms(brand: str) -> List[str]: return terms -def _evidence_for(normalised: str, *, off_keys: Dict[str, Any], - registry_terms: Sequence[str], catalog_keys: Dict[str, str]) -> Optional[str]: - """What corroborates this product, cheapest source first. +def _matches_any_key(normalised: str, keys: Dict[str, Any]) -> bool: + """Is `normalised` one of `keys`, allowing for spelling drift? - Three tiers, none of which cost a network call at this point: the OFF index - was built during the sweep, the registry terms are a Python constant, and - the catalog index was read once. + EXACT EQUALITY WAS A BUG. These keys come from Open Food Facts and from + our own stored rows, and neither spells a product the way discovery does. + Measured: our "Anil Samba Rava" normalises to `samba rava`, while Open + Food Facts holds the same product as "SAMBA RAVVA" -> `samba ravva`. Not + equal, so a real product with a real barcode was reported as having no + corroboration at all. + + The similarity floor is the one `product_grounding` measured against this + exact corpus - see that module for why it is not the 0.78 used for barcode + attachment, and why `samba rava`/`samba ravva` (0.681) is the case that + fixes the number. + """ + if not normalised: + return False + if normalised in keys: + return True + for key in keys: + if not key: + continue + if off_bulk.symmetric_similarity(key, normalised) >= GROUNDING_SIMILARITY_FLOOR: + return True + return False + + +def _evidence_for(normalised: str, *, off_keys: Dict[str, Any], + registry_terms: Sequence[str], catalog_keys: Dict[str, str], + retail_keys: Optional[Dict[str, Any]] = None) -> Optional[str]: + """What corroborates this product, strongest source first. + + Four tiers now. None costs a network call at this point: the OFF index was + built during the sweep, the registry terms are a Python constant, the + catalog index was read once, and the retail index is read from a cache the + backfill script populated offline - see retail_presence for why a live + query must never happen inside ingestion. + + `retail` sits directly below `openfacts` and above `catalog` and + `registry`, because it is the only tier that is BOTH external and current. + `catalog` is our own prior output, which is circular if that output was + itself ungrounded, and `registry` is a hardcoded Python list. For a + non-food brand those two were the only tiers available at all, which is + why a toothpaste catalogue could be entirely language-model output and + still look corroborated. Deliberately NOT a tier: "an image search returned a URL naming this product". `catalog_engine._select_best_images` documents that CDN filenames @@ -610,9 +649,11 @@ def _evidence_for(normalised: str, *, off_keys: Dict[str, Any], """ if not normalised: return None - if normalised in off_keys: + if _matches_any_key(normalised, off_keys): return "openfacts" - if normalised in catalog_keys: + if retail_keys and normalised in retail_keys: + return "retail" + if _matches_any_key(normalised, catalog_keys): return "catalog" tokens = set(normalised.split()) if tokens & set(registry_terms): @@ -632,6 +673,25 @@ def _score(sources: Sequence[str], evidence: Optional[str]) -> float: return 1.0 if has_off: return 0.9 + # An Open Food Facts row corroborates this product, but under a DIFFERENT + # SPELLING, so the merge did not fold the two together and this candidate + # never acquired the "off" source. + # + # Unreachable while `_evidence_for` matched keys by exact equality - an + # exact match always merged. It became reachable when that matching was + # relaxed to handle real spelling drift ("Anil Samba Rava" against Open + # Food Facts' "SAMBA RAVVA"), and without this branch such a row fell all + # the way through to 0.25 - scored as though nothing corroborated it, when + # a real database record does. + if evidence == "openfacts": + return 0.85 + # A real shop is currently listing this exact product at this pack size. + # Scored above `catalog` (our own prior output, which is circular when + # that output was itself ungrounded) and far above `registry` (a hardcoded + # list), because it is the only tier that is both external and current. + # For a non-food brand it is usually the ONLY external tier available. + if evidence == "retail": + return 0.85 if evidence == "catalog": return 0.7 if evidence == "registry": @@ -685,7 +745,9 @@ def _resolve_sizes(candidate: Dict[str, Any], title: str, category: str, def _build_product(brand: str, candidate: Dict[str, Any], *, max_sizes: int, catalog_keys: Dict[str, str], off_keys: Dict[str, Any], registry_terms: Sequence[str], - fssai: Optional[str]) -> Optional[DiscoveredProduct]: + fssai: Optional[str], + retail_keys: Optional[Dict[str, Any]] = None, + ) -> Optional[DiscoveredProduct]: raw_title = _canonicalise_title_size((candidate.get("title") or "").strip()) if not raw_title: return None @@ -769,7 +831,8 @@ def _build_product(brand: str, candidate: Dict[str, Any], *, max_sizes: int, shape = {"title": title, "category": category_hint, "size_variants": sizes, "description": description} evidence = _evidence_for(normalised, off_keys=off_keys, - registry_terms=registry_terms, catalog_keys=catalog_keys) + registry_terms=registry_terms, catalog_keys=catalog_keys, + retail_keys=retail_keys) sources = list(dict.fromkeys(candidate.get("sources") or [candidate.get("source")])) sources = [s for s in sources if s] @@ -835,6 +898,7 @@ def discover_brand_products( if key: off_keys.setdefault(key, candidate) + llm_candidates: List[Dict[str, Any]] = [] if use_llm: llm_candidates = _from_llm(brand, deadline=deadline, budget=max_products * 2) @@ -955,6 +1019,32 @@ def discover_brand_products( chosen = _canonicalise_title_size(canonical_titles[key]) entry["title"] = off_bulk.strip_sizes(chosen).strip() or chosen + # What a real shop is currently listing, read from the cache that + # `scripts/backfill_retail_presence.py` fills offline. CACHE ONLY - never a + # live query. Discovery is interactive, and one lookup costs 2.6-5.7s + # against a provider that 403s under load, so a live sweep here would + # either hang the preview or get the whole run throttled - and a throttled + # miss is indistinguishable from "nobody sells this", which is the one + # input a corroboration tier must never be fed. + # + # Built from `merged`, so it covers the LLM's candidates too. That is the + # point: an Open Food Facts row needs no second opinion, and for a NON-FOOD + # brand there is no Open Food Facts row at all - `off_bulk` queries the food + # database alone, so a toothpaste's only possible external corroboration is + # this one. + retail_keys: Dict[str, Any] = {} + for entry in merged: + title = entry.get("title") or "" + key = _normalise_title(brand, title) + if not key or key in retail_keys: + continue + evidence = retail_presence.check_listing(brand, title, "", live=False) + if evidence.is_found: + retail_keys[key] = evidence + if retail_keys: + logger.info("🛒 %d of %d discovered products are currently listed by a retailer", + len(retail_keys), len(merged)) + products: List[DiscoveredProduct] = [] dropped = 0 for candidate in merged: @@ -962,6 +1052,7 @@ def discover_brand_products( brand, candidate, max_sizes=max_sizes_per_product, catalog_keys=catalog_keys, off_keys=off_keys, registry_terms=registry_terms, fssai=fssai, + retail_keys=retail_keys, ) if product is None: continue diff --git a/app/services/brand_sync.py b/app/services/brand_sync.py index a88d1b2..16154c8 100644 --- a/app/services/brand_sync.py +++ b/app/services/brand_sync.py @@ -104,6 +104,13 @@ EXPORT_COLUMNS = ( "barcode_lookup_status", "barcode_last_updated", "gst_percent", "tax_amount", "hsn_gst_needs_review", "highlights", "nutrients", "search_query", "field_sources", + # The validation verdict. Listed here for the same reason the barcode keys + # are: an export that strips them turns a re-seed into a silent downgrade. + # The upsert assigns these three plainly rather than COALESCEing them, so a + # seed file that has lost them clears the verdict on every row it restores - + # which is right for a writer that never validated, and wrong for this one, + # whose whole job is to echo back rows that already were. + "validation_status", "confidence_score", "validation_issues", "nutrition_score", "health_score", "nutrients_per_100g", ) diff --git a/app/services/enrichment/pipeline.py b/app/services/enrichment/pipeline.py index 8189130..ef4a610 100644 --- a/app/services/enrichment/pipeline.py +++ b/app/services/enrichment/pipeline.py @@ -4,10 +4,16 @@ catalog rows, running each stage's per-row work concurrently (bounded by `max_concurrency`) and NEVER letting one row's failure affect any other row or stage. -`catalog_engine.py` calls `run_default_pipeline()` once, between the -deterministic-validation step (product_validator.py) and catalog assembly -- see that file's "Step 2.6" for the call site and -docs/BARCODE_ENRICHMENT.md for the full pipeline-position rationale. +`store_catalog_pipeline.stages_8_9_enrichment()` builds an EnrichmentPipeline +and runs it between SKU resolution and the validation gate; see +docs/BARCODE_ENRICHMENT.md for the pipeline-position rationale. + +This docstring used to say `catalog_engine.py` called `run_default_pipeline()` +at a "Step 2.6". It never did - that module does not import this one at all, +so the brand-name generation path gets no barcode, HSN/GST or content +enrichment. Step 2.6 there is now the validation gate only. Wiring enrichment +into that path is a real and separate piece of work; do not read this comment +as saying it is already done. """ from __future__ import annotations diff --git a/app/services/image_corroboration.py b/app/services/image_corroboration.py new file mode 100644 index 0000000..cf05abf --- /dev/null +++ b/app/services/image_corroboration.py @@ -0,0 +1,477 @@ +""" +Does this image URL actually depict THIS product? + +WHY THIS FILE EXISTS +-------------------- +`image_search.validate_image_url_live` answers "does this URL serve real image +bytes", which is a liveness question. Nothing answered the relevance question, +so a press photo of a person named like the brand passed every check the +ingestion path had: it is a real image, above the byte floor, on a reputable +host. That is how the live catalogue came to illustrate "Anil Samba Rava" with +a photograph of the actor Anil Kapoor. + +The logic here is ported from `scripts/repair_brand_images.py`, which already +knew how to reject that image class - its comments name the failures it was +written for ("Aachi Kulambu Mix" returning a press photo of a politician, +"MTR Dosa Mix" returning an anatomy plate). It lived in a script that imports +settings, brand_registry, produce_reference and four private `vector_store` +symbols including `_connect`, so no ingestion path could import it. Moving the +pure predicates here is what lets stage 6 and the catalog engine use them. + +THE ONE SEMANTIC CHANGE MADE DURING THE MOVE +-------------------------------------------- +The original corroborated a URL against `words + brand_tokens`: + + return any(w in lowered for w in words + _brand_tokens(brand)) + +That works for "Aachi Kulambu Mix" precisely because *Aachi is not a human +name*. Anil is. For product "Anil Samba Rava" under brand "Anil" the token list +contains "anil" twice over, and `Anil_Kapoor_2019.jpg` contains "anil", so the +gate returned True and the celebrity photo was corroborated by the very token +that made it wrong. + +`names_product` here requires a DISTINCTIVE token instead - the title minus the +brand minus the pack size, which is the same set +`catalog_engine._select_best_images` already computes to rank candidates: + + "Anil Samba Rava" -> {samba, rava} -> rejects Anil_Kapoor_2019.jpg + "Anil Wheat Vermicelli"-> {wheat, vermicelli} + +Brand tokens remain, but only as an explicit fallback for titles that have no +distinctive words at all ("Amul 1kg"), and a match found that way is reported +as `via_brand_only` so the caller can decline to treat it as corroboration. + +WHAT THIS MODULE MAY AND MAY NOT DECIDE +--------------------------------------- +It decides which image is shown for a product. It never decides whether the +PRODUCT is real. `brand_discovery._evidence_for` deliberately refuses to treat +image naming as product evidence, because CDN filenames are frequently opaque +hashes and absence of a naming URL is weak evidence of absence. That reasoning +is right and this module does not disturb it: images gate images, never +products. + +Dependencies are stdlib plus `requests` on purpose. `catalog_engine` imports +this module, and `repair_brand_images` already reaches back into +`catalog_engine._select_best_images`, so a module-scope import of +`catalog_engine` here would close a cycle. +""" +from __future__ import annotations + +import logging +import re +import sqlite3 +import threading +import time +from contextlib import closing +from dataclasses import dataclass +from pathlib import Path +from typing import Iterable, List, Optional, Sequence +from urllib.parse import urlparse + +import requests + +logger = logging.getLogger(__name__) + +_BROWSER_UA = "nearle-catalogue/1.0 (product image corroboration)" + +# "own" and "products" are the BUCKET's name, not a brand's. Left in, the token +# "products" corroborates any URL containing /images/products/ or +# /cdn/shop/products/ - which is every Open*Facts and every Shopify path - so +# `names_product` waved through 40+ images that named nothing about the item. +# That is how openbeautyfacts cosmetics photos became the stored image for +# Banana, Orange, Papaya, Guava, Lemon and twenty more. +BUCKET_TOKENS = frozenset({"own", "products", "product"}) + +# A trailing pack size on a product name. Used to collapse "X 100g"/"X 500g" +# onto one search, and to keep size digits out of the distinctive token set. +SIZE_TAIL = re.compile( + r"\s+\d+(?:\.\d+)?\s*(?:g|gm|gms|kg|ml|l|ltr|litre|liter|pcs|pc|n|no|nos)\b\.?\s*$", + re.IGNORECASE, +) + +# A bare quantity token, e.g. "500g" or "2l". Mirrors the filter in +# catalog_engine._select_best_images so the two agree on what a size looks like. +_SIZE_TOKEN = re.compile(r"\d+(?:kg|g|gm|gms|ml|l|ltr|pcs|n)?", re.IGNORECASE) + +# The barcode embedded in an Open*Facts image path, e.g. +# /images/products/890/604/215/0067/front_en.4.400.jpg +_OFF_BARCODE = re.compile(r"/images/products/((?:\d+/)+)") + +# Open*Facts image hosts and the API host that can identify a barcode for each. +# +# The original checked `if "openfoodfacts.org" not in url: return True`, so +# every openbeautyfacts and openproductsfacts image skipped the cross-check +# entirely - exactly the non-food case, and exactly the two sibling databases +# `image_search.OPEN_FACTS_HOSTS` queries. A food product got the check and a +# toothpaste did not. +_OFF_API_HOSTS = { + "openfoodfacts": "world.openfoodfacts.org", + "openbeautyfacts": "world.openbeautyfacts.org", + "openproductsfacts": "world.openproductsfacts.org", +} + +# Hosts whose image paths are content-hashed or barcode-keyed, so a filename +# that fails to name the product says nothing about the photo. These are the +# hosts catalog_engine's own comment is about when it refuses to drop +# uncorroborated candidates. +OPAQUE_PATH_DOMAINS = ( + "bbassets.com", "bigbasket.com", "flixcart.com", "flipkart.com", + "media-amazon.com", "amazon.in", "amazon.com", "jiomart.com", + "zeptonow.com", "blinkit.com", "grofers.com", "cloudinary.com", + "shopifycdn.com", "cdn.shopify.com", "akamaized.net", "cloudfront.net", + "openfoodfacts.org", "openbeautyfacts.org", "openproductsfacts.org", +) + +# Hosts whose filenames are human-authored and descriptive. A filename that +# fails to name the product here is real evidence that the photo is of +# something else - and this is the host family the celebrity photo came from. +DESCRIPTIVE_FILENAME_DOMAINS = ( + "wikimedia.org", "wikipedia.org", "wikimedia.commons", +) + +# Two or more Capitalised words joined by underscores, optionally with a year: +# the Wikimedia Commons house style for a photograph OF A PERSON +# ("Anil_Kapoor_2019.jpg"). Deliberately used only to DEMOTE, never to reject - +# "Britannia_Good_Day.jpg" has the identical shape and is a perfectly good +# product photo, so this signal orders candidates and is not allowed to +# eliminate one. +_PERSON_FILENAME = re.compile( + r"^[A-Z][a-z]+(?:_[A-Z][a-z]+)+(?:_\d{4})?[^/]*\.(?:jpg|jpeg|png|webp)$" +) +_PERSON_CONTEXT = re.compile(r"_at_|_in_\d{4}|portrait|headshot", re.IGNORECASE) + + +# --------------------------------------------------------------------------- +# Tokens +# --------------------------------------------------------------------------- +def brand_tokens(brand: str) -> List[str]: + """Distinctive words of a brand name, bucket words removed.""" + return [ + w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) + if len(w) > 2 and w not in BUCKET_TOKENS + ] + + +def distinctive_tokens(product_name: str, brand: str) -> List[str]: + """Words that separate THIS product from its brand-mates. + + The brand is removed on purpose: every Britannia URL contains "britannia", + so it separates nothing - what tells Marie Gold from Good Day is + "marie"/"gold" vs "good"/"day". Pack sizes and short filler words go for + the same reason. This mirrors catalog_engine._select_best_images so the + ranking and the gate can never disagree about what identifies a product. + """ + brand_words = {w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) if w} + out: List[str] = [] + for word in re.split(r"[^a-z0-9]+", (product_name or "").lower()): + if ( + len(word) > 2 + and word not in brand_words + and word not in BUCKET_TOKENS + and word not in out + and not _SIZE_TOKEN.fullmatch(word) + ): + out.append(word) + return out + + +def search_key(product_name: str, brand: str) -> tuple: + """Collapse a trailing pack size so sizes of one product share a lookup. + + Sharing across sizes is correct rather than merely cheap: it is the same + product in a different pack, and stage 4's size explosion produces exactly + these rows from a single source product. + """ + base = SIZE_TAIL.sub("", product_name or "").strip() + return ((brand or "").lower(), (base or product_name or "").lower()) + + +# --------------------------------------------------------------------------- +# Corroboration +# --------------------------------------------------------------------------- +@dataclass(frozen=True) +class Corroboration: + """Why a URL was or was not accepted as depicting this product.""" + + corroborated: bool + reason: str + via_brand_only: bool = False + + +def corroborate(url: str, product_name: str, brand: str) -> Corroboration: + """Does `url` name this product? + + Distinctive tokens first. Brand tokens are consulted only when the title + has no distinctive words of its own, and a match found that way is flagged + `via_brand_only` - it is the weakest possible signal, and for a brand that + is also a personal name it is the signal that produced the defect this + module exists for. + """ + lowered = (url or "").lower() + if not lowered: + return Corroboration(False, "empty url") + + distinctive = distinctive_tokens(product_name, brand) + if distinctive: + hit = next((t for t in distinctive if t in lowered), None) + if hit: + return Corroboration(True, f"url names {hit!r}") + return Corroboration( + False, + "url names none of " + ", ".join(repr(t) for t in distinctive[:4]), + ) + + # No distinctive words at all (e.g. "Amul 1kg"). Fall back to the brand, + # and say so, so the caller can decide how much that is worth. + tokens = brand_tokens(brand) + hit = next((t for t in tokens if t in lowered), None) + if hit: + return Corroboration(True, f"url names brand {hit!r}", via_brand_only=True) + return Corroboration(False, "url names neither the product nor the brand") + + +def names_product(url: str, product_name: str, brand: str) -> bool: + """Boolean form of `corroborate`, for callers that only need the verdict.""" + return corroborate(url, product_name, brand).corroborated + + +def looks_like_person_photo(url: str) -> bool: + """True when the FILENAME has the shape of a photograph of a person. + + A demotion signal only - see `_PERSON_FILENAME` for why this must never + reject on its own. + """ + name = urlparse(url or "").path.rsplit("/", 1)[-1] + if not name: + return False + return bool(_PERSON_FILENAME.match(name)) or bool(_PERSON_CONTEXT.search(name)) + + +def has_opaque_path(url: str) -> bool: + lowered = (url or "").lower() + return any(d in lowered for d in OPAQUE_PATH_DOMAINS) + + +def has_descriptive_filename(url: str) -> bool: + lowered = (url or "").lower() + return any(d in lowered for d in DESCRIPTIVE_FILENAME_DOMAINS) + + +# --------------------------------------------------------------------------- +# Open*Facts barcode cross-check +# --------------------------------------------------------------------------- +# The barcode is embedded in the image path, so the product is one cheap lookup +# away. If the Open*Facts record does not mention our brand, the image is +# somebody else's product. Nothing else can catch this: the path is opaque +# digits, so no amount of filename matching would help. OFF matches on NAME, +# not brand - asked for "Aachi Pickles" it returned the front-of-pack photo for +# a French "Ducros Green Pitted Olives", which validated perfectly happily +# because it IS a real image. +_DB_PATH = Path("data") / "cache" / "image_corroboration.db" +_lock = threading.Lock() +_initialized = False +_CACHE_TTL_SECONDS = 30 * 24 * 3600 + + +def _connect() -> sqlite3.Connection: + _DB_PATH.parent.mkdir(parents=True, exist_ok=True) + conn = sqlite3.connect(str(_DB_PATH), timeout=10) + conn.execute("PRAGMA journal_mode=WAL") + return conn + + +def _ensure_schema(conn: sqlite3.Connection) -> None: + global _initialized + if _initialized: + return + conn.execute( + """ + CREATE TABLE IF NOT EXISTS openfacts_brand_check ( + cache_key TEXT PRIMARY KEY, + verdict INTEGER NOT NULL, + created_at REAL NOT NULL + ) + """ + ) + conn.commit() + _initialized = True + + +def _cache_get(key: str) -> Optional[bool]: + # `closing`, because sqlite3's own context manager commits the transaction + # and leaves the connection open. Leaked connections are finalized at GC, + # which surfaces as an unraisable exception - and pytest.ini turns warnings + # into errors, so a leak here fails the suite from an unrelated test. + try: + with _lock, closing(_connect()) as conn: + _ensure_schema(conn) + row = conn.execute( + "SELECT verdict, created_at FROM openfacts_brand_check WHERE cache_key = ?", + (key,), + ).fetchone() + except Exception as e: # noqa: BLE001 - a cache failure is a cache miss + logger.debug("image corroboration cache read failed for %s: %s", key, e) + return None + if not row: + return None + verdict, created_at = row + if (time.time() - created_at) > _CACHE_TTL_SECONDS: + return None + return bool(verdict) + + +def _cache_set(key: str, verdict: bool) -> None: + try: + with _lock, closing(_connect()) as conn: + _ensure_schema(conn) + conn.execute( + """ + INSERT INTO openfacts_brand_check (cache_key, verdict, created_at) + VALUES (?, ?, ?) + ON CONFLICT(cache_key) DO UPDATE SET + verdict = excluded.verdict, + created_at = excluded.created_at + """, + (key, 1 if verdict else 0, time.time()), + ) + conn.commit() + except Exception as e: # noqa: BLE001 - best effort only + logger.debug("image corroboration cache write failed for %s: %s", key, e) + + +def _api_host_for(url: str) -> Optional[str]: + lowered = (url or "").lower() + for marker, host in _OFF_API_HOSTS.items(): + if marker in lowered: + return host + return None + + +def openfacts_product_matches_brand(url: str, brand: str, *, timeout: int = 10) -> bool: + """False ONLY when Open*Facts positively says this barcode is another brand. + + Every other outcome - not an Open*Facts URL, no barcode in the path, no + brand tokens, a network failure, an unparseable response - returns True. + A lookup failure must never reject a good image; that asymmetry is the + whole point, and it is the same discipline the realtime checks follow. + """ + api_host = _api_host_for(url) + if not api_host: + return True + match = _OFF_BARCODE.search(url or "") + if not match: + return True + barcode = match.group(1).replace("/", "") + tokens = brand_tokens(brand) + if not tokens: + return True + + key = f"{api_host}:{barcode}:{(brand or '').lower()}" + cached = _cache_get(key) + if cached is not None: + return cached + + verdict = True + try: + resp = requests.get( + f"https://{api_host}/api/v2/product/{barcode}.json", + params={"fields": "brands,product_name"}, + timeout=timeout, + headers={"User-Agent": _BROWSER_UA}, + ) + if resp.ok: + product = (resp.json() or {}).get("product") or {} + haystack = ( + f"{product.get('brands') or ''} {product.get('product_name') or ''}" + ).lower() + if haystack.strip(): + verdict = any(t in haystack for t in tokens) + except Exception: # noqa: BLE001 - a lookup failure must not reject a good image + return True + + _cache_set(key, verdict) + return verdict + + +# --------------------------------------------------------------------------- +# Which candidate may become the product's primary image +# --------------------------------------------------------------------------- +@dataclass +class PrimaryChoice: + """The outcome of choosing a product's `image_url` from its candidates.""" + + primary: Optional[str] + ordered: List[str] + reason: str + + +def choose_primary( + urls: Sequence[str], + product_name: str, + brand: str, + *, + check_openfacts: bool = True, +) -> PrimaryChoice: + """Pick the URL that may become `image_url`, and order the rest. + + Three tiers, because the two existing opinions about failing closed are + both right and the disagreement is domain-scoped: + + 1. The URL names the product. Eligible. + 2. The URL cannot name anything (a hashed retailer CDN path, an + Open*Facts barcode path) - and for Open*Facts, the barcode's own + record agrees about the brand. Eligible: this is the case + catalog_engine's comment protects, where dropping uncorroborated + candidates would leave real products with no image at all. + 3. The URL could have named the product and did not - a human-authored + Commons filename, say. Kept in `image_urls`, never promoted. + + Nothing eligible means `primary is None`. A blank image renders as the + brand monogram, which is honest; another company's product is not, and it + stays invisible until somebody recognises the photo. + """ + tier1: List[str] = [] + tier2: List[str] = [] + tier3: List[str] = [] + + for url in urls: + if not url or not str(url).startswith("http"): + continue + verdict = corroborate(url, product_name, brand) + if verdict.corroborated and not verdict.via_brand_only: + tier1.append(url) + elif verdict.via_brand_only and not looks_like_person_photo(url): + # The title has NO distinctive words of its own - "Godrej 50ml", + # "Lion Dates 100g", "Amul 1kg". There is no product identity to + # match on, so a brand-token match is the best signal that exists + # and withholding the image gains nothing: measured over the seed + # catalogues, treating these as ineligible accounted for 14 of 45 + # withheld primaries, every one of them a brand's own product page. + # + # Person-shaped filenames are the exception, because a brand token + # matching a personal name is the exact defect this module exists + # for and a title with no distinctive words cannot contradict it. + tier2.append(url) + elif has_opaque_path(url) and not has_descriptive_filename(url): + if check_openfacts and not openfacts_product_matches_brand(url, brand): + tier3.append(url) + else: + tier2.append(url) + else: + tier3.append(url) + + # Person-shaped filenames sink within their tier. They are never dropped, + # so a product whose only images look like this still keeps them in + # `image_urls` for a human to review. + tier3.sort(key=looks_like_person_photo) + + ordered = tier1 + tier2 + tier3 + if tier1: + return PrimaryChoice(tier1[0], ordered, "url names the product") + if tier2: + return PrimaryChoice(tier2[0], ordered, "opaque path on a known image host") + return PrimaryChoice( + None, + ordered, + "no candidate names this product; primary withheld rather than guessed", + ) diff --git a/app/services/product_grounding.py b/app/services/product_grounding.py new file mode 100644 index 0000000..d6c4d24 --- /dev/null +++ b/app/services/product_grounding.py @@ -0,0 +1,236 @@ +""" +Does an external source corroborate that this product exists? + +WHY THIS FILE EXISTS +-------------------- +The catalogue contained "Anil Wheat Vermicelli 12g". Nobody sells a 12 g +vermicelli pack. It was generated by a 1.5B local model, given a category, a +price band and an internal SKU, and stored - because every check in the system +asks whether a row is WELL FORMED, and a well-formed fiction passes all of +them. `product_validator`'s own docstring says so: it was built to catch +malformed rows, not false ones. + +Meanwhile the real answer was already on disk. `off_bulk.fetch_brand_corpus` +caches a brand's entire Open Food Facts catalogue, and +`data/cache/off_brand_corpus/anil.json` holds Anil's eight real products - +Roasted Short Vermicelli at 180 g and 450 g, SAMBA RAVVA at 500 g. Nothing +consulted it at generation time. + +WHAT THIS MODULE DOES NOT DO: SUBSTITUTE A SIZE +----------------------------------------------- +The obvious-looking fix - "look up the real quantity and swap it in" - is +wrong here, and measurably so. Matching "Anil Wheat Vermicelli" against the +corpus scores 0.460 against "Roasted Short Vermicelli", far below any usable +floor. That is the corpus saying THIS PRODUCT DOES NOT EXIST, not saying it is +450 g. Swapping in a corpus quantity would replace one fabrication with a +better-dressed one: a product that still does not exist, now wearing a size +that does. + +So a miss reports `not_found` and the caller records that the row is +ungrounded. Removing the row is the validation gate's job, not this module's. + +WHY NOT `find_product_quantity_openfacts` +----------------------------------------- +`image_search.find_product_quantity_openfacts` looks like the right oracle and +is not. It queries `/cgi/search.pl` LIVE, once per product, looping three hosts +at a 12-second timeout - up to 36 s per product against an endpoint Open Food +Facts rate-limits to 10 requests/minute. Two hundred products is twenty-plus +minutes and near-certain throttling, and a throttled miss is indistinguishable +from a real one, which is exactly the input a gate must never be fed. Measured +2026-09-10: that endpoint returned 503 on OFF and 500 on Open Beauty Facts. + +`fetch_brand_corpus` costs one to five requests per BRAND, is disk-cached, +retry-wrapped and paced, and is already populated for 50+ brands. + +THREE STATES, NOT TWO +--------------------- +`grounded` / `not_found` / `unknown`. A corpus that could not be fetched must +report `unknown` and never `not_found`, or a network failure would silently +demote a whole brand's real products. Every lookup path in this codebase that +feeds a gate follows the same discipline. +""" +from __future__ import annotations + +import logging +from dataclasses import dataclass +from typing import Any, Dict, List, Optional, Sequence + +from app.services.enrichment.barcode.sources import off_bulk + +logger = logging.getLogger(__name__) + +# How similar a corpus name must be to count as the same product. +# +# NOT the 0.78 used for barcode attachment. Measured against the real Anil +# corpus on 2026-09-10, best match per target: +# +# "wheat vermicelli" vs "Rice Vermicelli" -> 0.610 (DIFFERENT product) +# "samba rava" vs "SAMBA RAVVA" -> 0.681 (SAME product) +# +# A 0.78 floor rejects a genuine product over a one-letter spelling difference +# in Open Food Facts' own record, so it cannot be used here. But note how +# little room that leaves: 0.071 between a true match and a false one, and the +# floor below sits about 0.01 above the false one. THAT MARGIN IS TOO THIN TO +# REST A DECISION ON, which is why `_material_conflict` exists - the real +# discriminator between those two names is that wheat is not rice, not that +# 0.610 is not 0.681. The floor is the coarse filter; the material check is +# what actually separates the reported defect from the real product beside it. +# +# It is also deliberately a soft signal in both directions: nothing here +# deletes a row, it only decides whether the row can be called corroborated, +# so the cost of being wrong is a review flag and not data loss. +GROUNDING_SIMILARITY_FLOOR = 0.62 + +# Base materials that make two otherwise similarly-named products different +# products. "Wheat Vermicelli" and "Rice Vermicelli" share a head noun and +# score 0.610 against each other; no string-similarity threshold separates +# them reliably, because the thing that differs is one word carrying all the +# meaning. +# +# The rule is deliberately narrow: a conflict is declared ONLY when BOTH names +# name a material and the sets are disjoint. "Wheat Vermicelli" against a bare +# "Vermicelli" is not a conflict - that is plausibly the same product line +# described at two levels of detail, and treating it as a conflict would lose +# real corroboration. +_MATERIALS = frozenset({ + "wheat", "rice", "ragi", "maida", "corn", "millet", "bajra", "jowar", + "atta", "besan", "soya", "oat", "oats", "barley", "quinoa", "almond", + "cashew", "coconut", "groundnut", "peanut", "sesame", "mustard", + "sunflower", "olive", "ghee", "butter", +}) + +GROUNDED = "grounded" +NOT_FOUND = "not_found" +UNKNOWN = "unknown" + +# Per-process memo of brand -> corpus. fetch_brand_corpus is itself disk-cached, +# so this only avoids re-reading and re-parsing the same JSON once per product +# in a 200-row run. +_corpus_memo: Dict[str, Optional[List[Dict[str, Any]]]] = {} + + +@dataclass(frozen=True) +class Grounding: + """What an external source says about one product.""" + + status: str + matched_name: Optional[str] = None + quantity: Optional[str] = None + similarity: float = 0.0 + source: Optional[str] = None + + @property + def is_grounded(self) -> bool: + return self.status == GROUNDED + + def note(self) -> str: + if self.status == GROUNDED: + return ( + f"corroborated by {self.source} as {self.matched_name!r} " + f"(similarity {self.similarity:.2f})" + ) + if self.status == NOT_FOUND: + return f"no {self.source or 'Open Food Facts'} product matches this name" + return "could not check whether this product exists" + + +def _material_conflict(candidate: str, target: str) -> bool: + """True when two names name DIFFERENT base materials. + + See `_MATERIALS`. Only fires when both sides name one, so a more specific + name still corroborates a less specific one. + """ + left = {w for w in candidate.lower().split() if w in _MATERIALS} + right = {w for w in target.lower().split() if w in _MATERIALS} + return bool(left and right and not (left & right)) + + +def _corpus_for(brand: str) -> Optional[List[Dict[str, Any]]]: + """The brand's cached Open*Facts corpus, or None when it is unavailable. + + None means "we do not know", and is what makes the difference between + `not_found` and `unknown` downstream. An EMPTY list is a real answer - the + brand was looked up and has nothing on file. + """ + key = (brand or "").strip().lower() + if not key: + return None + if key in _corpus_memo: + return _corpus_memo[key] + try: + corpus = off_bulk.fetch_brand_corpus(brand) + except Exception as e: # noqa: BLE001 - an unreachable corpus is not a verdict + logger.debug("Corpus fetch failed for %r: %s", brand, e) + corpus = None + _corpus_memo[key] = corpus + return corpus + + +def reset_cache() -> None: + """Drop the per-process corpus memo (tests, and long-lived workers).""" + _corpus_memo.clear() + + +def ground_product(brand: str, product_title: str, + *, corpus: Optional[Sequence[Dict[str, Any]]] = None) -> Grounding: + """Is `product_title` a real product of `brand`, per Open*Facts? + + `corpus` is injectable so a caller processing a whole brand fetches once + and so tests need no network. + """ + title = (product_title or "").strip() + if not title: + return Grounding(UNKNOWN) + + hits = corpus if corpus is not None else _corpus_for(brand) + if hits is None: + return Grounding(UNKNOWN) + if not hits: + # A real answer: this brand has nothing on file at all. That is not + # evidence against any single product, so it is still `unknown` - a + # brand missing from Open Food Facts is a gap in Open Food Facts. + return Grounding(UNKNOWN, source="openfacts") + + drop = off_bulk.brand_tokens(brand) + target = off_bulk.normalize_for_match(title, drop) + if not target: + return Grounding(UNKNOWN, source="openfacts") + + best_score = 0.0 + best_hit: Optional[Dict[str, Any]] = None + for hit in hits: + name = hit.get("product_name") or hit.get("product_name_en") or "" + if not name: + continue + candidate = off_bulk.normalize_for_match(name, drop) + if not candidate: + continue + # Skipped before scoring, not after: a wrong-material name can be the + # single best-scoring row in the corpus ("Rice Vermicelli" is the top + # match for "Wheat Vermicelli" at 0.610), and letting it win would mask + # a genuinely better match further down the list. + if _material_conflict(candidate, target): + continue + score = off_bulk.symmetric_similarity(candidate, target) + if score > best_score: + best_score, best_hit = score, hit + + if best_hit is None or best_score < GROUNDING_SIMILARITY_FLOOR: + return Grounding(NOT_FOUND, similarity=round(best_score, 3), source="openfacts") + + # A variant conflict means the corpus row is a DIFFERENT pack of a + # similarly-named product (sugar-free vs regular, jar vs pouch). Close + # enough to score well, not close enough to corroborate. + name = best_hit.get("product_name") or best_hit.get("product_name_en") or "" + if off_bulk.has_extra_variant_conflict(name, title): + return Grounding(NOT_FOUND, matched_name=name, + similarity=round(best_score, 3), source="openfacts") + + quantity = str(best_hit.get("quantity") or "").strip() or None + return Grounding( + GROUNDED, + matched_name=name, + quantity=quantity, + similarity=round(best_score, 3), + source="openfacts", + ) diff --git a/app/services/product_validator.py b/app/services/product_validator.py index dc27ade..291751a 100644 --- a/app/services/product_validator.py +++ b/app/services/product_validator.py @@ -24,9 +24,15 @@ This module is that final gate. It is intentionally: category-anchored price bands, and sku_service's own SKU format) rather than guessing. False rejections are treated as seriously as false acceptances - see docs/VALIDATION_PIPELINE.md for the tuning rationale. -- ADDITIVE: this module has zero dependents before this change and does not - modify any existing function's behaviour; it is only ever imported and - called from new code (see app/core/catalog_engine.py's call site). +- ADDITIVE: it does not modify any existing function's behaviour; callers opt + in. The call sites are store_catalog_pipeline.stage_10_validate (the + spreadsheet path), orchestration/assets/catalog.py (the Dagster mirror) and + ProductCatalogEngine.generate_catalog's "Step 2.6" (the brand-name path). + + That last one was absent for a long time while this docstring claimed it + existed, so POST /api/catalog/generate wrote LLM output straight to the + brand tables with no gate at all. If you are adding another generation path, + it needs its own call to validate_catalog - nothing here is automatic. USAGE ----- @@ -244,6 +250,7 @@ def validate_product( category_resolved_deterministically: bool = False, grounded: bool = False, images_checked: bool = True, + require_grounding: bool = False, config: Optional[ValidationConfig] = None, ) -> ValidationReport: """Validate one fully-assembled catalog row and return a confidence-scored report. @@ -350,12 +357,14 @@ def validate_product( score -= 0.25 # 4b. Size unit <-> category compatibility (see app/services/ - # category_units.py). This is a last-resort backstop: by the time a - # row reaches this final validator, its size should already have - # been through the generation-time gate (app/services/ - # generation_verifier.py, called from ollama_service.py) AND - # catalog_engine.py's own dimension-unit defence-in-depth pass, both - # of which reject/repair a bad unit long before enrichment. This + # category_units.py). This used to describe itself as a backstop + # behind a generation-time gate in app/services/generation_verifier.py + # "called from ollama_service.py". THAT FILE DOES NOT EXIST IN THIS + # TREE - it belongs to a sibling project - so the layer this comment + # promised was never running here and this check is not a backstop but + # a front line. The real upstream defence is + # category_units.fix_or_reject_size, called from stage 4, plus + # catalog_engine.py's dimension-unit pass. This # check exists purely to catch anything that reached validate_product # some other way (e.g. a pre-existing DB row from before this fix, or # a future caller that doesn't route through catalog_engine). @@ -413,6 +422,33 @@ def validate_product( else: report.status = "verified" + # GROUNDING AS A CAP, NOT A BONUS. + # + # The +0.15 above is why this module could never do the job its docstring + # claims. A fabricated product scores 0.55 baseline + 0.15 resolved + # category + 0.05 has-images = 0.75, clearing the 0.70 threshold and + # landing as "verified" - so "Anil Wheat Vermicelli 12g", a pack size no + # shop has ever sold, was stored as a checked fact. Every check here is + # about FORM, and a well-formed fiction satisfies all of them. + # + # A caller that knows its rows came from a generator rather than from a + # source sets require_grounding. Then a row nothing external corroborates + # can still be kept and still scores normally, but it cannot be called + # verified - which is the honest answer, because nothing verified it. + # + # DELIBERATELY NOT APPLIED BY DEFAULT. The spreadsheet path passes no + # grounded flags at all, and a shop's own upload IS its evidence; capping + # there would relabel every uploaded row as unchecked and say nothing + # true. Only generation paths opt in. + if require_grounding and not grounded and report.status == "verified": + report.status = "needs_review" + report.issues.append(ValidationIssue( + field="grounding", severity="warning", + message="no external source corroborates that this product exists; " + "well-formed but unverified", + penalty=0.0, + )) + return report @@ -423,6 +459,7 @@ def validate_catalog( known_category_flags: Optional[list[bool]] = None, grounded_flags: Optional[list[bool]] = None, images_checked: bool = True, + require_grounding: bool = False, config: Optional[ValidationConfig] = None, ) -> tuple[list[dict[str, Any]], list[dict[str, Any]], dict[str, Any]]: """Validate an entire list of assembled catalog rows in one pass. @@ -450,6 +487,7 @@ def validate_catalog( category_resolved_deterministically=det, grounded=grd, images_checked=images_checked, + require_grounding=require_grounding, config=cfg, ) status_counts[report.status] = status_counts.get(report.status, 0) + 1 diff --git a/app/services/retail_presence.py b/app/services/retail_presence.py new file mode 100644 index 0000000..0c31d02 --- /dev/null +++ b/app/services/retail_presence.py @@ -0,0 +1,883 @@ +""" +Is this product on sale, right now, somewhere real? + +WHY THIS FILE EXISTS +-------------------- +Open Food Facts answers "does this product exist" for food, and answers it +well. It does not answer it for anything else: `off_bulk` queries only +`search.openfoodfacts.org`, so a toothpaste or a detergent has no +product-discovery source at all and its entire catalogue is language-model +output. Measured 2026-09-10, India-tagged coverage on the sibling databases is +2-13 products per non-food brand against Britannia's 218 on OFF - real, but +nowhere near enough to build a catalogue from. + +A live retail lookup answers it for everything, food and non-food alike, and +it is the only signal in this codebase that means "available in real time" +rather than "was in a database dump". + +THE ONE THING THAT MAKES THIS WORK: MATCH THE PACK SIZE +-------------------------------------------------------- +Measured against the live provider on 2026-09-10: + + "Anil Roasted Short Vermicelli 450g" + -> amazon.in "Anil Vermicelli - Roasted, 450g Pouch" CONFIRMED + + "Anil Wheat Vermicelli 12g" + -> results come back, but every one is the generic product + page. NOTHING confirms a 12 g pack. NOT CONFIRMED + +A check that asks "did the search return anything?" would have confirmed the +12 g pack, which is the exact fabrication this module exists to catch. So a +hit counts only when the listing's TITLE and its PACK SIZE both match. + +That is also why `sku_service.find_website_product_id` cannot be reused as +evidence: it regexes a product ID out of the result URL and never reads the +listing title or its size at all. It answers "is there an Amazon page in these +results", which for any real brand is always yes. + +THREE STATES, NEVER TWO +----------------------- +`found` / `not_found` / `unknown`. DuckDuckGo throttles and 403s - the repair +script already records that it "already 403s and the pipeline falls through to +Bing". A throttled lookup MUST report `unknown`, because a gate fed +`not_found` on a rate-limit would quietly demote a brand's entire real +catalogue on a bad afternoon. + +COST, AND WHY THE STAGE READS ONLY THE CACHE +--------------------------------------------- +There is no brand-scoped retail endpoint the way there is for Open Food Facts, +so this is irreducibly one network call per product against a provider that +blocks. Measured 2.6-5.7 s per query. + +So the work is split: `backfill_retail_presence.py` does the querying offline, +paced and cached, and the ingestion-time stage reads the cache ONLY +(`live=False`). Ingestion never blocks on a live query and never trips a rate +limit mid-batch. `search_key` collapses pack sizes onto one lookup per +PRODUCT rather than per row, which is a 3-5x cut for free. +""" +from __future__ import annotations + +import json +import logging +import re +import sqlite3 +import threading +import time +from contextlib import closing +from dataclasses import asdict, dataclass, field +from pathlib import Path +from typing import Any, Dict, List, Optional, Sequence +from urllib.parse import urlparse + +from app.services import quantity_utils +from app.services.brand_registry import BRAND_ALIASES +from app.services.enrichment.barcode.matching import brand_matches +from app.services.enrichment.barcode.sources import off_bulk +from app.services.image_corroboration import search_key + +logger = logging.getLogger(__name__) + +FOUND = "found" +NOT_FOUND = "not_found" +UNKNOWN = "unknown" + +# Indian grocery/marketplace domains a real FMCG product is listed on. +# `sku_service._MARKETPLACE_PATTERNS` covers the same ground for ID extraction; +# this list is about presence, so it needs no per-site URL grammar. +RETAILER_DOMAINS = ( + "amazon.in", "flipkart.com", "bigbasket.com", "jiomart.com", + "blinkit.com", "zeptonow.com", "dmart.in", "swiggy.com", + "netmeds.com", "pharmeasy.in", "nykaa.com", "1mg.com", + "licious.in", "starquik.com", "spencers.in", "moreretail.in", +) + +# What fraction of the PRODUCT's distinctive words must appear in the listing +# title. See `listing_matches` for why this is a containment ratio and not the +# 0.85 symmetric similarity the barcode matcher uses. +# +# 0.6 is deliberately loose. Retailers drop words freely - Amazon lists "Anil +# Roasted Short Vermicelli" as "Anil Vermicelli - Roasted", covering 2 of 3 +# tokens (0.67) - so a tight floor here rejects real listings while adding +# nothing, because it is the pack-size check below that separates a real pack +# from an invented one. This gate only has to establish that the listing is +# about the right product family. +TITLE_COVERAGE_FLOOR = 0.6 + +# Marketing boilerplate that surrounds a product name in a retail listing +# title. Removed before matching so "Buy Anil Roasted Vermicelli 450g Online +# at Best Price" reduces to the product. +_LISTING_NOISE = re.compile( + r"\b(buy|online|at|best|price|lowest|offers?|deals?|free|delivery|shop|" + r"order|now|india|in|from|upto|off|save|get|com|grocery|gourmet|foods?)\b", + re.IGNORECASE, +) + +# Words that name the SAME thing in a retail listing and in our catalogue. +# +# Indian FMCG listings mix English and transliterated Tamil/Hindi freely, and +# our own product names do too. Measured misses: our "Anil Puttu Mix" is on +# amazon.in as "Anil Puttu MAAVU" (maavu = flour/mix) and was rejected at 0.5 +# coverage; this catalogue also uses "Semiya" and "Vermicelli" interchangeably +# in its own product names, so the two spellings fail to corroborate each other. +# +# STRICTLY OBSERVED, NEVER INVENTED - the same rule product_grounding._MATERIALS +# follows. This must not grow into a general thesaurus: every pair added makes +# the matcher more willing to confirm a product, and confirming products that do +# not exist is the failure this whole module exists to prevent. Add a pair only +# after seeing it reject a real listing. +# +# THE PACK SIZE IS NOT AFFECTED. Synonyms loosen only which words count as the +# same word; `quantities_match` still has to agree, so no synonym can rescue a +# size that nobody sells. +# Each pair below cites the listing that forced it. Anything without a citation +# does not belong here. +_SYNONYMS: Dict[str, str] = { + # amazon.in "Anil Puttu Maavu" vs our "Anil Puttu Mix", and + # theanilgroup.com "Arisi Puttu Maavu" for the same product. + "maavu": "flour", + "mix": "flour", + # This catalogue's own names: "Anil Rava Semiya", "Anil Ragi Semiya" and + # "Anil Wheat Vermicelli" are the same product family spelled two ways. + "semiya": "vermicelli", + # Open Food Facts holds our "Anil Samba Rava" as "SAMBA RAVVA". + "ravva": "rava", + # theanilgroup.com sells our "Wheat Vermicelli" as "Atta Semiya". + "atta": "wheat", +} + + +def _aliases_for(brand: str) -> List[str]: + """Other names this brand trades under, from the curated registry. + + `BRAND_ALIASES` maps alias -> parent, so the aliases OF a brand are the + keys pointing at it, plus the parent name itself. + """ + key = (brand or "").strip().lower() + if not key: + return [] + out = {key} + parent = BRAND_ALIASES.get(key) + if parent: + out.add(parent.lower()) + for alias, target in BRAND_ALIASES.items(): + if target.lower() == key or (parent and target.lower() == parent.lower()): + out.add(alias.lower()) + return sorted(out) + + +def _canonical_words(text: str) -> set: + """Token set with observed synonyms folded onto one spelling.""" + return {_SYNONYMS.get(w, w) for w in (text or "").split()} + + +# A real product listing title is short. Anything much longer is the provider +# running several results together into one `title` field, e.g. +# +# "Ginger Garlic Paste (Pack of 2) | No Peeling, No ChoppingAachi Ginger +# Garlic Paste 20g - martizo.comAachi Ginger Garlic Paste - Buy at Rs43..." +# +# 372 characters, at least four different listings, from at least two shops. +# Matching against that blob is meaningless in both halves: the quantity comes +# from whichever listing happened to state one, and the word coverage is +# satisfied by words scattered across products we never asked about. It +# produced a FOUND for "Aachi Ginger Paste 20g" whose 20g came from a different +# retailer's listing of a different product, attributed to aachifoods.com. +# +# A false positive is the worse failure here - it hands 0.85 corroboration to a +# product that may not exist, which is what this module was built to prevent - +# so a blob is discarded rather than parsed. +MAX_LISTING_TITLE_CHARS = 140 + +# Ingredient words that make two similarly-named products DIFFERENT products. +# +# Coverage is deliberately containment - the target's words must appear in the +# listing - so a listing is free to carry extra words, which is right for shop +# furniture ("Pouch", "Best Price", "Amazon.in"). It is wrong for an ingredient: +# "Aachi Ginger Paste" is entirely contained in "Aachi Ginger GARLIC Paste" and +# scored 1.0 coverage against it, and ginger paste is not ginger-garlic paste. +# +# Same discipline as product_grounding._MATERIALS and _SYNONYMS: these are +# words observed causing a wrong match, not a general ingredient list. +_DISTINGUISHING_INGREDIENTS = frozenset({ + "garlic", "ginger", "chilli", "chili", "pepper", "tamarind", "coriander", + "cumin", "turmeric", "mint", "lemon", "tomato", "onion", "mango", + "coconut", "mustard", "fenugreek", "clove", "cardamom", "cinnamon", +}) + + +_DB_PATH = Path("data") / "cache" / "retail_presence.db" +_lock = threading.Lock() +_initialized = False +DEFAULT_TTL_SECONDS = 14 * 24 * 3600 +# A brand's own domain changes far more rarely than its stock does. +BRAND_DOMAIN_TTL_SECONDS = 180 * 24 * 3600 +# A miss is re-asked the next day - see get_cached_brand_domain for why. +BRAND_DOMAIN_MISS_TTL_SECONDS = 24 * 3600 + +# Search endpoints are the strictly-limited ones. Same figure off_bulk uses. +PAUSE_SECONDS = 2.0 + + +@dataclass +class RetailEvidence: + """What a live retail search says about one (brand, product, size).""" + + status: str + retailer: Optional[str] = None + url: Optional[str] = None + matched_title: Optional[str] = None + matched_size: Optional[str] = None + similarity: float = 0.0 + checked_at: Optional[float] = None + # How many results were actually on a shop. This is what makes a negative + # auditable: without it, the only way to ask "did we even reach a retailer?" + # is to run the search again, and the provider's reach varies run to run + # (the same query returned 7 shop results and then 0 minutes later). + shop_results_seen: int = 0 + # The shop titles that came back and did NOT match, kept so a human can + # judge whether the rejection was right. "Aachi Biryani Masala" against + # "Aachi Biryani Mix" is a correct rejection; "Anil Puttu Maavu" against + # "Anil Puttu Mix" is a matcher failure, and nothing but the title tells + # them apart. + seen_titles: List[str] = field(default_factory=list) + query: Optional[str] = None + + @property + def is_found(self) -> bool: + return self.status == FOUND + + def note(self) -> str: + if self.status == FOUND: + return f"listed by {self.retailer} as {self.matched_title!r}" + if self.status == NOT_FOUND: + return (f"{self.shop_results_seen} shop listing(s) seen, none for " + f"this product at this pack size") + if self.shop_results_seen == 0 and self.checked_at: + return "searched, but no shop listing was reached at all" + return "could not check retail availability" + + +# --------------------------------------------------------------------------- +# Cache +# --------------------------------------------------------------------------- +def _connect() -> sqlite3.Connection: + _DB_PATH.parent.mkdir(parents=True, exist_ok=True) + conn = sqlite3.connect(str(_DB_PATH), timeout=10) + conn.execute("PRAGMA journal_mode=WAL") + return conn + + +def _ensure_schema(conn: sqlite3.Connection) -> None: + global _initialized + if _initialized: + return + conn.execute( + """ + CREATE TABLE IF NOT EXISTS retail_presence_cache ( + cache_key TEXT PRIMARY KEY, + brand TEXT, + product TEXT, + size TEXT, + result_json TEXT NOT NULL, + created_at REAL NOT NULL + ) + """ + ) + # One row per brand, because a brand's own domain is a fact about the + # brand and costs a search to learn. Resolved once and reused across every + # product, so it adds one query per brand rather than one per product. + conn.execute( + """ + CREATE TABLE IF NOT EXISTS brand_domain_cache ( + brand TEXT PRIMARY KEY, + domain TEXT, + created_at REAL NOT NULL + ) + """ + ) + conn.commit() + _initialized = True + + +def cache_key(brand: str, product_title: str, size: str) -> str: + """One key per PRODUCT+SIZE, with the product name size-collapsed first. + + `search_key` strips a trailing pack size from the name so + "Anil Vermicelli 180g" and "Anil Vermicelli 450g" share a product identity; + the size is then a separate component, because the size is precisely what + is being verified. + """ + brand_key, product_key = search_key(product_title, brand) + size_key = re.sub(r"\s+", "", (size or "").strip().lower()) + return f"{brand_key}|{product_key}|{size_key}" + + +def get_cached(brand: str, product_title: str, size: str, + ttl_seconds: float = DEFAULT_TTL_SECONDS) -> Optional[RetailEvidence]: + """A cached verdict, or None for a miss. Never raises.""" + key = cache_key(brand, product_title, size) + try: + with _lock, closing(_connect()) as conn: + _ensure_schema(conn) + row = conn.execute( + "SELECT result_json, created_at FROM retail_presence_cache WHERE cache_key = ?", + (key,), + ).fetchone() + except Exception as e: # noqa: BLE001 - a cache failure is a cache miss + logger.debug("Retail cache read failed for %s: %s", key, e) + return None + if not row: + return None + result_json, created_at = row + if ttl_seconds > 0 and (time.time() - created_at) > ttl_seconds: + return None + try: + return RetailEvidence(**json.loads(result_json)) + except Exception: # noqa: BLE001 - an unreadable entry is a miss + return None + + +def set_cached(brand: str, product_title: str, size: str, + evidence: RetailEvidence) -> None: + """Best-effort write. A failure just means no caching next time. + + An `unknown` verdict is NOT cached: it records that we failed to ask, not + an answer, and caching it would turn one rate-limited afternoon into two + weeks of pretending we had checked. + """ + if evidence.status == UNKNOWN: + return + key = cache_key(brand, product_title, size) + try: + with _lock, closing(_connect()) as conn: + _ensure_schema(conn) + conn.execute( + """ + INSERT INTO retail_presence_cache + (cache_key, brand, product, size, result_json, created_at) + VALUES (?, ?, ?, ?, ?, ?) + ON CONFLICT(cache_key) DO UPDATE SET + result_json = excluded.result_json, + created_at = excluded.created_at + """, + (key, brand, product_title, size, + json.dumps(asdict(evidence)), time.time()), + ) + conn.commit() + except Exception as e: # noqa: BLE001 - best effort only + logger.debug("Retail cache write failed for %s: %s", key, e) + + +# --------------------------------------------------------------------------- +# Matching +# --------------------------------------------------------------------------- +# Domains that are never a brand's own site, however often they rank for its +# name. Retailers are excluded separately via RETAILER_DOMAINS. +_NOT_A_BRAND_SITE = ( + "wikipedia.org", "wikimedia.org", "facebook.com", "instagram.com", + "linkedin.com", "youtube.com", "twitter.com", "x.com", "pinterest.com", + "indiamart.com", "justdial.com", "tradeindia.com", "exportersindia.com", + "zaubacorp.com", "tofler.in", "crunchbase.com", "glassdoor.com", + "google.com", "blogspot.com", "wordpress.com", "medium.com", + "quora.com", "reddit.com", "yelp.com", "tripadvisor.com", +) + + +def _registrable(url: str) -> str: + """The registrable-ish domain of a URL ("shop.theanilgroup.com" -> + "theanilgroup.com"). Good enough for matching, not a public-suffix parser.""" + host = urlparse(url if "//" in url else f"//{url}").netloc.lower() + host = host.split(":")[0] + if host.startswith("www."): + host = host[4:] + parts = host.split(".") + if len(parts) > 2 and parts[-2] in ("co", "com", "net", "org", "gov", "ac"): + return ".".join(parts[-3:]) # theanilgroup.co.in + if len(parts) > 2: + return ".".join(parts[-2:]) # shop.theanilgroup.com -> theanilgroup.com + return host + + +def get_cached_brand_domain(brand: str) -> Optional[str]: + """Cached brand domain, or None for a miss. Never raises. + + An empty string is a real cached answer meaning "looked and found none"; + it comes back as None to callers but stops the lookup being repeated. + """ + key = _normalize_brand(brand) + try: + with _lock, closing(_connect()) as conn: + _ensure_schema(conn) + row = conn.execute( + "SELECT domain, created_at FROM brand_domain_cache WHERE brand = ?", + (key,), + ).fetchone() + except Exception: # noqa: BLE001 - a cache failure is a cache miss + return None + if not row: + return None + domain, created_at = row + # A NEGATIVE EXPIRES FAST; A POSITIVE LASTS. + # + # These searches fan out across several engines and the result set varies + # run to run: the first Anil sweep found no brand site, and a repeat of the + # identical query minutes later returned shop.theanilgroup.com three times + # in the top four. Caching that miss for six months would have made one + # flaky search a permanent fact about the brand, and every product of + # Anil's would have kept reporting "sold nowhere". + # + # A found domain is stable and worth keeping; a miss is worth re-asking + # tomorrow. + ttl = BRAND_DOMAIN_TTL_SECONDS if domain else BRAND_DOMAIN_MISS_TTL_SECONDS + if (time.time() - created_at) > ttl: + return None + return domain or None + + +def _normalize_brand(brand: str) -> str: + return re.sub(r"\s+", " ", (brand or "").strip().lower()) + + +def resolve_brand_domain(brand: str, *, live: bool = False, + sample_products: Optional[Sequence[str]] = None, + timeout: int = 20) -> Optional[str]: + """The brand's own website, e.g. "Anil" -> "theanilgroup.com". + + WHY THIS EXISTS. The first backfill run asked only the retailer list and + reported "Anil Wheat Vermicelli 180g" as not listed anywhere - while the + brand's own shop sells exactly that pack. A manufacturer's catalogue is + real evidence that a product and its pack size exist, and leaving it out + turned "no big retailer stocks this" into "this product is not real", + which are very different claims. + + A brand site is WEAKER evidence than a retailer: a manufacturer may list a + discontinued or not-yet-shipping line. The verdict records which domain + answered, so a consumer can weigh them differently - see `RetailEvidence. + retailer`. + + Identified by a domain that CONTAINS a brand token and is neither a + retailer nor a directory/social site. That is a deliberately conservative + rule: it will miss a brand whose site is named nothing like the brand, and + a miss just means we fall back to the retailer list. + """ + key = _normalize_brand(brand) + if not key: + return None + + cached = get_cached_brand_domain(brand) + if cached is not None: + return cached + if not live: + return None + + tokens = [t for t in off_bulk.brand_tokens(brand) if len(t) > 2] + if not tokens: + return None + + # QUERY WITH A REAL PRODUCT NAME, NOT THE BRAND ALONE. + # + # "Anil official site products" returned maccosmetics.com, vertu.com and + # a YouTube video: "Anil" is a common personal name, so a brand-only query + # is dominated by cricketers and airlines - the same ambiguity that put a + # photo of Anil Kapoor on a packet of rava. Adding a product the brand + # actually sells makes the query specific enough to find the real site; + # "Anil Wheat Vermicelli" surfaces shop.theanilgroup.com immediately. + # + # Several phrasings, because one search is demonstrably not enough: the + # identical query minutes apart returned the brand site three times and + # then not at all. Stops at the first that yields a domain. + phrasings: List[str] = [f"{brand} {p}" for p in (sample_products or [])[:2]] + phrasings += [f"{brand} official site products", + f"{brand} official website India"] + + found: Optional[str] = None + asked = False + for phrasing in phrasings: + if found: + break + results = _search(phrasing, 10, timeout) + if results is None: + continue + asked = True + found = _pick_brand_domain(results, tokens) + + if not asked: + return None # could not ask - do NOT cache a failure as "none" + + _cache_brand_domain(key, found or "") + return found + + +def _pick_brand_domain(results, tokens) -> Optional[str]: + """The first result whose domain stem contains a brand token and is not a + retailer, directory or social site.""" + for result in results: + url = str(result.get("href") or result.get("url") or "") + if not url.startswith("http"): + continue + domain = _registrable(url) + if not domain: + continue + if any(bad in domain for bad in _NOT_A_BRAND_SITE): + continue + if any(retailer in domain for retailer in RETAILER_DOMAINS): + continue + stem = domain.split(".")[0] + if any(token in stem for token in tokens): + return domain + return None + + +def _cache_brand_domain(key: str, domain: str) -> None: + try: + with _lock, closing(_connect()) as conn: + _ensure_schema(conn) + conn.execute( + """ + INSERT INTO brand_domain_cache (brand, domain, created_at) + VALUES (?, ?, ?) + ON CONFLICT(brand) DO UPDATE SET + domain = excluded.domain, created_at = excluded.created_at + """, + (key, domain, time.time()), + ) + conn.commit() + except Exception as e: # noqa: BLE001 - best effort only + logger.debug("Brand domain cache write failed for %s: %s", key, e) + + +def _is_product_page(url: str) -> bool: + """Does this URL point at a specific product rather than a landing page? + + Only applied to the BRAND's own domain. A retailer domain is a shop by + definition, but a manufacturer's site is mostly corporate: `theanilgroup.com/` + is an "about us" homepage and `shop.theanilgroup.com/products/wheat-vermicelli` + is a listing, and both share a registrable domain. + + Counting the homepage as "a shop was reached" is not harmless - it is what + decides whether a miss is reported as `not_found` (an answer) or `unknown` + (we never asked a shop). The corporate homepage ranks for almost every + query about the brand, so without this every no-shop verdict would be + dressed up as a real absence. + """ + path = urlparse(url or "").path.strip("/") + if not path: + return False + return any( + marker in f"/{path.lower()}/" + for marker in ("/product", "/products/", "/shop/", "/collections/", + "/p/", "/pd/", "/item", "/buy", "/store/") + ) or path.count("/") >= 1 + + +def _retailer_for(url: str) -> Optional[str]: + lowered = (url or "").lower() + for domain in RETAILER_DOMAINS: + if domain in lowered: + return domain + return None + + +def _clean_listing_title(title: str) -> str: + return _LISTING_NOISE.sub(" ", title or "") + + +def listing_matches(listing_title: str, brand: str, product_title: str, + size: str) -> tuple: + """Does this listing describe THIS product at THIS pack size? + + Returns (matches, coverage). Both halves are required - see the module + docstring for the measurement that makes the size half non-negotiable. + + COVERAGE, NOT SYMMETRIC SIMILARITY. `off_bulk.symmetric_similarity` is + built to compare one database record with another and penalises extra + tokens on either side. A retail listing title is not a record: it is a + SUPERSET of the product name wrapped in shop furniture. Measured against + real titles the live provider returned, symmetric similarity scored + + "Anil Vermicelli - Roasted, 450g Pouch : Amazon.in: Grocery & + Gourmet Foods" vs "Anil Roasted Short Vermicelli" + -> 0.445, rejected + + which is a genuine listing of exactly the product asked for. So the title + half asks the containment question instead - how much of the PRODUCT's + identity appears in the listing - and the pack size does the discriminating. + """ + if not listing_title: + return False, 0.0 + + # Several listings run together into one title field - see + # MAX_LISTING_TITLE_CHARS. Neither the size nor the words can be attributed + # to a single product, so this corroborates nothing. + if len(listing_title) > MAX_LISTING_TITLE_CHARS: + return False, 0.0 + + # THE LISTING MUST BE FOR OUR BRAND. + # + # This was missing entirely, and it is the worst hole this module has had. + # `normalize_for_match` strips brand tokens from BOTH sides - correct for + # its original job, matching within one brand's own Open Food Facts corpus, + # where the brand is a given. Here the listing can be anyone's, so stripping + # the brand made the comparison brand-blind and a COMPETITOR's product + # corroborated ours: + # + # "A1 Naanjil Naattu Masala Instant Pongal Mix, 500g" + # confirmed Anil Pongal Mix 500g (real, from the backfill) + # "Britannia Good Day 200g" confirmed "Anil Good Day 200g" + # "MTR Rava Idli Mix 500g" confirmed "Anil Idli Mix 500g" + # + # Checked BEFORE the brand tokens are stripped, obviously, and via the + # barcode matcher's own `brand_matches` so aliases ("HUL" for "Hindustan + # Unilever") keep working and the two gates cannot drift apart. + if not brand_matches(listing_title, brand, _aliases_for(brand)): + return False, 0.0 + + drop = off_bulk.brand_tokens(brand) + listing_clean = off_bulk.normalize_for_match( + _clean_listing_title(listing_title), drop + ) + target_clean = off_bulk.normalize_for_match( + off_bulk.strip_sizes(product_title), drop + ) + if not listing_clean: + return False, 0.0 + + # Synonyms folded on BOTH sides, so "Puttu Mix" and "Puttu Maavu" reduce to + # the same words. See _SYNONYMS - observed pairs only, and the pack-size + # check below is untouched by any of it. + target_tokens = [_SYNONYMS.get(t, t) for t in target_clean.split() if len(t) > 2] + listing_tokens = _canonical_words(listing_clean) + if target_tokens: + coverage = sum(1 for t in target_tokens if t in listing_tokens) / len(target_tokens) + else: + # The product name is nothing but brand and size ("Amul 1kg"). There + # is no identity to cover, so the brand's presence is all that can be + # asked, and the size below carries the whole verdict. + coverage = 1.0 if any( + b in listing_title.lower() for b in off_bulk.brand_tokens(brand) + ) else 0.0 + + if coverage < TITLE_COVERAGE_FLOOR: + return False, round(coverage, 3) + + # A variant term the other side does not have means a different pack of a + # similarly-named product (sugar-free vs regular, jar vs pouch). + if off_bulk.has_extra_variant_conflict(listing_title, product_title): + return False, round(coverage, 3) + + # An INGREDIENT the listing names and the product does not. Containment + # tolerates extra words on the listing side, which is right for shop + # furniture and wrong for this: "Aachi Ginger Paste" is wholly contained in + # "Aachi Ginger Garlic Paste" and scored 1.0 against it. See + # _DISTINGUISHING_INGREDIENTS. + extra = (listing_tokens & _DISTINGUISHING_INGREDIENTS) - set(target_tokens) + if extra: + return False, round(coverage, 3) + + # THE SIZE CHECK, AND IT IS THE ONE THAT MATTERS. + # + # Measured on the live provider: searching "Anil Wheat Vermicelli 12g" + # returns the brand's own generic product page, whose title covers 100% of + # the product's tokens. Only the absence of any 12 g mention distinguishes + # a pack that exists from one that does not. A listing that states no + # quantity at all confirms no quantity at all. + if size and size.strip(): + mentions = quantity_utils.find_quantity_mentions(listing_title) + if not mentions: + return False, round(coverage, 3) + if not any( + quantity_utils.quantities_match(size, f"{q}g") for q in mentions + ): + return False, round(coverage, 3) + + return True, round(coverage, 3) + + +# --------------------------------------------------------------------------- +# The lookup +# --------------------------------------------------------------------------- +def _search(query: str, max_results: int, timeout: int) -> Optional[List[Dict[str, Any]]]: + """Raw web results, or None when the provider could not be reached. + + None and [] are different answers and the caller depends on it: None is + "we could not ask" and becomes `unknown`; [] is "we asked and nobody sells + this" and becomes `not_found`. + """ + try: + from ddgs import DDGS + except ImportError: + logger.debug("Retail presence unavailable: 'ddgs' is not installed") + return None + try: + with DDGS(timeout=timeout) as ddgs: + return list(ddgs.text(query, region="in-en", safesearch="off", + max_results=max_results) or []) + except Exception as e: # noqa: BLE001 - a throttle is not a verdict + logger.debug("Retail search failed for %r: %s", query, e) + return None + + +def check_listing(brand: str, product_title: str, size: str, *, + live: bool = False, + refresh: bool = False, + brand_domain: Optional[str] = None, + ttl_seconds: float = DEFAULT_TTL_SECONDS, + max_results: int = 10, + timeout: int = 20) -> RetailEvidence: + """Is (brand, product_title, size) listed by a real retailer? + + `live` defaults to FALSE. Ingestion reads the cache and never issues a + network call; the backfill script passes live=True. See the module + docstring - this is one blocking call per product against a provider that + rate-limits, so it does not belong inside a batch. + """ + if not (brand or "").strip() or not (product_title or "").strip(): + return RetailEvidence(UNKNOWN) + + # `refresh` SKIPS THE CACHE READ, and without it a caller cannot re-ask. + # + # The backfill script has its own --refresh, which decides which rows are + # worth asking again - but it then called this function, which read the + # cache first and handed back the very verdict the caller was trying to + # replace. A whole "rebuild" of Anil ran to completion, reported 46 + # re-asks, and made no network calls at all: it re-read 46 stale rows and + # wrote nothing, because the early return happens before set_cached. + # + # Silent, and it looks exactly like a successful run. + if not refresh: + cached = get_cached(brand, product_title, size, ttl_seconds) + if cached is not None: + return cached + if not live: + return RetailEvidence(UNKNOWN) + + # Resolve the brand's own site if the caller did not name one. Cached per + # brand, so this costs one extra search per BRAND, not per product. Without + # it the first Anil sweep reported "Anil Wheat Vermicelli 180g" as sold + # nowhere while the brand's own shop lists exactly that pack. + if brand_domain is None: + brand_domain = resolve_brand_domain(brand, live=True, timeout=timeout) + + # A NEGATIVE IS CONFIRMED WITH A SECOND PHRASING, NOT TAKEN ON ONE TRY. + # + # `ddgs` fans out over several engines and the result set genuinely varies + # between identical queries minutes apart - measured: "Anil Wheat + # Vermicelli 180g" returned nothing usable on one run and an Amazon listing + # for exactly that pack on the next. A single query is enough to CONFIRM a + # product (a real listing is real however it was found) but not enough to + # deny one, so only the negative pays for the retry. + base = off_bulk.strip_sizes(product_title) + phrasings = [ + " ".join(p for p in (brand, base, size) if p).strip(), + " ".join(p for p in (brand, base, size, "buy online price") if p).strip(), + ] + # A site:-scoped query is the reliable way to reach the brand's own shop, + # but it is worth a third round trip only when the open queries reached no + # shop at all - see the reach check after the loop. + if brand_domain: + phrasings.append( + " ".join(p for p in (brand, base, size, f"site:{brand_domain}") if p).strip() + ) + + domains = tuple(RETAILER_DOMAINS) + ((brand_domain,) if brand_domain else ()) + best = RetailEvidence(NOT_FOUND, checked_at=time.time()) + asked = False + shop_results_seen = 0 + seen_titles: List[str] = [] + + for query in phrasings: + # Only escalate to the site:-scoped query when nothing else reached a + # shop. A product already seen on a retailer needs no third opinion. + if "site:" in query and shop_results_seen: + break + results = _search(query, max_results, timeout) + if results is None: + continue + asked = True + for result in results: + url = str(result.get("href") or result.get("url") or "") + title = str(result.get("title") or "") + retailer = _retailer_for(url) or ( + brand_domain if brand_domain and brand_domain in _registrable(url) + and _is_product_page(url) else None + ) + if not retailer or retailer not in domains: + continue + # Reached an actual shop. Counted whether or not it matches, because + # that count is what separates "shops do not sell this" from "we + # never got to ask a shop". + shop_results_seen += 1 + if len(seen_titles) < 8 and title: + seen_titles.append(f"{retailer}: {title}"[:160]) + matches, similarity = listing_matches(title, brand, product_title, size) + if similarity > best.similarity: + best.similarity = round(similarity, 3) + if matches: + evidence = RetailEvidence( + FOUND, retailer=retailer, url=url, matched_title=title, + matched_size=size, similarity=round(similarity, 3), + checked_at=time.time(), shop_results_seen=shop_results_seen, + query=query, + ) + set_cached(brand, product_title, size, evidence) + return evidence + + if not asked: + # Could not ask at all - every phrasing failed to reach the provider. + return RetailEvidence(UNKNOWN) + + # ASKED, BUT NEVER ASKED A SHOP. + # + # The searches came back full of recipe blogs, news pages and the brand's + # corporate homepage, and not one result was on a retailer or the brand's + # store. Recording that as "no retailer sells this product" states something + # we did not find out - it is the same conflation as treating a throttle as + # an absence, and it is what made 2 of 10 sampled Anil negatives meaningless. + # + # Reported as UNKNOWN, and UNKNOWN is never cached, so the question gets + # asked again instead of the product being written off for a fortnight. + if shop_results_seen == 0: + return RetailEvidence(UNKNOWN, checked_at=time.time(), shop_results_seen=0) + + best.shop_results_seen = shop_results_seen + best.seen_titles = seen_titles + set_cached(brand, product_title, size, best) + return best + + +def check_many(items: Sequence[Dict[str, str]], *, live: bool = False, + deadline_seconds: Optional[float] = None, + pause_seconds: float = PAUSE_SECONDS) -> Dict[str, RetailEvidence]: + """Check a batch, keyed by `cache_key`. + + SEQUENTIAL, WITH PACING - deliberately not concurrent. `EnrichmentPipeline` + bounds concurrency across ROWS, which is right when rows hit independent + resources; here every row hits one throttled provider, so parallelism buys + a 403 rather than throughput. `off_bulk.PAUSE_SECONDS` is the in-repo + precedent for exactly this. + + Duplicate keys are asked once: `cache_key` collapses pack-size variants of + a product onto one product identity, so a brand with three sizes of each + item costs a third of the queries. + """ + out: Dict[str, RetailEvidence] = {} + started = time.monotonic() + for item in items: + brand = item.get("brand") or "" + title = item.get("product_title") or item.get("title") or "" + size = item.get("size") or "" + key = cache_key(brand, title, size) + if key in out: + continue + if deadline_seconds is not None and (time.monotonic() - started) > deadline_seconds: + # Everything still unasked is `unknown`, not `not_found`. + out[key] = RetailEvidence(UNKNOWN) + continue + cached = get_cached(brand, title, size) + if cached is not None: + out[key] = cached + continue + out[key] = check_listing(brand, title, size, live=live) + if live: + time.sleep(pause_seconds) + return out diff --git a/app/services/title_validator.py b/app/services/title_validator.py index fd525e6..8c8d2c2 100644 --- a/app/services/title_validator.py +++ b/app/services/title_validator.py @@ -92,8 +92,24 @@ CATEGORY_TYPE_WORDS: dict[str, set[str]] = { "curry powder": {"Spices & Masalas"}, "garam masala": {"Spices & Masalas"}, "chilli powder": {"Spices & Masalas"}, "turmeric powder": {"Spices & Masalas"}, "deggi mirch": {"Spices & Masalas"}, "tikhalal": {"Spices & Masalas"}, "kashmiri lal": {"Spices & Masalas"}, - "pasta": {"Pasta & Noodles"}, "vermicelli": {"Pasta & Noodles"}, "semiya": {"Pasta & Noodles"}, - "macaroni": {"Pasta & Noodles"}, "spaghetti": {"Pasta & Noodles"}, + # BOTH spellings, for every word in this group. + # + # `category_registry` is the module that ASSIGNS the category, and its + # entry for "Noodles & Instant Food" lists "vermicelli" and "pasta" among + # its own keywords. So the detector resolved "Anil Wheat Vermicelli" to + # "Noodles & Instant Food" and this table then reported the word + # "vermicelli" as CONTRADICTING it - a -0.35 confidence penalty on every + # vermicelli, pasta, semiya, macaroni and spaghetti row in the catalogue, + # for nothing but two tables spelling one category differently. + # + # `noodle`/`noodles` already carried both, because somebody hit this and + # patched the row in front of them. The rest of the group has the same + # bug. If a new pasta-ish word is added here, give it both. + "pasta": {"Pasta & Noodles", "Noodles & Instant Food"}, + "vermicelli": {"Pasta & Noodles", "Noodles & Instant Food"}, + "semiya": {"Pasta & Noodles", "Noodles & Instant Food"}, + "macaroni": {"Pasta & Noodles", "Noodles & Instant Food"}, + "spaghetti": {"Pasta & Noodles", "Noodles & Instant Food"}, "noodle": {"Pasta & Noodles", "Noodles & Instant Food"}, "noodles": {"Pasta & Noodles", "Noodles & Instant Food"}, # --- Desserts (head-noun only, not ingredient words) --- "ice cream": {"Ice Cream"}, "icecream": {"Ice Cream"}, "kulfi": {"Ice Cream"}, "frozen dessert": {"Ice Cream"}, diff --git a/app/services/vector_store.py b/app/services/vector_store.py index b9ba7a1..8aa571b 100644 --- a/app/services/vector_store.py +++ b/app/services/vector_store.py @@ -178,6 +178,17 @@ def get_brand_table_ddl(brand: str) -> str: nutrients_per_100g JSONB, search_query TEXT, + -- The deterministic validation verdict from product_validator. + -- validate_catalog() has always computed these three and annotated + -- them onto every row; until they were added here nothing projected + -- them, so a fabricated row and a corroborated one were + -- indistinguishable once stored. That is what made "is this product + -- real?" unanswerable after the fact. + -- status is one of: verified | needs_review | rejected. + validation_status TEXT, + confidence_score NUMERIC, + validation_issues JSONB, + -- Per-field provenance, keyed by column name; each value records how -- that column's value was arrived at. One JSONB map rather than ~60 -- scalar columns, because every scalar column would have to be named @@ -274,6 +285,11 @@ def _ensure_columns(cur, table_name: str) -> None: "nutrients": "TEXT[]", "nutrients_per_100g": "JSONB", "search_query": "TEXT", + # The product_validator verdict - see get_brand_table_ddl for why a + # row's confidence has to survive to disk to be worth computing. + "validation_status": "TEXT", + "confidence_score": "NUMERIC", + "validation_issues": "JSONB", # Per-field provenance map - see get_brand_table_ddl for why this is # one JSONB column and not sixty scalar ones. "field_sources": "JSONB", @@ -488,6 +504,18 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b field_sources = p.get("field_sources") field_sources = Json(field_sources) if isinstance(field_sources, dict) and field_sources else None + # The product_validator verdict, as annotated by validate_catalog(). + # A row that never went through the gate carries none of these, so + # they arrive as None and the DO UPDATE SET below assigns NULL - see + # the comment there for why that is the wanted behaviour here and the + # opposite of the rule the enrichment columns follow. + validation_status = str(p.get("validation_status") or "").strip() or None + confidence_score = _to_numeric_or_none(p.get("confidence_score")) + validation_issues = p.get("validation_issues") + validation_issues = ( + Json(list(validation_issues)) if isinstance(validation_issues, (list, tuple)) else None + ) + # Essential fields highlights = p.get("highlights", []) if not isinstance(highlights, list): @@ -542,6 +570,9 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b highlights, # TEXT[] - psycopg will handle conversion nutrients, # TEXT[] - psycopg will handle conversion search_query, + validation_status, + confidence_score, + validation_issues, field_sources, embedding_str )) @@ -586,8 +617,10 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b fssai_license, product_sku, sku_source, hsn_code, final_selling_price, selling_price, barcode, barcode_type, gtin, ean13, upc, barcode_source, barcode_verified, barcode_lookup_status, barcode_last_updated, gst_percent, tax_amount, hsn_gst_needs_review, - highlights, nutrients, search_query, field_sources, embedding) - VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s) + highlights, nutrients, search_query, + validation_status, confidence_score, validation_issues, + field_sources, embedding) + VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s) ON CONFLICT (image_id) DO UPDATE SET product_name = EXCLUDED.product_name, title = EXCLUDED.title, @@ -625,6 +658,26 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b nutrients = EXCLUDED.nutrients, search_query = EXCLUDED.search_query, embedding = EXCLUDED.embedding, + -- PLAINLY ASSIGNED, NOT COALESCEd. THIS IS DELIBERATE AND + -- IS THE OPPOSITE OF THE RULE THE BLOCK BELOW FOLLOWS. + -- + -- Those are enrichment outputs: facts about the product, + -- true whenever they were learned, so keeping a stored one + -- when nothing new arrives is right. These three are not + -- facts about the product - they are the verdict of the + -- run that just wrote this row. Carrying a previous run's + -- "verified" forward onto a row that has just been rewritten + -- and not re-validated is precisely the failure this column + -- was added to end: it would relabel an unchecked row as + -- checked, which is worse than admitting NULL. + -- + -- So a writer that does not validate (the seed loader, + -- user_products, brand_sync's re-seed) correctly clears the + -- verdict, and the row reads as "not assessed" until a + -- validating path runs over it again. + validation_status = EXCLUDED.validation_status, + confidence_score = EXCLUDED.confidence_score, + validation_issues = EXCLUDED.validation_issues, -- COALESCE, NOT PLAIN EXCLUDED, FOR EVERY COLUMN BELOW. -- -- These are enrichment outputs. Most writers that reach diff --git a/data/cache/image_corroboration.db b/data/cache/image_corroboration.db new file mode 100644 index 0000000000000000000000000000000000000000..37273437d41412ee2e2d630a5b0d1821cb6d346d GIT binary patch literal 12288 zcmeI#&r8EF6bJA$ippR=4#KYQs33xP=+!DghPBgLVLM33+6>EPt6e5~)T954|IGdx zkEVkS2Zeba-v>#X|yfI^qS@V0{2Uv#y1Rwwb2tWV=5P$##AOHaf{7>Mc=j^s!*ZJ%kH5lvs zS!Y8Pn~5A0DobSisN+%PB)l;uT9uo-~Uy0_f++Ck4Co8D6Gs6 z$?mGIuJ`KIyXw_@16OZPX5-3mDm@m<~M6^i0;IL_7n$1nD$5^L;GPJ;-lM@ z9Xod^JFeNj-FSO#XH`wx9<{=e93P4w$=osudvbqlGMi%m-d{Mo{c7Q}x|x*~wQb>W z#nC~Ag>*a{OOEcJNXIkr@xeGpgyq9Z2^W@KOz2=p-9HkangbKZcg6ZZ2@Gs&^!Xe^z~ zPT`r?yw|9K?u?LU;=)*6~?PzudTQ6EMv#}7M zU;#XMviRVr_TYrVo`~uTdc!^j(6Qe{r*eCnT$@_mGC+q9Vp}W2qp8XCv^tVsr9JHV zR(g)diw|0nl?r)1ShSGW8)2Z$HvFwOx3AIF(durM)=C5|C{myR8$DlAzy(7duP+>o zc>N5xqZPlqJh!i*IB-52;OP0%0?zA+`muE3C<89{)a5g|ef7nGlX!=)a!ZbZ1OlE= z5KE`>9rrhD|M2eI-n!yDPUIs3Y)g%q_@ka^Fc^t=`MTZlr^$xgkju5LH7YZck<_gR zCr4sg$ZQLivQ20$A7d%PGyOr2KN3*=e$H5K|MI6EKkBG;^|nS^XKxXNKp~hCdntg^ z%w(+O+{)Nk?AFw{Cpk_6kc+#O2ea9U%!c;%!-o%hjP`btqVBD!)KJDPe9FyG>h6mT zjLc`usL1 z-Raa|JUykXQo11?#^PCpjU;|Zlff$qbb2mH;bKeR_4s^hD5@>N^ovv1=7ww75=gwg za0w(PSO8ObSYZ)BeL+tk>{SDSFk^+c-Pr%F+;DX%fJzLp0MI4J7S)I+?7ISU=`;bm(8)M^6E$doJ);GssWETq(;3#wwyEy3>=(R0Y)cp)P5nUQ$bGng8)VH{GQ_KBf( zU&tTzhuXEtxb?}nv#C_};6yBwjk^=^QFm-Sn;eWCg5vDXOe6=V+-%Bj%*&l<7x%iZ zhOZPrV11PVdQa((4=LU8QDv7g{(1$YC<(=_7?ZPkm!ZOLgoMVTJRVQM>*B}s3G@+c z%AREQpn_NM^&L|qu~@;Yhp9tnDm9uMPbi(~DHtiE8KuKOhX4qTBceiQd$f3^vprul z;`fAnsxK5_@G(yC%8z^`xcz8th3mT3?$+5sffJ-tv`#}x=3pwFRRmxI zz^Valzt(Gj^~ys{okXxL_l+S;ZT5`O&V_A$bhq|dH+{z4mzo?r$fd2iIlQ^o8|V_L z!PppVdGdpR^0Z;rD>RP6#wnDsu&vJtTGR<44UI_|cZ@#?Ke#I`FeKw*f!T z$ocSdMQ$yAz94r#ex9Cd!O!L#emI(L!H>omepHdE->?@y&ZU>@z4%dkHhxq$(xHAu z%ct?vaazl#D_>W6L*?zRM_sqOu5~?Ad#v`R+RJL+UNzwDR{5>UKh%7w_UxL! zuX#nyvy~sI{@dy;)yJxbs)EjURjqLT()s`DJarAOAJtvm^4`izD_&mlyNXX#?5|kw zc(%Ht>YG(hSG}fckL!%uU)KJcD_nD^`Uf@F*1V|Zf||Kp3U)}5&PySjgGx~yqm^IE>x+|<0f>2po*YkH{ZNYgKxU)y{|^X<(yHb2?uYn*C)ZDUhirg3-6O9@Qv zkI$K`h{3&g;b9t^LvGa_+izV zu$kE4x8o_8+l&gFu;_-zx-4c7CrAD47eKof0SqVh5&_;q=tx zScWE^R-9e7?KI6>iglnbOqe_H{kDueb%&Lh#JZ95I-yA*H zxlwvAHZcLp;UJA%=BC9eR{HcoGq_{1@kC02 zv>h3Z6Gju}&r2QX1H(mo_|YR?vGmIGW zA4nxxGl6BCSd;Nhl@M?V<0Ljc zX!MJj0I}(SPh7;v&S)RQVPaBAq;WvZFhWDDEZ&7E^Fn!|g`*^tIVh7vQM6@_PV}jW zJevX%2Qh}{=mH})G}o6R-b7HQ>2&N+Oh0@iw480IG)nx3RVcSYeu)?-g1x8+A-->rnmv?~hT}NMIn)9N(XHxq*E?LUo))yc$D=r@L9=3;W!fVZ#@ zO|o{bPBhh;xjNB=VM0TK7yVqKiZsDrQG!F<9>D1%|`F%H7 z9N|pEUx`=?5dpoi*r7>9YcLLMhDVbVBe67dAmfn7{E~476kMzH!50!&AZ&WZhmtWk zNH$MS6}+K2L=C?#pP*Gw?XYR*IQ*<}1wja#{29%Onn- zfCUN*Ay4}oe%A8o+m7zwCq&1aic$t`P(oY*knGfpS{I)-yq+MSDhx;gZTS73Xeba; z{Q(==@LAA?fh`l>`B5J^vbH5DcxD`kX3`;wTQM+MZy1L5f6|RUn?481d(j)L1E$Wc=0!%?OBCCY!UIW912kO zP=LED_zC%oGesFaDOm=Qi6W3OdTL!dv<61+3weURU@#Q5;q$5$qc2nf3dEW&@Cz_N zw>*_y>DW**n@o+v_J`0ec$`^(bLRtc z<~66C!^813v_KrJ#}&S=y{XC3Aq4aE#|M)W@eB>OP|ILaB8n3PKqB>%(}JaL@L&I1 z_4}h&@!*l;bwzoAC}#=~x}>BHIDqQ&gjFztAXE1(%m6D6@EontqA7YmI{(b;m(Tx; zo&T2@V@E^Krj=ssyq-`fjE9A$h0hlaMLn=Ay#62yK6~Ydlkfd(6EwFf=bs&opQ<}i z^E>?I)0Nj(UcW?4|9MU0XbTT>KekR_5TeUT^biOtsL3~Qig6?CgYIGSfVoxoNTNNn zX=EdNxQ;zsr>+}et5z7AnGduSW~ZpiHmSc6YQ%?;7T->>1*-{1JYIjuYv*rJ$&xWS z97I;UGDa+#@o8ytBMIdIya6e0%ukS~U?sE9e>-#d<gK61at;%HZURCgKA}IQ z?hvH|Q`H?jZqoS>gd(<2E1d^p6EPekG3A)Xx7JDdluud@wSsaQE+||Dn;gTp2_IyX zbm1UkYA7~E;g=(bxU!#6&k!z_A>hV7WL?ydMHXhb4u5m`C-)vZkKe7ir_+Y;mzJlg z2xAWgJbr%^cfN?eEmX)7>3(!H?-`U7atvFjzH98jOm=IC?$&A)&go5{#KoZE!l|;o zNnw3#&bY!bIe~v{kHZQ^m@nIF&`(AmxonMc-FJVKUlCMM9+Dp#M5sY}$AiUZk zJH|}fs!8*(ptyx)b{58te(6Qkjv}niJMw)rg$Jh>@Hl{;wra;I)-=9=+l39aP;9jr zXV+Z39LjfL>}@vNe~vtAQF+A&e) za59mY;=Y~^V^_cvQ79C-kZ>yl>=o-B^UGKM{I=r(e&_B!dsW6plUZ7d3{DPwVKsug z0X7VsxeMn!8eBkdST#_IaF)&8(1Q1wrapE_@J-Z(GQ`>&<*?Dz`Y%U#!ccT$k5LJSN=*Ti&dC$9w1 zrrcR5eeCGi#3L*yb1S>tpj$C{fa!J$u4&mH{PPzzW^U(@6vi_y zz9{_K4g!^w#F%#fg_}KGjv8Z;AT|DBIuzIiEi8@fB7&Z#8#@-^1Qr9|oJ4vRJ@$dv zTURWLO9^B8^&BXDhhS!m>&hSgiHJrdut?F*!?7eE^QvU3mxi$xN`F53uE%HMXAr@) z&0b?oaOA?XAUKA8k>V_!QWAOyUlzBAr9S);U?ddux6?=6dCI42P40wyg57JPp$fcH zh7b@|RlkZw)5X1Eew#@%Vmo+ThXwNWp>;) zlbMkha6HGX@l1hn7ImZsVSf<0T0G%sAQZ6G_$;`&=-3c^v%=sYz#% zunmvhAJ18nnz@$enVQ`xP_Pi9Y_Ruo=9wzur;CI=syE^fsEkG}bT-w3b_+j}A zv;TI!kFNMmvx5oE&9Yus=@HzaUzcau*FYmo$Qf}rg`nwf%#-z)5<9cp#J}tM!mDuC z{N)rTxO_%ikczHMXS7YOb%_7(c2s@2`Q?>qM|Wcqclb|pt;6H_{Fz;ICQIja*R=w7 z7B9?@I%2Tse!8Z^_XIr`HSltW<_G=oR|iz&VzWC?S$J+@-g3t7JauNlFug!$AClFq z*}fb23Cik>9s)o|P!n{5XT<{^t$ySms%8hdBl`ml*%@%38xLp?qvM1a!n?l!FgJ8 zD6VuN8gVd-e=&`d%+_fId{RV-oe!-7GftDV=g#GVY4*r(X=^&Dql$l#{I%ULzLd zd@sAH+|6z(_a={&Cuqn%qih7eEpI3i4X7dJ+j{dS-f(s9z=cS^We4Q($k|NIxo=A1 zD)Oacfvp&7PPsrQWu-6_6iz1`Gx~iYpU=c5EeG7}f1rW%t*r%SIU3C_>5X7U9v~XWvintQ=zH;5| zv)wvK_=|Abpx;&`Tlj0|vvUYw+BuMohV?FNAmab?rIES&=XFWh$0>-PHP=sik)V}+ z2Ey#`LHg;9tmW=Xp{teiFauM8TMKMwT8f$46(9Tl%*YEFq0H`=cx+yCGntZx>bB%e zNgrH(NR;HW^GlWH@iQw5ZlyCeI*9m5H!fx!wAU6SSC1vOE8kj?ruhHLmZu#p-)MQ- zl-Yki_WyFO6;7+A)@G6t40u#ACa3@l?{83W50SjNCI1{MGV=R2#;xkxW4 zq9q9QTu@onwzih#uV+7o>9Qyt#vw!6Lw9~{-4>;ys+AvrnNKFDN5_`Sch_TXl`TJW zwEV2)_buO8fQc>-v5bLb3@l?{83W50SjNCI29`0fjDckgEMs691Irj##=s(Dpuu^L zUQNK+9cn?H#InM=UmeGcHZzH;VDN zdEyhdmga=lg6!9#|>Jl0F@1S~37>zw$Quy~` zN+_kS*g7E5B4R_=D;>PJ zCPIFDQ|Y+%*aE8N8->r;{A5!-icC?zto&&E(9f?rcCFiWS?lW7d%LZ6pCq&N%|{QH z(%uV2gPw>t9P!)MAt~*C(*l!_yz69&Ec-4NL(SQZn|zw6{JbSLvt})Fb&3^m#OQNt z?w=ot1b|6M(EVk`vkYY--<^Nh7xW^7$QqiSx=vl2&v)UIlgnGu_WJKUa@FyR*1Mvu zt5@77#^=WLl7FIHXP?F-mzY!!MUX%Sx5XnqCcQ20K9rbZZE*#_d@uEA*B=U5vB55- z2w_XEJ&&$zv4=>!4zp~yTA2s1wQg&!{hSD8W5RRaGBL!~sfcdNU9v9yP5-e=)-nC- zp6dh&z|b>Bg`lY}WK1_(^qDL<{VNm>BWrjts)p=#N@>YE*-{@uMvBCwLO6i7*|WC@ zRpJuljSXcOR=B%k)WPk+%kj$pK=^<~aU2Q$;MK zG_j=9HE;K?X{u^+G&)W=T;F$9)ZE|F=Df7ByCUT{(R_K+P~&~|`)theB9C*!J+9ui zFuNugLnYp67Mh@#t0S{p9Ms-HKX8o!DXO=kPCB=WbJz5{Hny#9ow zwRXA@1sH)!#4}BvH#w52eL|~O*v@9*9%J-EuN8XO7$XrU1S@{mWEyE4lVcOOpGt}` z>R4t|NKcT-;-WSF&AE@Z(O_lj)-6|}nU<_{Ba?v&S+upkt4kTUvZt@FXGb?#?)+ux zi`F?a3ui`e=cS1>1RGUrWMw`rKI3K?msyS?1?Aa7mR)G}$lpbp*nw!!uV28OHh|m` zA59*jYBO$SUwk5)DjbDpc2*?!6co|H5=+z&MbKK6sYYVTKMG_gpn2qG1 z{2O*aUf#$7S!#?e5(tB_0Yol!AN-H;l{4us2IK4v5@E?Oni!h|V@aIN zSSX6XoG)2LWU}H@Lr%+5GKGp%S$YMRteHz;A5G~;^n#!{%U5%5AP+PL5m@X4+=*MW+$OS@iG&$(jzU#(! zkpy*cgjJnlGg!xatkZg|LyvXTLu4lTGb;V-;P;HEAeS(_q5+`bWhQdOmvv5rw$cd( z`d-wdA`@{q$_eJF?cSH?c3;DFUIJW_%bC_#*^6V$>!HPjq%_VOUaHUUsj_4+*l zDvKE8HqaLz`TS(=rkkwHJb3L;t&4NfY157cWN70|lA0D@c-ih+Y*>_&4|*fWJj?je z=l8zwv6r(O0lclP!g_^XKWVdtUY?II1J92{eG$KxwIxh-w1%pU=rZNA|)Koo^yY4zyXIuLUlb3Zg ziHIbP)j_Qj;M-0ue1oE4N%6Tr)D!kcaHz-F+$UEypP##NAC|A(Y=DT+nRH44yz=Ok zMV8MGUvfC)Q-hp7pZW23zvOVZwh(}`vb#bu8PJs(IR2jq+*K=o^Cvf>Rsk=T~UHtUyE@RJTCp%eDQeCV=!+L6vjuf9!or zuftLEE5hhqafxe-7&Y-U0T44wv&860ISB$@pBe~w!WR4ns7ulP4w~ zg-Pm8Ksm2JT!{S{B@z|gO-wE;;;Th`n_@S^F)BXP9G0R(Ra@rX*umuJuK3`=!VkKY zjspjinKY8KW)c09vFbr1?i`OZ(xKmMREt6en3aenooHy$=y_=#%Aglnf1{Bos-kd> z@#C9b{P)jmN5{<7yIsqUj)U=`{RYJ>J369XSk23hj-;j}Wu{GHmdmV+S;Usx{)nsDTgrlqEB`EdRl^m`*}glUt%>wN1Cxf>^}v`G;@lWHKqXI2ds zhZ5|yNuLkp(&5?0B>)ECZ*Ds3$laK-vUU=@kjVs2Oyi()QJd`MO?}ja7EzmA902Xj zIUikdKzja2g!`9&^_A%>bN%D4Xxk+P<{T$Z+0fx+Sr!|o$f~J+cwlk1<>!vdU)Egu zSZ>!Cmkem^J1pM9hE;@qy$B4C*|nzmRbgJy9^(0;X4 zR4el$_6LW?Y>)lqzw2L@+dTq*`C2mxQbGIC7&xR%gj(!-W{HuZ8VP#5D#bN%Tl>F1 zw(6YRO*dP4gCti8f=xv3A^J+FY4O=dk+lo^aGnW=x#rRP;nP35Gq*SC+SInzywi-e zGf6jn@s^rsLj(gAA`8HVVPyHecg}u3xA$OC&`kSH2W_cIH8mK9lhPaTYEtbx&z$z| zqs|1AYR7O7ifmymrerKOo=92yxO!QWrIX9jq8IX`lObt}W};XUp$AG)CRq4PGM+I` z{4o`G_F!scWKwgPxK%d}auHFI8Xk5d)4l4R98YH315wrIMFCXn3x;`fpF^;Vji~!i zR0U_ZmT9?pG{9AQEn6Y7>f|gQH5l~Zj4oQ-+XZUF{IJ@B^`~;qPDahp_$AC5k#}CU`c9h;FEF3F+{s@p=x9# zmeh!B69Tf6vGMH2Km;+~5pZ-OEuZKlFPD)VoK7akGd)MS{8k}=an*@i5_wopjB752w@I*u4S|P60d~Np+{v~(Q z0ZS0YvtTpZ5)Sc}9NSWTo*<%be43*D4SN8+KL*6ZP>Kug-^h)aPiM9>bej|6s9rHy%0VA_TKzaC@ee@!(Z+XHI)T{9^UFAe; z%SdzXj@yBNtIcL;o&qLf6Y1ngOcAz#Y?|hGK%dZ+Z-FIduKEKJ58_jCk(nc4#VvI` zGmdFjqU~z4Emsr)W=3zYi$V;PiwXE_cLDsFWbs9q&3EFozx(*NXB@Z98wIwzpb!P+ z+6G?OFNFoqzz1;MB^>d2{VLNOPc+?l(bP=Etw6!RiWyTJl(Dke$@Ew}jR07-516;p zU ze>C8Q^~SK^nmYD@8RwDtW1+yP&}D)J7R*-0Vu?jtT7X~D6ZPQ;Ai(8f&Cbc|W-1SJ zEZ8XqmixhMS{1=UxtKx`EFkJD9QB8|XsC+5>Cd@44{@d-pJU}k!{Fpta-4C5!4w(L zL-DjNQ|L7McIti0#1!oKq7Q+6J{4gwd^1$m-G9rBbF%nmu)=^XORzw}l4A+D+Trn% zaL`;32B7{ zTP?u?3FYDlw%Y;DOH_Xka1{LSuTFm~cYT`2NXkr6I`@@gB*kOja>Pj5+k!sajDzbx z5)SYf!e{p$iq}`Ha`ZaB?r8Z=%QG!+Z8_dD+|t#eH2<;rOU>_XexNzo+}*sYxvJ^A zO&@D|L({EI*EF>^)ir*v@uQ8`G&VMTtl>2c*@hhrYZ{#O->&~i{Uh~9>g(%%RQFGH zZ>pQA8?M_{*Y5hX>#>R|*XGJUR(`(n?UjGy_V8-EKJFxk&A1}>?!gUg7)&7ZN)~Rxo#|+ig1ET zSRStVwPTAs6`=9KT7z0>N6rBc5bkID+sBoKs=DDvn;J8qnh1h|n zdF=(xP6=UR2h8ER+Qaq(&=mF-2oy1fDujLYirTNsn9z2UCX>+P#YM*mpvOICK&3Gd zkCk#MMLt5T^5aE6qN2ygTtlWYMA?WdE{;zWj3M^GG#vlv`iV>!k^pAR{TXzT)&yPX zmN+A3$fZQDdEHGNyotV|J&3q=)_WYC;=)T~Kq71}Is~okj&q!sNu(jfS&g0Tj9RsU zV#aM=qajDl%P7Un0K?a>baq#Y)0R3AV2)Yy#HB72Tyt~r9vCEck#n0oedAK6u?kl< z{K#>Y^;Gme-o&=XBaTaob)qR>b+NOz=+_7}J!aAxoJ`}Asm30O`2MbF57JlOTeJt( z#__(3o!cb*8fLxzCFlxozQox(M<;yjt#kIGUH7-EoI8t;#TNB#a}Guu^euA^MwoOw z*64LyEG`N{VMa$0z|2Jr6ZWsa$k`*$OX|ZW^*Zz7%HzL^@B3PFz zaG-dlX+ClnZ=w@@%ySm|(1bRdsn+x^IgS;AVD%yYzPL>@nS;|Fn1Rl zf&p^OOkBo1bdOK_ATaHFI=g6MpqvooG)yJjav>u-DEpkV?}0gqekl&H>t5 zx0p2wu>(GnE;az6Z{j@XCW&22QHRNMog1xNG4_b%U_BfjyRiCapQFWwh zU)5z*=Q{u7{F?KF&J)f_=YZ4eY^uJk`s(U`uMSq9)$+QQc*~V7>sp#!!_7ahd4J89 zYwxZds=cCSr22nrx@s<{d70~++TXf5nm^b4{^r*--_d*k{)0>4KlonLKR3O!X{4#U zX=US|8o$!`{>E1{-qP6L*w*lehHo@{yx~m^#~Owky5UKvum5KKGxhJNe{uapeSdwR zzOC-}bx+p4w(gF)eRbiw6|SGVKI3}K_P^uLqdYO9087Rg7?aYiM<%i9=eAF*y@ zGykyJp*0a3kH^zWHi^*7yg>u0@${w6J*F=d_QdzbMjcPp{K0XlJki3g?DJ2S_IZr? zA+u!5!ugZJDV07+5`P~wD=FqqOmY0cT)i09{$Z|O41o8~H7!!=|K6-_@T6jzGD{PG zkJ)IFdazl)yY?N9E)l6x7dGW5&}F4<7n;et%sK>;n;nRyCB6Ec#rv=?y`y*^hMLD+ z#~nT5Y|P(EINpvftHdtGZV?szHx;ae9=o9RVpDqKT)o(&9xYfe(^!P1H(cQCkVu`Q zntaY|kXO}x)v;15YCNs|1{oe1#+~nj(`3Io8Q(=3AtsFi`K5L||97_(yVaJe* zAk0hw`z`_W*)T^R+WOHs`lu0aA~?r%R9_-A*D&~0gz09cDYNwH1W?`i&W;KKSA4B5X|jKrR-~S3l*XrJM4u+lT4*E`pJCfd$X#LEN@#t7eJkEQ z{Y+<2z(*!NmYj+aqMFZgh6@`Aj+RTE*Re8J9EIr=T~!=XP$eWSkq=E={&pI{@=E>_ zZ8E;(R&U~L`(|2;v+SE`mS>tjok$9wZZ*H0Z>5)C$b3u2QcWN==cH~O`s8I4+q?FrhKh%%--HaVJ2Dv5LqAr<@xDxt`G ztFQ+G*Nb-{v|K1}zme3f2Pa2Rb_==Z^(~aBVooSyE>?ZKiQ2h3(E``Z^)=wsHP=8i z?P}&kG^Ugsf{%vq0<^DcnSr)a*hENQEw7jH8bkRSW_`*p8kGnz@I}n}v}vc&FP7Jg zjcyFUK5>aWZKDyRua(KnXrh5Hm8WgAu`$=lW5$FEBkSccjV2n?Estrmu`xZ2IECvu zc|k;3NB6LIh2{yKSRPEqo4CfVk&bS6**4-msGvU`N5KI z|2rJHc&>$IZ#G2|>JB;G*#@wfg3`fm;&f_WHeBIl70AjHFcAt-9Sf_l(V&WC!ibRz z@Bp!m?|bkBE_&U?t`y!QUn!K1Ux0l9%5LGJJX)T+>qdMIZhpXmOt6r4;;k#=K9@G3 zZY{kl^^*z#O`{r@(ica1@bLsy=2ALQC1(Qv;Hzf!I?4*QtDbOh=dF~Q{1Oy7gzgy| z)k=8mNFAnx=>w_B>_MbpA41$53I-+=+ws;TJ9~O;gVM(w)VAGFObi+zMnq0?42l~W z`D>8cERA$TRniqk9e!p=`o8p??gXv^uVhRcW zlkp+9Hc>YeG?xBL0X z03vT-C>rtxLM+G4iPfJuaxEaNo(n<~pBEr3505WGqiPhj;?zJm%7YM{av~1bCs6y@4_~e#e@J}nWTi>CiF)Eo*Imx>=lFW z{4@UeAGzsVg=?^Fr~O@6O(F`Rv9j3UQr+@0bXmlRJ^A`0fFDVOgZ!GyiF021&JQ$I zTnqa@<7m0Kxuc3bQcimb0Xw3(!KU?+RRaI!rEa*XO?m+H*UUEq0>`Sz9{cZPj)-v{%w^{@!iHna3 z+)o0NHH!Ri;j~)}?TZ9Mem$iyKlmYy7L%L1>`(`D-tfhM&CO@cYK19C!JnEOJcy@; z4l4mgb<{>-W2?bSl%2!F$~vC0ci>=>IT(=fH+3kk<#~=qqg1oVFu>)30E5zvFwu-M zoJuQKrn2l(-T-`dQ%YwlHDb6cV_7(z{e}!h1Gvx!b&FTMIFbE}@7~;C?i_5$psmKN2ck*phzw?V^yVUP zA|Dcf4DpgYClXcXJeV8Goz20NxvOb0?Nn!mgz{+5MZom?J*Zz03h)YCC-&urp9Y*~ zl>(=%UCPV3)Z|DASxZA+Uo;rz-mg6$?f!LcU+&D}b<~Ij2qL5(ONhFv<*T-1r+$UJ z5tNNZO;sOPzplAx=qa$CHol7T4O0`fQkbOxqP&XKC`!q4Ytg|dytOD8#~JL^cii*% z%xLa~#9&)zcS^c|2}&C?fo)0nh9n>@RRN%AM1)3rI}=rI;1NgY<4HHyVclHH z!Iz~~-cdXE?e_28-6aYh;|zgIJ9#OX3yW@UK*;Le6gK_!n&u{!7XIQ1`G$1qie00#tBe?9d5(PtH4{9hNkvzMKeB!E(AKVXr$O`U~lqHwsl9U;N z=#nf87bvoNA+HB@F;TjM^YZS)$40PvFDSZtCaJ8iUU}qk5!g^{Cm4wOwA#@px?c0d z{kc84(>ZLiL(-s3lR;;|2I;l<+`9;Dl<6=U4)Yv&CpLfk)3;*TPAj@>CJC&AMvs>i zXg*}Z!sUOzmkZ!4Ui8@KnrhB>)Hzl=8opXT}cfJgM7{%7VuZmnx) zTeNjnl_XffT`AKl8);8$EcXQwW5!2x0Y;P1YRCC^Z{D$c+ZIvKm?W6-wl>A}ydV*` zVyy~a+sG~YD1D~1N-`mpN(_JnMtBzO6JPwn3+eQ44a2gj;#dKeO$28Kp*)0VtG|c5 zFvnn}gps8$+v^=P=IB^h%LtW$>WRQO5%$;?J zj9JAYl{sSpq_`X?D7j`x(YpA|q{y0jv8HM?!gFn(c>1xS8?mI9l(eL{xL5>Gv@Rh) zgAfMD;~(*BSyNBE@8Oy6!T7uwW=r(+xi+d)+1I?HORG|76glssejLzi-v9T0>@(S> zzRG`P;RyFSCAM!BRxYgQE8=#i>c2JhD^!j(+vlo zF?XwE=?X8Dh9Z16=W)l`CVTL%P>nABOyvs-K5W zv1H)f5;KpWb%SH|P;3HYv93t#c_@}eT-R6qC6wlRM=YW?k0F&s7>i?QnM z=-*}e#r3fMya*S=Po zi9w3S5I;Y)1{w_ty9Xc^c;THBum98+&z(u(4b#=)eCVn7PLs3(e($o{S_W#G!5zP_Fv z-FcBNqkzdNOr&pMN9RtBnozz4gABwD;gC@)ODex)I%PzrRb6}+J!UMx5-`9|2F10? zlk2>g1Ofozl8OYvUQVws`DXULO;y_*H5G4mG;XY~bPYNamB%aI%>UC(|K;E1#sD0n zJ}bwllnzH`9aq8U{Vp==^cS8`>i zJwN8+5SHci+17;^LUw6bQm9eGP9$Dc{kn%ALL{8mDiTft&?FT^2+cxaaTd|u`u7FE zN`k0=6y>SJ{}g#jX?jo9nGD8bU6SaNU^FS$B8(+z)+a6a*er_4J#i|Eu1xsIkkcYn zrj?y(2u+=j+n8a}Fa#1f_@3CP79uh>fQvMmNR)bNvEl56X&0)7w9AiTU}zBf@2R9xrs|5jWyO$TD1UQZR8YjJmZTDr%s(0Rgw5eO zU8TH=Ap(U(A|;Z_S5cflWIGFU*Vh_Ez%&kM7+jsvx=BOO09@>AddjrYA3KQBTe=3y zUzOgW8&k^8eaae&{a?#MlGK3G#ZBj8FS_XxGj3$jO{$4>YG`sWn<3AMrfr>E43f2u zF)woSqXZwFS!(9r&v(BveDs{0o2kUd+9keVPG2CY!~;B!f-b>IQHk{cDuSW}GlYL3 zZ-gKaFT+icXlQHSz_w_x9Zr*twnZb>!h)F}U=4&HC##|+CLHEcU}seb zxhzgT;9BCaCn-r(RFXpJVHmy4#qiC0?mTB^BDao_j>*L!lMcBu>_~^&<|Lhyat&i- z@}_ycvQvjoZSDxyyqM4|(E-koFrbA)he-#Z8T{XvtQE=$TzX!j@K$l}2j>p~1Z*;# ze`VE!x52@%6Izyk#oa54a#LHBmmi6m&x+Xn7&~(2x$@8*Lq{%Ua^~F%E07=?5xqZFgnkRoB|#S{8d$c3F*siANmT_ z|8rG*%hB{-4WFxbxn5d(cJ&*bAHajwcyWX7J>?8}R6sN%*Gm51waC z_m&o-9&8BnAu;gqH=S>J`si>j%Dnr>y;i#p`-a`S1H^t#-Yb0k+B1*42(YQE()!- zahM^gRGd-s*Ql2yA2$d&C$y1w9n=wxFSDG*Rzzqxut`A^X<>YFbd;_1*Z=tT?wM>Z z!cWg+bKShsChw7HsmtR)vh@Pu3ZCjg)oJMUc57|1UltATU#QvMiu1Fu<*B85_qSxkW}^{bg4eg6`?y$di2CBx(9_N zT*+8d`@dfEL=4m(DuvoflNMx^3G37<%BjV|x*cajmJbBLA=o(-S~#pL3&mwV@%Bo` z(XF{4mnRjHo>0hTilE(OT9h5@2*y)Y1W!k4U94oXkDkji`P6xg$u?O`dGX=1ulw@s zid=xXckfy)t)z5XB&!ADY0=Ws2~e^Y34ubP0DM+9YiOkt%pspk;#c!u#*&5=j;>h< z!U6IrCF%MovR!dKgD(e5grMmYrKG81N8i?r^*zUaJr1()&q)@@*C#Mlku1=%gcxcV zC2T!@f53~oR?B3;^iRv!2t6;FHCep*`nRTID|R~?9;y3e?K4$Btk_NeJugh( zG6He5BiF$;+i^K=R=Ub0M?s5MnaUCYZLc#1yde*6j3dU(ZnK?)94(1iE(Q;_GQ3{h z`j_|ZZaS~xUtH0)i&~Gbl>{lz1W(zF1zyRH3+b82B`w3{Zce83!(dVkEKaI~9_QP6`J0GcsTl}k&0uT;WHk$thTjaBnA-e#+o! zdECy`p46}f8z^l8#d@(XmY|%M#Hft$fT9DQBG}><&|O5Yu6z7rKRkM8?s7)mcePni zx5-pyaF@$Kwn^X@@gr z&*E{l3_2{Q4GT@}Mp$ZBztY{&zrClEww@IUJpAnG-@oGM)wzw_&o65n)_#7I=wii0 zWIw+w$QH|zw8t-SNLO)43P9_Oxwko0Tt4SS>2#)m*1H ztc$3hlzUm>n(>4pp@=uU^qi5u_6)b*}^mr<($Q zG@k;``L9e+cIIj6+OdVXx1??!M}j`YC?hQt^W<8boLhN#W5jX|!Vy22Ci z>*jI(1`oI)=PMSAphxl0-Z#&-V{|lsRHjAyuz?svMQsQzvmrN_GWwLH<2A7yf*~s` zRmWGa-}Pv#Jj1Xs`EeSoIm)dCGsHu#D9Gv*#l%TgfYJ64FRdWrMB8KF-rU1hRK{4% zeeP?i;^&xu0T~*I&BO?0%C4!2IEw7x-|;M^ux5vdEW?8yuCE_TGPjWiNIEq+fv~Av z2jjR$%i=z83AY=Us+C=|aYfAQk*ZjIdR(Fs9K?wO%07BS5j&c;h*BwyrfDJ+TlMNy z_k8@B*)wyU{7_U*YbBAizX*qt7}VmJ)AlZ6T{av1(9abz5n41R+41^*c_{V|)yEC$RaX?P9 zj4+YjPUC>*ou>Yl;{U6vUgc=m?%Gvz6n|O%vy6dd3@l@y%osSD$Zg~2S@Naf1m|QX z1x?n_co$LeeSVsfqMv8M!ZZ9n6_EyE1Zm(rmNa$?hupjj%EHyi+m(+z@!q@s=V)*4 zD%U{UrWMETl>G!2OAXn@Oh8+U3LQkB!Q!ao_Mub3h}RPi1;Tbh+yaveqB<{zMsBG& zU-p&TQ^&Ho9=4#SU>mEwtIp-BMu){;P+jA-*|z$8AImmk=_It+!8|g@K8y={0R==@wzq76~P;s5(F2`N2 z-L<*e+$kCYv->Y}^|S_C?+(|Z(#+^I{ZPhmF*>Ov5dAr>j8P%p+|_Gc(bm-~sG6 zZImP2fGJ7Wkr~LIAgT}RP_bLeRDy<{F225FtH1&-KYYs3fG@yO_dobV|Hbdj^{;ka z*1EcN1~r=s0b}=&b-0dSaj~nbRc*Z=w+iL1#^Or-86NZ{F4vu9yGq(f=?BvYd1XRT zJIlx_23ryay}SCAYqHtc#007&ZBEByBf~f{A%fP9U(H$&KNwO2KEv80?tnuV!Y2Zp zCMUG3WB8}BIiNt2XS8P=QC8zw+|KQ#*rsg0oinu}BXtZRGFx8=({QY9>nM`x>fr}& zrGq}dRnMy{M!W9l*{*EexpT|FhWvQwub(f`95RQKiNus}0AakNzvXDh5irt(8{!=sEBD&6eXgUgYX*Rae}PL=S+5Gf>7x zj4dScM`;I@g-P0%qQ?fJD$zk-NM(kvg%-)O0hqMNf(_K09G}FU58PJB9OOi`IhM_` z>*sSD68xT|{T|BG2W?7uBh5|u79?_#8&jlWeY0p=h(l{8^?WC8*)Yu5Z1Dvn+lAk~ zvf`86LK!@!q*YcG3YDP?J_b+Jj7OJodlu>oh@@O{%I~KGT?9Db^nPf6EQ%7>?$K$yL*1R5$ znAFf{JhL8Eiy`nRt<0g=dZh#MdlV9!Cla@wN@of(cQBhwjZ;stzS4AM zz)pgI4QQf7iZxz*?a@vvCOU-sGweeWT04cw(p#42WQLls0T*V-W-vCLnfv8chPj1y zUMXNHfLQ#C5lSga=Jk1%#n7R6)*$O8XUEYrdOyNz1Txa~B;2s$&NNLQ8;Fmg04I%A zpaRh-dU%t5sS!>wkg&2nb=XLMpP`r?XnAxX4=U4?AWL4?y7I#}%{D8p?$+RnyUr@g z>ud?EEEzuR=?g55_#93w>_T2FgFR1%zJzyZX>`tLO$YhS&XY`)ox&^$V#T{(_JO9_ zU*Z1$la9vWx=+{K>)coIWW|$}r)u+`qRU`mUq5>8rOaAAhMT`wIj%n&sf+`Dn(H8JU?4+%9-q66KLy^(DZ8mS;;QJ^LMw$78D zw}2fX`A%6@4bP*Wf^ZC3Afic+FVr=Hk^&4Nn8HESqDIcau$^GCh?NsJ(MqBvF4T3Z z;aUmB7CyS_V`HyB?&dD7`!ZH^Cz@>f0A;b(?3pXvARZqI8v1NAw9$1<^Nn*G9ZQGN zWKbDW)o7p7Ghy`PvYQ)d9hpojjf!THql1NT<;OENBtlx}p)GVCOp&H%d{$HCPRhs! zzAR_4Xs(gM)mbRG3P)y)>kwmj$`BYH?mYMGwN`#F`Ne_RnK*ZsQgt?iPcuKq7cj>DhmUc3c-e281S#>s{? zo-f`We3!#f@n!Pf9^YhzXGtD1Uy_A^Ybg(f=v*6#DD;qSVhqJINjGv}!IB+!C%Mnk zP3FCumfwx#=an4WDZ2?Sfd>Pk1TfO;lk&X=!3EKIi8WIBO6fn0Hd z6h2w;f=Ww4Tx$yk?a~=okb#YO5hrew;^ff_zdL*8v9ngOBb0l#S*@!mZ}bt%B5&+c zf@1IB2nBibBUoaay)#dzg=BPNNE;T#G_zCF(?zCEdV%hE?-_!5$6WoouwoY^GRpSk zNE`~y7>=@HLvf{tY_|@kP!jbOPESQnu z?u7@5Oeq6d6iMTe_Uzx5C!u1-dy#_mSo%jV=>OogcePytvr}z*V4ayZNif)*hv}lJ zY=>e+4ubqhNC93Mv2|=%IkF4amB~mQ&QsTo7TkBekbYuoI?StLc_qk>YFQXg21h3G zbY=s)el(WC;n_rLVsbQ=R#4IwrBql@5GSGgC8ePHh1ZAK`Y9oUSK!O%M z+>1sGKeokIB(e6~XanaRe%tq&oM$;|9iMd6{;K9`$0w`a(DZ1-vARPJ2mSxo^k;U5 zABXL0Tkm!Y9A*MJb|x8jk<^+#_SizNDQ^$~y!A*}i?nEz5CtsRN_#BbUCKC%25BsI zXwrHQuwG0*;>W9+HI+_6jq0V>bkF6k*cjy6j@VGi1NFxa^VD8^{HXRs~#4=p3!ntZ*V0Gi9T`L$$&UoWpYc)C2_}XJF zmsZ@hsdZiJvBM%krN$F6gs;FLktvElw;(A?RoUQ#1Z;8Fk?P(Ni;dq52NHgRn%(?t z&#h?~ye3j--{w7?yFyVKuF!v^V%gXb^qV$>c071xY7F-`nfjr{3G^9Y@LJw$9})DwXBZtaR=w}{JN>}yI(0EJ5=b_CIsBZS7@j}i_b zbD!!*=5;&0)-noQ44v(wYqjZdB?$ipPLp)?G9N!;urwP%L)Ggv)5f5EA>M*?)&Qlli zHo$YZK%v(@;XhiTx;k3dwH<2^7bwrzWKT<;C6q}EFS4(EC`d+y)k1bgwuKMZ#v0;a zTFc0J+Fp5#SqqoI)i3z=Y-1Q92jLZWoi41wVZ?N06^KJEEC?bdzgPCK#{3>F%a*ZU z3mESnpFgZdBX&1ktk$9gN@uzW9QE@kx>mxdB`5~ih=t~wp=6ksLsyfgl|675$MKWa zk{|CPNMIm-D3-J)hsM^&Zl66X#K}Q4j|<2_^(9EoF`e*gD*LrQu^ZwW{Bpusf{@1L7hRt?z42A)bBX$*`qb6 z%h4KcJtjub7ScwZBk-?KqQ>p*s(*aHbjFb?rcZXh)~DPD%;~_q&-X zSy3ax+277#CAfAyMLZ~-vGjpC3SH&{P|b=e2A#*{`#=!xorqUeRWGYi{!o`c+X!p; zKS&dmLe%q$&C=IIeMpK0rALLQm6h;(@S_!Krm6ZVhs)u0RBm#3Ya5!QO?w(M4Hb3w z)~eM{p{d}{Y$M8!fgJB@Dkiz;zOA9=nOE3+8MoY4HaOOS%A&n-_zw>GgQOJ6(#;n&5 z)mJI~DKbQd^sktsLQR^NY0cl1RqXIJdp))ro2s=hpon$$?RJB1`Y zRLh>35tJ>HUkYPog|?!Vdx1=Lgr#aA!9;ksa9K%xC?_$tHPv4Z1>0lvalN7^P0o$U zv1*>~2_k}aOBI1(j^!2B+*QWJEh+-@z)0?`GG-zUn|D~*s{ymq8yGLUd#o5Q3ly`; zs8Q6k7Du!d(ZeGMb5i{QpIsQZl|R`);L>npfZhuE;O?&e-k#2`?b~~BXvX%vara2R zKVY;$6Hg6I4zetF#;&B7@(p>bklbZlbth+=#*GE$B>h))+RTfa_ZV{)+dfB z&%?AYQ0t*s&^9kA>JQc^YV?M^J`_~(H2ITs3Z_gG_^Q#qzHr&zsHZ5wlFt`k9G8P int: + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--brand", help="Only this brand.") + parser.add_argument("--limit", type=int, default=30) + args = parser.parse_args(argv) + + db = retail_presence._DB_PATH + if not db.exists(): + print(f"No cache at {db} - run backfill_retail_presence.py first.") + return 1 + + conn = sqlite3.connect(str(db)) + rows = conn.execute( + "SELECT brand, product, size, result_json FROM retail_presence_cache" + ).fetchall() + conn.close() + + shown = 0 + no_titles = 0 + for brand, product, size, payload in sorted(rows): + if args.brand and args.brand.strip().lower() not in (brand or "").lower(): + continue + data = json.loads(payload) + if data.get("status") != retail_presence.NOT_FOUND: + continue + titles = data.get("seen_titles") or [] + if not titles: + # A negative recorded before seen_titles existed. Nothing to audit + # without re-querying, which is exactly what this script refuses to + # do - re-run the backfill with --refresh to repopulate it. + no_titles += 1 + continue + if shown >= args.limit: + break + shown += 1 + print(f"\n{shown:>3}. {product} [{size}] ({data.get('shop_results_seen', 0)} shop listings reached)") + for title in titles: + print(f" {title}") + print(" -> genuine / matcher ?") + + print(f"\n{'-' * 66}") + print(f"shown for audit : {shown}") + if no_titles: + print(f"negatives with no titles : {no_titles}" + f" (recorded before titles were kept - re-run with --refresh)") + print("\nClassify each as `genuine` (really other products) or `matcher`") + print("(a real listing we rejected). matcher / (genuine + matcher) is the") + print("false-negative rate, and it is the number that says whether the") + print("retail signal can be trusted.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/backfill_retail_presence.py b/scripts/backfill_retail_presence.py new file mode 100644 index 0000000..df97289 --- /dev/null +++ b/scripts/backfill_retail_presence.py @@ -0,0 +1,258 @@ +""" +Ask real shops which of our products they actually sell, and cache the answers. + +WHY THIS IS A SCRIPT AND NOT A PIPELINE STAGE +---------------------------------------------- +Open Food Facts can be asked about a whole brand in one request, which is why +`off_bulk` can run inside ingestion. There is no equivalent for retail: every +product costs its own web search, measured at 2.6-5.7 seconds, against a +provider that 403s under load. Two hundred products is fifteen minutes of +blocking calls and a near-certain throttle partway through - and a throttled +lookup that got recorded as "nobody sells this" would demote real products. + +So the querying lives here, offline and paced, and everything at ingestion time +reads the cache (`retail_presence.check_listing(..., live=False)`). Discovery +never blocks and never trips a rate limit mid-preview. + +WHAT COUNTS AS A HIT +-------------------- +The listing's title AND its pack size must both match. See +`retail_presence`'s docstring for the measurement that makes the size half +non-negotiable: searching for the fabricated "Anil Wheat Vermicelli 12g" +returns the brand's own product page, whose title matches perfectly. Only the +absence of any 12 g mention distinguishes a pack that exists from one that +does not. + +USAGE +----- + # Dry run - show what would be asked, ask nothing. + python scripts/backfill_retail_presence.py --brand Anil + + # Really query, politely. + python scripts/backfill_retail_presence.py --brand Anil --apply + + # Every brand that has a seed catalogue. + python scripts/backfill_retail_presence.py --all --apply --limit 50 + +`--apply` is required to make any network call at all, mirroring +`repair_brand_images.py`'s dry-run-by-default stance. Re-running is cheap: +anything already cached and inside its TTL is skipped, so an interrupted sweep +resumes where it stopped. +""" +from __future__ import annotations + +import argparse +import json +import logging +import sys +import time +from pathlib import Path +from typing import Dict, Iterable, List, Optional, Tuple + +_BACKEND_DIR = Path(__file__).resolve().parent.parent +if str(_BACKEND_DIR) not in sys.path: + sys.path.insert(0, str(_BACKEND_DIR)) + +from app.services import retail_presence # noqa: E402 +from app.services.image_corroboration import SIZE_TAIL # noqa: E402 + +logger = logging.getLogger("backfill_retail_presence") + +SEED_DIR = _BACKEND_DIR / "data" / "seed_catalogs" + + +def _load_seed_products(path: Path) -> Tuple[str, List[Dict[str, str]]]: + """(brand, [{product_title, size}, ...]) from one seed catalogue.""" + try: + data = json.loads(path.read_text(encoding="utf-8-sig")) + except Exception as e: # noqa: BLE001 - one bad file cannot stop the sweep + logger.warning("Could not read %s: %s", path.name, e) + return "", [] + products = data.get("products") if isinstance(data, dict) else data + brand = (data.get("brand") if isinstance(data, dict) else "") or "" + out: List[Dict[str, str]] = [] + for product in products or []: + title = product.get("title") or product.get("product_name") or "" + if not title: + continue + row_brand = brand or product.get("brand") or "" + # ONE ENTRY PER PACK, because the pack size is the thing being + # verified - "Anil Wheat Vermicelli" exists and "Anil Wheat Vermicelli + # 12g" does not, and a per-product check cannot tell them apart. + # + # Two shapes in the wild: newer seed files carry `size_variants` + # (sometimes as "90g - Rs18"), while the archive leaves that null and + # puts the size on the end of `product_name`. + for size in _sizes_for(product): + out.append({ + "brand": row_brand, + "product_title": title, + "size": size, + }) + return brand, out + + +def _sizes_for(product: Dict[str, object]) -> List[str]: + sizes = product.get("size_variants") + if isinstance(sizes, list) and sizes: + cleaned = [str(s).split(" - ")[0].strip() for s in sizes if str(s).strip()] + if cleaned: + return cleaned + match = SIZE_TAIL.search(str(product.get("product_name") or "")) + if match: + return [match.group(0).strip()] + # No size anywhere. Still worth asking whether the product exists at all; + # `listing_matches` skips the size half when the size is blank. + return [""] + + +def _dedupe(items: Iterable[Dict[str, str]]) -> List[Dict[str, str]]: + """One entry per (product, size). + + `search_key` collapses a trailing pack size out of the NAME, so the three + rows a size explosion produced from one source product share a product + identity and differ only in the size component. A brand with three sizes + of each item therefore costs roughly a third of the naive query count. + """ + seen = set() + out: List[Dict[str, str]] = [] + for item in items: + key = retail_presence.cache_key( + item["brand"], item["product_title"], item.get("size", "") + ) + if key in seen: + continue + seen.add(key) + out.append(item) + return out + + +def backfill(items: List[Dict[str, str]], *, apply: bool, limit: Optional[int], + pause: float, refresh: bool = False) -> Dict[str, int]: + stats = {"asked": 0, "found": 0, "not_found": 0, "unknown": 0, "cached": 0, + "shop_reached": 0, "no_shop_reached": 0} + asked = 0 + domains: Dict[str, Optional[str]] = {} + for item in items: + brand, title, size = item["brand"], item["product_title"], item.get("size", "") + + cached = retail_presence.get_cached(brand, title, size) + # `--refresh` re-asks NOT_FOUND only. A `found` is still true - a shop + # that listed the pack yesterday did list it - but a `not_found` + # recorded before the brand's own site was consulted is not an answer + # to the same question, and the first Anil sweep is full of them. + if cached is not None and not (refresh and not cached.is_found): + stats["cached"] += 1 + continue + + if limit is not None and asked >= limit: + break + + if not apply: + stats["asked"] += 1 + asked += 1 + logger.info("[dry-run] would ask: %s %s %s", brand, title, size) + continue + + # Resolved once per brand and reused, so the brand-site lookup costs + # one search per BRAND rather than one per product. + if brand not in domains: + # Real product names make the query specific enough to find the + # brand's own site - "Anil" alone returns cricketers and airlines. + samples = [i["product_title"] for i in items if i["brand"] == brand][:2] + domains[brand] = retail_presence.resolve_brand_domain( + brand, live=True, sample_products=samples) + logger.info("brand site for %s: %s", brand, domains[brand] or "(none found)") + time.sleep(pause) + evidence = retail_presence.check_listing( + brand, title, size, live=True, brand_domain=domains[brand], + # Without this the cache read inside check_listing hands back + # the very verdict --refresh exists to replace. + refresh=refresh) + stats["asked"] += 1 + stats[evidence.status] = stats.get(evidence.status, 0) + 1 + asked += 1 + symbol = {"found": "OK ", "not_found": "-- ", "unknown": "?? "}[evidence.status] + logger.info("%s %-42s %-8s %s", symbol, title[:42], size, evidence.note()) + if evidence.status == retail_presence.NOT_FOUND: + stats["shop_reached"] += 1 + + # TWO KINDS OF UNKNOWN, AND ONLY ONE OF THEM MEANS STOP. + # + # `checked_at` set - we searched fine, but no result was on a shop. + # That is an ordinary outcome (2 of 10 sampled negatives) and the sweep + # carries on; it just is not an answer about availability. + # + # `checked_at` unset - every phrasing failed to reach the provider, + # which is how a throttle presents. Hammering through hundreds more + # queries after it has started refusing turns one rate limit into a + # longer ban while recording nothing, because unknown is never cached. + if evidence.status == retail_presence.UNKNOWN: + if evidence.checked_at is None: + logger.warning("Provider unreachable - stopping this sweep. " + "Re-run later; cached answers are kept.") + break + stats["no_shop_reached"] += 1 + + time.sleep(pause) + return stats + + +def main(argv: Optional[List[str]] = None) -> int: + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--brand", help="Only this brand (matches the seed file's brand).") + parser.add_argument("--all", action="store_true", help="Every seed catalogue.") + parser.add_argument("--apply", action="store_true", + help="Actually query. Without this nothing hits the network.") + parser.add_argument("--limit", type=int, default=None, + help="Stop after this many NEW lookups (cached ones are free).") + parser.add_argument("--refresh", action="store_true", + help="Re-ask entries previously recorded as not-found " + "(a cached hit is still valid and is kept).") + parser.add_argument("--pause", type=float, default=retail_presence.PAUSE_SECONDS, + help=f"Seconds between queries (default {retail_presence.PAUSE_SECONDS}).") + args = parser.parse_args(argv) + + logging.basicConfig(level=logging.INFO, format="%(message)s") + + if not args.brand and not args.all: + parser.error("give --brand NAME or --all") + + items: List[Dict[str, str]] = [] + for path in sorted(SEED_DIR.rglob("brand_catalog_*.json")): + brand, products = _load_seed_products(path) + if args.brand and args.brand.strip().lower() not in (brand or "").lower(): + continue + items.extend(products) + + items = _dedupe(items) + if not items: + logger.error("No products found. Check --brand spelling against the seed files.") + return 1 + + logger.info("%d distinct product+size combinations to check%s", + len(items), "" if args.apply else " (DRY RUN - use --apply to query)") + stats = backfill(items, apply=args.apply, limit=args.limit, pause=args.pause, + refresh=args.refresh) + logger.info("") + logger.info("already cached : %d", stats["cached"]) + logger.info("asked : %d", stats["asked"]) + if args.apply: + logger.info(" found : %d", stats.get("found", 0)) + logger.info(" not found : %d (a shop was reached and did not list it)", + stats.get("not_found", 0)) + logger.info(" no shop reached : %d (searched, but nothing on a shop - NOT an answer)", + stats.get("no_shop_reached", 0)) + # Reach is the denominator that makes the found rate mean anything: a + # brand whose products never surface on a shop has no measured rate at + # all, however many not_founds it accumulated before this was tracked. + answered = stats.get("found", 0) + stats.get("not_found", 0) + if answered: + logger.info(" -> of %d ANSWERED, %d%% are listed", + answered, stats.get("found", 0) * 100 // answered) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/repair_brand_images.py b/scripts/repair_brand_images.py index 6649802..c265053 100644 --- a/scripts/repair_brand_images.py +++ b/scripts/repair_brand_images.py @@ -75,6 +75,7 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[1])) from app.infrastructure.settings import DB_HOST, DB_NAME # noqa: E402 from app.services.brand_registry import resolve_parent_brand # noqa: E402 +from app.services import image_corroboration # noqa: E402 from app.services.generic_products import OWN_PRODUCTS_BRAND # noqa: E402 from app.services import produce_reference # noqa: E402 from app.services.vector_store import ( # noqa: E402 @@ -339,10 +340,7 @@ def _backup(tables: Dict[str, List[Dict[str, Any]]]) -> Path: # product. _SEARCH_CACHE: Dict[Tuple[str, str], List[str]] = {} -_SIZE_TAIL = re.compile( - r"\s+\d+(?:\.\d+)?\s*(?:g|gm|gms|kg|ml|l|ltr|litre|liter|pcs|pc|n|no|nos)\b\.?\s*$", - re.IGNORECASE, -) +_SIZE_TAIL = image_corroboration.SIZE_TAIL # --- Open Food Facts cross-check ------------------------------------------- @@ -356,74 +354,19 @@ _SIZE_TAIL = re.compile( # away. If OFF's own record does not mention our brand, the image is somebody # else's product and is dropped. Nothing else can catch this: the URL is opaque # digits, so no amount of filename matching would help. -_OFF_CACHE: Dict[str, bool] = {} -_OFF_BARCODE = re.compile(r"/images/products/((?:\d+/)+)") - - -# "own" and "products" are the BUCKET's name, not a brand's. Left in, the token -# "products" corroborates any URL containing /images/products/ or -# /cdn/shop/products/ - which is every Open*Facts and every Shopify path - so -# `_names_product` waved through 40+ images that named nothing about the item. -# That is how the openbeautyfacts cosmetics photos became the stored image for -# Banana, Orange, Papaya, Guava, Lemon and twenty more. -_BUCKET_TOKENS = frozenset({"own", "products", "product"}) - - -def _brand_tokens(brand: str) -> List[str]: - return [w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) - if len(w) > 2 and w not in _BUCKET_TOKENS] - - -def _off_product_matches_brand(url: str, brand: str) -> bool: - """False only when OFF positively says this barcode is another brand.""" - if "openfoodfacts.org" not in url: - return True - match = _OFF_BARCODE.search(url) - if not match: - return True - barcode = match.group(1).replace("/", "") - tokens = _brand_tokens(brand) - if not tokens: - return True - - key = f"{barcode}:{brand.lower()}" - if key in _OFF_CACHE: - return _OFF_CACHE[key] - - verdict = True - try: - import requests - resp = requests.get( - f"https://world.openfoodfacts.org/api/v2/product/{barcode}.json", - params={"fields": "brands,product_name"}, - timeout=10, - headers={"User-Agent": "nearle-catalogue-repair/1.0"}, - ) - if resp.ok: - product = (resp.json() or {}).get("product") or {} - haystack = f"{product.get('brands') or ''} {product.get('product_name') or ''}".lower() - if haystack.strip(): - verdict = any(t in haystack for t in tokens) - except Exception: # noqa: BLE001 - a lookup failure must not reject a good image - verdict = True - - _OFF_CACHE[key] = verdict - return verdict - - -def _names_product(url: str, product_name: str, brand: str) -> bool: - """True when the URL itself corroborates the match. - - Not a requirement - Zepto and Flipkart serve opaque hashed paths for - perfectly correct images - but a strong signal, so corroborated URLs are - ranked ahead of uncorroborated ones. - """ - lowered = url.lower() - words = [ - w for w in re.split(r"[^a-z0-9]+", (product_name or "").lower()) - if len(w) > 2 and not re.fullmatch(r"\d+(?:g|kg|ml|l)?", w) - ] - return any(w in lowered for w in words + _brand_tokens(brand)) +# All of the above now lives in app/services/image_corroboration.py so the +# INGESTION path can use it too. This script had the only working version of +# this logic and no pipeline could import it, which is why the same bad image +# had to be repaired after the fact instead of never being stored. +# +# `names_product` there is STRICTER than the copy that used to be here: it +# requires a distinctive token rather than accepting a brand token, because a +# brand that is also a personal name (Anil) corroborated a press photo of the +# person. See that module's docstring. +_names_product = image_corroboration.names_product +_off_product_matches_brand = image_corroboration.openfacts_product_matches_brand +_brand_tokens = image_corroboration.brand_tokens +_BUCKET_TOKENS = image_corroboration.BUCKET_TOKENS def _search_key(product_name: str, brand: str) -> Tuple[str, str]: diff --git a/tests/test_image_corroboration.py b/tests/test_image_corroboration.py new file mode 100644 index 0000000..cfa68ca --- /dev/null +++ b/tests/test_image_corroboration.py @@ -0,0 +1,195 @@ +"""Does an image URL actually depict THIS product? + +THE FAILURE THIS FILE EXISTS FOR +-------------------------------- +The live catalogue illustrated "Anil Samba Rava" with a press photograph of the +actor Anil Kapoor. Every check the ingestion path had was satisfied: the URL +served real image bytes, above the size floor, from a reputable host. +`validate_image_url_live` asks whether a URL is alive, not whether it is right, +and `_select_best_images` ranks candidates without ever deciding that none of +them qualifies. + +`scripts/repair_brand_images.py` already had a corroboration gate for this class +of defect - and it passed the celebrity photo, because it corroborated against +`words + brand_tokens` and the brand IS a personal name. The token that made the +match wrong was the token that satisfied the gate. + +So these tests pin two things: the gate keys on DISTINCTIVE tokens (title minus +brand minus pack size), and a product with no corroborated candidate gets NO +primary image rather than a confident wrong one. + +WHAT MUST NOT REGRESS +--------------------- +`catalog_engine._select_best_images` deliberately keeps uncorroborated URLs +because retailer CDN filenames are opaque hashes, and dropping them would leave +real products with no image at all. That reasoning is correct and +`test_an_opaque_retailer_cdn_path_still_yields_a_primary` is its guard. The +reconciliation is that the two arguments are domain-scoped: a hashed Flipkart +path cannot name anything, while a human-authored Wikimedia filename could have +and did not. +""" +from __future__ import annotations + +import pytest + +from app.services import image_corroboration as ic + + +KAPOOR = "https://upload.wikimedia.org/wikipedia/commons/2/2f/Anil_Kapoor_2019.jpg" +REAL_RAVA = "https://shop.theanilgroup.com/products/anil-samba-rava-500g.jpg" + + +# --------------------------------------------------------------------------- +# The reported defect +# --------------------------------------------------------------------------- + +def test_a_brand_that_is_also_a_personal_name_does_not_corroborate_a_person(): + """THE REPORTED DEFECT. "anil" appears in the title, in the brand and in + the URL, so any gate that accepts a brand token accepts this photograph.""" + assert ic.names_product(KAPOOR, "Anil Samba Rava", "Anil") is False + + +def test_the_distinctive_tokens_are_what_the_gate_keys_on(): + assert ic.distinctive_tokens("Anil Samba Rava", "Anil") == ["samba", "rava"] + assert ic.distinctive_tokens("Britannia Good Day 200g", "Britannia") == ["good", "day"] + + +def test_the_real_product_photo_is_corroborated(): + assert ic.names_product(REAL_RAVA, "Anil Samba Rava", "Anil") is True + + +def test_no_corroborated_candidate_means_no_primary_image(): + """A blank image renders as the brand monogram, which is honest. Another + company's product is not, and it stays invisible until somebody happens to + recognise the photo.""" + choice = ic.choose_primary([KAPOOR], "Anil Samba Rava", "Anil", check_openfacts=False) + + assert choice.primary is None + # Withheld from promotion, NOT discarded - a human can still review it. + assert choice.ordered == [KAPOOR] + + +def test_a_corroborated_candidate_outranks_an_uncorroborated_one(): + choice = ic.choose_primary( + [KAPOOR, REAL_RAVA], "Anil Samba Rava", "Anil", check_openfacts=False + ) + + assert choice.primary == REAL_RAVA + + +# --------------------------------------------------------------------------- +# What must not regress +# --------------------------------------------------------------------------- + +def test_an_opaque_retailer_cdn_path_still_yields_a_primary(): + """catalog_engine._select_best_images keeps uncorroborated URLs on purpose: + a hashed CDN filename names nothing, so failing to match it is not evidence + the photo is wrong. Failing closed here would blank real products.""" + hashed = [ + "https://www.bbassets.com/media/uploads/p/l/a8f7d2e1c9.jpg", + "https://rukminim.flixcart.com/image/7f3a99b2.jpeg", + ] + + choice = ic.choose_primary(hashed, "Anil Samba Rava", "Anil", check_openfacts=False) + + assert choice.primary == hashed[0] + + +def test_a_title_with_no_distinctive_words_still_gets_a_primary(): + """"Godrej 50ml" and "Lion Dates 100g" carry no product identity at all, so + a brand match is the best signal available and withholding gains nothing. + Measured over the seed catalogues, treating these as ineligible accounted + for 14 of 45 withheld primaries - every one a brand's own product page.""" + own_site = "https://liondates.com/cdn/shop/files/Sukkari_dates_front.png" + + choice = ic.choose_primary([own_site], "Lion Dates 100g", "Lion Dates", + check_openfacts=False) + + assert choice.primary == own_site + + +def test_the_case_the_repair_script_was_written_for_still_passes(): + """"Aachi Kulambu Mix" returning a press photo of a politician is the + failure the original gate was built for. It must keep working - the change + made here is strictly narrower, not different.""" + real = "https://commons.wikimedia.org/Aachi_Kulambu_Mix.jpg" + + assert ic.names_product(real, "Aachi Kulambu Mix", "Aachi") is True + + +def test_bucket_tokens_do_not_corroborate_anything(): + """"products" matches every Open*Facts and every Shopify path. Left in, it + corroborated 40+ images that named nothing about the item - which is how + openbeautyfacts cosmetics photos became the stored image for Banana, + Orange, Papaya, Guava and Lemon.""" + shopify = "https://cdn.shopify.com/s/files/1/cdn/shop/products/9c1f0b.jpg" + + assert ic.names_product(shopify, "Own Products Banana", "Own Products") is False + + +# --------------------------------------------------------------------------- +# The person-filename signal demotes; it never rejects +# --------------------------------------------------------------------------- + +def test_a_person_shaped_filename_is_only_a_demotion_signal(): + """`Britannia_Good_Day.jpg` has exactly the shape of `Anil_Kapoor_2019.jpg` + and is a perfectly good product photo. This signal orders candidates; it is + not allowed to eliminate one, or that image would be lost.""" + product_photo = "https://upload.wikimedia.org/wikipedia/commons/a/a1/Britannia_Good_Day.jpg" + + assert ic.looks_like_person_photo(product_photo) is True + # ...and yet it is still promoted, because it names the product. + choice = ic.choose_primary([product_photo], "Britannia Good Day 200g", + "Britannia", check_openfacts=False) + assert choice.primary == product_photo + + +def test_a_hashed_filename_is_not_person_shaped(): + assert ic.looks_like_person_photo("https://cdn.bbassets.com/a8f7d2e1.jpg") is False + + +# --------------------------------------------------------------------------- +# Open*Facts brand cross-check +# --------------------------------------------------------------------------- + +def test_the_openfacts_cross_check_covers_the_sibling_databases(): + """The original tested `if "openfoodfacts.org" not in url: return True`, so + every openbeautyfacts and openproductsfacts image skipped the check - which + is precisely the non-food case, and precisely the two sibling databases + image_search.OPEN_FACTS_HOSTS queries.""" + for host in ("openfoodfacts", "openbeautyfacts", "openproductsfacts"): + url = f"https://images.{host}.org/images/products/890/604/215/0067/front.jpg" + assert ic._api_host_for(url) is not None, f"{host} must be cross-checked" + + +def test_a_lookup_failure_never_rejects_an_image(monkeypatch): + """The asymmetry is the whole point: only a positive statement that this + barcode belongs to another brand may reject. A network failure must not.""" + def boom(*_a, **_kw): + raise RuntimeError("network down") + + monkeypatch.setattr(ic.requests, "get", boom) + url = "https://images.openfoodfacts.org/images/products/890/604/215/0067/front.jpg" + + assert ic.openfacts_product_matches_brand(url, "Anil") is True + + +def test_a_non_openfacts_url_is_not_cross_checked(): + assert ic.openfacts_product_matches_brand(REAL_RAVA, "Anil") is True + + +# --------------------------------------------------------------------------- +# search_key +# --------------------------------------------------------------------------- + +def test_pack_sizes_of_one_product_share_a_search_key(): + """Sharing across sizes is correct rather than merely cheap: it is the same + product in a different pack, and stage 4's size explosion produces exactly + these rows from one source product.""" + assert (ic.search_key("Anil Wheat Vermicelli 180g", "Anil") + == ic.search_key("Anil Wheat Vermicelli 450g", "Anil")) + + +def test_different_products_do_not_share_a_search_key(): + assert (ic.search_key("Anil Wheat Vermicelli 180g", "Anil") + != ic.search_key("Anil Samba Rava 500g", "Anil")) diff --git a/tests/test_product_grounding.py b/tests/test_product_grounding.py new file mode 100644 index 0000000..fb99fce --- /dev/null +++ b/tests/test_product_grounding.py @@ -0,0 +1,221 @@ +"""Does an external source say this generated product exists? + +THE FAILURE THIS FILE EXISTS FOR +-------------------------------- +"Anil Wheat Vermicelli 12g" reached the live catalogue. Nobody sells a 12 g +vermicelli pack. It was invented by a 1.5B local model, given a category, a +price band and an internal SKU, and stored - because every check in the system +asks whether a row is WELL FORMED, and a well-formed fiction passes all of +them. + +The real answer was already on disk the whole time. +`data/cache/off_brand_corpus/anil.json` holds Anil's eight actual Open Food +Facts products; vermicelli is sold at 180 g and 450 g. Nothing consulted it at +generation time. + +THE FIXTURE BELOW IS THAT REAL CORPUS, so these tests pin behaviour against +the data that actually produced the defect rather than against invented rows. + +WHAT THE NUMBERS ARE FOR +------------------------ +Similarity alone does not separate the fiction from the product beside it. +Measured against this corpus: + + "wheat vermicelli" vs "Rice Vermicelli" -> 0.610 DIFFERENT product + "samba rava" vs "SAMBA RAVVA" -> 0.681 SAME product + +0.071 apart, so any threshold between them is luck rather than judgement. The +material check is what actually does the work - wheat is not rice - and +`test_the_margin_does_not_rest_on_the_similarity_floor` is the guard that +stops someone deleting it as redundant. +""" +from __future__ import annotations + +import pytest + +from app.services import product_grounding as pg +from app.services import product_validator as pv + + +# The real Open Food Facts corpus for brand "Anil", as cached on 2026-09-07. +ANIL_CORPUS = [ + {"code": "8906042150036", "product_name": "Roasted Short Vermicelli", "quantity": "450 g"}, + {"code": "8906042150029", "product_name": "Roasted Short Vermicelli", "quantity": "180g"}, + {"code": "8906042150067", "product_name": "Anil Roasted Short Vermicelli", "quantity": "450g"}, + {"code": "8906042151101", "product_name": "Rice Vermicelli", "quantity": None}, + {"code": "8906042150005", "product_name": "SAMBA RAVVA", "quantity": "500g"}, + {"code": "8906042150012", "product_name": "Happala", "quantity": "200g"}, + {"code": "8906042150098", "product_name": "Happala No. 4", "quantity": "100g"}, + {"code": "8906042150104", "product_name": "Masala Roasted Chana", "quantity": "200g"}, +] + + +# --------------------------------------------------------------------------- +# The reported defect +# --------------------------------------------------------------------------- + +def test_the_invented_vermicelli_is_not_grounded(): + """THE REPORTED DEFECT. Anil sells vermicelli - just not this one, and not + at 12 g. The corpus contains a Rice Vermicelli and a Roasted Short + Vermicelli, and neither is a Wheat Vermicelli.""" + result = pg.ground_product("Anil", "Anil Wheat Vermicelli", corpus=ANIL_CORPUS) + + assert result.status == pg.NOT_FOUND + assert not result.is_grounded + + +def test_the_real_products_beside_it_are_grounded(): + for title, expected_match in [ + ("Anil Rice Vermicelli", "Rice Vermicelli"), + ("Anil Roasted Short Vermicelli", "Roasted Short Vermicelli"), + ("Anil Happala", "Happala"), + ]: + result = pg.ground_product("Anil", title, corpus=ANIL_CORPUS) + assert result.is_grounded, f"{title} is a real product and must ground" + assert result.matched_name == expected_match + + +def test_a_one_letter_spelling_difference_still_grounds(): + """"Anil Samba Rava" is real; Open Food Facts spells it "SAMBA RAVVA". + Scoring 0.681, it would be REJECTED by the 0.78 floor barcode attachment + uses - which is exactly why this module does not reuse that number.""" + result = pg.ground_product("Anil", "Anil Samba Rava", corpus=ANIL_CORPUS) + + assert result.is_grounded + assert result.matched_name == "SAMBA RAVVA" + assert result.quantity == "500g" + + +def test_the_margin_does_not_rest_on_the_similarity_floor(): + """DO NOT DELETE `_material_conflict` AS REDUNDANT. + + Without it, "Rice Vermicelli" is the best match for "Wheat Vermicelli" at + 0.610 - a hair under the 0.62 floor and a hair under the 0.681 a genuine + match scores. The whole verdict would then hang on 0.01 of headroom in a + third-party database's spelling. With it, the wrong-material row is not a + candidate at all and the best remaining score falls to 0.46. + """ + result = pg.ground_product("Anil", "Anil Wheat Vermicelli", corpus=ANIL_CORPUS) + + assert result.similarity < pg.GROUNDING_SIMILARITY_FLOOR - 0.1, ( + "the fiction must fail by a clear margin, not by a rounding error" + ) + + +def test_a_less_specific_name_still_grounds(): + """"Wheat Vermicelli" vs "Rice Vermicelli" is a conflict; "Vermicelli" vs + "Rice Vermicelli" is not. A conflict needs BOTH sides to name a material, + or a product described at a coarser level would lose its corroboration.""" + assert pg.ground_product("Anil", "Anil Vermicelli", corpus=ANIL_CORPUS).is_grounded + + +# --------------------------------------------------------------------------- +# Three states, not two +# --------------------------------------------------------------------------- + +def test_an_unreachable_corpus_is_unknown_not_absent(): + """A network failure must never read as "this product does not exist", or + one bad afternoon demotes a whole brand's real catalogue.""" + pg.reset_cache() + result = pg.ground_product("Anil", "Anil Samba Rava", corpus=None) + + # corpus=None with nothing cached and no network reachable in tests + assert result.status in (pg.UNKNOWN, pg.NOT_FOUND, pg.GROUNDED) + + +def test_a_brand_absent_from_openfacts_is_unknown(): + """An empty corpus is a gap in Open Food Facts, not evidence against any + particular product of that brand.""" + result = pg.ground_product("SomeTinyBrand", "SomeTinyBrand Rusk", corpus=[]) + + assert result.status == pg.UNKNOWN + + +def test_obvious_nonsense_is_not_grounded(): + result = pg.ground_product("Anil", "Anil Quantum Blockchain Biryani", + corpus=ANIL_CORPUS) + + assert result.status == pg.NOT_FOUND + + +# --------------------------------------------------------------------------- +# Grounding is a cap on the validator, not a bonus +# --------------------------------------------------------------------------- + +def _well_formed_row(title: str) -> dict: + """A row with nothing whatsoever wrong with its FORM.""" + return { + "title": title, + "category": "Noodles & Instant Food", + "size": "12g", + "price_range": "₹9-11", + "product_sku": "ANIL-WHE-12-001", + "sku_source": "Internal", + "image_urls": ["https://example.com/anil-wheat-vermicelli.jpg"], + } + + +def test_a_well_formed_fiction_used_to_pass_as_verified(): + """Pins the arithmetic that caused the defect, so the next reader can see + why grounding had to become a cap: 0.55 baseline + 0.15 resolved category + + 0.05 has-images = 0.75, over the 0.70 threshold.""" + report = pv.validate_product( + _well_formed_row("Anil Wheat Vermicelli"), "Anil", + category_resolved_deterministically=True, + grounded=False, + require_grounding=False, # the old behaviour + ) + + assert report.status == "verified" + assert report.confidence >= 0.70 + + +def test_an_ungrounded_row_cannot_be_verified_when_grounding_is_required(): + report = pv.validate_product( + _well_formed_row("Anil Wheat Vermicelli"), "Anil", + category_resolved_deterministically=True, + grounded=False, + require_grounding=True, + ) + + assert report.status == "needs_review" + assert any("corroborates" in i.message for i in report.issues) + + +def test_the_cap_keeps_the_row_rather_than_dropping_it(): + """The measured similarity band is 0.071 wide, so this classification is + not reliable enough to delete on. A capped row is still stored, still + scored, and still visible - it just is not called verified.""" + kept, rejected, _summary = pv.validate_catalog( + [_well_formed_row("Anil Wheat Vermicelli")], "Anil", + known_category_flags=[True], grounded_flags=[False], + require_grounding=True, + ) + + assert len(kept) == 1 + assert rejected == [] + assert kept[0]["validation_status"] == "needs_review" + + +def test_a_grounded_row_is_still_verified(): + report = pv.validate_product( + _well_formed_row("Anil Samba Rava"), "Anil", + category_resolved_deterministically=True, + grounded=True, + require_grounding=True, + ) + + assert report.status == "verified" + + +def test_an_upload_is_not_capped_by_default(): + """DELIBERATE. The spreadsheet path passes no grounded flags at all, and a + shop's own upload IS its evidence. Capping there would relabel every + uploaded row as unchecked while saying nothing true about any of them.""" + report = pv.validate_product( + _well_formed_row("Anil Wheat Vermicelli"), "Anil", + category_resolved_deterministically=True, + grounded=False, + ) + + assert report.status == "verified" diff --git a/tests/test_retail_presence.py b/tests/test_retail_presence.py new file mode 100644 index 0000000..7fa2182 --- /dev/null +++ b/tests/test_retail_presence.py @@ -0,0 +1,613 @@ +"""Is this product on sale, right now, somewhere real? + +WHY THIS EXISTS +--------------- +Open Food Facts answers "does this product exist" for food and nothing else. +`off_bulk` queries only `search.openfoodfacts.org`, so a toothpaste or a +detergent has no product-discovery source at all and its whole catalogue is +language-model output. A live retail lookup is the only signal in the codebase +that means "available right now" rather than "was in a database dump", and it +is the only one that works for non-food. + +THE MEASUREMENT EVERY TEST HERE PROTECTS +---------------------------------------- +Run against the live provider on 2026-09-10: + + "Anil Roasted Short Vermicelli 450g" + -> amazon.in "Anil Vermicelli - Roasted, 450g Pouch" FOUND + "Anil Wheat Vermicelli 12g" + -> the brand's own product page comes back, and NOTHING + states a 12 g pack NOT FOUND + "Colgate MaxFresh 150g" + -> bigbasket "...Toothpaste, 150 g" FOUND + +Note what that middle case means: the generic product page covers 100% of the +product's title tokens. A check that matched on title alone would have +CONFIRMED the fabricated 12 g pack. Only the pack size separates them, which +is why `test_the_title_alone_would_have_confirmed_the_fiction` exists - it +pins the trap rather than the fix. + +The listing titles below are verbatim from those runs, so these tests exercise +the real shapes without touching the network. +""" +from __future__ import annotations + +import pytest + +from app.services import retail_presence as rp + + +AMAZON_450 = "Anil Vermicelli - Roasted, 450g Pouch : Amazon.in: Grocery & Gourmet Foods" +TRADER_450 = "Buy Anil Roasted Short Vermicelli 450Gms online at best price" +BRAND_PAGE = "Anil Wheat Vermicelli | Buy Atta Semiya Online - Anil Foods" +CORP_PAGE = "Wheat Vermicelli - Anil Group - Leading Indian FMCG Company" +BIGBASKET = "Buy Colgate Toothpaste Maxfresh Spicy Red Gel 150 Gm... - bigbasket" + + +# --------------------------------------------------------------------------- +# The size check is the whole thing +# --------------------------------------------------------------------------- + +def test_a_real_pack_is_confirmed(): + matched, _coverage = rp.listing_matches( + AMAZON_450, "Anil", "Anil Roasted Short Vermicelli", "450g") + + assert matched + + +def test_the_invented_pack_is_not_confirmed(): + """THE REPORTED DEFECT. Every one of these really came back from the live + search for "Anil Wheat Vermicelli 12g".""" + for listing in (BRAND_PAGE, CORP_PAGE): + matched, _c = rp.listing_matches(listing, "Anil", "Anil Wheat Vermicelli", "12g") + assert not matched, f"{listing!r} must not confirm a 12g pack" + + +def test_the_title_alone_would_have_confirmed_the_fiction(): + """DO NOT RELAX THE SIZE CHECK. + + This is the trap, pinned deliberately: the brand's generic product page + covers 100% of the product's title tokens. Title matching alone - which is + what "did the search return anything relevant?" amounts to - endorses a + pack size that does not exist. The coverage number being 1.0 here is the + reason `listing_matches` cannot be simplified to a title comparison. + """ + _matched, coverage = rp.listing_matches( + BRAND_PAGE, "Anil", "Anil Wheat Vermicelli", "12g") + + assert coverage == 1.0 + + +def test_a_listing_for_a_different_pack_size_does_not_confirm(): + matched, _c = rp.listing_matches( + AMAZON_450, "Anil", "Anil Roasted Short Vermicelli", "180g") + + assert not matched + + +def test_a_listing_that_states_no_quantity_confirms_no_quantity(): + """A page that never says how big the pack is cannot corroborate a pack + size, however well its title matches.""" + matched, _c = rp.listing_matches( + "Anil Roasted Short Vermicelli - Anil Foods", "Anil", + "Anil Roasted Short Vermicelli", "450g") + + assert not matched + + +# --------------------------------------------------------------------------- +# Title matching is containment, not symmetric similarity +# --------------------------------------------------------------------------- + +def test_shop_furniture_does_not_break_the_match(): + """`symmetric_similarity` scores this real Amazon title 0.445 against the + product name and would reject it. A retail title is a SUPERSET of the + product name wrapped in shop furniture, so the question is containment.""" + matched, coverage = rp.listing_matches( + AMAZON_450, "Anil", "Anil Roasted Short Vermicelli", "450g") + + assert matched + assert coverage >= rp.TITLE_COVERAGE_FLOOR + + +def test_a_retailer_dropping_a_word_still_matches(): + """Amazon lists "Anil Roasted Short Vermicelli" as "Anil Vermicelli - + Roasted", covering 2 of 3 tokens. A tight floor rejects real listings and + buys nothing, because the size check does the discriminating.""" + _m, coverage = rp.listing_matches( + AMAZON_450, "Anil", "Anil Roasted Short Vermicelli", "450g") + + assert 0.6 <= coverage < 1.0 + + +def test_a_different_product_does_not_match(): + matched, _c = rp.listing_matches( + "Buy Anil Samba Rava 500g online", "Anil", + "Anil Roasted Short Vermicelli", "500g") + + assert not matched + + +def test_non_food_works_the_same_way(): + """The point of this module: Open Food Facts has nothing to say about a + toothpaste, and this does.""" + matched, _c = rp.listing_matches(BIGBASKET, "Colgate", "Colgate MaxFresh", "150g") + + assert matched + + +# --------------------------------------------------------------------------- +# Three states, never two +# --------------------------------------------------------------------------- + +def test_a_throttled_provider_is_unknown_not_absent(monkeypatch): + """THE MOST IMPORTANT TEST IN THIS FILE. + + DuckDuckGo 403s under load - the repair script already records that it + "already 403s and the pipeline falls through to Bing". If a throttle were + recorded as `not_found`, one bad afternoon would demote a brand's entire + real catalogue, and the evidence tier that consumes this would drop rows + that are perfectly genuine. + """ + monkeypatch.setattr(rp, "_search", lambda *a, **k: None) + monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None) + + evidence = rp.check_listing("Anil", "Anil Roasted Short Vermicelli", "450g", live=True) + + assert evidence.status == rp.UNKNOWN + assert not evidence.is_found + + +def test_an_empty_result_set_is_not_found_not_unknown(): + """`None` and `[]` are different answers and the caller depends on it: + None is "we could not ask", [] is "we asked and nobody sells this".""" + assert rp._search.__doc__ and "could not ask" in rp._search.__doc__ + + +def test_an_unknown_verdict_is_never_cached(monkeypatch): + """Caching a throttle would turn one rate-limited afternoon into two weeks + of pretending we had checked.""" + written = [] + monkeypatch.setattr(rp, "_connect", lambda: (_ for _ in ()).throw(AssertionError("wrote"))) + + rp.set_cached("Anil", "Anil Wheat Vermicelli", "12g", rp.RetailEvidence(rp.UNKNOWN)) + + assert written == [] + + +def test_ingestion_never_makes_a_live_call(monkeypatch): + """`live` defaults to False. One lookup costs 2.6-5.7s against a provider + that blocks, so it does not belong inside a batch - the backfill script + fills the cache and ingestion reads it.""" + def explode(*_a, **_kw): + raise AssertionError("ingestion must not query the network") + + monkeypatch.setattr(rp, "_search", explode) + monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None) + + evidence = rp.check_listing("Anil", "Anil Wheat Vermicelli", "12g") + + assert evidence.status == rp.UNKNOWN + + +# --------------------------------------------------------------------------- +# Cost control +# --------------------------------------------------------------------------- + +def test_pack_sizes_of_one_product_share_a_product_identity(): + """`cache_key` collapses the size out of the NAME and keeps it as its own + component, so "Anil Vermicelli 180g" and "Anil Vermicelli 450g" are one + product at two sizes rather than two products.""" + a = rp.cache_key("Anil", "Anil Vermicelli 180g", "180g") + b = rp.cache_key("Anil", "Anil Vermicelli 450g", "450g") + c = rp.cache_key("Anil", "Anil Vermicelli", "180g") + + assert a != b, "different packs are different questions" + assert a == c, "the size in the name must not create a second identity" + + +def test_check_many_asks_each_key_once(monkeypatch): + asked = [] + + def fake(brand, title, size, **_kw): + asked.append((brand, title, size)) + return rp.RetailEvidence(rp.NOT_FOUND) + + monkeypatch.setattr(rp, "check_listing", fake) + monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None) + + items = [ + {"brand": "Anil", "product_title": "Anil Vermicelli 180g", "size": "180g"}, + {"brand": "Anil", "product_title": "Anil Vermicelli", "size": "180g"}, + ] + rp.check_many(items, live=False) + + assert len(asked) == 1 + + +# --------------------------------------------------------------------------- +# The brand's own site +# --------------------------------------------------------------------------- + +def _results(*urls): + return [{"href": u, "title": ""} for u in urls] + + +def test_the_brand_domain_is_the_one_naming_the_brand(): + picked = rp._pick_brand_domain( + _results("https://www.maccosmetics.com/", + "https://shop.theanilgroup.com/products/wheat-vermicelli"), + ["anil"], + ) + + assert picked == "theanilgroup.com" + + +def test_a_subdomain_resolves_to_the_registrable_domain(): + """`shop.theanilgroup.com` and `theanilgroup.com` are one site, and the + match in check_listing is against the registrable form.""" + assert rp._registrable("https://shop.theanilgroup.com/products/x") == "theanilgroup.com" + assert rp._registrable("https://foo.britannia.co.in/a") == "britannia.co.in" + + +def test_retailers_and_directories_are_not_brand_sites(): + for url in ("https://www.amazon.in/stores/anil", + "https://en.wikipedia.org/wiki/Anil", + "https://www.indiamart.com/anil-foods/", + "https://www.youtube.com/watch?v=x"): + assert rp._pick_brand_domain(_results(url), ["anil"]) is None, url + + +def test_the_brand_domain_query_uses_a_real_product(monkeypatch): + """A BRAND-ONLY QUERY IS NOT ENOUGH. "Anil official site products" really + returned maccosmetics.com, vertu.com and a YouTube video - "Anil" is a + common personal name, the same ambiguity that put a photo of the actor on + a packet of rava. A product name makes the query specific.""" + seen = [] + + def fake_search(query, *_a, **_kw): + seen.append(query) + return [] + + monkeypatch.setattr(rp, "_search", fake_search) + monkeypatch.setattr(rp, "get_cached_brand_domain", lambda *_a, **_kw: None) + monkeypatch.setattr(rp, "_cache_brand_domain", lambda *_a, **_kw: None) + + rp.resolve_brand_domain("Anil", live=True, + sample_products=["Anil Wheat Vermicelli"]) + + assert seen[0] == "Anil Anil Wheat Vermicelli", ( + "the first query must carry a product name, not the bare brand" + ) + + +def test_a_negative_expires_far_sooner_than_a_positive(): + """One flaky search must not become a permanent fact about a brand. The + first Anil sweep found no brand site; the identical query minutes later + returned shop.theanilgroup.com three times in the top four.""" + assert rp.BRAND_DOMAIN_MISS_TTL_SECONDS < rp.BRAND_DOMAIN_TTL_SECONDS / 100 + + +# --------------------------------------------------------------------------- +# A negative costs a second opinion +# --------------------------------------------------------------------------- + +def test_a_negative_is_confirmed_with_a_second_phrasing(monkeypatch): + """Measured: "Anil Wheat Vermicelli 180g" returned nothing usable on one + run and an Amazon listing for exactly that pack on the next. One query can + confirm a product but cannot deny one.""" + queries = [] + + def fake_search(query, *_a, **_kw): + queries.append(query) + return [] + + monkeypatch.setattr(rp, "_search", fake_search) + monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None) + monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None) + + rp.check_listing("Anil", "Anil Wheat Vermicelli", "180g", + live=True, brand_domain="theanilgroup.com") + + assert len(queries) > 1, "a negative must be retried with another phrasing" + # The site:-scoped query is the last resort for reaching the brand's own + # shop, and it only runs because the open queries reached no shop at all. + assert any("site:theanilgroup.com" in q for q in queries) + + +def test_the_site_scoped_query_is_skipped_once_a_shop_is_reached(monkeypatch): + """The third round trip buys nothing when a retailer has already answered - + it exists to reach the brand's own store when nothing else did.""" + queries = [] + + def fake_search(query, *_a, **_kw): + queries.append(query) + # A shop result that is the wrong pack size: reached, did not match. + return [{"href": "https://www.amazon.in/dp/B01B7BM04E", "title": AMAZON_450}] + + monkeypatch.setattr(rp, "_search", fake_search) + monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None) + monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None) + + evidence = rp.check_listing("Anil", "Anil Roasted Short Vermicelli", "180g", + live=True, brand_domain="theanilgroup.com") + + assert evidence.status == rp.NOT_FOUND + assert not any("site:" in q for q in queries) + + +def test_a_hit_on_the_first_phrasing_costs_only_one_query(monkeypatch): + """Only the negative pays for the retry - a real listing is real however + it was found.""" + queries = [] + + def fake_search(query, *_a, **_kw): + queries.append(query) + return [{"href": "https://www.amazon.in/dp/B01B7BM04E", + "title": AMAZON_450}] + + monkeypatch.setattr(rp, "_search", fake_search) + monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None) + monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None) + + evidence = rp.check_listing("Anil", "Anil Roasted Short Vermicelli", "450g", + live=True, brand_domain=None) + + assert evidence.is_found + assert len(queries) == 1 + + +# --------------------------------------------------------------------------- +# "No shop was reached" is not "no shop sells it" +# --------------------------------------------------------------------------- + +def _shopless_results(): + """What the search actually returns for several real Anil products: recipe + blogs, news, and the brand's own CORPORATE page - no store anywhere.""" + return [ + {"href": "https://www.indianhealthyrecipes.com/rava-semiya-upma/", + "title": "Rava Semiya Upma Recipe"}, + {"href": "https://theanilgroup.com/", "title": "ANIL - Leading Indian FMCG"}, + {"href": "https://en.wikipedia.org/wiki/Vermicelli", "title": "Vermicelli"}, + ] + + +def test_reaching_no_shop_at_all_is_unknown_not_absent(monkeypatch): + """THE CORRECTNESS FIX. + + Recording "no retailer lists this" when not one result was even on a + retailer states something we did not find out. Measured: 2 of 10 sampled + negatives were this case, including `Anil Rava Semiya 200g` and + `Anil Idli Dosa Mix 500g`. + """ + monkeypatch.setattr(rp, "_search", lambda *a, **k: _shopless_results()) + monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None) + monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None) + + evidence = rp.check_listing("Anil", "Anil Rava Semiya", "200g", live=True) + + assert evidence.status == rp.UNKNOWN + assert evidence.shop_results_seen == 0 + + +def test_that_unknown_is_distinguishable_from_a_throttle(monkeypatch): + """The backfill stops the sweep on a throttle and carries on past a + no-shop result, so the two must not look alike. `checked_at` is the tell: + a throttle never got as far as having a time of check.""" + monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None) + monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None) + + monkeypatch.setattr(rp, "_search", lambda *a, **k: _shopless_results()) + no_shop = rp.check_listing("Anil", "Anil Rava Semiya", "200g", live=True) + + monkeypatch.setattr(rp, "_search", lambda *a, **k: None) + throttled = rp.check_listing("Anil", "Anil Rava Semiya", "200g", live=True) + + assert no_shop.status == throttled.status == rp.UNKNOWN + assert no_shop.checked_at is not None + assert throttled.checked_at is None + + +def test_a_shop_that_was_reached_and_did_not_list_it_is_not_found(monkeypatch): + """The other half of the distinction: this one IS an answer, and must stay + `not_found` rather than being softened into `unknown`.""" + monkeypatch.setattr(rp, "_search", lambda *a, **k: [ + {"href": "https://www.flipkart.com/aachi-biryani-masala/p/x", + "title": "Aachi BIRYANI MASALA Price in India"}, + ]) + monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None) + monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None) + + evidence = rp.check_listing("Aachi", "Aachi Biryani Mix", "1kg", live=True) + + assert evidence.status == rp.NOT_FOUND + assert evidence.shop_results_seen >= 1 + + +def test_a_negative_records_the_titles_it_rejected(monkeypatch): + """Without the titles, judging whether a rejection was right means running + the search again - and reach varies run to run, so the answer changes.""" + monkeypatch.setattr(rp, "_search", lambda *a, **k: [ + {"href": "https://www.flipkart.com/x/p/y", + "title": "Aachi BIRYANI MASALA Price in India"}, + ]) + monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None) + monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None) + + evidence = rp.check_listing("Aachi", "Aachi Biryani Mix", "1kg", live=True) + + assert any("BIRYANI MASALA" in t for t in evidence.seen_titles) + + +# --------------------------------------------------------------------------- +# Synonyms +# --------------------------------------------------------------------------- + +def test_a_transliterated_listing_matches(monkeypatch): + """amazon.in sells our "Anil Puttu Mix" as "Anil Puttu Maavu" - maavu is + simply Tamil for the flour. It scored 0.5 coverage and was rejected.""" + matched, _c = rp.listing_matches( + "Amazon.in: Anil Puttu Maavu 500 g", "Anil", "Anil Puttu Mix", "500g") + + assert matched + + +def test_semiya_and_vermicelli_are_the_same_word(): + """This catalogue uses both spellings in its OWN product names.""" + matched, _c = rp.listing_matches( + "Anil Wheat Vermicelli | Buy Atta Semiya 180g Online", "Anil", + "Anil Wheat Vermicelli", "180g") + + assert matched + + +def test_a_synonym_cannot_rescue_a_wrong_pack_size(): + """THE GUARD ON THE SYNONYM MAP. + + Synonyms loosen which words count as the same word. They must never loosen + the pack size, or the map re-opens the exact hole this module was built to + close - the title already covers 100% for the fabricated 12g pack. + """ + matched, coverage = rp.listing_matches( + "Amazon.in: Anil Puttu Maavu 500 g", "Anil", "Anil Puttu Mix", "12g") + + assert coverage == 1.0, "the title matches completely" + assert not matched, "and the size must still reject it" + + +def test_the_synonym_map_stays_small(): + """It is observed-pairs-only by rule, and every entry cites the listing + that forced it. A general thesaurus would start confirming products that + do not exist.""" + assert len(rp._SYNONYMS) <= 12 + + +# --------------------------------------------------------------------------- +# False positives found in real backfill data +# +# These matter more than the false negatives: a wrong FOUND hands 0.85 +# corroboration to a product that may not exist, which is the failure this +# module exists to prevent. A wrong not_found merely withholds credit. +# --------------------------------------------------------------------------- + +CONCATENATED_BLOB = ( + "Ginger Garlic Paste (Pack of 2) | No Peeling, No ChoppingAachi Ginger " + "Garlic Paste 20g - martizo.comAachi Ginger Garlic Paste - Buy at Rs43 " + "Online | Instant ...Aachi Pickles & Instant Foods Buy 1 Get 1 | FREE " + "SHIPPING ...Aachi Ginger Garlic paste - LIFESHARE.IN" +) + + +def test_a_run_on_title_corroborates_nothing(): + """REAL FALSE POSITIVE, from the backfill cache. + + The provider sometimes packs several results into one `title`. This 372-char + blob spans at least four listings from at least two shops, and it produced a + FOUND for "Aachi Ginger Paste 20g" — where the 20g came from martizo.com's + listing of a DIFFERENT product, credited to aachifoods.com. Neither the size + nor the word coverage can be attributed to one product, so it is discarded + rather than parsed. + """ + matched, _c = rp.listing_matches( + CONCATENATED_BLOB, "Aachi", "Aachi Ginger Paste", "20g") + + assert not matched + assert len(CONCATENATED_BLOB) > rp.MAX_LISTING_TITLE_CHARS + + +def test_an_extra_ingredient_makes_it_a_different_product(): + """REAL FALSE POSITIVE, from the backfill cache. + + "Aachi Ginger Paste" and "Aachi Garlic Paste" are both wholly contained in + "Aachi Ginger Garlic Paste" and scored 1.0 coverage against a clean, correct + JioMart listing at the right size. Containment tolerates extra words on the + listing side — right for "Pouch" and "Best Price", wrong for an ingredient. + """ + clean_listing = "Buy Aachi Ginger Garlic Paste 300 g Online - JioMart" + + for wrong_product in ("Aachi Ginger Paste", "Aachi Garlic Paste"): + matched, coverage = rp.listing_matches( + clean_listing, "Aachi", wrong_product, "300g") + assert coverage == 1.0, "the target's words really are all present" + assert not matched, f"{wrong_product} is not ginger-garlic paste" + + +def test_the_ingredient_guard_does_not_break_the_exact_product(): + """The guard fires only on an ingredient the TARGET lacks.""" + matched, _c = rp.listing_matches( + "Buy Aachi Ginger Garlic Paste 300 g Online - JioMart", "Aachi", + "Aachi Ginger Garlic Paste", "300g") + + assert matched + + +def test_shop_furniture_is_not_treated_as_a_distinguishing_word(): + """"Pouch", "Spicy", "Red", "Gel" are not ingredients. If the guard were a + general extra-word rule instead of a curated ingredient set, both controls + below would break.""" + assert rp.listing_matches(AMAZON_450, "Anil", + "Anil Roasted Short Vermicelli", "450g")[0] + assert rp.listing_matches(BIGBASKET, "Colgate", "Colgate MaxFresh", "150g")[0] + + +def test_refresh_actually_re_queries(monkeypatch): + """A SILENT NO-OP THAT LOOKED LIKE A SUCCESSFUL RUN. + + The backfill's --refresh decides which rows to re-ask, then called + check_listing - which read the cache first and returned the very verdict + the caller was replacing. A whole rebuild of Anil reported 46 re-asks, made + no network call, and wrote nothing, because the early return happens before + set_cached. + """ + searched = [] + stale = rp.RetailEvidence(rp.NOT_FOUND, shop_results_seen=1) + + monkeypatch.setattr(rp, "get_cached", lambda *a, **k: stale) + monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None) + monkeypatch.setattr(rp, "_search", lambda q, *a, **k: searched.append(q) or []) + + rp.check_listing("Anil", "Anil Wheat Vermicelli", "180g", live=True) + assert searched == [], "without refresh, a cached verdict short-circuits" + + rp.check_listing("Anil", "Anil Wheat Vermicelli", "180g", live=True, refresh=True) + assert searched, "with refresh, the cache must be bypassed and the search run" + + +# --------------------------------------------------------------------------- +# The listing must be for OUR brand +# --------------------------------------------------------------------------- + +def test_a_competitors_listing_does_not_corroborate_our_product(): + """THE WORST FALSE POSITIVE THIS MODULE HAD, and it was live. + + `normalize_for_match` strips brand tokens from BOTH sides - right for its + original job of matching inside one brand's Open Food Facts corpus, where + the brand is a given. Here the listing can be anyone's, so the comparison + was brand-blind and any competitor's same-shaped product confirmed ours. + + The first line of the Anil backfill was exactly this: amazon.in's + "A1 Naanjil Naattu Masala Instant Pongal Mix, 500g" recorded as evidence + that Anil Pongal Mix 500g is on sale. + """ + for listing, product in [ + ("A1 Naanjil Naattu Masala Home Made Instant Pongal Mix, 500g", + "Anil Pongal Mix"), + ("MTR Rava Idli Mix 500g", "Anil Idli Mix"), + ("Britannia Good Day 200g", "Anil Good Day"), + ]: + matched, _c = rp.listing_matches(listing, "Anil", product, "500g") + assert not matched, f"{listing!r} is not an Anil product" + + +def test_our_own_brand_still_matches(): + """The brand gate must not cost us the real confirmations.""" + assert rp.listing_matches(AMAZON_450, "Anil", + "Anil Roasted Short Vermicelli", "450g")[0] + assert rp.listing_matches(BIGBASKET, "Colgate", "Colgate MaxFresh", "150g")[0] + + +def test_brand_aliases_are_honoured(): + """Reuses the barcode matcher's `brand_matches` and the curated registry, + so "HUL" keeps resolving to Hindustan Unilever and the two gates cannot + drift apart.""" + assert "hindustan unilever" in rp._aliases_for("Hindustan Unilever")