product generation with validation check

This commit is contained in:
sriram
2026-09-10 16:17:28 +05:30
parent 10b24c6348
commit d5a23f6456
19 changed files with 3381 additions and 121 deletions

View File

@@ -0,0 +1,111 @@
"""
Show a human the shop listings a `not_found` verdict rejected, so the
false-negative rate can be measured instead of assumed.
WHY THIS CANNOT BE AUTOMATED
-----------------------------
The obvious way to check a negative is to run the search again and see whether
it matches this time. That is what the first investigation did, and the number
it produced (0 false negatives out of 10) was worthless, because the re-check
used the SAME `listing_matches` that produced the negatives. A matcher cannot
find its own blind spots: it scored `Anil Puttu Maavu` against
`Anil Puttu Mix` as a correct rejection, and *maavu* is simply Tamil for the
flour.
Re-running is also unreliable in its own right - the provider's reach varies
between identical queries minutes apart (7 shop results, then 0).
So the only honest measurement is a person reading the listing titles. This
script puts them in front of one. It makes no network calls and no judgements;
it reads what `check_listing` already stored.
READING THE OUTPUT
------------------
For each rejected row it prints the shop listings that were reached. Classify:
genuine the listings really are other products
e.g. "Aachi BIRYANI MASALA" is not "Aachi Biryani Mix",
and a 450g pouch does not confirm a 180g pack
matcher a listing IS this product and was rejected anyway
e.g. "Anil Puttu Maavu" for "Anil Puttu Mix"
A `matcher` verdict means a synonym is missing (see `_SYNONYMS`) or coverage is
too strict. A `genuine` verdict is what this whole pipeline is for.
Rows with `shop_results_seen == 0` are NOT shown: those are recorded as
`unknown`, not `not_found`, and are not claims about availability at all.
USAGE
python scripts/audit_retail_negatives.py --brand Anil --limit 30
"""
from __future__ import annotations
import argparse
import json
import sqlite3
import sys
from pathlib import Path
from typing import List, Optional
_BACKEND_DIR = Path(__file__).resolve().parent.parent
if str(_BACKEND_DIR) not in sys.path:
sys.path.insert(0, str(_BACKEND_DIR))
from app.services import retail_presence # noqa: E402
def main(argv: Optional[List[str]] = None) -> int:
parser = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--brand", help="Only this brand.")
parser.add_argument("--limit", type=int, default=30)
args = parser.parse_args(argv)
db = retail_presence._DB_PATH
if not db.exists():
print(f"No cache at {db} - run backfill_retail_presence.py first.")
return 1
conn = sqlite3.connect(str(db))
rows = conn.execute(
"SELECT brand, product, size, result_json FROM retail_presence_cache"
).fetchall()
conn.close()
shown = 0
no_titles = 0
for brand, product, size, payload in sorted(rows):
if args.brand and args.brand.strip().lower() not in (brand or "").lower():
continue
data = json.loads(payload)
if data.get("status") != retail_presence.NOT_FOUND:
continue
titles = data.get("seen_titles") or []
if not titles:
# A negative recorded before seen_titles existed. Nothing to audit
# without re-querying, which is exactly what this script refuses to
# do - re-run the backfill with --refresh to repopulate it.
no_titles += 1
continue
if shown >= args.limit:
break
shown += 1
print(f"\n{shown:>3}. {product} [{size}] ({data.get('shop_results_seen', 0)} shop listings reached)")
for title in titles:
print(f" {title}")
print(" -> genuine / matcher ?")
print(f"\n{'-' * 66}")
print(f"shown for audit : {shown}")
if no_titles:
print(f"negatives with no titles : {no_titles}"
f" (recorded before titles were kept - re-run with --refresh)")
print("\nClassify each as `genuine` (really other products) or `matcher`")
print("(a real listing we rejected). matcher / (genuine + matcher) is the")
print("false-negative rate, and it is the number that says whether the")
print("retail signal can be trusted.")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,258 @@
"""
Ask real shops which of our products they actually sell, and cache the answers.
WHY THIS IS A SCRIPT AND NOT A PIPELINE STAGE
----------------------------------------------
Open Food Facts can be asked about a whole brand in one request, which is why
`off_bulk` can run inside ingestion. There is no equivalent for retail: every
product costs its own web search, measured at 2.6-5.7 seconds, against a
provider that 403s under load. Two hundred products is fifteen minutes of
blocking calls and a near-certain throttle partway through - and a throttled
lookup that got recorded as "nobody sells this" would demote real products.
So the querying lives here, offline and paced, and everything at ingestion time
reads the cache (`retail_presence.check_listing(..., live=False)`). Discovery
never blocks and never trips a rate limit mid-preview.
WHAT COUNTS AS A HIT
--------------------
The listing's title AND its pack size must both match. See
`retail_presence`'s docstring for the measurement that makes the size half
non-negotiable: searching for the fabricated "Anil Wheat Vermicelli 12g"
returns the brand's own product page, whose title matches perfectly. Only the
absence of any 12 g mention distinguishes a pack that exists from one that
does not.
USAGE
-----
# Dry run - show what would be asked, ask nothing.
python scripts/backfill_retail_presence.py --brand Anil
# Really query, politely.
python scripts/backfill_retail_presence.py --brand Anil --apply
# Every brand that has a seed catalogue.
python scripts/backfill_retail_presence.py --all --apply --limit 50
`--apply` is required to make any network call at all, mirroring
`repair_brand_images.py`'s dry-run-by-default stance. Re-running is cheap:
anything already cached and inside its TTL is skipped, so an interrupted sweep
resumes where it stopped.
"""
from __future__ import annotations
import argparse
import json
import logging
import sys
import time
from pathlib import Path
from typing import Dict, Iterable, List, Optional, Tuple
_BACKEND_DIR = Path(__file__).resolve().parent.parent
if str(_BACKEND_DIR) not in sys.path:
sys.path.insert(0, str(_BACKEND_DIR))
from app.services import retail_presence # noqa: E402
from app.services.image_corroboration import SIZE_TAIL # noqa: E402
logger = logging.getLogger("backfill_retail_presence")
SEED_DIR = _BACKEND_DIR / "data" / "seed_catalogs"
def _load_seed_products(path: Path) -> Tuple[str, List[Dict[str, str]]]:
"""(brand, [{product_title, size}, ...]) from one seed catalogue."""
try:
data = json.loads(path.read_text(encoding="utf-8-sig"))
except Exception as e: # noqa: BLE001 - one bad file cannot stop the sweep
logger.warning("Could not read %s: %s", path.name, e)
return "", []
products = data.get("products") if isinstance(data, dict) else data
brand = (data.get("brand") if isinstance(data, dict) else "") or ""
out: List[Dict[str, str]] = []
for product in products or []:
title = product.get("title") or product.get("product_name") or ""
if not title:
continue
row_brand = brand or product.get("brand") or ""
# ONE ENTRY PER PACK, because the pack size is the thing being
# verified - "Anil Wheat Vermicelli" exists and "Anil Wheat Vermicelli
# 12g" does not, and a per-product check cannot tell them apart.
#
# Two shapes in the wild: newer seed files carry `size_variants`
# (sometimes as "90g - Rs18"), while the archive leaves that null and
# puts the size on the end of `product_name`.
for size in _sizes_for(product):
out.append({
"brand": row_brand,
"product_title": title,
"size": size,
})
return brand, out
def _sizes_for(product: Dict[str, object]) -> List[str]:
sizes = product.get("size_variants")
if isinstance(sizes, list) and sizes:
cleaned = [str(s).split(" - ")[0].strip() for s in sizes if str(s).strip()]
if cleaned:
return cleaned
match = SIZE_TAIL.search(str(product.get("product_name") or ""))
if match:
return [match.group(0).strip()]
# No size anywhere. Still worth asking whether the product exists at all;
# `listing_matches` skips the size half when the size is blank.
return [""]
def _dedupe(items: Iterable[Dict[str, str]]) -> List[Dict[str, str]]:
"""One entry per (product, size).
`search_key` collapses a trailing pack size out of the NAME, so the three
rows a size explosion produced from one source product share a product
identity and differ only in the size component. A brand with three sizes
of each item therefore costs roughly a third of the naive query count.
"""
seen = set()
out: List[Dict[str, str]] = []
for item in items:
key = retail_presence.cache_key(
item["brand"], item["product_title"], item.get("size", "")
)
if key in seen:
continue
seen.add(key)
out.append(item)
return out
def backfill(items: List[Dict[str, str]], *, apply: bool, limit: Optional[int],
pause: float, refresh: bool = False) -> Dict[str, int]:
stats = {"asked": 0, "found": 0, "not_found": 0, "unknown": 0, "cached": 0,
"shop_reached": 0, "no_shop_reached": 0}
asked = 0
domains: Dict[str, Optional[str]] = {}
for item in items:
brand, title, size = item["brand"], item["product_title"], item.get("size", "")
cached = retail_presence.get_cached(brand, title, size)
# `--refresh` re-asks NOT_FOUND only. A `found` is still true - a shop
# that listed the pack yesterday did list it - but a `not_found`
# recorded before the brand's own site was consulted is not an answer
# to the same question, and the first Anil sweep is full of them.
if cached is not None and not (refresh and not cached.is_found):
stats["cached"] += 1
continue
if limit is not None and asked >= limit:
break
if not apply:
stats["asked"] += 1
asked += 1
logger.info("[dry-run] would ask: %s %s %s", brand, title, size)
continue
# Resolved once per brand and reused, so the brand-site lookup costs
# one search per BRAND rather than one per product.
if brand not in domains:
# Real product names make the query specific enough to find the
# brand's own site - "Anil" alone returns cricketers and airlines.
samples = [i["product_title"] for i in items if i["brand"] == brand][:2]
domains[brand] = retail_presence.resolve_brand_domain(
brand, live=True, sample_products=samples)
logger.info("brand site for %s: %s", brand, domains[brand] or "(none found)")
time.sleep(pause)
evidence = retail_presence.check_listing(
brand, title, size, live=True, brand_domain=domains[brand],
# Without this the cache read inside check_listing hands back
# the very verdict --refresh exists to replace.
refresh=refresh)
stats["asked"] += 1
stats[evidence.status] = stats.get(evidence.status, 0) + 1
asked += 1
symbol = {"found": "OK ", "not_found": "-- ", "unknown": "?? "}[evidence.status]
logger.info("%s %-42s %-8s %s", symbol, title[:42], size, evidence.note())
if evidence.status == retail_presence.NOT_FOUND:
stats["shop_reached"] += 1
# TWO KINDS OF UNKNOWN, AND ONLY ONE OF THEM MEANS STOP.
#
# `checked_at` set - we searched fine, but no result was on a shop.
# That is an ordinary outcome (2 of 10 sampled negatives) and the sweep
# carries on; it just is not an answer about availability.
#
# `checked_at` unset - every phrasing failed to reach the provider,
# which is how a throttle presents. Hammering through hundreds more
# queries after it has started refusing turns one rate limit into a
# longer ban while recording nothing, because unknown is never cached.
if evidence.status == retail_presence.UNKNOWN:
if evidence.checked_at is None:
logger.warning("Provider unreachable - stopping this sweep. "
"Re-run later; cached answers are kept.")
break
stats["no_shop_reached"] += 1
time.sleep(pause)
return stats
def main(argv: Optional[List[str]] = None) -> int:
parser = argparse.ArgumentParser(description=__doc__,
formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--brand", help="Only this brand (matches the seed file's brand).")
parser.add_argument("--all", action="store_true", help="Every seed catalogue.")
parser.add_argument("--apply", action="store_true",
help="Actually query. Without this nothing hits the network.")
parser.add_argument("--limit", type=int, default=None,
help="Stop after this many NEW lookups (cached ones are free).")
parser.add_argument("--refresh", action="store_true",
help="Re-ask entries previously recorded as not-found "
"(a cached hit is still valid and is kept).")
parser.add_argument("--pause", type=float, default=retail_presence.PAUSE_SECONDS,
help=f"Seconds between queries (default {retail_presence.PAUSE_SECONDS}).")
args = parser.parse_args(argv)
logging.basicConfig(level=logging.INFO, format="%(message)s")
if not args.brand and not args.all:
parser.error("give --brand NAME or --all")
items: List[Dict[str, str]] = []
for path in sorted(SEED_DIR.rglob("brand_catalog_*.json")):
brand, products = _load_seed_products(path)
if args.brand and args.brand.strip().lower() not in (brand or "").lower():
continue
items.extend(products)
items = _dedupe(items)
if not items:
logger.error("No products found. Check --brand spelling against the seed files.")
return 1
logger.info("%d distinct product+size combinations to check%s",
len(items), "" if args.apply else " (DRY RUN - use --apply to query)")
stats = backfill(items, apply=args.apply, limit=args.limit, pause=args.pause,
refresh=args.refresh)
logger.info("")
logger.info("already cached : %d", stats["cached"])
logger.info("asked : %d", stats["asked"])
if args.apply:
logger.info(" found : %d", stats.get("found", 0))
logger.info(" not found : %d (a shop was reached and did not list it)",
stats.get("not_found", 0))
logger.info(" no shop reached : %d (searched, but nothing on a shop - NOT an answer)",
stats.get("no_shop_reached", 0))
# Reach is the denominator that makes the found rate mean anything: a
# brand whose products never surface on a shop has no measured rate at
# all, however many not_founds it accumulated before this was tracked.
answered = stats.get("found", 0) + stats.get("not_found", 0)
if answered:
logger.info(" -> of %d ANSWERED, %d%% are listed",
answered, stats.get("found", 0) * 100 // answered)
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -75,6 +75,7 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.infrastructure.settings import DB_HOST, DB_NAME # noqa: E402
from app.services.brand_registry import resolve_parent_brand # noqa: E402
from app.services import image_corroboration # noqa: E402
from app.services.generic_products import OWN_PRODUCTS_BRAND # noqa: E402
from app.services import produce_reference # noqa: E402
from app.services.vector_store import ( # noqa: E402
@@ -339,10 +340,7 @@ def _backup(tables: Dict[str, List[Dict[str, Any]]]) -> Path:
# product.
_SEARCH_CACHE: Dict[Tuple[str, str], List[str]] = {}
_SIZE_TAIL = re.compile(
r"\s+\d+(?:\.\d+)?\s*(?:g|gm|gms|kg|ml|l|ltr|litre|liter|pcs|pc|n|no|nos)\b\.?\s*$",
re.IGNORECASE,
)
_SIZE_TAIL = image_corroboration.SIZE_TAIL
# --- Open Food Facts cross-check -------------------------------------------
@@ -356,74 +354,19 @@ _SIZE_TAIL = re.compile(
# away. If OFF's own record does not mention our brand, the image is somebody
# else's product and is dropped. Nothing else can catch this: the URL is opaque
# digits, so no amount of filename matching would help.
_OFF_CACHE: Dict[str, bool] = {}
_OFF_BARCODE = re.compile(r"/images/products/((?:\d+/)+)")
# "own" and "products" are the BUCKET's name, not a brand's. Left in, the token
# "products" corroborates any URL containing /images/products/ or
# /cdn/shop/products/ - which is every Open*Facts and every Shopify path - so
# `_names_product` waved through 40+ images that named nothing about the item.
# That is how the openbeautyfacts cosmetics photos became the stored image for
# Banana, Orange, Papaya, Guava, Lemon and twenty more.
_BUCKET_TOKENS = frozenset({"own", "products", "product"})
def _brand_tokens(brand: str) -> List[str]:
return [w for w in re.split(r"[^a-z0-9]+", (brand or "").lower())
if len(w) > 2 and w not in _BUCKET_TOKENS]
def _off_product_matches_brand(url: str, brand: str) -> bool:
"""False only when OFF positively says this barcode is another brand."""
if "openfoodfacts.org" not in url:
return True
match = _OFF_BARCODE.search(url)
if not match:
return True
barcode = match.group(1).replace("/", "")
tokens = _brand_tokens(brand)
if not tokens:
return True
key = f"{barcode}:{brand.lower()}"
if key in _OFF_CACHE:
return _OFF_CACHE[key]
verdict = True
try:
import requests
resp = requests.get(
f"https://world.openfoodfacts.org/api/v2/product/{barcode}.json",
params={"fields": "brands,product_name"},
timeout=10,
headers={"User-Agent": "nearle-catalogue-repair/1.0"},
)
if resp.ok:
product = (resp.json() or {}).get("product") or {}
haystack = f"{product.get('brands') or ''} {product.get('product_name') or ''}".lower()
if haystack.strip():
verdict = any(t in haystack for t in tokens)
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
verdict = True
_OFF_CACHE[key] = verdict
return verdict
def _names_product(url: str, product_name: str, brand: str) -> bool:
"""True when the URL itself corroborates the match.
Not a requirement - Zepto and Flipkart serve opaque hashed paths for
perfectly correct images - but a strong signal, so corroborated URLs are
ranked ahead of uncorroborated ones.
"""
lowered = url.lower()
words = [
w for w in re.split(r"[^a-z0-9]+", (product_name or "").lower())
if len(w) > 2 and not re.fullmatch(r"\d+(?:g|kg|ml|l)?", w)
]
return any(w in lowered for w in words + _brand_tokens(brand))
# All of the above now lives in app/services/image_corroboration.py so the
# INGESTION path can use it too. This script had the only working version of
# this logic and no pipeline could import it, which is why the same bad image
# had to be repaired after the fact instead of never being stored.
#
# `names_product` there is STRICTER than the copy that used to be here: it
# requires a distinctive token rather than accepting a brand token, because a
# brand that is also a personal name (Anil) corroborated a press photo of the
# person. See that module's docstring.
_names_product = image_corroboration.names_product
_off_product_matches_brand = image_corroboration.openfacts_product_matches_brand
_brand_tokens = image_corroboration.brand_tokens
_BUCKET_TOKENS = image_corroboration.BUCKET_TOKENS
def _search_key(product_name: str, brand: str) -> Tuple[str, str]: