product generation with validation check
This commit is contained in:
111
scripts/audit_retail_negatives.py
Normal file
111
scripts/audit_retail_negatives.py
Normal file
@@ -0,0 +1,111 @@
|
||||
"""
|
||||
Show a human the shop listings a `not_found` verdict rejected, so the
|
||||
false-negative rate can be measured instead of assumed.
|
||||
|
||||
WHY THIS CANNOT BE AUTOMATED
|
||||
-----------------------------
|
||||
The obvious way to check a negative is to run the search again and see whether
|
||||
it matches this time. That is what the first investigation did, and the number
|
||||
it produced (0 false negatives out of 10) was worthless, because the re-check
|
||||
used the SAME `listing_matches` that produced the negatives. A matcher cannot
|
||||
find its own blind spots: it scored `Anil Puttu Maavu` against
|
||||
`Anil Puttu Mix` as a correct rejection, and *maavu* is simply Tamil for the
|
||||
flour.
|
||||
|
||||
Re-running is also unreliable in its own right - the provider's reach varies
|
||||
between identical queries minutes apart (7 shop results, then 0).
|
||||
|
||||
So the only honest measurement is a person reading the listing titles. This
|
||||
script puts them in front of one. It makes no network calls and no judgements;
|
||||
it reads what `check_listing` already stored.
|
||||
|
||||
READING THE OUTPUT
|
||||
------------------
|
||||
For each rejected row it prints the shop listings that were reached. Classify:
|
||||
|
||||
genuine the listings really are other products
|
||||
e.g. "Aachi BIRYANI MASALA" is not "Aachi Biryani Mix",
|
||||
and a 450g pouch does not confirm a 180g pack
|
||||
matcher a listing IS this product and was rejected anyway
|
||||
e.g. "Anil Puttu Maavu" for "Anil Puttu Mix"
|
||||
|
||||
A `matcher` verdict means a synonym is missing (see `_SYNONYMS`) or coverage is
|
||||
too strict. A `genuine` verdict is what this whole pipeline is for.
|
||||
|
||||
Rows with `shop_results_seen == 0` are NOT shown: those are recorded as
|
||||
`unknown`, not `not_found`, and are not claims about availability at all.
|
||||
|
||||
USAGE
|
||||
python scripts/audit_retail_negatives.py --brand Anil --limit 30
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sqlite3
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import List, Optional
|
||||
|
||||
_BACKEND_DIR = Path(__file__).resolve().parent.parent
|
||||
if str(_BACKEND_DIR) not in sys.path:
|
||||
sys.path.insert(0, str(_BACKEND_DIR))
|
||||
|
||||
from app.services import retail_presence # noqa: E402
|
||||
|
||||
|
||||
def main(argv: Optional[List[str]] = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
parser.add_argument("--brand", help="Only this brand.")
|
||||
parser.add_argument("--limit", type=int, default=30)
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
db = retail_presence._DB_PATH
|
||||
if not db.exists():
|
||||
print(f"No cache at {db} - run backfill_retail_presence.py first.")
|
||||
return 1
|
||||
|
||||
conn = sqlite3.connect(str(db))
|
||||
rows = conn.execute(
|
||||
"SELECT brand, product, size, result_json FROM retail_presence_cache"
|
||||
).fetchall()
|
||||
conn.close()
|
||||
|
||||
shown = 0
|
||||
no_titles = 0
|
||||
for brand, product, size, payload in sorted(rows):
|
||||
if args.brand and args.brand.strip().lower() not in (brand or "").lower():
|
||||
continue
|
||||
data = json.loads(payload)
|
||||
if data.get("status") != retail_presence.NOT_FOUND:
|
||||
continue
|
||||
titles = data.get("seen_titles") or []
|
||||
if not titles:
|
||||
# A negative recorded before seen_titles existed. Nothing to audit
|
||||
# without re-querying, which is exactly what this script refuses to
|
||||
# do - re-run the backfill with --refresh to repopulate it.
|
||||
no_titles += 1
|
||||
continue
|
||||
if shown >= args.limit:
|
||||
break
|
||||
shown += 1
|
||||
print(f"\n{shown:>3}. {product} [{size}] ({data.get('shop_results_seen', 0)} shop listings reached)")
|
||||
for title in titles:
|
||||
print(f" {title}")
|
||||
print(" -> genuine / matcher ?")
|
||||
|
||||
print(f"\n{'-' * 66}")
|
||||
print(f"shown for audit : {shown}")
|
||||
if no_titles:
|
||||
print(f"negatives with no titles : {no_titles}"
|
||||
f" (recorded before titles were kept - re-run with --refresh)")
|
||||
print("\nClassify each as `genuine` (really other products) or `matcher`")
|
||||
print("(a real listing we rejected). matcher / (genuine + matcher) is the")
|
||||
print("false-negative rate, and it is the number that says whether the")
|
||||
print("retail signal can be trusted.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
258
scripts/backfill_retail_presence.py
Normal file
258
scripts/backfill_retail_presence.py
Normal file
@@ -0,0 +1,258 @@
|
||||
"""
|
||||
Ask real shops which of our products they actually sell, and cache the answers.
|
||||
|
||||
WHY THIS IS A SCRIPT AND NOT A PIPELINE STAGE
|
||||
----------------------------------------------
|
||||
Open Food Facts can be asked about a whole brand in one request, which is why
|
||||
`off_bulk` can run inside ingestion. There is no equivalent for retail: every
|
||||
product costs its own web search, measured at 2.6-5.7 seconds, against a
|
||||
provider that 403s under load. Two hundred products is fifteen minutes of
|
||||
blocking calls and a near-certain throttle partway through - and a throttled
|
||||
lookup that got recorded as "nobody sells this" would demote real products.
|
||||
|
||||
So the querying lives here, offline and paced, and everything at ingestion time
|
||||
reads the cache (`retail_presence.check_listing(..., live=False)`). Discovery
|
||||
never blocks and never trips a rate limit mid-preview.
|
||||
|
||||
WHAT COUNTS AS A HIT
|
||||
--------------------
|
||||
The listing's title AND its pack size must both match. See
|
||||
`retail_presence`'s docstring for the measurement that makes the size half
|
||||
non-negotiable: searching for the fabricated "Anil Wheat Vermicelli 12g"
|
||||
returns the brand's own product page, whose title matches perfectly. Only the
|
||||
absence of any 12 g mention distinguishes a pack that exists from one that
|
||||
does not.
|
||||
|
||||
USAGE
|
||||
-----
|
||||
# Dry run - show what would be asked, ask nothing.
|
||||
python scripts/backfill_retail_presence.py --brand Anil
|
||||
|
||||
# Really query, politely.
|
||||
python scripts/backfill_retail_presence.py --brand Anil --apply
|
||||
|
||||
# Every brand that has a seed catalogue.
|
||||
python scripts/backfill_retail_presence.py --all --apply --limit 50
|
||||
|
||||
`--apply` is required to make any network call at all, mirroring
|
||||
`repair_brand_images.py`'s dry-run-by-default stance. Re-running is cheap:
|
||||
anything already cached and inside its TTL is skipped, so an interrupted sweep
|
||||
resumes where it stopped.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Dict, Iterable, List, Optional, Tuple
|
||||
|
||||
_BACKEND_DIR = Path(__file__).resolve().parent.parent
|
||||
if str(_BACKEND_DIR) not in sys.path:
|
||||
sys.path.insert(0, str(_BACKEND_DIR))
|
||||
|
||||
from app.services import retail_presence # noqa: E402
|
||||
from app.services.image_corroboration import SIZE_TAIL # noqa: E402
|
||||
|
||||
logger = logging.getLogger("backfill_retail_presence")
|
||||
|
||||
SEED_DIR = _BACKEND_DIR / "data" / "seed_catalogs"
|
||||
|
||||
|
||||
def _load_seed_products(path: Path) -> Tuple[str, List[Dict[str, str]]]:
|
||||
"""(brand, [{product_title, size}, ...]) from one seed catalogue."""
|
||||
try:
|
||||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
except Exception as e: # noqa: BLE001 - one bad file cannot stop the sweep
|
||||
logger.warning("Could not read %s: %s", path.name, e)
|
||||
return "", []
|
||||
products = data.get("products") if isinstance(data, dict) else data
|
||||
brand = (data.get("brand") if isinstance(data, dict) else "") or ""
|
||||
out: List[Dict[str, str]] = []
|
||||
for product in products or []:
|
||||
title = product.get("title") or product.get("product_name") or ""
|
||||
if not title:
|
||||
continue
|
||||
row_brand = brand or product.get("brand") or ""
|
||||
# ONE ENTRY PER PACK, because the pack size is the thing being
|
||||
# verified - "Anil Wheat Vermicelli" exists and "Anil Wheat Vermicelli
|
||||
# 12g" does not, and a per-product check cannot tell them apart.
|
||||
#
|
||||
# Two shapes in the wild: newer seed files carry `size_variants`
|
||||
# (sometimes as "90g - Rs18"), while the archive leaves that null and
|
||||
# puts the size on the end of `product_name`.
|
||||
for size in _sizes_for(product):
|
||||
out.append({
|
||||
"brand": row_brand,
|
||||
"product_title": title,
|
||||
"size": size,
|
||||
})
|
||||
return brand, out
|
||||
|
||||
|
||||
def _sizes_for(product: Dict[str, object]) -> List[str]:
|
||||
sizes = product.get("size_variants")
|
||||
if isinstance(sizes, list) and sizes:
|
||||
cleaned = [str(s).split(" - ")[0].strip() for s in sizes if str(s).strip()]
|
||||
if cleaned:
|
||||
return cleaned
|
||||
match = SIZE_TAIL.search(str(product.get("product_name") or ""))
|
||||
if match:
|
||||
return [match.group(0).strip()]
|
||||
# No size anywhere. Still worth asking whether the product exists at all;
|
||||
# `listing_matches` skips the size half when the size is blank.
|
||||
return [""]
|
||||
|
||||
|
||||
def _dedupe(items: Iterable[Dict[str, str]]) -> List[Dict[str, str]]:
|
||||
"""One entry per (product, size).
|
||||
|
||||
`search_key` collapses a trailing pack size out of the NAME, so the three
|
||||
rows a size explosion produced from one source product share a product
|
||||
identity and differ only in the size component. A brand with three sizes
|
||||
of each item therefore costs roughly a third of the naive query count.
|
||||
"""
|
||||
seen = set()
|
||||
out: List[Dict[str, str]] = []
|
||||
for item in items:
|
||||
key = retail_presence.cache_key(
|
||||
item["brand"], item["product_title"], item.get("size", "")
|
||||
)
|
||||
if key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
out.append(item)
|
||||
return out
|
||||
|
||||
|
||||
def backfill(items: List[Dict[str, str]], *, apply: bool, limit: Optional[int],
|
||||
pause: float, refresh: bool = False) -> Dict[str, int]:
|
||||
stats = {"asked": 0, "found": 0, "not_found": 0, "unknown": 0, "cached": 0,
|
||||
"shop_reached": 0, "no_shop_reached": 0}
|
||||
asked = 0
|
||||
domains: Dict[str, Optional[str]] = {}
|
||||
for item in items:
|
||||
brand, title, size = item["brand"], item["product_title"], item.get("size", "")
|
||||
|
||||
cached = retail_presence.get_cached(brand, title, size)
|
||||
# `--refresh` re-asks NOT_FOUND only. A `found` is still true - a shop
|
||||
# that listed the pack yesterday did list it - but a `not_found`
|
||||
# recorded before the brand's own site was consulted is not an answer
|
||||
# to the same question, and the first Anil sweep is full of them.
|
||||
if cached is not None and not (refresh and not cached.is_found):
|
||||
stats["cached"] += 1
|
||||
continue
|
||||
|
||||
if limit is not None and asked >= limit:
|
||||
break
|
||||
|
||||
if not apply:
|
||||
stats["asked"] += 1
|
||||
asked += 1
|
||||
logger.info("[dry-run] would ask: %s %s %s", brand, title, size)
|
||||
continue
|
||||
|
||||
# Resolved once per brand and reused, so the brand-site lookup costs
|
||||
# one search per BRAND rather than one per product.
|
||||
if brand not in domains:
|
||||
# Real product names make the query specific enough to find the
|
||||
# brand's own site - "Anil" alone returns cricketers and airlines.
|
||||
samples = [i["product_title"] for i in items if i["brand"] == brand][:2]
|
||||
domains[brand] = retail_presence.resolve_brand_domain(
|
||||
brand, live=True, sample_products=samples)
|
||||
logger.info("brand site for %s: %s", brand, domains[brand] or "(none found)")
|
||||
time.sleep(pause)
|
||||
evidence = retail_presence.check_listing(
|
||||
brand, title, size, live=True, brand_domain=domains[brand],
|
||||
# Without this the cache read inside check_listing hands back
|
||||
# the very verdict --refresh exists to replace.
|
||||
refresh=refresh)
|
||||
stats["asked"] += 1
|
||||
stats[evidence.status] = stats.get(evidence.status, 0) + 1
|
||||
asked += 1
|
||||
symbol = {"found": "OK ", "not_found": "-- ", "unknown": "?? "}[evidence.status]
|
||||
logger.info("%s %-42s %-8s %s", symbol, title[:42], size, evidence.note())
|
||||
if evidence.status == retail_presence.NOT_FOUND:
|
||||
stats["shop_reached"] += 1
|
||||
|
||||
# TWO KINDS OF UNKNOWN, AND ONLY ONE OF THEM MEANS STOP.
|
||||
#
|
||||
# `checked_at` set - we searched fine, but no result was on a shop.
|
||||
# That is an ordinary outcome (2 of 10 sampled negatives) and the sweep
|
||||
# carries on; it just is not an answer about availability.
|
||||
#
|
||||
# `checked_at` unset - every phrasing failed to reach the provider,
|
||||
# which is how a throttle presents. Hammering through hundreds more
|
||||
# queries after it has started refusing turns one rate limit into a
|
||||
# longer ban while recording nothing, because unknown is never cached.
|
||||
if evidence.status == retail_presence.UNKNOWN:
|
||||
if evidence.checked_at is None:
|
||||
logger.warning("Provider unreachable - stopping this sweep. "
|
||||
"Re-run later; cached answers are kept.")
|
||||
break
|
||||
stats["no_shop_reached"] += 1
|
||||
|
||||
time.sleep(pause)
|
||||
return stats
|
||||
|
||||
|
||||
def main(argv: Optional[List[str]] = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
parser.add_argument("--brand", help="Only this brand (matches the seed file's brand).")
|
||||
parser.add_argument("--all", action="store_true", help="Every seed catalogue.")
|
||||
parser.add_argument("--apply", action="store_true",
|
||||
help="Actually query. Without this nothing hits the network.")
|
||||
parser.add_argument("--limit", type=int, default=None,
|
||||
help="Stop after this many NEW lookups (cached ones are free).")
|
||||
parser.add_argument("--refresh", action="store_true",
|
||||
help="Re-ask entries previously recorded as not-found "
|
||||
"(a cached hit is still valid and is kept).")
|
||||
parser.add_argument("--pause", type=float, default=retail_presence.PAUSE_SECONDS,
|
||||
help=f"Seconds between queries (default {retail_presence.PAUSE_SECONDS}).")
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
||||
|
||||
if not args.brand and not args.all:
|
||||
parser.error("give --brand NAME or --all")
|
||||
|
||||
items: List[Dict[str, str]] = []
|
||||
for path in sorted(SEED_DIR.rglob("brand_catalog_*.json")):
|
||||
brand, products = _load_seed_products(path)
|
||||
if args.brand and args.brand.strip().lower() not in (brand or "").lower():
|
||||
continue
|
||||
items.extend(products)
|
||||
|
||||
items = _dedupe(items)
|
||||
if not items:
|
||||
logger.error("No products found. Check --brand spelling against the seed files.")
|
||||
return 1
|
||||
|
||||
logger.info("%d distinct product+size combinations to check%s",
|
||||
len(items), "" if args.apply else " (DRY RUN - use --apply to query)")
|
||||
stats = backfill(items, apply=args.apply, limit=args.limit, pause=args.pause,
|
||||
refresh=args.refresh)
|
||||
logger.info("")
|
||||
logger.info("already cached : %d", stats["cached"])
|
||||
logger.info("asked : %d", stats["asked"])
|
||||
if args.apply:
|
||||
logger.info(" found : %d", stats.get("found", 0))
|
||||
logger.info(" not found : %d (a shop was reached and did not list it)",
|
||||
stats.get("not_found", 0))
|
||||
logger.info(" no shop reached : %d (searched, but nothing on a shop - NOT an answer)",
|
||||
stats.get("no_shop_reached", 0))
|
||||
# Reach is the denominator that makes the found rate mean anything: a
|
||||
# brand whose products never surface on a shop has no measured rate at
|
||||
# all, however many not_founds it accumulated before this was tracked.
|
||||
answered = stats.get("found", 0) + stats.get("not_found", 0)
|
||||
if answered:
|
||||
logger.info(" -> of %d ANSWERED, %d%% are listed",
|
||||
answered, stats.get("found", 0) * 100 // answered)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -75,6 +75,7 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from app.infrastructure.settings import DB_HOST, DB_NAME # noqa: E402
|
||||
from app.services.brand_registry import resolve_parent_brand # noqa: E402
|
||||
from app.services import image_corroboration # noqa: E402
|
||||
from app.services.generic_products import OWN_PRODUCTS_BRAND # noqa: E402
|
||||
from app.services import produce_reference # noqa: E402
|
||||
from app.services.vector_store import ( # noqa: E402
|
||||
@@ -339,10 +340,7 @@ def _backup(tables: Dict[str, List[Dict[str, Any]]]) -> Path:
|
||||
# product.
|
||||
_SEARCH_CACHE: Dict[Tuple[str, str], List[str]] = {}
|
||||
|
||||
_SIZE_TAIL = re.compile(
|
||||
r"\s+\d+(?:\.\d+)?\s*(?:g|gm|gms|kg|ml|l|ltr|litre|liter|pcs|pc|n|no|nos)\b\.?\s*$",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_SIZE_TAIL = image_corroboration.SIZE_TAIL
|
||||
|
||||
|
||||
# --- Open Food Facts cross-check -------------------------------------------
|
||||
@@ -356,74 +354,19 @@ _SIZE_TAIL = re.compile(
|
||||
# away. If OFF's own record does not mention our brand, the image is somebody
|
||||
# else's product and is dropped. Nothing else can catch this: the URL is opaque
|
||||
# digits, so no amount of filename matching would help.
|
||||
_OFF_CACHE: Dict[str, bool] = {}
|
||||
_OFF_BARCODE = re.compile(r"/images/products/((?:\d+/)+)")
|
||||
|
||||
|
||||
# "own" and "products" are the BUCKET's name, not a brand's. Left in, the token
|
||||
# "products" corroborates any URL containing /images/products/ or
|
||||
# /cdn/shop/products/ - which is every Open*Facts and every Shopify path - so
|
||||
# `_names_product` waved through 40+ images that named nothing about the item.
|
||||
# That is how the openbeautyfacts cosmetics photos became the stored image for
|
||||
# Banana, Orange, Papaya, Guava, Lemon and twenty more.
|
||||
_BUCKET_TOKENS = frozenset({"own", "products", "product"})
|
||||
|
||||
|
||||
def _brand_tokens(brand: str) -> List[str]:
|
||||
return [w for w in re.split(r"[^a-z0-9]+", (brand or "").lower())
|
||||
if len(w) > 2 and w not in _BUCKET_TOKENS]
|
||||
|
||||
|
||||
def _off_product_matches_brand(url: str, brand: str) -> bool:
|
||||
"""False only when OFF positively says this barcode is another brand."""
|
||||
if "openfoodfacts.org" not in url:
|
||||
return True
|
||||
match = _OFF_BARCODE.search(url)
|
||||
if not match:
|
||||
return True
|
||||
barcode = match.group(1).replace("/", "")
|
||||
tokens = _brand_tokens(brand)
|
||||
if not tokens:
|
||||
return True
|
||||
|
||||
key = f"{barcode}:{brand.lower()}"
|
||||
if key in _OFF_CACHE:
|
||||
return _OFF_CACHE[key]
|
||||
|
||||
verdict = True
|
||||
try:
|
||||
import requests
|
||||
resp = requests.get(
|
||||
f"https://world.openfoodfacts.org/api/v2/product/{barcode}.json",
|
||||
params={"fields": "brands,product_name"},
|
||||
timeout=10,
|
||||
headers={"User-Agent": "nearle-catalogue-repair/1.0"},
|
||||
)
|
||||
if resp.ok:
|
||||
product = (resp.json() or {}).get("product") or {}
|
||||
haystack = f"{product.get('brands') or ''} {product.get('product_name') or ''}".lower()
|
||||
if haystack.strip():
|
||||
verdict = any(t in haystack for t in tokens)
|
||||
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
|
||||
verdict = True
|
||||
|
||||
_OFF_CACHE[key] = verdict
|
||||
return verdict
|
||||
|
||||
|
||||
def _names_product(url: str, product_name: str, brand: str) -> bool:
|
||||
"""True when the URL itself corroborates the match.
|
||||
|
||||
Not a requirement - Zepto and Flipkart serve opaque hashed paths for
|
||||
perfectly correct images - but a strong signal, so corroborated URLs are
|
||||
ranked ahead of uncorroborated ones.
|
||||
"""
|
||||
lowered = url.lower()
|
||||
words = [
|
||||
w for w in re.split(r"[^a-z0-9]+", (product_name or "").lower())
|
||||
if len(w) > 2 and not re.fullmatch(r"\d+(?:g|kg|ml|l)?", w)
|
||||
]
|
||||
return any(w in lowered for w in words + _brand_tokens(brand))
|
||||
# All of the above now lives in app/services/image_corroboration.py so the
|
||||
# INGESTION path can use it too. This script had the only working version of
|
||||
# this logic and no pipeline could import it, which is why the same bad image
|
||||
# had to be repaired after the fact instead of never being stored.
|
||||
#
|
||||
# `names_product` there is STRICTER than the copy that used to be here: it
|
||||
# requires a distinctive token rather than accepting a brand token, because a
|
||||
# brand that is also a personal name (Anil) corroborated a press photo of the
|
||||
# person. See that module's docstring.
|
||||
_names_product = image_corroboration.names_product
|
||||
_off_product_matches_brand = image_corroboration.openfacts_product_matches_brand
|
||||
_brand_tokens = image_corroboration.brand_tokens
|
||||
_BUCKET_TOKENS = image_corroboration.BUCKET_TOKENS
|
||||
|
||||
|
||||
def _search_key(product_name: str, brand: str) -> Tuple[str, str]:
|
||||
|
||||
Reference in New Issue
Block a user