From c0601b65fe733817551c5246d871cbc86e5733e5 Mon Sep 17 00:00:00 2001 From: sriram Date: Tue, 29 Sep 2026 16:42:49 +0530 Subject: [PATCH] Brand Discovery-LLM Updates --- .env.example | 12 + app/api/routers/brand_discovery.py | 63 +++ app/infrastructure/settings.py | 19 + app/services/brand_discovery.py | 92 +++- app/services/web_discovery/__init__.py | 37 ++ app/services/web_discovery/brands.py | 118 +++++ app/services/web_discovery/cache.py | 82 ++++ app/services/web_discovery/discover.py | 225 +++++++++ app/services/web_discovery/jobs.py | 225 +++++++++ app/services/web_discovery/listings.py | 292 +++++++++++ app/services/web_discovery/provenance.py | 117 +++++ app/services/web_discovery/search.py | 105 ++++ .../off_brand_corpus/reckitt_benckiser.json | 9 +- data/cache/web_discovery/search_cache.db | Bin 0 -> 147456 bytes tests/test_web_discovery.py | 453 ++++++++++++++++++ tests/test_web_discovery_api.py | 127 +++++ 16 files changed, 1971 insertions(+), 5 deletions(-) create mode 100644 app/services/web_discovery/__init__.py create mode 100644 app/services/web_discovery/brands.py create mode 100644 app/services/web_discovery/cache.py create mode 100644 app/services/web_discovery/discover.py create mode 100644 app/services/web_discovery/jobs.py create mode 100644 app/services/web_discovery/listings.py create mode 100644 app/services/web_discovery/provenance.py create mode 100644 app/services/web_discovery/search.py create mode 100644 data/cache/web_discovery/search_cache.db create mode 100644 tests/test_web_discovery.py create mode 100644 tests/test_web_discovery_api.py diff --git a/.env.example b/.env.example index d64c16a..3a7520a 100644 --- a/.env.example +++ b/.env.example @@ -294,6 +294,18 @@ BRAND_SYNC_INTERVAL_SECONDS=300 # Food Facts runs first and is never subject to it. #BRAND_DISCOVERY_DEADLINE_SECONDS=300 +# Brand Discovery's optional "Web & retail listings" source, for brands Open +# Food Facts does not carry. Reads search-engine results that point at retailer +# product pages (Blinkit, Zepto, Instamart, BigBasket, JioMart, Amazon, +# Flipkart...) - never the pages themselves, never a language model. Uses the +# Google Custom Search API when USE_GOOGLE_CSE=true, else DuckDuckGo. +#WEB_DISCOVERY_ENABLED=true +#WEB_DISCOVERY_MAX_QUERIES=80 +#WEB_DISCOVERY_PAUSE_SECONDS=2.0 +#WEB_DISCOVERY_RESULTS_PER_QUERY=20 +#WEB_DISCOVERY_CACHE_DAYS=7 +#WEB_DISCOVERY_DIR=/app/data/cache/web_discovery + USE_S3=true S3_ACCESS_KEY=your-do-spaces-key diff --git a/app/api/routers/brand_discovery.py b/app/api/routers/brand_discovery.py index b09b680..66d6e70 100644 --- a/app/api/routers/brand_discovery.py +++ b/app/api/routers/brand_discovery.py @@ -33,6 +33,7 @@ from app.infrastructure.settings import ( BATCH_MAX_TOTAL_ROWS, BRAND_DISCOVERY_DEADLINE_SECONDS, BRAND_DISCOVERY_MAX_PRODUCTS, + WEB_DISCOVERY_ENABLED, ) from app.services import active_brands, brand_discovery @@ -72,6 +73,11 @@ class DiscoveryPreviewRequest(BaseModel): deadline_seconds: float = Field(default=90.0, ge=0.0, le=BRAND_DISCOVERY_DEADLINE_SECONDS) refresh_corpus: bool = False + # Web & retail listings. Reads a FINISHED web-discovery job (start one at + # POST /web-jobs first); off by default, and off means the preview is + # exactly what it was before this source existed. + use_web: bool = False + web_job_id: Optional[str] = None class DiscoveredProductIn(BaseModel): @@ -93,6 +99,19 @@ class DiscoveredProductIn(BaseModel): fssai_license: Optional[str] = None barcode: Optional[str] = None image_url: Optional[str] = None + # Web-found rows only: the listings that justified the product. Not + # written by the pipeline; recorded afterwards into field_sources by + # web_discovery.provenance, which also fills a blank price_range from them. + listings: List[Dict[str, Any]] = Field(default_factory=list) + price_range: Optional[str] = None + retailer_count: int = 0 + + +class WebJobRequest(BaseModel): + brand: str + # Re-search even when a finished job for this brand is less than a day old. + # Answers already cached are still reused, so this is cheap. + refresh: bool = False class DiscoveryIngestRequest(BaseModel): @@ -131,6 +150,8 @@ async def preview_brand_discovery(payload: DiscoveryPreviewRequest) -> Dict[str, use_llm=payload.use_llm, require_evidence=payload.require_evidence, refresh_corpus=payload.refresh_corpus, + use_web=payload.use_web and WEB_DISCOVERY_ENABLED, + web_job_id=payload.web_job_id, ) except ValueError as exc: raise HTTPException(status_code=400, detail=str(exc)) from exc @@ -220,9 +241,51 @@ async def ingest_brand_discovery(payload: DiscoveryIngestRequest) -> batch_commo "press Resume on it once the current batch finishes." ), ) + # Web-found rows: record their retailer listings once the batch has stored + # them. Keyed by CSV position, which is how stage 11 reports source rows. + web_entries = { + index: {"listings": item.listings, "price_range": item.price_range or "", + "retailer_count": item.retailer_count} + for index, item in enumerate(payload.products) if item.listings + } + if web_entries: + from app.services.web_discovery import provenance + provenance.watch(manifest.batch_id, brand, web_entries) return batch_common.to_out(manifest) +# --------------------------------------------------------------------------- +# Web & retail listings - a background search the panel polls +# --------------------------------------------------------------------------- +def _require_web() -> None: + if not WEB_DISCOVERY_ENABLED: + raise HTTPException(status_code=404, detail="Web discovery is switched off (WEB_DISCOVERY_ENABLED).") + + +@router.post("/web-jobs", status_code=status.HTTP_202_ACCEPTED, + dependencies=[Depends(require_admin)]) +def start_web_job(payload: WebJobRequest) -> Dict[str, Any]: + """Start searching retailer listings for a brand, or return the job that is + already running or finished within the last day. Poll GET /web-jobs/{id}.""" + _require_web() + from app.services.web_discovery import jobs as web_jobs + try: + job = web_jobs.start_job(payload.brand, refresh=payload.refresh) + except ValueError as exc: + raise HTTPException(status_code=400, detail=str(exc)) from exc + return job.summary() + + +@router.get("/web-jobs/{job_id}", dependencies=[Depends(require_admin)]) +def get_web_job(job_id: str) -> Dict[str, Any]: + _require_web() + from app.services.web_discovery import jobs as web_jobs + job = web_jobs.get_job(job_id) + if job is None: + raise HTTPException(status_code=404, detail="No web discovery job with that id.") + return job.summary() + + # --------------------------------------------------------------------------- # The ACTIVE_BRANDS message, in one place # --------------------------------------------------------------------------- diff --git a/app/infrastructure/settings.py b/app/infrastructure/settings.py index 6f0036f..99a6718 100644 --- a/app/infrastructure/settings.py +++ b/app/infrastructure/settings.py @@ -347,6 +347,25 @@ BRAND_DISCOVERY_DEADLINE_SECONDS = float( os.getenv("BRAND_DISCOVERY_DEADLINE_SECONDS", "300") ) +# Web & retail listing discovery (app/services/web_discovery/) - the optional +# Brand Discovery source for brands Open Food Facts does not carry. It reads +# ONLY search-engine results (title, URL, snippet) that point at Indian +# retailers' product pages; it never fetches a retailer page and never asks a +# language model. A product is kept only when a real listing names the brand +# and states a pack size. Off in a request unless the admin ticks it. +WEB_DISCOVERY_ENABLED = _bool("WEB_DISCOVERY_ENABLED", "true") +# Searches per brand job. A parent brand fans out to its sub-brands (Reckitt -> +# Dettol, Harpic, Lizol, ...) times the retailers, so this is the cost ceiling. +WEB_DISCOVERY_MAX_QUERIES = int(os.getenv("WEB_DISCOVERY_MAX_QUERIES", "80")) +# Seconds between two LIVE searches (cached ones are free). DuckDuckGo throttles +# a burst; a throttled search is "could not ask", never "nothing found". +WEB_DISCOVERY_PAUSE_SECONDS = float(os.getenv("WEB_DISCOVERY_PAUSE_SECONDS", "2.0")) +WEB_DISCOVERY_RESULTS_PER_QUERY = int(os.getenv("WEB_DISCOVERY_RESULTS_PER_QUERY", "20")) +# How long a search answer is reused. An EMPTY answer is kept one day only: +# search reach is unstable, and a flaky empty page must not become a week-long fact. +WEB_DISCOVERY_CACHE_DAYS = float(os.getenv("WEB_DISCOVERY_CACHE_DAYS", "7")) +WEB_DISCOVERY_DIR = _dir("WEB_DISCOVERY_DIR", DATA_DIR / "cache" / "web_discovery") + # --------------------------------------------------------------------------- # S3 / DigitalOcean Spaces (product image storage) - optional # --------------------------------------------------------------------------- diff --git a/app/services/brand_discovery.py b/app/services/brand_discovery.py index 4b7ea02..49c1269 100644 --- a/app/services/brand_discovery.py +++ b/app/services/brand_discovery.py @@ -227,8 +227,23 @@ class DiscoveredProduct: confidence: float = 0.0 matches_existing: Optional[str] = None notes: List[str] = field(default_factory=list) + # Web & retail listings (app/services/web_discovery). Empty unless that + # source found this product, and then shown in the preview and handed back + # on ingest for the provenance post-pass - never written to the CSV. + listings: List[Dict[str, Any]] = field(default_factory=list) + price_range: str = "" + retailer_count: int = 0 def as_preview(self) -> Dict[str, Any]: + body = self._preview_body() + # Only for rows the web source found, so a preview without it is + # exactly what it was before that source existed. + if self.listings: + body.update({"listings": list(self.listings), "price_range": self.price_range, + "retailer_count": self.retailer_count}) + return body + + def _preview_body(self) -> Dict[str, Any]: return { "brand": self.brand, "product_name": self.product_name, @@ -918,6 +933,8 @@ def discover_brand_products( use_store: bool = True, require_evidence: bool = True, refresh_corpus: bool = False, + use_web: bool = False, + web_job_id: Optional[str] = None, ) -> DiscoveryResult: """Discover a brand's products. Reads the catalog; writes nothing. @@ -987,7 +1004,15 @@ def discover_brand_products( "brand_registry.BRAND_STORE_DOMAINS and run " "scripts/backfill_brand_stores.py to populate a real catalogue." ) - if not off_candidates and not store_candidates: + # Web & retail listings: what a finished web-discovery job found. Read, + # never run, here - the job takes minutes and the panel starts it first. + web_candidates: List[Dict[str, Any]] = [] + if use_web: + web_candidates = _from_web(brand, web_job_id, warnings) + # With the web source on, this warning is only true when the model is on + # too; without it, it is left exactly as it has always been. + if (not off_candidates and not store_candidates and not web_candidates + and (use_llm or not use_web)): warnings.append( "Every row below rests on the language model alone - review them " "individually." @@ -1018,12 +1043,17 @@ def discover_brand_products( # STORE FIRST, then Open Food Facts, then the model. The merge fills a # blank from whatever comes later, so order IS precedence: the brand's own # shop outranks a crowd-sourced database, which outranks a guess. - for candidate in store_candidates + off_candidates + llm_candidates: + for candidate in store_candidates + off_candidates + web_candidates + llm_candidates: title = candidate["title"] key = _normalise_title(brand, title) target = None if key: for index, existing in enumerate(keys): + # Two web rows were already told apart by web_discovery's + # stricter rule (a variant word on one side only - "Citrus" + # vs "Floral" scores 0.857 here, over the 0.85 floor). + if candidate.get("source") == "web" and merged[index].get("sources") == ["web"]: + continue if not _same_product(key, existing, title, merged[index]["title"]): continue held = merged[index].get("size") @@ -1050,6 +1080,16 @@ def discover_brand_products( entry["sizes"] = candidate["sizes"] if not entry.get("providers") and candidate.get("providers"): entry["providers"] = candidate["providers"] + if candidate.get("listings"): + # A web listing joining an OFF / store row: keep the evidence and + # the retailers, the other source keeps the name and the barcode. + entry["listings"] = (entry.get("listings") or []) + list(candidate["listings"]) + entry["retailer_count"] = max(int(entry.get("retailer_count") or 0), + int(candidate.get("retailer_count") or 0)) + entry["providers"] = list(dict.fromkeys( + list(candidate.get("providers") or []) + list(entry.get("providers") or []))) + if not entry.get("price_range") and candidate.get("price_range"): + entry["price_range"] = candidate["price_range"] # EVERY PACK OF ONE PRODUCT MUST BE NAMED THE SAME WAY. # @@ -1154,6 +1194,8 @@ def discover_brand_products( ) if product is None: continue + if "web" in product.sources: + _apply_web(product, candidate) if (require_evidence and "off" not in product.sources and "store" not in product.sources and not product.evidence): dropped += 1 @@ -1173,6 +1215,8 @@ def discover_brand_products( "with_barcode": sum(1 for p in products if p.barcode), "dropped_without_evidence": dropped, } + if use_web: + counts["from_web"] = sum(1 for p in products if "web" in p.sources) return DiscoveryResult( brand=brand, @@ -1186,6 +1230,50 @@ def discover_brand_products( ) +# --------------------------------------------------------------------------- +# Web & retail listings - the optional fourth source +# --------------------------------------------------------------------------- +# A product a real Indian retailer is listing now, found from search results +# only (app/services/web_discovery). Two retailers or more is strong evidence; +# one is real but thin, so it is shown and left unticked. +WEB_CONFIDENCE_MULTI = 0.9 +WEB_CONFIDENCE_SINGLE = 0.75 + + +def _from_web(brand: str, job_id: Optional[str], warnings: List[str]) -> List[Dict[str, Any]]: + from app.services.web_discovery import jobs as web_jobs + + job = web_jobs.candidates_for(brand, job_id) + if job is None: + warnings.append( + "Web & retail listings have not been searched for this brand yet - " + "tick the box and run Discover again to start the search." + ) + return [] + kept = len(job.candidates) + note = (f"Web & retail listings: {kept} product(s) from {job.listings_kept} retailer " + f"listing(s), {job.queries_done} of {job.queries_total} searches via {job.backend}.") + if job.status != web_jobs.DONE: + note += f" Incomplete: {job.detail}" + warnings.append(note) + return [dict(c) for c in job.candidates] + + +def _apply_web(product: DiscoveredProduct, candidate: Dict[str, Any]) -> None: + product.listings = list(candidate.get("listings") or []) + product.price_range = candidate.get("price_range") or "" + product.retailer_count = int(candidate.get("retailer_count") or 0) + # A live retailer listing outranks our own earlier output ("catalog") and a + # hardcoded list ("registry"); only Open Food Facts stays ahead of it. + if product.evidence in (None, "catalog", "registry"): + product.evidence = "retail" + if product.sources == ["web"]: + product.confidence = (WEB_CONFIDENCE_MULTI if product.retailer_count >= 2 + else WEB_CONFIDENCE_SINGLE) + retailers = ", ".join(product.providers[:product.retailer_count] or product.providers) + product.notes.append(f"listed by {retailers}") + + # --------------------------------------------------------------------------- # Serialisation - the bridge itself # --------------------------------------------------------------------------- diff --git a/app/services/web_discovery/__init__.py b/app/services/web_discovery/__init__.py new file mode 100644 index 0000000..74ffbd9 --- /dev/null +++ b/app/services/web_discovery/__init__.py @@ -0,0 +1,37 @@ +"""Find a brand's real products from web search results that point at Indian retailers. + +WHY THIS EXISTS +--------------- +Brand Discovery's sources are Open Food Facts, the brand's own storefront and a +language model. For a non-food brand such as Reckitt Benckiser the first two are +empty - OFF is a FOOD database and no storefront is registered - which left the +model inventing products no shop has ever sold. This package is a fourth +source that only reports what a real retailer is listing. + +WHAT IT READS, AND WHAT IT NEVER DOES +------------------------------------- +It reads the search ENGINE's answer - result title, URL and snippet - for +queries like `"Dettol site:blinkit.com"`. It never fetches a Blinkit, Zepto, +Instamart, Amazon or Flipkart page, never calls their private APIs, and never +asks a language model anything. A product is kept only when + + 1. the result URL is a PRODUCT page on an allowlisted retailer (not a search, + category or brand page) - listings.is_product_url; + 2. the listing title names the brand or one of its sub-brands, as a word; and + 3. the title states an explicit pack size. No size, no product: the size is + what separates a pack that exists from an invented one (see + retail_presence's "Anil Wheat Vermicelli 12g" measurement). + +MODULES +------- + search.py DuckDuckGo, or the Google Custom Search API when configured + listings.py retailer + product-URL detection, title cleaning, size, price + brands.py a parent brand -> the sub-brands it trades under + cache.py SQLite cache of search answers (a throttle is never cached) + discover.py queries -> listings -> one candidate per (product, pack size) + jobs.py one background worker; the admin panel polls its progress + provenance.py after an ingest batch: listing URLs into field_sources + +Nothing here is imported by the 11-stage pipeline, and nothing here runs +unless an admin ticks "Web & retail listings" in the Brand Discovery tab. +""" diff --git a/app/services/web_discovery/brands.py b/app/services/web_discovery/brands.py new file mode 100644 index 0000000..fd34e56 --- /dev/null +++ b/app/services/web_discovery/brands.py @@ -0,0 +1,118 @@ +"""A brand name -> the names its products are actually sold under. + +Nobody shops for "Reckitt Benckiser". Its products are listed as Dettol, +Harpic, Lizol, Mortein, Durex..., so a search for the parent finds almost +nothing. `brand_registry.BRAND_ALIASES` already knows the family, but spells +it with an abbreviation - "rb dettol" -> "reckitt benckiser" - and +`get_known_sub_brands` only accepts aliases that START with the parent name, +so for Reckitt it returns []. This reads the same registry without changing it. + +WHICH WORDS COUNT AS "THE BRAND" +-------------------------------- +A listing is kept only if it names the brand (listings.names_brand). For an +abbreviation alias the sub-brand itself is the brand word: "rb dettol" -> +"dettol". For an alias that starts with the parent ("nestle cerelac", +"amul butter") the parent always counts, and the rest counts too ONLY when it +is a product-line name rather than a kind of food: "Cerelac" is sold without +"Nestle" in the title, but "butter" is not a brand, and accepting it would +let another company's butter in under Amul. A rest is a line name when it is +not a category keyword, is at least 5 characters, and belongs to exactly one +parent in the registry. +""" +from __future__ import annotations + +import re +from collections import defaultdict +from dataclasses import dataclass +from typing import Dict, List, Set, Tuple + +from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand + + +@dataclass(frozen=True) +class Target: + """One name to search for, and the words a listing must lead with to count.""" + query: str # what goes before `site:` - "Dettol", "Amul Butter" + brand_terms: Tuple[str, ...] # "dettol"; or ("nestle", "cerelac") + + +def _words(text: str) -> List[str]: + return re.findall(r"[a-z0-9]+", (text or "").lower()) + + +def _is_abbreviation(token: str, parent: str) -> bool: + """"rb" for Reckitt Benckiser, "hul" for Hindustan Unilever, "jnj" for + Johnson & Johnson. Short, not itself a word of the parent, and starting + with the parent's first letter - which is what keeps "kit kat" (Nestle) + from being read as the abbreviation "kit" plus a sub-brand "kat".""" + parent_words = _words(parent) + return (bool(parent_words) and 1 < len(token) <= 4 and token not in parent_words + and token[0] == parent_words[0][0]) + + +def _sub_brand_part(alias: str, parent: str) -> str: + words = alias.split() + if len(words) > 1 and _is_abbreviation(words[0], parent): + return " ".join(words[1:]) + return alias + + +def _rest_owners() -> Dict[str, Set[str]]: + """For every " " alias, which parents use that rest.""" + owners: Dict[str, Set[str]] = defaultdict(set) + for alias, owner in BRAND_ALIASES.items(): + parent = owner.lower().strip() + if alias.startswith(parent + " "): + owners[alias[len(parent) + 1:].strip()].add(parent) + return owners + + +def _is_line_name(rest: str, owners: Dict[str, Set[str]]) -> bool: + from app.services.category_registry import detect_category_from_text + + if len(rest.replace(" ", "")) < 5 or len(owners.get(rest, ())) != 1: + return False + return detect_category_from_text(rest, exact_only=True) is None + + +def _family(parent: str) -> List[Target]: + parent_key = parent.lower().strip() + owners = _rest_owners() + out: List[Target] = [] + seen: Set[str] = set() + for alias, owner in sorted(BRAND_ALIASES.items()): + if owner.lower().strip() != parent_key or alias == parent_key: + continue + if alias.startswith(parent_key + " "): + rest = alias[len(parent_key) + 1:].strip() + terms = (parent_key, rest) if _is_line_name(rest, owners) else (parent_key,) + target = Target(query=alias.title(), brand_terms=terms) + else: + rest = _sub_brand_part(alias, parent) + target = Target(query=rest.title(), brand_terms=(rest,)) + if target.query.lower() not in seen: + seen.add(target.query.lower()) + out.append(target) + return out + + +def targets_for(brand: str) -> List[Target]: + """Sub-brands first, the brand itself last. Deterministic order. + + A sub-brand typed on its own ("Dettol") searches only that sub-brand.""" + name = (brand or "").strip() + if not name: + return [] + parent = resolve_parent_brand(name) or name + parent_key = parent.lower().strip() + typed_key = " ".join(_words(name)) + family = _family(parent) + + if typed_key != " ".join(_words(parent_key)): + own = [t for t in family + if " ".join(_words(t.query)) == typed_key + or typed_key in {" ".join(_words(term)) for term in t.brand_terms[1:] or t.brand_terms}] + return own or [Target(query=name, brand_terms=(name.lower(),))] + + display = parent.title() if parent.islower() else parent + return family + [Target(query=display, brand_terms=(parent_key,))] diff --git a/app/services/web_discovery/cache.py b/app/services/web_discovery/cache.py new file mode 100644 index 0000000..81fb602 --- /dev/null +++ b/app/services/web_discovery/cache.py @@ -0,0 +1,82 @@ +"""Search answers, kept so a re-run costs nothing and a throttled run can finish. + +Only ANSWERS are stored. A search that could not be asked (search.search +returned None) is never written: caching it would turn "we could not ask" into +"there is nothing", which is the one confusion this whole source must avoid. +An empty answer is kept for a day only - the same query minutes apart has been +measured returning 7 shop results and then 0 (see retail_presence). +""" +from __future__ import annotations + +import json +import logging +import sqlite3 +import threading +import time +from contextlib import closing +from pathlib import Path +from typing import List, Optional + +from app.infrastructure import settings +from app.services.web_discovery.search import Hit + +logger = logging.getLogger(__name__) + +_lock = threading.Lock() +_EMPTY_TTL_SECONDS = 24 * 3600 + + +def _db_path() -> Path: + return Path(settings.WEB_DISCOVERY_DIR) / "search_cache.db" + + +def _connect() -> sqlite3.Connection: + path = _db_path() + path.parent.mkdir(parents=True, exist_ok=True) + conn = sqlite3.connect(str(path), timeout=10) + conn.execute( + "CREATE TABLE IF NOT EXISTS search_cache (" + " query TEXT NOT NULL, backend TEXT NOT NULL, hits TEXT NOT NULL," + " created_at REAL NOT NULL, PRIMARY KEY (query, backend))" + ) + return conn + + +def _key(query: str) -> str: + return " ".join((query or "").lower().split()) + + +def get(query: str, backend: str) -> Optional[List[Hit]]: + """The cached answer, or None for a miss or an expired entry. Never raises.""" + try: + with _lock, closing(_connect()) as conn: + row = conn.execute( + "SELECT hits, created_at FROM search_cache WHERE query = ? AND backend = ?", + (_key(query), backend), + ).fetchone() + except Exception as exc: # noqa: BLE001 - a cache failure is a cache miss + logger.debug("web discovery cache read failed: %s", exc) + return None + if not row: + return None + hits = [Hit.from_dict(h) for h in json.loads(row[0])] + ttl = settings.WEB_DISCOVERY_CACHE_DAYS * 86400 if hits else _EMPTY_TTL_SECONDS + if time.time() - row[1] > ttl: + return None + return hits + + +def put(query: str, backend: str, hits: List[Hit]) -> None: + """Store an ANSWER. Callers must never pass the None of a failed search.""" + if hits is None: # defensive: see the module docstring + return + try: + with _lock, closing(_connect()) as conn: + conn.execute( + "INSERT OR REPLACE INTO search_cache (query, backend, hits, created_at) " + "VALUES (?, ?, ?, ?)", + (_key(query), backend, json.dumps([h.as_dict() for h in hits]), time.time()), + ) + conn.commit() + except Exception as exc: # noqa: BLE001 + logger.debug("web discovery cache write failed: %s", exc) diff --git a/app/services/web_discovery/discover.py b/app/services/web_discovery/discover.py new file mode 100644 index 0000000..f75619e --- /dev/null +++ b/app/services/web_discovery/discover.py @@ -0,0 +1,225 @@ +"""Queries -> listings -> one candidate per (product, pack size). + +For each sub-brand (brands.targets_for) and each retailer +(listings.SEARCH_ORDER) one query, `" site:"`, capped at +WEB_DISCOVERY_MAX_QUERIES. The order is retailer-major, so when the budget +runs out it is the last retailers that go unasked for every sub-brand, not +the last sub-brands that go unasked everywhere. + +Each hit either becomes a listings.Listing or is refused with a reason; the +reasons are counted and reported, so a filter never throws products away +silently. Listings of the same pack on several retailers are folded into one +candidate, and the number of distinct retailers is its corroboration. + +A search that could not be asked is counted as such. Three in a row end the +run as PARTIAL - a throttled provider only gets worse if pushed - and what was +learned so far is kept; the cached answers make the next run pick up there. +""" +from __future__ import annotations + +import logging +import re +import time +from collections import Counter +from dataclasses import dataclass, field +from typing import Callable, Dict, List, Optional, Sequence + +from app.infrastructure import settings +from app.services.web_discovery import cache, listings, search +from app.services.web_discovery.brands import Target, targets_for + +logger = logging.getLogger(__name__) + +DONE = "done" +PARTIAL = "partial" + +_MAX_CONSECUTIVE_FAILURES = 3 +_MAX_LISTINGS_PER_CANDIDATE = 10 +# Two names are one product when their words (brand and packaging words set +# aside) overlap this much, Jaccard. Deliberately strict: "Dettol Soap" and +# "Dettol Original Soap" stay apart (0.5), because folding a generic title into +# a variant would name two products as one. Two rows the admin can see are +# better than one silently wrong row. +_SAME_NAME_JACCARD = 0.6 +_PACKAGING_WORDS = frozenset({"bar", "pouch", "bottle", "jar", "box", "tin", "carton", "pack"}) + + +@dataclass +class Query: + text: str + target: Target + retailer: str + + +@dataclass +class RunResult: + status: str = DONE + candidates: List[Dict] = field(default_factory=list) + queries_total: int = 0 + queries_done: int = 0 + queries_failed: int = 0 + queries_cached: int = 0 + listings_kept: int = 0 + rejected: Dict[str, int] = field(default_factory=dict) + backend: str = "" + detail: str = "" + + +def plan_queries(brand: str, max_queries: Optional[int] = None) -> List[Query]: + budget = max_queries if max_queries is not None else settings.WEB_DISCOVERY_MAX_QUERIES + out: List[Query] = [] + for retailer in listings.SEARCH_ORDER: + site = listings.RETAILERS[retailer][1] + for target in targets_for(brand): + out.append(Query(text=f"{target.query} site:{site}", target=target, retailer=retailer)) + return out[:max(0, budget)] + + +def _identity(listing: listings.Listing) -> List[str]: + brand_words = set() + for term in (listing.brand_term,): + brand_words.update(listings._words(term)) + return [t for t in listing.tokens if t not in brand_words and t not in _PACKAGING_WORDS] + + +def _variant_words(listing: listings.Listing) -> set: + """Words the retailer put in brackets - how Blinkit and others mark a + variant: "Lizol Floor Cleaner (Citrus)", "(Floral)", "(Original)".""" + return {w for group in re.findall(r"\(([^()]*)\)", listing.name) for w in listings._words(group)} + + +def _same_product(a: listings.Listing, b: listings.Listing) -> bool: + """Same brand word, same pack, near-identical name, and no bracketed + variant word on one side only. Measured: "(Citrus) 2l" and "(Floral) 2l" + share 4 of 6 words - enough for the overlap test alone, and two products.""" + if a.brand_term != b.brand_term or a.size != b.size: + return False + ident_a, ident_b = _identity(a), _identity(b) + if _jaccard(ident_a, ident_b) < _SAME_NAME_JACCARD: + return False + differing = set(ident_a) ^ set(ident_b) + return not (differing & (_variant_words(a) | _variant_words(b))) + + +def _jaccard(a: Sequence[str], b: Sequence[str]) -> float: + sa, sb = set(a), set(b) + if not sa and not sb: + return 1.0 + return len(sa & sb) / len(sa | sb) + + +def _price_range(prices: Sequence[float]) -> str: + if not prices: + return "" + low, high = min(prices), max(prices) + fmt = lambda v: f"₹{int(v)}" if float(v).is_integer() else f"₹{v:.2f}" # noqa: E731 + return fmt(low) if low == high else f"{fmt(low)} - {fmt(high)}" + + +def cluster(found: Sequence[listings.Listing]) -> List[Dict]: + """Fold listings of the same pack into candidates, best-corroborated first.""" + groups: List[List[listings.Listing]] = [] + seen_urls = set() + for listing in found: + if listing.url in seen_urls: + continue + seen_urls.add(listing.url) + for members in groups: + # Against the group's FIRST listing, so a chain of near-misses + # cannot drift a group from one product to another. + if _same_product(members[0], listing): + members.append(listing) + break + else: + groups.append([listing]) + + candidates: List[Dict] = [] + for members in groups: + brand_term, size = members[0].brand_term, members[0].size + names = Counter(m.name for m in members) + # The spelling most retailers agree on; then the shortest; then A-Z. + name = sorted(names, key=lambda n: (-names[n], len(n), n))[0] + retailers = sorted({m.retailer for m in members}, key=listings.SEARCH_ORDER.index) + prices = [m.price for m in members if m.price] + # Brackets come off, their words stay: Brand Discovery's name + # normaliser (and off_bulk.strip_sizes) drops bracketed text, which + # turned "Lizol Floor Cleaner (Citrus)" and "(Floral)" into one name. + plain = re.sub(r"\s+", " ", re.sub(r"[()]", " ", name)).strip() + candidates.append({ + "title": f"{plain} {size}", + "size": size, + "sizes": [size], + "providers": [listings.display_name(r) for r in retailers], + "retailer_count": len(retailers), + "price_range": _price_range(prices), + "listings": [m.as_dict() for m in members[:_MAX_LISTINGS_PER_CANDIDATE]], + "brand_term": brand_term, + "source": "web", + }) + candidates.sort(key=lambda c: (-c["retailer_count"], c["title"].lower())) + return candidates + + +def run(brand: str, *, progress: Optional[Callable[[RunResult], None]] = None, + should_stop: Optional[Callable[[], bool]] = None, + max_queries: Optional[int] = None, sleep: Callable[[float], None] = time.sleep) -> RunResult: + """Search, parse, fold. Never raises for a search failure.""" + queries = plan_queries(brand, max_queries) + backend = search.backend_name() + result = RunResult(queries_total=len(queries), backend=backend) + rejected: Counter = Counter() + kept: List[listings.Listing] = [] + consecutive_failures = 0 + + for query in queries: + if should_stop and should_stop(): + result.status = PARTIAL + result.detail = "stopped" + break + hits = cache.get(query.text, backend) + if hits is not None: + result.queries_cached += 1 + else: + hits = search.search(query.text) + if hits is None: + result.queries_failed += 1 + consecutive_failures += 1 + if consecutive_failures >= _MAX_CONSECUTIVE_FAILURES: + result.status = PARTIAL + result.detail = (f"the search provider stopped answering after " + f"{result.queries_done} of {len(queries)} searches; " + f"run it again later - answers so far are cached") + result.queries_done += 1 + break + result.queries_done += 1 + if progress: + progress(result) + sleep(settings.WEB_DISCOVERY_PAUSE_SECONDS) + continue + cache.put(query.text, backend, hits) + sleep(settings.WEB_DISCOVERY_PAUSE_SECONDS) + consecutive_failures = 0 + + for hit in hits: + listing, reason = listings.parse_hit(hit.title, hit.url, hit.snippet, + query.target.brand_terms) + if listing is None: + rejected[reason] += 1 + else: + kept.append(listing) + result.queries_done += 1 + if progress: + progress(result) + + result.listings_kept = len(kept) + result.rejected = dict(rejected) + result.candidates = cluster(kept) + if result.status == DONE and result.queries_failed: + # Some searches went unasked, so absence of a product proves nothing. + result.status = PARTIAL + result.detail = (f"{result.queries_failed} search(es) could not be asked; run it again " + f"to fill them in - answered searches are cached") + logger.info("web discovery %s: %d candidates from %d listings (%d/%d searches, %d cached, %d failed)", + brand, len(result.candidates), len(kept), result.queries_done, len(queries), + result.queries_cached, result.queries_failed) + return result diff --git a/app/services/web_discovery/jobs.py b/app/services/web_discovery/jobs.py new file mode 100644 index 0000000..a6fec87 --- /dev/null +++ b/app/services/web_discovery/jobs.py @@ -0,0 +1,225 @@ +"""Web discovery as a background job the Brand Discovery panel polls. + +A brand fans out to ~80 searches paced 2 s apart, which is minutes - far past +what a preview request may block for. So "Discover products" with the web box +ticked starts (or reuses) a job, the panel polls its progress, and the preview +then reads the finished job's candidates. + +One worker thread: the search provider throttles bursts, and two brands +searched in parallel would only get both throttled. Job records are JSON under +WEB_DISCOVERY_DIR/jobs so a restart does not lose a finished job; a job that +was running when the process died is marked `interrupted`. + +Reuse: a finished job for the same brand younger than REUSE_SECONDS is +returned as is unless the caller asks to refresh. Its searches are cached for +days anyway, so even a refresh is cheap for what was already answered. +""" +from __future__ import annotations + +import json +import logging +import queue +import threading +import time +import uuid +from dataclasses import asdict, dataclass, field +from pathlib import Path +from typing import Dict, List, Optional + +from app.infrastructure import settings +from app.services.web_discovery import discover + +logger = logging.getLogger(__name__) + +QUEUED = "queued" +RUNNING = "running" +DONE = discover.DONE +PARTIAL = discover.PARTIAL +FAILED = "failed" +INTERRUPTED = "interrupted" +FINISHED = {DONE, PARTIAL, FAILED, INTERRUPTED} + +REUSE_SECONDS = 24 * 3600 + + +@dataclass +class WebJob: + job_id: str + brand: str + status: str = QUEUED + created_at: float = field(default_factory=time.time) + updated_at: float = field(default_factory=time.time) + backend: str = "" + queries_total: int = 0 + queries_done: int = 0 + queries_failed: int = 0 + queries_cached: int = 0 + listings_kept: int = 0 + rejected: Dict[str, int] = field(default_factory=dict) + candidates: List[Dict] = field(default_factory=list) + detail: str = "" + + def summary(self) -> Dict: + """Everything but the candidates - what the progress poll needs.""" + body = asdict(self) + body.pop("candidates") + body["candidate_count"] = len(self.candidates) + return body + + +_jobs: Dict[str, WebJob] = {} +_lock = threading.Lock() +_queue: "queue.Queue[str]" = queue.Queue() +_worker: Optional[threading.Thread] = None + + +def _brand_key(brand: str) -> str: + return " ".join((brand or "").lower().split()) + + +def _jobs_dir() -> Path: + path = Path(settings.WEB_DISCOVERY_DIR) / "jobs" + path.mkdir(parents=True, exist_ok=True) + return path + + +def _save(job: WebJob) -> None: + job.updated_at = time.time() + try: + (_jobs_dir() / f"{job.job_id}.json").write_text(json.dumps(asdict(job)), encoding="utf-8") + except Exception as exc: # noqa: BLE001 - the in-memory job is still served + logger.warning("web discovery: could not save job %s: %s", job.job_id, exc) + + +def _load(job_id: str) -> Optional[WebJob]: + if not job_id or not all(c in "0123456789abcdef" for c in job_id): + return None + path = _jobs_dir() / f"{job_id}.json" + if not path.exists(): + return None + try: + job = WebJob(**json.loads(path.read_text(encoding="utf-8"))) + except Exception: # noqa: BLE001 - an unreadable record is no record + return None + if job.status in (QUEUED, RUNNING): + # Loaded from disk, so no worker of THIS process is running it. + job.status = INTERRUPTED + job.detail = "the server restarted while this was running; start it again" + return job + + +def get_job(job_id: str) -> Optional[WebJob]: + with _lock: + job = _jobs.get(job_id) + if job is None: + job = _load(job_id) + if job is not None: + with _lock: + _jobs.setdefault(job.job_id, job) + return job + + +def latest_for(brand: str) -> Optional[WebJob]: + """The newest FINISHED job with results for `brand`, in memory or on disk.""" + key = _brand_key(brand) + best: Optional[WebJob] = None + with _lock: + known = list(_jobs.values()) + try: + for path in _jobs_dir().glob("*.json"): + if path.stem not in {j.job_id for j in known}: + loaded = _load(path.stem) + if loaded: + known.append(loaded) + except Exception: # noqa: BLE001 + pass + for job in known: + if _brand_key(job.brand) != key or job.status not in (DONE, PARTIAL): + continue + if best is None or job.updated_at > best.updated_at: + best = job + return best + + +def start_job(brand: str, *, refresh: bool = False) -> WebJob: + """Queue a job for `brand`, or return the one already running or recent.""" + brand = (brand or "").strip() + if not brand: + raise ValueError("a brand name is required") + key = _brand_key(brand) + with _lock: + for job in _jobs.values(): + if _brand_key(job.brand) == key and job.status in (QUEUED, RUNNING): + return job + if not refresh: + recent = latest_for(brand) + if recent and time.time() - recent.updated_at < REUSE_SECONDS: + return recent + job = WebJob(job_id=uuid.uuid4().hex, brand=brand) + with _lock: + _jobs[job.job_id] = job + _save(job) + _ensure_worker() + _queue.put(job.job_id) + return job + + +def candidates_for(brand: str, job_id: Optional[str] = None) -> Optional[WebJob]: + """The job whose candidates a preview should use: the named one if it is + for this brand and finished, else the brand's latest finished job.""" + if job_id: + job = get_job(job_id) + if job and _brand_key(job.brand) == _brand_key(brand) and job.status in (DONE, PARTIAL): + return job + return latest_for(brand) + + +def _ensure_worker() -> None: + global _worker + with _lock: + if _worker is not None and _worker.is_alive(): + return + _worker = threading.Thread(target=_work, name="web-discovery", daemon=True) + _worker.start() + + +def _work() -> None: + while True: + job_id = _queue.get() + job = get_job(job_id) + if job is None: + continue + run_job(job) + + +def run_job(job: WebJob) -> WebJob: + """Run one job to the end on the calling thread. The worker's body; also + what tests call directly.""" + job.status = RUNNING + _save(job) + + def progress(partial: discover.RunResult) -> None: + job.backend = partial.backend + job.queries_total = partial.queries_total + job.queries_done = partial.queries_done + job.queries_failed = partial.queries_failed + job.queries_cached = partial.queries_cached + job.updated_at = time.time() + + try: + result = discover.run(job.brand, progress=progress) + except Exception as exc: # noqa: BLE001 - a failed job must say so, not vanish + logger.exception("web discovery failed for %r", job.brand) + job.status = FAILED + job.detail = str(exc) + _save(job) + return job + + progress(result) + job.status = result.status + job.detail = result.detail + job.listings_kept = result.listings_kept + job.rejected = result.rejected + job.candidates = result.candidates + _save(job) + return job diff --git a/app/services/web_discovery/listings.py b/app/services/web_discovery/listings.py new file mode 100644 index 0000000..1acdaab --- /dev/null +++ b/app/services/web_discovery/listings.py @@ -0,0 +1,292 @@ +"""A search hit -> a retail listing we can trust, or a reason it was refused. + +Every rule here is a refusal. A listing is accepted only when it is a PRODUCT +page on a known retailer, names the brand as a word, states exactly one pack +size, and is not a multipack or a combo. Nothing is fetched: the title, URL and +snippet are what the search engine returned. +""" +from __future__ import annotations + +import re +from dataclasses import dataclass, field +from typing import Dict, List, Optional, Sequence, Tuple +from urllib.parse import urlparse + +# --------------------------------------------------------------------------- +# Retailers +# --------------------------------------------------------------------------- +# domain -> (display name as the `providers` column spells it, `site:` scope +# for the query, the path shape of ONE product's page). A search, category, +# brand or offer page is not a listing of a product and must not count as one; +# these patterns are what tell them apart. Taken from the URL grammar each site +# uses today (sku_service._MARKETPLACE_PATTERNS covers several of the same). +RETAILERS: Dict[str, Tuple[str, str, re.Pattern]] = { + "blinkit.com": ("Blinkit", "blinkit.com", re.compile(r"/prn/[^/]+/prid/\d+", re.I)), + "zeptonow.com": ("Zepto", "zeptonow.com", re.compile(r"/pn/[^/]+/pvid/[0-9a-f-]{8,}", re.I)), + "swiggy.com": ("Swiggy Instamart", "swiggy.com/instamart", + re.compile(r"/instamart/item/[A-Za-z0-9]+", re.I)), + "bigbasket.com": ("BigBasket", "bigbasket.com", re.compile(r"/pd/\d+", re.I)), + "jiomart.com": ("JioMart", "jiomart.com", re.compile(r"/p/[a-z-]+/[^/]+/\d+", re.I)), + "amazon.in": ("Amazon", "amazon.in", re.compile(r"/(?:dp|gp/product)/[A-Z0-9]{10}", re.I)), + "flipkart.com": ("Flipkart", "flipkart.com", re.compile(r"/p/itm[a-z0-9]+", re.I)), +} + +# The retailers queried with `site:`, in the order they are asked. Quick +# commerce first: it lists what is in Indian shops this week. +SEARCH_ORDER: Tuple[str, ...] = ( + "blinkit.com", "zeptonow.com", "swiggy.com", "bigbasket.com", + "jiomart.com", "amazon.in", "flipkart.com", +) + + +def retailer_for(url: str) -> Optional[str]: + """The allowlisted retailer domain `url` is on, or None.""" + host = urlparse(url or "").netloc.lower().split(":")[0] + for domain in RETAILERS: + if host == domain or host.endswith("." + domain): + return domain + return None + + +def is_product_url(retailer: str, url: str) -> bool: + spec = RETAILERS.get(retailer) + if not spec: + return False + parsed = urlparse(url or "") + return bool(spec[2].search(parsed.path + ("?" + parsed.query if parsed.query else ""))) + + +def display_name(retailer: str) -> str: + return RETAILERS.get(retailer, (retailer,))[0] + + +# --------------------------------------------------------------------------- +# Titles +# --------------------------------------------------------------------------- +# What surrounds the product name in a result title. Each retailer wraps it +# differently: +# blinkit "Dettol Original Soap (125 g) - Buy Online at Best Price | Blinkit" +# zepto "Buy Dettol Original Soap 125 g Online at Best Price | Zepto" +# bigbasket "Buy Dettol Original Soap 125 g Online at Best Price of Rs 58 - bigbasket" +# amazon "Dettol Original Soap Bar, 125g : Amazon.in: Beauty" +# instamart "Buy Dettol Original Soap 125 g Online | Swiggy Instamart" +# blinkit "Harpic Toilet Cleaner - (Original) - 500 ml Price - Buy Online at ₹120 in India" +# so the name can run over several segments: they are joined from the first +# one that leads with the brand until the first piece of shop furniture. +_SEGMENT_SPLIT = re.compile(r"\s+[|:]\s+|\s+[-–—]\s+|\s*\|\s*") +_LEADING_BUY = re.compile(r"^\s*(?:buy|shop|order)\s+", re.I) +# Where the product name ends and the shop's wording begins. Everything from +# the first of these on is furniture: "... Online at Best Price", "... Price - +# Buy", "...: Amazon.in: Beauty", "... | Blinkit". +_FURNITURE = re.compile( + r"\s*\b(?:online|price|blinkit|zepto|zeptonow|swiggy|instamart|bigbasket|jiomart|" + r"amazon(?:\.in)?|flipkart(?:\.com)?|grofers|free\s+delivery|delivered\s+in|" + r"in\s+\d+\s+minutes?)\b.*$", + re.I, +) +# A listing of several packs, or of a bundle, is a different SKU from the pack +# itself; taken as the pack it would put a carton's size or a combo's name on +# a catalogue row. +_MULTIPACK = re.compile( + r"\b(?:pack\s+of\s+(?:[2-9]|\d{2,})|set\s+of\s+\d+|combo|bundle|" + r"(?:[2-9]|\d{2,})\s*x\s*\d|\d+\s*x\s*(?:[2-9]|\d{2,})\b|multipack|value\s+pack\s+of)\b" + r"|(?:\d\s*(?:g|gm|gms|kg|ml|l|ltr)\s*[x×*]\s*\d)", + re.I, +) +# A long title is several listings run together by the provider - see +# retail_presence.MAX_LISTING_TITLE_CHARS for the measured case. +MAX_TITLE_CHARS = 160 + +# Pack sizes. Mass and volume in the units the catalogue stores; counts become +# "pcs" because that is the count unit brand_discovery's title parser reads. +_MASS_VOLUME = re.compile( + r"(\d+(?:\.\d+)?)\s*(kilograms?|kgs?|grams?|gms?|gm|g|millilitres?|milliliters?|mls?|ml|" + r"litres?|liters?|ltrs?|ltr|l)\b", re.I) +_COUNT = re.compile( + r"(\d+)\s*(pcs|pc|pieces?|count|ct|tablets?|capsules?|lozenges?|sachets?|rolls?|" + r"sticks?|condoms?|wipes|napkins|pads|units?|n)\b", re.I) +_UNIT_CANON = { + "kilogram": "kg", "kilograms": "kg", "kg": "kg", "kgs": "kg", + "gram": "g", "grams": "g", "gm": "g", "gms": "g", "g": "g", + "millilitre": "ml", "millilitres": "ml", "milliliter": "ml", "milliliters": "ml", + "ml": "ml", "mls": "ml", + "litre": "l", "litres": "l", "liter": "l", "liters": "l", "ltr": "l", "ltrs": "l", "l": "l", +} +_PRICE = re.compile(r"(?:₹|rs\.?|inr)\s*(\d{1,6}(?:\.\d{1,2})?)", re.I) + + +def _number(raw: str) -> str: + value = float(raw) + return str(int(value)) if value.is_integer() else str(value) + + +def extract_size(text: str) -> Tuple[Optional[str], int]: + """(the pack size in canonical form, how many DIFFERENT sizes were named). + + "Dettol Soap (125 g)" -> ("125g", 1); "Harpic 500ml + 200ml" -> (.., 2), + which the caller refuses: one listing, one pack. + """ + found: List[str] = [] + for num, unit in _MASS_VOLUME.findall(text or ""): + canon = _UNIT_CANON.get(unit.lower()) + if canon and float(num) > 0: + size = f"{_number(num)}{canon}" + if size not in found: + found.append(size) + if not found: + for num, _unit in _COUNT.findall(text or ""): + if int(num) > 0: + size = f"{int(num)}pcs" + if size not in found: + found.append(size) + return (found[0] if found else None), len(found) + + +def extract_price(text: str) -> Optional[float]: + """The first rupee amount in a snippet, if it is a plausible shelf price.""" + for raw in _PRICE.findall(text or ""): + value = float(raw) + if 1 <= value <= 100_000: + return value + return None + + +def _strip_sizes(text: str) -> str: + """The name without its pack size - keeping a variant that shares a + bracket with it: "(Lavender - 500 ml)" -> "(Lavender)", "(125 g)" -> "".""" + text = re.sub(r"\bpack\s+of\s+1\b", " ", text, flags=re.I) + text = _MASS_VOLUME.sub(" ", text) + text = _COUNT.sub(" ", text) + text = re.sub(r"\(\s*[,;\-]*\s*", "(", text) + text = re.sub(r"\s*[,;\-]*\s*\)", ")", text) + text = re.sub(r"\(\s*\)", " ", text) + return text + + +def _tidy(text: str) -> str: + text = re.sub(r"\s+", " ", text) + return text.strip(" ,;:.-/&+–—") + + +def _words(text: str) -> List[str]: + return re.findall(r"[a-z0-9]+", (text or "").lower()) + + +# Words a retail title may put before the brand ("New Dettol ..."). +_LEAD_IN = {"new", "the", "original"} + + +def names_brand(title: str, brand_terms: Sequence[str]) -> bool: + """True when `title` LEADS with one of `brand_terms`, as whole words. + + Retail titles put the brand first ("Dettol Original Soap", "Maggi 2-Minute + Noodles"). Requiring it there - optionally after "New"/"The" - is what + keeps "Haldiram's Soan Papdi Mithai" from counting as an "Amul Mithai" + listing, or a Savlon listing that mentions Dettol from counting as Dettol. + """ + words = _words(title) + starts = [0] + ([1] if words and words[0] in _LEAD_IN else []) + for term in brand_terms: + term_words = _words(term) + if not term_words: + continue + for start in starts: + if words[start:start + len(term_words)] == term_words: + return True + return False + + +def _protect_brackets(title: str) -> str: + """A separator INSIDE brackets is part of the variant, not a segment break: + "(Lavender - 500 ml)" must not split into "(Lavender" and "500 ml)".""" + return re.sub(r"\(([^()]*)\)", + lambda m: "(" + re.sub(r"\s+[-–—|:]\s+", " ", m.group(1)) + ")", + title or "") + + +def clean_title(title: str, brand_terms: Sequence[str]) -> str: + """The product name inside a result title: shop furniture removed. + + Starts at the first segment that leads with the brand and keeps following + segments - a variant ("(Original)") or the size ("500 ml") - until the + shop's own wording begins. "Buy X Online | Blinkit", "X : Amazon.in: + Beauty" and "X - (Original) - 500 ml Price - Buy Online" all reduce to X + with its variant and size. + """ + collected: List[str] = [] + for segment in _SEGMENT_SPLIT.split(_protect_brackets(title)): + segment = _LEADING_BUY.sub("", segment) + cut = _FURNITURE.search(segment) + content = _tidy(segment[:cut.start()] if cut else segment) + if not collected and not names_brand(content, brand_terms): + continue # before the name: "Buy", a site banner + if content: + collected.append(content) + if cut: + break + return _tidy(" ".join(collected)) + + +# --------------------------------------------------------------------------- +# One hit -> one listing +# --------------------------------------------------------------------------- +@dataclass +class Listing: + retailer: str # domain, e.g. "blinkit.com" + url: str + raw_title: str + name: str # cleaned product name, size removed + size: str # canonical, e.g. "125g", "10pcs" + brand_term: str # which brand/sub-brand it named + price: Optional[float] = None + tokens: List[str] = field(default_factory=list) + + def as_dict(self) -> dict: + return {"retailer": display_name(self.retailer), "domain": self.retailer, + "url": self.url, "title": self.raw_title, "price": self.price} + + +# Why a hit was refused. Counted per job so the admin can see what was thrown +# away and why, rather than trusting a silent filter. +NOT_A_RETAILER = "not_a_retailer" +NOT_A_PRODUCT_PAGE = "not_a_product_page" +TITLE_TOO_LONG = "title_too_long" +WRONG_BRAND = "wrong_brand" +NO_PACK_SIZE = "no_pack_size" +SEVERAL_SIZES = "several_sizes" +MULTIPACK = "multipack_or_combo" +NO_NAME = "no_product_name" + + +def parse_hit(title: str, url: str, snippet: str, + brand_terms: Sequence[str]) -> Tuple[Optional[Listing], Optional[str]]: + """(listing, None) when the hit is a trustworthy product listing, else (None, reason).""" + retailer = retailer_for(url) + if retailer is None: + return None, NOT_A_RETAILER + if not is_product_url(retailer, url): + return None, NOT_A_PRODUCT_PAGE + if len(title or "") > MAX_TITLE_CHARS: + return None, TITLE_TOO_LONG + cleaned = clean_title(title, brand_terms) + if not cleaned: + return None, WRONG_BRAND + if _MULTIPACK.search(title or ""): + return None, MULTIPACK + size, distinct = extract_size(cleaned) + if size is None: + # Some retailers put the size only in the title's tail ("... - 125 g | Zepto"). + size, distinct = extract_size(title) + if size is None: + return None, NO_PACK_SIZE + if distinct > 1: + return None, SEVERAL_SIZES + name = _tidy(_strip_sizes(cleaned)) + if not name or not re.search(r"[a-z]", name, re.I): + return None, NO_NAME + term = next((t for t in brand_terms if names_brand(cleaned, [t])), brand_terms[0]) + return Listing( + retailer=retailer, url=url, raw_title=title.strip(), name=name, size=size, + brand_term=term, price=extract_price(snippet) or extract_price(title), + tokens=_words(name), + ), None diff --git a/app/services/web_discovery/provenance.py b/app/services/web_discovery/provenance.py new file mode 100644 index 0000000..8094bc2 --- /dev/null +++ b/app/services/web_discovery/provenance.py @@ -0,0 +1,117 @@ +"""After a Brand Discovery ingest: record WHERE each web-found product was seen. + +The ingest CSV has no column for it, and the 11-stage pipeline is not changed +to carry one, so this runs AFTER the batch, the same way +capture_discovery.mark_captured_row does. For each row the batch stored from a +web-found product it + +* merges `field_sources.web_listings` = the retailer listings (retailer, URL, + listing title, price) that justified the product, with when they were seen; +* fills `price_range` from the listing prices when the pipeline left it blank; +* sets `validation_status = 'needs_review'` when only ONE retailer listed it + (never lifting a `rejected` verdict). Two or more retailers keep whatever the + validator decided. + +The join is exact: stage 11 reports every stored row's `image_id` with the +1-based sheet row it came from (header = row 1), and the ingest wrote the web +products in a known order, so CSV row i is `source_row` i + 2. +""" +from __future__ import annotations + +import logging +import threading +import time +from typing import Any, Dict, List, Optional + +logger = logging.getLogger(__name__) + +_POLL_SECONDS = 5.0 +_GIVE_UP_SECONDS = 6 * 3600 + + +def web_updates(manifest_files: List[Any], entries: Dict[int, Dict[str, Any]]) -> List[Dict[str, Any]]: + """(image_id, table-brand, entry) for every stored row that came from a + web-found product. `entries` is keyed by the product's 0-based CSV index.""" + out: List[Dict[str, Any]] = [] + for file in manifest_files: + result = getattr(file, "result", None) or {} + for product in result.get("products") or []: + row = product.get("source_row") + image_id = product.get("image_id") + if not image_id or not isinstance(row, int): + continue + entry = entries.get(row - 2) + if entry: + out.append({"image_id": image_id, "brand": product.get("brand"), "entry": entry}) + return out + + +def apply(brand: str, updates: List[Dict[str, Any]]) -> int: + """Write the provenance. Returns rows updated. Never raises.""" + if not updates: + return 0 + from psycopg.types.json import Json + + from app.services.vector_store import _connect, _table_name + + conn = _connect() + if conn is None: + logger.warning("web discovery provenance: no database connection") + return 0 + seen_at = time.strftime("%Y-%m-%dT%H:%M:%S%z") + written = 0 + try: + with conn, conn.cursor() as cur: + for update in updates: + entry = update["entry"] + table = _table_name(update.get("brand") or brand) + single = int(entry.get("retailer_count") or 0) < 2 + provenance = {"web_listings": { + "seen_at": seen_at, + "retailer_count": int(entry.get("retailer_count") or 0), + "listings": list(entry.get("listings") or [])[:10], + }} + cur.execute( + f"UPDATE {table} SET " + f"field_sources = COALESCE(field_sources, '{{}}'::jsonb) || %s, " + f"price_range = CASE WHEN COALESCE(price_range, '') = '' AND %s <> '' " + f"THEN %s ELSE price_range END, " + f"validation_status = CASE WHEN %s AND COALESCE(validation_status, '') <> 'rejected' " + f"THEN 'needs_review' ELSE validation_status END " + f"WHERE image_id = %s", + (Json(provenance), entry.get("price_range") or "", entry.get("price_range") or "", + single, update["image_id"]), + ) + written += cur.rowcount + except Exception as exc: # noqa: BLE001 - provenance must not take the batch down + logger.exception("web discovery provenance failed for %s: %s", brand, exc) + finally: + conn.close() + logger.info("web discovery provenance: %d row(s) of %s marked with their listings", written, brand) + return written + + +def watch_batch(batch_id: str, brand: str, entries: Dict[int, Dict[str, Any]], *, + read_manifest=None, sleep=time.sleep, give_up_seconds: float = _GIVE_UP_SECONDS) -> Optional[int]: + """Block until the batch ends, then apply. Returns rows written, or None + when the batch never finished in time. Run it on a thread - see `watch`.""" + if read_manifest is None: + from app.core.batch_ingest import read_manifest + from app.core.batch_ingest import TERMINAL_BATCH_STATES + + deadline = time.monotonic() + give_up_seconds + while time.monotonic() < deadline: + manifest = read_manifest(batch_id) + if manifest is not None and manifest.status in TERMINAL_BATCH_STATES: + return apply(brand, web_updates(manifest.files, entries)) + sleep(_POLL_SECONDS) + logger.warning("web discovery provenance: batch %s did not finish in time", batch_id) + return None + + +def watch(batch_id: str, brand: str, entries: Dict[int, Dict[str, Any]]) -> None: + """Fire-and-forget `watch_batch` on a daemon thread.""" + if not entries: + return + threading.Thread(target=watch_batch, args=(batch_id, brand, entries), + name=f"web-provenance-{batch_id[:8]}", daemon=True).start() diff --git a/app/services/web_discovery/search.py b/app/services/web_discovery/search.py new file mode 100644 index 0000000..cae3805 --- /dev/null +++ b/app/services/web_discovery/search.py @@ -0,0 +1,105 @@ +"""One web search, from whichever backend is available. + + search(query) -> list[Hit] the engine answered (possibly with nothing) + -> None the engine could not be asked (throttled, + unreachable, not installed) + +The two are different answers and every caller depends on it: None is "we do +not know" and must never be recorded or reported as "no products", which is +the same rule retail_presence._search keeps. + +Backends, in order: + +* Google Custom Search JSON API, when USE_GOOGLE_CSE is on and both keys are + set. The official API - we never scrape google.com result pages. 100 free + queries a day; a quota error falls through to DuckDuckGo. +* DuckDuckGo through the `ddgs` package the project already uses for images + and retail presence. Free, no key, throttles bursts. +""" +from __future__ import annotations + +import logging +from dataclasses import dataclass +from typing import List, Optional + +from app.infrastructure import settings + +logger = logging.getLogger(__name__) + +_GOOGLE_URL = "https://www.googleapis.com/customsearch/v1" +_GOOGLE_MAX_NUM = 10 # the API's per-request ceiling +_TIMEOUT_SECONDS = 15 + + +@dataclass(frozen=True) +class Hit: + title: str + url: str + snippet: str = "" + + def as_dict(self) -> dict: + return {"title": self.title, "url": self.url, "snippet": self.snippet} + + @classmethod + def from_dict(cls, data: dict) -> "Hit": + return cls(title=str(data.get("title") or ""), url=str(data.get("url") or ""), + snippet=str(data.get("snippet") or "")) + + +def google_configured() -> bool: + return bool(settings.USE_GOOGLE_CSE and settings.GOOGLE_API_KEY and settings.GOOGLE_CSE_ID) + + +def backend_name() -> str: + return "google_cse" if google_configured() else "duckduckgo" + + +def _google(query: str, max_results: int) -> Optional[List[Hit]]: + import requests + + params = { + "key": settings.GOOGLE_API_KEY, "cx": settings.GOOGLE_CSE_ID, "q": query, + "num": min(_GOOGLE_MAX_NUM, max(1, max_results)), "gl": "in", + } + try: + response = requests.get(_GOOGLE_URL, params=params, timeout=_TIMEOUT_SECONDS) + except Exception as exc: # noqa: BLE001 - unreachable is "could not ask" + logger.debug("Google CSE unreachable for %r: %s", query, exc) + return None + if response.status_code != 200: + # 429 / 403 dailyLimitExceeded: a quota, not an answer. + logger.info("Google CSE answered %s for %r; falling back", response.status_code, query) + return None + try: + items = response.json().get("items") or [] + except ValueError: + return None + return [Hit(title=str(i.get("title") or ""), url=str(i.get("link") or ""), + snippet=str(i.get("snippet") or "")) for i in items if i.get("link")] + + +def _duckduckgo(query: str, max_results: int) -> Optional[List[Hit]]: + try: + from ddgs import DDGS + except ImportError: + logger.debug("web discovery: 'ddgs' is not installed") + return None + try: + with DDGS(timeout=_TIMEOUT_SECONDS) as ddgs: + rows = list(ddgs.text(query, region="in-en", safesearch="off", + max_results=max_results) or []) + except Exception as exc: # noqa: BLE001 - a throttle is not a verdict + logger.debug("DuckDuckGo failed for %r: %s", query, exc) + return None + return [Hit(title=str(r.get("title") or ""), url=str(r.get("href") or r.get("url") or ""), + snippet=str(r.get("body") or "")) for r in rows if (r.get("href") or r.get("url"))] + + +def search(query: str, max_results: Optional[int] = None) -> Optional[List[Hit]]: + """The hits for `query`, or None when no backend could be asked.""" + limit = max_results or settings.WEB_DISCOVERY_RESULTS_PER_QUERY + if google_configured(): + hits = _google(query, limit) + if hits is not None: + return hits + return _duckduckgo(query, limit) diff --git a/data/cache/off_brand_corpus/reckitt_benckiser.json b/data/cache/off_brand_corpus/reckitt_benckiser.json index 7426b8d..b2e3171 100644 --- a/data/cache/off_brand_corpus/reckitt_benckiser.json +++ b/data/cache/off_brand_corpus/reckitt_benckiser.json @@ -1,7 +1,10 @@ { - "brand": "reckitt benckiser", + "schema": 2, + "endpoint": "https://world.openfoodfacts.org/api/v2/search", + "brand": "Reckitt Benckiser", + "brand_tag": "reckitt-benckiser", "country": "india", - "fetched_at": 1788851995.867008, - "fetched_at_human": "2026-09-08 12:49:55", + "fetched_at": 1790675563.0787868, + "fetched_at_human": "2026-09-29 15:22:43", "hits": [] } \ No newline at end of file diff --git a/data/cache/web_discovery/search_cache.db b/data/cache/web_discovery/search_cache.db new file mode 100644 index 0000000000000000000000000000000000000000..fe009f67f9da708190ffcc79361f14f05b2d9d03 GIT binary patch literal 147456 zcmeFaYiuN0mL66;)7!N(J2N%=9NF^YR`qmG^)!RbhGg0TNBUB zU`A$SM`SXYoSv0tA^V2`0g@pY5M*NwVFAkmyne`rZ5Z~#uwbuYU~NklU@Rcpm^Ezw z&=1=`{E#8Q-#IrTACXBi$s)Vi-9c25`G|;n@45G$d!FCfeYWAYZQ=MqLvM@hy>H)p z?X`P965`&yd*8wT@8W;_^BP{f9>2j~dVkr^cka#p3%~Kr2mjB#uRM7D-hfW^Ls4R~C1c=l7Py-u%MGvKYQfJQg2!?4T$1mS60N%`N=j-`JQDhHlpFrj`7- z=C;G+XJ%mQZQFVYB^<463_WCfXT37Nb10rJABxA^O#i{}yi@$j8{c{Q)@%3Nre&Xo zAA8W{U+SH<&+lIj_wl7V{O+ru!%SR9XzX$)@k6jjHcg zJ^Q5@+8_S(QQK{M_M`X2qb0s}Q*U&Ef)Qz8S$tSc=VWByWI-ko0}VIZ>|+I=PbM3_B|zR2YT1AgP^B4fnE*trfDmV zuQ%LgRjKJstE-1K+~8Vs#ay;n%+sU8rrT=SZF)R4Jw;FbCOT#dw<*?}ma7Y_JkRgi zVO!wQ7y}v-Rl6v1G(EueZkbdCC5?BR3!E&=t@x9B-jh6R}=rr4|N4pEl3X2F!PWafL?HV>_ z*r6kXzB8~vXPLWPb3rWn9V~M!n%mm`KmF)Y*fCA)n~xs7*A6=N%tw!~ksY^+_X6J< zppX@AxM7<<`0%4gRk<|8n%-)Gb?Lp29^szRPuH;dMn0~9_1eMrLa|u$u_EPnVkNM{ z8o%E99*(PD`geZowKsn7@J~OyFL~>CVu_Kui)3Oe+|EWjr9Fry^mkt zdIP_{POki~edX);{afGoi}xP<$+yIVfA^dJ?}NYe&Hwg+^Nqj%;78y5TMz!sH~!+c z{`YVG{99|^_{ZP+&%gQnTfg^>f8$$c-^}2a^e3f2N`aIDDFsprq!dUgkWwI}KuUp> z0@tFzI`?>~I)L~pvVOkQx^Z}N3Dn2TKV zqU(M`dwuueYk_UT>DPv3y@?kgTr(Fu`q$rjy^Jo{FneBb)vta3^~Hy;!KY=nLKlw0 z3oiXtIrxqVUtHvdoILn%zy10T9=_(eXTEpQ^B(Zn;X$~BM&IJcSp(*Z_1GdJ-v z+OXu(xGNU^d3eD>?0)Rr4R(XTb*SaZ*mJnxHhB4l*s3Trj&*U zlMS5fs#YrJG6}y5JJ;oQxS*@I=3I-+VB<~0C)2jg$Vo`*asJdJ8Td_xyyw>pjnedLnO-X(0JDL0d&axpVb}|}aRxCGRk-9<>i$ZD_$4fsU1yRtXB7nvhHQQ^2 zqJf9D@n2ricbnIGRo8CXGgz%noX!->VbLZxbaIz_dZ%dxy&2(nh((zcY>{vovTIoE z*09L2@>dW~a~o6F_tEksDF{%Z<~V#u&X}g=GP!bM{SR@KzSDU^e{Y+uT+}u>fCa)B zKzfQ*Yk;c|31xVK@3dj*=9mQo(Q_7up6F4O4oOrX0|If@DYnQyL5NLw{)9yvZi08X z-#zw>Z4Xv!|JoU$O3(Z!f>3CLU2UaqvKY3G}C2R1umMVp?B- zNAR=`voR1G+=fH(5~tTNuUEDsWnxyogVrED^^Z}J$%rGe>WrhNGty{9!O1+-XZQj|-jrfK^n%v7s+JekbC$;|_1KTYd_pwT@2#4Q@R{7(XbX=l@e+xv8SljUVuTW?k8dXd*8nK^ZRxMArq*-1OyPd$%5&W;I_ag7X60d-v$YAytG{mS9@`pjpv(sb)HuVB7ZjtL}|P(iK$1ye8C%C4z<6bAcv!*2M^^LCr; zy#^L?_3q2IS%e_bvLV}6EUY5gwt_HbzbVIr2=Q`$oOS$o{!jkk7w`TL-v0*m{|$~< z<)F!3h-nlc&2O|E#??e%Hou!kIG`Jx=(;7}BK~iEUB1K$kS|Wq;(oOnJ@u*E z^laIar)~)5#}9kj^aDAfjqhziXOL5|q4%P@Hr(T5UCuicYEw44;gac>@2}|9s$3uy z-Ks@1Q!y8zD$4Jd*5qFfWXEE4<-(}A2==$+<7AjHf{NW}g;ab5T9dN5WOwD=dprDh z)5i93d!ggWnb_5xXio6Ti5Cns`#7>Cv7Z?VrNYK6TRNC3ZuZ zIX6cxlEnFUXY%aQGv;%A#*qI%RUsc$QWX+sIgV~|Dm$?KF zy)hLwF_Q6Q_zB%>MH%xKb?}xo<3zVPqJwqPs|0k=4GFs5m3R&w%Zd#P-X83;lQ234 zF1>3S!Kn$wK@*H*0lk%AKoq0XQ&hNP+xU%006Q`;XS2m@(il6?W5+rg8B<~t@y$VW zEH_~qf@6j3AFxb7GcY+=hC?*u2Qct75UWA1dbZlor4n*r&`mkmFc}ymFIzr^=~(Qz zh5;kV+hWOwU4`O|#$ZxzW(sU-P6@a<#p1@I$%qZ%s#Y!}*6Du0mKcge_?e%o2o1bYR_U_c-Yu zMj`aL8G;Rf%6Kjjrr$1XSHSu~zy*_!Oi2Jpfw_r>24hsz8AR>CI0dr8I@Qom-39_} zfTiNXBF{Z!BNz3K9>9|uF`eX;Me$IVV|Ad%R(KJ|y4$w4WnPqf6uz}j@H12J1?j56 z;<=KmhS>ry2ZhlV;2n&>ZbnySK3~ozEn5k$JiJJ`=Nmo{Mgp;+)6%S7yrpFCvl>Ak zOz$U4_FgW|h>4Xm#qwp2NO}M}VHLP5*cnSEXdv-`6-O-KaaOck8+O-=ePn7eclE9Z zXiPF?*>feU$hIPTVn)s6wFEDiNOkJ^`9rwKGnvn4uikHV1A}7xcWe-9W_%QvOvK=D znRkN;OFk!BW*B4s^nrjT2rI)v6h|uZ5`kAmId5g>R2Xfiq4Ch3yvoXcmGgRNV|u)q z6i;rZ`25CtF)4E!VZd2h?^7+7YYrsZqS>p(5+%1Nje|!q!e6-am#mQ+;V)O(q^Coa zovL%!--bD0ab217C3oHs*O;oi171f9T8SC;WR$Og_!BLJ+$;6C4>lEbTFaEn$#Bz0 zKV$#D@<;#k??e6n&MkcuvE44*$B54ahG9>vczPF5RiTI;2PZD#2M^#EMLZ{> zf2D{+!HNC*H}pD=_H~31gf2Q0-#K*oWB4}#*OyV>8Qe7mzmr)R6wSkYZ;?pLV<6dqWAY`Q zSaj<(Cj8b$GepBdw=P*qx7wL&EYHU=#@~YNR_%9gZ|aFyS6cqJMX7`!6|JisxczN0 z?^-kPz*DRgR6-!84zZ6_$cFumeD2n{NajlU*SXlD4zqLU$Cw2LfilQUz$7Thpy=~7 zAYxYSIQJ=2Dwif%Dy}q}x3$7z#(K`00{0`u8*!C%^vCQD!A_ik0F8j+e<+;+Fnhq8 z0}bM!^e7^o(qZs5C&;ksOvJ)P@<@cP&8jp#n4(LDxYW>UPC5NmFk^>km~#5d0Cw*< zryoqfQCZloV7PNW|y zk#J*?eyDVzA0mV>)(>MKY1{<)JJ#e2#nL2&?HVKav}QfQO0Fd7_xZ>;g~|Uv21M_X z7L9EjY0)u#hlq-VPUIV8R7uZ54jg#`L%vccZ)-az2sf4wegU}sv<6YLIp+if(XE`= z@cD5O348+dVuz0~RK$o)UjLSL(BH#rsW{CT+-_)l=&Kg;kp^Ye<5vefNx1^$$ zEfnXp+?Z*3A9&`-4#7P$@ZG`K-Gy-qr;*>4!08|$#bx&}D3;#&nszOqG5a0(_u${O zbhze_q}=s`dI*GelYl4)!X0L(E=9MKS`qo)pvNL=nTjIFNbd&bp~o$ABSd@$fcIo% z(fb7jA`OBastpA42S}$7$WgJz{Bcm`ql>o~sC|6rj_g{Atsw76uBy9%t*|1MCwSf9 zZN{hvLeP-Xv`1Sf#!Ud%0u|)&$@P6bgTJH+c`3ME1fE9*&AxO3y?5A}SJbjaEvfn@ zdoN8G<2~m&C)JXhaXMb4MAE_;QW|;tR^f99Nr7C)=5!lKTR!2%WC8ESgijaukRMmr z$ju0_8N(3EFhG75{^^-hTQunZEv zz|uipoLvW)22EQuRg@2gq7tF+br_z5mJ`8X1U^CrW(4r7P^+kkSLEkMG(pho>bEtt z@wCHEO}(I<3J$5Dstt7_k~A_UEtm9{)1+TIHj*@%?~WF$nFme@>dt_@^Bij`kso2`jYAs9UKWkAkoVh zH)k;dnYpm9g)4cy5`B$f2qHL}h&dEbD}{$d zL%@oXO;1*v69-dnlHxphOw+EFS2|M17K;+YjYOL@S1OMeCYjvfz9l#55O~ZdZNT0F zuyZ5Ya_b-B3mnM^jDR38)Lx|AO!dF^xLY9hN-fnRd+ zLc88vU6tDs)*g3bh~FkvyTnkRbP&OIMqxd`>}rc$ia{F@AolAW&jvfa$#ETXqZI2v zjVd7;9TgrrveAeg;MgdPXZEufeC|K{wcjKy_{;f2H3GhJ@BhB{#{Y<4U!I?|{@?96 zqGdc$^847zB=`_5b1uaU(o(#5gLcdgWASJY(uqv{%Z%h%yz@`hT1b zM=D(($gm%h1qCgw|CiSP8;+$->;Da^Dh}`$)B1n6UzhQ&lsKERrdBnr|Hp7+Y5l)X zqiE?}*&J#8KR`TD71;=`wGYP0*u9^6am*=s)u&wl@4A(Jq3*xV`u`h${oc2}jbG_c zN`aIDDFsprq!jq-T@~@aO+>@BXEgb3Xj(BLq%)_M`X2Bk~5X$=XkFgAIyy zM9A2CJ_01#BKAu@?n@M~ludt|B1Ue!R#4gWtv%jF+I}4+OR7Sm3P-mhm3n=3cWh)12kBXZZ;?N<2S+(D=vT&VH^N%l-#Zqn-fRa z)Zy|9MY%j%eoL0bnVA@8Bfa-^8z8JQm`4soN~j!qAxoCHrc4T%eu9UV6Qw1>!AJ#< zZY?$?>R4hcAZ><%-~jUq2np2VA-Gc+4G6z@G*q55S_@6v1`Y(GTt1t^JscV~J$=VE z(C2}+LyLa;jegr}rq5^wUE1SNv|R>O-r?}HKuGSxHDfr)fp%LQup~p`5b89HULxb# zu<$yD;Rh%K$dPS`u>@2H!vG0c@5|r_S!-~}Vm+wfH+<-> z$mq^~7{_!;o|p8Hj*+nt4UWi|kheG-&WR;Qkz@H&8SSYcnhasQ;qWD)?zIs;FetB( z1;R@*uxxx(E#zh)GFV4h#~yQhoNtKAqlipSCd8)*+9oziIq{TENGQ_yLm|aj)By3J zXfqw*c^qlnFNf4cOf2dowFupZg3zdVq>hNODwddtB1O5t7NQ~f0dpt`W<VL^jR|L$i7-~e3#0f! z+6E~1Wpz;CaKhwvyWLq}a#g8X?3`{=&4Ld+>wUYlzB_MP>%oe#rq+G^X}(k|pJw-- z?=SA3u7~-j*_wU0=hh74u(8!Ja|ij8lf$;Fcbg|q6|do1&tD!NS>15^^tlzDsWtPg zRrYi3Ab+Nv<(o=STi7g>ca?mww%XflcU!^7weHI&A8*=?M&l@a@$u6W?d)V{S6|rd zJgv=#J)`S28yk&-YB7AVx9q8|yI$Bnunx{TXS=KR(`OrneQ)JuX|c1kd|GW*78e&P zXKODvI-B)stA3`hFYniLyE~g!$8}Zz?aur9+D^t?t9TV{*{d`T@&71mWzEjvfm*X0 z<<3T9^TgQQ?;SO94Sla2wT^c8&yH4}@9!K}y~^=XaP-2<8rdgxvxm=DYc;dE&^qkx z_YNCR!lQ$gV|{VIx7qMIN2||sPxdzJhc6b+p6sn_*6K==-Wvzi?xCCMRp`3r6Yt2y zbJiB>FS48Ai?ijty|!~)X+G(gTFXJ(c-Bh&DEp+VA3Q^Uo}E_CDyN%!%cqsS{odBz zdS+{>o~7sOYe%)CRj*EMY%NuKmA!fv*Q%RKhuN*YI`!SR8hOuJGx40AljxpP;{`o; z$BW*((R)|6UDGfQjmLp8ZZ>L~v06UX5AqrMEV?H&w9T%*hBnZ@&82!CeUt4qcWP$C zY8@>St=ie6#Z0bpw%)6tPnD%-+ScB3wu12=9-N*W_A+^W7LxFgQQZTuaenX#Z_mO{MOR^>0#qpZgbDfZ7uCI zj`mg>n@mgeU)yZy4N0fUp4qD$&#Rlq``OCzI<7xdx0d#^o5yB*Bf9?RMXhVB=BO`S z8t;+D{p!7qN*~tyxvW~q8ls^|2_yD= z1WP|mFqI)bJRc`ejL3Bffi8O1v>9GQ8m%5ZepI!U`cZs(M8T8gqFTN{m7S`i(OEI7 z!AdcEBngDF0HCLhD6ujdfGVQuQb|Q>6c^P*b=^pzV^P{c!Wgo@{2*-GtvGpPOtgqt z^axca8&!tX*(t3BQEGQCqpD+t&W5J;YHfe6RO!Teo}Soas!4R~;Oy^<1Jj95S7>@n zotkpGYf~GC2p*FyUn51Wv{ESv<$1MLGDfsp)8t}KnBYIxF}4g1ar(*qlF6E15`(~< z+M~#5CLcW{aD?EKiFovAjEX~!o{&iF9og6;x;VD?SRKBh#XrRDGBpVy%? zcGSEtB^wy|CumATcAPmGP(^N@P5^o38W#Ht4EE7DUY+Uf!IXl<)$vet9C9WsB(W!i zP0<5^28F-POHPSFEqFe1J0CB)?VxiBmTkFs`Ls0bTA+74cX&!-p-oBbAs9mygPK=NZe31@ zs%qI}%1$(f_(3sU@BqONVB$7myK4F;fK^1EuZg@znSC13P+rIRjk0VzPXN=KDQ`E2 zDe8C~MD0Y)uJFvT+eH*0zmIs}BKoZ$Hx&rMOE%|YzVq2m0&2E(>c~j8(rNhFj2LpR z-r!!yU%nTJE6Em>c=suCpDA(?DLdE!ynyc9(pR;&#Ma8nY%Kq+H_>(ISMY62-T)f_ zoNvNx{m!gVQLMLm&JI|ijfIqBsBo7)asp49j_5ThpYtZ>wg}V7l-*A1H%aE1Ah`vEX1`7vakFWOX$h=b$0t~^+#6#qt~s#BpMtM#Vqg-3hK z8ge=+M$-G>@M$=W9NjNweX3}~LUguVDrL{vg;P@OJtbydasT?RTNJ*4ABuKNrwKGJ z-kb7E+F<0eA~=#BP!e2^MFfa#U|Cd_0(C`XiB+;rMQf1-2(A!Io0;8bungLLm+?89 zaAX{lO;HHb(NQg7Wb?f;rc5-2?Ab=~(oU#Ac0nr%)+qO&mK`@%M8i%T1lO?%M6gGG zV4)}pBw1#sfI3B{8Pyb_S|yk$qNsst#11g3C^c;eJw`^l=-zM74A~B)Es-XuKQq^# zf{{Jd4{jNbNQ#WH*T(pnw529%o$+oNhzm)ipULcgij|fa<~AX|5U|Tg?A2hT5llz< zO2GNk4^&Ye$76j*QIg0ES+$hOCoR+09{D6?`;vKYx8>e~vt_X-!sOwDCzX3ImC=B4C>8Qgk!PM}V6m`5~%qp}T7gGuZ$1;WuV6M@v zsPbK@;N&uSD1HYk@mk1bq4BSJf2o(Ovzr>0@{P|;mNMuv;cqDRP zUNR#RWQ}h~c*5>T;Dnr&ku?kH2eep}(<&Zw_^8-LX{oWW&;e_@9?}TYWmIATd|BI7 zPy-!chgb;F+VT)1h7vcj4}CdREhKFgG&@%wkuQCO@gBRml4&bgV|}14TJG@%q=F1S?N5lbq#zE$g&|5FwlHbIr`oPTc=r5-1B^Qc8wdT`-<_;SPbegRgK!HZoF00lbWf z#+IqENM05`rL?MM1nGGpBq4yz1p{%FJe-?{nT%bpj? z#o}ev)UGhYeM@>C1RsI~xKNH(OIItOHTC>R!@82kB2Ta~LN_a1B1K+h*kIg{-#BIg z1D~uONg)yth*uFJryeUrK=dF!kqh~;V8_G;y2|KcQ>0{A=M3+b52Sc5hg5-lxs*#{ zh+cn`#3C+X)bGY74+Nb~6L>zNRen0gC~5iZ^KiL*Ci8xv%Z>4TiP&tQ0;>m^ZxCUf z1z_turiTLucYrg#xLAlz@c8)OTyTcDp7^chKWUgVxo zhBav#>o~05h{xzEOyY?iHZG7?CHKn}inqf3j%DM~X^ep-^^xoIh5nIt=}OIC$vhxe zYQ)yC6o(XetjGRHCIF!_@?rMt|L6}n|NpBy_wN4{{7QdP3ZxWxMGE{;=8gOBKm7S$ z{+0U;z_#1a*+RE%zc&3bI(=MbS%#f}$neK(Uc~uW+lxxKBP{8=jv@4;g87@~zYL`(43xal#h&KQXL3D9w z2WSJfXfo+IE<)g862{^o3%2dpfEN~eRY#SDHflPw@k|#{N|9|_K0%u6Q2OHxrXpCO zhqsHIzdqG3SfjLizs-6SBoI#;?1mAK`Q`JVqgq*$?DSWg zgkhGX=HZtm<`*{m$!%Y9mUPugxEYp|P6Py+S71pL;Ru1NjzWllA5f}wuA6h-`2cZ) z)c89Q73v0WT`JFw8RnTiJ`0ZoKEh?6!W-@P)%m%Ms2R?f-i=l%5afy5W-?K zkjVK+UcvxjRV3_^Yj8-u1XUfn0Y)tp^`k%s1j%#Es+4aLdll;}k&HY+dYo3+KUdU@ z)DdYszAYtfT+-;t8imyyRclTWed7xdz`D3_33hP8*S=}Z4bdQsvKyCZQavC?+# z55Q*P##VLKe&l z!4}0+R`rJ^=vR)ijpvPYZ#}AzMQtgOS%Do*Qi?{>xRwlOjJwKdN2FR>6Z#wR{Rq#k z%R$GE-e7c}Nb84=CEC3f*!Li)4fzK$sBMVZOUpxKQ zKF&duwnpLVfsNw2&9Q*15i-UIVf~|GVAL7XA8&Z|N72mja)T)?5*+0E&lXTqJ*mNr z5%Gea-vSNA{0WJnR7VjDnnO^|S)E&yj;U-Qd1ulJ9XEa7)B@!+P~ZxnqWCyrhppm> zgwV2T;%prz?=&Nrmc)qLzn~;F^Ab+2xMZ#J`81sjHc-7U*S5**7lC!;L6c&VDqjp6 z?l}u}`ZIr+6M(2O1K|f;tf!0|p~+jPo6pak+aTg@k!A=bO0nz{7Y5Gg3wd5l+N6db zBgdrjTH7b7;|fcSRAO>=_C%@{-&WZ8@F{Txmi#Vk6zg1yHg<1*$<^X3EbA{L2W1^; zqO3j2LXA#O$tl1jMZg+T`yyn^oSH9aNg4bK6i;nyoF#l)5?*RsBi)*SbU%IBcu1Ba z`Tr8o`^IEbQ)I9hjf0T(D9#?9pnXIl5A9ct_p8eLRR*CCMb+kunPMrKH*oDk=Uf5z zQ?VgmBy~?K`3^!}ZhE>j!oW}mLg*lVDw zs~Q@sZX%a4!YZYhJcP?7hmOCB}apr zsi90j-$J}dC@7N$^J+Vb5l9e?XDQs`U|=R%l>JAxWjQI@NBRG`!yo?v1pvJE+DG^B zFa5b;3j8Y%@4xZ>!~f8~-{_bKDLtJsm~&D^rEuHULg|p&JdxzWKTb% z5<`iRN0Iho-kMZ`fT1D6*`Xl+B*71CgXdqukRilbIuIF@%!W*YU}r^dwGyq~2EFen zeIYgNwovl~;6<_cv6zOvmfO*+L z{L%YcdsRT&5!gSGm@c|#3r7!<{^FP)uxTQ|LIoIFAPm~opzSQ>_L?9}Al?2z zuR%@<(Gb*)FA{!>M%!t+ZA^Ev+q=kSi8liH@tZ0P!URY`H5WWYl>zK)7r6^qN_|1- z!p#y=c1&>5_@4N1)W+kYH9w;vG$^7#X(co@Z7NogoLVBthT8&;2{6Lm2BwFP-59AQSD^`b9>DiH z?H}m0&E1Wq zv5gHgQKpr07gyCnojQ5VP=>?;&?aTnHZk@V)sSi9jKiy{1k(j*8$+-QIvMu}>W#p9 zaMY7TgeqN9PJmu~qT{vvXqJQYaNJuqWDv{`%TK@ohPXp23S#geI=DcHvmqeJcx=DR zCpU{@dRd-egoxCz18`t|K2S$D+y?S+Eum%9f}+7>7K?9-R4@#|cMwEy=t&ge{xOL7 zJRd>a#7Fq(#z78O1msAZ!8qs>qQE(@OW*C^6A5ph^vxz0`wC&s`S|_*-3f(!^vM9l9f+C(Ikhis zgQMXRRzBhiUP*DY-;me>`vO<0Af4JorjR8z{4-?NfVq+WwCjTf?xBvjewngOQGk7d zDL{=P@Z(1ORK<&)s^F{4b2%nZ38no2!J3;wgS{z4SJ->Nu@h91HH|t z0?_S}OuxShR&fwQ<9l0HX+M{^7b>E|CL|(rq;5-Y&ZJ8M?hHtGUi&dQpeaM)B$(){ z5rdaqM2}evPDc>~-JpUhNM-_sEI-8I1M&?r_4yV^4G5Iy+Pg`jUCXfbN-FVlR9C;lpG;m02J;8)J(eMk5Q_SEpL*ulh*(Rf}T)2cMPjD{f4)7t4<0ngp$YO2=g0 zAkz(2lNkWAglti;NkBl_+S^8U9a~gVYstXg!#?oKtR+*e%hsJUS3QRi4FxJbV1UgD zfat+NBaf8*0|ByLH8Mq21s(j@12MkAoY%Dr)mBEQ7 z*-{RrS#r4qEpCtoJ1#h)E;c%$=~6Q9e5(b#XO|d58+LDe6Q9?QcRSEpWG3z+F;KAt zH7|T;y3xb%hISMn@NhdyNDq@CQt+b4CKZ^5(IgPan-@9}HHd-&^8-nCI>FY*0Ah$UkUF80d2a+_#gi2Qr$giNtwIT_T z&*wAcB=V(ft&o8dzO*#FHKIPog3VxC_h&O9KVG~=E}#bN18gU*;4C2yGESE3)4`O_ z2D{*A4+t@4+lxXbO1DUegfue2-gNAP#S9A|klsAMXWI%Lyzmfp6We&|vpLfL%Wv)e z*R21)@gLm7zw{@iKuUp>0x1Pj3fy@L{EN$9d*cTWf4_YH1P&PZ`=&G6J(m}6ow{P4 zLqf!gZa|cTaEe$8v4hGaq z4Jrs@K&?mHf{comA+~eNQY6Xd%h@q~Wgo!MR-~hj= z#}#>{)sXMtGP)ENV|oNg5R%_qme!?iC5=BMaZyjD$*0FbUY4@xu<#bRzO ztv;QI(VAwL?+63y^*X!ZqFIQR5P^SUrBQGZOu;&0m|)!0Eej4^LiNOY4}sblmuG>= z7djRB4FTc_k1sD%-g?0FB+nwNp}ZscEBI%7LX|s9+HL^FLHiib_D_4*A#nv7t~9E9 zp-;uEfD+mV6v3O=bLduU{cnh9qfts%+7shi(NwGnc!hxuB;gQrfYB1WUX~ZV;_kSB zOC1o@2^ATPG8bYpan#mK%x@&O3yDLZxGQj1neY@)VyCoIpq}xh9)8Rn*kdoUBvERl^oy*JPcCDc0GI`)O;4+9^mRRV`LJ*bY9kj$1 zBgOU!CV!(1=?x~dW*cW6POYuLg=H_feLf{hTOEvwKyc*=k5hHJs$s(zxGAwYY4Gx8 zUF~=cA5~BlEI_KqlCEz=S-VZb<)pai^D4JRoCKS#T$Gj2`p3oPG`DJc=lE)bz6@G; ze0M+Q?G1NOyuT4Gc73j*NS-oaDMnAc7xR<>ONkq0exhl`8L?Hb!FMynyOtqywc*&` zuA!1E(P~s?oxo$Bx(JhiRM)Ri75%@aDqKbM-YoMkYVy-0nFX%`Bm}U z#V(ap6YJ~iVE z&=3FggN47;{Q4WkhkyOS{SzBjkLip`*9v7J(c%Iy30Z3RIXaP$K|ZSxlJU*IOerYX zl+ewhYzYW2gLDZ5IYqv!ONROZ(xu;@@7f_S8TWvRx!XpVY+}?z2&4_^a1t-*^}tOs zBKjVM^$-@!COk-82l^?w_K^%B9&e(q2tL|Hj?uL_kluji45I%CF>bxF$Lyd6Jqb6_c)L9>|WZTRo-iD`aXK z8ijx|(ev6EcY^CVqSi`n~ zv@`@fQXm%!poBHVz|jkJHs5GOZ)+eb64BhaG`HxqsFfT_c-x5gfcOcdaUgeB#}^2X z1!a%z(V1DKG&k{>9LAr@#Lx(&eHGQwDc!Qsp9g#pVo}7?kuN0?2X@CI1G0T%11-6E z&xcjk4s;580=FQ;5G}I_BW9p+QJs}k348~!g_sho%l>pDm~NOPLsLI-z27=%M=ja3 z$X|t)(pICz`ta?LFEI^+&6A`c1=)@;j3|N^k|E`EMfMzECT!z)NBjk4uHi_p*q3lK zB!&zVNhXD43&pu&CK+vUy_EoaB*~ZJaY?jX8+oMf$R%}^r5O4D|J*PBGL;;AQ)CJa0pt~Av_$Z*Ta>gg~D|}5Hrj^J*ml-iI`$a}ta&wS4NtYX#vXentm%Q`? zL{;f+*((Cr?+tQLL-1*&T_QaadjP9Ed)IQKgBTRj@MH7U4s2 zD3(zgZ3oTG3WDvN;h0B=aVRAoLqa!Db3s~~ZQb#Ggkw`|G4hyU#6-#~f}lKT^aq?G zOZh0$7lVOu0-}T19LhW*Q#C4k6SpJ|&LQ7%HZ82mY`7*O1ekK-GBhRKtrbmZ7Ozz` zyCGDXFvpS$ssz5ho)M81#*<6zR89qTe=N)3w|=$Enk2CVyjb114W<@enh5 z>0SPwA88aEepBRIExRm^8%$YBp@xAUi-hGZ!r6{Lii5=ZiLt}+`Qhv;P6aYie}9k) zI-E<28uU>*DbKilYGPHfI|0=tungGCz`H2PuvQtRp*JdPsTm4l{61tYz6bq1qaBDG zRm$3$z^c04?(Fa-a~)U`kVG1`WqPobQE2ZXuFiU2FX>scpp}$zCg&)?8OkUnqhKoe ztZ9}DMbpX|NeM=F1X~Udq2%Y&wnBLcQGP6XC9|>+q=&e11S>*{B*{hY<^dMPjNFAY zgB>_Sl26Ddrim>E9x+Jwhe`(Mh#C^-Y5(}lE^rDT!v+`AJ4sK|YK~B10@2Y?clvGmBO8YHAV$L;9oPXKxqdZd&W`p6W5vf7$lo>sTduUKav&s_PKoC%x8;6 zNwLd$P08s1x-I3@qJl)NvYE*m*|Jquh{5t3_p@)P}3`0)|Y3WY>o^ z%L`gAM8g(hvBgJ^pfsCHKN9a2CAJ|797J@hFYQ3<)oJ0pxz$MI2oE%sK^^9Ai zZ|J%;$u^k17<`>!$WwtCpQVZK;X<{*oXyTQ-^{<*R+!;fG7r#U?&3&h$M+cg+Tvym zVvJx!{A8PWN5s&;G8QF?B(uxDL)RH)zFP7q$!R2uF(k8d_?UHae3fic6F;JNy844` zgC9XXRm<8bzriyw{+#;$$Ksig3{WG+4(6weGyW@isT_KFhRJ8kK4kE&bDCHkF>SE1ms=(bDu84=6R za07PXcj!aCy!=uua#N-xHqbD(CsZy-?!9TYC+)NlJ1v}hNqh$xN|g9BM=3-49P_1@ zq;Z5Vmk^P>K#3dSY-=`?tLF1T(5n{0m;P)QMTwHQ0({Q$`gAfxJIykGB=8<1r}v4& zmS&ADy6f~^$^G{oO`qF9J`p{lxO3QiHcj%4HtIFo(p!09X^7$&c z9O)zM1F#3~Bo8AJ`%2=3d+cPI@A3;>u49XCmkYpDv+|pwm>GO(N#c~N^ijk!E|n+~ zwnX5D*q1Tnv1{faZu4=57;neMIfl~&Y%^rB+j4BRP*zJyNi|d@S2nXsIj0&3#If@E zqEj@rVsVa+xjD-&=1Rb_SL{N$tmN`VO(|#d86{iJ=d^OZV5zyJ@P3$z!*@&^POPU} zWgrv)-)S)WP+zKc^Q7%*L8F{GpM9ia`pt{!i5ZV2bcw(jzuS&6%~*e@ZTrapKnF4+ zB=QPmXg-3*ZL(wXvNX&_51-Qt<*aI#vx;MvZ6%ijpD~^DS97JS8)k{gn5%=q} zRV~=y{iDDCkN$}L|8L~({qi<`r9UYJQVOILNGXs~-~tN#{;z%$vG@PQv-=4CiBV05 zjY?JDuX^@NGqgYa>7%yW_UuRRiAT3-zyh)ltnNr~blM6^0!ialAzLmOwwh7$js-J| zl~G}4F-r<+r5QQLEb93&VB94h(4r5d0ALD!B35ivgJVnGHViCql`SG1qKaTfn6=*% z@rXVVi!QQHZTSA^ro|e09r3pVB;ZfPTBlyug@bqDC-L@>Z7Rg8E&jJU%uHqj;@F{a@rFBpiA6U$Pus~6B=mOHfV*>O9LRB*fxX!K*{8EDU#uB z#I_^;UvAqTJTjbxg9uur&rUh{$MlSy zPZR9&W7m+CmTmDqvOSPT80bzroI_-iZ2&zeRA8PbiHeQ38LODnGG$;YX?6~W3GB7y zlATdZ@*QXZFDVwY(r8=C!GvpN3QESw;7VP~!MtlZid86-oSfm zVuYU%A_W6CQy7B{j4zkNq$Cz3V!8Mb;q#^I_ah(R48~2JG4z-Hz(ASX41Ppoz9uwn zHh(}Ul!(><+A4>MF6hlVe6$P%Ie~Cv#wMJ@9Xu6imH0teW=+wwe3xnq(L*VR6v1W& z9t!AH#DvJ>xLnE^Svyn2A)Lo?>^P2Ms3k{H3z>W|Z^J!dnAG@?U-61vhgVk$t_pGw zki{yldOsY815CTf=?ys@zU=Qy5bwUroWdgDBu8Ww`H0y^9^o)A>ICi!Kv6{60(6al zT$#v34vP06c+MerI=sRNRC$angM--APhhH!hY-KJ zr2#Z$V>j6S{U(&MI?V7;oY_W6&rYOxU8GEOqWcjrYT2i=wRR&}2@Q?`1pcE2GKJcY zSwl&k1*ap8whNqW6i|TrN`@Lz)^sFe`DPuJhmfBHF2nx+g+Dp|KcW8r6EGO$#0P?t z9w>19ScZ7z0tBJ`1X!me-f4yJ^QVRkA=m9VKn>Au8F574X=dS!uT}MZzSc4MM3C{A zx+XPf^SOA)kz-Mi9KGhaO0^h8tG$%5b!v&PGxHztB{HywbCXBe(Ghu!RD|wooz*Sd2MeydNw^wMtvH^lM1~v3yqR^>!;LY zk>c$8(QUf@oZi{|43k@yv4J2*i@uQ|kupHt$#*d{^v32;NJ$R|k-PK}{mybX9g4F< z4E_(Y>u-w%dWa=ECaK6VqZUW^8Cj0A*ZDTsY@TpmIM7r#j^^o|!{5np+bDXKCYio6 z3mgwN_R{+h#cW5Iz8YZM1(9(eL9W5zt3wNj$?}m?h|)!Gkn4@fLr_z1QWW*&lN}X%wWW zbAtv&uV!rHC-ev;6mU>K6ELE@e4c|vhOZsLp}vmx+x!-RkQiiH{BVKqmeXf)HxMXH zm&n8cBmV<#Mt(&hm7F2-QS_ItvzC`}1D;mer2EYL1e+N-YMyq<9I=QjYLuI4gdJEj zR^d}aiouGS7jt5se^>B(fs%U`8HQ|vi)l>T8T%l~VkU5*Np_|>br4aMp^cI@+(Fb7 zp$;mJnpbrEZ{WXy&cXbIC`(wQ;OI0z5L{@kS4jW+=O4>zSffR>j8?swW_?z}1aa1s`R?8I34XoT- zKcm$!%tIaKp|nJ_;UpU@6>E=HjNvQ{anjj7Cp{J(uS7$y@LyYQfAAYDHr}OjKRC3# zu!r7jihX)sv`ggn;Crz8(NY{*j39qph6rX+I;e98^!J}W^XQdr$ZQ}r%cF6$pYW$e z;sj`chidSa#NG6jNxKV-vB@0A;;&2@POob+D?nG@+@S0Nbd17}!LsSK!J){8Ouhm3 ziT6|HXBK~88E?`xmhmI{V`pmgN3ti|;LUGlu@%ugKTDSZ$&N0VDr$~dftAZ}FtX{C zItRS@BL+td8$8L(fw2)2z*`W59_R2wi7lgX$+Wlu`H92a)iQrtrO`yUm@xJ7t$d(Z z)RQO}9FLH_li!q?@#92h8ZmFiBjjY3c^#O!Q#>g;1Hnd$>_4*{9CbAC*X*~ZV9 zf7_C8N>W<<$1d|p0AN45P%?P4Z2d&`Q7({}webiI=Cd}XYS04Zpm5@9yK!IesTv&` zbP8r{`)41B%8!4zNmRz7kSC9vd71EmJS%3pNH?%#g-l6arUzx1QRZ42J$>hr32)Xn zBE506Vp7pE=3c3235J`BmVC_IkZ3t3{Y{Ye#;*BtxMQb~gOiucA)1Krc#$-wAQ#(i zhe`|C19J|l9ENjvQdIU!i?2<>ogyodHPXtc z9NaGoGSI031sN%}Yyeds!Zz1tqvtDe%s8r+@$(f_fcUtB%62h4ilSRZCtpC8O36|` zXb1K8Tc{c4nlD>^*nY`ZHQP!XAP3Z`p-8^7kG(hK%YC!yooBo*bm9NHDn-MBO*gAw=IJ3J|e)j8Zh(j4%TqqO?WbvZ-h! zWAX#A1i^a*jFdRvH(*;9%P4MGi#(Y_AKa2B&_k07d}Qrff|fDRR#~)6plqRlIK}|z zLK2`C*9SjUJAMok5-h>I@o)@*#oxb0D8v|N-6 zG0PT2qns_mDyabg1sTxE@o(sLB-qG-%h%J47#U6Hc~j3Dxkn8phAbPziY17MXquwn zT%fZyd@7?N2{0ozuTI@7>xRc2r4()I$gA2$jXMyWu`1CUE}HNh0O1IL70XGop;xHN%vI$j?S?*> zhJaUCG7kBGxPX0vdLv^?Y}wuY)%hI|Bm@SLYliX4PJBAn(<^jK)^;@G77~4vbAXwm zNk9lNSW#j!MtYB#(J(VA%}lcEs~#1N-Ufsg{oS>LP6NBX1$;~R}?|jn$avJoMe~siF=gF4;V%u0U+8$-rS1EF-ZSrosQ67`EiKJtK&(~twI--d>N<` zq59G7cn}~57!#JC-U2K-V8bDCM#;;_kRv5jhVDK6y}$79{Wpbg-GBe#&;Ql$+=txO zuYt&h&IkZ>G7tkP9-BUj_F&`|kv;Im-oY-ZT?T!?7^FK02?jDJX>O1=@$?z-tfO1x zz5iUqGc{>|z+<;U1xbg#((Z<&2>v#&tf z3rrtq;X`S5wvI3+9dL0aM=vTlH3Q5zOS2S&r4(!x=y8R7DY5)U;+w8L0~MGqmNyrX z@0QE20hL(x5n?xG!ci{;=NH~|xYJ1I)8H(Qut|fn(%>vWr;^p)NrSUer1x7xdcU3E zEXa3%hV=i}?y2|a-&8-pxAE;aK6v<_EZzs;Hq1q*(+Ymq0LPhNA{q-_PVvjn2YVm$ zTt*^ln!28*uBWN%to?Bm_-){qkNGAhFtf31($w`dbv>@`lJK%D*`Zs7CJRRkfZ2f9 z1EhNwHORzMl)BZS{q3Q)4Pa8ZK1RiEelJAH86^HeUtg^GsPljvWB4sP@WoOM85ac} z2r94#*EXt9TP{4-1fgQWyER{Mm|IoV;P3sHvw4ktUT)TzTLlhN-k za;BRol!o3aJ@^;lK=%~CXQS9utA(N&1TkkA8meNH3QE}l!cVSThO%AOG$oTM!RTj5{_g zI_II`h6sd;#{A@DGfd^ z=j7+>R7JRu%OWfbm71~zC0ojtl&ng>N?Fuy$`($`60Bc>jdZ zHK#1d^qj6;(_J{6$jUEsmEdkt#C$_PqiP7qQ?2^I$(y_w&1r<%4d1@A>}Xez*I~TRA*!SSL?b!li9vTYH|* zp6oc@NnkGKR>F_h&*pt&qh)M*nHPDzt!tIF{MLSM_hjvC^Hg;+Ss&??^H06%;=-P8 z9j`b?A2+=h3)zF!QoE!ab~7j1Qn#mfcf4}rWyh-;+ve)RGpkfDH$VPZ%jVRu^n4+= zv%L81`9^j9V4+msUksm|t?xa3+Uq^p4VIpshKrjAg%^eD+nx9IwVjN)R`Dv@vR7#w z;{Q?B%9@?S1GQ#1%AJkI=83Vp-#cpJ8v0&4Y8~zFpB=3{-`_c|dX?j&;OK>is_#$g zW)GjQ9eGDDp5fXj{^5bwd$RY$J8n(z;q%I_zzpnYgZcT3Nz#kB@w`UDprLKkZqra#rtd9@oz*OUvEL zUcI;J?)NquHO*)|&sa2;)snoYxze`ixvTrpGoD26pK#my+7ZSE_;kcXMw!w|RVs zzo6yuv)tCwdVXUUbIR90&zY;I9qNB&k;Y`X{JQ#ffRwcN&0M*NvgF00j?ylLtX@D_ zlybgoI$1bA0mDWcZ%DFV#nxT2sd5_`J~xo^yiKL$Lb26EwYSOpdcGZP>{%p5ANxHN zx3l4p96ScK&`jil!b^;&pn~0e5a>PZ|3UmdYQQzbbChy{)m$tDXnU+i3b&5p?Hemc*-iKGVR1v>NrS0K~a_H`=msCjhM1A8Eec9@LcyOeH%%a zt0*T((eG5^BnI^b15EZJ`TiU>I2@Qi2M0NEW5O5e(?aPoEr38ov{3498+b#3(qvjF zA!va@FwwWtEnX-WR5O!zl&qm=Q3A(+Tw`cO#j;Czy$CU;m?GCHhGJfJkk*5=9a~4h16X6TjVu@BWcQ#(V5=|r;6Nze+19~* ziozj|xr_e@Zh*2f#z2zWKdBU4O&1ztM`Ir(+nbWzfH|W)fk$Ho*WodvfK)VQ#ln~Y zOTlASiU`&LZvp_C;ya3_wA^M?TS{|m)3oy?#VVpOSI)`XN(ms?ilwVrt5C8XE1yp= zoNLos1LxmH%VU8zul~fFh_qP)J;h2Vpxh`fPIXSjDMZxV5Ndl`Sd#KA(!!F=y>Fth zBpB&GMG7yd%IL;MQD!w+9JxaHLqu=-ErIBP&i*5AT>U|oagP}d8i@A+vq z_5F{g$1n3S`r=5wT3x=3OHSkR&RibR2EayBk zDC0asVR0cxZUw%U2_bGeMIf@(&H^w4o6fItOz6kTMYCR$5v7$bgU8h#G3$Jib5*W&;?>jQx-?X zm3A5E>s<1ISfE6%m*~Al<6V*k6AqhesH4*eZ=B+U2__M(sv{X)mbcfl`5YX(sIA}_ zB_&sM%8FieP!829=Q2R3H_B$RsMDS0UI{`@A>t-4Ndq0yKnL2gX`lmYWgJUMjDDxN zO^;G}=B~*z33h*Fi|d3_^P=+rX;RXnt7wJutE;`LjFlKvvP0AKIgQ>R7S0+cRofyD zBmH(PE_frW2634lwU9<{G+E%6A|3QiQVg$H|8MUX>)#{&Uo5iT&!wa?X7(MTxgiS| zaap0%I>6M&P#gMENCUAoXH`WECa`F$tai#BGJrpp=eq;Bk3slywEchYy{Ow#1YALXCaF0 z&|Bot$@h7!Dq0`f%X|sDS519~-h5sKo9j$?iZs(8UJHMSE`HnqfVwCdKNPYirj>EI zLZX@1*h^x_!ii>4w6!dcxWqM67+ ziilP&ilfLlS<-Hjn;`jqjG-nkxFxOQGm9Io=*^C!o9#}3`J1;a&@x0)fSlQd{>-i* z?ANsMMHPi|I61&Z^g_kKJH7oLDrjgT=-3F20JL@|2*u(qJ}YJOGeRq6G9?sNS)1S4 z+gaWfi?(H-&S27_2PRv3yxz22{T;D=(ng7uHX7Kmfls~S1}I3h*Yy$Jhf+Lf0%3a# zdK=*roBpgI%%Qk0g>Zz>oZnUKuLLXV0_#Xa3Wuxy3WmKnG_%82v<1WadQC*BX7*3hcvA>|Aq{Y%P5~xw*0@zq#4KJz2 zp<)e5$1>HZl+R{zB|}GQ0BYIhvSwD%&4Q{F)OjEyVXce8d0YOAVp2|JHWIGxSndM!{ePBp{@iTov6`Whh|qo7E|Q{XbsA-% zmdo)=EtgM4le;IH+%UqO)9$tO6Bm`OT&LkWYQxpp`wIjByRAR0SzgdW^G)?c6myvGbYjsgP1PU6f z()WQsE%($FZW_c2qNJMDZD>{#M7llS^^n;4HW`R&fsgd{MI=Z9d6bgeBa`46s=@l_ z*Vq^bs%g%d^bpvV6vDhlDmgOo=i1zq!e8O}j7_8JMLSo}ONw1U(snMN0oY8r0F;_C ze9QTajh}+Vrm-^xEtAg`fiznzD>;Z?3TiDElw7Htaf-T`&li)i{%bmIRan6qK#Na~ zFSdey0;5?2pxemTk14QAX|Exwn8H9f%aemB1AF_%P!_hYEcDi zL5ad#zKDVes7a_~%lVvE&KE2-mrbZ^X_)+73tmjDrzCrrz+01-q*+m^K$!}ZXQXM| zys191J;#7SuDCxt&6aM~ZR`9Dl^aX(j`06J{0qu|O#c6`{!-!IH-7!=Z{kn-lTsk1 zKuUp>0vA%?UpIaWMNt2)e;<$>Z5vn|Q=)2;AE^@EGH;<&mI>skWk&jdu3qs&cwdmm zwWFIKcVL)+QVtNLZ00QrNgC6wlfvHE0d$QRq$he)4|`Nkj=CDZ17e@mF#-LzNflX0 zQTPOrRh|pogbK(Zghy1!XfPpgYQiBnW->ugQ$$BakBEZOJCYVkAZejUJQP%CH}tlW z%_yGVtl9y+LzGWXQI(d7QgaQ%m;R~@@+_R34P#0%3)@P8aDP?Z1VjRqA>GIoi&oNN zG6Icvty$XOKGC#MpHQv_-MMLyyi`rQYiimT*|!BQ_}#zp=YI2z_aFY9x9(GEKSWSY z$?KlWtK)b}8(X4El>n6pR1{{(b1pfDXH3QsZUm;c{c>}+qHI=9yw&Gt-FCP{IKpR* zHFtB<(KagwPqx=jmwMZ-@w9lbqE*8a=c&8);_NIq-a4%BtYprP=8xCZC{Sv zuGX<$bi#!f&$XA;dS%BtaCe_)yS3+w8@7L_mwKfq$A#L-{N};dddu27c9u>1bY;7? zZYV7~wO7`z*hmowWc>(zAp~kc8w2}Ekahs;fBA- z6O7KBtJ>tpj>WMLlC7MxyoL?^jL+%;0lo$VY{kKNaJKMxlu(XGp^{;W(dj9E(^im7 zK!4@YqpMmWTPoN}(ZH`<4!}GmN6jl%83D9*k&JlBqv!e6X*ULoc%5h51Q238!Txyw zh-o=b4gtP}6x>(z@nEw*Br#We-T^u{P^Z1l(dzRYAx|H^SO8-7Iw5N}>Ai7K?H;Z-CAcQ5|UL zdPC9)sO#Cyy-KEXwvOkoYn#VF7C)r5rMl zi%t&ZFjPY?p@_1j7tI_#%anm#PYiM>j*FPJ)rS*DeP)TVOl5`R6I_Ok;i~wM$5*mW zKopj5B5EFhsR7D%p15|GAPpI_j*Bct8Pv0sp+U%p&_HkWJ*2?E&IHEDusnxJ}4g6f!^Yr=ZEh~Iyzj+T)k z)}-Go`hLBuBiK3)F@g-f02q-d_ponaQ)F=bgt3j;&rl1i<+nPX4s#fn-Iz1uMmWw0 z(b;-QQ_D(`0NVi$>L6mElvNC?XqQ!0vr5IN2384Q>IMo*6r2L0yh|El3p86(^5rt( z35o^1oU;Q_y&5+dO;uF*WWB+?P$ddyd;#5{P2tijAxmjxGYCb6M!k4%EH zuK0eN3rS0pX8&yz;B^l5Ke-b1n|2pv$CNH4oaWrf8`HJs9F*ZOin+34m2#-k;h2t6 zwu>1h1DVUPOvgs84jHvkES3s-&dMr9vkc2+E?0uR(z28cN|+&QOIMM(l^}nAJm*rs zf5j^hzyFM@(LOEOE>IhQD@Ty+%=7eFu~$R73sOpTi;Q6wg$M?KY7PnioAQx4MOAP1 zM86EZ@DQOFB-b@c<(M_Q3A;)QyhJYU(a+9b;jq(B$yboLO|2-pYlU<1n=UoUoW;(V zMKg;ukgP+|h<3?P%4XhBw4!0y1+!$@8BH#2wUE(D23CDJ>yS~zBD7+}IvQHm&g62o zs+LQMrM(4WXh;c0q6DWwU=vY+prKKpyXHqPdxeER`k+f!Kd05d{F}f1&y!mHx%p93 zL>Z0{`SO-ExUtBRyPMJ1cV;RuN&789HtN{F8KG5?wSl9)N6n(MWB@^Woc(=480@GK z)B@ig@qR{C7`Wb=eQLr(1>Y!;2OJ6|XBp*)!pLJp0Lq3?HKbGn)5#VDy{B1b^dVQjr9kS*b%kX82ow|6xy5yL?Aw+P;(-KI^` z^dx)nX9#JM9(MgGF1Q}Vqc`vVT;EK(C}|hUqNtEdp=sKrUu5RZyf+5&Q3qC;71zgE zWfyc+9b%8b> zR0BPlFePS>Q$dk1J&XtnIkxB;wzx`>)7iu~p~E~AQ`;v#hgj*jSym?HS|nIbDB2vQ zV4V=~Ua}6d0RKQ`rh>R%X^>~DorD!#Q5vonStn(Q=KYZqMin}CeHbJwrs-#K|KcBN4|9eY*&&67Ap@?g? z%ZI#pxqFb?_GT+LM&)}LMkNjPCCC({lE|7#hSB#MM*kNwJ(%bxCnU~%(h~rcQPb5p z)XHgCbVRW7Z4VGsd6#MJ4S-XD0I3tkb^3BQJfbi&;-yB_eGOz4P^nbC2B-@7)LWpc Y%9d5r#?jQz8~QyrWX%76y?=lE0D-ffM*si- literal 0 HcmV?d00001 diff --git a/tests/test_web_discovery.py b/tests/test_web_discovery.py new file mode 100644 index 0000000..0cef51e --- /dev/null +++ b/tests/test_web_discovery.py @@ -0,0 +1,453 @@ +"""Web & retail listing discovery (app/services/web_discovery) - no network. + +Every search is a recorded fixture. What is pinned here is the anti-invention +contract: a product exists only if a real retailer's PRODUCT page names the +brand and a pack size; a throttled search is "could not ask", never "nothing +there"; and with the source switched off Brand Discovery is exactly what it was. +""" +from __future__ import annotations + +from typing import Dict, List, Optional + +import pytest + +from app.infrastructure import settings +from app.services import brand_discovery as bd +from app.services.web_discovery import brands, cache, discover, jobs, listings, provenance, search +from app.services.web_discovery.search import Hit + +DETTOL = ("dettol",) + + +# --------------------------------------------------------------------------- +# listings.parse_hit - one hit -> one trustworthy listing, or a reason +# --------------------------------------------------------------------------- +@pytest.mark.parametrize("title,url,snippet,name,size,price", [ + ("Dettol Original Soap (125 g) - Buy Online at Best Price | Blinkit", + "https://blinkit.com/prn/dettol-original-soap/prid/12345", "", "Dettol Original Soap", "125g", None), + ("Buy Dettol Original Soap 125 g Online at Best Price | Zepto", + "https://www.zeptonow.com/pn/dettol-original-soap/pvid/1a2b3c4d-1111", "", "Dettol Original Soap", "125g", None), + ("Buy Dettol Original Soap 125 g Online at Best Price of Rs 58 - bigbasket", + "https://www.bigbasket.com/pd/40001234/dettol-soap/", "MRP Rs 58", "Dettol Original Soap", "125g", 58.0), + ("Dettol Original Soap Bar, 125g : Amazon.in: Beauty", + "https://www.amazon.in/Dettol-Original/dp/B00ABCDEFG", "₹ 55", "Dettol Original Soap Bar", "125g", 55.0), + ("Buy Dettol Antiseptic Liquid 1 L Online | Swiggy Instamart", + "https://www.swiggy.com/instamart/item/ABC123XYZ", "", "Dettol Antiseptic Liquid", "1l", None), + ("Dettol Antiseptic Liquid 550 ml - JioMart", + "https://www.jiomart.com/p/groceries/dettol-antiseptic/490001234", "", "Dettol Antiseptic Liquid", "550ml", None), +]) +def test_each_retailers_title_shape_reduces_to_the_product(title, url, snippet, name, size, price): + listing, reason = listings.parse_hit(title, url, snippet, DETTOL) + assert reason is None + assert (listing.name, listing.size, listing.price) == (name, size, price) + + +@pytest.mark.parametrize("title,url,reason", [ + ("Dettol Soaps - Buy Dettol Soaps Online | Blinkit", + "https://blinkit.com/cn/dettol/cid/1", listings.NOT_A_PRODUCT_PAGE), + ("Dettol Original Soap 125g", "https://www.example-blog.com/dettol-review", listings.NOT_A_RETAILER), + ("Dettol Original Soap 125g", "https://www.amazon.in/s?k=dettol", listings.NOT_A_PRODUCT_PAGE), + ("Dettol Liquid Handwash Refill | Blinkit", "https://blinkit.com/prn/x/prid/7", listings.NO_PACK_SIZE), + ("Dettol Original Soap 125g (Pack of 4) : Amazon.in", "https://www.amazon.in/x/dp/B00ABCDEFH", + listings.MULTIPACK), + ("Dettol Soap 125g x 3 | Zepto", "https://www.zeptonow.com/pn/x/pvid/1a2b3c4d-9999", listings.MULTIPACK), + ("Dettol Handwash 200ml + Refill 175ml | Blinkit", "https://blinkit.com/prn/x/prid/8", + listings.SEVERAL_SIZES), +]) +def test_what_is_not_a_listing_of_one_pack_is_refused(title, url, reason): + assert listings.parse_hit(title, url, "", DETTOL) == (None, reason) + + +def test_another_brand_is_refused_even_when_it_mentions_ours(): + listing, reason = listings.parse_hit("Savlon Soap 125g better than Dettol | Blinkit", + "https://blinkit.com/prn/savlon/prid/9", "", DETTOL) + assert listing is None and reason == listings.WRONG_BRAND + + +def test_the_brand_must_lead_the_title(): + assert listings.names_brand("Dettol Original Soap", DETTOL) + assert listings.names_brand("New Dettol Original Soap", DETTOL) + assert not listings.names_brand("Haldiram Mithai with Dettol", DETTOL) + + +def test_counts_become_pcs_the_unit_discovery_reads(): + listing, _ = listings.parse_hit("Durex Extra Thin Condoms 10 Count : Amazon.in", + "https://www.amazon.in/x/dp/B00ABCDEFJ", "", ("durex",)) + assert listing.size == "10pcs" and listing.name == "Durex Extra Thin Condoms" + + +def test_a_run_together_title_is_refused(): + title = "Dettol Original Soap 125g " + "x" * 200 + assert listings.parse_hit(title, "https://blinkit.com/prn/x/prid/1", "", DETTOL)[1] == listings.TITLE_TOO_LONG + + +# --------------------------------------------------------------------------- +# brands.targets_for +# --------------------------------------------------------------------------- +def test_reckitt_fans_out_to_the_names_its_products_are_sold_under(): + queries = [t.query for t in brands.targets_for("Reckitt Benckiser")] + for sub in ("Dettol", "Harpic", "Lizol", "Mortein", "Durex"): + assert sub in queries + assert queries[-1] == "Reckitt Benckiser" + + +def test_a_typed_sub_brand_searches_only_itself(): + assert [(t.query, t.brand_terms) for t in brands.targets_for("Dettol")] == [("Dettol", ("dettol",))] + + +def test_a_generic_product_word_is_never_a_brand_word(): + butter = next(t for t in brands.targets_for("Amul") if t.query == "Amul Butter") + assert butter.brand_terms == ("amul",) + + +def test_a_line_name_sold_without_the_parent_counts_on_its_own(): + cerelac = next(t for t in brands.targets_for("Nestle") if t.query == "Nestle Cerelac") + assert "cerelac" in cerelac.brand_terms + + +def test_kit_kat_is_not_an_abbreviation_plus_a_sub_brand(): + assert not brands._is_abbreviation("kit", "nestle") + assert brands._is_abbreviation("rb", "reckitt benckiser") + + +# --------------------------------------------------------------------------- +# discover.run - the loop, with a fake search engine +# --------------------------------------------------------------------------- +@pytest.fixture(autouse=True) +def _isolated_dir(tmp_path, monkeypatch): + monkeypatch.setattr(settings, "WEB_DISCOVERY_DIR", tmp_path / "web_discovery") + monkeypatch.setattr(settings, "WEB_DISCOVERY_PAUSE_SECONDS", 0.0) + monkeypatch.setattr(search, "google_configured", lambda: False) + jobs._jobs.clear() + yield + jobs._jobs.clear() + + +class FakeEngine: + def __init__(self, answers: Dict[str, Optional[List[Hit]]]): + self.answers = answers + self.asked: List[str] = [] + + def __call__(self, query, max_results=None): + self.asked.append(query) + for key, hits in self.answers.items(): + if key in query: + return hits + return [] + + +BLINKIT_SOAP = Hit("Dettol Original Soap (125 g) - Buy Online at Best Price | Blinkit", + "https://blinkit.com/prn/dettol-original-soap/prid/1", "₹ 50") +ZEPTO_SOAP = Hit("Buy Dettol Original Soap 125 g Online at Best Price | Zepto", + "https://www.zeptonow.com/pn/dettol-original-soap/pvid/1a2b3c4d-0001", "₹ 55") +ZEPTO_LIQUID = Hit("Buy Dettol Antiseptic Liquid 250 ml Online | Zepto", + "https://www.zeptonow.com/pn/dettol-liquid/pvid/1a2b3c4d-0002", "") + + +def test_the_same_pack_on_two_retailers_is_one_candidate_with_both(monkeypatch): + engine = FakeEngine({"site:blinkit.com": [BLINKIT_SOAP], "site:zeptonow.com": [ZEPTO_SOAP, ZEPTO_LIQUID]}) + monkeypatch.setattr(search, "search", engine) + + result = discover.run("Dettol", sleep=lambda s: None) + + assert result.status == discover.DONE + by_title = {c["title"]: c for c in result.candidates} + soap = by_title["Dettol Original Soap 125g"] + assert soap["providers"] == ["Blinkit", "Zepto"] and soap["retailer_count"] == 2 + assert soap["price_range"] == "₹50 - ₹55" + assert by_title["Dettol Antiseptic Liquid 250ml"]["retailer_count"] == 1 + assert result.candidates[0]["title"] == "Dettol Original Soap 125g" # best corroborated first + + +def test_a_generic_title_does_not_swallow_a_variant(): + a, _ = listings.parse_hit("Dettol Soap 125g | Blinkit", "https://blinkit.com/prn/a/prid/1", "", DETTOL) + b, _ = listings.parse_hit("Dettol Cool Soap 125g | Zepto", + "https://www.zeptonow.com/pn/b/pvid/1a2b3c4d-0003", "", DETTOL) + assert len(discover.cluster([a, b])) == 2 + + +def test_three_unanswered_searches_end_the_run_as_partial_and_are_not_cached(monkeypatch): + engine = FakeEngine({"site:": None}) + monkeypatch.setattr(search, "search", engine) + + result = discover.run("Dettol", sleep=lambda s: None) + + assert result.status == discover.PARTIAL and result.queries_failed == 3 + assert len(engine.asked) == 3 and result.candidates == [] + assert cache.get(engine.asked[0], search.backend_name()) is None # "could not ask" is never stored + + +def test_a_single_unanswered_search_still_makes_the_run_partial(monkeypatch): + answers = {"site:blinkit.com": None, "site:zeptonow.com": [ZEPTO_SOAP]} + monkeypatch.setattr(search, "search", FakeEngine(answers)) + + result = discover.run("Dettol", sleep=lambda s: None) + + assert result.status == discover.PARTIAL and result.queries_failed == 1 + assert [c["title"] for c in result.candidates] == ["Dettol Original Soap 125g"] + + +def test_answers_are_cached_so_a_rerun_asks_nothing(monkeypatch): + engine = FakeEngine({"site:blinkit.com": [BLINKIT_SOAP]}) + monkeypatch.setattr(search, "search", engine) + discover.run("Dettol", sleep=lambda s: None) + asked = len(engine.asked) + + again = discover.run("Dettol", sleep=lambda s: None) + + assert len(engine.asked) == asked and again.queries_cached == again.queries_total + + +def test_the_budget_cuts_the_last_retailers_not_the_last_sub_brands(): + planned = discover.plan_queries("Reckitt Benckiser", max_queries=11) + assert {q.retailer for q in planned} == {"blinkit.com"} + assert len({q.target.query for q in planned}) == 11 + + +def test_refused_hits_are_counted_by_reason(monkeypatch): + category = Hit("Dettol - Buy Online | Blinkit", "https://blinkit.com/cn/dettol/cid/1", "") + monkeypatch.setattr(search, "search", FakeEngine({"site:blinkit.com": [category, BLINKIT_SOAP]})) + + result = discover.run("Dettol", sleep=lambda s: None) + + assert result.rejected == {listings.NOT_A_PRODUCT_PAGE: 1} + assert result.listings_kept == 1 + + +# --------------------------------------------------------------------------- +# jobs +# --------------------------------------------------------------------------- +def test_a_finished_job_is_reused_and_found_from_disk(monkeypatch): + monkeypatch.setattr(search, "search", FakeEngine({"site:blinkit.com": [BLINKIT_SOAP]})) + job = jobs.WebJob(job_id="a" * 32, brand="Dettol") + jobs._jobs[job.job_id] = job + jobs.run_job(job) + jobs._jobs.clear() # as after a restart + + found = jobs.candidates_for("dettol") + assert found.job_id == job.job_id and found.candidates[0]["title"] == "Dettol Original Soap 125g" + assert jobs.start_job("Dettol").job_id == job.job_id + + +def test_a_job_that_was_running_when_the_process_died_is_interrupted(): + job = jobs.WebJob(job_id="b" * 32, brand="Dettol", status=jobs.RUNNING) + jobs._save(job) + assert jobs._load(job.job_id).status == jobs.INTERRUPTED + + +def test_a_job_id_that_is_not_hex_is_never_a_path(): + assert jobs.get_job("../../etc/passwd") is None + + +# --------------------------------------------------------------------------- +# brand_discovery with the web source +# --------------------------------------------------------------------------- +@pytest.fixture +def no_other_sources(monkeypatch): + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: []) + monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: []) + monkeypatch.setattr(bd, "_from_brand_store", lambda brand: []) + monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: []) + + +def _finished_job(candidates, status=jobs.DONE) -> jobs.WebJob: + job = jobs.WebJob(job_id="c" * 32, brand="Reckitt Benckiser", status=status, backend="duckduckgo", + queries_total=77, queries_done=77, listings_kept=3, candidates=candidates) + jobs._jobs[job.job_id] = job + return job + + +def _candidate(title, size, providers): + return {"title": title, "size": size, "sizes": [size], "providers": providers, + "retailer_count": len(providers), "price_range": "₹58", + "listings": [{"retailer": p, "url": f"https://{p.lower()}.example/{i}", "title": title} + for i, p in enumerate(providers)], + "brand_term": "dettol", "source": "web"} + + +def test_web_rows_arrive_with_their_listings_and_honest_confidence(no_other_sources): + job = _finished_job([_candidate("Dettol Original Soap 125g", "125g", ["Blinkit", "Zepto"]), + _candidate("Harpic Power Plus 500ml", "500ml", ["Amazon"])]) + + result = bd.discover_brand_products("Reckitt Benckiser", use_llm=False, use_web=True, + web_job_id=job.job_id) + + rows = {p.product_name: p for p in result.products} + soap = rows["Dettol Original Soap"] + assert soap.size_variants == ["125g"] and soap.evidence == "retail" + assert soap.confidence == bd.WEB_CONFIDENCE_MULTI and soap.as_preview()["selected"] is True + harpic = rows["Harpic Power Plus"] + assert harpic.confidence == bd.WEB_CONFIDENCE_SINGLE and harpic.as_preview()["retailer_count"] == 1 + assert soap.as_preview()["listings"][0]["retailer"] == "Blinkit" + assert result.counts["from_web"] == 2 + assert not any("language model alone" in w for w in result.warnings) + assert any(w.startswith("Web & retail listings: 2 product(s)") for w in result.warnings) + + +def test_web_on_but_never_searched_says_so(no_other_sources): + result = bd.discover_brand_products("Reckitt Benckiser", use_llm=False, use_web=True) + assert result.products == [] + assert any("have not been searched" in w for w in result.warnings) + + +def test_a_partial_job_says_it_is_incomplete(no_other_sources): + job = _finished_job([_candidate("Dettol Original Soap 125g", "125g", ["Blinkit"])], status=jobs.PARTIAL) + job.detail = "the search provider stopped answering" + result = bd.discover_brand_products("Reckitt Benckiser", use_llm=False, use_web=True) + assert any("Incomplete: the search provider stopped answering" in w for w in result.warnings) + + +def test_with_the_web_source_off_nothing_changes(no_other_sources): + _finished_job([_candidate("Dettol Original Soap 125g", "125g", ["Blinkit", "Zepto"])]) + + result = bd.discover_brand_products("Reckitt Benckiser", use_llm=False) + + assert result.products == [] and "from_web" not in result.counts + assert "Every row below rests on the language model alone - review them individually." in result.warnings + + +def test_a_web_listing_joins_an_open_food_facts_row_instead_of_duplicating_it(no_other_sources, monkeypatch): + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [ + {"title": "Dettol Original Soap", "barcode": "8901396393009", "size": "125g", "source": "off"}]) + _finished_job([_candidate("Dettol Original Soap 125g", "125g", ["Blinkit", "Zepto"])]) + + result = bd.discover_brand_products("Reckitt Benckiser", use_llm=False, use_web=True) + + [row] = [p for p in result.products if "Soap" in p.product_name] + assert row.sources == ["off", "web"] and row.barcode == "8901396393009" + assert row.providers[:2] == ["Blinkit", "Zepto"] and row.retailer_count == 2 + assert row.evidence == "openfacts" + + +# --------------------------------------------------------------------------- +# provenance +# --------------------------------------------------------------------------- +class _File: + def __init__(self, products): + self.result = {"products": products} + + +class _Manifest: + def __init__(self, status, files): + self.status, self.files = status, files + + +def test_stored_rows_are_joined_back_by_csv_position(): + entries = {0: {"retailer_count": 2}, 2: {"retailer_count": 1}} + files = [_File([ + {"image_id": "rb_dettol_soap_125g", "brand": "Reckitt Benckiser", "source_row": 2}, + {"image_id": "rb_other", "brand": "Reckitt Benckiser", "source_row": 3}, + {"image_id": "rb_harpic_500ml", "brand": "Reckitt Benckiser", "source_row": 4}, + ])] + updates = provenance.web_updates(files, entries) + assert [(u["image_id"], u["entry"]["retailer_count"]) for u in updates] == [ + ("rb_dettol_soap_125g", 2), ("rb_harpic_500ml", 1)] + + +def test_the_watcher_waits_for_the_batch_then_applies(monkeypatch): + states = iter([_Manifest("running", []), _Manifest("done", [_File([ + {"image_id": "x", "brand": "Reckitt Benckiser", "source_row": 2}])])]) + applied = [] + monkeypatch.setattr(provenance, "apply", lambda brand, updates: applied.append(updates) or len(updates)) + + written = provenance.watch_batch("batch1", "Reckitt Benckiser", {0: {"retailer_count": 1}}, + read_manifest=lambda _id: next(states), sleep=lambda s: None) + + assert written == 1 and applied[0][0]["image_id"] == "x" + + +class _Cursor: + def __init__(self): + self.calls = [] + self.rowcount = 1 + + def __enter__(self): + return self + + def __exit__(self, *exc): + return False + + def execute(self, sql, params): + self.calls.append((sql, params)) + + +class _Conn(_Cursor): + def __init__(self): + super().__init__() + self.cur = _Cursor() + + def cursor(self): + return self.cur + + def close(self): + pass + + +def test_apply_touches_only_provenance_price_and_review_status(monkeypatch): + conn = _Conn() + monkeypatch.setattr("app.services.vector_store._connect", lambda: conn) + provenance.apply("Reckitt Benckiser", [ + {"image_id": "one_shop", "brand": "Reckitt Benckiser", + "entry": {"retailer_count": 1, "price_range": "₹58", "listings": [{"url": "u"}]}}, + ]) + sql, params = conn.cur.calls[0] + assert sql.startswith("UPDATE brand_reckitt_benckiser SET field_sources") + assert "price_range = CASE WHEN COALESCE(price_range, '') = ''" in sql + assert "'rejected'" in sql and params[-2] is True and params[-1] == "one_shop" + for column in ("product_name", "barcode", "image_url", "category"): + assert f"{column} =" not in sql + + +# --------------------------------------------------------------------------- +# Found on the first live run (Reckitt Benckiser, Blinkit titles) +# --------------------------------------------------------------------------- +@pytest.mark.parametrize("title,name,size", [ + ("Dettol Original Hand Wash Refill 675 ml Price - Buy Online at \u20b992 in...", + "Dettol Original Hand Wash Refill", "675ml"), + ("Harpic Disinfectant Liquid Toilet Cleaner - (Original) - 500 ml Price - Buy Online at Best", + "Harpic Disinfectant Liquid Toilet Cleaner (Original)", "500ml"), + ("Lizol Disinfectant Surface & Floor Cleaner (Lavender - 500 ml) Price - Buy Online at Best", + "Lizol Disinfectant Surface & Floor Cleaner (Lavender)", "500ml"), + ("Buy Lizol Disinfectant Surface & Floor Cleaner (Citrus, 625 ml) Online", + "Lizol Disinfectant Surface & Floor Cleaner (Citrus)", "625ml"), +]) +def test_blinkit_names_keep_their_variant_and_lose_the_price_wording(title, name, size): + term = (title.split()[1] if title.startswith("Buy") else title.split()[0]).lower() + listing, reason = listings.parse_hit(title, "https://blinkit.com/prn/x/prid/1", "", (term,)) + assert reason is None and (listing.name, listing.size) == (name, size) + + +def test_two_scents_of_one_product_stay_two_products(): + citrus, _ = listings.parse_hit("Lizol Disinfectant Surface & Floor Cleaner (Citrus) 2 l Price - Buy", + "https://blinkit.com/prn/a/prid/1", "", ("lizol",)) + floral, _ = listings.parse_hit("Lizol Disinfectant Surface & Floor Cleaner (Floral) - 2 l Price - Buy", + "https://blinkit.com/prn/b/prid/2", "", ("lizol",)) + assert len(discover.cluster([citrus, floral])) == 2 + + +def test_discovery_keeps_two_web_variants_its_own_merge_would_fold(no_other_sources): + """Its title similarity scores "... Cleaner Citrus" vs "... Cleaner Floral" + at 0.857, over its 0.85 floor; web rows were already told apart.""" + _finished_job([_candidate("Lizol Floor Cleaner Citrus 2l", "2l", ["Blinkit"]), + _candidate("Lizol Floor Cleaner Floral 2l", "2l", ["Blinkit"])]) + + result = bd.discover_brand_products("Reckitt Benckiser", use_llm=False, use_web=True) + + assert sorted(p.product_name for p in result.products) == [ + "Lizol Floor Cleaner Citrus", "Lizol Floor Cleaner Floral"] + + +def test_a_retail_listing_outranks_our_own_earlier_output_as_evidence(no_other_sources, monkeypatch): + monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [ + {"product_name": "Dettol Antiseptic Liquid", "title": "Dettol Antiseptic Liquid"}]) + _finished_job([_candidate("Dettol Antiseptic Liquid 250ml", "250ml", ["Blinkit"])]) + + [row] = bd.discover_brand_products("Reckitt Benckiser", use_llm=False, use_web=True).products + + assert row.evidence == "retail" and row.matches_existing == "Dettol Antiseptic Liquid" + + +def test_bracketed_variant_words_survive_into_the_candidate_title(): + a, _ = listings.parse_hit("Lizol Floor Cleaner (Citrus) 2 l Price - Buy", + "https://blinkit.com/prn/a/prid/1", "", ("lizol",)) + assert discover.cluster([a])[0]["title"] == "Lizol Floor Cleaner Citrus 2l" diff --git a/tests/test_web_discovery_api.py b/tests/test_web_discovery_api.py new file mode 100644 index 0000000..6f6eed0 --- /dev/null +++ b/tests/test_web_discovery_api.py @@ -0,0 +1,127 @@ +"""HTTP tests for the web & retail listings additions to /api/admin/brand-discovery. + +The batch fixtures mirror test_brand_discovery_api.py (read the docstring +there before removing `batch_root` or `no_background_worker`: without them a +test writes batch manifests into the repository's data directory). +""" +from __future__ import annotations + +import pytest + +from app.api.routers import brand_discovery as router_module +from app.core import batch_ingest +from app.infrastructure import settings +from app.services import brand_discovery as bd +from app.services.web_discovery import jobs, provenance + +PREVIEW = "/api/admin/brand-discovery/preview" +INGEST = "/api/admin/brand-discovery/ingest" +WEB_JOBS = "/api/admin/brand-discovery/web-jobs" + + +@pytest.fixture(autouse=True) +def _isolate(tmp_path, monkeypatch): + from app.services import active_brands, sku_service + from app.core import batch_worker + + monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences") + monkeypatch.setattr(batch_ingest, "BATCH_UPLOAD_DIR", tmp_path / "batch_uploads") + monkeypatch.setattr(batch_worker, "submit", lambda batch_id: None) + monkeypatch.setattr(active_brands, "filtering_enabled", lambda: False) + monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: True) + monkeypatch.setattr(settings, "WEB_DISCOVERY_DIR", tmp_path / "web_discovery") + monkeypatch.setattr(router_module, "WEB_DISCOVERY_ENABLED", True) + from app.api.batch_job_store import batch_job_store + batch_job_store._batches.clear() + batch_job_store._cancelled.clear() + jobs._jobs.clear() + yield + batch_job_store._batches.clear() + batch_job_store._cancelled.clear() + jobs._jobs.clear() + + +@pytest.fixture +def no_worker(monkeypatch): + """start_job must not spawn a real search thread in a test.""" + queued = [] + monkeypatch.setattr(jobs, "_ensure_worker", lambda: None) + monkeypatch.setattr(jobs._queue, "put", queued.append) + return queued + + +def test_the_web_job_routes_are_admin_only(client, user_headers): + assert client.post(WEB_JOBS, json={"brand": "Dettol"}).status_code in (401, 403) + assert client.post(WEB_JOBS, json={"brand": "Dettol"}, headers=user_headers).status_code == 403 + assert client.get(f"{WEB_JOBS}/{'a' * 32}").status_code in (401, 403) + + +def test_starting_a_job_queues_it_and_polling_returns_progress(client, admin_headers, no_worker): + started = client.post(WEB_JOBS, json={"brand": "Reckitt Benckiser"}, headers=admin_headers) + + assert started.status_code == 202, started.text + body = started.json() + assert body["status"] == jobs.QUEUED and body["brand"] == "Reckitt Benckiser" + assert "candidates" not in body and body["candidate_count"] == 0 + assert no_worker == [body["job_id"]] + polled = client.get(f"{WEB_JOBS}/{body['job_id']}", headers=admin_headers) + assert polled.status_code == 200 and polled.json()["job_id"] == body["job_id"] + + +def test_a_second_start_while_running_returns_the_same_job(client, admin_headers, no_worker): + first = client.post(WEB_JOBS, json={"brand": "Dettol"}, headers=admin_headers).json() + second = client.post(WEB_JOBS, json={"brand": "dettol"}, headers=admin_headers).json() + assert first["job_id"] == second["job_id"] and len(no_worker) == 1 + + +def test_an_unknown_job_is_404(client, admin_headers): + assert client.get(f"{WEB_JOBS}/{'f' * 32}", headers=admin_headers).status_code == 404 + + +def test_switched_off_the_routes_say_so(client, admin_headers, monkeypatch): + monkeypatch.setattr(router_module, "WEB_DISCOVERY_ENABLED", False) + assert client.post(WEB_JOBS, json={"brand": "Dettol"}, headers=admin_headers).status_code == 404 + + +def test_preview_passes_the_web_flag_through_and_defaults_it_off(client, admin_headers, monkeypatch): + seen = [] + + def fake_discover(brand, **kwargs): + seen.append(kwargs) + return bd.DiscoveryResult(brand=brand, parent_brand=brand.lower(), table="brand_x", + brand_active=True, filtering_enabled=False) + + monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products", fake_discover) + client.post(PREVIEW, json={"brand": "Dettol"}, headers=admin_headers) + client.post(PREVIEW, json={"brand": "Dettol", "use_web": True, "web_job_id": "c" * 32}, + headers=admin_headers) + + assert (seen[0]["use_web"], seen[0]["web_job_id"]) == (False, None) + assert (seen[1]["use_web"], seen[1]["web_job_id"]) == (True, "c" * 32) + + +def test_ingest_hands_web_rows_to_the_provenance_watcher_by_csv_position(client, admin_headers, monkeypatch): + watched = [] + monkeypatch.setattr(provenance, "watch", lambda batch_id, brand, entries: watched.append( + (batch_id, brand, entries))) + listing = {"retailer": "Blinkit", "url": "https://blinkit.com/prn/x/prid/1", "title": "Dettol Soap 125g"} + + resp = client.post(INGEST, json={"brand": "Reckitt Benckiser", "products": [ + {"product_name": "Dettol Original Soap", "size_variants": ["125g"]}, + {"product_name": "Harpic Power Plus", "size_variants": ["500ml"], "listings": [listing], + "price_range": "₹99", "retailer_count": 1}, + ]}, headers=admin_headers) + + assert resp.status_code == 202, resp.text + [(batch_id, brand, entries)] = watched + assert batch_id == resp.json()["batch_id"] and brand == "Reckitt Benckiser" + assert list(entries) == [1] and entries[1]["listings"] == [listing] + assert entries[1]["price_range"] == "₹99" + + +def test_ingest_without_web_rows_starts_no_watcher(client, admin_headers, monkeypatch): + watched = [] + monkeypatch.setattr(provenance, "watch", lambda *a: watched.append(a)) + resp = client.post(INGEST, json={"brand": "Britannia", "products": [ + {"product_name": "Marie Gold", "size_variants": ["250g"]}]}, headers=admin_headers) + assert resp.status_code == 202 and watched == []