From 34c65a7cbdface1287f0700f617ef00ba09da599 Mon Sep 17 00:00:00 2001 From: sriram Date: Tue, 6 Oct 2026 11:57:01 +0530 Subject: [PATCH] Product discovery on names --- app/api/routers/brand_discovery.py | 8 +- app/core/store_catalog_pipeline.py | 12 ++ app/services/brand_discovery.py | 51 +++++++ app/services/brand_registry.py | 68 +++++++++ app/services/brand_sync.py | 1 + app/services/vector_store.py | 26 +++- app/services/web_discovery/brands.py | 30 ++-- app/services/web_discovery/discover.py | 27 +++- app/services/web_discovery/listings.py | 8 + app/services/web_discovery/provenance.py | 12 +- app/services/web_discovery/variants.py | 162 ++++++++++++++++++++ tests/test_web_discovery.py | 6 +- tests/test_web_variants.py | 182 +++++++++++++++++++++++ 13 files changed, 569 insertions(+), 24 deletions(-) create mode 100644 app/services/web_discovery/variants.py create mode 100644 tests/test_web_variants.py diff --git a/app/api/routers/brand_discovery.py b/app/api/routers/brand_discovery.py index 66d6e70..7a3c8da 100644 --- a/app/api/routers/brand_discovery.py +++ b/app/api/routers/brand_discovery.py @@ -105,6 +105,11 @@ class DiscoveredProductIn(BaseModel): listings: List[Dict[str, Any]] = Field(default_factory=list) price_range: Optional[str] = None retailer_count: int = 0 + # The line and variant the preview grouped this row under. Web rows only + # need them sent back: their variant came from the retailer's brackets, + # which the plain product name no longer carries. + product_line: Optional[str] = None + variant: Optional[str] = None class WebJobRequest(BaseModel): @@ -245,7 +250,8 @@ async def ingest_brand_discovery(payload: DiscoveryIngestRequest) -> batch_commo # them. Keyed by CSV position, which is how stage 11 reports source rows. web_entries = { index: {"listings": item.listings, "price_range": item.price_range or "", - "retailer_count": item.retailer_count} + "retailer_count": item.retailer_count, + "product_line": item.product_line or "", "variant": item.variant or ""} for index, item in enumerate(payload.products) if item.listings } if web_entries: diff --git a/app/core/store_catalog_pipeline.py b/app/core/store_catalog_pipeline.py index cd5f02f..e91a6bd 100644 --- a/app/core/store_catalog_pipeline.py +++ b/app/core/store_catalog_pipeline.py @@ -1003,9 +1003,21 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]: "confidence_score": row.get("confidence_score"), "validation_issues": list(row.get("validation_issues") or []), "field_sources": dict(row.get("field_sources") or {}), + # Line and variant read off the size-free name. Web discovery's + # provenance pass may set a better variant later; the upsert keeps + # whichever arrived first, so this never overwrites it. + **_line_and_variant(row, name), } +def _line_and_variant(row: Dict[str, Any], name: str) -> Dict[str, Optional[str]]: + from app.services.web_discovery.variants import split_variant + + line, variant = split_variant(name or "") + return {"product_line": row.get("product_line") or line or None, + "variant": row.get("variant") or variant or None} + + def _record_product(result: PipelineResult, brand: str, row: Dict[str, Any], disposition: str, source_rows: Optional[Dict[str, int]] = None) -> None: diff --git a/app/services/brand_discovery.py b/app/services/brand_discovery.py index 49c1269..4819087 100644 --- a/app/services/brand_discovery.py +++ b/app/services/brand_discovery.py @@ -233,6 +233,10 @@ class DiscoveredProduct: listings: List[Dict[str, Any]] = field(default_factory=list) price_range: str = "" retailer_count: int = 0 + # "Lizol Disinfectant Surface & Floor Cleaner" / "Lavender". Shown grouped + # in the preview and sent back on ingest; stored by the pipeline/provenance. + product_line: str = "" + variant: str = "" def as_preview(self) -> Dict[str, Any]: body = self._preview_body() @@ -263,6 +267,8 @@ class DiscoveredProduct: "confidence": round(self.confidence, 2), "matches_existing": self.matches_existing, "notes": list(self.notes), + "product_line": self.product_line, + "variant": self.variant, "selected": self.confidence >= 0.5, } @@ -286,6 +292,7 @@ class DiscoveryResult: "brand_active": self.brand_active, "filtering_enabled": self.filtering_enabled, "products": [p.as_preview() for p in self.products], + "families": families_of(self.products), "counts": dict(self.counts), "warnings": list(self.warnings), } @@ -370,6 +377,25 @@ def _normalise_title(brand: str, title: Optional[str]) -> str: return off_bulk.normalize_for_match(title, off_bulk.brand_tokens(brand)) +def families_of(products: Sequence[DiscoveredProduct]) -> List[Dict[str, Any]]: + """The preview grouped as product line -> variant -> pack sizes. + + Each size carries `rows`, the indexes into `products` it came from, so the + panel can tick and untick the underlying rows from the grouped view.""" + from app.services.web_discovery.variants import group_families + + candidates = [] + for index, p in enumerate(products): + for size in p.size_variants or [""]: + candidates.append({ + "product_line": p.product_line, "variant": p.variant, "size": size, + "retailer_count": p.retailer_count, "price_range": p.price_range, + "providers": list(p.providers[:p.retailer_count] if p.retailer_count else []), + "row_index": index, + }) + return group_families(candidates) + + def _same_product(left: str, right: str, left_raw: str, right_raw: str) -> bool: """Whether two normalised titles name the same retail product. @@ -392,9 +418,19 @@ def _same_product(left: str, right: str, left_raw: str, right_raw: str) -> bool: return True if off_bulk.has_extra_variant_conflict(left_raw, right_raw): return False + # Citrus vs Floral scores 0.857 - over the floor. Two names that differ in + # a scent or flavour are two packs on the shelf, never one product. + if _scents(left_raw) != _scents(right_raw): + return False return off_bulk.symmetric_similarity(left, right) >= _TITLE_MATCH_THRESHOLD +def _scents(title: str) -> frozenset: + from app.services.web_discovery.variants import variant_words + + return variant_words(title) + + def _match_existing(normalised: str, title: str, catalog_keys: Dict[str, str]) -> Optional[str]: """The stored base name for this product, or None. @@ -1196,6 +1232,7 @@ def discover_brand_products( continue if "web" in product.sources: _apply_web(product, candidate) + _apply_line(product, candidate) if (require_evidence and "off" not in product.sources and "store" not in product.sources and not product.evidence): dropped += 1 @@ -1259,6 +1296,20 @@ def _from_web(brand: str, job_id: Optional[str], warnings: List[str]) -> List[Di return [dict(c) for c in job.candidates] +def _apply_line(product: DiscoveredProduct, candidate: Dict[str, Any]) -> None: + """Product line and variant. A web candidate brings its own, read from the + retailer's brackets - kept while its variant words still appear in the + final name (a stored-name rename could have changed it). Everything else is + split from the final product name.""" + from app.services.web_discovery.variants import split_variant + + line, variant = candidate.get("product_line") or "", candidate.get("variant") or "" + name_words = set(re.findall(r"[a-z0-9]+", product.product_name.lower())) + if not (line and set(re.findall(r"[a-z0-9]+", variant.lower())) <= name_words): + line, variant = split_variant(product.product_name) + product.product_line, product.variant = line, variant + + def _apply_web(product: DiscoveredProduct, candidate: Dict[str, Any]) -> None: product.listings = list(candidate.get("listings") or []) product.price_range = candidate.get("price_range") or "" diff --git a/app/services/brand_registry.py b/app/services/brand_registry.py index 03da3a7..87888f1 100644 --- a/app/services/brand_registry.py +++ b/app/services/brand_registry.py @@ -367,9 +367,77 @@ def resolve_parent_brand(brand: str) -> str: prefix = _known_leading_brand(key) if prefix: return resolve_parent_brand(prefix) + line = _known_leading_line(key) + if line: + return line + parent = _unique_parent_prefix(key) + if parent: + return parent return brand +def is_abbreviation(token: str, parent: str) -> bool: + """"rb" for Reckitt Benckiser, "hul" for Hindustan Unilever, "jnj" for + Johnson & Johnson. Short, not itself a word of the parent, and starting + with the parent's first letter - which is what keeps "kit kat" (Nestle) + from being read as the abbreviation "kit" plus a sub-brand "kat".""" + parent_words = re.findall(r"[a-z0-9]+", (parent or "").lower()) + return (bool(parent_words) and 1 < len(token) <= 4 and token not in parent_words + and token[0] == parent_words[0][0]) + + +# Line names that are also everyday words or another maker's name. Read as a +# leading word they would misfile: "Laxmi Chilli Powder" is a spice brand, not +# HUL; "Boost Energy Drink", "Finish Line", "Baby Oil" (any maker) likewise. +_AMBIGUOUS_LINES = frozenset({"always", "boost", "finish", "wheel", "fairy", "ivory", + "vanish", "laxmi", "whisper"}) + + +@lru_cache(maxsize=1) +def _abbreviated_lines() -> dict: + """"lizol" -> "reckitt benckiser", read off the alias "rb lizol". + + Only line names of 5+ characters that exactly one parent registers: the + bare "sun" off "hul sun" would otherwise send "Sun Pharma" to Unilever.""" + owners: dict = {} + for alias, parent in BRAND_ALIASES.items(): + words = alias.split() + if len(words) > 1 and is_abbreviation(words[0], parent): + rest = " ".join(words[1:]) + if (len(rest.replace(" ", "")) >= 5 and rest not in _AMBIGUOUS_LINES + and not rest.startswith("baby ")): + owners.setdefault(rest, set()).add(parent) + return {rest: next(iter(ps)) for rest, ps in owners.items() if len(ps) == 1} + + +def _known_leading_line(key: str) -> Optional[str]: + """"Lizol Floor Cleaner" -> "reckitt benckiser". + + The registry spells Reckitt's lines with an abbreviation ("rb lizol"), so + the bare line name is neither a key nor a parent and `_known_leading_brand` + could not see it: a product typed with a line after it built a + brand_lizol_floor_cleaner table. Reached only after every older rule failed.""" + lines = _abbreviated_lines() + words = key.split() + for n in range(len(words), 0, -1): + parent = lines.get(" ".join(words[:n])) + if parent: + return parent + return None + + +def _unique_parent_prefix(key: str) -> Optional[str]: + """"Reckitt" -> "reckitt benckiser": the typed words are the opening words + of exactly one parent. Two parents sharing the opening ("tata ...") give + nothing, and a single word must be 5+ characters.""" + words = key.split() + if not words or (len(words) == 1 and len(words[0]) < 5): + return None + hits = {p for p in set(BRAND_ALIASES.values()) + if p.split()[:len(words)] == words and len(p.split()) > len(words)} + return next(iter(hits)) if len(hits) == 1 else None + + def _known_leading_brand(key: str) -> Optional[str]: """The longest leading run of words in `key` that is itself a known brand. diff --git a/app/services/brand_sync.py b/app/services/brand_sync.py index 16154c8..8e6dc78 100644 --- a/app/services/brand_sync.py +++ b/app/services/brand_sync.py @@ -104,6 +104,7 @@ EXPORT_COLUMNS = ( "barcode_lookup_status", "barcode_last_updated", "gst_percent", "tax_amount", "hsn_gst_needs_review", "highlights", "nutrients", "search_query", "field_sources", + "product_line", "variant", # The validation verdict. Listed here for the same reason the barcode keys # are: an export that strips them turns a re-seed into a silent downgrade. # The upsert assigns these three plainly rather than COALESCEing them, so a diff --git a/app/services/vector_store.py b/app/services/vector_store.py index 1a73172..28e0de0 100644 --- a/app/services/vector_store.py +++ b/app/services/vector_store.py @@ -179,6 +179,14 @@ def get_brand_table_ddl(brand: str) -> str: nutrients_per_100g JSONB, search_query TEXT, + -- "Lizol Disinfectant Surface & Floor Cleaner" / "Lavender": which + -- product line a row belongs to and which scent or flavour it is, so + -- every pack of every variant of one product can be listed together. + -- Fill-only-blanks in the upsert: a later re-derivation from the bare + -- name can never overwrite the bracket-aware value web discovery set. + product_line TEXT, + variant TEXT, + -- The deterministic validation verdict from product_validator. -- validate_catalog() has always computed these three and annotated -- them onto every row; until they were added here nothing projected @@ -382,6 +390,8 @@ def _ensure_columns(cur, table_name: str) -> None: "nutrients": "TEXT[]", "nutrients_per_100g": "JSONB", "search_query": "TEXT", + "product_line": "TEXT", + "variant": "TEXT", # The product_validator verdict - see get_brand_table_ddl for why a # row's confidence has to survive to disk to be worth computing. "validation_status": "TEXT", @@ -452,7 +462,7 @@ def _ensure_columns(cur, table_name: str) -> None: # `nutrients_per_100g` joins them: it is mirrored from nutrition_facts by # the same sync, not written by the INSERT, and is likewise a current # column rather than legacy debris. - inserted_cols = {"id", "product_name", "title", "description", "category", "image_id", "image_url", "image_urls", "price_range", "size_variants", "providers", "fssai_license", "product_sku", "sku_source", "hsn_code", "final_selling_price", "selling_price", "barcode", "barcode_type", "gtin", "ean13", "upc", "barcode_source", "barcode_verified", "barcode_lookup_status", "barcode_last_updated", "gst_percent", "tax_amount", "hsn_gst_needs_review", "highlights", "nutrients", "nutrients_per_100g", "field_sources", "search_query", "nutrition_score", "health_score", "embedding", "img_vector", "img_vector_src", "created_at", "updated_at"} + inserted_cols = {"id", "product_name", "title", "description", "category", "image_id", "image_url", "image_urls", "price_range", "size_variants", "providers", "fssai_license", "product_sku", "sku_source", "hsn_code", "final_selling_price", "selling_price", "barcode", "barcode_type", "gtin", "ean13", "upc", "barcode_source", "barcode_verified", "barcode_lookup_status", "barcode_last_updated", "gst_percent", "tax_amount", "hsn_gst_needs_review", "highlights", "nutrients", "nutrients_per_100g", "field_sources", "search_query", "product_line", "variant", "nutrition_score", "health_score", "embedding", "img_vector", "img_vector_src", "created_at", "updated_at"} for col, is_nullable, col_def in col_info: if col not in inserted_cols and is_nullable == 'NO' and col_def is None: cur.execute(f"ALTER TABLE {table_name} ALTER COLUMN {col} DROP NOT NULL") @@ -657,6 +667,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b fssai_license = str(p.get("fssai_license", "")) if p.get("fssai_license") else "" search_query = p.get("search_query") or "" + product_line = str(p.get("product_line") or "").strip() or None + variant = str(p.get("variant") or "").strip() or None # Convert embedding to PostgreSQL vector format embedding = p.get("embedding") @@ -703,6 +715,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b confidence_score, validation_issues, field_sources, + product_line, + variant, embedding_str )) @@ -753,8 +767,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b gst_percent, tax_amount, hsn_gst_needs_review, highlights, nutrients, search_query, validation_status, confidence_score, validation_issues, - field_sources, embedding) - VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s) + field_sources, product_line, variant, embedding) + VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s) ON CONFLICT (image_id) DO UPDATE SET product_name = EXCLUDED.product_name, title = EXCLUDED.title, @@ -848,6 +862,12 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b -- depth here - each key's value is one flat record. field_sources = COALESCE({table_name}.field_sources, '{{}}'::jsonb) || COALESCE(EXCLUDED.field_sources, '{{}}'::jsonb), + -- STORED FIRST: fill-only-blanks. The pipeline derives these + -- from the bare name on every run; the web provenance pass + -- writes the retailer's bracketed variant ("(Original)"), + -- which the bare name has lost. The better value must win. + product_line = COALESCE({table_name}.product_line, EXCLUDED.product_line), + variant = COALESCE({table_name}.variant, EXCLUDED.variant), updated_at = CURRENT_TIMESTAMP """, rows, diff --git a/app/services/web_discovery/brands.py b/app/services/web_discovery/brands.py index fd34e56..70750f6 100644 --- a/app/services/web_discovery/brands.py +++ b/app/services/web_discovery/brands.py @@ -26,7 +26,7 @@ from collections import defaultdict from dataclasses import dataclass from typing import Dict, List, Set, Tuple -from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand +from app.services.brand_registry import BRAND_ALIASES, is_abbreviation, resolve_parent_brand @dataclass(frozen=True) @@ -34,20 +34,16 @@ class Target: """One name to search for, and the words a listing must lead with to count.""" query: str # what goes before `site:` - "Dettol", "Amul Butter" brand_terms: Tuple[str, ...] # "dettol"; or ("nestle", "cerelac") + # "Lizol Floor Cleaner" typed: the listing must lead with "lizol" AND carry + # every one of these words somewhere, so other Lizol lines drop out. + line_terms: Tuple[str, ...] = () def _words(text: str) -> List[str]: return re.findall(r"[a-z0-9]+", (text or "").lower()) -def _is_abbreviation(token: str, parent: str) -> bool: - """"rb" for Reckitt Benckiser, "hul" for Hindustan Unilever, "jnj" for - Johnson & Johnson. Short, not itself a word of the parent, and starting - with the parent's first letter - which is what keeps "kit kat" (Nestle) - from being read as the abbreviation "kit" plus a sub-brand "kat".""" - parent_words = _words(parent) - return (bool(parent_words) and 1 < len(token) <= 4 and token not in parent_words - and token[0] == parent_words[0][0]) +_is_abbreviation = is_abbreviation def _sub_brand_part(alias: str, parent: str) -> str: @@ -96,6 +92,20 @@ def _family(parent: str) -> List[Target]: return out +def _line_target(name: str, typed_key: str, family: List[Target]) -> List[Target]: + """"Lizol Floor Cleaner" -> search that phrase, but judge listings by the + family member's own brand words ("lizol") plus the line words. Taking the + whole phrase as the brand would refuse every real title, which reads + "Lizol Disinfectant Surface & Floor Cleaner".""" + typed = typed_key.split() + for target in sorted(family, key=lambda t: -len(_words(t.query))): + lead = _words(target.query) + if len(typed) > len(lead) and typed[:len(lead)] == lead: + return [Target(query=name, brand_terms=target.brand_terms, + line_terms=tuple(typed[len(lead):]))] + return [] + + def targets_for(brand: str) -> List[Target]: """Sub-brands first, the brand itself last. Deterministic order. @@ -112,7 +122,7 @@ def targets_for(brand: str) -> List[Target]: own = [t for t in family if " ".join(_words(t.query)) == typed_key or typed_key in {" ".join(_words(term)) for term in t.brand_terms[1:] or t.brand_terms}] - return own or [Target(query=name, brand_terms=(name.lower(),))] + return own or _line_target(name, typed_key, family) or [Target(query=name, brand_terms=(name.lower(),))] display = parent.title() if parent.islower() else parent return family + [Target(query=display, brand_terms=(parent_key,))] diff --git a/app/services/web_discovery/discover.py b/app/services/web_discovery/discover.py index f75619e..4efa272 100644 --- a/app/services/web_discovery/discover.py +++ b/app/services/web_discovery/discover.py @@ -25,7 +25,7 @@ from dataclasses import dataclass, field from typing import Callable, Dict, List, Optional, Sequence from app.infrastructure import settings -from app.services.web_discovery import cache, listings, search +from app.services.web_discovery import cache, listings, search, variants from app.services.web_discovery.brands import Target, targets_for logger = logging.getLogger(__name__) @@ -89,16 +89,28 @@ def _variant_words(listing: listings.Listing) -> set: def _same_product(a: listings.Listing, b: listings.Listing) -> bool: - """Same brand word, same pack, near-identical name, and no bracketed - variant word on one side only. Measured: "(Citrus) 2l" and "(Floral) 2l" - share 4 of 6 words - enough for the overlap test alone, and two products.""" + """Same brand word, same pack, near-identical name, and no variant word on + one side only - bracketed, or a bare scent/flavour from + variants.VARIANT_TERMS. Measured: "(Citrus) 2l" and "(Floral) 2l" share 4 + of 6 words, and "Cleaner Floral 500 ml" / "Cleaner Pine 500 ml" share 3 of + 5 (exactly the 0.6 bar) - enough for the overlap test alone, and two + products each time.""" if a.brand_term != b.brand_term or a.size != b.size: return False ident_a, ident_b = _identity(a), _identity(b) if _jaccard(ident_a, ident_b) < _SAME_NAME_JACCARD: return False differing = set(ident_a) ^ set(ident_b) - return not (differing & (_variant_words(a) | _variant_words(b))) + return not (differing & (_variant_words(a) | _variant_words(b) | variants.VARIANT_TERMS)) + + +# "Lizol Floor Cleaner" typed: a Lizol listing of another line (toilet cleaner, +# wipes) is a real product, just not the one asked for. +OTHER_LINE = "other_product_line" + + +def _on_line(listing: listings.Listing, target: Target) -> bool: + return set(target.line_terms) <= set(listing.tokens) def _jaccard(a: Sequence[str], b: Sequence[str]) -> float: @@ -145,6 +157,7 @@ def cluster(found: Sequence[listings.Listing]) -> List[Dict]: # normaliser (and off_bulk.strip_sizes) drops bracketed text, which # turned "Lizol Floor Cleaner (Citrus)" and "(Floral)" into one name. plain = re.sub(r"\s+", " ", re.sub(r"[()]", " ", name)).strip() + line, variant = variants.split_variant(name) candidates.append({ "title": f"{plain} {size}", "size": size, @@ -154,6 +167,8 @@ def cluster(found: Sequence[listings.Listing]) -> List[Dict]: "price_range": _price_range(prices), "listings": [m.as_dict() for m in members[:_MAX_LISTINGS_PER_CANDIDATE]], "brand_term": brand_term, + "product_line": line, + "variant": variant, "source": "web", }) candidates.sort(key=lambda c: (-c["retailer_count"], c["title"].lower())) @@ -205,6 +220,8 @@ def run(brand: str, *, progress: Optional[Callable[[RunResult], None]] = None, query.target.brand_terms) if listing is None: rejected[reason] += 1 + elif not _on_line(listing, query.target): + rejected[OTHER_LINE] += 1 else: kept.append(listing) result.queries_done += 1 diff --git a/app/services/web_discovery/listings.py b/app/services/web_discovery/listings.py index 1acdaab..90702c1 100644 --- a/app/services/web_discovery/listings.py +++ b/app/services/web_discovery/listings.py @@ -277,6 +277,14 @@ def parse_hit(title: str, url: str, snippet: str, if size is None: # Some retailers put the size only in the title's tail ("... - 125 g | Zepto"). size, distinct = extract_size(title) + if size is None and not _MULTIPACK.search(snippet or ""): + # Blinkit often prints the pack only in the snippet ("... 500 ml ... + # Price"): 63 of 80 real Reckitt listings were lost for want of it. + # Taken only when the snippet names exactly ONE size - a snippet that + # lists the other packs ("also in 1 L, 2 L") says nothing about this one. + size, distinct = extract_size(snippet) + if distinct > 1: + return None, NO_PACK_SIZE if size is None: return None, NO_PACK_SIZE if distinct > 1: diff --git a/app/services/web_discovery/provenance.py b/app/services/web_discovery/provenance.py index 8094bc2..4893732 100644 --- a/app/services/web_discovery/provenance.py +++ b/app/services/web_discovery/provenance.py @@ -10,7 +10,10 @@ web-found product it * fills `price_range` from the listing prices when the pipeline left it blank; * sets `validation_status = 'needs_review'` when only ONE retailer listed it (never lifting a `rejected` verdict). Two or more retailers keep whatever the - validator decided. + validator decided; +* sets `product_line` / `variant` to what the listing titles said. These + OVERWRITE the pipeline's value: the pipeline read them off the bare name, + which has lost the retailer's brackets ("Harpic ... (Original)"). The join is exact: stage 11 reports every stored row's `image_id` with the 1-based sheet row it came from (header = row 1), and the ingest wrote the web @@ -77,10 +80,13 @@ def apply(brand: str, updates: List[Dict[str, Any]]) -> int: f"price_range = CASE WHEN COALESCE(price_range, '') = '' AND %s <> '' " f"THEN %s ELSE price_range END, " f"validation_status = CASE WHEN %s AND COALESCE(validation_status, '') <> 'rejected' " - f"THEN 'needs_review' ELSE validation_status END " + f"THEN 'needs_review' ELSE validation_status END, " + f"product_line = COALESCE(NULLIF(%s, ''), product_line), " + f"variant = COALESCE(NULLIF(%s, ''), variant) " f"WHERE image_id = %s", (Json(provenance), entry.get("price_range") or "", entry.get("price_range") or "", - single, update["image_id"]), + single, entry.get("product_line") or "", entry.get("variant") or "", + update["image_id"]), ) written += cur.rowcount except Exception as exc: # noqa: BLE001 - provenance must not take the batch down diff --git a/app/services/web_discovery/variants.py b/app/services/web_discovery/variants.py new file mode 100644 index 0000000..5140d01 --- /dev/null +++ b/app/services/web_discovery/variants.py @@ -0,0 +1,162 @@ +"""A product name -> (product line, variant), and candidates -> line -> variant -> sizes. + +"Lizol Disinfectant Surface & Floor Cleaner (Lavender)" is the line +"Lizol Disinfectant Surface & Floor Cleaner" in the variant "Lavender". Retail +titles mark the variant two ways: + + * in brackets - Blinkit's "(Citrus, 625 ml)", "(Original)". Whatever a + bracket holds once the size is gone IS the variant, whatever the word. + * as a bare word - "Lizol ... Cleaner Floral 500 ml". Only a word in + VARIANT_TERMS is read as a variant here; any other bare word stays in the + line, because guessing would split one product into two lines. + +VARIANT_TERMS is scents and flavours only. Generic words ("original", "fresh", +"chocolate", "classic") are left out on purpose: "Dettol Original Soap" and +"Cadbury Dairy Milk Chocolate" name a line, not a variant. A bracket still +catches "(Original)". Add a word only when a real listing needs it. +""" +from __future__ import annotations + +import re +from collections import Counter, defaultdict +from typing import Dict, Iterable, List, Tuple + +VARIANT_TERMS = frozenset({ + # scents - household and personal care + "citrus", "floral", "pine", "lavender", "jasmine", "rose", "lemon", "lime", + "orange", "sandal", "sandalwood", "neem", "mint", "eucalyptus", "aloe", + "lily", "marine", "ocean", "mogra", "tulsi", "vanilla", "berry", "musk", + # flavours that name a pack rather than a product + "strawberry", "mango", "elaichi", "masala", "pineapple", "litchi", "guava", +}) + +# A word that belongs to the variant when it sits next to one: "Lemon Fresh", +# "Ocean Breeze", "Lavender Blossom". Never a variant on its own. +_VARIANT_MODIFIERS = frozenset({"fresh", "blossom", "breeze", "burst", "bloom", "splash", "twist"}) + +_PACKAGING_WORDS = frozenset({"bar", "pouch", "bottle", "jar", "box", "tin", "carton", "pack", "refill"}) +_BRACKET = re.compile(r"\(([^()]*)\)") +_WORD = re.compile(r"[A-Za-z0-9]+(?:'[A-Za-z]+)?") + + +def _words(text: str) -> List[str]: + return re.findall(r"[a-z0-9]+", (text or "").lower()) + + +def _tidy(text: str) -> str: + return re.sub(r"\s+", " ", text or "").strip(" ,;:.-/&+–—") + + +def _title(word: str) -> str: + return word if any(c.isupper() for c in word) else word.capitalize() + + +def variant_words(text: str) -> frozenset: + """The VARIANT_TERMS words in `text` - for telling two names apart.""" + return frozenset(w for w in _words(text) if w in VARIANT_TERMS) + + +def split_variant(name: str) -> Tuple[str, str]: + """(line, variant). A pack size in the name is ignored ("... Floral 500 ml"). + + ("Lizol Floor Cleaner (Citrus)") -> ("Lizol Floor Cleaner", "Citrus") + ("Lizol Floor Cleaner Floral") -> ("Lizol Floor Cleaner", "Floral") + ("Harpic Toilet Cleaner (Original)") -> ("Harpic Toilet Cleaner", "Original") + ("Dettol Original Soap") -> ("Dettol Original Soap", "") + """ + from app.services.web_discovery.listings import _strip_sizes + + name = _tidy(re.sub(r"\s+", " ", _strip_sizes(name or ""))) + groups = [_tidy(g) for g in _BRACKET.findall(name)] + groups = [g for g in groups if re.search(r"[A-Za-z]", g)] + if groups: + line = _tidy(re.sub(r"\s+", " ", _BRACKET.sub(" ", name))) + return line, " ".join(_title(w) for g in groups for w in _WORD.findall(g)) + + tokens = _WORD.findall(name) + lowered = [t.lower() for t in tokens] + keep = [l in VARIANT_TERMS for l in lowered] + if not any(keep) or all(keep): + return name, "" + # A modifier joins the variant only when it touches a variant word. + for i, l in enumerate(lowered): + if l in _VARIANT_MODIFIERS and ((i > 0 and keep[i - 1]) or (i + 1 < len(keep) and keep[i + 1])): + keep[i] = True + variant = " ".join(_title(t) for t, k in zip(tokens, keep) if k) + line = name + for t, k in zip(tokens, keep): + if k: + line = re.sub(r"(? Tuple[str, ...]: + """Lines are one line when these words match - order, '&' and packaging aside.""" + brand = set(_words(brand_term)) + return tuple(sorted({w for w in _words(line) if w not in _PACKAGING_WORDS} - brand)) or tuple(_words(line)) + + +_SIZE = re.compile(r"^(\d+(?:\.\d+)?)(kg|g|ml|l|pcs)$") +_SIZE_SCALE = {"g": (0, 1), "kg": (0, 1000), "ml": (1, 1), "l": (1, 1000), "pcs": (2, 1)} + + +def size_sort_key(size: str) -> Tuple[int, float, str]: + """Mass, then volume, then counts, each smallest first: 500ml < 625ml < 1l.""" + m = _SIZE.match((size or "").lower()) + if not m: + return (9, 0.0, size or "") + group, scale = _SIZE_SCALE[m.group(2)] + return (group, float(m.group(1)) * scale, size) + + +def group_families(candidates: Iterable[Dict]) -> List[Dict]: + """Candidates (one per name and pack) -> product line -> variant -> sizes. + + Each candidate needs `product_line`, `variant` and `size`; `retailer_count`, + `price_range`, `providers` and `listings` are carried through when present, + and a `row_index` is collected into the size's `rows`. + Lines come out best-corroborated first; variants and sizes in a stable order. + """ + lines: Dict[Tuple, List[Dict]] = defaultdict(list) + for c in candidates: + line = c.get("product_line") or "" + if not line: + continue + lines[(c.get("brand_term") or "", line_key(line, c.get("brand_term") or ""))].append(c) + + families: List[Dict] = [] + for members in lines.values(): + spellings = Counter(m["product_line"] for m in members) + display = sorted(spellings, key=lambda s: (-spellings[s], len(s), s))[0] + by_variant: Dict[str, List[Dict]] = defaultdict(list) + for m in members: + by_variant[m.get("variant") or ""].append(m) + variants = [] + for variant, rows in sorted(by_variant.items(), key=lambda kv: (kv[0] == "", kv[0].lower())): + sizes: Dict[str, Dict] = {} + for r in rows: + size = r.get("size") or ((r.get("sizes") or [None])[0]) + if not size: + continue + entry = sizes.setdefault(size, {"size": size, "retailer_count": 0, "price_range": "", + "providers": [], "listings": [], "rows": []}) + if r.get("row_index") is not None: + entry["rows"].append(r["row_index"]) + entry["retailer_count"] = max(entry["retailer_count"], int(r.get("retailer_count") or 0)) + entry["price_range"] = entry["price_range"] or r.get("price_range") or "" + for p in r.get("providers") or []: + if p not in entry["providers"]: + entry["providers"].append(p) + entry["listings"].extend(r.get("listings") or []) + variants.append({"variant": variant, + "sizes": sorted(sizes.values(), key=lambda s: size_sort_key(s["size"]))}) + families.append({ + "product_line": display, + "brand_term": members[0].get("brand_term") or "", + "variants": variants, + "retailer_count": max((int(m.get("retailer_count") or 0) for m in members), default=0), + }) + families.sort(key=lambda f: (-f["retailer_count"], -len(f["variants"]), f["product_line"].lower())) + return families + diff --git a/tests/test_web_discovery.py b/tests/test_web_discovery.py index 0cef51e..3fb8155 100644 --- a/tests/test_web_discovery.py +++ b/tests/test_web_discovery.py @@ -383,7 +383,7 @@ class _Conn(_Cursor): pass -def test_apply_touches_only_provenance_price_and_review_status(monkeypatch): +def test_apply_touches_only_provenance_price_review_status_and_line(monkeypatch): conn = _Conn() monkeypatch.setattr("app.services.vector_store._connect", lambda: conn) provenance.apply("Reckitt Benckiser", [ @@ -393,7 +393,9 @@ def test_apply_touches_only_provenance_price_and_review_status(monkeypatch): sql, params = conn.cur.calls[0] assert sql.startswith("UPDATE brand_reckitt_benckiser SET field_sources") assert "price_range = CASE WHEN COALESCE(price_range, '') = ''" in sql - assert "'rejected'" in sql and params[-2] is True and params[-1] == "one_shop" + assert "'rejected'" in sql and params[3] is True and params[-1] == "one_shop" + # A blank line/variant from the entry keeps what the pipeline stored. + assert "variant = COALESCE(NULLIF(%s, ''), variant)" in sql and params[-2] == "" for column in ("product_name", "barcode", "image_url", "category"): assert f"{column} =" not in sql diff --git a/tests/test_web_variants.py b/tests/test_web_variants.py new file mode 100644 index 0000000..b6ee56a --- /dev/null +++ b/tests/test_web_variants.py @@ -0,0 +1,182 @@ +"""A product name typed into Brand Discovery -> its line, variants and pack sizes. + +Typing "Lizol" (a Reckitt line) must find Lizol, and the result must read as +"Lizol ... Floor Cleaner -> Citrus / Floral / Lavender -> 500 ml, 625 ml, 1 l" +rather than a flat list - without ever folding two scents into one product. +No network: every listing is a recorded Blinkit/Amazon title shape. +""" +from __future__ import annotations + +import pytest + +from app.services import brand_discovery as bd +from app.services import brand_registry +from app.services.web_discovery import brands, discover, jobs, listings, variants + +LIZOL = ("lizol",) +LINE = "Lizol Disinfectant Surface & Floor Cleaner" + + +def _listing(title, n, terms=LIZOL, snippet=""): + listing, reason = listings.parse_hit(title, f"https://blinkit.com/prn/x/prid/{n}", snippet, terms) + assert reason is None, (title, reason) + return listing + + +# --------------------------------------------------------------------------- +# A product name resolves to the maker +# --------------------------------------------------------------------------- +@pytest.mark.parametrize("typed", ["Lizol", "Lizol Floor Cleaner", "lizol disinfectant surface cleaner", + "Harpic Power Plus", "Dettol Liquid", "Reckitt"]) +def test_a_product_name_resolves_to_its_maker(typed): + assert brand_registry.resolve_parent_brand(typed) == "reckitt benckiser" + + +@pytest.mark.parametrize("typed", ["Laxmi Chilli Powder", "Boost Energy Drink", "Baby Oil Himalaya", + "Sun Pharma", "Lizzy Cleaner"]) +def test_an_everyday_word_is_not_read_as_a_line(typed): + assert brand_registry.resolve_parent_brand(typed) == typed + + +def test_every_name_that_resolved_before_still_resolves_the_same(): + """The new rules run only after every old one failed, so an alias key can + never change its answer.""" + for alias, parent in brand_registry.BRAND_ALIASES.items(): + assert brand_registry.resolve_parent_brand(alias) == parent + + +def test_a_product_line_searches_that_phrase_but_judges_by_the_line_word(): + [target] = brands.targets_for("Lizol Floor Cleaner") + assert target.query == "Lizol Floor Cleaner" + assert target.brand_terms == ("lizol",) and target.line_terms == ("floor", "cleaner") + assert brands.targets_for("Lizol") == [brands.Target("Lizol", ("lizol",))] + + +def test_another_line_of_the_same_product_is_counted_and_dropped(): + target = brands.Target("Lizol Floor Cleaner", LIZOL, ("floor", "cleaner")) + floor = _listing(f"{LINE} (Citrus - 500 ml) Price - Buy Online", 1) + toilet = _listing("Lizol Power Toilet Cleaner 500 ml Price - Buy Online", 2) + assert discover._on_line(floor, target) and not discover._on_line(toilet, target) + + +# --------------------------------------------------------------------------- +# Line and variant +# --------------------------------------------------------------------------- +@pytest.mark.parametrize("name,line,variant", [ + (f"{LINE} (Lavender)", LINE, "Lavender"), + ("Lizol Disinfectant Surface Cleaner Floral 500 ml", "Lizol Disinfectant Surface Cleaner", "Floral"), + ("Harpic Disinfectant Liquid Toilet Cleaner (Original)", "Harpic Disinfectant Liquid Toilet Cleaner", + "Original"), + ("Lizol Floor Cleaner Lemon Fresh", "Lizol Floor Cleaner", "Lemon Fresh"), + # Generic words are a line, not a variant - only a bracket makes them one. + ("Dettol Original Soap", "Dettol Original Soap", ""), + ("Cadbury Dairy Milk Chocolate", "Cadbury Dairy Milk Chocolate", ""), +]) +def test_split_variant(name, line, variant): + assert variants.split_variant(name) == (line, variant) + + +def test_sizes_sort_by_quantity_not_text(): + sizes = ["1l", "500ml", "2l", "625ml", "200ml"] + assert sorted(sizes, key=variants.size_sort_key) == ["200ml", "500ml", "625ml", "1l", "2l"] + + +# --------------------------------------------------------------------------- +# Scents never merge +# --------------------------------------------------------------------------- +def test_unbracketed_scents_of_one_pack_stay_two_products(): + """3 of 5 words shared - exactly the 0.6 overlap bar, which used to fold them.""" + floral = _listing("Lizol Disinfectant Surface Cleaner Floral 500 ml : Amazon.in: Health", 1) + pine = _listing("Lizol Disinfectant Surface Cleaner Pine 500 ml : Amazon.in: Health", 2) + assert len(discover.cluster([floral, pine])) == 2 + + +def test_the_same_scent_on_two_retailers_is_still_one_product(): + a = _listing(f"{LINE} (Citrus - 500 ml) Price - Buy Online", 1) + b, _ = listings.parse_hit(f"{LINE} Citrus 500 ml : Amazon.in", "https://www.amazon.in/dp/B000000001", + "", LIZOL) + [one] = discover.cluster([a, b]) + assert one["retailer_count"] == 2 and one["variant"] == "Citrus" + + +def test_a_scent_cannot_be_renamed_to_a_stored_other_scent(): + """Citrus vs Floral scores 0.857 - over the 0.85 floor `_match_existing` uses.""" + stored = {bd._normalise_title("Reckitt Benckiser", "Lizol Floor Cleaner Citrus"): "Lizol Floor Cleaner Citrus"} + key = bd._normalise_title("Reckitt Benckiser", "Lizol Floor Cleaner Floral") + assert bd._match_existing(key, "Lizol Floor Cleaner Floral", stored) is None + + +# --------------------------------------------------------------------------- +# Size from the snippet +# --------------------------------------------------------------------------- +def test_a_size_only_in_the_snippet_is_taken(): + listing = _listing("Lizol Disinfectant Floor Cleaner (Jasmine) Price - Buy Online", 1, + snippet="Lizol Disinfectant Floor Cleaner Jasmine 975 ml. Delivered in minutes.") + assert listing.size == "975ml" + + +def test_a_snippet_naming_several_packs_proves_nothing(): + listing, reason = listings.parse_hit( + "Lizol Disinfectant Floor Cleaner (Jasmine) Price - Buy Online", "https://blinkit.com/prn/x/prid/1", + "Available in 500 ml, 975 ml and 2 l packs.", LIZOL) + assert listing is None and reason == listings.NO_PACK_SIZE + + +def test_a_multipack_snippet_is_not_a_pack_size(): + listing, reason = listings.parse_hit( + "Lizol Disinfectant Floor Cleaner (Jasmine) Price - Buy Online", "https://blinkit.com/prn/x/prid/1", + "Pack of 2 - 500 ml each", LIZOL) + assert listing is None and reason == listings.NO_PACK_SIZE + + +# --------------------------------------------------------------------------- +# End to end: line -> variants -> sizes +# --------------------------------------------------------------------------- +_TITLES = [ + f"{LINE} (Citrus - 500 ml) Price - Buy Online at Best", + f"Buy {LINE} (Citrus, 625 ml) Online", + f"{LINE} (Citrus - 1 l) Price - Buy Online at Best", + f"{LINE} (Floral - 500 ml) Price - Buy Online at Best", + f"{LINE} (Floral - 2 l) Price - Buy Online at Best", + f"{LINE} (Lavender - 500 ml) Price - Buy Online at Best", + "Lizol Power Toilet Cleaner Original 500 ml Price - Buy Online", +] + + +def test_lizol_comes_out_as_line_then_variants_then_sizes(): + found = [_listing(t, i) for i, t in enumerate(_TITLES)] + families = variants.group_families(discover.cluster(found)) + + floor = next(f for f in families if f["product_line"] == LINE) + assert {v["variant"]: [s["size"] for s in v["sizes"]] for v in floor["variants"]} == { + "Citrus": ["500ml", "625ml", "1l"], + "Floral": ["500ml", "2l"], + "Lavender": ["500ml"], + } + assert any(f["product_line"].startswith("Lizol Power Toilet Cleaner") for f in families) + + +@pytest.fixture +def no_other_sources(monkeypatch): + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: []) + monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: []) + monkeypatch.setattr(bd, "_from_brand_store", lambda brand: []) + monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: []) + + +def test_the_preview_carries_families_and_each_row_its_line(no_other_sources): + found = [_listing(t, i) for i, t in enumerate(_TITLES[:5])] + job = jobs.WebJob(job_id="d" * 32, brand="Lizol", status=jobs.DONE, backend="duckduckgo", + queries_total=7, queries_done=7, listings_kept=5, candidates=discover.cluster(found)) + jobs._jobs[job.job_id] = job + + body = bd.discover_brand_products("Lizol", use_llm=False, use_web=True, web_job_id=job.job_id).as_dict() + + assert body["parent_brand"] == "reckitt benckiser" and body["table"] == "brand_reckitt_benckiser" + assert {(p["product_line"], p["variant"]) for p in body["products"]} == {(LINE, "Citrus"), (LINE, "Floral")} + [family] = body["families"] + sizes = {v["variant"]: [s["size"] for s in v["sizes"]] for v in family["variants"]} + assert sizes == {"Citrus": ["500ml", "625ml", "1l"], "Floral": ["500ml", "2l"]} + for v in family["variants"]: + for s in v["sizes"]: + assert all(body["products"][i]["variant"] == v["variant"] for i in s["rows"])