Product discovery on names

This commit is contained in:
sriram
2026-10-06 11:57:01 +05:30
parent c0601b65fe
commit 34c65a7cbd
13 changed files with 569 additions and 24 deletions

View File

@@ -105,6 +105,11 @@ class DiscoveredProductIn(BaseModel):
listings: List[Dict[str, Any]] = Field(default_factory=list)
price_range: Optional[str] = None
retailer_count: int = 0
# The line and variant the preview grouped this row under. Web rows only
# need them sent back: their variant came from the retailer's brackets,
# which the plain product name no longer carries.
product_line: Optional[str] = None
variant: Optional[str] = None
class WebJobRequest(BaseModel):
@@ -245,7 +250,8 @@ async def ingest_brand_discovery(payload: DiscoveryIngestRequest) -> batch_commo
# them. Keyed by CSV position, which is how stage 11 reports source rows.
web_entries = {
index: {"listings": item.listings, "price_range": item.price_range or "",
"retailer_count": item.retailer_count}
"retailer_count": item.retailer_count,
"product_line": item.product_line or "", "variant": item.variant or ""}
for index, item in enumerate(payload.products) if item.listings
}
if web_entries:

View File

@@ -1003,9 +1003,21 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
"confidence_score": row.get("confidence_score"),
"validation_issues": list(row.get("validation_issues") or []),
"field_sources": dict(row.get("field_sources") or {}),
# Line and variant read off the size-free name. Web discovery's
# provenance pass may set a better variant later; the upsert keeps
# whichever arrived first, so this never overwrites it.
**_line_and_variant(row, name),
}
def _line_and_variant(row: Dict[str, Any], name: str) -> Dict[str, Optional[str]]:
from app.services.web_discovery.variants import split_variant
line, variant = split_variant(name or "")
return {"product_line": row.get("product_line") or line or None,
"variant": row.get("variant") or variant or None}
def _record_product(result: PipelineResult, brand: str, row: Dict[str, Any],
disposition: str,
source_rows: Optional[Dict[str, int]] = None) -> None:

View File

@@ -233,6 +233,10 @@ class DiscoveredProduct:
listings: List[Dict[str, Any]] = field(default_factory=list)
price_range: str = ""
retailer_count: int = 0
# "Lizol Disinfectant Surface & Floor Cleaner" / "Lavender". Shown grouped
# in the preview and sent back on ingest; stored by the pipeline/provenance.
product_line: str = ""
variant: str = ""
def as_preview(self) -> Dict[str, Any]:
body = self._preview_body()
@@ -263,6 +267,8 @@ class DiscoveredProduct:
"confidence": round(self.confidence, 2),
"matches_existing": self.matches_existing,
"notes": list(self.notes),
"product_line": self.product_line,
"variant": self.variant,
"selected": self.confidence >= 0.5,
}
@@ -286,6 +292,7 @@ class DiscoveryResult:
"brand_active": self.brand_active,
"filtering_enabled": self.filtering_enabled,
"products": [p.as_preview() for p in self.products],
"families": families_of(self.products),
"counts": dict(self.counts),
"warnings": list(self.warnings),
}
@@ -370,6 +377,25 @@ def _normalise_title(brand: str, title: Optional[str]) -> str:
return off_bulk.normalize_for_match(title, off_bulk.brand_tokens(brand))
def families_of(products: Sequence[DiscoveredProduct]) -> List[Dict[str, Any]]:
"""The preview grouped as product line -> variant -> pack sizes.
Each size carries `rows`, the indexes into `products` it came from, so the
panel can tick and untick the underlying rows from the grouped view."""
from app.services.web_discovery.variants import group_families
candidates = []
for index, p in enumerate(products):
for size in p.size_variants or [""]:
candidates.append({
"product_line": p.product_line, "variant": p.variant, "size": size,
"retailer_count": p.retailer_count, "price_range": p.price_range,
"providers": list(p.providers[:p.retailer_count] if p.retailer_count else []),
"row_index": index,
})
return group_families(candidates)
def _same_product(left: str, right: str, left_raw: str, right_raw: str) -> bool:
"""Whether two normalised titles name the same retail product.
@@ -392,9 +418,19 @@ def _same_product(left: str, right: str, left_raw: str, right_raw: str) -> bool:
return True
if off_bulk.has_extra_variant_conflict(left_raw, right_raw):
return False
# Citrus vs Floral scores 0.857 - over the floor. Two names that differ in
# a scent or flavour are two packs on the shelf, never one product.
if _scents(left_raw) != _scents(right_raw):
return False
return off_bulk.symmetric_similarity(left, right) >= _TITLE_MATCH_THRESHOLD
def _scents(title: str) -> frozenset:
from app.services.web_discovery.variants import variant_words
return variant_words(title)
def _match_existing(normalised: str, title: str,
catalog_keys: Dict[str, str]) -> Optional[str]:
"""The stored base name for this product, or None.
@@ -1196,6 +1232,7 @@ def discover_brand_products(
continue
if "web" in product.sources:
_apply_web(product, candidate)
_apply_line(product, candidate)
if (require_evidence and "off" not in product.sources
and "store" not in product.sources and not product.evidence):
dropped += 1
@@ -1259,6 +1296,20 @@ def _from_web(brand: str, job_id: Optional[str], warnings: List[str]) -> List[Di
return [dict(c) for c in job.candidates]
def _apply_line(product: DiscoveredProduct, candidate: Dict[str, Any]) -> None:
"""Product line and variant. A web candidate brings its own, read from the
retailer's brackets - kept while its variant words still appear in the
final name (a stored-name rename could have changed it). Everything else is
split from the final product name."""
from app.services.web_discovery.variants import split_variant
line, variant = candidate.get("product_line") or "", candidate.get("variant") or ""
name_words = set(re.findall(r"[a-z0-9]+", product.product_name.lower()))
if not (line and set(re.findall(r"[a-z0-9]+", variant.lower())) <= name_words):
line, variant = split_variant(product.product_name)
product.product_line, product.variant = line, variant
def _apply_web(product: DiscoveredProduct, candidate: Dict[str, Any]) -> None:
product.listings = list(candidate.get("listings") or [])
product.price_range = candidate.get("price_range") or ""

View File

@@ -367,9 +367,77 @@ def resolve_parent_brand(brand: str) -> str:
prefix = _known_leading_brand(key)
if prefix:
return resolve_parent_brand(prefix)
line = _known_leading_line(key)
if line:
return line
parent = _unique_parent_prefix(key)
if parent:
return parent
return brand
def is_abbreviation(token: str, parent: str) -> bool:
""""rb" for Reckitt Benckiser, "hul" for Hindustan Unilever, "jnj" for
Johnson & Johnson. Short, not itself a word of the parent, and starting
with the parent's first letter - which is what keeps "kit kat" (Nestle)
from being read as the abbreviation "kit" plus a sub-brand "kat"."""
parent_words = re.findall(r"[a-z0-9]+", (parent or "").lower())
return (bool(parent_words) and 1 < len(token) <= 4 and token not in parent_words
and token[0] == parent_words[0][0])
# Line names that are also everyday words or another maker's name. Read as a
# leading word they would misfile: "Laxmi Chilli Powder" is a spice brand, not
# HUL; "Boost Energy Drink", "Finish Line", "Baby Oil" (any maker) likewise.
_AMBIGUOUS_LINES = frozenset({"always", "boost", "finish", "wheel", "fairy", "ivory",
"vanish", "laxmi", "whisper"})
@lru_cache(maxsize=1)
def _abbreviated_lines() -> dict:
""""lizol" -> "reckitt benckiser", read off the alias "rb lizol".
Only line names of 5+ characters that exactly one parent registers: the
bare "sun" off "hul sun" would otherwise send "Sun Pharma" to Unilever."""
owners: dict = {}
for alias, parent in BRAND_ALIASES.items():
words = alias.split()
if len(words) > 1 and is_abbreviation(words[0], parent):
rest = " ".join(words[1:])
if (len(rest.replace(" ", "")) >= 5 and rest not in _AMBIGUOUS_LINES
and not rest.startswith("baby ")):
owners.setdefault(rest, set()).add(parent)
return {rest: next(iter(ps)) for rest, ps in owners.items() if len(ps) == 1}
def _known_leading_line(key: str) -> Optional[str]:
""""Lizol Floor Cleaner" -> "reckitt benckiser".
The registry spells Reckitt's lines with an abbreviation ("rb lizol"), so
the bare line name is neither a key nor a parent and `_known_leading_brand`
could not see it: a product typed with a line after it built a
brand_lizol_floor_cleaner table. Reached only after every older rule failed."""
lines = _abbreviated_lines()
words = key.split()
for n in range(len(words), 0, -1):
parent = lines.get(" ".join(words[:n]))
if parent:
return parent
return None
def _unique_parent_prefix(key: str) -> Optional[str]:
""""Reckitt" -> "reckitt benckiser": the typed words are the opening words
of exactly one parent. Two parents sharing the opening ("tata ...") give
nothing, and a single word must be 5+ characters."""
words = key.split()
if not words or (len(words) == 1 and len(words[0]) < 5):
return None
hits = {p for p in set(BRAND_ALIASES.values())
if p.split()[:len(words)] == words and len(p.split()) > len(words)}
return next(iter(hits)) if len(hits) == 1 else None
def _known_leading_brand(key: str) -> Optional[str]:
"""The longest leading run of words in `key` that is itself a known brand.

View File

@@ -104,6 +104,7 @@ EXPORT_COLUMNS = (
"barcode_lookup_status", "barcode_last_updated",
"gst_percent", "tax_amount", "hsn_gst_needs_review",
"highlights", "nutrients", "search_query", "field_sources",
"product_line", "variant",
# The validation verdict. Listed here for the same reason the barcode keys
# are: an export that strips them turns a re-seed into a silent downgrade.
# The upsert assigns these three plainly rather than COALESCEing them, so a

View File

@@ -179,6 +179,14 @@ def get_brand_table_ddl(brand: str) -> str:
nutrients_per_100g JSONB,
search_query TEXT,
-- "Lizol Disinfectant Surface & Floor Cleaner" / "Lavender": which
-- product line a row belongs to and which scent or flavour it is, so
-- every pack of every variant of one product can be listed together.
-- Fill-only-blanks in the upsert: a later re-derivation from the bare
-- name can never overwrite the bracket-aware value web discovery set.
product_line TEXT,
variant TEXT,
-- The deterministic validation verdict from product_validator.
-- validate_catalog() has always computed these three and annotated
-- them onto every row; until they were added here nothing projected
@@ -382,6 +390,8 @@ def _ensure_columns(cur, table_name: str) -> None:
"nutrients": "TEXT[]",
"nutrients_per_100g": "JSONB",
"search_query": "TEXT",
"product_line": "TEXT",
"variant": "TEXT",
# The product_validator verdict - see get_brand_table_ddl for why a
# row's confidence has to survive to disk to be worth computing.
"validation_status": "TEXT",
@@ -452,7 +462,7 @@ def _ensure_columns(cur, table_name: str) -> None:
# `nutrients_per_100g` joins them: it is mirrored from nutrition_facts by
# the same sync, not written by the INSERT, and is likewise a current
# column rather than legacy debris.
inserted_cols = {"id", "product_name", "title", "description", "category", "image_id", "image_url", "image_urls", "price_range", "size_variants", "providers", "fssai_license", "product_sku", "sku_source", "hsn_code", "final_selling_price", "selling_price", "barcode", "barcode_type", "gtin", "ean13", "upc", "barcode_source", "barcode_verified", "barcode_lookup_status", "barcode_last_updated", "gst_percent", "tax_amount", "hsn_gst_needs_review", "highlights", "nutrients", "nutrients_per_100g", "field_sources", "search_query", "nutrition_score", "health_score", "embedding", "img_vector", "img_vector_src", "created_at", "updated_at"}
inserted_cols = {"id", "product_name", "title", "description", "category", "image_id", "image_url", "image_urls", "price_range", "size_variants", "providers", "fssai_license", "product_sku", "sku_source", "hsn_code", "final_selling_price", "selling_price", "barcode", "barcode_type", "gtin", "ean13", "upc", "barcode_source", "barcode_verified", "barcode_lookup_status", "barcode_last_updated", "gst_percent", "tax_amount", "hsn_gst_needs_review", "highlights", "nutrients", "nutrients_per_100g", "field_sources", "search_query", "product_line", "variant", "nutrition_score", "health_score", "embedding", "img_vector", "img_vector_src", "created_at", "updated_at"}
for col, is_nullable, col_def in col_info:
if col not in inserted_cols and is_nullable == 'NO' and col_def is None:
cur.execute(f"ALTER TABLE {table_name} ALTER COLUMN {col} DROP NOT NULL")
@@ -657,6 +667,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
fssai_license = str(p.get("fssai_license", "")) if p.get("fssai_license") else ""
search_query = p.get("search_query") or ""
product_line = str(p.get("product_line") or "").strip() or None
variant = str(p.get("variant") or "").strip() or None
# Convert embedding to PostgreSQL vector format
embedding = p.get("embedding")
@@ -703,6 +715,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
confidence_score,
validation_issues,
field_sources,
product_line,
variant,
embedding_str
))
@@ -753,8 +767,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
gst_percent, tax_amount, hsn_gst_needs_review,
highlights, nutrients, search_query,
validation_status, confidence_score, validation_issues,
field_sources, embedding)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
field_sources, product_line, variant, embedding)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
ON CONFLICT (image_id) DO UPDATE SET
product_name = EXCLUDED.product_name,
title = EXCLUDED.title,
@@ -848,6 +862,12 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
-- depth here - each key's value is one flat record.
field_sources = COALESCE({table_name}.field_sources, '{{}}'::jsonb)
|| COALESCE(EXCLUDED.field_sources, '{{}}'::jsonb),
-- STORED FIRST: fill-only-blanks. The pipeline derives these
-- from the bare name on every run; the web provenance pass
-- writes the retailer's bracketed variant ("(Original)"),
-- which the bare name has lost. The better value must win.
product_line = COALESCE({table_name}.product_line, EXCLUDED.product_line),
variant = COALESCE({table_name}.variant, EXCLUDED.variant),
updated_at = CURRENT_TIMESTAMP
""",
rows,

View File

@@ -26,7 +26,7 @@ from collections import defaultdict
from dataclasses import dataclass
from typing import Dict, List, Set, Tuple
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
from app.services.brand_registry import BRAND_ALIASES, is_abbreviation, resolve_parent_brand
@dataclass(frozen=True)
@@ -34,20 +34,16 @@ class Target:
"""One name to search for, and the words a listing must lead with to count."""
query: str # what goes before `site:` - "Dettol", "Amul Butter"
brand_terms: Tuple[str, ...] # "dettol"; or ("nestle", "cerelac")
# "Lizol Floor Cleaner" typed: the listing must lead with "lizol" AND carry
# every one of these words somewhere, so other Lizol lines drop out.
line_terms: Tuple[str, ...] = ()
def _words(text: str) -> List[str]:
return re.findall(r"[a-z0-9]+", (text or "").lower())
def _is_abbreviation(token: str, parent: str) -> bool:
""""rb" for Reckitt Benckiser, "hul" for Hindustan Unilever, "jnj" for
Johnson & Johnson. Short, not itself a word of the parent, and starting
with the parent's first letter - which is what keeps "kit kat" (Nestle)
from being read as the abbreviation "kit" plus a sub-brand "kat"."""
parent_words = _words(parent)
return (bool(parent_words) and 1 < len(token) <= 4 and token not in parent_words
and token[0] == parent_words[0][0])
_is_abbreviation = is_abbreviation
def _sub_brand_part(alias: str, parent: str) -> str:
@@ -96,6 +92,20 @@ def _family(parent: str) -> List[Target]:
return out
def _line_target(name: str, typed_key: str, family: List[Target]) -> List[Target]:
""""Lizol Floor Cleaner" -> search that phrase, but judge listings by the
family member's own brand words ("lizol") plus the line words. Taking the
whole phrase as the brand would refuse every real title, which reads
"Lizol Disinfectant Surface & Floor Cleaner"."""
typed = typed_key.split()
for target in sorted(family, key=lambda t: -len(_words(t.query))):
lead = _words(target.query)
if len(typed) > len(lead) and typed[:len(lead)] == lead:
return [Target(query=name, brand_terms=target.brand_terms,
line_terms=tuple(typed[len(lead):]))]
return []
def targets_for(brand: str) -> List[Target]:
"""Sub-brands first, the brand itself last. Deterministic order.
@@ -112,7 +122,7 @@ def targets_for(brand: str) -> List[Target]:
own = [t for t in family
if " ".join(_words(t.query)) == typed_key
or typed_key in {" ".join(_words(term)) for term in t.brand_terms[1:] or t.brand_terms}]
return own or [Target(query=name, brand_terms=(name.lower(),))]
return own or _line_target(name, typed_key, family) or [Target(query=name, brand_terms=(name.lower(),))]
display = parent.title() if parent.islower() else parent
return family + [Target(query=display, brand_terms=(parent_key,))]

View File

@@ -25,7 +25,7 @@ from dataclasses import dataclass, field
from typing import Callable, Dict, List, Optional, Sequence
from app.infrastructure import settings
from app.services.web_discovery import cache, listings, search
from app.services.web_discovery import cache, listings, search, variants
from app.services.web_discovery.brands import Target, targets_for
logger = logging.getLogger(__name__)
@@ -89,16 +89,28 @@ def _variant_words(listing: listings.Listing) -> set:
def _same_product(a: listings.Listing, b: listings.Listing) -> bool:
"""Same brand word, same pack, near-identical name, and no bracketed
variant word on one side only. Measured: "(Citrus) 2l" and "(Floral) 2l"
share 4 of 6 words - enough for the overlap test alone, and two products."""
"""Same brand word, same pack, near-identical name, and no variant word on
one side only - bracketed, or a bare scent/flavour from
variants.VARIANT_TERMS. Measured: "(Citrus) 2l" and "(Floral) 2l" share 4
of 6 words, and "Cleaner Floral 500 ml" / "Cleaner Pine 500 ml" share 3 of
5 (exactly the 0.6 bar) - enough for the overlap test alone, and two
products each time."""
if a.brand_term != b.brand_term or a.size != b.size:
return False
ident_a, ident_b = _identity(a), _identity(b)
if _jaccard(ident_a, ident_b) < _SAME_NAME_JACCARD:
return False
differing = set(ident_a) ^ set(ident_b)
return not (differing & (_variant_words(a) | _variant_words(b)))
return not (differing & (_variant_words(a) | _variant_words(b) | variants.VARIANT_TERMS))
# "Lizol Floor Cleaner" typed: a Lizol listing of another line (toilet cleaner,
# wipes) is a real product, just not the one asked for.
OTHER_LINE = "other_product_line"
def _on_line(listing: listings.Listing, target: Target) -> bool:
return set(target.line_terms) <= set(listing.tokens)
def _jaccard(a: Sequence[str], b: Sequence[str]) -> float:
@@ -145,6 +157,7 @@ def cluster(found: Sequence[listings.Listing]) -> List[Dict]:
# normaliser (and off_bulk.strip_sizes) drops bracketed text, which
# turned "Lizol Floor Cleaner (Citrus)" and "(Floral)" into one name.
plain = re.sub(r"\s+", " ", re.sub(r"[()]", " ", name)).strip()
line, variant = variants.split_variant(name)
candidates.append({
"title": f"{plain} {size}",
"size": size,
@@ -154,6 +167,8 @@ def cluster(found: Sequence[listings.Listing]) -> List[Dict]:
"price_range": _price_range(prices),
"listings": [m.as_dict() for m in members[:_MAX_LISTINGS_PER_CANDIDATE]],
"brand_term": brand_term,
"product_line": line,
"variant": variant,
"source": "web",
})
candidates.sort(key=lambda c: (-c["retailer_count"], c["title"].lower()))
@@ -205,6 +220,8 @@ def run(brand: str, *, progress: Optional[Callable[[RunResult], None]] = None,
query.target.brand_terms)
if listing is None:
rejected[reason] += 1
elif not _on_line(listing, query.target):
rejected[OTHER_LINE] += 1
else:
kept.append(listing)
result.queries_done += 1

View File

@@ -277,6 +277,14 @@ def parse_hit(title: str, url: str, snippet: str,
if size is None:
# Some retailers put the size only in the title's tail ("... - 125 g | Zepto").
size, distinct = extract_size(title)
if size is None and not _MULTIPACK.search(snippet or ""):
# Blinkit often prints the pack only in the snippet ("... 500 ml ...
# Price"): 63 of 80 real Reckitt listings were lost for want of it.
# Taken only when the snippet names exactly ONE size - a snippet that
# lists the other packs ("also in 1 L, 2 L") says nothing about this one.
size, distinct = extract_size(snippet)
if distinct > 1:
return None, NO_PACK_SIZE
if size is None:
return None, NO_PACK_SIZE
if distinct > 1:

View File

@@ -10,7 +10,10 @@ web-found product it
* fills `price_range` from the listing prices when the pipeline left it blank;
* sets `validation_status = 'needs_review'` when only ONE retailer listed it
(never lifting a `rejected` verdict). Two or more retailers keep whatever the
validator decided.
validator decided;
* sets `product_line` / `variant` to what the listing titles said. These
OVERWRITE the pipeline's value: the pipeline read them off the bare name,
which has lost the retailer's brackets ("Harpic ... (Original)").
The join is exact: stage 11 reports every stored row's `image_id` with the
1-based sheet row it came from (header = row 1), and the ingest wrote the web
@@ -77,10 +80,13 @@ def apply(brand: str, updates: List[Dict[str, Any]]) -> int:
f"price_range = CASE WHEN COALESCE(price_range, '') = '' AND %s <> '' "
f"THEN %s ELSE price_range END, "
f"validation_status = CASE WHEN %s AND COALESCE(validation_status, '') <> 'rejected' "
f"THEN 'needs_review' ELSE validation_status END "
f"THEN 'needs_review' ELSE validation_status END, "
f"product_line = COALESCE(NULLIF(%s, ''), product_line), "
f"variant = COALESCE(NULLIF(%s, ''), variant) "
f"WHERE image_id = %s",
(Json(provenance), entry.get("price_range") or "", entry.get("price_range") or "",
single, update["image_id"]),
single, entry.get("product_line") or "", entry.get("variant") or "",
update["image_id"]),
)
written += cur.rowcount
except Exception as exc: # noqa: BLE001 - provenance must not take the batch down

View File

@@ -0,0 +1,162 @@
"""A product name -> (product line, variant), and candidates -> line -> variant -> sizes.
"Lizol Disinfectant Surface & Floor Cleaner (Lavender)" is the line
"Lizol Disinfectant Surface & Floor Cleaner" in the variant "Lavender". Retail
titles mark the variant two ways:
* in brackets - Blinkit's "(Citrus, 625 ml)", "(Original)". Whatever a
bracket holds once the size is gone IS the variant, whatever the word.
* as a bare word - "Lizol ... Cleaner Floral 500 ml". Only a word in
VARIANT_TERMS is read as a variant here; any other bare word stays in the
line, because guessing would split one product into two lines.
VARIANT_TERMS is scents and flavours only. Generic words ("original", "fresh",
"chocolate", "classic") are left out on purpose: "Dettol Original Soap" and
"Cadbury Dairy Milk Chocolate" name a line, not a variant. A bracket still
catches "(Original)". Add a word only when a real listing needs it.
"""
from __future__ import annotations
import re
from collections import Counter, defaultdict
from typing import Dict, Iterable, List, Tuple
VARIANT_TERMS = frozenset({
# scents - household and personal care
"citrus", "floral", "pine", "lavender", "jasmine", "rose", "lemon", "lime",
"orange", "sandal", "sandalwood", "neem", "mint", "eucalyptus", "aloe",
"lily", "marine", "ocean", "mogra", "tulsi", "vanilla", "berry", "musk",
# flavours that name a pack rather than a product
"strawberry", "mango", "elaichi", "masala", "pineapple", "litchi", "guava",
})
# A word that belongs to the variant when it sits next to one: "Lemon Fresh",
# "Ocean Breeze", "Lavender Blossom". Never a variant on its own.
_VARIANT_MODIFIERS = frozenset({"fresh", "blossom", "breeze", "burst", "bloom", "splash", "twist"})
_PACKAGING_WORDS = frozenset({"bar", "pouch", "bottle", "jar", "box", "tin", "carton", "pack", "refill"})
_BRACKET = re.compile(r"\(([^()]*)\)")
_WORD = re.compile(r"[A-Za-z0-9]+(?:'[A-Za-z]+)?")
def _words(text: str) -> List[str]:
return re.findall(r"[a-z0-9]+", (text or "").lower())
def _tidy(text: str) -> str:
return re.sub(r"\s+", " ", text or "").strip(" ,;:.-/&+–—")
def _title(word: str) -> str:
return word if any(c.isupper() for c in word) else word.capitalize()
def variant_words(text: str) -> frozenset:
"""The VARIANT_TERMS words in `text` - for telling two names apart."""
return frozenset(w for w in _words(text) if w in VARIANT_TERMS)
def split_variant(name: str) -> Tuple[str, str]:
"""(line, variant). A pack size in the name is ignored ("... Floral 500 ml").
("Lizol Floor Cleaner (Citrus)") -> ("Lizol Floor Cleaner", "Citrus")
("Lizol Floor Cleaner Floral") -> ("Lizol Floor Cleaner", "Floral")
("Harpic Toilet Cleaner (Original)") -> ("Harpic Toilet Cleaner", "Original")
("Dettol Original Soap") -> ("Dettol Original Soap", "")
"""
from app.services.web_discovery.listings import _strip_sizes
name = _tidy(re.sub(r"\s+", " ", _strip_sizes(name or "")))
groups = [_tidy(g) for g in _BRACKET.findall(name)]
groups = [g for g in groups if re.search(r"[A-Za-z]", g)]
if groups:
line = _tidy(re.sub(r"\s+", " ", _BRACKET.sub(" ", name)))
return line, " ".join(_title(w) for g in groups for w in _WORD.findall(g))
tokens = _WORD.findall(name)
lowered = [t.lower() for t in tokens]
keep = [l in VARIANT_TERMS for l in lowered]
if not any(keep) or all(keep):
return name, ""
# A modifier joins the variant only when it touches a variant word.
for i, l in enumerate(lowered):
if l in _VARIANT_MODIFIERS and ((i > 0 and keep[i - 1]) or (i + 1 < len(keep) and keep[i + 1])):
keep[i] = True
variant = " ".join(_title(t) for t, k in zip(tokens, keep) if k)
line = name
for t, k in zip(tokens, keep):
if k:
line = re.sub(r"(?<![A-Za-z0-9])" + re.escape(t) + r"(?![A-Za-z0-9])", " ", line, count=1)
line = _tidy(re.sub(r"\s*&\s*$", "", re.sub(r"\s+", " ", line)))
return (line or name), variant
def line_key(line: str, brand_term: str = "") -> Tuple[str, ...]:
"""Lines are one line when these words match - order, '&' and packaging aside."""
brand = set(_words(brand_term))
return tuple(sorted({w for w in _words(line) if w not in _PACKAGING_WORDS} - brand)) or tuple(_words(line))
_SIZE = re.compile(r"^(\d+(?:\.\d+)?)(kg|g|ml|l|pcs)$")
_SIZE_SCALE = {"g": (0, 1), "kg": (0, 1000), "ml": (1, 1), "l": (1, 1000), "pcs": (2, 1)}
def size_sort_key(size: str) -> Tuple[int, float, str]:
"""Mass, then volume, then counts, each smallest first: 500ml < 625ml < 1l."""
m = _SIZE.match((size or "").lower())
if not m:
return (9, 0.0, size or "")
group, scale = _SIZE_SCALE[m.group(2)]
return (group, float(m.group(1)) * scale, size)
def group_families(candidates: Iterable[Dict]) -> List[Dict]:
"""Candidates (one per name and pack) -> product line -> variant -> sizes.
Each candidate needs `product_line`, `variant` and `size`; `retailer_count`,
`price_range`, `providers` and `listings` are carried through when present,
and a `row_index` is collected into the size's `rows`.
Lines come out best-corroborated first; variants and sizes in a stable order.
"""
lines: Dict[Tuple, List[Dict]] = defaultdict(list)
for c in candidates:
line = c.get("product_line") or ""
if not line:
continue
lines[(c.get("brand_term") or "", line_key(line, c.get("brand_term") or ""))].append(c)
families: List[Dict] = []
for members in lines.values():
spellings = Counter(m["product_line"] for m in members)
display = sorted(spellings, key=lambda s: (-spellings[s], len(s), s))[0]
by_variant: Dict[str, List[Dict]] = defaultdict(list)
for m in members:
by_variant[m.get("variant") or ""].append(m)
variants = []
for variant, rows in sorted(by_variant.items(), key=lambda kv: (kv[0] == "", kv[0].lower())):
sizes: Dict[str, Dict] = {}
for r in rows:
size = r.get("size") or ((r.get("sizes") or [None])[0])
if not size:
continue
entry = sizes.setdefault(size, {"size": size, "retailer_count": 0, "price_range": "",
"providers": [], "listings": [], "rows": []})
if r.get("row_index") is not None:
entry["rows"].append(r["row_index"])
entry["retailer_count"] = max(entry["retailer_count"], int(r.get("retailer_count") or 0))
entry["price_range"] = entry["price_range"] or r.get("price_range") or ""
for p in r.get("providers") or []:
if p not in entry["providers"]:
entry["providers"].append(p)
entry["listings"].extend(r.get("listings") or [])
variants.append({"variant": variant,
"sizes": sorted(sizes.values(), key=lambda s: size_sort_key(s["size"]))})
families.append({
"product_line": display,
"brand_term": members[0].get("brand_term") or "",
"variants": variants,
"retailer_count": max((int(m.get("retailer_count") or 0) for m in members), default=0),
})
families.sort(key=lambda f: (-f["retailer_count"], -len(f["variants"]), f["product_line"].lower()))
return families

View File

@@ -383,7 +383,7 @@ class _Conn(_Cursor):
pass
def test_apply_touches_only_provenance_price_and_review_status(monkeypatch):
def test_apply_touches_only_provenance_price_review_status_and_line(monkeypatch):
conn = _Conn()
monkeypatch.setattr("app.services.vector_store._connect", lambda: conn)
provenance.apply("Reckitt Benckiser", [
@@ -393,7 +393,9 @@ def test_apply_touches_only_provenance_price_and_review_status(monkeypatch):
sql, params = conn.cur.calls[0]
assert sql.startswith("UPDATE brand_reckitt_benckiser SET field_sources")
assert "price_range = CASE WHEN COALESCE(price_range, '') = ''" in sql
assert "'rejected'" in sql and params[-2] is True and params[-1] == "one_shop"
assert "'rejected'" in sql and params[3] is True and params[-1] == "one_shop"
# A blank line/variant from the entry keeps what the pipeline stored.
assert "variant = COALESCE(NULLIF(%s, ''), variant)" in sql and params[-2] == ""
for column in ("product_name", "barcode", "image_url", "category"):
assert f"{column} =" not in sql

182
tests/test_web_variants.py Normal file
View File

@@ -0,0 +1,182 @@
"""A product name typed into Brand Discovery -> its line, variants and pack sizes.
Typing "Lizol" (a Reckitt line) must find Lizol, and the result must read as
"Lizol ... Floor Cleaner -> Citrus / Floral / Lavender -> 500 ml, 625 ml, 1 l"
rather than a flat list - without ever folding two scents into one product.
No network: every listing is a recorded Blinkit/Amazon title shape.
"""
from __future__ import annotations
import pytest
from app.services import brand_discovery as bd
from app.services import brand_registry
from app.services.web_discovery import brands, discover, jobs, listings, variants
LIZOL = ("lizol",)
LINE = "Lizol Disinfectant Surface & Floor Cleaner"
def _listing(title, n, terms=LIZOL, snippet=""):
listing, reason = listings.parse_hit(title, f"https://blinkit.com/prn/x/prid/{n}", snippet, terms)
assert reason is None, (title, reason)
return listing
# ---------------------------------------------------------------------------
# A product name resolves to the maker
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("typed", ["Lizol", "Lizol Floor Cleaner", "lizol disinfectant surface cleaner",
"Harpic Power Plus", "Dettol Liquid", "Reckitt"])
def test_a_product_name_resolves_to_its_maker(typed):
assert brand_registry.resolve_parent_brand(typed) == "reckitt benckiser"
@pytest.mark.parametrize("typed", ["Laxmi Chilli Powder", "Boost Energy Drink", "Baby Oil Himalaya",
"Sun Pharma", "Lizzy Cleaner"])
def test_an_everyday_word_is_not_read_as_a_line(typed):
assert brand_registry.resolve_parent_brand(typed) == typed
def test_every_name_that_resolved_before_still_resolves_the_same():
"""The new rules run only after every old one failed, so an alias key can
never change its answer."""
for alias, parent in brand_registry.BRAND_ALIASES.items():
assert brand_registry.resolve_parent_brand(alias) == parent
def test_a_product_line_searches_that_phrase_but_judges_by_the_line_word():
[target] = brands.targets_for("Lizol Floor Cleaner")
assert target.query == "Lizol Floor Cleaner"
assert target.brand_terms == ("lizol",) and target.line_terms == ("floor", "cleaner")
assert brands.targets_for("Lizol") == [brands.Target("Lizol", ("lizol",))]
def test_another_line_of_the_same_product_is_counted_and_dropped():
target = brands.Target("Lizol Floor Cleaner", LIZOL, ("floor", "cleaner"))
floor = _listing(f"{LINE} (Citrus - 500 ml) Price - Buy Online", 1)
toilet = _listing("Lizol Power Toilet Cleaner 500 ml Price - Buy Online", 2)
assert discover._on_line(floor, target) and not discover._on_line(toilet, target)
# ---------------------------------------------------------------------------
# Line and variant
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("name,line,variant", [
(f"{LINE} (Lavender)", LINE, "Lavender"),
("Lizol Disinfectant Surface Cleaner Floral 500 ml", "Lizol Disinfectant Surface Cleaner", "Floral"),
("Harpic Disinfectant Liquid Toilet Cleaner (Original)", "Harpic Disinfectant Liquid Toilet Cleaner",
"Original"),
("Lizol Floor Cleaner Lemon Fresh", "Lizol Floor Cleaner", "Lemon Fresh"),
# Generic words are a line, not a variant - only a bracket makes them one.
("Dettol Original Soap", "Dettol Original Soap", ""),
("Cadbury Dairy Milk Chocolate", "Cadbury Dairy Milk Chocolate", ""),
])
def test_split_variant(name, line, variant):
assert variants.split_variant(name) == (line, variant)
def test_sizes_sort_by_quantity_not_text():
sizes = ["1l", "500ml", "2l", "625ml", "200ml"]
assert sorted(sizes, key=variants.size_sort_key) == ["200ml", "500ml", "625ml", "1l", "2l"]
# ---------------------------------------------------------------------------
# Scents never merge
# ---------------------------------------------------------------------------
def test_unbracketed_scents_of_one_pack_stay_two_products():
"""3 of 5 words shared - exactly the 0.6 overlap bar, which used to fold them."""
floral = _listing("Lizol Disinfectant Surface Cleaner Floral 500 ml : Amazon.in: Health", 1)
pine = _listing("Lizol Disinfectant Surface Cleaner Pine 500 ml : Amazon.in: Health", 2)
assert len(discover.cluster([floral, pine])) == 2
def test_the_same_scent_on_two_retailers_is_still_one_product():
a = _listing(f"{LINE} (Citrus - 500 ml) Price - Buy Online", 1)
b, _ = listings.parse_hit(f"{LINE} Citrus 500 ml : Amazon.in", "https://www.amazon.in/dp/B000000001",
"", LIZOL)
[one] = discover.cluster([a, b])
assert one["retailer_count"] == 2 and one["variant"] == "Citrus"
def test_a_scent_cannot_be_renamed_to_a_stored_other_scent():
"""Citrus vs Floral scores 0.857 - over the 0.85 floor `_match_existing` uses."""
stored = {bd._normalise_title("Reckitt Benckiser", "Lizol Floor Cleaner Citrus"): "Lizol Floor Cleaner Citrus"}
key = bd._normalise_title("Reckitt Benckiser", "Lizol Floor Cleaner Floral")
assert bd._match_existing(key, "Lizol Floor Cleaner Floral", stored) is None
# ---------------------------------------------------------------------------
# Size from the snippet
# ---------------------------------------------------------------------------
def test_a_size_only_in_the_snippet_is_taken():
listing = _listing("Lizol Disinfectant Floor Cleaner (Jasmine) Price - Buy Online", 1,
snippet="Lizol Disinfectant Floor Cleaner Jasmine 975 ml. Delivered in minutes.")
assert listing.size == "975ml"
def test_a_snippet_naming_several_packs_proves_nothing():
listing, reason = listings.parse_hit(
"Lizol Disinfectant Floor Cleaner (Jasmine) Price - Buy Online", "https://blinkit.com/prn/x/prid/1",
"Available in 500 ml, 975 ml and 2 l packs.", LIZOL)
assert listing is None and reason == listings.NO_PACK_SIZE
def test_a_multipack_snippet_is_not_a_pack_size():
listing, reason = listings.parse_hit(
"Lizol Disinfectant Floor Cleaner (Jasmine) Price - Buy Online", "https://blinkit.com/prn/x/prid/1",
"Pack of 2 - 500 ml each", LIZOL)
assert listing is None and reason == listings.NO_PACK_SIZE
# ---------------------------------------------------------------------------
# End to end: line -> variants -> sizes
# ---------------------------------------------------------------------------
_TITLES = [
f"{LINE} (Citrus - 500 ml) Price - Buy Online at Best",
f"Buy {LINE} (Citrus, 625 ml) Online",
f"{LINE} (Citrus - 1 l) Price - Buy Online at Best",
f"{LINE} (Floral - 500 ml) Price - Buy Online at Best",
f"{LINE} (Floral - 2 l) Price - Buy Online at Best",
f"{LINE} (Lavender - 500 ml) Price - Buy Online at Best",
"Lizol Power Toilet Cleaner Original 500 ml Price - Buy Online",
]
def test_lizol_comes_out_as_line_then_variants_then_sizes():
found = [_listing(t, i) for i, t in enumerate(_TITLES)]
families = variants.group_families(discover.cluster(found))
floor = next(f for f in families if f["product_line"] == LINE)
assert {v["variant"]: [s["size"] for s in v["sizes"]] for v in floor["variants"]} == {
"Citrus": ["500ml", "625ml", "1l"],
"Floral": ["500ml", "2l"],
"Lavender": ["500ml"],
}
assert any(f["product_line"].startswith("Lizol Power Toilet Cleaner") for f in families)
@pytest.fixture
def no_other_sources(monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [])
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [])
monkeypatch.setattr(bd, "_from_brand_store", lambda brand: [])
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [])
def test_the_preview_carries_families_and_each_row_its_line(no_other_sources):
found = [_listing(t, i) for i, t in enumerate(_TITLES[:5])]
job = jobs.WebJob(job_id="d" * 32, brand="Lizol", status=jobs.DONE, backend="duckduckgo",
queries_total=7, queries_done=7, listings_kept=5, candidates=discover.cluster(found))
jobs._jobs[job.job_id] = job
body = bd.discover_brand_products("Lizol", use_llm=False, use_web=True, web_job_id=job.job_id).as_dict()
assert body["parent_brand"] == "reckitt benckiser" and body["table"] == "brand_reckitt_benckiser"
assert {(p["product_line"], p["variant"]) for p in body["products"]} == {(LINE, "Citrus"), (LINE, "Floral")}
[family] = body["families"]
sizes = {v["variant"]: [s["size"] for s in v["sizes"]] for v in family["variants"]}
assert sizes == {"Citrus": ["500ml", "625ml", "1l"], "Floral": ["500ml", "2l"]}
for v in family["variants"]:
for s in v["sizes"]:
assert all(body["products"][i]["variant"] == v["variant"] for i in s["rows"])