Product discovery on names

This commit is contained in:
sriram
2026-10-06 11:57:01 +05:30
parent c0601b65fe
commit 34c65a7cbd
13 changed files with 569 additions and 24 deletions

View File

@@ -105,6 +105,11 @@ class DiscoveredProductIn(BaseModel):
listings: List[Dict[str, Any]] = Field(default_factory=list)
price_range: Optional[str] = None
retailer_count: int = 0
# The line and variant the preview grouped this row under. Web rows only
# need them sent back: their variant came from the retailer's brackets,
# which the plain product name no longer carries.
product_line: Optional[str] = None
variant: Optional[str] = None
class WebJobRequest(BaseModel):
@@ -245,7 +250,8 @@ async def ingest_brand_discovery(payload: DiscoveryIngestRequest) -> batch_commo
# them. Keyed by CSV position, which is how stage 11 reports source rows.
web_entries = {
index: {"listings": item.listings, "price_range": item.price_range or "",
"retailer_count": item.retailer_count}
"retailer_count": item.retailer_count,
"product_line": item.product_line or "", "variant": item.variant or ""}
for index, item in enumerate(payload.products) if item.listings
}
if web_entries:

View File

@@ -1003,9 +1003,21 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
"confidence_score": row.get("confidence_score"),
"validation_issues": list(row.get("validation_issues") or []),
"field_sources": dict(row.get("field_sources") or {}),
# Line and variant read off the size-free name. Web discovery's
# provenance pass may set a better variant later; the upsert keeps
# whichever arrived first, so this never overwrites it.
**_line_and_variant(row, name),
}
def _line_and_variant(row: Dict[str, Any], name: str) -> Dict[str, Optional[str]]:
from app.services.web_discovery.variants import split_variant
line, variant = split_variant(name or "")
return {"product_line": row.get("product_line") or line or None,
"variant": row.get("variant") or variant or None}
def _record_product(result: PipelineResult, brand: str, row: Dict[str, Any],
disposition: str,
source_rows: Optional[Dict[str, int]] = None) -> None:

View File

@@ -233,6 +233,10 @@ class DiscoveredProduct:
listings: List[Dict[str, Any]] = field(default_factory=list)
price_range: str = ""
retailer_count: int = 0
# "Lizol Disinfectant Surface & Floor Cleaner" / "Lavender". Shown grouped
# in the preview and sent back on ingest; stored by the pipeline/provenance.
product_line: str = ""
variant: str = ""
def as_preview(self) -> Dict[str, Any]:
body = self._preview_body()
@@ -263,6 +267,8 @@ class DiscoveredProduct:
"confidence": round(self.confidence, 2),
"matches_existing": self.matches_existing,
"notes": list(self.notes),
"product_line": self.product_line,
"variant": self.variant,
"selected": self.confidence >= 0.5,
}
@@ -286,6 +292,7 @@ class DiscoveryResult:
"brand_active": self.brand_active,
"filtering_enabled": self.filtering_enabled,
"products": [p.as_preview() for p in self.products],
"families": families_of(self.products),
"counts": dict(self.counts),
"warnings": list(self.warnings),
}
@@ -370,6 +377,25 @@ def _normalise_title(brand: str, title: Optional[str]) -> str:
return off_bulk.normalize_for_match(title, off_bulk.brand_tokens(brand))
def families_of(products: Sequence[DiscoveredProduct]) -> List[Dict[str, Any]]:
"""The preview grouped as product line -> variant -> pack sizes.
Each size carries `rows`, the indexes into `products` it came from, so the
panel can tick and untick the underlying rows from the grouped view."""
from app.services.web_discovery.variants import group_families
candidates = []
for index, p in enumerate(products):
for size in p.size_variants or [""]:
candidates.append({
"product_line": p.product_line, "variant": p.variant, "size": size,
"retailer_count": p.retailer_count, "price_range": p.price_range,
"providers": list(p.providers[:p.retailer_count] if p.retailer_count else []),
"row_index": index,
})
return group_families(candidates)
def _same_product(left: str, right: str, left_raw: str, right_raw: str) -> bool:
"""Whether two normalised titles name the same retail product.
@@ -392,9 +418,19 @@ def _same_product(left: str, right: str, left_raw: str, right_raw: str) -> bool:
return True
if off_bulk.has_extra_variant_conflict(left_raw, right_raw):
return False
# Citrus vs Floral scores 0.857 - over the floor. Two names that differ in
# a scent or flavour are two packs on the shelf, never one product.
if _scents(left_raw) != _scents(right_raw):
return False
return off_bulk.symmetric_similarity(left, right) >= _TITLE_MATCH_THRESHOLD
def _scents(title: str) -> frozenset:
from app.services.web_discovery.variants import variant_words
return variant_words(title)
def _match_existing(normalised: str, title: str,
catalog_keys: Dict[str, str]) -> Optional[str]:
"""The stored base name for this product, or None.
@@ -1196,6 +1232,7 @@ def discover_brand_products(
continue
if "web" in product.sources:
_apply_web(product, candidate)
_apply_line(product, candidate)
if (require_evidence and "off" not in product.sources
and "store" not in product.sources and not product.evidence):
dropped += 1
@@ -1259,6 +1296,20 @@ def _from_web(brand: str, job_id: Optional[str], warnings: List[str]) -> List[Di
return [dict(c) for c in job.candidates]
def _apply_line(product: DiscoveredProduct, candidate: Dict[str, Any]) -> None:
"""Product line and variant. A web candidate brings its own, read from the
retailer's brackets - kept while its variant words still appear in the
final name (a stored-name rename could have changed it). Everything else is
split from the final product name."""
from app.services.web_discovery.variants import split_variant
line, variant = candidate.get("product_line") or "", candidate.get("variant") or ""
name_words = set(re.findall(r"[a-z0-9]+", product.product_name.lower()))
if not (line and set(re.findall(r"[a-z0-9]+", variant.lower())) <= name_words):
line, variant = split_variant(product.product_name)
product.product_line, product.variant = line, variant
def _apply_web(product: DiscoveredProduct, candidate: Dict[str, Any]) -> None:
product.listings = list(candidate.get("listings") or [])
product.price_range = candidate.get("price_range") or ""

View File

@@ -367,9 +367,77 @@ def resolve_parent_brand(brand: str) -> str:
prefix = _known_leading_brand(key)
if prefix:
return resolve_parent_brand(prefix)
line = _known_leading_line(key)
if line:
return line
parent = _unique_parent_prefix(key)
if parent:
return parent
return brand
def is_abbreviation(token: str, parent: str) -> bool:
""""rb" for Reckitt Benckiser, "hul" for Hindustan Unilever, "jnj" for
Johnson & Johnson. Short, not itself a word of the parent, and starting
with the parent's first letter - which is what keeps "kit kat" (Nestle)
from being read as the abbreviation "kit" plus a sub-brand "kat"."""
parent_words = re.findall(r"[a-z0-9]+", (parent or "").lower())
return (bool(parent_words) and 1 < len(token) <= 4 and token not in parent_words
and token[0] == parent_words[0][0])
# Line names that are also everyday words or another maker's name. Read as a
# leading word they would misfile: "Laxmi Chilli Powder" is a spice brand, not
# HUL; "Boost Energy Drink", "Finish Line", "Baby Oil" (any maker) likewise.
_AMBIGUOUS_LINES = frozenset({"always", "boost", "finish", "wheel", "fairy", "ivory",
"vanish", "laxmi", "whisper"})
@lru_cache(maxsize=1)
def _abbreviated_lines() -> dict:
""""lizol" -> "reckitt benckiser", read off the alias "rb lizol".
Only line names of 5+ characters that exactly one parent registers: the
bare "sun" off "hul sun" would otherwise send "Sun Pharma" to Unilever."""
owners: dict = {}
for alias, parent in BRAND_ALIASES.items():
words = alias.split()
if len(words) > 1 and is_abbreviation(words[0], parent):
rest = " ".join(words[1:])
if (len(rest.replace(" ", "")) >= 5 and rest not in _AMBIGUOUS_LINES
and not rest.startswith("baby ")):
owners.setdefault(rest, set()).add(parent)
return {rest: next(iter(ps)) for rest, ps in owners.items() if len(ps) == 1}
def _known_leading_line(key: str) -> Optional[str]:
""""Lizol Floor Cleaner" -> "reckitt benckiser".
The registry spells Reckitt's lines with an abbreviation ("rb lizol"), so
the bare line name is neither a key nor a parent and `_known_leading_brand`
could not see it: a product typed with a line after it built a
brand_lizol_floor_cleaner table. Reached only after every older rule failed."""
lines = _abbreviated_lines()
words = key.split()
for n in range(len(words), 0, -1):
parent = lines.get(" ".join(words[:n]))
if parent:
return parent
return None
def _unique_parent_prefix(key: str) -> Optional[str]:
""""Reckitt" -> "reckitt benckiser": the typed words are the opening words
of exactly one parent. Two parents sharing the opening ("tata ...") give
nothing, and a single word must be 5+ characters."""
words = key.split()
if not words or (len(words) == 1 and len(words[0]) < 5):
return None
hits = {p for p in set(BRAND_ALIASES.values())
if p.split()[:len(words)] == words and len(p.split()) > len(words)}
return next(iter(hits)) if len(hits) == 1 else None
def _known_leading_brand(key: str) -> Optional[str]:
"""The longest leading run of words in `key` that is itself a known brand.

View File

@@ -104,6 +104,7 @@ EXPORT_COLUMNS = (
"barcode_lookup_status", "barcode_last_updated",
"gst_percent", "tax_amount", "hsn_gst_needs_review",
"highlights", "nutrients", "search_query", "field_sources",
"product_line", "variant",
# The validation verdict. Listed here for the same reason the barcode keys
# are: an export that strips them turns a re-seed into a silent downgrade.
# The upsert assigns these three plainly rather than COALESCEing them, so a

View File

@@ -179,6 +179,14 @@ def get_brand_table_ddl(brand: str) -> str:
nutrients_per_100g JSONB,
search_query TEXT,
-- "Lizol Disinfectant Surface & Floor Cleaner" / "Lavender": which
-- product line a row belongs to and which scent or flavour it is, so
-- every pack of every variant of one product can be listed together.
-- Fill-only-blanks in the upsert: a later re-derivation from the bare
-- name can never overwrite the bracket-aware value web discovery set.
product_line TEXT,
variant TEXT,
-- The deterministic validation verdict from product_validator.
-- validate_catalog() has always computed these three and annotated
-- them onto every row; until they were added here nothing projected
@@ -382,6 +390,8 @@ def _ensure_columns(cur, table_name: str) -> None:
"nutrients": "TEXT[]",
"nutrients_per_100g": "JSONB",
"search_query": "TEXT",
"product_line": "TEXT",
"variant": "TEXT",
# The product_validator verdict - see get_brand_table_ddl for why a
# row's confidence has to survive to disk to be worth computing.
"validation_status": "TEXT",
@@ -452,7 +462,7 @@ def _ensure_columns(cur, table_name: str) -> None:
# `nutrients_per_100g` joins them: it is mirrored from nutrition_facts by
# the same sync, not written by the INSERT, and is likewise a current
# column rather than legacy debris.
inserted_cols = {"id", "product_name", "title", "description", "category", "image_id", "image_url", "image_urls", "price_range", "size_variants", "providers", "fssai_license", "product_sku", "sku_source", "hsn_code", "final_selling_price", "selling_price", "barcode", "barcode_type", "gtin", "ean13", "upc", "barcode_source", "barcode_verified", "barcode_lookup_status", "barcode_last_updated", "gst_percent", "tax_amount", "hsn_gst_needs_review", "highlights", "nutrients", "nutrients_per_100g", "field_sources", "search_query", "nutrition_score", "health_score", "embedding", "img_vector", "img_vector_src", "created_at", "updated_at"}
inserted_cols = {"id", "product_name", "title", "description", "category", "image_id", "image_url", "image_urls", "price_range", "size_variants", "providers", "fssai_license", "product_sku", "sku_source", "hsn_code", "final_selling_price", "selling_price", "barcode", "barcode_type", "gtin", "ean13", "upc", "barcode_source", "barcode_verified", "barcode_lookup_status", "barcode_last_updated", "gst_percent", "tax_amount", "hsn_gst_needs_review", "highlights", "nutrients", "nutrients_per_100g", "field_sources", "search_query", "product_line", "variant", "nutrition_score", "health_score", "embedding", "img_vector", "img_vector_src", "created_at", "updated_at"}
for col, is_nullable, col_def in col_info:
if col not in inserted_cols and is_nullable == 'NO' and col_def is None:
cur.execute(f"ALTER TABLE {table_name} ALTER COLUMN {col} DROP NOT NULL")
@@ -657,6 +667,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
fssai_license = str(p.get("fssai_license", "")) if p.get("fssai_license") else ""
search_query = p.get("search_query") or ""
product_line = str(p.get("product_line") or "").strip() or None
variant = str(p.get("variant") or "").strip() or None
# Convert embedding to PostgreSQL vector format
embedding = p.get("embedding")
@@ -703,6 +715,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
confidence_score,
validation_issues,
field_sources,
product_line,
variant,
embedding_str
))
@@ -753,8 +767,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
gst_percent, tax_amount, hsn_gst_needs_review,
highlights, nutrients, search_query,
validation_status, confidence_score, validation_issues,
field_sources, embedding)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
field_sources, product_line, variant, embedding)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
ON CONFLICT (image_id) DO UPDATE SET
product_name = EXCLUDED.product_name,
title = EXCLUDED.title,
@@ -848,6 +862,12 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
-- depth here - each key's value is one flat record.
field_sources = COALESCE({table_name}.field_sources, '{{}}'::jsonb)
|| COALESCE(EXCLUDED.field_sources, '{{}}'::jsonb),
-- STORED FIRST: fill-only-blanks. The pipeline derives these
-- from the bare name on every run; the web provenance pass
-- writes the retailer's bracketed variant ("(Original)"),
-- which the bare name has lost. The better value must win.
product_line = COALESCE({table_name}.product_line, EXCLUDED.product_line),
variant = COALESCE({table_name}.variant, EXCLUDED.variant),
updated_at = CURRENT_TIMESTAMP
""",
rows,

View File

@@ -26,7 +26,7 @@ from collections import defaultdict
from dataclasses import dataclass
from typing import Dict, List, Set, Tuple
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
from app.services.brand_registry import BRAND_ALIASES, is_abbreviation, resolve_parent_brand
@dataclass(frozen=True)
@@ -34,20 +34,16 @@ class Target:
"""One name to search for, and the words a listing must lead with to count."""
query: str # what goes before `site:` - "Dettol", "Amul Butter"
brand_terms: Tuple[str, ...] # "dettol"; or ("nestle", "cerelac")
# "Lizol Floor Cleaner" typed: the listing must lead with "lizol" AND carry
# every one of these words somewhere, so other Lizol lines drop out.
line_terms: Tuple[str, ...] = ()
def _words(text: str) -> List[str]:
return re.findall(r"[a-z0-9]+", (text or "").lower())
def _is_abbreviation(token: str, parent: str) -> bool:
""""rb" for Reckitt Benckiser, "hul" for Hindustan Unilever, "jnj" for
Johnson & Johnson. Short, not itself a word of the parent, and starting
with the parent's first letter - which is what keeps "kit kat" (Nestle)
from being read as the abbreviation "kit" plus a sub-brand "kat"."""
parent_words = _words(parent)
return (bool(parent_words) and 1 < len(token) <= 4 and token not in parent_words
and token[0] == parent_words[0][0])
_is_abbreviation = is_abbreviation
def _sub_brand_part(alias: str, parent: str) -> str:
@@ -96,6 +92,20 @@ def _family(parent: str) -> List[Target]:
return out
def _line_target(name: str, typed_key: str, family: List[Target]) -> List[Target]:
""""Lizol Floor Cleaner" -> search that phrase, but judge listings by the
family member's own brand words ("lizol") plus the line words. Taking the
whole phrase as the brand would refuse every real title, which reads
"Lizol Disinfectant Surface & Floor Cleaner"."""
typed = typed_key.split()
for target in sorted(family, key=lambda t: -len(_words(t.query))):
lead = _words(target.query)
if len(typed) > len(lead) and typed[:len(lead)] == lead:
return [Target(query=name, brand_terms=target.brand_terms,
line_terms=tuple(typed[len(lead):]))]
return []
def targets_for(brand: str) -> List[Target]:
"""Sub-brands first, the brand itself last. Deterministic order.
@@ -112,7 +122,7 @@ def targets_for(brand: str) -> List[Target]:
own = [t for t in family
if " ".join(_words(t.query)) == typed_key
or typed_key in {" ".join(_words(term)) for term in t.brand_terms[1:] or t.brand_terms}]
return own or [Target(query=name, brand_terms=(name.lower(),))]
return own or _line_target(name, typed_key, family) or [Target(query=name, brand_terms=(name.lower(),))]
display = parent.title() if parent.islower() else parent
return family + [Target(query=display, brand_terms=(parent_key,))]

View File

@@ -25,7 +25,7 @@ from dataclasses import dataclass, field
from typing import Callable, Dict, List, Optional, Sequence
from app.infrastructure import settings
from app.services.web_discovery import cache, listings, search
from app.services.web_discovery import cache, listings, search, variants
from app.services.web_discovery.brands import Target, targets_for
logger = logging.getLogger(__name__)
@@ -89,16 +89,28 @@ def _variant_words(listing: listings.Listing) -> set:
def _same_product(a: listings.Listing, b: listings.Listing) -> bool:
"""Same brand word, same pack, near-identical name, and no bracketed
variant word on one side only. Measured: "(Citrus) 2l" and "(Floral) 2l"
share 4 of 6 words - enough for the overlap test alone, and two products."""
"""Same brand word, same pack, near-identical name, and no variant word on
one side only - bracketed, or a bare scent/flavour from
variants.VARIANT_TERMS. Measured: "(Citrus) 2l" and "(Floral) 2l" share 4
of 6 words, and "Cleaner Floral 500 ml" / "Cleaner Pine 500 ml" share 3 of
5 (exactly the 0.6 bar) - enough for the overlap test alone, and two
products each time."""
if a.brand_term != b.brand_term or a.size != b.size:
return False
ident_a, ident_b = _identity(a), _identity(b)
if _jaccard(ident_a, ident_b) < _SAME_NAME_JACCARD:
return False
differing = set(ident_a) ^ set(ident_b)
return not (differing & (_variant_words(a) | _variant_words(b)))
return not (differing & (_variant_words(a) | _variant_words(b) | variants.VARIANT_TERMS))
# "Lizol Floor Cleaner" typed: a Lizol listing of another line (toilet cleaner,
# wipes) is a real product, just not the one asked for.
OTHER_LINE = "other_product_line"
def _on_line(listing: listings.Listing, target: Target) -> bool:
return set(target.line_terms) <= set(listing.tokens)
def _jaccard(a: Sequence[str], b: Sequence[str]) -> float:
@@ -145,6 +157,7 @@ def cluster(found: Sequence[listings.Listing]) -> List[Dict]:
# normaliser (and off_bulk.strip_sizes) drops bracketed text, which
# turned "Lizol Floor Cleaner (Citrus)" and "(Floral)" into one name.
plain = re.sub(r"\s+", " ", re.sub(r"[()]", " ", name)).strip()
line, variant = variants.split_variant(name)
candidates.append({
"title": f"{plain} {size}",
"size": size,
@@ -154,6 +167,8 @@ def cluster(found: Sequence[listings.Listing]) -> List[Dict]:
"price_range": _price_range(prices),
"listings": [m.as_dict() for m in members[:_MAX_LISTINGS_PER_CANDIDATE]],
"brand_term": brand_term,
"product_line": line,
"variant": variant,
"source": "web",
})
candidates.sort(key=lambda c: (-c["retailer_count"], c["title"].lower()))
@@ -205,6 +220,8 @@ def run(brand: str, *, progress: Optional[Callable[[RunResult], None]] = None,
query.target.brand_terms)
if listing is None:
rejected[reason] += 1
elif not _on_line(listing, query.target):
rejected[OTHER_LINE] += 1
else:
kept.append(listing)
result.queries_done += 1

View File

@@ -277,6 +277,14 @@ def parse_hit(title: str, url: str, snippet: str,
if size is None:
# Some retailers put the size only in the title's tail ("... - 125 g | Zepto").
size, distinct = extract_size(title)
if size is None and not _MULTIPACK.search(snippet or ""):
# Blinkit often prints the pack only in the snippet ("... 500 ml ...
# Price"): 63 of 80 real Reckitt listings were lost for want of it.
# Taken only when the snippet names exactly ONE size - a snippet that
# lists the other packs ("also in 1 L, 2 L") says nothing about this one.
size, distinct = extract_size(snippet)
if distinct > 1:
return None, NO_PACK_SIZE
if size is None:
return None, NO_PACK_SIZE
if distinct > 1:

View File

@@ -10,7 +10,10 @@ web-found product it
* fills `price_range` from the listing prices when the pipeline left it blank;
* sets `validation_status = 'needs_review'` when only ONE retailer listed it
(never lifting a `rejected` verdict). Two or more retailers keep whatever the
validator decided.
validator decided;
* sets `product_line` / `variant` to what the listing titles said. These
OVERWRITE the pipeline's value: the pipeline read them off the bare name,
which has lost the retailer's brackets ("Harpic ... (Original)").
The join is exact: stage 11 reports every stored row's `image_id` with the
1-based sheet row it came from (header = row 1), and the ingest wrote the web
@@ -77,10 +80,13 @@ def apply(brand: str, updates: List[Dict[str, Any]]) -> int:
f"price_range = CASE WHEN COALESCE(price_range, '') = '' AND %s <> '' "
f"THEN %s ELSE price_range END, "
f"validation_status = CASE WHEN %s AND COALESCE(validation_status, '') <> 'rejected' "
f"THEN 'needs_review' ELSE validation_status END "
f"THEN 'needs_review' ELSE validation_status END, "
f"product_line = COALESCE(NULLIF(%s, ''), product_line), "
f"variant = COALESCE(NULLIF(%s, ''), variant) "
f"WHERE image_id = %s",
(Json(provenance), entry.get("price_range") or "", entry.get("price_range") or "",
single, update["image_id"]),
single, entry.get("product_line") or "", entry.get("variant") or "",
update["image_id"]),
)
written += cur.rowcount
except Exception as exc: # noqa: BLE001 - provenance must not take the batch down

View File

@@ -0,0 +1,162 @@
"""A product name -> (product line, variant), and candidates -> line -> variant -> sizes.
"Lizol Disinfectant Surface & Floor Cleaner (Lavender)" is the line
"Lizol Disinfectant Surface & Floor Cleaner" in the variant "Lavender". Retail
titles mark the variant two ways:
* in brackets - Blinkit's "(Citrus, 625 ml)", "(Original)". Whatever a
bracket holds once the size is gone IS the variant, whatever the word.
* as a bare word - "Lizol ... Cleaner Floral 500 ml". Only a word in
VARIANT_TERMS is read as a variant here; any other bare word stays in the
line, because guessing would split one product into two lines.
VARIANT_TERMS is scents and flavours only. Generic words ("original", "fresh",
"chocolate", "classic") are left out on purpose: "Dettol Original Soap" and
"Cadbury Dairy Milk Chocolate" name a line, not a variant. A bracket still
catches "(Original)". Add a word only when a real listing needs it.
"""
from __future__ import annotations
import re
from collections import Counter, defaultdict
from typing import Dict, Iterable, List, Tuple
VARIANT_TERMS = frozenset({
# scents - household and personal care
"citrus", "floral", "pine", "lavender", "jasmine", "rose", "lemon", "lime",
"orange", "sandal", "sandalwood", "neem", "mint", "eucalyptus", "aloe",
"lily", "marine", "ocean", "mogra", "tulsi", "vanilla", "berry", "musk",
# flavours that name a pack rather than a product
"strawberry", "mango", "elaichi", "masala", "pineapple", "litchi", "guava",
})
# A word that belongs to the variant when it sits next to one: "Lemon Fresh",
# "Ocean Breeze", "Lavender Blossom". Never a variant on its own.
_VARIANT_MODIFIERS = frozenset({"fresh", "blossom", "breeze", "burst", "bloom", "splash", "twist"})
_PACKAGING_WORDS = frozenset({"bar", "pouch", "bottle", "jar", "box", "tin", "carton", "pack", "refill"})
_BRACKET = re.compile(r"\(([^()]*)\)")
_WORD = re.compile(r"[A-Za-z0-9]+(?:'[A-Za-z]+)?")
def _words(text: str) -> List[str]:
return re.findall(r"[a-z0-9]+", (text or "").lower())
def _tidy(text: str) -> str:
return re.sub(r"\s+", " ", text or "").strip(" ,;:.-/&+–—")
def _title(word: str) -> str:
return word if any(c.isupper() for c in word) else word.capitalize()
def variant_words(text: str) -> frozenset:
"""The VARIANT_TERMS words in `text` - for telling two names apart."""
return frozenset(w for w in _words(text) if w in VARIANT_TERMS)
def split_variant(name: str) -> Tuple[str, str]:
"""(line, variant). A pack size in the name is ignored ("... Floral 500 ml").
("Lizol Floor Cleaner (Citrus)") -> ("Lizol Floor Cleaner", "Citrus")
("Lizol Floor Cleaner Floral") -> ("Lizol Floor Cleaner", "Floral")
("Harpic Toilet Cleaner (Original)") -> ("Harpic Toilet Cleaner", "Original")
("Dettol Original Soap") -> ("Dettol Original Soap", "")
"""
from app.services.web_discovery.listings import _strip_sizes
name = _tidy(re.sub(r"\s+", " ", _strip_sizes(name or "")))
groups = [_tidy(g) for g in _BRACKET.findall(name)]
groups = [g for g in groups if re.search(r"[A-Za-z]", g)]
if groups:
line = _tidy(re.sub(r"\s+", " ", _BRACKET.sub(" ", name)))
return line, " ".join(_title(w) for g in groups for w in _WORD.findall(g))
tokens = _WORD.findall(name)
lowered = [t.lower() for t in tokens]
keep = [l in VARIANT_TERMS for l in lowered]
if not any(keep) or all(keep):
return name, ""
# A modifier joins the variant only when it touches a variant word.
for i, l in enumerate(lowered):
if l in _VARIANT_MODIFIERS and ((i > 0 and keep[i - 1]) or (i + 1 < len(keep) and keep[i + 1])):
keep[i] = True
variant = " ".join(_title(t) for t, k in zip(tokens, keep) if k)
line = name
for t, k in zip(tokens, keep):
if k:
line = re.sub(r"(?<![A-Za-z0-9])" + re.escape(t) + r"(?![A-Za-z0-9])", " ", line, count=1)
line = _tidy(re.sub(r"\s*&\s*$", "", re.sub(r"\s+", " ", line)))
return (line or name), variant
def line_key(line: str, brand_term: str = "") -> Tuple[str, ...]:
"""Lines are one line when these words match - order, '&' and packaging aside."""
brand = set(_words(brand_term))
return tuple(sorted({w for w in _words(line) if w not in _PACKAGING_WORDS} - brand)) or tuple(_words(line))
_SIZE = re.compile(r"^(\d+(?:\.\d+)?)(kg|g|ml|l|pcs)$")
_SIZE_SCALE = {"g": (0, 1), "kg": (0, 1000), "ml": (1, 1), "l": (1, 1000), "pcs": (2, 1)}
def size_sort_key(size: str) -> Tuple[int, float, str]:
"""Mass, then volume, then counts, each smallest first: 500ml < 625ml < 1l."""
m = _SIZE.match((size or "").lower())
if not m:
return (9, 0.0, size or "")
group, scale = _SIZE_SCALE[m.group(2)]
return (group, float(m.group(1)) * scale, size)
def group_families(candidates: Iterable[Dict]) -> List[Dict]:
"""Candidates (one per name and pack) -> product line -> variant -> sizes.
Each candidate needs `product_line`, `variant` and `size`; `retailer_count`,
`price_range`, `providers` and `listings` are carried through when present,
and a `row_index` is collected into the size's `rows`.
Lines come out best-corroborated first; variants and sizes in a stable order.
"""
lines: Dict[Tuple, List[Dict]] = defaultdict(list)
for c in candidates:
line = c.get("product_line") or ""
if not line:
continue
lines[(c.get("brand_term") or "", line_key(line, c.get("brand_term") or ""))].append(c)
families: List[Dict] = []
for members in lines.values():
spellings = Counter(m["product_line"] for m in members)
display = sorted(spellings, key=lambda s: (-spellings[s], len(s), s))[0]
by_variant: Dict[str, List[Dict]] = defaultdict(list)
for m in members:
by_variant[m.get("variant") or ""].append(m)
variants = []
for variant, rows in sorted(by_variant.items(), key=lambda kv: (kv[0] == "", kv[0].lower())):
sizes: Dict[str, Dict] = {}
for r in rows:
size = r.get("size") or ((r.get("sizes") or [None])[0])
if not size:
continue
entry = sizes.setdefault(size, {"size": size, "retailer_count": 0, "price_range": "",
"providers": [], "listings": [], "rows": []})
if r.get("row_index") is not None:
entry["rows"].append(r["row_index"])
entry["retailer_count"] = max(entry["retailer_count"], int(r.get("retailer_count") or 0))
entry["price_range"] = entry["price_range"] or r.get("price_range") or ""
for p in r.get("providers") or []:
if p not in entry["providers"]:
entry["providers"].append(p)
entry["listings"].extend(r.get("listings") or [])
variants.append({"variant": variant,
"sizes": sorted(sizes.values(), key=lambda s: size_sort_key(s["size"]))})
families.append({
"product_line": display,
"brand_term": members[0].get("brand_term") or "",
"variants": variants,
"retailer_count": max((int(m.get("retailer_count") or 0) for m in members), default=0),
})
families.sort(key=lambda f: (-f["retailer_count"], -len(f["variants"]), f["product_line"].lower()))
return families