Product discovery on names
This commit is contained in:
@@ -233,6 +233,10 @@ class DiscoveredProduct:
|
||||
listings: List[Dict[str, Any]] = field(default_factory=list)
|
||||
price_range: str = ""
|
||||
retailer_count: int = 0
|
||||
# "Lizol Disinfectant Surface & Floor Cleaner" / "Lavender". Shown grouped
|
||||
# in the preview and sent back on ingest; stored by the pipeline/provenance.
|
||||
product_line: str = ""
|
||||
variant: str = ""
|
||||
|
||||
def as_preview(self) -> Dict[str, Any]:
|
||||
body = self._preview_body()
|
||||
@@ -263,6 +267,8 @@ class DiscoveredProduct:
|
||||
"confidence": round(self.confidence, 2),
|
||||
"matches_existing": self.matches_existing,
|
||||
"notes": list(self.notes),
|
||||
"product_line": self.product_line,
|
||||
"variant": self.variant,
|
||||
"selected": self.confidence >= 0.5,
|
||||
}
|
||||
|
||||
@@ -286,6 +292,7 @@ class DiscoveryResult:
|
||||
"brand_active": self.brand_active,
|
||||
"filtering_enabled": self.filtering_enabled,
|
||||
"products": [p.as_preview() for p in self.products],
|
||||
"families": families_of(self.products),
|
||||
"counts": dict(self.counts),
|
||||
"warnings": list(self.warnings),
|
||||
}
|
||||
@@ -370,6 +377,25 @@ def _normalise_title(brand: str, title: Optional[str]) -> str:
|
||||
return off_bulk.normalize_for_match(title, off_bulk.brand_tokens(brand))
|
||||
|
||||
|
||||
def families_of(products: Sequence[DiscoveredProduct]) -> List[Dict[str, Any]]:
|
||||
"""The preview grouped as product line -> variant -> pack sizes.
|
||||
|
||||
Each size carries `rows`, the indexes into `products` it came from, so the
|
||||
panel can tick and untick the underlying rows from the grouped view."""
|
||||
from app.services.web_discovery.variants import group_families
|
||||
|
||||
candidates = []
|
||||
for index, p in enumerate(products):
|
||||
for size in p.size_variants or [""]:
|
||||
candidates.append({
|
||||
"product_line": p.product_line, "variant": p.variant, "size": size,
|
||||
"retailer_count": p.retailer_count, "price_range": p.price_range,
|
||||
"providers": list(p.providers[:p.retailer_count] if p.retailer_count else []),
|
||||
"row_index": index,
|
||||
})
|
||||
return group_families(candidates)
|
||||
|
||||
|
||||
def _same_product(left: str, right: str, left_raw: str, right_raw: str) -> bool:
|
||||
"""Whether two normalised titles name the same retail product.
|
||||
|
||||
@@ -392,9 +418,19 @@ def _same_product(left: str, right: str, left_raw: str, right_raw: str) -> bool:
|
||||
return True
|
||||
if off_bulk.has_extra_variant_conflict(left_raw, right_raw):
|
||||
return False
|
||||
# Citrus vs Floral scores 0.857 - over the floor. Two names that differ in
|
||||
# a scent or flavour are two packs on the shelf, never one product.
|
||||
if _scents(left_raw) != _scents(right_raw):
|
||||
return False
|
||||
return off_bulk.symmetric_similarity(left, right) >= _TITLE_MATCH_THRESHOLD
|
||||
|
||||
|
||||
def _scents(title: str) -> frozenset:
|
||||
from app.services.web_discovery.variants import variant_words
|
||||
|
||||
return variant_words(title)
|
||||
|
||||
|
||||
def _match_existing(normalised: str, title: str,
|
||||
catalog_keys: Dict[str, str]) -> Optional[str]:
|
||||
"""The stored base name for this product, or None.
|
||||
@@ -1196,6 +1232,7 @@ def discover_brand_products(
|
||||
continue
|
||||
if "web" in product.sources:
|
||||
_apply_web(product, candidate)
|
||||
_apply_line(product, candidate)
|
||||
if (require_evidence and "off" not in product.sources
|
||||
and "store" not in product.sources and not product.evidence):
|
||||
dropped += 1
|
||||
@@ -1259,6 +1296,20 @@ def _from_web(brand: str, job_id: Optional[str], warnings: List[str]) -> List[Di
|
||||
return [dict(c) for c in job.candidates]
|
||||
|
||||
|
||||
def _apply_line(product: DiscoveredProduct, candidate: Dict[str, Any]) -> None:
|
||||
"""Product line and variant. A web candidate brings its own, read from the
|
||||
retailer's brackets - kept while its variant words still appear in the
|
||||
final name (a stored-name rename could have changed it). Everything else is
|
||||
split from the final product name."""
|
||||
from app.services.web_discovery.variants import split_variant
|
||||
|
||||
line, variant = candidate.get("product_line") or "", candidate.get("variant") or ""
|
||||
name_words = set(re.findall(r"[a-z0-9]+", product.product_name.lower()))
|
||||
if not (line and set(re.findall(r"[a-z0-9]+", variant.lower())) <= name_words):
|
||||
line, variant = split_variant(product.product_name)
|
||||
product.product_line, product.variant = line, variant
|
||||
|
||||
|
||||
def _apply_web(product: DiscoveredProduct, candidate: Dict[str, Any]) -> None:
|
||||
product.listings = list(candidate.get("listings") or [])
|
||||
product.price_range = candidate.get("price_range") or ""
|
||||
|
||||
@@ -367,9 +367,77 @@ def resolve_parent_brand(brand: str) -> str:
|
||||
prefix = _known_leading_brand(key)
|
||||
if prefix:
|
||||
return resolve_parent_brand(prefix)
|
||||
line = _known_leading_line(key)
|
||||
if line:
|
||||
return line
|
||||
parent = _unique_parent_prefix(key)
|
||||
if parent:
|
||||
return parent
|
||||
return brand
|
||||
|
||||
|
||||
def is_abbreviation(token: str, parent: str) -> bool:
|
||||
""""rb" for Reckitt Benckiser, "hul" for Hindustan Unilever, "jnj" for
|
||||
Johnson & Johnson. Short, not itself a word of the parent, and starting
|
||||
with the parent's first letter - which is what keeps "kit kat" (Nestle)
|
||||
from being read as the abbreviation "kit" plus a sub-brand "kat"."""
|
||||
parent_words = re.findall(r"[a-z0-9]+", (parent or "").lower())
|
||||
return (bool(parent_words) and 1 < len(token) <= 4 and token not in parent_words
|
||||
and token[0] == parent_words[0][0])
|
||||
|
||||
|
||||
# Line names that are also everyday words or another maker's name. Read as a
|
||||
# leading word they would misfile: "Laxmi Chilli Powder" is a spice brand, not
|
||||
# HUL; "Boost Energy Drink", "Finish Line", "Baby Oil" (any maker) likewise.
|
||||
_AMBIGUOUS_LINES = frozenset({"always", "boost", "finish", "wheel", "fairy", "ivory",
|
||||
"vanish", "laxmi", "whisper"})
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def _abbreviated_lines() -> dict:
|
||||
""""lizol" -> "reckitt benckiser", read off the alias "rb lizol".
|
||||
|
||||
Only line names of 5+ characters that exactly one parent registers: the
|
||||
bare "sun" off "hul sun" would otherwise send "Sun Pharma" to Unilever."""
|
||||
owners: dict = {}
|
||||
for alias, parent in BRAND_ALIASES.items():
|
||||
words = alias.split()
|
||||
if len(words) > 1 and is_abbreviation(words[0], parent):
|
||||
rest = " ".join(words[1:])
|
||||
if (len(rest.replace(" ", "")) >= 5 and rest not in _AMBIGUOUS_LINES
|
||||
and not rest.startswith("baby ")):
|
||||
owners.setdefault(rest, set()).add(parent)
|
||||
return {rest: next(iter(ps)) for rest, ps in owners.items() if len(ps) == 1}
|
||||
|
||||
|
||||
def _known_leading_line(key: str) -> Optional[str]:
|
||||
""""Lizol Floor Cleaner" -> "reckitt benckiser".
|
||||
|
||||
The registry spells Reckitt's lines with an abbreviation ("rb lizol"), so
|
||||
the bare line name is neither a key nor a parent and `_known_leading_brand`
|
||||
could not see it: a product typed with a line after it built a
|
||||
brand_lizol_floor_cleaner table. Reached only after every older rule failed."""
|
||||
lines = _abbreviated_lines()
|
||||
words = key.split()
|
||||
for n in range(len(words), 0, -1):
|
||||
parent = lines.get(" ".join(words[:n]))
|
||||
if parent:
|
||||
return parent
|
||||
return None
|
||||
|
||||
|
||||
def _unique_parent_prefix(key: str) -> Optional[str]:
|
||||
""""Reckitt" -> "reckitt benckiser": the typed words are the opening words
|
||||
of exactly one parent. Two parents sharing the opening ("tata ...") give
|
||||
nothing, and a single word must be 5+ characters."""
|
||||
words = key.split()
|
||||
if not words or (len(words) == 1 and len(words[0]) < 5):
|
||||
return None
|
||||
hits = {p for p in set(BRAND_ALIASES.values())
|
||||
if p.split()[:len(words)] == words and len(p.split()) > len(words)}
|
||||
return next(iter(hits)) if len(hits) == 1 else None
|
||||
|
||||
|
||||
def _known_leading_brand(key: str) -> Optional[str]:
|
||||
"""The longest leading run of words in `key` that is itself a known brand.
|
||||
|
||||
|
||||
@@ -104,6 +104,7 @@ EXPORT_COLUMNS = (
|
||||
"barcode_lookup_status", "barcode_last_updated",
|
||||
"gst_percent", "tax_amount", "hsn_gst_needs_review",
|
||||
"highlights", "nutrients", "search_query", "field_sources",
|
||||
"product_line", "variant",
|
||||
# The validation verdict. Listed here for the same reason the barcode keys
|
||||
# are: an export that strips them turns a re-seed into a silent downgrade.
|
||||
# The upsert assigns these three plainly rather than COALESCEing them, so a
|
||||
|
||||
@@ -179,6 +179,14 @@ def get_brand_table_ddl(brand: str) -> str:
|
||||
nutrients_per_100g JSONB,
|
||||
search_query TEXT,
|
||||
|
||||
-- "Lizol Disinfectant Surface & Floor Cleaner" / "Lavender": which
|
||||
-- product line a row belongs to and which scent or flavour it is, so
|
||||
-- every pack of every variant of one product can be listed together.
|
||||
-- Fill-only-blanks in the upsert: a later re-derivation from the bare
|
||||
-- name can never overwrite the bracket-aware value web discovery set.
|
||||
product_line TEXT,
|
||||
variant TEXT,
|
||||
|
||||
-- The deterministic validation verdict from product_validator.
|
||||
-- validate_catalog() has always computed these three and annotated
|
||||
-- them onto every row; until they were added here nothing projected
|
||||
@@ -382,6 +390,8 @@ def _ensure_columns(cur, table_name: str) -> None:
|
||||
"nutrients": "TEXT[]",
|
||||
"nutrients_per_100g": "JSONB",
|
||||
"search_query": "TEXT",
|
||||
"product_line": "TEXT",
|
||||
"variant": "TEXT",
|
||||
# The product_validator verdict - see get_brand_table_ddl for why a
|
||||
# row's confidence has to survive to disk to be worth computing.
|
||||
"validation_status": "TEXT",
|
||||
@@ -452,7 +462,7 @@ def _ensure_columns(cur, table_name: str) -> None:
|
||||
# `nutrients_per_100g` joins them: it is mirrored from nutrition_facts by
|
||||
# the same sync, not written by the INSERT, and is likewise a current
|
||||
# column rather than legacy debris.
|
||||
inserted_cols = {"id", "product_name", "title", "description", "category", "image_id", "image_url", "image_urls", "price_range", "size_variants", "providers", "fssai_license", "product_sku", "sku_source", "hsn_code", "final_selling_price", "selling_price", "barcode", "barcode_type", "gtin", "ean13", "upc", "barcode_source", "barcode_verified", "barcode_lookup_status", "barcode_last_updated", "gst_percent", "tax_amount", "hsn_gst_needs_review", "highlights", "nutrients", "nutrients_per_100g", "field_sources", "search_query", "nutrition_score", "health_score", "embedding", "img_vector", "img_vector_src", "created_at", "updated_at"}
|
||||
inserted_cols = {"id", "product_name", "title", "description", "category", "image_id", "image_url", "image_urls", "price_range", "size_variants", "providers", "fssai_license", "product_sku", "sku_source", "hsn_code", "final_selling_price", "selling_price", "barcode", "barcode_type", "gtin", "ean13", "upc", "barcode_source", "barcode_verified", "barcode_lookup_status", "barcode_last_updated", "gst_percent", "tax_amount", "hsn_gst_needs_review", "highlights", "nutrients", "nutrients_per_100g", "field_sources", "search_query", "product_line", "variant", "nutrition_score", "health_score", "embedding", "img_vector", "img_vector_src", "created_at", "updated_at"}
|
||||
for col, is_nullable, col_def in col_info:
|
||||
if col not in inserted_cols and is_nullable == 'NO' and col_def is None:
|
||||
cur.execute(f"ALTER TABLE {table_name} ALTER COLUMN {col} DROP NOT NULL")
|
||||
@@ -657,6 +667,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
|
||||
|
||||
fssai_license = str(p.get("fssai_license", "")) if p.get("fssai_license") else ""
|
||||
search_query = p.get("search_query") or ""
|
||||
product_line = str(p.get("product_line") or "").strip() or None
|
||||
variant = str(p.get("variant") or "").strip() or None
|
||||
|
||||
# Convert embedding to PostgreSQL vector format
|
||||
embedding = p.get("embedding")
|
||||
@@ -703,6 +715,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
|
||||
confidence_score,
|
||||
validation_issues,
|
||||
field_sources,
|
||||
product_line,
|
||||
variant,
|
||||
embedding_str
|
||||
))
|
||||
|
||||
@@ -753,8 +767,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
|
||||
gst_percent, tax_amount, hsn_gst_needs_review,
|
||||
highlights, nutrients, search_query,
|
||||
validation_status, confidence_score, validation_issues,
|
||||
field_sources, embedding)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
|
||||
field_sources, product_line, variant, embedding)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
|
||||
ON CONFLICT (image_id) DO UPDATE SET
|
||||
product_name = EXCLUDED.product_name,
|
||||
title = EXCLUDED.title,
|
||||
@@ -848,6 +862,12 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
|
||||
-- depth here - each key's value is one flat record.
|
||||
field_sources = COALESCE({table_name}.field_sources, '{{}}'::jsonb)
|
||||
|| COALESCE(EXCLUDED.field_sources, '{{}}'::jsonb),
|
||||
-- STORED FIRST: fill-only-blanks. The pipeline derives these
|
||||
-- from the bare name on every run; the web provenance pass
|
||||
-- writes the retailer's bracketed variant ("(Original)"),
|
||||
-- which the bare name has lost. The better value must win.
|
||||
product_line = COALESCE({table_name}.product_line, EXCLUDED.product_line),
|
||||
variant = COALESCE({table_name}.variant, EXCLUDED.variant),
|
||||
updated_at = CURRENT_TIMESTAMP
|
||||
""",
|
||||
rows,
|
||||
|
||||
@@ -26,7 +26,7 @@ from collections import defaultdict
|
||||
from dataclasses import dataclass
|
||||
from typing import Dict, List, Set, Tuple
|
||||
|
||||
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
|
||||
from app.services.brand_registry import BRAND_ALIASES, is_abbreviation, resolve_parent_brand
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
@@ -34,20 +34,16 @@ class Target:
|
||||
"""One name to search for, and the words a listing must lead with to count."""
|
||||
query: str # what goes before `site:` - "Dettol", "Amul Butter"
|
||||
brand_terms: Tuple[str, ...] # "dettol"; or ("nestle", "cerelac")
|
||||
# "Lizol Floor Cleaner" typed: the listing must lead with "lizol" AND carry
|
||||
# every one of these words somewhere, so other Lizol lines drop out.
|
||||
line_terms: Tuple[str, ...] = ()
|
||||
|
||||
|
||||
def _words(text: str) -> List[str]:
|
||||
return re.findall(r"[a-z0-9]+", (text or "").lower())
|
||||
|
||||
|
||||
def _is_abbreviation(token: str, parent: str) -> bool:
|
||||
""""rb" for Reckitt Benckiser, "hul" for Hindustan Unilever, "jnj" for
|
||||
Johnson & Johnson. Short, not itself a word of the parent, and starting
|
||||
with the parent's first letter - which is what keeps "kit kat" (Nestle)
|
||||
from being read as the abbreviation "kit" plus a sub-brand "kat"."""
|
||||
parent_words = _words(parent)
|
||||
return (bool(parent_words) and 1 < len(token) <= 4 and token not in parent_words
|
||||
and token[0] == parent_words[0][0])
|
||||
_is_abbreviation = is_abbreviation
|
||||
|
||||
|
||||
def _sub_brand_part(alias: str, parent: str) -> str:
|
||||
@@ -96,6 +92,20 @@ def _family(parent: str) -> List[Target]:
|
||||
return out
|
||||
|
||||
|
||||
def _line_target(name: str, typed_key: str, family: List[Target]) -> List[Target]:
|
||||
""""Lizol Floor Cleaner" -> search that phrase, but judge listings by the
|
||||
family member's own brand words ("lizol") plus the line words. Taking the
|
||||
whole phrase as the brand would refuse every real title, which reads
|
||||
"Lizol Disinfectant Surface & Floor Cleaner"."""
|
||||
typed = typed_key.split()
|
||||
for target in sorted(family, key=lambda t: -len(_words(t.query))):
|
||||
lead = _words(target.query)
|
||||
if len(typed) > len(lead) and typed[:len(lead)] == lead:
|
||||
return [Target(query=name, brand_terms=target.brand_terms,
|
||||
line_terms=tuple(typed[len(lead):]))]
|
||||
return []
|
||||
|
||||
|
||||
def targets_for(brand: str) -> List[Target]:
|
||||
"""Sub-brands first, the brand itself last. Deterministic order.
|
||||
|
||||
@@ -112,7 +122,7 @@ def targets_for(brand: str) -> List[Target]:
|
||||
own = [t for t in family
|
||||
if " ".join(_words(t.query)) == typed_key
|
||||
or typed_key in {" ".join(_words(term)) for term in t.brand_terms[1:] or t.brand_terms}]
|
||||
return own or [Target(query=name, brand_terms=(name.lower(),))]
|
||||
return own or _line_target(name, typed_key, family) or [Target(query=name, brand_terms=(name.lower(),))]
|
||||
|
||||
display = parent.title() if parent.islower() else parent
|
||||
return family + [Target(query=display, brand_terms=(parent_key,))]
|
||||
|
||||
@@ -25,7 +25,7 @@ from dataclasses import dataclass, field
|
||||
from typing import Callable, Dict, List, Optional, Sequence
|
||||
|
||||
from app.infrastructure import settings
|
||||
from app.services.web_discovery import cache, listings, search
|
||||
from app.services.web_discovery import cache, listings, search, variants
|
||||
from app.services.web_discovery.brands import Target, targets_for
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -89,16 +89,28 @@ def _variant_words(listing: listings.Listing) -> set:
|
||||
|
||||
|
||||
def _same_product(a: listings.Listing, b: listings.Listing) -> bool:
|
||||
"""Same brand word, same pack, near-identical name, and no bracketed
|
||||
variant word on one side only. Measured: "(Citrus) 2l" and "(Floral) 2l"
|
||||
share 4 of 6 words - enough for the overlap test alone, and two products."""
|
||||
"""Same brand word, same pack, near-identical name, and no variant word on
|
||||
one side only - bracketed, or a bare scent/flavour from
|
||||
variants.VARIANT_TERMS. Measured: "(Citrus) 2l" and "(Floral) 2l" share 4
|
||||
of 6 words, and "Cleaner Floral 500 ml" / "Cleaner Pine 500 ml" share 3 of
|
||||
5 (exactly the 0.6 bar) - enough for the overlap test alone, and two
|
||||
products each time."""
|
||||
if a.brand_term != b.brand_term or a.size != b.size:
|
||||
return False
|
||||
ident_a, ident_b = _identity(a), _identity(b)
|
||||
if _jaccard(ident_a, ident_b) < _SAME_NAME_JACCARD:
|
||||
return False
|
||||
differing = set(ident_a) ^ set(ident_b)
|
||||
return not (differing & (_variant_words(a) | _variant_words(b)))
|
||||
return not (differing & (_variant_words(a) | _variant_words(b) | variants.VARIANT_TERMS))
|
||||
|
||||
|
||||
# "Lizol Floor Cleaner" typed: a Lizol listing of another line (toilet cleaner,
|
||||
# wipes) is a real product, just not the one asked for.
|
||||
OTHER_LINE = "other_product_line"
|
||||
|
||||
|
||||
def _on_line(listing: listings.Listing, target: Target) -> bool:
|
||||
return set(target.line_terms) <= set(listing.tokens)
|
||||
|
||||
|
||||
def _jaccard(a: Sequence[str], b: Sequence[str]) -> float:
|
||||
@@ -145,6 +157,7 @@ def cluster(found: Sequence[listings.Listing]) -> List[Dict]:
|
||||
# normaliser (and off_bulk.strip_sizes) drops bracketed text, which
|
||||
# turned "Lizol Floor Cleaner (Citrus)" and "(Floral)" into one name.
|
||||
plain = re.sub(r"\s+", " ", re.sub(r"[()]", " ", name)).strip()
|
||||
line, variant = variants.split_variant(name)
|
||||
candidates.append({
|
||||
"title": f"{plain} {size}",
|
||||
"size": size,
|
||||
@@ -154,6 +167,8 @@ def cluster(found: Sequence[listings.Listing]) -> List[Dict]:
|
||||
"price_range": _price_range(prices),
|
||||
"listings": [m.as_dict() for m in members[:_MAX_LISTINGS_PER_CANDIDATE]],
|
||||
"brand_term": brand_term,
|
||||
"product_line": line,
|
||||
"variant": variant,
|
||||
"source": "web",
|
||||
})
|
||||
candidates.sort(key=lambda c: (-c["retailer_count"], c["title"].lower()))
|
||||
@@ -205,6 +220,8 @@ def run(brand: str, *, progress: Optional[Callable[[RunResult], None]] = None,
|
||||
query.target.brand_terms)
|
||||
if listing is None:
|
||||
rejected[reason] += 1
|
||||
elif not _on_line(listing, query.target):
|
||||
rejected[OTHER_LINE] += 1
|
||||
else:
|
||||
kept.append(listing)
|
||||
result.queries_done += 1
|
||||
|
||||
@@ -277,6 +277,14 @@ def parse_hit(title: str, url: str, snippet: str,
|
||||
if size is None:
|
||||
# Some retailers put the size only in the title's tail ("... - 125 g | Zepto").
|
||||
size, distinct = extract_size(title)
|
||||
if size is None and not _MULTIPACK.search(snippet or ""):
|
||||
# Blinkit often prints the pack only in the snippet ("... 500 ml ...
|
||||
# Price"): 63 of 80 real Reckitt listings were lost for want of it.
|
||||
# Taken only when the snippet names exactly ONE size - a snippet that
|
||||
# lists the other packs ("also in 1 L, 2 L") says nothing about this one.
|
||||
size, distinct = extract_size(snippet)
|
||||
if distinct > 1:
|
||||
return None, NO_PACK_SIZE
|
||||
if size is None:
|
||||
return None, NO_PACK_SIZE
|
||||
if distinct > 1:
|
||||
|
||||
@@ -10,7 +10,10 @@ web-found product it
|
||||
* fills `price_range` from the listing prices when the pipeline left it blank;
|
||||
* sets `validation_status = 'needs_review'` when only ONE retailer listed it
|
||||
(never lifting a `rejected` verdict). Two or more retailers keep whatever the
|
||||
validator decided.
|
||||
validator decided;
|
||||
* sets `product_line` / `variant` to what the listing titles said. These
|
||||
OVERWRITE the pipeline's value: the pipeline read them off the bare name,
|
||||
which has lost the retailer's brackets ("Harpic ... (Original)").
|
||||
|
||||
The join is exact: stage 11 reports every stored row's `image_id` with the
|
||||
1-based sheet row it came from (header = row 1), and the ingest wrote the web
|
||||
@@ -77,10 +80,13 @@ def apply(brand: str, updates: List[Dict[str, Any]]) -> int:
|
||||
f"price_range = CASE WHEN COALESCE(price_range, '') = '' AND %s <> '' "
|
||||
f"THEN %s ELSE price_range END, "
|
||||
f"validation_status = CASE WHEN %s AND COALESCE(validation_status, '') <> 'rejected' "
|
||||
f"THEN 'needs_review' ELSE validation_status END "
|
||||
f"THEN 'needs_review' ELSE validation_status END, "
|
||||
f"product_line = COALESCE(NULLIF(%s, ''), product_line), "
|
||||
f"variant = COALESCE(NULLIF(%s, ''), variant) "
|
||||
f"WHERE image_id = %s",
|
||||
(Json(provenance), entry.get("price_range") or "", entry.get("price_range") or "",
|
||||
single, update["image_id"]),
|
||||
single, entry.get("product_line") or "", entry.get("variant") or "",
|
||||
update["image_id"]),
|
||||
)
|
||||
written += cur.rowcount
|
||||
except Exception as exc: # noqa: BLE001 - provenance must not take the batch down
|
||||
|
||||
162
app/services/web_discovery/variants.py
Normal file
162
app/services/web_discovery/variants.py
Normal file
@@ -0,0 +1,162 @@
|
||||
"""A product name -> (product line, variant), and candidates -> line -> variant -> sizes.
|
||||
|
||||
"Lizol Disinfectant Surface & Floor Cleaner (Lavender)" is the line
|
||||
"Lizol Disinfectant Surface & Floor Cleaner" in the variant "Lavender". Retail
|
||||
titles mark the variant two ways:
|
||||
|
||||
* in brackets - Blinkit's "(Citrus, 625 ml)", "(Original)". Whatever a
|
||||
bracket holds once the size is gone IS the variant, whatever the word.
|
||||
* as a bare word - "Lizol ... Cleaner Floral 500 ml". Only a word in
|
||||
VARIANT_TERMS is read as a variant here; any other bare word stays in the
|
||||
line, because guessing would split one product into two lines.
|
||||
|
||||
VARIANT_TERMS is scents and flavours only. Generic words ("original", "fresh",
|
||||
"chocolate", "classic") are left out on purpose: "Dettol Original Soap" and
|
||||
"Cadbury Dairy Milk Chocolate" name a line, not a variant. A bracket still
|
||||
catches "(Original)". Add a word only when a real listing needs it.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from collections import Counter, defaultdict
|
||||
from typing import Dict, Iterable, List, Tuple
|
||||
|
||||
VARIANT_TERMS = frozenset({
|
||||
# scents - household and personal care
|
||||
"citrus", "floral", "pine", "lavender", "jasmine", "rose", "lemon", "lime",
|
||||
"orange", "sandal", "sandalwood", "neem", "mint", "eucalyptus", "aloe",
|
||||
"lily", "marine", "ocean", "mogra", "tulsi", "vanilla", "berry", "musk",
|
||||
# flavours that name a pack rather than a product
|
||||
"strawberry", "mango", "elaichi", "masala", "pineapple", "litchi", "guava",
|
||||
})
|
||||
|
||||
# A word that belongs to the variant when it sits next to one: "Lemon Fresh",
|
||||
# "Ocean Breeze", "Lavender Blossom". Never a variant on its own.
|
||||
_VARIANT_MODIFIERS = frozenset({"fresh", "blossom", "breeze", "burst", "bloom", "splash", "twist"})
|
||||
|
||||
_PACKAGING_WORDS = frozenset({"bar", "pouch", "bottle", "jar", "box", "tin", "carton", "pack", "refill"})
|
||||
_BRACKET = re.compile(r"\(([^()]*)\)")
|
||||
_WORD = re.compile(r"[A-Za-z0-9]+(?:'[A-Za-z]+)?")
|
||||
|
||||
|
||||
def _words(text: str) -> List[str]:
|
||||
return re.findall(r"[a-z0-9]+", (text or "").lower())
|
||||
|
||||
|
||||
def _tidy(text: str) -> str:
|
||||
return re.sub(r"\s+", " ", text or "").strip(" ,;:.-/&+–—")
|
||||
|
||||
|
||||
def _title(word: str) -> str:
|
||||
return word if any(c.isupper() for c in word) else word.capitalize()
|
||||
|
||||
|
||||
def variant_words(text: str) -> frozenset:
|
||||
"""The VARIANT_TERMS words in `text` - for telling two names apart."""
|
||||
return frozenset(w for w in _words(text) if w in VARIANT_TERMS)
|
||||
|
||||
|
||||
def split_variant(name: str) -> Tuple[str, str]:
|
||||
"""(line, variant). A pack size in the name is ignored ("... Floral 500 ml").
|
||||
|
||||
("Lizol Floor Cleaner (Citrus)") -> ("Lizol Floor Cleaner", "Citrus")
|
||||
("Lizol Floor Cleaner Floral") -> ("Lizol Floor Cleaner", "Floral")
|
||||
("Harpic Toilet Cleaner (Original)") -> ("Harpic Toilet Cleaner", "Original")
|
||||
("Dettol Original Soap") -> ("Dettol Original Soap", "")
|
||||
"""
|
||||
from app.services.web_discovery.listings import _strip_sizes
|
||||
|
||||
name = _tidy(re.sub(r"\s+", " ", _strip_sizes(name or "")))
|
||||
groups = [_tidy(g) for g in _BRACKET.findall(name)]
|
||||
groups = [g for g in groups if re.search(r"[A-Za-z]", g)]
|
||||
if groups:
|
||||
line = _tidy(re.sub(r"\s+", " ", _BRACKET.sub(" ", name)))
|
||||
return line, " ".join(_title(w) for g in groups for w in _WORD.findall(g))
|
||||
|
||||
tokens = _WORD.findall(name)
|
||||
lowered = [t.lower() for t in tokens]
|
||||
keep = [l in VARIANT_TERMS for l in lowered]
|
||||
if not any(keep) or all(keep):
|
||||
return name, ""
|
||||
# A modifier joins the variant only when it touches a variant word.
|
||||
for i, l in enumerate(lowered):
|
||||
if l in _VARIANT_MODIFIERS and ((i > 0 and keep[i - 1]) or (i + 1 < len(keep) and keep[i + 1])):
|
||||
keep[i] = True
|
||||
variant = " ".join(_title(t) for t, k in zip(tokens, keep) if k)
|
||||
line = name
|
||||
for t, k in zip(tokens, keep):
|
||||
if k:
|
||||
line = re.sub(r"(?<![A-Za-z0-9])" + re.escape(t) + r"(?![A-Za-z0-9])", " ", line, count=1)
|
||||
line = _tidy(re.sub(r"\s*&\s*$", "", re.sub(r"\s+", " ", line)))
|
||||
return (line or name), variant
|
||||
|
||||
|
||||
def line_key(line: str, brand_term: str = "") -> Tuple[str, ...]:
|
||||
"""Lines are one line when these words match - order, '&' and packaging aside."""
|
||||
brand = set(_words(brand_term))
|
||||
return tuple(sorted({w for w in _words(line) if w not in _PACKAGING_WORDS} - brand)) or tuple(_words(line))
|
||||
|
||||
|
||||
_SIZE = re.compile(r"^(\d+(?:\.\d+)?)(kg|g|ml|l|pcs)$")
|
||||
_SIZE_SCALE = {"g": (0, 1), "kg": (0, 1000), "ml": (1, 1), "l": (1, 1000), "pcs": (2, 1)}
|
||||
|
||||
|
||||
def size_sort_key(size: str) -> Tuple[int, float, str]:
|
||||
"""Mass, then volume, then counts, each smallest first: 500ml < 625ml < 1l."""
|
||||
m = _SIZE.match((size or "").lower())
|
||||
if not m:
|
||||
return (9, 0.0, size or "")
|
||||
group, scale = _SIZE_SCALE[m.group(2)]
|
||||
return (group, float(m.group(1)) * scale, size)
|
||||
|
||||
|
||||
def group_families(candidates: Iterable[Dict]) -> List[Dict]:
|
||||
"""Candidates (one per name and pack) -> product line -> variant -> sizes.
|
||||
|
||||
Each candidate needs `product_line`, `variant` and `size`; `retailer_count`,
|
||||
`price_range`, `providers` and `listings` are carried through when present,
|
||||
and a `row_index` is collected into the size's `rows`.
|
||||
Lines come out best-corroborated first; variants and sizes in a stable order.
|
||||
"""
|
||||
lines: Dict[Tuple, List[Dict]] = defaultdict(list)
|
||||
for c in candidates:
|
||||
line = c.get("product_line") or ""
|
||||
if not line:
|
||||
continue
|
||||
lines[(c.get("brand_term") or "", line_key(line, c.get("brand_term") or ""))].append(c)
|
||||
|
||||
families: List[Dict] = []
|
||||
for members in lines.values():
|
||||
spellings = Counter(m["product_line"] for m in members)
|
||||
display = sorted(spellings, key=lambda s: (-spellings[s], len(s), s))[0]
|
||||
by_variant: Dict[str, List[Dict]] = defaultdict(list)
|
||||
for m in members:
|
||||
by_variant[m.get("variant") or ""].append(m)
|
||||
variants = []
|
||||
for variant, rows in sorted(by_variant.items(), key=lambda kv: (kv[0] == "", kv[0].lower())):
|
||||
sizes: Dict[str, Dict] = {}
|
||||
for r in rows:
|
||||
size = r.get("size") or ((r.get("sizes") or [None])[0])
|
||||
if not size:
|
||||
continue
|
||||
entry = sizes.setdefault(size, {"size": size, "retailer_count": 0, "price_range": "",
|
||||
"providers": [], "listings": [], "rows": []})
|
||||
if r.get("row_index") is not None:
|
||||
entry["rows"].append(r["row_index"])
|
||||
entry["retailer_count"] = max(entry["retailer_count"], int(r.get("retailer_count") or 0))
|
||||
entry["price_range"] = entry["price_range"] or r.get("price_range") or ""
|
||||
for p in r.get("providers") or []:
|
||||
if p not in entry["providers"]:
|
||||
entry["providers"].append(p)
|
||||
entry["listings"].extend(r.get("listings") or [])
|
||||
variants.append({"variant": variant,
|
||||
"sizes": sorted(sizes.values(), key=lambda s: size_sort_key(s["size"]))})
|
||||
families.append({
|
||||
"product_line": display,
|
||||
"brand_term": members[0].get("brand_term") or "",
|
||||
"variants": variants,
|
||||
"retailer_count": max((int(m.get("retailer_count") or 0) for m in members), default=0),
|
||||
})
|
||||
families.sort(key=lambda f: (-f["retailer_count"], -len(f["variants"]), f["product_line"].lower()))
|
||||
return families
|
||||
|
||||
Reference in New Issue
Block a user