Product discovery on names
This commit is contained in:
@@ -105,6 +105,11 @@ class DiscoveredProductIn(BaseModel):
|
||||
listings: List[Dict[str, Any]] = Field(default_factory=list)
|
||||
price_range: Optional[str] = None
|
||||
retailer_count: int = 0
|
||||
# The line and variant the preview grouped this row under. Web rows only
|
||||
# need them sent back: their variant came from the retailer's brackets,
|
||||
# which the plain product name no longer carries.
|
||||
product_line: Optional[str] = None
|
||||
variant: Optional[str] = None
|
||||
|
||||
|
||||
class WebJobRequest(BaseModel):
|
||||
@@ -245,7 +250,8 @@ async def ingest_brand_discovery(payload: DiscoveryIngestRequest) -> batch_commo
|
||||
# them. Keyed by CSV position, which is how stage 11 reports source rows.
|
||||
web_entries = {
|
||||
index: {"listings": item.listings, "price_range": item.price_range or "",
|
||||
"retailer_count": item.retailer_count}
|
||||
"retailer_count": item.retailer_count,
|
||||
"product_line": item.product_line or "", "variant": item.variant or ""}
|
||||
for index, item in enumerate(payload.products) if item.listings
|
||||
}
|
||||
if web_entries:
|
||||
|
||||
@@ -1003,9 +1003,21 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"confidence_score": row.get("confidence_score"),
|
||||
"validation_issues": list(row.get("validation_issues") or []),
|
||||
"field_sources": dict(row.get("field_sources") or {}),
|
||||
# Line and variant read off the size-free name. Web discovery's
|
||||
# provenance pass may set a better variant later; the upsert keeps
|
||||
# whichever arrived first, so this never overwrites it.
|
||||
**_line_and_variant(row, name),
|
||||
}
|
||||
|
||||
|
||||
def _line_and_variant(row: Dict[str, Any], name: str) -> Dict[str, Optional[str]]:
|
||||
from app.services.web_discovery.variants import split_variant
|
||||
|
||||
line, variant = split_variant(name or "")
|
||||
return {"product_line": row.get("product_line") or line or None,
|
||||
"variant": row.get("variant") or variant or None}
|
||||
|
||||
|
||||
def _record_product(result: PipelineResult, brand: str, row: Dict[str, Any],
|
||||
disposition: str,
|
||||
source_rows: Optional[Dict[str, int]] = None) -> None:
|
||||
|
||||
@@ -233,6 +233,10 @@ class DiscoveredProduct:
|
||||
listings: List[Dict[str, Any]] = field(default_factory=list)
|
||||
price_range: str = ""
|
||||
retailer_count: int = 0
|
||||
# "Lizol Disinfectant Surface & Floor Cleaner" / "Lavender". Shown grouped
|
||||
# in the preview and sent back on ingest; stored by the pipeline/provenance.
|
||||
product_line: str = ""
|
||||
variant: str = ""
|
||||
|
||||
def as_preview(self) -> Dict[str, Any]:
|
||||
body = self._preview_body()
|
||||
@@ -263,6 +267,8 @@ class DiscoveredProduct:
|
||||
"confidence": round(self.confidence, 2),
|
||||
"matches_existing": self.matches_existing,
|
||||
"notes": list(self.notes),
|
||||
"product_line": self.product_line,
|
||||
"variant": self.variant,
|
||||
"selected": self.confidence >= 0.5,
|
||||
}
|
||||
|
||||
@@ -286,6 +292,7 @@ class DiscoveryResult:
|
||||
"brand_active": self.brand_active,
|
||||
"filtering_enabled": self.filtering_enabled,
|
||||
"products": [p.as_preview() for p in self.products],
|
||||
"families": families_of(self.products),
|
||||
"counts": dict(self.counts),
|
||||
"warnings": list(self.warnings),
|
||||
}
|
||||
@@ -370,6 +377,25 @@ def _normalise_title(brand: str, title: Optional[str]) -> str:
|
||||
return off_bulk.normalize_for_match(title, off_bulk.brand_tokens(brand))
|
||||
|
||||
|
||||
def families_of(products: Sequence[DiscoveredProduct]) -> List[Dict[str, Any]]:
|
||||
"""The preview grouped as product line -> variant -> pack sizes.
|
||||
|
||||
Each size carries `rows`, the indexes into `products` it came from, so the
|
||||
panel can tick and untick the underlying rows from the grouped view."""
|
||||
from app.services.web_discovery.variants import group_families
|
||||
|
||||
candidates = []
|
||||
for index, p in enumerate(products):
|
||||
for size in p.size_variants or [""]:
|
||||
candidates.append({
|
||||
"product_line": p.product_line, "variant": p.variant, "size": size,
|
||||
"retailer_count": p.retailer_count, "price_range": p.price_range,
|
||||
"providers": list(p.providers[:p.retailer_count] if p.retailer_count else []),
|
||||
"row_index": index,
|
||||
})
|
||||
return group_families(candidates)
|
||||
|
||||
|
||||
def _same_product(left: str, right: str, left_raw: str, right_raw: str) -> bool:
|
||||
"""Whether two normalised titles name the same retail product.
|
||||
|
||||
@@ -392,9 +418,19 @@ def _same_product(left: str, right: str, left_raw: str, right_raw: str) -> bool:
|
||||
return True
|
||||
if off_bulk.has_extra_variant_conflict(left_raw, right_raw):
|
||||
return False
|
||||
# Citrus vs Floral scores 0.857 - over the floor. Two names that differ in
|
||||
# a scent or flavour are two packs on the shelf, never one product.
|
||||
if _scents(left_raw) != _scents(right_raw):
|
||||
return False
|
||||
return off_bulk.symmetric_similarity(left, right) >= _TITLE_MATCH_THRESHOLD
|
||||
|
||||
|
||||
def _scents(title: str) -> frozenset:
|
||||
from app.services.web_discovery.variants import variant_words
|
||||
|
||||
return variant_words(title)
|
||||
|
||||
|
||||
def _match_existing(normalised: str, title: str,
|
||||
catalog_keys: Dict[str, str]) -> Optional[str]:
|
||||
"""The stored base name for this product, or None.
|
||||
@@ -1196,6 +1232,7 @@ def discover_brand_products(
|
||||
continue
|
||||
if "web" in product.sources:
|
||||
_apply_web(product, candidate)
|
||||
_apply_line(product, candidate)
|
||||
if (require_evidence and "off" not in product.sources
|
||||
and "store" not in product.sources and not product.evidence):
|
||||
dropped += 1
|
||||
@@ -1259,6 +1296,20 @@ def _from_web(brand: str, job_id: Optional[str], warnings: List[str]) -> List[Di
|
||||
return [dict(c) for c in job.candidates]
|
||||
|
||||
|
||||
def _apply_line(product: DiscoveredProduct, candidate: Dict[str, Any]) -> None:
|
||||
"""Product line and variant. A web candidate brings its own, read from the
|
||||
retailer's brackets - kept while its variant words still appear in the
|
||||
final name (a stored-name rename could have changed it). Everything else is
|
||||
split from the final product name."""
|
||||
from app.services.web_discovery.variants import split_variant
|
||||
|
||||
line, variant = candidate.get("product_line") or "", candidate.get("variant") or ""
|
||||
name_words = set(re.findall(r"[a-z0-9]+", product.product_name.lower()))
|
||||
if not (line and set(re.findall(r"[a-z0-9]+", variant.lower())) <= name_words):
|
||||
line, variant = split_variant(product.product_name)
|
||||
product.product_line, product.variant = line, variant
|
||||
|
||||
|
||||
def _apply_web(product: DiscoveredProduct, candidate: Dict[str, Any]) -> None:
|
||||
product.listings = list(candidate.get("listings") or [])
|
||||
product.price_range = candidate.get("price_range") or ""
|
||||
|
||||
@@ -367,9 +367,77 @@ def resolve_parent_brand(brand: str) -> str:
|
||||
prefix = _known_leading_brand(key)
|
||||
if prefix:
|
||||
return resolve_parent_brand(prefix)
|
||||
line = _known_leading_line(key)
|
||||
if line:
|
||||
return line
|
||||
parent = _unique_parent_prefix(key)
|
||||
if parent:
|
||||
return parent
|
||||
return brand
|
||||
|
||||
|
||||
def is_abbreviation(token: str, parent: str) -> bool:
|
||||
""""rb" for Reckitt Benckiser, "hul" for Hindustan Unilever, "jnj" for
|
||||
Johnson & Johnson. Short, not itself a word of the parent, and starting
|
||||
with the parent's first letter - which is what keeps "kit kat" (Nestle)
|
||||
from being read as the abbreviation "kit" plus a sub-brand "kat"."""
|
||||
parent_words = re.findall(r"[a-z0-9]+", (parent or "").lower())
|
||||
return (bool(parent_words) and 1 < len(token) <= 4 and token not in parent_words
|
||||
and token[0] == parent_words[0][0])
|
||||
|
||||
|
||||
# Line names that are also everyday words or another maker's name. Read as a
|
||||
# leading word they would misfile: "Laxmi Chilli Powder" is a spice brand, not
|
||||
# HUL; "Boost Energy Drink", "Finish Line", "Baby Oil" (any maker) likewise.
|
||||
_AMBIGUOUS_LINES = frozenset({"always", "boost", "finish", "wheel", "fairy", "ivory",
|
||||
"vanish", "laxmi", "whisper"})
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def _abbreviated_lines() -> dict:
|
||||
""""lizol" -> "reckitt benckiser", read off the alias "rb lizol".
|
||||
|
||||
Only line names of 5+ characters that exactly one parent registers: the
|
||||
bare "sun" off "hul sun" would otherwise send "Sun Pharma" to Unilever."""
|
||||
owners: dict = {}
|
||||
for alias, parent in BRAND_ALIASES.items():
|
||||
words = alias.split()
|
||||
if len(words) > 1 and is_abbreviation(words[0], parent):
|
||||
rest = " ".join(words[1:])
|
||||
if (len(rest.replace(" ", "")) >= 5 and rest not in _AMBIGUOUS_LINES
|
||||
and not rest.startswith("baby ")):
|
||||
owners.setdefault(rest, set()).add(parent)
|
||||
return {rest: next(iter(ps)) for rest, ps in owners.items() if len(ps) == 1}
|
||||
|
||||
|
||||
def _known_leading_line(key: str) -> Optional[str]:
|
||||
""""Lizol Floor Cleaner" -> "reckitt benckiser".
|
||||
|
||||
The registry spells Reckitt's lines with an abbreviation ("rb lizol"), so
|
||||
the bare line name is neither a key nor a parent and `_known_leading_brand`
|
||||
could not see it: a product typed with a line after it built a
|
||||
brand_lizol_floor_cleaner table. Reached only after every older rule failed."""
|
||||
lines = _abbreviated_lines()
|
||||
words = key.split()
|
||||
for n in range(len(words), 0, -1):
|
||||
parent = lines.get(" ".join(words[:n]))
|
||||
if parent:
|
||||
return parent
|
||||
return None
|
||||
|
||||
|
||||
def _unique_parent_prefix(key: str) -> Optional[str]:
|
||||
""""Reckitt" -> "reckitt benckiser": the typed words are the opening words
|
||||
of exactly one parent. Two parents sharing the opening ("tata ...") give
|
||||
nothing, and a single word must be 5+ characters."""
|
||||
words = key.split()
|
||||
if not words or (len(words) == 1 and len(words[0]) < 5):
|
||||
return None
|
||||
hits = {p for p in set(BRAND_ALIASES.values())
|
||||
if p.split()[:len(words)] == words and len(p.split()) > len(words)}
|
||||
return next(iter(hits)) if len(hits) == 1 else None
|
||||
|
||||
|
||||
def _known_leading_brand(key: str) -> Optional[str]:
|
||||
"""The longest leading run of words in `key` that is itself a known brand.
|
||||
|
||||
|
||||
@@ -104,6 +104,7 @@ EXPORT_COLUMNS = (
|
||||
"barcode_lookup_status", "barcode_last_updated",
|
||||
"gst_percent", "tax_amount", "hsn_gst_needs_review",
|
||||
"highlights", "nutrients", "search_query", "field_sources",
|
||||
"product_line", "variant",
|
||||
# The validation verdict. Listed here for the same reason the barcode keys
|
||||
# are: an export that strips them turns a re-seed into a silent downgrade.
|
||||
# The upsert assigns these three plainly rather than COALESCEing them, so a
|
||||
|
||||
@@ -179,6 +179,14 @@ def get_brand_table_ddl(brand: str) -> str:
|
||||
nutrients_per_100g JSONB,
|
||||
search_query TEXT,
|
||||
|
||||
-- "Lizol Disinfectant Surface & Floor Cleaner" / "Lavender": which
|
||||
-- product line a row belongs to and which scent or flavour it is, so
|
||||
-- every pack of every variant of one product can be listed together.
|
||||
-- Fill-only-blanks in the upsert: a later re-derivation from the bare
|
||||
-- name can never overwrite the bracket-aware value web discovery set.
|
||||
product_line TEXT,
|
||||
variant TEXT,
|
||||
|
||||
-- The deterministic validation verdict from product_validator.
|
||||
-- validate_catalog() has always computed these three and annotated
|
||||
-- them onto every row; until they were added here nothing projected
|
||||
@@ -382,6 +390,8 @@ def _ensure_columns(cur, table_name: str) -> None:
|
||||
"nutrients": "TEXT[]",
|
||||
"nutrients_per_100g": "JSONB",
|
||||
"search_query": "TEXT",
|
||||
"product_line": "TEXT",
|
||||
"variant": "TEXT",
|
||||
# The product_validator verdict - see get_brand_table_ddl for why a
|
||||
# row's confidence has to survive to disk to be worth computing.
|
||||
"validation_status": "TEXT",
|
||||
@@ -452,7 +462,7 @@ def _ensure_columns(cur, table_name: str) -> None:
|
||||
# `nutrients_per_100g` joins them: it is mirrored from nutrition_facts by
|
||||
# the same sync, not written by the INSERT, and is likewise a current
|
||||
# column rather than legacy debris.
|
||||
inserted_cols = {"id", "product_name", "title", "description", "category", "image_id", "image_url", "image_urls", "price_range", "size_variants", "providers", "fssai_license", "product_sku", "sku_source", "hsn_code", "final_selling_price", "selling_price", "barcode", "barcode_type", "gtin", "ean13", "upc", "barcode_source", "barcode_verified", "barcode_lookup_status", "barcode_last_updated", "gst_percent", "tax_amount", "hsn_gst_needs_review", "highlights", "nutrients", "nutrients_per_100g", "field_sources", "search_query", "nutrition_score", "health_score", "embedding", "img_vector", "img_vector_src", "created_at", "updated_at"}
|
||||
inserted_cols = {"id", "product_name", "title", "description", "category", "image_id", "image_url", "image_urls", "price_range", "size_variants", "providers", "fssai_license", "product_sku", "sku_source", "hsn_code", "final_selling_price", "selling_price", "barcode", "barcode_type", "gtin", "ean13", "upc", "barcode_source", "barcode_verified", "barcode_lookup_status", "barcode_last_updated", "gst_percent", "tax_amount", "hsn_gst_needs_review", "highlights", "nutrients", "nutrients_per_100g", "field_sources", "search_query", "product_line", "variant", "nutrition_score", "health_score", "embedding", "img_vector", "img_vector_src", "created_at", "updated_at"}
|
||||
for col, is_nullable, col_def in col_info:
|
||||
if col not in inserted_cols and is_nullable == 'NO' and col_def is None:
|
||||
cur.execute(f"ALTER TABLE {table_name} ALTER COLUMN {col} DROP NOT NULL")
|
||||
@@ -657,6 +667,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
|
||||
|
||||
fssai_license = str(p.get("fssai_license", "")) if p.get("fssai_license") else ""
|
||||
search_query = p.get("search_query") or ""
|
||||
product_line = str(p.get("product_line") or "").strip() or None
|
||||
variant = str(p.get("variant") or "").strip() or None
|
||||
|
||||
# Convert embedding to PostgreSQL vector format
|
||||
embedding = p.get("embedding")
|
||||
@@ -703,6 +715,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
|
||||
confidence_score,
|
||||
validation_issues,
|
||||
field_sources,
|
||||
product_line,
|
||||
variant,
|
||||
embedding_str
|
||||
))
|
||||
|
||||
@@ -753,8 +767,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
|
||||
gst_percent, tax_amount, hsn_gst_needs_review,
|
||||
highlights, nutrients, search_query,
|
||||
validation_status, confidence_score, validation_issues,
|
||||
field_sources, embedding)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
|
||||
field_sources, product_line, variant, embedding)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
|
||||
ON CONFLICT (image_id) DO UPDATE SET
|
||||
product_name = EXCLUDED.product_name,
|
||||
title = EXCLUDED.title,
|
||||
@@ -848,6 +862,12 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
|
||||
-- depth here - each key's value is one flat record.
|
||||
field_sources = COALESCE({table_name}.field_sources, '{{}}'::jsonb)
|
||||
|| COALESCE(EXCLUDED.field_sources, '{{}}'::jsonb),
|
||||
-- STORED FIRST: fill-only-blanks. The pipeline derives these
|
||||
-- from the bare name on every run; the web provenance pass
|
||||
-- writes the retailer's bracketed variant ("(Original)"),
|
||||
-- which the bare name has lost. The better value must win.
|
||||
product_line = COALESCE({table_name}.product_line, EXCLUDED.product_line),
|
||||
variant = COALESCE({table_name}.variant, EXCLUDED.variant),
|
||||
updated_at = CURRENT_TIMESTAMP
|
||||
""",
|
||||
rows,
|
||||
|
||||
@@ -26,7 +26,7 @@ from collections import defaultdict
|
||||
from dataclasses import dataclass
|
||||
from typing import Dict, List, Set, Tuple
|
||||
|
||||
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
|
||||
from app.services.brand_registry import BRAND_ALIASES, is_abbreviation, resolve_parent_brand
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
@@ -34,20 +34,16 @@ class Target:
|
||||
"""One name to search for, and the words a listing must lead with to count."""
|
||||
query: str # what goes before `site:` - "Dettol", "Amul Butter"
|
||||
brand_terms: Tuple[str, ...] # "dettol"; or ("nestle", "cerelac")
|
||||
# "Lizol Floor Cleaner" typed: the listing must lead with "lizol" AND carry
|
||||
# every one of these words somewhere, so other Lizol lines drop out.
|
||||
line_terms: Tuple[str, ...] = ()
|
||||
|
||||
|
||||
def _words(text: str) -> List[str]:
|
||||
return re.findall(r"[a-z0-9]+", (text or "").lower())
|
||||
|
||||
|
||||
def _is_abbreviation(token: str, parent: str) -> bool:
|
||||
""""rb" for Reckitt Benckiser, "hul" for Hindustan Unilever, "jnj" for
|
||||
Johnson & Johnson. Short, not itself a word of the parent, and starting
|
||||
with the parent's first letter - which is what keeps "kit kat" (Nestle)
|
||||
from being read as the abbreviation "kit" plus a sub-brand "kat"."""
|
||||
parent_words = _words(parent)
|
||||
return (bool(parent_words) and 1 < len(token) <= 4 and token not in parent_words
|
||||
and token[0] == parent_words[0][0])
|
||||
_is_abbreviation = is_abbreviation
|
||||
|
||||
|
||||
def _sub_brand_part(alias: str, parent: str) -> str:
|
||||
@@ -96,6 +92,20 @@ def _family(parent: str) -> List[Target]:
|
||||
return out
|
||||
|
||||
|
||||
def _line_target(name: str, typed_key: str, family: List[Target]) -> List[Target]:
|
||||
""""Lizol Floor Cleaner" -> search that phrase, but judge listings by the
|
||||
family member's own brand words ("lizol") plus the line words. Taking the
|
||||
whole phrase as the brand would refuse every real title, which reads
|
||||
"Lizol Disinfectant Surface & Floor Cleaner"."""
|
||||
typed = typed_key.split()
|
||||
for target in sorted(family, key=lambda t: -len(_words(t.query))):
|
||||
lead = _words(target.query)
|
||||
if len(typed) > len(lead) and typed[:len(lead)] == lead:
|
||||
return [Target(query=name, brand_terms=target.brand_terms,
|
||||
line_terms=tuple(typed[len(lead):]))]
|
||||
return []
|
||||
|
||||
|
||||
def targets_for(brand: str) -> List[Target]:
|
||||
"""Sub-brands first, the brand itself last. Deterministic order.
|
||||
|
||||
@@ -112,7 +122,7 @@ def targets_for(brand: str) -> List[Target]:
|
||||
own = [t for t in family
|
||||
if " ".join(_words(t.query)) == typed_key
|
||||
or typed_key in {" ".join(_words(term)) for term in t.brand_terms[1:] or t.brand_terms}]
|
||||
return own or [Target(query=name, brand_terms=(name.lower(),))]
|
||||
return own or _line_target(name, typed_key, family) or [Target(query=name, brand_terms=(name.lower(),))]
|
||||
|
||||
display = parent.title() if parent.islower() else parent
|
||||
return family + [Target(query=display, brand_terms=(parent_key,))]
|
||||
|
||||
@@ -25,7 +25,7 @@ from dataclasses import dataclass, field
|
||||
from typing import Callable, Dict, List, Optional, Sequence
|
||||
|
||||
from app.infrastructure import settings
|
||||
from app.services.web_discovery import cache, listings, search
|
||||
from app.services.web_discovery import cache, listings, search, variants
|
||||
from app.services.web_discovery.brands import Target, targets_for
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -89,16 +89,28 @@ def _variant_words(listing: listings.Listing) -> set:
|
||||
|
||||
|
||||
def _same_product(a: listings.Listing, b: listings.Listing) -> bool:
|
||||
"""Same brand word, same pack, near-identical name, and no bracketed
|
||||
variant word on one side only. Measured: "(Citrus) 2l" and "(Floral) 2l"
|
||||
share 4 of 6 words - enough for the overlap test alone, and two products."""
|
||||
"""Same brand word, same pack, near-identical name, and no variant word on
|
||||
one side only - bracketed, or a bare scent/flavour from
|
||||
variants.VARIANT_TERMS. Measured: "(Citrus) 2l" and "(Floral) 2l" share 4
|
||||
of 6 words, and "Cleaner Floral 500 ml" / "Cleaner Pine 500 ml" share 3 of
|
||||
5 (exactly the 0.6 bar) - enough for the overlap test alone, and two
|
||||
products each time."""
|
||||
if a.brand_term != b.brand_term or a.size != b.size:
|
||||
return False
|
||||
ident_a, ident_b = _identity(a), _identity(b)
|
||||
if _jaccard(ident_a, ident_b) < _SAME_NAME_JACCARD:
|
||||
return False
|
||||
differing = set(ident_a) ^ set(ident_b)
|
||||
return not (differing & (_variant_words(a) | _variant_words(b)))
|
||||
return not (differing & (_variant_words(a) | _variant_words(b) | variants.VARIANT_TERMS))
|
||||
|
||||
|
||||
# "Lizol Floor Cleaner" typed: a Lizol listing of another line (toilet cleaner,
|
||||
# wipes) is a real product, just not the one asked for.
|
||||
OTHER_LINE = "other_product_line"
|
||||
|
||||
|
||||
def _on_line(listing: listings.Listing, target: Target) -> bool:
|
||||
return set(target.line_terms) <= set(listing.tokens)
|
||||
|
||||
|
||||
def _jaccard(a: Sequence[str], b: Sequence[str]) -> float:
|
||||
@@ -145,6 +157,7 @@ def cluster(found: Sequence[listings.Listing]) -> List[Dict]:
|
||||
# normaliser (and off_bulk.strip_sizes) drops bracketed text, which
|
||||
# turned "Lizol Floor Cleaner (Citrus)" and "(Floral)" into one name.
|
||||
plain = re.sub(r"\s+", " ", re.sub(r"[()]", " ", name)).strip()
|
||||
line, variant = variants.split_variant(name)
|
||||
candidates.append({
|
||||
"title": f"{plain} {size}",
|
||||
"size": size,
|
||||
@@ -154,6 +167,8 @@ def cluster(found: Sequence[listings.Listing]) -> List[Dict]:
|
||||
"price_range": _price_range(prices),
|
||||
"listings": [m.as_dict() for m in members[:_MAX_LISTINGS_PER_CANDIDATE]],
|
||||
"brand_term": brand_term,
|
||||
"product_line": line,
|
||||
"variant": variant,
|
||||
"source": "web",
|
||||
})
|
||||
candidates.sort(key=lambda c: (-c["retailer_count"], c["title"].lower()))
|
||||
@@ -205,6 +220,8 @@ def run(brand: str, *, progress: Optional[Callable[[RunResult], None]] = None,
|
||||
query.target.brand_terms)
|
||||
if listing is None:
|
||||
rejected[reason] += 1
|
||||
elif not _on_line(listing, query.target):
|
||||
rejected[OTHER_LINE] += 1
|
||||
else:
|
||||
kept.append(listing)
|
||||
result.queries_done += 1
|
||||
|
||||
@@ -277,6 +277,14 @@ def parse_hit(title: str, url: str, snippet: str,
|
||||
if size is None:
|
||||
# Some retailers put the size only in the title's tail ("... - 125 g | Zepto").
|
||||
size, distinct = extract_size(title)
|
||||
if size is None and not _MULTIPACK.search(snippet or ""):
|
||||
# Blinkit often prints the pack only in the snippet ("... 500 ml ...
|
||||
# Price"): 63 of 80 real Reckitt listings were lost for want of it.
|
||||
# Taken only when the snippet names exactly ONE size - a snippet that
|
||||
# lists the other packs ("also in 1 L, 2 L") says nothing about this one.
|
||||
size, distinct = extract_size(snippet)
|
||||
if distinct > 1:
|
||||
return None, NO_PACK_SIZE
|
||||
if size is None:
|
||||
return None, NO_PACK_SIZE
|
||||
if distinct > 1:
|
||||
|
||||
@@ -10,7 +10,10 @@ web-found product it
|
||||
* fills `price_range` from the listing prices when the pipeline left it blank;
|
||||
* sets `validation_status = 'needs_review'` when only ONE retailer listed it
|
||||
(never lifting a `rejected` verdict). Two or more retailers keep whatever the
|
||||
validator decided.
|
||||
validator decided;
|
||||
* sets `product_line` / `variant` to what the listing titles said. These
|
||||
OVERWRITE the pipeline's value: the pipeline read them off the bare name,
|
||||
which has lost the retailer's brackets ("Harpic ... (Original)").
|
||||
|
||||
The join is exact: stage 11 reports every stored row's `image_id` with the
|
||||
1-based sheet row it came from (header = row 1), and the ingest wrote the web
|
||||
@@ -77,10 +80,13 @@ def apply(brand: str, updates: List[Dict[str, Any]]) -> int:
|
||||
f"price_range = CASE WHEN COALESCE(price_range, '') = '' AND %s <> '' "
|
||||
f"THEN %s ELSE price_range END, "
|
||||
f"validation_status = CASE WHEN %s AND COALESCE(validation_status, '') <> 'rejected' "
|
||||
f"THEN 'needs_review' ELSE validation_status END "
|
||||
f"THEN 'needs_review' ELSE validation_status END, "
|
||||
f"product_line = COALESCE(NULLIF(%s, ''), product_line), "
|
||||
f"variant = COALESCE(NULLIF(%s, ''), variant) "
|
||||
f"WHERE image_id = %s",
|
||||
(Json(provenance), entry.get("price_range") or "", entry.get("price_range") or "",
|
||||
single, update["image_id"]),
|
||||
single, entry.get("product_line") or "", entry.get("variant") or "",
|
||||
update["image_id"]),
|
||||
)
|
||||
written += cur.rowcount
|
||||
except Exception as exc: # noqa: BLE001 - provenance must not take the batch down
|
||||
|
||||
162
app/services/web_discovery/variants.py
Normal file
162
app/services/web_discovery/variants.py
Normal file
@@ -0,0 +1,162 @@
|
||||
"""A product name -> (product line, variant), and candidates -> line -> variant -> sizes.
|
||||
|
||||
"Lizol Disinfectant Surface & Floor Cleaner (Lavender)" is the line
|
||||
"Lizol Disinfectant Surface & Floor Cleaner" in the variant "Lavender". Retail
|
||||
titles mark the variant two ways:
|
||||
|
||||
* in brackets - Blinkit's "(Citrus, 625 ml)", "(Original)". Whatever a
|
||||
bracket holds once the size is gone IS the variant, whatever the word.
|
||||
* as a bare word - "Lizol ... Cleaner Floral 500 ml". Only a word in
|
||||
VARIANT_TERMS is read as a variant here; any other bare word stays in the
|
||||
line, because guessing would split one product into two lines.
|
||||
|
||||
VARIANT_TERMS is scents and flavours only. Generic words ("original", "fresh",
|
||||
"chocolate", "classic") are left out on purpose: "Dettol Original Soap" and
|
||||
"Cadbury Dairy Milk Chocolate" name a line, not a variant. A bracket still
|
||||
catches "(Original)". Add a word only when a real listing needs it.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from collections import Counter, defaultdict
|
||||
from typing import Dict, Iterable, List, Tuple
|
||||
|
||||
VARIANT_TERMS = frozenset({
|
||||
# scents - household and personal care
|
||||
"citrus", "floral", "pine", "lavender", "jasmine", "rose", "lemon", "lime",
|
||||
"orange", "sandal", "sandalwood", "neem", "mint", "eucalyptus", "aloe",
|
||||
"lily", "marine", "ocean", "mogra", "tulsi", "vanilla", "berry", "musk",
|
||||
# flavours that name a pack rather than a product
|
||||
"strawberry", "mango", "elaichi", "masala", "pineapple", "litchi", "guava",
|
||||
})
|
||||
|
||||
# A word that belongs to the variant when it sits next to one: "Lemon Fresh",
|
||||
# "Ocean Breeze", "Lavender Blossom". Never a variant on its own.
|
||||
_VARIANT_MODIFIERS = frozenset({"fresh", "blossom", "breeze", "burst", "bloom", "splash", "twist"})
|
||||
|
||||
_PACKAGING_WORDS = frozenset({"bar", "pouch", "bottle", "jar", "box", "tin", "carton", "pack", "refill"})
|
||||
_BRACKET = re.compile(r"\(([^()]*)\)")
|
||||
_WORD = re.compile(r"[A-Za-z0-9]+(?:'[A-Za-z]+)?")
|
||||
|
||||
|
||||
def _words(text: str) -> List[str]:
|
||||
return re.findall(r"[a-z0-9]+", (text or "").lower())
|
||||
|
||||
|
||||
def _tidy(text: str) -> str:
|
||||
return re.sub(r"\s+", " ", text or "").strip(" ,;:.-/&+–—")
|
||||
|
||||
|
||||
def _title(word: str) -> str:
|
||||
return word if any(c.isupper() for c in word) else word.capitalize()
|
||||
|
||||
|
||||
def variant_words(text: str) -> frozenset:
|
||||
"""The VARIANT_TERMS words in `text` - for telling two names apart."""
|
||||
return frozenset(w for w in _words(text) if w in VARIANT_TERMS)
|
||||
|
||||
|
||||
def split_variant(name: str) -> Tuple[str, str]:
|
||||
"""(line, variant). A pack size in the name is ignored ("... Floral 500 ml").
|
||||
|
||||
("Lizol Floor Cleaner (Citrus)") -> ("Lizol Floor Cleaner", "Citrus")
|
||||
("Lizol Floor Cleaner Floral") -> ("Lizol Floor Cleaner", "Floral")
|
||||
("Harpic Toilet Cleaner (Original)") -> ("Harpic Toilet Cleaner", "Original")
|
||||
("Dettol Original Soap") -> ("Dettol Original Soap", "")
|
||||
"""
|
||||
from app.services.web_discovery.listings import _strip_sizes
|
||||
|
||||
name = _tidy(re.sub(r"\s+", " ", _strip_sizes(name or "")))
|
||||
groups = [_tidy(g) for g in _BRACKET.findall(name)]
|
||||
groups = [g for g in groups if re.search(r"[A-Za-z]", g)]
|
||||
if groups:
|
||||
line = _tidy(re.sub(r"\s+", " ", _BRACKET.sub(" ", name)))
|
||||
return line, " ".join(_title(w) for g in groups for w in _WORD.findall(g))
|
||||
|
||||
tokens = _WORD.findall(name)
|
||||
lowered = [t.lower() for t in tokens]
|
||||
keep = [l in VARIANT_TERMS for l in lowered]
|
||||
if not any(keep) or all(keep):
|
||||
return name, ""
|
||||
# A modifier joins the variant only when it touches a variant word.
|
||||
for i, l in enumerate(lowered):
|
||||
if l in _VARIANT_MODIFIERS and ((i > 0 and keep[i - 1]) or (i + 1 < len(keep) and keep[i + 1])):
|
||||
keep[i] = True
|
||||
variant = " ".join(_title(t) for t, k in zip(tokens, keep) if k)
|
||||
line = name
|
||||
for t, k in zip(tokens, keep):
|
||||
if k:
|
||||
line = re.sub(r"(?<![A-Za-z0-9])" + re.escape(t) + r"(?![A-Za-z0-9])", " ", line, count=1)
|
||||
line = _tidy(re.sub(r"\s*&\s*$", "", re.sub(r"\s+", " ", line)))
|
||||
return (line or name), variant
|
||||
|
||||
|
||||
def line_key(line: str, brand_term: str = "") -> Tuple[str, ...]:
|
||||
"""Lines are one line when these words match - order, '&' and packaging aside."""
|
||||
brand = set(_words(brand_term))
|
||||
return tuple(sorted({w for w in _words(line) if w not in _PACKAGING_WORDS} - brand)) or tuple(_words(line))
|
||||
|
||||
|
||||
_SIZE = re.compile(r"^(\d+(?:\.\d+)?)(kg|g|ml|l|pcs)$")
|
||||
_SIZE_SCALE = {"g": (0, 1), "kg": (0, 1000), "ml": (1, 1), "l": (1, 1000), "pcs": (2, 1)}
|
||||
|
||||
|
||||
def size_sort_key(size: str) -> Tuple[int, float, str]:
|
||||
"""Mass, then volume, then counts, each smallest first: 500ml < 625ml < 1l."""
|
||||
m = _SIZE.match((size or "").lower())
|
||||
if not m:
|
||||
return (9, 0.0, size or "")
|
||||
group, scale = _SIZE_SCALE[m.group(2)]
|
||||
return (group, float(m.group(1)) * scale, size)
|
||||
|
||||
|
||||
def group_families(candidates: Iterable[Dict]) -> List[Dict]:
|
||||
"""Candidates (one per name and pack) -> product line -> variant -> sizes.
|
||||
|
||||
Each candidate needs `product_line`, `variant` and `size`; `retailer_count`,
|
||||
`price_range`, `providers` and `listings` are carried through when present,
|
||||
and a `row_index` is collected into the size's `rows`.
|
||||
Lines come out best-corroborated first; variants and sizes in a stable order.
|
||||
"""
|
||||
lines: Dict[Tuple, List[Dict]] = defaultdict(list)
|
||||
for c in candidates:
|
||||
line = c.get("product_line") or ""
|
||||
if not line:
|
||||
continue
|
||||
lines[(c.get("brand_term") or "", line_key(line, c.get("brand_term") or ""))].append(c)
|
||||
|
||||
families: List[Dict] = []
|
||||
for members in lines.values():
|
||||
spellings = Counter(m["product_line"] for m in members)
|
||||
display = sorted(spellings, key=lambda s: (-spellings[s], len(s), s))[0]
|
||||
by_variant: Dict[str, List[Dict]] = defaultdict(list)
|
||||
for m in members:
|
||||
by_variant[m.get("variant") or ""].append(m)
|
||||
variants = []
|
||||
for variant, rows in sorted(by_variant.items(), key=lambda kv: (kv[0] == "", kv[0].lower())):
|
||||
sizes: Dict[str, Dict] = {}
|
||||
for r in rows:
|
||||
size = r.get("size") or ((r.get("sizes") or [None])[0])
|
||||
if not size:
|
||||
continue
|
||||
entry = sizes.setdefault(size, {"size": size, "retailer_count": 0, "price_range": "",
|
||||
"providers": [], "listings": [], "rows": []})
|
||||
if r.get("row_index") is not None:
|
||||
entry["rows"].append(r["row_index"])
|
||||
entry["retailer_count"] = max(entry["retailer_count"], int(r.get("retailer_count") or 0))
|
||||
entry["price_range"] = entry["price_range"] or r.get("price_range") or ""
|
||||
for p in r.get("providers") or []:
|
||||
if p not in entry["providers"]:
|
||||
entry["providers"].append(p)
|
||||
entry["listings"].extend(r.get("listings") or [])
|
||||
variants.append({"variant": variant,
|
||||
"sizes": sorted(sizes.values(), key=lambda s: size_sort_key(s["size"]))})
|
||||
families.append({
|
||||
"product_line": display,
|
||||
"brand_term": members[0].get("brand_term") or "",
|
||||
"variants": variants,
|
||||
"retailer_count": max((int(m.get("retailer_count") or 0) for m in members), default=0),
|
||||
})
|
||||
families.sort(key=lambda f: (-f["retailer_count"], -len(f["variants"]), f["product_line"].lower()))
|
||||
return families
|
||||
|
||||
@@ -383,7 +383,7 @@ class _Conn(_Cursor):
|
||||
pass
|
||||
|
||||
|
||||
def test_apply_touches_only_provenance_price_and_review_status(monkeypatch):
|
||||
def test_apply_touches_only_provenance_price_review_status_and_line(monkeypatch):
|
||||
conn = _Conn()
|
||||
monkeypatch.setattr("app.services.vector_store._connect", lambda: conn)
|
||||
provenance.apply("Reckitt Benckiser", [
|
||||
@@ -393,7 +393,9 @@ def test_apply_touches_only_provenance_price_and_review_status(monkeypatch):
|
||||
sql, params = conn.cur.calls[0]
|
||||
assert sql.startswith("UPDATE brand_reckitt_benckiser SET field_sources")
|
||||
assert "price_range = CASE WHEN COALESCE(price_range, '') = ''" in sql
|
||||
assert "'rejected'" in sql and params[-2] is True and params[-1] == "one_shop"
|
||||
assert "'rejected'" in sql and params[3] is True and params[-1] == "one_shop"
|
||||
# A blank line/variant from the entry keeps what the pipeline stored.
|
||||
assert "variant = COALESCE(NULLIF(%s, ''), variant)" in sql and params[-2] == ""
|
||||
for column in ("product_name", "barcode", "image_url", "category"):
|
||||
assert f"{column} =" not in sql
|
||||
|
||||
|
||||
182
tests/test_web_variants.py
Normal file
182
tests/test_web_variants.py
Normal file
@@ -0,0 +1,182 @@
|
||||
"""A product name typed into Brand Discovery -> its line, variants and pack sizes.
|
||||
|
||||
Typing "Lizol" (a Reckitt line) must find Lizol, and the result must read as
|
||||
"Lizol ... Floor Cleaner -> Citrus / Floral / Lavender -> 500 ml, 625 ml, 1 l"
|
||||
rather than a flat list - without ever folding two scents into one product.
|
||||
No network: every listing is a recorded Blinkit/Amazon title shape.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from app.services import brand_discovery as bd
|
||||
from app.services import brand_registry
|
||||
from app.services.web_discovery import brands, discover, jobs, listings, variants
|
||||
|
||||
LIZOL = ("lizol",)
|
||||
LINE = "Lizol Disinfectant Surface & Floor Cleaner"
|
||||
|
||||
|
||||
def _listing(title, n, terms=LIZOL, snippet=""):
|
||||
listing, reason = listings.parse_hit(title, f"https://blinkit.com/prn/x/prid/{n}", snippet, terms)
|
||||
assert reason is None, (title, reason)
|
||||
return listing
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# A product name resolves to the maker
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.mark.parametrize("typed", ["Lizol", "Lizol Floor Cleaner", "lizol disinfectant surface cleaner",
|
||||
"Harpic Power Plus", "Dettol Liquid", "Reckitt"])
|
||||
def test_a_product_name_resolves_to_its_maker(typed):
|
||||
assert brand_registry.resolve_parent_brand(typed) == "reckitt benckiser"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("typed", ["Laxmi Chilli Powder", "Boost Energy Drink", "Baby Oil Himalaya",
|
||||
"Sun Pharma", "Lizzy Cleaner"])
|
||||
def test_an_everyday_word_is_not_read_as_a_line(typed):
|
||||
assert brand_registry.resolve_parent_brand(typed) == typed
|
||||
|
||||
|
||||
def test_every_name_that_resolved_before_still_resolves_the_same():
|
||||
"""The new rules run only after every old one failed, so an alias key can
|
||||
never change its answer."""
|
||||
for alias, parent in brand_registry.BRAND_ALIASES.items():
|
||||
assert brand_registry.resolve_parent_brand(alias) == parent
|
||||
|
||||
|
||||
def test_a_product_line_searches_that_phrase_but_judges_by_the_line_word():
|
||||
[target] = brands.targets_for("Lizol Floor Cleaner")
|
||||
assert target.query == "Lizol Floor Cleaner"
|
||||
assert target.brand_terms == ("lizol",) and target.line_terms == ("floor", "cleaner")
|
||||
assert brands.targets_for("Lizol") == [brands.Target("Lizol", ("lizol",))]
|
||||
|
||||
|
||||
def test_another_line_of_the_same_product_is_counted_and_dropped():
|
||||
target = brands.Target("Lizol Floor Cleaner", LIZOL, ("floor", "cleaner"))
|
||||
floor = _listing(f"{LINE} (Citrus - 500 ml) Price - Buy Online", 1)
|
||||
toilet = _listing("Lizol Power Toilet Cleaner 500 ml Price - Buy Online", 2)
|
||||
assert discover._on_line(floor, target) and not discover._on_line(toilet, target)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Line and variant
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.mark.parametrize("name,line,variant", [
|
||||
(f"{LINE} (Lavender)", LINE, "Lavender"),
|
||||
("Lizol Disinfectant Surface Cleaner Floral 500 ml", "Lizol Disinfectant Surface Cleaner", "Floral"),
|
||||
("Harpic Disinfectant Liquid Toilet Cleaner (Original)", "Harpic Disinfectant Liquid Toilet Cleaner",
|
||||
"Original"),
|
||||
("Lizol Floor Cleaner Lemon Fresh", "Lizol Floor Cleaner", "Lemon Fresh"),
|
||||
# Generic words are a line, not a variant - only a bracket makes them one.
|
||||
("Dettol Original Soap", "Dettol Original Soap", ""),
|
||||
("Cadbury Dairy Milk Chocolate", "Cadbury Dairy Milk Chocolate", ""),
|
||||
])
|
||||
def test_split_variant(name, line, variant):
|
||||
assert variants.split_variant(name) == (line, variant)
|
||||
|
||||
|
||||
def test_sizes_sort_by_quantity_not_text():
|
||||
sizes = ["1l", "500ml", "2l", "625ml", "200ml"]
|
||||
assert sorted(sizes, key=variants.size_sort_key) == ["200ml", "500ml", "625ml", "1l", "2l"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Scents never merge
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_unbracketed_scents_of_one_pack_stay_two_products():
|
||||
"""3 of 5 words shared - exactly the 0.6 overlap bar, which used to fold them."""
|
||||
floral = _listing("Lizol Disinfectant Surface Cleaner Floral 500 ml : Amazon.in: Health", 1)
|
||||
pine = _listing("Lizol Disinfectant Surface Cleaner Pine 500 ml : Amazon.in: Health", 2)
|
||||
assert len(discover.cluster([floral, pine])) == 2
|
||||
|
||||
|
||||
def test_the_same_scent_on_two_retailers_is_still_one_product():
|
||||
a = _listing(f"{LINE} (Citrus - 500 ml) Price - Buy Online", 1)
|
||||
b, _ = listings.parse_hit(f"{LINE} Citrus 500 ml : Amazon.in", "https://www.amazon.in/dp/B000000001",
|
||||
"", LIZOL)
|
||||
[one] = discover.cluster([a, b])
|
||||
assert one["retailer_count"] == 2 and one["variant"] == "Citrus"
|
||||
|
||||
|
||||
def test_a_scent_cannot_be_renamed_to_a_stored_other_scent():
|
||||
"""Citrus vs Floral scores 0.857 - over the 0.85 floor `_match_existing` uses."""
|
||||
stored = {bd._normalise_title("Reckitt Benckiser", "Lizol Floor Cleaner Citrus"): "Lizol Floor Cleaner Citrus"}
|
||||
key = bd._normalise_title("Reckitt Benckiser", "Lizol Floor Cleaner Floral")
|
||||
assert bd._match_existing(key, "Lizol Floor Cleaner Floral", stored) is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Size from the snippet
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_size_only_in_the_snippet_is_taken():
|
||||
listing = _listing("Lizol Disinfectant Floor Cleaner (Jasmine) Price - Buy Online", 1,
|
||||
snippet="Lizol Disinfectant Floor Cleaner Jasmine 975 ml. Delivered in minutes.")
|
||||
assert listing.size == "975ml"
|
||||
|
||||
|
||||
def test_a_snippet_naming_several_packs_proves_nothing():
|
||||
listing, reason = listings.parse_hit(
|
||||
"Lizol Disinfectant Floor Cleaner (Jasmine) Price - Buy Online", "https://blinkit.com/prn/x/prid/1",
|
||||
"Available in 500 ml, 975 ml and 2 l packs.", LIZOL)
|
||||
assert listing is None and reason == listings.NO_PACK_SIZE
|
||||
|
||||
|
||||
def test_a_multipack_snippet_is_not_a_pack_size():
|
||||
listing, reason = listings.parse_hit(
|
||||
"Lizol Disinfectant Floor Cleaner (Jasmine) Price - Buy Online", "https://blinkit.com/prn/x/prid/1",
|
||||
"Pack of 2 - 500 ml each", LIZOL)
|
||||
assert listing is None and reason == listings.NO_PACK_SIZE
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# End to end: line -> variants -> sizes
|
||||
# ---------------------------------------------------------------------------
|
||||
_TITLES = [
|
||||
f"{LINE} (Citrus - 500 ml) Price - Buy Online at Best",
|
||||
f"Buy {LINE} (Citrus, 625 ml) Online",
|
||||
f"{LINE} (Citrus - 1 l) Price - Buy Online at Best",
|
||||
f"{LINE} (Floral - 500 ml) Price - Buy Online at Best",
|
||||
f"{LINE} (Floral - 2 l) Price - Buy Online at Best",
|
||||
f"{LINE} (Lavender - 500 ml) Price - Buy Online at Best",
|
||||
"Lizol Power Toilet Cleaner Original 500 ml Price - Buy Online",
|
||||
]
|
||||
|
||||
|
||||
def test_lizol_comes_out_as_line_then_variants_then_sizes():
|
||||
found = [_listing(t, i) for i, t in enumerate(_TITLES)]
|
||||
families = variants.group_families(discover.cluster(found))
|
||||
|
||||
floor = next(f for f in families if f["product_line"] == LINE)
|
||||
assert {v["variant"]: [s["size"] for s in v["sizes"]] for v in floor["variants"]} == {
|
||||
"Citrus": ["500ml", "625ml", "1l"],
|
||||
"Floral": ["500ml", "2l"],
|
||||
"Lavender": ["500ml"],
|
||||
}
|
||||
assert any(f["product_line"].startswith("Lizol Power Toilet Cleaner") for f in families)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def no_other_sources(monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [])
|
||||
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [])
|
||||
monkeypatch.setattr(bd, "_from_brand_store", lambda brand: [])
|
||||
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [])
|
||||
|
||||
|
||||
def test_the_preview_carries_families_and_each_row_its_line(no_other_sources):
|
||||
found = [_listing(t, i) for i, t in enumerate(_TITLES[:5])]
|
||||
job = jobs.WebJob(job_id="d" * 32, brand="Lizol", status=jobs.DONE, backend="duckduckgo",
|
||||
queries_total=7, queries_done=7, listings_kept=5, candidates=discover.cluster(found))
|
||||
jobs._jobs[job.job_id] = job
|
||||
|
||||
body = bd.discover_brand_products("Lizol", use_llm=False, use_web=True, web_job_id=job.job_id).as_dict()
|
||||
|
||||
assert body["parent_brand"] == "reckitt benckiser" and body["table"] == "brand_reckitt_benckiser"
|
||||
assert {(p["product_line"], p["variant"]) for p in body["products"]} == {(LINE, "Citrus"), (LINE, "Floral")}
|
||||
[family] = body["families"]
|
||||
sizes = {v["variant"]: [s["size"] for s in v["sizes"]] for v in family["variants"]}
|
||||
assert sizes == {"Citrus": ["500ml", "625ml", "1l"], "Floral": ["500ml", "2l"]}
|
||||
for v in family["variants"]:
|
||||
for s in v["sizes"]:
|
||||
assert all(body["products"][i]["variant"] == v["variant"] for i in s["rows"])
|
||||
Reference in New Issue
Block a user