Product discovery on names

This commit is contained in:
sriram
2026-10-06 11:57:01 +05:30
parent c0601b65fe
commit 34c65a7cbd
13 changed files with 569 additions and 24 deletions

View File

@@ -383,7 +383,7 @@ class _Conn(_Cursor):
pass
def test_apply_touches_only_provenance_price_and_review_status(monkeypatch):
def test_apply_touches_only_provenance_price_review_status_and_line(monkeypatch):
conn = _Conn()
monkeypatch.setattr("app.services.vector_store._connect", lambda: conn)
provenance.apply("Reckitt Benckiser", [
@@ -393,7 +393,9 @@ def test_apply_touches_only_provenance_price_and_review_status(monkeypatch):
sql, params = conn.cur.calls[0]
assert sql.startswith("UPDATE brand_reckitt_benckiser SET field_sources")
assert "price_range = CASE WHEN COALESCE(price_range, '') = ''" in sql
assert "'rejected'" in sql and params[-2] is True and params[-1] == "one_shop"
assert "'rejected'" in sql and params[3] is True and params[-1] == "one_shop"
# A blank line/variant from the entry keeps what the pipeline stored.
assert "variant = COALESCE(NULLIF(%s, ''), variant)" in sql and params[-2] == ""
for column in ("product_name", "barcode", "image_url", "category"):
assert f"{column} =" not in sql

182
tests/test_web_variants.py Normal file
View File

@@ -0,0 +1,182 @@
"""A product name typed into Brand Discovery -> its line, variants and pack sizes.
Typing "Lizol" (a Reckitt line) must find Lizol, and the result must read as
"Lizol ... Floor Cleaner -> Citrus / Floral / Lavender -> 500 ml, 625 ml, 1 l"
rather than a flat list - without ever folding two scents into one product.
No network: every listing is a recorded Blinkit/Amazon title shape.
"""
from __future__ import annotations
import pytest
from app.services import brand_discovery as bd
from app.services import brand_registry
from app.services.web_discovery import brands, discover, jobs, listings, variants
LIZOL = ("lizol",)
LINE = "Lizol Disinfectant Surface & Floor Cleaner"
def _listing(title, n, terms=LIZOL, snippet=""):
listing, reason = listings.parse_hit(title, f"https://blinkit.com/prn/x/prid/{n}", snippet, terms)
assert reason is None, (title, reason)
return listing
# ---------------------------------------------------------------------------
# A product name resolves to the maker
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("typed", ["Lizol", "Lizol Floor Cleaner", "lizol disinfectant surface cleaner",
"Harpic Power Plus", "Dettol Liquid", "Reckitt"])
def test_a_product_name_resolves_to_its_maker(typed):
assert brand_registry.resolve_parent_brand(typed) == "reckitt benckiser"
@pytest.mark.parametrize("typed", ["Laxmi Chilli Powder", "Boost Energy Drink", "Baby Oil Himalaya",
"Sun Pharma", "Lizzy Cleaner"])
def test_an_everyday_word_is_not_read_as_a_line(typed):
assert brand_registry.resolve_parent_brand(typed) == typed
def test_every_name_that_resolved_before_still_resolves_the_same():
"""The new rules run only after every old one failed, so an alias key can
never change its answer."""
for alias, parent in brand_registry.BRAND_ALIASES.items():
assert brand_registry.resolve_parent_brand(alias) == parent
def test_a_product_line_searches_that_phrase_but_judges_by_the_line_word():
[target] = brands.targets_for("Lizol Floor Cleaner")
assert target.query == "Lizol Floor Cleaner"
assert target.brand_terms == ("lizol",) and target.line_terms == ("floor", "cleaner")
assert brands.targets_for("Lizol") == [brands.Target("Lizol", ("lizol",))]
def test_another_line_of_the_same_product_is_counted_and_dropped():
target = brands.Target("Lizol Floor Cleaner", LIZOL, ("floor", "cleaner"))
floor = _listing(f"{LINE} (Citrus - 500 ml) Price - Buy Online", 1)
toilet = _listing("Lizol Power Toilet Cleaner 500 ml Price - Buy Online", 2)
assert discover._on_line(floor, target) and not discover._on_line(toilet, target)
# ---------------------------------------------------------------------------
# Line and variant
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("name,line,variant", [
(f"{LINE} (Lavender)", LINE, "Lavender"),
("Lizol Disinfectant Surface Cleaner Floral 500 ml", "Lizol Disinfectant Surface Cleaner", "Floral"),
("Harpic Disinfectant Liquid Toilet Cleaner (Original)", "Harpic Disinfectant Liquid Toilet Cleaner",
"Original"),
("Lizol Floor Cleaner Lemon Fresh", "Lizol Floor Cleaner", "Lemon Fresh"),
# Generic words are a line, not a variant - only a bracket makes them one.
("Dettol Original Soap", "Dettol Original Soap", ""),
("Cadbury Dairy Milk Chocolate", "Cadbury Dairy Milk Chocolate", ""),
])
def test_split_variant(name, line, variant):
assert variants.split_variant(name) == (line, variant)
def test_sizes_sort_by_quantity_not_text():
sizes = ["1l", "500ml", "2l", "625ml", "200ml"]
assert sorted(sizes, key=variants.size_sort_key) == ["200ml", "500ml", "625ml", "1l", "2l"]
# ---------------------------------------------------------------------------
# Scents never merge
# ---------------------------------------------------------------------------
def test_unbracketed_scents_of_one_pack_stay_two_products():
"""3 of 5 words shared - exactly the 0.6 overlap bar, which used to fold them."""
floral = _listing("Lizol Disinfectant Surface Cleaner Floral 500 ml : Amazon.in: Health", 1)
pine = _listing("Lizol Disinfectant Surface Cleaner Pine 500 ml : Amazon.in: Health", 2)
assert len(discover.cluster([floral, pine])) == 2
def test_the_same_scent_on_two_retailers_is_still_one_product():
a = _listing(f"{LINE} (Citrus - 500 ml) Price - Buy Online", 1)
b, _ = listings.parse_hit(f"{LINE} Citrus 500 ml : Amazon.in", "https://www.amazon.in/dp/B000000001",
"", LIZOL)
[one] = discover.cluster([a, b])
assert one["retailer_count"] == 2 and one["variant"] == "Citrus"
def test_a_scent_cannot_be_renamed_to_a_stored_other_scent():
"""Citrus vs Floral scores 0.857 - over the 0.85 floor `_match_existing` uses."""
stored = {bd._normalise_title("Reckitt Benckiser", "Lizol Floor Cleaner Citrus"): "Lizol Floor Cleaner Citrus"}
key = bd._normalise_title("Reckitt Benckiser", "Lizol Floor Cleaner Floral")
assert bd._match_existing(key, "Lizol Floor Cleaner Floral", stored) is None
# ---------------------------------------------------------------------------
# Size from the snippet
# ---------------------------------------------------------------------------
def test_a_size_only_in_the_snippet_is_taken():
listing = _listing("Lizol Disinfectant Floor Cleaner (Jasmine) Price - Buy Online", 1,
snippet="Lizol Disinfectant Floor Cleaner Jasmine 975 ml. Delivered in minutes.")
assert listing.size == "975ml"
def test_a_snippet_naming_several_packs_proves_nothing():
listing, reason = listings.parse_hit(
"Lizol Disinfectant Floor Cleaner (Jasmine) Price - Buy Online", "https://blinkit.com/prn/x/prid/1",
"Available in 500 ml, 975 ml and 2 l packs.", LIZOL)
assert listing is None and reason == listings.NO_PACK_SIZE
def test_a_multipack_snippet_is_not_a_pack_size():
listing, reason = listings.parse_hit(
"Lizol Disinfectant Floor Cleaner (Jasmine) Price - Buy Online", "https://blinkit.com/prn/x/prid/1",
"Pack of 2 - 500 ml each", LIZOL)
assert listing is None and reason == listings.NO_PACK_SIZE
# ---------------------------------------------------------------------------
# End to end: line -> variants -> sizes
# ---------------------------------------------------------------------------
_TITLES = [
f"{LINE} (Citrus - 500 ml) Price - Buy Online at Best",
f"Buy {LINE} (Citrus, 625 ml) Online",
f"{LINE} (Citrus - 1 l) Price - Buy Online at Best",
f"{LINE} (Floral - 500 ml) Price - Buy Online at Best",
f"{LINE} (Floral - 2 l) Price - Buy Online at Best",
f"{LINE} (Lavender - 500 ml) Price - Buy Online at Best",
"Lizol Power Toilet Cleaner Original 500 ml Price - Buy Online",
]
def test_lizol_comes_out_as_line_then_variants_then_sizes():
found = [_listing(t, i) for i, t in enumerate(_TITLES)]
families = variants.group_families(discover.cluster(found))
floor = next(f for f in families if f["product_line"] == LINE)
assert {v["variant"]: [s["size"] for s in v["sizes"]] for v in floor["variants"]} == {
"Citrus": ["500ml", "625ml", "1l"],
"Floral": ["500ml", "2l"],
"Lavender": ["500ml"],
}
assert any(f["product_line"].startswith("Lizol Power Toilet Cleaner") for f in families)
@pytest.fixture
def no_other_sources(monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [])
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [])
monkeypatch.setattr(bd, "_from_brand_store", lambda brand: [])
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [])
def test_the_preview_carries_families_and_each_row_its_line(no_other_sources):
found = [_listing(t, i) for i, t in enumerate(_TITLES[:5])]
job = jobs.WebJob(job_id="d" * 32, brand="Lizol", status=jobs.DONE, backend="duckduckgo",
queries_total=7, queries_done=7, listings_kept=5, candidates=discover.cluster(found))
jobs._jobs[job.job_id] = job
body = bd.discover_brand_products("Lizol", use_llm=False, use_web=True, web_job_id=job.job_id).as_dict()
assert body["parent_brand"] == "reckitt benckiser" and body["table"] == "brand_reckitt_benckiser"
assert {(p["product_line"], p["variant"]) for p in body["products"]} == {(LINE, "Citrus"), (LINE, "Floral")}
[family] = body["families"]
sizes = {v["variant"]: [s["size"] for s in v["sizes"]] for v in family["variants"]}
assert sizes == {"Citrus": ["500ml", "625ml", "1l"], "Floral": ["500ml", "2l"]}
for v in family["variants"]:
for s in v["sizes"]:
assert all(body["products"][i]["variant"] == v["variant"] for i in s["rows"])