Brand Discovery-LLM Updates

This commit is contained in:
sriram
2026-09-29 16:42:49 +05:30
parent c0489d89d6
commit c0601b65fe
16 changed files with 1971 additions and 5 deletions

453
tests/test_web_discovery.py Normal file
View File

@@ -0,0 +1,453 @@
"""Web & retail listing discovery (app/services/web_discovery) - no network.
Every search is a recorded fixture. What is pinned here is the anti-invention
contract: a product exists only if a real retailer's PRODUCT page names the
brand and a pack size; a throttled search is "could not ask", never "nothing
there"; and with the source switched off Brand Discovery is exactly what it was.
"""
from __future__ import annotations
from typing import Dict, List, Optional
import pytest
from app.infrastructure import settings
from app.services import brand_discovery as bd
from app.services.web_discovery import brands, cache, discover, jobs, listings, provenance, search
from app.services.web_discovery.search import Hit
DETTOL = ("dettol",)
# ---------------------------------------------------------------------------
# listings.parse_hit - one hit -> one trustworthy listing, or a reason
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("title,url,snippet,name,size,price", [
("Dettol Original Soap (125 g) - Buy Online at Best Price | Blinkit",
"https://blinkit.com/prn/dettol-original-soap/prid/12345", "", "Dettol Original Soap", "125g", None),
("Buy Dettol Original Soap 125 g Online at Best Price | Zepto",
"https://www.zeptonow.com/pn/dettol-original-soap/pvid/1a2b3c4d-1111", "", "Dettol Original Soap", "125g", None),
("Buy Dettol Original Soap 125 g Online at Best Price of Rs 58 - bigbasket",
"https://www.bigbasket.com/pd/40001234/dettol-soap/", "MRP Rs 58", "Dettol Original Soap", "125g", 58.0),
("Dettol Original Soap Bar, 125g : Amazon.in: Beauty",
"https://www.amazon.in/Dettol-Original/dp/B00ABCDEFG", "₹ 55", "Dettol Original Soap Bar", "125g", 55.0),
("Buy Dettol Antiseptic Liquid 1 L Online | Swiggy Instamart",
"https://www.swiggy.com/instamart/item/ABC123XYZ", "", "Dettol Antiseptic Liquid", "1l", None),
("Dettol Antiseptic Liquid 550 ml - JioMart",
"https://www.jiomart.com/p/groceries/dettol-antiseptic/490001234", "", "Dettol Antiseptic Liquid", "550ml", None),
])
def test_each_retailers_title_shape_reduces_to_the_product(title, url, snippet, name, size, price):
listing, reason = listings.parse_hit(title, url, snippet, DETTOL)
assert reason is None
assert (listing.name, listing.size, listing.price) == (name, size, price)
@pytest.mark.parametrize("title,url,reason", [
("Dettol Soaps - Buy Dettol Soaps Online | Blinkit",
"https://blinkit.com/cn/dettol/cid/1", listings.NOT_A_PRODUCT_PAGE),
("Dettol Original Soap 125g", "https://www.example-blog.com/dettol-review", listings.NOT_A_RETAILER),
("Dettol Original Soap 125g", "https://www.amazon.in/s?k=dettol", listings.NOT_A_PRODUCT_PAGE),
("Dettol Liquid Handwash Refill | Blinkit", "https://blinkit.com/prn/x/prid/7", listings.NO_PACK_SIZE),
("Dettol Original Soap 125g (Pack of 4) : Amazon.in", "https://www.amazon.in/x/dp/B00ABCDEFH",
listings.MULTIPACK),
("Dettol Soap 125g x 3 | Zepto", "https://www.zeptonow.com/pn/x/pvid/1a2b3c4d-9999", listings.MULTIPACK),
("Dettol Handwash 200ml + Refill 175ml | Blinkit", "https://blinkit.com/prn/x/prid/8",
listings.SEVERAL_SIZES),
])
def test_what_is_not_a_listing_of_one_pack_is_refused(title, url, reason):
assert listings.parse_hit(title, url, "", DETTOL) == (None, reason)
def test_another_brand_is_refused_even_when_it_mentions_ours():
listing, reason = listings.parse_hit("Savlon Soap 125g better than Dettol | Blinkit",
"https://blinkit.com/prn/savlon/prid/9", "", DETTOL)
assert listing is None and reason == listings.WRONG_BRAND
def test_the_brand_must_lead_the_title():
assert listings.names_brand("Dettol Original Soap", DETTOL)
assert listings.names_brand("New Dettol Original Soap", DETTOL)
assert not listings.names_brand("Haldiram Mithai with Dettol", DETTOL)
def test_counts_become_pcs_the_unit_discovery_reads():
listing, _ = listings.parse_hit("Durex Extra Thin Condoms 10 Count : Amazon.in",
"https://www.amazon.in/x/dp/B00ABCDEFJ", "", ("durex",))
assert listing.size == "10pcs" and listing.name == "Durex Extra Thin Condoms"
def test_a_run_together_title_is_refused():
title = "Dettol Original Soap 125g " + "x" * 200
assert listings.parse_hit(title, "https://blinkit.com/prn/x/prid/1", "", DETTOL)[1] == listings.TITLE_TOO_LONG
# ---------------------------------------------------------------------------
# brands.targets_for
# ---------------------------------------------------------------------------
def test_reckitt_fans_out_to_the_names_its_products_are_sold_under():
queries = [t.query for t in brands.targets_for("Reckitt Benckiser")]
for sub in ("Dettol", "Harpic", "Lizol", "Mortein", "Durex"):
assert sub in queries
assert queries[-1] == "Reckitt Benckiser"
def test_a_typed_sub_brand_searches_only_itself():
assert [(t.query, t.brand_terms) for t in brands.targets_for("Dettol")] == [("Dettol", ("dettol",))]
def test_a_generic_product_word_is_never_a_brand_word():
butter = next(t for t in brands.targets_for("Amul") if t.query == "Amul Butter")
assert butter.brand_terms == ("amul",)
def test_a_line_name_sold_without_the_parent_counts_on_its_own():
cerelac = next(t for t in brands.targets_for("Nestle") if t.query == "Nestle Cerelac")
assert "cerelac" in cerelac.brand_terms
def test_kit_kat_is_not_an_abbreviation_plus_a_sub_brand():
assert not brands._is_abbreviation("kit", "nestle")
assert brands._is_abbreviation("rb", "reckitt benckiser")
# ---------------------------------------------------------------------------
# discover.run - the loop, with a fake search engine
# ---------------------------------------------------------------------------
@pytest.fixture(autouse=True)
def _isolated_dir(tmp_path, monkeypatch):
monkeypatch.setattr(settings, "WEB_DISCOVERY_DIR", tmp_path / "web_discovery")
monkeypatch.setattr(settings, "WEB_DISCOVERY_PAUSE_SECONDS", 0.0)
monkeypatch.setattr(search, "google_configured", lambda: False)
jobs._jobs.clear()
yield
jobs._jobs.clear()
class FakeEngine:
def __init__(self, answers: Dict[str, Optional[List[Hit]]]):
self.answers = answers
self.asked: List[str] = []
def __call__(self, query, max_results=None):
self.asked.append(query)
for key, hits in self.answers.items():
if key in query:
return hits
return []
BLINKIT_SOAP = Hit("Dettol Original Soap (125 g) - Buy Online at Best Price | Blinkit",
"https://blinkit.com/prn/dettol-original-soap/prid/1", "₹ 50")
ZEPTO_SOAP = Hit("Buy Dettol Original Soap 125 g Online at Best Price | Zepto",
"https://www.zeptonow.com/pn/dettol-original-soap/pvid/1a2b3c4d-0001", "₹ 55")
ZEPTO_LIQUID = Hit("Buy Dettol Antiseptic Liquid 250 ml Online | Zepto",
"https://www.zeptonow.com/pn/dettol-liquid/pvid/1a2b3c4d-0002", "")
def test_the_same_pack_on_two_retailers_is_one_candidate_with_both(monkeypatch):
engine = FakeEngine({"site:blinkit.com": [BLINKIT_SOAP], "site:zeptonow.com": [ZEPTO_SOAP, ZEPTO_LIQUID]})
monkeypatch.setattr(search, "search", engine)
result = discover.run("Dettol", sleep=lambda s: None)
assert result.status == discover.DONE
by_title = {c["title"]: c for c in result.candidates}
soap = by_title["Dettol Original Soap 125g"]
assert soap["providers"] == ["Blinkit", "Zepto"] and soap["retailer_count"] == 2
assert soap["price_range"] == "₹50 - ₹55"
assert by_title["Dettol Antiseptic Liquid 250ml"]["retailer_count"] == 1
assert result.candidates[0]["title"] == "Dettol Original Soap 125g" # best corroborated first
def test_a_generic_title_does_not_swallow_a_variant():
a, _ = listings.parse_hit("Dettol Soap 125g | Blinkit", "https://blinkit.com/prn/a/prid/1", "", DETTOL)
b, _ = listings.parse_hit("Dettol Cool Soap 125g | Zepto",
"https://www.zeptonow.com/pn/b/pvid/1a2b3c4d-0003", "", DETTOL)
assert len(discover.cluster([a, b])) == 2
def test_three_unanswered_searches_end_the_run_as_partial_and_are_not_cached(monkeypatch):
engine = FakeEngine({"site:": None})
monkeypatch.setattr(search, "search", engine)
result = discover.run("Dettol", sleep=lambda s: None)
assert result.status == discover.PARTIAL and result.queries_failed == 3
assert len(engine.asked) == 3 and result.candidates == []
assert cache.get(engine.asked[0], search.backend_name()) is None # "could not ask" is never stored
def test_a_single_unanswered_search_still_makes_the_run_partial(monkeypatch):
answers = {"site:blinkit.com": None, "site:zeptonow.com": [ZEPTO_SOAP]}
monkeypatch.setattr(search, "search", FakeEngine(answers))
result = discover.run("Dettol", sleep=lambda s: None)
assert result.status == discover.PARTIAL and result.queries_failed == 1
assert [c["title"] for c in result.candidates] == ["Dettol Original Soap 125g"]
def test_answers_are_cached_so_a_rerun_asks_nothing(monkeypatch):
engine = FakeEngine({"site:blinkit.com": [BLINKIT_SOAP]})
monkeypatch.setattr(search, "search", engine)
discover.run("Dettol", sleep=lambda s: None)
asked = len(engine.asked)
again = discover.run("Dettol", sleep=lambda s: None)
assert len(engine.asked) == asked and again.queries_cached == again.queries_total
def test_the_budget_cuts_the_last_retailers_not_the_last_sub_brands():
planned = discover.plan_queries("Reckitt Benckiser", max_queries=11)
assert {q.retailer for q in planned} == {"blinkit.com"}
assert len({q.target.query for q in planned}) == 11
def test_refused_hits_are_counted_by_reason(monkeypatch):
category = Hit("Dettol - Buy Online | Blinkit", "https://blinkit.com/cn/dettol/cid/1", "")
monkeypatch.setattr(search, "search", FakeEngine({"site:blinkit.com": [category, BLINKIT_SOAP]}))
result = discover.run("Dettol", sleep=lambda s: None)
assert result.rejected == {listings.NOT_A_PRODUCT_PAGE: 1}
assert result.listings_kept == 1
# ---------------------------------------------------------------------------
# jobs
# ---------------------------------------------------------------------------
def test_a_finished_job_is_reused_and_found_from_disk(monkeypatch):
monkeypatch.setattr(search, "search", FakeEngine({"site:blinkit.com": [BLINKIT_SOAP]}))
job = jobs.WebJob(job_id="a" * 32, brand="Dettol")
jobs._jobs[job.job_id] = job
jobs.run_job(job)
jobs._jobs.clear() # as after a restart
found = jobs.candidates_for("dettol")
assert found.job_id == job.job_id and found.candidates[0]["title"] == "Dettol Original Soap 125g"
assert jobs.start_job("Dettol").job_id == job.job_id
def test_a_job_that_was_running_when_the_process_died_is_interrupted():
job = jobs.WebJob(job_id="b" * 32, brand="Dettol", status=jobs.RUNNING)
jobs._save(job)
assert jobs._load(job.job_id).status == jobs.INTERRUPTED
def test_a_job_id_that_is_not_hex_is_never_a_path():
assert jobs.get_job("../../etc/passwd") is None
# ---------------------------------------------------------------------------
# brand_discovery with the web source
# ---------------------------------------------------------------------------
@pytest.fixture
def no_other_sources(monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [])
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [])
monkeypatch.setattr(bd, "_from_brand_store", lambda brand: [])
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [])
def _finished_job(candidates, status=jobs.DONE) -> jobs.WebJob:
job = jobs.WebJob(job_id="c" * 32, brand="Reckitt Benckiser", status=status, backend="duckduckgo",
queries_total=77, queries_done=77, listings_kept=3, candidates=candidates)
jobs._jobs[job.job_id] = job
return job
def _candidate(title, size, providers):
return {"title": title, "size": size, "sizes": [size], "providers": providers,
"retailer_count": len(providers), "price_range": "₹58",
"listings": [{"retailer": p, "url": f"https://{p.lower()}.example/{i}", "title": title}
for i, p in enumerate(providers)],
"brand_term": "dettol", "source": "web"}
def test_web_rows_arrive_with_their_listings_and_honest_confidence(no_other_sources):
job = _finished_job([_candidate("Dettol Original Soap 125g", "125g", ["Blinkit", "Zepto"]),
_candidate("Harpic Power Plus 500ml", "500ml", ["Amazon"])])
result = bd.discover_brand_products("Reckitt Benckiser", use_llm=False, use_web=True,
web_job_id=job.job_id)
rows = {p.product_name: p for p in result.products}
soap = rows["Dettol Original Soap"]
assert soap.size_variants == ["125g"] and soap.evidence == "retail"
assert soap.confidence == bd.WEB_CONFIDENCE_MULTI and soap.as_preview()["selected"] is True
harpic = rows["Harpic Power Plus"]
assert harpic.confidence == bd.WEB_CONFIDENCE_SINGLE and harpic.as_preview()["retailer_count"] == 1
assert soap.as_preview()["listings"][0]["retailer"] == "Blinkit"
assert result.counts["from_web"] == 2
assert not any("language model alone" in w for w in result.warnings)
assert any(w.startswith("Web & retail listings: 2 product(s)") for w in result.warnings)
def test_web_on_but_never_searched_says_so(no_other_sources):
result = bd.discover_brand_products("Reckitt Benckiser", use_llm=False, use_web=True)
assert result.products == []
assert any("have not been searched" in w for w in result.warnings)
def test_a_partial_job_says_it_is_incomplete(no_other_sources):
job = _finished_job([_candidate("Dettol Original Soap 125g", "125g", ["Blinkit"])], status=jobs.PARTIAL)
job.detail = "the search provider stopped answering"
result = bd.discover_brand_products("Reckitt Benckiser", use_llm=False, use_web=True)
assert any("Incomplete: the search provider stopped answering" in w for w in result.warnings)
def test_with_the_web_source_off_nothing_changes(no_other_sources):
_finished_job([_candidate("Dettol Original Soap 125g", "125g", ["Blinkit", "Zepto"])])
result = bd.discover_brand_products("Reckitt Benckiser", use_llm=False)
assert result.products == [] and "from_web" not in result.counts
assert "Every row below rests on the language model alone - review them individually." in result.warnings
def test_a_web_listing_joins_an_open_food_facts_row_instead_of_duplicating_it(no_other_sources, monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
{"title": "Dettol Original Soap", "barcode": "8901396393009", "size": "125g", "source": "off"}])
_finished_job([_candidate("Dettol Original Soap 125g", "125g", ["Blinkit", "Zepto"])])
result = bd.discover_brand_products("Reckitt Benckiser", use_llm=False, use_web=True)
[row] = [p for p in result.products if "Soap" in p.product_name]
assert row.sources == ["off", "web"] and row.barcode == "8901396393009"
assert row.providers[:2] == ["Blinkit", "Zepto"] and row.retailer_count == 2
assert row.evidence == "openfacts"
# ---------------------------------------------------------------------------
# provenance
# ---------------------------------------------------------------------------
class _File:
def __init__(self, products):
self.result = {"products": products}
class _Manifest:
def __init__(self, status, files):
self.status, self.files = status, files
def test_stored_rows_are_joined_back_by_csv_position():
entries = {0: {"retailer_count": 2}, 2: {"retailer_count": 1}}
files = [_File([
{"image_id": "rb_dettol_soap_125g", "brand": "Reckitt Benckiser", "source_row": 2},
{"image_id": "rb_other", "brand": "Reckitt Benckiser", "source_row": 3},
{"image_id": "rb_harpic_500ml", "brand": "Reckitt Benckiser", "source_row": 4},
])]
updates = provenance.web_updates(files, entries)
assert [(u["image_id"], u["entry"]["retailer_count"]) for u in updates] == [
("rb_dettol_soap_125g", 2), ("rb_harpic_500ml", 1)]
def test_the_watcher_waits_for_the_batch_then_applies(monkeypatch):
states = iter([_Manifest("running", []), _Manifest("done", [_File([
{"image_id": "x", "brand": "Reckitt Benckiser", "source_row": 2}])])])
applied = []
monkeypatch.setattr(provenance, "apply", lambda brand, updates: applied.append(updates) or len(updates))
written = provenance.watch_batch("batch1", "Reckitt Benckiser", {0: {"retailer_count": 1}},
read_manifest=lambda _id: next(states), sleep=lambda s: None)
assert written == 1 and applied[0][0]["image_id"] == "x"
class _Cursor:
def __init__(self):
self.calls = []
self.rowcount = 1
def __enter__(self):
return self
def __exit__(self, *exc):
return False
def execute(self, sql, params):
self.calls.append((sql, params))
class _Conn(_Cursor):
def __init__(self):
super().__init__()
self.cur = _Cursor()
def cursor(self):
return self.cur
def close(self):
pass
def test_apply_touches_only_provenance_price_and_review_status(monkeypatch):
conn = _Conn()
monkeypatch.setattr("app.services.vector_store._connect", lambda: conn)
provenance.apply("Reckitt Benckiser", [
{"image_id": "one_shop", "brand": "Reckitt Benckiser",
"entry": {"retailer_count": 1, "price_range": "₹58", "listings": [{"url": "u"}]}},
])
sql, params = conn.cur.calls[0]
assert sql.startswith("UPDATE brand_reckitt_benckiser SET field_sources")
assert "price_range = CASE WHEN COALESCE(price_range, '') = ''" in sql
assert "'rejected'" in sql and params[-2] is True and params[-1] == "one_shop"
for column in ("product_name", "barcode", "image_url", "category"):
assert f"{column} =" not in sql
# ---------------------------------------------------------------------------
# Found on the first live run (Reckitt Benckiser, Blinkit titles)
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("title,name,size", [
("Dettol Original Hand Wash Refill 675 ml Price - Buy Online at \u20b992 in...",
"Dettol Original Hand Wash Refill", "675ml"),
("Harpic Disinfectant Liquid Toilet Cleaner - (Original) - 500 ml Price - Buy Online at Best",
"Harpic Disinfectant Liquid Toilet Cleaner (Original)", "500ml"),
("Lizol Disinfectant Surface & Floor Cleaner (Lavender - 500 ml) Price - Buy Online at Best",
"Lizol Disinfectant Surface & Floor Cleaner (Lavender)", "500ml"),
("Buy Lizol Disinfectant Surface & Floor Cleaner (Citrus, 625 ml) Online",
"Lizol Disinfectant Surface & Floor Cleaner (Citrus)", "625ml"),
])
def test_blinkit_names_keep_their_variant_and_lose_the_price_wording(title, name, size):
term = (title.split()[1] if title.startswith("Buy") else title.split()[0]).lower()
listing, reason = listings.parse_hit(title, "https://blinkit.com/prn/x/prid/1", "", (term,))
assert reason is None and (listing.name, listing.size) == (name, size)
def test_two_scents_of_one_product_stay_two_products():
citrus, _ = listings.parse_hit("Lizol Disinfectant Surface & Floor Cleaner (Citrus) 2 l Price - Buy",
"https://blinkit.com/prn/a/prid/1", "", ("lizol",))
floral, _ = listings.parse_hit("Lizol Disinfectant Surface & Floor Cleaner (Floral) - 2 l Price - Buy",
"https://blinkit.com/prn/b/prid/2", "", ("lizol",))
assert len(discover.cluster([citrus, floral])) == 2
def test_discovery_keeps_two_web_variants_its_own_merge_would_fold(no_other_sources):
"""Its title similarity scores "... Cleaner Citrus" vs "... Cleaner Floral"
at 0.857, over its 0.85 floor; web rows were already told apart."""
_finished_job([_candidate("Lizol Floor Cleaner Citrus 2l", "2l", ["Blinkit"]),
_candidate("Lizol Floor Cleaner Floral 2l", "2l", ["Blinkit"])])
result = bd.discover_brand_products("Reckitt Benckiser", use_llm=False, use_web=True)
assert sorted(p.product_name for p in result.products) == [
"Lizol Floor Cleaner Citrus", "Lizol Floor Cleaner Floral"]
def test_a_retail_listing_outranks_our_own_earlier_output_as_evidence(no_other_sources, monkeypatch):
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [
{"product_name": "Dettol Antiseptic Liquid", "title": "Dettol Antiseptic Liquid"}])
_finished_job([_candidate("Dettol Antiseptic Liquid 250ml", "250ml", ["Blinkit"])])
[row] = bd.discover_brand_products("Reckitt Benckiser", use_llm=False, use_web=True).products
assert row.evidence == "retail" and row.matches_existing == "Dettol Antiseptic Liquid"
def test_bracketed_variant_words_survive_into_the_candidate_title():
a, _ = listings.parse_hit("Lizol Floor Cleaner (Citrus) 2 l Price - Buy",
"https://blinkit.com/prn/a/prid/1", "", ("lizol",))
assert discover.cluster([a])[0]["title"] == "Lizol Floor Cleaner Citrus 2l"

View File

@@ -0,0 +1,127 @@
"""HTTP tests for the web & retail listings additions to /api/admin/brand-discovery.
The batch fixtures mirror test_brand_discovery_api.py (read the docstring
there before removing `batch_root` or `no_background_worker`: without them a
test writes batch manifests into the repository's data directory).
"""
from __future__ import annotations
import pytest
from app.api.routers import brand_discovery as router_module
from app.core import batch_ingest
from app.infrastructure import settings
from app.services import brand_discovery as bd
from app.services.web_discovery import jobs, provenance
PREVIEW = "/api/admin/brand-discovery/preview"
INGEST = "/api/admin/brand-discovery/ingest"
WEB_JOBS = "/api/admin/brand-discovery/web-jobs"
@pytest.fixture(autouse=True)
def _isolate(tmp_path, monkeypatch):
from app.services import active_brands, sku_service
from app.core import batch_worker
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
monkeypatch.setattr(batch_ingest, "BATCH_UPLOAD_DIR", tmp_path / "batch_uploads")
monkeypatch.setattr(batch_worker, "submit", lambda batch_id: None)
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: False)
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: True)
monkeypatch.setattr(settings, "WEB_DISCOVERY_DIR", tmp_path / "web_discovery")
monkeypatch.setattr(router_module, "WEB_DISCOVERY_ENABLED", True)
from app.api.batch_job_store import batch_job_store
batch_job_store._batches.clear()
batch_job_store._cancelled.clear()
jobs._jobs.clear()
yield
batch_job_store._batches.clear()
batch_job_store._cancelled.clear()
jobs._jobs.clear()
@pytest.fixture
def no_worker(monkeypatch):
"""start_job must not spawn a real search thread in a test."""
queued = []
monkeypatch.setattr(jobs, "_ensure_worker", lambda: None)
monkeypatch.setattr(jobs._queue, "put", queued.append)
return queued
def test_the_web_job_routes_are_admin_only(client, user_headers):
assert client.post(WEB_JOBS, json={"brand": "Dettol"}).status_code in (401, 403)
assert client.post(WEB_JOBS, json={"brand": "Dettol"}, headers=user_headers).status_code == 403
assert client.get(f"{WEB_JOBS}/{'a' * 32}").status_code in (401, 403)
def test_starting_a_job_queues_it_and_polling_returns_progress(client, admin_headers, no_worker):
started = client.post(WEB_JOBS, json={"brand": "Reckitt Benckiser"}, headers=admin_headers)
assert started.status_code == 202, started.text
body = started.json()
assert body["status"] == jobs.QUEUED and body["brand"] == "Reckitt Benckiser"
assert "candidates" not in body and body["candidate_count"] == 0
assert no_worker == [body["job_id"]]
polled = client.get(f"{WEB_JOBS}/{body['job_id']}", headers=admin_headers)
assert polled.status_code == 200 and polled.json()["job_id"] == body["job_id"]
def test_a_second_start_while_running_returns_the_same_job(client, admin_headers, no_worker):
first = client.post(WEB_JOBS, json={"brand": "Dettol"}, headers=admin_headers).json()
second = client.post(WEB_JOBS, json={"brand": "dettol"}, headers=admin_headers).json()
assert first["job_id"] == second["job_id"] and len(no_worker) == 1
def test_an_unknown_job_is_404(client, admin_headers):
assert client.get(f"{WEB_JOBS}/{'f' * 32}", headers=admin_headers).status_code == 404
def test_switched_off_the_routes_say_so(client, admin_headers, monkeypatch):
monkeypatch.setattr(router_module, "WEB_DISCOVERY_ENABLED", False)
assert client.post(WEB_JOBS, json={"brand": "Dettol"}, headers=admin_headers).status_code == 404
def test_preview_passes_the_web_flag_through_and_defaults_it_off(client, admin_headers, monkeypatch):
seen = []
def fake_discover(brand, **kwargs):
seen.append(kwargs)
return bd.DiscoveryResult(brand=brand, parent_brand=brand.lower(), table="brand_x",
brand_active=True, filtering_enabled=False)
monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products", fake_discover)
client.post(PREVIEW, json={"brand": "Dettol"}, headers=admin_headers)
client.post(PREVIEW, json={"brand": "Dettol", "use_web": True, "web_job_id": "c" * 32},
headers=admin_headers)
assert (seen[0]["use_web"], seen[0]["web_job_id"]) == (False, None)
assert (seen[1]["use_web"], seen[1]["web_job_id"]) == (True, "c" * 32)
def test_ingest_hands_web_rows_to_the_provenance_watcher_by_csv_position(client, admin_headers, monkeypatch):
watched = []
monkeypatch.setattr(provenance, "watch", lambda batch_id, brand, entries: watched.append(
(batch_id, brand, entries)))
listing = {"retailer": "Blinkit", "url": "https://blinkit.com/prn/x/prid/1", "title": "Dettol Soap 125g"}
resp = client.post(INGEST, json={"brand": "Reckitt Benckiser", "products": [
{"product_name": "Dettol Original Soap", "size_variants": ["125g"]},
{"product_name": "Harpic Power Plus", "size_variants": ["500ml"], "listings": [listing],
"price_range": "₹99", "retailer_count": 1},
]}, headers=admin_headers)
assert resp.status_code == 202, resp.text
[(batch_id, brand, entries)] = watched
assert batch_id == resp.json()["batch_id"] and brand == "Reckitt Benckiser"
assert list(entries) == [1] and entries[1]["listings"] == [listing]
assert entries[1]["price_range"] == "₹99"
def test_ingest_without_web_rows_starts_no_watcher(client, admin_headers, monkeypatch):
watched = []
monkeypatch.setattr(provenance, "watch", lambda *a: watched.append(a))
resp = client.post(INGEST, json={"brand": "Britannia", "products": [
{"product_name": "Marie Gold", "size_variants": ["250g"]}]}, headers=admin_headers)
assert resp.status_code == 202 and watched == []