Brand Ingestion

This commit is contained in:
sriram
2026-09-07 17:52:45 +05:30
parent 75dd3eb3ce
commit 2749bee1a3
11 changed files with 5069 additions and 10 deletions

View File

@@ -0,0 +1,551 @@
"""Tests for brand discovery - the brand-name -> 11-stage-pipeline bridge.
Follows the pattern in test_store_catalog_pipeline.py: monkeypatch the network
and storage boundary *on the module object* (brand_discovery imports those names
directly), and assert on the rows that come out rather than on a status code.
Nothing here reaches Open Food Facts, Ollama or a database. The two source
functions are stubbed and every pipeline run uses `use_llm=False,
fetch_images=False`.
Several of these tests pin behaviour that was WRONG in the first working
version of this module and was found by round-tripping the real Britannia
corpus. They are regression tests with a known failure, not speculative ones -
each names the defect it prevents.
"""
from __future__ import annotations
import pytest
from app.api.routers.user_products import map_spreadsheet_columns, row_to_request
from app.core import store_catalog_pipeline as pipeline
from app.services import brand_discovery as bd
# ---------------------------------------------------------------------------
# Fixtures
# ---------------------------------------------------------------------------
@pytest.fixture(autouse=True)
def _isolate_sku_counter(tmp_path, monkeypatch):
"""Keep the SKU sequence counter out of the repo - see the same fixture in
test_store_catalog_pipeline.py."""
from app.services import sku_service
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
@pytest.fixture(autouse=True)
def _no_network(monkeypatch):
"""Neither source may reach the outside world by default.
A test that wants products stubs one of these explicitly. Without this an
accidental real call would hit Open Food Facts from the suite.
"""
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [])
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [])
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [])
@pytest.fixture
def store(monkeypatch):
"""A fake brand table. Returns the dict of image_id -> stored row."""
table: dict = {}
def fake_upsert(brand, rows, cleanup=False):
assert cleanup is False, "cleanup=True would delete the brand's existing catalog"
for row in rows:
table[row["image_id"]] = row
return len(rows)
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
monkeypatch.setattr(pipeline, "get_products_by_brand",
lambda brand, **kw: list(table.values()))
monkeypatch.setattr(pipeline, "embed_texts",
lambda texts: [[0.0] * 384 for _ in texts])
return table
def _off(title, *, code=None, quantity=None):
"""One Open Food Facts corpus hit, in the shape `_from_open_facts` returns."""
return {"title": title, "barcode": code, "size": bd._canonical_size(quantity),
"source": "off"}
def _run(products, filename="discovered.csv"):
"""Discovered products -> CSV -> the real 11 stages."""
return pipeline.run_pipeline(filename, bd.rows_to_csv_bytes(products),
use_llm=False, fetch_images=False)
# ---------------------------------------------------------------------------
# The contract the whole CSV bridge rests on
# ---------------------------------------------------------------------------
def test_the_csv_headers_all_map_onto_catalog_fields():
"""Every emitted header must be understood by the real column mapper.
This is THE contract: discovery writes a spreadsheet and the pipeline reads
it with `map_spreadsheet_columns`. A header that does not resolve is dropped
in silence, so the column simply never arrives and the run still reports
success. Asserting it here means a rename on either side fails loudly.
"""
mapping = map_spreadsheet_columns(bd.CSV_HEADERS)
assert mapping.unrecognised == []
assert mapping.ignored == []
for header in bd.CSV_HEADERS:
assert header in mapping.columns, f"{header!r} did not resolve to a field"
def test_the_emitted_csv_parses_with_the_real_spreadsheet_reader(monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts",
lambda brand, refresh=False: [_off("Marie Gold", quantity="250 g")])
result = bd.discover_brand_products("Britannia", use_llm=False)
df, mapping = pipeline.parse_spreadsheet("d.csv", bd.rows_to_csv_bytes(result.products))
assert len(df) == 1
assert mapping.unrecognised == []
request = row_to_request(df.to_dict(orient="records")[0], mapping)
assert request.brand == "Britannia"
assert request.product_name == "Marie Gold"
assert request.size_variants == ["250g"]
# ---------------------------------------------------------------------------
# Regression: list cells were shredded by the separator
# ---------------------------------------------------------------------------
def test_a_list_item_containing_a_separator_survives_the_round_trip():
"""`_string_list` splits on [,;|], and the highlight generator emits commas.
"Available in 3 sizes: 100g, 250g, 500g" and "Baked, Not Fried" became five
fragments instead of two highlights. The damage is invisible in the database
- the column is populated, just wrong - so it needs a test.
"""
cell = bd._safe_list_cell([
"Available in 3 sizes: 100g, 250g, 500g",
"Baked, Not Fried",
"Pipe | separated | too",
])
mapping = map_spreadsheet_columns(("highlights",))
from app.api.routers.user_products import _string_list
recovered = _string_list({"highlights": cell}, mapping, "highlights")
assert len(recovered) == 3
assert recovered[0] == "Available in 3 sizes: 100g / 250g / 500g"
assert recovered[1] == "Baked / Not Fried"
# ---------------------------------------------------------------------------
# Regression: pack sizes doubled into the image_id
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("raw, expected", [
("200 g", "200g"),
("200g", "200g"),
("1.5 kg", "1.5kg"),
("75 g", "75g"),
])
def test_a_spaced_pack_size_is_canonicalised(raw, expected):
"""OFF writes both "100g" and "100 g" for the same brand.
`build_image_id` slugifies "200 g" to "200_g", which is not a substring of a
name containing "200g", so the size is appended anyway and the same product
lands on two permanent rows depending on which spelling was discovered.
"""
assert bd._canonical_size(raw) == expected
@pytest.mark.parametrize("junk", ["India", "12", "200", "", None, "6"])
def test_a_quantity_that_is_not_a_pack_size_is_rejected(junk):
"""The Britannia corpus carries "India" and bare counts in `quantity`."""
assert bd._canonical_size(junk) is None
def test_a_size_glued_to_a_letter_in_the_title_is_separated():
"""OFF holds "Jim Jam92 g", where there is no word boundary before the 92.
`off_bulk._SIZE_RE` anchors on one and so cannot see it, which left the
unnormalised size in the product name and produced the image_id
`britannia_jim_jam92_g_40g`.
"""
assert bd._canonicalise_title_size("Jim Jam92 g") == "Jim Jam 92g"
assert bd._canonicalise_title_size("Britannia Good Day 200 g") == "Britannia Good Day 200g"
def test_no_stored_image_id_carries_the_pack_size_twice(store):
result = _run([
bd.DiscoveredProduct(brand="Britannia", product_name="Jim Jam",
title="Jim Jam", category="", category_hint="Biscuits & Cookies",
description="", size_variants=["92g"]),
])
assert result.inserted == 1
assert list(store) == ["britannia_jim_jam_92g"]
# ---------------------------------------------------------------------------
# Regression: the pack size belongs in ONE place
# ---------------------------------------------------------------------------
def test_a_size_in_the_title_is_not_applied_twice(monkeypatch):
"""OFF holds "Britannia Toastea 200g" with a `quantity` of "250g".
Taking the quantity and keeping the title produced the product name
"Britannia Toastea 200g 250g" - two sizes, one name, a pack that does not
exist. The title's own size wins and is removed from the name.
"""
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Britannia Toastea 200g", code="8901063342934", quantity="250 g"),
])
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.product_name == "Britannia Toastea"
assert product.size_variants == ["200g"]
def test_a_case_pack_count_is_not_part_of_the_product_name(monkeypatch):
"""OFF titles carry carton counts: "Good Day Butter Cookies (25)".
That is how many units ship in a box, not part of what the product is
called, and leaving it in put the carton count into the image_id.
"""
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Good Day Butter Cookies (25)", quantity="60 g"),
])
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.product_name == "Good Day Butter Cookies"
assert product.size_variants == ["60g"]
# ---------------------------------------------------------------------------
# Regression: one GTIN, one pack
# ---------------------------------------------------------------------------
def test_a_barcode_is_dropped_when_no_real_pack_size_is_known(monkeypatch):
"""A GTIN identifies one pack, and stage 4 invents three when it has none.
Every column is copied into each exploded variant, so a surviving barcode
would be stamped onto two packs that do not exist - wrong data that looks
authoritative.
"""
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("50 50 Gol Maal", code="8901063017702"),
])
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.size_variants == []
assert product.barcode is None
assert any("barcode dropped" in note for note in product.notes)
def test_a_barcode_survives_when_the_pack_size_is_real(monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Milk Bikis", code="8901063012516", quantity="100 g"),
])
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.barcode == "8901063012516"
assert product.size_variants == ["100g"]
def test_no_gtin_is_written_to_more_than_one_row(store, monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Milk Bikis", code="8901063012516", quantity="100 g"),
_off("50 50 Gol Maal", code="8901063017702"),
])
result = bd.discover_brand_products("Britannia", use_llm=False)
_run(result.products)
barcodes = [row["barcode"] for row in store.values() if row["barcode"]]
assert len(barcodes) == len(set(barcodes))
# ---------------------------------------------------------------------------
# Regression: the generated description poisoned the category and the tax code
# ---------------------------------------------------------------------------
def test_no_category_is_written_when_no_source_stated_one(monkeypatch):
"""Stage 3 owns category detection, and treats a supplied value as final.
It sets `_category_deterministic` and then rewrites the title against the
category, so a wrong guess here does not merely mislabel a row - it renames
the product.
"""
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("50 50 Gol Maal", quantity="100 g"),
])
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.category == ""
assert product.category_hint # still reported, for the preview
def test_a_stated_category_is_carried_through(monkeypatch):
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
{"title": "Good Day Cashew", "category": "Biscuits & Cookies",
"description": "Cashew cookies", "sizes": ["75g"], "providers": [],
"source": "llm"},
])
result = bd.discover_brand_products("Britannia", require_evidence=False)
assert result.products[0].category == "Biscuits & Cookies"
def test_no_boilerplate_description_is_generated(monkeypatch):
"""`generate_detailed_description` contains the word "taste".
`detect_category_from_text` matches keywords fuzzily and "taste" is one edit
from the Oral Care keyword "paste", so every weakly-titled product was
classified Oral Care and stage 9 stamped it with HSN 3306 - the tax code for
dentifrices. A blank description is honest, and `_to_storage_row` already
substitutes "<name> <size> from <brand>."
"""
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("50 50 Gol Maal", quantity="100 g"),
])
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.description == ""
assert "taste" not in (product.description or "").lower()
def test_a_weakly_titled_product_is_not_classified_as_oral_care(store, monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("50 50 Gol Maal", quantity="100 g"),
])
result = bd.discover_brand_products("Britannia", use_llm=False)
_run(result.products)
stored = next(iter(store.values()))
assert stored["category"] != "Oral Care"
assert stored["hsn_code"] != "3306"
# ---------------------------------------------------------------------------
# Regression: catalogue growth by re-run
# ---------------------------------------------------------------------------
def test_a_known_product_is_reemitted_under_its_stored_name(monkeypatch):
"""A brand prefix appearing or disappearing is the realistic drift.
The stored name has no brand prefix and the discovered one does; both
normalise identically once `brand_tokens` are stripped, so the row folds and
the image_id stays put.
"""
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [
{"product_name": "Good Day Cashew Cookies 75g"},
])
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Britannia Good Day Cashew Cookies", quantity="75 g"),
])
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.matches_existing == "Good Day Cashew Cookies"
assert product.product_name == "Good Day Cashew Cookies"
def test_a_shorter_title_does_not_fold_onto_a_longer_stored_one(monkeypatch):
"""The documented limit of the 0.85 threshold, pinned so it is a decision
rather than a surprise.
"Good Day Cashew" scores 0.75 against "Good Day Cashew Cookies" and stays a
separate product. Relaxing the threshold far enough to fold it would also
fold "Dairy Milk Silk" onto "Dairy Milk Silk Minis", which is a different
product - see `_same_product`.
"""
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [
{"product_name": "Good Day Cashew Cookies 75g"},
])
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Good Day Cashew", quantity="75 g"),
])
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.matches_existing is None
def test_two_real_packs_sharing_a_name_stay_on_their_own_rows(monkeypatch):
"""OFF holds "Jim Jam" at 25g and "Jim jam" at 92g.
Both normalise to the same size-free key. Re-emitting the first stored
DISPLAY name gave the 92g pack the 25g name, which `_to_storage_row` then
extended to "Jim Jam 25g 92g" under a brand-new image_id. The index hands
back the base name so the size can be re-applied per pack.
"""
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [
{"product_name": "Jim Jam 25g"},
{"product_name": "Jim jam 92g"},
])
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Jim Jam", quantity="25 g"),
_off("Jim jam", quantity="92 g"),
])
products = bd.discover_brand_products("Britannia", use_llm=False).products
names = {p.product_name for p in products}
assert names == {"Jim Jam"}
assert sorted(s for p in products for s in p.size_variants) == ["25g", "92g"]
assert not any("25g 92g" in p.product_name for p in products)
def test_re_running_discovery_and_ingest_writes_nothing_new(store, monkeypatch):
"""The whole feature's safety property, end to end.
Deterministic image_id + fill-only-blanks + `cleanup=False` make a re-run a
no-op, but only if discovery re-emits the same names. This exercises the
real pipeline twice with the catalog fed back in between.
"""
corpus = [
_off("Britannia Toastea 200g", code="8901063342934", quantity="250 g"),
_off("Good Day Butter Cookies (25)", quantity="60 g"),
_off("Jim Jam92 g", code="8901063019027"),
_off("50 50 Gol Maal", code="8901063017702"),
]
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: corpus)
monkeypatch.setattr(bd, "get_products_by_brand",
lambda brand, **kw: list(store.values()))
first = _run(bd.discover_brand_products("Britannia", use_llm=False).products)
after_first = dict(store)
assert first.inserted > 0
second = _run(bd.discover_brand_products("Britannia", use_llm=False).products)
assert second.inserted == 0
assert second.backfilled == 0
assert second.skipped_existing == len(after_first)
assert set(store) == set(after_first)
# ---------------------------------------------------------------------------
# Sources, evidence and scoring
# ---------------------------------------------------------------------------
def test_open_food_facts_results_survive_an_unreachable_ollama(monkeypatch):
"""OFF runs first and unconditionally, so a dead LLM degrades the result
rather than emptying it - the common case on a CPU-only box."""
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Marie Gold", code="8901063014206", quantity="250 g"),
])
monkeypatch.setattr(bd.ollama_service, "_generate", lambda *a, **k: "")
monkeypatch.setattr(bd.ollama_service, "get_categories_for_brand", lambda brand: [])
result = bd.discover_brand_products("Britannia", use_llm=True)
assert len(result.products) == 1
assert result.products[0].sources == ["off"]
assert any("language model returned no products" in w for w in result.warnings)
def test_a_product_found_by_both_sources_is_one_row_and_scores_highest(monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Marie Gold", code="8901063014206", quantity="250 g"),
])
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
{"title": "Marie Gold", "category": "Biscuits & Cookies",
"description": "Tea-time biscuit", "sizes": ["250g"],
"providers": ["Amazon"], "source": "llm"},
])
result = bd.discover_brand_products("Britannia")
assert len(result.products) == 1
product = result.products[0]
assert sorted(product.sources) == ["llm", "off"]
assert product.confidence == 1.0
assert result.counts["corroborated"] == 1
def test_an_uncorroborated_llm_product_is_dropped_when_evidence_is_required(monkeypatch):
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
{"title": "Britannia Quantum Wafer", "category": None, "description": None,
"sizes": [], "providers": [], "source": "llm"},
])
result = bd.discover_brand_products("Britannia", require_evidence=True)
assert result.products == []
assert result.counts["dropped_without_evidence"] == 1
def test_an_uncorroborated_llm_product_is_kept_but_unticked_when_evidence_is_optional(monkeypatch):
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
{"title": "Britannia Quantum Wafer", "category": None, "description": None,
"sizes": [], "providers": [], "source": "llm"},
])
result = bd.discover_brand_products("Britannia", require_evidence=False)
assert len(result.products) == 1
product = result.products[0]
assert product.evidence is None
assert product.confidence < 0.5
assert product.as_preview()["selected"] is False
def test_a_sub_brand_match_counts_as_registry_evidence(monkeypatch):
""""Good Day" is a registered Britannia sub-brand, so an LLM row naming it
is grounded without needing Open Food Facts."""
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
{"title": "Good Day Chocochip", "category": None, "description": None,
"sizes": ["100g"], "providers": [], "source": "llm"},
])
result = bd.discover_brand_products("Britannia", require_evidence=True)
assert len(result.products) == 1
assert result.products[0].evidence == "registry"
def test_discovery_stops_at_max_products(monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off(f"Product {n}", quantity="100 g") for n in range(50)
])
result = bd.discover_brand_products("Britannia", max_products=10, use_llm=False)
assert len(result.products) == 10
def test_the_brand_table_and_active_state_are_reported(monkeypatch):
monkeypatch.setattr(bd.active_brands, "is_active_brand", lambda brand: False)
monkeypatch.setattr(bd.active_brands, "filtering_enabled", lambda: True)
result = bd.discover_brand_products("Britannia", use_llm=False)
assert result.table == "brand_britannia"
assert result.parent_brand == "britannia"
assert result.brand_active is False
assert result.filtering_enabled is True
def test_an_empty_brand_name_is_refused():
with pytest.raises(ValueError):
bd.discover_brand_products(" ")
# ---------------------------------------------------------------------------
# Field completeness through the real stages
# ---------------------------------------------------------------------------
def test_every_column_the_pipeline_can_fill_is_filled(store, monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Milk Bikis", code="8901063012516", quantity="100 g"),
])
result = bd.discover_brand_products("Britannia", use_llm=False)
_run(result.products)
row = next(iter(store.values()))
for column in ("product_name", "title", "description", "category", "image_id",
"price_range", "size_variants", "providers", "highlights",
"nutrients", "fssai_license", "product_sku", "sku_source",
"search_query"):
assert row.get(column), f"{column} was left empty"
assert row["barcode"] == "8901063012516"
assert row["fssai_license"] == "10012022000103"

View File

@@ -0,0 +1,361 @@
"""HTTP tests for /api/admin/brand-discovery/*.
Reuses the fixture set from test_batch_catalog_ingest.py verbatim - in
particular `batch_root` and `no_background_worker`, whose absence once wrote
batch manifests into the repository's own data directory. Read that fixture's
docstring before removing either from a test here.
Discovery itself is stubbed at the module boundary; the point of this file is
the routes, the ACTIVE_BRANDS gate and the staging handoff, not the merge logic
(which test_brand_discovery.py covers).
"""
from __future__ import annotations
import pytest
from app.api import batch_common
from app.core import batch_ingest
from app.core import store_catalog_pipeline as pipeline
from app.services import brand_discovery as bd
PREVIEW = "/api/admin/brand-discovery/preview"
INGEST = "/api/admin/brand-discovery/ingest"
# ---------------------------------------------------------------------------
# Fixtures - see test_batch_catalog_ingest.py for the rationale behind each
# ---------------------------------------------------------------------------
@pytest.fixture(autouse=True)
def _isolate_sku_counter(tmp_path, monkeypatch):
from app.services import sku_service
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
@pytest.fixture(autouse=True)
def batch_root(tmp_path, monkeypatch):
root = tmp_path / "batch_uploads"
monkeypatch.setattr(batch_ingest, "BATCH_UPLOAD_DIR", root)
return root
@pytest.fixture(autouse=True)
def no_background_worker(monkeypatch):
"""Stub the worker. Its absence once wrote manifests into the working tree -
the long docstring in test_batch_catalog_ingest.py explains how."""
from app.core import batch_worker
submitted: list = []
monkeypatch.setattr(batch_worker, "submit", submitted.append)
return submitted
@pytest.fixture(autouse=True)
def _clean_job_store():
"""The job store is a module singleton and outlives a test - see the same
fixture in test_uploads_api.py."""
from app.api.batch_job_store import batch_job_store
batch_job_store._batches.clear()
batch_job_store._cancelled.clear()
yield
batch_job_store._batches.clear()
batch_job_store._cancelled.clear()
@pytest.fixture(autouse=True)
def _active_brands_allow_everything(monkeypatch):
"""Filtering off by default, so only the tests that are about the
ACTIVE_BRANDS gate have to think about it."""
from app.services import active_brands
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: False)
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: True)
@pytest.fixture
def discovered(monkeypatch):
"""A fixed two-product discovery result."""
def fake_discover(brand, **kwargs):
return bd.DiscoveryResult(
brand=brand,
parent_brand="britannia",
table="brand_britannia",
brand_active=True,
filtering_enabled=False,
products=[
bd.DiscoveredProduct(
brand=brand, product_name="Marie Gold", title="Marie Gold",
category="", category_hint="Biscuits & Cookies", description="",
size_variants=["250g"], barcode="8901063014206",
sources=["off"], evidence="openfacts", confidence=0.9,
),
bd.DiscoveredProduct(
brand=brand, product_name="Milk Bikis", title="Milk Bikis",
category="", category_hint="Biscuits & Cookies", description="",
size_variants=["100g"], barcode="8901063012516",
sources=["off"], evidence="openfacts", confidence=0.9,
),
],
counts={"discovered": 2},
)
from app.api.routers import brand_discovery as router_module
monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products",
fake_discover)
return fake_discover
def _payload(**overrides):
body = {
"brand": "Britannia",
"products": [
{"product_name": "Marie Gold", "size_variants": ["250g"],
"barcode": "8901063014206", "providers": ["Amazon"],
"highlights": ["Tea-time favourite"], "nutrients": ["Iron - Blood health"]},
],
}
body.update(overrides)
return body
# ---------------------------------------------------------------------------
# Auth
# ---------------------------------------------------------------------------
def test_preview_requires_admin(client):
assert client.post(PREVIEW, json={"brand": "Britannia"}).status_code in (401, 403)
def test_ingest_requires_admin(client):
assert client.post(INGEST, json=_payload()).status_code in (401, 403)
def test_a_plain_user_cannot_discover(client, user_headers):
resp = client.post(PREVIEW, json={"brand": "Britannia"}, headers=user_headers)
assert resp.status_code == 403
# ---------------------------------------------------------------------------
# Preview writes nothing
# ---------------------------------------------------------------------------
def test_preview_returns_products_and_the_stage_list(client, admin_headers, discovered):
resp = client.post(PREVIEW, json={"brand": "Britannia"}, headers=admin_headers)
assert resp.status_code == 200
body = resp.json()
assert len(body["products"]) == 2
assert body["table"] == "brand_britannia"
assert body["stages"] == list(pipeline.STAGE_NAMES)
assert len(body["stages"]) == 11
assert body["products"][0]["selected"] is True
def test_preview_stages_nothing(client, admin_headers, discovered, batch_root,
no_background_worker):
"""The whole reason preview is a separate route."""
client.post(PREVIEW, json={"brand": "Britannia"}, headers=admin_headers)
assert no_background_worker == []
assert not batch_root.exists() or list(batch_root.iterdir()) == []
def test_preview_reports_a_bad_brand_name_as_client_error(client, admin_headers):
resp = client.post(PREVIEW, json={"brand": " "}, headers=admin_headers)
assert resp.status_code == 400
def test_preview_surfaces_a_discovery_failure_rather_than_500(client, admin_headers,
monkeypatch):
from app.api.routers import brand_discovery as router_module
def boom(brand, **kwargs):
raise RuntimeError("Open Food Facts is unreachable")
monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products", boom)
resp = client.post(PREVIEW, json={"brand": "Britannia"}, headers=admin_headers)
assert resp.status_code == 502
assert "unreachable" in resp.json()["detail"]
# ---------------------------------------------------------------------------
# The ACTIVE_BRANDS gate
# ---------------------------------------------------------------------------
def test_preview_warns_when_the_brand_is_not_active(client, admin_headers, discovered,
monkeypatch):
from app.services import active_brands
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: True)
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: False)
monkeypatch.setattr(active_brands, "active_display_names", lambda: ["Amul"])
from app.api.routers import brand_discovery as router_module
def inactive(brand, **kwargs):
result = discovered(brand, **kwargs)
result.brand_active = False
result.filtering_enabled = True
return result
monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products", inactive)
body = client.post(PREVIEW, json={"brand": "Britannia"},
headers=admin_headers).json()
assert body["brand_active"] is False
assert any("ACTIVE_BRANDS" in w for w in body["warnings"])
def test_ingest_is_refused_for_an_inactive_brand(client, admin_headers, monkeypatch):
"""A green run over a catalog no endpoint can read is not an acceptable
outcome to hand back silently."""
from app.services import active_brands
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: True)
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: False)
monkeypatch.setattr(active_brands, "active_display_names", lambda: ["Amul", "Cadbury"])
resp = client.post(INGEST, json=_payload(), headers=admin_headers)
assert resp.status_code == 409
detail = resp.json()["detail"]
assert "ACTIVE_BRANDS=Amul,Cadbury,Britannia" in detail
assert "restart" in detail.lower()
def test_ingest_proceeds_for_an_inactive_brand_when_acknowledged(client, admin_headers,
monkeypatch):
from app.services import active_brands
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: True)
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: False)
monkeypatch.setattr(active_brands, "active_display_names", lambda: ["Amul"])
resp = client.post(INGEST, json=_payload(acknowledge_inactive=True),
headers=admin_headers)
assert resp.status_code == 202
# ---------------------------------------------------------------------------
# Ingest
# ---------------------------------------------------------------------------
def test_ingest_stages_one_batch_and_returns_a_pollable_id(client, admin_headers,
no_background_worker):
resp = client.post(INGEST, json=_payload(), headers=admin_headers)
assert resp.status_code == 202
body = resp.json()
assert body["batch_id"]
assert body["files_total"] == 1
assert len(body["stage_names"]) == 11
assert no_background_worker == [body["batch_id"]]
def test_the_staged_file_is_a_real_csv_the_pipeline_can_read(client, admin_headers,
batch_root):
"""Not a placeholder: the bytes on the batch volume are the exact input that
produced the rows, and they must parse with the same reader an upload uses."""
batch_id = client.post(INGEST, json=_payload(), headers=admin_headers).json()["batch_id"]
manifest = batch_ingest.read_manifest(batch_id)
entry = manifest.files[0]
assert entry.filename.startswith("discovered-britannia-")
assert entry.filename.endswith(".csv")
contents = (batch_ingest.batch_dir(batch_id) / entry.stored_name).read_bytes()
df, mapping = pipeline.parse_spreadsheet(entry.filename, contents)
assert len(df) == 1
assert mapping.unrecognised == []
assert "product_name" in mapping.columns
def test_the_batch_records_where_it_came_from(client, admin_headers):
batch_id = client.post(INGEST, json=_payload(), headers=admin_headers).json()["batch_id"]
manifest = batch_ingest.read_manifest(batch_id)
assert manifest.submitted_by == "brand-discovery: Britannia"
def test_ingest_runs_the_eleven_stages_end_to_end(client, admin_headers, store,
no_background_worker, monkeypatch):
"""Drive the real worker function over the staged file.
`use_llm` and `fetch_images` are turned OFF for this one, matching every
other pipeline test in the suite: with them on, `run_batch` really calls
Ollama once per row and really searches the open web for images. That is the
correct production default and a terrible test - slow, and it fails when the
machine is offline.
"""
body = _payload(use_llm=False, fetch_images=False)
batch_id = client.post(INGEST, json=body, headers=admin_headers).json()["batch_id"]
batch_ingest.run_batch(batch_id)
assert len(store) == 1
row = next(iter(store.values()))
assert row["image_id"] == "britannia_marie_gold_250g"
assert row["barcode"] == "8901063014206"
assert row["product_sku"]
assert row["highlights"] == ["Tea-time favourite"]
def test_an_empty_selection_is_refused(client, admin_headers):
resp = client.post(INGEST, json=_payload(products=[]), headers=admin_headers)
assert resp.status_code == 400
assert "nothing to ingest" in resp.json()["detail"]
def test_a_blank_brand_is_refused(client, admin_headers):
resp = client.post(INGEST, json=_payload(brand=" "), headers=admin_headers)
assert resp.status_code == 400
def test_a_full_queue_is_reported_as_429(client, admin_headers, monkeypatch):
import queue
def full(batch_id):
raise queue.Full()
from app.core import batch_worker
monkeypatch.setattr(batch_worker, "submit", full)
resp = client.post(INGEST, json=_payload(), headers=admin_headers)
assert resp.status_code == 429
assert "Resume" in resp.json()["detail"]
def test_ingest_defaults_turn_on_images_and_the_per_row_llm(client, admin_headers,
monkeypatch):
"""The two enrichment stages this feature runs with. Both are per-run
arguments already, so neither needs a settings change."""
seen = {}
real = batch_common.stage_and_queue
def spy(valid, invalid, *, use_llm, fetch_images, submitted_by=None):
seen["use_llm"] = use_llm
seen["fetch_images"] = fetch_images
return real(valid, invalid, use_llm=use_llm, fetch_images=fetch_images,
submitted_by=submitted_by)
from app.api.routers import brand_discovery as router_module
monkeypatch.setattr(router_module.batch_common, "stage_and_queue", spy)
client.post(INGEST, json=_payload(), headers=admin_headers)
assert seen == {"use_llm": True, "fetch_images": True}
@pytest.fixture
def store(monkeypatch):
table: dict = {}
def fake_upsert(brand, rows, cleanup=False):
assert cleanup is False, "cleanup=True would delete the brand's existing catalog"
for row in rows:
table[row["image_id"]] = dict(row)
return len(rows)
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: list(table.values()))
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
return table