Brand Ingestion
This commit is contained in:
551
tests/test_brand_discovery.py
Normal file
551
tests/test_brand_discovery.py
Normal file
@@ -0,0 +1,551 @@
|
||||
"""Tests for brand discovery - the brand-name -> 11-stage-pipeline bridge.
|
||||
|
||||
Follows the pattern in test_store_catalog_pipeline.py: monkeypatch the network
|
||||
and storage boundary *on the module object* (brand_discovery imports those names
|
||||
directly), and assert on the rows that come out rather than on a status code.
|
||||
|
||||
Nothing here reaches Open Food Facts, Ollama or a database. The two source
|
||||
functions are stubbed and every pipeline run uses `use_llm=False,
|
||||
fetch_images=False`.
|
||||
|
||||
Several of these tests pin behaviour that was WRONG in the first working
|
||||
version of this module and was found by round-tripping the real Britannia
|
||||
corpus. They are regression tests with a known failure, not speculative ones -
|
||||
each names the defect it prevents.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from app.api.routers.user_products import map_spreadsheet_columns, row_to_request
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.services import brand_discovery as bd
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Fixtures
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolate_sku_counter(tmp_path, monkeypatch):
|
||||
"""Keep the SKU sequence counter out of the repo - see the same fixture in
|
||||
test_store_catalog_pipeline.py."""
|
||||
from app.services import sku_service
|
||||
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _no_network(monkeypatch):
|
||||
"""Neither source may reach the outside world by default.
|
||||
|
||||
A test that wants products stubs one of these explicitly. Without this an
|
||||
accidental real call would hit Open Food Facts from the suite.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [])
|
||||
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [])
|
||||
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [])
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store(monkeypatch):
|
||||
"""A fake brand table. Returns the dict of image_id -> stored row."""
|
||||
table: dict = {}
|
||||
|
||||
def fake_upsert(brand, rows, cleanup=False):
|
||||
assert cleanup is False, "cleanup=True would delete the brand's existing catalog"
|
||||
for row in rows:
|
||||
table[row["image_id"]] = row
|
||||
return len(rows)
|
||||
|
||||
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
|
||||
monkeypatch.setattr(pipeline, "get_products_by_brand",
|
||||
lambda brand, **kw: list(table.values()))
|
||||
monkeypatch.setattr(pipeline, "embed_texts",
|
||||
lambda texts: [[0.0] * 384 for _ in texts])
|
||||
return table
|
||||
|
||||
|
||||
def _off(title, *, code=None, quantity=None):
|
||||
"""One Open Food Facts corpus hit, in the shape `_from_open_facts` returns."""
|
||||
return {"title": title, "barcode": code, "size": bd._canonical_size(quantity),
|
||||
"source": "off"}
|
||||
|
||||
|
||||
def _run(products, filename="discovered.csv"):
|
||||
"""Discovered products -> CSV -> the real 11 stages."""
|
||||
return pipeline.run_pipeline(filename, bd.rows_to_csv_bytes(products),
|
||||
use_llm=False, fetch_images=False)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The contract the whole CSV bridge rests on
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_the_csv_headers_all_map_onto_catalog_fields():
|
||||
"""Every emitted header must be understood by the real column mapper.
|
||||
|
||||
This is THE contract: discovery writes a spreadsheet and the pipeline reads
|
||||
it with `map_spreadsheet_columns`. A header that does not resolve is dropped
|
||||
in silence, so the column simply never arrives and the run still reports
|
||||
success. Asserting it here means a rename on either side fails loudly.
|
||||
"""
|
||||
mapping = map_spreadsheet_columns(bd.CSV_HEADERS)
|
||||
|
||||
assert mapping.unrecognised == []
|
||||
assert mapping.ignored == []
|
||||
for header in bd.CSV_HEADERS:
|
||||
assert header in mapping.columns, f"{header!r} did not resolve to a field"
|
||||
|
||||
|
||||
def test_the_emitted_csv_parses_with_the_real_spreadsheet_reader(monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_open_facts",
|
||||
lambda brand, refresh=False: [_off("Marie Gold", quantity="250 g")])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", use_llm=False)
|
||||
df, mapping = pipeline.parse_spreadsheet("d.csv", bd.rows_to_csv_bytes(result.products))
|
||||
|
||||
assert len(df) == 1
|
||||
assert mapping.unrecognised == []
|
||||
request = row_to_request(df.to_dict(orient="records")[0], mapping)
|
||||
assert request.brand == "Britannia"
|
||||
assert request.product_name == "Marie Gold"
|
||||
assert request.size_variants == ["250g"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Regression: list cells were shredded by the separator
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_list_item_containing_a_separator_survives_the_round_trip():
|
||||
"""`_string_list` splits on [,;|], and the highlight generator emits commas.
|
||||
|
||||
"Available in 3 sizes: 100g, 250g, 500g" and "Baked, Not Fried" became five
|
||||
fragments instead of two highlights. The damage is invisible in the database
|
||||
- the column is populated, just wrong - so it needs a test.
|
||||
"""
|
||||
cell = bd._safe_list_cell([
|
||||
"Available in 3 sizes: 100g, 250g, 500g",
|
||||
"Baked, Not Fried",
|
||||
"Pipe | separated | too",
|
||||
])
|
||||
mapping = map_spreadsheet_columns(("highlights",))
|
||||
from app.api.routers.user_products import _string_list
|
||||
|
||||
recovered = _string_list({"highlights": cell}, mapping, "highlights")
|
||||
|
||||
assert len(recovered) == 3
|
||||
assert recovered[0] == "Available in 3 sizes: 100g / 250g / 500g"
|
||||
assert recovered[1] == "Baked / Not Fried"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Regression: pack sizes doubled into the image_id
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.mark.parametrize("raw, expected", [
|
||||
("200 g", "200g"),
|
||||
("200g", "200g"),
|
||||
("1.5 kg", "1.5kg"),
|
||||
("75 g", "75g"),
|
||||
])
|
||||
def test_a_spaced_pack_size_is_canonicalised(raw, expected):
|
||||
"""OFF writes both "100g" and "100 g" for the same brand.
|
||||
|
||||
`build_image_id` slugifies "200 g" to "200_g", which is not a substring of a
|
||||
name containing "200g", so the size is appended anyway and the same product
|
||||
lands on two permanent rows depending on which spelling was discovered.
|
||||
"""
|
||||
assert bd._canonical_size(raw) == expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize("junk", ["India", "12", "200", "", None, "6"])
|
||||
def test_a_quantity_that_is_not_a_pack_size_is_rejected(junk):
|
||||
"""The Britannia corpus carries "India" and bare counts in `quantity`."""
|
||||
assert bd._canonical_size(junk) is None
|
||||
|
||||
|
||||
def test_a_size_glued_to_a_letter_in_the_title_is_separated():
|
||||
"""OFF holds "Jim Jam92 g", where there is no word boundary before the 92.
|
||||
|
||||
`off_bulk._SIZE_RE` anchors on one and so cannot see it, which left the
|
||||
unnormalised size in the product name and produced the image_id
|
||||
`britannia_jim_jam92_g_40g`.
|
||||
"""
|
||||
assert bd._canonicalise_title_size("Jim Jam92 g") == "Jim Jam 92g"
|
||||
assert bd._canonicalise_title_size("Britannia Good Day 200 g") == "Britannia Good Day 200g"
|
||||
|
||||
|
||||
def test_no_stored_image_id_carries_the_pack_size_twice(store):
|
||||
result = _run([
|
||||
bd.DiscoveredProduct(brand="Britannia", product_name="Jim Jam",
|
||||
title="Jim Jam", category="", category_hint="Biscuits & Cookies",
|
||||
description="", size_variants=["92g"]),
|
||||
])
|
||||
assert result.inserted == 1
|
||||
assert list(store) == ["britannia_jim_jam_92g"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Regression: the pack size belongs in ONE place
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_size_in_the_title_is_not_applied_twice(monkeypatch):
|
||||
"""OFF holds "Britannia Toastea 200g" with a `quantity` of "250g".
|
||||
|
||||
Taking the quantity and keeping the title produced the product name
|
||||
"Britannia Toastea 200g 250g" - two sizes, one name, a pack that does not
|
||||
exist. The title's own size wins and is removed from the name.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Britannia Toastea 200g", code="8901063342934", quantity="250 g"),
|
||||
])
|
||||
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.product_name == "Britannia Toastea"
|
||||
assert product.size_variants == ["200g"]
|
||||
|
||||
|
||||
def test_a_case_pack_count_is_not_part_of_the_product_name(monkeypatch):
|
||||
"""OFF titles carry carton counts: "Good Day Butter Cookies (25)".
|
||||
|
||||
That is how many units ship in a box, not part of what the product is
|
||||
called, and leaving it in put the carton count into the image_id.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Good Day Butter Cookies (25)", quantity="60 g"),
|
||||
])
|
||||
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.product_name == "Good Day Butter Cookies"
|
||||
assert product.size_variants == ["60g"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Regression: one GTIN, one pack
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_barcode_is_dropped_when_no_real_pack_size_is_known(monkeypatch):
|
||||
"""A GTIN identifies one pack, and stage 4 invents three when it has none.
|
||||
|
||||
Every column is copied into each exploded variant, so a surviving barcode
|
||||
would be stamped onto two packs that do not exist - wrong data that looks
|
||||
authoritative.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("50 50 Gol Maal", code="8901063017702"),
|
||||
])
|
||||
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.size_variants == []
|
||||
assert product.barcode is None
|
||||
assert any("barcode dropped" in note for note in product.notes)
|
||||
|
||||
|
||||
def test_a_barcode_survives_when_the_pack_size_is_real(monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Milk Bikis", code="8901063012516", quantity="100 g"),
|
||||
])
|
||||
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.barcode == "8901063012516"
|
||||
assert product.size_variants == ["100g"]
|
||||
|
||||
|
||||
def test_no_gtin_is_written_to_more_than_one_row(store, monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Milk Bikis", code="8901063012516", quantity="100 g"),
|
||||
_off("50 50 Gol Maal", code="8901063017702"),
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", use_llm=False)
|
||||
_run(result.products)
|
||||
|
||||
barcodes = [row["barcode"] for row in store.values() if row["barcode"]]
|
||||
assert len(barcodes) == len(set(barcodes))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Regression: the generated description poisoned the category and the tax code
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_no_category_is_written_when_no_source_stated_one(monkeypatch):
|
||||
"""Stage 3 owns category detection, and treats a supplied value as final.
|
||||
|
||||
It sets `_category_deterministic` and then rewrites the title against the
|
||||
category, so a wrong guess here does not merely mislabel a row - it renames
|
||||
the product.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("50 50 Gol Maal", quantity="100 g"),
|
||||
])
|
||||
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.category == ""
|
||||
assert product.category_hint # still reported, for the preview
|
||||
|
||||
|
||||
def test_a_stated_category_is_carried_through(monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
|
||||
{"title": "Good Day Cashew", "category": "Biscuits & Cookies",
|
||||
"description": "Cashew cookies", "sizes": ["75g"], "providers": [],
|
||||
"source": "llm"},
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", require_evidence=False)
|
||||
|
||||
assert result.products[0].category == "Biscuits & Cookies"
|
||||
|
||||
|
||||
def test_no_boilerplate_description_is_generated(monkeypatch):
|
||||
"""`generate_detailed_description` contains the word "taste".
|
||||
|
||||
`detect_category_from_text` matches keywords fuzzily and "taste" is one edit
|
||||
from the Oral Care keyword "paste", so every weakly-titled product was
|
||||
classified Oral Care and stage 9 stamped it with HSN 3306 - the tax code for
|
||||
dentifrices. A blank description is honest, and `_to_storage_row` already
|
||||
substitutes "<name> <size> from <brand>."
|
||||
"""
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("50 50 Gol Maal", quantity="100 g"),
|
||||
])
|
||||
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.description == ""
|
||||
assert "taste" not in (product.description or "").lower()
|
||||
|
||||
|
||||
def test_a_weakly_titled_product_is_not_classified_as_oral_care(store, monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("50 50 Gol Maal", quantity="100 g"),
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", use_llm=False)
|
||||
_run(result.products)
|
||||
|
||||
stored = next(iter(store.values()))
|
||||
assert stored["category"] != "Oral Care"
|
||||
assert stored["hsn_code"] != "3306"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Regression: catalogue growth by re-run
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_known_product_is_reemitted_under_its_stored_name(monkeypatch):
|
||||
"""A brand prefix appearing or disappearing is the realistic drift.
|
||||
|
||||
The stored name has no brand prefix and the discovered one does; both
|
||||
normalise identically once `brand_tokens` are stripped, so the row folds and
|
||||
the image_id stays put.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [
|
||||
{"product_name": "Good Day Cashew Cookies 75g"},
|
||||
])
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Britannia Good Day Cashew Cookies", quantity="75 g"),
|
||||
])
|
||||
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.matches_existing == "Good Day Cashew Cookies"
|
||||
assert product.product_name == "Good Day Cashew Cookies"
|
||||
|
||||
|
||||
def test_a_shorter_title_does_not_fold_onto_a_longer_stored_one(monkeypatch):
|
||||
"""The documented limit of the 0.85 threshold, pinned so it is a decision
|
||||
rather than a surprise.
|
||||
|
||||
"Good Day Cashew" scores 0.75 against "Good Day Cashew Cookies" and stays a
|
||||
separate product. Relaxing the threshold far enough to fold it would also
|
||||
fold "Dairy Milk Silk" onto "Dairy Milk Silk Minis", which is a different
|
||||
product - see `_same_product`.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [
|
||||
{"product_name": "Good Day Cashew Cookies 75g"},
|
||||
])
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Good Day Cashew", quantity="75 g"),
|
||||
])
|
||||
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.matches_existing is None
|
||||
|
||||
|
||||
def test_two_real_packs_sharing_a_name_stay_on_their_own_rows(monkeypatch):
|
||||
"""OFF holds "Jim Jam" at 25g and "Jim jam" at 92g.
|
||||
|
||||
Both normalise to the same size-free key. Re-emitting the first stored
|
||||
DISPLAY name gave the 92g pack the 25g name, which `_to_storage_row` then
|
||||
extended to "Jim Jam 25g 92g" under a brand-new image_id. The index hands
|
||||
back the base name so the size can be re-applied per pack.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [
|
||||
{"product_name": "Jim Jam 25g"},
|
||||
{"product_name": "Jim jam 92g"},
|
||||
])
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Jim Jam", quantity="25 g"),
|
||||
_off("Jim jam", quantity="92 g"),
|
||||
])
|
||||
|
||||
products = bd.discover_brand_products("Britannia", use_llm=False).products
|
||||
names = {p.product_name for p in products}
|
||||
|
||||
assert names == {"Jim Jam"}
|
||||
assert sorted(s for p in products for s in p.size_variants) == ["25g", "92g"]
|
||||
assert not any("25g 92g" in p.product_name for p in products)
|
||||
|
||||
|
||||
def test_re_running_discovery_and_ingest_writes_nothing_new(store, monkeypatch):
|
||||
"""The whole feature's safety property, end to end.
|
||||
|
||||
Deterministic image_id + fill-only-blanks + `cleanup=False` make a re-run a
|
||||
no-op, but only if discovery re-emits the same names. This exercises the
|
||||
real pipeline twice with the catalog fed back in between.
|
||||
"""
|
||||
corpus = [
|
||||
_off("Britannia Toastea 200g", code="8901063342934", quantity="250 g"),
|
||||
_off("Good Day Butter Cookies (25)", quantity="60 g"),
|
||||
_off("Jim Jam92 g", code="8901063019027"),
|
||||
_off("50 50 Gol Maal", code="8901063017702"),
|
||||
]
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: corpus)
|
||||
monkeypatch.setattr(bd, "get_products_by_brand",
|
||||
lambda brand, **kw: list(store.values()))
|
||||
|
||||
first = _run(bd.discover_brand_products("Britannia", use_llm=False).products)
|
||||
after_first = dict(store)
|
||||
assert first.inserted > 0
|
||||
|
||||
second = _run(bd.discover_brand_products("Britannia", use_llm=False).products)
|
||||
|
||||
assert second.inserted == 0
|
||||
assert second.backfilled == 0
|
||||
assert second.skipped_existing == len(after_first)
|
||||
assert set(store) == set(after_first)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Sources, evidence and scoring
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_open_food_facts_results_survive_an_unreachable_ollama(monkeypatch):
|
||||
"""OFF runs first and unconditionally, so a dead LLM degrades the result
|
||||
rather than emptying it - the common case on a CPU-only box."""
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Marie Gold", code="8901063014206", quantity="250 g"),
|
||||
])
|
||||
monkeypatch.setattr(bd.ollama_service, "_generate", lambda *a, **k: "")
|
||||
monkeypatch.setattr(bd.ollama_service, "get_categories_for_brand", lambda brand: [])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", use_llm=True)
|
||||
|
||||
assert len(result.products) == 1
|
||||
assert result.products[0].sources == ["off"]
|
||||
assert any("language model returned no products" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_a_product_found_by_both_sources_is_one_row_and_scores_highest(monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Marie Gold", code="8901063014206", quantity="250 g"),
|
||||
])
|
||||
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
|
||||
{"title": "Marie Gold", "category": "Biscuits & Cookies",
|
||||
"description": "Tea-time biscuit", "sizes": ["250g"],
|
||||
"providers": ["Amazon"], "source": "llm"},
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia")
|
||||
|
||||
assert len(result.products) == 1
|
||||
product = result.products[0]
|
||||
assert sorted(product.sources) == ["llm", "off"]
|
||||
assert product.confidence == 1.0
|
||||
assert result.counts["corroborated"] == 1
|
||||
|
||||
|
||||
def test_an_uncorroborated_llm_product_is_dropped_when_evidence_is_required(monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
|
||||
{"title": "Britannia Quantum Wafer", "category": None, "description": None,
|
||||
"sizes": [], "providers": [], "source": "llm"},
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", require_evidence=True)
|
||||
|
||||
assert result.products == []
|
||||
assert result.counts["dropped_without_evidence"] == 1
|
||||
|
||||
|
||||
def test_an_uncorroborated_llm_product_is_kept_but_unticked_when_evidence_is_optional(monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
|
||||
{"title": "Britannia Quantum Wafer", "category": None, "description": None,
|
||||
"sizes": [], "providers": [], "source": "llm"},
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", require_evidence=False)
|
||||
|
||||
assert len(result.products) == 1
|
||||
product = result.products[0]
|
||||
assert product.evidence is None
|
||||
assert product.confidence < 0.5
|
||||
assert product.as_preview()["selected"] is False
|
||||
|
||||
|
||||
def test_a_sub_brand_match_counts_as_registry_evidence(monkeypatch):
|
||||
""""Good Day" is a registered Britannia sub-brand, so an LLM row naming it
|
||||
is grounded without needing Open Food Facts."""
|
||||
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
|
||||
{"title": "Good Day Chocochip", "category": None, "description": None,
|
||||
"sizes": ["100g"], "providers": [], "source": "llm"},
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", require_evidence=True)
|
||||
|
||||
assert len(result.products) == 1
|
||||
assert result.products[0].evidence == "registry"
|
||||
|
||||
|
||||
def test_discovery_stops_at_max_products(monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off(f"Product {n}", quantity="100 g") for n in range(50)
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", max_products=10, use_llm=False)
|
||||
|
||||
assert len(result.products) == 10
|
||||
|
||||
|
||||
def test_the_brand_table_and_active_state_are_reported(monkeypatch):
|
||||
monkeypatch.setattr(bd.active_brands, "is_active_brand", lambda brand: False)
|
||||
monkeypatch.setattr(bd.active_brands, "filtering_enabled", lambda: True)
|
||||
|
||||
result = bd.discover_brand_products("Britannia", use_llm=False)
|
||||
|
||||
assert result.table == "brand_britannia"
|
||||
assert result.parent_brand == "britannia"
|
||||
assert result.brand_active is False
|
||||
assert result.filtering_enabled is True
|
||||
|
||||
|
||||
def test_an_empty_brand_name_is_refused():
|
||||
with pytest.raises(ValueError):
|
||||
bd.discover_brand_products(" ")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Field completeness through the real stages
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_every_column_the_pipeline_can_fill_is_filled(store, monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Milk Bikis", code="8901063012516", quantity="100 g"),
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", use_llm=False)
|
||||
_run(result.products)
|
||||
|
||||
row = next(iter(store.values()))
|
||||
for column in ("product_name", "title", "description", "category", "image_id",
|
||||
"price_range", "size_variants", "providers", "highlights",
|
||||
"nutrients", "fssai_license", "product_sku", "sku_source",
|
||||
"search_query"):
|
||||
assert row.get(column), f"{column} was left empty"
|
||||
assert row["barcode"] == "8901063012516"
|
||||
assert row["fssai_license"] == "10012022000103"
|
||||
361
tests/test_brand_discovery_api.py
Normal file
361
tests/test_brand_discovery_api.py
Normal file
@@ -0,0 +1,361 @@
|
||||
"""HTTP tests for /api/admin/brand-discovery/*.
|
||||
|
||||
Reuses the fixture set from test_batch_catalog_ingest.py verbatim - in
|
||||
particular `batch_root` and `no_background_worker`, whose absence once wrote
|
||||
batch manifests into the repository's own data directory. Read that fixture's
|
||||
docstring before removing either from a test here.
|
||||
|
||||
Discovery itself is stubbed at the module boundary; the point of this file is
|
||||
the routes, the ACTIVE_BRANDS gate and the staging handoff, not the merge logic
|
||||
(which test_brand_discovery.py covers).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from app.api import batch_common
|
||||
from app.core import batch_ingest
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.services import brand_discovery as bd
|
||||
|
||||
PREVIEW = "/api/admin/brand-discovery/preview"
|
||||
INGEST = "/api/admin/brand-discovery/ingest"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Fixtures - see test_batch_catalog_ingest.py for the rationale behind each
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolate_sku_counter(tmp_path, monkeypatch):
|
||||
from app.services import sku_service
|
||||
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def batch_root(tmp_path, monkeypatch):
|
||||
root = tmp_path / "batch_uploads"
|
||||
monkeypatch.setattr(batch_ingest, "BATCH_UPLOAD_DIR", root)
|
||||
return root
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def no_background_worker(monkeypatch):
|
||||
"""Stub the worker. Its absence once wrote manifests into the working tree -
|
||||
the long docstring in test_batch_catalog_ingest.py explains how."""
|
||||
from app.core import batch_worker
|
||||
|
||||
submitted: list = []
|
||||
monkeypatch.setattr(batch_worker, "submit", submitted.append)
|
||||
return submitted
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _clean_job_store():
|
||||
"""The job store is a module singleton and outlives a test - see the same
|
||||
fixture in test_uploads_api.py."""
|
||||
from app.api.batch_job_store import batch_job_store
|
||||
batch_job_store._batches.clear()
|
||||
batch_job_store._cancelled.clear()
|
||||
yield
|
||||
batch_job_store._batches.clear()
|
||||
batch_job_store._cancelled.clear()
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _active_brands_allow_everything(monkeypatch):
|
||||
"""Filtering off by default, so only the tests that are about the
|
||||
ACTIVE_BRANDS gate have to think about it."""
|
||||
from app.services import active_brands
|
||||
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: False)
|
||||
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: True)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def discovered(monkeypatch):
|
||||
"""A fixed two-product discovery result."""
|
||||
def fake_discover(brand, **kwargs):
|
||||
return bd.DiscoveryResult(
|
||||
brand=brand,
|
||||
parent_brand="britannia",
|
||||
table="brand_britannia",
|
||||
brand_active=True,
|
||||
filtering_enabled=False,
|
||||
products=[
|
||||
bd.DiscoveredProduct(
|
||||
brand=brand, product_name="Marie Gold", title="Marie Gold",
|
||||
category="", category_hint="Biscuits & Cookies", description="",
|
||||
size_variants=["250g"], barcode="8901063014206",
|
||||
sources=["off"], evidence="openfacts", confidence=0.9,
|
||||
),
|
||||
bd.DiscoveredProduct(
|
||||
brand=brand, product_name="Milk Bikis", title="Milk Bikis",
|
||||
category="", category_hint="Biscuits & Cookies", description="",
|
||||
size_variants=["100g"], barcode="8901063012516",
|
||||
sources=["off"], evidence="openfacts", confidence=0.9,
|
||||
),
|
||||
],
|
||||
counts={"discovered": 2},
|
||||
)
|
||||
|
||||
from app.api.routers import brand_discovery as router_module
|
||||
monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products",
|
||||
fake_discover)
|
||||
return fake_discover
|
||||
|
||||
|
||||
def _payload(**overrides):
|
||||
body = {
|
||||
"brand": "Britannia",
|
||||
"products": [
|
||||
{"product_name": "Marie Gold", "size_variants": ["250g"],
|
||||
"barcode": "8901063014206", "providers": ["Amazon"],
|
||||
"highlights": ["Tea-time favourite"], "nutrients": ["Iron - Blood health"]},
|
||||
],
|
||||
}
|
||||
body.update(overrides)
|
||||
return body
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Auth
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_preview_requires_admin(client):
|
||||
assert client.post(PREVIEW, json={"brand": "Britannia"}).status_code in (401, 403)
|
||||
|
||||
|
||||
def test_ingest_requires_admin(client):
|
||||
assert client.post(INGEST, json=_payload()).status_code in (401, 403)
|
||||
|
||||
|
||||
def test_a_plain_user_cannot_discover(client, user_headers):
|
||||
resp = client.post(PREVIEW, json={"brand": "Britannia"}, headers=user_headers)
|
||||
assert resp.status_code == 403
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Preview writes nothing
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_preview_returns_products_and_the_stage_list(client, admin_headers, discovered):
|
||||
resp = client.post(PREVIEW, json={"brand": "Britannia"}, headers=admin_headers)
|
||||
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert len(body["products"]) == 2
|
||||
assert body["table"] == "brand_britannia"
|
||||
assert body["stages"] == list(pipeline.STAGE_NAMES)
|
||||
assert len(body["stages"]) == 11
|
||||
assert body["products"][0]["selected"] is True
|
||||
|
||||
|
||||
def test_preview_stages_nothing(client, admin_headers, discovered, batch_root,
|
||||
no_background_worker):
|
||||
"""The whole reason preview is a separate route."""
|
||||
client.post(PREVIEW, json={"brand": "Britannia"}, headers=admin_headers)
|
||||
|
||||
assert no_background_worker == []
|
||||
assert not batch_root.exists() or list(batch_root.iterdir()) == []
|
||||
|
||||
|
||||
def test_preview_reports_a_bad_brand_name_as_client_error(client, admin_headers):
|
||||
resp = client.post(PREVIEW, json={"brand": " "}, headers=admin_headers)
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_preview_surfaces_a_discovery_failure_rather_than_500(client, admin_headers,
|
||||
monkeypatch):
|
||||
from app.api.routers import brand_discovery as router_module
|
||||
|
||||
def boom(brand, **kwargs):
|
||||
raise RuntimeError("Open Food Facts is unreachable")
|
||||
|
||||
monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products", boom)
|
||||
|
||||
resp = client.post(PREVIEW, json={"brand": "Britannia"}, headers=admin_headers)
|
||||
|
||||
assert resp.status_code == 502
|
||||
assert "unreachable" in resp.json()["detail"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The ACTIVE_BRANDS gate
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_preview_warns_when_the_brand_is_not_active(client, admin_headers, discovered,
|
||||
monkeypatch):
|
||||
from app.services import active_brands
|
||||
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: True)
|
||||
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: False)
|
||||
monkeypatch.setattr(active_brands, "active_display_names", lambda: ["Amul"])
|
||||
|
||||
from app.api.routers import brand_discovery as router_module
|
||||
|
||||
def inactive(brand, **kwargs):
|
||||
result = discovered(brand, **kwargs)
|
||||
result.brand_active = False
|
||||
result.filtering_enabled = True
|
||||
return result
|
||||
|
||||
monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products", inactive)
|
||||
|
||||
body = client.post(PREVIEW, json={"brand": "Britannia"},
|
||||
headers=admin_headers).json()
|
||||
|
||||
assert body["brand_active"] is False
|
||||
assert any("ACTIVE_BRANDS" in w for w in body["warnings"])
|
||||
|
||||
|
||||
def test_ingest_is_refused_for_an_inactive_brand(client, admin_headers, monkeypatch):
|
||||
"""A green run over a catalog no endpoint can read is not an acceptable
|
||||
outcome to hand back silently."""
|
||||
from app.services import active_brands
|
||||
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: True)
|
||||
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: False)
|
||||
monkeypatch.setattr(active_brands, "active_display_names", lambda: ["Amul", "Cadbury"])
|
||||
|
||||
resp = client.post(INGEST, json=_payload(), headers=admin_headers)
|
||||
|
||||
assert resp.status_code == 409
|
||||
detail = resp.json()["detail"]
|
||||
assert "ACTIVE_BRANDS=Amul,Cadbury,Britannia" in detail
|
||||
assert "restart" in detail.lower()
|
||||
|
||||
|
||||
def test_ingest_proceeds_for_an_inactive_brand_when_acknowledged(client, admin_headers,
|
||||
monkeypatch):
|
||||
from app.services import active_brands
|
||||
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: True)
|
||||
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: False)
|
||||
monkeypatch.setattr(active_brands, "active_display_names", lambda: ["Amul"])
|
||||
|
||||
resp = client.post(INGEST, json=_payload(acknowledge_inactive=True),
|
||||
headers=admin_headers)
|
||||
|
||||
assert resp.status_code == 202
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Ingest
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_ingest_stages_one_batch_and_returns_a_pollable_id(client, admin_headers,
|
||||
no_background_worker):
|
||||
resp = client.post(INGEST, json=_payload(), headers=admin_headers)
|
||||
|
||||
assert resp.status_code == 202
|
||||
body = resp.json()
|
||||
assert body["batch_id"]
|
||||
assert body["files_total"] == 1
|
||||
assert len(body["stage_names"]) == 11
|
||||
assert no_background_worker == [body["batch_id"]]
|
||||
|
||||
|
||||
def test_the_staged_file_is_a_real_csv_the_pipeline_can_read(client, admin_headers,
|
||||
batch_root):
|
||||
"""Not a placeholder: the bytes on the batch volume are the exact input that
|
||||
produced the rows, and they must parse with the same reader an upload uses."""
|
||||
batch_id = client.post(INGEST, json=_payload(), headers=admin_headers).json()["batch_id"]
|
||||
|
||||
manifest = batch_ingest.read_manifest(batch_id)
|
||||
entry = manifest.files[0]
|
||||
assert entry.filename.startswith("discovered-britannia-")
|
||||
assert entry.filename.endswith(".csv")
|
||||
|
||||
contents = (batch_ingest.batch_dir(batch_id) / entry.stored_name).read_bytes()
|
||||
df, mapping = pipeline.parse_spreadsheet(entry.filename, contents)
|
||||
|
||||
assert len(df) == 1
|
||||
assert mapping.unrecognised == []
|
||||
assert "product_name" in mapping.columns
|
||||
|
||||
|
||||
def test_the_batch_records_where_it_came_from(client, admin_headers):
|
||||
batch_id = client.post(INGEST, json=_payload(), headers=admin_headers).json()["batch_id"]
|
||||
|
||||
manifest = batch_ingest.read_manifest(batch_id)
|
||||
|
||||
assert manifest.submitted_by == "brand-discovery: Britannia"
|
||||
|
||||
|
||||
def test_ingest_runs_the_eleven_stages_end_to_end(client, admin_headers, store,
|
||||
no_background_worker, monkeypatch):
|
||||
"""Drive the real worker function over the staged file.
|
||||
|
||||
`use_llm` and `fetch_images` are turned OFF for this one, matching every
|
||||
other pipeline test in the suite: with them on, `run_batch` really calls
|
||||
Ollama once per row and really searches the open web for images. That is the
|
||||
correct production default and a terrible test - slow, and it fails when the
|
||||
machine is offline.
|
||||
"""
|
||||
body = _payload(use_llm=False, fetch_images=False)
|
||||
batch_id = client.post(INGEST, json=body, headers=admin_headers).json()["batch_id"]
|
||||
|
||||
batch_ingest.run_batch(batch_id)
|
||||
|
||||
assert len(store) == 1
|
||||
row = next(iter(store.values()))
|
||||
assert row["image_id"] == "britannia_marie_gold_250g"
|
||||
assert row["barcode"] == "8901063014206"
|
||||
assert row["product_sku"]
|
||||
assert row["highlights"] == ["Tea-time favourite"]
|
||||
|
||||
|
||||
def test_an_empty_selection_is_refused(client, admin_headers):
|
||||
resp = client.post(INGEST, json=_payload(products=[]), headers=admin_headers)
|
||||
|
||||
assert resp.status_code == 400
|
||||
assert "nothing to ingest" in resp.json()["detail"]
|
||||
|
||||
|
||||
def test_a_blank_brand_is_refused(client, admin_headers):
|
||||
resp = client.post(INGEST, json=_payload(brand=" "), headers=admin_headers)
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_a_full_queue_is_reported_as_429(client, admin_headers, monkeypatch):
|
||||
import queue
|
||||
|
||||
def full(batch_id):
|
||||
raise queue.Full()
|
||||
|
||||
from app.core import batch_worker
|
||||
monkeypatch.setattr(batch_worker, "submit", full)
|
||||
|
||||
resp = client.post(INGEST, json=_payload(), headers=admin_headers)
|
||||
|
||||
assert resp.status_code == 429
|
||||
assert "Resume" in resp.json()["detail"]
|
||||
|
||||
|
||||
def test_ingest_defaults_turn_on_images_and_the_per_row_llm(client, admin_headers,
|
||||
monkeypatch):
|
||||
"""The two enrichment stages this feature runs with. Both are per-run
|
||||
arguments already, so neither needs a settings change."""
|
||||
seen = {}
|
||||
real = batch_common.stage_and_queue
|
||||
|
||||
def spy(valid, invalid, *, use_llm, fetch_images, submitted_by=None):
|
||||
seen["use_llm"] = use_llm
|
||||
seen["fetch_images"] = fetch_images
|
||||
return real(valid, invalid, use_llm=use_llm, fetch_images=fetch_images,
|
||||
submitted_by=submitted_by)
|
||||
|
||||
from app.api.routers import brand_discovery as router_module
|
||||
monkeypatch.setattr(router_module.batch_common, "stage_and_queue", spy)
|
||||
|
||||
client.post(INGEST, json=_payload(), headers=admin_headers)
|
||||
|
||||
assert seen == {"use_llm": True, "fetch_images": True}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store(monkeypatch):
|
||||
table: dict = {}
|
||||
|
||||
def fake_upsert(brand, rows, cleanup=False):
|
||||
assert cleanup is False, "cleanup=True would delete the brand's existing catalog"
|
||||
for row in rows:
|
||||
table[row["image_id"]] = dict(row)
|
||||
return len(rows)
|
||||
|
||||
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
|
||||
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: list(table.values()))
|
||||
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
|
||||
return table
|
||||
Reference in New Issue
Block a user