Brand valid image generation
This commit is contained in:
@@ -55,6 +55,11 @@ os.environ.setdefault("USE_PGVECTOR", "true")
|
||||
os.environ.setdefault("DB_PASSWORD", "test-password-not-real")
|
||||
os.environ.setdefault("USE_S3", "false")
|
||||
os.environ.setdefault("USE_GOOGLE_CSE", "false")
|
||||
# Unconditional: the LLM description stage is on by default for every batch
|
||||
# and a developer machine often has `ollama serve` running, so without this a
|
||||
# pipeline test would make real, slow, non-deterministic model calls. Tests
|
||||
# that exercise the probe itself monkeypatch `ollama_service.USE_OLLAMA`.
|
||||
os.environ["USE_OLLAMA"] = "false"
|
||||
|
||||
# Unconditional, NOT setdefault. The suite pins brand-name behaviour all over
|
||||
# the place ("any Nestle chocolates?", the suggest ranking fixtures), and a
|
||||
|
||||
@@ -5,8 +5,8 @@ boundary *on the pipeline module object*, build real spreadsheets in memory,
|
||||
and point the batch directory at tmp_path so nothing is written into the repo.
|
||||
|
||||
Nothing here loads sentence-transformers or torch, and nothing reaches the
|
||||
network: every batch runs with use_llm=False and fetch_images=False, which are
|
||||
also the defaults the endpoint ships.
|
||||
network: every batch runs with fetch_images=False (the endpoint's default) and
|
||||
the LLM stage, on by default, finds Ollama disabled and falls back at once.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -486,7 +486,7 @@ def test_ingest_returns_202_and_a_batch_id(client, admin_headers, store, batch_r
|
||||
assert response.status_code == 202
|
||||
body = response.json()
|
||||
assert body["files_total"] == 2
|
||||
assert body["use_llm"] is False, "the LLM stage must be opt-in for a batch"
|
||||
assert body["use_llm"] is True, "LLM descriptions are on by default; a batch opts OUT"
|
||||
assert body["fetch_images"] is False, "image search must be opt-in for a batch"
|
||||
assert len(body["files"]) == 2
|
||||
assert body["files"][0]["total_stages"] == pipeline.TOTAL_STAGES
|
||||
|
||||
@@ -21,6 +21,9 @@ from app.api.routers.user_products import map_spreadsheet_columns, row_to_reques
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.services import brand_discovery as bd
|
||||
|
||||
# Captured before the autouse `_no_network` fixture replaces it on the module.
|
||||
_REAL_FROM_OPEN_FACTS = bd._from_open_facts
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Fixtures
|
||||
@@ -220,12 +223,11 @@ def test_a_case_pack_count_is_not_part_of_the_product_name(monkeypatch):
|
||||
# ---------------------------------------------------------------------------
|
||||
# Regression: one GTIN, one pack
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_barcode_is_dropped_when_no_real_pack_size_is_known(monkeypatch):
|
||||
"""A GTIN identifies one pack, and stage 4 invents three when it has none.
|
||||
|
||||
Every column is copied into each exploded variant, so a surviving barcode
|
||||
would be stamped onto two packs that do not exist - wrong data that looks
|
||||
authoritative.
|
||||
def test_a_barcode_survives_when_no_pack_size_is_known(monkeypatch):
|
||||
"""A GTIN identifies one pack, and stage 4 now stores exactly one row for
|
||||
a product with no size (it used to invent three and stamp the same GTIN
|
||||
on all of them, which is why the barcode was dropped before). One GTIN on
|
||||
one unsized row is what a GTIN means; the row is only flagged for review.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("50 50 Gol Maal", code="8901063017702"),
|
||||
@@ -234,8 +236,8 @@ def test_a_barcode_is_dropped_when_no_real_pack_size_is_known(monkeypatch):
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.size_variants == []
|
||||
assert product.barcode is None
|
||||
assert any("barcode dropped" in note for note in product.notes)
|
||||
assert product.barcode == "8901063017702"
|
||||
assert any("pack size unknown" in note for note in product.notes)
|
||||
|
||||
|
||||
def test_a_barcode_survives_when_the_pack_size_is_real(monkeypatch):
|
||||
@@ -549,3 +551,113 @@ def test_every_column_the_pipeline_can_fill_is_filled(store, monkeypatch):
|
||||
assert row.get(column), f"{column} was left empty"
|
||||
assert row["barcode"] == "8901063012516"
|
||||
assert row["fssai_license"] == "10012022000103"
|
||||
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# What the warnings claim - "unreachable" vs "empty" vs "no storefront"
|
||||
# ---------------------------------------------------------------------------
|
||||
# Naga (2026-09-11): Open Food Facts holds two real rows for the brand, yet the
|
||||
# tab said "Neither Open Food Facts nor a brand storefront has anything for
|
||||
# this brand". The lookup had failed / been mis-indexed, no storefront was
|
||||
# registered, and one warning text covered all of it. Each claim now has to
|
||||
# be true on its own.
|
||||
|
||||
def _store_rows(monkeypatch, rows):
|
||||
monkeypatch.setattr(bd, "_from_brand_store", lambda brand: list(rows))
|
||||
|
||||
|
||||
def test_an_unreachable_off_is_reported_as_unreachable_not_empty(monkeypatch):
|
||||
def down(brand, refresh=False):
|
||||
raise bd.OpenFactsUnavailable("HTTP 503 from Open Food Facts")
|
||||
|
||||
monkeypatch.setattr(bd, "_from_open_facts", down)
|
||||
_store_rows(monkeypatch, [])
|
||||
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
|
||||
{"title": "Naga Sooji", "category": None, "description": None,
|
||||
"sizes": ["500g"], "providers": [], "source": "llm"},
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Naga", require_evidence=False)
|
||||
|
||||
reached = [w for w in result.warnings if "could not be reached" in w]
|
||||
assert len(reached) == 1 and "503" in reached[0]
|
||||
assert not any("has no products tagged" in w for w in result.warnings)
|
||||
assert [p.title for p in result.products] == ["Naga Sooji"], "the LLM rows must still come through when OFF is down"
|
||||
|
||||
|
||||
def test_a_genuinely_empty_off_says_so_without_blaming_the_network(monkeypatch):
|
||||
_store_rows(monkeypatch, [])
|
||||
result = bd.discover_brand_products("Udhaiyam")
|
||||
assert any("has no products tagged" in w for w in result.warnings)
|
||||
assert not any("could not be reached" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_unregistered_storefront_is_not_described_as_empty(monkeypatch):
|
||||
_store_rows(monkeypatch, [])
|
||||
monkeypatch.setattr(bd, "get_brand_store_domain", lambda brand: None)
|
||||
|
||||
result = bd.discover_brand_products("Naga")
|
||||
|
||||
assert any("No verified storefront is registered" in w for w in result.warnings)
|
||||
assert not any("Neither Open Food Facts" in w for w in result.warnings)
|
||||
assert not any("has not been fetched" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_registered_but_unfetched_storefront_names_the_backfill(monkeypatch):
|
||||
_store_rows(monkeypatch, [])
|
||||
monkeypatch.setattr(bd, "get_brand_store_domain", lambda brand: "gopuramproducts.com")
|
||||
|
||||
result = bd.discover_brand_products("Gopuram")
|
||||
|
||||
hit = [w for w in result.warnings if "has not been fetched" in w]
|
||||
assert len(hit) == 1
|
||||
assert "gopuramproducts.com" in hit[0] and "backfill_brand_stores.py" in hit[0]
|
||||
assert not any("No verified storefront" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_the_llm_only_caveat_appears_exactly_once(monkeypatch):
|
||||
_store_rows(monkeypatch, [])
|
||||
result = bd.discover_brand_products("Naga")
|
||||
assert sum("rests on the language model alone" in w for w in result.warnings) == 1
|
||||
|
||||
|
||||
def test_no_storefront_warning_when_the_shop_or_off_has_rows(monkeypatch):
|
||||
_store_rows(monkeypatch, [{"title": "Naga Maida 1kg", "source": "store", "size": "1kg"}])
|
||||
result = bd.discover_brand_products("Naga")
|
||||
assert not any("storefront" in w.lower() for w in result.warnings)
|
||||
assert not any("rests on the language model alone" in w for w in result.warnings)
|
||||
|
||||
_store_rows(monkeypatch, [])
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Sooji", code="8906011830068", quantity="500 g"),
|
||||
])
|
||||
result = bd.discover_brand_products("Naga")
|
||||
assert not any("storefront" in w.lower() for w in result.warnings)
|
||||
assert not any("rests on the language model alone" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_from_open_facts_raises_when_the_fetch_failed_with_nothing(monkeypatch):
|
||||
from app.services.enrichment.barcode.sources import off_bulk
|
||||
|
||||
monkeypatch.setattr(off_bulk, "fetch_brand_corpus_result",
|
||||
lambda brand, refresh=False: off_bulk.CorpusFetch(hits=[], error="dns"))
|
||||
with pytest.raises(bd.OpenFactsUnavailable):
|
||||
_REAL_FROM_OPEN_FACTS("Naga")
|
||||
|
||||
|
||||
def test_from_open_facts_maps_v2_hits_to_candidates(monkeypatch):
|
||||
from app.services.enrichment.barcode.sources import off_bulk
|
||||
|
||||
monkeypatch.setattr(off_bulk, "fetch_brand_corpus_result",
|
||||
lambda brand, refresh=False: off_bulk.CorpusFetch(hits=[
|
||||
{"code": "8906011830068", "product_name": "Sooji",
|
||||
"product_name_en": "Sooji", "quantity": "500 g"},
|
||||
{"code": "8906011831713", "product_name": "Maida", "quantity": "500 g"},
|
||||
]))
|
||||
|
||||
rows = _REAL_FROM_OPEN_FACTS("Naga")
|
||||
|
||||
assert [(r["title"], r["barcode"], r["source"]) for r in rows] == [
|
||||
("Sooji", "8906011830068", "off"), ("Maida", "8906011831713", "off"),
|
||||
]
|
||||
|
||||
@@ -174,6 +174,94 @@ def test_a_lookup_failure_never_rejects_an_image(monkeypatch):
|
||||
assert ic.openfacts_product_matches_brand(url, "Anil") is True
|
||||
|
||||
|
||||
class _Resp:
|
||||
def __init__(self, ok=True, payload=None):
|
||||
self.ok = ok
|
||||
self.status_code = 200 if ok else 503
|
||||
self._payload = payload or {}
|
||||
|
||||
def json(self):
|
||||
return self._payload
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def verdict_cache(tmp_path, monkeypatch):
|
||||
"""A throwaway sqlite cache, so these tests neither read nor poison the
|
||||
real one in data/cache."""
|
||||
monkeypatch.setattr(ic, "_DB_PATH", tmp_path / "verdicts.db")
|
||||
monkeypatch.setattr(ic, "_initialized", False) # the schema is per file
|
||||
monkeypatch.setattr(ic.time, "sleep", lambda s: None) # pacing/retry waits
|
||||
return tmp_path / "verdicts.db"
|
||||
|
||||
|
||||
KELLOGGS_HONEY = "https://images.openfoodfacts.org/images/products/505/931/902/3762/front_fr.32.400.jpg"
|
||||
|
||||
|
||||
def test_a_throttled_lookup_fails_open_but_is_not_remembered(monkeypatch, verdict_cache):
|
||||
"""The Dabur Honey defect. A hundred lookups in two minutes got the tail
|
||||
of them a 503; the fail-open True was cached for thirty days, and
|
||||
Kellogg's "Miel Pops" became a Dabur product for a month."""
|
||||
throttled = _Resp(ok=False)
|
||||
throttled.status_code = 503
|
||||
answers = iter([throttled, throttled, _Resp(payload={"status": 1, "product": {
|
||||
"brands": "KELLOG'S", "product_name": "Miel Pops"}})])
|
||||
monkeypatch.setattr(ic.requests, "get", lambda *a, **k: next(answers))
|
||||
|
||||
assert ic.openfacts_product_matches_brand(KELLOGGS_HONEY, "Dabur") is True, "throttled on both attempts: fails open now"
|
||||
assert ic.openfacts_product_matches_brand(KELLOGGS_HONEY, "Dabur") is False, "asks again next time"
|
||||
|
||||
|
||||
def test_one_throttled_attempt_is_retried_into_a_real_answer(monkeypatch, verdict_cache):
|
||||
throttled = _Resp(ok=False)
|
||||
throttled.status_code = 429
|
||||
answers = iter([throttled, _Resp(payload={"status": 1, "product": {
|
||||
"brands": "KELLOG'S", "product_name": "Miel Pops"}})])
|
||||
monkeypatch.setattr(ic.requests, "get", lambda *a, **k: next(answers))
|
||||
|
||||
assert ic.openfacts_product_matches_brand(KELLOGGS_HONEY, "Dabur") is False
|
||||
|
||||
|
||||
def test_a_real_verdict_is_cached(monkeypatch, verdict_cache):
|
||||
calls = []
|
||||
|
||||
def get(*a, **k):
|
||||
calls.append(1)
|
||||
return _Resp(payload={"status": 1, "product": {"brands": "Toblerone",
|
||||
"product_name": "Milk Chocolate"}})
|
||||
|
||||
monkeypatch.setattr(ic.requests, "get", get)
|
||||
url = "https://images.openfoodfacts.org/images/products/761/450/001/0013/front_en.362.400.jpg"
|
||||
|
||||
assert ic.openfacts_product_matches_brand(url, "Dabur") is False
|
||||
assert ic.openfacts_product_matches_brand(url, "Dabur") is False
|
||||
assert len(calls) == 1
|
||||
|
||||
|
||||
def test_an_unknown_barcode_is_cached_as_open(monkeypatch, verdict_cache):
|
||||
calls = []
|
||||
|
||||
def get(*a, **k):
|
||||
calls.append(1)
|
||||
return _Resp(payload={"status": 0, "status_verbose": "product not found"})
|
||||
|
||||
monkeypatch.setattr(ic.requests, "get", get)
|
||||
url = "https://images.openfoodfacts.org/images/products/000/000/000/0001/front.jpg"
|
||||
|
||||
assert ic.openfacts_product_matches_brand(url, "Dabur") is True
|
||||
assert ic.openfacts_product_matches_brand(url, "Dabur") is True
|
||||
assert len(calls) == 1
|
||||
|
||||
|
||||
def test_pre_fix_cache_entries_are_ignored(monkeypatch, verdict_cache):
|
||||
"""Entries written by the old code (no "v2:" prefix) may be fail-open
|
||||
Trues; they must not be trusted."""
|
||||
ic._cache_set("world.openfoodfacts.org:5059319023762:dabur", True)
|
||||
monkeypatch.setattr(ic.requests, "get", lambda *a, **k: _Resp(payload={
|
||||
"status": 1, "product": {"brands": "KELLOG'S", "product_name": "Miel Pops"}}))
|
||||
|
||||
assert ic.openfacts_product_matches_brand(KELLOGGS_HONEY, "Dabur") is False
|
||||
|
||||
|
||||
def test_a_non_openfacts_url_is_not_cross_checked():
|
||||
assert ic.openfacts_product_matches_brand(REAL_RAVA, "Anil") is True
|
||||
|
||||
@@ -193,3 +281,20 @@ def test_pack_sizes_of_one_product_share_a_search_key():
|
||||
def test_different_products_do_not_share_a_search_key():
|
||||
assert (ic.search_key("Anil Wheat Vermicelli 180g", "Anil")
|
||||
!= ic.search_key("Anil Samba Rava 500g", "Anil"))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Plural titles, singular filenames
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_plural_title_is_named_by_a_singular_filename():
|
||||
"""The all-brands re-gate would have dropped the correct
|
||||
`Aachi-Appalam-100-g-1.webp` from "Aachi Appalams 500g" and promoted an
|
||||
opaque Amazon URL over it."""
|
||||
url = "https://thedesifood.com/media/Aachi-Appalam-100-g-1.webp"
|
||||
assert ic.corroborate(url, "Aachi Appalams 500g", "Aachi").corroborated
|
||||
assert ic.corroborate("https://x.in/img/aachi-pickle-jar.jpg", "Aachi Pickles 200g", "Aachi").corroborated
|
||||
|
||||
|
||||
def test_the_singular_fallback_does_not_widen_short_tokens():
|
||||
# "gems" -> "gem" is too short a stem to trust; four letters is the floor.
|
||||
assert not ic.corroborate("https://x.in/gem-ring.jpg", "Cadbury Gems 20g", "Cadbury").corroborated
|
||||
|
||||
297
tests/test_image_precision.py
Normal file
297
tests/test_image_precision.py
Normal file
@@ -0,0 +1,297 @@
|
||||
"""The Dabur Honey regression: one product, several pack sizes, one photo.
|
||||
|
||||
Brand Discovery for Dabur (2026-09-11) stored, for the same product:
|
||||
|
||||
Dabur Honey 225g Amazon photo of Dabur Honey correct
|
||||
Dabur Honey 1kg image_url empty; image_urls = 4 honeys from the UK,
|
||||
Dabur Honey 250g France, Switzerland and Spain (Open Food Facts photos
|
||||
Dabur Honey 50g whose barcodes belong to other brands)
|
||||
Dabur Honey 500g a Dabur Odomos mosquito repellent
|
||||
|
||||
Three causes, each pinned below:
|
||||
|
||||
1. Stage 6 searched once PER SIZE. The query was identical - the size is not
|
||||
in it - but the providers are live and not deterministic, so four asks
|
||||
came back four ways. One search per product, shared by every size.
|
||||
2. `find_images_openfacts` threw away the `brands` field of the records it
|
||||
fetched, so a title-only fallback ("Honey 1kg") handed over every honey on
|
||||
the site. The brand is checked at the source now.
|
||||
3. `choose_primary` kept rejected candidates in `image_urls` "for review" -
|
||||
and the product card falls back to `image_urls[0]` when the primary is
|
||||
withheld, so the review pile was what the shopper saw. Only eligible
|
||||
candidates are stored.
|
||||
|
||||
Everything runs offline: the search function and the Open*Facts barcode
|
||||
cross-check are monkeypatched; storage and embeddings are stubbed.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
|
||||
import pytest
|
||||
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.services import image_corroboration as ic
|
||||
from app.services import image_search
|
||||
|
||||
openpyxl = pytest.importorskip("openpyxl")
|
||||
|
||||
AMAZON = "https://m.media-amazon.com/images/I/71O4OnjaHVL.jpg"
|
||||
ODOMOS = ("https://www.indianproductsstore.com/uploads/products/"
|
||||
"dabur-odomos-naturals-mosquito-repellent-gel.jpg")
|
||||
FOREIGN_HONEYS = [
|
||||
"https://images.openfoodfacts.org/images/products/505/931/902/3762/front_fr.32.400.jpg",
|
||||
"https://images.openfoodfacts.org/images/products/308/854/000/4440/front_fr.129.400.jpg",
|
||||
]
|
||||
DABUR_OFF = "https://images.openfoodfacts.org/images/products/890/120/702/6553/front_en.4.400.jpg"
|
||||
NAMED = "https://cdn.example.in/dabur-honey-squeezy-bottle.jpg"
|
||||
|
||||
|
||||
def _sheet(headers, rows) -> bytes:
|
||||
wb = openpyxl.Workbook()
|
||||
ws = wb.active
|
||||
ws.append(headers)
|
||||
for row in rows:
|
||||
ws.append(row)
|
||||
buf = io.BytesIO()
|
||||
wb.save(buf)
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolate_sku_counter(tmp_path, monkeypatch):
|
||||
from app.services import sku_service
|
||||
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _off_barcodes_offline(monkeypatch):
|
||||
"""Open*Facts says: the 890120... barcode is Dabur, the others are not."""
|
||||
monkeypatch.setattr(
|
||||
ic, "openfacts_product_matches_brand",
|
||||
lambda url, brand, timeout=10: "/890/120/" in url or "openfoodfacts" not in url,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store(monkeypatch):
|
||||
table: dict = {}
|
||||
|
||||
def fake_upsert(brand, rows, cleanup=False):
|
||||
for row in rows:
|
||||
table[row["image_id"]] = dict(row)
|
||||
return len(rows)
|
||||
|
||||
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
|
||||
monkeypatch.setattr(pipeline, "get_products_by_brand",
|
||||
lambda b, **kw: [dict(r, brand=b) for r in table.values()])
|
||||
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
|
||||
return table
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def search(monkeypatch):
|
||||
"""A scripted `find_all_image_urls`; records every call it receives."""
|
||||
calls = []
|
||||
|
||||
def install(results):
|
||||
def fake(title, brand=None, country_hint=None, validate=True, max_results=24, produce=False):
|
||||
calls.append((title, brand))
|
||||
return list(results)
|
||||
monkeypatch.setattr(image_search, "find_all_image_urls", fake)
|
||||
return calls
|
||||
|
||||
return install
|
||||
|
||||
|
||||
def _run(headers, rows):
|
||||
return pipeline.run_pipeline("store.xlsx", _sheet(headers, rows),
|
||||
use_llm=False, fetch_images=True)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. One search per product
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_every_pack_size_of_a_product_gets_the_same_image_from_one_search(store, search):
|
||||
calls = search([AMAZON, DABUR_OFF])
|
||||
|
||||
_run(["Product Name", "Brand", "Pack Size"], [["Dabur Honey", "Dabur", "1kg, 250g, 50g, 225g"]])
|
||||
|
||||
assert len(store) == 4
|
||||
assert {r["image_url"] for r in store.values()} == {AMAZON}
|
||||
assert len(calls) == 1, "four pack sizes, one search"
|
||||
assert calls[0][0] == "Dabur Honey", "the size never enters the query"
|
||||
|
||||
|
||||
def test_sizes_written_into_the_name_still_share_the_search(store, search):
|
||||
calls = search([AMAZON])
|
||||
|
||||
_run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"], ["Dabur Honey 250g", "Dabur"]])
|
||||
|
||||
assert len(calls) == 1
|
||||
assert calls[0][0] == "Dabur Honey"
|
||||
assert {r["image_url"] for r in store.values()} == {AMAZON}
|
||||
|
||||
|
||||
def test_different_products_of_one_brand_are_searched_separately(store, search):
|
||||
calls = search([NAMED])
|
||||
_run(["Product Name", "Brand"], [["Dabur Honey", "Dabur"], ["Dabur Chyawanprash", "Dabur"]])
|
||||
assert sorted(c[0] for c in calls) == ["Dabur Chyawanprash", "Dabur Honey"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Open*Facts records are filtered by brand at the source
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_openfacts_photos_of_other_brands_honey_are_not_candidates(monkeypatch):
|
||||
records = [
|
||||
{"brands": "Dabur", "brands_tags": ["dabur"], "image_front_url": DABUR_OFF},
|
||||
{"brands": "Rowse", "brands_tags": ["rowse"], "image_front_url": FOREIGN_HONEYS[0]},
|
||||
{"brands": "Lune de Miel", "image_front_url": FOREIGN_HONEYS[1]},
|
||||
{"brands": "", "image_front_url": "https://images.openfoodfacts.org/x/unlabelled.jpg"},
|
||||
]
|
||||
monkeypatch.setattr(image_search, "_query_openfacts", lambda query, n: records)
|
||||
monkeypatch.setattr(image_search, "USE_OPEN_FACTS", True)
|
||||
|
||||
urls = image_search.find_images_openfacts("Honey", "Dabur")
|
||||
|
||||
assert DABUR_OFF in urls
|
||||
assert not any(u in urls for u in FOREIGN_HONEYS)
|
||||
assert "https://images.openfoodfacts.org/x/unlabelled.jpg" in urls, \
|
||||
"a record with no brand at all is not evidence of another brand"
|
||||
|
||||
|
||||
def test_the_title_only_fallback_is_also_brand_filtered(monkeypatch):
|
||||
seen = []
|
||||
|
||||
def query(q, n):
|
||||
seen.append(q)
|
||||
if q.startswith("dabur"):
|
||||
return []
|
||||
return [{"brands": "Rowse", "image_front_url": FOREIGN_HONEYS[0]}]
|
||||
|
||||
monkeypatch.setattr(image_search, "_query_openfacts", query)
|
||||
monkeypatch.setattr(image_search, "USE_OPEN_FACTS", True)
|
||||
|
||||
assert image_search.find_images_openfacts("Honey 1kg", "dabur") == []
|
||||
assert seen == ["dabur Honey 1kg", "Honey 1kg"]
|
||||
|
||||
|
||||
def test_without_a_brand_nothing_is_filtered(monkeypatch):
|
||||
monkeypatch.setattr(image_search, "_query_openfacts",
|
||||
lambda q, n: [{"brands": "Rowse", "image_front_url": FOREIGN_HONEYS[0]}])
|
||||
monkeypatch.setattr(image_search, "USE_OPEN_FACTS", True)
|
||||
assert image_search.find_images_openfacts("Honey", None) == FOREIGN_HONEYS[:1]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. Rejected candidates are not stored where the UI will show them
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_choose_primary_separates_eligible_from_rejected():
|
||||
choice = ic.choose_primary([ODOMOS, FOREIGN_HONEYS[0], AMAZON, NAMED], "Dabur Honey 500g", "Dabur")
|
||||
|
||||
assert choice.primary == NAMED # tier 1: names "honey"
|
||||
assert choice.eligible == [NAMED, AMAZON] # tier 1 then tier 2
|
||||
assert choice.rejected == [ODOMOS, FOREIGN_HONEYS[0]]
|
||||
assert choice.ordered == choice.eligible + choice.rejected
|
||||
|
||||
|
||||
def test_a_same_brand_wrong_product_photo_is_never_stored(store, search):
|
||||
"""The 500g row: a Dabur Odomos repellent as the primary image of honey."""
|
||||
search([ODOMOS] + FOREIGN_HONEYS)
|
||||
|
||||
result = _run(["Product Name", "Brand"], [["Dabur Honey 500g", "Dabur"]])
|
||||
|
||||
row = next(iter(store.values()))
|
||||
assert row["image_url"] is None
|
||||
assert row["image_urls"] == [], "nothing the gate rejected may reach image_urls"
|
||||
assert any("rejected" in w for w in result.warnings)
|
||||
assert any("no primary image" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_foreign_honeys_ride_along_only_when_a_real_primary_exists(store, search):
|
||||
search([AMAZON, DABUR_OFF] + FOREIGN_HONEYS)
|
||||
_run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"]])
|
||||
|
||||
row = next(iter(store.values()))
|
||||
assert row["image_url"] == AMAZON
|
||||
assert row["image_urls"] == [AMAZON, DABUR_OFF]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4. A sibling pack size already in the table lends its photo - through the gate
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_new_pack_size_adopts_the_siblings_image_without_searching(store, search):
|
||||
calls = search([AMAZON])
|
||||
_run(["Product Name", "Brand"], [["Dabur Honey 225g", "Dabur"]])
|
||||
assert len(calls) == 1 and store["dabur_dabur_honey_225g"]["image_url"] == AMAZON
|
||||
|
||||
result = _run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"]])
|
||||
|
||||
assert len(calls) == 1, "the sibling's photo made the search unnecessary"
|
||||
assert store["dabur_dabur_honey_1kg"]["image_url"] == AMAZON
|
||||
assert store["dabur_dabur_honey_1kg"]["image_urls"] == [AMAZON]
|
||||
assert any("adopted from another pack size" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_a_siblings_wrong_image_is_not_adopted(store, search):
|
||||
"""A stored primary that would not pass the gate for this title stays put."""
|
||||
store["dabur_dabur_honey_500g"] = {
|
||||
"image_id": "dabur_dabur_honey_500g", "product_name": "Dabur Honey 500g",
|
||||
"image_url": ODOMOS, "image_urls": [ODOMOS],
|
||||
}
|
||||
calls = search([NAMED])
|
||||
|
||||
_run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"]])
|
||||
|
||||
assert len(calls) == 1, "an ineligible sibling image forces a fresh search"
|
||||
assert store["dabur_dabur_honey_1kg"]["image_url"] == NAMED
|
||||
|
||||
|
||||
def test_a_different_product_of_the_brand_is_not_a_sibling(store, search):
|
||||
store["dabur_dabur_chyawanprash_500g"] = {
|
||||
"image_id": "dabur_dabur_chyawanprash_500g", "product_name": "Dabur Chyawanprash 500g",
|
||||
"image_url": NAMED, "image_urls": [NAMED],
|
||||
}
|
||||
calls = search([AMAZON])
|
||||
_run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"]])
|
||||
assert len(calls) == 1
|
||||
assert store["dabur_dabur_honey_1kg"]["image_url"] == AMAZON
|
||||
|
||||
|
||||
def test_stage_six_without_a_cache_still_works():
|
||||
"""Direct callers (and the produce tests) pass no cache."""
|
||||
row = {"brand": "Dabur", "product_name": "Dabur Honey 1kg", "image_urls": [AMAZON]}
|
||||
assert pipeline.stage_6_images(row) is row
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 5. The one-off repair for rows written before the fix
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_regate_rows_reproduces_the_dabur_repair():
|
||||
from scripts.regate_brand_images import regate_rows
|
||||
|
||||
rows = [
|
||||
{"image_id": "dabur_dabur_honey_225g", "product_name": "Dabur Honey 225g",
|
||||
"image_url": AMAZON, "image_urls": [AMAZON, DABUR_OFF] + FOREIGN_HONEYS},
|
||||
{"image_id": "dabur_dabur_honey_1kg", "product_name": "Dabur Honey 1kg",
|
||||
"image_url": None, "image_urls": list(FOREIGN_HONEYS)},
|
||||
{"image_id": "dabur_dabur_honey_500g", "product_name": "Dabur Honey 500g",
|
||||
"image_url": ODOMOS, "image_urls": [ODOMOS]},
|
||||
{"image_id": "dabur_dabur_chyawanprash_500g", "product_name": "Dabur Chyawanprash 500g",
|
||||
"image_url": None, "image_urls": [ODOMOS]},
|
||||
]
|
||||
|
||||
by_id = {v["image_id"]: v for v in regate_rows(rows, "dabur")}
|
||||
|
||||
# 225g keeps its photo and loses the foreign honeys from its gallery.
|
||||
assert by_id["dabur_dabur_honey_225g"]["new_url"] == AMAZON
|
||||
assert by_id["dabur_dabur_honey_225g"]["new_urls"] == [AMAZON, DABUR_OFF]
|
||||
# 1kg had nothing eligible and adopts the 225g photo.
|
||||
assert by_id["dabur_dabur_honey_1kg"]["new_url"] == AMAZON
|
||||
assert by_id["dabur_dabur_honey_1kg"]["why"].startswith("adopted")
|
||||
# 500g drops the mosquito repellent and adopts the sibling photo too.
|
||||
assert by_id["dabur_dabur_honey_500g"]["new_url"] == AMAZON
|
||||
assert ODOMOS in by_id["dabur_dabur_honey_500g"]["rejected"]
|
||||
# A different product with only a wrong image is blanked, not lent a honey.
|
||||
assert by_id["dabur_dabur_chyawanprash_500g"]["new_url"] is None
|
||||
assert by_id["dabur_dabur_chyawanprash_500g"]["new_urls"] == []
|
||||
308
tests/test_off_bulk_fetch.py
Normal file
308
tests/test_off_bulk_fetch.py
Normal file
@@ -0,0 +1,308 @@
|
||||
"""The network half of `off_bulk`: what is asked of Open Food Facts, how the
|
||||
answer is paged, and - the part that produced a user-visible lie - what gets
|
||||
written to the corpus cache when the answer never arrives.
|
||||
|
||||
Background, measured 2026-09-11. Brand Discovery told the user that Open Food
|
||||
Facts had nothing for "Naga". The v2 API returns two real Naga Limited rows
|
||||
(Sooji 500 g / 8906011830068, Maida 500 g / 8906011831713). The old query hit
|
||||
the Search-a-licious endpoint with a free-text `brands:Naga`, whose index
|
||||
still filed 8906011830068 brandless under Kuwait, and any 503 it met on the
|
||||
way was cached to disk as `hits: []`. Both failure modes are pinned here, with
|
||||
no network: `requests.get` is replaced for every test.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
import requests
|
||||
|
||||
from app.services.enrichment.barcode.sources import off_bulk
|
||||
from app.services.enrichment.barcode.sources.off_bulk import (
|
||||
CACHE_SCHEMA,
|
||||
CorpusFetch,
|
||||
OffUnavailable,
|
||||
brand_tag_slug,
|
||||
fetch_brand_corpus,
|
||||
fetch_brand_corpus_result,
|
||||
)
|
||||
|
||||
NAGA_PRODUCTS = [
|
||||
{"code": "8906011830068", "product_name": "Sooji", "product_name_en": "Sooji",
|
||||
"brands": "Naga", "quantity": "500 g", "countries_tags": ["en:india"]},
|
||||
{"code": "8906011831713", "product_name": "Maida", "product_name_en": "Maida",
|
||||
"brands": "Naga", "quantity": "500 g", "countries_tags": ["en:india"]},
|
||||
]
|
||||
|
||||
|
||||
class _Resp:
|
||||
def __init__(self, status=200, payload=None, text="", content_type="application/json"):
|
||||
self.status_code = status
|
||||
self._payload = payload
|
||||
self.text = text
|
||||
self.headers = {"content-type": content_type}
|
||||
|
||||
def json(self):
|
||||
if self._payload is None:
|
||||
raise ValueError("not json")
|
||||
return self._payload
|
||||
|
||||
|
||||
def _v2(products, count=None, page=1, page_size=None):
|
||||
"""A v2 search body. `page_count` is deliberately the size of THIS page,
|
||||
which is what the real API sends and what a naive paginator misreads;
|
||||
`page_size` is the size the server APPLIED, not the one requested."""
|
||||
return {"count": len(products) if count is None else count, "page": page,
|
||||
"page_count": len(products),
|
||||
"page_size": off_bulk.PAGE_SIZE if page_size is None else page_size,
|
||||
"products": products, "skip": 0}
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _no_sleep(monkeypatch):
|
||||
monkeypatch.setattr(off_bulk.time, "sleep", lambda s: None)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def calls(monkeypatch):
|
||||
"""Replace requests.get with a scripted responder. Returns the list of
|
||||
(url, params) actually sent, so tests can assert on the query."""
|
||||
sent = []
|
||||
|
||||
def install(responder):
|
||||
def fake_get(url, params=None, headers=None, timeout=None):
|
||||
sent.append((url, dict(params or {})))
|
||||
r = responder(len(sent), dict(params or {}))
|
||||
if isinstance(r, Exception):
|
||||
raise r
|
||||
return r
|
||||
monkeypatch.setattr(off_bulk.requests, "get", fake_get)
|
||||
return sent
|
||||
|
||||
return install
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The query
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("brand, slug", [
|
||||
("Naga", "naga"),
|
||||
("Hindustan Unilever", "hindustan-unilever"),
|
||||
("P&G", "p-g"),
|
||||
(" Double Horse ", "double-horse"),
|
||||
("Coca-Cola", "coca-cola"),
|
||||
])
|
||||
def test_brand_tag_slug_matches_off_tag_form(brand, slug):
|
||||
assert brand_tag_slug(brand) == slug
|
||||
|
||||
|
||||
def test_query_is_an_exact_brand_tag_filter_on_the_v2_api(calls, tmp_path):
|
||||
sent = calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
|
||||
|
||||
result = fetch_brand_corpus_result("Naga", country="india", cache_dir=tmp_path)
|
||||
|
||||
assert result.error is None and not result.from_cache
|
||||
assert [h["code"] for h in result.hits] == ["8906011830068", "8906011831713"]
|
||||
url, params = sent[0]
|
||||
assert url == "https://world.openfoodfacts.org/api/v2/search"
|
||||
assert params["brands_tags"] == "naga"
|
||||
assert params["countries_tags"] == "en:india"
|
||||
assert "q" not in params, "free-text brands: matching is what returned Mr Naga"
|
||||
|
||||
|
||||
def test_no_country_filter_when_country_is_blank(calls, tmp_path):
|
||||
sent = calls(lambda n, p: _Resp(payload=_v2([])))
|
||||
fetch_brand_corpus_result("Naga", country="", cache_dir=tmp_path)
|
||||
assert "countries_tags" not in sent[0][1]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Pagination
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_pagination_uses_count_not_page_count(calls, tmp_path):
|
||||
"""150 products at PAGE_SIZE 100 is two pages. The first page's
|
||||
`page_count` is 100 - reading that as "pages" would fetch 100 pages."""
|
||||
page1 = [{"code": f"890{i:010d}", "product_name": f"P{i}"} for i in range(100)]
|
||||
page2 = [{"code": f"891{i:010d}", "product_name": f"Q{i}"} for i in range(50)]
|
||||
|
||||
def responder(n, params):
|
||||
return _Resp(payload=_v2(page1 if params["page"] == 1 else page2,
|
||||
count=150, page=params["page"]))
|
||||
|
||||
sent = calls(responder)
|
||||
result = fetch_brand_corpus_result("Amul", cache_dir=tmp_path)
|
||||
|
||||
assert [p["page"] for _, p in sent] == [1, 2]
|
||||
assert len(result.hits) == 150
|
||||
|
||||
|
||||
def test_pagination_follows_the_page_size_the_server_applied(calls, tmp_path, monkeypatch):
|
||||
"""The real Amul case: 216 products, 250 requested, 100 served per page.
|
||||
Dividing by what we ASKED for stops after one page and loses 116 rows."""
|
||||
monkeypatch.setattr(off_bulk, "PAGE_SIZE", 250)
|
||||
pages = {
|
||||
1: [{"code": f"890{i:010d}", "product_name": f"P{i}"} for i in range(100)],
|
||||
2: [{"code": f"891{i:010d}", "product_name": f"Q{i}"} for i in range(100)],
|
||||
3: [{"code": f"892{i:010d}", "product_name": f"R{i}"} for i in range(16)],
|
||||
}
|
||||
sent = calls(lambda n, p: _Resp(payload=_v2(pages[p["page"]], count=216,
|
||||
page=p["page"], page_size=100)))
|
||||
|
||||
result = fetch_brand_corpus_result("Amul", cache_dir=tmp_path)
|
||||
|
||||
assert [p["page"] for _, p in sent] == [1, 2, 3]
|
||||
assert len(result.hits) == 216
|
||||
|
||||
|
||||
def test_pagination_is_capped_at_max_pages(calls, tmp_path):
|
||||
body = [{"code": "8900000000000", "product_name": "X"}] * off_bulk.PAGE_SIZE
|
||||
sent = calls(lambda n, p: _Resp(payload=_v2(body, count=10 ** 6, page=p["page"])))
|
||||
fetch_brand_corpus_result("Nestle", cache_dir=tmp_path)
|
||||
assert len(sent) == off_bulk.MAX_PAGES
|
||||
|
||||
|
||||
def test_nameless_products_are_dropped_from_the_corpus(calls, tmp_path):
|
||||
calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS + [
|
||||
{"code": "8906011839999", "product_name": ""},
|
||||
{"code": "8906011839998"},
|
||||
])))
|
||||
assert len(fetch_brand_corpus_result("Naga", cache_dir=tmp_path).hits) == 2
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Failure is not emptiness, and is never cached
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_503_is_retried_then_reported_and_not_cached(calls, tmp_path):
|
||||
sent = calls(lambda n, p: _Resp(status=503, text="unavailable", content_type="text/html"))
|
||||
|
||||
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
|
||||
|
||||
assert len(sent) == 3, "with_retry gives a 5xx three attempts"
|
||||
assert result.hits == []
|
||||
assert result.error and "503" in result.error
|
||||
assert list(tmp_path.glob("*.json")) == [], "a failed fetch must not become hits: []"
|
||||
|
||||
|
||||
def test_503_then_success_recovers(calls, tmp_path):
|
||||
def responder(n, params):
|
||||
return _Resp(status=503) if n == 1 else _Resp(payload=_v2(NAGA_PRODUCTS))
|
||||
|
||||
calls(responder)
|
||||
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
|
||||
assert result.error is None and len(result.hits) == 2
|
||||
|
||||
|
||||
def test_html_200_error_page_is_a_failure_not_an_empty_brand(calls, tmp_path):
|
||||
calls(lambda n, p: _Resp(status=200, text="<html>temporarily unavailable</html>",
|
||||
content_type="text/html"))
|
||||
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
|
||||
assert result.error and result.hits == []
|
||||
assert list(tmp_path.glob("*.json")) == []
|
||||
|
||||
|
||||
def test_transport_error_after_retries_is_reported(calls, tmp_path):
|
||||
calls(lambda n, p: requests.exceptions.ConnectionError("dns"))
|
||||
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
|
||||
assert result.error and result.hits == []
|
||||
assert list(tmp_path.glob("*.json")) == []
|
||||
|
||||
|
||||
def test_partial_fetch_keeps_what_arrived_but_flags_it(calls, tmp_path):
|
||||
page1 = [{"code": f"890{i:010d}", "product_name": f"P{i}"} for i in range(100)]
|
||||
|
||||
def responder(n, params):
|
||||
if params["page"] == 1:
|
||||
return _Resp(payload=_v2(page1, count=150))
|
||||
return _Resp(status=502)
|
||||
|
||||
calls(responder)
|
||||
result = fetch_brand_corpus_result("Amul", cache_dir=tmp_path)
|
||||
assert len(result.hits) == 100 and result.error
|
||||
assert list(tmp_path.glob("*.json")) == []
|
||||
|
||||
|
||||
def test_list_form_still_returns_empty_on_failure(calls, tmp_path):
|
||||
"""Backfill scripts and ingestion call `fetch_brand_corpus` and rely on
|
||||
[] rather than an exception so one dead brand does not abort a run."""
|
||||
calls(lambda n, p: _Resp(status=503))
|
||||
assert fetch_brand_corpus("Naga", cache_dir=tmp_path) == []
|
||||
|
||||
|
||||
def test_off_unavailable_is_a_connection_error_for_the_retry_decorator():
|
||||
assert issubclass(OffUnavailable, requests.exceptions.ConnectionError)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The cache
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_successful_fetch_is_cached_with_schema_and_served_from_cache(calls, tmp_path):
|
||||
sent = calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
|
||||
|
||||
first = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
|
||||
second = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
|
||||
|
||||
assert len(sent) == 1
|
||||
assert not first.from_cache and second.from_cache
|
||||
assert second.hits == first.hits
|
||||
on_disk = json.loads((tmp_path / "naga.json").read_text(encoding="utf-8"))
|
||||
assert on_disk["schema"] == CACHE_SCHEMA
|
||||
assert on_disk["brand_tag"] == "naga"
|
||||
assert on_disk["endpoint"] == off_bulk.SEARCH_URL
|
||||
|
||||
|
||||
def test_genuinely_empty_brand_is_cached_and_is_not_an_error(calls, tmp_path):
|
||||
sent = calls(lambda n, p: _Resp(payload=_v2([])))
|
||||
|
||||
result = fetch_brand_corpus_result("Udhaiyam", cache_dir=tmp_path)
|
||||
again = fetch_brand_corpus_result("Udhaiyam", cache_dir=tmp_path)
|
||||
|
||||
assert result == CorpusFetch(hits=[], error=None, from_cache=False)
|
||||
assert again.from_cache and again.hits == []
|
||||
assert len(sent) == 1
|
||||
|
||||
|
||||
def test_empty_pre_schema_cache_is_refetched(calls, tmp_path):
|
||||
"""The exact prod artefact: an old-endpoint file saying Naga has nothing."""
|
||||
(tmp_path / "naga.json").write_text(json.dumps({
|
||||
"brand": "Naga", "country": "india", "fetched_at": 0,
|
||||
"fetched_at_human": "2026-09-10 12:00:00", "hits": [],
|
||||
}), encoding="utf-8")
|
||||
sent = calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
|
||||
|
||||
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
|
||||
|
||||
assert len(sent) == 1 and not result.from_cache
|
||||
assert len(result.hits) == 2
|
||||
assert json.loads((tmp_path / "naga.json").read_text(encoding="utf-8"))["schema"] == CACHE_SCHEMA
|
||||
|
||||
|
||||
def test_non_empty_pre_schema_cache_is_still_trusted(calls, tmp_path):
|
||||
(tmp_path / "amul.json").write_text(json.dumps({
|
||||
"brand": "Amul", "country": "india", "fetched_at": 0,
|
||||
"hits": [{"code": "8901262010115", "product_name": "Amul Butter"}],
|
||||
}), encoding="utf-8")
|
||||
sent = calls(lambda n, p: _Resp(status=503))
|
||||
|
||||
result = fetch_brand_corpus_result("Amul", cache_dir=tmp_path)
|
||||
|
||||
assert sent == [] and result.from_cache
|
||||
assert result.hits[0]["product_name"] == "Amul Butter"
|
||||
|
||||
|
||||
def test_refresh_bypasses_the_cache(calls, tmp_path):
|
||||
(tmp_path / "naga.json").write_text(json.dumps({
|
||||
"schema": CACHE_SCHEMA, "brand": "Naga", "hits": [],
|
||||
}), encoding="utf-8")
|
||||
sent = calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
|
||||
assert len(fetch_brand_corpus_result("Naga", refresh=True, cache_dir=tmp_path).hits) == 2
|
||||
assert len(sent) == 1
|
||||
|
||||
|
||||
def test_unreadable_cache_is_refetched(calls, tmp_path):
|
||||
(tmp_path / "naga.json").write_text("{not json", encoding="utf-8")
|
||||
calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
|
||||
assert len(fetch_brand_corpus_result("Naga", cache_dir=tmp_path).hits) == 2
|
||||
@@ -159,7 +159,11 @@ def test_the_sheets_values_are_kept_and_nothing_else_is_invented(store):
|
||||
# What the sheet said.
|
||||
assert row["product_name"] == "Apple 500g"
|
||||
assert row["size_variants"] == ["500g"]
|
||||
assert row["final_selling_price"] == 155
|
||||
# "Selling Price" is the shop's price and lands in selling_price. The
|
||||
# final (tax-inclusive) column is filled from it by upsert_brand_products,
|
||||
# which this fixture stubs, so here it stays what the sheet said: nothing.
|
||||
assert row["selling_price"] == 155
|
||||
assert row["final_selling_price"] is None
|
||||
assert row["price_range"] == "₹143-167" # +/-8% of the sheet's own price
|
||||
|
||||
# What it did not say, and what we therefore do not claim.
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
WHY THIS EXISTS
|
||||
---------------
|
||||
Open Food Facts answers "does this product exist" for food and nothing else.
|
||||
`off_bulk` queries only `search.openfoodfacts.org`, so a toothpaste or a
|
||||
`off_bulk` queries only Open Food Facts' food database, so a toothpaste or a
|
||||
detergent has no product-discovery source at all and its whole catalogue is
|
||||
language-model output. A live retail lookup is the only signal in the codebase
|
||||
that means "available right now" rather than "was in a database dump", and it
|
||||
|
||||
312
tests/test_upload_sheet_fields.py
Normal file
312
tests/test_upload_sheet_fields.py
Normal file
@@ -0,0 +1,312 @@
|
||||
"""What an uploaded sheet's own facts become in the brand table.
|
||||
|
||||
Three rules, each of which was violated before this file existed:
|
||||
|
||||
1. THE RETAIL PRICE IS THE SELLING PRICE. Every price header used to land in
|
||||
`final_selling_price` and `selling_price` was NULL for every uploaded row;
|
||||
worse, the HSN/GST stage then overwrote whatever price the sheet gave with
|
||||
the ceiling of the derived band (155 became 167).
|
||||
2. A PACK SIZE IS RECORDED ONLY WHEN THE SHEET OR THE NAME STATES IT.
|
||||
"India Gate Basmati Rice 1kg" is one product in one size. "India Gate
|
||||
Basmati Rice" used to become three products - 1kg, 5kg, 25kg - none of
|
||||
which the sheet listed.
|
||||
3. A BLANK DESCRIPTION IS WRITTEN, NOT PADDED. The LLM writes it when Ollama
|
||||
is up; when it is not, the row gets a short factual line built only from
|
||||
facts the row holds - never "<name> from <brand>." and never the marketing
|
||||
essay in catalog_engine.generate_detailed_description.
|
||||
|
||||
Everything runs offline: storage and embeddings are monkeypatched on the
|
||||
pipeline module, and USE_OLLAMA is false for the whole suite (conftest.py).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
|
||||
import pytest
|
||||
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.services.enrichment.hsn_gst.models import enrich_pricing_fields
|
||||
from app.services.generic_products import OWN_PRODUCTS_BRAND
|
||||
|
||||
openpyxl = pytest.importorskip("openpyxl")
|
||||
|
||||
|
||||
def _sheet(headers, rows) -> bytes:
|
||||
wb = openpyxl.Workbook()
|
||||
ws = wb.active
|
||||
ws.append(headers)
|
||||
for row in rows:
|
||||
ws.append(row)
|
||||
buf = io.BytesIO()
|
||||
wb.save(buf)
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolate_sku_counter(tmp_path, monkeypatch):
|
||||
from app.services import sku_service
|
||||
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store(monkeypatch):
|
||||
"""A fake brand table: image_id -> stored row, across every brand."""
|
||||
table: dict = {}
|
||||
|
||||
def fake_upsert(brand, rows, cleanup=False):
|
||||
assert cleanup is False
|
||||
for row in rows:
|
||||
table[row["image_id"]] = dict(row)
|
||||
return len(rows)
|
||||
|
||||
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
|
||||
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: list(table.values()))
|
||||
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
|
||||
return table
|
||||
|
||||
|
||||
def _run(headers, rows, **kw):
|
||||
kw.setdefault("use_llm", False)
|
||||
kw.setdefault("fetch_images", False)
|
||||
return pipeline.run_pipeline("store.xlsx", _sheet(headers, rows), **kw)
|
||||
|
||||
|
||||
# Real-looking names: the validation gate rejects placeholder titles such as
|
||||
# "Product 3", which would empty the store and hide what a test is about.
|
||||
_AMUL_NAMES = ["Amul Butter 100g", "Amul Cheese 200g", "Amul Ghee 500ml", "Amul Taaza 1L",
|
||||
"Amul Kool 200ml", "Amul Lassi 200ml", "Amul Masti Dahi 400g"]
|
||||
|
||||
|
||||
def _only(store):
|
||||
assert len(store) == 1, list(store)
|
||||
return next(iter(store.values()))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. Price
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_retail_price_lands_in_selling_price_and_drives_the_band(store):
|
||||
_run(["ProductName", "ProductSKU", "ProductBrand", "RetailPrice"],
|
||||
[["India Gate Basmati Rice 1kg", "IG-BR-1", "India Gate", 120]])
|
||||
|
||||
row = _only(store)
|
||||
assert row["selling_price"] == 120
|
||||
assert row["price_range"] == "₹110-130" # +/-8% of the sheet's price
|
||||
assert row["product_sku"] == "IG-BR-1" and row["sku_source"] == "sheet"
|
||||
|
||||
|
||||
def test_a_final_price_column_is_kept_separate_and_wins_the_band(store):
|
||||
_run(["Item Name", "Brand", "Selling Price", "Final Price"],
|
||||
[["Amul Butter 500g", "Amul", 250, 262]])
|
||||
|
||||
row = _only(store)
|
||||
assert row["selling_price"] == 250
|
||||
assert row["final_selling_price"] == 262
|
||||
assert row["price_range"] == "₹241-283" # band from the final price
|
||||
|
||||
|
||||
def test_reingesting_a_sheet_with_a_selling_price_is_a_no_op(store):
|
||||
headers, rows = ["Item Name", "Brand", "Retail Price"], [["Amul Butter 500g", "Amul", 250]]
|
||||
first = _run(headers, rows)
|
||||
assert first.inserted == 1
|
||||
|
||||
second = _run(headers, rows)
|
||||
assert (second.inserted, second.backfilled, second.skipped_existing) == (0, 0, 1)
|
||||
|
||||
|
||||
def test_a_held_final_price_is_not_clobbered_by_a_different_retail_price(store):
|
||||
_run(["Item Name", "Brand", "Retail Price"], [["Amul Butter 500g", "Amul", 250]])
|
||||
key = next(iter(store))
|
||||
store[key]["final_selling_price"] = 262 # what the table already holds
|
||||
|
||||
_run(["Item Name", "Brand", "Retail Price"], [["Amul Butter 500g", "Amul", 199]])
|
||||
|
||||
assert store[key]["final_selling_price"] == 262
|
||||
assert store[key]["selling_price"] == 250, "fill-only-blanks: a held price is kept"
|
||||
|
||||
|
||||
class TestHsnStageKeepsHeldPrices:
|
||||
"""The overwrite that turned a sheet's 155 into 167."""
|
||||
|
||||
def test_a_held_selling_price_is_returned_as_none_so_apply_keeps_it(self):
|
||||
fields = enrich_pricing_fields({
|
||||
"selling_price": 155, "final_selling_price": None,
|
||||
"price_range": "₹143-167", "gst_percent": 18,
|
||||
})
|
||||
assert fields["selling_price"] is None
|
||||
assert fields["tax_amount"] == 27.9 # 18% of the HELD 155, not of 167
|
||||
assert fields["final_selling_price"] == 182.9
|
||||
|
||||
def test_a_held_final_price_is_also_kept(self):
|
||||
fields = enrich_pricing_fields({
|
||||
"selling_price": 155, "final_selling_price": 160,
|
||||
"price_range": "₹143-167", "gst_percent": 18,
|
||||
})
|
||||
assert fields["selling_price"] is None
|
||||
assert fields["final_selling_price"] is None
|
||||
|
||||
def test_without_a_sheet_price_the_band_ceiling_is_still_the_base(self):
|
||||
fields = enrich_pricing_fields({"price_range": "₹143-167", "gst_percent": 5})
|
||||
assert fields["selling_price"] == 167
|
||||
assert fields["final_selling_price"] == 175.35
|
||||
|
||||
def test_a_held_final_price_alone_seeds_the_selling_price(self):
|
||||
fields = enrich_pricing_fields({"final_selling_price": 200, "gst_percent": 5,
|
||||
"price_range": "₹300-400"})
|
||||
assert fields["selling_price"] == 200
|
||||
assert fields["final_selling_price"] is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Size
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.mark.parametrize("name, size", [
|
||||
("India Gate Basmati Rice 1kg", "1kg"),
|
||||
("Fortune Sunflower Oil 1L", "1L"),
|
||||
("Amul Taaza Milk 1 Litre", "1 Litre"),
|
||||
("Aashirvaad Atta 5 Kg", "5 Kg"),
|
||||
("Tata Salt 1.5kg", "1.5kg"),
|
||||
("Parle-G Biscuits 800 gm", "800 gm"),
|
||||
("Saffola Gold 500ml", "500ml"),
|
||||
])
|
||||
def test_the_size_in_the_name_is_the_size_and_nothing_is_appended(store, name, size):
|
||||
_run(["Product Name", "Brand"], [[name, name.split()[0]]])
|
||||
|
||||
row = _only(store)
|
||||
assert row["size_variants"] == [size]
|
||||
assert row["product_name"] == name, "the name already carries the size"
|
||||
|
||||
|
||||
def test_a_branded_product_with_no_size_is_one_clean_row(store):
|
||||
result = _run(["Product Name", "Brand"], [["India Gate Basmati Rice", "India Gate"]])
|
||||
|
||||
row = _only(store)
|
||||
assert row["size_variants"] == []
|
||||
assert row["product_name"] == "India Gate Basmati Rice"
|
||||
assert "Standard" not in row["product_name"]
|
||||
assert next(iter(store)) == "india_gate_india_gate_basmati_rice"
|
||||
assert any("no pack size" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_a_commodity_with_no_size_still_uses_the_standard_sentinel(store):
|
||||
"""Own Products dedupes against a seeded base list built on "Standard"."""
|
||||
_run(["Product Name"], [["Apple"]])
|
||||
|
||||
row = _only(store)
|
||||
assert row["size_variants"] == ["Standard"]
|
||||
assert next(iter(store)) == pipeline.build_image_id(OWN_PRODUCTS_BRAND, "Apple", "Standard")
|
||||
|
||||
|
||||
def test_the_llm_may_not_supply_a_pack_size(store, monkeypatch):
|
||||
from app.services import ollama_service
|
||||
|
||||
monkeypatch.setattr(ollama_service, "fetch_product_details",
|
||||
lambda brand, title, category=None, size=None: {
|
||||
"description": "Refined wheat flour milled for baking.",
|
||||
"size_variants": ["500g", "1kg", "5kg"],
|
||||
})
|
||||
|
||||
_run(["Product Name", "Brand"], [["Naga Maida", "Naga"]], use_llm=True)
|
||||
|
||||
row = _only(store)
|
||||
assert row["size_variants"] == []
|
||||
assert row["description"] == "Refined wheat flour milled for baking."
|
||||
|
||||
|
||||
def test_a_sheet_size_column_still_wins_over_the_name(store):
|
||||
_run(["Product Name", "Brand", "Pack Size"], [["Amul Butter 100g", "Amul", "500g"]])
|
||||
assert _only(store)["size_variants"] == ["500g"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. Description
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_the_llm_description_is_used_when_ollama_answers(store, monkeypatch):
|
||||
from app.services import ollama_service
|
||||
|
||||
seen = {}
|
||||
|
||||
def fake(brand, title, category=None, size=None):
|
||||
seen.update(brand=brand, title=title, category=category, size=size)
|
||||
return {"description": "Long-grain aged basmati rice for biryani and pulao."}
|
||||
|
||||
monkeypatch.setattr(ollama_service, "fetch_product_details", fake)
|
||||
|
||||
_run(["Product Name", "Brand"], [["India Gate Basmati Rice 1kg", "India Gate"]], use_llm=True)
|
||||
|
||||
assert _only(store)["description"] == "Long-grain aged basmati rice for biryani and pulao."
|
||||
assert seen["size"] == "1kg", "the pack size in the name is handed to the prompt"
|
||||
|
||||
|
||||
def test_an_overlong_llm_description_is_clamped(store, monkeypatch):
|
||||
from app.services import ollama_service
|
||||
|
||||
monkeypatch.setattr(ollama_service, "fetch_product_details",
|
||||
lambda *a, **k: {"description": "x" * 2000})
|
||||
_run(["Product Name", "Brand"], [["Amul Butter 500g", "Amul"]], use_llm=True)
|
||||
assert len(_only(store)["description"]) == pipeline._MAX_LLM_DESCRIPTION_CHARS
|
||||
|
||||
|
||||
def test_the_fallback_description_is_factual_and_short(store):
|
||||
_run(["Product Name", "Brand", "Category"],
|
||||
[["Fortune Sunflower Oil 1L", "Fortune", "Edible Oils"]])
|
||||
|
||||
desc = _only(store)["description"]
|
||||
assert desc == "Fortune Sunflower Oil 1L, Edible Oils, by Fortune."
|
||||
assert "from Fortune." not in desc and "Introducing" not in desc
|
||||
|
||||
|
||||
def test_the_fallback_description_names_a_sheet_size_the_name_lacks(store):
|
||||
_run(["Product Name", "Brand", "Pack Size"], [["Amul Butter", "Amul", "500g"]])
|
||||
assert _only(store)["description"] == "Amul Butter 500g, Dairy, by Amul."
|
||||
|
||||
|
||||
def test_a_sheet_description_is_never_overwritten(store, monkeypatch):
|
||||
from app.services import ollama_service
|
||||
|
||||
monkeypatch.setattr(ollama_service, "fetch_product_details",
|
||||
lambda *a, **k: pytest.fail("the LLM must not be asked"))
|
||||
_run(["Product Name", "Brand", "Description"],
|
||||
[["Amul Butter 500g", "Amul", "Salted table butter."]], use_llm=True)
|
||||
assert _only(store)["description"] == "Salted table butter."
|
||||
|
||||
|
||||
def test_the_llm_is_switched_off_after_three_consecutive_misses(store, monkeypatch):
|
||||
from app.services import ollama_service
|
||||
|
||||
calls = []
|
||||
monkeypatch.setattr(ollama_service, "fetch_product_details",
|
||||
lambda brand, title, **k: calls.append(title) or None)
|
||||
|
||||
rows = [[name, "Amul"] for name in _AMUL_NAMES[:6]]
|
||||
result = _run(["Product Name", "Brand"], rows, use_llm=True)
|
||||
|
||||
assert len(calls) == 3
|
||||
assert any("consecutive" in w for w in result.warnings)
|
||||
assert len(store) == 6, "the rows themselves are unaffected"
|
||||
|
||||
|
||||
def test_one_answer_resets_the_breaker(store, monkeypatch):
|
||||
from app.services import ollama_service
|
||||
|
||||
answers = iter([None, None, {"description": "ok"}, None, None, None, None])
|
||||
calls = []
|
||||
|
||||
def fake(brand, title, **k):
|
||||
calls.append(title)
|
||||
return next(answers)
|
||||
|
||||
monkeypatch.setattr(ollama_service, "fetch_product_details", fake)
|
||||
_run(["Product Name", "Brand"], [[name, "Amul"] for name in _AMUL_NAMES], use_llm=True)
|
||||
|
||||
# miss, miss, hit (reset), miss, miss, miss (trip) -> the 7th row is not asked.
|
||||
assert len(calls) == 6
|
||||
|
||||
|
||||
def test_stage_two_leaves_a_blank_description_blank_on_a_miss():
|
||||
"""The fallback lives in _to_storage_row; stage 2 must not pre-empt it."""
|
||||
row = {"brand": "Amul", "product_name": "Amul Butter 500g", "description": ""}
|
||||
breaker = pipeline.LlmBreaker()
|
||||
pipeline.stage_2_row_intake(row, use_llm=True, breaker=breaker)
|
||||
assert row["description"] == ""
|
||||
assert breaker.consecutive_misses == 1
|
||||
@@ -187,6 +187,45 @@ def test_template_headers_all_map_to_their_field():
|
||||
assert mapping.ignored == []
|
||||
|
||||
|
||||
@pytest.mark.parametrize("header, canonical", [
|
||||
# The camel-cased headers a store's export tool produces, which
|
||||
# _normalize_header collapses to one word - locked as exact aliases so
|
||||
# they do not depend on the keyword rules' ordering.
|
||||
("ProductName", "product_name"),
|
||||
("ProductBrand", "brand"),
|
||||
("Product Brand", "brand"),
|
||||
("ProductSKU", "product_sku"),
|
||||
("Product SKU", "product_sku"),
|
||||
# What the shop charges -> selling_price.
|
||||
("RetailPrice", "selling_price"),
|
||||
("Retail Price", "selling_price"),
|
||||
("Selling Price", "selling_price"),
|
||||
("Sale Price", "selling_price"),
|
||||
("SP", "selling_price"),
|
||||
("Price", "selling_price"),
|
||||
("Unit Price (Rs)", "selling_price"),
|
||||
# The tax-inclusive ceiling -> final_selling_price. "Final Selling Price"
|
||||
# contains "selling price" and must still win: rule order regression.
|
||||
("Final Selling Price", "final_selling_price"),
|
||||
("Final Price (₹)", "final_selling_price"),
|
||||
("MRP", "final_selling_price"),
|
||||
("Maximum MRP", "final_selling_price"),
|
||||
])
|
||||
def test_price_and_identity_headers_map_where_the_sheet_means_them(header, canonical):
|
||||
"""Before the selling_price canonical existed every price header landed in
|
||||
final_selling_price and the table's selling_price column stayed NULL."""
|
||||
mapping = user_products.map_spreadsheet_columns([header])
|
||||
assert mapping.columns == {canonical: header}
|
||||
|
||||
|
||||
def test_a_sheet_may_carry_both_a_retail_and_a_final_price():
|
||||
mapping = user_products.map_spreadsheet_columns(
|
||||
["Item Name", "Brand", "Retail Price", "MRP"])
|
||||
assert mapping.columns["selling_price"] == "Retail Price"
|
||||
assert mapping.columns["final_selling_price"] == "MRP"
|
||||
assert mapping.ignored == []
|
||||
|
||||
|
||||
def test_a_product_sku_column_is_not_mistaken_for_the_product_name():
|
||||
"""'Product SKU' contains 'product'. Matched loosely, it used to become a
|
||||
second product_name column, and duplicate columns are what turned a row
|
||||
|
||||
Reference in New Issue
Block a user