Brand valid image generation

This commit is contained in:
sriram
2026-09-11 15:47:56 +05:30
parent e1a5962f82
commit ce4fa70dee
31 changed files with 5586 additions and 1495 deletions

View File

@@ -55,6 +55,11 @@ os.environ.setdefault("USE_PGVECTOR", "true")
os.environ.setdefault("DB_PASSWORD", "test-password-not-real")
os.environ.setdefault("USE_S3", "false")
os.environ.setdefault("USE_GOOGLE_CSE", "false")
# Unconditional: the LLM description stage is on by default for every batch
# and a developer machine often has `ollama serve` running, so without this a
# pipeline test would make real, slow, non-deterministic model calls. Tests
# that exercise the probe itself monkeypatch `ollama_service.USE_OLLAMA`.
os.environ["USE_OLLAMA"] = "false"
# Unconditional, NOT setdefault. The suite pins brand-name behaviour all over
# the place ("any Nestle chocolates?", the suggest ranking fixtures), and a

View File

@@ -5,8 +5,8 @@ boundary *on the pipeline module object*, build real spreadsheets in memory,
and point the batch directory at tmp_path so nothing is written into the repo.
Nothing here loads sentence-transformers or torch, and nothing reaches the
network: every batch runs with use_llm=False and fetch_images=False, which are
also the defaults the endpoint ships.
network: every batch runs with fetch_images=False (the endpoint's default) and
the LLM stage, on by default, finds Ollama disabled and falls back at once.
"""
from __future__ import annotations
@@ -486,7 +486,7 @@ def test_ingest_returns_202_and_a_batch_id(client, admin_headers, store, batch_r
assert response.status_code == 202
body = response.json()
assert body["files_total"] == 2
assert body["use_llm"] is False, "the LLM stage must be opt-in for a batch"
assert body["use_llm"] is True, "LLM descriptions are on by default; a batch opts OUT"
assert body["fetch_images"] is False, "image search must be opt-in for a batch"
assert len(body["files"]) == 2
assert body["files"][0]["total_stages"] == pipeline.TOTAL_STAGES

View File

@@ -21,6 +21,9 @@ from app.api.routers.user_products import map_spreadsheet_columns, row_to_reques
from app.core import store_catalog_pipeline as pipeline
from app.services import brand_discovery as bd
# Captured before the autouse `_no_network` fixture replaces it on the module.
_REAL_FROM_OPEN_FACTS = bd._from_open_facts
# ---------------------------------------------------------------------------
# Fixtures
@@ -220,12 +223,11 @@ def test_a_case_pack_count_is_not_part_of_the_product_name(monkeypatch):
# ---------------------------------------------------------------------------
# Regression: one GTIN, one pack
# ---------------------------------------------------------------------------
def test_a_barcode_is_dropped_when_no_real_pack_size_is_known(monkeypatch):
"""A GTIN identifies one pack, and stage 4 invents three when it has none.
Every column is copied into each exploded variant, so a surviving barcode
would be stamped onto two packs that do not exist - wrong data that looks
authoritative.
def test_a_barcode_survives_when_no_pack_size_is_known(monkeypatch):
"""A GTIN identifies one pack, and stage 4 now stores exactly one row for
a product with no size (it used to invent three and stamp the same GTIN
on all of them, which is why the barcode was dropped before). One GTIN on
one unsized row is what a GTIN means; the row is only flagged for review.
"""
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("50 50 Gol Maal", code="8901063017702"),
@@ -234,8 +236,8 @@ def test_a_barcode_is_dropped_when_no_real_pack_size_is_known(monkeypatch):
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.size_variants == []
assert product.barcode is None
assert any("barcode dropped" in note for note in product.notes)
assert product.barcode == "8901063017702"
assert any("pack size unknown" in note for note in product.notes)
def test_a_barcode_survives_when_the_pack_size_is_real(monkeypatch):
@@ -549,3 +551,113 @@ def test_every_column_the_pipeline_can_fill_is_filled(store, monkeypatch):
assert row.get(column), f"{column} was left empty"
assert row["barcode"] == "8901063012516"
assert row["fssai_license"] == "10012022000103"
# ---------------------------------------------------------------------------
# What the warnings claim - "unreachable" vs "empty" vs "no storefront"
# ---------------------------------------------------------------------------
# Naga (2026-09-11): Open Food Facts holds two real rows for the brand, yet the
# tab said "Neither Open Food Facts nor a brand storefront has anything for
# this brand". The lookup had failed / been mis-indexed, no storefront was
# registered, and one warning text covered all of it. Each claim now has to
# be true on its own.
def _store_rows(monkeypatch, rows):
monkeypatch.setattr(bd, "_from_brand_store", lambda brand: list(rows))
def test_an_unreachable_off_is_reported_as_unreachable_not_empty(monkeypatch):
def down(brand, refresh=False):
raise bd.OpenFactsUnavailable("HTTP 503 from Open Food Facts")
monkeypatch.setattr(bd, "_from_open_facts", down)
_store_rows(monkeypatch, [])
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
{"title": "Naga Sooji", "category": None, "description": None,
"sizes": ["500g"], "providers": [], "source": "llm"},
])
result = bd.discover_brand_products("Naga", require_evidence=False)
reached = [w for w in result.warnings if "could not be reached" in w]
assert len(reached) == 1 and "503" in reached[0]
assert not any("has no products tagged" in w for w in result.warnings)
assert [p.title for p in result.products] == ["Naga Sooji"], "the LLM rows must still come through when OFF is down"
def test_a_genuinely_empty_off_says_so_without_blaming_the_network(monkeypatch):
_store_rows(monkeypatch, [])
result = bd.discover_brand_products("Udhaiyam")
assert any("has no products tagged" in w for w in result.warnings)
assert not any("could not be reached" in w for w in result.warnings)
def test_unregistered_storefront_is_not_described_as_empty(monkeypatch):
_store_rows(monkeypatch, [])
monkeypatch.setattr(bd, "get_brand_store_domain", lambda brand: None)
result = bd.discover_brand_products("Naga")
assert any("No verified storefront is registered" in w for w in result.warnings)
assert not any("Neither Open Food Facts" in w for w in result.warnings)
assert not any("has not been fetched" in w for w in result.warnings)
def test_registered_but_unfetched_storefront_names_the_backfill(monkeypatch):
_store_rows(monkeypatch, [])
monkeypatch.setattr(bd, "get_brand_store_domain", lambda brand: "gopuramproducts.com")
result = bd.discover_brand_products("Gopuram")
hit = [w for w in result.warnings if "has not been fetched" in w]
assert len(hit) == 1
assert "gopuramproducts.com" in hit[0] and "backfill_brand_stores.py" in hit[0]
assert not any("No verified storefront" in w for w in result.warnings)
def test_the_llm_only_caveat_appears_exactly_once(monkeypatch):
_store_rows(monkeypatch, [])
result = bd.discover_brand_products("Naga")
assert sum("rests on the language model alone" in w for w in result.warnings) == 1
def test_no_storefront_warning_when_the_shop_or_off_has_rows(monkeypatch):
_store_rows(monkeypatch, [{"title": "Naga Maida 1kg", "source": "store", "size": "1kg"}])
result = bd.discover_brand_products("Naga")
assert not any("storefront" in w.lower() for w in result.warnings)
assert not any("rests on the language model alone" in w for w in result.warnings)
_store_rows(monkeypatch, [])
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Sooji", code="8906011830068", quantity="500 g"),
])
result = bd.discover_brand_products("Naga")
assert not any("storefront" in w.lower() for w in result.warnings)
assert not any("rests on the language model alone" in w for w in result.warnings)
def test_from_open_facts_raises_when_the_fetch_failed_with_nothing(monkeypatch):
from app.services.enrichment.barcode.sources import off_bulk
monkeypatch.setattr(off_bulk, "fetch_brand_corpus_result",
lambda brand, refresh=False: off_bulk.CorpusFetch(hits=[], error="dns"))
with pytest.raises(bd.OpenFactsUnavailable):
_REAL_FROM_OPEN_FACTS("Naga")
def test_from_open_facts_maps_v2_hits_to_candidates(monkeypatch):
from app.services.enrichment.barcode.sources import off_bulk
monkeypatch.setattr(off_bulk, "fetch_brand_corpus_result",
lambda brand, refresh=False: off_bulk.CorpusFetch(hits=[
{"code": "8906011830068", "product_name": "Sooji",
"product_name_en": "Sooji", "quantity": "500 g"},
{"code": "8906011831713", "product_name": "Maida", "quantity": "500 g"},
]))
rows = _REAL_FROM_OPEN_FACTS("Naga")
assert [(r["title"], r["barcode"], r["source"]) for r in rows] == [
("Sooji", "8906011830068", "off"), ("Maida", "8906011831713", "off"),
]

View File

@@ -174,6 +174,94 @@ def test_a_lookup_failure_never_rejects_an_image(monkeypatch):
assert ic.openfacts_product_matches_brand(url, "Anil") is True
class _Resp:
def __init__(self, ok=True, payload=None):
self.ok = ok
self.status_code = 200 if ok else 503
self._payload = payload or {}
def json(self):
return self._payload
@pytest.fixture
def verdict_cache(tmp_path, monkeypatch):
"""A throwaway sqlite cache, so these tests neither read nor poison the
real one in data/cache."""
monkeypatch.setattr(ic, "_DB_PATH", tmp_path / "verdicts.db")
monkeypatch.setattr(ic, "_initialized", False) # the schema is per file
monkeypatch.setattr(ic.time, "sleep", lambda s: None) # pacing/retry waits
return tmp_path / "verdicts.db"
KELLOGGS_HONEY = "https://images.openfoodfacts.org/images/products/505/931/902/3762/front_fr.32.400.jpg"
def test_a_throttled_lookup_fails_open_but_is_not_remembered(monkeypatch, verdict_cache):
"""The Dabur Honey defect. A hundred lookups in two minutes got the tail
of them a 503; the fail-open True was cached for thirty days, and
Kellogg's "Miel Pops" became a Dabur product for a month."""
throttled = _Resp(ok=False)
throttled.status_code = 503
answers = iter([throttled, throttled, _Resp(payload={"status": 1, "product": {
"brands": "KELLOG'S", "product_name": "Miel Pops"}})])
monkeypatch.setattr(ic.requests, "get", lambda *a, **k: next(answers))
assert ic.openfacts_product_matches_brand(KELLOGGS_HONEY, "Dabur") is True, "throttled on both attempts: fails open now"
assert ic.openfacts_product_matches_brand(KELLOGGS_HONEY, "Dabur") is False, "asks again next time"
def test_one_throttled_attempt_is_retried_into_a_real_answer(monkeypatch, verdict_cache):
throttled = _Resp(ok=False)
throttled.status_code = 429
answers = iter([throttled, _Resp(payload={"status": 1, "product": {
"brands": "KELLOG'S", "product_name": "Miel Pops"}})])
monkeypatch.setattr(ic.requests, "get", lambda *a, **k: next(answers))
assert ic.openfacts_product_matches_brand(KELLOGGS_HONEY, "Dabur") is False
def test_a_real_verdict_is_cached(monkeypatch, verdict_cache):
calls = []
def get(*a, **k):
calls.append(1)
return _Resp(payload={"status": 1, "product": {"brands": "Toblerone",
"product_name": "Milk Chocolate"}})
monkeypatch.setattr(ic.requests, "get", get)
url = "https://images.openfoodfacts.org/images/products/761/450/001/0013/front_en.362.400.jpg"
assert ic.openfacts_product_matches_brand(url, "Dabur") is False
assert ic.openfacts_product_matches_brand(url, "Dabur") is False
assert len(calls) == 1
def test_an_unknown_barcode_is_cached_as_open(monkeypatch, verdict_cache):
calls = []
def get(*a, **k):
calls.append(1)
return _Resp(payload={"status": 0, "status_verbose": "product not found"})
monkeypatch.setattr(ic.requests, "get", get)
url = "https://images.openfoodfacts.org/images/products/000/000/000/0001/front.jpg"
assert ic.openfacts_product_matches_brand(url, "Dabur") is True
assert ic.openfacts_product_matches_brand(url, "Dabur") is True
assert len(calls) == 1
def test_pre_fix_cache_entries_are_ignored(monkeypatch, verdict_cache):
"""Entries written by the old code (no "v2:" prefix) may be fail-open
Trues; they must not be trusted."""
ic._cache_set("world.openfoodfacts.org:5059319023762:dabur", True)
monkeypatch.setattr(ic.requests, "get", lambda *a, **k: _Resp(payload={
"status": 1, "product": {"brands": "KELLOG'S", "product_name": "Miel Pops"}}))
assert ic.openfacts_product_matches_brand(KELLOGGS_HONEY, "Dabur") is False
def test_a_non_openfacts_url_is_not_cross_checked():
assert ic.openfacts_product_matches_brand(REAL_RAVA, "Anil") is True
@@ -193,3 +281,20 @@ def test_pack_sizes_of_one_product_share_a_search_key():
def test_different_products_do_not_share_a_search_key():
assert (ic.search_key("Anil Wheat Vermicelli 180g", "Anil")
!= ic.search_key("Anil Samba Rava 500g", "Anil"))
# ---------------------------------------------------------------------------
# Plural titles, singular filenames
# ---------------------------------------------------------------------------
def test_a_plural_title_is_named_by_a_singular_filename():
"""The all-brands re-gate would have dropped the correct
`Aachi-Appalam-100-g-1.webp` from "Aachi Appalams 500g" and promoted an
opaque Amazon URL over it."""
url = "https://thedesifood.com/media/Aachi-Appalam-100-g-1.webp"
assert ic.corroborate(url, "Aachi Appalams 500g", "Aachi").corroborated
assert ic.corroborate("https://x.in/img/aachi-pickle-jar.jpg", "Aachi Pickles 200g", "Aachi").corroborated
def test_the_singular_fallback_does_not_widen_short_tokens():
# "gems" -> "gem" is too short a stem to trust; four letters is the floor.
assert not ic.corroborate("https://x.in/gem-ring.jpg", "Cadbury Gems 20g", "Cadbury").corroborated

View File

@@ -0,0 +1,297 @@
"""The Dabur Honey regression: one product, several pack sizes, one photo.
Brand Discovery for Dabur (2026-09-11) stored, for the same product:
Dabur Honey 225g Amazon photo of Dabur Honey correct
Dabur Honey 1kg image_url empty; image_urls = 4 honeys from the UK,
Dabur Honey 250g France, Switzerland and Spain (Open Food Facts photos
Dabur Honey 50g whose barcodes belong to other brands)
Dabur Honey 500g a Dabur Odomos mosquito repellent
Three causes, each pinned below:
1. Stage 6 searched once PER SIZE. The query was identical - the size is not
in it - but the providers are live and not deterministic, so four asks
came back four ways. One search per product, shared by every size.
2. `find_images_openfacts` threw away the `brands` field of the records it
fetched, so a title-only fallback ("Honey 1kg") handed over every honey on
the site. The brand is checked at the source now.
3. `choose_primary` kept rejected candidates in `image_urls` "for review" -
and the product card falls back to `image_urls[0]` when the primary is
withheld, so the review pile was what the shopper saw. Only eligible
candidates are stored.
Everything runs offline: the search function and the Open*Facts barcode
cross-check are monkeypatched; storage and embeddings are stubbed.
"""
from __future__ import annotations
import io
import pytest
from app.core import store_catalog_pipeline as pipeline
from app.services import image_corroboration as ic
from app.services import image_search
openpyxl = pytest.importorskip("openpyxl")
AMAZON = "https://m.media-amazon.com/images/I/71O4OnjaHVL.jpg"
ODOMOS = ("https://www.indianproductsstore.com/uploads/products/"
"dabur-odomos-naturals-mosquito-repellent-gel.jpg")
FOREIGN_HONEYS = [
"https://images.openfoodfacts.org/images/products/505/931/902/3762/front_fr.32.400.jpg",
"https://images.openfoodfacts.org/images/products/308/854/000/4440/front_fr.129.400.jpg",
]
DABUR_OFF = "https://images.openfoodfacts.org/images/products/890/120/702/6553/front_en.4.400.jpg"
NAMED = "https://cdn.example.in/dabur-honey-squeezy-bottle.jpg"
def _sheet(headers, rows) -> bytes:
wb = openpyxl.Workbook()
ws = wb.active
ws.append(headers)
for row in rows:
ws.append(row)
buf = io.BytesIO()
wb.save(buf)
return buf.getvalue()
@pytest.fixture(autouse=True)
def _isolate_sku_counter(tmp_path, monkeypatch):
from app.services import sku_service
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
@pytest.fixture(autouse=True)
def _off_barcodes_offline(monkeypatch):
"""Open*Facts says: the 890120... barcode is Dabur, the others are not."""
monkeypatch.setattr(
ic, "openfacts_product_matches_brand",
lambda url, brand, timeout=10: "/890/120/" in url or "openfoodfacts" not in url,
)
@pytest.fixture
def store(monkeypatch):
table: dict = {}
def fake_upsert(brand, rows, cleanup=False):
for row in rows:
table[row["image_id"]] = dict(row)
return len(rows)
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
monkeypatch.setattr(pipeline, "get_products_by_brand",
lambda b, **kw: [dict(r, brand=b) for r in table.values()])
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
return table
@pytest.fixture
def search(monkeypatch):
"""A scripted `find_all_image_urls`; records every call it receives."""
calls = []
def install(results):
def fake(title, brand=None, country_hint=None, validate=True, max_results=24, produce=False):
calls.append((title, brand))
return list(results)
monkeypatch.setattr(image_search, "find_all_image_urls", fake)
return calls
return install
def _run(headers, rows):
return pipeline.run_pipeline("store.xlsx", _sheet(headers, rows),
use_llm=False, fetch_images=True)
# ---------------------------------------------------------------------------
# 1. One search per product
# ---------------------------------------------------------------------------
def test_every_pack_size_of_a_product_gets_the_same_image_from_one_search(store, search):
calls = search([AMAZON, DABUR_OFF])
_run(["Product Name", "Brand", "Pack Size"], [["Dabur Honey", "Dabur", "1kg, 250g, 50g, 225g"]])
assert len(store) == 4
assert {r["image_url"] for r in store.values()} == {AMAZON}
assert len(calls) == 1, "four pack sizes, one search"
assert calls[0][0] == "Dabur Honey", "the size never enters the query"
def test_sizes_written_into_the_name_still_share_the_search(store, search):
calls = search([AMAZON])
_run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"], ["Dabur Honey 250g", "Dabur"]])
assert len(calls) == 1
assert calls[0][0] == "Dabur Honey"
assert {r["image_url"] for r in store.values()} == {AMAZON}
def test_different_products_of_one_brand_are_searched_separately(store, search):
calls = search([NAMED])
_run(["Product Name", "Brand"], [["Dabur Honey", "Dabur"], ["Dabur Chyawanprash", "Dabur"]])
assert sorted(c[0] for c in calls) == ["Dabur Chyawanprash", "Dabur Honey"]
# ---------------------------------------------------------------------------
# 2. Open*Facts records are filtered by brand at the source
# ---------------------------------------------------------------------------
def test_openfacts_photos_of_other_brands_honey_are_not_candidates(monkeypatch):
records = [
{"brands": "Dabur", "brands_tags": ["dabur"], "image_front_url": DABUR_OFF},
{"brands": "Rowse", "brands_tags": ["rowse"], "image_front_url": FOREIGN_HONEYS[0]},
{"brands": "Lune de Miel", "image_front_url": FOREIGN_HONEYS[1]},
{"brands": "", "image_front_url": "https://images.openfoodfacts.org/x/unlabelled.jpg"},
]
monkeypatch.setattr(image_search, "_query_openfacts", lambda query, n: records)
monkeypatch.setattr(image_search, "USE_OPEN_FACTS", True)
urls = image_search.find_images_openfacts("Honey", "Dabur")
assert DABUR_OFF in urls
assert not any(u in urls for u in FOREIGN_HONEYS)
assert "https://images.openfoodfacts.org/x/unlabelled.jpg" in urls, \
"a record with no brand at all is not evidence of another brand"
def test_the_title_only_fallback_is_also_brand_filtered(monkeypatch):
seen = []
def query(q, n):
seen.append(q)
if q.startswith("dabur"):
return []
return [{"brands": "Rowse", "image_front_url": FOREIGN_HONEYS[0]}]
monkeypatch.setattr(image_search, "_query_openfacts", query)
monkeypatch.setattr(image_search, "USE_OPEN_FACTS", True)
assert image_search.find_images_openfacts("Honey 1kg", "dabur") == []
assert seen == ["dabur Honey 1kg", "Honey 1kg"]
def test_without_a_brand_nothing_is_filtered(monkeypatch):
monkeypatch.setattr(image_search, "_query_openfacts",
lambda q, n: [{"brands": "Rowse", "image_front_url": FOREIGN_HONEYS[0]}])
monkeypatch.setattr(image_search, "USE_OPEN_FACTS", True)
assert image_search.find_images_openfacts("Honey", None) == FOREIGN_HONEYS[:1]
# ---------------------------------------------------------------------------
# 3. Rejected candidates are not stored where the UI will show them
# ---------------------------------------------------------------------------
def test_choose_primary_separates_eligible_from_rejected():
choice = ic.choose_primary([ODOMOS, FOREIGN_HONEYS[0], AMAZON, NAMED], "Dabur Honey 500g", "Dabur")
assert choice.primary == NAMED # tier 1: names "honey"
assert choice.eligible == [NAMED, AMAZON] # tier 1 then tier 2
assert choice.rejected == [ODOMOS, FOREIGN_HONEYS[0]]
assert choice.ordered == choice.eligible + choice.rejected
def test_a_same_brand_wrong_product_photo_is_never_stored(store, search):
"""The 500g row: a Dabur Odomos repellent as the primary image of honey."""
search([ODOMOS] + FOREIGN_HONEYS)
result = _run(["Product Name", "Brand"], [["Dabur Honey 500g", "Dabur"]])
row = next(iter(store.values()))
assert row["image_url"] is None
assert row["image_urls"] == [], "nothing the gate rejected may reach image_urls"
assert any("rejected" in w for w in result.warnings)
assert any("no primary image" in w for w in result.warnings)
def test_foreign_honeys_ride_along_only_when_a_real_primary_exists(store, search):
search([AMAZON, DABUR_OFF] + FOREIGN_HONEYS)
_run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"]])
row = next(iter(store.values()))
assert row["image_url"] == AMAZON
assert row["image_urls"] == [AMAZON, DABUR_OFF]
# ---------------------------------------------------------------------------
# 4. A sibling pack size already in the table lends its photo - through the gate
# ---------------------------------------------------------------------------
def test_a_new_pack_size_adopts_the_siblings_image_without_searching(store, search):
calls = search([AMAZON])
_run(["Product Name", "Brand"], [["Dabur Honey 225g", "Dabur"]])
assert len(calls) == 1 and store["dabur_dabur_honey_225g"]["image_url"] == AMAZON
result = _run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"]])
assert len(calls) == 1, "the sibling's photo made the search unnecessary"
assert store["dabur_dabur_honey_1kg"]["image_url"] == AMAZON
assert store["dabur_dabur_honey_1kg"]["image_urls"] == [AMAZON]
assert any("adopted from another pack size" in w for w in result.warnings)
def test_a_siblings_wrong_image_is_not_adopted(store, search):
"""A stored primary that would not pass the gate for this title stays put."""
store["dabur_dabur_honey_500g"] = {
"image_id": "dabur_dabur_honey_500g", "product_name": "Dabur Honey 500g",
"image_url": ODOMOS, "image_urls": [ODOMOS],
}
calls = search([NAMED])
_run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"]])
assert len(calls) == 1, "an ineligible sibling image forces a fresh search"
assert store["dabur_dabur_honey_1kg"]["image_url"] == NAMED
def test_a_different_product_of_the_brand_is_not_a_sibling(store, search):
store["dabur_dabur_chyawanprash_500g"] = {
"image_id": "dabur_dabur_chyawanprash_500g", "product_name": "Dabur Chyawanprash 500g",
"image_url": NAMED, "image_urls": [NAMED],
}
calls = search([AMAZON])
_run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"]])
assert len(calls) == 1
assert store["dabur_dabur_honey_1kg"]["image_url"] == AMAZON
def test_stage_six_without_a_cache_still_works():
"""Direct callers (and the produce tests) pass no cache."""
row = {"brand": "Dabur", "product_name": "Dabur Honey 1kg", "image_urls": [AMAZON]}
assert pipeline.stage_6_images(row) is row
# ---------------------------------------------------------------------------
# 5. The one-off repair for rows written before the fix
# ---------------------------------------------------------------------------
def test_regate_rows_reproduces_the_dabur_repair():
from scripts.regate_brand_images import regate_rows
rows = [
{"image_id": "dabur_dabur_honey_225g", "product_name": "Dabur Honey 225g",
"image_url": AMAZON, "image_urls": [AMAZON, DABUR_OFF] + FOREIGN_HONEYS},
{"image_id": "dabur_dabur_honey_1kg", "product_name": "Dabur Honey 1kg",
"image_url": None, "image_urls": list(FOREIGN_HONEYS)},
{"image_id": "dabur_dabur_honey_500g", "product_name": "Dabur Honey 500g",
"image_url": ODOMOS, "image_urls": [ODOMOS]},
{"image_id": "dabur_dabur_chyawanprash_500g", "product_name": "Dabur Chyawanprash 500g",
"image_url": None, "image_urls": [ODOMOS]},
]
by_id = {v["image_id"]: v for v in regate_rows(rows, "dabur")}
# 225g keeps its photo and loses the foreign honeys from its gallery.
assert by_id["dabur_dabur_honey_225g"]["new_url"] == AMAZON
assert by_id["dabur_dabur_honey_225g"]["new_urls"] == [AMAZON, DABUR_OFF]
# 1kg had nothing eligible and adopts the 225g photo.
assert by_id["dabur_dabur_honey_1kg"]["new_url"] == AMAZON
assert by_id["dabur_dabur_honey_1kg"]["why"].startswith("adopted")
# 500g drops the mosquito repellent and adopts the sibling photo too.
assert by_id["dabur_dabur_honey_500g"]["new_url"] == AMAZON
assert ODOMOS in by_id["dabur_dabur_honey_500g"]["rejected"]
# A different product with only a wrong image is blanked, not lent a honey.
assert by_id["dabur_dabur_chyawanprash_500g"]["new_url"] is None
assert by_id["dabur_dabur_chyawanprash_500g"]["new_urls"] == []

View File

@@ -0,0 +1,308 @@
"""The network half of `off_bulk`: what is asked of Open Food Facts, how the
answer is paged, and - the part that produced a user-visible lie - what gets
written to the corpus cache when the answer never arrives.
Background, measured 2026-09-11. Brand Discovery told the user that Open Food
Facts had nothing for "Naga". The v2 API returns two real Naga Limited rows
(Sooji 500 g / 8906011830068, Maida 500 g / 8906011831713). The old query hit
the Search-a-licious endpoint with a free-text `brands:Naga`, whose index
still filed 8906011830068 brandless under Kuwait, and any 503 it met on the
way was cached to disk as `hits: []`. Both failure modes are pinned here, with
no network: `requests.get` is replaced for every test.
"""
from __future__ import annotations
import json
import pytest
import requests
from app.services.enrichment.barcode.sources import off_bulk
from app.services.enrichment.barcode.sources.off_bulk import (
CACHE_SCHEMA,
CorpusFetch,
OffUnavailable,
brand_tag_slug,
fetch_brand_corpus,
fetch_brand_corpus_result,
)
NAGA_PRODUCTS = [
{"code": "8906011830068", "product_name": "Sooji", "product_name_en": "Sooji",
"brands": "Naga", "quantity": "500 g", "countries_tags": ["en:india"]},
{"code": "8906011831713", "product_name": "Maida", "product_name_en": "Maida",
"brands": "Naga", "quantity": "500 g", "countries_tags": ["en:india"]},
]
class _Resp:
def __init__(self, status=200, payload=None, text="", content_type="application/json"):
self.status_code = status
self._payload = payload
self.text = text
self.headers = {"content-type": content_type}
def json(self):
if self._payload is None:
raise ValueError("not json")
return self._payload
def _v2(products, count=None, page=1, page_size=None):
"""A v2 search body. `page_count` is deliberately the size of THIS page,
which is what the real API sends and what a naive paginator misreads;
`page_size` is the size the server APPLIED, not the one requested."""
return {"count": len(products) if count is None else count, "page": page,
"page_count": len(products),
"page_size": off_bulk.PAGE_SIZE if page_size is None else page_size,
"products": products, "skip": 0}
@pytest.fixture(autouse=True)
def _no_sleep(monkeypatch):
monkeypatch.setattr(off_bulk.time, "sleep", lambda s: None)
@pytest.fixture
def calls(monkeypatch):
"""Replace requests.get with a scripted responder. Returns the list of
(url, params) actually sent, so tests can assert on the query."""
sent = []
def install(responder):
def fake_get(url, params=None, headers=None, timeout=None):
sent.append((url, dict(params or {})))
r = responder(len(sent), dict(params or {}))
if isinstance(r, Exception):
raise r
return r
monkeypatch.setattr(off_bulk.requests, "get", fake_get)
return sent
return install
# ---------------------------------------------------------------------------
# The query
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("brand, slug", [
("Naga", "naga"),
("Hindustan Unilever", "hindustan-unilever"),
("P&G", "p-g"),
(" Double Horse ", "double-horse"),
("Coca-Cola", "coca-cola"),
])
def test_brand_tag_slug_matches_off_tag_form(brand, slug):
assert brand_tag_slug(brand) == slug
def test_query_is_an_exact_brand_tag_filter_on_the_v2_api(calls, tmp_path):
sent = calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
result = fetch_brand_corpus_result("Naga", country="india", cache_dir=tmp_path)
assert result.error is None and not result.from_cache
assert [h["code"] for h in result.hits] == ["8906011830068", "8906011831713"]
url, params = sent[0]
assert url == "https://world.openfoodfacts.org/api/v2/search"
assert params["brands_tags"] == "naga"
assert params["countries_tags"] == "en:india"
assert "q" not in params, "free-text brands: matching is what returned Mr Naga"
def test_no_country_filter_when_country_is_blank(calls, tmp_path):
sent = calls(lambda n, p: _Resp(payload=_v2([])))
fetch_brand_corpus_result("Naga", country="", cache_dir=tmp_path)
assert "countries_tags" not in sent[0][1]
# ---------------------------------------------------------------------------
# Pagination
# ---------------------------------------------------------------------------
def test_pagination_uses_count_not_page_count(calls, tmp_path):
"""150 products at PAGE_SIZE 100 is two pages. The first page's
`page_count` is 100 - reading that as "pages" would fetch 100 pages."""
page1 = [{"code": f"890{i:010d}", "product_name": f"P{i}"} for i in range(100)]
page2 = [{"code": f"891{i:010d}", "product_name": f"Q{i}"} for i in range(50)]
def responder(n, params):
return _Resp(payload=_v2(page1 if params["page"] == 1 else page2,
count=150, page=params["page"]))
sent = calls(responder)
result = fetch_brand_corpus_result("Amul", cache_dir=tmp_path)
assert [p["page"] for _, p in sent] == [1, 2]
assert len(result.hits) == 150
def test_pagination_follows_the_page_size_the_server_applied(calls, tmp_path, monkeypatch):
"""The real Amul case: 216 products, 250 requested, 100 served per page.
Dividing by what we ASKED for stops after one page and loses 116 rows."""
monkeypatch.setattr(off_bulk, "PAGE_SIZE", 250)
pages = {
1: [{"code": f"890{i:010d}", "product_name": f"P{i}"} for i in range(100)],
2: [{"code": f"891{i:010d}", "product_name": f"Q{i}"} for i in range(100)],
3: [{"code": f"892{i:010d}", "product_name": f"R{i}"} for i in range(16)],
}
sent = calls(lambda n, p: _Resp(payload=_v2(pages[p["page"]], count=216,
page=p["page"], page_size=100)))
result = fetch_brand_corpus_result("Amul", cache_dir=tmp_path)
assert [p["page"] for _, p in sent] == [1, 2, 3]
assert len(result.hits) == 216
def test_pagination_is_capped_at_max_pages(calls, tmp_path):
body = [{"code": "8900000000000", "product_name": "X"}] * off_bulk.PAGE_SIZE
sent = calls(lambda n, p: _Resp(payload=_v2(body, count=10 ** 6, page=p["page"])))
fetch_brand_corpus_result("Nestle", cache_dir=tmp_path)
assert len(sent) == off_bulk.MAX_PAGES
def test_nameless_products_are_dropped_from_the_corpus(calls, tmp_path):
calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS + [
{"code": "8906011839999", "product_name": ""},
{"code": "8906011839998"},
])))
assert len(fetch_brand_corpus_result("Naga", cache_dir=tmp_path).hits) == 2
# ---------------------------------------------------------------------------
# Failure is not emptiness, and is never cached
# ---------------------------------------------------------------------------
def test_503_is_retried_then_reported_and_not_cached(calls, tmp_path):
sent = calls(lambda n, p: _Resp(status=503, text="unavailable", content_type="text/html"))
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
assert len(sent) == 3, "with_retry gives a 5xx three attempts"
assert result.hits == []
assert result.error and "503" in result.error
assert list(tmp_path.glob("*.json")) == [], "a failed fetch must not become hits: []"
def test_503_then_success_recovers(calls, tmp_path):
def responder(n, params):
return _Resp(status=503) if n == 1 else _Resp(payload=_v2(NAGA_PRODUCTS))
calls(responder)
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
assert result.error is None and len(result.hits) == 2
def test_html_200_error_page_is_a_failure_not_an_empty_brand(calls, tmp_path):
calls(lambda n, p: _Resp(status=200, text="<html>temporarily unavailable</html>",
content_type="text/html"))
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
assert result.error and result.hits == []
assert list(tmp_path.glob("*.json")) == []
def test_transport_error_after_retries_is_reported(calls, tmp_path):
calls(lambda n, p: requests.exceptions.ConnectionError("dns"))
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
assert result.error and result.hits == []
assert list(tmp_path.glob("*.json")) == []
def test_partial_fetch_keeps_what_arrived_but_flags_it(calls, tmp_path):
page1 = [{"code": f"890{i:010d}", "product_name": f"P{i}"} for i in range(100)]
def responder(n, params):
if params["page"] == 1:
return _Resp(payload=_v2(page1, count=150))
return _Resp(status=502)
calls(responder)
result = fetch_brand_corpus_result("Amul", cache_dir=tmp_path)
assert len(result.hits) == 100 and result.error
assert list(tmp_path.glob("*.json")) == []
def test_list_form_still_returns_empty_on_failure(calls, tmp_path):
"""Backfill scripts and ingestion call `fetch_brand_corpus` and rely on
[] rather than an exception so one dead brand does not abort a run."""
calls(lambda n, p: _Resp(status=503))
assert fetch_brand_corpus("Naga", cache_dir=tmp_path) == []
def test_off_unavailable_is_a_connection_error_for_the_retry_decorator():
assert issubclass(OffUnavailable, requests.exceptions.ConnectionError)
# ---------------------------------------------------------------------------
# The cache
# ---------------------------------------------------------------------------
def test_successful_fetch_is_cached_with_schema_and_served_from_cache(calls, tmp_path):
sent = calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
first = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
second = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
assert len(sent) == 1
assert not first.from_cache and second.from_cache
assert second.hits == first.hits
on_disk = json.loads((tmp_path / "naga.json").read_text(encoding="utf-8"))
assert on_disk["schema"] == CACHE_SCHEMA
assert on_disk["brand_tag"] == "naga"
assert on_disk["endpoint"] == off_bulk.SEARCH_URL
def test_genuinely_empty_brand_is_cached_and_is_not_an_error(calls, tmp_path):
sent = calls(lambda n, p: _Resp(payload=_v2([])))
result = fetch_brand_corpus_result("Udhaiyam", cache_dir=tmp_path)
again = fetch_brand_corpus_result("Udhaiyam", cache_dir=tmp_path)
assert result == CorpusFetch(hits=[], error=None, from_cache=False)
assert again.from_cache and again.hits == []
assert len(sent) == 1
def test_empty_pre_schema_cache_is_refetched(calls, tmp_path):
"""The exact prod artefact: an old-endpoint file saying Naga has nothing."""
(tmp_path / "naga.json").write_text(json.dumps({
"brand": "Naga", "country": "india", "fetched_at": 0,
"fetched_at_human": "2026-09-10 12:00:00", "hits": [],
}), encoding="utf-8")
sent = calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
assert len(sent) == 1 and not result.from_cache
assert len(result.hits) == 2
assert json.loads((tmp_path / "naga.json").read_text(encoding="utf-8"))["schema"] == CACHE_SCHEMA
def test_non_empty_pre_schema_cache_is_still_trusted(calls, tmp_path):
(tmp_path / "amul.json").write_text(json.dumps({
"brand": "Amul", "country": "india", "fetched_at": 0,
"hits": [{"code": "8901262010115", "product_name": "Amul Butter"}],
}), encoding="utf-8")
sent = calls(lambda n, p: _Resp(status=503))
result = fetch_brand_corpus_result("Amul", cache_dir=tmp_path)
assert sent == [] and result.from_cache
assert result.hits[0]["product_name"] == "Amul Butter"
def test_refresh_bypasses_the_cache(calls, tmp_path):
(tmp_path / "naga.json").write_text(json.dumps({
"schema": CACHE_SCHEMA, "brand": "Naga", "hits": [],
}), encoding="utf-8")
sent = calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
assert len(fetch_brand_corpus_result("Naga", refresh=True, cache_dir=tmp_path).hits) == 2
assert len(sent) == 1
def test_unreadable_cache_is_refetched(calls, tmp_path):
(tmp_path / "naga.json").write_text("{not json", encoding="utf-8")
calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
assert len(fetch_brand_corpus_result("Naga", cache_dir=tmp_path).hits) == 2

View File

@@ -159,7 +159,11 @@ def test_the_sheets_values_are_kept_and_nothing_else_is_invented(store):
# What the sheet said.
assert row["product_name"] == "Apple 500g"
assert row["size_variants"] == ["500g"]
assert row["final_selling_price"] == 155
# "Selling Price" is the shop's price and lands in selling_price. The
# final (tax-inclusive) column is filled from it by upsert_brand_products,
# which this fixture stubs, so here it stays what the sheet said: nothing.
assert row["selling_price"] == 155
assert row["final_selling_price"] is None
assert row["price_range"] == "₹143-167" # +/-8% of the sheet's own price
# What it did not say, and what we therefore do not claim.

View File

@@ -3,7 +3,7 @@
WHY THIS EXISTS
---------------
Open Food Facts answers "does this product exist" for food and nothing else.
`off_bulk` queries only `search.openfoodfacts.org`, so a toothpaste or a
`off_bulk` queries only Open Food Facts' food database, so a toothpaste or a
detergent has no product-discovery source at all and its whole catalogue is
language-model output. A live retail lookup is the only signal in the codebase
that means "available right now" rather than "was in a database dump", and it

View File

@@ -0,0 +1,312 @@
"""What an uploaded sheet's own facts become in the brand table.
Three rules, each of which was violated before this file existed:
1. THE RETAIL PRICE IS THE SELLING PRICE. Every price header used to land in
`final_selling_price` and `selling_price` was NULL for every uploaded row;
worse, the HSN/GST stage then overwrote whatever price the sheet gave with
the ceiling of the derived band (155 became 167).
2. A PACK SIZE IS RECORDED ONLY WHEN THE SHEET OR THE NAME STATES IT.
"India Gate Basmati Rice 1kg" is one product in one size. "India Gate
Basmati Rice" used to become three products - 1kg, 5kg, 25kg - none of
which the sheet listed.
3. A BLANK DESCRIPTION IS WRITTEN, NOT PADDED. The LLM writes it when Ollama
is up; when it is not, the row gets a short factual line built only from
facts the row holds - never "<name> from <brand>." and never the marketing
essay in catalog_engine.generate_detailed_description.
Everything runs offline: storage and embeddings are monkeypatched on the
pipeline module, and USE_OLLAMA is false for the whole suite (conftest.py).
"""
from __future__ import annotations
import io
import pytest
from app.core import store_catalog_pipeline as pipeline
from app.services.enrichment.hsn_gst.models import enrich_pricing_fields
from app.services.generic_products import OWN_PRODUCTS_BRAND
openpyxl = pytest.importorskip("openpyxl")
def _sheet(headers, rows) -> bytes:
wb = openpyxl.Workbook()
ws = wb.active
ws.append(headers)
for row in rows:
ws.append(row)
buf = io.BytesIO()
wb.save(buf)
return buf.getvalue()
@pytest.fixture(autouse=True)
def _isolate_sku_counter(tmp_path, monkeypatch):
from app.services import sku_service
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
@pytest.fixture
def store(monkeypatch):
"""A fake brand table: image_id -> stored row, across every brand."""
table: dict = {}
def fake_upsert(brand, rows, cleanup=False):
assert cleanup is False
for row in rows:
table[row["image_id"]] = dict(row)
return len(rows)
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: list(table.values()))
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
return table
def _run(headers, rows, **kw):
kw.setdefault("use_llm", False)
kw.setdefault("fetch_images", False)
return pipeline.run_pipeline("store.xlsx", _sheet(headers, rows), **kw)
# Real-looking names: the validation gate rejects placeholder titles such as
# "Product 3", which would empty the store and hide what a test is about.
_AMUL_NAMES = ["Amul Butter 100g", "Amul Cheese 200g", "Amul Ghee 500ml", "Amul Taaza 1L",
"Amul Kool 200ml", "Amul Lassi 200ml", "Amul Masti Dahi 400g"]
def _only(store):
assert len(store) == 1, list(store)
return next(iter(store.values()))
# ---------------------------------------------------------------------------
# 1. Price
# ---------------------------------------------------------------------------
def test_retail_price_lands_in_selling_price_and_drives_the_band(store):
_run(["ProductName", "ProductSKU", "ProductBrand", "RetailPrice"],
[["India Gate Basmati Rice 1kg", "IG-BR-1", "India Gate", 120]])
row = _only(store)
assert row["selling_price"] == 120
assert row["price_range"] == "₹110-130" # +/-8% of the sheet's price
assert row["product_sku"] == "IG-BR-1" and row["sku_source"] == "sheet"
def test_a_final_price_column_is_kept_separate_and_wins_the_band(store):
_run(["Item Name", "Brand", "Selling Price", "Final Price"],
[["Amul Butter 500g", "Amul", 250, 262]])
row = _only(store)
assert row["selling_price"] == 250
assert row["final_selling_price"] == 262
assert row["price_range"] == "₹241-283" # band from the final price
def test_reingesting_a_sheet_with_a_selling_price_is_a_no_op(store):
headers, rows = ["Item Name", "Brand", "Retail Price"], [["Amul Butter 500g", "Amul", 250]]
first = _run(headers, rows)
assert first.inserted == 1
second = _run(headers, rows)
assert (second.inserted, second.backfilled, second.skipped_existing) == (0, 0, 1)
def test_a_held_final_price_is_not_clobbered_by_a_different_retail_price(store):
_run(["Item Name", "Brand", "Retail Price"], [["Amul Butter 500g", "Amul", 250]])
key = next(iter(store))
store[key]["final_selling_price"] = 262 # what the table already holds
_run(["Item Name", "Brand", "Retail Price"], [["Amul Butter 500g", "Amul", 199]])
assert store[key]["final_selling_price"] == 262
assert store[key]["selling_price"] == 250, "fill-only-blanks: a held price is kept"
class TestHsnStageKeepsHeldPrices:
"""The overwrite that turned a sheet's 155 into 167."""
def test_a_held_selling_price_is_returned_as_none_so_apply_keeps_it(self):
fields = enrich_pricing_fields({
"selling_price": 155, "final_selling_price": None,
"price_range": "₹143-167", "gst_percent": 18,
})
assert fields["selling_price"] is None
assert fields["tax_amount"] == 27.9 # 18% of the HELD 155, not of 167
assert fields["final_selling_price"] == 182.9
def test_a_held_final_price_is_also_kept(self):
fields = enrich_pricing_fields({
"selling_price": 155, "final_selling_price": 160,
"price_range": "₹143-167", "gst_percent": 18,
})
assert fields["selling_price"] is None
assert fields["final_selling_price"] is None
def test_without_a_sheet_price_the_band_ceiling_is_still_the_base(self):
fields = enrich_pricing_fields({"price_range": "₹143-167", "gst_percent": 5})
assert fields["selling_price"] == 167
assert fields["final_selling_price"] == 175.35
def test_a_held_final_price_alone_seeds_the_selling_price(self):
fields = enrich_pricing_fields({"final_selling_price": 200, "gst_percent": 5,
"price_range": "₹300-400"})
assert fields["selling_price"] == 200
assert fields["final_selling_price"] is None
# ---------------------------------------------------------------------------
# 2. Size
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("name, size", [
("India Gate Basmati Rice 1kg", "1kg"),
("Fortune Sunflower Oil 1L", "1L"),
("Amul Taaza Milk 1 Litre", "1 Litre"),
("Aashirvaad Atta 5 Kg", "5 Kg"),
("Tata Salt 1.5kg", "1.5kg"),
("Parle-G Biscuits 800 gm", "800 gm"),
("Saffola Gold 500ml", "500ml"),
])
def test_the_size_in_the_name_is_the_size_and_nothing_is_appended(store, name, size):
_run(["Product Name", "Brand"], [[name, name.split()[0]]])
row = _only(store)
assert row["size_variants"] == [size]
assert row["product_name"] == name, "the name already carries the size"
def test_a_branded_product_with_no_size_is_one_clean_row(store):
result = _run(["Product Name", "Brand"], [["India Gate Basmati Rice", "India Gate"]])
row = _only(store)
assert row["size_variants"] == []
assert row["product_name"] == "India Gate Basmati Rice"
assert "Standard" not in row["product_name"]
assert next(iter(store)) == "india_gate_india_gate_basmati_rice"
assert any("no pack size" in w for w in result.warnings)
def test_a_commodity_with_no_size_still_uses_the_standard_sentinel(store):
"""Own Products dedupes against a seeded base list built on "Standard"."""
_run(["Product Name"], [["Apple"]])
row = _only(store)
assert row["size_variants"] == ["Standard"]
assert next(iter(store)) == pipeline.build_image_id(OWN_PRODUCTS_BRAND, "Apple", "Standard")
def test_the_llm_may_not_supply_a_pack_size(store, monkeypatch):
from app.services import ollama_service
monkeypatch.setattr(ollama_service, "fetch_product_details",
lambda brand, title, category=None, size=None: {
"description": "Refined wheat flour milled for baking.",
"size_variants": ["500g", "1kg", "5kg"],
})
_run(["Product Name", "Brand"], [["Naga Maida", "Naga"]], use_llm=True)
row = _only(store)
assert row["size_variants"] == []
assert row["description"] == "Refined wheat flour milled for baking."
def test_a_sheet_size_column_still_wins_over_the_name(store):
_run(["Product Name", "Brand", "Pack Size"], [["Amul Butter 100g", "Amul", "500g"]])
assert _only(store)["size_variants"] == ["500g"]
# ---------------------------------------------------------------------------
# 3. Description
# ---------------------------------------------------------------------------
def test_the_llm_description_is_used_when_ollama_answers(store, monkeypatch):
from app.services import ollama_service
seen = {}
def fake(brand, title, category=None, size=None):
seen.update(brand=brand, title=title, category=category, size=size)
return {"description": "Long-grain aged basmati rice for biryani and pulao."}
monkeypatch.setattr(ollama_service, "fetch_product_details", fake)
_run(["Product Name", "Brand"], [["India Gate Basmati Rice 1kg", "India Gate"]], use_llm=True)
assert _only(store)["description"] == "Long-grain aged basmati rice for biryani and pulao."
assert seen["size"] == "1kg", "the pack size in the name is handed to the prompt"
def test_an_overlong_llm_description_is_clamped(store, monkeypatch):
from app.services import ollama_service
monkeypatch.setattr(ollama_service, "fetch_product_details",
lambda *a, **k: {"description": "x" * 2000})
_run(["Product Name", "Brand"], [["Amul Butter 500g", "Amul"]], use_llm=True)
assert len(_only(store)["description"]) == pipeline._MAX_LLM_DESCRIPTION_CHARS
def test_the_fallback_description_is_factual_and_short(store):
_run(["Product Name", "Brand", "Category"],
[["Fortune Sunflower Oil 1L", "Fortune", "Edible Oils"]])
desc = _only(store)["description"]
assert desc == "Fortune Sunflower Oil 1L, Edible Oils, by Fortune."
assert "from Fortune." not in desc and "Introducing" not in desc
def test_the_fallback_description_names_a_sheet_size_the_name_lacks(store):
_run(["Product Name", "Brand", "Pack Size"], [["Amul Butter", "Amul", "500g"]])
assert _only(store)["description"] == "Amul Butter 500g, Dairy, by Amul."
def test_a_sheet_description_is_never_overwritten(store, monkeypatch):
from app.services import ollama_service
monkeypatch.setattr(ollama_service, "fetch_product_details",
lambda *a, **k: pytest.fail("the LLM must not be asked"))
_run(["Product Name", "Brand", "Description"],
[["Amul Butter 500g", "Amul", "Salted table butter."]], use_llm=True)
assert _only(store)["description"] == "Salted table butter."
def test_the_llm_is_switched_off_after_three_consecutive_misses(store, monkeypatch):
from app.services import ollama_service
calls = []
monkeypatch.setattr(ollama_service, "fetch_product_details",
lambda brand, title, **k: calls.append(title) or None)
rows = [[name, "Amul"] for name in _AMUL_NAMES[:6]]
result = _run(["Product Name", "Brand"], rows, use_llm=True)
assert len(calls) == 3
assert any("consecutive" in w for w in result.warnings)
assert len(store) == 6, "the rows themselves are unaffected"
def test_one_answer_resets_the_breaker(store, monkeypatch):
from app.services import ollama_service
answers = iter([None, None, {"description": "ok"}, None, None, None, None])
calls = []
def fake(brand, title, **k):
calls.append(title)
return next(answers)
monkeypatch.setattr(ollama_service, "fetch_product_details", fake)
_run(["Product Name", "Brand"], [[name, "Amul"] for name in _AMUL_NAMES], use_llm=True)
# miss, miss, hit (reset), miss, miss, miss (trip) -> the 7th row is not asked.
assert len(calls) == 6
def test_stage_two_leaves_a_blank_description_blank_on_a_miss():
"""The fallback lives in _to_storage_row; stage 2 must not pre-empt it."""
row = {"brand": "Amul", "product_name": "Amul Butter 500g", "description": ""}
breaker = pipeline.LlmBreaker()
pipeline.stage_2_row_intake(row, use_llm=True, breaker=breaker)
assert row["description"] == ""
assert breaker.consecutive_misses == 1

View File

@@ -187,6 +187,45 @@ def test_template_headers_all_map_to_their_field():
assert mapping.ignored == []
@pytest.mark.parametrize("header, canonical", [
# The camel-cased headers a store's export tool produces, which
# _normalize_header collapses to one word - locked as exact aliases so
# they do not depend on the keyword rules' ordering.
("ProductName", "product_name"),
("ProductBrand", "brand"),
("Product Brand", "brand"),
("ProductSKU", "product_sku"),
("Product SKU", "product_sku"),
# What the shop charges -> selling_price.
("RetailPrice", "selling_price"),
("Retail Price", "selling_price"),
("Selling Price", "selling_price"),
("Sale Price", "selling_price"),
("SP", "selling_price"),
("Price", "selling_price"),
("Unit Price (Rs)", "selling_price"),
# The tax-inclusive ceiling -> final_selling_price. "Final Selling Price"
# contains "selling price" and must still win: rule order regression.
("Final Selling Price", "final_selling_price"),
("Final Price (₹)", "final_selling_price"),
("MRP", "final_selling_price"),
("Maximum MRP", "final_selling_price"),
])
def test_price_and_identity_headers_map_where_the_sheet_means_them(header, canonical):
"""Before the selling_price canonical existed every price header landed in
final_selling_price and the table's selling_price column stayed NULL."""
mapping = user_products.map_spreadsheet_columns([header])
assert mapping.columns == {canonical: header}
def test_a_sheet_may_carry_both_a_retail_and_a_final_price():
mapping = user_products.map_spreadsheet_columns(
["Item Name", "Brand", "Retail Price", "MRP"])
assert mapping.columns["selling_price"] == "Retail Price"
assert mapping.columns["final_selling_price"] == "MRP"
assert mapping.ignored == []
def test_a_product_sku_column_is_not_mistaken_for_the_product_name():
"""'Product SKU' contains 'product'. Matched loosely, it used to become a
second product_name column, and duplicate columns are what turned a row