Health score updates in backend

This commit is contained in:
sriram
2026-09-04 12:06:40 +05:30
parent 1300d4c678
commit 0c3e23fac4
33 changed files with 59703 additions and 94 deletions

View File

@@ -67,6 +67,19 @@ os.environ.setdefault("USE_GOOGLE_CSE", "false")
# which sets the value explicitly and clears the parsed cache.
os.environ["ACTIVE_BRANDS"] = ""
# Unconditional, and OFF - the opposite of the production default.
#
# Finishing an ingestion batch queues nutrition enrichment on a background
# thread, and that thread calls out to a database and to Open Food Facts. Any
# test that runs `run_batch` would start it, so a `pytest` run reached for the
# host in the developer's backend/.env - which on this project is PRODUCTION.
# It got no further than a failed password, which is not a margin worth
# relying on.
#
# The behaviour itself is covered by tests/test_auto_enrich_on_upload.py, which
# turns it on explicitly, the same arrangement ACTIVE_BRANDS has above.
os.environ["AUTO_ENRICH_ON_UPLOAD"] = "false"
# Auth is set unconditionally (not setdefault): the suite asserts on the real
# guards, so it must never inherit a developer's AUTH_ENABLED=false.
os.environ["AUTH_ENABLED"] = "true"

View File

@@ -0,0 +1,295 @@
"""A newly uploaded product gets a health score without anyone asking for one.
Before this, a catalogue upload ran eleven pipeline stages, stored the products
and stopped. Nothing on that path touched nutrition, and `enrich_all_products`
had exactly one HTTP trigger in the whole application - the admin-only
`POST /api/admin/nutrition-intelligence/enrich`. So every uploaded product sat
unscored until somebody remembered to run it by hand.
The three things that can silently break this, each pinned below:
* `include_inactive=True` going missing. A brand that was just uploaded is
almost never in ACTIVE_BRANDS, and `_brands_to_enrich` honours that filter
by default - the job would report a clean run over an empty work list.
* The submission raising. The products ARE stored by that point; a scoring
step that could not start must not make a successful import look failed.
* The manifest gaining a required field. Manifests are JSON on the upload
volume and outlive a deploy, so an older one still has to load.
Fixture conventions follow test_batch_stage_timeline.py.
"""
from __future__ import annotations
import io
import pytest
from app.core import batch_ingest
from app.core import store_catalog_pipeline as pipeline
from app.services import nutrition_autoenrich
openpyxl = pytest.importorskip("openpyxl")
# ---------------------------------------------------------------------------
# The submission helper
# ---------------------------------------------------------------------------
@pytest.fixture
def captured(monkeypatch):
"""Capture the thunk instead of starting a thread, then run it against a
fake `enrich_all_products` so the kwargs it was built with are visible."""
box = {"kwargs": None, "names": []}
def fake_background(func, *, name):
box["names"].append(name)
func()
def fake_enrich(**kwargs):
box["kwargs"] = kwargs
return type("R", (), {
"total_products": 3, "verified": 2, "partial": 0, "unavailable": 1,
"skipped_non_consumable": 4, "duration_seconds": 1.0, "errors": [],
})()
# conftest pins this off for the suite; these tests are about what happens
# when it is on.
monkeypatch.setattr(nutrition_autoenrich.settings, "AUTO_ENRICH_ON_UPLOAD", True)
monkeypatch.setattr(nutrition_autoenrich, "run_in_background", fake_background)
monkeypatch.setattr(nutrition_autoenrich.nutrition_enrichment_service,
"enrich_all_products", fake_enrich)
return box
def test_the_uploaded_brands_are_what_gets_enriched(captured):
nutrition_autoenrich.submit_enrichment_for_brands(["Amul", "Britannia"], source="test")
assert captured["kwargs"]["brands"] == ["Amul", "Britannia"]
def test_active_brands_is_widened_or_nothing_would_be_scored(captured):
"""The single most load-bearing argument here. Without it a freshly
uploaded brand is filtered out before any product is looked at, and the job
still reports success."""
nutrition_autoenrich.submit_enrichment_for_brands(["NewBrand"], source="test")
assert captured["kwargs"]["include_inactive"] is True
def test_only_the_products_that_still_need_a_score_are_fetched(captured):
"""`skip_if_verified` is what makes re-uploading a scored brand cheap, and
is why no separate "which rows are new" bookkeeping is needed."""
nutrition_autoenrich.submit_enrichment_for_brands(["Amul"], source="test")
assert captured["kwargs"]["skip_if_verified"] is True
def test_the_llm_narrative_is_left_for_the_admin_job(captured):
"""It is the slow, token-costing step and adds no health score."""
nutrition_autoenrich.submit_enrichment_for_brands(["Amul"], source="test")
assert captured["kwargs"]["generate_narrative"] is False
def test_a_brand_uploaded_in_two_files_is_enriched_once(captured):
nutrition_autoenrich.submit_enrichment_for_brands(
["Amul", "Britannia", "Amul"], source="test")
assert captured["kwargs"]["brands"] == ["Amul", "Britannia"]
def test_blank_brands_do_not_become_work(captured):
nutrition_autoenrich.submit_enrichment_for_brands(["", " ", "Amul"], source="test")
assert captured["kwargs"]["brands"] == ["Amul"]
def test_a_batch_that_produced_no_brands_queues_nothing(captured):
assert nutrition_autoenrich.submit_enrichment_for_brands([], source="test") is None
assert captured["kwargs"] is None
def test_the_job_id_is_returned_so_the_caller_can_be_polled(captured):
job_id = nutrition_autoenrich.submit_enrichment_for_brands(["Amul"], source="test")
assert job_id
assert nutrition_autoenrich.nutrition_job_store.get(job_id) is not None
def test_the_job_records_what_the_run_actually_did(captured):
job_id = nutrition_autoenrich.submit_enrichment_for_brands(["Amul"], source="test")
job = nutrition_autoenrich.nutrition_job_store.get(job_id)
assert job.status == "done"
assert job.result["verified"] == 2
assert job.result["skipped_non_consumable"] == 4
def test_the_detail_names_the_brands_so_an_operator_can_read_it(captured):
job_id = nutrition_autoenrich.submit_enrichment_for_brands(
["Amul", "Britannia"], source="batch abc123")
job = nutrition_autoenrich.nutrition_job_store.get(job_id)
assert "batch abc123" in job.detail
assert "Amul, Britannia" in job.detail
def test_an_enrichment_that_raises_fails_the_job_not_the_caller(captured, monkeypatch):
def boom(**kwargs):
raise RuntimeError("Open Food Facts is down")
monkeypatch.setattr(nutrition_autoenrich.nutrition_enrichment_service,
"enrich_all_products", boom)
job_id = nutrition_autoenrich.submit_enrichment_for_brands(["Amul"], source="test")
job = nutrition_autoenrich.nutrition_job_store.get(job_id)
assert job.status == "failed"
assert "Open Food Facts is down" in job.detail
def test_the_submission_itself_never_raises_at_its_caller(monkeypatch):
"""Every caller is an upload handler finishing its work. An exception here
would turn a successful ingestion into a 500."""
monkeypatch.setattr(nutrition_autoenrich.settings, "AUTO_ENRICH_ON_UPLOAD", True)
monkeypatch.setattr(nutrition_autoenrich.nutrition_job_store, "create",
lambda kind: (_ for _ in ()).throw(RuntimeError("no job store")))
assert nutrition_autoenrich.submit_enrichment_for_brands(["Amul"], source="test") is None
# ---------------------------------------------------------------------------
# The off switch
# ---------------------------------------------------------------------------
def test_the_setting_turns_it_off_without_a_deploy(captured, monkeypatch):
monkeypatch.setattr(nutrition_autoenrich.settings, "AUTO_ENRICH_ON_UPLOAD", False)
assert nutrition_autoenrich.submit_enrichment_for_brands(["Amul"], source="test") is None
assert captured["kwargs"] is None
assert captured["names"] == [], "no thread should have been started"
# ---------------------------------------------------------------------------
# The batch hook
# ---------------------------------------------------------------------------
THREE_BRANDS = [
["Amul Butter 100g", "Dairy", "Amul"],
["Britannia Marie Gold 250g", "Biscuits", "Britannia"],
["Cadbury Dairy Milk 150g", "Chocolate", "Cadbury"],
]
def _sheet(rows=THREE_BRANDS) -> bytes:
wb = openpyxl.Workbook()
ws = wb.active
ws.append(["Product Name", "Category", "Brand"])
for row in rows:
ws.append(row)
buf = io.BytesIO()
wb.save(buf)
return buf.getvalue()
@pytest.fixture(autouse=True)
def _isolate_sku_counter(tmp_path, monkeypatch):
from app.services import sku_service
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
@pytest.fixture(autouse=True)
def batch_root(tmp_path, monkeypatch):
monkeypatch.setattr(batch_ingest, "BATCH_UPLOAD_DIR", tmp_path / "batch_uploads")
@pytest.fixture(autouse=True)
def store(monkeypatch):
"""A fake brand table, so the pipeline never reaches Postgres."""
table: dict = {}
def fake_upsert(brand, rows, **kwargs):
for row in rows:
table[(brand, row.get("image_id"))] = row
return {"inserted": len(rows), "backfilled": 0, "skipped_existing": 0}
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: list(table.values()))
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
return table
@pytest.fixture
def submissions(monkeypatch):
"""Record what the batch hook submits, without starting anything."""
seen = []
def fake_submit(brands, *, source):
seen.append((list(brands), source))
return "job-1234"
monkeypatch.setattr(nutrition_autoenrich, "submit_enrichment_for_brands", fake_submit)
return seen
def test_a_finished_batch_queues_scoring_for_every_brand_it_wrote(submissions):
manifest = batch_ingest.stage_batch([("catalog.xlsx", _sheet())])
result = batch_ingest.run_batch(manifest.batch_id)
assert len(submissions) == 1, "one job for the batch, not one per file"
brands, source = submissions[0]
assert set(brands) == {"Amul", "Britannia", "Cadbury"}
assert manifest.batch_id in source
assert result.nutrition_job_id == "job-1234"
def test_the_job_id_survives_a_reread_from_disk(submissions):
"""The client polls the batch through the manifest on the volume, which is
a different object from the one `run_batch` returned."""
manifest = batch_ingest.stage_batch([("catalog.xlsx", _sheet())])
batch_ingest.run_batch(manifest.batch_id)
assert batch_ingest.read_manifest(manifest.batch_id).nutrition_job_id == "job-1234"
def test_a_submission_that_raises_leaves_the_batch_successful(monkeypatch):
"""The products are stored by this point. Reporting the import as failed
because an optional follow-up could not start would be a lie."""
def boom(brands, *, source):
raise RuntimeError("job store unavailable")
monkeypatch.setattr(nutrition_autoenrich, "submit_enrichment_for_brands", boom)
manifest = batch_ingest.stage_batch([("catalog.xlsx", _sheet())])
result = batch_ingest.run_batch(manifest.batch_id)
assert result.files[0].status == batch_ingest.DONE
assert result.status not in (batch_ingest.FAILED,)
assert result.nutrition_job_id is None
def test_the_job_id_reaches_the_api_response(submissions):
"""`to_out` renders only fields declared on BatchOut, so a manifest field
that is not also declared there is dropped on the way out - silently."""
from app.api import batch_common
manifest = batch_ingest.stage_batch([("catalog.xlsx", _sheet())])
result = batch_ingest.run_batch(manifest.batch_id)
assert batch_common.to_out(result).nutrition_job_id == "job-1234"
# ---------------------------------------------------------------------------
# Backwards compatibility of the manifest on disk
# ---------------------------------------------------------------------------
def test_a_manifest_written_before_this_field_existed_still_loads():
"""manifest.json files live on the upload volume and outlive a deploy."""
older = {"batch_id": "b1", "status": batch_ingest.DONE, "files": []}
restored = batch_ingest.BatchManifest.from_dict(older)
assert restored.nutrition_job_id is None
def test_the_field_round_trips():
manifest = batch_ingest.BatchManifest(batch_id="b1", nutrition_job_id="job-9")
assert batch_ingest.BatchManifest.from_dict(
manifest.to_dict()).nutrition_job_id == "job-9"

271
tests/test_consumability.py Normal file
View File

@@ -0,0 +1,271 @@
"""
Walls off the defect where non-food products carried a health score.
Measured against the live catalogue on 2026-09-03, these rows had one:
Colgate-Palmolive Palmolive Naturals General soap
Cavinkare Nyle, Cavinkare Nature's Hair Care shampoo
P&G Pantene Hair Care shampoo
Godrej Hit Spray Personal Care - Mosquito 291 kcal
Every category string in LIVE_CATEGORIES below was read out of the production
brand tables, with its row count, so this file tests the taxonomy that exists
rather than the one `CATEGORY_REGISTRY` describes - the two have drifted, which
is the whole reason `consumability` is not just a field on the registry.
Pure unit tests: no database, no network, no LLM.
"""
from __future__ import annotations
import sys
from pathlib import Path
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.services import nutrition_data_service as nds # noqa: E402
from app.services.consumability import ( # noqa: E402
CATEGORY_VERDICTS,
Edibility,
classify_edibility,
is_consumable,
is_junk_category,
is_non_consumable,
)
from app.services.enrichment.hsn_gst.models import HSN_GST_TABLE # noqa: E402
# (category string, live row count, expected verdict). Read from the production
# database; the counts are kept so a future reader can tell a 189-row decision
# from a 1-row one.
LIVE_CATEGORIES = [
("Dairy", 189, Edibility.CONSUMABLE),
("Beverages", 174, Edibility.CONSUMABLE),
("Spices & Masalas", 136, Edibility.CONSUMABLE),
("Chocolates", 119, Edibility.CONSUMABLE),
("Fruits & Vegetables", 95, Edibility.CONSUMABLE),
("Food - Mixes", 83, Edibility.CONSUMABLE),
("Tea & Coffee", 76, Edibility.CONSUMABLE),
("Hair Care", 51, Edibility.NON_CONSUMABLE),
("Oral Care", 45, Edibility.NON_CONSUMABLE),
("Snacks", 42, Edibility.CONSUMABLE),
("Cooking Oils", 42, Edibility.CONSUMABLE),
("Detergents & Fabric Care", 38, Edibility.NON_CONSUMABLE),
("Pulses, Grains & Spices", 35, Edibility.CONSUMABLE),
("Health Drinks", 26, Edibility.CONSUMABLE),
("Pasta & Noodles", 24, Edibility.CONSUMABLE),
("Pickles & Chutneys", 24, Edibility.CONSUMABLE),
("Skin Care", 19, Edibility.NON_CONSUMABLE),
("Bath Soap", 17, Edibility.NON_CONSUMABLE),
("Fresh Herbs & Greens", 17, Edibility.CONSUMABLE),
("Atta & Staples", 16, Edibility.CONSUMABLE),
("Fish & Seafood", 15, Edibility.CONSUMABLE),
("Flowers", 14, Edibility.NON_CONSUMABLE),
("Candy & Confectionery", 13, Edibility.CONSUMABLE),
("Dairy - Desserts", 13, Edibility.CONSUMABLE),
("Biscuits & Cookies", 12, Edibility.CONSUMABLE),
("Dishwash", 12, Edibility.NON_CONSUMABLE),
("Health Foods", 12, Edibility.CONSUMABLE),
("Food - Spreads", 11, Edibility.CONSUMABLE),
("Skin & Bath Care", 10, Edibility.NON_CONSUMABLE),
("Noodles & Instant Food", 10, Edibility.CONSUMABLE),
("Cheese", 10, Edibility.CONSUMABLE),
("Salt & Staples", 9, Edibility.CONSUMABLE),
("Breakfast Cereal", 8, Edibility.CONSUMABLE),
("Feminine Hygiene", 6, Edibility.NON_CONSUMABLE),
("Ready to Eat", 6, Edibility.CONSUMABLE),
("Dry Fruits & Nuts", 6, Edibility.CONSUMABLE),
("Eggs", 5, Edibility.CONSUMABLE),
("Bakery & Breads", 5, Edibility.CONSUMABLE),
("Ice Cream", 5, Edibility.CONSUMABLE),
("Health Care - Cold & Cough", 5, Edibility.NON_CONSUMABLE),
("Personal Care - Mosquito Repellent", 5, Edibility.NON_CONSUMABLE),
("Food & Beverages", 4, Edibility.CONSUMABLE),
("Men's Grooming", 4, Edibility.NON_CONSUMABLE),
("Household - Lamp Oil", 4, Edibility.NON_CONSUMABLE),
("Fragrance & Deodorants", 3, Edibility.NON_CONSUMABLE),
("Household Cleaning", 2, Edibility.NON_CONSUMABLE),
("Biscuits", 2, Edibility.CONSUMABLE),
("Food - Soups & Sauces", 2, Edibility.CONSUMABLE),
("Crackers", 2, Edibility.CONSUMABLE),
("Health Care - Antiseptic", 2, Edibility.NON_CONSUMABLE),
("Namkeen", 1, Edibility.CONSUMABLE),
("Household - Air Freshener", 1, Edibility.NON_CONSUMABLE),
("Flour & Grains", 1, Edibility.CONSUMABLE),
("Staples", 1, Edibility.CONSUMABLE),
("Personal Care", 1, Edibility.NON_CONSUMABLE),
("Health Care - Digestive", 1, Edibility.NON_CONSUMABLE),
]
# The two category values that genuinely carry no signal. "General" holds 40
# rows of mixed food and non-food; "1".."5" are import damage on 17 rows.
# Both are meant to reach the title, so they are excluded here rather than
# given a verdict this file would have to invent.
UNDECIDABLE_CATEGORIES = ("General", "1", "2", "3", "4", "5")
# Categories holding both food and non-food, so they carry no verdict of their
# own and are excluded from the table above. Found by running the purge audit
# before deleting anything - see test_baby_food_is_food.
MIXED_CATEGORIES = ("Baby Care", "Health Care - Ayurvedic")
@pytest.mark.parametrize("category,count,expected", LIVE_CATEGORIES)
def test_every_live_category_has_the_expected_verdict(category, count, expected):
verdict = classify_edibility(category)
assert verdict.edibility is expected, (
f"{category!r} ({count} live rows) classified {verdict.edibility.value}: "
f"{verdict.reason}"
)
@pytest.mark.parametrize(
"category",
[c for c in sorted(HSN_GST_TABLE) if c not in MIXED_CATEGORIES],
)
def test_every_hsn_category_resolves(category):
"""Drift guard. Someone adding a category to the tax table and not here
would otherwise get a silent UNKNOWN, and UNKNOWN means no health score.
The mixed categories are excluded because their UNKNOWN is the intended
answer - they are decided by the title, not by the category.
"""
assert classify_edibility(category).edibility is not Edibility.UNKNOWN
def test_lamp_oil_is_not_a_cooking_oil():
"""These must not collapse into each other.
`resolve_hsn_gst`'s keyword fallbacks contain a greedy ("oil", "1517")
entry, which is chapter 15 - an edible fat. Reading HSN_GST_TABLE directly
instead of calling that resolver is what keeps lamp oil inedible.
"""
assert is_consumable("Cooking Oils", "Idhayam Sesame Oil 500ml") is True
assert is_non_consumable("Household - Lamp Oil", "Idhayam Lamp Oil 500ml") is True
def test_flowers_are_not_food():
"""HSN 0603 is chapter 06, which sits inside the naive 01-24 "food" range.
Flowers reach the Own Products table through the same produce lexicon as the
vegetables, so nothing upstream separates them - this is the separation."""
assert is_non_consumable("Flowers", "Jasmine Garland") is True
assert is_consumable("Flowers", "Jasmine Garland") is False
def test_medicines_are_not_scored():
"""Ingested, but they publish dosage rather than a nutrition panel."""
for category in ("Health Care - Cold & Cough", "Health Care - Digestive",
"Health Care - Antiseptic"):
assert is_non_consumable(category) is True, category
assert is_non_consumable("Health Care - Cold & Cough", "Dabur Honitus 1L") is True
def test_baby_food_is_food():
"""The error this classifier made on its first audit run, caught before a
single row was deleted.
"Baby Care" held 21 live rows and a blanket non-consumable verdict. They are
Nestle Cerelac, Nan Pro and Lactogen - infant formula and baby cereal, among
the most heavily nutrition-labelled products sold in India - sitting beside a
bottle of baby oil. One verdict cannot cover both, so the category defers to
the title.
"""
assert is_consumable("Baby Care", "Nestle Cerelac 125g") is True
assert is_consumable("Baby Care", "Nestle Nan Pro 400 g") is True
assert is_consumable("Baby Care", "Nestle Lactogen 400 g") is True
assert is_non_consumable("Baby Care", "Johnson Baby Soap 100g") is True
assert is_non_consumable("Baby Care", "Pampers Diaper Large 20s") is True
def test_chyawanprash_is_food_but_cough_syrup_is_not():
"""Same shape as Baby Care: "Health Care - Ayurvedic" mixes a spoonable food
supplement with medicine, so it too defers to the title."""
assert is_consumable("Health Care - Ayurvedic",
"Chyawanprash 100% Natural 70g") is True
@pytest.mark.parametrize("category", MIXED_CATEGORIES)
def test_a_mixed_category_carries_no_verdict_of_its_own(category):
"""With no title to go on there is nothing to decide, and UNKNOWN neither
enriches nor deletes."""
assert classify_edibility(category).edibility is Edibility.UNKNOWN
def test_health_drinks_are_food_despite_the_name():
assert is_consumable("Health Drinks", "Horlicks Classic Malt 500g") is True
assert is_consumable("Health Foods", "Manna Health Mix 500g") is True
def test_unknown_is_neither_consumable_nor_non_consumable():
"""The asymmetry the purge script depends on: an unrecognised product is
never enriched AND never deleted. If these two ever become inverses, a
classifier miss turns into data loss."""
verdict = classify_edibility("Zzz Unheard Of Category", "Mystery Item 1kg")
assert verdict.edibility is Edibility.UNKNOWN
assert is_consumable("Zzz Unheard Of Category", "Mystery Item 1kg") is False
assert is_non_consumable("Zzz Unheard Of Category", "Mystery Item 1kg") is False
def test_a_missing_category_falls_back_to_the_title():
assert is_consumable("", "Amul Butter 500g") is True
assert is_consumable(None, "Toor Dhal 1kg") is True
assert is_non_consumable("", "Lifebuoy Soap 100g") is True
assert classify_edibility(None, "").edibility is Edibility.UNKNOWN
@pytest.mark.parametrize("junk", UNDECIDABLE_CATEGORIES)
def test_an_uninformative_category_defers_to_the_title(junk):
assert is_consumable(junk, "Aachi Sambar Powder 100g") is True
assert is_non_consumable(junk, "Lifebuoy Soap 100g") is True
def test_numeric_categories_are_reported_as_junk():
"""17 live rows carry a bare digit as their category. They are recognised so
the purge audit can report them for catalogue repair, rather than being
quietly accommodated."""
assert is_junk_category("3") is True
assert is_junk_category("Dairy") is False
def test_a_toothpaste_brand_is_not_a_chocolate():
"""The regression that forced `exact_only` into `detect_category_from_text`.
Its fuzzy fallback scores "colgate" at >=0.8 against the misspelling keyword
"choclate", so "Colgate-Palmolive Palmolive Naturals" - whose stored
category is the uninformative "General" - resolved to Chocolates and would
have been handed a health score.
"""
verdict = classify_edibility("General", "Colgate-Palmolive Palmolive Naturals")
assert verdict.edibility is not Edibility.CONSUMABLE
def test_soapnut_is_not_a_soap():
"""The old gate matched NON_FOOD_KEYWORDS by substring, so "soap" hit
*soapnut* (reetha). Matching here is whole-word."""
assert is_non_consumable("Fruits & Vegetables", "Soapnut Reetha 500g") is False
@pytest.mark.parametrize("category,title", [
("Bath Soap", "Colgate-Palmolive Palmolive Naturals"),
("Hair Care", "Cavinkare Nyle"),
("Hair Care", "Cavinkare Nature's"),
("Hair Care", "P G Pg Pantene"),
("Personal Care - Mosquito Repellent", "Godrej Hit Spray"),
])
def test_the_products_that_are_wrongly_scored_today_all_classify_non_food(category, title):
"""Named individually because each one is a live row with a health score."""
assert is_non_consumable(category, title) is True
def test_the_legacy_gate_still_answers():
"""`_looks_non_food` is kept as a delegate; nothing outside its module
imports it, but silently changing a public-looking name is worse than
keeping it honest."""
assert nds._looks_non_food("Lifebuoy Soap", "Bath Soap") is True
assert nds._looks_non_food("Amul Butter", "Dairy") is False
def test_the_verdict_map_has_no_contradictions():
"""A category listed as both consumable and non-consumable would resolve by
dict-ordering luck."""
assert len(CATEGORY_VERDICTS) == len(set(CATEGORY_VERDICTS))
assert all(isinstance(v, Edibility) for v in CATEGORY_VERDICTS.values())

View File

@@ -0,0 +1,340 @@
"""GET /api/nutrition/health-scores - every scored consumable product.
The endpoint a consumer reads to display health scores across the catalogue.
Three promises are load-bearing, and each is tested against the SQL rather than
the response, because that is where each one can silently stop being true:
1. Only CONSUMABLE products. Enforced in the WHERE clause, not by filtering a
fetched page, so `total` and `items` describe the same population.
2. Only SCORED products - no null health_score, whatever the caller sorts by.
3. Stable paging. A total ordering, so OFFSET cannot repeat or skip a product
when scores tie.
No database and no network: the connection is a recorder, so the assertions are
about the exact SQL that would be sent.
"""
from __future__ import annotations
import pytest
from app.api.routers import nutrition as nutrition_router
from app.services import nutrition_db
# ---------------------------------------------------------------------------
# Recorder
# ---------------------------------------------------------------------------
class FakeCursor:
def __init__(self, calls, rows):
self.calls = calls
self._rows = rows
self.description = None
def execute(self, sql, params=None):
self.calls.append((" ".join(str(sql).split()), list(params or [])))
def fetchall(self):
return list(self._rows)
def fetchone(self):
return (len(self._rows),)
def __enter__(self):
return self
def __exit__(self, *exc):
return False
class FakeConn:
def __init__(self, rows=()):
self.calls = []
self.rows = list(rows)
def cursor(self, *a, **k):
return FakeCursor(self.calls, self.rows)
def __enter__(self):
return self
def __exit__(self, *exc):
return False
def close(self):
pass
@pytest.fixture
def db(monkeypatch):
conn = FakeConn()
monkeypatch.setattr(nutrition_db, "_connect", lambda: conn)
return conn
def _sql(conn):
assert conn.calls, "no statement was executed"
return conn.calls[-1][0]
# ---------------------------------------------------------------------------
# Promise 1: only consumables
# ---------------------------------------------------------------------------
def test_the_default_admits_only_confirmed_consumables():
where, params = nutrition_db._health_score_filters()
assert "f.edibility = %s" in where
assert nutrition_db.CONSUMABLE in params
def test_include_unknown_still_refuses_confirmed_non_consumables():
"""The opt-in is for products the classifier could not decide on. There is
deliberately no argument that lets soap back in."""
where, params = nutrition_db._health_score_filters(include_unknown=True)
assert "f.edibility IS DISTINCT FROM %s" in where
assert nutrition_db.NON_CONSUMABLE in params
assert nutrition_db.CONSUMABLE not in params
def test_a_null_edibility_is_treated_as_unknown_not_as_food():
"""Every row written before the column existed carries NULL. A plain !=
would drop those rows out of the comparison entirely; IS DISTINCT FROM
keeps them, which is what makes `include_unknown` mean what it says."""
where, _ = nutrition_db._health_score_filters(include_unknown=True)
edibility_clause = [c for c in where.split(" AND ") if "edibility" in c]
assert edibility_clause == ["f.edibility IS DISTINCT FROM %s"]
assert "!=" not in edibility_clause[0]
@pytest.mark.parametrize("kwargs", [
{}, {"include_unknown": True}, {"brand": "Amul"}, {"category": "Biscuits"},
{"min_score": 10}, {"max_score": 90}, {"brand": "Amul", "include_unknown": True},
])
def test_edibility_is_constrained_no_matter_which_filters_are_used(kwargs):
where, _ = nutrition_db._health_score_filters(**kwargs)
assert "f.edibility" in where
# ---------------------------------------------------------------------------
# Promise 2: only scored products
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("sort_by", [
"health_score", "nutrition_score", "protein", "fiber", "sugar", "sodium",
"calcium", "iron", "vitamin_c", "calories", "product_name", "brand", "category",
])
def test_no_sort_column_can_admit_an_unscored_product(db, sort_by):
"""`query_products` derives this predicate from the sort column, so sorting
it by protein returns rows with no health_score at all. Here the promise is
stated outright and cannot be sorted away."""
nutrition_db.query_health_scores(sort_by=sort_by)
assert "i.health_score IS NOT NULL" in _sql(db)
def test_the_join_is_inner_so_a_product_with_no_insights_row_cannot_appear(db):
nutrition_db.query_health_scores()
sql = _sql(db)
assert "LEFT JOIN" not in sql
assert "JOIN nutrition_insights" in sql
def test_an_unknown_sort_column_falls_back_rather_than_reaching_the_query(db):
"""The allow-list is what stops `sort_by` being an injection point."""
nutrition_db.query_health_scores(sort_by="; DROP TABLE nutrition_facts; --")
sql = _sql(db)
assert "DROP TABLE" not in sql
assert "ORDER BY i.health_score" in sql
# ---------------------------------------------------------------------------
# Promise 3: stable paging
# ---------------------------------------------------------------------------
def test_the_ordering_is_total_so_tied_scores_cannot_shuffle_between_pages(db):
"""Without the primary-key tiebreaker Postgres may return equal
health_scores in a different order on each call, and OFFSET paging then
hands the caller one product twice and never shows another."""
nutrition_db.query_health_scores()
sql = _sql(db)
order = sql[sql.index("ORDER BY"):]
assert "f.brand ASC" in order and "f.image_id ASC" in order
def test_the_count_and_the_listing_are_filtered_identically(db):
"""A total that describes a different population from the rows beside it is
worse than no total: it tells a client to keep paging past the end, or to
stop early."""
kwargs = {"brand": "Amul", "category": "Dairy", "min_score": 20}
nutrition_db.query_health_scores(**kwargs)
list_where = _sql(db).split("WHERE", 1)[1].split("ORDER BY", 1)[0].strip()
list_params = db.calls[-1][1][:-2] # minus LIMIT/OFFSET
nutrition_db.count_health_scores(**kwargs)
count_where = _sql(db).split("WHERE", 1)[1].strip()
assert list_where == count_where
assert list_params == db.calls[-1][1]
# ---------------------------------------------------------------------------
# The route
# ---------------------------------------------------------------------------
_ROW = {
"brand": "Own Products", "image_id": "own_products_apple", "product_name": "Apple",
"category": "Fruits & Vegetables", "edibility": "consumable",
"data_status": "verified", "data_source": "usda_fdc",
"source_url": "https://fdc.nal.usda.gov/food-details/171688",
"calories_kcal": 52.0, "protein_g": 0.26, "dietary_fiber_g": 2.4,
"total_sugar_g": 10.4, "sodium_mg": 1.0,
"nutrition_score": 61.2, "health_score": 57.4,
"scoring_version": "v1.0-fsa-eu-thresholds",
"diet_tags": ["High Fiber"], "allergens": [],
}
@pytest.fixture
def rows(monkeypatch):
"""Serve a fixed page, and record the filters the route passed down."""
seen = {}
def _query(**kwargs):
seen.update(kwargs)
return [dict(_ROW)]
def _export(**kwargs):
return iter([
dict(_ROW),
dict(_ROW, product_name="Banana", health_score=82.0, diet_tags=["A", "B"]),
])
monkeypatch.setattr(nutrition_router.nutrition_db, "query_health_scores", _query)
monkeypatch.setattr(nutrition_router.nutrition_db, "count_health_scores", lambda **k: 892)
monkeypatch.setattr(nutrition_router.nutrition_db, "iter_health_scores", _export)
return seen
def test_the_envelope_carries_the_total_of_all_matches_not_of_the_page(client, rows):
body = client.get("/api/nutrition/health-scores?limit=1").json()
assert body["total"] == 892
assert body["limit"] == 1
assert body["offset"] == 0
assert len(body["items"]) == 1
def test_the_filters_reach_the_query_layer(client, rows):
client.get("/api/nutrition/health-scores?brand=Amul&category=Dairy"
"&min_score=10&max_score=90&include_unknown=true")
assert rows["brand"] == "Amul"
assert rows["category"] == "Dairy"
assert rows["min_score"] == 10
assert rows["max_score"] == 90
assert rows["include_unknown"] is True
def test_provenance_travels_with_the_number(client, rows):
"""A score is only as good as the source it was computed from, and the
consumer has to be able to show that alongside it."""
item = client.get("/api/nutrition/health-scores").json()["items"][0]
assert item["data_source"] == "usda_fdc"
assert item["source_url"].startswith("https://fdc.nal.usda.gov/")
assert item["scoring_version"] == "v1.0-fsa-eu-thresholds"
@pytest.mark.parametrize("score,band", [
(100.0, "excellent"), (80.0, "excellent"), (79.9, "good"),
(60.0, "good"), (59.9, "fair"), (40.0, "fair"), (39.9, "poor"), (0.0, "poor"),
])
def test_the_band_boundaries(score, band):
assert nutrition_router._health_band(score) == band
def test_the_healthy_boundary_matches_the_one_the_rest_of_the_api_uses():
"""`store_healthy_distribution` counts health_score >= 60 as healthy. Two
parts of the same API disagreeing about that would be indefensible."""
assert nutrition_router._health_band(60.0) == "good"
assert nutrition_router._health_band(59.99) == "fair"
def test_over_the_page_ceiling_is_rejected_rather_than_silently_capped(client, rows):
assert client.get("/api/nutrition/health-scores?limit=501").status_code == 422
assert client.get("/api/nutrition/health-scores?limit=500").status_code == 200
def test_a_negative_offset_is_rejected(client, rows):
assert client.get("/api/nutrition/health-scores?offset=-1").status_code == 422
def test_an_out_of_range_score_filter_is_rejected(client, rows):
assert client.get("/api/nutrition/health-scores?min_score=101").status_code == 422
# ---------------------------------------------------------------------------
# CSV export
# ---------------------------------------------------------------------------
def test_csv_returns_every_row_and_ignores_the_page_size(client, rows):
"""One request for the whole list is the entire point of the format."""
response = client.get("/api/nutrition/health-scores?format=csv&limit=1")
assert response.status_code == 200
assert response.headers["content-type"].startswith("text/csv")
assert "attachment" in response.headers["content-disposition"]
lines = [line for line in response.text.splitlines() if line]
assert len(lines) == 3, "header plus both rows, despite limit=1"
def test_the_csv_header_names_the_columns(client, rows):
header = client.get("/api/nutrition/health-scores?format=csv").text.splitlines()[0]
assert header.split(",")[:4] == ["brand", "image_id", "product_name", "category"]
assert "health_score" in header and "health_band" in header
def test_csv_carries_the_derived_band(client, rows):
lines = client.get("/api/nutrition/health-scores?format=csv").text.splitlines()
assert "fair" in lines[1] # 57.4
assert "excellent" in lines[2] # 82.0
def test_a_postgres_array_becomes_a_readable_cell(client, rows):
"""str() on a TEXT[] renders it with commas inside, which a spreadsheet
then splits across columns."""
lines = client.get("/api/nutrition/health-scores?format=csv").text.splitlines()
assert "A|B" in lines[2]
assert "'A'" not in lines[2]
def test_an_unrecognised_format_is_rejected(client, rows):
assert client.get("/api/nutrition/health-scores?format=xml").status_code == 422
# ---------------------------------------------------------------------------
# The response model is the contract
# ---------------------------------------------------------------------------
def test_a_null_health_score_is_a_loud_failure_not_a_blank_cell(monkeypatch, client):
"""`health_score` is required on the response model precisely so that a
query which lost its `IS NOT NULL` guarantee cannot serve nulls quietly."""
monkeypatch.setattr(nutrition_router.nutrition_db, "count_health_scores", lambda **k: 1)
monkeypatch.setattr(nutrition_router.nutrition_db, "query_health_scores",
lambda **k: [dict(_ROW, health_score=None)])
with pytest.raises(Exception):
client.get("/api/nutrition/health-scores")
def test_the_route_is_not_shadowed_by_the_product_detail_catch_all(client, rows):
"""`/{brand}/{image_id}` sits below this route in the same router. A static
path registered after it resolves as brand='health-scores'."""
body = client.get("/api/nutrition/health-scores").json()
assert "items" in body and "total" in body

View File

@@ -0,0 +1,199 @@
"""The gate that stops a bar of soap acquiring a health score.
WHAT WENT WRONG
---------------
`nutrition_scoring` is careful: it refuses to score when the numbers are
missing, and every sentence it emits traces to a published threshold. None of
that helps if the product is not food. Measured on the live catalogue, these
rows had a health score:
Colgate-Palmolive Palmolive Naturals soap
Cavinkare Nyle / Nature's, P&G Pantene shampoo
Godrej Hit Spray mosquito repellent, recorded at 291 kcal
WHERE THE GATE HAD TO GO
------------------------
Three places, and the third is the one that matters:
* `fetch_verified_nutrition` - skip the network call
* `fetch_verified_nutrition_by_barcode` - had NO gate at all, and a barcode
match records `match_confidence` 0.95, higher than anything the name
search produces
* `enrich_one_product` - the only place that knows
(brand, image_id) and can therefore refuse to WRITE A ROW
No network, no database: both boundaries are stubbed.
"""
from __future__ import annotations
import pytest
from app.services import nutrition_data_service as nds
from app.services import nutrition_enrichment_service as nes
from app.services.enrichment.barcode.sources import open_food_facts as off
class _Resp:
def __init__(self, payload, status_code=200):
self._payload = payload
self.status_code = status_code
def json(self):
return self._payload
# ---------------------------------------------------------------------------
# The retrieval paths
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("title,category", [
("Colgate-Palmolive Palmolive Naturals", "Bath Soap"),
("Cavinkare Nyle", "Hair Care"),
("P G Pg Pantene", "Hair Care"),
("Godrej Hit Spray", "Personal Care - Mosquito Repellent"),
("Jasmine Garland", "Flowers"),
])
def test_the_name_search_makes_no_call_for_a_non_consumable(monkeypatch, title, category):
called = []
monkeypatch.setattr(nds, "_search_openfoodfacts",
lambda *a, **k: called.append(1) or [])
facts = nds.fetch_verified_nutrition("Whoever", title, category)
assert facts["data_status"] == "unavailable"
assert not called, f"queried Open Food Facts about {title!r}"
def test_the_barcode_path_also_refuses_a_non_consumable(monkeypatch):
"""The hole. This path checked only USE_OPEN_FACTS and barcode presence, so
a shampoo whose barcode Open Beauty Facts happens to know acquired a full
nutrition row at confidence 0.95."""
called = []
monkeypatch.setattr(off.requests, "get",
lambda *a, **k: called.append(1) or _Resp({"status": 0}))
facts = nds.fetch_verified_nutrition_by_barcode(
"8901030865278", "Dove", "Dove Shampoo 340ml", "340ml", "Hair Care")
assert facts["data_status"] == "unavailable"
assert not called
def test_a_real_food_still_reaches_open_food_facts(monkeypatch):
"""The gate must not be so eager that it starves the happy path."""
called = []
monkeypatch.setattr(nds, "_search_openfoodfacts",
lambda *a, **k: called.append(1) or [])
nds.fetch_verified_nutrition("Amul", "Amul Butter 500g", "Dairy")
assert called, "a dairy product should still be looked up"
# ---------------------------------------------------------------------------
# The write gate
# ---------------------------------------------------------------------------
@pytest.fixture
def no_writes(monkeypatch):
"""Records every attempted write instead of performing it."""
writes = {"facts": [], "insights": []}
monkeypatch.setattr(nes.nutrition_db, "upsert_nutrition_facts",
lambda f: writes["facts"].append(f) or True)
monkeypatch.setattr(nes.nutrition_db, "upsert_nutrition_insights",
lambda i: writes["insights"].append(i) or True)
monkeypatch.setattr(nes.nutrition_db, "get_nutrition_facts", lambda b, i: None)
return writes
def test_a_non_consumable_never_gets_a_row_written(no_writes, monkeypatch):
monkeypatch.setattr(nes.nutrition_data_service, "fetch_verified_nutrition",
lambda *a, **k: pytest.fail("retrieval should not be reached"))
status = nes.enrich_one_product(
"Colgate-Palmolive", "colgate_palmolive_palmolive_naturals",
"Colgate-Palmolive Palmolive Naturals", "Bath Soap")
assert status == nes.SKIPPED_NON_CONSUMABLE
assert no_writes["facts"] == []
assert no_writes["insights"] == [], (
"an insights row with a null score still makes the product appear in "
"query_products and the analytics leaderboards"
)
def test_skipping_is_not_reported_as_unavailable(no_writes, monkeypatch):
"""'We did not look' and 'we looked and found nothing' are different facts,
and the second invites a pointless retry."""
monkeypatch.setattr(nes, "_brands_to_enrich", lambda brands, inc: ["Acme"])
monkeypatch.setattr(nes, "get_products_by_brand", lambda brand: [
{"image_id": "a1", "title": "Acme Shampoo 200ml", "category": "Hair Care"},
{"image_id": "a2", "title": "Acme Marigold Garland", "category": "Flowers"},
])
result = nes.enrich_all_products(generate_narrative=False)
assert result.skipped_non_consumable == 2
assert result.unavailable == 0
assert result.total_products == 0, (
"total_products should describe the work actually done, so the progress "
"percentage means something"
)
def test_a_consumable_is_still_enriched(no_writes, monkeypatch):
monkeypatch.setattr(nes.nutrition_data_service, "fetch_verified_nutrition",
lambda *a, **k: {"data_status": "verified",
"data_source": "openfoodfacts",
"protein_g": 8.0, "calories_kcal": 717.0})
status = nes.enrich_one_product(
"Amul", "amul_butter_500g", "Amul Butter 500g", "Dairy",
generate_narrative=False)
assert status == "verified"
assert len(no_writes["facts"]) == 1
assert no_writes["facts"][0]["brand"] == "Amul"
def test_allergen_source_names_the_database_the_facts_came_from(no_writes, monkeypatch):
"""It was hardcoded to "openfoodfacts", which was true when OFF was the only
source and would credit the wrong database once USDA joined the cascade."""
monkeypatch.setattr(nes.nutrition_data_service, "fetch_verified_nutrition",
lambda *a, **k: {"data_status": "verified",
"data_source": "usda_fdc",
"allergens": ["Milk"],
"calories_kcal": 52.0})
nes.enrich_one_product("Own Products", "own_products_apple", "Apple",
"Fruits & Vegetables", generate_narrative=False)
assert no_writes["insights"][0]["allergen_source"] == "usda_fdc"
def test_no_allergens_still_reads_unavailable(no_writes, monkeypatch):
monkeypatch.setattr(nes.nutrition_data_service, "fetch_verified_nutrition",
lambda *a, **k: {"data_status": "verified",
"data_source": "usda_fdc",
"allergens": [], "calories_kcal": 52.0})
nes.enrich_one_product("Own Products", "own_products_apple", "Apple",
"Fruits & Vegetables", generate_narrative=False)
assert no_writes["insights"][0]["allergen_source"] == "unavailable"
# ---------------------------------------------------------------------------
# Reaching brands ACTIVE_BRANDS hides
# ---------------------------------------------------------------------------
def test_an_explicit_brand_list_bypasses_active_brands(monkeypatch):
"""ACTIVE_BRANDS is right for what the app serves and wrong for a backfill:
with it set to four brands, Britannia and Parle can never be enriched."""
monkeypatch.setattr(nes, "list_available_brands",
lambda: pytest.fail("should not consult ACTIVE_BRANDS"))
assert nes._brands_to_enrich(["Britannia", " Parle "], False) == ["Britannia", "Parle"]
def test_the_default_still_honours_active_brands(monkeypatch):
monkeypatch.setattr(nes, "list_available_brands", lambda: ["Amul", "Cadbury"])
assert nes._brands_to_enrich(None, False) == ["Amul", "Cadbury"]

View File

@@ -0,0 +1,290 @@
"""Nutrition for loose commodities, from USDA FoodData Central.
WHY THIS SOURCE EXISTS
----------------------
Open Food Facts catalogues PACKAGED products. It has no entry for a raw apple,
so every one of the 159 rows in `brand_own_products` - all the fruit,
vegetables, greens, fish and eggs the shop sells loose - had no nutrition, no
benefits and no health score at all.
NO NETWORK ANYWHERE IN THIS FILE. The pinned snapshot is built from USDA's open
bulk download, so the default path makes no calls and needs no API key. Several
tests below prove that by making `requests.get` raise.
"""
from __future__ import annotations
import pytest
from app.services import nutrition_data_service as nds
from app.services import nutrition_usda_service as usda
from app.services import produce_reference
def _name_path_keys():
"""The keys `fetch_verified_nutrition` produces on a successful match.
Duplicated from test_nutrition_by_barcode.py, matching that file's
self-contained style."""
return {
"data_status", "data_source", "source_ref", "source_url",
"match_confidence", "serving_size_g", "serving_size_label",
"extended_nutrients", "per_serving", "ingredients_text",
"off_nutriscore", "allergens", "off_labels_tags",
"off_ingredients_analysis_tags", "off_categories_tags", "fetched_at",
} | set(nds._build_flat_fields({}, "100g"))
@pytest.fixture
def no_network(monkeypatch):
"""Any outbound call is a failure, not a slow test."""
def _boom(*a, **k):
raise AssertionError("the USDA path must not touch the network")
monkeypatch.setattr(usda.requests, "get", _boom)
# ---------------------------------------------------------------------------
# Shape
# ---------------------------------------------------------------------------
def test_the_usda_path_returns_the_same_shape_as_the_name_path(no_network):
"""Both feed the same `upsert_nutrition_facts`, whose INSERT is built from a
fixed column list - a key either path grew alone would be silently dropped."""
facts = usda.fetch_verified_nutrition_usda("Apple", "Fruits & Vegetables")
assert set(facts) == _name_path_keys()
def test_the_off_only_fields_are_present_but_empty(no_network):
"""USDA has no Nutri-Score and no ingredient-derived vegan flag. `[]` says
"USDA told us nothing", which is what `classify_diet_tags` reads; omitting
the keys would break shape parity."""
facts = usda.fetch_verified_nutrition_usda("Apple", "Fruits & Vegetables")
assert facts["off_nutriscore"] is None
assert facts["ingredients_text"] is None
assert facts["allergens"] == []
assert facts["off_labels_tags"] == []
# ---------------------------------------------------------------------------
# Values
# ---------------------------------------------------------------------------
def test_a_curated_commodity_resolves_offline(no_network):
"""The strongest test here: it proves the offline path is genuinely offline.
Values are USDA 171688, "Apples, raw, with skin".
"""
facts = usda.fetch_verified_nutrition_usda("Apple", "Fruits & Vegetables")
assert facts["data_status"] == "verified"
assert facts["data_source"] == "usda_fdc"
assert facts["source_ref"] == "171688"
assert facts["source_url"] == "https://fdc.nal.usda.gov/food-details/171688"
assert facts["calories_kcal"] == 52.0
assert facts["dietary_fiber_g"] == 2.4
assert facts["vitamin_c_mg"] == 4.6
@pytest.mark.parametrize("name,category,kcal", [
("Banana", "Fruits & Vegetables", 89.0),
("Spinach", "Fresh Herbs & Greens", 23.0),
("Fish Salmon", "Fish & Seafood", 208.0),
("Egg", "Eggs", 143.0),
])
def test_the_headline_commodities_all_resolve(no_network, name, category, kcal):
facts = usda.fetch_verified_nutrition_usda(name, category)
assert facts["data_status"] == "verified"
assert facts["calories_kcal"] == kcal
def test_a_size_variant_finds_its_base_commodity(no_network):
"""Catalogue rows carry varietal and pack suffixes; the curated table is
keyed on the commodity."""
for name in ("Mango Alphonso", "Mango Totapuri 1kg", "Banana Robusta"):
assert usda.fetch_verified_nutrition_usda(name)["data_status"] == "verified", name
def test_an_uncurated_commodity_is_unavailable_not_invented(no_network):
"""USDA has no entry for amla (Phyllanthus emblica); its "Gooseberries,
raw" is Ribes, a different genus. Substituting it would put the wrong
vitamin C on the row by an order of magnitude."""
facts = usda.fetch_verified_nutrition_usda("Amla", "Fruits & Vegetables")
assert facts["data_status"] == "unavailable"
assert "calories_kcal" not in facts
def test_a_missing_nutrient_stays_none_rather_than_zero(no_network):
"""The Feature-15 rule: absent is not zero."""
facts = usda.fetch_verified_nutrition_usda("Apple", "Fruits & Vegetables")
assert facts["added_sugar_g"] is None
assert facts["calories_kcal"] is not None
# ---------------------------------------------------------------------------
# The mapping layer
# ---------------------------------------------------------------------------
def _food(nid, unit, amount):
return {"description": "test", "nutrients": [{"id": nid, "unit": unit, "amount": amount}]}
def test_the_unit_is_asserted_not_assumed():
"""USDA has served the same nutrient id in different units across dataset
revisions. Sodium read as grams and scaled would be wrong by 1000x and
would still look like a plausible number."""
good = usda._build_fields(_food(1093, "MG", 79.0))
assert good["sodium_mg"] == 79.0
bad = usda._build_fields(_food(1093, "G", 79.0))
assert bad["sodium_mg"] is None, "a wrong unit must drop the value, not rescale it"
def test_international_units_are_refused_rather_than_converted():
"""The IU-to-microgram factor depends on the vitamer (retinol vs
beta-carotene), so any single factor is an estimate."""
facts = usda._build_fields(_food(1104, "IU", 3000))
assert facts["vitamin_a_mcg"] is None
def test_both_api_response_shapes_are_understood():
"""/food/{id} nests the nutrient; /foods/search flattens it; the snapshot
stores a third, trimmed shape."""
nested = {"foodNutrients": [{"nutrient": {"id": 1008, "unitName": "KCAL"}, "amount": 52.0}]}
flat = {"foodNutrients": [{"nutrientId": 1008, "unitName": "KCAL", "value": 52.0}]}
pinned = {"nutrients": [{"id": 1008, "unit": "kcal", "amount": 52.0}]}
for shape in (nested, flat, pinned):
assert usda._build_fields(shape)["calories_kcal"] == 52.0
def test_micro_sign_and_ascii_micrograms_both_parse():
"""The bulk download writes "µg"; the live API writes "UG"."""
assert usda._build_fields(_food(1106, "µg", 12.0))["vitamin_a_mcg"] == 12.0
assert usda._build_fields(_food(1106, "UG", 12.0))["vitamin_a_mcg"] == 12.0
def test_per_serving_is_arithmetic_on_given_values(no_network):
facts = usda.fetch_verified_nutrition_usda("Apple", "Fruits & Vegetables")
if facts["serving_size_g"]:
factor = facts["serving_size_g"] / 100.0
assert facts["per_serving"]["calories_kcal"] == round(52.0 * factor, 3)
def test_disabled_usda_makes_no_lookup(monkeypatch):
monkeypatch.setattr(usda, "USE_USDA_FDC", False)
monkeypatch.setattr(usda, "get_food",
lambda *a, **k: pytest.fail("should not have looked up"))
assert usda.fetch_verified_nutrition_usda("Apple")["data_status"] == "unavailable"
def test_the_live_api_is_not_called_without_a_key(monkeypatch):
"""The snapshot is the default path; the key is only for ids it lacks."""
monkeypatch.setattr(usda, "USDA_FDC_API_KEY", "")
monkeypatch.setattr(usda.requests, "get",
lambda *a, **k: pytest.fail("called the API with no key"))
assert usda.get_food(999999999) is None
# ---------------------------------------------------------------------------
# Dispatch
# ---------------------------------------------------------------------------
def test_produce_goes_to_usda_and_not_to_open_food_facts(monkeypatch):
monkeypatch.setattr(nds, "_search_openfoodfacts",
lambda *a, **k: pytest.fail("OFF must not be asked about a raw banana"))
facts = nds.fetch_verified_nutrition("Own Products", "Banana", "Fruits & Vegetables")
assert facts["data_source"] == "usda_fdc"
def test_a_branded_packaged_product_never_reaches_usda(monkeypatch):
"""The reverse fallback is deliberately absent: a packaged biscuit must
never be handed a raw-commodity number."""
monkeypatch.setattr(usda, "fetch_verified_nutrition_usda",
lambda *a, **k: pytest.fail("USDA must not be asked about a biscuit"))
monkeypatch.setattr(nds, "_search_openfoodfacts", lambda *a, **k: [])
nds.fetch_verified_nutrition("Britannia", "Britannia Good Day 200g", "Biscuits & Cookies")
def test_unbranded_pantry_staples_still_use_open_food_facts(monkeypatch):
"""`Own Products` is the bucket for EVERY unbranded row, including "Toor
Dhal 1kg" and "Sugar 1kg". Dispatching on the brand rather than the category
would send those to the wrong source."""
called = []
monkeypatch.setattr(nds, "_search_openfoodfacts",
lambda *a, **k: called.append(1) or [])
nds.fetch_verified_nutrition("Own Products", "Toor Dhal 1kg", "Pulses, Grains & Spices")
assert called
def test_produce_misfiled_under_a_junk_category_still_reaches_usda(monkeypatch):
"""17 live rows carry a bare digit as their category. The commodity lexicon
is the second opinion that rescues them."""
monkeypatch.setattr(nds, "_search_openfoodfacts", lambda *a, **k: [])
facts = nds.fetch_verified_nutrition("Own Products", "Banana", "3")
assert facts["data_source"] == "usda_fdc"
# ---------------------------------------------------------------------------
# The curated table itself
# ---------------------------------------------------------------------------
def test_every_snapshot_reference_is_pinned():
"""A half-refreshed pair - an id in the map with no snapshot entry - would
silently fall back to the live API, or to nothing."""
for name, entry in produce_reference.all_entries().items():
fdc_id = entry.get("usda_fdc_id")
if fdc_id:
assert usda.get_food(fdc_id), f"{name} references unpinned FDC id {fdc_id}"
def test_every_pinned_food_description_matches_the_curated_one():
"""Catches a snapshot rebuilt against a USDA revision that reassigned an id."""
for name, entry in produce_reference.all_entries().items():
fdc_id = entry.get("usda_fdc_id")
if not fdc_id:
continue
assert usda.get_food(fdc_id)["description"] == entry["usda_description"], name
def test_every_catalogue_row_has_an_image():
"""159 rows, 159 images. The whole point of the table."""
entries = produce_reference.all_entries()
missing = [n for n, e in entries.items() if not e.get("image_url")]
assert missing == []
assert len(entries) == 159
def test_no_image_comes_from_a_cosmetics_or_packaged_goods_database():
"""openbeautyfacts supplied the stored photo for Banana, Orange, Papaya,
Guava, Lemon and twenty more."""
for name, entry in produce_reference.all_entries().items():
url = entry["image_url"].lower()
assert "openbeautyfacts" not in url, name
assert "openproductsfacts" not in url, name
assert "openfoodfacts" not in url, name
def test_every_image_is_https():
"""An http image is blocked as mixed content and renders blank."""
for name, entry in produce_reference.all_entries().items():
assert entry["image_url"].startswith("https://"), name
def test_lookup_ignores_pack_sizes():
assert produce_reference.usda_fdc_id("Apple 1kg") == 171688
assert produce_reference.usda_fdc_id("Apple") == 171688
def test_an_unknown_commodity_returns_nothing():
assert produce_reference.lookup("Britannia Good Day Biscuits") is None
assert produce_reference.lookup("") is None

View File

@@ -0,0 +1,163 @@
"""Why a search for an apple returned a bottle of juice.
THE DEFECT CHAIN, as measured against the live catalogue
--------------------------------------------------------
1. `image_search._AMBIQUITY_HINTS` maps the token "apple" to the context hint
"fruit juice". The table was written for packaged sub-brands - Cadbury Perk,
Britannia Tiger - where the word needs steering away from its everyday
meaning. For the Own Products bucket the word IS the item, so the hint
inverts the query.
2. The hint was applied by SUBSTRING, so `Pineapple`, `Custard Apple`,
`Buttermilk`, `Baby Corn` and `Eggs White` inherited hints too.
3. `repair_brand_images` passed the table suffix "own_products" as the brand, so
the query became "own_products Apple fruit juice" - and its corroboration
guard derived the token "products", which matches `/images/products/` and
`/cdn/shop/products/`, i.e. every Open*Facts and Shopify URL ever served.
The stored primary image for `Apple` was a 2-litre juice bottle. For `Palak` it
was a photograph of a person. Every URL asserted in this file is one that was
actually in the database.
No network: nothing here calls out.
"""
from __future__ import annotations
import pytest
from app.services import image_search
from app.services.image_search import _clean_search_title
# ---------------------------------------------------------------------------
# The hint table
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("title,unwanted", [
("Pineapple", "fruit juice"),
("Custard Apple", "fruit juice"),
("Apple Green", "fruit juice"),
("Buttermilk", "chocolate dairy"),
("Baby Corn", "snack flakes"),
("Sweet Corn", "snack flakes"),
("Eggs White", "toothpaste dental"),
])
def test_produce_never_inherits_a_commodity_hint(title, unwanted):
"""In produce mode the table is skipped wholesale, because every key in it
assumes the word names a brand rather than the thing itself."""
assert unwanted not in _clean_search_title(title, produce=True)
assert _clean_search_title(title, produce=True) == title
def test_a_substring_no_longer_drags_in_a_hint():
"""`"apple" in "pineapple"` is True, which is how a pineapple came to be
searched for as juice. Whole-word matching is the fix."""
assert _clean_search_title("Pineapple") == "Pineapple"
assert _clean_search_title("Buttermilk") == "Buttermilk"
def test_the_hint_table_still_works_for_the_brands_it_was_written_for():
"""The fix must not disarm the table - these are real disambiguations."""
assert "chocolate wafer" in _clean_search_title("Cadbury Perk Crunch")
assert "biscuit cookies" in _clean_search_title("Britannia Good Day Biscuits")
assert "fruit juice" in _clean_search_title("Apple"), (
"a bare 'Apple' outside produce mode is still an ambiguous brand word"
)
# ---------------------------------------------------------------------------
# Source selection
# ---------------------------------------------------------------------------
def test_produce_does_not_query_the_packaged_goods_databases(monkeypatch):
"""Open*Facts photographs PACKAGING. Asked about a banana it answers with
whatever carton mentions one - which is how openbeautyfacts, a COSMETICS
database, supplied the stored image for Banana, Orange, Papaya, Guava,
Lemon, Pear, Plum, Apricot, Cherry, Litchi and more.
"""
calls = []
monkeypatch.setattr(image_search, "find_images_openfacts",
lambda *a, **k: calls.append("openfacts") or [])
monkeypatch.setattr(image_search, "find_images_wikimedia", lambda *a, **k: [])
monkeypatch.setattr(image_search, "find_all_image_urls_ddg", lambda *a, **k: [])
monkeypatch.setattr(image_search, "find_all_image_urls_google_cse", lambda *a, **k: [])
monkeypatch.setattr(image_search, "find_images_playwright", lambda *a, **k: [])
image_search.find_all_image_urls("Banana", brand="", validate=False, produce=True)
assert "openfacts" not in calls
def test_a_branded_product_still_queries_open_facts(monkeypatch):
calls = []
monkeypatch.setattr(image_search, "find_images_openfacts",
lambda *a, **k: calls.append("openfacts") or [])
monkeypatch.setattr(image_search, "find_images_wikimedia", lambda *a, **k: [])
monkeypatch.setattr(image_search, "find_all_image_urls_ddg", lambda *a, **k: [])
monkeypatch.setattr(image_search, "find_all_image_urls_google_cse", lambda *a, **k: [])
monkeypatch.setattr(image_search, "find_images_playwright", lambda *a, **k: [])
image_search.find_all_image_urls("Amul Butter 500g", brand="Amul", validate=False)
assert "openfacts" in calls
def test_wikimedia_stops_excluding_what_produce_actually_is(monkeypatch):
"""The standing exclusion list contains -plant -garden -nature -tree -herb
-fish -animal -flower. Every one of those describes what a fruit, a
vegetable or a fish IS, so the query asks the index to rule out the answer.
"""
seen = {}
class _Resp:
status_code = 200
@staticmethod
def json():
return {"query": {"pages": {}}}
def _fake_get(url, params=None, **kwargs):
seen["search"] = (params or {}).get("gsrsearch", "")
return _Resp()
monkeypatch.setattr(image_search.requests, "get", _fake_get)
image_search.find_images_wikimedia("Tomato", brand="", produce=True)
assert "-plant" not in seen["search"]
assert "-fish" not in seen["search"]
image_search.find_images_wikimedia("Dettol Soap", brand="Dettol")
assert "-plant" in seen["search"], "branded searches keep their exclusions"
# ---------------------------------------------------------------------------
# The repair script's corroboration guard
# ---------------------------------------------------------------------------
def test_the_bucket_name_no_longer_corroborates_every_cdn_url():
"""`_brand_tokens("own_products")` used to yield ["own", "products"], and
"products" appears in the path of essentially every product CDN. The guard
it feeds was therefore satisfied by URLs naming nothing about the item."""
from scripts.repair_brand_images import _brand_tokens, _names_product
assert _brand_tokens("own_products") == []
off_url = "https://images.openbeautyfacts.org/images/products/360/054/214/0775/front_es.27.400.jpg"
assert _names_product(off_url, "Banana", "own_products") is False, (
"this exact URL was the stored image for Banana"
)
def test_a_url_that_names_the_item_still_corroborates():
from scripts.repair_brand_images import _names_product
good = "https://upload.wikimedia.org/wikipedia/commons/8/8a/Banana-Single.jpg"
assert _names_product(good, "Banana", "own_products") is True
def test_the_bucket_is_recognised_under_either_spelling():
"""`repair()` passes the table suffix; the pipeline and UI use the display
name. Both have to reach produce mode."""
from scripts.repair_brand_images import _is_own_products
assert _is_own_products("own_products") is True
assert _is_own_products("Own Products") is True
assert _is_own_products("Amul") is False

View File

@@ -265,13 +265,62 @@ NUTRITION_FULL = (
)
def _split_top_level(text):
"""Split a comma-separated SQL list, ignoring commas inside parentheses."""
parts, depth, current = [], 0, []
for ch in text:
if ch == "(":
depth += 1
elif ch == ")":
depth -= 1
if ch == "," and depth == 0:
parts.append("".join(current).strip())
current = []
else:
current.append(ch)
if "".join(current).strip():
parts.append("".join(current).strip())
return parts
def _by_column(sql, params):
"""Map an INSERT's column names to the values it will actually write.
Indexing recorded parameters by POSITION - which is what these tests used to
do - couples every assertion to the column order of a statement none of them
are about. Adding one column to nutrition_facts broke six unrelated tests
that had nothing to say about that column. Column names are the stable
thing, so match on those.
Walks the column list and the VALUES list together, consuming a parameter
only for a `%s` placeholder, so inline literals (`'excel_upload'`,
`CURRENT_TIMESTAMP`) stay aligned instead of shifting everything after them.
"""
head, _, rest = sql.partition("VALUES")
columns = _split_top_level(head[head.index("(") + 1:head.rindex(")")])
values = _split_top_level(rest[rest.index("(") + 1:rest.index(")")])
mapped, i = {}, 0
for column, value in zip(columns, values):
if value.strip() == "%s":
mapped[column] = params[i]
i += 1
else:
mapped[column] = value.strip()
return mapped
def _facts(db):
return _find(db, "INSERT INTO nutrition_facts")[0][1]
sql, params = _find(db, "INSERT INTO nutrition_facts")[0]
return _by_column(sql, params)
def _insights(db):
hits = _find(db, "INSERT INTO nutrition_insights")
return hits[0][1] if hits else None
if not hits:
return None
sql, params = hits[0]
return _by_column(sql, params)
def test_an_absent_allergen_column_is_null_never_none(client, admin_headers, db):
@@ -282,7 +331,7 @@ def test_an_absent_allergen_column_is_null_never_none(client, admin_headers, db)
params = _insights(db)
assert params is not None
allergens, allergen_source = params[8], params[9]
allergens, allergen_source = params["allergens"], params["allergen_source"]
assert allergens is None, "allergens were invented"
assert allergen_source == "unavailable"
@@ -291,7 +340,7 @@ def test_an_absent_diet_tag_column_is_null(client, admin_headers, db):
"""It used to default to ['High Protein', 'Gluten Free'] for everything."""
body = "brand,image_id,product_name,protein_g\namul,x,Butter,5.0\n"
client.post("/api/upload/nutrition", files=_csv(body), headers=admin_headers)
assert _insights(db)[7] is None
assert _insights(db)["diet_tags"] is None
def test_absent_nutrients_are_null_not_plausible_numbers(client, admin_headers, db):
@@ -299,16 +348,17 @@ def test_absent_nutrients_are_null_not_plausible_numbers(client, admin_headers,
client.post("/api/upload/nutrition", files=_csv(body), headers=admin_headers)
params = _facts(db)
# (brand, image_id, product_name, category, data_status, then 10 nutrients)
assert params[3] is None # category
assert all(v is None for v in params[5:]), "a nutrient was invented"
assert params["category"] is None
nutrients = [c for c in params if c.endswith(("_g", "_mg", "_kcal"))]
assert nutrients, "the column-name map found no nutrient columns to check"
assert all(params[c] is None for c in nutrients), "a nutrient was invented"
def test_a_row_with_no_nutrients_is_unavailable_not_verified(client, admin_headers, db):
body = "brand,image_id,product_name\namul,x,Butter\n"
response = client.post("/api/upload/nutrition", files=_csv(body), headers=admin_headers)
assert _facts(db)[4] == "unavailable"
assert _facts(db)["data_status"] == "unavailable"
assert response.json()["data_status_counts"]["unavailable"] == 1
@@ -316,19 +366,19 @@ def test_a_row_with_some_nutrients_is_partial(client, admin_headers, db):
body = "brand,image_id,product_name,protein_g,total_sugar_g\namul,x,Butter,5.0,2.0\n"
response = client.post("/api/upload/nutrition", files=_csv(body), headers=admin_headers)
assert _facts(db)[4] == "partial"
assert _facts(db)["data_status"] == "partial"
assert response.json()["data_status_counts"]["partial"] == 1
def test_only_a_complete_core_set_is_verified(client, admin_headers, db):
response = client.post("/api/upload/nutrition", files=_csv(NUTRITION_FULL),
headers=admin_headers)
assert _facts(db)[4] == "verified"
assert _facts(db)["data_status"] == "verified"
assert response.json()["data_status_counts"]["verified"] == 1
params = _insights(db)
assert params[8] == ["Milk"]
assert params[9] == "upload"
assert params["allergens"] == ["Milk"]
assert params["allergen_source"] == "upload"
def test_insights_are_built_only_from_supplied_values(client, admin_headers, db):
@@ -336,9 +386,13 @@ def test_insights_are_built_only_from_supplied_values(client, admin_headers, db)
client.post("/api/upload/nutrition", files=_csv(body), headers=admin_headers)
params = _insights(db)
positives, cautions = params[5], params[6]
assert positives == ["Contains 5.0g protein per 100g"]
assert cautions is None, "a sugar caution was invented from no sugar value"
# Wording comes from `nutrition_scoring.generate_positive_insights`, the
# same function that phrases every other product's insights. This endpoint
# used to hand-roll its own strings, so an uploaded product read differently
# from an identical one enriched from a source.
assert params["positive_insights"] == ["Good source of protein (5.0 g per 100 g)."]
assert params["nutritional_cautions"] is None, (
"a sugar caution was invented from no sugar value")
def test_no_insight_row_at_all_when_nothing_was_supplied(client, admin_headers, db):

View File

@@ -0,0 +1,295 @@
"""POST /api/upload/nutrition computes the health score; it does not copy one.
TWO DEFECTS THIS FILE PINS SHUT.
1. NO EDIBILITY GATE. Every other write path into `nutrition_facts` refuses
non-food. This one had no such check - it imported only `store_db`,
`nutrition_db` and `_connect` - so a spreadsheet could put a health score on
a shampoo, and the public listing endpoints and analytics leaderboards then
served it.
2. THE SCORE CAME OFF THE SHEET. `health_score = _opt_float(row, [...])` stored
whatever number a column happened to contain, into the same column that
holds scores computed by `nutrition_scoring.compute_scores` from USDA and
Open Food Facts data. Two scales in one column makes every ranking that
sorts on it meaningless: an uploaded 95 outranks a computed 80 without the
two having been measured the same way.
No database: `_connect` is a recorder, so the assertions are about the exact
parameters the handler would send.
"""
from __future__ import annotations
import io
import pytest
from app.api.routers import upload as upload_router
from app.services import nutrition_scoring
class FakeCursor:
def __init__(self, calls):
self.calls = calls
def execute(self, sql, params=None):
self.calls.append((" ".join(str(sql).split()), params))
def __enter__(self):
return self
def __exit__(self, *exc):
return False
class FakeConn:
def __init__(self):
self.calls = []
def cursor(self):
return FakeCursor(self.calls)
def __enter__(self):
return self
def __exit__(self, *exc):
return False
def close(self):
pass
@pytest.fixture
def db(monkeypatch):
conn = FakeConn()
monkeypatch.setattr(upload_router, "_connect", lambda: conn)
monkeypatch.setattr(upload_router.nutrition_db, "ensure_nutrition_schema", lambda: None)
return conn
def _csv(text: str):
return {"file": ("sheet.csv", io.BytesIO(text.encode()), "text/csv")}
def _split_top_level(text):
parts, depth, current = [], 0, []
for ch in text:
if ch == "(":
depth += 1
elif ch == ")":
depth -= 1
if ch == "," and depth == 0:
parts.append("".join(current).strip())
current = []
else:
current.append(ch)
if "".join(current).strip():
parts.append("".join(current).strip())
return parts
def _by_column(conn, table):
"""An INSERT's column names mapped to the values it would write.
Matched on column NAME, never on position: these tests are about two
specific columns, and should not break when a third is added beside them.
"""
hits = [(sql, params) for sql, params in conn.calls if "INSERT INTO " + table in sql]
if not hits:
return None
sql, params = hits[0]
head, _, rest = sql.partition("VALUES")
columns = _split_top_level(head[head.index("(") + 1:head.rindex(")")])
values = _split_top_level(rest[rest.index("(") + 1:rest.index(")")])
mapped, i = {}, 0
for column, value in zip(columns, values):
if value.strip() == "%s":
mapped[column] = params[i]
i += 1
else:
mapped[column] = value.strip()
return mapped
# A complete core set, so `compute_scores` has enough to work with. The
# health_score column is deliberately a number no rule could ever produce from
# these nutrients, so a test asserting it was NOT used cannot pass by accident.
SHEET_SCORE = 95.0
BISCUIT = (
"brand,image_id,product_name,category,calories_kcal,protein_g,carbohydrates_g,"
"total_sugar_g,dietary_fiber_g,total_fat_g,sodium_mg,health_score\n"
"britannia,b1,Britannia Good Day 200g,Biscuits & Cookies,"
"480,6.5,66.0,28.0,2.1,21.0,350,95\n"
)
BISCUIT_VALUES = {
"calories_kcal": 480.0, "protein_g": 6.5, "carbohydrates_g": 66.0,
"total_sugar_g": 28.0, "dietary_fiber_g": 2.1, "total_fat_g": 21.0,
"sodium_mg": 350.0, "calcium_mg": None, "iron_mg": None, "vitamin_c_mg": None,
}
# ---------------------------------------------------------------------------
# The gate
# ---------------------------------------------------------------------------
def test_a_shampoo_row_writes_nothing_at_all(client, admin_headers, db):
"""Not "writes a row with a null score" - no row. An insights row with a
null score still makes the product appear in the listing endpoints and the
leaderboards, which is the whole problem."""
body = ("brand,image_id,product_name,category,protein_g,health_score\n"
"pantene,p1,Pantene Pro-V Shampoo 340ml,Hair Care,5.0,88\n")
response = client.post("/api/upload/nutrition", files=_csv(body), headers=admin_headers)
assert response.status_code == 422, "nothing importable means nothing imported"
assert _by_column(db, "nutrition_facts") is None
assert _by_column(db, "nutrition_insights") is None
@pytest.mark.parametrize("category,product", [
("Bath Soap", "Palmolive Naturals"),
("Hair Care", "Nyle Anti-Hairfall Shampoo"),
("Personal Care - Mosquito Repellent", "Godrej Hit Spray"),
("Flowers", "Marigold Garland"),
])
def test_the_products_that_carry_a_live_health_score_today_are_all_refused(
client, admin_headers, db, category, product):
body = ("brand,image_id,product_name,category,calories_kcal,health_score\n"
"x,x1,{},{},291,72\n".format(product, category))
client.post("/api/upload/nutrition", files=_csv(body), headers=admin_headers)
assert _by_column(db, "nutrition_facts") is None
def test_a_refused_row_is_reported_rather_than_silently_dropped(client, admin_headers, db):
"""The uploader has to be able to see that a row in their sheet was not
imported, and why."""
body = ("brand,image_id,product_name,category,calories_kcal,protein_g,carbohydrates_g,"
"total_sugar_g,dietary_fiber_g,total_fat_g,sodium_mg\n"
"britannia,b1,Britannia Good Day 200g,Biscuits & Cookies,480,6.5,66,28,2.1,21,350\n"
"pantene,p1,Pantene Shampoo,Hair Care,5,1,1,1,1,1,1\n")
body_json = client.post("/api/upload/nutrition", files=_csv(body),
headers=admin_headers).json()
assert body_json["rows_imported"] == 1
assert body_json["skipped_non_consumable"] == 1
assert "not a food or drink" in body_json["errors"][0]["error"]
def test_the_refusal_count_is_separate_from_the_malformed_row_count(client, admin_headers, db):
"""`rows_skipped` reads as "your sheet was wrong". A non-consumable row was
perfectly well-formed and deliberately not imported - a different fact."""
body_json = client.post("/api/upload/nutrition", files=_csv(BISCUIT),
headers=admin_headers).json()
assert body_json["skipped_non_consumable"] == 0
assert body_json["rows_skipped"] == 0
def test_the_verdict_is_stored_so_the_listing_endpoint_can_filter_on_it(
client, admin_headers, db):
client.post("/api/upload/nutrition", files=_csv(BISCUIT), headers=admin_headers)
assert _by_column(db, "nutrition_facts")["edibility"] == "consumable"
# ---------------------------------------------------------------------------
# The score
# ---------------------------------------------------------------------------
def test_the_stored_score_is_computed_from_the_nutrients(client, admin_headers, db):
client.post("/api/upload/nutrition", files=_csv(BISCUIT), headers=admin_headers)
expected = nutrition_scoring.compute_scores(
dict(BISCUIT_VALUES, data_status="verified", allergens=[]))
insights = _by_column(db, "nutrition_insights")
assert insights["health_score"] == expected["health_score"]
assert insights["nutrition_score"] == expected["nutrition_score"]
def test_the_number_in_the_sheet_is_not_the_number_that_is_stored(client, admin_headers, db):
"""The regression this endpoint shipped with. 95 is not a score any rule
could derive from a biscuit with 28 g of sugar."""
client.post("/api/upload/nutrition", files=_csv(BISCUIT), headers=admin_headers)
assert _by_column(db, "nutrition_insights")["health_score"] != SHEET_SCORE
def test_the_scoring_version_identifies_the_rules_that_produced_it(client, admin_headers, db):
"""Without this, a computed score and a copied one are indistinguishable
once they are both sitting in the same column."""
client.post("/api/upload/nutrition", files=_csv(BISCUIT), headers=admin_headers)
assert _by_column(db, "nutrition_insights")["scoring_version"] == \
nutrition_scoring.SCORING_VERSION
def test_the_two_scores_are_no_longer_the_same_number_written_twice(client, admin_headers, db):
"""`nutrition_score` is nutrient density; `health_score` is that adjusted
for calorie density. This endpoint used to write the sheet's one number
into both columns, so the calorie penalty silently vanished."""
client.post("/api/upload/nutrition", files=_csv(BISCUIT), headers=admin_headers)
insights = _by_column(db, "nutrition_insights")
assert insights["health_score"] < insights["nutrition_score"], (
"480 kcal per 100 g should carry a calorie-density penalty")
def test_an_operator_score_survives_when_the_rules_cannot_produce_one(
client, admin_headers, db):
"""Too few nutrients to score fairly. Their number is kept - they may have
it from a source this sheet does not carry - but the provenance says so."""
body = ("brand,image_id,product_name,category,health_score\n"
"britannia,b1,Britannia Marie Gold,Biscuits & Cookies,71\n")
client.post("/api/upload/nutrition", files=_csv(body), headers=admin_headers)
insights = _by_column(db, "nutrition_insights")
assert insights["health_score"] == 71.0
assert insights["nutrition_score"] is None, "nothing was computed, so nothing is claimed"
assert insights["scoring_version"] == "excel_upload"
# ---------------------------------------------------------------------------
# The insights beside the score
# ---------------------------------------------------------------------------
def test_insights_come_from_the_same_rules_as_every_other_product(
client, admin_headers, db):
"""This endpoint used to hand-roll its own two sentences, so an uploaded
product read differently from an identical one enriched from a source."""
client.post("/api/upload/nutrition", files=_csv(BISCUIT), headers=admin_headers)
facts = dict(BISCUIT_VALUES, data_status="verified", allergens=[])
insights = _by_column(db, "nutrition_insights")
assert insights["nutritional_cautions"] == nutrition_scoring.generate_cautions(facts)
def test_a_sheet_supplied_diet_tag_is_kept_alongside_the_derived_ones(
client, admin_headers, db):
""""Vegan" is an ingredient fact; the rules work on numbers and cannot
derive it. Replacing the operator's tags with the derived set would throw
away the only place that information exists."""
body = BISCUIT.replace("health_score\n", "health_score,diet_tags\n")
body = body.replace(",350,95\n", ",350,95,Vegan\n")
client.post("/api/upload/nutrition", files=_csv(body), headers=admin_headers)
assert "Vegan" in _by_column(db, "nutrition_insights")["diet_tags"]
def test_an_absent_allergen_column_is_still_null_not_an_empty_list(
client, admin_headers, db):
"""The rule this module exists for, re-checked after the rewrite: "we were
not told" must not become "contains no allergens"."""
client.post("/api/upload/nutrition", files=_csv(BISCUIT), headers=admin_headers)
insights = _by_column(db, "nutrition_insights")
assert insights["allergens"] is None
assert insights["allergen_source"] == "unavailable"
def test_absent_nutrients_are_still_null_not_zero(client, admin_headers, db):
"""`compute_scores` reads absent as "do not score this component". A zero
would score it, badly, against a value nobody supplied."""
client.post("/api/upload/nutrition", files=_csv(BISCUIT), headers=admin_headers)
facts = _by_column(db, "nutrition_facts")
assert facts["calcium_mg"] is None
assert facts["vitamin_c_mg"] is None