diff --git a/app/services/consumability.py b/app/services/consumability.py index ff9136c..afe3a9c 100644 --- a/app/services/consumability.py +++ b/app/services/consumability.py @@ -54,9 +54,11 @@ bad import. Any single list is stale the moment somebody adds a category. catalogue actually contains. 2. Failing that, the HSN chapter the category resolves to. The tariff's own classification, already maintained here for tax purposes. - 3. Failing that, the product title, through the same commodity lexicon and + 3. Failing that, the product title: first a non-food brand or product-line + name (see `_NON_FOOD_LINE_WORDS`), then the same commodity lexicon and category detector the ingestion pipeline uses. - 4. Failing that, UNKNOWN - which enriches nothing and deletes nothing. + 4. Failing that, UNKNOWN - which deletes nothing, and which the write paths + still enrich, so every non-food line that reaches it is a leak. WHY THE HSN RANGE IS NOT SIMPLY 01-24 -------------------------------------- @@ -109,7 +111,7 @@ class Edibility(str, Enum): class EdibilityVerdict: edibility: Edibility reason: str # human-readable; the purge audit prints this - signal: str # category_map | hsn_chapter | title_lexicon | title_keyword | none + signal: str # category_map | hsn_chapter | title_brand_line | title_lexicon | title_keyword | none # --------------------------------------------------------------------------- @@ -227,6 +229,29 @@ _FOOD_TITLE_WORDS = frozenset({ "chyawanprash", "chyavanprash", "honey", "malt", }) +# Brand and product-line names sold ONLY as non-food, consulted before the food +# words above. These exist for the uninformative "General" category: a title like +# "Colgate-Palmolive Palmolive Naturals 350g" names no article at all, so neither +# word list nor the lexicon recognises it, the verdict is UNKNOWN - and the write +# paths refuse only NON_CONSUMABLE, so it was scored. Checked before the food +# words because Palmolive sells a "Milk & Honey" range and "honey" is food. +# +# Deliberately conservative: a name any food range also uses stays out, because +# a hit here outranks every food word. Left out on purpose: "dove" (Mars +# chocolate), "himalaya" (supplements), "parachute" (edible coconut oil), +# "hit", "wheel", "tide" (ordinary words). Only reached when the category says +# nothing, so a real category always wins. +_NON_FOOD_LINE_WORDS = frozenset({ + # personal care + "palmolive", "colgate", "lifebuoy", "lux", "dettol", "savlon", "santoor", + "cinthol", "pears", "hamam", "medimix", "margo", "fiama", "vivel", "rexona", + "pantene", "sunsilk", "nivea", "vaseline", "ponds", "closeup", "pepsodent", + "sensodyne", + # household + "ariel", "surf", "rin", "vim", "harpic", "lizol", "odonil", "goodknight", + "allout", "mortein", +}) + _WORD_RE = re.compile(r"[a-z0-9]+") @@ -306,6 +331,15 @@ def classify_edibility(category: Optional[str], title: str = "") -> EdibilityVer if text: words = _words(text) + line_hit = words & _NON_FOOD_LINE_WORDS + if line_hit: + return EdibilityVerdict( + Edibility.NON_CONSUMABLE, + f"title {title!r} names a non-food brand or line " + f"({sorted(line_hit)[0]!r})", + "title_brand_line", + ) + food_hit = words & _FOOD_TITLE_WORDS if food_hit: return EdibilityVerdict( diff --git a/tests/test_consumability.py b/tests/test_consumability.py index 19e1266..e93d9bd 100644 --- a/tests/test_consumability.py +++ b/tests/test_consumability.py @@ -235,7 +235,38 @@ def test_a_toothpaste_brand_is_not_a_chocolate(): have been handed a health score. """ verdict = classify_edibility("General", "Colgate-Palmolive Palmolive Naturals") - assert verdict.edibility is not Edibility.CONSUMABLE + assert verdict.edibility is Edibility.NON_CONSUMABLE + + +@pytest.mark.parametrize("title", [ + "Colgate-Palmolive Palmolive Naturals 350g", + "Colgate-Palmolive Palmolive Naturals 250g", + "Colgate-Palmolive Palmolive Naturals 100g", + # "honey" is a food word; the brand line has to outrank it + "Palmolive Naturals Milk & Honey 125g", +]) +def test_a_non_food_line_in_general_is_not_scored(title): + """Live rows in "General" whose titles name no article at all. Asserting + merely "not CONSUMABLE" let them through as UNKNOWN, and the write paths + score UNKNOWN - so the verdict has to be positively NON_CONSUMABLE.""" + verdict = classify_edibility("General", title) + assert verdict.edibility is Edibility.NON_CONSUMABLE + assert verdict.signal == "title_brand_line" + + +@pytest.mark.parametrize("title", [ + "Amul Butter 500g", + "Aachi Sambar Powder 100g", + "Dabur Honey 500g", + "Nestle Cerelac 300g", +]) +def test_food_in_general_is_still_consumable(title): + assert is_consumable("General", title) is True + + +def test_a_real_category_outranks_the_brand_line(): + """The brand-line list is only consulted when the category says nothing.""" + assert is_consumable("Dairy", "Pears Milk 500ml") is True def test_soapnut_is_not_a_soap(): diff --git a/tests/test_nutrition_consumable_gate.py b/tests/test_nutrition_consumable_gate.py index 1198ed6..3c1994a 100644 --- a/tests/test_nutrition_consumable_gate.py +++ b/tests/test_nutrition_consumable_gate.py @@ -120,6 +120,21 @@ def test_a_non_consumable_never_gets_a_row_written(no_writes, monkeypatch): ) +def test_a_non_food_line_filed_under_general_gets_no_row(no_writes, monkeypatch): + """The live shape: "General" carries no information and the title names no + article, only the brand line. It used to classify UNKNOWN and be scored.""" + monkeypatch.setattr(nes.nutrition_data_service, "fetch_verified_nutrition", + lambda *a, **k: pytest.fail("retrieval should not be reached")) + + status = nes.enrich_one_product( + "Colgate-Palmolive", "colgate_palmolive_palmolive_naturals_350g", + "Colgate-Palmolive Palmolive Naturals 350g", "General") + + assert status == nes.SKIPPED_NON_CONSUMABLE + assert no_writes["facts"] == [] + assert no_writes["insights"] == [] + + def test_skipping_is_not_reported_as_unavailable(no_writes, monkeypatch): """'We did not look' and 'we looked and found nothing' are different facts, and the second invites a pointless retry."""