Non-consumable products updation

This commit is contained in:
sriram
2026-09-25 11:48:52 +05:30
parent fe208f4715
commit 6628207810
3 changed files with 84 additions and 4 deletions

View File

@@ -54,9 +54,11 @@ bad import. Any single list is stale the moment somebody adds a category.
catalogue actually contains.
2. Failing that, the HSN chapter the category resolves to. The tariff's own
classification, already maintained here for tax purposes.
3. Failing that, the product title, through the same commodity lexicon and
3. Failing that, the product title: first a non-food brand or product-line
name (see `_NON_FOOD_LINE_WORDS`), then the same commodity lexicon and
category detector the ingestion pipeline uses.
4. Failing that, UNKNOWN - which enriches nothing and deletes nothing.
4. Failing that, UNKNOWN - which deletes nothing, and which the write paths
still enrich, so every non-food line that reaches it is a leak.
WHY THE HSN RANGE IS NOT SIMPLY 01-24
--------------------------------------
@@ -109,7 +111,7 @@ class Edibility(str, Enum):
class EdibilityVerdict:
edibility: Edibility
reason: str # human-readable; the purge audit prints this
signal: str # category_map | hsn_chapter | title_lexicon | title_keyword | none
signal: str # category_map | hsn_chapter | title_brand_line | title_lexicon | title_keyword | none
# ---------------------------------------------------------------------------
@@ -227,6 +229,29 @@ _FOOD_TITLE_WORDS = frozenset({
"chyawanprash", "chyavanprash", "honey", "malt",
})
# Brand and product-line names sold ONLY as non-food, consulted before the food
# words above. These exist for the uninformative "General" category: a title like
# "Colgate-Palmolive Palmolive Naturals 350g" names no article at all, so neither
# word list nor the lexicon recognises it, the verdict is UNKNOWN - and the write
# paths refuse only NON_CONSUMABLE, so it was scored. Checked before the food
# words because Palmolive sells a "Milk & Honey" range and "honey" is food.
#
# Deliberately conservative: a name any food range also uses stays out, because
# a hit here outranks every food word. Left out on purpose: "dove" (Mars
# chocolate), "himalaya" (supplements), "parachute" (edible coconut oil),
# "hit", "wheel", "tide" (ordinary words). Only reached when the category says
# nothing, so a real category always wins.
_NON_FOOD_LINE_WORDS = frozenset({
# personal care
"palmolive", "colgate", "lifebuoy", "lux", "dettol", "savlon", "santoor",
"cinthol", "pears", "hamam", "medimix", "margo", "fiama", "vivel", "rexona",
"pantene", "sunsilk", "nivea", "vaseline", "ponds", "closeup", "pepsodent",
"sensodyne",
# household
"ariel", "surf", "rin", "vim", "harpic", "lizol", "odonil", "goodknight",
"allout", "mortein",
})
_WORD_RE = re.compile(r"[a-z0-9]+")
@@ -306,6 +331,15 @@ def classify_edibility(category: Optional[str], title: str = "") -> EdibilityVer
if text:
words = _words(text)
line_hit = words & _NON_FOOD_LINE_WORDS
if line_hit:
return EdibilityVerdict(
Edibility.NON_CONSUMABLE,
f"title {title!r} names a non-food brand or line "
f"({sorted(line_hit)[0]!r})",
"title_brand_line",
)
food_hit = words & _FOOD_TITLE_WORDS
if food_hit:
return EdibilityVerdict(

View File

@@ -235,7 +235,38 @@ def test_a_toothpaste_brand_is_not_a_chocolate():
have been handed a health score.
"""
verdict = classify_edibility("General", "Colgate-Palmolive Palmolive Naturals")
assert verdict.edibility is not Edibility.CONSUMABLE
assert verdict.edibility is Edibility.NON_CONSUMABLE
@pytest.mark.parametrize("title", [
"Colgate-Palmolive Palmolive Naturals 350g",
"Colgate-Palmolive Palmolive Naturals 250g",
"Colgate-Palmolive Palmolive Naturals 100g",
# "honey" is a food word; the brand line has to outrank it
"Palmolive Naturals Milk & Honey 125g",
])
def test_a_non_food_line_in_general_is_not_scored(title):
"""Live rows in "General" whose titles name no article at all. Asserting
merely "not CONSUMABLE" let them through as UNKNOWN, and the write paths
score UNKNOWN - so the verdict has to be positively NON_CONSUMABLE."""
verdict = classify_edibility("General", title)
assert verdict.edibility is Edibility.NON_CONSUMABLE
assert verdict.signal == "title_brand_line"
@pytest.mark.parametrize("title", [
"Amul Butter 500g",
"Aachi Sambar Powder 100g",
"Dabur Honey 500g",
"Nestle Cerelac 300g",
])
def test_food_in_general_is_still_consumable(title):
assert is_consumable("General", title) is True
def test_a_real_category_outranks_the_brand_line():
"""The brand-line list is only consulted when the category says nothing."""
assert is_consumable("Dairy", "Pears Milk 500ml") is True
def test_soapnut_is_not_a_soap():

View File

@@ -120,6 +120,21 @@ def test_a_non_consumable_never_gets_a_row_written(no_writes, monkeypatch):
)
def test_a_non_food_line_filed_under_general_gets_no_row(no_writes, monkeypatch):
"""The live shape: "General" carries no information and the title names no
article, only the brand line. It used to classify UNKNOWN and be scored."""
monkeypatch.setattr(nes.nutrition_data_service, "fetch_verified_nutrition",
lambda *a, **k: pytest.fail("retrieval should not be reached"))
status = nes.enrich_one_product(
"Colgate-Palmolive", "colgate_palmolive_palmolive_naturals_350g",
"Colgate-Palmolive Palmolive Naturals 350g", "General")
assert status == nes.SKIPPED_NON_CONSUMABLE
assert no_writes["facts"] == []
assert no_writes["insights"] == []
def test_skipping_is_not_reported_as_unavailable(no_writes, monkeypatch):
"""'We did not look' and 'we looked and found nothing' are different facts,
and the second invites a pointless retry."""