Non-consumable products updation

This commit is contained in:
sriram
2026-09-25 11:48:52 +05:30
parent fe208f4715
commit 6628207810
3 changed files with 84 additions and 4 deletions

View File

@@ -54,9 +54,11 @@ bad import. Any single list is stale the moment somebody adds a category.
catalogue actually contains.
2. Failing that, the HSN chapter the category resolves to. The tariff's own
classification, already maintained here for tax purposes.
3. Failing that, the product title, through the same commodity lexicon and
3. Failing that, the product title: first a non-food brand or product-line
name (see `_NON_FOOD_LINE_WORDS`), then the same commodity lexicon and
category detector the ingestion pipeline uses.
4. Failing that, UNKNOWN - which enriches nothing and deletes nothing.
4. Failing that, UNKNOWN - which deletes nothing, and which the write paths
still enrich, so every non-food line that reaches it is a leak.
WHY THE HSN RANGE IS NOT SIMPLY 01-24
--------------------------------------
@@ -109,7 +111,7 @@ class Edibility(str, Enum):
class EdibilityVerdict:
edibility: Edibility
reason: str # human-readable; the purge audit prints this
signal: str # category_map | hsn_chapter | title_lexicon | title_keyword | none
signal: str # category_map | hsn_chapter | title_brand_line | title_lexicon | title_keyword | none
# ---------------------------------------------------------------------------
@@ -227,6 +229,29 @@ _FOOD_TITLE_WORDS = frozenset({
"chyawanprash", "chyavanprash", "honey", "malt",
})
# Brand and product-line names sold ONLY as non-food, consulted before the food
# words above. These exist for the uninformative "General" category: a title like
# "Colgate-Palmolive Palmolive Naturals 350g" names no article at all, so neither
# word list nor the lexicon recognises it, the verdict is UNKNOWN - and the write
# paths refuse only NON_CONSUMABLE, so it was scored. Checked before the food
# words because Palmolive sells a "Milk & Honey" range and "honey" is food.
#
# Deliberately conservative: a name any food range also uses stays out, because
# a hit here outranks every food word. Left out on purpose: "dove" (Mars
# chocolate), "himalaya" (supplements), "parachute" (edible coconut oil),
# "hit", "wheel", "tide" (ordinary words). Only reached when the category says
# nothing, so a real category always wins.
_NON_FOOD_LINE_WORDS = frozenset({
# personal care
"palmolive", "colgate", "lifebuoy", "lux", "dettol", "savlon", "santoor",
"cinthol", "pears", "hamam", "medimix", "margo", "fiama", "vivel", "rexona",
"pantene", "sunsilk", "nivea", "vaseline", "ponds", "closeup", "pepsodent",
"sensodyne",
# household
"ariel", "surf", "rin", "vim", "harpic", "lizol", "odonil", "goodknight",
"allout", "mortein",
})
_WORD_RE = re.compile(r"[a-z0-9]+")
@@ -306,6 +331,15 @@ def classify_edibility(category: Optional[str], title: str = "") -> EdibilityVer
if text:
words = _words(text)
line_hit = words & _NON_FOOD_LINE_WORDS
if line_hit:
return EdibilityVerdict(
Edibility.NON_CONSUMABLE,
f"title {title!r} names a non-food brand or line "
f"({sorted(line_hit)[0]!r})",
"title_brand_line",
)
food_hit = words & _FOOD_TITLE_WORDS
if food_hit:
return EdibilityVerdict(