417 lines
19 KiB
Python
417 lines
19 KiB
Python
"""
|
|
Is this product something a person eats or drinks?
|
|
|
|
WHY THIS FILE EXISTS
|
|
--------------------
|
|
The nutrition module's contract is that every number it publishes traces to a
|
|
verified source (see `nutrition_data_service`'s docstring). That contract says
|
|
nothing about products which have no nutrition panel *at all*, and the only
|
|
guard that existed was a seventeen-word substring list:
|
|
|
|
nutrition_data_service.NON_FOOD_KEYWORDS = ("soap", "detergent", "shampoo", ...)
|
|
|
|
matched against `f"{title} {category}".lower()`. It missed every category name it
|
|
was not literally spelled with. Measured against the live catalogue, these rows
|
|
carry a health score today:
|
|
|
|
Colgate-Palmolive Palmolive Naturals General score present
|
|
Cavinkare Nyle / Nature's Hair Care score present
|
|
P&G Pantene Hair Care score present
|
|
Godrej Hit Spray Personal Care - Mosquito 291 kcal (!)
|
|
|
|
A mosquito repellent with a calorie count is not a cosmetic defect. It is the
|
|
nutrition module asserting a fact about a product it has no business describing,
|
|
and a shopper has no way to tell that the number is meaningless.
|
|
|
|
Being a substring test, that list also has the opposite failure: "soap" matches
|
|
*soapnut* (reetha), a real commodity. Matching here is whole-word.
|
|
|
|
WHY THREE STATES AND NOT A BOOLEAN
|
|
----------------------------------
|
|
`is_consumable` and `is_non_consumable` are NOT inverses, and that is the single
|
|
most important thing in this file.
|
|
|
|
Two callers ask opposite questions of the same fact:
|
|
|
|
the WRITE gate - "may I attach nutrition to this?" unknown => NO
|
|
the DELETE gate - "may I destroy this row?" unknown => NO
|
|
|
|
Collapsing them into one boolean makes whichever caller loses the coin-toss act
|
|
destructively on a guess: a single `not is_consumable()` in the purge script
|
|
would delete every row the classifier merely failed to recognise. So the engine
|
|
returns `CONSUMABLE | NON_CONSUMABLE | UNKNOWN`, and both booleans are positive
|
|
tests that return False for UNKNOWN.
|
|
|
|
WHY A LAYERED DECISION AND NOT ONE KEYWORD LIST
|
|
-----------------------------------------------
|
|
The catalogue's category strings are not one vocabulary. Four maps have drifted
|
|
apart - CATEGORY_REGISTRY (31 names), HSN_GST_TABLE (60), CATEGORY_UNIT_TYPE
|
|
(~70) and CATEGORY_TYPE_WORDS - and the live tables hold 66 distinct values
|
|
including "General" (40 rows) and the bare strings "1".."5" (17 rows) left by a
|
|
bad import. Any single list is stale the moment somebody adds a category.
|
|
|
|
1. An explicit per-category verdict, hand-set, for every string the live
|
|
catalogue actually contains.
|
|
2. Failing that, the HSN chapter the category resolves to. The tariff's own
|
|
classification, already maintained here for tax purposes.
|
|
3. Failing that, the product title: first a non-food brand or product-line
|
|
name (see `_NON_FOOD_LINE_WORDS`), then the same commodity lexicon and
|
|
category detector the ingestion pipeline uses.
|
|
4. Failing that, UNKNOWN - which deletes nothing, and which the write paths
|
|
still enrich, so every non-food line that reaches it is a leak.
|
|
|
|
WHY THE HSN RANGE IS NOT SIMPLY 01-24
|
|
--------------------------------------
|
|
"Chapters 1 to 24 are the food chapters" is the obvious rule and it is wrong at
|
|
exactly the case this project cares about. Chapter 06 is live plants and cut
|
|
flowers, and `Flowers` is a live category here with 14 rows that reach the same
|
|
Own Products table as the vegetables. A naive range would score a jasmine
|
|
garland. Excluded for the same reason: 05 (inedible animal products), 14
|
|
(vegetable plaiting materials), 23 (animal feed) and 24 (tobacco - consumed, but
|
|
it publishes no nutrition panel).
|
|
|
|
EDGE CASES, AND WHY THEY WENT THE WAY THEY DID
|
|
----------------------------------------------
|
|
Flowers False. Sold loose beside the vegetables and routed by the same
|
|
produce lexicon, but a garland is not food.
|
|
Oral Care False. Toothpaste goes in the mouth and is spat out; it carries
|
|
no nutrition panel, and it is Open *Beauty* Facts that matches it.
|
|
Health Care - False. A cough syrup is ingested but is regulated as a drug and
|
|
Cold & Cough / publishes dosage, not nutrition. Scoring it would be the most
|
|
Digestive dangerous error available here.
|
|
Baby Care MIXED, so it defers to the title. The audit found 21 live rows
|
|
Health Care - that are Nestle Cerelac, Nan Pro and Lactogen sitting beside a
|
|
Ayurvedic bottle of baby oil; Dabur Chyawanprash sits beside cough syrup.
|
|
Infant formula is among the most nutrition-labelled food sold in
|
|
India, so a blanket "not food" here would have been a worse
|
|
defect than the one this module fixes.
|
|
Health Drinks True. Horlicks, Boost and Complan publish a real nutrition
|
|
panel, notwithstanding the word "Health".
|
|
Household - False, and listed explicitly so it can never be swept in by
|
|
Lamp Oil "Cooking Oils" being True. See `_hsn_verdict` for the second
|
|
guard on the same trap.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
from dataclasses import dataclass
|
|
from enum import Enum
|
|
from typing import Dict, Optional, Tuple
|
|
|
|
from app.services.category_registry import _normalize, detect_category_from_text
|
|
|
|
|
|
class Edibility(str, Enum):
|
|
CONSUMABLE = "consumable"
|
|
NON_CONSUMABLE = "non_consumable"
|
|
UNKNOWN = "unknown"
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class EdibilityVerdict:
|
|
edibility: Edibility
|
|
reason: str # human-readable; the purge audit prints this
|
|
signal: str # category_map | hsn_chapter | title_brand_line | title_lexicon | title_keyword | none
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# 1. Explicit verdicts
|
|
# ---------------------------------------------------------------------------
|
|
# Keyed on the NORMALIZED category ("Pulses, Grains & Spices" -> "pulses grains
|
|
# and spices") so a stored string differing only in punctuation or case still
|
|
# lands here rather than falling through to the HSN guess. `_normalize` is
|
|
# imported rather than reimplemented: a fifth normalizer that disagreed with the
|
|
# other four is exactly how this area got into trouble.
|
|
_CONSUMABLE_CATEGORIES: Tuple[str, ...] = (
|
|
# Dairy and dairy-adjacent
|
|
"Dairy", "Cheese", "Dairy - Desserts", "Ice Cream",
|
|
# Drinks
|
|
"Beverages", "Tea & Coffee", "Health Drinks", "Food & Beverages",
|
|
# Fresh, loose goods sold by weight or by the piece
|
|
"Fruits & Vegetables", "Fresh Herbs & Greens", "Fish & Seafood", "Eggs",
|
|
# Bakery and biscuit
|
|
"Biscuits & Cookies", "Biscuits", "Crackers", "Rusk", "Cakes & Muffins",
|
|
"Bakery & Breads", "Breakfast Cereal",
|
|
# Confectionery
|
|
"Chocolates", "Candy & Confectionery",
|
|
# Savoury
|
|
"Snacks", "Namkeen", "Ready to Eat", "Noodles & Instant Food",
|
|
"Pasta & Noodles",
|
|
# Staples, pulses, spices
|
|
"Atta & Staples", "Staples", "Flour & Grains", "Salt & Staples",
|
|
"Sugar & Jaggery", "Pulses, Grains & Spices", "Spices & Masalas",
|
|
"Cooking Oils", "Dry Fruits & Nuts", "Millets", "Rice & Pulses",
|
|
# Prepared foods
|
|
"Food - Mixes", "Food - Spreads", "Food - Soups & Sauces",
|
|
"Pickles & Chutneys", "Health Foods", "Sweets",
|
|
)
|
|
|
|
_NON_CONSUMABLE_CATEGORIES: Tuple[str, ...] = (
|
|
# Personal care
|
|
"Hair Care", "Skin Care", "Skin & Bath Care", "Bath Soap", "Beauty Care",
|
|
"Oral Care", "Fragrance & Deodorants", "Men's Grooming",
|
|
"Feminine Hygiene", "Personal Care",
|
|
# Household
|
|
"Detergents & Fabric Care", "Dishwash", "Household Cleaning",
|
|
"Household - Agarbatti", "Household - Lamp Oil", "Household - Air Freshener",
|
|
"Personal Care - Mosquito Repellent",
|
|
# Ingested, but regulated as medicines: they publish dosage, not nutrition
|
|
"Health Care - Cold & Cough", "Health Care - Digestive",
|
|
"Health Care - Antiseptic", "Health Care - First Aid",
|
|
# Not food, despite arriving through the produce lexicon
|
|
"Flowers",
|
|
)
|
|
|
|
# Categories holding BOTH food and non-food, where a single verdict is simply
|
|
# wrong. Deferred to the title, exactly like an uninformative category.
|
|
#
|
|
# Found by running the purge audit before deleting anything - which is what that
|
|
# dry run is for. "Baby Care" held 21 live rows, and they are Nestle Cerelac,
|
|
# Nan Pro and Lactogen alongside a bottle of baby oil. Infant formula and baby
|
|
# cereal are among the most heavily nutrition-labelled products sold in India;
|
|
# refusing them a score would have been a worse defect than the one being fixed.
|
|
#
|
|
# "Health Care - Ayurvedic" is the same shape: Dabur Chyawanprash is eaten by
|
|
# the spoonful and carries a nutrition panel, while a cough syrup does not.
|
|
_MIXED_CATEGORIES = frozenset(
|
|
{_normalize("Baby Care"), _normalize("Health Care - Ayurvedic")}
|
|
)
|
|
|
|
CATEGORY_VERDICTS: Dict[str, Edibility] = {
|
|
**{_normalize(c): Edibility.CONSUMABLE for c in _CONSUMABLE_CATEGORIES},
|
|
**{_normalize(c): Edibility.NON_CONSUMABLE for c in _NON_CONSUMABLE_CATEGORIES},
|
|
}
|
|
|
|
# Category strings carrying no information. Treated as MISSING (ask the title),
|
|
# not as UNKNOWN (refuse) - 40 live rows sit in "General" and many are real food.
|
|
# The bare numerics "1".."5" are import damage on 17 rows; repairing those is a
|
|
# catalogue fix, not a consumability rule, so they are recognised and reported
|
|
# rather than accommodated.
|
|
_UNINFORMATIVE_CATEGORIES = frozenset(
|
|
{"", "general", "uncategorized", "uncategorised", "other", "others",
|
|
"misc", "miscellaneous", "unknown", "na", "none"}
|
|
)
|
|
|
|
_JUNK_CATEGORY_RE = re.compile(r"^\d+$")
|
|
|
|
# HSN chapters that are food and drink, per the customs tariff. Deliberately NOT
|
|
# `range(1, 25)` - see the module docstring for why 05, 06, 14, 23 and 24 are out.
|
|
_HSN_FOOD_CHAPTERS = frozenset(
|
|
{1, 2, 3, 4, 7, 8, 9, 10, 11, 12, 13, 15, 16, 17, 18, 19, 20, 21, 22}
|
|
)
|
|
# Cosmetics, soap, pharma, insecticide, razors, hygiene articles.
|
|
_HSN_NON_FOOD_CHAPTERS = frozenset({5, 6, 14, 23, 24, 28, 29, 30, 33, 34, 38, 82, 96})
|
|
|
|
# Title words that settle an uninformative category on their own. Deliberately
|
|
# short: the commodity lexicon and the category detector do the real work, and a
|
|
# longer list here would start overriding them.
|
|
_NON_FOOD_TITLE_WORDS = frozenset({
|
|
"soap", "detergent", "shampoo", "conditioner", "toothpaste", "toothbrush",
|
|
"mouthwash", "deodorant", "perfume", "cosmetic", "lipstick", "kajal",
|
|
"talc", "lotion", "moisturizer", "moisturiser", "sunscreen", "facewash",
|
|
"handwash", "sanitizer", "sanitiser", "diaper", "sanitary", "napkin",
|
|
"razor", "shaving", "cleaner", "disinfectant", "phenyl", "bleach",
|
|
"repellent", "mosquito", "agarbatti", "incense", "camphor", "matchbox",
|
|
"battery", "bulb", "candle", "polish", "freshener", "dishwash",
|
|
})
|
|
|
|
# Title words that positively identify FOOD, consulted before the non-food list.
|
|
# These exist for the mixed categories above: a product line name is the only
|
|
# signal a title like "Nestle Cerelac 125g" carries, since the commodity lexicon
|
|
# knows nothing of it. Naming specific ranges is consistent with how this
|
|
# codebase already resolves ambiguity (BRAND_ALIASES, and the "bikis" keyword
|
|
# added to Biscuits & Cookies so Britannia Milk Bikis is not filed as Dairy).
|
|
_FOOD_TITLE_WORDS = frozenset({
|
|
# infant and toddler nutrition
|
|
"cerelac", "lactogen", "nangrow", "nan", "farex", "dexolac", "nusobee",
|
|
"formula", "infant", "weaning", "porridge", "cereal", "cereals",
|
|
# ayurvedic preparations eaten as food
|
|
"chyawanprash", "chyavanprash", "honey", "malt",
|
|
})
|
|
|
|
# Brand and product-line names sold ONLY as non-food, consulted before the food
|
|
# words above. These exist for the uninformative "General" category: a title like
|
|
# "Colgate-Palmolive Palmolive Naturals 350g" names no article at all, so neither
|
|
# word list nor the lexicon recognises it, the verdict is UNKNOWN - and the write
|
|
# paths refuse only NON_CONSUMABLE, so it was scored. Checked before the food
|
|
# words because Palmolive sells a "Milk & Honey" range and "honey" is food.
|
|
#
|
|
# Deliberately conservative: a name any food range also uses stays out, because
|
|
# a hit here outranks every food word. Left out on purpose: "dove" (Mars
|
|
# chocolate), "himalaya" (supplements), "parachute" (edible coconut oil),
|
|
# "hit", "wheel", "tide" (ordinary words). Only reached when the category says
|
|
# nothing, so a real category always wins.
|
|
_NON_FOOD_LINE_WORDS = frozenset({
|
|
# personal care
|
|
"palmolive", "colgate", "lifebuoy", "lux", "dettol", "savlon", "santoor",
|
|
"cinthol", "pears", "hamam", "medimix", "margo", "fiama", "vivel", "rexona",
|
|
"pantene", "sunsilk", "nivea", "vaseline", "ponds", "closeup", "pepsodent",
|
|
"sensodyne",
|
|
# household
|
|
"ariel", "surf", "rin", "vim", "harpic", "lizol", "odonil", "goodknight",
|
|
"allout", "mortein",
|
|
})
|
|
|
|
_WORD_RE = re.compile(r"[a-z0-9]+")
|
|
|
|
|
|
def _words(text: str) -> frozenset:
|
|
"""Whole words, lowercased. Whole-word matching is the point: the old
|
|
substring gate classified *soapnut* (reetha) as a soap."""
|
|
return frozenset(_WORD_RE.findall((text or "").lower()))
|
|
|
|
|
|
def _hsn_verdict(category: str) -> Optional[Edibility]:
|
|
"""The verdict implied by the HSN chapter this category maps to.
|
|
|
|
Reads `HSN_GST_TABLE` directly rather than calling `resolve_hsn_gst`, and
|
|
that is deliberate. `resolve_hsn_gst` falls through to `_KEYWORD_FALLBACKS`,
|
|
which scans `f"{product_title} {category}"` and contains a greedy
|
|
`("oil", "1517")` entry - enough to classify "Household - Lamp Oil" as an
|
|
edible oil. Reading the table means only an exact category name can match,
|
|
so the fallbacks can never fire here at all.
|
|
|
|
Imported lazily: the hsn_gst package pulls in the enrichment machinery, and
|
|
the nutrition path should not pay for that import on every call.
|
|
"""
|
|
from app.services.enrichment.hsn_gst.models import HSN_GST_TABLE
|
|
|
|
target = _normalize(category)
|
|
for name, entry in HSN_GST_TABLE.items():
|
|
if _normalize(name) != target:
|
|
continue
|
|
try:
|
|
chapter = int(str(entry[0])[:2])
|
|
except (TypeError, ValueError, IndexError):
|
|
return None
|
|
if chapter in _HSN_FOOD_CHAPTERS:
|
|
return Edibility.CONSUMABLE
|
|
if chapter in _HSN_NON_FOOD_CHAPTERS:
|
|
return Edibility.NON_CONSUMABLE
|
|
return None
|
|
return None
|
|
|
|
|
|
def classify_edibility(category: Optional[str], title: str = "") -> EdibilityVerdict:
|
|
"""The full verdict, with the reason that produced it.
|
|
|
|
The reason string is what the purge audit prints, so a row's fate can be
|
|
argued with rather than taken on trust.
|
|
"""
|
|
normalized = _normalize(category)
|
|
informative = (
|
|
normalized
|
|
and normalized not in _UNINFORMATIVE_CATEGORIES
|
|
and normalized not in _MIXED_CATEGORIES
|
|
and not _JUNK_CATEGORY_RE.match(normalized)
|
|
)
|
|
|
|
if informative:
|
|
verdict = CATEGORY_VERDICTS.get(normalized)
|
|
if verdict is not None:
|
|
return EdibilityVerdict(
|
|
verdict,
|
|
f"category {category!r} is listed as {verdict.value}",
|
|
"category_map",
|
|
)
|
|
|
|
hsn = _hsn_verdict(category or "")
|
|
if hsn is not None:
|
|
return EdibilityVerdict(
|
|
hsn,
|
|
f"category {category!r} maps to an HSN chapter that is "
|
|
+ ("food/beverage" if hsn is Edibility.CONSUMABLE else "not food"),
|
|
"hsn_chapter",
|
|
)
|
|
|
|
# The category told us nothing usable. Ask the title, through the same two
|
|
# resolvers the ingestion pipeline uses, so a row classified here agrees
|
|
# with the row the pipeline would have written.
|
|
text = (title or "").strip()
|
|
if text:
|
|
words = _words(text)
|
|
|
|
line_hit = words & _NON_FOOD_LINE_WORDS
|
|
if line_hit:
|
|
return EdibilityVerdict(
|
|
Edibility.NON_CONSUMABLE,
|
|
f"title {title!r} names a non-food brand or line "
|
|
f"({sorted(line_hit)[0]!r})",
|
|
"title_brand_line",
|
|
)
|
|
|
|
food_hit = words & _FOOD_TITLE_WORDS
|
|
if food_hit:
|
|
return EdibilityVerdict(
|
|
Edibility.CONSUMABLE,
|
|
f"title {title!r} names a food product ({sorted(food_hit)[0]!r})",
|
|
"title_keyword",
|
|
)
|
|
|
|
hit = words & _NON_FOOD_TITLE_WORDS
|
|
if hit:
|
|
return EdibilityVerdict(
|
|
Edibility.NON_CONSUMABLE,
|
|
f"title {title!r} names a non-food article ({sorted(hit)[0]!r})",
|
|
"title_keyword",
|
|
)
|
|
|
|
from app.services.generic_products import canonical_category
|
|
|
|
# `detect_category_from_text` is pinned to exact_only. Its fuzzy
|
|
# fallback scores "colgate" at 0.8 against the misspelling keyword
|
|
# "choclate", so without this a tube of toothpaste reads as Chocolates
|
|
# and earns a health score. Verified: that is the live behaviour for
|
|
# "Colgate-Palmolive Palmolive Naturals", whose category is "General".
|
|
for resolver in (
|
|
canonical_category,
|
|
lambda t: detect_category_from_text(t, exact_only=True),
|
|
):
|
|
detected = resolver(text)
|
|
if not detected:
|
|
continue
|
|
verdict = CATEGORY_VERDICTS.get(_normalize(detected))
|
|
if verdict is not None:
|
|
return EdibilityVerdict(
|
|
verdict,
|
|
f"title {title!r} reads as {detected!r}, which is {verdict.value}",
|
|
"title_lexicon",
|
|
)
|
|
|
|
return EdibilityVerdict(
|
|
Edibility.UNKNOWN,
|
|
f"neither category {category!r} nor title {title!r} identifies this "
|
|
"product; refusing to guess",
|
|
"none",
|
|
)
|
|
|
|
|
|
def is_consumable(category: Optional[str], title: str = "") -> bool:
|
|
"""True only when this is positively something a person eats or drinks.
|
|
|
|
The WRITE gate. False for UNKNOWN, so an unrecognised product gets no
|
|
nutrition row and no health score - the same asymmetry
|
|
`nutrition_data_service` already chose when it returns
|
|
`data_status="unavailable"` rather than zeroes.
|
|
"""
|
|
return classify_edibility(category, title).edibility is Edibility.CONSUMABLE
|
|
|
|
|
|
def is_non_consumable(category: Optional[str], title: str = "") -> bool:
|
|
"""True only when this is positively NOT food.
|
|
|
|
The DELETE gate, and deliberately not `not is_consumable(...)`. False for
|
|
UNKNOWN, so the purge script can never destroy a row it merely failed to
|
|
recognise. See the module docstring.
|
|
"""
|
|
return classify_edibility(category, title).edibility is Edibility.NON_CONSUMABLE
|
|
|
|
|
|
def is_junk_category(category: Optional[str]) -> bool:
|
|
"""True for the bare numeric category strings left by a bad import.
|
|
|
|
Reported by the purge audit so the 17 affected rows get repaired in the
|
|
catalogue rather than worked around here.
|
|
"""
|
|
return bool(_JUNK_CATEGORY_RE.match(_normalize(category)))
|