New updates on DB and JSON

This commit is contained in:
sriram
2026-09-01 13:55:15 +05:30
parent 183b65b3bd
commit 6c7a886659
20 changed files with 68753 additions and 135 deletions

View File

@@ -252,6 +252,15 @@ BRAND_ALIASES = {
"sunfeast bounce": "sunfeast",
"sunfeast yippee": "sunfeast",
"sunfeast cookies": "sunfeast",
# SPELLING VARIANTS, not sub-brands.
#
# "Haldiram" and "Haldirams" were resolving to themselves, so nothing knew
# they were one brand and uploads built brand_haldiram and brand_haldirams
# side by side. The data was merged into the plural, which is the correct
# name; THIS LINE is what stops the split reappearing on the next sheet
# that spells it without the s. Removing it re-opens the bug.
"haldiram": "haldirams",
"haldiram's": "haldirams",
}
DEFAULT_ALIASES = BRAND_ALIASES
@@ -352,7 +361,14 @@ FSSAI_LICENSES: Dict[str, str] = {
"lion dates": "10012042000244",
"brooke bond": "10013022001897",
"mother dairy": "10012011000015",
# Keyed on BOTH spellings deliberately. get_fssai_license() resolves to the
# canonical parent first, so once "haldiram" aliases to "haldirams" the
# lookup arrives as "haldirams" - and with only the singular key here it
# would return None and every future Haldiram row would ship with no
# licence at all. A silent loss, since a blank licence is a legitimate
# outcome elsewhere and nothing would flag it.
"haldiram": "10012011000140",
"haldirams": "10012011000140",
"fortune": "10012021000071",
"paper boat": "10012043000083",
"bisk farm": "10012031000012",

View File

@@ -88,6 +88,22 @@ CATEGORY_REGISTRY: List[Dict[str, object]] = [
{"category": "Salt & Staples", "keywords": ["salt", "rock salt", "sea salt", "table salt", "iodised salt", "iodized salt"], "generic_term": "salt"},
{"category": "Atta & Staples", "keywords": ["atta", "wheat flour", "flour", "rice", "dal", "pulses", "staples", "suji", "maida"], "generic_term": "staple product"},
{"category": "Dairy", "keywords": ["milk", "dairy", "cheese", "paneer", "panner", "paner", "paneerr", "curd", "yogurt", "butter", "ghee", "dahi"], "generic_term": "dairy product"},
# ---- Loose, unbranded fresh goods -----------------------------------
# Added for the produce a grocer sells by weight or by the piece. These
# rows carry no brand, so they land in the Own Products table via
# generic_products.is_unbranded(); the categories exist so that HSN/GST
# resolution and the pack-size unit rules have something to key off,
# rather than falling through to "General".
#
# Keyword lists stay SHORT here on purpose. Detection for these rows comes
# from the commodity lexicon, which is far more specific; a broad keyword
# such as "fresh" or "leaf" would pull branded products in through
# keyword matching, which is the failure this whole area exists to avoid.
{"category": "Fruits & Vegetables", "keywords": ["fruits", "vegetables", "vegetable", "fresh produce", "loose produce"], "generic_term": "fresh produce"},
{"category": "Fresh Herbs & Greens", "keywords": ["herbs", "greens", "curry leaves", "coriander leaves", "mint leaves", "spinach", "keerai"], "generic_term": "fresh greens"},
{"category": "Flowers", "keywords": ["flowers", "flower", "garland", "jasmine flower", "loose flowers"], "generic_term": "flowers"},
{"category": "Fish & Seafood", "keywords": ["fish", "seafood", "prawns", "prawn", "shrimp"], "generic_term": "fresh fish"},
{"category": "Eggs", "keywords": ["eggs", "egg", "country egg"], "generic_term": "eggs"},
{"category": "Oral Care", "keywords": ["toothpaste", "toothbrush", "mouthwash", "paste"], "generic_term": "oral care product"},
{"category": "Hair Care", "keywords": ["shampoo", "shampooo", "conditioner", "hair oil"], "generic_term": "hair care product"},
{"category": "Bath Soap", "keywords": ["bath soap", "soap bar", "soap", "soaps"], "generic_term": "soap"},

View File

@@ -70,6 +70,17 @@ HSN_GST_TABLE: Dict[str, Tuple[str, int, bool]] = {
# Chapter 17: cane/beet sugar and jaggery, 5% for ordinary retail sugar.
"Sugar & Jaggery": ("1701", 5, False),
"Cooking Oils": ("1517", 5, False),
# ---- Loose fresh goods, sold by weight or by the piece ---------------
# Chapters 3, 4, 6, 7 and 8. Fresh, unprocessed produce is NIL-rated under
# GST - it is not a reduced rate, it is exempt - so 0 here is the real
# figure and not a placeholder. The moment any of these is branded and
# packaged in a unit container the rate changes, but that product would
# resolve to its brand's category rather than these.
"Fruits & Vegetables": ("0709", 0, False),
"Fresh Herbs & Greens": ("0709", 0, False),
"Flowers": ("0603", 0, False),
"Fish & Seafood": ("0302", 0, False),
"Eggs": ("0407", 0, False),
"Pickles & Chutneys": ("2001", 12, True),
"Dry Fruits & Nuts": ("0801", 12, True),
"Food - Spreads": ("2007", 12, True),

View File

@@ -74,6 +74,16 @@ _SALT = "Salt & Staples"
_OILS = "Cooking Oils"
_DAIRY = "Dairy"
_BEVERAGE = "Beverages"
# Fresh, loose goods. These are sold by weight or by the piece and carry no
# brand at all, which is precisely why they were the worst offenders: before
# these existed, "Apple" resolved to a brand called Apple and "Red Rose" was
# whole-word matched into brand_brooke_bond, carrying Brooke Bond's FSSAI
# licence onto a flower.
_PRODUCE = "Fruits & Vegetables"
_GREENS = "Fresh Herbs & Greens"
_FLOWERS = "Flowers"
_SEAFOOD = "Fish & Seafood"
_EGGS = "Eggs"
COMMODITY_TERMS: Dict[str, str] = {}
@@ -127,6 +137,79 @@ _add(_DAIRY,
_add(_BEVERAGE, "tea", "coffee", "chai")
# ---------------------------------------------------------------------------
# Fresh produce
# ---------------------------------------------------------------------------
# WHY THIS BLOCK IS SAFE TO ADD.
#
# The all-tokens-must-be-commodities rule means every word added here makes the
# test MORE permissive, so the risk is real brands collapsing into this bucket.
# That was measured, not assumed, before these went in: all 1,414 products in
# the live catalogue were reclassified with this list applied, and exactly one
# changed - `24 Mantra Organic Moong Dal 500g`, which was ALREADY misfiled by
# the `_strip_sizes` bug fixed below and has nothing to do with produce.
#
# The other check was against BRAND_ALIASES: of 113 candidate terms only "amla"
# appears in an alias ("dabur amla"), and that is harmless because "dabur" is
# not a commodity, so "Dabur Amla" keeps every token it needs to stay branded.
# Bare "Amla" is fruit and belongs here.
#
# RE-RUN BOTH CHECKS BEFORE ADDING A WORD TO THIS BLOCK. See
# tests/test_generic_products_produce.py, which encodes them.
# Fruit. Indian sheets mix English, Tamil and Hindi names freely.
_add(_PRODUCE,
"apple", "orange", "banana", "grape", "grapes", "mango", "pineapple",
"papaya", "guava", "pomegranate", "watermelon", "muskmelon", "melon",
"lemon", "lime", "mosambi", "sathukudi", "sapota", "chikoo", "jackfruit",
"fig", "anjeer", "pear", "peach", "plum", "apricot", "cherry", "cherries",
"strawberry", "blueberry", "kiwi", "litchi", "lychee", "amla", "gooseberry",
"avocado", "dragonfruit", "rambutan", "mangosteen", "starfruit", "jamun",
"ber", "plantain", "vazhaikkai")
# Vegetables. "gourd" covers the whole family once _PHRASES has collapsed the
# two-word forms (bitter gourd, bottle gourd, snake gourd, ridge gourd).
_add(_PRODUCE,
"tomato", "potato", "onion", "carrot", "beetroot", "beet", "radish",
"turnip", "cabbage", "cauliflower", "broccoli", "brinjal", "eggplant",
"aubergine", "okra", "bhindi", "cucumber", "pumpkin", "gourd", "drumstick",
"beans", "bean", "capsicum", "garlic", "ginger", "yam", "colocasia",
"tapioca", "arbi", "zucchini", "chayote", "sweetcorn", "babycorn",
"mushroom", "leek", "celery", "lettuce", "shallot", "springonion")
# Greens and fresh herbs, sold in bunches and never branded. NOTE that
# "coriander" and "methi" are deliberately NOT here: they are already spices in
# the block above, `_add` is last-wins, and re-adding them would silently move
# dhania powder out of Spices & Masalas. The leaf forms are collapsed to
# distinct tokens by _PHRASES instead.
_add(_GREENS,
"spinach", "palak", "amaranth", "keerai", "greens", "cilantro", "mint",
"pudina", "curryleaves", "basil", "thulasi", "tulsi", "parsley", "dill",
"sorrel", "moringa", "methileaves", "bunch")
# Flowers. Sold loose or by the metre of garland; the reason "Red Rose" used to
# land in a tea catalogue.
_add(_FLOWERS,
"flower", "flowers", "rose", "jasmine", "malli", "lotus", "marigold",
"samanthi", "chrysanthemum", "kanakambaram", "arali", "garland", "poo")
# Fish and seafood, sold fresh by weight.
_add(_SEAFOOD,
"fish", "prawn", "prawns", "shrimp", "crab", "squid", "tuna", "mackerel",
"sardine", "pomfret", "seer", "vanjaram", "anchovy", "nethili", "sole",
"tilapia", "salmon", "shellfish", "clam", "mussel")
# Eggs.
_add(_EGGS, "egg", "eggs", "muttai", "quail")
# Stragglers found by running the produce base list through is_unbranded and
# fixing every row it refused. Kept in one block so the next person adding to
# the seed list knows where the tail ends up.
_add(_PRODUCE, "dates", "custard", "dragonfruit", "ivy", "raw")
_add(_GREENS, "agathi", "ponnanganni", "keerai")
_add(_FLOWERS, "tuberose", "lily")
# Words that describe a product without naming a brand. Stripped before the
# all-tokens-are-commodities test, so "Organic Toor Dal Whole 1kg" still reads
# as unbranded.
@@ -150,6 +233,16 @@ QUALIFIERS: Set[str] = {
"bottle", "refill", "combo", "assorted", "mixed", "mix",
# connectives
"and", "with", "of", "the", "in", "for",
# Form words for fresh goods. "Leaves" is the important one: without it
# "Mint Leaves" keeps an unknown token and reads as a brand.
"leaves", "leaf", "bunch", "sweet", "broad", "cluster", "full", "toned",
"seedless", "ripe", "tender", "baby", "country", "hybrid", "nati",
# Varietal names. A variety qualifies a commodity, it does not brand it:
# an Alphonso mango is a mango. None of these appears in BRAND_ALIASES -
# the produce test asserts that, so a future addition cannot smuggle a
# real brand in through this list.
"robusta", "yelakki", "nendran", "alphonso", "banganapalli", "totapuri",
"malgova", "sindoora", "shimla", "ooty", "kashmiri",
}
# Multi-word commodities collapsed to a single token before tokenising, so the
@@ -175,16 +268,70 @@ _PHRASES = {
"brown sugar": "sugar",
"palm jaggery": "jaggery",
"cane sugar": "sugar",
# Fresh produce. The two-word gourds collapse onto "gourd" so the whole
# family is one lexicon entry. The leaf forms get their OWN tokens rather
# than reusing "coriander" / "methi": those are spices, _add is last-wins,
# and re-adding them under a greens category would silently move dhania
# powder out of Spices & Masalas.
"bitter gourd": "gourd",
"bottle gourd": "gourd",
"snake gourd": "gourd",
"ridge gourd": "gourd",
"ash gourd": "gourd",
"bitter guard": "gourd", # misspellings seen in real merchant data
"bottle ground": "gourd",
"lady finger": "okra",
"ladies finger": "okra",
"spring onion": "springonion",
"spring onions": "springonion",
"sweet potato": "potato",
"curry leaves": "curryleaves",
"curry leaf": "curryleaves",
"coriander leaves": "cilantro",
"methi leaves": "methileaves",
"fenugreek leaves": "methileaves",
"french beans": "beans",
"cluster beans": "beans",
"green peas": "peas",
"baby corn": "babycorn",
"sweet corn": "sweetcorn",
"tender coconut": "coconut",
"dragon fruit": "dragonfruit",
"custard apple": "apple",
"sweet lime": "mosambi",
"ivy gourd": "gourd",
"broad beans": "beans",
"cluster bean": "beans",
"quail egg": "egg",
"spring garlic": "garlic",
}
_WORD_RE = re.compile(r"[a-z]+")
# Units a pack size is actually written in. The strip below is bounded to these
# rather than to "any letters", because [a-z]* after a number ate the NEXT WORD:
# "24 Mantra Organic Moong Dal" became "organic moong dal", the brand was
# destroyed, and the row was then filed as an unbranded commodity. Every brand
# whose name begins with a number hit this. Keep the list tight - a unit added
# here is a word that can be deleted from a product name.
_UNITS = (
"kg|kgs|g|gm|gms|gram|grams|mg|ml|l|ltr|ltrs|litre|litres|liter|liters"
"|pc|pcs|piece|pieces|pack|packs|pkt|n|no|nos|x|cm|mm|inch|dozen"
)
_SIZE_RE = re.compile(
# "1kg", "500 g", "1.5 L" - a number followed by a REAL unit, optionally
# spaced - or a bare number, which is a quantity and never a brand.
r"\b\d+(?:[.,]\d+)?\s*(?:" + _UNITS + r")\b"
r"|\b\d+(?:[.,]\d+)?\b",
re.IGNORECASE,
)
def _strip_sizes(text: str) -> str:
"""Remove pack sizes and bare numbers - they never name a brand."""
# "1kg", "500 g", "1.5 L", and any leftover bare number.
text = re.sub(r"\b\d+(?:[.,]\d+)?\s*[a-z]*\b", " ", text)
return text
return _SIZE_RE.sub(" ", text)
def canonical_category(name: str) -> Optional[str]:

View File

@@ -54,6 +54,7 @@ from app.infrastructure.settings import (
from app.services.title_validator import find_category_conflicts
from app.services import price_estimator
from app.services import category_units as cu
from app.services.generic_products import OWN_PRODUCTS_BRAND
logger = logging.getLogger(__name__)
@@ -277,6 +278,26 @@ def validate_product(
sku_source = str(product.get("sku_source") or "").strip()
images = product.get("image_urls") or []
# A COMMODITY IS NOT A DEFECTIVE BRANDED PRODUCT.
#
# The penalties below treat a missing price_range or SKU as evidence that a
# row was fabricated, which is right for a scraped brand catalogue: a real
# Amul product has a shelf price and an article number, so their absence
# means something went wrong. Loose produce has neither, by nature. A shop
# prices apples by the day and does not issue article numbers for them.
#
# Left unqualified, the arithmetic rejected every produce row outright:
# 0.55 baseline - 0.30 (no price_range) - 0.10 (no SKU) = 0.15, against a
# reject threshold of 0.35. That is the exact opposite of the requirement
# these rows exist to satisfy, so the two absences stop counting as faults.
#
# EVERYTHING ELSE STILL APPLIES. Title sanity, placeholder detection,
# category resolution, the title/category contradiction check, size
# validity and unit compatibility, and the image-presence check all run
# unchanged - a blank or junk product name is still caught, and this is not
# a way in for rows that would otherwise fail.
commodity = brand == OWN_PRODUCTS_BRAND
report = ValidationReport(product_name=title or "(untitled)", grounded=grounded)
score = 0.55 # neutral baseline - moves up/down based on evidence below
@@ -345,16 +366,22 @@ def validate_product(
score -= 0.15
# 5. Price range ----------------------------------------------------------
ok, msg = validate_price_range(price_range, size, title, brand, category)
if not ok:
report.issues.append(ValidationIssue(field="price_range", severity="error", message=msg, penalty=0.30))
score -= 0.30
# Skipped for a commodity only when there is none. A band that IS present is
# still checked for being well formed, so a malformed one cannot hide here.
if price_range or not commodity:
ok, msg = validate_price_range(price_range, size, title, brand, category)
if not ok:
report.issues.append(ValidationIssue(field="price_range", severity="error", message=msg, penalty=0.30))
score -= 0.30
# 6. SKU --------------------------------------------------------------------
ok, msg = validate_sku(sku, sku_source)
if not ok:
report.issues.append(ValidationIssue(field="product_sku", severity="warning", message=msg, penalty=0.10))
score -= 0.10
# Same rule: a commodity is not expected to carry one, but a SKU the sheet
# did supply must still look like a SKU.
if sku or not commodity:
ok, msg = validate_sku(sku, sku_source)
if not ok:
report.issues.append(ValidationIssue(field="product_sku", severity="warning", message=msg, penalty=0.10))
score -= 0.10
# 7. Image presence -----------------------------------------------------
# Only meaningful if image search actually ran. When the operator disables