backend store_catalog updates
This commit is contained in:
@@ -80,6 +80,7 @@ from app.services.category_registry import (
|
||||
sanitize_category_language,
|
||||
)
|
||||
from app.services.category_units import fix_or_reject_size, parse_unit
|
||||
from app.services.generic_products import OWN_PRODUCTS_BRAND, is_unbranded
|
||||
from app.services.embeddings_service import embed_texts
|
||||
from app.services.enrichment.barcode.stage import BarcodeEnrichmentStage
|
||||
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
|
||||
@@ -236,6 +237,20 @@ def infer_brand(product_name: str) -> Optional[str]:
|
||||
|
||||
def stage_1_brand_and_fssai(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
brand = row.get("brand") or ""
|
||||
# The own-products bucket is not a brand and must never be resolved like
|
||||
# one. resolve_parent_brand is identity for it today, but it matches
|
||||
# bidirectionally against every alias, so a future alias containing "own"
|
||||
# or "products" would silently redirect the whole table. Skipping is the
|
||||
# only version of this that stays true.
|
||||
#
|
||||
# It also keeps get_fssai_license out of the picture: a commodity has no
|
||||
# licence, and the bug being fixed here was precisely a real third-party
|
||||
# licence number being stamped onto unbranded rows.
|
||||
if brand == OWN_PRODUCTS_BRAND:
|
||||
row["brand"] = OWN_PRODUCTS_BRAND
|
||||
row["brand_name"] = OWN_PRODUCTS_BRAND
|
||||
row["_table"] = f"brand_{_sanitize_name(OWN_PRODUCTS_BRAND)}"
|
||||
return row
|
||||
parent = resolve_parent_brand(brand)
|
||||
row["brand"] = parent
|
||||
row["brand_name"] = parent
|
||||
@@ -433,12 +448,21 @@ def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, An
|
||||
from app.core.catalog_engine import catalog_engine
|
||||
from app.services.image_search import find_all_image_urls
|
||||
|
||||
# "Own Products" is a bucket, not a brand, so it must not enter the
|
||||
# search query or the relevance scoring - searching for "Own Products
|
||||
# Toor Dal" finds nothing. With it blank the query is just the product
|
||||
# name, and _select_best_images derives its distinctive tokens from the
|
||||
# whole name ({toor, dal}), which is exactly right for a commodity.
|
||||
brand = row.get("brand") or ""
|
||||
if brand == OWN_PRODUCTS_BRAND:
|
||||
brand = ""
|
||||
|
||||
candidates = find_all_image_urls(
|
||||
row.get("product_name") or "", brand=row.get("brand"), max_results=24
|
||||
row.get("product_name") or "", brand=brand or None, max_results=24
|
||||
)
|
||||
if candidates:
|
||||
best = catalog_engine._select_best_images(
|
||||
candidates, row.get("product_name") or "", row.get("brand") or "", max_images=10
|
||||
candidates, row.get("product_name") or "", brand, max_images=10
|
||||
)
|
||||
if best:
|
||||
row["image_urls"] = list(best)
|
||||
@@ -722,7 +746,23 @@ def run_pipeline(
|
||||
|
||||
raw_name = _text(record, mapping, "product_name") or _text(record, mapping, "title") or ""
|
||||
if not brand_column_supplied:
|
||||
record[_INFERRED_BRAND_COL] = infer_brand(raw_name) or ""
|
||||
# Commodity rows are diverted BEFORE infer_brand rather than after,
|
||||
# because both of the steps that follow damage them. infer_brand
|
||||
# falls back to the first word ("Sugar 1kg" -> brand "Sugar"), and
|
||||
# resolve_parent_brand then whole-word-matches that against ~230
|
||||
# aliases, which is how "Salt" reached colgate-palmolive and "Milk"
|
||||
# reached cadbury. Neither runs for these rows now.
|
||||
if is_unbranded(raw_name):
|
||||
record[_INFERRED_BRAND_COL] = OWN_PRODUCTS_BRAND
|
||||
else:
|
||||
record[_INFERRED_BRAND_COL] = infer_brand(raw_name) or ""
|
||||
elif _blank(_text(record, mapping, "brand")) and is_unbranded(
|
||||
raw_name, brand_column_supplied=True
|
||||
):
|
||||
# The sheet has a brand column and left this cell empty. That is an
|
||||
# explicit statement, and previously it was the one case that got
|
||||
# the row rejected outright as "no brand name in this row".
|
||||
record[mapping.columns["brand"]] = OWN_PRODUCTS_BRAND
|
||||
|
||||
try:
|
||||
req = row_to_request(record, mapping)
|
||||
|
||||
@@ -75,6 +75,17 @@ CATEGORY_REGISTRY: List[Dict[str, object]] = [
|
||||
# and grounds far more often than as a drink).
|
||||
{"category": "Beverages", "keywords": ["beverages", "beverage", "soft drink", "soft drinks", "cold drink", "cold drinks", "carbonated", "aerated drink", "cola", "coke", "juice", "juices", "squash", "sharbat", "energy drink", "sports drink", "mineral water", "packaged drinking water", "lemonade", "iced tea", "thums up", "sprite", "fanta", "limca", "maaza", "pepsi", "mirinda"], "generic_term": "beverage"},
|
||||
{"category": "Cooking Oils", "keywords": ["cooking oil", "edible oil", "sunflower oil", "mustard oil", "vanaspati", "refined oil", "oil", "oils"], "generic_term": "cooking oil"},
|
||||
# Loose-commodity categories. The names are chosen to match entries that
|
||||
# already exist in category_units.CATEGORY_UNIT_TYPE and in
|
||||
# enrichment/hsn_gst/models.HSN_GST_TABLE, so a bag of dal picks up its unit
|
||||
# rulebook and its HSN code without either table needing a new key. Listed
|
||||
# before "Cooking Oils", whose bare "oil" keyword is greedy, and before
|
||||
# "Atta & Staples", whose "dal"/"pulses"/"rice" keywords would otherwise
|
||||
# swallow every pulse.
|
||||
{"category": "Pulses, Grains & Spices", "keywords": ["dal", "dhal", "daal", "toor dal", "urad dal", "moong dal", "masoor dal", "chana dal", "arhar", "lentils", "lentil", "rajma", "pulses", "pulse"], "generic_term": "pulse"},
|
||||
{"category": "Spices & Masalas", "keywords": ["spices", "spice", "masala", "masalas", "turmeric", "haldi", "chilli powder", "coriander powder", "cumin", "jeera", "peppercorn", "black pepper", "cardamom", "asafoetida", "hing", "tamarind"], "generic_term": "spice"},
|
||||
{"category": "Sugar & Jaggery", "keywords": ["sugar", "jaggery", "gur", "brown sugar", "cane sugar", "misri"], "generic_term": "sweetener"},
|
||||
{"category": "Salt & Staples", "keywords": ["salt", "rock salt", "sea salt", "table salt", "iodised salt", "iodized salt"], "generic_term": "salt"},
|
||||
{"category": "Atta & Staples", "keywords": ["atta", "wheat flour", "flour", "rice", "dal", "pulses", "staples", "suji", "maida"], "generic_term": "staple product"},
|
||||
{"category": "Dairy", "keywords": ["milk", "dairy", "cheese", "paneer", "panner", "paner", "paneerr", "curd", "yogurt", "butter", "ghee", "dahi"], "generic_term": "dairy product"},
|
||||
{"category": "Oral Care", "keywords": ["toothpaste", "toothbrush", "mouthwash", "paste"], "generic_term": "oral care product"},
|
||||
|
||||
@@ -108,6 +108,9 @@ CATEGORY_UNIT_TYPE: dict[str, str] = {
|
||||
"candy & confectionery": "weight",
|
||||
"atta & staples": "weight",
|
||||
"spices & masalas": "weight",
|
||||
# Loose commodities, sold by weight in every case.
|
||||
"pulses, grains & spices": "weight",
|
||||
"sugar & jaggery": "weight",
|
||||
"pasta & noodles": "weight",
|
||||
"noodles & instant food": "weight",
|
||||
"breakfast cereal": "weight",
|
||||
|
||||
@@ -67,6 +67,8 @@ HSN_GST_TABLE: Dict[str, Tuple[str, int, bool]] = {
|
||||
"Salt & Staples": ("2501", 5, False),
|
||||
"Pulses, Grains & Spices": ("0713", 5, True),
|
||||
"Spices & Masalas": ("0910", 5, False),
|
||||
# Chapter 17: cane/beet sugar and jaggery, 5% for ordinary retail sugar.
|
||||
"Sugar & Jaggery": ("1701", 5, False),
|
||||
"Cooking Oils": ("1517", 5, False),
|
||||
"Pickles & Chutneys": ("2001", 12, True),
|
||||
"Dry Fruits & Nuts": ("0801", 12, True),
|
||||
|
||||
237
app/services/generic_products.py
Normal file
237
app/services/generic_products.py
Normal file
@@ -0,0 +1,237 @@
|
||||
"""
|
||||
Unbranded grocery commodities, and the one table they belong in.
|
||||
|
||||
WHY THIS FILE EXISTS
|
||||
--------------------
|
||||
A store sheet routinely lists items that simply have no brand: "Toor Dhal 1kg",
|
||||
"Sugar", "Salt 1kg", "Black Pepper 100g". Nothing rejected those rows - what
|
||||
happened was worse. `store_catalog_pipeline.infer_brand()` falls back to the
|
||||
first word of the name, and `brand_registry.resolve_parent_brand()` then does a
|
||||
bidirectional whole-word scan over ~230 aliases. Between them:
|
||||
|
||||
"Toor Dhal 1kg" -> brand "Toor" -> a junk table brand_toor
|
||||
"Sugar 1kg" -> brand "Sugar" -> a junk table brand_sugar
|
||||
"Salt 1kg" -> "colgate active salt" contains "salt"
|
||||
-> brand_colgate_palmolive
|
||||
"Milk 1L" -> "cadbury dairy milk" contains "milk"
|
||||
-> brand_cadbury
|
||||
"Butter 500g" -> "nestle butter" -> brand_nestle
|
||||
"Red Chilli Powder" -> "brooke bond red label" -> brand_brooke_bond
|
||||
|
||||
The junk tables are noise. The misroutes are real damage: generic groceries
|
||||
written into the live Amul, Nestle, Cadbury and HUL catalogs - and
|
||||
`stage_1_brand_and_fssai` then stamps that brand's FSSAI licence number onto the
|
||||
row, so unbranded chilli powder ships carrying Brooke Bond's real licence.
|
||||
|
||||
WHY A WORD LIST AND NOT "THE BRAND IS UNKNOWN"
|
||||
----------------------------------------------
|
||||
"Not in BRAND_ALIASES" is the obvious rule and it is wrong here. That map holds
|
||||
231 mostly-large FMCG names; Bikaji, Aachi, Idhayam, Naga and Lion Dates are all
|
||||
real brands absent from it. Treating unknown as unbranded would sweep every
|
||||
regional brand into one bucket.
|
||||
|
||||
So the test is POSITIVE and conservative: strip the pack size and the words that
|
||||
carry no brand signal, and require that *everything still standing* is a
|
||||
commodity noun. One unrecognised token means "this is a brand". Hence:
|
||||
|
||||
"Butter 500g" -> {butter} -> unbranded
|
||||
"Amul Butter 500g" -> {amul, butter} -> branded, unchanged
|
||||
"Aachi Sambar" -> {aachi, sambar} -> branded, unchanged
|
||||
|
||||
The lexicon doubles as a category map, so the same entry that identifies a
|
||||
commodity also says which canonical category it belongs to - a bag of dal should
|
||||
not have to go through keyword detection twice.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Dict, Optional, Set
|
||||
|
||||
from app.services.category_units import parse_unit
|
||||
|
||||
# The single brand every unbranded product is filed under. `_sanitize_name`
|
||||
# turns this into the table `brand_own_products`, and `display_name_for_suffix`
|
||||
# turns that back into "Own Products" for the UI, so the round trip the frontend
|
||||
# depends on (card -> /api/brands/{brand}/products) is an identity.
|
||||
#
|
||||
# Deliberately NOT registered in BRAND_ALIASES: an alias overlapping these words
|
||||
# would let resolve_parent_brand hijack the table, which is the exact class of
|
||||
# bug this module exists to end.
|
||||
OWN_PRODUCTS_BRAND = "Own Products"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The lexicon: commodity term -> canonical category
|
||||
# ---------------------------------------------------------------------------
|
||||
# Categories are the canonical names from category_registry.ALL_CATEGORIES, so
|
||||
# HSN/GST resolution and the pack-size unit rulebook both key off them without
|
||||
# a second translation step.
|
||||
_PULSES = "Pulses, Grains & Spices"
|
||||
_STAPLES = "Atta & Staples"
|
||||
_SPICES = "Spices & Masalas"
|
||||
_SUGAR = "Sugar & Jaggery"
|
||||
_SALT = "Salt & Staples"
|
||||
_OILS = "Cooking Oils"
|
||||
_DAIRY = "Dairy"
|
||||
_BEVERAGE = "Beverages"
|
||||
|
||||
COMMODITY_TERMS: Dict[str, str] = {}
|
||||
|
||||
|
||||
def _add(category: str, *terms: str) -> None:
|
||||
for term in terms:
|
||||
COMMODITY_TERMS[term] = category
|
||||
|
||||
|
||||
# Pulses and lentils. Indian sheets spell dal a dozen ways.
|
||||
_add(_PULSES,
|
||||
"dal", "dhal", "dhall", "daal", "dail", "pulse", "pulses", "lentil", "lentils",
|
||||
"toor", "tur", "arhar", "urad", "urid", "moong", "mung", "masoor", "masur",
|
||||
"chana", "channa", "gram", "rajma", "lobia", "kabuli", "peas", "matar",
|
||||
"soya", "soyabean")
|
||||
|
||||
# Grains, flours and other dry staples.
|
||||
_add(_STAPLES,
|
||||
"rice", "basmati", "sona", "masoori", "ponni", "idli", "sona masoori",
|
||||
"wheat", "atta", "maida", "sooji", "suji", "rava", "semolina", "besan",
|
||||
"poha", "aval", "ragi", "bajra", "jowar", "millet", "millets", "quinoa",
|
||||
"sabudana", "vermicelli", "corn", "oats", "flour")
|
||||
|
||||
# Sweeteners.
|
||||
_add(_SUGAR, "sugar", "jaggery", "gur", "misri", "honey", "sakkarai")
|
||||
|
||||
# Salt.
|
||||
_add(_SALT, "salt", "sendha", "iodised", "iodized")
|
||||
|
||||
# Spices, whole and ground.
|
||||
_add(_SPICES,
|
||||
"pepper", "peppercorn", "peppercorns", "turmeric", "haldi", "manjal",
|
||||
"chilli", "chili", "chillies", "chile", "mirchi", "coriander", "dhania",
|
||||
"cumin", "jeera", "mustard", "methi", "fenugreek", "cardamom", "elaichi",
|
||||
"clove", "cloves", "lavang", "cinnamon", "dalchini", "bay", "tejpatta",
|
||||
"asafoetida", "hing", "tamarind", "imli", "masala", "garam", "sambar",
|
||||
"rasam", "ajwain", "saunf", "fennel", "nutmeg", "mace", "star", "anise",
|
||||
"kalonji", "poppy", "khus")
|
||||
|
||||
# Edible oils. Bare "oil" currently resolves to Johnson & Johnson.
|
||||
_add(_OILS,
|
||||
"oil", "gingelly", "groundnut", "peanut", "sunflower", "sesame", "til",
|
||||
"coconut", "castor", "vanaspati")
|
||||
|
||||
# Loose dairy. These are the ones misrouting into real brand tables today.
|
||||
_add(_DAIRY,
|
||||
"milk", "butter", "ghee", "paneer", "curd", "dahi", "yogurt", "yoghurt",
|
||||
"cheese", "khoa", "khoya", "cream", "buttermilk", "lassi")
|
||||
|
||||
# Loose tea and coffee.
|
||||
_add(_BEVERAGE, "tea", "coffee", "chai")
|
||||
|
||||
|
||||
# Words that describe a product without naming a brand. Stripped before the
|
||||
# all-tokens-are-commodities test, so "Organic Toor Dal Whole 1kg" still reads
|
||||
# as unbranded.
|
||||
QUALIFIERS: Set[str] = {
|
||||
# quality / provenance
|
||||
"organic", "premium", "fresh", "natural", "pure", "best", "quality",
|
||||
"grade", "select", "special", "classic", "regular", "standard", "economy",
|
||||
"value", "farm", "country", "desi", "local", "homemade", "traditional",
|
||||
# processing / form
|
||||
"whole", "half", "split", "raw", "roasted", "unroasted", "polished",
|
||||
"unpolished", "sortex", "cleaned", "washed", "refined", "filtered",
|
||||
"double", "single", "extra", "fine", "coarse", "powder", "powdered",
|
||||
"ground", "crushed", "flakes", "seeds", "seed", "granules", "crystal",
|
||||
"crystals", "cube", "cubes", "stick", "sticks", "dried", "dry",
|
||||
"slice", "slices", "sliced", "block", "grated", "shredded", "chopped",
|
||||
# colour / variety, which qualify a commodity rather than brand it
|
||||
"black", "white", "red", "green", "yellow", "brown", "long", "short",
|
||||
"small", "big", "large", "medium",
|
||||
# packaging / retail noise
|
||||
"pack", "packet", "packed", "loose", "bag", "pouch", "box", "tin", "jar",
|
||||
"bottle", "refill", "combo", "assorted", "mixed", "mix",
|
||||
# connectives
|
||||
"and", "with", "of", "the", "in", "for",
|
||||
}
|
||||
|
||||
# Multi-word commodities collapsed to a single token before tokenising, so the
|
||||
# individual words do not have to stand alone in the lexicon.
|
||||
_PHRASES = {
|
||||
"rock salt": "salt",
|
||||
"sea salt": "salt",
|
||||
"table salt": "salt",
|
||||
"black pepper": "pepper",
|
||||
"white pepper": "pepper",
|
||||
"bengal gram": "chana",
|
||||
"green gram": "moong",
|
||||
"black gram": "urad",
|
||||
"horse gram": "chana",
|
||||
"red chilli": "chilli",
|
||||
"bay leaf": "bay",
|
||||
"star anise": "anise",
|
||||
"sona masoori": "rice",
|
||||
"wheat flour": "atta",
|
||||
"gram flour": "besan",
|
||||
"corn flour": "flour",
|
||||
"rice flour": "flour",
|
||||
"brown sugar": "sugar",
|
||||
"palm jaggery": "jaggery",
|
||||
"cane sugar": "sugar",
|
||||
}
|
||||
|
||||
_WORD_RE = re.compile(r"[a-z]+")
|
||||
|
||||
|
||||
def _strip_sizes(text: str) -> str:
|
||||
"""Remove pack sizes and bare numbers - they never name a brand."""
|
||||
# "1kg", "500 g", "1.5 L", and any leftover bare number.
|
||||
text = re.sub(r"\b\d+(?:[.,]\d+)?\s*[a-z]*\b", " ", text)
|
||||
return text
|
||||
|
||||
|
||||
def canonical_category(name: str) -> Optional[str]:
|
||||
"""The category implied by the commodity words in `name`, if any.
|
||||
|
||||
Returns the category of the FIRST commodity term found, scanning left to
|
||||
right, because an Indian product name leads with its head noun ("Toor Dhal",
|
||||
"Sugar", "Groundnut Oil").
|
||||
"""
|
||||
for token in _tokens(name):
|
||||
category = COMMODITY_TERMS.get(token)
|
||||
if category:
|
||||
return category
|
||||
return None
|
||||
|
||||
|
||||
def _tokens(name: str) -> list:
|
||||
text = (name or "").lower()
|
||||
text = text.replace("-", " ").replace("/", " ").replace("&", " ")
|
||||
for phrase, replacement in _PHRASES.items():
|
||||
text = text.replace(phrase, replacement)
|
||||
text = _strip_sizes(text)
|
||||
return _WORD_RE.findall(text)
|
||||
|
||||
|
||||
def is_unbranded(name: str, sheet_brand: Optional[str] = None,
|
||||
brand_column_supplied: bool = False) -> bool:
|
||||
"""True when `name` names a commodity rather than a branded product.
|
||||
|
||||
`sheet_brand` is whatever the spreadsheet's own Brand column said, and
|
||||
`brand_column_supplied` whether that column existed at all. A store that
|
||||
troubled itself to include the column and left the cell empty has said
|
||||
something explicit, and is believed.
|
||||
|
||||
Deliberately conservative: a single token that is not a known commodity or
|
||||
qualifier means the row keeps its normal brand resolution. Getting this
|
||||
wrong in the permissive direction would collapse real regional brands into
|
||||
one bucket, which is far harder to undo than a staple sitting in its own
|
||||
table.
|
||||
"""
|
||||
if sheet_brand and str(sheet_brand).strip():
|
||||
return False
|
||||
if brand_column_supplied:
|
||||
return True
|
||||
|
||||
tokens = _tokens(name)
|
||||
significant = [t for t in tokens if t not in QUALIFIERS and len(t) > 1]
|
||||
if not significant:
|
||||
return False
|
||||
return all(token in COMMODITY_TERMS for token in significant)
|
||||
@@ -59,6 +59,26 @@ CATEGORY_BANDS = {
|
||||
"skin_bath": (30, 85, 15), # soap, body wash, lotion
|
||||
"household_clean": (15, 40, 10),
|
||||
"baby_care": (40, 110, 25),
|
||||
# Loose commodities. These sit far below every branded band above because
|
||||
# the rates here are per 100g and staples are sold by the kilo - without
|
||||
# them a 1kg bag of sugar fell to "general" and priced at ₹120-280 against a
|
||||
# real shelf price near ₹50, and 100g of pepper priced at ₹20-45 against a
|
||||
# real ₹80-140. Spices are the one commodity that is genuinely expensive by
|
||||
# weight, hence the wide, high band.
|
||||
"pulses_dal": (12, 20, 25),
|
||||
"rice_grains": (7, 18, 30),
|
||||
"atta_flour": (3.5, 8, 25),
|
||||
"sugar_jaggery": (4, 7, 20),
|
||||
"salt": (2, 4, 10),
|
||||
# Spices span two orders of magnitude per 100g, so one band cannot serve
|
||||
# them: ground turmeric is ~₹30 and cardamom is ~₹400. Split by how they
|
||||
# are actually sold rather than lumping them.
|
||||
# The small-pack multiplier (1.25x at 100g, 1.7x at 50g) applies on top of
|
||||
# these, so the upper bounds are set below the shelf price they aim at.
|
||||
"spices_ground": (20, 55, 15),
|
||||
"spices_whole": (55, 120, 20),
|
||||
"spices_premium": (200, 450, 50),
|
||||
"edible_oil": (12, 22, 40),
|
||||
"general": (15, 35, 10),
|
||||
}
|
||||
|
||||
@@ -80,6 +100,31 @@ CATEGORY_KEYWORDS = [
|
||||
(["tea", "coffee"], "beverages_tea_coffee"),
|
||||
(["cereal", "muesli", "oats", "cornflakes"], "breakfast_cereal"),
|
||||
(["chips", "namkeen", "snack", "wafer", "mixture"], "snacks_namkeen"),
|
||||
# Commodities. Checked after the branded-FMCG rules above so a branded
|
||||
# product still wins its own band ("Aashirvaad Atta" is atta_flour either
|
||||
# way, but "Bourbon Cream Biscuit" must stay with biscuits), and ordered
|
||||
# narrow-to-broad within the group: "chilli powder" before "powder" would
|
||||
# matter if a bare "powder" rule existed, and "sugar" must precede nothing
|
||||
# that also contains it.
|
||||
(["dal", "dhal", "daal", "toor", "urad", "moong", "masoor", "chana", "arhar", "rajma", "lentil", "pulses"], "pulses_dal"),
|
||||
(["basmati", "sona masoori", "rice", "poha", "millet", "ragi", "bajra", "jowar"], "rice_grains"),
|
||||
(["atta", "maida", "besan", "sooji", "suji", "rava", "semolina", "flour"], "atta_flour"),
|
||||
(["jaggery", "gur", "sugar"], "sugar_jaggery"),
|
||||
(["salt"], "salt"),
|
||||
# Spices, most specific first. Note that `classify_category` prepends the
|
||||
# CATEGORY HINT to the title before matching, so the words "spice" and
|
||||
# "masala" must NOT appear in these three rules - the hint "Spices &
|
||||
# Masalas" contains both, and either one would drag every spice into
|
||||
# whichever rule mentioned it regardless of what the product actually is.
|
||||
# Only the product's own name distinguishes them; the generic blend
|
||||
# catch-all comes afterwards.
|
||||
(["cardamom", "elaichi", "saffron", "kesar", "javitri"], "spices_premium"),
|
||||
(["turmeric", "haldi", "chilli", "chili", "mirchi", "dhania"], "spices_ground"),
|
||||
(["pepper", "jeera", "cumin", "coriander", "mustard", "methi", "fenugreek", "saunf", "fennel", "clove", "cinnamon", "dalchini", "asafoetida", "hing", "ajwain"], "spices_whole"),
|
||||
# A packaged blend, and the fallback for anything the hint alone identifies
|
||||
# as a spice. Blends are ground, so they price with the ground band.
|
||||
(["masala", "sambar", "rasam", "spice"], "spices_ground"),
|
||||
(["groundnut oil", "sunflower oil", "sesame oil", "gingelly", "mustard oil", "coconut oil", "edible oil", "cooking oil"], "edible_oil"),
|
||||
]
|
||||
|
||||
# Small curated brand-tier list. Unknown brands default to "mainstream" (1.0x).
|
||||
@@ -313,6 +358,16 @@ def default_size_variants(category_hint: str, product_title: str = "") -> list:
|
||||
"beverages_juice": ["200ml", "1L", "2L"],
|
||||
"beverages_tea_coffee": ["100g", "250g", "500g"],
|
||||
"breakfast_cereal": ["250g", "500g", "1kg"],
|
||||
# Commodities are bought by the kilo, not in 100g packets. Sugar and
|
||||
# salt are near-universally a single 1kg pack; spices are the opposite,
|
||||
# sold in small quantities because they are expensive by weight.
|
||||
"pulses_dal": ["500g", "1kg", "5kg"],
|
||||
"rice_grains": ["1kg", "5kg", "25kg"],
|
||||
"atta_flour": ["500g", "1kg", "5kg"],
|
||||
"sugar_jaggery": ["500g", "1kg"],
|
||||
"salt": ["1kg"],
|
||||
"spices_whole": ["50g", "100g", "250g"],
|
||||
"edible_oil": ["500ml", "1L", "5L"],
|
||||
"general": ["100g", "250g", "500g"],
|
||||
}
|
||||
return presets.get(category, presets["general"])
|
||||
|
||||
Reference in New Issue
Block a user