backend store_catalog updates

This commit is contained in:
sriram
2026-08-31 11:03:56 +05:30
parent 998df898db
commit 8394316907
19 changed files with 892 additions and 8 deletions

View File

@@ -215,7 +215,12 @@ BRAND_SYNC_INTERVAL_SECONDS=300
# deleted, the other brand_* tables just stop being discovered.
# Names resolve through the brand aliases, so "Tata" activates
# brand_hindustan_unilever exactly as ingesting Tata products would.
#ACTIVE_BRANDS=Amul,Cadbury,Hindustan Unilever
# NOTE: this is a whitelist, so anything omitted is invisible in browse, search,
# suggestions, chat and the MCP tools. If you narrow it, include "Own Products"
# - that is the bucket unbranded commodities (dal, sugar, salt, spices) are
# filed under, and it is a normal brand table as far as every read path is
# concerned.
#ACTIVE_BRANDS=Amul,Cadbury,Hindustan Unilever,Own Products
USE_S3=true

View File

@@ -80,6 +80,7 @@ from app.services.category_registry import (
sanitize_category_language,
)
from app.services.category_units import fix_or_reject_size, parse_unit
from app.services.generic_products import OWN_PRODUCTS_BRAND, is_unbranded
from app.services.embeddings_service import embed_texts
from app.services.enrichment.barcode.stage import BarcodeEnrichmentStage
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
@@ -236,6 +237,20 @@ def infer_brand(product_name: str) -> Optional[str]:
def stage_1_brand_and_fssai(row: Dict[str, Any]) -> Dict[str, Any]:
brand = row.get("brand") or ""
# The own-products bucket is not a brand and must never be resolved like
# one. resolve_parent_brand is identity for it today, but it matches
# bidirectionally against every alias, so a future alias containing "own"
# or "products" would silently redirect the whole table. Skipping is the
# only version of this that stays true.
#
# It also keeps get_fssai_license out of the picture: a commodity has no
# licence, and the bug being fixed here was precisely a real third-party
# licence number being stamped onto unbranded rows.
if brand == OWN_PRODUCTS_BRAND:
row["brand"] = OWN_PRODUCTS_BRAND
row["brand_name"] = OWN_PRODUCTS_BRAND
row["_table"] = f"brand_{_sanitize_name(OWN_PRODUCTS_BRAND)}"
return row
parent = resolve_parent_brand(brand)
row["brand"] = parent
row["brand_name"] = parent
@@ -433,12 +448,21 @@ def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, An
from app.core.catalog_engine import catalog_engine
from app.services.image_search import find_all_image_urls
# "Own Products" is a bucket, not a brand, so it must not enter the
# search query or the relevance scoring - searching for "Own Products
# Toor Dal" finds nothing. With it blank the query is just the product
# name, and _select_best_images derives its distinctive tokens from the
# whole name ({toor, dal}), which is exactly right for a commodity.
brand = row.get("brand") or ""
if brand == OWN_PRODUCTS_BRAND:
brand = ""
candidates = find_all_image_urls(
row.get("product_name") or "", brand=row.get("brand"), max_results=24
row.get("product_name") or "", brand=brand or None, max_results=24
)
if candidates:
best = catalog_engine._select_best_images(
candidates, row.get("product_name") or "", row.get("brand") or "", max_images=10
candidates, row.get("product_name") or "", brand, max_images=10
)
if best:
row["image_urls"] = list(best)
@@ -722,7 +746,23 @@ def run_pipeline(
raw_name = _text(record, mapping, "product_name") or _text(record, mapping, "title") or ""
if not brand_column_supplied:
record[_INFERRED_BRAND_COL] = infer_brand(raw_name) or ""
# Commodity rows are diverted BEFORE infer_brand rather than after,
# because both of the steps that follow damage them. infer_brand
# falls back to the first word ("Sugar 1kg" -> brand "Sugar"), and
# resolve_parent_brand then whole-word-matches that against ~230
# aliases, which is how "Salt" reached colgate-palmolive and "Milk"
# reached cadbury. Neither runs for these rows now.
if is_unbranded(raw_name):
record[_INFERRED_BRAND_COL] = OWN_PRODUCTS_BRAND
else:
record[_INFERRED_BRAND_COL] = infer_brand(raw_name) or ""
elif _blank(_text(record, mapping, "brand")) and is_unbranded(
raw_name, brand_column_supplied=True
):
# The sheet has a brand column and left this cell empty. That is an
# explicit statement, and previously it was the one case that got
# the row rejected outright as "no brand name in this row".
record[mapping.columns["brand"]] = OWN_PRODUCTS_BRAND
try:
req = row_to_request(record, mapping)

View File

@@ -75,6 +75,17 @@ CATEGORY_REGISTRY: List[Dict[str, object]] = [
# and grounds far more often than as a drink).
{"category": "Beverages", "keywords": ["beverages", "beverage", "soft drink", "soft drinks", "cold drink", "cold drinks", "carbonated", "aerated drink", "cola", "coke", "juice", "juices", "squash", "sharbat", "energy drink", "sports drink", "mineral water", "packaged drinking water", "lemonade", "iced tea", "thums up", "sprite", "fanta", "limca", "maaza", "pepsi", "mirinda"], "generic_term": "beverage"},
{"category": "Cooking Oils", "keywords": ["cooking oil", "edible oil", "sunflower oil", "mustard oil", "vanaspati", "refined oil", "oil", "oils"], "generic_term": "cooking oil"},
# Loose-commodity categories. The names are chosen to match entries that
# already exist in category_units.CATEGORY_UNIT_TYPE and in
# enrichment/hsn_gst/models.HSN_GST_TABLE, so a bag of dal picks up its unit
# rulebook and its HSN code without either table needing a new key. Listed
# before "Cooking Oils", whose bare "oil" keyword is greedy, and before
# "Atta & Staples", whose "dal"/"pulses"/"rice" keywords would otherwise
# swallow every pulse.
{"category": "Pulses, Grains & Spices", "keywords": ["dal", "dhal", "daal", "toor dal", "urad dal", "moong dal", "masoor dal", "chana dal", "arhar", "lentils", "lentil", "rajma", "pulses", "pulse"], "generic_term": "pulse"},
{"category": "Spices & Masalas", "keywords": ["spices", "spice", "masala", "masalas", "turmeric", "haldi", "chilli powder", "coriander powder", "cumin", "jeera", "peppercorn", "black pepper", "cardamom", "asafoetida", "hing", "tamarind"], "generic_term": "spice"},
{"category": "Sugar & Jaggery", "keywords": ["sugar", "jaggery", "gur", "brown sugar", "cane sugar", "misri"], "generic_term": "sweetener"},
{"category": "Salt & Staples", "keywords": ["salt", "rock salt", "sea salt", "table salt", "iodised salt", "iodized salt"], "generic_term": "salt"},
{"category": "Atta & Staples", "keywords": ["atta", "wheat flour", "flour", "rice", "dal", "pulses", "staples", "suji", "maida"], "generic_term": "staple product"},
{"category": "Dairy", "keywords": ["milk", "dairy", "cheese", "paneer", "panner", "paner", "paneerr", "curd", "yogurt", "butter", "ghee", "dahi"], "generic_term": "dairy product"},
{"category": "Oral Care", "keywords": ["toothpaste", "toothbrush", "mouthwash", "paste"], "generic_term": "oral care product"},

View File

@@ -108,6 +108,9 @@ CATEGORY_UNIT_TYPE: dict[str, str] = {
"candy & confectionery": "weight",
"atta & staples": "weight",
"spices & masalas": "weight",
# Loose commodities, sold by weight in every case.
"pulses, grains & spices": "weight",
"sugar & jaggery": "weight",
"pasta & noodles": "weight",
"noodles & instant food": "weight",
"breakfast cereal": "weight",

View File

@@ -67,6 +67,8 @@ HSN_GST_TABLE: Dict[str, Tuple[str, int, bool]] = {
"Salt & Staples": ("2501", 5, False),
"Pulses, Grains & Spices": ("0713", 5, True),
"Spices & Masalas": ("0910", 5, False),
# Chapter 17: cane/beet sugar and jaggery, 5% for ordinary retail sugar.
"Sugar & Jaggery": ("1701", 5, False),
"Cooking Oils": ("1517", 5, False),
"Pickles & Chutneys": ("2001", 12, True),
"Dry Fruits & Nuts": ("0801", 12, True),

View File

@@ -0,0 +1,237 @@
"""
Unbranded grocery commodities, and the one table they belong in.
WHY THIS FILE EXISTS
--------------------
A store sheet routinely lists items that simply have no brand: "Toor Dhal 1kg",
"Sugar", "Salt 1kg", "Black Pepper 100g". Nothing rejected those rows - what
happened was worse. `store_catalog_pipeline.infer_brand()` falls back to the
first word of the name, and `brand_registry.resolve_parent_brand()` then does a
bidirectional whole-word scan over ~230 aliases. Between them:
"Toor Dhal 1kg" -> brand "Toor" -> a junk table brand_toor
"Sugar 1kg" -> brand "Sugar" -> a junk table brand_sugar
"Salt 1kg" -> "colgate active salt" contains "salt"
-> brand_colgate_palmolive
"Milk 1L" -> "cadbury dairy milk" contains "milk"
-> brand_cadbury
"Butter 500g" -> "nestle butter" -> brand_nestle
"Red Chilli Powder" -> "brooke bond red label" -> brand_brooke_bond
The junk tables are noise. The misroutes are real damage: generic groceries
written into the live Amul, Nestle, Cadbury and HUL catalogs - and
`stage_1_brand_and_fssai` then stamps that brand's FSSAI licence number onto the
row, so unbranded chilli powder ships carrying Brooke Bond's real licence.
WHY A WORD LIST AND NOT "THE BRAND IS UNKNOWN"
----------------------------------------------
"Not in BRAND_ALIASES" is the obvious rule and it is wrong here. That map holds
231 mostly-large FMCG names; Bikaji, Aachi, Idhayam, Naga and Lion Dates are all
real brands absent from it. Treating unknown as unbranded would sweep every
regional brand into one bucket.
So the test is POSITIVE and conservative: strip the pack size and the words that
carry no brand signal, and require that *everything still standing* is a
commodity noun. One unrecognised token means "this is a brand". Hence:
"Butter 500g" -> {butter} -> unbranded
"Amul Butter 500g" -> {amul, butter} -> branded, unchanged
"Aachi Sambar" -> {aachi, sambar} -> branded, unchanged
The lexicon doubles as a category map, so the same entry that identifies a
commodity also says which canonical category it belongs to - a bag of dal should
not have to go through keyword detection twice.
"""
from __future__ import annotations
import re
from typing import Dict, Optional, Set
from app.services.category_units import parse_unit
# The single brand every unbranded product is filed under. `_sanitize_name`
# turns this into the table `brand_own_products`, and `display_name_for_suffix`
# turns that back into "Own Products" for the UI, so the round trip the frontend
# depends on (card -> /api/brands/{brand}/products) is an identity.
#
# Deliberately NOT registered in BRAND_ALIASES: an alias overlapping these words
# would let resolve_parent_brand hijack the table, which is the exact class of
# bug this module exists to end.
OWN_PRODUCTS_BRAND = "Own Products"
# ---------------------------------------------------------------------------
# The lexicon: commodity term -> canonical category
# ---------------------------------------------------------------------------
# Categories are the canonical names from category_registry.ALL_CATEGORIES, so
# HSN/GST resolution and the pack-size unit rulebook both key off them without
# a second translation step.
_PULSES = "Pulses, Grains & Spices"
_STAPLES = "Atta & Staples"
_SPICES = "Spices & Masalas"
_SUGAR = "Sugar & Jaggery"
_SALT = "Salt & Staples"
_OILS = "Cooking Oils"
_DAIRY = "Dairy"
_BEVERAGE = "Beverages"
COMMODITY_TERMS: Dict[str, str] = {}
def _add(category: str, *terms: str) -> None:
for term in terms:
COMMODITY_TERMS[term] = category
# Pulses and lentils. Indian sheets spell dal a dozen ways.
_add(_PULSES,
"dal", "dhal", "dhall", "daal", "dail", "pulse", "pulses", "lentil", "lentils",
"toor", "tur", "arhar", "urad", "urid", "moong", "mung", "masoor", "masur",
"chana", "channa", "gram", "rajma", "lobia", "kabuli", "peas", "matar",
"soya", "soyabean")
# Grains, flours and other dry staples.
_add(_STAPLES,
"rice", "basmati", "sona", "masoori", "ponni", "idli", "sona masoori",
"wheat", "atta", "maida", "sooji", "suji", "rava", "semolina", "besan",
"poha", "aval", "ragi", "bajra", "jowar", "millet", "millets", "quinoa",
"sabudana", "vermicelli", "corn", "oats", "flour")
# Sweeteners.
_add(_SUGAR, "sugar", "jaggery", "gur", "misri", "honey", "sakkarai")
# Salt.
_add(_SALT, "salt", "sendha", "iodised", "iodized")
# Spices, whole and ground.
_add(_SPICES,
"pepper", "peppercorn", "peppercorns", "turmeric", "haldi", "manjal",
"chilli", "chili", "chillies", "chile", "mirchi", "coriander", "dhania",
"cumin", "jeera", "mustard", "methi", "fenugreek", "cardamom", "elaichi",
"clove", "cloves", "lavang", "cinnamon", "dalchini", "bay", "tejpatta",
"asafoetida", "hing", "tamarind", "imli", "masala", "garam", "sambar",
"rasam", "ajwain", "saunf", "fennel", "nutmeg", "mace", "star", "anise",
"kalonji", "poppy", "khus")
# Edible oils. Bare "oil" currently resolves to Johnson & Johnson.
_add(_OILS,
"oil", "gingelly", "groundnut", "peanut", "sunflower", "sesame", "til",
"coconut", "castor", "vanaspati")
# Loose dairy. These are the ones misrouting into real brand tables today.
_add(_DAIRY,
"milk", "butter", "ghee", "paneer", "curd", "dahi", "yogurt", "yoghurt",
"cheese", "khoa", "khoya", "cream", "buttermilk", "lassi")
# Loose tea and coffee.
_add(_BEVERAGE, "tea", "coffee", "chai")
# Words that describe a product without naming a brand. Stripped before the
# all-tokens-are-commodities test, so "Organic Toor Dal Whole 1kg" still reads
# as unbranded.
QUALIFIERS: Set[str] = {
# quality / provenance
"organic", "premium", "fresh", "natural", "pure", "best", "quality",
"grade", "select", "special", "classic", "regular", "standard", "economy",
"value", "farm", "country", "desi", "local", "homemade", "traditional",
# processing / form
"whole", "half", "split", "raw", "roasted", "unroasted", "polished",
"unpolished", "sortex", "cleaned", "washed", "refined", "filtered",
"double", "single", "extra", "fine", "coarse", "powder", "powdered",
"ground", "crushed", "flakes", "seeds", "seed", "granules", "crystal",
"crystals", "cube", "cubes", "stick", "sticks", "dried", "dry",
"slice", "slices", "sliced", "block", "grated", "shredded", "chopped",
# colour / variety, which qualify a commodity rather than brand it
"black", "white", "red", "green", "yellow", "brown", "long", "short",
"small", "big", "large", "medium",
# packaging / retail noise
"pack", "packet", "packed", "loose", "bag", "pouch", "box", "tin", "jar",
"bottle", "refill", "combo", "assorted", "mixed", "mix",
# connectives
"and", "with", "of", "the", "in", "for",
}
# Multi-word commodities collapsed to a single token before tokenising, so the
# individual words do not have to stand alone in the lexicon.
_PHRASES = {
"rock salt": "salt",
"sea salt": "salt",
"table salt": "salt",
"black pepper": "pepper",
"white pepper": "pepper",
"bengal gram": "chana",
"green gram": "moong",
"black gram": "urad",
"horse gram": "chana",
"red chilli": "chilli",
"bay leaf": "bay",
"star anise": "anise",
"sona masoori": "rice",
"wheat flour": "atta",
"gram flour": "besan",
"corn flour": "flour",
"rice flour": "flour",
"brown sugar": "sugar",
"palm jaggery": "jaggery",
"cane sugar": "sugar",
}
_WORD_RE = re.compile(r"[a-z]+")
def _strip_sizes(text: str) -> str:
"""Remove pack sizes and bare numbers - they never name a brand."""
# "1kg", "500 g", "1.5 L", and any leftover bare number.
text = re.sub(r"\b\d+(?:[.,]\d+)?\s*[a-z]*\b", " ", text)
return text
def canonical_category(name: str) -> Optional[str]:
"""The category implied by the commodity words in `name`, if any.
Returns the category of the FIRST commodity term found, scanning left to
right, because an Indian product name leads with its head noun ("Toor Dhal",
"Sugar", "Groundnut Oil").
"""
for token in _tokens(name):
category = COMMODITY_TERMS.get(token)
if category:
return category
return None
def _tokens(name: str) -> list:
text = (name or "").lower()
text = text.replace("-", " ").replace("/", " ").replace("&", " ")
for phrase, replacement in _PHRASES.items():
text = text.replace(phrase, replacement)
text = _strip_sizes(text)
return _WORD_RE.findall(text)
def is_unbranded(name: str, sheet_brand: Optional[str] = None,
brand_column_supplied: bool = False) -> bool:
"""True when `name` names a commodity rather than a branded product.
`sheet_brand` is whatever the spreadsheet's own Brand column said, and
`brand_column_supplied` whether that column existed at all. A store that
troubled itself to include the column and left the cell empty has said
something explicit, and is believed.
Deliberately conservative: a single token that is not a known commodity or
qualifier means the row keeps its normal brand resolution. Getting this
wrong in the permissive direction would collapse real regional brands into
one bucket, which is far harder to undo than a staple sitting in its own
table.
"""
if sheet_brand and str(sheet_brand).strip():
return False
if brand_column_supplied:
return True
tokens = _tokens(name)
significant = [t for t in tokens if t not in QUALIFIERS and len(t) > 1]
if not significant:
return False
return all(token in COMMODITY_TERMS for token in significant)

View File

@@ -59,6 +59,26 @@ CATEGORY_BANDS = {
"skin_bath": (30, 85, 15), # soap, body wash, lotion
"household_clean": (15, 40, 10),
"baby_care": (40, 110, 25),
# Loose commodities. These sit far below every branded band above because
# the rates here are per 100g and staples are sold by the kilo - without
# them a 1kg bag of sugar fell to "general" and priced at ₹120-280 against a
# real shelf price near ₹50, and 100g of pepper priced at ₹20-45 against a
# real ₹80-140. Spices are the one commodity that is genuinely expensive by
# weight, hence the wide, high band.
"pulses_dal": (12, 20, 25),
"rice_grains": (7, 18, 30),
"atta_flour": (3.5, 8, 25),
"sugar_jaggery": (4, 7, 20),
"salt": (2, 4, 10),
# Spices span two orders of magnitude per 100g, so one band cannot serve
# them: ground turmeric is ~₹30 and cardamom is ~₹400. Split by how they
# are actually sold rather than lumping them.
# The small-pack multiplier (1.25x at 100g, 1.7x at 50g) applies on top of
# these, so the upper bounds are set below the shelf price they aim at.
"spices_ground": (20, 55, 15),
"spices_whole": (55, 120, 20),
"spices_premium": (200, 450, 50),
"edible_oil": (12, 22, 40),
"general": (15, 35, 10),
}
@@ -80,6 +100,31 @@ CATEGORY_KEYWORDS = [
(["tea", "coffee"], "beverages_tea_coffee"),
(["cereal", "muesli", "oats", "cornflakes"], "breakfast_cereal"),
(["chips", "namkeen", "snack", "wafer", "mixture"], "snacks_namkeen"),
# Commodities. Checked after the branded-FMCG rules above so a branded
# product still wins its own band ("Aashirvaad Atta" is atta_flour either
# way, but "Bourbon Cream Biscuit" must stay with biscuits), and ordered
# narrow-to-broad within the group: "chilli powder" before "powder" would
# matter if a bare "powder" rule existed, and "sugar" must precede nothing
# that also contains it.
(["dal", "dhal", "daal", "toor", "urad", "moong", "masoor", "chana", "arhar", "rajma", "lentil", "pulses"], "pulses_dal"),
(["basmati", "sona masoori", "rice", "poha", "millet", "ragi", "bajra", "jowar"], "rice_grains"),
(["atta", "maida", "besan", "sooji", "suji", "rava", "semolina", "flour"], "atta_flour"),
(["jaggery", "gur", "sugar"], "sugar_jaggery"),
(["salt"], "salt"),
# Spices, most specific first. Note that `classify_category` prepends the
# CATEGORY HINT to the title before matching, so the words "spice" and
# "masala" must NOT appear in these three rules - the hint "Spices &
# Masalas" contains both, and either one would drag every spice into
# whichever rule mentioned it regardless of what the product actually is.
# Only the product's own name distinguishes them; the generic blend
# catch-all comes afterwards.
(["cardamom", "elaichi", "saffron", "kesar", "javitri"], "spices_premium"),
(["turmeric", "haldi", "chilli", "chili", "mirchi", "dhania"], "spices_ground"),
(["pepper", "jeera", "cumin", "coriander", "mustard", "methi", "fenugreek", "saunf", "fennel", "clove", "cinnamon", "dalchini", "asafoetida", "hing", "ajwain"], "spices_whole"),
# A packaged blend, and the fallback for anything the hint alone identifies
# as a spice. Blends are ground, so they price with the ground band.
(["masala", "sambar", "rasam", "spice"], "spices_ground"),
(["groundnut oil", "sunflower oil", "sesame oil", "gingelly", "mustard oil", "coconut oil", "edible oil", "cooking oil"], "edible_oil"),
]
# Small curated brand-tier list. Unknown brands default to "mainstream" (1.0x).
@@ -313,6 +358,16 @@ def default_size_variants(category_hint: str, product_title: str = "") -> list:
"beverages_juice": ["200ml", "1L", "2L"],
"beverages_tea_coffee": ["100g", "250g", "500g"],
"breakfast_cereal": ["250g", "500g", "1kg"],
# Commodities are bought by the kilo, not in 100g packets. Sugar and
# salt are near-universally a single 1kg pack; spices are the opposite,
# sold in small quantities because they are expensive by weight.
"pulses_dal": ["500g", "1kg", "5kg"],
"rice_grains": ["1kg", "5kg", "25kg"],
"atta_flour": ["500g", "1kg", "5kg"],
"sugar_jaggery": ["500g", "1kg"],
"salt": ["1kg"],
"spices_whole": ["50g", "100g", "250g"],
"edible_oil": ["500ml", "1L", "5L"],
"general": ["100g", "250g", "500g"],
}
return presets.get(category, presets["general"])

View File

@@ -1,5 +1,5 @@
{
"_sku_sequences": {
"BIKAJI-ALO-200": 8
"BIKAJI-ALO-200": 11
}
}

View File

@@ -0,0 +1,5 @@
{
"_sku_sequences": {
"BLACK-PEP-100": 2
}
}

View File

@@ -0,0 +1,12 @@
{
"_sku_sequences": {
"OWN-TOO-1": 3,
"OWN-SUG-1": 3,
"OWN-SAL-1": 3,
"OWN-BLA-100": 3,
"OWN-RIC-5": 1,
"OWN-MIL-1": 3,
"OWN-BUT-500": 3,
"OWN-RED-100": 3
}
}

View File

@@ -0,0 +1,5 @@
{
"_sku_sequences": {
"RICE-5KG-5": 1
}
}

View File

@@ -0,0 +1,8 @@
{
"_sku_sequences": {
"SUGAR-PRD-100": 1,
"SUGAR-PRD-250": 1,
"SUGAR-PRD-500": 1,
"SUGAR-1KG-1": 2
}
}

View File

@@ -0,0 +1,5 @@
{
"_sku_sequences": {
"TOOR-DHA-1": 3
}
}

View File

@@ -1,5 +1,5 @@
{
"_sku_sequences": {
"AMUL-BUT-500": 8
"AMUL-BUT-500": 11
}
}

View File

@@ -1,7 +1,7 @@
{
"_sku_sequences": {
"BRITAN-GOO-100": 10,
"BRITAN-GOO-200": 10,
"BRITAN-GOO-200": 12,
"BRITAN-GOO-500": 9,
"BRITAN-GOO-250": 1,
"BRITAN-MAR-100": 2,
@@ -15,6 +15,7 @@
"BRITAN-MAR-375": 1,
"BRITAN-MIL-100": 1,
"BRITAN-MIL-375": 1,
"BRITAN-MIL-150": 1
"BRITAN-MIL-150": 1,
"BRITAN-50-200": 1
}
}

View File

@@ -0,0 +1,5 @@
{
"_sku_sequences": {
"COLGAT-SAL-1": 1
}
}

View File

@@ -0,0 +1,5 @@
{
"_sku_sequences": {
"HINDUS-TAT-1": 2
}
}

View File

@@ -0,0 +1,217 @@
#!/usr/bin/env python3
"""
Move already-stored unbranded commodities into `brand_own_products`.
Fixing the pipeline only helps the next upload. Rows written before the fix are
still sitting in two kinds of wrong place:
* **Junk tables** - `brand_toor`, `brand_sugar`, `brand_rice`, `brand_black`,
one per leading word, created because `infer_brand` fell back to the first
token of the product name.
* **Real brand catalogs** - and this is the damaging half. A single commodity
noun matching a whole word inside a multi-word alias sent salt into
`brand_colgate_palmolive` ("colgate active salt"), milk into `brand_cadbury`
("cadbury dairy milk"), butter/ghee/paneer into `brand_nestle`, cheese into
`brand_amul`, tea into `brand_hindustan_unilever` and red chilli powder into
`brand_brooke_bond` ("brooke bond red label"). `stage_1_brand_and_fssai`
then stamped that brand's REAL FSSAI licence number onto the row, so those
rows are carrying another company's regulatory identifier.
The same `is_unbranded` predicate the pipeline now uses does both jobs, and it
is what keeps genuine products safe: "Amul Butter 500g" contains "amul", which
is not a commodity term, so it is never touched. Only the bare "Butter 500g" is.
What a moved row gets:
* `fssai_license` cleared - it belonged to somebody else.
* a recomputed `image_id` (`own_products_<slug>`), so a later re-ingest of the
same sheet UPDATES the row instead of inserting a duplicate beside it.
* `image_url` / `image_urls` cleared, so the fixed image search refetches
them. The old S3 folder is orphaned by this; those images were mostly
absent or belonged to the wrong product anyway.
* `category` and `price_range` re-resolved with the staple bands.
Usage:
python -m scripts.migrate_own_products --dry-run # default
python -m scripts.migrate_own_products --apply
python -m scripts.migrate_own_products --apply --drop-empty
`--dry-run` is the default and `--apply` must be explicit: this rewrites
whatever database `backend/.env` points at, which is production. The target host
is printed on startup so it can be checked before committing.
"""
from __future__ import annotations
import argparse
import logging
import sys
from pathlib import Path
from typing import Any, Dict, List, Tuple
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.infrastructure.settings import DB_HOST, DB_NAME
from app.services import price_estimator
from app.services.category_registry import detect_category_from_text
from app.services.generic_products import OWN_PRODUCTS_BRAND, is_unbranded
from app.services.vector_store import (
_connect,
_sanitize_name,
ensure_brand_schema,
)
logging.basicConfig(level=logging.INFO, format="%(asctime)s - %(levelname)s - %(message)s")
logger = logging.getLogger(__name__)
TARGET_SUFFIX = _sanitize_name(OWN_PRODUCTS_BRAND) # "own_products"
TARGET_TABLE = f"brand_{TARGET_SUFFIX}"
# Columns copied across. Deliberately explicit rather than SELECT *: `id` is a
# per-table BIGSERIAL and must not travel, and `embedding` is recomputed by the
# next ingest rather than carried.
_COPY_COLUMNS = (
"product_name", "title", "description", "category", "image_id",
"image_url", "image_urls", "price_range", "size_variants", "providers",
"fssai_license", "product_sku", "sku_source", "hsn_code",
"final_selling_price", "selling_price", "barcode", "barcode_type",
"highlights", "nutrients", "search_query",
)
def _slugify(text: str) -> str:
import re
return re.sub(r"_+", "_", re.sub(r"[^a-z0-9]+", "_", (text or "").lower())).strip("_")
def _brand_tables(cur) -> List[str]:
cur.execute(
"SELECT table_name FROM information_schema.tables "
"WHERE table_schema = 'public' AND table_name LIKE 'brand_%' "
"ORDER BY table_name"
)
return [r[0] for r in cur.fetchall()]
def _rebuild_row(row: Dict[str, Any]) -> Dict[str, Any]:
"""Re-resolve the fields that were wrong because the brand was wrong."""
name = row.get("product_name") or ""
category = detect_category_from_text(name) or "General"
size = (row.get("size_variants") or ["Standard"])[0]
updated = dict(row)
updated["category"] = category
updated["image_id"] = f"{TARGET_SUFFIX}_{_slugify(name)}"
# Belonged to the brand this row was wrongly filed under.
updated["fssai_license"] = None
# Cleared so the fixed search refetches with a query that is not poisoned
# by a brand name the product never had.
updated["image_url"] = None
updated["image_urls"] = []
lo, hi = price_estimator.estimate_price_range_for_size(
size, name, OWN_PRODUCTS_BRAND, category
)
updated["price_range"] = f"₹{lo}-{hi}"
return updated
def scan(cur) -> List[Tuple[str, Dict[str, Any]]]:
"""Every unbranded row currently sitting somewhere else."""
found: List[Tuple[str, Dict[str, Any]]] = []
for table in _brand_tables(cur):
if table == TARGET_TABLE:
continue
try:
cur.execute(f"SELECT {', '.join(_COPY_COLUMNS)} FROM {table}")
except Exception as exc: # noqa: BLE001 - a legacy table may lack columns
logger.warning("Skipping %s: %s", table, exc)
continue
for values in cur.fetchall():
row = dict(zip(_COPY_COLUMNS, values))
if is_unbranded(row.get("product_name") or ""):
found.append((table, row))
return found
def migrate(apply: bool, drop_empty: bool) -> int:
conn = _connect()
if not conn:
logger.error("Could not connect to PostgreSQL (USE_PGVECTOR off, or DB unreachable).")
return 1
try:
with conn.cursor() as cur:
candidates = scan(cur)
if not candidates:
logger.info("Nothing to migrate - no unbranded rows outside %s.", TARGET_TABLE)
return 0
by_table: Dict[str, int] = {}
for table, row in candidates:
by_table[table] = by_table.get(table, 0) + 1
logger.info("%s: %r -> %s", table, row["product_name"], TARGET_TABLE)
logger.info("--- %d row(s) across %d table(s) ---", len(candidates), len(by_table))
for table, count in sorted(by_table.items()):
logger.info(" %-34s %d row(s)", table, count)
if not apply:
logger.info("Dry run - nothing written. Re-run with --apply to commit.")
return 0
ensure_brand_schema(OWN_PRODUCTS_BRAND)
moved = 0
with conn.cursor() as cur:
for table, row in candidates:
rebuilt = _rebuild_row(row)
columns = ", ".join(_COPY_COLUMNS)
placeholders = ", ".join(["%s"] * len(_COPY_COLUMNS))
# ON CONFLICT: the same commodity may sit in several junk
# tables; the first one wins and the rest are dropped rather
# than failing the whole migration.
cur.execute(
f"INSERT INTO {TARGET_TABLE} ({columns}) VALUES ({placeholders}) "
f"ON CONFLICT (image_id) DO NOTHING",
[rebuilt[c] for c in _COPY_COLUMNS],
)
cur.execute(
f"DELETE FROM {table} WHERE image_id = %s", [row["image_id"]]
)
moved += 1
if drop_empty:
for table in sorted(by_table):
cur.execute(f"SELECT COUNT(*) FROM {table}")
if cur.fetchone()[0] == 0:
logger.info("Dropping now-empty %s", table)
cur.execute(f"DROP TABLE {table}")
logger.info("Moved %d row(s) into %s.", moved, TARGET_TABLE)
logger.info(
"Remember: %r must be in ACTIVE_BRANDS or the table stays invisible.",
OWN_PRODUCTS_BRAND,
)
return 0
finally:
conn.close()
def main() -> int:
parser = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
)
parser.add_argument("--apply", action="store_true",
help="write the changes (default is a dry run)")
parser.add_argument("--dry-run", action="store_true",
help="explicit no-op; this is already the default")
parser.add_argument("--drop-empty", action="store_true",
help="drop junk brand tables left with no rows (requires --apply)")
args = parser.parse_args()
logger.info("Target database: %s/%s", DB_HOST, DB_NAME)
logger.info("Mode: %s", "APPLY - rows will be moved" if args.apply else "DRY RUN - no writes")
return migrate(apply=args.apply, drop_empty=args.drop_empty)
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,268 @@
"""Tests for unbranded-commodity detection and the own-products bucket.
There was no test anywhere that fed `infer_brand` a name with no brand in it,
which is exactly why "Salt 1kg" had been filing itself under Colgate-Palmolive
and "Red Chilli Powder" under Brooke Bond - complete with Brooke Bond's real
FSSAI licence number. The first two sections below are the regression wall for
that, and the third is the property that keeps this from over-reaching.
"""
from __future__ import annotations
import io
import pytest
from app.core import store_catalog_pipeline as pipeline
from app.services import price_estimator
from app.services.category_registry import detect_category_from_text
from app.services.generic_products import (
OWN_PRODUCTS_BRAND,
canonical_category,
is_unbranded,
)
openpyxl = pytest.importorskip("openpyxl")
def _sheet(names) -> bytes:
wb = openpyxl.Workbook()
ws = wb.active
ws.append(["Item Name"])
for name in names:
ws.append([name])
buf = io.BytesIO()
wb.save(buf)
return buf.getvalue()
@pytest.fixture(autouse=True)
def _isolate_sku_counter(tmp_path, monkeypatch):
from app.services import sku_service
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
@pytest.fixture
def store(monkeypatch):
"""Captures (brand, row) pairs so a test can assert where a row landed."""
written: list = []
def fake_upsert(brand, rows, cleanup=False):
assert cleanup is False, "cleanup=True would delete the brand's existing catalog"
for row in rows:
written.append((brand, row))
return len(rows)
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: [])
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
return written
def _brand_of(names, store):
"""Run one sheet and return the brand the (single) input row landed under.
Matches on the ORIGINAL name rather than the stored `product_name`: stage 4
explodes a row into one per pack size and appends that size to the name, so
"Idhayam Sesame Oil" is stored as "Idhayam Sesame Oil 500ml".
"""
pipeline.run_pipeline("store.xlsx", _sheet(names), use_llm=False, fetch_images=False)
brands = {brand for brand, _row in store}
assert len(brands) == 1, f"expected one destination brand, got {brands}"
return brands.pop()
# ---------------------------------------------------------------------------
# The misroutes this exists to stop
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("name", [
"Salt 1kg", # matched alias "colgate active salt"
"Milk 1L", # matched alias "cadbury dairy milk"
"Butter 500g", # matched alias "nestle butter"
"Ghee 1L",
"Paneer 200g",
"Cheese Slices",
"Tea Powder 250g",
"Red Chilli Powder 100g", # matched alias "brooke bond red label"
"Oil 1L",
])
def test_a_commodity_is_not_filed_under_a_real_brand(name, store):
"""Each of these used to be written into a live brand's catalog because one
commodity noun matched a whole word inside a multi-word alias."""
assert _brand_of([name], store) == OWN_PRODUCTS_BRAND
def test_a_commodity_never_carries_someone_elses_fssai_licence(store):
"""The compliance half of the bug: the misresolved parent's real licence
number was copied onto the row."""
pipeline.run_pipeline("s.xlsx", _sheet(["Red Chilli Powder 100g", "Milk 1L", "Butter 500g"]), use_llm=False, fetch_images=False)
assert [row["fssai_license"] for _brand, row in store] == [None, None, None]
@pytest.mark.parametrize("name", [
"Toor Dhal 1kg", "Sugar 1kg", "Black Pepper 100g", "Rice 5kg",
"Urad Dal 500g", "Wheat Flour 5kg", "Jaggery 500g", "Groundnut Oil 1L",
])
def test_a_commodity_does_not_spawn_its_own_brand_table(name, store):
"""These used to become brand_toor, brand_sugar, brand_black, brand_rice -
one junk table per leading word."""
assert _brand_of([name], store) == OWN_PRODUCTS_BRAND
# ---------------------------------------------------------------------------
# The property that keeps this from over-reaching
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("name,expected", [
("Amul Butter 500g", "amul"),
("Cadbury Dairy Milk Silk", "cadbury"),
("Britannia Good Day Cashew Cookies 200g", "britannia"),
("Tata Salt 1kg", "hindustan unilever"), # a real alias, not a misroute
("Colgate Active Salt 200g", "colgate-palmolive"),
])
def test_a_branded_product_keeps_its_own_brand(name, expected, store):
"""One token that is not a commodity means "this is a brand". Without this,
the fix would be worse than the bug."""
assert _brand_of([name], store) == expected
@pytest.mark.parametrize("name", [
"Bikaji Aloo Bhujia 200g", "Aachi Sambar Powder", "Idhayam Sesame Oil",
"Lion Dates 450g", "Naga Maida 2kg",
])
def test_a_brand_absent_from_the_alias_map_is_not_swallowed(name, store):
"""BRAND_ALIASES holds only 231 mostly-large FMCG names. These are real
brands missing from it, and "unknown" must never mean "unbranded"."""
assert _brand_of([name], store) != OWN_PRODUCTS_BRAND
def test_qualifiers_do_not_make_a_commodity_look_branded():
assert is_unbranded("Organic Toor Dal Whole 1kg")
assert is_unbranded("Premium Sortex Basmati Rice 5kg")
assert is_unbranded("Rock Salt 1kg")
def test_a_supplied_brand_always_wins():
assert not is_unbranded("Sugar 1kg", sheet_brand="Madhur")
def test_a_blank_brand_column_is_an_explicit_statement():
"""The sheet troubled itself to include the column and left it empty. This
was previously the one case rejected outright."""
assert is_unbranded("Anything At All", brand_column_supplied=True)
# ---------------------------------------------------------------------------
# The rest of the row: category and price
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("name,category", [
("Toor Dhal 1kg", "Pulses, Grains & Spices"),
("Sugar 1kg", "Sugar & Jaggery"),
("Salt 1kg", "Salt & Staples"),
("Black Pepper 100g", "Spices & Masalas"),
("Rice 5kg", "Atta & Staples"),
])
def test_a_commodity_resolves_to_a_real_category(name, category):
"""These all fell through to "General" before, which disabled HSN lookup and
the pack-size unit rulebook along with it."""
assert detect_category_from_text(name) == category
assert canonical_category(name) is not None
def test_the_category_carries_an_hsn_code(store):
"""The names were chosen to match HSN_GST_TABLE keys, so tax enrichment
works without a second mapping."""
_brand_of(["Toor Dhal 1kg"], store)
assert store[0][1]["hsn_code"] == "0713"
@pytest.mark.parametrize("name,size,ceiling", [
("Sugar 1kg", "1kg", 100), # was ₹120-280 under the "general" band
("Salt 1kg", "1kg", 60), # was ₹120-280
("Turmeric Powder 100g", "100g", 80),
])
def test_a_staple_is_no_longer_priced_like_branded_fmcg(name, size, ceiling):
lo, _hi = price_estimator.estimate_price_range_for_size(
size, name, OWN_PRODUCTS_BRAND, detect_category_from_text(name) or ""
)
assert lo < ceiling, f"{name} still prices at ₹{lo}, above the {ceiling} sanity ceiling"
def test_pepper_is_not_priced_like_a_cheap_staple():
"""The opposite failure: spices are genuinely expensive by weight and used
to come out at ₹20-45 per 100g."""
lo, _hi = price_estimator.estimate_price_range_for_size(
"100g", "Black Pepper 100g", OWN_PRODUCTS_BRAND, "Spices & Masalas"
)
assert lo > 50
def test_staples_get_staple_pack_sizes():
"""100g/250g/500g is not how dal, rice or salt is sold."""
assert price_estimator.default_size_variants("Pulses, Grains & Spices", "Toor Dal") == ["500g", "1kg", "5kg"]
assert price_estimator.default_size_variants("Salt & Staples", "Salt") == ["1kg"]
# ---------------------------------------------------------------------------
# Table naming and visibility
# ---------------------------------------------------------------------------
def test_the_bucket_round_trips_to_its_table_and_back():
"""The frontend passes the display name straight back to
/api/brands/{brand}/products, so this has to be an identity."""
from app.services.brand_registry import resolve_parent_brand
from app.services.vector_store import _sanitize_name, display_name_for_suffix
suffix = _sanitize_name(resolve_parent_brand(OWN_PRODUCTS_BRAND))
assert suffix == "own_products"
assert display_name_for_suffix(suffix) == OWN_PRODUCTS_BRAND
def test_the_bucket_survives_an_active_brands_whitelist(monkeypatch):
"""conftest blanks ACTIVE_BRANDS for the whole suite, so nothing else here
runs under the value production actually ships. ACTIVE_BRANDS is a
whitelist: without the bucket in it, the table is invisible everywhere."""
from app.services import active_brands
monkeypatch.setattr(active_brands, "_RAW_ACTIVE_BRANDS", "Amul,Cadbury,Own Products")
active_brands.invalidate()
try:
assert active_brands.is_active_suffix("own_products")
assert not active_brands.is_active_suffix("toor")
finally:
active_brands.invalidate()
# ---------------------------------------------------------------------------
# The migration for rows already misfiled
# ---------------------------------------------------------------------------
def test_the_migration_selects_only_unbranded_rows():
"""The safety property. It runs over the REAL brand catalogs looking for
misfiled commodities, so it must never pick up a genuine product."""
from scripts.migrate_own_products import is_unbranded as predicate
# These are the rows sitting in brand_nestle / brand_amul / brand_cadbury.
assert predicate("Butter 500g")
assert predicate("Milk 1L")
assert predicate("Salt 1kg")
# These are their legitimate neighbours in the same tables.
assert not predicate("Amul Butter 500g")
assert not predicate("Nestle Munch Chocolate")
assert not predicate("Cadbury Dairy Milk Silk 150g")
def test_a_migrated_row_is_rebuilt_for_its_new_home():
from scripts.migrate_own_products import TARGET_SUFFIX, _rebuild_row
rebuilt = _rebuild_row({
"product_name": "Red Chilli Powder 100g",
"size_variants": ["100g"],
"category": "General",
"image_id": "brooke_bond_red_chilli_powder_100g",
"image_url": "https://cdn.example.com/brooke_bond/x.jpg",
"image_urls": ["https://cdn.example.com/brooke_bond/x.jpg"],
"fssai_license": "10013022001897", # Brooke Bond's, not this product's
})
assert rebuilt["fssai_license"] is None, "another company's licence must not travel"
assert rebuilt["image_id"].startswith(TARGET_SUFFIX)
assert rebuilt["image_urls"] == [] and rebuilt["image_url"] is None
assert rebuilt["category"] == "Spices & Masalas"
assert rebuilt["price_range"].startswith("₹")