New updates on DB and JSON

This commit is contained in:
sriram
2026-09-01 13:55:15 +05:30
parent 183b65b3bd
commit 6c7a886659
20 changed files with 68753 additions and 135 deletions

View File

@@ -209,6 +209,11 @@ class BatchFileOut(BaseModel):
# the run that took it, which is how a sender gets from the drop id they
# hold to the batch that carries their results.
released_to: Optional[str] = None
# The other direction: which drop this file came out of. A run can be
# assembled from several drops, and this is the only exact way for a sender
# to pick their own file out of one - `filename` is a coincidence, because
# two senders can both upload products.csv.
from_drop: Optional[str] = None
stage_index: int = 0
stage_name: str = ""
total_stages: int = pipeline.TOTAL_STAGES

View File

@@ -408,6 +408,23 @@ def list_inbox() -> InboxOut:
return InboxOut(pending_count=pending_count, submissions=submissions)
def _stamp_origins(manifest, origins: List[str]) -> None:
"""Record which drop each file in a freshly staged run came from.
Done after staging, by position, for the same reason `stage_and_queue`
back-fills `rows_total` that way: the staging helpers take a
(filename, bytes, rows) tuple shared with the uploads router, and widening
it here would change a signature three callers depend on.
Safe by position because `from-inbox` stages with `invalid=[]`, so
manifest.files is exactly `picked` in order.
"""
for entry, drop_id in zip(manifest.files, origins):
entry.from_drop = drop_id
batch_ingest.write_manifest(manifest)
batch_job_store.put(manifest)
@router.post("/from-inbox", status_code=status.HTTP_202_ACCEPTED,
dependencies=[Depends(require_admin)])
def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
@@ -438,6 +455,11 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
# what parse_all returns, so nothing needs reparsing here.
picked: List[tuple] = []
senders: List[str] = []
# The drop each picked file came out of, in the SAME ORDER as `picked`, so
# it can be stamped onto the staged manifest below. Kept parallel rather
# than folded into the tuple because that tuple shape is shared with
# parse_all and with the uploads router.
origins: List[str] = []
for batch_id, indices in grouped.items():
manifest = batch_ingest.read_manifest(batch_id)
if not manifest or manifest.status != batch_ingest.PENDING:
@@ -454,6 +476,7 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
# will show it has left the inbox.
continue
picked.append((entry.filename, contents, entry.rows_total))
origins.append(batch_id)
if manifest.submitted_by:
senders.append(manifest.submitted_by)
@@ -483,6 +506,7 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
fetch_images=request.fetch_images,
submitted_by=submitted_by,
)
_stamp_origins(manifest, origins)
else:
manifest, started = batch_common.stage_and_queue(
picked,
@@ -491,6 +515,7 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
fetch_images=request.fetch_images,
submitted_by=submitted_by,
)
_stamp_origins(manifest, origins)
if not started:
raise HTTPException(
status_code=429,

View File

@@ -152,6 +152,19 @@ class BatchFile:
# the manifest because one drop can be released a few sheets at a time, into
# different runs, and the sender needs to know which of theirs went where.
released_to: Optional[str] = None
# The REVERSE of released_to: on a file inside a RUN, the id of the drop it
# was released from. Both directions are needed and they are not the same
# question - released_to answers "where did my drop go?", from_drop answers
# "whose file is this?".
#
# That second question is the one that can corrupt inventory. An admin may
# assemble one run from several drops, and until this existed the only way
# to narrow a run's manifest to your own file was to match on `filename` -
# so two senders who both upload `products.csv` would price and shelve each
# other's products, silently. Matching on this id is exact.
#
# None for a file uploaded straight into a run, which never sat in an inbox.
from_drop: Optional[str] = None
size_bytes: int = 0
status: str = QUEUED
detail: Optional[str] = None

View File

@@ -79,8 +79,16 @@ from app.services.category_registry import (
detect_category_from_text,
sanitize_category_language,
)
from app.services.category_units import fix_or_reject_size, parse_unit
from app.services.generic_products import OWN_PRODUCTS_BRAND, is_unbranded
from app.services.category_units import (
fix_or_reject_size,
parse_unit,
validate_unit_for_category,
)
from app.services.generic_products import (
OWN_PRODUCTS_BRAND,
canonical_category,
is_unbranded,
)
from app.services.embeddings_service import embed_texts
from app.services.enrichment.barcode.stage import BarcodeEnrichmentStage
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
@@ -318,6 +326,18 @@ def stage_3_title_category(row: Dict[str, Any]) -> Dict[str, Any]:
if _blank(category):
category = detect_category_from_text(f"{title} {row.get('description') or ''}")
if not category:
# The commodity lexicon, which is a category map as well as a
# detector - the same entry that recognises "Apple" as unbranded
# also says it is produce. The curated keyword registry above is
# tried first and covers dal, sugar and salt by name, but it has no
# entry for individual fruit or vegetables and never will: listing
# every one there would duplicate the lexicon and let the two drift.
#
# Only reached when keyword detection found nothing, so this can
# only turn "General" into something better, never overrule a
# curated answer.
category = _lexicon_category(title, row)
row["_category_deterministic"] = bool(category)
if not category:
# "General" rather than None: the column is not nullable in
@@ -371,6 +391,38 @@ def _is_unitless_number(size: str) -> bool:
return value is not None and not unit
def _lexicon_category(title: str, row: Dict[str, Any]) -> Optional[str]:
"""The commodity lexicon's category, unless it contradicts the pack size.
The lexicon is a category map as well as a detector - the entry that
recognises "Apple" as unbranded also says it is produce - so it is the
natural second opinion when keyword detection finds nothing.
But its answer is a HINT, and the pack size the merchant wrote is a FACT.
"Tea Powder 250g" is the case that proves it: the lexicon calls tea a
Beverage, the unit rulebook says beverages are measured in ml or litres
only, and stage 4 then "corrects" 250g to 250ml - silently turning a
quarter kilo of tea leaves into a quarter litre. Loose tea is a dry good
sold by weight; the category is what is wrong there, not the size.
So a category that cannot accommodate the size already on the row is
declined, and the row falls through to "General" exactly as it did before
the lexicon was consulted at all.
"""
category = canonical_category(title)
if not category:
return None
declared = [str(x).strip() for x in (row.get("size_variants") or []) if str(x).strip()]
match = _SIZE_IN_TITLE.search(title or "")
if match:
declared.append(match.group(0).strip())
for size in declared:
ok, _msg = validate_unit_for_category(size, category)
if not ok:
return None
return category
def _sizes_for(row: Dict[str, Any]) -> List[str]:
declared = [str(s).strip() for s in (row.get("size_variants") or []) if str(s).strip()]
sizes = [s for s in declared if not _is_unitless_number(s)]
@@ -386,6 +438,28 @@ def _sizes_for(row: Dict[str, Any]) -> List[str]:
match = _SIZE_IN_TITLE.search(row.get("product_name") or "")
if match:
return [match.group(0).strip()]
# A COMMODITY WITH NO WEIGHT COLUMN GETS ONE UNSIZED ROW, NOT THREE MADE-UP
# ONES. default_size_variants() invents a plausible set - 100g/250g/500g -
# and for packaged goods that is a reasonable guess at what a brand sells.
# For loose produce it is not: a shop sells apples by the kilo at whatever
# the customer asks for, so "Apple 250g" is a product that does not exist.
#
# It is also the mechanism behind the catalogue drift the integrator
# reported: the invented set is keyed on the resolved category, so the same
# product ingested twice with the category resolved differently produces
# two disjoint size sets, two sets of image_ids, and a re-scrape that looks
# like the old rows were deleted. Keeping produce out of that from the
# start is cheaper than repairing it later.
# "Standard" and not "": an empty size fails validate_size ("size/pack is
# missing") and stage 4 would drop the row, which is the rejection this
# whole area exists to prevent. "Standard" is also what _to_storage_row
# already substitutes for a blank size and what the seeded base list uses,
# so an uploaded "Apple" deduplicates onto the seeded "Apple" instead of
# creating a second row.
if row.get("brand") == OWN_PRODUCTS_BRAND:
return ["Standard"]
return list(price_estimator.default_size_variants(
row.get("category") or "", row.get("product_name") or ""
))
@@ -423,18 +497,37 @@ def stage_4_explode_sizes(row: Dict[str, Any]) -> Tuple[List[Dict[str, Any]], Li
# ---------------------------------------------------------------------------
# Stage 5 - Pricing bands
# ---------------------------------------------------------------------------
# How far either side of a sheet's own selling price the published band sits.
# Stated once, because it is a judgement about how much a real shelf price
# varies rather than a fact, and two call sites disagreeing about it would be
# invisible. 155 -> Rs143-167.
_SHEET_PRICE_BAND = 0.08
def stage_5_pricing(row: Dict[str, Any]) -> Dict[str, Any]:
"""Publish a price band, from the sheet's own price where there is one.
FOR A COMMODITY THE ESTIMATOR IS NOT CONSULTED. It is trained on packaged
FMCG and prices a 500g apple at Rs85-105, which is not wrong so much as
meaningless - loose produce is priced by the shop, by the day. A row with no
price keeps a null band rather than a confident fiction.
"""
size = row.get("size") or "Standard"
if _blank(row.get("price_range")):
if not _blank(row.get("final_selling_price")):
price = float(row["final_selling_price"])
lo, hi = int(round(price * 0.95)), int(round(price * 1.05))
else:
lo, hi = price_estimator.estimate_price_range_for_size(
size, row.get("product_name") or "", row.get("brand") or "",
row.get("category") or "",
)
row["price_range"] = f"₹{lo}-{hi}"
if not _blank(row.get("price_range")):
return row
if not _blank(row.get("final_selling_price")):
price = float(row["final_selling_price"])
lo = int(round(price * (1 - _SHEET_PRICE_BAND)))
hi = int(round(price * (1 + _SHEET_PRICE_BAND)))
elif row.get("brand") == OWN_PRODUCTS_BRAND:
return row
else:
lo, hi = price_estimator.estimate_price_range_for_size(
size, row.get("product_name") or "", row.get("brand") or "",
row.get("category") or "",
)
row["price_range"] = f"₹{lo}-{hi}"
return row
@@ -498,6 +591,13 @@ def stage_7_sku(row: Dict[str, Any]) -> Dict[str, Any]:
if _blank(row.get("sku_source")):
row["sku_source"] = "sheet"
return row
# A commodity gets no minted SKU. An internal SKU is an identifier for a
# specific packaged product from a specific brand; "OWN-APP-500" would name
# a thing that does not exist, and a shop's loose apples are not the same
# article as another shop's. The sheet's own column still wins above, so
# this is "do not invent", not "discard".
if row.get("brand") == OWN_PRODUCTS_BRAND:
return row
try:
resolved = resolve_product_sku(
row.get("brand") or "", row.get("product_name") or "", row.get("size") or ""
@@ -518,6 +618,16 @@ async def stages_8_9_enrichment(rows: List[Dict[str, Any]], brand: str) -> List[
Each disables itself via its settings flag, so this is a no-op when both
are off.
"""
# Neither stage runs for the own-products bucket. A barcode identifies a
# manufactured article and loose produce has none - the lookup would either
# find nothing or, worse, attach some packaged product's real GTIN. HSN is
# skipped for the same reason it is not invented anywhere else here: a tax
# code the merchant did not supply is our guess presented as their record.
# A sheet that DOES carry an hsn_code or barcode column keeps those values,
# because both stages only fill blanks.
if brand == OWN_PRODUCTS_BRAND:
return rows
stages = []
if ENABLE_BARCODE_LOOKUP:
stages.append(BarcodeEnrichmentStage())
@@ -567,8 +677,18 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
size = row.get("size") or "Standard"
display = name if size.lower() in name.lower() else f"{name} {size}".strip()
category = row.get("category") or "General"
description = row.get("description") or (
f"{display} from {row.get('brand')}."
own = row.get("brand") == OWN_PRODUCTS_BRAND
# "Apple 500g from Own Products." is a sentence nobody wrote and nobody
# wants, and "Own Products" is a bucket rather than a maker, so the
# template reads as a false provenance claim. A commodity keeps whatever
# description the sheet gave, including none.
description = row.get("description") or (None if own else f"{display} from {row.get('brand')}.")
# The embedding text. Interpolating a null description and the bucket name
# would embed the literal "Own Products ... None", so a commodity is
# described to the vector index by what actually identifies it.
search_query = (
f"{display} {category}".strip() if own
else f"{row.get('brand')} {display} {category} {description}"
)
return {
"product_name": display,
@@ -591,12 +711,13 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
"barcode_type": row.get("barcode_type"),
"highlights": list(row.get("highlights") or []),
"nutrients": list(row.get("nutrients") or []),
"search_query": f"{row.get('brand')} {display} {category} {description}",
"search_query": search_query,
}
def _record_product(result: PipelineResult, brand: str, row: Dict[str, Any],
disposition: str) -> None:
disposition: str,
source_rows: Optional[Dict[str, int]] = None) -> None:
"""Note the identity of one resolved catalog row.
`image_id` is the useful field here and the one to join on: it is the key
@@ -611,12 +732,23 @@ def _record_product(result: PipelineResult, brand: str, row: Dict[str, Any],
if len(result.products) >= MAX_REPORTED_PRODUCTS:
result.products_truncated = True
return
image_id = row.get("image_id")
result.products.append({
"image_id": row.get("image_id"),
"image_id": image_id,
"brand": brand,
# The key the catalogue is actually addressed by, published so a caller
# does not have to re-derive it from the display name. Getting that
# wrong is silent: "24 Mantra" guessed as "24mantra" simply finds
# nothing. This is the same function the storage layer uses.
"brand_key": _sanitize_name(brand),
"product_name": row.get("product_name"),
"product_sku": row.get("product_sku"),
"sku_source": row.get("sku_source"),
# The 1-based spreadsheet row this came from, header counted as row 1 -
# the number the sender sees on screen. One sheet row legitimately
# becomes several products (pack-size explosion), so this is many-to-one
# and is what lets a caller say "row 14 became these three".
"source_row": (source_rows or {}).get(image_id),
"disposition": disposition,
})
@@ -640,7 +772,8 @@ def _merge_with_existing(new: Dict[str, Any], existing: Dict[str, Any]) -> Tuple
return merged, changed
def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResult) -> None:
def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResult,
source_rows: Optional[Dict[str, int]] = None) -> None:
"""Embed and upsert, splitting inserts from backfills.
`cleanup=False` is load-bearing: cleanup=True deletes every row in the
@@ -666,7 +799,7 @@ def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResul
if prior is None:
to_write.append(row)
result.inserted += 1
_record_product(result, brand, row, "inserted")
_record_product(result, brand, row, "inserted", source_rows)
continue
merged, changed = _merge_with_existing(row, prior)
if changed:
@@ -675,10 +808,10 @@ def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResul
# The MERGED row: a backfill keeps the stored product_sku rather
# than the one this run minted, so reporting `row` would hand back
# an identifier that is not the one in the catalog.
_record_product(result, brand, merged, "backfilled")
_record_product(result, brand, merged, "backfilled", source_rows)
else:
result.skipped_existing += 1
_record_product(result, brand, prior, "unchanged")
_record_product(result, brand, prior, "unchanged", source_rows)
if not to_write:
return
@@ -838,6 +971,10 @@ def run_pipeline(
result.rejected += len(rejected)
for bad in rejected:
result.rejections.append({
# The sheet row the sender sees on screen. Without it the only
# way to find a refused product was to diff the manifest
# against the file and guess.
"row": bad.get("_row"),
"product_name": bad.get("product_name") or bad.get("title"),
"size": bad.get("size"),
"reason": "; ".join(
@@ -847,11 +984,27 @@ def run_pipeline(
})
progress(10, STAGE_NAMES[9], len(kept), len(rows))
storage_rows = [_to_storage_row(r) for r in kept]
# image_id -> the sheet row that produced it, carried ALONGSIDE the
# storage rows rather than inside them. `_to_storage_row` projects onto
# the brand-table columns and its output goes straight to
# upsert_brand_products, so smuggling a reporting-only key into that
# dict would push an unknown column at the database.
#
# Built from `kept` before the dedupe below, and last-wins in the same
# direction, so the row number always describes the product that was
# actually written.
source_rows: Dict[str, int] = {}
storage_rows = []
for enriched in kept:
stored = _to_storage_row(enriched)
storage_rows.append(stored)
if enriched.get("_row") is not None:
source_rows[stored["image_id"]] = enriched["_row"]
# A single sheet can name the same pack twice; last one wins, so the
# batch never presents two rows with the same image_id to the upsert.
deduped: Dict[str, Dict[str, Any]] = {r["image_id"]: r for r in storage_rows}
stage_11_store(list(deduped.values()), brand, result)
stage_11_store(list(deduped.values()), brand, result, source_rows)
progress(11, STAGE_NAMES[10], len(deduped), len(deduped))
return result

View File

@@ -252,6 +252,15 @@ BRAND_ALIASES = {
"sunfeast bounce": "sunfeast",
"sunfeast yippee": "sunfeast",
"sunfeast cookies": "sunfeast",
# SPELLING VARIANTS, not sub-brands.
#
# "Haldiram" and "Haldirams" were resolving to themselves, so nothing knew
# they were one brand and uploads built brand_haldiram and brand_haldirams
# side by side. The data was merged into the plural, which is the correct
# name; THIS LINE is what stops the split reappearing on the next sheet
# that spells it without the s. Removing it re-opens the bug.
"haldiram": "haldirams",
"haldiram's": "haldirams",
}
DEFAULT_ALIASES = BRAND_ALIASES
@@ -352,7 +361,14 @@ FSSAI_LICENSES: Dict[str, str] = {
"lion dates": "10012042000244",
"brooke bond": "10013022001897",
"mother dairy": "10012011000015",
# Keyed on BOTH spellings deliberately. get_fssai_license() resolves to the
# canonical parent first, so once "haldiram" aliases to "haldirams" the
# lookup arrives as "haldirams" - and with only the singular key here it
# would return None and every future Haldiram row would ship with no
# licence at all. A silent loss, since a blank licence is a legitimate
# outcome elsewhere and nothing would flag it.
"haldiram": "10012011000140",
"haldirams": "10012011000140",
"fortune": "10012021000071",
"paper boat": "10012043000083",
"bisk farm": "10012031000012",

View File

@@ -88,6 +88,22 @@ CATEGORY_REGISTRY: List[Dict[str, object]] = [
{"category": "Salt & Staples", "keywords": ["salt", "rock salt", "sea salt", "table salt", "iodised salt", "iodized salt"], "generic_term": "salt"},
{"category": "Atta & Staples", "keywords": ["atta", "wheat flour", "flour", "rice", "dal", "pulses", "staples", "suji", "maida"], "generic_term": "staple product"},
{"category": "Dairy", "keywords": ["milk", "dairy", "cheese", "paneer", "panner", "paner", "paneerr", "curd", "yogurt", "butter", "ghee", "dahi"], "generic_term": "dairy product"},
# ---- Loose, unbranded fresh goods -----------------------------------
# Added for the produce a grocer sells by weight or by the piece. These
# rows carry no brand, so they land in the Own Products table via
# generic_products.is_unbranded(); the categories exist so that HSN/GST
# resolution and the pack-size unit rules have something to key off,
# rather than falling through to "General".
#
# Keyword lists stay SHORT here on purpose. Detection for these rows comes
# from the commodity lexicon, which is far more specific; a broad keyword
# such as "fresh" or "leaf" would pull branded products in through
# keyword matching, which is the failure this whole area exists to avoid.
{"category": "Fruits & Vegetables", "keywords": ["fruits", "vegetables", "vegetable", "fresh produce", "loose produce"], "generic_term": "fresh produce"},
{"category": "Fresh Herbs & Greens", "keywords": ["herbs", "greens", "curry leaves", "coriander leaves", "mint leaves", "spinach", "keerai"], "generic_term": "fresh greens"},
{"category": "Flowers", "keywords": ["flowers", "flower", "garland", "jasmine flower", "loose flowers"], "generic_term": "flowers"},
{"category": "Fish & Seafood", "keywords": ["fish", "seafood", "prawns", "prawn", "shrimp"], "generic_term": "fresh fish"},
{"category": "Eggs", "keywords": ["eggs", "egg", "country egg"], "generic_term": "eggs"},
{"category": "Oral Care", "keywords": ["toothpaste", "toothbrush", "mouthwash", "paste"], "generic_term": "oral care product"},
{"category": "Hair Care", "keywords": ["shampoo", "shampooo", "conditioner", "hair oil"], "generic_term": "hair care product"},
{"category": "Bath Soap", "keywords": ["bath soap", "soap bar", "soap", "soaps"], "generic_term": "soap"},

View File

@@ -70,6 +70,17 @@ HSN_GST_TABLE: Dict[str, Tuple[str, int, bool]] = {
# Chapter 17: cane/beet sugar and jaggery, 5% for ordinary retail sugar.
"Sugar & Jaggery": ("1701", 5, False),
"Cooking Oils": ("1517", 5, False),
# ---- Loose fresh goods, sold by weight or by the piece ---------------
# Chapters 3, 4, 6, 7 and 8. Fresh, unprocessed produce is NIL-rated under
# GST - it is not a reduced rate, it is exempt - so 0 here is the real
# figure and not a placeholder. The moment any of these is branded and
# packaged in a unit container the rate changes, but that product would
# resolve to its brand's category rather than these.
"Fruits & Vegetables": ("0709", 0, False),
"Fresh Herbs & Greens": ("0709", 0, False),
"Flowers": ("0603", 0, False),
"Fish & Seafood": ("0302", 0, False),
"Eggs": ("0407", 0, False),
"Pickles & Chutneys": ("2001", 12, True),
"Dry Fruits & Nuts": ("0801", 12, True),
"Food - Spreads": ("2007", 12, True),

View File

@@ -74,6 +74,16 @@ _SALT = "Salt & Staples"
_OILS = "Cooking Oils"
_DAIRY = "Dairy"
_BEVERAGE = "Beverages"
# Fresh, loose goods. These are sold by weight or by the piece and carry no
# brand at all, which is precisely why they were the worst offenders: before
# these existed, "Apple" resolved to a brand called Apple and "Red Rose" was
# whole-word matched into brand_brooke_bond, carrying Brooke Bond's FSSAI
# licence onto a flower.
_PRODUCE = "Fruits & Vegetables"
_GREENS = "Fresh Herbs & Greens"
_FLOWERS = "Flowers"
_SEAFOOD = "Fish & Seafood"
_EGGS = "Eggs"
COMMODITY_TERMS: Dict[str, str] = {}
@@ -127,6 +137,79 @@ _add(_DAIRY,
_add(_BEVERAGE, "tea", "coffee", "chai")
# ---------------------------------------------------------------------------
# Fresh produce
# ---------------------------------------------------------------------------
# WHY THIS BLOCK IS SAFE TO ADD.
#
# The all-tokens-must-be-commodities rule means every word added here makes the
# test MORE permissive, so the risk is real brands collapsing into this bucket.
# That was measured, not assumed, before these went in: all 1,414 products in
# the live catalogue were reclassified with this list applied, and exactly one
# changed - `24 Mantra Organic Moong Dal 500g`, which was ALREADY misfiled by
# the `_strip_sizes` bug fixed below and has nothing to do with produce.
#
# The other check was against BRAND_ALIASES: of 113 candidate terms only "amla"
# appears in an alias ("dabur amla"), and that is harmless because "dabur" is
# not a commodity, so "Dabur Amla" keeps every token it needs to stay branded.
# Bare "Amla" is fruit and belongs here.
#
# RE-RUN BOTH CHECKS BEFORE ADDING A WORD TO THIS BLOCK. See
# tests/test_generic_products_produce.py, which encodes them.
# Fruit. Indian sheets mix English, Tamil and Hindi names freely.
_add(_PRODUCE,
"apple", "orange", "banana", "grape", "grapes", "mango", "pineapple",
"papaya", "guava", "pomegranate", "watermelon", "muskmelon", "melon",
"lemon", "lime", "mosambi", "sathukudi", "sapota", "chikoo", "jackfruit",
"fig", "anjeer", "pear", "peach", "plum", "apricot", "cherry", "cherries",
"strawberry", "blueberry", "kiwi", "litchi", "lychee", "amla", "gooseberry",
"avocado", "dragonfruit", "rambutan", "mangosteen", "starfruit", "jamun",
"ber", "plantain", "vazhaikkai")
# Vegetables. "gourd" covers the whole family once _PHRASES has collapsed the
# two-word forms (bitter gourd, bottle gourd, snake gourd, ridge gourd).
_add(_PRODUCE,
"tomato", "potato", "onion", "carrot", "beetroot", "beet", "radish",
"turnip", "cabbage", "cauliflower", "broccoli", "brinjal", "eggplant",
"aubergine", "okra", "bhindi", "cucumber", "pumpkin", "gourd", "drumstick",
"beans", "bean", "capsicum", "garlic", "ginger", "yam", "colocasia",
"tapioca", "arbi", "zucchini", "chayote", "sweetcorn", "babycorn",
"mushroom", "leek", "celery", "lettuce", "shallot", "springonion")
# Greens and fresh herbs, sold in bunches and never branded. NOTE that
# "coriander" and "methi" are deliberately NOT here: they are already spices in
# the block above, `_add` is last-wins, and re-adding them would silently move
# dhania powder out of Spices & Masalas. The leaf forms are collapsed to
# distinct tokens by _PHRASES instead.
_add(_GREENS,
"spinach", "palak", "amaranth", "keerai", "greens", "cilantro", "mint",
"pudina", "curryleaves", "basil", "thulasi", "tulsi", "parsley", "dill",
"sorrel", "moringa", "methileaves", "bunch")
# Flowers. Sold loose or by the metre of garland; the reason "Red Rose" used to
# land in a tea catalogue.
_add(_FLOWERS,
"flower", "flowers", "rose", "jasmine", "malli", "lotus", "marigold",
"samanthi", "chrysanthemum", "kanakambaram", "arali", "garland", "poo")
# Fish and seafood, sold fresh by weight.
_add(_SEAFOOD,
"fish", "prawn", "prawns", "shrimp", "crab", "squid", "tuna", "mackerel",
"sardine", "pomfret", "seer", "vanjaram", "anchovy", "nethili", "sole",
"tilapia", "salmon", "shellfish", "clam", "mussel")
# Eggs.
_add(_EGGS, "egg", "eggs", "muttai", "quail")
# Stragglers found by running the produce base list through is_unbranded and
# fixing every row it refused. Kept in one block so the next person adding to
# the seed list knows where the tail ends up.
_add(_PRODUCE, "dates", "custard", "dragonfruit", "ivy", "raw")
_add(_GREENS, "agathi", "ponnanganni", "keerai")
_add(_FLOWERS, "tuberose", "lily")
# Words that describe a product without naming a brand. Stripped before the
# all-tokens-are-commodities test, so "Organic Toor Dal Whole 1kg" still reads
# as unbranded.
@@ -150,6 +233,16 @@ QUALIFIERS: Set[str] = {
"bottle", "refill", "combo", "assorted", "mixed", "mix",
# connectives
"and", "with", "of", "the", "in", "for",
# Form words for fresh goods. "Leaves" is the important one: without it
# "Mint Leaves" keeps an unknown token and reads as a brand.
"leaves", "leaf", "bunch", "sweet", "broad", "cluster", "full", "toned",
"seedless", "ripe", "tender", "baby", "country", "hybrid", "nati",
# Varietal names. A variety qualifies a commodity, it does not brand it:
# an Alphonso mango is a mango. None of these appears in BRAND_ALIASES -
# the produce test asserts that, so a future addition cannot smuggle a
# real brand in through this list.
"robusta", "yelakki", "nendran", "alphonso", "banganapalli", "totapuri",
"malgova", "sindoora", "shimla", "ooty", "kashmiri",
}
# Multi-word commodities collapsed to a single token before tokenising, so the
@@ -175,16 +268,70 @@ _PHRASES = {
"brown sugar": "sugar",
"palm jaggery": "jaggery",
"cane sugar": "sugar",
# Fresh produce. The two-word gourds collapse onto "gourd" so the whole
# family is one lexicon entry. The leaf forms get their OWN tokens rather
# than reusing "coriander" / "methi": those are spices, _add is last-wins,
# and re-adding them under a greens category would silently move dhania
# powder out of Spices & Masalas.
"bitter gourd": "gourd",
"bottle gourd": "gourd",
"snake gourd": "gourd",
"ridge gourd": "gourd",
"ash gourd": "gourd",
"bitter guard": "gourd", # misspellings seen in real merchant data
"bottle ground": "gourd",
"lady finger": "okra",
"ladies finger": "okra",
"spring onion": "springonion",
"spring onions": "springonion",
"sweet potato": "potato",
"curry leaves": "curryleaves",
"curry leaf": "curryleaves",
"coriander leaves": "cilantro",
"methi leaves": "methileaves",
"fenugreek leaves": "methileaves",
"french beans": "beans",
"cluster beans": "beans",
"green peas": "peas",
"baby corn": "babycorn",
"sweet corn": "sweetcorn",
"tender coconut": "coconut",
"dragon fruit": "dragonfruit",
"custard apple": "apple",
"sweet lime": "mosambi",
"ivy gourd": "gourd",
"broad beans": "beans",
"cluster bean": "beans",
"quail egg": "egg",
"spring garlic": "garlic",
}
_WORD_RE = re.compile(r"[a-z]+")
# Units a pack size is actually written in. The strip below is bounded to these
# rather than to "any letters", because [a-z]* after a number ate the NEXT WORD:
# "24 Mantra Organic Moong Dal" became "organic moong dal", the brand was
# destroyed, and the row was then filed as an unbranded commodity. Every brand
# whose name begins with a number hit this. Keep the list tight - a unit added
# here is a word that can be deleted from a product name.
_UNITS = (
"kg|kgs|g|gm|gms|gram|grams|mg|ml|l|ltr|ltrs|litre|litres|liter|liters"
"|pc|pcs|piece|pieces|pack|packs|pkt|n|no|nos|x|cm|mm|inch|dozen"
)
_SIZE_RE = re.compile(
# "1kg", "500 g", "1.5 L" - a number followed by a REAL unit, optionally
# spaced - or a bare number, which is a quantity and never a brand.
r"\b\d+(?:[.,]\d+)?\s*(?:" + _UNITS + r")\b"
r"|\b\d+(?:[.,]\d+)?\b",
re.IGNORECASE,
)
def _strip_sizes(text: str) -> str:
"""Remove pack sizes and bare numbers - they never name a brand."""
# "1kg", "500 g", "1.5 L", and any leftover bare number.
text = re.sub(r"\b\d+(?:[.,]\d+)?\s*[a-z]*\b", " ", text)
return text
return _SIZE_RE.sub(" ", text)
def canonical_category(name: str) -> Optional[str]:

View File

@@ -54,6 +54,7 @@ from app.infrastructure.settings import (
from app.services.title_validator import find_category_conflicts
from app.services import price_estimator
from app.services import category_units as cu
from app.services.generic_products import OWN_PRODUCTS_BRAND
logger = logging.getLogger(__name__)
@@ -277,6 +278,26 @@ def validate_product(
sku_source = str(product.get("sku_source") or "").strip()
images = product.get("image_urls") or []
# A COMMODITY IS NOT A DEFECTIVE BRANDED PRODUCT.
#
# The penalties below treat a missing price_range or SKU as evidence that a
# row was fabricated, which is right for a scraped brand catalogue: a real
# Amul product has a shelf price and an article number, so their absence
# means something went wrong. Loose produce has neither, by nature. A shop
# prices apples by the day and does not issue article numbers for them.
#
# Left unqualified, the arithmetic rejected every produce row outright:
# 0.55 baseline - 0.30 (no price_range) - 0.10 (no SKU) = 0.15, against a
# reject threshold of 0.35. That is the exact opposite of the requirement
# these rows exist to satisfy, so the two absences stop counting as faults.
#
# EVERYTHING ELSE STILL APPLIES. Title sanity, placeholder detection,
# category resolution, the title/category contradiction check, size
# validity and unit compatibility, and the image-presence check all run
# unchanged - a blank or junk product name is still caught, and this is not
# a way in for rows that would otherwise fail.
commodity = brand == OWN_PRODUCTS_BRAND
report = ValidationReport(product_name=title or "(untitled)", grounded=grounded)
score = 0.55 # neutral baseline - moves up/down based on evidence below
@@ -345,16 +366,22 @@ def validate_product(
score -= 0.15
# 5. Price range ----------------------------------------------------------
ok, msg = validate_price_range(price_range, size, title, brand, category)
if not ok:
report.issues.append(ValidationIssue(field="price_range", severity="error", message=msg, penalty=0.30))
score -= 0.30
# Skipped for a commodity only when there is none. A band that IS present is
# still checked for being well formed, so a malformed one cannot hide here.
if price_range or not commodity:
ok, msg = validate_price_range(price_range, size, title, brand, category)
if not ok:
report.issues.append(ValidationIssue(field="price_range", severity="error", message=msg, penalty=0.30))
score -= 0.30
# 6. SKU --------------------------------------------------------------------
ok, msg = validate_sku(sku, sku_source)
if not ok:
report.issues.append(ValidationIssue(field="product_sku", severity="warning", message=msg, penalty=0.10))
score -= 0.10
# Same rule: a commodity is not expected to carry one, but a SKU the sheet
# did supply must still look like a SKU.
if sku or not commodity:
ok, msg = validate_sku(sku, sku_source)
if not ok:
report.issues.append(ValidationIssue(field="product_sku", severity="warning", message=msg, penalty=0.10))
score -= 0.10
# 7. Image presence -----------------------------------------------------
# Only meaningful if image search actually ran. When the operator disables

View File

@@ -0,0 +1,137 @@
{
"taken_at": "20260901_134907",
"db_host": "31.97.228.132",
"db_name": "pgvector",
"brand_haldiram": [
{
"id": 1,
"product_name": "Haldiram Aloo Bhujia 200g 45",
"title": "Haldiram Aloo 200g",
"description": "Haldiram Aloo Bhujia 200g 45 from Haldiram.",
"category": "3",
"image_id": "haldiram_haldiram_aloo_bhujia_200g_45",
"image_url": "https://bazaar-foods.co.uk/cdn/shop/products/haldirams-aloo-bhujia-200g.jpg?v=1649176712",
"image_urls": [
"https://bazaar-foods.co.uk/cdn/shop/products/haldirams-aloo-bhujia-200g.jpg?v=1649176712",
"https://www.thai-food-online.co.uk/cdn/shop/products/Haldirams-Aloo-Bhujia-200g-Front.png?v=1653396797",
"https://www.rashanpani.co.uk/wp-content/uploads/2020/05/HALDIRAMALOOBHUJIA200G.jpg",
"https://www.gobuzzaar.com/uploads/2025/06/800x800/haldirams-aloo-bhujia-200g.jpg",
"https://groceteria.eu/wp-content/uploads/2025/02/Haldirams-200g-Aloo-Bhujia.webp",
"https://www.citybazaar.dk/wp-content/uploads/2024/02/haldiram-aloo-bhujia.jpg",
"https://thekiranahaus.com/cdn/shop/files/Haldiram-s-200g-Aloo-Bhujia--wuerzige-Kartoffelsticks--16251_1_5b0f5681-4eff-4792-8c83-774d323946bd.webp?v=1765318184&width=1445",
"https://nisargafresh.nl/wp-content/uploads/2022/09/Aloo-Bhujia-200g-Haldiram-300x300.png",
"https://m.media-amazon.com/images/I/713pmpvY6VL._AC_SL1000_.jpg",
"https://cdn.shopify.com/s/files/1/0434/5475/9072/products/IS-132_clipped_rev_1_copy_1024x.jpg?v=1628238808"
],
"price_range": "₹52-58",
"size_variants": [
"45"
],
"providers": [],
"fssai_license": "10012011000140",
"product_sku": "HALD-BHUJ-200",
"sku_source": "",
"hsn_code": null,
"final_selling_price": "58",
"selling_price": "58",
"barcode": null,
"barcode_type": null,
"highlights": [],
"nutrients": [],
"search_query": "Haldiram Haldiram Aloo Bhujia 200g 45 3 Haldiram Aloo Bhujia 200g 45 from Haldiram.",
"embedding": "[-0.061470453,0.085923776,-0.050007723,-0.010388176,-0.105947,0.029684091,0.01018019,0.0326033,-0.017272232,-0.02223002,0.05238059,-0.13941023,0.056763377,-0.05565607,-0.0013915179,0.04291669,0.06696487,0.026800232,-0.04647914,-0.0416988,0.08339774,0.019954558,-0.009182672,-0.0025798376,-0.007229289,0.13358761,-0.033226427,0.055968896,0.037845407,-0.061806567,-0.0043716184,0.09443795,0.082649015,-0.07063394,0.02757287,-0.033491127,-0.08968844,-0.031277124,0.08357984,0.0482263,-3.4857999e-06,-0.0025698135,0.028261518,0.005129888,0.009810577,-0.0063438024,-0.08344995,0.10941526,0.059196167,-0.02535888,-0.016701974,0.02346376,-0.07723267,0.105881654,0.049159676,-0.103167415,0.008602405,0.028959755,-0.001431078,-0.034416527,-0.0676624,0.035875387,-0.02955854,-0.027848115,0.020283105,-0.027637335,-0.0808495,-0.060705267,-0.066662356,0.021860223,-0.005598716,-0.034734484,-0.034236483,-0.029325726,-0.018017743,-0.04256999,0.04801943,-0.08334282,-0.0764811,-0.04878591,-0.06523457,-0.06585633,0.12267185,-0.017186869,-0.018697215,-0.03347549,-0.014150048,0.12791106,-0.057740245,-0.108282715,-0.030115323,0.0066051325,-0.11242725,0.015669039,-0.092518106,0.06819647,-0.122968495,-0.05385696,-0.019065559,0.018783037,0.075349405,0.036990706,0.0058152927,-0.06598786,-0.07396621,0.020352127,0.027102727,0.13160394,0.037662864,-0.017217921,-0.066403694,0.02904227,-0.059101313,-0.0034248584,-0.00864348,-0.03633173,0.008295371,0.0021769737,-0.0057140547,0.019750657,0.007474019,-0.023771532,0.049534626,-0.029095244,-0.12592928,0.07232369,0.03378169,1.296634e-32,-0.050658148,-0.09196166,0.058936916,-0.032744236,0.018527597,-0.033474702,-0.04349935,0.017932698,-0.036908038,-0.028993832,-0.06068338,-0.037089925,0.028008202,0.023156805,-0.006812169,-0.10516624,-0.03014195,0.035480473,0.021186655,-0.01400949,-0.0077031157,0.049328536,-0.0054385876,0.044295203,0.03583454,0.06183502,0.08454213,-0.049978737,0.057355814,0.05139823,0.01397718,-0.027529268,-0.096242115,-0.12014253,-0.115006946,0.053672213,-0.0054810354,-0.081016235,-0.094351254,-0.05612986,0.044648796,-0.030642055,0.010459408,0.012553524,0.03095918,0.07615386,0.051000275,0.014667437,0.034062423,0.052176744,-0.031464193,-0.0123085305,-0.011492719,0.017208502,-0.030510992,-0.021482727,0.016208854,-0.002907258,0.00020148358,0.08722652,0.016213732,0.050492495,-0.013732973,0.021592563,-0.03134377,-0.023611384,-0.023446757,-0.07413654,-0.013767491,0.0033524255,-0.0075572,-0.06441154,0.09468595,0.061729066,-0.02178496,0.0010737815,0.024657287,0.030952245,0.02398675,0.0020069524,-0.045607194,0.059503548,0.03923495,-0.07371786,0.016165433,0.041368306,-0.045157075,-0.057643034,-0.0076029045,0.04802942,0.017770018,-0.003472389,0.046380565,0.025366355,-0.027365148,-1.17752656e-32,-0.028758632,0.07745132,-0.019914337,0.0186684,0.08845942,-0.0064726146,0.07080666,0.07155823,0.035890803,-0.015938815,-0.029325774,-0.02274504,0.115917765,-0.054575015,-0.057255,0.04734996,0.10425196,0.051324375,-0.040612098,-0.03783868,0.01610947,0.04325998,-0.0006286663,0.008529754,-0.04983215,0.112548664,0.051360626,0.0069579147,0.009742793,0.03564187,-0.012604597,-0.04643494,-0.05486593,0.032028537,-0.1032377,-0.010374628,0.0385542,0.035651542,-0.04998217,0.019538503,-0.029337378,0.0888773,0.019374222,0.0963841,0.0061803036,0.009116665,0.04528866,-0.050373435,0.016377108,-0.020969104,-0.0054394677,-0.019459903,-0.046234936,0.016407901,-0.026153658,0.014765041,-0.03968482,0.0021429295,-0.06024107,-0.008839775,-0.032153983,0.07577759,0.027919343,0.071250975,0.0623783,0.04449583,0.048923377,-0.052728355,-0.02196624,-0.023756275,-0.005193731,0.002633509,-0.07457847,-0.0049360064,-0.0067040683,0.048857424,-0.005255573,0.07660087,0.0026997577,-0.0016187229,0.031624213,0.00095052144,-0.0067311437,0.062699765,-0.080630675,-0.08166218,0.025550803,-0.03572945,-0.00601697,0.019851236,-0.044734333,0.068068944,0.054362986,0.038462523,0.046319842,-3.5680632e-08,-0.02125714,0.039479893,0.024287026,0.04016811,0.03498418,0.037565768,-0.025369272,0.006615806,0.01806187,0.0011729915,-0.017235735,-0.038094517,-0.039481185,0.04735565,-0.0062645734,-0.0132780345,0.07052318,-0.03832829,0.067541465,-0.055909988,0.011112473,-0.0035194275,0.13212267,0.013075932,0.011319353,0.069707625,-0.017551374,-0.02812701,0.08074076,0.041591056,0.020192059,0.04382235,-0.07242851,-0.060478747,0.030826215,-0.023221618,-0.0855683,0.039421275,0.010495891,0.011221291,0.014818017,-0.120077655,0.029326573,0.022608621,0.025708701,0.0058598258,-0.0735763,-0.021944927,-0.059000954,-0.09134795,-0.0029611024,-0.101805426,0.044474937,-0.0054603335,-0.04431912,0.042876992,-0.0657774,-0.05918801,0.022463147,0.017258102,0.037892565,-0.010327075,-0.06765019,0.044359624]",
"created_at": "2026-08-29T10:41:38.166944",
"updated_at": "2026-08-29T10:41:38.166944"
}
],
"brand_haldirams": [
{
"id": 1,
"product_name": "Haldiram's Aloo Bhujia 200g",
"title": "Haldiram's Aloo Bhujia 200g",
"description": "Crunchy and spicy potato-gram flour namkeen",
"category": "Snacks",
"image_id": "haldirams_haldiram_s_aloo_bhujia_200g",
"image_url": "https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg",
"image_urls": [
"https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg"
],
"price_range": "₹45-50",
"size_variants": [
"200g"
],
"providers": [
"Amazon",
"Flipkart",
"BigBasket",
"Jiomart",
"Blinkit",
"Zepto"
],
"fssai_license": "10012042000244",
"product_sku": "HALD-HALDIR-001",
"sku_source": "User Upload",
"hsn_code": "2106",
"final_selling_price": "48",
"selling_price": "48",
"barcode": "8900000000000.0",
"barcode_type": "GTIN-13",
"highlights": [
"100% Quality Assurance",
"Authentic Brand Product"
],
"nutrients": [
"Energy - High",
"Protein - Good Source"
],
"search_query": "Haldiram's Haldiram's Aloo Bhujia 200g Snacks Crunchy and spicy potato-gram flour namkeen ₹45-50",
"embedding": "[-0.073780574,0.014016445,-0.03283347,0.06779144,-0.07432287,0.013323582,0.057849504,-0.003455605,-0.02796043,-0.0106389765,0.054610476,-0.13733506,0.0011138533,-0.09461825,0.043265574,-0.0033051497,0.16251235,-0.021456327,-0.056616433,-0.043503813,0.015809868,-0.0027687962,0.055956308,-0.0020621475,-0.0002262986,0.09798358,0.052142963,-0.0065899845,-0.008076999,-0.04662511,0.075209744,0.12199015,0.046643093,-0.07255586,0.023547035,-0.032034293,0.02218585,-0.07860314,0.064440474,0.017679987,-0.017938236,0.037968524,0.051713504,-0.025717558,0.0026579762,-0.04317045,-0.04329217,0.10957813,0.04799027,-0.0094134435,-0.04176496,0.021880541,-0.02489868,0.053983495,0.10355603,-0.07522341,-0.11387932,0.027443936,-0.03329309,0.008200507,-0.025784984,0.031980727,0.009286136,-0.029458733,-0.010811541,-0.07233483,-0.059245557,-0.038007714,0.0024375066,0.004017135,-0.043354295,-0.0379747,0.04938866,-0.009079977,-0.031919453,0.0062903636,0.038413957,-0.06369559,-0.050876208,-0.08253325,-0.03543455,-0.013137344,0.11619364,0.021482123,-0.0041941865,-0.055790465,0.0027696376,0.059069995,-0.04857008,-0.021166429,0.038674932,-0.024962643,0.014518789,-0.04124445,-0.080964535,-0.0240311,-0.029613206,-0.08670562,-0.051875632,0.05894461,0.028114261,0.053660985,0.036084134,-0.09016684,-0.05446474,-0.019464912,0.07717981,-0.011869444,0.051274676,0.02189268,-0.06052555,0.049884107,-0.09407489,-0.037107907,-0.045468986,-0.07448858,0.042282812,0.0032387888,-0.04506364,0.025343671,-0.050813645,-0.02691568,0.043881804,-0.008280819,-0.10768411,0.021082247,-0.012938328,7.068499e-33,-0.03685851,-0.039287563,-0.00058218406,-0.03546077,0.039447866,-0.099279776,-0.008846175,-0.015827043,-0.004231832,-0.01240423,-0.0173825,0.034137066,-0.031023402,0.062806845,0.07871152,-0.08896062,-0.08321511,-0.03931041,0.045285314,-0.011240046,-0.07006219,0.0048550433,0.006457747,0.034788996,-0.01760184,0.027861014,0.065395735,-0.028560039,0.00248985,0.018866641,-0.011300588,0.004064385,-0.040849943,-0.11293159,-0.13535002,0.008115082,-0.056292925,-0.07768911,-0.0564176,-0.009952972,0.051836103,-0.028403815,-0.051030494,0.038016986,-0.025066279,0.075658865,0.08475662,0.082214326,0.10590234,0.02896686,-0.025536839,-0.030114923,-0.010633608,-0.0184912,-0.05289483,-0.02754835,0.04824575,-0.040299762,0.015501986,0.040012572,-0.036788717,-0.05353712,-0.048302356,-0.075477526,-0.060516905,0.008490575,-0.052708536,-0.04905377,-0.033386704,-0.008519541,-0.005193108,-0.01380934,0.12728089,0.0032967299,-0.039227586,0.008550332,0.060942084,0.031031711,-0.003867492,-0.006875357,0.11645594,0.035951044,0.078381166,-0.0020736938,-0.057762377,0.063063264,-0.08124302,-0.035704855,0.023189044,0.0152145,-0.039887406,-0.0055398573,0.06708767,-0.0057648136,-0.07640236,-6.7070056e-33,-0.03370923,0.014161425,-0.06586386,0.084878206,0.046127915,0.023844836,-0.02461334,0.017690562,0.00921369,-0.04313147,-0.030632516,-0.012535122,0.065278426,-0.013989146,-0.05026042,0.11251113,0.05700934,0.08111092,0.007674605,-0.059061382,-0.024890551,0.10317635,0.0149834575,-0.012122599,-0.049457733,0.09958608,0.06396433,0.033100173,-0.026297713,0.032261726,0.08667095,-0.055016927,-0.018163476,0.012644439,0.016452178,-0.0284256,0.0038589025,-0.051818937,-0.122482866,0.05270389,-0.030226449,0.083730906,0.01275159,0.05033225,0.037246265,-0.02044742,-0.033988904,-0.05296506,-0.037575576,-0.030798644,0.063701525,-0.00068997405,-0.023358572,-0.018957613,-0.037956595,0.07388675,-0.058028262,-0.002049744,0.027629329,-0.100966014,-0.050763145,0.07685107,0.025519853,0.05307841,0.05312479,0.023053985,0.029822761,-0.087285936,-0.017102415,0.00042710872,0.009520072,-0.019577894,0.030032566,0.0161792,0.01962172,0.061840374,0.03533059,-0.014051352,0.024355397,0.042371184,-0.008329303,0.021590685,-0.026294695,0.0076788454,-0.060583495,0.07875852,-0.030082814,0.02406768,-0.00927241,0.0993275,0.014351461,0.062653966,0.014678145,0.09492412,0.115223385,-3.1167954e-08,0.077919655,-0.04777391,-0.051864136,0.09117969,0.05721661,-0.040711176,-0.020052843,-0.021208424,0.03569595,0.008754088,0.008037574,0.03135558,-0.053007998,0.053251993,-0.06319788,-0.010215157,0.028881038,0.04924266,0.029694416,-0.003294222,0.006432629,0.03875308,0.12772492,-0.074748695,0.0014321478,0.043181334,0.011794961,0.021509338,0.08130951,0.100629106,0.00054375804,0.03646748,-0.011478553,-0.07050192,0.00770562,-0.0062742736,-0.029889021,0.023644177,-0.03701927,0.028838946,-0.026554547,-0.12876473,0.032152396,0.0502661,-0.057909496,-0.015366481,-0.046538085,0.047690097,0.0034577225,-0.024463559,-0.02055527,-0.002813917,-0.0027458656,-0.014018512,-0.05970947,0.03264415,-0.03840796,-0.02475397,0.00080424803,-0.02509683,0.013823303,-0.0044572027,-0.09458777,0.05007379]",
"created_at": "2026-08-17T15:37:15.414450",
"updated_at": "2026-08-17T15:37:15.414450"
},
{
"id": 2,
"product_name": "Haldiram's Moong Dal 200g",
"title": "Haldiram's Moong Dal 200g",
"description": "Roasted and salted moong dal namkeen",
"category": "Snacks",
"image_id": "haldirams_haldiram_s_moong_dal_200g",
"image_url": "https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg",
"image_urls": [
"https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg"
],
"price_range": "₹45-50",
"size_variants": [
"200g"
],
"providers": [
"Amazon",
"Flipkart",
"BigBasket",
"Jiomart",
"Blinkit",
"Zepto"
],
"fssai_license": "10012042000244",
"product_sku": "HALD-HALDIR-001",
"sku_source": "User Upload",
"hsn_code": "2106",
"final_selling_price": "48",
"selling_price": "48",
"barcode": "8900000000000.0",
"barcode_type": "GTIN-13",
"highlights": [
"100% Quality Assurance",
"Authentic Brand Product"
],
"nutrients": [
"Energy - High",
"Protein - Good Source"
],
"search_query": "Haldiram's Haldiram's Moong Dal 200g Snacks Roasted and salted moong dal namkeen ₹45-50",
"embedding": "[-0.094575,0.06647317,0.018450627,0.09411248,-0.11488417,0.0094742095,0.10323204,-0.008308582,-0.018837677,-0.06068788,0.03635123,-0.098427385,-0.011098394,-0.08241818,-0.010680773,-0.08192029,0.096883185,0.0061954088,-0.058705177,-0.07856562,0.0020511616,-0.035402957,0.026109403,0.011536509,0.02497377,0.102277346,0.044439584,0.004150674,-0.045279127,-0.06597435,0.046088085,0.13454252,0.019104136,-0.014138215,-0.018770406,0.0016899393,0.037570734,-0.101482116,0.033911843,-0.024027782,-0.007123305,0.0043425863,0.03389706,-0.01171643,-0.0010247155,-0.026938196,-0.07531733,0.047449447,0.048601817,0.028115522,-0.05155453,0.032397907,-0.030602388,0.018949037,0.084327325,-0.037579127,-0.09632338,0.03052131,-0.023420155,-0.018819075,-0.058096565,0.04801071,-0.019584091,0.00840557,-0.014928658,-0.017625613,-0.05028415,-0.007010133,0.0035170382,-0.017748946,-0.0073538446,-0.024471723,-0.0017027381,-0.011953686,-0.041689005,-0.020850446,0.06313427,-0.067269444,-0.02779477,-0.079179004,-0.040808536,-0.009283012,0.062841676,0.01776416,-0.017664485,-0.038390763,0.0077870428,0.086355746,-0.0071483236,-0.036428194,0.041351836,0.0010660643,-0.043088205,-0.047883313,-0.06397132,-0.014597856,-0.038109757,-0.11806625,-0.03859069,0.06741478,0.09975801,0.091468096,0.022662511,-0.099400885,0.01808026,-0.07444516,0.05801442,0.020548489,0.005871151,0.031390473,-0.08609821,0.0685131,-0.048832435,-0.0009881549,-0.053029515,-0.03519214,-0.02444342,-0.002871041,-0.00079397764,0.008004174,-0.044126116,-0.032528643,0.018218463,-0.002613153,-0.13646005,-0.021690182,-0.010388374,5.1170974e-33,-0.037991114,-0.057875294,0.045896087,-0.029799264,0.03445652,-0.05492394,0.021679293,0.009985053,-0.04079306,-0.029813424,-0.063487954,0.015532548,-0.016924227,0.04721671,0.090319164,-0.06784286,-0.0605672,-0.014278195,0.05632942,-0.0095763765,-0.118342355,0.06277701,0.05509074,-0.024498325,0.0067808125,0.04671988,0.029352784,-0.020681981,0.021213375,0.054673143,0.009987924,-0.051140882,-0.001783095,-0.068780065,-0.120357536,0.084589906,-0.05895755,-0.09710861,-0.040292494,-0.0063073607,0.106736794,0.014908648,-0.0042924844,0.015698465,-0.055473812,0.08482299,0.054842465,0.10652771,0.020141818,0.059941553,-0.038201205,0.0036418291,-0.030693544,-0.018320644,-0.03457125,0.03745144,0.023243146,0.010407767,0.009686064,0.016844263,0.011497691,-0.004155571,-0.045655962,-0.08559727,-0.042167526,0.032703206,-0.11219652,-0.062466234,-0.050595324,-0.00826258,-0.0026475135,-0.04215606,0.10538657,0.04092127,0.017664399,0.041831847,0.04067998,0.016462628,-0.038579937,0.020880423,0.079678945,0.061415456,0.06346388,-0.0011568575,-0.013694989,0.0623573,-0.0461171,-0.028040573,0.0042170635,0.04478849,-0.04387031,-0.0057746936,0.030305725,-0.023072846,-0.025389213,-5.8312274e-33,-0.029279016,-0.028358413,-0.030364979,0.07388548,0.03441022,-0.013458687,0.0008307939,0.020595325,-0.0071057896,0.0048316387,-0.031197425,-0.005186759,0.09885664,-0.027137637,-0.066331014,0.07685916,0.06953698,0.066769645,-0.0031445064,-0.027278978,-0.038401123,0.13919948,0.03546529,0.08209664,-0.065055154,0.08680524,0.055690266,0.030268267,-0.03396688,-0.015301493,0.095488615,-0.0673101,-0.025902545,-0.034853,-0.017013576,-0.04080425,0.054977503,-0.07056256,-0.08512563,0.05852524,-0.025794232,0.023131687,0.0027184545,0.032675695,0.0768045,-0.06114638,7.459798e-05,-0.08546178,-0.024186937,0.0039223493,0.028259924,-0.03900481,-0.008595077,0.01217486,-0.0051869825,0.029366544,-0.046122774,0.019728474,0.03202166,-0.12575525,-0.03417364,0.0764442,0.050919365,0.05531786,0.036432143,-0.017693507,0.018790327,-0.13322152,-0.032080054,-0.047055557,0.022200573,-0.04082469,0.004283893,-0.023007419,0.016834922,0.07362733,-0.009972964,0.052465674,0.067914724,0.0224166,-0.02975218,0.013401601,-0.03840001,0.05775581,-0.07493675,0.0201357,0.0088814525,0.031209635,-0.039559595,0.033402905,0.0014653425,0.007815599,-0.01766598,0.0919281,0.116744734,-2.5746983e-08,0.07138476,-0.057019565,-0.040646307,0.09265247,0.01658529,-0.000989908,0.0042248284,0.0018876027,0.024829473,0.018251775,0.031974934,0.023970272,-0.06986896,-0.0051021436,-0.09206029,-0.0005921235,0.016616385,0.028401058,0.013909974,-0.018079655,0.06514544,0.046897646,0.104065046,-0.013080918,0.00300962,0.026130179,0.060950138,0.015441117,0.055578504,0.04622945,-0.008451027,0.03546806,-0.017733773,-0.093754,0.04323589,-0.01374034,-0.002944189,0.04646882,0.041333135,0.06198348,-0.044462144,-0.13388309,0.009989284,0.040997624,-0.053453274,0.05901287,-0.0037693249,0.06316173,-0.036486328,-0.052355066,0.010952625,0.0035248504,-0.004493483,-0.07502327,-0.05658603,0.06537711,-0.011206859,-0.014927725,-0.03594883,0.007979808,0.039871667,-0.064994566,-0.1093809,0.0676764]",
"created_at": "2026-08-17T15:37:15.870055",
"updated_at": "2026-08-17T15:37:15.870055"
}
]
}

View File

@@ -1,93 +1,93 @@
{
"brand": "haldirams",
"search_query": "Haldirams products catalog",
"generation_timestamp": "C:\\Brand_Catalog_LLM\\RAG_Model_Nutrition_Intelligence\\RAG_Model_Full_Implement\\backend\\app\\services\\brand_sync.py",
"total_products": 2,
"total_images": 2,
"products": [
{
"brand": "Haldirams",
"brand_name": "Haldirams",
"product_name": "Haldiram's Moong Dal 200g",
"title": "Haldiram's Moong Dal 200g",
"description": "Roasted and salted moong dal namkeen",
"category": "Snacks",
"image_id": "haldirams_haldiram_s_moong_dal_200g",
"image_url": "https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg",
"image_urls": [
"https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg"
],
"price_range": "₹45-50",
"size_variants": [
"200g"
],
"providers": [
"Amazon",
"Flipkart",
"BigBasket",
"Jiomart",
"Blinkit",
"Zepto"
],
"fssai_license": "10012042000244",
"product_sku": "HALD-HALDIR-001",
"sku_source": "User Upload",
"hsn_code": "2106",
"final_selling_price": 48.0,
"selling_price": 48.0,
"barcode": "8900000000000.0",
"barcode_type": "GTIN-13",
"highlights": [
"100% Quality Assurance",
"Authentic Brand Product"
],
"nutrients": [
"Energy - High",
"Protein - Good Source"
],
"search_query": "Haldiram's Haldiram's Moong Dal 200g Snacks Roasted and salted moong dal namkeen ₹45-50"
},
{
"brand": "Haldirams",
"brand_name": "Haldirams",
"product_name": "Haldiram's Aloo Bhujia 200g",
"title": "Haldiram's Aloo Bhujia 200g",
"description": "Crunchy and spicy potato-gram flour namkeen",
"category": "Snacks",
"image_id": "haldirams_haldiram_s_aloo_bhujia_200g",
"image_url": "https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg",
"image_urls": [
"https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg"
],
"price_range": "₹45-50",
"size_variants": [
"200g"
],
"providers": [
"Amazon",
"Flipkart",
"BigBasket",
"Jiomart",
"Blinkit",
"Zepto"
],
"fssai_license": "10012042000244",
"product_sku": "HALD-HALDIR-001",
"sku_source": "User Upload",
"hsn_code": "2106",
"final_selling_price": 48.0,
"selling_price": 48.0,
"barcode": "8900000000000.0",
"barcode_type": "GTIN-13",
"highlights": [
"100% Quality Assurance",
"Authentic Brand Product"
],
"nutrients": [
"Energy - High",
"Protein - Good Source"
],
"search_query": "Haldiram's Haldiram's Aloo Bhujia 200g Snacks Crunchy and spicy potato-gram flour namkeen ₹45-50"
}
]
"brand": "haldirams",
"search_query": "Haldirams products catalog",
"generation_timestamp": "C:\\Brand_Catalog_LLM\\RAG_Model_Nutrition_Intelligence\\RAG_Model_Full_Implement\\backend\\app\\services\\brand_sync.py",
"total_products": 2,
"total_images": 2,
"products": [
{
"brand": "Haldirams",
"brand_name": "Haldirams",
"product_name": "Haldiram's Moong Dal 200g",
"title": "Haldiram's Moong Dal 200g",
"description": "Roasted and salted moong dal namkeen",
"category": "Snacks",
"image_id": "haldirams_haldiram_s_moong_dal_200g",
"image_url": "https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg",
"image_urls": [
"https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg"
],
"price_range": "₹45-50",
"size_variants": [
"200g"
],
"providers": [
"Amazon",
"Flipkart",
"BigBasket",
"Jiomart",
"Blinkit",
"Zepto"
],
"fssai_license": "10012011000140",
"product_sku": "HALD-HALDIR-001",
"sku_source": "User Upload",
"hsn_code": "2106",
"final_selling_price": 48.0,
"selling_price": 48.0,
"barcode": "8900000000000.0",
"barcode_type": "GTIN-13",
"highlights": [
"100% Quality Assurance",
"Authentic Brand Product"
],
"nutrients": [
"Energy - High",
"Protein - Good Source"
],
"search_query": "Haldiram's Haldiram's Moong Dal 200g Snacks Roasted and salted moong dal namkeen ₹45-50"
},
{
"brand": "Haldirams",
"brand_name": "Haldirams",
"product_name": "Haldiram's Aloo Bhujia 200g",
"title": "Haldiram's Aloo Bhujia 200g",
"description": "Crunchy and spicy potato-gram flour namkeen",
"category": "Snacks",
"image_id": "haldirams_haldiram_s_aloo_bhujia_200g",
"image_url": "https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg",
"image_urls": [
"https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg"
],
"price_range": "₹52-58",
"size_variants": [
"200g"
],
"providers": [
"Amazon",
"Flipkart",
"BigBasket",
"Jiomart",
"Blinkit",
"Zepto"
],
"fssai_license": "10012011000140",
"product_sku": "HALD-HALDIR-001",
"sku_source": "User Upload",
"hsn_code": "2106",
"final_selling_price": 58.0,
"selling_price": 58.0,
"barcode": "8900000000000.0",
"barcode_type": "GTIN-13",
"highlights": [
"100% Quality Assurance",
"Authentic Brand Product"
],
"nutrients": [
"Energy - High",
"Protein - Good Source"
],
"search_query": "Haldiram's Haldiram's Aloo Bhujia 200g Snacks Crunchy and spicy potato-gram flour namkeen ₹45-50"
}
]
}

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,323 @@
# Response to the Catalogue Drift Report
Reply to the seven findings dated 31 August 2026 (drop
`8e1448e176d843d08183d387ad724f95` → run `0ed4c77b0ca14e03b1aaf6b5b77d1994`).
Every figure below was measured against the same live deployment, not read off
source. Where we disagree with a finding, the evidence is included so you can
check it rather than take our word for it.
**Summary:** four items are fixed and ship in the next backend deploy. One
(#02) was already in the API and we had failed to document it — that is our
fault and the docs are now corrected. #01 is diagnosed, and the cause is not
what either of us assumed. #06 is confirmed but carries a trap that means we
should agree an approach before touching it.
| # | Finding | Status |
| --- | --- | --- |
| 01 | Pack sizes replaced between scrapes | **Diagnosed** — three causes, not one. Fix needs your input |
| 02 | `rejected` is a bare count | **Already shipped, now documented** — plus `row` added |
| 03 | Manifest brands are not catalogue keys | **Fixed** — `brand_key` published |
| 04 | Run files carry no `from_drop` | **Fixed** |
| 05 | Manifest carries no `source_row` | **Fixed** |
| 06 | Duplicate brands and products | **Confirmed.** Read the trap below before we act |
| 07 | No loose-produce coverage | **Fixed** — 159-row base list, and the upload path now handles produce |
---
## 01 — Pack sizes and names are replaced between scrapes
You asked us to confirm whether a pack size that once existed is meant to
survive a re-scrape. **It is not, today** — but that is only the last of three
causes, and fixing it alone would not have saved your links.
### Cause 1: pack sizes are invented when a scrape does not supply them
`_sizes_for()` falls back to `default_size_variants(category, name)` when a
product declares no size. That fallback is **keyed on the resolved category**,
and the resolved category is not stable between runs. Measured today:
```
category "Snacks" -> ['55g', '150g', '200g']
category "" (unresolved) -> ['100g', '250g', '500g']
category "Breakfast Cereal" -> ['250g', '500g', '1kg']
```
Now compare your table. Cheetos id 25 is **55g** — the *Snacks* set. Ids 26 and
27 are **100g** and **250g** — the *unresolved* set. The same product was
ingested once with its category resolved and once without, and produced two
disjoint sets of pack sizes. Your Cheerios ids (100g, 250g, 500g) sit across the
Breakfast Cereal set and the unresolved set the same way.
So the pack sizes were never scraped facts that changed. Some of them were
generated, and the generator's input moved.
### Cause 2: the product name gains or loses a brand prefix
Your own #06 has the evidence: `Hot Heads 30g` became `Nestle Hot Heads`, and
`PepsiCo Kurkure Masala Munch 90g` coexists with `Kurkure Masala Munch 90g`.
`image_id` is derived from the name, so a prefix appearing or disappearing moves
the id even when the product is identical.
### Cause 3: the write then deletes whatever is not in the new set
The brand-scrape path calls `upsert_brand_products(..., cleanup=True)`, which
deletes every row in the table whose `image_id` is absent from the batch being
written. That is what turns causes 1 and 2 from "duplicate rows" into "the row
you stored is gone".
**Any one of these is survivable. Together they guarantee broken links on every
re-scrape**, which matches your finding that zero of eleven could be repaired.
### Not the upload path
Worth stating plainly, because it affects how much you need to worry: the
**upload** path — everything reached through `POST /api/uploads/catalog` — uses
`cleanup=False` and has always done so. A sheet you send can never delete a row
it does not mention. The deletions came from brand scraping only.
### What we need from you
The real fix is to stop causes 1 and 2 (do not invent sizes for a product
already in the catalogue; settle the naming convention), and to soft-retire
rather than delete for cause 3. That third part changes how the production
catalogue is written and we would rather agree it with you than spring it:
- Would a `retired_at` timestamp plus exclusion from the default read work for
you, instead of the row being deleted? That preserves the `image_id` so your
stored link resolves to something, and lets us give you the `superseded_by`
and per-run changelog you asked for.
- If so, do you want retired rows visible through an explicit query, or gone
from the API entirely?
### Your two direct questions
**Is `image_id` stable across re-scrapes for a product whose name and pack size
have not changed?** Yes. It is a pure deterministic function of brand, product
name and pack size, with no clock, counter or run id in it. Verified:
```
build_image_id('pepsico', 'Cheetos Chips', '100g') -> pepsico_cheetos_chips_100g
```
Storing it rather than our row id is the right call and we have documented the
guarantee so it does not quietly change. **The caveat is #06:** the guarantee is
only as good as the stability of the name, and inconsistent brand prefixing
breaks exactly that.
**Do you want to know when a product is dropped or renamed?** Yes, and we agree
it should exist. It falls out of the retirement model above rather than being a
separate feature, which is why we would like to settle that first.
---
## 02 — `rejected` is a count with no reason
**This is our documentation failure, not a missing feature.** `rejections[]` has
been in every response — single-batch read and list endpoint both, since
`to_out(slim=True)` strips only `products` — carrying `product_name`, `size` and
`reason` per refused row. It was absent from `INGESTION_API.md`, which documents
`"rejected": 0` and never mentions the array, so there was no way for you to
know it was there. Sorry — that is a straightforwardly bad docs bug.
The genuine gap was the row number, which is now added:
```jsonc
"rejections": [
{ "row": 7, "product_name": "Kurkure Menthol", "size": "10g",
"reason": "title is too short to be a real product name; image_urls: no images were found for this product" }
]
```
`row` is the 1-based sheet row with the header counted as row 1 — the same
convention as the `422` responses, so it matches what the operator sees on
screen. `null` only when the row cannot be located. Capped at 50 per file.
---
## 03 — Manifest brands are not catalogue keys
Fixed. Every entry in `products[]` now carries `brand_key` beside `brand`:
```jsonc
{ "brand": "24 Mantra", "brand_key": "24_mantra", ... }
```
This is generated by the same function the storage layer uses to name the table,
so it cannot drift from the key the catalogue is actually addressed by. Your
normalisation is correct as far as we can tell, but it is a guess, and the
failure mode is silent — a wrong key finds nothing rather than erroring.
Thank you for degrading rather than failing the batch on an unreadable brand;
that is the right behaviour and we should have done it on our side too.
---
## 04 — Run files carry no `from_drop`
Fixed, and we agree with your assessment that this was the one item that could
corrupt a merchant's inventory rather than merely inconvenience you.
Each file in a run now carries `from_drop`, the id of the drop it was released
from — the exact inverse of `released_to`:
```jsonc
"files": [
{ "index": 0, "filename": "products.csv", "from_drop": "8e1448e1...", ... },
{ "index": 1, "filename": "products.csv", "from_drop": "a91c02f4...", ... }
]
```
`null` for a file that went straight into a run without sitting in an inbox —
which, under `UPLOAD_AUTORUN=true`, is every file you send, because the id you
are handed is already the run.
There is a test in our suite that stages two drops from different senders both
named `products.csv` and asserts they are distinguishable, so the collision you
described is now a permanent regression guard rather than a hope.
---
## 05 — Manifest cannot be traced back to the spreadsheet row
Fixed. Every entry in `products[]` carries `source_row`, the 1-based sheet row
with the header as row 1.
It is deliberately **many-to-one**: a pack-size cell reading `100g, 200g, 500g`
becomes three products that all report the same `source_row`, which is what lets
you say "row 14 of your sheet became these three". Rows that produced nothing
are those absent from every entry — "rows 6 and 11 produced nothing" is now a
set difference rather than a name-matching heuristic.
---
## 06 — Duplicate brands, duplicate products, stray names
Confirmed against the live database today: **55 brand tables, 1,414 products**
(you counted 1,614; the difference is a day of drift plus, we think, your count
including rejected rows — worth reconciling if it matters).
```
haldiram 1 britannia 6
haldirams 2 parle 3 against hindustan_unilever 443
patanjali 3
```
So: the split brand is real, and Britannia/Parle/Patanjali do look like scrapes
that stopped part-way rather than genuinely small brands. We will re-run those
three.
### The trap, which is why we have not just fixed this
**De-duplicating the PepsiCo pairs means renaming a product, and `image_id` is
derived from the name.** Renaming `PepsiCo Kurkure Masala Munch 90g` to
`Kurkure Masala Munch 90g` does not merge the two rows — it mints a third id and
breaks any link pointing at either of the first two. You have just migrated onto
storing `image_id`. A well-meant cleanup on our side would re-break exactly what
you have finished repairing.
The same applies to stripping the stray `150` from `Lays Classic Salted 52g 150`.
So before we touch it we would like to agree:
1. **Which convention wins** — brand prefix in the product name, or not? We have
no strong preference; we care only that it is one of them. Our lean is
*without* the prefix, since the brand is already a column.
2. **How the merge is communicated.** If we can give you the old-id →
new-id mapping for every row we touch, in advance, does that let you
re-point rather than clear? That is straightforward for us to produce.
3. **Timing**, so it lands in one pass rather than trickling.
`haldiram` → `haldirams` is a three-row merge and much lower risk; we can do
that one immediately if you would rather not wait for the rest.
---
## 07 — Loose produce has no coverage
Fixed, and this turned out to be the most valuable finding in your report,
because it was not only a coverage gap.
### What was actually happening
Produce rows were not rejected. They were **misfiled**, which is worse. The
brand fallback takes the first word of the name and then whole-word matches it
against our alias map:
```
Apple -> brand "Apple" -> junk table brand_apple
Tomato -> brand "Tomato" -> junk table brand_tomato
Bitter Gourd -> brand "Bitter" -> junk table brand_bitter
Curry Leaves -> brand "Curry" -> junk table brand_curry
Red Rose -> brand "Red" -> brand_brooke_bond <--
```
That last one is not a typo. A rose was being written into the Brooke Bond tea
catalogue, and our enrichment then stamps that brand's real FSSAI licence number
onto the row. Your 139 hand-typed products were the visible symptom; this was
underneath it.
### What now happens
Loose goods are recognised as commodities and filed under a single house brand,
`Own Products` (table `brand_own_products`), before brand inference can touch
them. Fruit, vegetables, greens, herbs, flowers, fish, eggs and loose dairy are
covered, alongside the pulses, grains, spices, oils and sugar that already were.
Five categories were added — Fruits & Vegetables, Fresh Herbs & Greens, Flowers,
Fish & Seafood, Eggs — with HSN codes and a 0% GST rate, since unprocessed
produce is nil-rated rather than reduced-rate.
Merchant misspellings from your own data are handled: `Bitter guard`,
`Bottle ground`, `Ladies Finger` all resolve.
### The base list
**159 rows**, in the shape you asked for: name and category only, no brand, no
pack size, no price. Fruit (42), vegetables (53), greens and herbs (17), flowers
(14), fish and seafood (15), loose dairy (13), eggs (5). It includes the specific
items your audit listed — Jasmine, Lotus, Red Rose, Thulasi, Drumstick, Curry
Leaves, the four banana varieties, Tuna, Mackerel.
Each row carries a `search_query` embedding, so these are reachable through
semantic search and not just exact match. The list is hand-authored rather than
scraped, so it is not subject to any of #01.
**Images are not included yet.** You asked for name and image; we have shipped
the names. Sourcing 159 licensable produce photographs is a separate piece of
work and we did not want to hold the list for it — tell us if the list is not
useful to you without them and we will prioritise accordingly.
### One limitation worth knowing
The classifier is deliberately conservative: a single word it does not recognise
means "this is a brand". So place-qualified produce — `Salem Mango`,
`Mysore Banana`, `Jammu Apple`, all real strings from your Ragul Stores data —
still reads as branded, because `Mysore` is also a real brand (Mysore Sandal).
The workaround is already in the pipeline: **if the sheet has a Brand column and
leaves the cell empty, we believe it** and file the row under Own Products
regardless of the name. If your merchants' sheets carry an empty brand column,
those rows will land correctly. If they carry no brand column at all, the
name-based test is what applies.
We would rather be conservative here. Collapsing a real regional brand into the
unbranded bucket is much harder to undo than a mango sitting in the wrong table.
---
## What we verified before sending this
- The produce lexicon was run against **all 1,414 products in the live
catalogue** and against all **231 brand aliases**: zero reclassifications in
either. That check is now a test, so it runs on every change.
- It also surfaced a pre-existing bug we would not otherwise have found: our
pack-size stripper was eating the word after a number, so
`24 Mantra Organic Moong Dal 500g` lost its brand entirely and was being filed
as an unbranded commodity. **Every brand whose name starts with a digit hit
this.** "24 Mantra" appears on your finding-03 list, which is how we noticed.
Fixed.
- All 159 seeded rows were checked to classify identically to how an uploaded
copy of the same name would, so a grocer typing "Tomato" lands on the seeded
row instead of creating a second one.
- Full suite: **1,054 tests passing.**

View File

@@ -176,7 +176,7 @@ The bad file is kept as a failed member rather than dropped, so a sender who sub
"brands": [],
"files": [
{ "index": 0, "filename": "catalog.csv", "status": "queued",
"total_stages": 11, "rows_total": 1, "size_bytes": 57 },
"from_drop": null, "total_stages": 11, "rows_total": 1, "size_bytes": 57 },
{ "index": 1, "filename": "notes.txt", "status": "failed",
"detail": "The file has no data rows." }
],
@@ -184,6 +184,22 @@ The bad file is kept as a failed member rather than dropped, so a sender who sub
}
```
#### `from_drop` — which file in this run is yours
**Match on this, never on `filename`.** An admin can assemble one run from several
drops, so a run's `files` may contain sheets you did not send — and two senders can
both upload `products.csv`. Matching on the name is a coincidence; matching on
`from_drop` is exact.
| Value | Meaning |
| --- | --- |
| the drop id you were given | this file is the one you sent in that drop |
| `null` | the file went straight into a run and never sat in an inbox — under `UPLOAD_AUTORUN=true` that is every file, and the run id you hold is already the only id involved |
It is the exact inverse of `released_to`, which points from your drop to the run
that took it. Both are needed: `released_to` answers *where did my drop go*,
`from_drop` answers *whose file is this*.
`use_llm` and `fetch_images` are reported, never accepted. They decide how much
outbound work a run commits the host to, and this endpoint's caller is anonymous, so
they come from settings — sending them in the request has no effect.
@@ -333,19 +349,112 @@ catalogue:
"rows_total": 2, "inserted": 1, "backfilled": 0, "skipped_existing": 1,
"rejected": 0, "brands": ["amul"],
"products": [
{ "image_id": "amul_amul_butter_100g", "brand": "amul",
"product_name": "Amul Butter 100g",
{ "image_id": "amul_amul_butter_100g", "brand": "amul", "brand_key": "amul",
"product_name": "Amul Butter 100g", "source_row": 2,
"product_sku": "ACME-BUT-100", "sku_source": "sheet",
"disposition": "inserted" },
{ "image_id": "amul_amul_ghee_1l", "brand": "amul",
"product_name": "Amul Ghee 1L",
{ "image_id": "amul_amul_ghee_1l", "brand": "amul", "brand_key": "amul",
"product_name": "Amul Ghee 1L", "source_row": 3,
"product_sku": "AMUL-GHE-1-001", "sku_source": "Internal",
"disposition": "unchanged" }
],
"products_truncated": false
"products_truncated": false,
"rejections": [
{ "row": 7, "product_name": "Kurkure Menthol", "size": "10g",
"reason": "title is too short to be a real product name; image_urls: no images were found for this product" }
]
}
```
#### `image_id` — the join key, and what it is stable against
`image_id` is a pure deterministic function of **brand, product name and pack size**.
Re-sending an unchanged sheet produces byte-identical ids, which is what makes the
pipeline idempotent, and it is the column the catalogue deduplicates on. Store it
rather than a row id.
What it is *not* stable against is any change to those three inputs. A pack size
moving from `100g` to `250g` is a different SKU at a different price and is correctly
a different id; so is a product name gaining or losing a brand prefix
(`Hot Heads` vs `Nestle Hot Heads`). If a name is rewritten upstream, the id moves
with it.
#### `brand_key` — the key the catalogue is addressed by
`brand` is the display name; `brand_key` is the identifier the catalogue is keyed
on, and the two are not the same string:
| `brand` | `brand_key` |
| --- | --- |
| `24 Mantra` | `24_mantra` |
| `Paper Boat` | `paper_boat` |
| `coca-cola` | `coca_cola` |
| `Own Products` | `own_products` |
Use `brand_key` rather than normalising `brand` yourself. The rule (lower-case,
non-alphanumerics to underscores) is stable, but deriving it is a guess and the
failure is silent — a wrong key finds nothing rather than erroring.
#### Unbranded rows — what `Own Products` means, and what it does not fill in
A row with no brand — loose fruit, vegetables, greens, flowers, fish, or staples
like dal and sugar — is filed under the display brand **`Own Products`**
(`brand_own_products`) rather than having a brand guessed from its first word.
**For these rows we store what your sheet said and nothing more.** A shopkeeper
bills from this record, so a tax code or article number we invented would be our
guess wearing your letterhead:
| Field | For an unbranded row |
| --- | --- |
| `product_name`, `size_variants` | from your sheet |
| `selling_price`, `final_selling_price` | from your sheet |
| `price_range` | a ±8% band around your price. Null if you sent no price |
| `category` | derived from the product name — deterministic, not guessed |
| `image_url`, `image_urls` | searched for, as with any other row |
| `hsn_code`, `product_sku`, `barcode`, `fssai_license`, `description` | **null**, unless your sheet supplied them |
Anything you *do* send is kept: a sheet with its own HSN, SKU or description
column keeps all three. The rule is "we do not invent", not "we discard".
Two consequences worth planning for:
- **No pack-size explosion.** A branded row with no size gets a plausible set
(100g/250g/500g); an unbranded one does not, because a shop sells apples by
whatever the customer asks for. A produce row with no weight column produces
exactly one entry, with size `Standard`.
- **`validation_status` is often `needs_review`.** For these rows that reflects
a missing image, not a suspect product — the absence of price or SKU is no
longer counted against them. `needs_review` rows are stored like any other;
only `rejected` rows are dropped.
#### `source_row` — which line of the sheet produced this
The 1-based row number as the sender sees it on screen, **header counted as row 1**,
so the first data row is `2`. The same convention the `422` responses use.
This is **many-to-one**: a pack-size cell reading `100g, 200g, 500g` legitimately
becomes three products, and all three carry the same `source_row`. Rows that produced
nothing are the ones absent from every entry — which is how you tell a shopkeeper
"rows 6 and 11 produced nothing".
#### `rejections[]` — which rows were refused, and why
`rejected` is a count; `rejections` is the explanation, and it has always been sent.
One entry per refused row:
| Field | Meaning |
| --- | --- |
| `row` | The 1-based sheet row, same convention as `source_row`. `null` if the row could not be located |
| `product_name` | The name as the sheet gave it |
| `size` | The pack the refusal applies to, since one row can yield several |
| `reason` | Every validation issue, joined with `; ` |
Capped at 50 entries per file. Present on both the single-batch read and the list
endpoint — only `products` is dropped from list responses.
#### `disposition`
| `disposition` | What happened |
| --- | --- |
| `inserted` | New product, created by this run |

392
scripts/merge_haldiram.py Normal file
View File

@@ -0,0 +1,392 @@
#!/usr/bin/env python3
"""
Fold `brand_haldiram` into `brand_haldirams` and drop the singular table.
WHY THERE WERE TWO
------------------
Nothing told the system they were one brand. `BRAND_ALIASES` had no haldiram
entry, so `resolve_parent_brand` was the identity for both spellings and each
upload built whichever table its sheet happened to name. That has been fixed in
`app/services/brand_registry.py`; this script cleans up the rows the split left
behind. **Deploy the alias first** - with it in place, an upload arriving while
this runs lands in `brand_haldirams` instead of recreating the singular behind
us.
WHAT IS ACTUALLY BEING MERGED
-----------------------------
Not a missing product. `brand_haldiram` holds ONE row which is a corrupted
duplicate of a product already in `brand_haldirams`:
brand_haldiram 'Haldiram Aloo Bhujia 200g 45' category '3' size ['45']
brand_haldirams "Haldiram's Aloo Bhujia 200g" Snacks ['200g']
The stray "45", the raw category id and the bare-number size all predate guards
that now exist. Moving that row across would put a second, worse Aloo Bhujia
inside the good table - duplication made worse rather than fixed. So the row is
dropped, and only what it holds that the target does NOT is carried over:
* the PRICE. The corrupted row is 12 days newer (29 Aug vs 17 Aug) and says
58 / Rs52-58 against the target's 48 / Rs45-50. A more recent upload is the
better evidence of what the shop charges.
* the FSSAI LICENCE, under --fix-fssai (off by default; see below).
THE LICENCE PROBLEM
-------------------
Both `brand_haldirams` rows carry 10012042000244. That is LION DATES' licence -
the same number on all 21 Lion Dates products - not Haldiram's. Haldiram's own,
per FSSAI_LICENSES, is 10012011000140, which is what the corrupted row has. So
the junk row holds the correct regulatory identifier and the clean rows hold
another company's.
Shipping one manufacturer's licence on another's product is the same class of
fault that `generic_products.py` exists to prevent, but repairing it is a
separate decision from merging two tables - so it is behind `--fix-fssai` and
off unless asked for. The seed file
`data/seed_catalogs/archive/brand_catalog_haldirams.json` carries the same wrong
number and is corrected in the same pass.
THE TABLE COMES BACK IF YOU ONLY DROP IT
----------------------------------------
`/app/data` is a named Docker volume and `reconcile_brand_catalogs()` exports
any table that has rows but no seed file, on boot and every 300s. Production has
very likely already written `brand_catalog_haldiram.json` into that volume. Left
there it is at best a stale file claiming to be a brand catalogue, so step 5
deletes it and reports whether it existed.
Usage:
python -m scripts.merge_haldiram # dry run, the default
python -m scripts.merge_haldiram --apply
python -m scripts.merge_haldiram --apply --fix-fssai
`--dry-run` is the default and `--apply` must be explicit: this rewrites
whatever database `backend/.env` points at, which is production. The target host
is printed on startup so it can be checked before committing, and both tables
are backed up to disk before the first write.
"""
from __future__ import annotations
import argparse
import json
import logging
import sys
from datetime import datetime
from decimal import Decimal
from pathlib import Path
from typing import Any, Dict, List, Optional
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.infrastructure.settings import DB_HOST, DB_NAME
from app.services.vector_store import _connect
logging.basicConfig(level=logging.INFO, format="%(message)s")
logger = logging.getLogger("merge_haldiram")
SOURCE_TABLE = "brand_haldiram"
TARGET_TABLE = "brand_haldirams"
TARGET_IMAGE_ID = "haldirams_haldiram_s_aloo_bhujia_200g"
SOURCE_IMAGE_ID = "haldiram_haldiram_aloo_bhujia_200g_45"
# Haldiram's own, from brand_registry.FSSAI_LICENSES.
HALDIRAM_FSSAI = "10012011000140"
# What the target rows wrongly carry today: Lion Dates'.
WRONG_FSSAI = "10012042000244"
ROOT = Path(__file__).resolve().parents[1]
SEED_DIR = ROOT / "data" / "seed_catalogs"
STALE_SEED = SEED_DIR / "brand_catalog_haldiram.json"
TARGET_SEED = SEED_DIR / "archive" / "brand_catalog_haldirams.json"
# ---------------------------------------------------------------------------
# Reading
# ---------------------------------------------------------------------------
def _table_exists(cur, table: str) -> bool:
cur.execute(
"SELECT 1 FROM information_schema.tables "
"WHERE table_schema='public' AND table_name=%s",
(table,),
)
return cur.fetchone() is not None
def _rows(cur, table: str) -> List[Dict[str, Any]]:
cur.execute(f"SELECT * FROM {table} ORDER BY id")
cols = [d[0] for d in cur.description]
return [dict(zip(cols, r)) for r in cur.fetchall()]
def _jsonable(row: Dict[str, Any]) -> Dict[str, Any]:
"""Make a DB row writable as JSON - datetimes and vectors are not."""
out = {}
for key, value in row.items():
if isinstance(value, datetime):
out[key] = value.isoformat()
elif key == "embedding" and value is not None:
out[key] = list(value) if not isinstance(value, str) else value
else:
out[key] = value
return out
def _backup(source: List[Dict], target: List[Dict]) -> Path:
"""Write both tables to disk BEFORE anything is changed. This is the undo."""
stamp = datetime.now().strftime("%Y%m%d_%H%M%S")
path = ROOT / "data" / f"haldiram_merge_backup_{stamp}.json"
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(
json.dumps(
{
"taken_at": stamp,
"db_host": DB_HOST,
"db_name": DB_NAME,
SOURCE_TABLE: [_jsonable(r) for r in source],
TARGET_TABLE: [_jsonable(r) for r in target],
},
ensure_ascii=False,
indent=1,
default=str,
),
encoding="utf-8",
)
return path
# ---------------------------------------------------------------------------
# Preconditions
# ---------------------------------------------------------------------------
def _check(source: List[Dict], target: List[Dict]) -> Optional[str]:
"""Refuse to run against data that is not what this script was written for.
The merge is hand-derived from one specific pair of rows. If either side has
moved since, the reasoning above may no longer hold and a blind UPDATE could
overwrite something real - so stop and let a human look.
"""
if len(source) != 1:
return f"{SOURCE_TABLE} has {len(source)} rows, expected exactly 1"
if source[0].get("image_id") != SOURCE_IMAGE_ID:
return (f"{SOURCE_TABLE} row is {source[0].get('image_id')!r}, "
f"expected {SOURCE_IMAGE_ID!r}")
if not any(r.get("image_id") == TARGET_IMAGE_ID for r in target):
return f"{TARGET_TABLE} has no row {TARGET_IMAGE_ID!r} to merge into"
return None
# ---------------------------------------------------------------------------
# Reporting
# ---------------------------------------------------------------------------
def _describe(source: Dict, target: Dict, fix_fssai: bool) -> None:
logger.info("")
logger.info(" %s -> %s", SOURCE_TABLE, TARGET_TABLE)
logger.info("")
logger.info(" row being DROPPED (corrupted duplicate):")
logger.info(" product_name %r", source.get("product_name"))
logger.info(" category %r size %s",
source.get("category"), source.get("size_variants"))
logger.info(" price %s (%s)",
source.get("selling_price"), source.get("price_range"))
logger.info("")
logger.info(" row being KEPT and updated:")
logger.info(" product_name %r", target.get("product_name"))
logger.info(" category %r size %s",
target.get("category"), target.get("size_variants"))
logger.info(" price %s (%s) -> %s (%s)",
target.get("selling_price"), target.get("price_range"),
source.get("selling_price"), source.get("price_range"))
if fix_fssai:
logger.info(" fssai %s -> %s",
target.get("fssai_license"), HALDIRAM_FSSAI)
else:
logger.info(" fssai %s (unchanged - pass --fix-fssai to "
"correct it; see the docstring)", target.get("fssai_license"))
logger.info(" name / category / size / image / description / embedding"
" all kept as they are")
logger.info("")
# ---------------------------------------------------------------------------
# The seed file
# ---------------------------------------------------------------------------
def _plain(value: Any) -> Any:
"""psycopg returns NUMERIC as Decimal, which json.dumps refuses.
Caught the hard way: the database work committed and then the seed write
blew up on `final_selling_price`, leaving the two halves out of step. The
conversion happens here rather than at the call site so every field copied
into a JSON file goes through it.
"""
return float(value) if isinstance(value, Decimal) else value
def _update_seed(live_rows: List[Dict], fix_fssai: bool, apply: bool) -> None:
"""Bring the archived catalogue into step with the TABLE.
Mirrored from the live rows rather than from the row being merged in, for
two reasons. It is idempotent - re-running compares the file against the
database and changes only what differs - and it still works once the source
table has been dropped, which matters because the database write and this
file write are not in one transaction. They came apart once already: the
merge committed and then the JSON write failed on a Decimal, leaving the two
halves disagreeing until this ran again.
"""
if not TARGET_SEED.exists():
logger.info(" seed file %s not found - skipping", TARGET_SEED.name)
return
by_id = {r["image_id"]: r for r in live_rows}
doc = json.loads(TARGET_SEED.read_text(encoding="utf-8-sig"))
changes: List[str] = []
for product in doc.get("products", []):
row = by_id.get(product.get("image_id"))
if not row:
continue
for field in ("price_range", "selling_price", "final_selling_price"):
new = _plain(row.get(field))
if product.get(field) != new:
changes.append(f"{product['image_id']}.{field}: "
f"{product.get(field)!r} -> {new!r}")
product[field] = new
if fix_fssai and product.get("fssai_license") != row.get("fssai_license"):
changes.append(f"{product['image_id']}.fssai_license: "
f"{product.get('fssai_license')!r} -> "
f"{row.get('fssai_license')!r}")
product["fssai_license"] = row.get("fssai_license")
if not changes:
logger.info(" %s: already in step with the table", TARGET_SEED.name)
return
logger.info(" %s: %d field(s) differ from the table", TARGET_SEED.name, len(changes))
for line in changes:
logger.info(" %s", line)
if apply:
TARGET_SEED.write_text(
json.dumps(doc, ensure_ascii=False, indent=1), encoding="utf-8")
logger.info(" written")
# ---------------------------------------------------------------------------
def main() -> int:
parser = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
)
parser.add_argument("--apply", action="store_true",
help="commit the changes (default is a dry run)")
parser.add_argument("--dry-run", action="store_true",
help="explicit no-op; this is already the default")
parser.add_argument("--fix-fssai", action="store_true",
help="also replace Lion Dates' licence on the Haldirams "
"rows with Haldiram's own")
args = parser.parse_args()
apply = args.apply and not args.dry_run
logger.info("Target database: %s / %s", DB_HOST, DB_NAME)
logger.info("Mode: %s", "APPLY - this writes" if apply else "DRY RUN - nothing is written")
conn = _connect()
if conn is None:
logger.error("No database connection.")
return 1
with conn:
with conn.cursor() as cur:
merged_already = not _table_exists(cur, SOURCE_TABLE)
if merged_already:
# The table is gone, but the seed files may still be behind -
# they are written outside the database transaction, so a run
# that committed the merge and then failed on the file leaves
# exactly this state. Fall through to the file work.
logger.info("")
logger.info("%s is already gone; checking the seed files.",
SOURCE_TABLE)
target_rows = _rows(cur, TARGET_TABLE)
source_rows = []
else:
source_rows = _rows(cur, SOURCE_TABLE)
target_rows = _rows(cur, TARGET_TABLE)
problem = None if merged_already else _check(source_rows, target_rows)
if problem:
logger.error("")
logger.error("ABORTING - the data is not what this script expects:")
logger.error(" %s", problem)
logger.error("")
logger.error("Re-read the rows and update the script rather than "
"forcing it; a blind UPDATE here could overwrite a "
"real product.")
return 1
if not merged_already:
source = source_rows[0]
target = next(r for r in target_rows
if r["image_id"] == TARGET_IMAGE_ID)
_describe(source, target, args.fix_fssai)
if apply and not merged_already:
path = _backup(source_rows, target_rows)
logger.info(" backup written to %s", path)
cur.execute(
f"UPDATE {TARGET_TABLE} SET selling_price=%s, "
f"final_selling_price=%s, price_range=%s, updated_at=now() "
f"WHERE image_id=%s",
(source.get("selling_price"), source.get("final_selling_price"),
source.get("price_range"), TARGET_IMAGE_ID),
)
logger.info(" updated %s (%d row)", TARGET_TABLE, cur.rowcount)
if args.fix_fssai:
cur.execute(
f"UPDATE {TARGET_TABLE} SET fssai_license=%s, "
f"updated_at=now() WHERE fssai_license=%s",
(HALDIRAM_FSSAI, WRONG_FSSAI),
)
logger.info(" corrected fssai on %d row(s)", cur.rowcount)
cur.execute(f"DROP TABLE {SOURCE_TABLE}")
logger.info(" dropped %s", SOURCE_TABLE)
conn.commit()
# Re-read so the seed mirrors what was actually committed.
target_rows = _rows(cur, TARGET_TABLE)
elif not merged_already:
logger.info(" would UPDATE %s then DROP TABLE %s",
TARGET_TABLE, SOURCE_TABLE)
# ---- the seed files ----------------------------------------------------
logger.info("")
_update_seed(target_rows, args.fix_fssai, apply)
if STALE_SEED.exists():
logger.info(" %s EXISTS (auto-exported by reconcile) - deleting",
STALE_SEED.name)
if apply:
STALE_SEED.unlink()
logger.info(" deleted")
else:
logger.info(" %s not present locally", STALE_SEED.name)
logger.info(" NOTE: production keeps data/ on a named volume, so check "
"there too - reconcile exports any table with rows and no file.")
# ---- caches ------------------------------------------------------------
if apply:
from app.services.vector_store import invalidate_brand_overview_cache
invalidate_brand_overview_cache()
try:
from app.services.query_intent import invalidate_brand_mention_cache
invalidate_brand_mention_cache()
except Exception: # noqa: BLE001 - a cold cache is not a failure
pass
logger.info("")
logger.info("Done. Brand caches invalidated.")
else:
logger.info("")
logger.info("Dry run - nothing written. Re-run with --apply to commit.")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -16,7 +16,11 @@ from pathlib import Path
import pytest
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
from app.services.brand_registry import (
BRAND_ALIASES,
get_fssai_license,
resolve_parent_brand,
)
from app.services.brand_sync import seed_catalog_paths
from app.services.vector_store import _sanitize_name
@@ -84,6 +88,13 @@ EXPECTED_SEED_TABLES = {
"parle": "brand_parle",
"pepsico": "brand_pepsico",
"tata": "brand_hindustan_unilever",
# The loose-produce base list: fruit, vegetables, greens, flowers, fish and
# loose dairy, none of which has a brand. It is the one seed catalog that
# was hand-authored rather than scraped, and it MUST land in
# brand_own_products - the same table generic_products.OWN_PRODUCTS_BRAND
# sends unbranded upload rows to, so an uploaded "Apple" deduplicates
# against the seeded one instead of creating a second row.
"Own Products": "brand_own_products",
}
@@ -112,6 +123,38 @@ def test_seed_catalogs_keep_their_current_tables() -> None:
assert actual == EXPECTED_SEED_TABLES
def test_both_haldiram_spellings_reach_one_table() -> None:
"""The regression that produced two tables for one brand.
Neither spelling was in BRAND_ALIASES, so resolve_parent_brand was the
identity for both and every upload built whichever table its sheet happened
to name - brand_haldiram (1 row) beside brand_haldirams (2). The rows were
merged into the plural, which is the correct name, so there is no longer a
singular table to assert against. This pins the PROPERTY instead, which is
what actually stops it recurring.
"""
from app.services.vector_store import _sanitize_name
for spelling in ("Haldiram", "haldiram", "HALDIRAM", "Haldiram's", "haldirams"):
table = f"brand_{_sanitize_name(resolve_parent_brand(spelling))}"
assert table == "brand_haldirams", f"{spelling!r} routed to {table}"
def test_the_haldiram_licence_survives_the_alias() -> None:
"""FSSAI_LICENSES has to be keyed on the parent, not the alias.
get_fssai_license resolves to the canonical parent before looking up, so
once "haldiram" aliases to "haldirams" a table keyed only on the singular
returns None - and a blank licence is a legitimate outcome elsewhere, so
nothing would flag it. Every future Haldiram row would simply ship without
one.
"""
assert get_fssai_license("Haldiram") == "10012011000140"
assert get_fssai_license("Haldirams") == "10012011000140"
# Not Lion Dates', which is what the stored rows wrongly carried.
assert get_fssai_license("Haldirams") != get_fssai_license("lion dates")
def test_known_sub_brands_still_route_to_their_family() -> None:
"""Word-boundary matching must not break legitimate sub-brand routing."""
assert resolve_parent_brand("Dove") == "hindustan unilever"

View File

@@ -326,3 +326,171 @@ def test_a_reupload_does_not_renumber_an_existing_products_sku(client, admin_hea
second = _run_one()
assert first["product_sku"] == second["product_sku"]
# ---------------------------------------------------------------------------
# 4. Reconciling a run against the sheet that produced it
# ---------------------------------------------------------------------------
# Four fields an integrator asked for after wiring the upload path end to end.
# Each answers a question that previously had only an approximate answer:
#
# source_row which of MY rows produced this product, and which produced none
# brand_key what key is the catalogue addressed by, given a display name
# row which rows were refused, and why
# from_drop which file in this run is mine, when a run spans several drops
#
# All four are ADDITIVE. Nothing existing changed meaning, because a consumer
# reading the old fields must keep working across this deploy.
def test_a_product_names_the_sheet_row_it_came_from(client, admin_headers, store):
"""source_row is the number the sender sees on screen, header counted as 1."""
rows = [["Amul Butter 100g", "Butter", "Amul"],
["Amul Ghee 1L", "Ghee", "Amul"]]
drop_id = client.post(UPLOAD, files=_files(("a.csv", _csv(rows=rows)))
).json()["batch_id"]
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
headers=admin_headers).json()["batch_id"]
products = batch_ingest.run_batch(run_id).files[0].result["products"]
by_name = {p["product_name"]: p["source_row"] for p in products}
# Row 1 is the header, so the first data row is 2.
assert by_name["Amul Butter 100g"] == 2
assert by_name["Amul Ghee 1L"] == 3
def test_one_sheet_row_can_own_several_products(client, admin_headers, store):
"""Pack-size explosion is many-to-one, and that is the point of the field.
"100g, 200g, 500g" in one cell is three catalog rows at three prices. The
sender needs to be able to say "row 2 of your sheet became these three"
rather than reconstruct it by matching names.
"""
headers = ["Product Name", "Category", "Brand", "Pack Size"]
rows = [["Amul Butter", "Butter", "Amul", "100g; 200g; 500g"]]
drop_id = client.post(
UPLOAD, files=_files(("a.csv", _csv(headers=headers, rows=rows)))
).json()["batch_id"]
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
headers=admin_headers).json()["batch_id"]
products = batch_ingest.run_batch(run_id).files[0].result["products"]
assert len(products) > 1, "the sheet did not explode; test proves nothing"
assert {p["source_row"] for p in products} == {2}
def test_a_product_carries_the_key_the_catalogue_is_addressed_by(client,
admin_headers,
store):
"""brand_key beside the display brand.
The manifest reports "24 Mantra" while the catalogue is keyed 24_mantra, and
an integrator normalising that themselves is guessing. Publishing the key
removes a whole class of "Unknown brand" failure.
"""
from app.services.vector_store import _sanitize_name
drop_id = _drop(client, "priya")
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
headers=admin_headers).json()["batch_id"]
product = batch_ingest.run_batch(run_id).files[0].result["products"][0]
assert product["brand_key"] == _sanitize_name(product["brand"])
assert " " not in product["brand_key"]
def test_a_run_says_which_drop_each_file_came_from(client, admin_headers):
"""from_drop is the exact answer to "which file in this run is mine".
Both drops here send a file called a.csv, which is precisely the collision
that made filename-matching unsafe: without from_drop a sender could match
the wrong merchant's file and price it as their own.
"""
first = _drop(client, "priya")
second = _drop(client, "arun")
run = client.post(
FROM_INBOX,
json={"file_ids": [f"{first}:0", f"{second}:0"]},
headers=admin_headers,
).json()
assert [f["filename"] for f in run["files"]] == ["a.csv", "a.csv"]
assert {f["from_drop"] for f in run["files"]} == {first, second}
def test_from_drop_survives_a_reread_of_the_run(client, admin_headers):
"""It is stamped after staging, so it has to reach disk, not just the reply."""
drop_id = _drop(client, "priya")
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
headers=admin_headers).json()["batch_id"]
assert batch_ingest.read_manifest(run_id).files[0].from_drop == drop_id
served = client.get(f"/api/admin/catalog-batch/batches/{run_id}",
headers=admin_headers).json()
assert served["files"][0]["from_drop"] == drop_id
def test_a_file_uploaded_straight_into_a_run_has_no_drop(client, monkeypatch):
"""Nothing to point at when the file never sat in an inbox.
None rather than the run's own id: saying a run came from itself would make
the field useless for the question it exists to answer.
"""
from app.api.routers import uploads
monkeypatch.setattr(uploads, "UPLOAD_AUTORUN", True)
run = client.post(UPLOAD, files=_files(("a.csv", _csv()))).json()
assert run["files"][0]["from_drop"] is None
def test_a_refused_row_names_itself(client, admin_headers, store, monkeypatch):
"""rejections[] carries the sheet row, not just a count.
The count alone was unusable for support: "rejected: 2" out of 19 products
left the only diagnosis being to diff the manifest against the file and
guess. A refused row is a product a shopkeeper expects on the shelf and
will not have, so it has to name itself and say why.
Everything here except `row` already shipped; this pins the whole shape
together so a future change cannot quietly drop one field of it.
"""
monkeypatch.setattr(pipeline, "ENABLE_PRODUCT_VALIDATION", True)
rows = [["Amul Butter 100g", "Butter", "Amul"],
["X", "", "Amul"]]
drop_id = client.post(UPLOAD, files=_files(("a.csv", _csv(rows=rows)))
).json()["batch_id"]
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
headers=admin_headers).json()["batch_id"]
result = batch_ingest.run_batch(run_id).files[0].result
assert result["rejected"] == len(result["rejections"]), (
"the count and the list must agree, or the list is not the explanation"
)
if result["rejections"]:
bad = result["rejections"][0]
assert set(bad) >= {"row", "product_name", "size", "reason"}
assert bad["reason"], "a rejection with no reason explains nothing"
# Row 1 is the header, so any real refusal is row 2 or later.
assert bad["row"] is None or bad["row"] >= 2
def test_the_rejection_row_matches_the_offending_sheet_line():
"""Straight at the gate, so the row number is checked without needing a
sheet that reliably fails validation end to end."""
rows = [
{"product_name": "X", "brand": "amul", "size": "100g",
"category": "Dairy", "image_id": "a", "_row": 7},
{"product_name": "", "brand": "amul", "size": "",
"category": "", "image_id": "b", "_row": 9},
]
_kept, rejected, _summary = pipeline.stage_10_validate(rows, "amul")
assert [r["_row"] for r in rejected] == [7, 9]

View File

@@ -167,10 +167,58 @@ def test_a_commodity_resolves_to_a_real_category(name, category):
assert canonical_category(name) is not None
def test_the_category_carries_an_hsn_code(store):
"""The names were chosen to match HSN_GST_TABLE keys, so tax enrichment
works without a second mapping."""
@pytest.mark.parametrize("name,hsn", [
("Toor Dhal 1kg", "0713"),
("Sugar 1kg", "1701"),
("Salt 1kg", "2501"),
# Fruit and vegetables share one category, so one code has to serve both.
# 0709 ("other vegetables, fresh or chilled") is the one carried; strictly
# a fruit is chapter 08. Both are nil-rated, so the GST is right either
# way, and this only ever surfaces in the seeded base list because an
# uploaded commodity is given no HSN at all.
("Apple", "0709"),
("Tomato", "0709"),
])
def test_a_commodity_category_is_a_key_the_hsn_table_knows(name, hsn):
"""The commodity categories line up with HSN_GST_TABLE keys.
Checked against the table directly rather than through an ingest. It used
to be asserted end to end, but the upload path no longer STAMPS an HSN onto
a commodity (see the test below), so running a sheet would now prove
nothing about the mapping. The property itself is still worth pinning: it
is what lets a category name serve as the tax key without a second lookup,
and it is how a sheet that DOES declare its tax treatment stays consistent
with ours.
"""
from app.services.enrichment.hsn_gst.models import HSN_GST_TABLE
category = canonical_category(name)
assert category in HSN_GST_TABLE, f"{category!r} has no HSN entry"
assert HSN_GST_TABLE[category][0] == hsn
def test_an_uploaded_commodity_is_given_no_hsn_code(store):
"""We do not invent a tax code the merchant did not supply.
An HSN we chose is our guess presented as their record, and a shopkeeper
bills from this. Fresh produce being nil-rated makes a wrong code cheap to
ignore and expensive to notice, which is the worst combination.
"""
_brand_of(["Toor Dhal 1kg"], store)
assert store[0][1]["hsn_code"] is None
def test_a_commodity_keeps_an_hsn_the_sheet_supplied(store):
"""The rule is "do not invent", not "discard"."""
wb = openpyxl.Workbook()
ws = wb.active
ws.append(["Item Name", "HSN Code"])
ws.append(["Toor Dhal 1kg", "0713"])
buf = io.BytesIO()
wb.save(buf)
pipeline.run_pipeline("store.xlsx", buf.getvalue(), use_llm=False, fetch_images=False)
assert store[0][1]["hsn_code"] == "0713"

View File

@@ -0,0 +1,201 @@
"""Loose produce reaches Own Products, and no real brand follows it there.
WHY THIS FILE IS THE GATE ON THE COMMODITY LEXICON
--------------------------------------------------
`is_unbranded()` works by requiring that EVERY significant token in a product
name is a known commodity or qualifier. That makes it asymmetric: adding a word
to the lexicon can only ever make the test more permissive, so the failure mode
is real brands quietly collapsing into one bucket - far harder to undo than a
staple sitting in the wrong table.
So the tests that matter most here are the negative ones. Before produce was
added, the candidate list was measured against the live catalogue and the alias
map; these encode that measurement, so the next person to add a word finds out
immediately if it swallows something it should not.
The positive tests exist because the alternative to landing in Own Products is
not "rejected" - it is being MISFILED. "Red Rose" resolved to brand "Red", which
resolve_parent_brand whole-word matched to Brooke Bond, so a flower was written
into the tea catalogue and stamped with Brooke Bond's FSSAI licence.
"""
from __future__ import annotations
import json
from pathlib import Path
import pytest
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
from app.services.generic_products import (
OWN_PRODUCTS_BRAND,
canonical_category,
is_unbranded,
)
SEED_DIR = Path(__file__).resolve().parents[1] / "data" / "seed_catalogs"
OWN_PRODUCTS_SEED = SEED_DIR / "brand_catalog_own_products.json"
# ---------------------------------------------------------------------------
# The negative tests: nothing branded may fall in here
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("alias", sorted(BRAND_ALIASES))
def test_no_brand_alias_is_read_as_a_commodity(alias: str) -> None:
"""Every alias must stay branded.
"amla" is the near miss: a real fruit that also appears inside the alias
"dabur amla". That stays branded because "dabur" is not a commodity - which
is exactly the conservatism the rule rests on: one unknown token is enough
to mean "this is a brand".
"""
assert not is_unbranded(alias)
@pytest.mark.parametrize(
"name",
[
"Amul Butter 500g",
"Aachi Sambar Powder",
"Brooke Bond Red Label 500g",
"Colgate Active Salt",
"Milky Mist Paneer",
"Nature Fresh Atta",
"Dabur Amla Hair Oil",
"Mother Dairy Milk 1L",
"24 Mantra Organic Moong Dal 500g",
],
)
def test_branded_products_stay_branded(name: str) -> None:
assert not is_unbranded(name)
def test_a_brand_whose_name_starts_with_a_number_survives() -> None:
"""The _strip_sizes regression.
The size strip used to be a number followed by "any letters", which ate the
word AFTER a number: "24 Mantra Organic Moong Dal" became "organic moong
dal", every remaining token was a commodity, and a real branded product was
filed as unbranded. Every brand beginning with a digit hit this.
"""
assert not is_unbranded("24 Mantra Organic Moong Dal 500g")
assert not is_unbranded("24 Mantra Organic Sona Masuri Rice 1kg")
# ... while the thing the strip actually exists for still works.
assert is_unbranded("Toor Dhal 1kg")
assert is_unbranded("Sugar 1kg")
assert is_unbranded("Black Pepper 100g")
# ---------------------------------------------------------------------------
# The positive tests: produce must reach Own Products
# ---------------------------------------------------------------------------
@pytest.mark.parametrize(
"name,category",
[
("Apple", "Fruits & Vegetables"),
("Orange", "Fruits & Vegetables"),
("Tomato", "Fruits & Vegetables"),
("Onion", "Fruits & Vegetables"),
("Potato", "Fruits & Vegetables"),
("Banana", "Fruits & Vegetables"),
("Drumstick", "Fruits & Vegetables"),
("Bitter Gourd", "Fruits & Vegetables"),
("Lady Finger", "Fruits & Vegetables"),
("Curry Leaves", "Fresh Herbs & Greens"),
("Mint Leaves", "Fresh Herbs & Greens"),
("Thulasi", "Fresh Herbs & Greens"),
("Jasmine", "Flowers"),
("Red Rose", "Flowers"),
("Tuna", "Fish & Seafood"),
("Prawns", "Fish & Seafood"),
("Egg", "Eggs"),
# Pantry staples that were already covered, pinned so the produce work
# cannot regress them.
("Toor Dhal 1kg", "Pulses, Grains & Spices"),
("Sugar 1kg", "Sugar & Jaggery"),
("Salt 1kg", "Salt & Staples"),
],
)
def test_loose_goods_are_unbranded_and_categorised(name: str, category: str) -> None:
assert is_unbranded(name), f"{name} would be given a junk brand"
assert canonical_category(name) == category
def test_a_flower_no_longer_lands_in_the_tea_catalogue() -> None:
"""The specific misroute this work exists to end.
"Red Rose" -> infer_brand -> "Red" -> resolve_parent_brand -> "brooke bond",
so a rose was written into brand_brooke_bond carrying Brooke Bond's real
FSSAI licence. The fix is upstream: the row never reaches infer_brand.
"""
assert is_unbranded("Red Rose")
# The hijack is still there for anything that DOES reach it, so this test
# fails loudly if the diversion is removed rather than passing for the
# wrong reason.
assert resolve_parent_brand("Red") == "brooke bond"
def test_merchant_typos_still_resolve() -> None:
"""Real strings from merchant data, misspellings included."""
for name in ("Bitter guard", "Bottle ground", "Ladies Finger"):
assert is_unbranded(name), name
def test_an_empty_brand_column_is_believed() -> None:
"""A sheet WITH a brand column that left the cell blank has said something.
This is how place-qualified produce gets in. "Salem Mango" keeps an unknown
token, so the word test alone calls it branded - deliberately, because
"Mysore" is also a real brand (Mysore Sandal). An explicit empty cell
overrides that.
"""
assert not is_unbranded("Salem Mango")
assert is_unbranded("Salem Mango", brand_column_supplied=True)
# A filled cell is believed just as much.
assert not is_unbranded("Apple", sheet_brand="Washington")
# ---------------------------------------------------------------------------
# The seeded base list
# ---------------------------------------------------------------------------
def _seed_doc():
if not OWN_PRODUCTS_SEED.exists():
pytest.skip("own-products seed catalog not present")
return json.loads(OWN_PRODUCTS_SEED.read_text(encoding="utf-8-sig"))
def test_the_seed_catalog_is_filed_under_own_products() -> None:
doc = _seed_doc()
assert doc["brand"] == OWN_PRODUCTS_BRAND
assert doc["total_products"] == len(doc["products"])
assert doc["products"], "the base list is empty"
def test_every_seeded_row_would_also_be_recognised_on_upload() -> None:
"""The round trip that makes the base list worth having.
A grocer who types "Tomato" into their own sheet must land on the SAME row
that was seeded rather than create a second one. That only holds if every
seeded name is itself classified unbranded - otherwise the uploaded copy
goes to a junk brand table and the two never meet.
"""
missed = [
p["product_name"] for p in _seed_doc()["products"]
if not is_unbranded(p["product_name"])
]
assert not missed, f"seeded rows a real upload would misfile: {missed}"
def test_seeded_image_ids_are_unique() -> None:
"""image_id is the deduplication key, and the column is UNIQUE NOT NULL."""
ids = [p["image_id"] for p in _seed_doc()["products"]]
assert len(ids) == len(set(ids))
assert all(ids)
def test_seeded_rows_carry_no_brand_pack_size_or_price() -> None:
"""Loose produce has none of those, and inventing them would be a lie."""
for p in _seed_doc()["products"]:
assert p["brand_name"] == OWN_PRODUCTS_BRAND
assert p["size_variants"] == []
assert p["price_range"] is None
assert p["fssai_license"] is None

View File

@@ -0,0 +1,311 @@
"""What actually gets written for an unbranded row: the sheet's values, and nothing else.
THE RULE
--------
A merchant sends `Apple, 500g, 155`. Those three values are what we know. FSSAI,
HSN, SKU, barcode and description are things we would be *making up*, and a
shopkeeper bills from this record - an invented tax code is our guess wearing
their letterhead.
WHY THIS NEEDED A VALIDATION CHANGE AS WELL
-------------------------------------------
Leaving those fields empty is not free. The validation gate scores a row down
for each missing field and drops it below the reject threshold:
0.55 baseline - 0.30 (no price_range) - 0.10 (no SKU) = 0.15 vs 0.35
So the literal instruction "leave these null" would have deleted every produce
row - the exact opposite of the requirement. `validate_product` now treats a
commodity's missing price and SKU as normal rather than as evidence of a
fabricated row. The first test below is the wall around that, and the branded
counterpart proves the exemption did not leak.
"""
from __future__ import annotations
import io
import pytest
from app.core import store_catalog_pipeline as pipeline
from app.services.generic_products import OWN_PRODUCTS_BRAND
from app.services.product_validator import validate_product
openpyxl = pytest.importorskip("openpyxl")
def _sheet(headers, rows) -> bytes:
wb = openpyxl.Workbook()
ws = wb.active
ws.append(headers)
for row in rows:
ws.append(row)
buf = io.BytesIO()
wb.save(buf)
return buf.getvalue()
@pytest.fixture(autouse=True)
def _isolate_sku_counter(tmp_path, monkeypatch):
from app.services import sku_service
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
@pytest.fixture
def store(monkeypatch):
"""Capture what would be written, keyed the way the real table is."""
written: list = []
def _upsert(brand, products, cleanup=False):
for product in products:
written.append((brand, dict(product)))
return len(products)
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda brand: [])
monkeypatch.setattr(pipeline, "upsert_brand_products", _upsert)
monkeypatch.setattr(pipeline, "USE_EMBEDDINGS", False)
return written
def _run(headers, rows):
pipeline.run_pipeline("store.xlsx", _sheet(headers, rows),
use_llm=False, fetch_images=False)
# ---------------------------------------------------------------------------
# The gate, which is what makes the rest of this possible
# ---------------------------------------------------------------------------
WORKED_EXAMPLE = {
"product_name": "Apple", "title": "Apple",
"category": "Fruits & Vegetables", "size": "500g",
"selling_price": 155, "final_selling_price": 155,
"price_range": "₹143-167",
"product_sku": None, "sku_source": None,
"hsn_code": None, "fssai_license": None, "description": None,
}
@pytest.mark.parametrize("images,expected", [
(["https://example.com/apple.jpg"], "verified"),
([], "needs_review"),
])
def test_the_worked_example_is_never_rejected(images, expected):
"""`Apple / 500g / 155` with everything else null must survive.
Both outcomes are KEPT: validate_catalog returns verified and needs_review
rows together, and only `rejected` is dropped. The no-image case stays
needs_review deliberately - a missing picture is the one absence here that
still says something, since image search did run and found nothing.
"""
row = dict(WORKED_EXAMPLE, image_urls=images)
report = validate_product(row, OWN_PRODUCTS_BRAND,
category_resolved_deterministically=True,
images_checked=True)
assert report.status == expected
assert report.status != "rejected"
def test_a_branded_row_with_the_same_gaps_is_still_rejected():
"""The exemption is scoped to commodities and must not leak.
Same row, same absences, under a real brand: a branded product with no
price and no SKU IS evidence that something went wrong upstream, and that
judgement is unchanged.
"""
row = dict(WORKED_EXAMPLE, image_urls=[], price_range=None)
report = validate_product(row, "amul",
category_resolved_deterministically=True,
images_checked=True)
assert report.status == "rejected"
def test_a_malformed_price_on_a_commodity_is_still_caught():
"""Absent is excused; wrong is not."""
row = dict(WORKED_EXAMPLE, image_urls=[], price_range="one fifty five")
report = validate_product(row, OWN_PRODUCTS_BRAND,
category_resolved_deterministically=True,
images_checked=True)
assert any(issue.field == "price_range" for issue in report.issues)
def test_a_blank_product_name_is_still_caught():
"""This is not a way in for junk. Everything except the two excused
absences is still enforced."""
row = dict(WORKED_EXAMPLE, product_name="", title="", image_urls=[])
report = validate_product(row, OWN_PRODUCTS_BRAND,
category_resolved_deterministically=True,
images_checked=True)
assert report.status == "rejected"
# ---------------------------------------------------------------------------
# The field policy, end to end
# ---------------------------------------------------------------------------
def test_the_sheets_values_are_kept_and_nothing_else_is_invented(store):
"""The requirement, as one assertion per field."""
_run(["Product Name", "Weight", "Selling Price"], [["Apple", "500g", 155]])
assert len(store) == 1
brand, row = store[0]
assert brand == OWN_PRODUCTS_BRAND
# What the sheet said.
assert row["product_name"] == "Apple 500g"
assert row["size_variants"] == ["500g"]
assert row["final_selling_price"] == 155
assert row["price_range"] == "₹143-167" # +/-8% of the sheet's own price
# What it did not say, and what we therefore do not claim.
for field in ("hsn_code", "product_sku", "sku_source",
"fssai_license", "barcode", "barcode_type", "description"):
assert row[field] is None, f"{field} was invented: {row[field]!r}"
def test_the_price_band_is_eight_percent_of_the_sheet_price(store):
_run(["Product Name", "Weight", "Selling Price"], [["Tomato", "1kg", 100]])
assert store[0][1]["price_range"] == "₹92-108"
def test_a_commodity_with_no_price_gets_no_band(store):
"""The market estimator is not consulted for loose produce.
It is trained on packaged FMCG and prices a 500g apple at around Rs85-105,
which is not so much wrong as meaningless - a shop prices produce by the
day. A null band is the honest answer.
"""
_run(["Product Name", "Weight"], [["Apple", "500g"]])
assert store[0][1]["price_range"] is None
def test_a_branded_row_is_untouched_by_all_of_this(store):
"""The blast radius check. A real brand still gets its full enrichment."""
_run(["Product Name", "Brand", "Category", "Weight", "Selling Price"],
[["Amul Butter", "Amul", "Dairy", "500g", 100]])
_brand, row = store[0]
assert row["fssai_license"], "a branded row lost its FSSAI licence"
assert row["product_sku"], "a branded row lost its minted SKU"
assert row["description"], "a branded row lost its description"
assert row["price_range"] == "₹92-108"
# ---------------------------------------------------------------------------
# Pack sizes
# ---------------------------------------------------------------------------
def test_a_commodity_with_no_weight_yields_exactly_one_row(store):
"""No invented 100g/250g/500g.
A shop sells apples by whatever the customer asks for, so "Apple 250g" is a
product that does not exist. This is also the mechanism behind the
catalogue drift reported by our integrator: the invented set is keyed on
the resolved category, so the same product ingested twice with the category
resolved differently produces two disjoint size sets and two sets of ids.
"""
_run(["Product Name"], [["Apple"]])
assert len(store) == 1
assert store[0][1]["size_variants"] == ["Standard"]
def test_an_uploaded_commodity_lands_on_the_seeded_row(store):
"""The dedupe that makes the base list worth having.
A grocer typing "Apple" must land on the seeded "Apple" rather than create
a second one. That holds only if both sides build the same image_id, which
means both must use "Standard" for an absent size.
"""
_run(["Product Name"], [["Apple"]])
assert store[0][1]["image_id"] == pipeline.build_image_id(
OWN_PRODUCTS_BRAND, "Apple", "Standard")
def test_a_declared_pack_size_is_still_honoured(store):
"""Not inventing sizes must not mean ignoring the ones we were given."""
_run(["Product Name", "Weight"], [["Apple", "500g"]])
assert store[0][1]["size_variants"] == ["500g"]
# ---------------------------------------------------------------------------
# "Do not invent" is not "discard"
# ---------------------------------------------------------------------------
def test_values_the_sheet_supplied_survive(store):
"""A merchant who fills in HSN, SKU and a description keeps all three."""
_run(["Product Name", "Weight", "Selling Price", "HSN Code", "SKU", "Description"],
[["Apple", "500g", 155, "0808", "SHOP-APL-1", "Shimla apples, loose"]])
row = store[0][1]
assert row["hsn_code"] == "0808"
assert row["product_sku"] == "SHOP-APL-1"
assert row["sku_source"] == "sheet"
assert row["description"] == "Shimla apples, loose"
def test_a_supplied_price_range_wins_over_the_derived_band(store):
_run(["Product Name", "Weight", "Selling Price", "Price Range"],
[["Apple", "500g", 155, "₹150-160"]])
assert store[0][1]["price_range"] == "₹150-160"
# ---------------------------------------------------------------------------
# The embedding text
# ---------------------------------------------------------------------------
def test_the_search_text_does_not_embed_the_bucket_name_or_a_null(store):
"""`search_query` feeds the vector index.
Interpolated the branded way it would read "Own Products Apple 500g
Fruits & Vegetables None" - the bucket is not a maker, and "None" is the
string repr of the description we deliberately left empty. Both would be
embedded and both would pull unrelated produce together.
"""
_run(["Product Name", "Weight"], [["Apple", "500g"]])
query = store[0][1]["search_query"]
assert "Own Products" not in query
assert "None" not in query
assert "Apple" in query and "Fruits & Vegetables" in query
# ---------------------------------------------------------------------------
# Category: the lexicon is a hint, the pack size is a fact
# ---------------------------------------------------------------------------
def test_a_commodity_gets_its_category_from_the_lexicon(store):
"""Keyword detection has no entry for individual fruit and never will.
Listing every vegetable in the curated registry would duplicate the
commodity lexicon and let the two drift, so the lexicon is consulted as a
second opinion. Without it every produce row landed in "General", which
loses the unit rules, the grouping and the validation credit.
"""
_run(["Product Name", "Weight"], [["Apple", "500g"]])
assert store[0][1]["category"] == "Fruits & Vegetables"
def test_a_lexicon_category_never_overrules_a_declared_pack_size(store):
"""The regression that made this guard necessary.
The lexicon calls tea a Beverage; the unit rulebook says beverages are
measured in ml or litres ONLY; stage 4 therefore "corrected" 250g to 250ml
and turned a quarter kilo of tea leaves into a quarter litre. Loose tea is
a dry good sold by weight - the category was wrong, not the size - so a
category that cannot hold the declared size is declined.
"""
_run(["Product Name"], [["Tea Powder 250g"]])
assert len(store) == 1, "the row was dropped or silently unit-converted"
brand, row = store[0]
assert brand == OWN_PRODUCTS_BRAND
assert row["size_variants"] == ["250g"], "250g became something else"
assert row["category"] != "Beverages"