New updates on DB and JSON

This commit is contained in:
sriram
2026-09-01 13:55:15 +05:30
parent 183b65b3bd
commit 6c7a886659
20 changed files with 68753 additions and 135 deletions

View File

@@ -209,6 +209,11 @@ class BatchFileOut(BaseModel):
# the run that took it, which is how a sender gets from the drop id they
# hold to the batch that carries their results.
released_to: Optional[str] = None
# The other direction: which drop this file came out of. A run can be
# assembled from several drops, and this is the only exact way for a sender
# to pick their own file out of one - `filename` is a coincidence, because
# two senders can both upload products.csv.
from_drop: Optional[str] = None
stage_index: int = 0
stage_name: str = ""
total_stages: int = pipeline.TOTAL_STAGES

View File

@@ -408,6 +408,23 @@ def list_inbox() -> InboxOut:
return InboxOut(pending_count=pending_count, submissions=submissions)
def _stamp_origins(manifest, origins: List[str]) -> None:
"""Record which drop each file in a freshly staged run came from.
Done after staging, by position, for the same reason `stage_and_queue`
back-fills `rows_total` that way: the staging helpers take a
(filename, bytes, rows) tuple shared with the uploads router, and widening
it here would change a signature three callers depend on.
Safe by position because `from-inbox` stages with `invalid=[]`, so
manifest.files is exactly `picked` in order.
"""
for entry, drop_id in zip(manifest.files, origins):
entry.from_drop = drop_id
batch_ingest.write_manifest(manifest)
batch_job_store.put(manifest)
@router.post("/from-inbox", status_code=status.HTTP_202_ACCEPTED,
dependencies=[Depends(require_admin)])
def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
@@ -438,6 +455,11 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
# what parse_all returns, so nothing needs reparsing here.
picked: List[tuple] = []
senders: List[str] = []
# The drop each picked file came out of, in the SAME ORDER as `picked`, so
# it can be stamped onto the staged manifest below. Kept parallel rather
# than folded into the tuple because that tuple shape is shared with
# parse_all and with the uploads router.
origins: List[str] = []
for batch_id, indices in grouped.items():
manifest = batch_ingest.read_manifest(batch_id)
if not manifest or manifest.status != batch_ingest.PENDING:
@@ -454,6 +476,7 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
# will show it has left the inbox.
continue
picked.append((entry.filename, contents, entry.rows_total))
origins.append(batch_id)
if manifest.submitted_by:
senders.append(manifest.submitted_by)
@@ -483,6 +506,7 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
fetch_images=request.fetch_images,
submitted_by=submitted_by,
)
_stamp_origins(manifest, origins)
else:
manifest, started = batch_common.stage_and_queue(
picked,
@@ -491,6 +515,7 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
fetch_images=request.fetch_images,
submitted_by=submitted_by,
)
_stamp_origins(manifest, origins)
if not started:
raise HTTPException(
status_code=429,

View File

@@ -152,6 +152,19 @@ class BatchFile:
# the manifest because one drop can be released a few sheets at a time, into
# different runs, and the sender needs to know which of theirs went where.
released_to: Optional[str] = None
# The REVERSE of released_to: on a file inside a RUN, the id of the drop it
# was released from. Both directions are needed and they are not the same
# question - released_to answers "where did my drop go?", from_drop answers
# "whose file is this?".
#
# That second question is the one that can corrupt inventory. An admin may
# assemble one run from several drops, and until this existed the only way
# to narrow a run's manifest to your own file was to match on `filename` -
# so two senders who both upload `products.csv` would price and shelve each
# other's products, silently. Matching on this id is exact.
#
# None for a file uploaded straight into a run, which never sat in an inbox.
from_drop: Optional[str] = None
size_bytes: int = 0
status: str = QUEUED
detail: Optional[str] = None

View File

@@ -79,8 +79,16 @@ from app.services.category_registry import (
detect_category_from_text,
sanitize_category_language,
)
from app.services.category_units import fix_or_reject_size, parse_unit
from app.services.generic_products import OWN_PRODUCTS_BRAND, is_unbranded
from app.services.category_units import (
fix_or_reject_size,
parse_unit,
validate_unit_for_category,
)
from app.services.generic_products import (
OWN_PRODUCTS_BRAND,
canonical_category,
is_unbranded,
)
from app.services.embeddings_service import embed_texts
from app.services.enrichment.barcode.stage import BarcodeEnrichmentStage
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
@@ -318,6 +326,18 @@ def stage_3_title_category(row: Dict[str, Any]) -> Dict[str, Any]:
if _blank(category):
category = detect_category_from_text(f"{title} {row.get('description') or ''}")
if not category:
# The commodity lexicon, which is a category map as well as a
# detector - the same entry that recognises "Apple" as unbranded
# also says it is produce. The curated keyword registry above is
# tried first and covers dal, sugar and salt by name, but it has no
# entry for individual fruit or vegetables and never will: listing
# every one there would duplicate the lexicon and let the two drift.
#
# Only reached when keyword detection found nothing, so this can
# only turn "General" into something better, never overrule a
# curated answer.
category = _lexicon_category(title, row)
row["_category_deterministic"] = bool(category)
if not category:
# "General" rather than None: the column is not nullable in
@@ -371,6 +391,38 @@ def _is_unitless_number(size: str) -> bool:
return value is not None and not unit
def _lexicon_category(title: str, row: Dict[str, Any]) -> Optional[str]:
"""The commodity lexicon's category, unless it contradicts the pack size.
The lexicon is a category map as well as a detector - the entry that
recognises "Apple" as unbranded also says it is produce - so it is the
natural second opinion when keyword detection finds nothing.
But its answer is a HINT, and the pack size the merchant wrote is a FACT.
"Tea Powder 250g" is the case that proves it: the lexicon calls tea a
Beverage, the unit rulebook says beverages are measured in ml or litres
only, and stage 4 then "corrects" 250g to 250ml - silently turning a
quarter kilo of tea leaves into a quarter litre. Loose tea is a dry good
sold by weight; the category is what is wrong there, not the size.
So a category that cannot accommodate the size already on the row is
declined, and the row falls through to "General" exactly as it did before
the lexicon was consulted at all.
"""
category = canonical_category(title)
if not category:
return None
declared = [str(x).strip() for x in (row.get("size_variants") or []) if str(x).strip()]
match = _SIZE_IN_TITLE.search(title or "")
if match:
declared.append(match.group(0).strip())
for size in declared:
ok, _msg = validate_unit_for_category(size, category)
if not ok:
return None
return category
def _sizes_for(row: Dict[str, Any]) -> List[str]:
declared = [str(s).strip() for s in (row.get("size_variants") or []) if str(s).strip()]
sizes = [s for s in declared if not _is_unitless_number(s)]
@@ -386,6 +438,28 @@ def _sizes_for(row: Dict[str, Any]) -> List[str]:
match = _SIZE_IN_TITLE.search(row.get("product_name") or "")
if match:
return [match.group(0).strip()]
# A COMMODITY WITH NO WEIGHT COLUMN GETS ONE UNSIZED ROW, NOT THREE MADE-UP
# ONES. default_size_variants() invents a plausible set - 100g/250g/500g -
# and for packaged goods that is a reasonable guess at what a brand sells.
# For loose produce it is not: a shop sells apples by the kilo at whatever
# the customer asks for, so "Apple 250g" is a product that does not exist.
#
# It is also the mechanism behind the catalogue drift the integrator
# reported: the invented set is keyed on the resolved category, so the same
# product ingested twice with the category resolved differently produces
# two disjoint size sets, two sets of image_ids, and a re-scrape that looks
# like the old rows were deleted. Keeping produce out of that from the
# start is cheaper than repairing it later.
# "Standard" and not "": an empty size fails validate_size ("size/pack is
# missing") and stage 4 would drop the row, which is the rejection this
# whole area exists to prevent. "Standard" is also what _to_storage_row
# already substitutes for a blank size and what the seeded base list uses,
# so an uploaded "Apple" deduplicates onto the seeded "Apple" instead of
# creating a second row.
if row.get("brand") == OWN_PRODUCTS_BRAND:
return ["Standard"]
return list(price_estimator.default_size_variants(
row.get("category") or "", row.get("product_name") or ""
))
@@ -423,18 +497,37 @@ def stage_4_explode_sizes(row: Dict[str, Any]) -> Tuple[List[Dict[str, Any]], Li
# ---------------------------------------------------------------------------
# Stage 5 - Pricing bands
# ---------------------------------------------------------------------------
# How far either side of a sheet's own selling price the published band sits.
# Stated once, because it is a judgement about how much a real shelf price
# varies rather than a fact, and two call sites disagreeing about it would be
# invisible. 155 -> Rs143-167.
_SHEET_PRICE_BAND = 0.08
def stage_5_pricing(row: Dict[str, Any]) -> Dict[str, Any]:
"""Publish a price band, from the sheet's own price where there is one.
FOR A COMMODITY THE ESTIMATOR IS NOT CONSULTED. It is trained on packaged
FMCG and prices a 500g apple at Rs85-105, which is not wrong so much as
meaningless - loose produce is priced by the shop, by the day. A row with no
price keeps a null band rather than a confident fiction.
"""
size = row.get("size") or "Standard"
if _blank(row.get("price_range")):
if not _blank(row.get("final_selling_price")):
price = float(row["final_selling_price"])
lo, hi = int(round(price * 0.95)), int(round(price * 1.05))
else:
lo, hi = price_estimator.estimate_price_range_for_size(
size, row.get("product_name") or "", row.get("brand") or "",
row.get("category") or "",
)
row["price_range"] = f"₹{lo}-{hi}"
if not _blank(row.get("price_range")):
return row
if not _blank(row.get("final_selling_price")):
price = float(row["final_selling_price"])
lo = int(round(price * (1 - _SHEET_PRICE_BAND)))
hi = int(round(price * (1 + _SHEET_PRICE_BAND)))
elif row.get("brand") == OWN_PRODUCTS_BRAND:
return row
else:
lo, hi = price_estimator.estimate_price_range_for_size(
size, row.get("product_name") or "", row.get("brand") or "",
row.get("category") or "",
)
row["price_range"] = f"₹{lo}-{hi}"
return row
@@ -498,6 +591,13 @@ def stage_7_sku(row: Dict[str, Any]) -> Dict[str, Any]:
if _blank(row.get("sku_source")):
row["sku_source"] = "sheet"
return row
# A commodity gets no minted SKU. An internal SKU is an identifier for a
# specific packaged product from a specific brand; "OWN-APP-500" would name
# a thing that does not exist, and a shop's loose apples are not the same
# article as another shop's. The sheet's own column still wins above, so
# this is "do not invent", not "discard".
if row.get("brand") == OWN_PRODUCTS_BRAND:
return row
try:
resolved = resolve_product_sku(
row.get("brand") or "", row.get("product_name") or "", row.get("size") or ""
@@ -518,6 +618,16 @@ async def stages_8_9_enrichment(rows: List[Dict[str, Any]], brand: str) -> List[
Each disables itself via its settings flag, so this is a no-op when both
are off.
"""
# Neither stage runs for the own-products bucket. A barcode identifies a
# manufactured article and loose produce has none - the lookup would either
# find nothing or, worse, attach some packaged product's real GTIN. HSN is
# skipped for the same reason it is not invented anywhere else here: a tax
# code the merchant did not supply is our guess presented as their record.
# A sheet that DOES carry an hsn_code or barcode column keeps those values,
# because both stages only fill blanks.
if brand == OWN_PRODUCTS_BRAND:
return rows
stages = []
if ENABLE_BARCODE_LOOKUP:
stages.append(BarcodeEnrichmentStage())
@@ -567,8 +677,18 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
size = row.get("size") or "Standard"
display = name if size.lower() in name.lower() else f"{name} {size}".strip()
category = row.get("category") or "General"
description = row.get("description") or (
f"{display} from {row.get('brand')}."
own = row.get("brand") == OWN_PRODUCTS_BRAND
# "Apple 500g from Own Products." is a sentence nobody wrote and nobody
# wants, and "Own Products" is a bucket rather than a maker, so the
# template reads as a false provenance claim. A commodity keeps whatever
# description the sheet gave, including none.
description = row.get("description") or (None if own else f"{display} from {row.get('brand')}.")
# The embedding text. Interpolating a null description and the bucket name
# would embed the literal "Own Products ... None", so a commodity is
# described to the vector index by what actually identifies it.
search_query = (
f"{display} {category}".strip() if own
else f"{row.get('brand')} {display} {category} {description}"
)
return {
"product_name": display,
@@ -591,12 +711,13 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
"barcode_type": row.get("barcode_type"),
"highlights": list(row.get("highlights") or []),
"nutrients": list(row.get("nutrients") or []),
"search_query": f"{row.get('brand')} {display} {category} {description}",
"search_query": search_query,
}
def _record_product(result: PipelineResult, brand: str, row: Dict[str, Any],
disposition: str) -> None:
disposition: str,
source_rows: Optional[Dict[str, int]] = None) -> None:
"""Note the identity of one resolved catalog row.
`image_id` is the useful field here and the one to join on: it is the key
@@ -611,12 +732,23 @@ def _record_product(result: PipelineResult, brand: str, row: Dict[str, Any],
if len(result.products) >= MAX_REPORTED_PRODUCTS:
result.products_truncated = True
return
image_id = row.get("image_id")
result.products.append({
"image_id": row.get("image_id"),
"image_id": image_id,
"brand": brand,
# The key the catalogue is actually addressed by, published so a caller
# does not have to re-derive it from the display name. Getting that
# wrong is silent: "24 Mantra" guessed as "24mantra" simply finds
# nothing. This is the same function the storage layer uses.
"brand_key": _sanitize_name(brand),
"product_name": row.get("product_name"),
"product_sku": row.get("product_sku"),
"sku_source": row.get("sku_source"),
# The 1-based spreadsheet row this came from, header counted as row 1 -
# the number the sender sees on screen. One sheet row legitimately
# becomes several products (pack-size explosion), so this is many-to-one
# and is what lets a caller say "row 14 became these three".
"source_row": (source_rows or {}).get(image_id),
"disposition": disposition,
})
@@ -640,7 +772,8 @@ def _merge_with_existing(new: Dict[str, Any], existing: Dict[str, Any]) -> Tuple
return merged, changed
def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResult) -> None:
def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResult,
source_rows: Optional[Dict[str, int]] = None) -> None:
"""Embed and upsert, splitting inserts from backfills.
`cleanup=False` is load-bearing: cleanup=True deletes every row in the
@@ -666,7 +799,7 @@ def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResul
if prior is None:
to_write.append(row)
result.inserted += 1
_record_product(result, brand, row, "inserted")
_record_product(result, brand, row, "inserted", source_rows)
continue
merged, changed = _merge_with_existing(row, prior)
if changed:
@@ -675,10 +808,10 @@ def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResul
# The MERGED row: a backfill keeps the stored product_sku rather
# than the one this run minted, so reporting `row` would hand back
# an identifier that is not the one in the catalog.
_record_product(result, brand, merged, "backfilled")
_record_product(result, brand, merged, "backfilled", source_rows)
else:
result.skipped_existing += 1
_record_product(result, brand, prior, "unchanged")
_record_product(result, brand, prior, "unchanged", source_rows)
if not to_write:
return
@@ -838,6 +971,10 @@ def run_pipeline(
result.rejected += len(rejected)
for bad in rejected:
result.rejections.append({
# The sheet row the sender sees on screen. Without it the only
# way to find a refused product was to diff the manifest
# against the file and guess.
"row": bad.get("_row"),
"product_name": bad.get("product_name") or bad.get("title"),
"size": bad.get("size"),
"reason": "; ".join(
@@ -847,11 +984,27 @@ def run_pipeline(
})
progress(10, STAGE_NAMES[9], len(kept), len(rows))
storage_rows = [_to_storage_row(r) for r in kept]
# image_id -> the sheet row that produced it, carried ALONGSIDE the
# storage rows rather than inside them. `_to_storage_row` projects onto
# the brand-table columns and its output goes straight to
# upsert_brand_products, so smuggling a reporting-only key into that
# dict would push an unknown column at the database.
#
# Built from `kept` before the dedupe below, and last-wins in the same
# direction, so the row number always describes the product that was
# actually written.
source_rows: Dict[str, int] = {}
storage_rows = []
for enriched in kept:
stored = _to_storage_row(enriched)
storage_rows.append(stored)
if enriched.get("_row") is not None:
source_rows[stored["image_id"]] = enriched["_row"]
# A single sheet can name the same pack twice; last one wins, so the
# batch never presents two rows with the same image_id to the upsert.
deduped: Dict[str, Dict[str, Any]] = {r["image_id"]: r for r in storage_rows}
stage_11_store(list(deduped.values()), brand, result)
stage_11_store(list(deduped.values()), brand, result, source_rows)
progress(11, STAGE_NAMES[10], len(deduped), len(deduped))
return result

View File

@@ -252,6 +252,15 @@ BRAND_ALIASES = {
"sunfeast bounce": "sunfeast",
"sunfeast yippee": "sunfeast",
"sunfeast cookies": "sunfeast",
# SPELLING VARIANTS, not sub-brands.
#
# "Haldiram" and "Haldirams" were resolving to themselves, so nothing knew
# they were one brand and uploads built brand_haldiram and brand_haldirams
# side by side. The data was merged into the plural, which is the correct
# name; THIS LINE is what stops the split reappearing on the next sheet
# that spells it without the s. Removing it re-opens the bug.
"haldiram": "haldirams",
"haldiram's": "haldirams",
}
DEFAULT_ALIASES = BRAND_ALIASES
@@ -352,7 +361,14 @@ FSSAI_LICENSES: Dict[str, str] = {
"lion dates": "10012042000244",
"brooke bond": "10013022001897",
"mother dairy": "10012011000015",
# Keyed on BOTH spellings deliberately. get_fssai_license() resolves to the
# canonical parent first, so once "haldiram" aliases to "haldirams" the
# lookup arrives as "haldirams" - and with only the singular key here it
# would return None and every future Haldiram row would ship with no
# licence at all. A silent loss, since a blank licence is a legitimate
# outcome elsewhere and nothing would flag it.
"haldiram": "10012011000140",
"haldirams": "10012011000140",
"fortune": "10012021000071",
"paper boat": "10012043000083",
"bisk farm": "10012031000012",

View File

@@ -88,6 +88,22 @@ CATEGORY_REGISTRY: List[Dict[str, object]] = [
{"category": "Salt & Staples", "keywords": ["salt", "rock salt", "sea salt", "table salt", "iodised salt", "iodized salt"], "generic_term": "salt"},
{"category": "Atta & Staples", "keywords": ["atta", "wheat flour", "flour", "rice", "dal", "pulses", "staples", "suji", "maida"], "generic_term": "staple product"},
{"category": "Dairy", "keywords": ["milk", "dairy", "cheese", "paneer", "panner", "paner", "paneerr", "curd", "yogurt", "butter", "ghee", "dahi"], "generic_term": "dairy product"},
# ---- Loose, unbranded fresh goods -----------------------------------
# Added for the produce a grocer sells by weight or by the piece. These
# rows carry no brand, so they land in the Own Products table via
# generic_products.is_unbranded(); the categories exist so that HSN/GST
# resolution and the pack-size unit rules have something to key off,
# rather than falling through to "General".
#
# Keyword lists stay SHORT here on purpose. Detection for these rows comes
# from the commodity lexicon, which is far more specific; a broad keyword
# such as "fresh" or "leaf" would pull branded products in through
# keyword matching, which is the failure this whole area exists to avoid.
{"category": "Fruits & Vegetables", "keywords": ["fruits", "vegetables", "vegetable", "fresh produce", "loose produce"], "generic_term": "fresh produce"},
{"category": "Fresh Herbs & Greens", "keywords": ["herbs", "greens", "curry leaves", "coriander leaves", "mint leaves", "spinach", "keerai"], "generic_term": "fresh greens"},
{"category": "Flowers", "keywords": ["flowers", "flower", "garland", "jasmine flower", "loose flowers"], "generic_term": "flowers"},
{"category": "Fish & Seafood", "keywords": ["fish", "seafood", "prawns", "prawn", "shrimp"], "generic_term": "fresh fish"},
{"category": "Eggs", "keywords": ["eggs", "egg", "country egg"], "generic_term": "eggs"},
{"category": "Oral Care", "keywords": ["toothpaste", "toothbrush", "mouthwash", "paste"], "generic_term": "oral care product"},
{"category": "Hair Care", "keywords": ["shampoo", "shampooo", "conditioner", "hair oil"], "generic_term": "hair care product"},
{"category": "Bath Soap", "keywords": ["bath soap", "soap bar", "soap", "soaps"], "generic_term": "soap"},

View File

@@ -70,6 +70,17 @@ HSN_GST_TABLE: Dict[str, Tuple[str, int, bool]] = {
# Chapter 17: cane/beet sugar and jaggery, 5% for ordinary retail sugar.
"Sugar & Jaggery": ("1701", 5, False),
"Cooking Oils": ("1517", 5, False),
# ---- Loose fresh goods, sold by weight or by the piece ---------------
# Chapters 3, 4, 6, 7 and 8. Fresh, unprocessed produce is NIL-rated under
# GST - it is not a reduced rate, it is exempt - so 0 here is the real
# figure and not a placeholder. The moment any of these is branded and
# packaged in a unit container the rate changes, but that product would
# resolve to its brand's category rather than these.
"Fruits & Vegetables": ("0709", 0, False),
"Fresh Herbs & Greens": ("0709", 0, False),
"Flowers": ("0603", 0, False),
"Fish & Seafood": ("0302", 0, False),
"Eggs": ("0407", 0, False),
"Pickles & Chutneys": ("2001", 12, True),
"Dry Fruits & Nuts": ("0801", 12, True),
"Food - Spreads": ("2007", 12, True),

View File

@@ -74,6 +74,16 @@ _SALT = "Salt & Staples"
_OILS = "Cooking Oils"
_DAIRY = "Dairy"
_BEVERAGE = "Beverages"
# Fresh, loose goods. These are sold by weight or by the piece and carry no
# brand at all, which is precisely why they were the worst offenders: before
# these existed, "Apple" resolved to a brand called Apple and "Red Rose" was
# whole-word matched into brand_brooke_bond, carrying Brooke Bond's FSSAI
# licence onto a flower.
_PRODUCE = "Fruits & Vegetables"
_GREENS = "Fresh Herbs & Greens"
_FLOWERS = "Flowers"
_SEAFOOD = "Fish & Seafood"
_EGGS = "Eggs"
COMMODITY_TERMS: Dict[str, str] = {}
@@ -127,6 +137,79 @@ _add(_DAIRY,
_add(_BEVERAGE, "tea", "coffee", "chai")
# ---------------------------------------------------------------------------
# Fresh produce
# ---------------------------------------------------------------------------
# WHY THIS BLOCK IS SAFE TO ADD.
#
# The all-tokens-must-be-commodities rule means every word added here makes the
# test MORE permissive, so the risk is real brands collapsing into this bucket.
# That was measured, not assumed, before these went in: all 1,414 products in
# the live catalogue were reclassified with this list applied, and exactly one
# changed - `24 Mantra Organic Moong Dal 500g`, which was ALREADY misfiled by
# the `_strip_sizes` bug fixed below and has nothing to do with produce.
#
# The other check was against BRAND_ALIASES: of 113 candidate terms only "amla"
# appears in an alias ("dabur amla"), and that is harmless because "dabur" is
# not a commodity, so "Dabur Amla" keeps every token it needs to stay branded.
# Bare "Amla" is fruit and belongs here.
#
# RE-RUN BOTH CHECKS BEFORE ADDING A WORD TO THIS BLOCK. See
# tests/test_generic_products_produce.py, which encodes them.
# Fruit. Indian sheets mix English, Tamil and Hindi names freely.
_add(_PRODUCE,
"apple", "orange", "banana", "grape", "grapes", "mango", "pineapple",
"papaya", "guava", "pomegranate", "watermelon", "muskmelon", "melon",
"lemon", "lime", "mosambi", "sathukudi", "sapota", "chikoo", "jackfruit",
"fig", "anjeer", "pear", "peach", "plum", "apricot", "cherry", "cherries",
"strawberry", "blueberry", "kiwi", "litchi", "lychee", "amla", "gooseberry",
"avocado", "dragonfruit", "rambutan", "mangosteen", "starfruit", "jamun",
"ber", "plantain", "vazhaikkai")
# Vegetables. "gourd" covers the whole family once _PHRASES has collapsed the
# two-word forms (bitter gourd, bottle gourd, snake gourd, ridge gourd).
_add(_PRODUCE,
"tomato", "potato", "onion", "carrot", "beetroot", "beet", "radish",
"turnip", "cabbage", "cauliflower", "broccoli", "brinjal", "eggplant",
"aubergine", "okra", "bhindi", "cucumber", "pumpkin", "gourd", "drumstick",
"beans", "bean", "capsicum", "garlic", "ginger", "yam", "colocasia",
"tapioca", "arbi", "zucchini", "chayote", "sweetcorn", "babycorn",
"mushroom", "leek", "celery", "lettuce", "shallot", "springonion")
# Greens and fresh herbs, sold in bunches and never branded. NOTE that
# "coriander" and "methi" are deliberately NOT here: they are already spices in
# the block above, `_add` is last-wins, and re-adding them would silently move
# dhania powder out of Spices & Masalas. The leaf forms are collapsed to
# distinct tokens by _PHRASES instead.
_add(_GREENS,
"spinach", "palak", "amaranth", "keerai", "greens", "cilantro", "mint",
"pudina", "curryleaves", "basil", "thulasi", "tulsi", "parsley", "dill",
"sorrel", "moringa", "methileaves", "bunch")
# Flowers. Sold loose or by the metre of garland; the reason "Red Rose" used to
# land in a tea catalogue.
_add(_FLOWERS,
"flower", "flowers", "rose", "jasmine", "malli", "lotus", "marigold",
"samanthi", "chrysanthemum", "kanakambaram", "arali", "garland", "poo")
# Fish and seafood, sold fresh by weight.
_add(_SEAFOOD,
"fish", "prawn", "prawns", "shrimp", "crab", "squid", "tuna", "mackerel",
"sardine", "pomfret", "seer", "vanjaram", "anchovy", "nethili", "sole",
"tilapia", "salmon", "shellfish", "clam", "mussel")
# Eggs.
_add(_EGGS, "egg", "eggs", "muttai", "quail")
# Stragglers found by running the produce base list through is_unbranded and
# fixing every row it refused. Kept in one block so the next person adding to
# the seed list knows where the tail ends up.
_add(_PRODUCE, "dates", "custard", "dragonfruit", "ivy", "raw")
_add(_GREENS, "agathi", "ponnanganni", "keerai")
_add(_FLOWERS, "tuberose", "lily")
# Words that describe a product without naming a brand. Stripped before the
# all-tokens-are-commodities test, so "Organic Toor Dal Whole 1kg" still reads
# as unbranded.
@@ -150,6 +233,16 @@ QUALIFIERS: Set[str] = {
"bottle", "refill", "combo", "assorted", "mixed", "mix",
# connectives
"and", "with", "of", "the", "in", "for",
# Form words for fresh goods. "Leaves" is the important one: without it
# "Mint Leaves" keeps an unknown token and reads as a brand.
"leaves", "leaf", "bunch", "sweet", "broad", "cluster", "full", "toned",
"seedless", "ripe", "tender", "baby", "country", "hybrid", "nati",
# Varietal names. A variety qualifies a commodity, it does not brand it:
# an Alphonso mango is a mango. None of these appears in BRAND_ALIASES -
# the produce test asserts that, so a future addition cannot smuggle a
# real brand in through this list.
"robusta", "yelakki", "nendran", "alphonso", "banganapalli", "totapuri",
"malgova", "sindoora", "shimla", "ooty", "kashmiri",
}
# Multi-word commodities collapsed to a single token before tokenising, so the
@@ -175,16 +268,70 @@ _PHRASES = {
"brown sugar": "sugar",
"palm jaggery": "jaggery",
"cane sugar": "sugar",
# Fresh produce. The two-word gourds collapse onto "gourd" so the whole
# family is one lexicon entry. The leaf forms get their OWN tokens rather
# than reusing "coriander" / "methi": those are spices, _add is last-wins,
# and re-adding them under a greens category would silently move dhania
# powder out of Spices & Masalas.
"bitter gourd": "gourd",
"bottle gourd": "gourd",
"snake gourd": "gourd",
"ridge gourd": "gourd",
"ash gourd": "gourd",
"bitter guard": "gourd", # misspellings seen in real merchant data
"bottle ground": "gourd",
"lady finger": "okra",
"ladies finger": "okra",
"spring onion": "springonion",
"spring onions": "springonion",
"sweet potato": "potato",
"curry leaves": "curryleaves",
"curry leaf": "curryleaves",
"coriander leaves": "cilantro",
"methi leaves": "methileaves",
"fenugreek leaves": "methileaves",
"french beans": "beans",
"cluster beans": "beans",
"green peas": "peas",
"baby corn": "babycorn",
"sweet corn": "sweetcorn",
"tender coconut": "coconut",
"dragon fruit": "dragonfruit",
"custard apple": "apple",
"sweet lime": "mosambi",
"ivy gourd": "gourd",
"broad beans": "beans",
"cluster bean": "beans",
"quail egg": "egg",
"spring garlic": "garlic",
}
_WORD_RE = re.compile(r"[a-z]+")
# Units a pack size is actually written in. The strip below is bounded to these
# rather than to "any letters", because [a-z]* after a number ate the NEXT WORD:
# "24 Mantra Organic Moong Dal" became "organic moong dal", the brand was
# destroyed, and the row was then filed as an unbranded commodity. Every brand
# whose name begins with a number hit this. Keep the list tight - a unit added
# here is a word that can be deleted from a product name.
_UNITS = (
"kg|kgs|g|gm|gms|gram|grams|mg|ml|l|ltr|ltrs|litre|litres|liter|liters"
"|pc|pcs|piece|pieces|pack|packs|pkt|n|no|nos|x|cm|mm|inch|dozen"
)
_SIZE_RE = re.compile(
# "1kg", "500 g", "1.5 L" - a number followed by a REAL unit, optionally
# spaced - or a bare number, which is a quantity and never a brand.
r"\b\d+(?:[.,]\d+)?\s*(?:" + _UNITS + r")\b"
r"|\b\d+(?:[.,]\d+)?\b",
re.IGNORECASE,
)
def _strip_sizes(text: str) -> str:
"""Remove pack sizes and bare numbers - they never name a brand."""
# "1kg", "500 g", "1.5 L", and any leftover bare number.
text = re.sub(r"\b\d+(?:[.,]\d+)?\s*[a-z]*\b", " ", text)
return text
return _SIZE_RE.sub(" ", text)
def canonical_category(name: str) -> Optional[str]:

View File

@@ -54,6 +54,7 @@ from app.infrastructure.settings import (
from app.services.title_validator import find_category_conflicts
from app.services import price_estimator
from app.services import category_units as cu
from app.services.generic_products import OWN_PRODUCTS_BRAND
logger = logging.getLogger(__name__)
@@ -277,6 +278,26 @@ def validate_product(
sku_source = str(product.get("sku_source") or "").strip()
images = product.get("image_urls") or []
# A COMMODITY IS NOT A DEFECTIVE BRANDED PRODUCT.
#
# The penalties below treat a missing price_range or SKU as evidence that a
# row was fabricated, which is right for a scraped brand catalogue: a real
# Amul product has a shelf price and an article number, so their absence
# means something went wrong. Loose produce has neither, by nature. A shop
# prices apples by the day and does not issue article numbers for them.
#
# Left unqualified, the arithmetic rejected every produce row outright:
# 0.55 baseline - 0.30 (no price_range) - 0.10 (no SKU) = 0.15, against a
# reject threshold of 0.35. That is the exact opposite of the requirement
# these rows exist to satisfy, so the two absences stop counting as faults.
#
# EVERYTHING ELSE STILL APPLIES. Title sanity, placeholder detection,
# category resolution, the title/category contradiction check, size
# validity and unit compatibility, and the image-presence check all run
# unchanged - a blank or junk product name is still caught, and this is not
# a way in for rows that would otherwise fail.
commodity = brand == OWN_PRODUCTS_BRAND
report = ValidationReport(product_name=title or "(untitled)", grounded=grounded)
score = 0.55 # neutral baseline - moves up/down based on evidence below
@@ -345,16 +366,22 @@ def validate_product(
score -= 0.15
# 5. Price range ----------------------------------------------------------
ok, msg = validate_price_range(price_range, size, title, brand, category)
if not ok:
report.issues.append(ValidationIssue(field="price_range", severity="error", message=msg, penalty=0.30))
score -= 0.30
# Skipped for a commodity only when there is none. A band that IS present is
# still checked for being well formed, so a malformed one cannot hide here.
if price_range or not commodity:
ok, msg = validate_price_range(price_range, size, title, brand, category)
if not ok:
report.issues.append(ValidationIssue(field="price_range", severity="error", message=msg, penalty=0.30))
score -= 0.30
# 6. SKU --------------------------------------------------------------------
ok, msg = validate_sku(sku, sku_source)
if not ok:
report.issues.append(ValidationIssue(field="product_sku", severity="warning", message=msg, penalty=0.10))
score -= 0.10
# Same rule: a commodity is not expected to carry one, but a SKU the sheet
# did supply must still look like a SKU.
if sku or not commodity:
ok, msg = validate_sku(sku, sku_source)
if not ok:
report.issues.append(ValidationIssue(field="product_sku", severity="warning", message=msg, penalty=0.10))
score -= 0.10
# 7. Image presence -----------------------------------------------------
# Only meaningful if image search actually ran. When the operator disables