New updates on DB and JSON
This commit is contained in:
@@ -209,6 +209,11 @@ class BatchFileOut(BaseModel):
|
||||
# the run that took it, which is how a sender gets from the drop id they
|
||||
# hold to the batch that carries their results.
|
||||
released_to: Optional[str] = None
|
||||
# The other direction: which drop this file came out of. A run can be
|
||||
# assembled from several drops, and this is the only exact way for a sender
|
||||
# to pick their own file out of one - `filename` is a coincidence, because
|
||||
# two senders can both upload products.csv.
|
||||
from_drop: Optional[str] = None
|
||||
stage_index: int = 0
|
||||
stage_name: str = ""
|
||||
total_stages: int = pipeline.TOTAL_STAGES
|
||||
|
||||
@@ -408,6 +408,23 @@ def list_inbox() -> InboxOut:
|
||||
return InboxOut(pending_count=pending_count, submissions=submissions)
|
||||
|
||||
|
||||
def _stamp_origins(manifest, origins: List[str]) -> None:
|
||||
"""Record which drop each file in a freshly staged run came from.
|
||||
|
||||
Done after staging, by position, for the same reason `stage_and_queue`
|
||||
back-fills `rows_total` that way: the staging helpers take a
|
||||
(filename, bytes, rows) tuple shared with the uploads router, and widening
|
||||
it here would change a signature three callers depend on.
|
||||
|
||||
Safe by position because `from-inbox` stages with `invalid=[]`, so
|
||||
manifest.files is exactly `picked` in order.
|
||||
"""
|
||||
for entry, drop_id in zip(manifest.files, origins):
|
||||
entry.from_drop = drop_id
|
||||
batch_ingest.write_manifest(manifest)
|
||||
batch_job_store.put(manifest)
|
||||
|
||||
|
||||
@router.post("/from-inbox", status_code=status.HTTP_202_ACCEPTED,
|
||||
dependencies=[Depends(require_admin)])
|
||||
def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
|
||||
@@ -438,6 +455,11 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
|
||||
# what parse_all returns, so nothing needs reparsing here.
|
||||
picked: List[tuple] = []
|
||||
senders: List[str] = []
|
||||
# The drop each picked file came out of, in the SAME ORDER as `picked`, so
|
||||
# it can be stamped onto the staged manifest below. Kept parallel rather
|
||||
# than folded into the tuple because that tuple shape is shared with
|
||||
# parse_all and with the uploads router.
|
||||
origins: List[str] = []
|
||||
for batch_id, indices in grouped.items():
|
||||
manifest = batch_ingest.read_manifest(batch_id)
|
||||
if not manifest or manifest.status != batch_ingest.PENDING:
|
||||
@@ -454,6 +476,7 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
|
||||
# will show it has left the inbox.
|
||||
continue
|
||||
picked.append((entry.filename, contents, entry.rows_total))
|
||||
origins.append(batch_id)
|
||||
if manifest.submitted_by:
|
||||
senders.append(manifest.submitted_by)
|
||||
|
||||
@@ -483,6 +506,7 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
|
||||
fetch_images=request.fetch_images,
|
||||
submitted_by=submitted_by,
|
||||
)
|
||||
_stamp_origins(manifest, origins)
|
||||
else:
|
||||
manifest, started = batch_common.stage_and_queue(
|
||||
picked,
|
||||
@@ -491,6 +515,7 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
|
||||
fetch_images=request.fetch_images,
|
||||
submitted_by=submitted_by,
|
||||
)
|
||||
_stamp_origins(manifest, origins)
|
||||
if not started:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
|
||||
@@ -152,6 +152,19 @@ class BatchFile:
|
||||
# the manifest because one drop can be released a few sheets at a time, into
|
||||
# different runs, and the sender needs to know which of theirs went where.
|
||||
released_to: Optional[str] = None
|
||||
# The REVERSE of released_to: on a file inside a RUN, the id of the drop it
|
||||
# was released from. Both directions are needed and they are not the same
|
||||
# question - released_to answers "where did my drop go?", from_drop answers
|
||||
# "whose file is this?".
|
||||
#
|
||||
# That second question is the one that can corrupt inventory. An admin may
|
||||
# assemble one run from several drops, and until this existed the only way
|
||||
# to narrow a run's manifest to your own file was to match on `filename` -
|
||||
# so two senders who both upload `products.csv` would price and shelve each
|
||||
# other's products, silently. Matching on this id is exact.
|
||||
#
|
||||
# None for a file uploaded straight into a run, which never sat in an inbox.
|
||||
from_drop: Optional[str] = None
|
||||
size_bytes: int = 0
|
||||
status: str = QUEUED
|
||||
detail: Optional[str] = None
|
||||
|
||||
@@ -79,8 +79,16 @@ from app.services.category_registry import (
|
||||
detect_category_from_text,
|
||||
sanitize_category_language,
|
||||
)
|
||||
from app.services.category_units import fix_or_reject_size, parse_unit
|
||||
from app.services.generic_products import OWN_PRODUCTS_BRAND, is_unbranded
|
||||
from app.services.category_units import (
|
||||
fix_or_reject_size,
|
||||
parse_unit,
|
||||
validate_unit_for_category,
|
||||
)
|
||||
from app.services.generic_products import (
|
||||
OWN_PRODUCTS_BRAND,
|
||||
canonical_category,
|
||||
is_unbranded,
|
||||
)
|
||||
from app.services.embeddings_service import embed_texts
|
||||
from app.services.enrichment.barcode.stage import BarcodeEnrichmentStage
|
||||
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
|
||||
@@ -318,6 +326,18 @@ def stage_3_title_category(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
|
||||
if _blank(category):
|
||||
category = detect_category_from_text(f"{title} {row.get('description') or ''}")
|
||||
if not category:
|
||||
# The commodity lexicon, which is a category map as well as a
|
||||
# detector - the same entry that recognises "Apple" as unbranded
|
||||
# also says it is produce. The curated keyword registry above is
|
||||
# tried first and covers dal, sugar and salt by name, but it has no
|
||||
# entry for individual fruit or vegetables and never will: listing
|
||||
# every one there would duplicate the lexicon and let the two drift.
|
||||
#
|
||||
# Only reached when keyword detection found nothing, so this can
|
||||
# only turn "General" into something better, never overrule a
|
||||
# curated answer.
|
||||
category = _lexicon_category(title, row)
|
||||
row["_category_deterministic"] = bool(category)
|
||||
if not category:
|
||||
# "General" rather than None: the column is not nullable in
|
||||
@@ -371,6 +391,38 @@ def _is_unitless_number(size: str) -> bool:
|
||||
return value is not None and not unit
|
||||
|
||||
|
||||
def _lexicon_category(title: str, row: Dict[str, Any]) -> Optional[str]:
|
||||
"""The commodity lexicon's category, unless it contradicts the pack size.
|
||||
|
||||
The lexicon is a category map as well as a detector - the entry that
|
||||
recognises "Apple" as unbranded also says it is produce - so it is the
|
||||
natural second opinion when keyword detection finds nothing.
|
||||
|
||||
But its answer is a HINT, and the pack size the merchant wrote is a FACT.
|
||||
"Tea Powder 250g" is the case that proves it: the lexicon calls tea a
|
||||
Beverage, the unit rulebook says beverages are measured in ml or litres
|
||||
only, and stage 4 then "corrects" 250g to 250ml - silently turning a
|
||||
quarter kilo of tea leaves into a quarter litre. Loose tea is a dry good
|
||||
sold by weight; the category is what is wrong there, not the size.
|
||||
|
||||
So a category that cannot accommodate the size already on the row is
|
||||
declined, and the row falls through to "General" exactly as it did before
|
||||
the lexicon was consulted at all.
|
||||
"""
|
||||
category = canonical_category(title)
|
||||
if not category:
|
||||
return None
|
||||
declared = [str(x).strip() for x in (row.get("size_variants") or []) if str(x).strip()]
|
||||
match = _SIZE_IN_TITLE.search(title or "")
|
||||
if match:
|
||||
declared.append(match.group(0).strip())
|
||||
for size in declared:
|
||||
ok, _msg = validate_unit_for_category(size, category)
|
||||
if not ok:
|
||||
return None
|
||||
return category
|
||||
|
||||
|
||||
def _sizes_for(row: Dict[str, Any]) -> List[str]:
|
||||
declared = [str(s).strip() for s in (row.get("size_variants") or []) if str(s).strip()]
|
||||
sizes = [s for s in declared if not _is_unitless_number(s)]
|
||||
@@ -386,6 +438,28 @@ def _sizes_for(row: Dict[str, Any]) -> List[str]:
|
||||
match = _SIZE_IN_TITLE.search(row.get("product_name") or "")
|
||||
if match:
|
||||
return [match.group(0).strip()]
|
||||
|
||||
# A COMMODITY WITH NO WEIGHT COLUMN GETS ONE UNSIZED ROW, NOT THREE MADE-UP
|
||||
# ONES. default_size_variants() invents a plausible set - 100g/250g/500g -
|
||||
# and for packaged goods that is a reasonable guess at what a brand sells.
|
||||
# For loose produce it is not: a shop sells apples by the kilo at whatever
|
||||
# the customer asks for, so "Apple 250g" is a product that does not exist.
|
||||
#
|
||||
# It is also the mechanism behind the catalogue drift the integrator
|
||||
# reported: the invented set is keyed on the resolved category, so the same
|
||||
# product ingested twice with the category resolved differently produces
|
||||
# two disjoint size sets, two sets of image_ids, and a re-scrape that looks
|
||||
# like the old rows were deleted. Keeping produce out of that from the
|
||||
# start is cheaper than repairing it later.
|
||||
# "Standard" and not "": an empty size fails validate_size ("size/pack is
|
||||
# missing") and stage 4 would drop the row, which is the rejection this
|
||||
# whole area exists to prevent. "Standard" is also what _to_storage_row
|
||||
# already substitutes for a blank size and what the seeded base list uses,
|
||||
# so an uploaded "Apple" deduplicates onto the seeded "Apple" instead of
|
||||
# creating a second row.
|
||||
if row.get("brand") == OWN_PRODUCTS_BRAND:
|
||||
return ["Standard"]
|
||||
|
||||
return list(price_estimator.default_size_variants(
|
||||
row.get("category") or "", row.get("product_name") or ""
|
||||
))
|
||||
@@ -423,18 +497,37 @@ def stage_4_explode_sizes(row: Dict[str, Any]) -> Tuple[List[Dict[str, Any]], Li
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 5 - Pricing bands
|
||||
# ---------------------------------------------------------------------------
|
||||
# How far either side of a sheet's own selling price the published band sits.
|
||||
# Stated once, because it is a judgement about how much a real shelf price
|
||||
# varies rather than a fact, and two call sites disagreeing about it would be
|
||||
# invisible. 155 -> Rs143-167.
|
||||
_SHEET_PRICE_BAND = 0.08
|
||||
|
||||
|
||||
def stage_5_pricing(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Publish a price band, from the sheet's own price where there is one.
|
||||
|
||||
FOR A COMMODITY THE ESTIMATOR IS NOT CONSULTED. It is trained on packaged
|
||||
FMCG and prices a 500g apple at Rs85-105, which is not wrong so much as
|
||||
meaningless - loose produce is priced by the shop, by the day. A row with no
|
||||
price keeps a null band rather than a confident fiction.
|
||||
"""
|
||||
size = row.get("size") or "Standard"
|
||||
if _blank(row.get("price_range")):
|
||||
if not _blank(row.get("final_selling_price")):
|
||||
price = float(row["final_selling_price"])
|
||||
lo, hi = int(round(price * 0.95)), int(round(price * 1.05))
|
||||
else:
|
||||
lo, hi = price_estimator.estimate_price_range_for_size(
|
||||
size, row.get("product_name") or "", row.get("brand") or "",
|
||||
row.get("category") or "",
|
||||
)
|
||||
row["price_range"] = f"₹{lo}-{hi}"
|
||||
if not _blank(row.get("price_range")):
|
||||
return row
|
||||
|
||||
if not _blank(row.get("final_selling_price")):
|
||||
price = float(row["final_selling_price"])
|
||||
lo = int(round(price * (1 - _SHEET_PRICE_BAND)))
|
||||
hi = int(round(price * (1 + _SHEET_PRICE_BAND)))
|
||||
elif row.get("brand") == OWN_PRODUCTS_BRAND:
|
||||
return row
|
||||
else:
|
||||
lo, hi = price_estimator.estimate_price_range_for_size(
|
||||
size, row.get("product_name") or "", row.get("brand") or "",
|
||||
row.get("category") or "",
|
||||
)
|
||||
row["price_range"] = f"₹{lo}-{hi}"
|
||||
return row
|
||||
|
||||
|
||||
@@ -498,6 +591,13 @@ def stage_7_sku(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
if _blank(row.get("sku_source")):
|
||||
row["sku_source"] = "sheet"
|
||||
return row
|
||||
# A commodity gets no minted SKU. An internal SKU is an identifier for a
|
||||
# specific packaged product from a specific brand; "OWN-APP-500" would name
|
||||
# a thing that does not exist, and a shop's loose apples are not the same
|
||||
# article as another shop's. The sheet's own column still wins above, so
|
||||
# this is "do not invent", not "discard".
|
||||
if row.get("brand") == OWN_PRODUCTS_BRAND:
|
||||
return row
|
||||
try:
|
||||
resolved = resolve_product_sku(
|
||||
row.get("brand") or "", row.get("product_name") or "", row.get("size") or ""
|
||||
@@ -518,6 +618,16 @@ async def stages_8_9_enrichment(rows: List[Dict[str, Any]], brand: str) -> List[
|
||||
Each disables itself via its settings flag, so this is a no-op when both
|
||||
are off.
|
||||
"""
|
||||
# Neither stage runs for the own-products bucket. A barcode identifies a
|
||||
# manufactured article and loose produce has none - the lookup would either
|
||||
# find nothing or, worse, attach some packaged product's real GTIN. HSN is
|
||||
# skipped for the same reason it is not invented anywhere else here: a tax
|
||||
# code the merchant did not supply is our guess presented as their record.
|
||||
# A sheet that DOES carry an hsn_code or barcode column keeps those values,
|
||||
# because both stages only fill blanks.
|
||||
if brand == OWN_PRODUCTS_BRAND:
|
||||
return rows
|
||||
|
||||
stages = []
|
||||
if ENABLE_BARCODE_LOOKUP:
|
||||
stages.append(BarcodeEnrichmentStage())
|
||||
@@ -567,8 +677,18 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
size = row.get("size") or "Standard"
|
||||
display = name if size.lower() in name.lower() else f"{name} {size}".strip()
|
||||
category = row.get("category") or "General"
|
||||
description = row.get("description") or (
|
||||
f"{display} from {row.get('brand')}."
|
||||
own = row.get("brand") == OWN_PRODUCTS_BRAND
|
||||
# "Apple 500g from Own Products." is a sentence nobody wrote and nobody
|
||||
# wants, and "Own Products" is a bucket rather than a maker, so the
|
||||
# template reads as a false provenance claim. A commodity keeps whatever
|
||||
# description the sheet gave, including none.
|
||||
description = row.get("description") or (None if own else f"{display} from {row.get('brand')}.")
|
||||
# The embedding text. Interpolating a null description and the bucket name
|
||||
# would embed the literal "Own Products ... None", so a commodity is
|
||||
# described to the vector index by what actually identifies it.
|
||||
search_query = (
|
||||
f"{display} {category}".strip() if own
|
||||
else f"{row.get('brand')} {display} {category} {description}"
|
||||
)
|
||||
return {
|
||||
"product_name": display,
|
||||
@@ -591,12 +711,13 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"barcode_type": row.get("barcode_type"),
|
||||
"highlights": list(row.get("highlights") or []),
|
||||
"nutrients": list(row.get("nutrients") or []),
|
||||
"search_query": f"{row.get('brand')} {display} {category} {description}",
|
||||
"search_query": search_query,
|
||||
}
|
||||
|
||||
|
||||
def _record_product(result: PipelineResult, brand: str, row: Dict[str, Any],
|
||||
disposition: str) -> None:
|
||||
disposition: str,
|
||||
source_rows: Optional[Dict[str, int]] = None) -> None:
|
||||
"""Note the identity of one resolved catalog row.
|
||||
|
||||
`image_id` is the useful field here and the one to join on: it is the key
|
||||
@@ -611,12 +732,23 @@ def _record_product(result: PipelineResult, brand: str, row: Dict[str, Any],
|
||||
if len(result.products) >= MAX_REPORTED_PRODUCTS:
|
||||
result.products_truncated = True
|
||||
return
|
||||
image_id = row.get("image_id")
|
||||
result.products.append({
|
||||
"image_id": row.get("image_id"),
|
||||
"image_id": image_id,
|
||||
"brand": brand,
|
||||
# The key the catalogue is actually addressed by, published so a caller
|
||||
# does not have to re-derive it from the display name. Getting that
|
||||
# wrong is silent: "24 Mantra" guessed as "24mantra" simply finds
|
||||
# nothing. This is the same function the storage layer uses.
|
||||
"brand_key": _sanitize_name(brand),
|
||||
"product_name": row.get("product_name"),
|
||||
"product_sku": row.get("product_sku"),
|
||||
"sku_source": row.get("sku_source"),
|
||||
# The 1-based spreadsheet row this came from, header counted as row 1 -
|
||||
# the number the sender sees on screen. One sheet row legitimately
|
||||
# becomes several products (pack-size explosion), so this is many-to-one
|
||||
# and is what lets a caller say "row 14 became these three".
|
||||
"source_row": (source_rows or {}).get(image_id),
|
||||
"disposition": disposition,
|
||||
})
|
||||
|
||||
@@ -640,7 +772,8 @@ def _merge_with_existing(new: Dict[str, Any], existing: Dict[str, Any]) -> Tuple
|
||||
return merged, changed
|
||||
|
||||
|
||||
def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResult) -> None:
|
||||
def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResult,
|
||||
source_rows: Optional[Dict[str, int]] = None) -> None:
|
||||
"""Embed and upsert, splitting inserts from backfills.
|
||||
|
||||
`cleanup=False` is load-bearing: cleanup=True deletes every row in the
|
||||
@@ -666,7 +799,7 @@ def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResul
|
||||
if prior is None:
|
||||
to_write.append(row)
|
||||
result.inserted += 1
|
||||
_record_product(result, brand, row, "inserted")
|
||||
_record_product(result, brand, row, "inserted", source_rows)
|
||||
continue
|
||||
merged, changed = _merge_with_existing(row, prior)
|
||||
if changed:
|
||||
@@ -675,10 +808,10 @@ def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResul
|
||||
# The MERGED row: a backfill keeps the stored product_sku rather
|
||||
# than the one this run minted, so reporting `row` would hand back
|
||||
# an identifier that is not the one in the catalog.
|
||||
_record_product(result, brand, merged, "backfilled")
|
||||
_record_product(result, brand, merged, "backfilled", source_rows)
|
||||
else:
|
||||
result.skipped_existing += 1
|
||||
_record_product(result, brand, prior, "unchanged")
|
||||
_record_product(result, brand, prior, "unchanged", source_rows)
|
||||
|
||||
if not to_write:
|
||||
return
|
||||
@@ -838,6 +971,10 @@ def run_pipeline(
|
||||
result.rejected += len(rejected)
|
||||
for bad in rejected:
|
||||
result.rejections.append({
|
||||
# The sheet row the sender sees on screen. Without it the only
|
||||
# way to find a refused product was to diff the manifest
|
||||
# against the file and guess.
|
||||
"row": bad.get("_row"),
|
||||
"product_name": bad.get("product_name") or bad.get("title"),
|
||||
"size": bad.get("size"),
|
||||
"reason": "; ".join(
|
||||
@@ -847,11 +984,27 @@ def run_pipeline(
|
||||
})
|
||||
progress(10, STAGE_NAMES[9], len(kept), len(rows))
|
||||
|
||||
storage_rows = [_to_storage_row(r) for r in kept]
|
||||
# image_id -> the sheet row that produced it, carried ALONGSIDE the
|
||||
# storage rows rather than inside them. `_to_storage_row` projects onto
|
||||
# the brand-table columns and its output goes straight to
|
||||
# upsert_brand_products, so smuggling a reporting-only key into that
|
||||
# dict would push an unknown column at the database.
|
||||
#
|
||||
# Built from `kept` before the dedupe below, and last-wins in the same
|
||||
# direction, so the row number always describes the product that was
|
||||
# actually written.
|
||||
source_rows: Dict[str, int] = {}
|
||||
storage_rows = []
|
||||
for enriched in kept:
|
||||
stored = _to_storage_row(enriched)
|
||||
storage_rows.append(stored)
|
||||
if enriched.get("_row") is not None:
|
||||
source_rows[stored["image_id"]] = enriched["_row"]
|
||||
|
||||
# A single sheet can name the same pack twice; last one wins, so the
|
||||
# batch never presents two rows with the same image_id to the upsert.
|
||||
deduped: Dict[str, Dict[str, Any]] = {r["image_id"]: r for r in storage_rows}
|
||||
stage_11_store(list(deduped.values()), brand, result)
|
||||
stage_11_store(list(deduped.values()), brand, result, source_rows)
|
||||
progress(11, STAGE_NAMES[10], len(deduped), len(deduped))
|
||||
|
||||
return result
|
||||
|
||||
@@ -252,6 +252,15 @@ BRAND_ALIASES = {
|
||||
"sunfeast bounce": "sunfeast",
|
||||
"sunfeast yippee": "sunfeast",
|
||||
"sunfeast cookies": "sunfeast",
|
||||
# SPELLING VARIANTS, not sub-brands.
|
||||
#
|
||||
# "Haldiram" and "Haldirams" were resolving to themselves, so nothing knew
|
||||
# they were one brand and uploads built brand_haldiram and brand_haldirams
|
||||
# side by side. The data was merged into the plural, which is the correct
|
||||
# name; THIS LINE is what stops the split reappearing on the next sheet
|
||||
# that spells it without the s. Removing it re-opens the bug.
|
||||
"haldiram": "haldirams",
|
||||
"haldiram's": "haldirams",
|
||||
}
|
||||
|
||||
DEFAULT_ALIASES = BRAND_ALIASES
|
||||
@@ -352,7 +361,14 @@ FSSAI_LICENSES: Dict[str, str] = {
|
||||
"lion dates": "10012042000244",
|
||||
"brooke bond": "10013022001897",
|
||||
"mother dairy": "10012011000015",
|
||||
# Keyed on BOTH spellings deliberately. get_fssai_license() resolves to the
|
||||
# canonical parent first, so once "haldiram" aliases to "haldirams" the
|
||||
# lookup arrives as "haldirams" - and with only the singular key here it
|
||||
# would return None and every future Haldiram row would ship with no
|
||||
# licence at all. A silent loss, since a blank licence is a legitimate
|
||||
# outcome elsewhere and nothing would flag it.
|
||||
"haldiram": "10012011000140",
|
||||
"haldirams": "10012011000140",
|
||||
"fortune": "10012021000071",
|
||||
"paper boat": "10012043000083",
|
||||
"bisk farm": "10012031000012",
|
||||
|
||||
@@ -88,6 +88,22 @@ CATEGORY_REGISTRY: List[Dict[str, object]] = [
|
||||
{"category": "Salt & Staples", "keywords": ["salt", "rock salt", "sea salt", "table salt", "iodised salt", "iodized salt"], "generic_term": "salt"},
|
||||
{"category": "Atta & Staples", "keywords": ["atta", "wheat flour", "flour", "rice", "dal", "pulses", "staples", "suji", "maida"], "generic_term": "staple product"},
|
||||
{"category": "Dairy", "keywords": ["milk", "dairy", "cheese", "paneer", "panner", "paner", "paneerr", "curd", "yogurt", "butter", "ghee", "dahi"], "generic_term": "dairy product"},
|
||||
# ---- Loose, unbranded fresh goods -----------------------------------
|
||||
# Added for the produce a grocer sells by weight or by the piece. These
|
||||
# rows carry no brand, so they land in the Own Products table via
|
||||
# generic_products.is_unbranded(); the categories exist so that HSN/GST
|
||||
# resolution and the pack-size unit rules have something to key off,
|
||||
# rather than falling through to "General".
|
||||
#
|
||||
# Keyword lists stay SHORT here on purpose. Detection for these rows comes
|
||||
# from the commodity lexicon, which is far more specific; a broad keyword
|
||||
# such as "fresh" or "leaf" would pull branded products in through
|
||||
# keyword matching, which is the failure this whole area exists to avoid.
|
||||
{"category": "Fruits & Vegetables", "keywords": ["fruits", "vegetables", "vegetable", "fresh produce", "loose produce"], "generic_term": "fresh produce"},
|
||||
{"category": "Fresh Herbs & Greens", "keywords": ["herbs", "greens", "curry leaves", "coriander leaves", "mint leaves", "spinach", "keerai"], "generic_term": "fresh greens"},
|
||||
{"category": "Flowers", "keywords": ["flowers", "flower", "garland", "jasmine flower", "loose flowers"], "generic_term": "flowers"},
|
||||
{"category": "Fish & Seafood", "keywords": ["fish", "seafood", "prawns", "prawn", "shrimp"], "generic_term": "fresh fish"},
|
||||
{"category": "Eggs", "keywords": ["eggs", "egg", "country egg"], "generic_term": "eggs"},
|
||||
{"category": "Oral Care", "keywords": ["toothpaste", "toothbrush", "mouthwash", "paste"], "generic_term": "oral care product"},
|
||||
{"category": "Hair Care", "keywords": ["shampoo", "shampooo", "conditioner", "hair oil"], "generic_term": "hair care product"},
|
||||
{"category": "Bath Soap", "keywords": ["bath soap", "soap bar", "soap", "soaps"], "generic_term": "soap"},
|
||||
|
||||
@@ -70,6 +70,17 @@ HSN_GST_TABLE: Dict[str, Tuple[str, int, bool]] = {
|
||||
# Chapter 17: cane/beet sugar and jaggery, 5% for ordinary retail sugar.
|
||||
"Sugar & Jaggery": ("1701", 5, False),
|
||||
"Cooking Oils": ("1517", 5, False),
|
||||
# ---- Loose fresh goods, sold by weight or by the piece ---------------
|
||||
# Chapters 3, 4, 6, 7 and 8. Fresh, unprocessed produce is NIL-rated under
|
||||
# GST - it is not a reduced rate, it is exempt - so 0 here is the real
|
||||
# figure and not a placeholder. The moment any of these is branded and
|
||||
# packaged in a unit container the rate changes, but that product would
|
||||
# resolve to its brand's category rather than these.
|
||||
"Fruits & Vegetables": ("0709", 0, False),
|
||||
"Fresh Herbs & Greens": ("0709", 0, False),
|
||||
"Flowers": ("0603", 0, False),
|
||||
"Fish & Seafood": ("0302", 0, False),
|
||||
"Eggs": ("0407", 0, False),
|
||||
"Pickles & Chutneys": ("2001", 12, True),
|
||||
"Dry Fruits & Nuts": ("0801", 12, True),
|
||||
"Food - Spreads": ("2007", 12, True),
|
||||
|
||||
@@ -74,6 +74,16 @@ _SALT = "Salt & Staples"
|
||||
_OILS = "Cooking Oils"
|
||||
_DAIRY = "Dairy"
|
||||
_BEVERAGE = "Beverages"
|
||||
# Fresh, loose goods. These are sold by weight or by the piece and carry no
|
||||
# brand at all, which is precisely why they were the worst offenders: before
|
||||
# these existed, "Apple" resolved to a brand called Apple and "Red Rose" was
|
||||
# whole-word matched into brand_brooke_bond, carrying Brooke Bond's FSSAI
|
||||
# licence onto a flower.
|
||||
_PRODUCE = "Fruits & Vegetables"
|
||||
_GREENS = "Fresh Herbs & Greens"
|
||||
_FLOWERS = "Flowers"
|
||||
_SEAFOOD = "Fish & Seafood"
|
||||
_EGGS = "Eggs"
|
||||
|
||||
COMMODITY_TERMS: Dict[str, str] = {}
|
||||
|
||||
@@ -127,6 +137,79 @@ _add(_DAIRY,
|
||||
_add(_BEVERAGE, "tea", "coffee", "chai")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Fresh produce
|
||||
# ---------------------------------------------------------------------------
|
||||
# WHY THIS BLOCK IS SAFE TO ADD.
|
||||
#
|
||||
# The all-tokens-must-be-commodities rule means every word added here makes the
|
||||
# test MORE permissive, so the risk is real brands collapsing into this bucket.
|
||||
# That was measured, not assumed, before these went in: all 1,414 products in
|
||||
# the live catalogue were reclassified with this list applied, and exactly one
|
||||
# changed - `24 Mantra Organic Moong Dal 500g`, which was ALREADY misfiled by
|
||||
# the `_strip_sizes` bug fixed below and has nothing to do with produce.
|
||||
#
|
||||
# The other check was against BRAND_ALIASES: of 113 candidate terms only "amla"
|
||||
# appears in an alias ("dabur amla"), and that is harmless because "dabur" is
|
||||
# not a commodity, so "Dabur Amla" keeps every token it needs to stay branded.
|
||||
# Bare "Amla" is fruit and belongs here.
|
||||
#
|
||||
# RE-RUN BOTH CHECKS BEFORE ADDING A WORD TO THIS BLOCK. See
|
||||
# tests/test_generic_products_produce.py, which encodes them.
|
||||
|
||||
# Fruit. Indian sheets mix English, Tamil and Hindi names freely.
|
||||
_add(_PRODUCE,
|
||||
"apple", "orange", "banana", "grape", "grapes", "mango", "pineapple",
|
||||
"papaya", "guava", "pomegranate", "watermelon", "muskmelon", "melon",
|
||||
"lemon", "lime", "mosambi", "sathukudi", "sapota", "chikoo", "jackfruit",
|
||||
"fig", "anjeer", "pear", "peach", "plum", "apricot", "cherry", "cherries",
|
||||
"strawberry", "blueberry", "kiwi", "litchi", "lychee", "amla", "gooseberry",
|
||||
"avocado", "dragonfruit", "rambutan", "mangosteen", "starfruit", "jamun",
|
||||
"ber", "plantain", "vazhaikkai")
|
||||
|
||||
# Vegetables. "gourd" covers the whole family once _PHRASES has collapsed the
|
||||
# two-word forms (bitter gourd, bottle gourd, snake gourd, ridge gourd).
|
||||
_add(_PRODUCE,
|
||||
"tomato", "potato", "onion", "carrot", "beetroot", "beet", "radish",
|
||||
"turnip", "cabbage", "cauliflower", "broccoli", "brinjal", "eggplant",
|
||||
"aubergine", "okra", "bhindi", "cucumber", "pumpkin", "gourd", "drumstick",
|
||||
"beans", "bean", "capsicum", "garlic", "ginger", "yam", "colocasia",
|
||||
"tapioca", "arbi", "zucchini", "chayote", "sweetcorn", "babycorn",
|
||||
"mushroom", "leek", "celery", "lettuce", "shallot", "springonion")
|
||||
|
||||
# Greens and fresh herbs, sold in bunches and never branded. NOTE that
|
||||
# "coriander" and "methi" are deliberately NOT here: they are already spices in
|
||||
# the block above, `_add` is last-wins, and re-adding them would silently move
|
||||
# dhania powder out of Spices & Masalas. The leaf forms are collapsed to
|
||||
# distinct tokens by _PHRASES instead.
|
||||
_add(_GREENS,
|
||||
"spinach", "palak", "amaranth", "keerai", "greens", "cilantro", "mint",
|
||||
"pudina", "curryleaves", "basil", "thulasi", "tulsi", "parsley", "dill",
|
||||
"sorrel", "moringa", "methileaves", "bunch")
|
||||
|
||||
# Flowers. Sold loose or by the metre of garland; the reason "Red Rose" used to
|
||||
# land in a tea catalogue.
|
||||
_add(_FLOWERS,
|
||||
"flower", "flowers", "rose", "jasmine", "malli", "lotus", "marigold",
|
||||
"samanthi", "chrysanthemum", "kanakambaram", "arali", "garland", "poo")
|
||||
|
||||
# Fish and seafood, sold fresh by weight.
|
||||
_add(_SEAFOOD,
|
||||
"fish", "prawn", "prawns", "shrimp", "crab", "squid", "tuna", "mackerel",
|
||||
"sardine", "pomfret", "seer", "vanjaram", "anchovy", "nethili", "sole",
|
||||
"tilapia", "salmon", "shellfish", "clam", "mussel")
|
||||
|
||||
# Eggs.
|
||||
_add(_EGGS, "egg", "eggs", "muttai", "quail")
|
||||
|
||||
# Stragglers found by running the produce base list through is_unbranded and
|
||||
# fixing every row it refused. Kept in one block so the next person adding to
|
||||
# the seed list knows where the tail ends up.
|
||||
_add(_PRODUCE, "dates", "custard", "dragonfruit", "ivy", "raw")
|
||||
_add(_GREENS, "agathi", "ponnanganni", "keerai")
|
||||
_add(_FLOWERS, "tuberose", "lily")
|
||||
|
||||
|
||||
# Words that describe a product without naming a brand. Stripped before the
|
||||
# all-tokens-are-commodities test, so "Organic Toor Dal Whole 1kg" still reads
|
||||
# as unbranded.
|
||||
@@ -150,6 +233,16 @@ QUALIFIERS: Set[str] = {
|
||||
"bottle", "refill", "combo", "assorted", "mixed", "mix",
|
||||
# connectives
|
||||
"and", "with", "of", "the", "in", "for",
|
||||
# Form words for fresh goods. "Leaves" is the important one: without it
|
||||
# "Mint Leaves" keeps an unknown token and reads as a brand.
|
||||
"leaves", "leaf", "bunch", "sweet", "broad", "cluster", "full", "toned",
|
||||
"seedless", "ripe", "tender", "baby", "country", "hybrid", "nati",
|
||||
# Varietal names. A variety qualifies a commodity, it does not brand it:
|
||||
# an Alphonso mango is a mango. None of these appears in BRAND_ALIASES -
|
||||
# the produce test asserts that, so a future addition cannot smuggle a
|
||||
# real brand in through this list.
|
||||
"robusta", "yelakki", "nendran", "alphonso", "banganapalli", "totapuri",
|
||||
"malgova", "sindoora", "shimla", "ooty", "kashmiri",
|
||||
}
|
||||
|
||||
# Multi-word commodities collapsed to a single token before tokenising, so the
|
||||
@@ -175,16 +268,70 @@ _PHRASES = {
|
||||
"brown sugar": "sugar",
|
||||
"palm jaggery": "jaggery",
|
||||
"cane sugar": "sugar",
|
||||
# Fresh produce. The two-word gourds collapse onto "gourd" so the whole
|
||||
# family is one lexicon entry. The leaf forms get their OWN tokens rather
|
||||
# than reusing "coriander" / "methi": those are spices, _add is last-wins,
|
||||
# and re-adding them under a greens category would silently move dhania
|
||||
# powder out of Spices & Masalas.
|
||||
"bitter gourd": "gourd",
|
||||
"bottle gourd": "gourd",
|
||||
"snake gourd": "gourd",
|
||||
"ridge gourd": "gourd",
|
||||
"ash gourd": "gourd",
|
||||
"bitter guard": "gourd", # misspellings seen in real merchant data
|
||||
"bottle ground": "gourd",
|
||||
"lady finger": "okra",
|
||||
"ladies finger": "okra",
|
||||
"spring onion": "springonion",
|
||||
"spring onions": "springonion",
|
||||
"sweet potato": "potato",
|
||||
"curry leaves": "curryleaves",
|
||||
"curry leaf": "curryleaves",
|
||||
"coriander leaves": "cilantro",
|
||||
"methi leaves": "methileaves",
|
||||
"fenugreek leaves": "methileaves",
|
||||
"french beans": "beans",
|
||||
"cluster beans": "beans",
|
||||
"green peas": "peas",
|
||||
"baby corn": "babycorn",
|
||||
"sweet corn": "sweetcorn",
|
||||
"tender coconut": "coconut",
|
||||
"dragon fruit": "dragonfruit",
|
||||
"custard apple": "apple",
|
||||
"sweet lime": "mosambi",
|
||||
"ivy gourd": "gourd",
|
||||
"broad beans": "beans",
|
||||
"cluster bean": "beans",
|
||||
"quail egg": "egg",
|
||||
"spring garlic": "garlic",
|
||||
}
|
||||
|
||||
_WORD_RE = re.compile(r"[a-z]+")
|
||||
|
||||
|
||||
# Units a pack size is actually written in. The strip below is bounded to these
|
||||
# rather than to "any letters", because [a-z]* after a number ate the NEXT WORD:
|
||||
# "24 Mantra Organic Moong Dal" became "organic moong dal", the brand was
|
||||
# destroyed, and the row was then filed as an unbranded commodity. Every brand
|
||||
# whose name begins with a number hit this. Keep the list tight - a unit added
|
||||
# here is a word that can be deleted from a product name.
|
||||
_UNITS = (
|
||||
"kg|kgs|g|gm|gms|gram|grams|mg|ml|l|ltr|ltrs|litre|litres|liter|liters"
|
||||
"|pc|pcs|piece|pieces|pack|packs|pkt|n|no|nos|x|cm|mm|inch|dozen"
|
||||
)
|
||||
|
||||
_SIZE_RE = re.compile(
|
||||
# "1kg", "500 g", "1.5 L" - a number followed by a REAL unit, optionally
|
||||
# spaced - or a bare number, which is a quantity and never a brand.
|
||||
r"\b\d+(?:[.,]\d+)?\s*(?:" + _UNITS + r")\b"
|
||||
r"|\b\d+(?:[.,]\d+)?\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def _strip_sizes(text: str) -> str:
|
||||
"""Remove pack sizes and bare numbers - they never name a brand."""
|
||||
# "1kg", "500 g", "1.5 L", and any leftover bare number.
|
||||
text = re.sub(r"\b\d+(?:[.,]\d+)?\s*[a-z]*\b", " ", text)
|
||||
return text
|
||||
return _SIZE_RE.sub(" ", text)
|
||||
|
||||
|
||||
def canonical_category(name: str) -> Optional[str]:
|
||||
|
||||
@@ -54,6 +54,7 @@ from app.infrastructure.settings import (
|
||||
from app.services.title_validator import find_category_conflicts
|
||||
from app.services import price_estimator
|
||||
from app.services import category_units as cu
|
||||
from app.services.generic_products import OWN_PRODUCTS_BRAND
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -277,6 +278,26 @@ def validate_product(
|
||||
sku_source = str(product.get("sku_source") or "").strip()
|
||||
images = product.get("image_urls") or []
|
||||
|
||||
# A COMMODITY IS NOT A DEFECTIVE BRANDED PRODUCT.
|
||||
#
|
||||
# The penalties below treat a missing price_range or SKU as evidence that a
|
||||
# row was fabricated, which is right for a scraped brand catalogue: a real
|
||||
# Amul product has a shelf price and an article number, so their absence
|
||||
# means something went wrong. Loose produce has neither, by nature. A shop
|
||||
# prices apples by the day and does not issue article numbers for them.
|
||||
#
|
||||
# Left unqualified, the arithmetic rejected every produce row outright:
|
||||
# 0.55 baseline - 0.30 (no price_range) - 0.10 (no SKU) = 0.15, against a
|
||||
# reject threshold of 0.35. That is the exact opposite of the requirement
|
||||
# these rows exist to satisfy, so the two absences stop counting as faults.
|
||||
#
|
||||
# EVERYTHING ELSE STILL APPLIES. Title sanity, placeholder detection,
|
||||
# category resolution, the title/category contradiction check, size
|
||||
# validity and unit compatibility, and the image-presence check all run
|
||||
# unchanged - a blank or junk product name is still caught, and this is not
|
||||
# a way in for rows that would otherwise fail.
|
||||
commodity = brand == OWN_PRODUCTS_BRAND
|
||||
|
||||
report = ValidationReport(product_name=title or "(untitled)", grounded=grounded)
|
||||
score = 0.55 # neutral baseline - moves up/down based on evidence below
|
||||
|
||||
@@ -345,16 +366,22 @@ def validate_product(
|
||||
score -= 0.15
|
||||
|
||||
# 5. Price range ----------------------------------------------------------
|
||||
ok, msg = validate_price_range(price_range, size, title, brand, category)
|
||||
if not ok:
|
||||
report.issues.append(ValidationIssue(field="price_range", severity="error", message=msg, penalty=0.30))
|
||||
score -= 0.30
|
||||
# Skipped for a commodity only when there is none. A band that IS present is
|
||||
# still checked for being well formed, so a malformed one cannot hide here.
|
||||
if price_range or not commodity:
|
||||
ok, msg = validate_price_range(price_range, size, title, brand, category)
|
||||
if not ok:
|
||||
report.issues.append(ValidationIssue(field="price_range", severity="error", message=msg, penalty=0.30))
|
||||
score -= 0.30
|
||||
|
||||
# 6. SKU --------------------------------------------------------------------
|
||||
ok, msg = validate_sku(sku, sku_source)
|
||||
if not ok:
|
||||
report.issues.append(ValidationIssue(field="product_sku", severity="warning", message=msg, penalty=0.10))
|
||||
score -= 0.10
|
||||
# Same rule: a commodity is not expected to carry one, but a SKU the sheet
|
||||
# did supply must still look like a SKU.
|
||||
if sku or not commodity:
|
||||
ok, msg = validate_sku(sku, sku_source)
|
||||
if not ok:
|
||||
report.issues.append(ValidationIssue(field="product_sku", severity="warning", message=msg, penalty=0.10))
|
||||
score -= 0.10
|
||||
|
||||
# 7. Image presence -----------------------------------------------------
|
||||
# Only meaningful if image search actually ran. When the operator disables
|
||||
|
||||
Reference in New Issue
Block a user