New updates on DB and JSON
This commit is contained in:
@@ -209,6 +209,11 @@ class BatchFileOut(BaseModel):
|
||||
# the run that took it, which is how a sender gets from the drop id they
|
||||
# hold to the batch that carries their results.
|
||||
released_to: Optional[str] = None
|
||||
# The other direction: which drop this file came out of. A run can be
|
||||
# assembled from several drops, and this is the only exact way for a sender
|
||||
# to pick their own file out of one - `filename` is a coincidence, because
|
||||
# two senders can both upload products.csv.
|
||||
from_drop: Optional[str] = None
|
||||
stage_index: int = 0
|
||||
stage_name: str = ""
|
||||
total_stages: int = pipeline.TOTAL_STAGES
|
||||
|
||||
@@ -408,6 +408,23 @@ def list_inbox() -> InboxOut:
|
||||
return InboxOut(pending_count=pending_count, submissions=submissions)
|
||||
|
||||
|
||||
def _stamp_origins(manifest, origins: List[str]) -> None:
|
||||
"""Record which drop each file in a freshly staged run came from.
|
||||
|
||||
Done after staging, by position, for the same reason `stage_and_queue`
|
||||
back-fills `rows_total` that way: the staging helpers take a
|
||||
(filename, bytes, rows) tuple shared with the uploads router, and widening
|
||||
it here would change a signature three callers depend on.
|
||||
|
||||
Safe by position because `from-inbox` stages with `invalid=[]`, so
|
||||
manifest.files is exactly `picked` in order.
|
||||
"""
|
||||
for entry, drop_id in zip(manifest.files, origins):
|
||||
entry.from_drop = drop_id
|
||||
batch_ingest.write_manifest(manifest)
|
||||
batch_job_store.put(manifest)
|
||||
|
||||
|
||||
@router.post("/from-inbox", status_code=status.HTTP_202_ACCEPTED,
|
||||
dependencies=[Depends(require_admin)])
|
||||
def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
|
||||
@@ -438,6 +455,11 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
|
||||
# what parse_all returns, so nothing needs reparsing here.
|
||||
picked: List[tuple] = []
|
||||
senders: List[str] = []
|
||||
# The drop each picked file came out of, in the SAME ORDER as `picked`, so
|
||||
# it can be stamped onto the staged manifest below. Kept parallel rather
|
||||
# than folded into the tuple because that tuple shape is shared with
|
||||
# parse_all and with the uploads router.
|
||||
origins: List[str] = []
|
||||
for batch_id, indices in grouped.items():
|
||||
manifest = batch_ingest.read_manifest(batch_id)
|
||||
if not manifest or manifest.status != batch_ingest.PENDING:
|
||||
@@ -454,6 +476,7 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
|
||||
# will show it has left the inbox.
|
||||
continue
|
||||
picked.append((entry.filename, contents, entry.rows_total))
|
||||
origins.append(batch_id)
|
||||
if manifest.submitted_by:
|
||||
senders.append(manifest.submitted_by)
|
||||
|
||||
@@ -483,6 +506,7 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
|
||||
fetch_images=request.fetch_images,
|
||||
submitted_by=submitted_by,
|
||||
)
|
||||
_stamp_origins(manifest, origins)
|
||||
else:
|
||||
manifest, started = batch_common.stage_and_queue(
|
||||
picked,
|
||||
@@ -491,6 +515,7 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
|
||||
fetch_images=request.fetch_images,
|
||||
submitted_by=submitted_by,
|
||||
)
|
||||
_stamp_origins(manifest, origins)
|
||||
if not started:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
|
||||
@@ -152,6 +152,19 @@ class BatchFile:
|
||||
# the manifest because one drop can be released a few sheets at a time, into
|
||||
# different runs, and the sender needs to know which of theirs went where.
|
||||
released_to: Optional[str] = None
|
||||
# The REVERSE of released_to: on a file inside a RUN, the id of the drop it
|
||||
# was released from. Both directions are needed and they are not the same
|
||||
# question - released_to answers "where did my drop go?", from_drop answers
|
||||
# "whose file is this?".
|
||||
#
|
||||
# That second question is the one that can corrupt inventory. An admin may
|
||||
# assemble one run from several drops, and until this existed the only way
|
||||
# to narrow a run's manifest to your own file was to match on `filename` -
|
||||
# so two senders who both upload `products.csv` would price and shelve each
|
||||
# other's products, silently. Matching on this id is exact.
|
||||
#
|
||||
# None for a file uploaded straight into a run, which never sat in an inbox.
|
||||
from_drop: Optional[str] = None
|
||||
size_bytes: int = 0
|
||||
status: str = QUEUED
|
||||
detail: Optional[str] = None
|
||||
|
||||
@@ -79,8 +79,16 @@ from app.services.category_registry import (
|
||||
detect_category_from_text,
|
||||
sanitize_category_language,
|
||||
)
|
||||
from app.services.category_units import fix_or_reject_size, parse_unit
|
||||
from app.services.generic_products import OWN_PRODUCTS_BRAND, is_unbranded
|
||||
from app.services.category_units import (
|
||||
fix_or_reject_size,
|
||||
parse_unit,
|
||||
validate_unit_for_category,
|
||||
)
|
||||
from app.services.generic_products import (
|
||||
OWN_PRODUCTS_BRAND,
|
||||
canonical_category,
|
||||
is_unbranded,
|
||||
)
|
||||
from app.services.embeddings_service import embed_texts
|
||||
from app.services.enrichment.barcode.stage import BarcodeEnrichmentStage
|
||||
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
|
||||
@@ -318,6 +326,18 @@ def stage_3_title_category(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
|
||||
if _blank(category):
|
||||
category = detect_category_from_text(f"{title} {row.get('description') or ''}")
|
||||
if not category:
|
||||
# The commodity lexicon, which is a category map as well as a
|
||||
# detector - the same entry that recognises "Apple" as unbranded
|
||||
# also says it is produce. The curated keyword registry above is
|
||||
# tried first and covers dal, sugar and salt by name, but it has no
|
||||
# entry for individual fruit or vegetables and never will: listing
|
||||
# every one there would duplicate the lexicon and let the two drift.
|
||||
#
|
||||
# Only reached when keyword detection found nothing, so this can
|
||||
# only turn "General" into something better, never overrule a
|
||||
# curated answer.
|
||||
category = _lexicon_category(title, row)
|
||||
row["_category_deterministic"] = bool(category)
|
||||
if not category:
|
||||
# "General" rather than None: the column is not nullable in
|
||||
@@ -371,6 +391,38 @@ def _is_unitless_number(size: str) -> bool:
|
||||
return value is not None and not unit
|
||||
|
||||
|
||||
def _lexicon_category(title: str, row: Dict[str, Any]) -> Optional[str]:
|
||||
"""The commodity lexicon's category, unless it contradicts the pack size.
|
||||
|
||||
The lexicon is a category map as well as a detector - the entry that
|
||||
recognises "Apple" as unbranded also says it is produce - so it is the
|
||||
natural second opinion when keyword detection finds nothing.
|
||||
|
||||
But its answer is a HINT, and the pack size the merchant wrote is a FACT.
|
||||
"Tea Powder 250g" is the case that proves it: the lexicon calls tea a
|
||||
Beverage, the unit rulebook says beverages are measured in ml or litres
|
||||
only, and stage 4 then "corrects" 250g to 250ml - silently turning a
|
||||
quarter kilo of tea leaves into a quarter litre. Loose tea is a dry good
|
||||
sold by weight; the category is what is wrong there, not the size.
|
||||
|
||||
So a category that cannot accommodate the size already on the row is
|
||||
declined, and the row falls through to "General" exactly as it did before
|
||||
the lexicon was consulted at all.
|
||||
"""
|
||||
category = canonical_category(title)
|
||||
if not category:
|
||||
return None
|
||||
declared = [str(x).strip() for x in (row.get("size_variants") or []) if str(x).strip()]
|
||||
match = _SIZE_IN_TITLE.search(title or "")
|
||||
if match:
|
||||
declared.append(match.group(0).strip())
|
||||
for size in declared:
|
||||
ok, _msg = validate_unit_for_category(size, category)
|
||||
if not ok:
|
||||
return None
|
||||
return category
|
||||
|
||||
|
||||
def _sizes_for(row: Dict[str, Any]) -> List[str]:
|
||||
declared = [str(s).strip() for s in (row.get("size_variants") or []) if str(s).strip()]
|
||||
sizes = [s for s in declared if not _is_unitless_number(s)]
|
||||
@@ -386,6 +438,28 @@ def _sizes_for(row: Dict[str, Any]) -> List[str]:
|
||||
match = _SIZE_IN_TITLE.search(row.get("product_name") or "")
|
||||
if match:
|
||||
return [match.group(0).strip()]
|
||||
|
||||
# A COMMODITY WITH NO WEIGHT COLUMN GETS ONE UNSIZED ROW, NOT THREE MADE-UP
|
||||
# ONES. default_size_variants() invents a plausible set - 100g/250g/500g -
|
||||
# and for packaged goods that is a reasonable guess at what a brand sells.
|
||||
# For loose produce it is not: a shop sells apples by the kilo at whatever
|
||||
# the customer asks for, so "Apple 250g" is a product that does not exist.
|
||||
#
|
||||
# It is also the mechanism behind the catalogue drift the integrator
|
||||
# reported: the invented set is keyed on the resolved category, so the same
|
||||
# product ingested twice with the category resolved differently produces
|
||||
# two disjoint size sets, two sets of image_ids, and a re-scrape that looks
|
||||
# like the old rows were deleted. Keeping produce out of that from the
|
||||
# start is cheaper than repairing it later.
|
||||
# "Standard" and not "": an empty size fails validate_size ("size/pack is
|
||||
# missing") and stage 4 would drop the row, which is the rejection this
|
||||
# whole area exists to prevent. "Standard" is also what _to_storage_row
|
||||
# already substitutes for a blank size and what the seeded base list uses,
|
||||
# so an uploaded "Apple" deduplicates onto the seeded "Apple" instead of
|
||||
# creating a second row.
|
||||
if row.get("brand") == OWN_PRODUCTS_BRAND:
|
||||
return ["Standard"]
|
||||
|
||||
return list(price_estimator.default_size_variants(
|
||||
row.get("category") or "", row.get("product_name") or ""
|
||||
))
|
||||
@@ -423,18 +497,37 @@ def stage_4_explode_sizes(row: Dict[str, Any]) -> Tuple[List[Dict[str, Any]], Li
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 5 - Pricing bands
|
||||
# ---------------------------------------------------------------------------
|
||||
# How far either side of a sheet's own selling price the published band sits.
|
||||
# Stated once, because it is a judgement about how much a real shelf price
|
||||
# varies rather than a fact, and two call sites disagreeing about it would be
|
||||
# invisible. 155 -> Rs143-167.
|
||||
_SHEET_PRICE_BAND = 0.08
|
||||
|
||||
|
||||
def stage_5_pricing(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Publish a price band, from the sheet's own price where there is one.
|
||||
|
||||
FOR A COMMODITY THE ESTIMATOR IS NOT CONSULTED. It is trained on packaged
|
||||
FMCG and prices a 500g apple at Rs85-105, which is not wrong so much as
|
||||
meaningless - loose produce is priced by the shop, by the day. A row with no
|
||||
price keeps a null band rather than a confident fiction.
|
||||
"""
|
||||
size = row.get("size") or "Standard"
|
||||
if _blank(row.get("price_range")):
|
||||
if not _blank(row.get("final_selling_price")):
|
||||
price = float(row["final_selling_price"])
|
||||
lo, hi = int(round(price * 0.95)), int(round(price * 1.05))
|
||||
else:
|
||||
lo, hi = price_estimator.estimate_price_range_for_size(
|
||||
size, row.get("product_name") or "", row.get("brand") or "",
|
||||
row.get("category") or "",
|
||||
)
|
||||
row["price_range"] = f"₹{lo}-{hi}"
|
||||
if not _blank(row.get("price_range")):
|
||||
return row
|
||||
|
||||
if not _blank(row.get("final_selling_price")):
|
||||
price = float(row["final_selling_price"])
|
||||
lo = int(round(price * (1 - _SHEET_PRICE_BAND)))
|
||||
hi = int(round(price * (1 + _SHEET_PRICE_BAND)))
|
||||
elif row.get("brand") == OWN_PRODUCTS_BRAND:
|
||||
return row
|
||||
else:
|
||||
lo, hi = price_estimator.estimate_price_range_for_size(
|
||||
size, row.get("product_name") or "", row.get("brand") or "",
|
||||
row.get("category") or "",
|
||||
)
|
||||
row["price_range"] = f"₹{lo}-{hi}"
|
||||
return row
|
||||
|
||||
|
||||
@@ -498,6 +591,13 @@ def stage_7_sku(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
if _blank(row.get("sku_source")):
|
||||
row["sku_source"] = "sheet"
|
||||
return row
|
||||
# A commodity gets no minted SKU. An internal SKU is an identifier for a
|
||||
# specific packaged product from a specific brand; "OWN-APP-500" would name
|
||||
# a thing that does not exist, and a shop's loose apples are not the same
|
||||
# article as another shop's. The sheet's own column still wins above, so
|
||||
# this is "do not invent", not "discard".
|
||||
if row.get("brand") == OWN_PRODUCTS_BRAND:
|
||||
return row
|
||||
try:
|
||||
resolved = resolve_product_sku(
|
||||
row.get("brand") or "", row.get("product_name") or "", row.get("size") or ""
|
||||
@@ -518,6 +618,16 @@ async def stages_8_9_enrichment(rows: List[Dict[str, Any]], brand: str) -> List[
|
||||
Each disables itself via its settings flag, so this is a no-op when both
|
||||
are off.
|
||||
"""
|
||||
# Neither stage runs for the own-products bucket. A barcode identifies a
|
||||
# manufactured article and loose produce has none - the lookup would either
|
||||
# find nothing or, worse, attach some packaged product's real GTIN. HSN is
|
||||
# skipped for the same reason it is not invented anywhere else here: a tax
|
||||
# code the merchant did not supply is our guess presented as their record.
|
||||
# A sheet that DOES carry an hsn_code or barcode column keeps those values,
|
||||
# because both stages only fill blanks.
|
||||
if brand == OWN_PRODUCTS_BRAND:
|
||||
return rows
|
||||
|
||||
stages = []
|
||||
if ENABLE_BARCODE_LOOKUP:
|
||||
stages.append(BarcodeEnrichmentStage())
|
||||
@@ -567,8 +677,18 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
size = row.get("size") or "Standard"
|
||||
display = name if size.lower() in name.lower() else f"{name} {size}".strip()
|
||||
category = row.get("category") or "General"
|
||||
description = row.get("description") or (
|
||||
f"{display} from {row.get('brand')}."
|
||||
own = row.get("brand") == OWN_PRODUCTS_BRAND
|
||||
# "Apple 500g from Own Products." is a sentence nobody wrote and nobody
|
||||
# wants, and "Own Products" is a bucket rather than a maker, so the
|
||||
# template reads as a false provenance claim. A commodity keeps whatever
|
||||
# description the sheet gave, including none.
|
||||
description = row.get("description") or (None if own else f"{display} from {row.get('brand')}.")
|
||||
# The embedding text. Interpolating a null description and the bucket name
|
||||
# would embed the literal "Own Products ... None", so a commodity is
|
||||
# described to the vector index by what actually identifies it.
|
||||
search_query = (
|
||||
f"{display} {category}".strip() if own
|
||||
else f"{row.get('brand')} {display} {category} {description}"
|
||||
)
|
||||
return {
|
||||
"product_name": display,
|
||||
@@ -591,12 +711,13 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"barcode_type": row.get("barcode_type"),
|
||||
"highlights": list(row.get("highlights") or []),
|
||||
"nutrients": list(row.get("nutrients") or []),
|
||||
"search_query": f"{row.get('brand')} {display} {category} {description}",
|
||||
"search_query": search_query,
|
||||
}
|
||||
|
||||
|
||||
def _record_product(result: PipelineResult, brand: str, row: Dict[str, Any],
|
||||
disposition: str) -> None:
|
||||
disposition: str,
|
||||
source_rows: Optional[Dict[str, int]] = None) -> None:
|
||||
"""Note the identity of one resolved catalog row.
|
||||
|
||||
`image_id` is the useful field here and the one to join on: it is the key
|
||||
@@ -611,12 +732,23 @@ def _record_product(result: PipelineResult, brand: str, row: Dict[str, Any],
|
||||
if len(result.products) >= MAX_REPORTED_PRODUCTS:
|
||||
result.products_truncated = True
|
||||
return
|
||||
image_id = row.get("image_id")
|
||||
result.products.append({
|
||||
"image_id": row.get("image_id"),
|
||||
"image_id": image_id,
|
||||
"brand": brand,
|
||||
# The key the catalogue is actually addressed by, published so a caller
|
||||
# does not have to re-derive it from the display name. Getting that
|
||||
# wrong is silent: "24 Mantra" guessed as "24mantra" simply finds
|
||||
# nothing. This is the same function the storage layer uses.
|
||||
"brand_key": _sanitize_name(brand),
|
||||
"product_name": row.get("product_name"),
|
||||
"product_sku": row.get("product_sku"),
|
||||
"sku_source": row.get("sku_source"),
|
||||
# The 1-based spreadsheet row this came from, header counted as row 1 -
|
||||
# the number the sender sees on screen. One sheet row legitimately
|
||||
# becomes several products (pack-size explosion), so this is many-to-one
|
||||
# and is what lets a caller say "row 14 became these three".
|
||||
"source_row": (source_rows or {}).get(image_id),
|
||||
"disposition": disposition,
|
||||
})
|
||||
|
||||
@@ -640,7 +772,8 @@ def _merge_with_existing(new: Dict[str, Any], existing: Dict[str, Any]) -> Tuple
|
||||
return merged, changed
|
||||
|
||||
|
||||
def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResult) -> None:
|
||||
def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResult,
|
||||
source_rows: Optional[Dict[str, int]] = None) -> None:
|
||||
"""Embed and upsert, splitting inserts from backfills.
|
||||
|
||||
`cleanup=False` is load-bearing: cleanup=True deletes every row in the
|
||||
@@ -666,7 +799,7 @@ def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResul
|
||||
if prior is None:
|
||||
to_write.append(row)
|
||||
result.inserted += 1
|
||||
_record_product(result, brand, row, "inserted")
|
||||
_record_product(result, brand, row, "inserted", source_rows)
|
||||
continue
|
||||
merged, changed = _merge_with_existing(row, prior)
|
||||
if changed:
|
||||
@@ -675,10 +808,10 @@ def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResul
|
||||
# The MERGED row: a backfill keeps the stored product_sku rather
|
||||
# than the one this run minted, so reporting `row` would hand back
|
||||
# an identifier that is not the one in the catalog.
|
||||
_record_product(result, brand, merged, "backfilled")
|
||||
_record_product(result, brand, merged, "backfilled", source_rows)
|
||||
else:
|
||||
result.skipped_existing += 1
|
||||
_record_product(result, brand, prior, "unchanged")
|
||||
_record_product(result, brand, prior, "unchanged", source_rows)
|
||||
|
||||
if not to_write:
|
||||
return
|
||||
@@ -838,6 +971,10 @@ def run_pipeline(
|
||||
result.rejected += len(rejected)
|
||||
for bad in rejected:
|
||||
result.rejections.append({
|
||||
# The sheet row the sender sees on screen. Without it the only
|
||||
# way to find a refused product was to diff the manifest
|
||||
# against the file and guess.
|
||||
"row": bad.get("_row"),
|
||||
"product_name": bad.get("product_name") or bad.get("title"),
|
||||
"size": bad.get("size"),
|
||||
"reason": "; ".join(
|
||||
@@ -847,11 +984,27 @@ def run_pipeline(
|
||||
})
|
||||
progress(10, STAGE_NAMES[9], len(kept), len(rows))
|
||||
|
||||
storage_rows = [_to_storage_row(r) for r in kept]
|
||||
# image_id -> the sheet row that produced it, carried ALONGSIDE the
|
||||
# storage rows rather than inside them. `_to_storage_row` projects onto
|
||||
# the brand-table columns and its output goes straight to
|
||||
# upsert_brand_products, so smuggling a reporting-only key into that
|
||||
# dict would push an unknown column at the database.
|
||||
#
|
||||
# Built from `kept` before the dedupe below, and last-wins in the same
|
||||
# direction, so the row number always describes the product that was
|
||||
# actually written.
|
||||
source_rows: Dict[str, int] = {}
|
||||
storage_rows = []
|
||||
for enriched in kept:
|
||||
stored = _to_storage_row(enriched)
|
||||
storage_rows.append(stored)
|
||||
if enriched.get("_row") is not None:
|
||||
source_rows[stored["image_id"]] = enriched["_row"]
|
||||
|
||||
# A single sheet can name the same pack twice; last one wins, so the
|
||||
# batch never presents two rows with the same image_id to the upsert.
|
||||
deduped: Dict[str, Dict[str, Any]] = {r["image_id"]: r for r in storage_rows}
|
||||
stage_11_store(list(deduped.values()), brand, result)
|
||||
stage_11_store(list(deduped.values()), brand, result, source_rows)
|
||||
progress(11, STAGE_NAMES[10], len(deduped), len(deduped))
|
||||
|
||||
return result
|
||||
|
||||
@@ -252,6 +252,15 @@ BRAND_ALIASES = {
|
||||
"sunfeast bounce": "sunfeast",
|
||||
"sunfeast yippee": "sunfeast",
|
||||
"sunfeast cookies": "sunfeast",
|
||||
# SPELLING VARIANTS, not sub-brands.
|
||||
#
|
||||
# "Haldiram" and "Haldirams" were resolving to themselves, so nothing knew
|
||||
# they were one brand and uploads built brand_haldiram and brand_haldirams
|
||||
# side by side. The data was merged into the plural, which is the correct
|
||||
# name; THIS LINE is what stops the split reappearing on the next sheet
|
||||
# that spells it without the s. Removing it re-opens the bug.
|
||||
"haldiram": "haldirams",
|
||||
"haldiram's": "haldirams",
|
||||
}
|
||||
|
||||
DEFAULT_ALIASES = BRAND_ALIASES
|
||||
@@ -352,7 +361,14 @@ FSSAI_LICENSES: Dict[str, str] = {
|
||||
"lion dates": "10012042000244",
|
||||
"brooke bond": "10013022001897",
|
||||
"mother dairy": "10012011000015",
|
||||
# Keyed on BOTH spellings deliberately. get_fssai_license() resolves to the
|
||||
# canonical parent first, so once "haldiram" aliases to "haldirams" the
|
||||
# lookup arrives as "haldirams" - and with only the singular key here it
|
||||
# would return None and every future Haldiram row would ship with no
|
||||
# licence at all. A silent loss, since a blank licence is a legitimate
|
||||
# outcome elsewhere and nothing would flag it.
|
||||
"haldiram": "10012011000140",
|
||||
"haldirams": "10012011000140",
|
||||
"fortune": "10012021000071",
|
||||
"paper boat": "10012043000083",
|
||||
"bisk farm": "10012031000012",
|
||||
|
||||
@@ -88,6 +88,22 @@ CATEGORY_REGISTRY: List[Dict[str, object]] = [
|
||||
{"category": "Salt & Staples", "keywords": ["salt", "rock salt", "sea salt", "table salt", "iodised salt", "iodized salt"], "generic_term": "salt"},
|
||||
{"category": "Atta & Staples", "keywords": ["atta", "wheat flour", "flour", "rice", "dal", "pulses", "staples", "suji", "maida"], "generic_term": "staple product"},
|
||||
{"category": "Dairy", "keywords": ["milk", "dairy", "cheese", "paneer", "panner", "paner", "paneerr", "curd", "yogurt", "butter", "ghee", "dahi"], "generic_term": "dairy product"},
|
||||
# ---- Loose, unbranded fresh goods -----------------------------------
|
||||
# Added for the produce a grocer sells by weight or by the piece. These
|
||||
# rows carry no brand, so they land in the Own Products table via
|
||||
# generic_products.is_unbranded(); the categories exist so that HSN/GST
|
||||
# resolution and the pack-size unit rules have something to key off,
|
||||
# rather than falling through to "General".
|
||||
#
|
||||
# Keyword lists stay SHORT here on purpose. Detection for these rows comes
|
||||
# from the commodity lexicon, which is far more specific; a broad keyword
|
||||
# such as "fresh" or "leaf" would pull branded products in through
|
||||
# keyword matching, which is the failure this whole area exists to avoid.
|
||||
{"category": "Fruits & Vegetables", "keywords": ["fruits", "vegetables", "vegetable", "fresh produce", "loose produce"], "generic_term": "fresh produce"},
|
||||
{"category": "Fresh Herbs & Greens", "keywords": ["herbs", "greens", "curry leaves", "coriander leaves", "mint leaves", "spinach", "keerai"], "generic_term": "fresh greens"},
|
||||
{"category": "Flowers", "keywords": ["flowers", "flower", "garland", "jasmine flower", "loose flowers"], "generic_term": "flowers"},
|
||||
{"category": "Fish & Seafood", "keywords": ["fish", "seafood", "prawns", "prawn", "shrimp"], "generic_term": "fresh fish"},
|
||||
{"category": "Eggs", "keywords": ["eggs", "egg", "country egg"], "generic_term": "eggs"},
|
||||
{"category": "Oral Care", "keywords": ["toothpaste", "toothbrush", "mouthwash", "paste"], "generic_term": "oral care product"},
|
||||
{"category": "Hair Care", "keywords": ["shampoo", "shampooo", "conditioner", "hair oil"], "generic_term": "hair care product"},
|
||||
{"category": "Bath Soap", "keywords": ["bath soap", "soap bar", "soap", "soaps"], "generic_term": "soap"},
|
||||
|
||||
@@ -70,6 +70,17 @@ HSN_GST_TABLE: Dict[str, Tuple[str, int, bool]] = {
|
||||
# Chapter 17: cane/beet sugar and jaggery, 5% for ordinary retail sugar.
|
||||
"Sugar & Jaggery": ("1701", 5, False),
|
||||
"Cooking Oils": ("1517", 5, False),
|
||||
# ---- Loose fresh goods, sold by weight or by the piece ---------------
|
||||
# Chapters 3, 4, 6, 7 and 8. Fresh, unprocessed produce is NIL-rated under
|
||||
# GST - it is not a reduced rate, it is exempt - so 0 here is the real
|
||||
# figure and not a placeholder. The moment any of these is branded and
|
||||
# packaged in a unit container the rate changes, but that product would
|
||||
# resolve to its brand's category rather than these.
|
||||
"Fruits & Vegetables": ("0709", 0, False),
|
||||
"Fresh Herbs & Greens": ("0709", 0, False),
|
||||
"Flowers": ("0603", 0, False),
|
||||
"Fish & Seafood": ("0302", 0, False),
|
||||
"Eggs": ("0407", 0, False),
|
||||
"Pickles & Chutneys": ("2001", 12, True),
|
||||
"Dry Fruits & Nuts": ("0801", 12, True),
|
||||
"Food - Spreads": ("2007", 12, True),
|
||||
|
||||
@@ -74,6 +74,16 @@ _SALT = "Salt & Staples"
|
||||
_OILS = "Cooking Oils"
|
||||
_DAIRY = "Dairy"
|
||||
_BEVERAGE = "Beverages"
|
||||
# Fresh, loose goods. These are sold by weight or by the piece and carry no
|
||||
# brand at all, which is precisely why they were the worst offenders: before
|
||||
# these existed, "Apple" resolved to a brand called Apple and "Red Rose" was
|
||||
# whole-word matched into brand_brooke_bond, carrying Brooke Bond's FSSAI
|
||||
# licence onto a flower.
|
||||
_PRODUCE = "Fruits & Vegetables"
|
||||
_GREENS = "Fresh Herbs & Greens"
|
||||
_FLOWERS = "Flowers"
|
||||
_SEAFOOD = "Fish & Seafood"
|
||||
_EGGS = "Eggs"
|
||||
|
||||
COMMODITY_TERMS: Dict[str, str] = {}
|
||||
|
||||
@@ -127,6 +137,79 @@ _add(_DAIRY,
|
||||
_add(_BEVERAGE, "tea", "coffee", "chai")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Fresh produce
|
||||
# ---------------------------------------------------------------------------
|
||||
# WHY THIS BLOCK IS SAFE TO ADD.
|
||||
#
|
||||
# The all-tokens-must-be-commodities rule means every word added here makes the
|
||||
# test MORE permissive, so the risk is real brands collapsing into this bucket.
|
||||
# That was measured, not assumed, before these went in: all 1,414 products in
|
||||
# the live catalogue were reclassified with this list applied, and exactly one
|
||||
# changed - `24 Mantra Organic Moong Dal 500g`, which was ALREADY misfiled by
|
||||
# the `_strip_sizes` bug fixed below and has nothing to do with produce.
|
||||
#
|
||||
# The other check was against BRAND_ALIASES: of 113 candidate terms only "amla"
|
||||
# appears in an alias ("dabur amla"), and that is harmless because "dabur" is
|
||||
# not a commodity, so "Dabur Amla" keeps every token it needs to stay branded.
|
||||
# Bare "Amla" is fruit and belongs here.
|
||||
#
|
||||
# RE-RUN BOTH CHECKS BEFORE ADDING A WORD TO THIS BLOCK. See
|
||||
# tests/test_generic_products_produce.py, which encodes them.
|
||||
|
||||
# Fruit. Indian sheets mix English, Tamil and Hindi names freely.
|
||||
_add(_PRODUCE,
|
||||
"apple", "orange", "banana", "grape", "grapes", "mango", "pineapple",
|
||||
"papaya", "guava", "pomegranate", "watermelon", "muskmelon", "melon",
|
||||
"lemon", "lime", "mosambi", "sathukudi", "sapota", "chikoo", "jackfruit",
|
||||
"fig", "anjeer", "pear", "peach", "plum", "apricot", "cherry", "cherries",
|
||||
"strawberry", "blueberry", "kiwi", "litchi", "lychee", "amla", "gooseberry",
|
||||
"avocado", "dragonfruit", "rambutan", "mangosteen", "starfruit", "jamun",
|
||||
"ber", "plantain", "vazhaikkai")
|
||||
|
||||
# Vegetables. "gourd" covers the whole family once _PHRASES has collapsed the
|
||||
# two-word forms (bitter gourd, bottle gourd, snake gourd, ridge gourd).
|
||||
_add(_PRODUCE,
|
||||
"tomato", "potato", "onion", "carrot", "beetroot", "beet", "radish",
|
||||
"turnip", "cabbage", "cauliflower", "broccoli", "brinjal", "eggplant",
|
||||
"aubergine", "okra", "bhindi", "cucumber", "pumpkin", "gourd", "drumstick",
|
||||
"beans", "bean", "capsicum", "garlic", "ginger", "yam", "colocasia",
|
||||
"tapioca", "arbi", "zucchini", "chayote", "sweetcorn", "babycorn",
|
||||
"mushroom", "leek", "celery", "lettuce", "shallot", "springonion")
|
||||
|
||||
# Greens and fresh herbs, sold in bunches and never branded. NOTE that
|
||||
# "coriander" and "methi" are deliberately NOT here: they are already spices in
|
||||
# the block above, `_add` is last-wins, and re-adding them would silently move
|
||||
# dhania powder out of Spices & Masalas. The leaf forms are collapsed to
|
||||
# distinct tokens by _PHRASES instead.
|
||||
_add(_GREENS,
|
||||
"spinach", "palak", "amaranth", "keerai", "greens", "cilantro", "mint",
|
||||
"pudina", "curryleaves", "basil", "thulasi", "tulsi", "parsley", "dill",
|
||||
"sorrel", "moringa", "methileaves", "bunch")
|
||||
|
||||
# Flowers. Sold loose or by the metre of garland; the reason "Red Rose" used to
|
||||
# land in a tea catalogue.
|
||||
_add(_FLOWERS,
|
||||
"flower", "flowers", "rose", "jasmine", "malli", "lotus", "marigold",
|
||||
"samanthi", "chrysanthemum", "kanakambaram", "arali", "garland", "poo")
|
||||
|
||||
# Fish and seafood, sold fresh by weight.
|
||||
_add(_SEAFOOD,
|
||||
"fish", "prawn", "prawns", "shrimp", "crab", "squid", "tuna", "mackerel",
|
||||
"sardine", "pomfret", "seer", "vanjaram", "anchovy", "nethili", "sole",
|
||||
"tilapia", "salmon", "shellfish", "clam", "mussel")
|
||||
|
||||
# Eggs.
|
||||
_add(_EGGS, "egg", "eggs", "muttai", "quail")
|
||||
|
||||
# Stragglers found by running the produce base list through is_unbranded and
|
||||
# fixing every row it refused. Kept in one block so the next person adding to
|
||||
# the seed list knows where the tail ends up.
|
||||
_add(_PRODUCE, "dates", "custard", "dragonfruit", "ivy", "raw")
|
||||
_add(_GREENS, "agathi", "ponnanganni", "keerai")
|
||||
_add(_FLOWERS, "tuberose", "lily")
|
||||
|
||||
|
||||
# Words that describe a product without naming a brand. Stripped before the
|
||||
# all-tokens-are-commodities test, so "Organic Toor Dal Whole 1kg" still reads
|
||||
# as unbranded.
|
||||
@@ -150,6 +233,16 @@ QUALIFIERS: Set[str] = {
|
||||
"bottle", "refill", "combo", "assorted", "mixed", "mix",
|
||||
# connectives
|
||||
"and", "with", "of", "the", "in", "for",
|
||||
# Form words for fresh goods. "Leaves" is the important one: without it
|
||||
# "Mint Leaves" keeps an unknown token and reads as a brand.
|
||||
"leaves", "leaf", "bunch", "sweet", "broad", "cluster", "full", "toned",
|
||||
"seedless", "ripe", "tender", "baby", "country", "hybrid", "nati",
|
||||
# Varietal names. A variety qualifies a commodity, it does not brand it:
|
||||
# an Alphonso mango is a mango. None of these appears in BRAND_ALIASES -
|
||||
# the produce test asserts that, so a future addition cannot smuggle a
|
||||
# real brand in through this list.
|
||||
"robusta", "yelakki", "nendran", "alphonso", "banganapalli", "totapuri",
|
||||
"malgova", "sindoora", "shimla", "ooty", "kashmiri",
|
||||
}
|
||||
|
||||
# Multi-word commodities collapsed to a single token before tokenising, so the
|
||||
@@ -175,16 +268,70 @@ _PHRASES = {
|
||||
"brown sugar": "sugar",
|
||||
"palm jaggery": "jaggery",
|
||||
"cane sugar": "sugar",
|
||||
# Fresh produce. The two-word gourds collapse onto "gourd" so the whole
|
||||
# family is one lexicon entry. The leaf forms get their OWN tokens rather
|
||||
# than reusing "coriander" / "methi": those are spices, _add is last-wins,
|
||||
# and re-adding them under a greens category would silently move dhania
|
||||
# powder out of Spices & Masalas.
|
||||
"bitter gourd": "gourd",
|
||||
"bottle gourd": "gourd",
|
||||
"snake gourd": "gourd",
|
||||
"ridge gourd": "gourd",
|
||||
"ash gourd": "gourd",
|
||||
"bitter guard": "gourd", # misspellings seen in real merchant data
|
||||
"bottle ground": "gourd",
|
||||
"lady finger": "okra",
|
||||
"ladies finger": "okra",
|
||||
"spring onion": "springonion",
|
||||
"spring onions": "springonion",
|
||||
"sweet potato": "potato",
|
||||
"curry leaves": "curryleaves",
|
||||
"curry leaf": "curryleaves",
|
||||
"coriander leaves": "cilantro",
|
||||
"methi leaves": "methileaves",
|
||||
"fenugreek leaves": "methileaves",
|
||||
"french beans": "beans",
|
||||
"cluster beans": "beans",
|
||||
"green peas": "peas",
|
||||
"baby corn": "babycorn",
|
||||
"sweet corn": "sweetcorn",
|
||||
"tender coconut": "coconut",
|
||||
"dragon fruit": "dragonfruit",
|
||||
"custard apple": "apple",
|
||||
"sweet lime": "mosambi",
|
||||
"ivy gourd": "gourd",
|
||||
"broad beans": "beans",
|
||||
"cluster bean": "beans",
|
||||
"quail egg": "egg",
|
||||
"spring garlic": "garlic",
|
||||
}
|
||||
|
||||
_WORD_RE = re.compile(r"[a-z]+")
|
||||
|
||||
|
||||
# Units a pack size is actually written in. The strip below is bounded to these
|
||||
# rather than to "any letters", because [a-z]* after a number ate the NEXT WORD:
|
||||
# "24 Mantra Organic Moong Dal" became "organic moong dal", the brand was
|
||||
# destroyed, and the row was then filed as an unbranded commodity. Every brand
|
||||
# whose name begins with a number hit this. Keep the list tight - a unit added
|
||||
# here is a word that can be deleted from a product name.
|
||||
_UNITS = (
|
||||
"kg|kgs|g|gm|gms|gram|grams|mg|ml|l|ltr|ltrs|litre|litres|liter|liters"
|
||||
"|pc|pcs|piece|pieces|pack|packs|pkt|n|no|nos|x|cm|mm|inch|dozen"
|
||||
)
|
||||
|
||||
_SIZE_RE = re.compile(
|
||||
# "1kg", "500 g", "1.5 L" - a number followed by a REAL unit, optionally
|
||||
# spaced - or a bare number, which is a quantity and never a brand.
|
||||
r"\b\d+(?:[.,]\d+)?\s*(?:" + _UNITS + r")\b"
|
||||
r"|\b\d+(?:[.,]\d+)?\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def _strip_sizes(text: str) -> str:
|
||||
"""Remove pack sizes and bare numbers - they never name a brand."""
|
||||
# "1kg", "500 g", "1.5 L", and any leftover bare number.
|
||||
text = re.sub(r"\b\d+(?:[.,]\d+)?\s*[a-z]*\b", " ", text)
|
||||
return text
|
||||
return _SIZE_RE.sub(" ", text)
|
||||
|
||||
|
||||
def canonical_category(name: str) -> Optional[str]:
|
||||
|
||||
@@ -54,6 +54,7 @@ from app.infrastructure.settings import (
|
||||
from app.services.title_validator import find_category_conflicts
|
||||
from app.services import price_estimator
|
||||
from app.services import category_units as cu
|
||||
from app.services.generic_products import OWN_PRODUCTS_BRAND
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -277,6 +278,26 @@ def validate_product(
|
||||
sku_source = str(product.get("sku_source") or "").strip()
|
||||
images = product.get("image_urls") or []
|
||||
|
||||
# A COMMODITY IS NOT A DEFECTIVE BRANDED PRODUCT.
|
||||
#
|
||||
# The penalties below treat a missing price_range or SKU as evidence that a
|
||||
# row was fabricated, which is right for a scraped brand catalogue: a real
|
||||
# Amul product has a shelf price and an article number, so their absence
|
||||
# means something went wrong. Loose produce has neither, by nature. A shop
|
||||
# prices apples by the day and does not issue article numbers for them.
|
||||
#
|
||||
# Left unqualified, the arithmetic rejected every produce row outright:
|
||||
# 0.55 baseline - 0.30 (no price_range) - 0.10 (no SKU) = 0.15, against a
|
||||
# reject threshold of 0.35. That is the exact opposite of the requirement
|
||||
# these rows exist to satisfy, so the two absences stop counting as faults.
|
||||
#
|
||||
# EVERYTHING ELSE STILL APPLIES. Title sanity, placeholder detection,
|
||||
# category resolution, the title/category contradiction check, size
|
||||
# validity and unit compatibility, and the image-presence check all run
|
||||
# unchanged - a blank or junk product name is still caught, and this is not
|
||||
# a way in for rows that would otherwise fail.
|
||||
commodity = brand == OWN_PRODUCTS_BRAND
|
||||
|
||||
report = ValidationReport(product_name=title or "(untitled)", grounded=grounded)
|
||||
score = 0.55 # neutral baseline - moves up/down based on evidence below
|
||||
|
||||
@@ -345,16 +366,22 @@ def validate_product(
|
||||
score -= 0.15
|
||||
|
||||
# 5. Price range ----------------------------------------------------------
|
||||
ok, msg = validate_price_range(price_range, size, title, brand, category)
|
||||
if not ok:
|
||||
report.issues.append(ValidationIssue(field="price_range", severity="error", message=msg, penalty=0.30))
|
||||
score -= 0.30
|
||||
# Skipped for a commodity only when there is none. A band that IS present is
|
||||
# still checked for being well formed, so a malformed one cannot hide here.
|
||||
if price_range or not commodity:
|
||||
ok, msg = validate_price_range(price_range, size, title, brand, category)
|
||||
if not ok:
|
||||
report.issues.append(ValidationIssue(field="price_range", severity="error", message=msg, penalty=0.30))
|
||||
score -= 0.30
|
||||
|
||||
# 6. SKU --------------------------------------------------------------------
|
||||
ok, msg = validate_sku(sku, sku_source)
|
||||
if not ok:
|
||||
report.issues.append(ValidationIssue(field="product_sku", severity="warning", message=msg, penalty=0.10))
|
||||
score -= 0.10
|
||||
# Same rule: a commodity is not expected to carry one, but a SKU the sheet
|
||||
# did supply must still look like a SKU.
|
||||
if sku or not commodity:
|
||||
ok, msg = validate_sku(sku, sku_source)
|
||||
if not ok:
|
||||
report.issues.append(ValidationIssue(field="product_sku", severity="warning", message=msg, penalty=0.10))
|
||||
score -= 0.10
|
||||
|
||||
# 7. Image presence -----------------------------------------------------
|
||||
# Only meaningful if image search actually ran. When the operator disables
|
||||
|
||||
137
data/haldiram_merge_backup_20260901_134907.json
Normal file
137
data/haldiram_merge_backup_20260901_134907.json
Normal file
@@ -0,0 +1,137 @@
|
||||
{
|
||||
"taken_at": "20260901_134907",
|
||||
"db_host": "31.97.228.132",
|
||||
"db_name": "pgvector",
|
||||
"brand_haldiram": [
|
||||
{
|
||||
"id": 1,
|
||||
"product_name": "Haldiram Aloo Bhujia 200g 45",
|
||||
"title": "Haldiram Aloo 200g",
|
||||
"description": "Haldiram Aloo Bhujia 200g 45 from Haldiram.",
|
||||
"category": "3",
|
||||
"image_id": "haldiram_haldiram_aloo_bhujia_200g_45",
|
||||
"image_url": "https://bazaar-foods.co.uk/cdn/shop/products/haldirams-aloo-bhujia-200g.jpg?v=1649176712",
|
||||
"image_urls": [
|
||||
"https://bazaar-foods.co.uk/cdn/shop/products/haldirams-aloo-bhujia-200g.jpg?v=1649176712",
|
||||
"https://www.thai-food-online.co.uk/cdn/shop/products/Haldirams-Aloo-Bhujia-200g-Front.png?v=1653396797",
|
||||
"https://www.rashanpani.co.uk/wp-content/uploads/2020/05/HALDIRAMALOOBHUJIA200G.jpg",
|
||||
"https://www.gobuzzaar.com/uploads/2025/06/800x800/haldirams-aloo-bhujia-200g.jpg",
|
||||
"https://groceteria.eu/wp-content/uploads/2025/02/Haldirams-200g-Aloo-Bhujia.webp",
|
||||
"https://www.citybazaar.dk/wp-content/uploads/2024/02/haldiram-aloo-bhujia.jpg",
|
||||
"https://thekiranahaus.com/cdn/shop/files/Haldiram-s-200g-Aloo-Bhujia--wuerzige-Kartoffelsticks--16251_1_5b0f5681-4eff-4792-8c83-774d323946bd.webp?v=1765318184&width=1445",
|
||||
"https://nisargafresh.nl/wp-content/uploads/2022/09/Aloo-Bhujia-200g-Haldiram-300x300.png",
|
||||
"https://m.media-amazon.com/images/I/713pmpvY6VL._AC_SL1000_.jpg",
|
||||
"https://cdn.shopify.com/s/files/1/0434/5475/9072/products/IS-132_clipped_rev_1_copy_1024x.jpg?v=1628238808"
|
||||
],
|
||||
"price_range": "₹52-58",
|
||||
"size_variants": [
|
||||
"45"
|
||||
],
|
||||
"providers": [],
|
||||
"fssai_license": "10012011000140",
|
||||
"product_sku": "HALD-BHUJ-200",
|
||||
"sku_source": "",
|
||||
"hsn_code": null,
|
||||
"final_selling_price": "58",
|
||||
"selling_price": "58",
|
||||
"barcode": null,
|
||||
"barcode_type": null,
|
||||
"highlights": [],
|
||||
"nutrients": [],
|
||||
"search_query": "Haldiram Haldiram Aloo Bhujia 200g 45 3 Haldiram Aloo Bhujia 200g 45 from Haldiram.",
|
||||
"embedding": "[-0.061470453,0.085923776,-0.050007723,-0.010388176,-0.105947,0.029684091,0.01018019,0.0326033,-0.017272232,-0.02223002,0.05238059,-0.13941023,0.056763377,-0.05565607,-0.0013915179,0.04291669,0.06696487,0.026800232,-0.04647914,-0.0416988,0.08339774,0.019954558,-0.009182672,-0.0025798376,-0.007229289,0.13358761,-0.033226427,0.055968896,0.037845407,-0.061806567,-0.0043716184,0.09443795,0.082649015,-0.07063394,0.02757287,-0.033491127,-0.08968844,-0.031277124,0.08357984,0.0482263,-3.4857999e-06,-0.0025698135,0.028261518,0.005129888,0.009810577,-0.0063438024,-0.08344995,0.10941526,0.059196167,-0.02535888,-0.016701974,0.02346376,-0.07723267,0.105881654,0.049159676,-0.103167415,0.008602405,0.028959755,-0.001431078,-0.034416527,-0.0676624,0.035875387,-0.02955854,-0.027848115,0.020283105,-0.027637335,-0.0808495,-0.060705267,-0.066662356,0.021860223,-0.005598716,-0.034734484,-0.034236483,-0.029325726,-0.018017743,-0.04256999,0.04801943,-0.08334282,-0.0764811,-0.04878591,-0.06523457,-0.06585633,0.12267185,-0.017186869,-0.018697215,-0.03347549,-0.014150048,0.12791106,-0.057740245,-0.108282715,-0.030115323,0.0066051325,-0.11242725,0.015669039,-0.092518106,0.06819647,-0.122968495,-0.05385696,-0.019065559,0.018783037,0.075349405,0.036990706,0.0058152927,-0.06598786,-0.07396621,0.020352127,0.027102727,0.13160394,0.037662864,-0.017217921,-0.066403694,0.02904227,-0.059101313,-0.0034248584,-0.00864348,-0.03633173,0.008295371,0.0021769737,-0.0057140547,0.019750657,0.007474019,-0.023771532,0.049534626,-0.029095244,-0.12592928,0.07232369,0.03378169,1.296634e-32,-0.050658148,-0.09196166,0.058936916,-0.032744236,0.018527597,-0.033474702,-0.04349935,0.017932698,-0.036908038,-0.028993832,-0.06068338,-0.037089925,0.028008202,0.023156805,-0.006812169,-0.10516624,-0.03014195,0.035480473,0.021186655,-0.01400949,-0.0077031157,0.049328536,-0.0054385876,0.044295203,0.03583454,0.06183502,0.08454213,-0.049978737,0.057355814,0.05139823,0.01397718,-0.027529268,-0.096242115,-0.12014253,-0.115006946,0.053672213,-0.0054810354,-0.081016235,-0.094351254,-0.05612986,0.044648796,-0.030642055,0.010459408,0.012553524,0.03095918,0.07615386,0.051000275,0.014667437,0.034062423,0.052176744,-0.031464193,-0.0123085305,-0.011492719,0.017208502,-0.030510992,-0.021482727,0.016208854,-0.002907258,0.00020148358,0.08722652,0.016213732,0.050492495,-0.013732973,0.021592563,-0.03134377,-0.023611384,-0.023446757,-0.07413654,-0.013767491,0.0033524255,-0.0075572,-0.06441154,0.09468595,0.061729066,-0.02178496,0.0010737815,0.024657287,0.030952245,0.02398675,0.0020069524,-0.045607194,0.059503548,0.03923495,-0.07371786,0.016165433,0.041368306,-0.045157075,-0.057643034,-0.0076029045,0.04802942,0.017770018,-0.003472389,0.046380565,0.025366355,-0.027365148,-1.17752656e-32,-0.028758632,0.07745132,-0.019914337,0.0186684,0.08845942,-0.0064726146,0.07080666,0.07155823,0.035890803,-0.015938815,-0.029325774,-0.02274504,0.115917765,-0.054575015,-0.057255,0.04734996,0.10425196,0.051324375,-0.040612098,-0.03783868,0.01610947,0.04325998,-0.0006286663,0.008529754,-0.04983215,0.112548664,0.051360626,0.0069579147,0.009742793,0.03564187,-0.012604597,-0.04643494,-0.05486593,0.032028537,-0.1032377,-0.010374628,0.0385542,0.035651542,-0.04998217,0.019538503,-0.029337378,0.0888773,0.019374222,0.0963841,0.0061803036,0.009116665,0.04528866,-0.050373435,0.016377108,-0.020969104,-0.0054394677,-0.019459903,-0.046234936,0.016407901,-0.026153658,0.014765041,-0.03968482,0.0021429295,-0.06024107,-0.008839775,-0.032153983,0.07577759,0.027919343,0.071250975,0.0623783,0.04449583,0.048923377,-0.052728355,-0.02196624,-0.023756275,-0.005193731,0.002633509,-0.07457847,-0.0049360064,-0.0067040683,0.048857424,-0.005255573,0.07660087,0.0026997577,-0.0016187229,0.031624213,0.00095052144,-0.0067311437,0.062699765,-0.080630675,-0.08166218,0.025550803,-0.03572945,-0.00601697,0.019851236,-0.044734333,0.068068944,0.054362986,0.038462523,0.046319842,-3.5680632e-08,-0.02125714,0.039479893,0.024287026,0.04016811,0.03498418,0.037565768,-0.025369272,0.006615806,0.01806187,0.0011729915,-0.017235735,-0.038094517,-0.039481185,0.04735565,-0.0062645734,-0.0132780345,0.07052318,-0.03832829,0.067541465,-0.055909988,0.011112473,-0.0035194275,0.13212267,0.013075932,0.011319353,0.069707625,-0.017551374,-0.02812701,0.08074076,0.041591056,0.020192059,0.04382235,-0.07242851,-0.060478747,0.030826215,-0.023221618,-0.0855683,0.039421275,0.010495891,0.011221291,0.014818017,-0.120077655,0.029326573,0.022608621,0.025708701,0.0058598258,-0.0735763,-0.021944927,-0.059000954,-0.09134795,-0.0029611024,-0.101805426,0.044474937,-0.0054603335,-0.04431912,0.042876992,-0.0657774,-0.05918801,0.022463147,0.017258102,0.037892565,-0.010327075,-0.06765019,0.044359624]",
|
||||
"created_at": "2026-08-29T10:41:38.166944",
|
||||
"updated_at": "2026-08-29T10:41:38.166944"
|
||||
}
|
||||
],
|
||||
"brand_haldirams": [
|
||||
{
|
||||
"id": 1,
|
||||
"product_name": "Haldiram's Aloo Bhujia 200g",
|
||||
"title": "Haldiram's Aloo Bhujia 200g",
|
||||
"description": "Crunchy and spicy potato-gram flour namkeen",
|
||||
"category": "Snacks",
|
||||
"image_id": "haldirams_haldiram_s_aloo_bhujia_200g",
|
||||
"image_url": "https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg",
|
||||
"image_urls": [
|
||||
"https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg"
|
||||
],
|
||||
"price_range": "₹45-50",
|
||||
"size_variants": [
|
||||
"200g"
|
||||
],
|
||||
"providers": [
|
||||
"Amazon",
|
||||
"Flipkart",
|
||||
"BigBasket",
|
||||
"Jiomart",
|
||||
"Blinkit",
|
||||
"Zepto"
|
||||
],
|
||||
"fssai_license": "10012042000244",
|
||||
"product_sku": "HALD-HALDIR-001",
|
||||
"sku_source": "User Upload",
|
||||
"hsn_code": "2106",
|
||||
"final_selling_price": "48",
|
||||
"selling_price": "48",
|
||||
"barcode": "8900000000000.0",
|
||||
"barcode_type": "GTIN-13",
|
||||
"highlights": [
|
||||
"100% Quality Assurance",
|
||||
"Authentic Brand Product"
|
||||
],
|
||||
"nutrients": [
|
||||
"Energy - High",
|
||||
"Protein - Good Source"
|
||||
],
|
||||
"search_query": "Haldiram's Haldiram's Aloo Bhujia 200g Snacks Crunchy and spicy potato-gram flour namkeen ₹45-50",
|
||||
"embedding": "[-0.073780574,0.014016445,-0.03283347,0.06779144,-0.07432287,0.013323582,0.057849504,-0.003455605,-0.02796043,-0.0106389765,0.054610476,-0.13733506,0.0011138533,-0.09461825,0.043265574,-0.0033051497,0.16251235,-0.021456327,-0.056616433,-0.043503813,0.015809868,-0.0027687962,0.055956308,-0.0020621475,-0.0002262986,0.09798358,0.052142963,-0.0065899845,-0.008076999,-0.04662511,0.075209744,0.12199015,0.046643093,-0.07255586,0.023547035,-0.032034293,0.02218585,-0.07860314,0.064440474,0.017679987,-0.017938236,0.037968524,0.051713504,-0.025717558,0.0026579762,-0.04317045,-0.04329217,0.10957813,0.04799027,-0.0094134435,-0.04176496,0.021880541,-0.02489868,0.053983495,0.10355603,-0.07522341,-0.11387932,0.027443936,-0.03329309,0.008200507,-0.025784984,0.031980727,0.009286136,-0.029458733,-0.010811541,-0.07233483,-0.059245557,-0.038007714,0.0024375066,0.004017135,-0.043354295,-0.0379747,0.04938866,-0.009079977,-0.031919453,0.0062903636,0.038413957,-0.06369559,-0.050876208,-0.08253325,-0.03543455,-0.013137344,0.11619364,0.021482123,-0.0041941865,-0.055790465,0.0027696376,0.059069995,-0.04857008,-0.021166429,0.038674932,-0.024962643,0.014518789,-0.04124445,-0.080964535,-0.0240311,-0.029613206,-0.08670562,-0.051875632,0.05894461,0.028114261,0.053660985,0.036084134,-0.09016684,-0.05446474,-0.019464912,0.07717981,-0.011869444,0.051274676,0.02189268,-0.06052555,0.049884107,-0.09407489,-0.037107907,-0.045468986,-0.07448858,0.042282812,0.0032387888,-0.04506364,0.025343671,-0.050813645,-0.02691568,0.043881804,-0.008280819,-0.10768411,0.021082247,-0.012938328,7.068499e-33,-0.03685851,-0.039287563,-0.00058218406,-0.03546077,0.039447866,-0.099279776,-0.008846175,-0.015827043,-0.004231832,-0.01240423,-0.0173825,0.034137066,-0.031023402,0.062806845,0.07871152,-0.08896062,-0.08321511,-0.03931041,0.045285314,-0.011240046,-0.07006219,0.0048550433,0.006457747,0.034788996,-0.01760184,0.027861014,0.065395735,-0.028560039,0.00248985,0.018866641,-0.011300588,0.004064385,-0.040849943,-0.11293159,-0.13535002,0.008115082,-0.056292925,-0.07768911,-0.0564176,-0.009952972,0.051836103,-0.028403815,-0.051030494,0.038016986,-0.025066279,0.075658865,0.08475662,0.082214326,0.10590234,0.02896686,-0.025536839,-0.030114923,-0.010633608,-0.0184912,-0.05289483,-0.02754835,0.04824575,-0.040299762,0.015501986,0.040012572,-0.036788717,-0.05353712,-0.048302356,-0.075477526,-0.060516905,0.008490575,-0.052708536,-0.04905377,-0.033386704,-0.008519541,-0.005193108,-0.01380934,0.12728089,0.0032967299,-0.039227586,0.008550332,0.060942084,0.031031711,-0.003867492,-0.006875357,0.11645594,0.035951044,0.078381166,-0.0020736938,-0.057762377,0.063063264,-0.08124302,-0.035704855,0.023189044,0.0152145,-0.039887406,-0.0055398573,0.06708767,-0.0057648136,-0.07640236,-6.7070056e-33,-0.03370923,0.014161425,-0.06586386,0.084878206,0.046127915,0.023844836,-0.02461334,0.017690562,0.00921369,-0.04313147,-0.030632516,-0.012535122,0.065278426,-0.013989146,-0.05026042,0.11251113,0.05700934,0.08111092,0.007674605,-0.059061382,-0.024890551,0.10317635,0.0149834575,-0.012122599,-0.049457733,0.09958608,0.06396433,0.033100173,-0.026297713,0.032261726,0.08667095,-0.055016927,-0.018163476,0.012644439,0.016452178,-0.0284256,0.0038589025,-0.051818937,-0.122482866,0.05270389,-0.030226449,0.083730906,0.01275159,0.05033225,0.037246265,-0.02044742,-0.033988904,-0.05296506,-0.037575576,-0.030798644,0.063701525,-0.00068997405,-0.023358572,-0.018957613,-0.037956595,0.07388675,-0.058028262,-0.002049744,0.027629329,-0.100966014,-0.050763145,0.07685107,0.025519853,0.05307841,0.05312479,0.023053985,0.029822761,-0.087285936,-0.017102415,0.00042710872,0.009520072,-0.019577894,0.030032566,0.0161792,0.01962172,0.061840374,0.03533059,-0.014051352,0.024355397,0.042371184,-0.008329303,0.021590685,-0.026294695,0.0076788454,-0.060583495,0.07875852,-0.030082814,0.02406768,-0.00927241,0.0993275,0.014351461,0.062653966,0.014678145,0.09492412,0.115223385,-3.1167954e-08,0.077919655,-0.04777391,-0.051864136,0.09117969,0.05721661,-0.040711176,-0.020052843,-0.021208424,0.03569595,0.008754088,0.008037574,0.03135558,-0.053007998,0.053251993,-0.06319788,-0.010215157,0.028881038,0.04924266,0.029694416,-0.003294222,0.006432629,0.03875308,0.12772492,-0.074748695,0.0014321478,0.043181334,0.011794961,0.021509338,0.08130951,0.100629106,0.00054375804,0.03646748,-0.011478553,-0.07050192,0.00770562,-0.0062742736,-0.029889021,0.023644177,-0.03701927,0.028838946,-0.026554547,-0.12876473,0.032152396,0.0502661,-0.057909496,-0.015366481,-0.046538085,0.047690097,0.0034577225,-0.024463559,-0.02055527,-0.002813917,-0.0027458656,-0.014018512,-0.05970947,0.03264415,-0.03840796,-0.02475397,0.00080424803,-0.02509683,0.013823303,-0.0044572027,-0.09458777,0.05007379]",
|
||||
"created_at": "2026-08-17T15:37:15.414450",
|
||||
"updated_at": "2026-08-17T15:37:15.414450"
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"product_name": "Haldiram's Moong Dal 200g",
|
||||
"title": "Haldiram's Moong Dal 200g",
|
||||
"description": "Roasted and salted moong dal namkeen",
|
||||
"category": "Snacks",
|
||||
"image_id": "haldirams_haldiram_s_moong_dal_200g",
|
||||
"image_url": "https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg",
|
||||
"image_urls": [
|
||||
"https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg"
|
||||
],
|
||||
"price_range": "₹45-50",
|
||||
"size_variants": [
|
||||
"200g"
|
||||
],
|
||||
"providers": [
|
||||
"Amazon",
|
||||
"Flipkart",
|
||||
"BigBasket",
|
||||
"Jiomart",
|
||||
"Blinkit",
|
||||
"Zepto"
|
||||
],
|
||||
"fssai_license": "10012042000244",
|
||||
"product_sku": "HALD-HALDIR-001",
|
||||
"sku_source": "User Upload",
|
||||
"hsn_code": "2106",
|
||||
"final_selling_price": "48",
|
||||
"selling_price": "48",
|
||||
"barcode": "8900000000000.0",
|
||||
"barcode_type": "GTIN-13",
|
||||
"highlights": [
|
||||
"100% Quality Assurance",
|
||||
"Authentic Brand Product"
|
||||
],
|
||||
"nutrients": [
|
||||
"Energy - High",
|
||||
"Protein - Good Source"
|
||||
],
|
||||
"search_query": "Haldiram's Haldiram's Moong Dal 200g Snacks Roasted and salted moong dal namkeen ₹45-50",
|
||||
"embedding": "[-0.094575,0.06647317,0.018450627,0.09411248,-0.11488417,0.0094742095,0.10323204,-0.008308582,-0.018837677,-0.06068788,0.03635123,-0.098427385,-0.011098394,-0.08241818,-0.010680773,-0.08192029,0.096883185,0.0061954088,-0.058705177,-0.07856562,0.0020511616,-0.035402957,0.026109403,0.011536509,0.02497377,0.102277346,0.044439584,0.004150674,-0.045279127,-0.06597435,0.046088085,0.13454252,0.019104136,-0.014138215,-0.018770406,0.0016899393,0.037570734,-0.101482116,0.033911843,-0.024027782,-0.007123305,0.0043425863,0.03389706,-0.01171643,-0.0010247155,-0.026938196,-0.07531733,0.047449447,0.048601817,0.028115522,-0.05155453,0.032397907,-0.030602388,0.018949037,0.084327325,-0.037579127,-0.09632338,0.03052131,-0.023420155,-0.018819075,-0.058096565,0.04801071,-0.019584091,0.00840557,-0.014928658,-0.017625613,-0.05028415,-0.007010133,0.0035170382,-0.017748946,-0.0073538446,-0.024471723,-0.0017027381,-0.011953686,-0.041689005,-0.020850446,0.06313427,-0.067269444,-0.02779477,-0.079179004,-0.040808536,-0.009283012,0.062841676,0.01776416,-0.017664485,-0.038390763,0.0077870428,0.086355746,-0.0071483236,-0.036428194,0.041351836,0.0010660643,-0.043088205,-0.047883313,-0.06397132,-0.014597856,-0.038109757,-0.11806625,-0.03859069,0.06741478,0.09975801,0.091468096,0.022662511,-0.099400885,0.01808026,-0.07444516,0.05801442,0.020548489,0.005871151,0.031390473,-0.08609821,0.0685131,-0.048832435,-0.0009881549,-0.053029515,-0.03519214,-0.02444342,-0.002871041,-0.00079397764,0.008004174,-0.044126116,-0.032528643,0.018218463,-0.002613153,-0.13646005,-0.021690182,-0.010388374,5.1170974e-33,-0.037991114,-0.057875294,0.045896087,-0.029799264,0.03445652,-0.05492394,0.021679293,0.009985053,-0.04079306,-0.029813424,-0.063487954,0.015532548,-0.016924227,0.04721671,0.090319164,-0.06784286,-0.0605672,-0.014278195,0.05632942,-0.0095763765,-0.118342355,0.06277701,0.05509074,-0.024498325,0.0067808125,0.04671988,0.029352784,-0.020681981,0.021213375,0.054673143,0.009987924,-0.051140882,-0.001783095,-0.068780065,-0.120357536,0.084589906,-0.05895755,-0.09710861,-0.040292494,-0.0063073607,0.106736794,0.014908648,-0.0042924844,0.015698465,-0.055473812,0.08482299,0.054842465,0.10652771,0.020141818,0.059941553,-0.038201205,0.0036418291,-0.030693544,-0.018320644,-0.03457125,0.03745144,0.023243146,0.010407767,0.009686064,0.016844263,0.011497691,-0.004155571,-0.045655962,-0.08559727,-0.042167526,0.032703206,-0.11219652,-0.062466234,-0.050595324,-0.00826258,-0.0026475135,-0.04215606,0.10538657,0.04092127,0.017664399,0.041831847,0.04067998,0.016462628,-0.038579937,0.020880423,0.079678945,0.061415456,0.06346388,-0.0011568575,-0.013694989,0.0623573,-0.0461171,-0.028040573,0.0042170635,0.04478849,-0.04387031,-0.0057746936,0.030305725,-0.023072846,-0.025389213,-5.8312274e-33,-0.029279016,-0.028358413,-0.030364979,0.07388548,0.03441022,-0.013458687,0.0008307939,0.020595325,-0.0071057896,0.0048316387,-0.031197425,-0.005186759,0.09885664,-0.027137637,-0.066331014,0.07685916,0.06953698,0.066769645,-0.0031445064,-0.027278978,-0.038401123,0.13919948,0.03546529,0.08209664,-0.065055154,0.08680524,0.055690266,0.030268267,-0.03396688,-0.015301493,0.095488615,-0.0673101,-0.025902545,-0.034853,-0.017013576,-0.04080425,0.054977503,-0.07056256,-0.08512563,0.05852524,-0.025794232,0.023131687,0.0027184545,0.032675695,0.0768045,-0.06114638,7.459798e-05,-0.08546178,-0.024186937,0.0039223493,0.028259924,-0.03900481,-0.008595077,0.01217486,-0.0051869825,0.029366544,-0.046122774,0.019728474,0.03202166,-0.12575525,-0.03417364,0.0764442,0.050919365,0.05531786,0.036432143,-0.017693507,0.018790327,-0.13322152,-0.032080054,-0.047055557,0.022200573,-0.04082469,0.004283893,-0.023007419,0.016834922,0.07362733,-0.009972964,0.052465674,0.067914724,0.0224166,-0.02975218,0.013401601,-0.03840001,0.05775581,-0.07493675,0.0201357,0.0088814525,0.031209635,-0.039559595,0.033402905,0.0014653425,0.007815599,-0.01766598,0.0919281,0.116744734,-2.5746983e-08,0.07138476,-0.057019565,-0.040646307,0.09265247,0.01658529,-0.000989908,0.0042248284,0.0018876027,0.024829473,0.018251775,0.031974934,0.023970272,-0.06986896,-0.0051021436,-0.09206029,-0.0005921235,0.016616385,0.028401058,0.013909974,-0.018079655,0.06514544,0.046897646,0.104065046,-0.013080918,0.00300962,0.026130179,0.060950138,0.015441117,0.055578504,0.04622945,-0.008451027,0.03546806,-0.017733773,-0.093754,0.04323589,-0.01374034,-0.002944189,0.04646882,0.041333135,0.06198348,-0.044462144,-0.13388309,0.009989284,0.040997624,-0.053453274,0.05901287,-0.0037693249,0.06316173,-0.036486328,-0.052355066,0.010952625,0.0035248504,-0.004493483,-0.07502327,-0.05658603,0.06537711,-0.011206859,-0.014927725,-0.03594883,0.007979808,0.039871667,-0.064994566,-0.1093809,0.0676764]",
|
||||
"created_at": "2026-08-17T15:37:15.870055",
|
||||
"updated_at": "2026-08-17T15:37:15.870055"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -1,93 +1,93 @@
|
||||
{
|
||||
"brand": "haldirams",
|
||||
"search_query": "Haldirams products catalog",
|
||||
"generation_timestamp": "C:\\Brand_Catalog_LLM\\RAG_Model_Nutrition_Intelligence\\RAG_Model_Full_Implement\\backend\\app\\services\\brand_sync.py",
|
||||
"total_products": 2,
|
||||
"total_images": 2,
|
||||
"products": [
|
||||
{
|
||||
"brand": "Haldirams",
|
||||
"brand_name": "Haldirams",
|
||||
"product_name": "Haldiram's Moong Dal 200g",
|
||||
"title": "Haldiram's Moong Dal 200g",
|
||||
"description": "Roasted and salted moong dal namkeen",
|
||||
"category": "Snacks",
|
||||
"image_id": "haldirams_haldiram_s_moong_dal_200g",
|
||||
"image_url": "https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg",
|
||||
"image_urls": [
|
||||
"https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg"
|
||||
],
|
||||
"price_range": "₹45-50",
|
||||
"size_variants": [
|
||||
"200g"
|
||||
],
|
||||
"providers": [
|
||||
"Amazon",
|
||||
"Flipkart",
|
||||
"BigBasket",
|
||||
"Jiomart",
|
||||
"Blinkit",
|
||||
"Zepto"
|
||||
],
|
||||
"fssai_license": "10012042000244",
|
||||
"product_sku": "HALD-HALDIR-001",
|
||||
"sku_source": "User Upload",
|
||||
"hsn_code": "2106",
|
||||
"final_selling_price": 48.0,
|
||||
"selling_price": 48.0,
|
||||
"barcode": "8900000000000.0",
|
||||
"barcode_type": "GTIN-13",
|
||||
"highlights": [
|
||||
"100% Quality Assurance",
|
||||
"Authentic Brand Product"
|
||||
],
|
||||
"nutrients": [
|
||||
"Energy - High",
|
||||
"Protein - Good Source"
|
||||
],
|
||||
"search_query": "Haldiram's Haldiram's Moong Dal 200g Snacks Roasted and salted moong dal namkeen ₹45-50"
|
||||
},
|
||||
{
|
||||
"brand": "Haldirams",
|
||||
"brand_name": "Haldirams",
|
||||
"product_name": "Haldiram's Aloo Bhujia 200g",
|
||||
"title": "Haldiram's Aloo Bhujia 200g",
|
||||
"description": "Crunchy and spicy potato-gram flour namkeen",
|
||||
"category": "Snacks",
|
||||
"image_id": "haldirams_haldiram_s_aloo_bhujia_200g",
|
||||
"image_url": "https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg",
|
||||
"image_urls": [
|
||||
"https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg"
|
||||
],
|
||||
"price_range": "₹45-50",
|
||||
"size_variants": [
|
||||
"200g"
|
||||
],
|
||||
"providers": [
|
||||
"Amazon",
|
||||
"Flipkart",
|
||||
"BigBasket",
|
||||
"Jiomart",
|
||||
"Blinkit",
|
||||
"Zepto"
|
||||
],
|
||||
"fssai_license": "10012042000244",
|
||||
"product_sku": "HALD-HALDIR-001",
|
||||
"sku_source": "User Upload",
|
||||
"hsn_code": "2106",
|
||||
"final_selling_price": 48.0,
|
||||
"selling_price": 48.0,
|
||||
"barcode": "8900000000000.0",
|
||||
"barcode_type": "GTIN-13",
|
||||
"highlights": [
|
||||
"100% Quality Assurance",
|
||||
"Authentic Brand Product"
|
||||
],
|
||||
"nutrients": [
|
||||
"Energy - High",
|
||||
"Protein - Good Source"
|
||||
],
|
||||
"search_query": "Haldiram's Haldiram's Aloo Bhujia 200g Snacks Crunchy and spicy potato-gram flour namkeen ₹45-50"
|
||||
}
|
||||
]
|
||||
"brand": "haldirams",
|
||||
"search_query": "Haldirams products catalog",
|
||||
"generation_timestamp": "C:\\Brand_Catalog_LLM\\RAG_Model_Nutrition_Intelligence\\RAG_Model_Full_Implement\\backend\\app\\services\\brand_sync.py",
|
||||
"total_products": 2,
|
||||
"total_images": 2,
|
||||
"products": [
|
||||
{
|
||||
"brand": "Haldirams",
|
||||
"brand_name": "Haldirams",
|
||||
"product_name": "Haldiram's Moong Dal 200g",
|
||||
"title": "Haldiram's Moong Dal 200g",
|
||||
"description": "Roasted and salted moong dal namkeen",
|
||||
"category": "Snacks",
|
||||
"image_id": "haldirams_haldiram_s_moong_dal_200g",
|
||||
"image_url": "https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg",
|
||||
"image_urls": [
|
||||
"https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg"
|
||||
],
|
||||
"price_range": "₹45-50",
|
||||
"size_variants": [
|
||||
"200g"
|
||||
],
|
||||
"providers": [
|
||||
"Amazon",
|
||||
"Flipkart",
|
||||
"BigBasket",
|
||||
"Jiomart",
|
||||
"Blinkit",
|
||||
"Zepto"
|
||||
],
|
||||
"fssai_license": "10012011000140",
|
||||
"product_sku": "HALD-HALDIR-001",
|
||||
"sku_source": "User Upload",
|
||||
"hsn_code": "2106",
|
||||
"final_selling_price": 48.0,
|
||||
"selling_price": 48.0,
|
||||
"barcode": "8900000000000.0",
|
||||
"barcode_type": "GTIN-13",
|
||||
"highlights": [
|
||||
"100% Quality Assurance",
|
||||
"Authentic Brand Product"
|
||||
],
|
||||
"nutrients": [
|
||||
"Energy - High",
|
||||
"Protein - Good Source"
|
||||
],
|
||||
"search_query": "Haldiram's Haldiram's Moong Dal 200g Snacks Roasted and salted moong dal namkeen ₹45-50"
|
||||
},
|
||||
{
|
||||
"brand": "Haldirams",
|
||||
"brand_name": "Haldirams",
|
||||
"product_name": "Haldiram's Aloo Bhujia 200g",
|
||||
"title": "Haldiram's Aloo Bhujia 200g",
|
||||
"description": "Crunchy and spicy potato-gram flour namkeen",
|
||||
"category": "Snacks",
|
||||
"image_id": "haldirams_haldiram_s_aloo_bhujia_200g",
|
||||
"image_url": "https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg",
|
||||
"image_urls": [
|
||||
"https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/haldirams/haldirams_haldiram_s_aloo_bhujia_200g/image_000.jpg"
|
||||
],
|
||||
"price_range": "₹52-58",
|
||||
"size_variants": [
|
||||
"200g"
|
||||
],
|
||||
"providers": [
|
||||
"Amazon",
|
||||
"Flipkart",
|
||||
"BigBasket",
|
||||
"Jiomart",
|
||||
"Blinkit",
|
||||
"Zepto"
|
||||
],
|
||||
"fssai_license": "10012011000140",
|
||||
"product_sku": "HALD-HALDIR-001",
|
||||
"sku_source": "User Upload",
|
||||
"hsn_code": "2106",
|
||||
"final_selling_price": 58.0,
|
||||
"selling_price": 58.0,
|
||||
"barcode": "8900000000000.0",
|
||||
"barcode_type": "GTIN-13",
|
||||
"highlights": [
|
||||
"100% Quality Assurance",
|
||||
"Authentic Brand Product"
|
||||
],
|
||||
"nutrients": [
|
||||
"Energy - High",
|
||||
"Protein - Good Source"
|
||||
],
|
||||
"search_query": "Haldiram's Haldiram's Aloo Bhujia 200g Snacks Crunchy and spicy potato-gram flour namkeen ₹45-50"
|
||||
}
|
||||
]
|
||||
}
|
||||
66473
data/seed_catalogs/brand_catalog_own_products.json
Normal file
66473
data/seed_catalogs/brand_catalog_own_products.json
Normal file
File diff suppressed because it is too large
Load Diff
323
docs/DRIFT_REPORT_RESPONSE.md
Normal file
323
docs/DRIFT_REPORT_RESPONSE.md
Normal file
@@ -0,0 +1,323 @@
|
||||
# Response to the Catalogue Drift Report
|
||||
|
||||
Reply to the seven findings dated 31 August 2026 (drop
|
||||
`8e1448e176d843d08183d387ad724f95` → run `0ed4c77b0ca14e03b1aaf6b5b77d1994`).
|
||||
|
||||
Every figure below was measured against the same live deployment, not read off
|
||||
source. Where we disagree with a finding, the evidence is included so you can
|
||||
check it rather than take our word for it.
|
||||
|
||||
**Summary:** four items are fixed and ship in the next backend deploy. One
|
||||
(#02) was already in the API and we had failed to document it — that is our
|
||||
fault and the docs are now corrected. #01 is diagnosed, and the cause is not
|
||||
what either of us assumed. #06 is confirmed but carries a trap that means we
|
||||
should agree an approach before touching it.
|
||||
|
||||
| # | Finding | Status |
|
||||
| --- | --- | --- |
|
||||
| 01 | Pack sizes replaced between scrapes | **Diagnosed** — three causes, not one. Fix needs your input |
|
||||
| 02 | `rejected` is a bare count | **Already shipped, now documented** — plus `row` added |
|
||||
| 03 | Manifest brands are not catalogue keys | **Fixed** — `brand_key` published |
|
||||
| 04 | Run files carry no `from_drop` | **Fixed** |
|
||||
| 05 | Manifest carries no `source_row` | **Fixed** |
|
||||
| 06 | Duplicate brands and products | **Confirmed.** Read the trap below before we act |
|
||||
| 07 | No loose-produce coverage | **Fixed** — 159-row base list, and the upload path now handles produce |
|
||||
|
||||
---
|
||||
|
||||
## 01 — Pack sizes and names are replaced between scrapes
|
||||
|
||||
You asked us to confirm whether a pack size that once existed is meant to
|
||||
survive a re-scrape. **It is not, today** — but that is only the last of three
|
||||
causes, and fixing it alone would not have saved your links.
|
||||
|
||||
### Cause 1: pack sizes are invented when a scrape does not supply them
|
||||
|
||||
`_sizes_for()` falls back to `default_size_variants(category, name)` when a
|
||||
product declares no size. That fallback is **keyed on the resolved category**,
|
||||
and the resolved category is not stable between runs. Measured today:
|
||||
|
||||
```
|
||||
category "Snacks" -> ['55g', '150g', '200g']
|
||||
category "" (unresolved) -> ['100g', '250g', '500g']
|
||||
category "Breakfast Cereal" -> ['250g', '500g', '1kg']
|
||||
```
|
||||
|
||||
Now compare your table. Cheetos id 25 is **55g** — the *Snacks* set. Ids 26 and
|
||||
27 are **100g** and **250g** — the *unresolved* set. The same product was
|
||||
ingested once with its category resolved and once without, and produced two
|
||||
disjoint sets of pack sizes. Your Cheerios ids (100g, 250g, 500g) sit across the
|
||||
Breakfast Cereal set and the unresolved set the same way.
|
||||
|
||||
So the pack sizes were never scraped facts that changed. Some of them were
|
||||
generated, and the generator's input moved.
|
||||
|
||||
### Cause 2: the product name gains or loses a brand prefix
|
||||
|
||||
Your own #06 has the evidence: `Hot Heads 30g` became `Nestle Hot Heads`, and
|
||||
`PepsiCo Kurkure Masala Munch 90g` coexists with `Kurkure Masala Munch 90g`.
|
||||
`image_id` is derived from the name, so a prefix appearing or disappearing moves
|
||||
the id even when the product is identical.
|
||||
|
||||
### Cause 3: the write then deletes whatever is not in the new set
|
||||
|
||||
The brand-scrape path calls `upsert_brand_products(..., cleanup=True)`, which
|
||||
deletes every row in the table whose `image_id` is absent from the batch being
|
||||
written. That is what turns causes 1 and 2 from "duplicate rows" into "the row
|
||||
you stored is gone".
|
||||
|
||||
**Any one of these is survivable. Together they guarantee broken links on every
|
||||
re-scrape**, which matches your finding that zero of eleven could be repaired.
|
||||
|
||||
### Not the upload path
|
||||
|
||||
Worth stating plainly, because it affects how much you need to worry: the
|
||||
**upload** path — everything reached through `POST /api/uploads/catalog` — uses
|
||||
`cleanup=False` and has always done so. A sheet you send can never delete a row
|
||||
it does not mention. The deletions came from brand scraping only.
|
||||
|
||||
### What we need from you
|
||||
|
||||
The real fix is to stop causes 1 and 2 (do not invent sizes for a product
|
||||
already in the catalogue; settle the naming convention), and to soft-retire
|
||||
rather than delete for cause 3. That third part changes how the production
|
||||
catalogue is written and we would rather agree it with you than spring it:
|
||||
|
||||
- Would a `retired_at` timestamp plus exclusion from the default read work for
|
||||
you, instead of the row being deleted? That preserves the `image_id` so your
|
||||
stored link resolves to something, and lets us give you the `superseded_by`
|
||||
and per-run changelog you asked for.
|
||||
- If so, do you want retired rows visible through an explicit query, or gone
|
||||
from the API entirely?
|
||||
|
||||
### Your two direct questions
|
||||
|
||||
**Is `image_id` stable across re-scrapes for a product whose name and pack size
|
||||
have not changed?** Yes. It is a pure deterministic function of brand, product
|
||||
name and pack size, with no clock, counter or run id in it. Verified:
|
||||
|
||||
```
|
||||
build_image_id('pepsico', 'Cheetos Chips', '100g') -> pepsico_cheetos_chips_100g
|
||||
```
|
||||
|
||||
Storing it rather than our row id is the right call and we have documented the
|
||||
guarantee so it does not quietly change. **The caveat is #06:** the guarantee is
|
||||
only as good as the stability of the name, and inconsistent brand prefixing
|
||||
breaks exactly that.
|
||||
|
||||
**Do you want to know when a product is dropped or renamed?** Yes, and we agree
|
||||
it should exist. It falls out of the retirement model above rather than being a
|
||||
separate feature, which is why we would like to settle that first.
|
||||
|
||||
---
|
||||
|
||||
## 02 — `rejected` is a count with no reason
|
||||
|
||||
**This is our documentation failure, not a missing feature.** `rejections[]` has
|
||||
been in every response — single-batch read and list endpoint both, since
|
||||
`to_out(slim=True)` strips only `products` — carrying `product_name`, `size` and
|
||||
`reason` per refused row. It was absent from `INGESTION_API.md`, which documents
|
||||
`"rejected": 0` and never mentions the array, so there was no way for you to
|
||||
know it was there. Sorry — that is a straightforwardly bad docs bug.
|
||||
|
||||
The genuine gap was the row number, which is now added:
|
||||
|
||||
```jsonc
|
||||
"rejections": [
|
||||
{ "row": 7, "product_name": "Kurkure Menthol", "size": "10g",
|
||||
"reason": "title is too short to be a real product name; image_urls: no images were found for this product" }
|
||||
]
|
||||
```
|
||||
|
||||
`row` is the 1-based sheet row with the header counted as row 1 — the same
|
||||
convention as the `422` responses, so it matches what the operator sees on
|
||||
screen. `null` only when the row cannot be located. Capped at 50 per file.
|
||||
|
||||
---
|
||||
|
||||
## 03 — Manifest brands are not catalogue keys
|
||||
|
||||
Fixed. Every entry in `products[]` now carries `brand_key` beside `brand`:
|
||||
|
||||
```jsonc
|
||||
{ "brand": "24 Mantra", "brand_key": "24_mantra", ... }
|
||||
```
|
||||
|
||||
This is generated by the same function the storage layer uses to name the table,
|
||||
so it cannot drift from the key the catalogue is actually addressed by. Your
|
||||
normalisation is correct as far as we can tell, but it is a guess, and the
|
||||
failure mode is silent — a wrong key finds nothing rather than erroring.
|
||||
|
||||
Thank you for degrading rather than failing the batch on an unreadable brand;
|
||||
that is the right behaviour and we should have done it on our side too.
|
||||
|
||||
---
|
||||
|
||||
## 04 — Run files carry no `from_drop`
|
||||
|
||||
Fixed, and we agree with your assessment that this was the one item that could
|
||||
corrupt a merchant's inventory rather than merely inconvenience you.
|
||||
|
||||
Each file in a run now carries `from_drop`, the id of the drop it was released
|
||||
from — the exact inverse of `released_to`:
|
||||
|
||||
```jsonc
|
||||
"files": [
|
||||
{ "index": 0, "filename": "products.csv", "from_drop": "8e1448e1...", ... },
|
||||
{ "index": 1, "filename": "products.csv", "from_drop": "a91c02f4...", ... }
|
||||
]
|
||||
```
|
||||
|
||||
`null` for a file that went straight into a run without sitting in an inbox —
|
||||
which, under `UPLOAD_AUTORUN=true`, is every file you send, because the id you
|
||||
are handed is already the run.
|
||||
|
||||
There is a test in our suite that stages two drops from different senders both
|
||||
named `products.csv` and asserts they are distinguishable, so the collision you
|
||||
described is now a permanent regression guard rather than a hope.
|
||||
|
||||
---
|
||||
|
||||
## 05 — Manifest cannot be traced back to the spreadsheet row
|
||||
|
||||
Fixed. Every entry in `products[]` carries `source_row`, the 1-based sheet row
|
||||
with the header as row 1.
|
||||
|
||||
It is deliberately **many-to-one**: a pack-size cell reading `100g, 200g, 500g`
|
||||
becomes three products that all report the same `source_row`, which is what lets
|
||||
you say "row 14 of your sheet became these three". Rows that produced nothing
|
||||
are those absent from every entry — "rows 6 and 11 produced nothing" is now a
|
||||
set difference rather than a name-matching heuristic.
|
||||
|
||||
---
|
||||
|
||||
## 06 — Duplicate brands, duplicate products, stray names
|
||||
|
||||
Confirmed against the live database today: **55 brand tables, 1,414 products**
|
||||
(you counted 1,614; the difference is a day of drift plus, we think, your count
|
||||
including rejected rows — worth reconciling if it matters).
|
||||
|
||||
```
|
||||
haldiram 1 britannia 6
|
||||
haldirams 2 parle 3 against hindustan_unilever 443
|
||||
patanjali 3
|
||||
```
|
||||
|
||||
So: the split brand is real, and Britannia/Parle/Patanjali do look like scrapes
|
||||
that stopped part-way rather than genuinely small brands. We will re-run those
|
||||
three.
|
||||
|
||||
### The trap, which is why we have not just fixed this
|
||||
|
||||
**De-duplicating the PepsiCo pairs means renaming a product, and `image_id` is
|
||||
derived from the name.** Renaming `PepsiCo Kurkure Masala Munch 90g` to
|
||||
`Kurkure Masala Munch 90g` does not merge the two rows — it mints a third id and
|
||||
breaks any link pointing at either of the first two. You have just migrated onto
|
||||
storing `image_id`. A well-meant cleanup on our side would re-break exactly what
|
||||
you have finished repairing.
|
||||
|
||||
The same applies to stripping the stray `150` from `Lays Classic Salted 52g 150`.
|
||||
|
||||
So before we touch it we would like to agree:
|
||||
|
||||
1. **Which convention wins** — brand prefix in the product name, or not? We have
|
||||
no strong preference; we care only that it is one of them. Our lean is
|
||||
*without* the prefix, since the brand is already a column.
|
||||
2. **How the merge is communicated.** If we can give you the old-id →
|
||||
new-id mapping for every row we touch, in advance, does that let you
|
||||
re-point rather than clear? That is straightforward for us to produce.
|
||||
3. **Timing**, so it lands in one pass rather than trickling.
|
||||
|
||||
`haldiram` → `haldirams` is a three-row merge and much lower risk; we can do
|
||||
that one immediately if you would rather not wait for the rest.
|
||||
|
||||
---
|
||||
|
||||
## 07 — Loose produce has no coverage
|
||||
|
||||
Fixed, and this turned out to be the most valuable finding in your report,
|
||||
because it was not only a coverage gap.
|
||||
|
||||
### What was actually happening
|
||||
|
||||
Produce rows were not rejected. They were **misfiled**, which is worse. The
|
||||
brand fallback takes the first word of the name and then whole-word matches it
|
||||
against our alias map:
|
||||
|
||||
```
|
||||
Apple -> brand "Apple" -> junk table brand_apple
|
||||
Tomato -> brand "Tomato" -> junk table brand_tomato
|
||||
Bitter Gourd -> brand "Bitter" -> junk table brand_bitter
|
||||
Curry Leaves -> brand "Curry" -> junk table brand_curry
|
||||
Red Rose -> brand "Red" -> brand_brooke_bond <--
|
||||
```
|
||||
|
||||
That last one is not a typo. A rose was being written into the Brooke Bond tea
|
||||
catalogue, and our enrichment then stamps that brand's real FSSAI licence number
|
||||
onto the row. Your 139 hand-typed products were the visible symptom; this was
|
||||
underneath it.
|
||||
|
||||
### What now happens
|
||||
|
||||
Loose goods are recognised as commodities and filed under a single house brand,
|
||||
`Own Products` (table `brand_own_products`), before brand inference can touch
|
||||
them. Fruit, vegetables, greens, herbs, flowers, fish, eggs and loose dairy are
|
||||
covered, alongside the pulses, grains, spices, oils and sugar that already were.
|
||||
|
||||
Five categories were added — Fruits & Vegetables, Fresh Herbs & Greens, Flowers,
|
||||
Fish & Seafood, Eggs — with HSN codes and a 0% GST rate, since unprocessed
|
||||
produce is nil-rated rather than reduced-rate.
|
||||
|
||||
Merchant misspellings from your own data are handled: `Bitter guard`,
|
||||
`Bottle ground`, `Ladies Finger` all resolve.
|
||||
|
||||
### The base list
|
||||
|
||||
**159 rows**, in the shape you asked for: name and category only, no brand, no
|
||||
pack size, no price. Fruit (42), vegetables (53), greens and herbs (17), flowers
|
||||
(14), fish and seafood (15), loose dairy (13), eggs (5). It includes the specific
|
||||
items your audit listed — Jasmine, Lotus, Red Rose, Thulasi, Drumstick, Curry
|
||||
Leaves, the four banana varieties, Tuna, Mackerel.
|
||||
|
||||
Each row carries a `search_query` embedding, so these are reachable through
|
||||
semantic search and not just exact match. The list is hand-authored rather than
|
||||
scraped, so it is not subject to any of #01.
|
||||
|
||||
**Images are not included yet.** You asked for name and image; we have shipped
|
||||
the names. Sourcing 159 licensable produce photographs is a separate piece of
|
||||
work and we did not want to hold the list for it — tell us if the list is not
|
||||
useful to you without them and we will prioritise accordingly.
|
||||
|
||||
### One limitation worth knowing
|
||||
|
||||
The classifier is deliberately conservative: a single word it does not recognise
|
||||
means "this is a brand". So place-qualified produce — `Salem Mango`,
|
||||
`Mysore Banana`, `Jammu Apple`, all real strings from your Ragul Stores data —
|
||||
still reads as branded, because `Mysore` is also a real brand (Mysore Sandal).
|
||||
|
||||
The workaround is already in the pipeline: **if the sheet has a Brand column and
|
||||
leaves the cell empty, we believe it** and file the row under Own Products
|
||||
regardless of the name. If your merchants' sheets carry an empty brand column,
|
||||
those rows will land correctly. If they carry no brand column at all, the
|
||||
name-based test is what applies.
|
||||
|
||||
We would rather be conservative here. Collapsing a real regional brand into the
|
||||
unbranded bucket is much harder to undo than a mango sitting in the wrong table.
|
||||
|
||||
---
|
||||
|
||||
## What we verified before sending this
|
||||
|
||||
- The produce lexicon was run against **all 1,414 products in the live
|
||||
catalogue** and against all **231 brand aliases**: zero reclassifications in
|
||||
either. That check is now a test, so it runs on every change.
|
||||
- It also surfaced a pre-existing bug we would not otherwise have found: our
|
||||
pack-size stripper was eating the word after a number, so
|
||||
`24 Mantra Organic Moong Dal 500g` lost its brand entirely and was being filed
|
||||
as an unbranded commodity. **Every brand whose name starts with a digit hit
|
||||
this.** "24 Mantra" appears on your finding-03 list, which is how we noticed.
|
||||
Fixed.
|
||||
- All 159 seeded rows were checked to classify identically to how an uploaded
|
||||
copy of the same name would, so a grocer typing "Tomato" lands on the seeded
|
||||
row instead of creating a second one.
|
||||
- Full suite: **1,054 tests passing.**
|
||||
@@ -176,7 +176,7 @@ The bad file is kept as a failed member rather than dropped, so a sender who sub
|
||||
"brands": [],
|
||||
"files": [
|
||||
{ "index": 0, "filename": "catalog.csv", "status": "queued",
|
||||
"total_stages": 11, "rows_total": 1, "size_bytes": 57 },
|
||||
"from_drop": null, "total_stages": 11, "rows_total": 1, "size_bytes": 57 },
|
||||
{ "index": 1, "filename": "notes.txt", "status": "failed",
|
||||
"detail": "The file has no data rows." }
|
||||
],
|
||||
@@ -184,6 +184,22 @@ The bad file is kept as a failed member rather than dropped, so a sender who sub
|
||||
}
|
||||
```
|
||||
|
||||
#### `from_drop` — which file in this run is yours
|
||||
|
||||
**Match on this, never on `filename`.** An admin can assemble one run from several
|
||||
drops, so a run's `files` may contain sheets you did not send — and two senders can
|
||||
both upload `products.csv`. Matching on the name is a coincidence; matching on
|
||||
`from_drop` is exact.
|
||||
|
||||
| Value | Meaning |
|
||||
| --- | --- |
|
||||
| the drop id you were given | this file is the one you sent in that drop |
|
||||
| `null` | the file went straight into a run and never sat in an inbox — under `UPLOAD_AUTORUN=true` that is every file, and the run id you hold is already the only id involved |
|
||||
|
||||
It is the exact inverse of `released_to`, which points from your drop to the run
|
||||
that took it. Both are needed: `released_to` answers *where did my drop go*,
|
||||
`from_drop` answers *whose file is this*.
|
||||
|
||||
`use_llm` and `fetch_images` are reported, never accepted. They decide how much
|
||||
outbound work a run commits the host to, and this endpoint's caller is anonymous, so
|
||||
they come from settings — sending them in the request has no effect.
|
||||
@@ -333,19 +349,112 @@ catalogue:
|
||||
"rows_total": 2, "inserted": 1, "backfilled": 0, "skipped_existing": 1,
|
||||
"rejected": 0, "brands": ["amul"],
|
||||
"products": [
|
||||
{ "image_id": "amul_amul_butter_100g", "brand": "amul",
|
||||
"product_name": "Amul Butter 100g",
|
||||
{ "image_id": "amul_amul_butter_100g", "brand": "amul", "brand_key": "amul",
|
||||
"product_name": "Amul Butter 100g", "source_row": 2,
|
||||
"product_sku": "ACME-BUT-100", "sku_source": "sheet",
|
||||
"disposition": "inserted" },
|
||||
{ "image_id": "amul_amul_ghee_1l", "brand": "amul",
|
||||
"product_name": "Amul Ghee 1L",
|
||||
{ "image_id": "amul_amul_ghee_1l", "brand": "amul", "brand_key": "amul",
|
||||
"product_name": "Amul Ghee 1L", "source_row": 3,
|
||||
"product_sku": "AMUL-GHE-1-001", "sku_source": "Internal",
|
||||
"disposition": "unchanged" }
|
||||
],
|
||||
"products_truncated": false
|
||||
"products_truncated": false,
|
||||
"rejections": [
|
||||
{ "row": 7, "product_name": "Kurkure Menthol", "size": "10g",
|
||||
"reason": "title is too short to be a real product name; image_urls: no images were found for this product" }
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
#### `image_id` — the join key, and what it is stable against
|
||||
|
||||
`image_id` is a pure deterministic function of **brand, product name and pack size**.
|
||||
Re-sending an unchanged sheet produces byte-identical ids, which is what makes the
|
||||
pipeline idempotent, and it is the column the catalogue deduplicates on. Store it
|
||||
rather than a row id.
|
||||
|
||||
What it is *not* stable against is any change to those three inputs. A pack size
|
||||
moving from `100g` to `250g` is a different SKU at a different price and is correctly
|
||||
a different id; so is a product name gaining or losing a brand prefix
|
||||
(`Hot Heads` vs `Nestle Hot Heads`). If a name is rewritten upstream, the id moves
|
||||
with it.
|
||||
|
||||
#### `brand_key` — the key the catalogue is addressed by
|
||||
|
||||
`brand` is the display name; `brand_key` is the identifier the catalogue is keyed
|
||||
on, and the two are not the same string:
|
||||
|
||||
| `brand` | `brand_key` |
|
||||
| --- | --- |
|
||||
| `24 Mantra` | `24_mantra` |
|
||||
| `Paper Boat` | `paper_boat` |
|
||||
| `coca-cola` | `coca_cola` |
|
||||
| `Own Products` | `own_products` |
|
||||
|
||||
Use `brand_key` rather than normalising `brand` yourself. The rule (lower-case,
|
||||
non-alphanumerics to underscores) is stable, but deriving it is a guess and the
|
||||
failure is silent — a wrong key finds nothing rather than erroring.
|
||||
|
||||
#### Unbranded rows — what `Own Products` means, and what it does not fill in
|
||||
|
||||
A row with no brand — loose fruit, vegetables, greens, flowers, fish, or staples
|
||||
like dal and sugar — is filed under the display brand **`Own Products`**
|
||||
(`brand_own_products`) rather than having a brand guessed from its first word.
|
||||
|
||||
**For these rows we store what your sheet said and nothing more.** A shopkeeper
|
||||
bills from this record, so a tax code or article number we invented would be our
|
||||
guess wearing your letterhead:
|
||||
|
||||
| Field | For an unbranded row |
|
||||
| --- | --- |
|
||||
| `product_name`, `size_variants` | from your sheet |
|
||||
| `selling_price`, `final_selling_price` | from your sheet |
|
||||
| `price_range` | a ±8% band around your price. Null if you sent no price |
|
||||
| `category` | derived from the product name — deterministic, not guessed |
|
||||
| `image_url`, `image_urls` | searched for, as with any other row |
|
||||
| `hsn_code`, `product_sku`, `barcode`, `fssai_license`, `description` | **null**, unless your sheet supplied them |
|
||||
|
||||
Anything you *do* send is kept: a sheet with its own HSN, SKU or description
|
||||
column keeps all three. The rule is "we do not invent", not "we discard".
|
||||
|
||||
Two consequences worth planning for:
|
||||
|
||||
- **No pack-size explosion.** A branded row with no size gets a plausible set
|
||||
(100g/250g/500g); an unbranded one does not, because a shop sells apples by
|
||||
whatever the customer asks for. A produce row with no weight column produces
|
||||
exactly one entry, with size `Standard`.
|
||||
- **`validation_status` is often `needs_review`.** For these rows that reflects
|
||||
a missing image, not a suspect product — the absence of price or SKU is no
|
||||
longer counted against them. `needs_review` rows are stored like any other;
|
||||
only `rejected` rows are dropped.
|
||||
|
||||
#### `source_row` — which line of the sheet produced this
|
||||
|
||||
The 1-based row number as the sender sees it on screen, **header counted as row 1**,
|
||||
so the first data row is `2`. The same convention the `422` responses use.
|
||||
|
||||
This is **many-to-one**: a pack-size cell reading `100g, 200g, 500g` legitimately
|
||||
becomes three products, and all three carry the same `source_row`. Rows that produced
|
||||
nothing are the ones absent from every entry — which is how you tell a shopkeeper
|
||||
"rows 6 and 11 produced nothing".
|
||||
|
||||
#### `rejections[]` — which rows were refused, and why
|
||||
|
||||
`rejected` is a count; `rejections` is the explanation, and it has always been sent.
|
||||
One entry per refused row:
|
||||
|
||||
| Field | Meaning |
|
||||
| --- | --- |
|
||||
| `row` | The 1-based sheet row, same convention as `source_row`. `null` if the row could not be located |
|
||||
| `product_name` | The name as the sheet gave it |
|
||||
| `size` | The pack the refusal applies to, since one row can yield several |
|
||||
| `reason` | Every validation issue, joined with `; ` |
|
||||
|
||||
Capped at 50 entries per file. Present on both the single-batch read and the list
|
||||
endpoint — only `products` is dropped from list responses.
|
||||
|
||||
#### `disposition`
|
||||
|
||||
| `disposition` | What happened |
|
||||
| --- | --- |
|
||||
| `inserted` | New product, created by this run |
|
||||
|
||||
392
scripts/merge_haldiram.py
Normal file
392
scripts/merge_haldiram.py
Normal file
@@ -0,0 +1,392 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Fold `brand_haldiram` into `brand_haldirams` and drop the singular table.
|
||||
|
||||
WHY THERE WERE TWO
|
||||
------------------
|
||||
Nothing told the system they were one brand. `BRAND_ALIASES` had no haldiram
|
||||
entry, so `resolve_parent_brand` was the identity for both spellings and each
|
||||
upload built whichever table its sheet happened to name. That has been fixed in
|
||||
`app/services/brand_registry.py`; this script cleans up the rows the split left
|
||||
behind. **Deploy the alias first** - with it in place, an upload arriving while
|
||||
this runs lands in `brand_haldirams` instead of recreating the singular behind
|
||||
us.
|
||||
|
||||
WHAT IS ACTUALLY BEING MERGED
|
||||
-----------------------------
|
||||
Not a missing product. `brand_haldiram` holds ONE row which is a corrupted
|
||||
duplicate of a product already in `brand_haldirams`:
|
||||
|
||||
brand_haldiram 'Haldiram Aloo Bhujia 200g 45' category '3' size ['45']
|
||||
brand_haldirams "Haldiram's Aloo Bhujia 200g" Snacks ['200g']
|
||||
|
||||
The stray "45", the raw category id and the bare-number size all predate guards
|
||||
that now exist. Moving that row across would put a second, worse Aloo Bhujia
|
||||
inside the good table - duplication made worse rather than fixed. So the row is
|
||||
dropped, and only what it holds that the target does NOT is carried over:
|
||||
|
||||
* the PRICE. The corrupted row is 12 days newer (29 Aug vs 17 Aug) and says
|
||||
58 / Rs52-58 against the target's 48 / Rs45-50. A more recent upload is the
|
||||
better evidence of what the shop charges.
|
||||
|
||||
* the FSSAI LICENCE, under --fix-fssai (off by default; see below).
|
||||
|
||||
THE LICENCE PROBLEM
|
||||
-------------------
|
||||
Both `brand_haldirams` rows carry 10012042000244. That is LION DATES' licence -
|
||||
the same number on all 21 Lion Dates products - not Haldiram's. Haldiram's own,
|
||||
per FSSAI_LICENSES, is 10012011000140, which is what the corrupted row has. So
|
||||
the junk row holds the correct regulatory identifier and the clean rows hold
|
||||
another company's.
|
||||
|
||||
Shipping one manufacturer's licence on another's product is the same class of
|
||||
fault that `generic_products.py` exists to prevent, but repairing it is a
|
||||
separate decision from merging two tables - so it is behind `--fix-fssai` and
|
||||
off unless asked for. The seed file
|
||||
`data/seed_catalogs/archive/brand_catalog_haldirams.json` carries the same wrong
|
||||
number and is corrected in the same pass.
|
||||
|
||||
THE TABLE COMES BACK IF YOU ONLY DROP IT
|
||||
----------------------------------------
|
||||
`/app/data` is a named Docker volume and `reconcile_brand_catalogs()` exports
|
||||
any table that has rows but no seed file, on boot and every 300s. Production has
|
||||
very likely already written `brand_catalog_haldiram.json` into that volume. Left
|
||||
there it is at best a stale file claiming to be a brand catalogue, so step 5
|
||||
deletes it and reports whether it existed.
|
||||
|
||||
Usage:
|
||||
|
||||
python -m scripts.merge_haldiram # dry run, the default
|
||||
python -m scripts.merge_haldiram --apply
|
||||
python -m scripts.merge_haldiram --apply --fix-fssai
|
||||
|
||||
`--dry-run` is the default and `--apply` must be explicit: this rewrites
|
||||
whatever database `backend/.env` points at, which is production. The target host
|
||||
is printed on startup so it can be checked before committing, and both tables
|
||||
are backed up to disk before the first write.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import sys
|
||||
from datetime import datetime
|
||||
from decimal import Decimal
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from app.infrastructure.settings import DB_HOST, DB_NAME
|
||||
from app.services.vector_store import _connect
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
||||
logger = logging.getLogger("merge_haldiram")
|
||||
|
||||
SOURCE_TABLE = "brand_haldiram"
|
||||
TARGET_TABLE = "brand_haldirams"
|
||||
TARGET_IMAGE_ID = "haldirams_haldiram_s_aloo_bhujia_200g"
|
||||
SOURCE_IMAGE_ID = "haldiram_haldiram_aloo_bhujia_200g_45"
|
||||
|
||||
# Haldiram's own, from brand_registry.FSSAI_LICENSES.
|
||||
HALDIRAM_FSSAI = "10012011000140"
|
||||
# What the target rows wrongly carry today: Lion Dates'.
|
||||
WRONG_FSSAI = "10012042000244"
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
SEED_DIR = ROOT / "data" / "seed_catalogs"
|
||||
STALE_SEED = SEED_DIR / "brand_catalog_haldiram.json"
|
||||
TARGET_SEED = SEED_DIR / "archive" / "brand_catalog_haldirams.json"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Reading
|
||||
# ---------------------------------------------------------------------------
|
||||
def _table_exists(cur, table: str) -> bool:
|
||||
cur.execute(
|
||||
"SELECT 1 FROM information_schema.tables "
|
||||
"WHERE table_schema='public' AND table_name=%s",
|
||||
(table,),
|
||||
)
|
||||
return cur.fetchone() is not None
|
||||
|
||||
|
||||
def _rows(cur, table: str) -> List[Dict[str, Any]]:
|
||||
cur.execute(f"SELECT * FROM {table} ORDER BY id")
|
||||
cols = [d[0] for d in cur.description]
|
||||
return [dict(zip(cols, r)) for r in cur.fetchall()]
|
||||
|
||||
|
||||
def _jsonable(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Make a DB row writable as JSON - datetimes and vectors are not."""
|
||||
out = {}
|
||||
for key, value in row.items():
|
||||
if isinstance(value, datetime):
|
||||
out[key] = value.isoformat()
|
||||
elif key == "embedding" and value is not None:
|
||||
out[key] = list(value) if not isinstance(value, str) else value
|
||||
else:
|
||||
out[key] = value
|
||||
return out
|
||||
|
||||
|
||||
def _backup(source: List[Dict], target: List[Dict]) -> Path:
|
||||
"""Write both tables to disk BEFORE anything is changed. This is the undo."""
|
||||
stamp = datetime.now().strftime("%Y%m%d_%H%M%S")
|
||||
path = ROOT / "data" / f"haldiram_merge_backup_{stamp}.json"
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"taken_at": stamp,
|
||||
"db_host": DB_HOST,
|
||||
"db_name": DB_NAME,
|
||||
SOURCE_TABLE: [_jsonable(r) for r in source],
|
||||
TARGET_TABLE: [_jsonable(r) for r in target],
|
||||
},
|
||||
ensure_ascii=False,
|
||||
indent=1,
|
||||
default=str,
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
return path
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Preconditions
|
||||
# ---------------------------------------------------------------------------
|
||||
def _check(source: List[Dict], target: List[Dict]) -> Optional[str]:
|
||||
"""Refuse to run against data that is not what this script was written for.
|
||||
|
||||
The merge is hand-derived from one specific pair of rows. If either side has
|
||||
moved since, the reasoning above may no longer hold and a blind UPDATE could
|
||||
overwrite something real - so stop and let a human look.
|
||||
"""
|
||||
if len(source) != 1:
|
||||
return f"{SOURCE_TABLE} has {len(source)} rows, expected exactly 1"
|
||||
if source[0].get("image_id") != SOURCE_IMAGE_ID:
|
||||
return (f"{SOURCE_TABLE} row is {source[0].get('image_id')!r}, "
|
||||
f"expected {SOURCE_IMAGE_ID!r}")
|
||||
if not any(r.get("image_id") == TARGET_IMAGE_ID for r in target):
|
||||
return f"{TARGET_TABLE} has no row {TARGET_IMAGE_ID!r} to merge into"
|
||||
return None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Reporting
|
||||
# ---------------------------------------------------------------------------
|
||||
def _describe(source: Dict, target: Dict, fix_fssai: bool) -> None:
|
||||
logger.info("")
|
||||
logger.info(" %s -> %s", SOURCE_TABLE, TARGET_TABLE)
|
||||
logger.info("")
|
||||
logger.info(" row being DROPPED (corrupted duplicate):")
|
||||
logger.info(" product_name %r", source.get("product_name"))
|
||||
logger.info(" category %r size %s",
|
||||
source.get("category"), source.get("size_variants"))
|
||||
logger.info(" price %s (%s)",
|
||||
source.get("selling_price"), source.get("price_range"))
|
||||
logger.info("")
|
||||
logger.info(" row being KEPT and updated:")
|
||||
logger.info(" product_name %r", target.get("product_name"))
|
||||
logger.info(" category %r size %s",
|
||||
target.get("category"), target.get("size_variants"))
|
||||
logger.info(" price %s (%s) -> %s (%s)",
|
||||
target.get("selling_price"), target.get("price_range"),
|
||||
source.get("selling_price"), source.get("price_range"))
|
||||
if fix_fssai:
|
||||
logger.info(" fssai %s -> %s",
|
||||
target.get("fssai_license"), HALDIRAM_FSSAI)
|
||||
else:
|
||||
logger.info(" fssai %s (unchanged - pass --fix-fssai to "
|
||||
"correct it; see the docstring)", target.get("fssai_license"))
|
||||
logger.info(" name / category / size / image / description / embedding"
|
||||
" all kept as they are")
|
||||
logger.info("")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The seed file
|
||||
# ---------------------------------------------------------------------------
|
||||
def _plain(value: Any) -> Any:
|
||||
"""psycopg returns NUMERIC as Decimal, which json.dumps refuses.
|
||||
|
||||
Caught the hard way: the database work committed and then the seed write
|
||||
blew up on `final_selling_price`, leaving the two halves out of step. The
|
||||
conversion happens here rather than at the call site so every field copied
|
||||
into a JSON file goes through it.
|
||||
"""
|
||||
return float(value) if isinstance(value, Decimal) else value
|
||||
|
||||
|
||||
def _update_seed(live_rows: List[Dict], fix_fssai: bool, apply: bool) -> None:
|
||||
"""Bring the archived catalogue into step with the TABLE.
|
||||
|
||||
Mirrored from the live rows rather than from the row being merged in, for
|
||||
two reasons. It is idempotent - re-running compares the file against the
|
||||
database and changes only what differs - and it still works once the source
|
||||
table has been dropped, which matters because the database write and this
|
||||
file write are not in one transaction. They came apart once already: the
|
||||
merge committed and then the JSON write failed on a Decimal, leaving the two
|
||||
halves disagreeing until this ran again.
|
||||
"""
|
||||
if not TARGET_SEED.exists():
|
||||
logger.info(" seed file %s not found - skipping", TARGET_SEED.name)
|
||||
return
|
||||
|
||||
by_id = {r["image_id"]: r for r in live_rows}
|
||||
doc = json.loads(TARGET_SEED.read_text(encoding="utf-8-sig"))
|
||||
changes: List[str] = []
|
||||
|
||||
for product in doc.get("products", []):
|
||||
row = by_id.get(product.get("image_id"))
|
||||
if not row:
|
||||
continue
|
||||
for field in ("price_range", "selling_price", "final_selling_price"):
|
||||
new = _plain(row.get(field))
|
||||
if product.get(field) != new:
|
||||
changes.append(f"{product['image_id']}.{field}: "
|
||||
f"{product.get(field)!r} -> {new!r}")
|
||||
product[field] = new
|
||||
if fix_fssai and product.get("fssai_license") != row.get("fssai_license"):
|
||||
changes.append(f"{product['image_id']}.fssai_license: "
|
||||
f"{product.get('fssai_license')!r} -> "
|
||||
f"{row.get('fssai_license')!r}")
|
||||
product["fssai_license"] = row.get("fssai_license")
|
||||
|
||||
if not changes:
|
||||
logger.info(" %s: already in step with the table", TARGET_SEED.name)
|
||||
return
|
||||
|
||||
logger.info(" %s: %d field(s) differ from the table", TARGET_SEED.name, len(changes))
|
||||
for line in changes:
|
||||
logger.info(" %s", line)
|
||||
if apply:
|
||||
TARGET_SEED.write_text(
|
||||
json.dumps(doc, ensure_ascii=False, indent=1), encoding="utf-8")
|
||||
logger.info(" written")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(
|
||||
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
|
||||
)
|
||||
parser.add_argument("--apply", action="store_true",
|
||||
help="commit the changes (default is a dry run)")
|
||||
parser.add_argument("--dry-run", action="store_true",
|
||||
help="explicit no-op; this is already the default")
|
||||
parser.add_argument("--fix-fssai", action="store_true",
|
||||
help="also replace Lion Dates' licence on the Haldirams "
|
||||
"rows with Haldiram's own")
|
||||
args = parser.parse_args()
|
||||
apply = args.apply and not args.dry_run
|
||||
|
||||
logger.info("Target database: %s / %s", DB_HOST, DB_NAME)
|
||||
logger.info("Mode: %s", "APPLY - this writes" if apply else "DRY RUN - nothing is written")
|
||||
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
logger.error("No database connection.")
|
||||
return 1
|
||||
|
||||
with conn:
|
||||
with conn.cursor() as cur:
|
||||
merged_already = not _table_exists(cur, SOURCE_TABLE)
|
||||
if merged_already:
|
||||
# The table is gone, but the seed files may still be behind -
|
||||
# they are written outside the database transaction, so a run
|
||||
# that committed the merge and then failed on the file leaves
|
||||
# exactly this state. Fall through to the file work.
|
||||
logger.info("")
|
||||
logger.info("%s is already gone; checking the seed files.",
|
||||
SOURCE_TABLE)
|
||||
target_rows = _rows(cur, TARGET_TABLE)
|
||||
source_rows = []
|
||||
else:
|
||||
source_rows = _rows(cur, SOURCE_TABLE)
|
||||
target_rows = _rows(cur, TARGET_TABLE)
|
||||
|
||||
problem = None if merged_already else _check(source_rows, target_rows)
|
||||
if problem:
|
||||
logger.error("")
|
||||
logger.error("ABORTING - the data is not what this script expects:")
|
||||
logger.error(" %s", problem)
|
||||
logger.error("")
|
||||
logger.error("Re-read the rows and update the script rather than "
|
||||
"forcing it; a blind UPDATE here could overwrite a "
|
||||
"real product.")
|
||||
return 1
|
||||
|
||||
if not merged_already:
|
||||
source = source_rows[0]
|
||||
target = next(r for r in target_rows
|
||||
if r["image_id"] == TARGET_IMAGE_ID)
|
||||
_describe(source, target, args.fix_fssai)
|
||||
|
||||
if apply and not merged_already:
|
||||
path = _backup(source_rows, target_rows)
|
||||
logger.info(" backup written to %s", path)
|
||||
|
||||
cur.execute(
|
||||
f"UPDATE {TARGET_TABLE} SET selling_price=%s, "
|
||||
f"final_selling_price=%s, price_range=%s, updated_at=now() "
|
||||
f"WHERE image_id=%s",
|
||||
(source.get("selling_price"), source.get("final_selling_price"),
|
||||
source.get("price_range"), TARGET_IMAGE_ID),
|
||||
)
|
||||
logger.info(" updated %s (%d row)", TARGET_TABLE, cur.rowcount)
|
||||
|
||||
if args.fix_fssai:
|
||||
cur.execute(
|
||||
f"UPDATE {TARGET_TABLE} SET fssai_license=%s, "
|
||||
f"updated_at=now() WHERE fssai_license=%s",
|
||||
(HALDIRAM_FSSAI, WRONG_FSSAI),
|
||||
)
|
||||
logger.info(" corrected fssai on %d row(s)", cur.rowcount)
|
||||
|
||||
cur.execute(f"DROP TABLE {SOURCE_TABLE}")
|
||||
logger.info(" dropped %s", SOURCE_TABLE)
|
||||
conn.commit()
|
||||
# Re-read so the seed mirrors what was actually committed.
|
||||
target_rows = _rows(cur, TARGET_TABLE)
|
||||
elif not merged_already:
|
||||
logger.info(" would UPDATE %s then DROP TABLE %s",
|
||||
TARGET_TABLE, SOURCE_TABLE)
|
||||
|
||||
# ---- the seed files ----------------------------------------------------
|
||||
logger.info("")
|
||||
_update_seed(target_rows, args.fix_fssai, apply)
|
||||
|
||||
if STALE_SEED.exists():
|
||||
logger.info(" %s EXISTS (auto-exported by reconcile) - deleting",
|
||||
STALE_SEED.name)
|
||||
if apply:
|
||||
STALE_SEED.unlink()
|
||||
logger.info(" deleted")
|
||||
else:
|
||||
logger.info(" %s not present locally", STALE_SEED.name)
|
||||
logger.info(" NOTE: production keeps data/ on a named volume, so check "
|
||||
"there too - reconcile exports any table with rows and no file.")
|
||||
|
||||
# ---- caches ------------------------------------------------------------
|
||||
if apply:
|
||||
from app.services.vector_store import invalidate_brand_overview_cache
|
||||
invalidate_brand_overview_cache()
|
||||
try:
|
||||
from app.services.query_intent import invalidate_brand_mention_cache
|
||||
invalidate_brand_mention_cache()
|
||||
except Exception: # noqa: BLE001 - a cold cache is not a failure
|
||||
pass
|
||||
logger.info("")
|
||||
logger.info("Done. Brand caches invalidated.")
|
||||
else:
|
||||
logger.info("")
|
||||
logger.info("Dry run - nothing written. Re-run with --apply to commit.")
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -16,7 +16,11 @@ from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
|
||||
from app.services.brand_registry import (
|
||||
BRAND_ALIASES,
|
||||
get_fssai_license,
|
||||
resolve_parent_brand,
|
||||
)
|
||||
from app.services.brand_sync import seed_catalog_paths
|
||||
from app.services.vector_store import _sanitize_name
|
||||
|
||||
@@ -84,6 +88,13 @@ EXPECTED_SEED_TABLES = {
|
||||
"parle": "brand_parle",
|
||||
"pepsico": "brand_pepsico",
|
||||
"tata": "brand_hindustan_unilever",
|
||||
# The loose-produce base list: fruit, vegetables, greens, flowers, fish and
|
||||
# loose dairy, none of which has a brand. It is the one seed catalog that
|
||||
# was hand-authored rather than scraped, and it MUST land in
|
||||
# brand_own_products - the same table generic_products.OWN_PRODUCTS_BRAND
|
||||
# sends unbranded upload rows to, so an uploaded "Apple" deduplicates
|
||||
# against the seeded one instead of creating a second row.
|
||||
"Own Products": "brand_own_products",
|
||||
}
|
||||
|
||||
|
||||
@@ -112,6 +123,38 @@ def test_seed_catalogs_keep_their_current_tables() -> None:
|
||||
assert actual == EXPECTED_SEED_TABLES
|
||||
|
||||
|
||||
def test_both_haldiram_spellings_reach_one_table() -> None:
|
||||
"""The regression that produced two tables for one brand.
|
||||
|
||||
Neither spelling was in BRAND_ALIASES, so resolve_parent_brand was the
|
||||
identity for both and every upload built whichever table its sheet happened
|
||||
to name - brand_haldiram (1 row) beside brand_haldirams (2). The rows were
|
||||
merged into the plural, which is the correct name, so there is no longer a
|
||||
singular table to assert against. This pins the PROPERTY instead, which is
|
||||
what actually stops it recurring.
|
||||
"""
|
||||
from app.services.vector_store import _sanitize_name
|
||||
|
||||
for spelling in ("Haldiram", "haldiram", "HALDIRAM", "Haldiram's", "haldirams"):
|
||||
table = f"brand_{_sanitize_name(resolve_parent_brand(spelling))}"
|
||||
assert table == "brand_haldirams", f"{spelling!r} routed to {table}"
|
||||
|
||||
|
||||
def test_the_haldiram_licence_survives_the_alias() -> None:
|
||||
"""FSSAI_LICENSES has to be keyed on the parent, not the alias.
|
||||
|
||||
get_fssai_license resolves to the canonical parent before looking up, so
|
||||
once "haldiram" aliases to "haldirams" a table keyed only on the singular
|
||||
returns None - and a blank licence is a legitimate outcome elsewhere, so
|
||||
nothing would flag it. Every future Haldiram row would simply ship without
|
||||
one.
|
||||
"""
|
||||
assert get_fssai_license("Haldiram") == "10012011000140"
|
||||
assert get_fssai_license("Haldirams") == "10012011000140"
|
||||
# Not Lion Dates', which is what the stored rows wrongly carried.
|
||||
assert get_fssai_license("Haldirams") != get_fssai_license("lion dates")
|
||||
|
||||
|
||||
def test_known_sub_brands_still_route_to_their_family() -> None:
|
||||
"""Word-boundary matching must not break legitimate sub-brand routing."""
|
||||
assert resolve_parent_brand("Dove") == "hindustan unilever"
|
||||
|
||||
@@ -326,3 +326,171 @@ def test_a_reupload_does_not_renumber_an_existing_products_sku(client, admin_hea
|
||||
second = _run_one()
|
||||
|
||||
assert first["product_sku"] == second["product_sku"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4. Reconciling a run against the sheet that produced it
|
||||
# ---------------------------------------------------------------------------
|
||||
# Four fields an integrator asked for after wiring the upload path end to end.
|
||||
# Each answers a question that previously had only an approximate answer:
|
||||
#
|
||||
# source_row which of MY rows produced this product, and which produced none
|
||||
# brand_key what key is the catalogue addressed by, given a display name
|
||||
# row which rows were refused, and why
|
||||
# from_drop which file in this run is mine, when a run spans several drops
|
||||
#
|
||||
# All four are ADDITIVE. Nothing existing changed meaning, because a consumer
|
||||
# reading the old fields must keep working across this deploy.
|
||||
|
||||
|
||||
def test_a_product_names_the_sheet_row_it_came_from(client, admin_headers, store):
|
||||
"""source_row is the number the sender sees on screen, header counted as 1."""
|
||||
rows = [["Amul Butter 100g", "Butter", "Amul"],
|
||||
["Amul Ghee 1L", "Ghee", "Amul"]]
|
||||
drop_id = client.post(UPLOAD, files=_files(("a.csv", _csv(rows=rows)))
|
||||
).json()["batch_id"]
|
||||
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
|
||||
headers=admin_headers).json()["batch_id"]
|
||||
|
||||
products = batch_ingest.run_batch(run_id).files[0].result["products"]
|
||||
|
||||
by_name = {p["product_name"]: p["source_row"] for p in products}
|
||||
# Row 1 is the header, so the first data row is 2.
|
||||
assert by_name["Amul Butter 100g"] == 2
|
||||
assert by_name["Amul Ghee 1L"] == 3
|
||||
|
||||
|
||||
def test_one_sheet_row_can_own_several_products(client, admin_headers, store):
|
||||
"""Pack-size explosion is many-to-one, and that is the point of the field.
|
||||
|
||||
"100g, 200g, 500g" in one cell is three catalog rows at three prices. The
|
||||
sender needs to be able to say "row 2 of your sheet became these three"
|
||||
rather than reconstruct it by matching names.
|
||||
"""
|
||||
headers = ["Product Name", "Category", "Brand", "Pack Size"]
|
||||
rows = [["Amul Butter", "Butter", "Amul", "100g; 200g; 500g"]]
|
||||
drop_id = client.post(
|
||||
UPLOAD, files=_files(("a.csv", _csv(headers=headers, rows=rows)))
|
||||
).json()["batch_id"]
|
||||
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
|
||||
headers=admin_headers).json()["batch_id"]
|
||||
|
||||
products = batch_ingest.run_batch(run_id).files[0].result["products"]
|
||||
|
||||
assert len(products) > 1, "the sheet did not explode; test proves nothing"
|
||||
assert {p["source_row"] for p in products} == {2}
|
||||
|
||||
|
||||
def test_a_product_carries_the_key_the_catalogue_is_addressed_by(client,
|
||||
admin_headers,
|
||||
store):
|
||||
"""brand_key beside the display brand.
|
||||
|
||||
The manifest reports "24 Mantra" while the catalogue is keyed 24_mantra, and
|
||||
an integrator normalising that themselves is guessing. Publishing the key
|
||||
removes a whole class of "Unknown brand" failure.
|
||||
"""
|
||||
from app.services.vector_store import _sanitize_name
|
||||
|
||||
drop_id = _drop(client, "priya")
|
||||
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
|
||||
headers=admin_headers).json()["batch_id"]
|
||||
|
||||
product = batch_ingest.run_batch(run_id).files[0].result["products"][0]
|
||||
|
||||
assert product["brand_key"] == _sanitize_name(product["brand"])
|
||||
assert " " not in product["brand_key"]
|
||||
|
||||
|
||||
def test_a_run_says_which_drop_each_file_came_from(client, admin_headers):
|
||||
"""from_drop is the exact answer to "which file in this run is mine".
|
||||
|
||||
Both drops here send a file called a.csv, which is precisely the collision
|
||||
that made filename-matching unsafe: without from_drop a sender could match
|
||||
the wrong merchant's file and price it as their own.
|
||||
"""
|
||||
first = _drop(client, "priya")
|
||||
second = _drop(client, "arun")
|
||||
|
||||
run = client.post(
|
||||
FROM_INBOX,
|
||||
json={"file_ids": [f"{first}:0", f"{second}:0"]},
|
||||
headers=admin_headers,
|
||||
).json()
|
||||
|
||||
assert [f["filename"] for f in run["files"]] == ["a.csv", "a.csv"]
|
||||
assert {f["from_drop"] for f in run["files"]} == {first, second}
|
||||
|
||||
|
||||
def test_from_drop_survives_a_reread_of_the_run(client, admin_headers):
|
||||
"""It is stamped after staging, so it has to reach disk, not just the reply."""
|
||||
drop_id = _drop(client, "priya")
|
||||
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
|
||||
headers=admin_headers).json()["batch_id"]
|
||||
|
||||
assert batch_ingest.read_manifest(run_id).files[0].from_drop == drop_id
|
||||
|
||||
served = client.get(f"/api/admin/catalog-batch/batches/{run_id}",
|
||||
headers=admin_headers).json()
|
||||
assert served["files"][0]["from_drop"] == drop_id
|
||||
|
||||
|
||||
def test_a_file_uploaded_straight_into_a_run_has_no_drop(client, monkeypatch):
|
||||
"""Nothing to point at when the file never sat in an inbox.
|
||||
|
||||
None rather than the run's own id: saying a run came from itself would make
|
||||
the field useless for the question it exists to answer.
|
||||
"""
|
||||
from app.api.routers import uploads
|
||||
monkeypatch.setattr(uploads, "UPLOAD_AUTORUN", True)
|
||||
|
||||
run = client.post(UPLOAD, files=_files(("a.csv", _csv()))).json()
|
||||
|
||||
assert run["files"][0]["from_drop"] is None
|
||||
|
||||
|
||||
def test_a_refused_row_names_itself(client, admin_headers, store, monkeypatch):
|
||||
"""rejections[] carries the sheet row, not just a count.
|
||||
|
||||
The count alone was unusable for support: "rejected: 2" out of 19 products
|
||||
left the only diagnosis being to diff the manifest against the file and
|
||||
guess. A refused row is a product a shopkeeper expects on the shelf and
|
||||
will not have, so it has to name itself and say why.
|
||||
|
||||
Everything here except `row` already shipped; this pins the whole shape
|
||||
together so a future change cannot quietly drop one field of it.
|
||||
"""
|
||||
monkeypatch.setattr(pipeline, "ENABLE_PRODUCT_VALIDATION", True)
|
||||
rows = [["Amul Butter 100g", "Butter", "Amul"],
|
||||
["X", "", "Amul"]]
|
||||
drop_id = client.post(UPLOAD, files=_files(("a.csv", _csv(rows=rows)))
|
||||
).json()["batch_id"]
|
||||
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
|
||||
headers=admin_headers).json()["batch_id"]
|
||||
|
||||
result = batch_ingest.run_batch(run_id).files[0].result
|
||||
|
||||
assert result["rejected"] == len(result["rejections"]), (
|
||||
"the count and the list must agree, or the list is not the explanation"
|
||||
)
|
||||
if result["rejections"]:
|
||||
bad = result["rejections"][0]
|
||||
assert set(bad) >= {"row", "product_name", "size", "reason"}
|
||||
assert bad["reason"], "a rejection with no reason explains nothing"
|
||||
# Row 1 is the header, so any real refusal is row 2 or later.
|
||||
assert bad["row"] is None or bad["row"] >= 2
|
||||
|
||||
|
||||
def test_the_rejection_row_matches_the_offending_sheet_line():
|
||||
"""Straight at the gate, so the row number is checked without needing a
|
||||
sheet that reliably fails validation end to end."""
|
||||
rows = [
|
||||
{"product_name": "X", "brand": "amul", "size": "100g",
|
||||
"category": "Dairy", "image_id": "a", "_row": 7},
|
||||
{"product_name": "", "brand": "amul", "size": "",
|
||||
"category": "", "image_id": "b", "_row": 9},
|
||||
]
|
||||
|
||||
_kept, rejected, _summary = pipeline.stage_10_validate(rows, "amul")
|
||||
|
||||
assert [r["_row"] for r in rejected] == [7, 9]
|
||||
|
||||
@@ -167,10 +167,58 @@ def test_a_commodity_resolves_to_a_real_category(name, category):
|
||||
assert canonical_category(name) is not None
|
||||
|
||||
|
||||
def test_the_category_carries_an_hsn_code(store):
|
||||
"""The names were chosen to match HSN_GST_TABLE keys, so tax enrichment
|
||||
works without a second mapping."""
|
||||
@pytest.mark.parametrize("name,hsn", [
|
||||
("Toor Dhal 1kg", "0713"),
|
||||
("Sugar 1kg", "1701"),
|
||||
("Salt 1kg", "2501"),
|
||||
# Fruit and vegetables share one category, so one code has to serve both.
|
||||
# 0709 ("other vegetables, fresh or chilled") is the one carried; strictly
|
||||
# a fruit is chapter 08. Both are nil-rated, so the GST is right either
|
||||
# way, and this only ever surfaces in the seeded base list because an
|
||||
# uploaded commodity is given no HSN at all.
|
||||
("Apple", "0709"),
|
||||
("Tomato", "0709"),
|
||||
])
|
||||
def test_a_commodity_category_is_a_key_the_hsn_table_knows(name, hsn):
|
||||
"""The commodity categories line up with HSN_GST_TABLE keys.
|
||||
|
||||
Checked against the table directly rather than through an ingest. It used
|
||||
to be asserted end to end, but the upload path no longer STAMPS an HSN onto
|
||||
a commodity (see the test below), so running a sheet would now prove
|
||||
nothing about the mapping. The property itself is still worth pinning: it
|
||||
is what lets a category name serve as the tax key without a second lookup,
|
||||
and it is how a sheet that DOES declare its tax treatment stays consistent
|
||||
with ours.
|
||||
"""
|
||||
from app.services.enrichment.hsn_gst.models import HSN_GST_TABLE
|
||||
|
||||
category = canonical_category(name)
|
||||
assert category in HSN_GST_TABLE, f"{category!r} has no HSN entry"
|
||||
assert HSN_GST_TABLE[category][0] == hsn
|
||||
|
||||
|
||||
def test_an_uploaded_commodity_is_given_no_hsn_code(store):
|
||||
"""We do not invent a tax code the merchant did not supply.
|
||||
|
||||
An HSN we chose is our guess presented as their record, and a shopkeeper
|
||||
bills from this. Fresh produce being nil-rated makes a wrong code cheap to
|
||||
ignore and expensive to notice, which is the worst combination.
|
||||
"""
|
||||
_brand_of(["Toor Dhal 1kg"], store)
|
||||
assert store[0][1]["hsn_code"] is None
|
||||
|
||||
|
||||
def test_a_commodity_keeps_an_hsn_the_sheet_supplied(store):
|
||||
"""The rule is "do not invent", not "discard"."""
|
||||
wb = openpyxl.Workbook()
|
||||
ws = wb.active
|
||||
ws.append(["Item Name", "HSN Code"])
|
||||
ws.append(["Toor Dhal 1kg", "0713"])
|
||||
buf = io.BytesIO()
|
||||
wb.save(buf)
|
||||
|
||||
pipeline.run_pipeline("store.xlsx", buf.getvalue(), use_llm=False, fetch_images=False)
|
||||
|
||||
assert store[0][1]["hsn_code"] == "0713"
|
||||
|
||||
|
||||
|
||||
201
tests/test_generic_products_produce.py
Normal file
201
tests/test_generic_products_produce.py
Normal file
@@ -0,0 +1,201 @@
|
||||
"""Loose produce reaches Own Products, and no real brand follows it there.
|
||||
|
||||
WHY THIS FILE IS THE GATE ON THE COMMODITY LEXICON
|
||||
--------------------------------------------------
|
||||
`is_unbranded()` works by requiring that EVERY significant token in a product
|
||||
name is a known commodity or qualifier. That makes it asymmetric: adding a word
|
||||
to the lexicon can only ever make the test more permissive, so the failure mode
|
||||
is real brands quietly collapsing into one bucket - far harder to undo than a
|
||||
staple sitting in the wrong table.
|
||||
|
||||
So the tests that matter most here are the negative ones. Before produce was
|
||||
added, the candidate list was measured against the live catalogue and the alias
|
||||
map; these encode that measurement, so the next person to add a word finds out
|
||||
immediately if it swallows something it should not.
|
||||
|
||||
The positive tests exist because the alternative to landing in Own Products is
|
||||
not "rejected" - it is being MISFILED. "Red Rose" resolved to brand "Red", which
|
||||
resolve_parent_brand whole-word matched to Brooke Bond, so a flower was written
|
||||
into the tea catalogue and stamped with Brooke Bond's FSSAI licence.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
|
||||
from app.services.generic_products import (
|
||||
OWN_PRODUCTS_BRAND,
|
||||
canonical_category,
|
||||
is_unbranded,
|
||||
)
|
||||
|
||||
SEED_DIR = Path(__file__).resolve().parents[1] / "data" / "seed_catalogs"
|
||||
OWN_PRODUCTS_SEED = SEED_DIR / "brand_catalog_own_products.json"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The negative tests: nothing branded may fall in here
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.mark.parametrize("alias", sorted(BRAND_ALIASES))
|
||||
def test_no_brand_alias_is_read_as_a_commodity(alias: str) -> None:
|
||||
"""Every alias must stay branded.
|
||||
|
||||
"amla" is the near miss: a real fruit that also appears inside the alias
|
||||
"dabur amla". That stays branded because "dabur" is not a commodity - which
|
||||
is exactly the conservatism the rule rests on: one unknown token is enough
|
||||
to mean "this is a brand".
|
||||
"""
|
||||
assert not is_unbranded(alias)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"name",
|
||||
[
|
||||
"Amul Butter 500g",
|
||||
"Aachi Sambar Powder",
|
||||
"Brooke Bond Red Label 500g",
|
||||
"Colgate Active Salt",
|
||||
"Milky Mist Paneer",
|
||||
"Nature Fresh Atta",
|
||||
"Dabur Amla Hair Oil",
|
||||
"Mother Dairy Milk 1L",
|
||||
"24 Mantra Organic Moong Dal 500g",
|
||||
],
|
||||
)
|
||||
def test_branded_products_stay_branded(name: str) -> None:
|
||||
assert not is_unbranded(name)
|
||||
|
||||
|
||||
def test_a_brand_whose_name_starts_with_a_number_survives() -> None:
|
||||
"""The _strip_sizes regression.
|
||||
|
||||
The size strip used to be a number followed by "any letters", which ate the
|
||||
word AFTER a number: "24 Mantra Organic Moong Dal" became "organic moong
|
||||
dal", every remaining token was a commodity, and a real branded product was
|
||||
filed as unbranded. Every brand beginning with a digit hit this.
|
||||
"""
|
||||
assert not is_unbranded("24 Mantra Organic Moong Dal 500g")
|
||||
assert not is_unbranded("24 Mantra Organic Sona Masuri Rice 1kg")
|
||||
# ... while the thing the strip actually exists for still works.
|
||||
assert is_unbranded("Toor Dhal 1kg")
|
||||
assert is_unbranded("Sugar 1kg")
|
||||
assert is_unbranded("Black Pepper 100g")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The positive tests: produce must reach Own Products
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.mark.parametrize(
|
||||
"name,category",
|
||||
[
|
||||
("Apple", "Fruits & Vegetables"),
|
||||
("Orange", "Fruits & Vegetables"),
|
||||
("Tomato", "Fruits & Vegetables"),
|
||||
("Onion", "Fruits & Vegetables"),
|
||||
("Potato", "Fruits & Vegetables"),
|
||||
("Banana", "Fruits & Vegetables"),
|
||||
("Drumstick", "Fruits & Vegetables"),
|
||||
("Bitter Gourd", "Fruits & Vegetables"),
|
||||
("Lady Finger", "Fruits & Vegetables"),
|
||||
("Curry Leaves", "Fresh Herbs & Greens"),
|
||||
("Mint Leaves", "Fresh Herbs & Greens"),
|
||||
("Thulasi", "Fresh Herbs & Greens"),
|
||||
("Jasmine", "Flowers"),
|
||||
("Red Rose", "Flowers"),
|
||||
("Tuna", "Fish & Seafood"),
|
||||
("Prawns", "Fish & Seafood"),
|
||||
("Egg", "Eggs"),
|
||||
# Pantry staples that were already covered, pinned so the produce work
|
||||
# cannot regress them.
|
||||
("Toor Dhal 1kg", "Pulses, Grains & Spices"),
|
||||
("Sugar 1kg", "Sugar & Jaggery"),
|
||||
("Salt 1kg", "Salt & Staples"),
|
||||
],
|
||||
)
|
||||
def test_loose_goods_are_unbranded_and_categorised(name: str, category: str) -> None:
|
||||
assert is_unbranded(name), f"{name} would be given a junk brand"
|
||||
assert canonical_category(name) == category
|
||||
|
||||
|
||||
def test_a_flower_no_longer_lands_in_the_tea_catalogue() -> None:
|
||||
"""The specific misroute this work exists to end.
|
||||
|
||||
"Red Rose" -> infer_brand -> "Red" -> resolve_parent_brand -> "brooke bond",
|
||||
so a rose was written into brand_brooke_bond carrying Brooke Bond's real
|
||||
FSSAI licence. The fix is upstream: the row never reaches infer_brand.
|
||||
"""
|
||||
assert is_unbranded("Red Rose")
|
||||
# The hijack is still there for anything that DOES reach it, so this test
|
||||
# fails loudly if the diversion is removed rather than passing for the
|
||||
# wrong reason.
|
||||
assert resolve_parent_brand("Red") == "brooke bond"
|
||||
|
||||
|
||||
def test_merchant_typos_still_resolve() -> None:
|
||||
"""Real strings from merchant data, misspellings included."""
|
||||
for name in ("Bitter guard", "Bottle ground", "Ladies Finger"):
|
||||
assert is_unbranded(name), name
|
||||
|
||||
|
||||
def test_an_empty_brand_column_is_believed() -> None:
|
||||
"""A sheet WITH a brand column that left the cell blank has said something.
|
||||
|
||||
This is how place-qualified produce gets in. "Salem Mango" keeps an unknown
|
||||
token, so the word test alone calls it branded - deliberately, because
|
||||
"Mysore" is also a real brand (Mysore Sandal). An explicit empty cell
|
||||
overrides that.
|
||||
"""
|
||||
assert not is_unbranded("Salem Mango")
|
||||
assert is_unbranded("Salem Mango", brand_column_supplied=True)
|
||||
# A filled cell is believed just as much.
|
||||
assert not is_unbranded("Apple", sheet_brand="Washington")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The seeded base list
|
||||
# ---------------------------------------------------------------------------
|
||||
def _seed_doc():
|
||||
if not OWN_PRODUCTS_SEED.exists():
|
||||
pytest.skip("own-products seed catalog not present")
|
||||
return json.loads(OWN_PRODUCTS_SEED.read_text(encoding="utf-8-sig"))
|
||||
|
||||
|
||||
def test_the_seed_catalog_is_filed_under_own_products() -> None:
|
||||
doc = _seed_doc()
|
||||
assert doc["brand"] == OWN_PRODUCTS_BRAND
|
||||
assert doc["total_products"] == len(doc["products"])
|
||||
assert doc["products"], "the base list is empty"
|
||||
|
||||
|
||||
def test_every_seeded_row_would_also_be_recognised_on_upload() -> None:
|
||||
"""The round trip that makes the base list worth having.
|
||||
|
||||
A grocer who types "Tomato" into their own sheet must land on the SAME row
|
||||
that was seeded rather than create a second one. That only holds if every
|
||||
seeded name is itself classified unbranded - otherwise the uploaded copy
|
||||
goes to a junk brand table and the two never meet.
|
||||
"""
|
||||
missed = [
|
||||
p["product_name"] for p in _seed_doc()["products"]
|
||||
if not is_unbranded(p["product_name"])
|
||||
]
|
||||
assert not missed, f"seeded rows a real upload would misfile: {missed}"
|
||||
|
||||
|
||||
def test_seeded_image_ids_are_unique() -> None:
|
||||
"""image_id is the deduplication key, and the column is UNIQUE NOT NULL."""
|
||||
ids = [p["image_id"] for p in _seed_doc()["products"]]
|
||||
assert len(ids) == len(set(ids))
|
||||
assert all(ids)
|
||||
|
||||
|
||||
def test_seeded_rows_carry_no_brand_pack_size_or_price() -> None:
|
||||
"""Loose produce has none of those, and inventing them would be a lie."""
|
||||
for p in _seed_doc()["products"]:
|
||||
assert p["brand_name"] == OWN_PRODUCTS_BRAND
|
||||
assert p["size_variants"] == []
|
||||
assert p["price_range"] is None
|
||||
assert p["fssai_license"] is None
|
||||
311
tests/test_own_products_fields.py
Normal file
311
tests/test_own_products_fields.py
Normal file
@@ -0,0 +1,311 @@
|
||||
"""What actually gets written for an unbranded row: the sheet's values, and nothing else.
|
||||
|
||||
THE RULE
|
||||
--------
|
||||
A merchant sends `Apple, 500g, 155`. Those three values are what we know. FSSAI,
|
||||
HSN, SKU, barcode and description are things we would be *making up*, and a
|
||||
shopkeeper bills from this record - an invented tax code is our guess wearing
|
||||
their letterhead.
|
||||
|
||||
WHY THIS NEEDED A VALIDATION CHANGE AS WELL
|
||||
-------------------------------------------
|
||||
Leaving those fields empty is not free. The validation gate scores a row down
|
||||
for each missing field and drops it below the reject threshold:
|
||||
|
||||
0.55 baseline - 0.30 (no price_range) - 0.10 (no SKU) = 0.15 vs 0.35
|
||||
|
||||
So the literal instruction "leave these null" would have deleted every produce
|
||||
row - the exact opposite of the requirement. `validate_product` now treats a
|
||||
commodity's missing price and SKU as normal rather than as evidence of a
|
||||
fabricated row. The first test below is the wall around that, and the branded
|
||||
counterpart proves the exemption did not leak.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
|
||||
import pytest
|
||||
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.services.generic_products import OWN_PRODUCTS_BRAND
|
||||
from app.services.product_validator import validate_product
|
||||
|
||||
openpyxl = pytest.importorskip("openpyxl")
|
||||
|
||||
|
||||
def _sheet(headers, rows) -> bytes:
|
||||
wb = openpyxl.Workbook()
|
||||
ws = wb.active
|
||||
ws.append(headers)
|
||||
for row in rows:
|
||||
ws.append(row)
|
||||
buf = io.BytesIO()
|
||||
wb.save(buf)
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolate_sku_counter(tmp_path, monkeypatch):
|
||||
from app.services import sku_service
|
||||
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store(monkeypatch):
|
||||
"""Capture what would be written, keyed the way the real table is."""
|
||||
written: list = []
|
||||
|
||||
def _upsert(brand, products, cleanup=False):
|
||||
for product in products:
|
||||
written.append((brand, dict(product)))
|
||||
return len(products)
|
||||
|
||||
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda brand: [])
|
||||
monkeypatch.setattr(pipeline, "upsert_brand_products", _upsert)
|
||||
monkeypatch.setattr(pipeline, "USE_EMBEDDINGS", False)
|
||||
return written
|
||||
|
||||
|
||||
def _run(headers, rows):
|
||||
pipeline.run_pipeline("store.xlsx", _sheet(headers, rows),
|
||||
use_llm=False, fetch_images=False)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The gate, which is what makes the rest of this possible
|
||||
# ---------------------------------------------------------------------------
|
||||
WORKED_EXAMPLE = {
|
||||
"product_name": "Apple", "title": "Apple",
|
||||
"category": "Fruits & Vegetables", "size": "500g",
|
||||
"selling_price": 155, "final_selling_price": 155,
|
||||
"price_range": "₹143-167",
|
||||
"product_sku": None, "sku_source": None,
|
||||
"hsn_code": None, "fssai_license": None, "description": None,
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("images,expected", [
|
||||
(["https://example.com/apple.jpg"], "verified"),
|
||||
([], "needs_review"),
|
||||
])
|
||||
def test_the_worked_example_is_never_rejected(images, expected):
|
||||
"""`Apple / 500g / 155` with everything else null must survive.
|
||||
|
||||
Both outcomes are KEPT: validate_catalog returns verified and needs_review
|
||||
rows together, and only `rejected` is dropped. The no-image case stays
|
||||
needs_review deliberately - a missing picture is the one absence here that
|
||||
still says something, since image search did run and found nothing.
|
||||
"""
|
||||
row = dict(WORKED_EXAMPLE, image_urls=images)
|
||||
|
||||
report = validate_product(row, OWN_PRODUCTS_BRAND,
|
||||
category_resolved_deterministically=True,
|
||||
images_checked=True)
|
||||
|
||||
assert report.status == expected
|
||||
assert report.status != "rejected"
|
||||
|
||||
|
||||
def test_a_branded_row_with_the_same_gaps_is_still_rejected():
|
||||
"""The exemption is scoped to commodities and must not leak.
|
||||
|
||||
Same row, same absences, under a real brand: a branded product with no
|
||||
price and no SKU IS evidence that something went wrong upstream, and that
|
||||
judgement is unchanged.
|
||||
"""
|
||||
row = dict(WORKED_EXAMPLE, image_urls=[], price_range=None)
|
||||
|
||||
report = validate_product(row, "amul",
|
||||
category_resolved_deterministically=True,
|
||||
images_checked=True)
|
||||
|
||||
assert report.status == "rejected"
|
||||
|
||||
|
||||
def test_a_malformed_price_on_a_commodity_is_still_caught():
|
||||
"""Absent is excused; wrong is not."""
|
||||
row = dict(WORKED_EXAMPLE, image_urls=[], price_range="one fifty five")
|
||||
|
||||
report = validate_product(row, OWN_PRODUCTS_BRAND,
|
||||
category_resolved_deterministically=True,
|
||||
images_checked=True)
|
||||
|
||||
assert any(issue.field == "price_range" for issue in report.issues)
|
||||
|
||||
|
||||
def test_a_blank_product_name_is_still_caught():
|
||||
"""This is not a way in for junk. Everything except the two excused
|
||||
absences is still enforced."""
|
||||
row = dict(WORKED_EXAMPLE, product_name="", title="", image_urls=[])
|
||||
|
||||
report = validate_product(row, OWN_PRODUCTS_BRAND,
|
||||
category_resolved_deterministically=True,
|
||||
images_checked=True)
|
||||
|
||||
assert report.status == "rejected"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The field policy, end to end
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_the_sheets_values_are_kept_and_nothing_else_is_invented(store):
|
||||
"""The requirement, as one assertion per field."""
|
||||
_run(["Product Name", "Weight", "Selling Price"], [["Apple", "500g", 155]])
|
||||
|
||||
assert len(store) == 1
|
||||
brand, row = store[0]
|
||||
assert brand == OWN_PRODUCTS_BRAND
|
||||
|
||||
# What the sheet said.
|
||||
assert row["product_name"] == "Apple 500g"
|
||||
assert row["size_variants"] == ["500g"]
|
||||
assert row["final_selling_price"] == 155
|
||||
assert row["price_range"] == "₹143-167" # +/-8% of the sheet's own price
|
||||
|
||||
# What it did not say, and what we therefore do not claim.
|
||||
for field in ("hsn_code", "product_sku", "sku_source",
|
||||
"fssai_license", "barcode", "barcode_type", "description"):
|
||||
assert row[field] is None, f"{field} was invented: {row[field]!r}"
|
||||
|
||||
|
||||
def test_the_price_band_is_eight_percent_of_the_sheet_price(store):
|
||||
_run(["Product Name", "Weight", "Selling Price"], [["Tomato", "1kg", 100]])
|
||||
|
||||
assert store[0][1]["price_range"] == "₹92-108"
|
||||
|
||||
|
||||
def test_a_commodity_with_no_price_gets_no_band(store):
|
||||
"""The market estimator is not consulted for loose produce.
|
||||
|
||||
It is trained on packaged FMCG and prices a 500g apple at around Rs85-105,
|
||||
which is not so much wrong as meaningless - a shop prices produce by the
|
||||
day. A null band is the honest answer.
|
||||
"""
|
||||
_run(["Product Name", "Weight"], [["Apple", "500g"]])
|
||||
|
||||
assert store[0][1]["price_range"] is None
|
||||
|
||||
|
||||
def test_a_branded_row_is_untouched_by_all_of_this(store):
|
||||
"""The blast radius check. A real brand still gets its full enrichment."""
|
||||
_run(["Product Name", "Brand", "Category", "Weight", "Selling Price"],
|
||||
[["Amul Butter", "Amul", "Dairy", "500g", 100]])
|
||||
|
||||
_brand, row = store[0]
|
||||
assert row["fssai_license"], "a branded row lost its FSSAI licence"
|
||||
assert row["product_sku"], "a branded row lost its minted SKU"
|
||||
assert row["description"], "a branded row lost its description"
|
||||
assert row["price_range"] == "₹92-108"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Pack sizes
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_commodity_with_no_weight_yields_exactly_one_row(store):
|
||||
"""No invented 100g/250g/500g.
|
||||
|
||||
A shop sells apples by whatever the customer asks for, so "Apple 250g" is a
|
||||
product that does not exist. This is also the mechanism behind the
|
||||
catalogue drift reported by our integrator: the invented set is keyed on
|
||||
the resolved category, so the same product ingested twice with the category
|
||||
resolved differently produces two disjoint size sets and two sets of ids.
|
||||
"""
|
||||
_run(["Product Name"], [["Apple"]])
|
||||
|
||||
assert len(store) == 1
|
||||
assert store[0][1]["size_variants"] == ["Standard"]
|
||||
|
||||
|
||||
def test_an_uploaded_commodity_lands_on_the_seeded_row(store):
|
||||
"""The dedupe that makes the base list worth having.
|
||||
|
||||
A grocer typing "Apple" must land on the seeded "Apple" rather than create
|
||||
a second one. That holds only if both sides build the same image_id, which
|
||||
means both must use "Standard" for an absent size.
|
||||
"""
|
||||
_run(["Product Name"], [["Apple"]])
|
||||
|
||||
assert store[0][1]["image_id"] == pipeline.build_image_id(
|
||||
OWN_PRODUCTS_BRAND, "Apple", "Standard")
|
||||
|
||||
|
||||
def test_a_declared_pack_size_is_still_honoured(store):
|
||||
"""Not inventing sizes must not mean ignoring the ones we were given."""
|
||||
_run(["Product Name", "Weight"], [["Apple", "500g"]])
|
||||
|
||||
assert store[0][1]["size_variants"] == ["500g"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# "Do not invent" is not "discard"
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_values_the_sheet_supplied_survive(store):
|
||||
"""A merchant who fills in HSN, SKU and a description keeps all three."""
|
||||
_run(["Product Name", "Weight", "Selling Price", "HSN Code", "SKU", "Description"],
|
||||
[["Apple", "500g", 155, "0808", "SHOP-APL-1", "Shimla apples, loose"]])
|
||||
|
||||
row = store[0][1]
|
||||
assert row["hsn_code"] == "0808"
|
||||
assert row["product_sku"] == "SHOP-APL-1"
|
||||
assert row["sku_source"] == "sheet"
|
||||
assert row["description"] == "Shimla apples, loose"
|
||||
|
||||
|
||||
def test_a_supplied_price_range_wins_over_the_derived_band(store):
|
||||
_run(["Product Name", "Weight", "Selling Price", "Price Range"],
|
||||
[["Apple", "500g", 155, "₹150-160"]])
|
||||
|
||||
assert store[0][1]["price_range"] == "₹150-160"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The embedding text
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_the_search_text_does_not_embed_the_bucket_name_or_a_null(store):
|
||||
"""`search_query` feeds the vector index.
|
||||
|
||||
Interpolated the branded way it would read "Own Products Apple 500g
|
||||
Fruits & Vegetables None" - the bucket is not a maker, and "None" is the
|
||||
string repr of the description we deliberately left empty. Both would be
|
||||
embedded and both would pull unrelated produce together.
|
||||
"""
|
||||
_run(["Product Name", "Weight"], [["Apple", "500g"]])
|
||||
|
||||
query = store[0][1]["search_query"]
|
||||
assert "Own Products" not in query
|
||||
assert "None" not in query
|
||||
assert "Apple" in query and "Fruits & Vegetables" in query
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Category: the lexicon is a hint, the pack size is a fact
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_commodity_gets_its_category_from_the_lexicon(store):
|
||||
"""Keyword detection has no entry for individual fruit and never will.
|
||||
|
||||
Listing every vegetable in the curated registry would duplicate the
|
||||
commodity lexicon and let the two drift, so the lexicon is consulted as a
|
||||
second opinion. Without it every produce row landed in "General", which
|
||||
loses the unit rules, the grouping and the validation credit.
|
||||
"""
|
||||
_run(["Product Name", "Weight"], [["Apple", "500g"]])
|
||||
|
||||
assert store[0][1]["category"] == "Fruits & Vegetables"
|
||||
|
||||
|
||||
def test_a_lexicon_category_never_overrules_a_declared_pack_size(store):
|
||||
"""The regression that made this guard necessary.
|
||||
|
||||
The lexicon calls tea a Beverage; the unit rulebook says beverages are
|
||||
measured in ml or litres ONLY; stage 4 therefore "corrected" 250g to 250ml
|
||||
and turned a quarter kilo of tea leaves into a quarter litre. Loose tea is
|
||||
a dry good sold by weight - the category was wrong, not the size - so a
|
||||
category that cannot hold the declared size is declined.
|
||||
"""
|
||||
_run(["Product Name"], [["Tea Powder 250g"]])
|
||||
|
||||
assert len(store) == 1, "the row was dropped or silently unit-converted"
|
||||
brand, row = store[0]
|
||||
assert brand == OWN_PRODUCTS_BRAND
|
||||
assert row["size_variants"] == ["250g"], "250g became something else"
|
||||
assert row["category"] != "Beverages"
|
||||
Reference in New Issue
Block a user