New updates on DB and JSON

This commit is contained in:
sriram
2026-09-01 13:55:15 +05:30
parent 183b65b3bd
commit 6c7a886659
20 changed files with 68753 additions and 135 deletions

View File

@@ -16,7 +16,11 @@ from pathlib import Path
import pytest
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
from app.services.brand_registry import (
BRAND_ALIASES,
get_fssai_license,
resolve_parent_brand,
)
from app.services.brand_sync import seed_catalog_paths
from app.services.vector_store import _sanitize_name
@@ -84,6 +88,13 @@ EXPECTED_SEED_TABLES = {
"parle": "brand_parle",
"pepsico": "brand_pepsico",
"tata": "brand_hindustan_unilever",
# The loose-produce base list: fruit, vegetables, greens, flowers, fish and
# loose dairy, none of which has a brand. It is the one seed catalog that
# was hand-authored rather than scraped, and it MUST land in
# brand_own_products - the same table generic_products.OWN_PRODUCTS_BRAND
# sends unbranded upload rows to, so an uploaded "Apple" deduplicates
# against the seeded one instead of creating a second row.
"Own Products": "brand_own_products",
}
@@ -112,6 +123,38 @@ def test_seed_catalogs_keep_their_current_tables() -> None:
assert actual == EXPECTED_SEED_TABLES
def test_both_haldiram_spellings_reach_one_table() -> None:
"""The regression that produced two tables for one brand.
Neither spelling was in BRAND_ALIASES, so resolve_parent_brand was the
identity for both and every upload built whichever table its sheet happened
to name - brand_haldiram (1 row) beside brand_haldirams (2). The rows were
merged into the plural, which is the correct name, so there is no longer a
singular table to assert against. This pins the PROPERTY instead, which is
what actually stops it recurring.
"""
from app.services.vector_store import _sanitize_name
for spelling in ("Haldiram", "haldiram", "HALDIRAM", "Haldiram's", "haldirams"):
table = f"brand_{_sanitize_name(resolve_parent_brand(spelling))}"
assert table == "brand_haldirams", f"{spelling!r} routed to {table}"
def test_the_haldiram_licence_survives_the_alias() -> None:
"""FSSAI_LICENSES has to be keyed on the parent, not the alias.
get_fssai_license resolves to the canonical parent before looking up, so
once "haldiram" aliases to "haldirams" a table keyed only on the singular
returns None - and a blank licence is a legitimate outcome elsewhere, so
nothing would flag it. Every future Haldiram row would simply ship without
one.
"""
assert get_fssai_license("Haldiram") == "10012011000140"
assert get_fssai_license("Haldirams") == "10012011000140"
# Not Lion Dates', which is what the stored rows wrongly carried.
assert get_fssai_license("Haldirams") != get_fssai_license("lion dates")
def test_known_sub_brands_still_route_to_their_family() -> None:
"""Word-boundary matching must not break legitimate sub-brand routing."""
assert resolve_parent_brand("Dove") == "hindustan unilever"

View File

@@ -326,3 +326,171 @@ def test_a_reupload_does_not_renumber_an_existing_products_sku(client, admin_hea
second = _run_one()
assert first["product_sku"] == second["product_sku"]
# ---------------------------------------------------------------------------
# 4. Reconciling a run against the sheet that produced it
# ---------------------------------------------------------------------------
# Four fields an integrator asked for after wiring the upload path end to end.
# Each answers a question that previously had only an approximate answer:
#
# source_row which of MY rows produced this product, and which produced none
# brand_key what key is the catalogue addressed by, given a display name
# row which rows were refused, and why
# from_drop which file in this run is mine, when a run spans several drops
#
# All four are ADDITIVE. Nothing existing changed meaning, because a consumer
# reading the old fields must keep working across this deploy.
def test_a_product_names_the_sheet_row_it_came_from(client, admin_headers, store):
"""source_row is the number the sender sees on screen, header counted as 1."""
rows = [["Amul Butter 100g", "Butter", "Amul"],
["Amul Ghee 1L", "Ghee", "Amul"]]
drop_id = client.post(UPLOAD, files=_files(("a.csv", _csv(rows=rows)))
).json()["batch_id"]
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
headers=admin_headers).json()["batch_id"]
products = batch_ingest.run_batch(run_id).files[0].result["products"]
by_name = {p["product_name"]: p["source_row"] for p in products}
# Row 1 is the header, so the first data row is 2.
assert by_name["Amul Butter 100g"] == 2
assert by_name["Amul Ghee 1L"] == 3
def test_one_sheet_row_can_own_several_products(client, admin_headers, store):
"""Pack-size explosion is many-to-one, and that is the point of the field.
"100g, 200g, 500g" in one cell is three catalog rows at three prices. The
sender needs to be able to say "row 2 of your sheet became these three"
rather than reconstruct it by matching names.
"""
headers = ["Product Name", "Category", "Brand", "Pack Size"]
rows = [["Amul Butter", "Butter", "Amul", "100g; 200g; 500g"]]
drop_id = client.post(
UPLOAD, files=_files(("a.csv", _csv(headers=headers, rows=rows)))
).json()["batch_id"]
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
headers=admin_headers).json()["batch_id"]
products = batch_ingest.run_batch(run_id).files[0].result["products"]
assert len(products) > 1, "the sheet did not explode; test proves nothing"
assert {p["source_row"] for p in products} == {2}
def test_a_product_carries_the_key_the_catalogue_is_addressed_by(client,
admin_headers,
store):
"""brand_key beside the display brand.
The manifest reports "24 Mantra" while the catalogue is keyed 24_mantra, and
an integrator normalising that themselves is guessing. Publishing the key
removes a whole class of "Unknown brand" failure.
"""
from app.services.vector_store import _sanitize_name
drop_id = _drop(client, "priya")
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
headers=admin_headers).json()["batch_id"]
product = batch_ingest.run_batch(run_id).files[0].result["products"][0]
assert product["brand_key"] == _sanitize_name(product["brand"])
assert " " not in product["brand_key"]
def test_a_run_says_which_drop_each_file_came_from(client, admin_headers):
"""from_drop is the exact answer to "which file in this run is mine".
Both drops here send a file called a.csv, which is precisely the collision
that made filename-matching unsafe: without from_drop a sender could match
the wrong merchant's file and price it as their own.
"""
first = _drop(client, "priya")
second = _drop(client, "arun")
run = client.post(
FROM_INBOX,
json={"file_ids": [f"{first}:0", f"{second}:0"]},
headers=admin_headers,
).json()
assert [f["filename"] for f in run["files"]] == ["a.csv", "a.csv"]
assert {f["from_drop"] for f in run["files"]} == {first, second}
def test_from_drop_survives_a_reread_of_the_run(client, admin_headers):
"""It is stamped after staging, so it has to reach disk, not just the reply."""
drop_id = _drop(client, "priya")
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
headers=admin_headers).json()["batch_id"]
assert batch_ingest.read_manifest(run_id).files[0].from_drop == drop_id
served = client.get(f"/api/admin/catalog-batch/batches/{run_id}",
headers=admin_headers).json()
assert served["files"][0]["from_drop"] == drop_id
def test_a_file_uploaded_straight_into_a_run_has_no_drop(client, monkeypatch):
"""Nothing to point at when the file never sat in an inbox.
None rather than the run's own id: saying a run came from itself would make
the field useless for the question it exists to answer.
"""
from app.api.routers import uploads
monkeypatch.setattr(uploads, "UPLOAD_AUTORUN", True)
run = client.post(UPLOAD, files=_files(("a.csv", _csv()))).json()
assert run["files"][0]["from_drop"] is None
def test_a_refused_row_names_itself(client, admin_headers, store, monkeypatch):
"""rejections[] carries the sheet row, not just a count.
The count alone was unusable for support: "rejected: 2" out of 19 products
left the only diagnosis being to diff the manifest against the file and
guess. A refused row is a product a shopkeeper expects on the shelf and
will not have, so it has to name itself and say why.
Everything here except `row` already shipped; this pins the whole shape
together so a future change cannot quietly drop one field of it.
"""
monkeypatch.setattr(pipeline, "ENABLE_PRODUCT_VALIDATION", True)
rows = [["Amul Butter 100g", "Butter", "Amul"],
["X", "", "Amul"]]
drop_id = client.post(UPLOAD, files=_files(("a.csv", _csv(rows=rows)))
).json()["batch_id"]
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
headers=admin_headers).json()["batch_id"]
result = batch_ingest.run_batch(run_id).files[0].result
assert result["rejected"] == len(result["rejections"]), (
"the count and the list must agree, or the list is not the explanation"
)
if result["rejections"]:
bad = result["rejections"][0]
assert set(bad) >= {"row", "product_name", "size", "reason"}
assert bad["reason"], "a rejection with no reason explains nothing"
# Row 1 is the header, so any real refusal is row 2 or later.
assert bad["row"] is None or bad["row"] >= 2
def test_the_rejection_row_matches_the_offending_sheet_line():
"""Straight at the gate, so the row number is checked without needing a
sheet that reliably fails validation end to end."""
rows = [
{"product_name": "X", "brand": "amul", "size": "100g",
"category": "Dairy", "image_id": "a", "_row": 7},
{"product_name": "", "brand": "amul", "size": "",
"category": "", "image_id": "b", "_row": 9},
]
_kept, rejected, _summary = pipeline.stage_10_validate(rows, "amul")
assert [r["_row"] for r in rejected] == [7, 9]

View File

@@ -167,10 +167,58 @@ def test_a_commodity_resolves_to_a_real_category(name, category):
assert canonical_category(name) is not None
def test_the_category_carries_an_hsn_code(store):
"""The names were chosen to match HSN_GST_TABLE keys, so tax enrichment
works without a second mapping."""
@pytest.mark.parametrize("name,hsn", [
("Toor Dhal 1kg", "0713"),
("Sugar 1kg", "1701"),
("Salt 1kg", "2501"),
# Fruit and vegetables share one category, so one code has to serve both.
# 0709 ("other vegetables, fresh or chilled") is the one carried; strictly
# a fruit is chapter 08. Both are nil-rated, so the GST is right either
# way, and this only ever surfaces in the seeded base list because an
# uploaded commodity is given no HSN at all.
("Apple", "0709"),
("Tomato", "0709"),
])
def test_a_commodity_category_is_a_key_the_hsn_table_knows(name, hsn):
"""The commodity categories line up with HSN_GST_TABLE keys.
Checked against the table directly rather than through an ingest. It used
to be asserted end to end, but the upload path no longer STAMPS an HSN onto
a commodity (see the test below), so running a sheet would now prove
nothing about the mapping. The property itself is still worth pinning: it
is what lets a category name serve as the tax key without a second lookup,
and it is how a sheet that DOES declare its tax treatment stays consistent
with ours.
"""
from app.services.enrichment.hsn_gst.models import HSN_GST_TABLE
category = canonical_category(name)
assert category in HSN_GST_TABLE, f"{category!r} has no HSN entry"
assert HSN_GST_TABLE[category][0] == hsn
def test_an_uploaded_commodity_is_given_no_hsn_code(store):
"""We do not invent a tax code the merchant did not supply.
An HSN we chose is our guess presented as their record, and a shopkeeper
bills from this. Fresh produce being nil-rated makes a wrong code cheap to
ignore and expensive to notice, which is the worst combination.
"""
_brand_of(["Toor Dhal 1kg"], store)
assert store[0][1]["hsn_code"] is None
def test_a_commodity_keeps_an_hsn_the_sheet_supplied(store):
"""The rule is "do not invent", not "discard"."""
wb = openpyxl.Workbook()
ws = wb.active
ws.append(["Item Name", "HSN Code"])
ws.append(["Toor Dhal 1kg", "0713"])
buf = io.BytesIO()
wb.save(buf)
pipeline.run_pipeline("store.xlsx", buf.getvalue(), use_llm=False, fetch_images=False)
assert store[0][1]["hsn_code"] == "0713"

View File

@@ -0,0 +1,201 @@
"""Loose produce reaches Own Products, and no real brand follows it there.
WHY THIS FILE IS THE GATE ON THE COMMODITY LEXICON
--------------------------------------------------
`is_unbranded()` works by requiring that EVERY significant token in a product
name is a known commodity or qualifier. That makes it asymmetric: adding a word
to the lexicon can only ever make the test more permissive, so the failure mode
is real brands quietly collapsing into one bucket - far harder to undo than a
staple sitting in the wrong table.
So the tests that matter most here are the negative ones. Before produce was
added, the candidate list was measured against the live catalogue and the alias
map; these encode that measurement, so the next person to add a word finds out
immediately if it swallows something it should not.
The positive tests exist because the alternative to landing in Own Products is
not "rejected" - it is being MISFILED. "Red Rose" resolved to brand "Red", which
resolve_parent_brand whole-word matched to Brooke Bond, so a flower was written
into the tea catalogue and stamped with Brooke Bond's FSSAI licence.
"""
from __future__ import annotations
import json
from pathlib import Path
import pytest
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
from app.services.generic_products import (
OWN_PRODUCTS_BRAND,
canonical_category,
is_unbranded,
)
SEED_DIR = Path(__file__).resolve().parents[1] / "data" / "seed_catalogs"
OWN_PRODUCTS_SEED = SEED_DIR / "brand_catalog_own_products.json"
# ---------------------------------------------------------------------------
# The negative tests: nothing branded may fall in here
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("alias", sorted(BRAND_ALIASES))
def test_no_brand_alias_is_read_as_a_commodity(alias: str) -> None:
"""Every alias must stay branded.
"amla" is the near miss: a real fruit that also appears inside the alias
"dabur amla". That stays branded because "dabur" is not a commodity - which
is exactly the conservatism the rule rests on: one unknown token is enough
to mean "this is a brand".
"""
assert not is_unbranded(alias)
@pytest.mark.parametrize(
"name",
[
"Amul Butter 500g",
"Aachi Sambar Powder",
"Brooke Bond Red Label 500g",
"Colgate Active Salt",
"Milky Mist Paneer",
"Nature Fresh Atta",
"Dabur Amla Hair Oil",
"Mother Dairy Milk 1L",
"24 Mantra Organic Moong Dal 500g",
],
)
def test_branded_products_stay_branded(name: str) -> None:
assert not is_unbranded(name)
def test_a_brand_whose_name_starts_with_a_number_survives() -> None:
"""The _strip_sizes regression.
The size strip used to be a number followed by "any letters", which ate the
word AFTER a number: "24 Mantra Organic Moong Dal" became "organic moong
dal", every remaining token was a commodity, and a real branded product was
filed as unbranded. Every brand beginning with a digit hit this.
"""
assert not is_unbranded("24 Mantra Organic Moong Dal 500g")
assert not is_unbranded("24 Mantra Organic Sona Masuri Rice 1kg")
# ... while the thing the strip actually exists for still works.
assert is_unbranded("Toor Dhal 1kg")
assert is_unbranded("Sugar 1kg")
assert is_unbranded("Black Pepper 100g")
# ---------------------------------------------------------------------------
# The positive tests: produce must reach Own Products
# ---------------------------------------------------------------------------
@pytest.mark.parametrize(
"name,category",
[
("Apple", "Fruits & Vegetables"),
("Orange", "Fruits & Vegetables"),
("Tomato", "Fruits & Vegetables"),
("Onion", "Fruits & Vegetables"),
("Potato", "Fruits & Vegetables"),
("Banana", "Fruits & Vegetables"),
("Drumstick", "Fruits & Vegetables"),
("Bitter Gourd", "Fruits & Vegetables"),
("Lady Finger", "Fruits & Vegetables"),
("Curry Leaves", "Fresh Herbs & Greens"),
("Mint Leaves", "Fresh Herbs & Greens"),
("Thulasi", "Fresh Herbs & Greens"),
("Jasmine", "Flowers"),
("Red Rose", "Flowers"),
("Tuna", "Fish & Seafood"),
("Prawns", "Fish & Seafood"),
("Egg", "Eggs"),
# Pantry staples that were already covered, pinned so the produce work
# cannot regress them.
("Toor Dhal 1kg", "Pulses, Grains & Spices"),
("Sugar 1kg", "Sugar & Jaggery"),
("Salt 1kg", "Salt & Staples"),
],
)
def test_loose_goods_are_unbranded_and_categorised(name: str, category: str) -> None:
assert is_unbranded(name), f"{name} would be given a junk brand"
assert canonical_category(name) == category
def test_a_flower_no_longer_lands_in_the_tea_catalogue() -> None:
"""The specific misroute this work exists to end.
"Red Rose" -> infer_brand -> "Red" -> resolve_parent_brand -> "brooke bond",
so a rose was written into brand_brooke_bond carrying Brooke Bond's real
FSSAI licence. The fix is upstream: the row never reaches infer_brand.
"""
assert is_unbranded("Red Rose")
# The hijack is still there for anything that DOES reach it, so this test
# fails loudly if the diversion is removed rather than passing for the
# wrong reason.
assert resolve_parent_brand("Red") == "brooke bond"
def test_merchant_typos_still_resolve() -> None:
"""Real strings from merchant data, misspellings included."""
for name in ("Bitter guard", "Bottle ground", "Ladies Finger"):
assert is_unbranded(name), name
def test_an_empty_brand_column_is_believed() -> None:
"""A sheet WITH a brand column that left the cell blank has said something.
This is how place-qualified produce gets in. "Salem Mango" keeps an unknown
token, so the word test alone calls it branded - deliberately, because
"Mysore" is also a real brand (Mysore Sandal). An explicit empty cell
overrides that.
"""
assert not is_unbranded("Salem Mango")
assert is_unbranded("Salem Mango", brand_column_supplied=True)
# A filled cell is believed just as much.
assert not is_unbranded("Apple", sheet_brand="Washington")
# ---------------------------------------------------------------------------
# The seeded base list
# ---------------------------------------------------------------------------
def _seed_doc():
if not OWN_PRODUCTS_SEED.exists():
pytest.skip("own-products seed catalog not present")
return json.loads(OWN_PRODUCTS_SEED.read_text(encoding="utf-8-sig"))
def test_the_seed_catalog_is_filed_under_own_products() -> None:
doc = _seed_doc()
assert doc["brand"] == OWN_PRODUCTS_BRAND
assert doc["total_products"] == len(doc["products"])
assert doc["products"], "the base list is empty"
def test_every_seeded_row_would_also_be_recognised_on_upload() -> None:
"""The round trip that makes the base list worth having.
A grocer who types "Tomato" into their own sheet must land on the SAME row
that was seeded rather than create a second one. That only holds if every
seeded name is itself classified unbranded - otherwise the uploaded copy
goes to a junk brand table and the two never meet.
"""
missed = [
p["product_name"] for p in _seed_doc()["products"]
if not is_unbranded(p["product_name"])
]
assert not missed, f"seeded rows a real upload would misfile: {missed}"
def test_seeded_image_ids_are_unique() -> None:
"""image_id is the deduplication key, and the column is UNIQUE NOT NULL."""
ids = [p["image_id"] for p in _seed_doc()["products"]]
assert len(ids) == len(set(ids))
assert all(ids)
def test_seeded_rows_carry_no_brand_pack_size_or_price() -> None:
"""Loose produce has none of those, and inventing them would be a lie."""
for p in _seed_doc()["products"]:
assert p["brand_name"] == OWN_PRODUCTS_BRAND
assert p["size_variants"] == []
assert p["price_range"] is None
assert p["fssai_license"] is None

View File

@@ -0,0 +1,311 @@
"""What actually gets written for an unbranded row: the sheet's values, and nothing else.
THE RULE
--------
A merchant sends `Apple, 500g, 155`. Those three values are what we know. FSSAI,
HSN, SKU, barcode and description are things we would be *making up*, and a
shopkeeper bills from this record - an invented tax code is our guess wearing
their letterhead.
WHY THIS NEEDED A VALIDATION CHANGE AS WELL
-------------------------------------------
Leaving those fields empty is not free. The validation gate scores a row down
for each missing field and drops it below the reject threshold:
0.55 baseline - 0.30 (no price_range) - 0.10 (no SKU) = 0.15 vs 0.35
So the literal instruction "leave these null" would have deleted every produce
row - the exact opposite of the requirement. `validate_product` now treats a
commodity's missing price and SKU as normal rather than as evidence of a
fabricated row. The first test below is the wall around that, and the branded
counterpart proves the exemption did not leak.
"""
from __future__ import annotations
import io
import pytest
from app.core import store_catalog_pipeline as pipeline
from app.services.generic_products import OWN_PRODUCTS_BRAND
from app.services.product_validator import validate_product
openpyxl = pytest.importorskip("openpyxl")
def _sheet(headers, rows) -> bytes:
wb = openpyxl.Workbook()
ws = wb.active
ws.append(headers)
for row in rows:
ws.append(row)
buf = io.BytesIO()
wb.save(buf)
return buf.getvalue()
@pytest.fixture(autouse=True)
def _isolate_sku_counter(tmp_path, monkeypatch):
from app.services import sku_service
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
@pytest.fixture
def store(monkeypatch):
"""Capture what would be written, keyed the way the real table is."""
written: list = []
def _upsert(brand, products, cleanup=False):
for product in products:
written.append((brand, dict(product)))
return len(products)
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda brand: [])
monkeypatch.setattr(pipeline, "upsert_brand_products", _upsert)
monkeypatch.setattr(pipeline, "USE_EMBEDDINGS", False)
return written
def _run(headers, rows):
pipeline.run_pipeline("store.xlsx", _sheet(headers, rows),
use_llm=False, fetch_images=False)
# ---------------------------------------------------------------------------
# The gate, which is what makes the rest of this possible
# ---------------------------------------------------------------------------
WORKED_EXAMPLE = {
"product_name": "Apple", "title": "Apple",
"category": "Fruits & Vegetables", "size": "500g",
"selling_price": 155, "final_selling_price": 155,
"price_range": "₹143-167",
"product_sku": None, "sku_source": None,
"hsn_code": None, "fssai_license": None, "description": None,
}
@pytest.mark.parametrize("images,expected", [
(["https://example.com/apple.jpg"], "verified"),
([], "needs_review"),
])
def test_the_worked_example_is_never_rejected(images, expected):
"""`Apple / 500g / 155` with everything else null must survive.
Both outcomes are KEPT: validate_catalog returns verified and needs_review
rows together, and only `rejected` is dropped. The no-image case stays
needs_review deliberately - a missing picture is the one absence here that
still says something, since image search did run and found nothing.
"""
row = dict(WORKED_EXAMPLE, image_urls=images)
report = validate_product(row, OWN_PRODUCTS_BRAND,
category_resolved_deterministically=True,
images_checked=True)
assert report.status == expected
assert report.status != "rejected"
def test_a_branded_row_with_the_same_gaps_is_still_rejected():
"""The exemption is scoped to commodities and must not leak.
Same row, same absences, under a real brand: a branded product with no
price and no SKU IS evidence that something went wrong upstream, and that
judgement is unchanged.
"""
row = dict(WORKED_EXAMPLE, image_urls=[], price_range=None)
report = validate_product(row, "amul",
category_resolved_deterministically=True,
images_checked=True)
assert report.status == "rejected"
def test_a_malformed_price_on_a_commodity_is_still_caught():
"""Absent is excused; wrong is not."""
row = dict(WORKED_EXAMPLE, image_urls=[], price_range="one fifty five")
report = validate_product(row, OWN_PRODUCTS_BRAND,
category_resolved_deterministically=True,
images_checked=True)
assert any(issue.field == "price_range" for issue in report.issues)
def test_a_blank_product_name_is_still_caught():
"""This is not a way in for junk. Everything except the two excused
absences is still enforced."""
row = dict(WORKED_EXAMPLE, product_name="", title="", image_urls=[])
report = validate_product(row, OWN_PRODUCTS_BRAND,
category_resolved_deterministically=True,
images_checked=True)
assert report.status == "rejected"
# ---------------------------------------------------------------------------
# The field policy, end to end
# ---------------------------------------------------------------------------
def test_the_sheets_values_are_kept_and_nothing_else_is_invented(store):
"""The requirement, as one assertion per field."""
_run(["Product Name", "Weight", "Selling Price"], [["Apple", "500g", 155]])
assert len(store) == 1
brand, row = store[0]
assert brand == OWN_PRODUCTS_BRAND
# What the sheet said.
assert row["product_name"] == "Apple 500g"
assert row["size_variants"] == ["500g"]
assert row["final_selling_price"] == 155
assert row["price_range"] == "₹143-167" # +/-8% of the sheet's own price
# What it did not say, and what we therefore do not claim.
for field in ("hsn_code", "product_sku", "sku_source",
"fssai_license", "barcode", "barcode_type", "description"):
assert row[field] is None, f"{field} was invented: {row[field]!r}"
def test_the_price_band_is_eight_percent_of_the_sheet_price(store):
_run(["Product Name", "Weight", "Selling Price"], [["Tomato", "1kg", 100]])
assert store[0][1]["price_range"] == "₹92-108"
def test_a_commodity_with_no_price_gets_no_band(store):
"""The market estimator is not consulted for loose produce.
It is trained on packaged FMCG and prices a 500g apple at around Rs85-105,
which is not so much wrong as meaningless - a shop prices produce by the
day. A null band is the honest answer.
"""
_run(["Product Name", "Weight"], [["Apple", "500g"]])
assert store[0][1]["price_range"] is None
def test_a_branded_row_is_untouched_by_all_of_this(store):
"""The blast radius check. A real brand still gets its full enrichment."""
_run(["Product Name", "Brand", "Category", "Weight", "Selling Price"],
[["Amul Butter", "Amul", "Dairy", "500g", 100]])
_brand, row = store[0]
assert row["fssai_license"], "a branded row lost its FSSAI licence"
assert row["product_sku"], "a branded row lost its minted SKU"
assert row["description"], "a branded row lost its description"
assert row["price_range"] == "₹92-108"
# ---------------------------------------------------------------------------
# Pack sizes
# ---------------------------------------------------------------------------
def test_a_commodity_with_no_weight_yields_exactly_one_row(store):
"""No invented 100g/250g/500g.
A shop sells apples by whatever the customer asks for, so "Apple 250g" is a
product that does not exist. This is also the mechanism behind the
catalogue drift reported by our integrator: the invented set is keyed on
the resolved category, so the same product ingested twice with the category
resolved differently produces two disjoint size sets and two sets of ids.
"""
_run(["Product Name"], [["Apple"]])
assert len(store) == 1
assert store[0][1]["size_variants"] == ["Standard"]
def test_an_uploaded_commodity_lands_on_the_seeded_row(store):
"""The dedupe that makes the base list worth having.
A grocer typing "Apple" must land on the seeded "Apple" rather than create
a second one. That holds only if both sides build the same image_id, which
means both must use "Standard" for an absent size.
"""
_run(["Product Name"], [["Apple"]])
assert store[0][1]["image_id"] == pipeline.build_image_id(
OWN_PRODUCTS_BRAND, "Apple", "Standard")
def test_a_declared_pack_size_is_still_honoured(store):
"""Not inventing sizes must not mean ignoring the ones we were given."""
_run(["Product Name", "Weight"], [["Apple", "500g"]])
assert store[0][1]["size_variants"] == ["500g"]
# ---------------------------------------------------------------------------
# "Do not invent" is not "discard"
# ---------------------------------------------------------------------------
def test_values_the_sheet_supplied_survive(store):
"""A merchant who fills in HSN, SKU and a description keeps all three."""
_run(["Product Name", "Weight", "Selling Price", "HSN Code", "SKU", "Description"],
[["Apple", "500g", 155, "0808", "SHOP-APL-1", "Shimla apples, loose"]])
row = store[0][1]
assert row["hsn_code"] == "0808"
assert row["product_sku"] == "SHOP-APL-1"
assert row["sku_source"] == "sheet"
assert row["description"] == "Shimla apples, loose"
def test_a_supplied_price_range_wins_over_the_derived_band(store):
_run(["Product Name", "Weight", "Selling Price", "Price Range"],
[["Apple", "500g", 155, "₹150-160"]])
assert store[0][1]["price_range"] == "₹150-160"
# ---------------------------------------------------------------------------
# The embedding text
# ---------------------------------------------------------------------------
def test_the_search_text_does_not_embed_the_bucket_name_or_a_null(store):
"""`search_query` feeds the vector index.
Interpolated the branded way it would read "Own Products Apple 500g
Fruits & Vegetables None" - the bucket is not a maker, and "None" is the
string repr of the description we deliberately left empty. Both would be
embedded and both would pull unrelated produce together.
"""
_run(["Product Name", "Weight"], [["Apple", "500g"]])
query = store[0][1]["search_query"]
assert "Own Products" not in query
assert "None" not in query
assert "Apple" in query and "Fruits & Vegetables" in query
# ---------------------------------------------------------------------------
# Category: the lexicon is a hint, the pack size is a fact
# ---------------------------------------------------------------------------
def test_a_commodity_gets_its_category_from_the_lexicon(store):
"""Keyword detection has no entry for individual fruit and never will.
Listing every vegetable in the curated registry would duplicate the
commodity lexicon and let the two drift, so the lexicon is consulted as a
second opinion. Without it every produce row landed in "General", which
loses the unit rules, the grouping and the validation credit.
"""
_run(["Product Name", "Weight"], [["Apple", "500g"]])
assert store[0][1]["category"] == "Fruits & Vegetables"
def test_a_lexicon_category_never_overrules_a_declared_pack_size(store):
"""The regression that made this guard necessary.
The lexicon calls tea a Beverage; the unit rulebook says beverages are
measured in ml or litres ONLY; stage 4 therefore "corrected" 250g to 250ml
and turned a quarter kilo of tea leaves into a quarter litre. Loose tea is
a dry good sold by weight - the category was wrong, not the size - so a
category that cannot hold the declared size is declined.
"""
_run(["Product Name"], [["Tea Powder 250g"]])
assert len(store) == 1, "the row was dropped or silently unit-converted"
brand, row = store[0]
assert brand == OWN_PRODUCTS_BRAND
assert row["size_variants"] == ["250g"], "250g became something else"
assert row["category"] != "Beverages"