New updates on DB and JSON
This commit is contained in:
@@ -16,7 +16,11 @@ from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
|
||||
from app.services.brand_registry import (
|
||||
BRAND_ALIASES,
|
||||
get_fssai_license,
|
||||
resolve_parent_brand,
|
||||
)
|
||||
from app.services.brand_sync import seed_catalog_paths
|
||||
from app.services.vector_store import _sanitize_name
|
||||
|
||||
@@ -84,6 +88,13 @@ EXPECTED_SEED_TABLES = {
|
||||
"parle": "brand_parle",
|
||||
"pepsico": "brand_pepsico",
|
||||
"tata": "brand_hindustan_unilever",
|
||||
# The loose-produce base list: fruit, vegetables, greens, flowers, fish and
|
||||
# loose dairy, none of which has a brand. It is the one seed catalog that
|
||||
# was hand-authored rather than scraped, and it MUST land in
|
||||
# brand_own_products - the same table generic_products.OWN_PRODUCTS_BRAND
|
||||
# sends unbranded upload rows to, so an uploaded "Apple" deduplicates
|
||||
# against the seeded one instead of creating a second row.
|
||||
"Own Products": "brand_own_products",
|
||||
}
|
||||
|
||||
|
||||
@@ -112,6 +123,38 @@ def test_seed_catalogs_keep_their_current_tables() -> None:
|
||||
assert actual == EXPECTED_SEED_TABLES
|
||||
|
||||
|
||||
def test_both_haldiram_spellings_reach_one_table() -> None:
|
||||
"""The regression that produced two tables for one brand.
|
||||
|
||||
Neither spelling was in BRAND_ALIASES, so resolve_parent_brand was the
|
||||
identity for both and every upload built whichever table its sheet happened
|
||||
to name - brand_haldiram (1 row) beside brand_haldirams (2). The rows were
|
||||
merged into the plural, which is the correct name, so there is no longer a
|
||||
singular table to assert against. This pins the PROPERTY instead, which is
|
||||
what actually stops it recurring.
|
||||
"""
|
||||
from app.services.vector_store import _sanitize_name
|
||||
|
||||
for spelling in ("Haldiram", "haldiram", "HALDIRAM", "Haldiram's", "haldirams"):
|
||||
table = f"brand_{_sanitize_name(resolve_parent_brand(spelling))}"
|
||||
assert table == "brand_haldirams", f"{spelling!r} routed to {table}"
|
||||
|
||||
|
||||
def test_the_haldiram_licence_survives_the_alias() -> None:
|
||||
"""FSSAI_LICENSES has to be keyed on the parent, not the alias.
|
||||
|
||||
get_fssai_license resolves to the canonical parent before looking up, so
|
||||
once "haldiram" aliases to "haldirams" a table keyed only on the singular
|
||||
returns None - and a blank licence is a legitimate outcome elsewhere, so
|
||||
nothing would flag it. Every future Haldiram row would simply ship without
|
||||
one.
|
||||
"""
|
||||
assert get_fssai_license("Haldiram") == "10012011000140"
|
||||
assert get_fssai_license("Haldirams") == "10012011000140"
|
||||
# Not Lion Dates', which is what the stored rows wrongly carried.
|
||||
assert get_fssai_license("Haldirams") != get_fssai_license("lion dates")
|
||||
|
||||
|
||||
def test_known_sub_brands_still_route_to_their_family() -> None:
|
||||
"""Word-boundary matching must not break legitimate sub-brand routing."""
|
||||
assert resolve_parent_brand("Dove") == "hindustan unilever"
|
||||
|
||||
@@ -326,3 +326,171 @@ def test_a_reupload_does_not_renumber_an_existing_products_sku(client, admin_hea
|
||||
second = _run_one()
|
||||
|
||||
assert first["product_sku"] == second["product_sku"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4. Reconciling a run against the sheet that produced it
|
||||
# ---------------------------------------------------------------------------
|
||||
# Four fields an integrator asked for after wiring the upload path end to end.
|
||||
# Each answers a question that previously had only an approximate answer:
|
||||
#
|
||||
# source_row which of MY rows produced this product, and which produced none
|
||||
# brand_key what key is the catalogue addressed by, given a display name
|
||||
# row which rows were refused, and why
|
||||
# from_drop which file in this run is mine, when a run spans several drops
|
||||
#
|
||||
# All four are ADDITIVE. Nothing existing changed meaning, because a consumer
|
||||
# reading the old fields must keep working across this deploy.
|
||||
|
||||
|
||||
def test_a_product_names_the_sheet_row_it_came_from(client, admin_headers, store):
|
||||
"""source_row is the number the sender sees on screen, header counted as 1."""
|
||||
rows = [["Amul Butter 100g", "Butter", "Amul"],
|
||||
["Amul Ghee 1L", "Ghee", "Amul"]]
|
||||
drop_id = client.post(UPLOAD, files=_files(("a.csv", _csv(rows=rows)))
|
||||
).json()["batch_id"]
|
||||
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
|
||||
headers=admin_headers).json()["batch_id"]
|
||||
|
||||
products = batch_ingest.run_batch(run_id).files[0].result["products"]
|
||||
|
||||
by_name = {p["product_name"]: p["source_row"] for p in products}
|
||||
# Row 1 is the header, so the first data row is 2.
|
||||
assert by_name["Amul Butter 100g"] == 2
|
||||
assert by_name["Amul Ghee 1L"] == 3
|
||||
|
||||
|
||||
def test_one_sheet_row_can_own_several_products(client, admin_headers, store):
|
||||
"""Pack-size explosion is many-to-one, and that is the point of the field.
|
||||
|
||||
"100g, 200g, 500g" in one cell is three catalog rows at three prices. The
|
||||
sender needs to be able to say "row 2 of your sheet became these three"
|
||||
rather than reconstruct it by matching names.
|
||||
"""
|
||||
headers = ["Product Name", "Category", "Brand", "Pack Size"]
|
||||
rows = [["Amul Butter", "Butter", "Amul", "100g; 200g; 500g"]]
|
||||
drop_id = client.post(
|
||||
UPLOAD, files=_files(("a.csv", _csv(headers=headers, rows=rows)))
|
||||
).json()["batch_id"]
|
||||
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
|
||||
headers=admin_headers).json()["batch_id"]
|
||||
|
||||
products = batch_ingest.run_batch(run_id).files[0].result["products"]
|
||||
|
||||
assert len(products) > 1, "the sheet did not explode; test proves nothing"
|
||||
assert {p["source_row"] for p in products} == {2}
|
||||
|
||||
|
||||
def test_a_product_carries_the_key_the_catalogue_is_addressed_by(client,
|
||||
admin_headers,
|
||||
store):
|
||||
"""brand_key beside the display brand.
|
||||
|
||||
The manifest reports "24 Mantra" while the catalogue is keyed 24_mantra, and
|
||||
an integrator normalising that themselves is guessing. Publishing the key
|
||||
removes a whole class of "Unknown brand" failure.
|
||||
"""
|
||||
from app.services.vector_store import _sanitize_name
|
||||
|
||||
drop_id = _drop(client, "priya")
|
||||
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
|
||||
headers=admin_headers).json()["batch_id"]
|
||||
|
||||
product = batch_ingest.run_batch(run_id).files[0].result["products"][0]
|
||||
|
||||
assert product["brand_key"] == _sanitize_name(product["brand"])
|
||||
assert " " not in product["brand_key"]
|
||||
|
||||
|
||||
def test_a_run_says_which_drop_each_file_came_from(client, admin_headers):
|
||||
"""from_drop is the exact answer to "which file in this run is mine".
|
||||
|
||||
Both drops here send a file called a.csv, which is precisely the collision
|
||||
that made filename-matching unsafe: without from_drop a sender could match
|
||||
the wrong merchant's file and price it as their own.
|
||||
"""
|
||||
first = _drop(client, "priya")
|
||||
second = _drop(client, "arun")
|
||||
|
||||
run = client.post(
|
||||
FROM_INBOX,
|
||||
json={"file_ids": [f"{first}:0", f"{second}:0"]},
|
||||
headers=admin_headers,
|
||||
).json()
|
||||
|
||||
assert [f["filename"] for f in run["files"]] == ["a.csv", "a.csv"]
|
||||
assert {f["from_drop"] for f in run["files"]} == {first, second}
|
||||
|
||||
|
||||
def test_from_drop_survives_a_reread_of_the_run(client, admin_headers):
|
||||
"""It is stamped after staging, so it has to reach disk, not just the reply."""
|
||||
drop_id = _drop(client, "priya")
|
||||
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
|
||||
headers=admin_headers).json()["batch_id"]
|
||||
|
||||
assert batch_ingest.read_manifest(run_id).files[0].from_drop == drop_id
|
||||
|
||||
served = client.get(f"/api/admin/catalog-batch/batches/{run_id}",
|
||||
headers=admin_headers).json()
|
||||
assert served["files"][0]["from_drop"] == drop_id
|
||||
|
||||
|
||||
def test_a_file_uploaded_straight_into_a_run_has_no_drop(client, monkeypatch):
|
||||
"""Nothing to point at when the file never sat in an inbox.
|
||||
|
||||
None rather than the run's own id: saying a run came from itself would make
|
||||
the field useless for the question it exists to answer.
|
||||
"""
|
||||
from app.api.routers import uploads
|
||||
monkeypatch.setattr(uploads, "UPLOAD_AUTORUN", True)
|
||||
|
||||
run = client.post(UPLOAD, files=_files(("a.csv", _csv()))).json()
|
||||
|
||||
assert run["files"][0]["from_drop"] is None
|
||||
|
||||
|
||||
def test_a_refused_row_names_itself(client, admin_headers, store, monkeypatch):
|
||||
"""rejections[] carries the sheet row, not just a count.
|
||||
|
||||
The count alone was unusable for support: "rejected: 2" out of 19 products
|
||||
left the only diagnosis being to diff the manifest against the file and
|
||||
guess. A refused row is a product a shopkeeper expects on the shelf and
|
||||
will not have, so it has to name itself and say why.
|
||||
|
||||
Everything here except `row` already shipped; this pins the whole shape
|
||||
together so a future change cannot quietly drop one field of it.
|
||||
"""
|
||||
monkeypatch.setattr(pipeline, "ENABLE_PRODUCT_VALIDATION", True)
|
||||
rows = [["Amul Butter 100g", "Butter", "Amul"],
|
||||
["X", "", "Amul"]]
|
||||
drop_id = client.post(UPLOAD, files=_files(("a.csv", _csv(rows=rows)))
|
||||
).json()["batch_id"]
|
||||
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
|
||||
headers=admin_headers).json()["batch_id"]
|
||||
|
||||
result = batch_ingest.run_batch(run_id).files[0].result
|
||||
|
||||
assert result["rejected"] == len(result["rejections"]), (
|
||||
"the count and the list must agree, or the list is not the explanation"
|
||||
)
|
||||
if result["rejections"]:
|
||||
bad = result["rejections"][0]
|
||||
assert set(bad) >= {"row", "product_name", "size", "reason"}
|
||||
assert bad["reason"], "a rejection with no reason explains nothing"
|
||||
# Row 1 is the header, so any real refusal is row 2 or later.
|
||||
assert bad["row"] is None or bad["row"] >= 2
|
||||
|
||||
|
||||
def test_the_rejection_row_matches_the_offending_sheet_line():
|
||||
"""Straight at the gate, so the row number is checked without needing a
|
||||
sheet that reliably fails validation end to end."""
|
||||
rows = [
|
||||
{"product_name": "X", "brand": "amul", "size": "100g",
|
||||
"category": "Dairy", "image_id": "a", "_row": 7},
|
||||
{"product_name": "", "brand": "amul", "size": "",
|
||||
"category": "", "image_id": "b", "_row": 9},
|
||||
]
|
||||
|
||||
_kept, rejected, _summary = pipeline.stage_10_validate(rows, "amul")
|
||||
|
||||
assert [r["_row"] for r in rejected] == [7, 9]
|
||||
|
||||
@@ -167,10 +167,58 @@ def test_a_commodity_resolves_to_a_real_category(name, category):
|
||||
assert canonical_category(name) is not None
|
||||
|
||||
|
||||
def test_the_category_carries_an_hsn_code(store):
|
||||
"""The names were chosen to match HSN_GST_TABLE keys, so tax enrichment
|
||||
works without a second mapping."""
|
||||
@pytest.mark.parametrize("name,hsn", [
|
||||
("Toor Dhal 1kg", "0713"),
|
||||
("Sugar 1kg", "1701"),
|
||||
("Salt 1kg", "2501"),
|
||||
# Fruit and vegetables share one category, so one code has to serve both.
|
||||
# 0709 ("other vegetables, fresh or chilled") is the one carried; strictly
|
||||
# a fruit is chapter 08. Both are nil-rated, so the GST is right either
|
||||
# way, and this only ever surfaces in the seeded base list because an
|
||||
# uploaded commodity is given no HSN at all.
|
||||
("Apple", "0709"),
|
||||
("Tomato", "0709"),
|
||||
])
|
||||
def test_a_commodity_category_is_a_key_the_hsn_table_knows(name, hsn):
|
||||
"""The commodity categories line up with HSN_GST_TABLE keys.
|
||||
|
||||
Checked against the table directly rather than through an ingest. It used
|
||||
to be asserted end to end, but the upload path no longer STAMPS an HSN onto
|
||||
a commodity (see the test below), so running a sheet would now prove
|
||||
nothing about the mapping. The property itself is still worth pinning: it
|
||||
is what lets a category name serve as the tax key without a second lookup,
|
||||
and it is how a sheet that DOES declare its tax treatment stays consistent
|
||||
with ours.
|
||||
"""
|
||||
from app.services.enrichment.hsn_gst.models import HSN_GST_TABLE
|
||||
|
||||
category = canonical_category(name)
|
||||
assert category in HSN_GST_TABLE, f"{category!r} has no HSN entry"
|
||||
assert HSN_GST_TABLE[category][0] == hsn
|
||||
|
||||
|
||||
def test_an_uploaded_commodity_is_given_no_hsn_code(store):
|
||||
"""We do not invent a tax code the merchant did not supply.
|
||||
|
||||
An HSN we chose is our guess presented as their record, and a shopkeeper
|
||||
bills from this. Fresh produce being nil-rated makes a wrong code cheap to
|
||||
ignore and expensive to notice, which is the worst combination.
|
||||
"""
|
||||
_brand_of(["Toor Dhal 1kg"], store)
|
||||
assert store[0][1]["hsn_code"] is None
|
||||
|
||||
|
||||
def test_a_commodity_keeps_an_hsn_the_sheet_supplied(store):
|
||||
"""The rule is "do not invent", not "discard"."""
|
||||
wb = openpyxl.Workbook()
|
||||
ws = wb.active
|
||||
ws.append(["Item Name", "HSN Code"])
|
||||
ws.append(["Toor Dhal 1kg", "0713"])
|
||||
buf = io.BytesIO()
|
||||
wb.save(buf)
|
||||
|
||||
pipeline.run_pipeline("store.xlsx", buf.getvalue(), use_llm=False, fetch_images=False)
|
||||
|
||||
assert store[0][1]["hsn_code"] == "0713"
|
||||
|
||||
|
||||
|
||||
201
tests/test_generic_products_produce.py
Normal file
201
tests/test_generic_products_produce.py
Normal file
@@ -0,0 +1,201 @@
|
||||
"""Loose produce reaches Own Products, and no real brand follows it there.
|
||||
|
||||
WHY THIS FILE IS THE GATE ON THE COMMODITY LEXICON
|
||||
--------------------------------------------------
|
||||
`is_unbranded()` works by requiring that EVERY significant token in a product
|
||||
name is a known commodity or qualifier. That makes it asymmetric: adding a word
|
||||
to the lexicon can only ever make the test more permissive, so the failure mode
|
||||
is real brands quietly collapsing into one bucket - far harder to undo than a
|
||||
staple sitting in the wrong table.
|
||||
|
||||
So the tests that matter most here are the negative ones. Before produce was
|
||||
added, the candidate list was measured against the live catalogue and the alias
|
||||
map; these encode that measurement, so the next person to add a word finds out
|
||||
immediately if it swallows something it should not.
|
||||
|
||||
The positive tests exist because the alternative to landing in Own Products is
|
||||
not "rejected" - it is being MISFILED. "Red Rose" resolved to brand "Red", which
|
||||
resolve_parent_brand whole-word matched to Brooke Bond, so a flower was written
|
||||
into the tea catalogue and stamped with Brooke Bond's FSSAI licence.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
|
||||
from app.services.generic_products import (
|
||||
OWN_PRODUCTS_BRAND,
|
||||
canonical_category,
|
||||
is_unbranded,
|
||||
)
|
||||
|
||||
SEED_DIR = Path(__file__).resolve().parents[1] / "data" / "seed_catalogs"
|
||||
OWN_PRODUCTS_SEED = SEED_DIR / "brand_catalog_own_products.json"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The negative tests: nothing branded may fall in here
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.mark.parametrize("alias", sorted(BRAND_ALIASES))
|
||||
def test_no_brand_alias_is_read_as_a_commodity(alias: str) -> None:
|
||||
"""Every alias must stay branded.
|
||||
|
||||
"amla" is the near miss: a real fruit that also appears inside the alias
|
||||
"dabur amla". That stays branded because "dabur" is not a commodity - which
|
||||
is exactly the conservatism the rule rests on: one unknown token is enough
|
||||
to mean "this is a brand".
|
||||
"""
|
||||
assert not is_unbranded(alias)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"name",
|
||||
[
|
||||
"Amul Butter 500g",
|
||||
"Aachi Sambar Powder",
|
||||
"Brooke Bond Red Label 500g",
|
||||
"Colgate Active Salt",
|
||||
"Milky Mist Paneer",
|
||||
"Nature Fresh Atta",
|
||||
"Dabur Amla Hair Oil",
|
||||
"Mother Dairy Milk 1L",
|
||||
"24 Mantra Organic Moong Dal 500g",
|
||||
],
|
||||
)
|
||||
def test_branded_products_stay_branded(name: str) -> None:
|
||||
assert not is_unbranded(name)
|
||||
|
||||
|
||||
def test_a_brand_whose_name_starts_with_a_number_survives() -> None:
|
||||
"""The _strip_sizes regression.
|
||||
|
||||
The size strip used to be a number followed by "any letters", which ate the
|
||||
word AFTER a number: "24 Mantra Organic Moong Dal" became "organic moong
|
||||
dal", every remaining token was a commodity, and a real branded product was
|
||||
filed as unbranded. Every brand beginning with a digit hit this.
|
||||
"""
|
||||
assert not is_unbranded("24 Mantra Organic Moong Dal 500g")
|
||||
assert not is_unbranded("24 Mantra Organic Sona Masuri Rice 1kg")
|
||||
# ... while the thing the strip actually exists for still works.
|
||||
assert is_unbranded("Toor Dhal 1kg")
|
||||
assert is_unbranded("Sugar 1kg")
|
||||
assert is_unbranded("Black Pepper 100g")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The positive tests: produce must reach Own Products
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.mark.parametrize(
|
||||
"name,category",
|
||||
[
|
||||
("Apple", "Fruits & Vegetables"),
|
||||
("Orange", "Fruits & Vegetables"),
|
||||
("Tomato", "Fruits & Vegetables"),
|
||||
("Onion", "Fruits & Vegetables"),
|
||||
("Potato", "Fruits & Vegetables"),
|
||||
("Banana", "Fruits & Vegetables"),
|
||||
("Drumstick", "Fruits & Vegetables"),
|
||||
("Bitter Gourd", "Fruits & Vegetables"),
|
||||
("Lady Finger", "Fruits & Vegetables"),
|
||||
("Curry Leaves", "Fresh Herbs & Greens"),
|
||||
("Mint Leaves", "Fresh Herbs & Greens"),
|
||||
("Thulasi", "Fresh Herbs & Greens"),
|
||||
("Jasmine", "Flowers"),
|
||||
("Red Rose", "Flowers"),
|
||||
("Tuna", "Fish & Seafood"),
|
||||
("Prawns", "Fish & Seafood"),
|
||||
("Egg", "Eggs"),
|
||||
# Pantry staples that were already covered, pinned so the produce work
|
||||
# cannot regress them.
|
||||
("Toor Dhal 1kg", "Pulses, Grains & Spices"),
|
||||
("Sugar 1kg", "Sugar & Jaggery"),
|
||||
("Salt 1kg", "Salt & Staples"),
|
||||
],
|
||||
)
|
||||
def test_loose_goods_are_unbranded_and_categorised(name: str, category: str) -> None:
|
||||
assert is_unbranded(name), f"{name} would be given a junk brand"
|
||||
assert canonical_category(name) == category
|
||||
|
||||
|
||||
def test_a_flower_no_longer_lands_in_the_tea_catalogue() -> None:
|
||||
"""The specific misroute this work exists to end.
|
||||
|
||||
"Red Rose" -> infer_brand -> "Red" -> resolve_parent_brand -> "brooke bond",
|
||||
so a rose was written into brand_brooke_bond carrying Brooke Bond's real
|
||||
FSSAI licence. The fix is upstream: the row never reaches infer_brand.
|
||||
"""
|
||||
assert is_unbranded("Red Rose")
|
||||
# The hijack is still there for anything that DOES reach it, so this test
|
||||
# fails loudly if the diversion is removed rather than passing for the
|
||||
# wrong reason.
|
||||
assert resolve_parent_brand("Red") == "brooke bond"
|
||||
|
||||
|
||||
def test_merchant_typos_still_resolve() -> None:
|
||||
"""Real strings from merchant data, misspellings included."""
|
||||
for name in ("Bitter guard", "Bottle ground", "Ladies Finger"):
|
||||
assert is_unbranded(name), name
|
||||
|
||||
|
||||
def test_an_empty_brand_column_is_believed() -> None:
|
||||
"""A sheet WITH a brand column that left the cell blank has said something.
|
||||
|
||||
This is how place-qualified produce gets in. "Salem Mango" keeps an unknown
|
||||
token, so the word test alone calls it branded - deliberately, because
|
||||
"Mysore" is also a real brand (Mysore Sandal). An explicit empty cell
|
||||
overrides that.
|
||||
"""
|
||||
assert not is_unbranded("Salem Mango")
|
||||
assert is_unbranded("Salem Mango", brand_column_supplied=True)
|
||||
# A filled cell is believed just as much.
|
||||
assert not is_unbranded("Apple", sheet_brand="Washington")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The seeded base list
|
||||
# ---------------------------------------------------------------------------
|
||||
def _seed_doc():
|
||||
if not OWN_PRODUCTS_SEED.exists():
|
||||
pytest.skip("own-products seed catalog not present")
|
||||
return json.loads(OWN_PRODUCTS_SEED.read_text(encoding="utf-8-sig"))
|
||||
|
||||
|
||||
def test_the_seed_catalog_is_filed_under_own_products() -> None:
|
||||
doc = _seed_doc()
|
||||
assert doc["brand"] == OWN_PRODUCTS_BRAND
|
||||
assert doc["total_products"] == len(doc["products"])
|
||||
assert doc["products"], "the base list is empty"
|
||||
|
||||
|
||||
def test_every_seeded_row_would_also_be_recognised_on_upload() -> None:
|
||||
"""The round trip that makes the base list worth having.
|
||||
|
||||
A grocer who types "Tomato" into their own sheet must land on the SAME row
|
||||
that was seeded rather than create a second one. That only holds if every
|
||||
seeded name is itself classified unbranded - otherwise the uploaded copy
|
||||
goes to a junk brand table and the two never meet.
|
||||
"""
|
||||
missed = [
|
||||
p["product_name"] for p in _seed_doc()["products"]
|
||||
if not is_unbranded(p["product_name"])
|
||||
]
|
||||
assert not missed, f"seeded rows a real upload would misfile: {missed}"
|
||||
|
||||
|
||||
def test_seeded_image_ids_are_unique() -> None:
|
||||
"""image_id is the deduplication key, and the column is UNIQUE NOT NULL."""
|
||||
ids = [p["image_id"] for p in _seed_doc()["products"]]
|
||||
assert len(ids) == len(set(ids))
|
||||
assert all(ids)
|
||||
|
||||
|
||||
def test_seeded_rows_carry_no_brand_pack_size_or_price() -> None:
|
||||
"""Loose produce has none of those, and inventing them would be a lie."""
|
||||
for p in _seed_doc()["products"]:
|
||||
assert p["brand_name"] == OWN_PRODUCTS_BRAND
|
||||
assert p["size_variants"] == []
|
||||
assert p["price_range"] is None
|
||||
assert p["fssai_license"] is None
|
||||
311
tests/test_own_products_fields.py
Normal file
311
tests/test_own_products_fields.py
Normal file
@@ -0,0 +1,311 @@
|
||||
"""What actually gets written for an unbranded row: the sheet's values, and nothing else.
|
||||
|
||||
THE RULE
|
||||
--------
|
||||
A merchant sends `Apple, 500g, 155`. Those three values are what we know. FSSAI,
|
||||
HSN, SKU, barcode and description are things we would be *making up*, and a
|
||||
shopkeeper bills from this record - an invented tax code is our guess wearing
|
||||
their letterhead.
|
||||
|
||||
WHY THIS NEEDED A VALIDATION CHANGE AS WELL
|
||||
-------------------------------------------
|
||||
Leaving those fields empty is not free. The validation gate scores a row down
|
||||
for each missing field and drops it below the reject threshold:
|
||||
|
||||
0.55 baseline - 0.30 (no price_range) - 0.10 (no SKU) = 0.15 vs 0.35
|
||||
|
||||
So the literal instruction "leave these null" would have deleted every produce
|
||||
row - the exact opposite of the requirement. `validate_product` now treats a
|
||||
commodity's missing price and SKU as normal rather than as evidence of a
|
||||
fabricated row. The first test below is the wall around that, and the branded
|
||||
counterpart proves the exemption did not leak.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
|
||||
import pytest
|
||||
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.services.generic_products import OWN_PRODUCTS_BRAND
|
||||
from app.services.product_validator import validate_product
|
||||
|
||||
openpyxl = pytest.importorskip("openpyxl")
|
||||
|
||||
|
||||
def _sheet(headers, rows) -> bytes:
|
||||
wb = openpyxl.Workbook()
|
||||
ws = wb.active
|
||||
ws.append(headers)
|
||||
for row in rows:
|
||||
ws.append(row)
|
||||
buf = io.BytesIO()
|
||||
wb.save(buf)
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolate_sku_counter(tmp_path, monkeypatch):
|
||||
from app.services import sku_service
|
||||
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store(monkeypatch):
|
||||
"""Capture what would be written, keyed the way the real table is."""
|
||||
written: list = []
|
||||
|
||||
def _upsert(brand, products, cleanup=False):
|
||||
for product in products:
|
||||
written.append((brand, dict(product)))
|
||||
return len(products)
|
||||
|
||||
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda brand: [])
|
||||
monkeypatch.setattr(pipeline, "upsert_brand_products", _upsert)
|
||||
monkeypatch.setattr(pipeline, "USE_EMBEDDINGS", False)
|
||||
return written
|
||||
|
||||
|
||||
def _run(headers, rows):
|
||||
pipeline.run_pipeline("store.xlsx", _sheet(headers, rows),
|
||||
use_llm=False, fetch_images=False)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The gate, which is what makes the rest of this possible
|
||||
# ---------------------------------------------------------------------------
|
||||
WORKED_EXAMPLE = {
|
||||
"product_name": "Apple", "title": "Apple",
|
||||
"category": "Fruits & Vegetables", "size": "500g",
|
||||
"selling_price": 155, "final_selling_price": 155,
|
||||
"price_range": "₹143-167",
|
||||
"product_sku": None, "sku_source": None,
|
||||
"hsn_code": None, "fssai_license": None, "description": None,
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("images,expected", [
|
||||
(["https://example.com/apple.jpg"], "verified"),
|
||||
([], "needs_review"),
|
||||
])
|
||||
def test_the_worked_example_is_never_rejected(images, expected):
|
||||
"""`Apple / 500g / 155` with everything else null must survive.
|
||||
|
||||
Both outcomes are KEPT: validate_catalog returns verified and needs_review
|
||||
rows together, and only `rejected` is dropped. The no-image case stays
|
||||
needs_review deliberately - a missing picture is the one absence here that
|
||||
still says something, since image search did run and found nothing.
|
||||
"""
|
||||
row = dict(WORKED_EXAMPLE, image_urls=images)
|
||||
|
||||
report = validate_product(row, OWN_PRODUCTS_BRAND,
|
||||
category_resolved_deterministically=True,
|
||||
images_checked=True)
|
||||
|
||||
assert report.status == expected
|
||||
assert report.status != "rejected"
|
||||
|
||||
|
||||
def test_a_branded_row_with_the_same_gaps_is_still_rejected():
|
||||
"""The exemption is scoped to commodities and must not leak.
|
||||
|
||||
Same row, same absences, under a real brand: a branded product with no
|
||||
price and no SKU IS evidence that something went wrong upstream, and that
|
||||
judgement is unchanged.
|
||||
"""
|
||||
row = dict(WORKED_EXAMPLE, image_urls=[], price_range=None)
|
||||
|
||||
report = validate_product(row, "amul",
|
||||
category_resolved_deterministically=True,
|
||||
images_checked=True)
|
||||
|
||||
assert report.status == "rejected"
|
||||
|
||||
|
||||
def test_a_malformed_price_on_a_commodity_is_still_caught():
|
||||
"""Absent is excused; wrong is not."""
|
||||
row = dict(WORKED_EXAMPLE, image_urls=[], price_range="one fifty five")
|
||||
|
||||
report = validate_product(row, OWN_PRODUCTS_BRAND,
|
||||
category_resolved_deterministically=True,
|
||||
images_checked=True)
|
||||
|
||||
assert any(issue.field == "price_range" for issue in report.issues)
|
||||
|
||||
|
||||
def test_a_blank_product_name_is_still_caught():
|
||||
"""This is not a way in for junk. Everything except the two excused
|
||||
absences is still enforced."""
|
||||
row = dict(WORKED_EXAMPLE, product_name="", title="", image_urls=[])
|
||||
|
||||
report = validate_product(row, OWN_PRODUCTS_BRAND,
|
||||
category_resolved_deterministically=True,
|
||||
images_checked=True)
|
||||
|
||||
assert report.status == "rejected"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The field policy, end to end
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_the_sheets_values_are_kept_and_nothing_else_is_invented(store):
|
||||
"""The requirement, as one assertion per field."""
|
||||
_run(["Product Name", "Weight", "Selling Price"], [["Apple", "500g", 155]])
|
||||
|
||||
assert len(store) == 1
|
||||
brand, row = store[0]
|
||||
assert brand == OWN_PRODUCTS_BRAND
|
||||
|
||||
# What the sheet said.
|
||||
assert row["product_name"] == "Apple 500g"
|
||||
assert row["size_variants"] == ["500g"]
|
||||
assert row["final_selling_price"] == 155
|
||||
assert row["price_range"] == "₹143-167" # +/-8% of the sheet's own price
|
||||
|
||||
# What it did not say, and what we therefore do not claim.
|
||||
for field in ("hsn_code", "product_sku", "sku_source",
|
||||
"fssai_license", "barcode", "barcode_type", "description"):
|
||||
assert row[field] is None, f"{field} was invented: {row[field]!r}"
|
||||
|
||||
|
||||
def test_the_price_band_is_eight_percent_of_the_sheet_price(store):
|
||||
_run(["Product Name", "Weight", "Selling Price"], [["Tomato", "1kg", 100]])
|
||||
|
||||
assert store[0][1]["price_range"] == "₹92-108"
|
||||
|
||||
|
||||
def test_a_commodity_with_no_price_gets_no_band(store):
|
||||
"""The market estimator is not consulted for loose produce.
|
||||
|
||||
It is trained on packaged FMCG and prices a 500g apple at around Rs85-105,
|
||||
which is not so much wrong as meaningless - a shop prices produce by the
|
||||
day. A null band is the honest answer.
|
||||
"""
|
||||
_run(["Product Name", "Weight"], [["Apple", "500g"]])
|
||||
|
||||
assert store[0][1]["price_range"] is None
|
||||
|
||||
|
||||
def test_a_branded_row_is_untouched_by_all_of_this(store):
|
||||
"""The blast radius check. A real brand still gets its full enrichment."""
|
||||
_run(["Product Name", "Brand", "Category", "Weight", "Selling Price"],
|
||||
[["Amul Butter", "Amul", "Dairy", "500g", 100]])
|
||||
|
||||
_brand, row = store[0]
|
||||
assert row["fssai_license"], "a branded row lost its FSSAI licence"
|
||||
assert row["product_sku"], "a branded row lost its minted SKU"
|
||||
assert row["description"], "a branded row lost its description"
|
||||
assert row["price_range"] == "₹92-108"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Pack sizes
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_commodity_with_no_weight_yields_exactly_one_row(store):
|
||||
"""No invented 100g/250g/500g.
|
||||
|
||||
A shop sells apples by whatever the customer asks for, so "Apple 250g" is a
|
||||
product that does not exist. This is also the mechanism behind the
|
||||
catalogue drift reported by our integrator: the invented set is keyed on
|
||||
the resolved category, so the same product ingested twice with the category
|
||||
resolved differently produces two disjoint size sets and two sets of ids.
|
||||
"""
|
||||
_run(["Product Name"], [["Apple"]])
|
||||
|
||||
assert len(store) == 1
|
||||
assert store[0][1]["size_variants"] == ["Standard"]
|
||||
|
||||
|
||||
def test_an_uploaded_commodity_lands_on_the_seeded_row(store):
|
||||
"""The dedupe that makes the base list worth having.
|
||||
|
||||
A grocer typing "Apple" must land on the seeded "Apple" rather than create
|
||||
a second one. That holds only if both sides build the same image_id, which
|
||||
means both must use "Standard" for an absent size.
|
||||
"""
|
||||
_run(["Product Name"], [["Apple"]])
|
||||
|
||||
assert store[0][1]["image_id"] == pipeline.build_image_id(
|
||||
OWN_PRODUCTS_BRAND, "Apple", "Standard")
|
||||
|
||||
|
||||
def test_a_declared_pack_size_is_still_honoured(store):
|
||||
"""Not inventing sizes must not mean ignoring the ones we were given."""
|
||||
_run(["Product Name", "Weight"], [["Apple", "500g"]])
|
||||
|
||||
assert store[0][1]["size_variants"] == ["500g"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# "Do not invent" is not "discard"
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_values_the_sheet_supplied_survive(store):
|
||||
"""A merchant who fills in HSN, SKU and a description keeps all three."""
|
||||
_run(["Product Name", "Weight", "Selling Price", "HSN Code", "SKU", "Description"],
|
||||
[["Apple", "500g", 155, "0808", "SHOP-APL-1", "Shimla apples, loose"]])
|
||||
|
||||
row = store[0][1]
|
||||
assert row["hsn_code"] == "0808"
|
||||
assert row["product_sku"] == "SHOP-APL-1"
|
||||
assert row["sku_source"] == "sheet"
|
||||
assert row["description"] == "Shimla apples, loose"
|
||||
|
||||
|
||||
def test_a_supplied_price_range_wins_over_the_derived_band(store):
|
||||
_run(["Product Name", "Weight", "Selling Price", "Price Range"],
|
||||
[["Apple", "500g", 155, "₹150-160"]])
|
||||
|
||||
assert store[0][1]["price_range"] == "₹150-160"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The embedding text
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_the_search_text_does_not_embed_the_bucket_name_or_a_null(store):
|
||||
"""`search_query` feeds the vector index.
|
||||
|
||||
Interpolated the branded way it would read "Own Products Apple 500g
|
||||
Fruits & Vegetables None" - the bucket is not a maker, and "None" is the
|
||||
string repr of the description we deliberately left empty. Both would be
|
||||
embedded and both would pull unrelated produce together.
|
||||
"""
|
||||
_run(["Product Name", "Weight"], [["Apple", "500g"]])
|
||||
|
||||
query = store[0][1]["search_query"]
|
||||
assert "Own Products" not in query
|
||||
assert "None" not in query
|
||||
assert "Apple" in query and "Fruits & Vegetables" in query
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Category: the lexicon is a hint, the pack size is a fact
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_commodity_gets_its_category_from_the_lexicon(store):
|
||||
"""Keyword detection has no entry for individual fruit and never will.
|
||||
|
||||
Listing every vegetable in the curated registry would duplicate the
|
||||
commodity lexicon and let the two drift, so the lexicon is consulted as a
|
||||
second opinion. Without it every produce row landed in "General", which
|
||||
loses the unit rules, the grouping and the validation credit.
|
||||
"""
|
||||
_run(["Product Name", "Weight"], [["Apple", "500g"]])
|
||||
|
||||
assert store[0][1]["category"] == "Fruits & Vegetables"
|
||||
|
||||
|
||||
def test_a_lexicon_category_never_overrules_a_declared_pack_size(store):
|
||||
"""The regression that made this guard necessary.
|
||||
|
||||
The lexicon calls tea a Beverage; the unit rulebook says beverages are
|
||||
measured in ml or litres ONLY; stage 4 therefore "corrected" 250g to 250ml
|
||||
and turned a quarter kilo of tea leaves into a quarter litre. Loose tea is
|
||||
a dry good sold by weight - the category was wrong, not the size - so a
|
||||
category that cannot hold the declared size is declined.
|
||||
"""
|
||||
_run(["Product Name"], [["Tea Powder 250g"]])
|
||||
|
||||
assert len(store) == 1, "the row was dropped or silently unit-converted"
|
||||
brand, row = store[0]
|
||||
assert brand == OWN_PRODUCTS_BRAND
|
||||
assert row["size_variants"] == ["250g"], "250g became something else"
|
||||
assert row["category"] != "Beverages"
|
||||
Reference in New Issue
Block a user