Catalog feature updates on column fields

This commit is contained in:
sriram
2026-09-08 15:18:29 +05:30
parent 2749bee1a3
commit 10b24c6348
60 changed files with 9224 additions and 31 deletions

View File

@@ -0,0 +1,198 @@
"""The offline stage that expands a barcode into the rest of its identity.
THE FAILURE THIS FILE EXISTS FOR
--------------------------------
Measured against the production database on 2026-09-08:
barcode 18.4% filled
gtin 8.7%
ean13 6.4%
upc 0.0%
Every one of those three could have been computed from the barcode already
sitting in the same row - they are arithmetic on the digits, not a lookup. They
were empty because the only code that produced them lived inside
`BarcodeEnrichmentStage`, which is off by default (`ENABLE_BARCODE_LOOKUP`,
false in production), and because the writer dropped the fields anyway.
`BarcodeIdentityStage` closes that. It performs NO lookup, so it needs no
settings flag and costs nothing, and it runs on every ingestion.
The three properties that matter, each pinned below:
1. It expands a valid barcode into barcode_type / gtin / ean13 / upc.
2. It NEVER touches a barcode that fails validation. A sheet-supplied barcode
is the merchant's assertion; silently "correcting" or deleting one would be
worse than leaving it visibly wrong. The failure goes into `field_sources`,
not into the data.
3. It never claims a value is "verified". That word is reserved for the
cascade's brand+size+name-matched result, and a barcode typed into a
spreadsheet has passed no such check.
No network and no database: the stage has neither.
"""
from __future__ import annotations
import asyncio
import pytest
from app.services.enrichment.barcode.identity_stage import BarcodeIdentityStage
def run(product, brand="Cadbury"):
"""Apply the stage the way EnrichmentPipeline does, returning the row."""
stage = BarcodeIdentityStage()
return asyncio.run(stage.apply(dict(product), brand))
# ---------------------------------------------------------------------------
# 1. Expansion
# ---------------------------------------------------------------------------
def test_an_ean13_expands_into_gtin_and_ean13():
"""8901233018362 is Cadbury Bournvita's real barcode - the one that had to
be repaired in production by hand on 2026-09-08."""
row = run({"product_name": "Cadbury Bournvita 500g", "barcode": "8901233018362"})
assert row["barcode"] == "8901233018362"
assert row["barcode_type"] == "EAN-13"
assert row["gtin"] == "8901233018362"
assert row["ean13"] == "8901233018362"
assert row["upc"] is None # a 13-digit code is not a UPC-A
def test_a_upc_a_expands_into_both_upc_and_a_padded_ean13():
"""UPC-A is numerically a GTIN-13 with a leading zero, so both fields are
real for the same pack - the zero-padded form is what an EAN-13 scanner
reports."""
row = run({"product_name": "Imported Bar 50g", "barcode": "036000291452"})
assert row["barcode_type"] == "UPC-A"
assert row["upc"] == "036000291452"
assert row["ean13"] == "0036000291452"
assert row["gtin"] == "036000291452"
def test_a_gtin8_is_not_padded_into_an_ean13():
"""An 8-digit GTIN is its own symbology, not a truncated EAN-13. Padding it
would invent a code that identifies nothing. 89009802 is the real GTIN-8
Open Food Facts holds for Nestle Munch."""
row = run({"product_name": "Nestle Munch 8.9g", "barcode": "89009802"})
assert row["barcode_type"] == "GTIN-8"
assert row["gtin"] == "89009802"
assert row["ean13"] is None
assert row["upc"] is None
def test_separators_are_stripped_but_the_value_is_not_otherwise_changed():
row = run({"product_name": "Amul Butter 100g", "barcode": " 8901262-010016 "})
assert row["barcode"] == "8901262010016"
assert row["gtin"] == "8901262010016"
# ---------------------------------------------------------------------------
# 2. It never damages what the merchant supplied
# ---------------------------------------------------------------------------
def test_an_invalid_barcode_is_left_exactly_as_typed():
"""The placeholder `8900000000000.0` sat in 38 production rows. It is not a
barcode, but it is also not this stage's to delete - a value visibly wrong
is findable, a value silently blanked is not."""
row = run({"product_name": "Cadbury 5 Star 24g", "barcode": "8900000000000.0"})
assert row["barcode"] == "8900000000000.0"
assert row.get("gtin") is None
assert row.get("ean13") is None
assert row.get("barcode_type") is None
def test_a_failed_checksum_is_recorded_in_provenance_not_in_the_data():
"""13 digits of the right length but the wrong check digit."""
row = run({"product_name": "Probe", "barcode": "8901233018363"})
assert row["barcode"] == "8901233018363"
assert row["field_sources"]["barcode"]["method"] == "unvalidated"
assert "failed checksum" in row["field_sources"]["barcode"]["note"]
def test_a_row_with_no_barcode_is_untouched():
row = run({"product_name": "Amul Butter 100g", "category": "Dairy"})
assert "gtin" not in row
assert "field_sources" not in row
def test_it_never_invents_a_barcode():
"""The stage has no source and no network. If the row has no barcode, it
cannot acquire one here - that is BarcodeEnrichmentStage's job."""
row = run({"product_name": "Unknown Product 1kg", "barcode": ""})
assert not row.get("barcode")
assert not row.get("gtin")
# ---------------------------------------------------------------------------
# 3. It does not overstate what it knows
# ---------------------------------------------------------------------------
def test_a_sheet_barcode_is_never_marked_verified():
row = run({"product_name": "Probe 100g", "barcode": "8901233018362"})
assert row["barcode_verified"] is False
assert row["barcode_lookup_status"] == "sheet_validated"
assert row["barcode_source"] == "sheet"
def test_the_derived_fields_are_flagged_derived_not_sourced():
"""gtin/ean13/upc are arithmetic on the barcode. Recording them as
`sourced` would claim a lookup confirmed them, which is the exact
overstatement the provenance map exists to prevent."""
row = run({"product_name": "Probe 100g", "barcode": "8901233018362"})
for field in ("gtin", "ean13", "upc", "barcode_type"):
assert row["field_sources"][field]["method"] == "derived", field
def test_an_existing_source_is_not_overwritten_by_sheet():
"""When the cascade found the barcode, its provenance is the real one and
must survive this stage running afterwards."""
row = run({
"product_name": "Probe 100g",
"barcode": "8901233018362",
"barcode_source": "Open Food Facts",
"barcode_verified": True,
"barcode_lookup_status": "verified",
})
assert row["barcode_source"] == "Open Food Facts"
assert row["barcode_verified"] is True
assert row["field_sources"]["barcode"]["method"] == "sourced"
# ---------------------------------------------------------------------------
# 4. Provenance accumulates across stages
# ---------------------------------------------------------------------------
def test_field_sources_from_an_earlier_stage_is_merged_not_replaced():
"""`base.apply()` assigns every key except this one. Assigning it would
mean the last stage to run erases what every earlier stage recorded, so a
value would end up in the database with no origin."""
row = run({
"product_name": "Probe 100g",
"barcode": "8901233018362",
"field_sources": {"hsn_code": {"method": "estimated", "source": "category"}},
})
assert row["field_sources"]["hsn_code"]["method"] == "estimated"
assert row["field_sources"]["gtin"]["method"] == "derived"
def test_the_stage_never_raises_on_a_malformed_row():
"""The EnrichmentStage contract: a stage bug degrades to "no fields added",
never an aborted catalog row."""
for barcode in (None, "", "abc", 12345, [], {"nested": 1}, 8901233018362):
row = run({"product_name": "Probe", "barcode": barcode})
assert isinstance(row, dict)

View File

@@ -0,0 +1,236 @@
"""Accepting an Open Food Facts record whose name is shorter than ours.
THE FAILURE THIS FILE EXISTS FOR
--------------------------------
Running `scripts/backfill_nutrition_from_barcodes` over the production catalog
on 2026-09-08 reported, of 300 barcoded rows:
not a valid GTIN 37
not in Open Food Facts 100
matched but empty 11
found, WRONG PRODUCT 149 <-- this file
would write 3
The 149 were not wrong products. The barcode resolved perfectly; Open Food
Facts simply stores a short name where we store a long one:
ours "Nestle Munch 8.9g" OFF "Munch" similarity 0.332
ours "Coca-Cola Maaza 750ml" OFF "Maaza" similarity 0.304
ours "Cadbury Perk 22 g" OFF "Perk" similarity 0.302
`name_similarity` divides token overlap by the TARGET's token count, so a
one-token candidate against a three-token target cannot exceed about 0.33 no
matter how correct it is.
WHY THE FIX IS NOT A LOWER THRESHOLD
The same run correctly rejected these, which sit BELOW the containment cases
but not far enough below to be separable by a number:
ours "Pepsico Lays 1kg" OFF "Spanish tomato tango" 0.133
ours "Coca-Cola Fanta 750ml" OFF "Orange" 0.089
and `settings.py:449-477` records the measurement that raised this floor to
0.78 in the first place (at 0.45: 15 accepted / 8 wrong; at 0.78: 2 / 0).
Lowering it re-admits exactly what it was raised to exclude.
Containment separates the groups structurally instead. It also has to reject
two cases a naive substring check would wave through, both of which are real
Open Food Facts titles: the bare brand name ("Colgate", "godrej"), which would
otherwise attach to every product of that brand, and a same-brand sibling
("Dairy Milk Silk" against our "Cadbury Dairy Milk").
The relaxation is OFF by default and enabled on exactly one call site -
`fetch_verified_nutrition_by_barcode` - because there the barcode has already
established identity and there are no competing candidates. On the search path,
where many candidates compete and the name is the only discriminator, "Munch"
would match every Nestle product containing that word.
"""
from __future__ import annotations
import pytest
from app.services.enrichment.barcode.matching import is_match, name_is_contained
from app.services.enrichment.barcode.models import BarcodeCandidate
# (our stored title, what OFF calls it, brand) - all measured on 2026-09-08
CONTAINED = [
("Nestle Munch 8.9g", "Munch", "Nestle"),
("Coca-Cola Maaza 750ml", "Maaza", "Coca-Cola"),
("Cadbury Perk 22 g", "Perk", "Cadbury"),
("Nestle Milo 25 g", "MILO", "Nestle"),
("Cadbury Fuse 25 g", "FUSE", "Cadbury"),
("Coca-Cola Limca 750g", "limca", "Coca-Cola"),
("Nestle Milkybar 25g", "Milkybar", "Nestle"),
]
NOT_CONTAINED = [
("Pepsico Lays 1kg", "Spanish tomato tango", "Pepsico"),
("Lion Dates Powder 100g", "PEPER NOTEN", "Lion Dates"),
("Coca-Cola Fanta 750ml", "Orange", "Coca-Cola"),
# Bare brand names. Both are real OFF product_name values.
("Colgate Total Toothpaste 150g", "Colgate", "Colgate"),
("Godrej No1 Soap 100g", "godrej", "Godrej"),
# Same brand, different product - the case the name gate exists for.
("Amul Butter 100g", "Amul Cheese", "Amul"),
("Cadbury Dairy Milk 50g", "Dairy Milk Silk", "Cadbury"),
]
@pytest.mark.parametrize("ours,theirs,brand", CONTAINED)
def test_a_short_off_name_is_recognised_as_ours(ours, theirs, brand):
assert name_is_contained(theirs, ours, brand) is True
@pytest.mark.parametrize("ours,theirs,brand", NOT_CONTAINED)
def test_a_different_product_is_still_refused(ours, theirs, brand):
assert name_is_contained(theirs, ours, brand) is False
def test_a_bare_brand_name_never_matches():
""""Colgate" as a product name identifies a brand, not a product. Accepting
it would attach one arbitrary pack's nutrition to every Colgate row."""
assert name_is_contained("Colgate", "Colgate MaxFresh Toothpaste 150g", "Colgate") is False
def test_size_tokens_do_not_decide_identity():
"""`size_matches` has already compared the pack size by the time this is
consulted, so a size token in our title must not make the names differ."""
assert name_is_contained("Munch", "Nestle Munch 8.9g", "Nestle") is True
assert name_is_contained("Munch", "Nestle Munch 38.5 g", "Nestle") is True
def test_an_extra_token_in_the_candidate_breaks_containment():
"""Containment is one-directional on purpose: every candidate token must be
ours. "Dairy Milk Silk" carries "silk", which our "Cadbury Dairy Milk" does
not, so it is a different product."""
assert name_is_contained("Dairy Milk Silk", "Cadbury Dairy Milk 50g", "Cadbury") is False
def test_empty_names_are_refused_rather_than_treated_as_contained():
"""The empty set is a subset of everything - the one case where the maths
says yes and the answer is obviously no."""
assert name_is_contained("", "Nestle Munch 8.9g", "Nestle") is False
assert name_is_contained("Munch", "", "Nestle") is False
# ---------------------------------------------------------------------------
# The gate as a whole
# ---------------------------------------------------------------------------
def _candidate(title, brand, size=""):
return BarcodeCandidate(barcode="8901058857245", source_name="Open Food Facts",
candidate_title=title, candidate_brand=brand,
candidate_size=size)
def test_containment_is_off_by_default():
"""The search path must not get this relaxation: there, many candidates
compete and "Munch" would match every Nestle product containing it."""
matched, _ = is_match(_candidate("Munch", "Nestle", "8.9g"),
"Nestle", "Nestle Munch 8.9g", "8.9g",
min_name_similarity=0.78)
assert matched is False
def test_containment_accepts_when_explicitly_enabled():
matched, similarity = is_match(_candidate("Munch", "Nestle", "8.9g"),
"Nestle", "Nestle Munch 8.9g", "8.9g",
min_name_similarity=0.78,
barcode_is_identity=True)
assert matched is True
# The reported confidence is still the honest similarity, not 1.0 - it is
# stored on the row for audit and must not be inflated by the relaxation.
assert similarity < 0.5
def test_containment_does_not_bypass_the_brand_gate():
"""Rules 1-3 still apply. A containment name match with the wrong brand is
still a wrong product."""
matched, _ = is_match(_candidate("Munch", "Britannia", "8.9g"),
"Nestle", "Nestle Munch 8.9g", "8.9g",
min_name_similarity=0.78, barcode_is_identity=True)
assert matched is False
def test_a_conflicting_size_is_still_refused():
"""A quantity that is PRESENT and different means our barcode is on the
wrong row. That is exactly what the sanity check is for."""
matched, _ = is_match(_candidate("Munch", "Nestle", "500g"),
"Nestle", "Nestle Munch 8.9g", "8.9g",
min_name_similarity=0.78, barcode_is_identity=True)
assert matched is False
# ---------------------------------------------------------------------------
# The gate that actually blocked most of the 149
# ---------------------------------------------------------------------------
def test_a_blank_candidate_size_no_longer_vetoes_under_a_barcode():
"""The real blocker, found only after measuring the containment fix.
`size_matches` returns False whenever EITHER side is blank, and Open Food
Facts leaves `quantity` null on a large share of records - 57 of 146 Amul
hits. Because `is_match` applies its rules in order, that rejected these
rows before the name rule was ever consulted, so fixing the name gate alone
moved the measured result from 3 rows to 8 rather than to ~149.
A record with no quantity does not disagree with our pack size. It says
nothing about it, and the barcode has already established identity.
"""
matched, _ = is_match(_candidate("Munch", "Nestle", ""),
"Nestle", "Nestle Munch 38.5 g", "38.5 g",
min_name_similarity=0.78, barcode_is_identity=True)
assert matched is True
def test_a_blank_candidate_size_still_vetoes_on_the_search_path():
"""Without a barcode a sizeless candidate is genuinely unidentifiable: it
could be any pack of that product, and a GTIN belongs to exactly one."""
matched, _ = is_match(_candidate("Nestle Munch", "Nestle", ""),
"Nestle", "Nestle Munch 38.5 g", "38.5 g",
min_name_similarity=0.45)
assert matched is False
def test_the_size_relaxation_does_not_also_relax_the_brand_gate():
matched, _ = is_match(_candidate("Munch", "Britannia", ""),
"Nestle", "Nestle Munch 38.5 g", "38.5 g",
min_name_similarity=0.78, barcode_is_identity=True)
assert matched is False
def test_the_size_relaxation_does_not_also_relax_variant_conflicts():
matched, _ = is_match(_candidate("Munch sugar free", "Nestle", ""),
"Nestle", "Nestle Munch 38.5 g", "38.5 g",
min_name_similarity=0.78, barcode_is_identity=True)
assert matched is False
def test_containment_does_not_bypass_the_variant_conflict_gate():
""""sugar free" on the candidate but not on ours is a different product
however well the rest of the name contains."""
matched, _ = is_match(_candidate("Munch sugar free", "Nestle", "8.9g"),
"Nestle", "Nestle Munch 8.9g", "8.9g",
min_name_similarity=0.78, barcode_is_identity=True)
assert matched is False
def test_a_normal_high_similarity_match_is_unaffected():
"""The relaxation is only consulted when the similarity floor fails, so it
cannot change any decision the existing gate already made."""
matched, similarity = is_match(_candidate("Nestle Munch", "Nestle", "8.9g"),
"Nestle", "Nestle Munch", "8.9g",
min_name_similarity=0.78)
assert matched is True
assert similarity >= 0.78

View File

@@ -0,0 +1,276 @@
"""The enrichment columns on the brand tables, and the write path that fills them.
THE FAILURE THIS FILE EXISTS FOR
--------------------------------
The barcode stage returns nine fields - `BarcodeResult.as_product_fields()` -
and `upsert_brand_products` named two of them. The other seven were computed on
every run and then dropped on the floor by the writer. Measured against the
production database on 2026-09-08, before this landed:
upc 0.0% (column existed on 7 of 56 tables, never written)
ean13 6.4% (added out-of-band, never written by any code)
gtin 8.7%
barcode 18.4%
The same was true of the HSN/GST stage: it computes `gst_percent`, `tax_amount`
and `hsn_gst_needs_review`, and `_to_storage_row` projected none of them.
Three distinct things have to hold for a value to survive, and each of them
broke independently at some point, so each gets a test here:
1. The column has to EXIST. `_ensure_columns` is the only migration mechanism
in the repo - there is no Alembic and no migrations directory - so a column
missing from its `col_defs` dict never appears on the 56 brand tables that
already exist.
2. The INSERT has to NAME it. Adding the column is not enough; that is exactly
how seven tables ended up carrying `gtin` and `ean13` columns that no code
ever wrote a value into.
3. A later re-seed must not BLANK it. `ON CONFLICT DO UPDATE SET x =
EXCLUDED.x` overwrites with whatever arrived, and the seed loader,
`user_products._build_product_dict` and `brand_sync`'s re-seed all build a
product dict from a spreadsheet with no enrichment keys in it - so their
EXCLUDED values are NULL. This is the same trap `test_brand_table_scores`
documents for the score columns, which is why those are omitted from the
statement entirely. These columns cannot be omitted (the pipeline is what
writes them), so they use COALESCE instead.
`barcode` and `barcode_type` are COALESCEd alongside the seven, though they are
older columns. Proven against a live table: a bare re-seed set `barcode` to
NULL while COALESCE kept `gtin` and `ean13`, leaving a row that claimed a GTIN
with no barcode. A half-erased identity is worse than either whole state, and
the rest of the barcode package already promises never to erase one -
`stage.py` skips a row that has a barcode and `enrichment/base.py` refuses to
blank a held value. The upsert was the one place that still could.
No database is involved: the cursor is a recorder, so the assertions are about
the exact SQL sent.
"""
from __future__ import annotations
import re
from app.services import brand_sync, vector_store
# The columns this change added, with the type each MUST be created as.
#
# THE TYPES ARE ADOPTED, NOT CHOSEN. Seven brand tables already carried these
# columns before any code created them, and `ADD COLUMN IF NOT EXISTS` does not
# reconcile a type difference - it silently leaves the old table alone. Picking
# a "better" type here (NUMERIC for the tax figures, DOUBLE PRECISION for the
# epoch) would leave 7 tables permanently disagreeing with 49. These are what
# `information_schema` reported for those 7 tables on 2026-09-08.
PIPELINE_OWNED = {
"gtin": "TEXT",
"ean13": "TEXT",
"upc": "TEXT",
"barcode_source": "TEXT",
"barcode_verified": "BOOLEAN",
"barcode_lookup_status": "TEXT",
"barcode_last_updated": "TIMESTAMP",
"gst_percent": "REAL",
"tax_amount": "REAL",
"hsn_gst_needs_review": "BOOLEAN",
"field_sources": "JSONB",
}
# Written only by nutrition_score_sync, never by the INSERT - the same
# category as nutrition_score / health_score.
MIRROR_OWNED = {"nutrients_per_100g": "JSONB"}
class MigrationCursor:
"""A brand table that exists but has none of the current columns."""
def __init__(self):
self.statements = []
def execute(self, sql, params=None):
self.statements.append(" ".join(str(sql).split()))
def fetchall(self):
return [] # no existing columns -> every column is missing
def _insert_statement() -> str:
"""The INSERT ... ON CONFLICT text, whitespace-normalised."""
source = vector_store.upsert_brand_products.__doc__ or ""
# The statement is built inline, so read it off the module source rather
# than reaching into a closure.
import inspect
body = inspect.getsource(vector_store.upsert_brand_products)
match = re.search(r"INSERT INTO \{table_name\}.*?updated_at = CURRENT_TIMESTAMP",
body, re.S)
assert match, "could not locate the INSERT statement in upsert_brand_products"
return " ".join(match.group(0).split())
# ---------------------------------------------------------------------------
# 1. The columns exist
# ---------------------------------------------------------------------------
def test_the_migration_adds_every_enrichment_column():
cur = MigrationCursor()
vector_store._ensure_columns(cur, "brand_cadbury")
for col, col_type in {**PIPELINE_OWNED, **MIRROR_OWNED}.items():
assert (f"ALTER TABLE brand_cadbury ADD COLUMN IF NOT EXISTS "
f"{col} {col_type}") in cur.statements, col
def test_the_types_match_the_tables_that_already_had_these_columns():
"""A wrong type here is invisible until a write fails on one of the seven
pre-existing tables, because ADD COLUMN IF NOT EXISTS skips them silently.
`barcode_last_updated` is the one most likely to be "corrected" by a future
reader: `BarcodeResult` carries a float epoch, so DOUBLE PRECISION looks
right. The column on disk is TIMESTAMP, and vector_store._epoch_to_timestamp
is what bridges the two.
"""
assert vector_store._ensure_columns.__doc__ is not None
import inspect
body = inspect.getsource(vector_store._ensure_columns)
assert '"barcode_last_updated": "TIMESTAMP"' in body
assert '"gst_percent": "REAL"' in body
assert '"tax_amount": "REAL"' in body
def test_the_create_table_ddl_carries_them_too():
"""`col_defs` migrates existing tables; the DDL is what a brand table
created from scratch gets. A column in one but not the other means a new
brand's table differs from every other brand's."""
ddl = vector_store.get_brand_table_ddl("Cadbury")
for col in {**PIPELINE_OWNED, **MIRROR_OWNED}:
assert re.search(rf"^\s*{col}\s", ddl, re.M), col
# ---------------------------------------------------------------------------
# 2. The INSERT names them - and does not name the mirror-owned ones
# ---------------------------------------------------------------------------
def test_the_insert_names_every_pipeline_owned_column():
statement = _insert_statement()
column_list = statement.split("VALUES")[0]
for col in PIPELINE_OWNED:
assert re.search(rf"[(,] ?{col}[,)]", column_list), col
def test_the_insert_does_not_name_the_mirror_owned_columns():
"""Same rule as nutrition_score / health_score: a column this statement
never names is a column it cannot damage."""
statement = _insert_statement()
for col in MIRROR_OWNED:
assert col not in statement, col
def test_the_placeholder_count_matches_the_column_count():
"""An arity mismatch here is a runtime error on every single write, so it
is worth catching at import time rather than on the next upload."""
statement = _insert_statement()
match = re.search(r"INSERT INTO \{table_name\} \((.*?)\) VALUES \((.*?)\)", statement)
columns = [c.strip() for c in match.group(1).split(",") if c.strip()]
assert len(columns) == match.group(2).count("%s")
# ---------------------------------------------------------------------------
# 3. A re-seed cannot blank them
# ---------------------------------------------------------------------------
def test_every_enrichment_column_is_coalesced_on_conflict():
"""This is the guard that makes "a re-seed wipes the enrichment"
structurally impossible rather than merely unlikely."""
statement = _insert_statement()
do_update = statement.split("DO UPDATE SET", 1)[1]
for col in PIPELINE_OWNED:
if col == "field_sources":
continue # merged, asserted separately below
assert (f"{col} = COALESCE(EXCLUDED.{col}, {{table_name}}.{col})"
in do_update), col
def test_the_barcode_pair_is_coalesced_with_its_identity_group():
"""`barcode` and `barcode_type` predate this change but belong to the same
identity group as gtin/ean13/upc. Leaving them on plain EXCLUDED produced a
row with a GTIN and no barcode after a bare re-seed."""
do_update = _insert_statement().split("DO UPDATE SET", 1)[1]
for col in ("barcode", "barcode_type"):
assert (f"{col} = COALESCE(EXCLUDED.{col}, {{table_name}}.{col})"
in do_update), col
def test_field_sources_is_merged_rather_than_replaced():
"""A run that learns the provenance of one field must not drop what is
already known about the others, so this one is `||`, not COALESCE."""
do_update = _insert_statement().split("DO UPDATE SET", 1)[1]
# Read off the module source, so the f-string's escaped braces are still
# doubled here - `'{{}}'` is what renders as the SQL literal `'{}'`.
assert "field_sources = COALESCE({table_name}.field_sources, '{{}}'::jsonb) " \
"|| COALESCE(EXCLUDED.field_sources, '{{}}'::jsonb)" in do_update
# ---------------------------------------------------------------------------
# 4. The seed export carries them
# ---------------------------------------------------------------------------
def test_the_seed_export_carries_every_new_column():
"""`export_brand_to_seed_file` rebuilds a product from EXPORT_COLUMNS and
replaces the whole dict. A column missing from that tuple is stripped out
of the catalog file on export - which is precisely why
scripts/backfill_barcodes_from_off.py refuses to call that helper today."""
for col in {**PIPELINE_OWNED, **MIRROR_OWNED}:
assert col in brand_sync.EXPORT_COLUMNS, col
def test_the_export_coerces_values_json_dumps_would_refuse():
"""`barcode_last_updated` comes back from psycopg as a datetime and
`field_sources` as a dict. json.dumps refuses the first outright and chokes
on a Decimal nested in the second."""
from datetime import datetime
from decimal import Decimal
assert brand_sync._jsonable(datetime(2026, 9, 8, 10, 51, 42)) == "2026-09-08T10:51:42"
assert brand_sync._jsonable({"barcode": {"confidence": Decimal("0.91")}}) == {
"barcode": {"confidence": 0.91}
}
# ---------------------------------------------------------------------------
# 5. The epoch/timestamp bridge
# ---------------------------------------------------------------------------
def test_the_timestamp_bridge_accepts_every_shape_that_reaches_it():
"""Three writers feed this column and they disagree about the type:
BarcodeResult emits a float epoch, a re-seeded catalog file carries the
ISO-8601 string brand_sync exported, and a DB read hands back a datetime.
All three have to load or a round-trip drops the value it just wrote."""
from datetime import datetime
assert vector_store._epoch_to_timestamp(1788773811.0) == datetime.fromtimestamp(1788773811.0)
assert vector_store._epoch_to_timestamp("2026-09-08T10:51:42") == datetime(2026, 9, 8, 10, 51, 42)
assert vector_store._epoch_to_timestamp(datetime(2026, 1, 1)) == datetime(2026, 1, 1)
assert vector_store._epoch_to_timestamp("not a date") is None
assert vector_store._epoch_to_timestamp("") is None
assert vector_store._epoch_to_timestamp(None) is None
def test_the_numeric_coercion_refuses_rather_than_raises():
"""A malformed tax figure must degrade to "no figure stored" and never
abort a whole batch's write."""
assert vector_store._to_numeric_or_none("18%") == 18.0
assert vector_store._to_numeric_or_none("₹1,250.50") == 1250.50
assert vector_store._to_numeric_or_none(12.5) == 12.5
assert vector_store._to_numeric_or_none("not a number") is None
assert vector_store._to_numeric_or_none(None) is None
# bool is an int subclass; True must not become 1.0 in a NUMERIC column
assert vector_store._to_numeric_or_none(True) is None

View File

@@ -0,0 +1,159 @@
"""The stage that fills `highlights` and `nutrients` on an uploaded row.
THE FAILURE THIS FILE EXISTS FOR
--------------------------------
`catalog_engine.generate_product_highlights` and `generate_nutrients_info` have
existed for a long time, and `brand_discovery._build_product` calls both. The
store-catalog pipeline never did - `_to_storage_row` passed through whatever
the sheet carried, and a colleague's sheet carries neither column. So every
single uploaded row landed `highlights=[]` and `nutrients=[]`.
That is why those columns look healthy in aggregate (95.3% / 69.7% measured on
2026-09-08) while being empty for exactly the rows this work is about: the
percentages come from the older brand-discovery path.
The second thing this file pins is the consumability gate.
`generate_nutrients_info` matches category keywords, so without a gate a Hair
Care row can acquire "Vitamin B Complex - Energy". Shampoo has no nutrients.
`scripts/purge_non_consumable_nutrition.py` exists because this already
happened once at the `nutrition_facts` level; the display column needs the same
refusal, and it must record `not_applicable` rather than leave a silent blank -
a permanent unexplained gap is what eventually gets "fixed" by fabricating.
No network, no database, no LLM: the generators are keyword functions over the
row dict.
"""
from __future__ import annotations
import asyncio
import pytest
from app.services.enrichment.content.stage import ContentEnrichmentStage, _has_entries
def run(product, brand="Cadbury"):
return asyncio.run(ContentEnrichmentStage().apply(dict(product), brand))
FOOD_ROW = {
"product_name": "Cadbury Dairy Milk 50g",
"title": "Cadbury Dairy Milk",
"category": "Chocolates",
"size": "50g",
"description": "Smooth milk chocolate bar.",
}
SHAMPOO_ROW = {
"product_name": "Dove Daily Shine Shampoo 340ml",
"title": "Dove Daily Shine Shampoo",
"category": "Hair Care",
"size": "340ml",
"description": "Nourishing shampoo for daily use.",
}
# ---------------------------------------------------------------------------
# It fills what the pipeline used to leave empty
# ---------------------------------------------------------------------------
def test_an_uploaded_food_row_gets_highlights():
row = run(FOOD_ROW)
assert row["highlights"], "every upload landed highlights=[] before this stage"
assert all(isinstance(h, str) and h.strip() for h in row["highlights"])
def test_an_uploaded_food_row_gets_nutrients():
row = run(FOOD_ROW)
assert row["nutrients"]
def test_highlights_are_flagged_derived_not_sourced():
"""They are marketing copy computed from fields we already hold. Calling
them sourced would claim something confirmed them."""
row = run(FOOD_ROW)
assert row["field_sources"]["highlights"]["method"] == "derived"
def test_the_keyword_nutrients_are_flagged_estimated():
"""They are category guesses standing in until a real lookup succeeds, and
the mirror from nutrition_facts overwrites them when one does."""
row = run(FOOD_ROW)
assert row["field_sources"]["nutrients"]["method"] == "estimated"
# ---------------------------------------------------------------------------
# The consumability gate
# ---------------------------------------------------------------------------
def test_a_shampoo_gets_no_nutrients():
"""The real defect: keyword matching gave personal-care rows entries like
"Vitamin B Complex - Energy"."""
row = run(SHAMPOO_ROW, brand="Dove")
assert not row.get("nutrients")
def test_a_shampoo_records_not_applicable_rather_than_a_silent_blank():
"""A gap nobody can explain is the one somebody eventually fills with
invented data. The coverage report reads this to exclude the row from the
nutrition denominator instead of reporting it missing forever."""
row = run(SHAMPOO_ROW, brand="Dove")
assert row["field_sources"]["nutrients"]["method"] == "not_applicable"
def test_a_shampoo_still_gets_highlights():
"""Non-consumable rules out nutrition, not description. A shampoo has
perfectly good highlights."""
row = run(SHAMPOO_ROW, brand="Dove")
assert row["highlights"]
# ---------------------------------------------------------------------------
# It fills blanks only
# ---------------------------------------------------------------------------
def test_values_the_sheet_supplied_are_never_overwritten():
"""The pipeline's first rule: the store's own data is authoritative."""
row = run({**FOOD_ROW,
"highlights": ["Fairtrade cocoa"],
"nutrients": ["Protein 7.3 g per 100 g"]})
assert row["highlights"] == ["Fairtrade cocoa"]
assert row["nutrients"] == ["Protein 7.3 g per 100 g"]
def test_a_column_of_empty_strings_counts_as_blank():
"""The spreadsheet parser produces [''] from a column that exists with no
value in it. Treating that as "already filled" keeps the row [''] forever."""
assert _has_entries([""]) is False
assert _has_entries(["", " "]) is False
assert _has_entries(["Real"]) is True
assert _has_entries([]) is False
row = run({**FOOD_ROW, "highlights": [""]})
assert row["highlights"] != [""]
# ---------------------------------------------------------------------------
# The stage contract
# ---------------------------------------------------------------------------
def test_provenance_from_an_earlier_stage_survives():
row = run({**FOOD_ROW,
"field_sources": {"gtin": {"method": "derived"}}})
assert row["field_sources"]["gtin"]["method"] == "derived"
assert row["field_sources"]["highlights"]["method"] == "derived"
def test_the_stage_never_raises_on_a_malformed_row():
for product in ({}, {"product_name": None}, {"category": 123},
{"title": "", "category": None, "highlights": "not a list"}):
assert isinstance(run(product), dict)

View File

@@ -0,0 +1,177 @@
"""Defaults that invent a regulatory identifier or a commercial claim.
THE FAILURE THIS FILE EXISTS FOR
--------------------------------
`user_products._build_product_dict` filled two blanks with constants:
fssai_license = req.fssai_license or sample.get("fssai_license") or "10012042000244"
providers = req.providers or list(sample.get("providers") or
["Amazon", "Flipkart", "BigBasket", "Jiomart", "Blinkit", "Zepto"])
The first constant is not a placeholder. `10012042000244` is Lion Dates' real,
registered FSSAI licence - it is still in `brand_registry.FSSAI_LICENSES` under
that brand. Every product uploaded for a brand with no existing row was stamped
with it, which attributes legal responsibility for that food to a business that
never made it. It reached production at least once:
`scripts/merge_haldiram.py` exists specifically to strip it back off
`brand_haldirams`.
The second asserts a product is stocked by six named marketplaces on the basis
of nothing at all.
Both are now sourced from the brand's own rows, or left empty. The tests below
pin three things:
1. The literal constants are gone from the module.
2. An unknown brand gets NO licence rather than someone else's.
3. Consensus refuses to answer when a brand's own rows disagree - because at
that point one of them is already wrong and a tie-break would just be
picking which product to mislabel.
No database: `consensus_value` and `fssai_for_brand` take the rows as an
argument, which is what makes them testable at all.
"""
from __future__ import annotations
import ast
import inspect
from app.api.routers import user_products
from app.services.enrichment.catalog_consensus import consensus_value, fssai_for_brand
LION_DATES_LICENCE = "10012042000244"
# ---------------------------------------------------------------------------
# 1. The constants are gone
# ---------------------------------------------------------------------------
def _executable_string_literals(module) -> list:
"""Every string constant the module can actually evaluate.
Docstrings and comments are excluded deliberately: the fix's own comment
has to be free to name the constant it removed, or the explanation of why
the bug mattered cannot be written down next to the code that had it.
"""
tree = ast.parse(inspect.getsource(module))
docstrings = set()
for node in ast.walk(tree):
if isinstance(node, (ast.Module, ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef)):
body = getattr(node, "body", [])
if (body and isinstance(body[0], ast.Expr)
and isinstance(body[0].value, ast.Constant)
and isinstance(body[0].value.value, str)):
docstrings.add(id(body[0].value))
return [n.value for n in ast.walk(tree)
if isinstance(n, ast.Constant) and isinstance(n.value, str)
and id(n) not in docstrings]
def test_lion_dates_licence_is_not_a_fallback_anywhere_in_the_upload_path():
literals = _executable_string_literals(user_products)
assert LION_DATES_LICENCE not in literals, (
"A real registered FSSAI licence must never appear as a default. "
"It belongs to Lion Dates and to no other brand."
)
def test_the_six_marketplace_default_is_gone():
literals = _executable_string_literals(user_products)
for marketplace in ("Blinkit", "BigBasket", "Jiomart", "Zepto"):
assert marketplace not in literals, (
f"{marketplace} appears as an evaluable literal. Listing "
f"marketplaces nobody verified is a false availability claim."
)
# ---------------------------------------------------------------------------
# 2. An unknown brand gets nothing
# ---------------------------------------------------------------------------
def test_an_unknown_brand_with_no_rows_gets_no_licence():
value, source = fssai_for_brand("Entirely Unknown Brand", rows=[])
assert value is None
assert source == "unknown"
def test_a_registry_brand_still_gets_its_mapped_licence():
"""The curated map remains the first and best source."""
value, source = fssai_for_brand("Lion Dates", rows=[])
assert value == LION_DATES_LICENCE
assert source == "brand_registry"
def test_an_unmapped_brand_inherits_from_its_own_agreeing_rows():
"""400 Britannia rows carrying one licence is good evidence for the 401st -
unlike a constant, this value genuinely belongs to the brand."""
rows = [{"fssai_license": "11223344556677"} for _ in range(20)]
value, source = fssai_for_brand("Some Unmapped Brand", rows=rows)
assert value == "11223344556677"
assert source == "catalog_consensus"
# ---------------------------------------------------------------------------
# 3. Consensus declines rather than guesses
# ---------------------------------------------------------------------------
def test_conflicting_licences_propagate_nothing():
"""Two different licences on one brand means one is already wrong. Picking
the more common one would just spread whichever error is ahead."""
rows = ([{"fssai_license": "11111111111111"}] * 10 +
[{"fssai_license": "22222222222222"}] * 8)
value, source = fssai_for_brand("Conflicted Brand", rows=rows)
assert value is None
assert source == "unknown"
def test_a_single_dissenting_row_does_not_veto_a_clear_majority():
"""One bad row among many should not block the other 19 from being useful."""
rows = [{"fssai_license": "11111111111111"}] * 19 + [{"fssai_license": "99999999999999"}]
value, _ = consensus_value("fssai_license", rows)
assert value == "11111111111111"
def test_too_few_rows_is_not_a_consensus():
"""Two rows agreeing proves nothing about a third."""
value, why = consensus_value("fssai_license", [{"fssai_license": "1"}, {"fssai_license": "1"}])
assert value is None
assert why["reason"] == "too few populated rows"
def test_blank_values_are_not_counted_as_agreement():
"""A column that is empty on every row must not come back as a consensus of
empties - that would read as "the brand agrees there is no licence"."""
value, _ = consensus_value("fssai_license", [{"fssai_license": None}] * 20)
assert value is None
def test_list_valued_columns_reach_consensus_too():
"""`providers` is a TEXT[]; lists are unhashable, so the modal calculation
has to key them as tuples or it raises."""
rows = [{"providers": ["Amazon", "Flipkart"]}] * 10
value, _ = consensus_value("providers", rows)
assert value == ["Amazon", "Flipkart"]
def test_consensus_never_raises_on_unusable_rows():
value, why = consensus_value("providers", [{"providers": {"unhashable": ["dict"]}}] * 5)
assert value is None
assert "reason" in why

View File

@@ -184,15 +184,36 @@ def test_a_matching_record_with_no_usable_values_is_unavailable(off_returns):
assert facts["data_status"] == "unavailable"
def test_the_floor_can_be_relaxed_per_call(off_returns):
"""The threshold is a judgement, not a constant, so it is a parameter.
def test_a_shorter_off_name_no_longer_needs_the_floor_relaxed(off_returns):
""""Aachi Biryani Masala" against our "Aachi Biryani Masala 50 g" is plainly
the same product and scores 0.716 - below the conservative 0.78 default.
"Aachi Biryani Masala" against our "Aachi Biryani Masala 50 g" is plainly
the same product and scores 0.716 - below the conservative default. The
backfill script exposes this as --min-similarity for exactly this reason.
It used to need `min_name_similarity=0.70` to get through. It no longer
does: this path passes `barcode_is_identity=True`, and the candidate's name
is ours minus the pack size, so the containment rule accepts it while the
floor stays where the measured yield table put it. See
tests/test_barcode_name_containment.py - 149 of 300 barcoded rows were
being refused this way.
"""
off_returns["body"] = {"status": 1, "product": _product(
"Aachi Biryani Masala", "Aachi", "50 g")}
facts = nds.fetch_verified_nutrition_by_barcode(
"8906021120272", "Aachi", "Aachi Biryani Masala 50 g", "50 g")
assert facts["data_status"] == "verified"
def test_the_floor_can_still_be_relaxed_per_call(off_returns):
"""The threshold is a judgement, not a constant, so it is a parameter, and
the backfill script exposes it as --min-similarity.
"Aachi Biryani Masala Mix" carries a token ours does not, so containment
does NOT rescue it - it is exactly the shape the floor exists to judge. It
scores 0.703: refused at the 0.78 default, accepted at 0.70.
"""
off_returns["body"] = {"status": 1, "product": _product(
"Aachi Biryani Masala Mix", "Aachi", "50 g")}
args = ("8906021120272", "Aachi", "Aachi Biryani Masala 50 g", "50 g")
assert nds.fetch_verified_nutrition_by_barcode(*args)["data_status"] == "unavailable"