Catalog feature updates on column fields
This commit is contained in:
198
tests/test_barcode_identity_stage.py
Normal file
198
tests/test_barcode_identity_stage.py
Normal file
@@ -0,0 +1,198 @@
|
||||
"""The offline stage that expands a barcode into the rest of its identity.
|
||||
|
||||
THE FAILURE THIS FILE EXISTS FOR
|
||||
--------------------------------
|
||||
Measured against the production database on 2026-09-08:
|
||||
|
||||
barcode 18.4% filled
|
||||
gtin 8.7%
|
||||
ean13 6.4%
|
||||
upc 0.0%
|
||||
|
||||
Every one of those three could have been computed from the barcode already
|
||||
sitting in the same row - they are arithmetic on the digits, not a lookup. They
|
||||
were empty because the only code that produced them lived inside
|
||||
`BarcodeEnrichmentStage`, which is off by default (`ENABLE_BARCODE_LOOKUP`,
|
||||
false in production), and because the writer dropped the fields anyway.
|
||||
|
||||
`BarcodeIdentityStage` closes that. It performs NO lookup, so it needs no
|
||||
settings flag and costs nothing, and it runs on every ingestion.
|
||||
|
||||
The three properties that matter, each pinned below:
|
||||
|
||||
1. It expands a valid barcode into barcode_type / gtin / ean13 / upc.
|
||||
2. It NEVER touches a barcode that fails validation. A sheet-supplied barcode
|
||||
is the merchant's assertion; silently "correcting" or deleting one would be
|
||||
worse than leaving it visibly wrong. The failure goes into `field_sources`,
|
||||
not into the data.
|
||||
3. It never claims a value is "verified". That word is reserved for the
|
||||
cascade's brand+size+name-matched result, and a barcode typed into a
|
||||
spreadsheet has passed no such check.
|
||||
|
||||
No network and no database: the stage has neither.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
|
||||
import pytest
|
||||
|
||||
from app.services.enrichment.barcode.identity_stage import BarcodeIdentityStage
|
||||
|
||||
|
||||
def run(product, brand="Cadbury"):
|
||||
"""Apply the stage the way EnrichmentPipeline does, returning the row."""
|
||||
stage = BarcodeIdentityStage()
|
||||
return asyncio.run(stage.apply(dict(product), brand))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. Expansion
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_an_ean13_expands_into_gtin_and_ean13():
|
||||
"""8901233018362 is Cadbury Bournvita's real barcode - the one that had to
|
||||
be repaired in production by hand on 2026-09-08."""
|
||||
row = run({"product_name": "Cadbury Bournvita 500g", "barcode": "8901233018362"})
|
||||
|
||||
assert row["barcode"] == "8901233018362"
|
||||
assert row["barcode_type"] == "EAN-13"
|
||||
assert row["gtin"] == "8901233018362"
|
||||
assert row["ean13"] == "8901233018362"
|
||||
assert row["upc"] is None # a 13-digit code is not a UPC-A
|
||||
|
||||
|
||||
def test_a_upc_a_expands_into_both_upc_and_a_padded_ean13():
|
||||
"""UPC-A is numerically a GTIN-13 with a leading zero, so both fields are
|
||||
real for the same pack - the zero-padded form is what an EAN-13 scanner
|
||||
reports."""
|
||||
row = run({"product_name": "Imported Bar 50g", "barcode": "036000291452"})
|
||||
|
||||
assert row["barcode_type"] == "UPC-A"
|
||||
assert row["upc"] == "036000291452"
|
||||
assert row["ean13"] == "0036000291452"
|
||||
assert row["gtin"] == "036000291452"
|
||||
|
||||
|
||||
def test_a_gtin8_is_not_padded_into_an_ean13():
|
||||
"""An 8-digit GTIN is its own symbology, not a truncated EAN-13. Padding it
|
||||
would invent a code that identifies nothing. 89009802 is the real GTIN-8
|
||||
Open Food Facts holds for Nestle Munch."""
|
||||
row = run({"product_name": "Nestle Munch 8.9g", "barcode": "89009802"})
|
||||
|
||||
assert row["barcode_type"] == "GTIN-8"
|
||||
assert row["gtin"] == "89009802"
|
||||
assert row["ean13"] is None
|
||||
assert row["upc"] is None
|
||||
|
||||
|
||||
def test_separators_are_stripped_but_the_value_is_not_otherwise_changed():
|
||||
row = run({"product_name": "Amul Butter 100g", "barcode": " 8901262-010016 "})
|
||||
|
||||
assert row["barcode"] == "8901262010016"
|
||||
assert row["gtin"] == "8901262010016"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. It never damages what the merchant supplied
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_an_invalid_barcode_is_left_exactly_as_typed():
|
||||
"""The placeholder `8900000000000.0` sat in 38 production rows. It is not a
|
||||
barcode, but it is also not this stage's to delete - a value visibly wrong
|
||||
is findable, a value silently blanked is not."""
|
||||
row = run({"product_name": "Cadbury 5 Star 24g", "barcode": "8900000000000.0"})
|
||||
|
||||
assert row["barcode"] == "8900000000000.0"
|
||||
assert row.get("gtin") is None
|
||||
assert row.get("ean13") is None
|
||||
assert row.get("barcode_type") is None
|
||||
|
||||
|
||||
def test_a_failed_checksum_is_recorded_in_provenance_not_in_the_data():
|
||||
"""13 digits of the right length but the wrong check digit."""
|
||||
row = run({"product_name": "Probe", "barcode": "8901233018363"})
|
||||
|
||||
assert row["barcode"] == "8901233018363"
|
||||
assert row["field_sources"]["barcode"]["method"] == "unvalidated"
|
||||
assert "failed checksum" in row["field_sources"]["barcode"]["note"]
|
||||
|
||||
|
||||
def test_a_row_with_no_barcode_is_untouched():
|
||||
row = run({"product_name": "Amul Butter 100g", "category": "Dairy"})
|
||||
|
||||
assert "gtin" not in row
|
||||
assert "field_sources" not in row
|
||||
|
||||
|
||||
def test_it_never_invents_a_barcode():
|
||||
"""The stage has no source and no network. If the row has no barcode, it
|
||||
cannot acquire one here - that is BarcodeEnrichmentStage's job."""
|
||||
row = run({"product_name": "Unknown Product 1kg", "barcode": ""})
|
||||
|
||||
assert not row.get("barcode")
|
||||
assert not row.get("gtin")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. It does not overstate what it knows
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_a_sheet_barcode_is_never_marked_verified():
|
||||
row = run({"product_name": "Probe 100g", "barcode": "8901233018362"})
|
||||
|
||||
assert row["barcode_verified"] is False
|
||||
assert row["barcode_lookup_status"] == "sheet_validated"
|
||||
assert row["barcode_source"] == "sheet"
|
||||
|
||||
|
||||
def test_the_derived_fields_are_flagged_derived_not_sourced():
|
||||
"""gtin/ean13/upc are arithmetic on the barcode. Recording them as
|
||||
`sourced` would claim a lookup confirmed them, which is the exact
|
||||
overstatement the provenance map exists to prevent."""
|
||||
row = run({"product_name": "Probe 100g", "barcode": "8901233018362"})
|
||||
|
||||
for field in ("gtin", "ean13", "upc", "barcode_type"):
|
||||
assert row["field_sources"][field]["method"] == "derived", field
|
||||
|
||||
|
||||
def test_an_existing_source_is_not_overwritten_by_sheet():
|
||||
"""When the cascade found the barcode, its provenance is the real one and
|
||||
must survive this stage running afterwards."""
|
||||
row = run({
|
||||
"product_name": "Probe 100g",
|
||||
"barcode": "8901233018362",
|
||||
"barcode_source": "Open Food Facts",
|
||||
"barcode_verified": True,
|
||||
"barcode_lookup_status": "verified",
|
||||
})
|
||||
|
||||
assert row["barcode_source"] == "Open Food Facts"
|
||||
assert row["barcode_verified"] is True
|
||||
assert row["field_sources"]["barcode"]["method"] == "sourced"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4. Provenance accumulates across stages
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_field_sources_from_an_earlier_stage_is_merged_not_replaced():
|
||||
"""`base.apply()` assigns every key except this one. Assigning it would
|
||||
mean the last stage to run erases what every earlier stage recorded, so a
|
||||
value would end up in the database with no origin."""
|
||||
row = run({
|
||||
"product_name": "Probe 100g",
|
||||
"barcode": "8901233018362",
|
||||
"field_sources": {"hsn_code": {"method": "estimated", "source": "category"}},
|
||||
})
|
||||
|
||||
assert row["field_sources"]["hsn_code"]["method"] == "estimated"
|
||||
assert row["field_sources"]["gtin"]["method"] == "derived"
|
||||
|
||||
|
||||
def test_the_stage_never_raises_on_a_malformed_row():
|
||||
"""The EnrichmentStage contract: a stage bug degrades to "no fields added",
|
||||
never an aborted catalog row."""
|
||||
for barcode in (None, "", "abc", 12345, [], {"nested": 1}, 8901233018362):
|
||||
row = run({"product_name": "Probe", "barcode": barcode})
|
||||
assert isinstance(row, dict)
|
||||
236
tests/test_barcode_name_containment.py
Normal file
236
tests/test_barcode_name_containment.py
Normal file
@@ -0,0 +1,236 @@
|
||||
"""Accepting an Open Food Facts record whose name is shorter than ours.
|
||||
|
||||
THE FAILURE THIS FILE EXISTS FOR
|
||||
--------------------------------
|
||||
Running `scripts/backfill_nutrition_from_barcodes` over the production catalog
|
||||
on 2026-09-08 reported, of 300 barcoded rows:
|
||||
|
||||
not a valid GTIN 37
|
||||
not in Open Food Facts 100
|
||||
matched but empty 11
|
||||
found, WRONG PRODUCT 149 <-- this file
|
||||
would write 3
|
||||
|
||||
The 149 were not wrong products. The barcode resolved perfectly; Open Food
|
||||
Facts simply stores a short name where we store a long one:
|
||||
|
||||
ours "Nestle Munch 8.9g" OFF "Munch" similarity 0.332
|
||||
ours "Coca-Cola Maaza 750ml" OFF "Maaza" similarity 0.304
|
||||
ours "Cadbury Perk 22 g" OFF "Perk" similarity 0.302
|
||||
|
||||
`name_similarity` divides token overlap by the TARGET's token count, so a
|
||||
one-token candidate against a three-token target cannot exceed about 0.33 no
|
||||
matter how correct it is.
|
||||
|
||||
WHY THE FIX IS NOT A LOWER THRESHOLD
|
||||
The same run correctly rejected these, which sit BELOW the containment cases
|
||||
but not far enough below to be separable by a number:
|
||||
|
||||
ours "Pepsico Lays 1kg" OFF "Spanish tomato tango" 0.133
|
||||
ours "Coca-Cola Fanta 750ml" OFF "Orange" 0.089
|
||||
|
||||
and `settings.py:449-477` records the measurement that raised this floor to
|
||||
0.78 in the first place (at 0.45: 15 accepted / 8 wrong; at 0.78: 2 / 0).
|
||||
Lowering it re-admits exactly what it was raised to exclude.
|
||||
|
||||
Containment separates the groups structurally instead. It also has to reject
|
||||
two cases a naive substring check would wave through, both of which are real
|
||||
Open Food Facts titles: the bare brand name ("Colgate", "godrej"), which would
|
||||
otherwise attach to every product of that brand, and a same-brand sibling
|
||||
("Dairy Milk Silk" against our "Cadbury Dairy Milk").
|
||||
|
||||
The relaxation is OFF by default and enabled on exactly one call site -
|
||||
`fetch_verified_nutrition_by_barcode` - because there the barcode has already
|
||||
established identity and there are no competing candidates. On the search path,
|
||||
where many candidates compete and the name is the only discriminator, "Munch"
|
||||
would match every Nestle product containing that word.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from app.services.enrichment.barcode.matching import is_match, name_is_contained
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate
|
||||
|
||||
|
||||
# (our stored title, what OFF calls it, brand) - all measured on 2026-09-08
|
||||
CONTAINED = [
|
||||
("Nestle Munch 8.9g", "Munch", "Nestle"),
|
||||
("Coca-Cola Maaza 750ml", "Maaza", "Coca-Cola"),
|
||||
("Cadbury Perk 22 g", "Perk", "Cadbury"),
|
||||
("Nestle Milo 25 g", "MILO", "Nestle"),
|
||||
("Cadbury Fuse 25 g", "FUSE", "Cadbury"),
|
||||
("Coca-Cola Limca 750g", "limca", "Coca-Cola"),
|
||||
("Nestle Milkybar 25g", "Milkybar", "Nestle"),
|
||||
]
|
||||
|
||||
NOT_CONTAINED = [
|
||||
("Pepsico Lays 1kg", "Spanish tomato tango", "Pepsico"),
|
||||
("Lion Dates Powder 100g", "PEPER NOTEN", "Lion Dates"),
|
||||
("Coca-Cola Fanta 750ml", "Orange", "Coca-Cola"),
|
||||
# Bare brand names. Both are real OFF product_name values.
|
||||
("Colgate Total Toothpaste 150g", "Colgate", "Colgate"),
|
||||
("Godrej No1 Soap 100g", "godrej", "Godrej"),
|
||||
# Same brand, different product - the case the name gate exists for.
|
||||
("Amul Butter 100g", "Amul Cheese", "Amul"),
|
||||
("Cadbury Dairy Milk 50g", "Dairy Milk Silk", "Cadbury"),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("ours,theirs,brand", CONTAINED)
|
||||
def test_a_short_off_name_is_recognised_as_ours(ours, theirs, brand):
|
||||
assert name_is_contained(theirs, ours, brand) is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize("ours,theirs,brand", NOT_CONTAINED)
|
||||
def test_a_different_product_is_still_refused(ours, theirs, brand):
|
||||
assert name_is_contained(theirs, ours, brand) is False
|
||||
|
||||
|
||||
def test_a_bare_brand_name_never_matches():
|
||||
""""Colgate" as a product name identifies a brand, not a product. Accepting
|
||||
it would attach one arbitrary pack's nutrition to every Colgate row."""
|
||||
assert name_is_contained("Colgate", "Colgate MaxFresh Toothpaste 150g", "Colgate") is False
|
||||
|
||||
|
||||
def test_size_tokens_do_not_decide_identity():
|
||||
"""`size_matches` has already compared the pack size by the time this is
|
||||
consulted, so a size token in our title must not make the names differ."""
|
||||
assert name_is_contained("Munch", "Nestle Munch 8.9g", "Nestle") is True
|
||||
assert name_is_contained("Munch", "Nestle Munch 38.5 g", "Nestle") is True
|
||||
|
||||
|
||||
def test_an_extra_token_in_the_candidate_breaks_containment():
|
||||
"""Containment is one-directional on purpose: every candidate token must be
|
||||
ours. "Dairy Milk Silk" carries "silk", which our "Cadbury Dairy Milk" does
|
||||
not, so it is a different product."""
|
||||
assert name_is_contained("Dairy Milk Silk", "Cadbury Dairy Milk 50g", "Cadbury") is False
|
||||
|
||||
|
||||
def test_empty_names_are_refused_rather_than_treated_as_contained():
|
||||
"""The empty set is a subset of everything - the one case where the maths
|
||||
says yes and the answer is obviously no."""
|
||||
assert name_is_contained("", "Nestle Munch 8.9g", "Nestle") is False
|
||||
assert name_is_contained("Munch", "", "Nestle") is False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The gate as a whole
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _candidate(title, brand, size=""):
|
||||
return BarcodeCandidate(barcode="8901058857245", source_name="Open Food Facts",
|
||||
candidate_title=title, candidate_brand=brand,
|
||||
candidate_size=size)
|
||||
|
||||
|
||||
def test_containment_is_off_by_default():
|
||||
"""The search path must not get this relaxation: there, many candidates
|
||||
compete and "Munch" would match every Nestle product containing it."""
|
||||
matched, _ = is_match(_candidate("Munch", "Nestle", "8.9g"),
|
||||
"Nestle", "Nestle Munch 8.9g", "8.9g",
|
||||
min_name_similarity=0.78)
|
||||
|
||||
assert matched is False
|
||||
|
||||
|
||||
def test_containment_accepts_when_explicitly_enabled():
|
||||
matched, similarity = is_match(_candidate("Munch", "Nestle", "8.9g"),
|
||||
"Nestle", "Nestle Munch 8.9g", "8.9g",
|
||||
min_name_similarity=0.78,
|
||||
barcode_is_identity=True)
|
||||
|
||||
assert matched is True
|
||||
# The reported confidence is still the honest similarity, not 1.0 - it is
|
||||
# stored on the row for audit and must not be inflated by the relaxation.
|
||||
assert similarity < 0.5
|
||||
|
||||
|
||||
def test_containment_does_not_bypass_the_brand_gate():
|
||||
"""Rules 1-3 still apply. A containment name match with the wrong brand is
|
||||
still a wrong product."""
|
||||
matched, _ = is_match(_candidate("Munch", "Britannia", "8.9g"),
|
||||
"Nestle", "Nestle Munch 8.9g", "8.9g",
|
||||
min_name_similarity=0.78, barcode_is_identity=True)
|
||||
|
||||
assert matched is False
|
||||
|
||||
|
||||
def test_a_conflicting_size_is_still_refused():
|
||||
"""A quantity that is PRESENT and different means our barcode is on the
|
||||
wrong row. That is exactly what the sanity check is for."""
|
||||
matched, _ = is_match(_candidate("Munch", "Nestle", "500g"),
|
||||
"Nestle", "Nestle Munch 8.9g", "8.9g",
|
||||
min_name_similarity=0.78, barcode_is_identity=True)
|
||||
|
||||
assert matched is False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The gate that actually blocked most of the 149
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_a_blank_candidate_size_no_longer_vetoes_under_a_barcode():
|
||||
"""The real blocker, found only after measuring the containment fix.
|
||||
|
||||
`size_matches` returns False whenever EITHER side is blank, and Open Food
|
||||
Facts leaves `quantity` null on a large share of records - 57 of 146 Amul
|
||||
hits. Because `is_match` applies its rules in order, that rejected these
|
||||
rows before the name rule was ever consulted, so fixing the name gate alone
|
||||
moved the measured result from 3 rows to 8 rather than to ~149.
|
||||
|
||||
A record with no quantity does not disagree with our pack size. It says
|
||||
nothing about it, and the barcode has already established identity.
|
||||
"""
|
||||
matched, _ = is_match(_candidate("Munch", "Nestle", ""),
|
||||
"Nestle", "Nestle Munch 38.5 g", "38.5 g",
|
||||
min_name_similarity=0.78, barcode_is_identity=True)
|
||||
|
||||
assert matched is True
|
||||
|
||||
|
||||
def test_a_blank_candidate_size_still_vetoes_on_the_search_path():
|
||||
"""Without a barcode a sizeless candidate is genuinely unidentifiable: it
|
||||
could be any pack of that product, and a GTIN belongs to exactly one."""
|
||||
matched, _ = is_match(_candidate("Nestle Munch", "Nestle", ""),
|
||||
"Nestle", "Nestle Munch 38.5 g", "38.5 g",
|
||||
min_name_similarity=0.45)
|
||||
|
||||
assert matched is False
|
||||
|
||||
|
||||
def test_the_size_relaxation_does_not_also_relax_the_brand_gate():
|
||||
matched, _ = is_match(_candidate("Munch", "Britannia", ""),
|
||||
"Nestle", "Nestle Munch 38.5 g", "38.5 g",
|
||||
min_name_similarity=0.78, barcode_is_identity=True)
|
||||
|
||||
assert matched is False
|
||||
|
||||
|
||||
def test_the_size_relaxation_does_not_also_relax_variant_conflicts():
|
||||
matched, _ = is_match(_candidate("Munch sugar free", "Nestle", ""),
|
||||
"Nestle", "Nestle Munch 38.5 g", "38.5 g",
|
||||
min_name_similarity=0.78, barcode_is_identity=True)
|
||||
|
||||
assert matched is False
|
||||
|
||||
|
||||
def test_containment_does_not_bypass_the_variant_conflict_gate():
|
||||
""""sugar free" on the candidate but not on ours is a different product
|
||||
however well the rest of the name contains."""
|
||||
matched, _ = is_match(_candidate("Munch sugar free", "Nestle", "8.9g"),
|
||||
"Nestle", "Nestle Munch 8.9g", "8.9g",
|
||||
min_name_similarity=0.78, barcode_is_identity=True)
|
||||
|
||||
assert matched is False
|
||||
|
||||
|
||||
def test_a_normal_high_similarity_match_is_unaffected():
|
||||
"""The relaxation is only consulted when the similarity floor fails, so it
|
||||
cannot change any decision the existing gate already made."""
|
||||
matched, similarity = is_match(_candidate("Nestle Munch", "Nestle", "8.9g"),
|
||||
"Nestle", "Nestle Munch", "8.9g",
|
||||
min_name_similarity=0.78)
|
||||
|
||||
assert matched is True
|
||||
assert similarity >= 0.78
|
||||
276
tests/test_brand_table_enrichment_columns.py
Normal file
276
tests/test_brand_table_enrichment_columns.py
Normal file
@@ -0,0 +1,276 @@
|
||||
"""The enrichment columns on the brand tables, and the write path that fills them.
|
||||
|
||||
THE FAILURE THIS FILE EXISTS FOR
|
||||
--------------------------------
|
||||
The barcode stage returns nine fields - `BarcodeResult.as_product_fields()` -
|
||||
and `upsert_brand_products` named two of them. The other seven were computed on
|
||||
every run and then dropped on the floor by the writer. Measured against the
|
||||
production database on 2026-09-08, before this landed:
|
||||
|
||||
upc 0.0% (column existed on 7 of 56 tables, never written)
|
||||
ean13 6.4% (added out-of-band, never written by any code)
|
||||
gtin 8.7%
|
||||
barcode 18.4%
|
||||
|
||||
The same was true of the HSN/GST stage: it computes `gst_percent`, `tax_amount`
|
||||
and `hsn_gst_needs_review`, and `_to_storage_row` projected none of them.
|
||||
|
||||
Three distinct things have to hold for a value to survive, and each of them
|
||||
broke independently at some point, so each gets a test here:
|
||||
|
||||
1. The column has to EXIST. `_ensure_columns` is the only migration mechanism
|
||||
in the repo - there is no Alembic and no migrations directory - so a column
|
||||
missing from its `col_defs` dict never appears on the 56 brand tables that
|
||||
already exist.
|
||||
|
||||
2. The INSERT has to NAME it. Adding the column is not enough; that is exactly
|
||||
how seven tables ended up carrying `gtin` and `ean13` columns that no code
|
||||
ever wrote a value into.
|
||||
|
||||
3. A later re-seed must not BLANK it. `ON CONFLICT DO UPDATE SET x =
|
||||
EXCLUDED.x` overwrites with whatever arrived, and the seed loader,
|
||||
`user_products._build_product_dict` and `brand_sync`'s re-seed all build a
|
||||
product dict from a spreadsheet with no enrichment keys in it - so their
|
||||
EXCLUDED values are NULL. This is the same trap `test_brand_table_scores`
|
||||
documents for the score columns, which is why those are omitted from the
|
||||
statement entirely. These columns cannot be omitted (the pipeline is what
|
||||
writes them), so they use COALESCE instead.
|
||||
|
||||
`barcode` and `barcode_type` are COALESCEd alongside the seven, though they are
|
||||
older columns. Proven against a live table: a bare re-seed set `barcode` to
|
||||
NULL while COALESCE kept `gtin` and `ean13`, leaving a row that claimed a GTIN
|
||||
with no barcode. A half-erased identity is worse than either whole state, and
|
||||
the rest of the barcode package already promises never to erase one -
|
||||
`stage.py` skips a row that has a barcode and `enrichment/base.py` refuses to
|
||||
blank a held value. The upsert was the one place that still could.
|
||||
|
||||
No database is involved: the cursor is a recorder, so the assertions are about
|
||||
the exact SQL sent.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from app.services import brand_sync, vector_store
|
||||
|
||||
|
||||
# The columns this change added, with the type each MUST be created as.
|
||||
#
|
||||
# THE TYPES ARE ADOPTED, NOT CHOSEN. Seven brand tables already carried these
|
||||
# columns before any code created them, and `ADD COLUMN IF NOT EXISTS` does not
|
||||
# reconcile a type difference - it silently leaves the old table alone. Picking
|
||||
# a "better" type here (NUMERIC for the tax figures, DOUBLE PRECISION for the
|
||||
# epoch) would leave 7 tables permanently disagreeing with 49. These are what
|
||||
# `information_schema` reported for those 7 tables on 2026-09-08.
|
||||
PIPELINE_OWNED = {
|
||||
"gtin": "TEXT",
|
||||
"ean13": "TEXT",
|
||||
"upc": "TEXT",
|
||||
"barcode_source": "TEXT",
|
||||
"barcode_verified": "BOOLEAN",
|
||||
"barcode_lookup_status": "TEXT",
|
||||
"barcode_last_updated": "TIMESTAMP",
|
||||
"gst_percent": "REAL",
|
||||
"tax_amount": "REAL",
|
||||
"hsn_gst_needs_review": "BOOLEAN",
|
||||
"field_sources": "JSONB",
|
||||
}
|
||||
|
||||
# Written only by nutrition_score_sync, never by the INSERT - the same
|
||||
# category as nutrition_score / health_score.
|
||||
MIRROR_OWNED = {"nutrients_per_100g": "JSONB"}
|
||||
|
||||
|
||||
class MigrationCursor:
|
||||
"""A brand table that exists but has none of the current columns."""
|
||||
|
||||
def __init__(self):
|
||||
self.statements = []
|
||||
|
||||
def execute(self, sql, params=None):
|
||||
self.statements.append(" ".join(str(sql).split()))
|
||||
|
||||
def fetchall(self):
|
||||
return [] # no existing columns -> every column is missing
|
||||
|
||||
|
||||
def _insert_statement() -> str:
|
||||
"""The INSERT ... ON CONFLICT text, whitespace-normalised."""
|
||||
source = vector_store.upsert_brand_products.__doc__ or ""
|
||||
# The statement is built inline, so read it off the module source rather
|
||||
# than reaching into a closure.
|
||||
import inspect
|
||||
body = inspect.getsource(vector_store.upsert_brand_products)
|
||||
match = re.search(r"INSERT INTO \{table_name\}.*?updated_at = CURRENT_TIMESTAMP",
|
||||
body, re.S)
|
||||
assert match, "could not locate the INSERT statement in upsert_brand_products"
|
||||
return " ".join(match.group(0).split())
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. The columns exist
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_the_migration_adds_every_enrichment_column():
|
||||
cur = MigrationCursor()
|
||||
|
||||
vector_store._ensure_columns(cur, "brand_cadbury")
|
||||
|
||||
for col, col_type in {**PIPELINE_OWNED, **MIRROR_OWNED}.items():
|
||||
assert (f"ALTER TABLE brand_cadbury ADD COLUMN IF NOT EXISTS "
|
||||
f"{col} {col_type}") in cur.statements, col
|
||||
|
||||
|
||||
def test_the_types_match_the_tables_that_already_had_these_columns():
|
||||
"""A wrong type here is invisible until a write fails on one of the seven
|
||||
pre-existing tables, because ADD COLUMN IF NOT EXISTS skips them silently.
|
||||
|
||||
`barcode_last_updated` is the one most likely to be "corrected" by a future
|
||||
reader: `BarcodeResult` carries a float epoch, so DOUBLE PRECISION looks
|
||||
right. The column on disk is TIMESTAMP, and vector_store._epoch_to_timestamp
|
||||
is what bridges the two.
|
||||
"""
|
||||
assert vector_store._ensure_columns.__doc__ is not None
|
||||
import inspect
|
||||
body = inspect.getsource(vector_store._ensure_columns)
|
||||
|
||||
assert '"barcode_last_updated": "TIMESTAMP"' in body
|
||||
assert '"gst_percent": "REAL"' in body
|
||||
assert '"tax_amount": "REAL"' in body
|
||||
|
||||
|
||||
def test_the_create_table_ddl_carries_them_too():
|
||||
"""`col_defs` migrates existing tables; the DDL is what a brand table
|
||||
created from scratch gets. A column in one but not the other means a new
|
||||
brand's table differs from every other brand's."""
|
||||
ddl = vector_store.get_brand_table_ddl("Cadbury")
|
||||
|
||||
for col in {**PIPELINE_OWNED, **MIRROR_OWNED}:
|
||||
assert re.search(rf"^\s*{col}\s", ddl, re.M), col
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. The INSERT names them - and does not name the mirror-owned ones
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_the_insert_names_every_pipeline_owned_column():
|
||||
statement = _insert_statement()
|
||||
column_list = statement.split("VALUES")[0]
|
||||
|
||||
for col in PIPELINE_OWNED:
|
||||
assert re.search(rf"[(,] ?{col}[,)]", column_list), col
|
||||
|
||||
|
||||
def test_the_insert_does_not_name_the_mirror_owned_columns():
|
||||
"""Same rule as nutrition_score / health_score: a column this statement
|
||||
never names is a column it cannot damage."""
|
||||
statement = _insert_statement()
|
||||
|
||||
for col in MIRROR_OWNED:
|
||||
assert col not in statement, col
|
||||
|
||||
|
||||
def test_the_placeholder_count_matches_the_column_count():
|
||||
"""An arity mismatch here is a runtime error on every single write, so it
|
||||
is worth catching at import time rather than on the next upload."""
|
||||
statement = _insert_statement()
|
||||
match = re.search(r"INSERT INTO \{table_name\} \((.*?)\) VALUES \((.*?)\)", statement)
|
||||
columns = [c.strip() for c in match.group(1).split(",") if c.strip()]
|
||||
|
||||
assert len(columns) == match.group(2).count("%s")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. A re-seed cannot blank them
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_every_enrichment_column_is_coalesced_on_conflict():
|
||||
"""This is the guard that makes "a re-seed wipes the enrichment"
|
||||
structurally impossible rather than merely unlikely."""
|
||||
statement = _insert_statement()
|
||||
do_update = statement.split("DO UPDATE SET", 1)[1]
|
||||
|
||||
for col in PIPELINE_OWNED:
|
||||
if col == "field_sources":
|
||||
continue # merged, asserted separately below
|
||||
assert (f"{col} = COALESCE(EXCLUDED.{col}, {{table_name}}.{col})"
|
||||
in do_update), col
|
||||
|
||||
|
||||
def test_the_barcode_pair_is_coalesced_with_its_identity_group():
|
||||
"""`barcode` and `barcode_type` predate this change but belong to the same
|
||||
identity group as gtin/ean13/upc. Leaving them on plain EXCLUDED produced a
|
||||
row with a GTIN and no barcode after a bare re-seed."""
|
||||
do_update = _insert_statement().split("DO UPDATE SET", 1)[1]
|
||||
|
||||
for col in ("barcode", "barcode_type"):
|
||||
assert (f"{col} = COALESCE(EXCLUDED.{col}, {{table_name}}.{col})"
|
||||
in do_update), col
|
||||
|
||||
|
||||
def test_field_sources_is_merged_rather_than_replaced():
|
||||
"""A run that learns the provenance of one field must not drop what is
|
||||
already known about the others, so this one is `||`, not COALESCE."""
|
||||
do_update = _insert_statement().split("DO UPDATE SET", 1)[1]
|
||||
|
||||
# Read off the module source, so the f-string's escaped braces are still
|
||||
# doubled here - `'{{}}'` is what renders as the SQL literal `'{}'`.
|
||||
assert "field_sources = COALESCE({table_name}.field_sources, '{{}}'::jsonb) " \
|
||||
"|| COALESCE(EXCLUDED.field_sources, '{{}}'::jsonb)" in do_update
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4. The seed export carries them
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_the_seed_export_carries_every_new_column():
|
||||
"""`export_brand_to_seed_file` rebuilds a product from EXPORT_COLUMNS and
|
||||
replaces the whole dict. A column missing from that tuple is stripped out
|
||||
of the catalog file on export - which is precisely why
|
||||
scripts/backfill_barcodes_from_off.py refuses to call that helper today."""
|
||||
for col in {**PIPELINE_OWNED, **MIRROR_OWNED}:
|
||||
assert col in brand_sync.EXPORT_COLUMNS, col
|
||||
|
||||
|
||||
def test_the_export_coerces_values_json_dumps_would_refuse():
|
||||
"""`barcode_last_updated` comes back from psycopg as a datetime and
|
||||
`field_sources` as a dict. json.dumps refuses the first outright and chokes
|
||||
on a Decimal nested in the second."""
|
||||
from datetime import datetime
|
||||
from decimal import Decimal
|
||||
|
||||
assert brand_sync._jsonable(datetime(2026, 9, 8, 10, 51, 42)) == "2026-09-08T10:51:42"
|
||||
assert brand_sync._jsonable({"barcode": {"confidence": Decimal("0.91")}}) == {
|
||||
"barcode": {"confidence": 0.91}
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 5. The epoch/timestamp bridge
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_the_timestamp_bridge_accepts_every_shape_that_reaches_it():
|
||||
"""Three writers feed this column and they disagree about the type:
|
||||
BarcodeResult emits a float epoch, a re-seeded catalog file carries the
|
||||
ISO-8601 string brand_sync exported, and a DB read hands back a datetime.
|
||||
All three have to load or a round-trip drops the value it just wrote."""
|
||||
from datetime import datetime
|
||||
|
||||
assert vector_store._epoch_to_timestamp(1788773811.0) == datetime.fromtimestamp(1788773811.0)
|
||||
assert vector_store._epoch_to_timestamp("2026-09-08T10:51:42") == datetime(2026, 9, 8, 10, 51, 42)
|
||||
assert vector_store._epoch_to_timestamp(datetime(2026, 1, 1)) == datetime(2026, 1, 1)
|
||||
assert vector_store._epoch_to_timestamp("not a date") is None
|
||||
assert vector_store._epoch_to_timestamp("") is None
|
||||
assert vector_store._epoch_to_timestamp(None) is None
|
||||
|
||||
|
||||
def test_the_numeric_coercion_refuses_rather_than_raises():
|
||||
"""A malformed tax figure must degrade to "no figure stored" and never
|
||||
abort a whole batch's write."""
|
||||
assert vector_store._to_numeric_or_none("18%") == 18.0
|
||||
assert vector_store._to_numeric_or_none("₹1,250.50") == 1250.50
|
||||
assert vector_store._to_numeric_or_none(12.5) == 12.5
|
||||
assert vector_store._to_numeric_or_none("not a number") is None
|
||||
assert vector_store._to_numeric_or_none(None) is None
|
||||
# bool is an int subclass; True must not become 1.0 in a NUMERIC column
|
||||
assert vector_store._to_numeric_or_none(True) is None
|
||||
159
tests/test_content_enrichment_stage.py
Normal file
159
tests/test_content_enrichment_stage.py
Normal file
@@ -0,0 +1,159 @@
|
||||
"""The stage that fills `highlights` and `nutrients` on an uploaded row.
|
||||
|
||||
THE FAILURE THIS FILE EXISTS FOR
|
||||
--------------------------------
|
||||
`catalog_engine.generate_product_highlights` and `generate_nutrients_info` have
|
||||
existed for a long time, and `brand_discovery._build_product` calls both. The
|
||||
store-catalog pipeline never did - `_to_storage_row` passed through whatever
|
||||
the sheet carried, and a colleague's sheet carries neither column. So every
|
||||
single uploaded row landed `highlights=[]` and `nutrients=[]`.
|
||||
|
||||
That is why those columns look healthy in aggregate (95.3% / 69.7% measured on
|
||||
2026-09-08) while being empty for exactly the rows this work is about: the
|
||||
percentages come from the older brand-discovery path.
|
||||
|
||||
The second thing this file pins is the consumability gate.
|
||||
`generate_nutrients_info` matches category keywords, so without a gate a Hair
|
||||
Care row can acquire "Vitamin B Complex - Energy". Shampoo has no nutrients.
|
||||
`scripts/purge_non_consumable_nutrition.py` exists because this already
|
||||
happened once at the `nutrition_facts` level; the display column needs the same
|
||||
refusal, and it must record `not_applicable` rather than leave a silent blank -
|
||||
a permanent unexplained gap is what eventually gets "fixed" by fabricating.
|
||||
|
||||
No network, no database, no LLM: the generators are keyword functions over the
|
||||
row dict.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
|
||||
import pytest
|
||||
|
||||
from app.services.enrichment.content.stage import ContentEnrichmentStage, _has_entries
|
||||
|
||||
|
||||
def run(product, brand="Cadbury"):
|
||||
return asyncio.run(ContentEnrichmentStage().apply(dict(product), brand))
|
||||
|
||||
|
||||
FOOD_ROW = {
|
||||
"product_name": "Cadbury Dairy Milk 50g",
|
||||
"title": "Cadbury Dairy Milk",
|
||||
"category": "Chocolates",
|
||||
"size": "50g",
|
||||
"description": "Smooth milk chocolate bar.",
|
||||
}
|
||||
|
||||
SHAMPOO_ROW = {
|
||||
"product_name": "Dove Daily Shine Shampoo 340ml",
|
||||
"title": "Dove Daily Shine Shampoo",
|
||||
"category": "Hair Care",
|
||||
"size": "340ml",
|
||||
"description": "Nourishing shampoo for daily use.",
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# It fills what the pipeline used to leave empty
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_an_uploaded_food_row_gets_highlights():
|
||||
row = run(FOOD_ROW)
|
||||
|
||||
assert row["highlights"], "every upload landed highlights=[] before this stage"
|
||||
assert all(isinstance(h, str) and h.strip() for h in row["highlights"])
|
||||
|
||||
|
||||
def test_an_uploaded_food_row_gets_nutrients():
|
||||
row = run(FOOD_ROW)
|
||||
|
||||
assert row["nutrients"]
|
||||
|
||||
|
||||
def test_highlights_are_flagged_derived_not_sourced():
|
||||
"""They are marketing copy computed from fields we already hold. Calling
|
||||
them sourced would claim something confirmed them."""
|
||||
row = run(FOOD_ROW)
|
||||
|
||||
assert row["field_sources"]["highlights"]["method"] == "derived"
|
||||
|
||||
|
||||
def test_the_keyword_nutrients_are_flagged_estimated():
|
||||
"""They are category guesses standing in until a real lookup succeeds, and
|
||||
the mirror from nutrition_facts overwrites them when one does."""
|
||||
row = run(FOOD_ROW)
|
||||
|
||||
assert row["field_sources"]["nutrients"]["method"] == "estimated"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The consumability gate
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_a_shampoo_gets_no_nutrients():
|
||||
"""The real defect: keyword matching gave personal-care rows entries like
|
||||
"Vitamin B Complex - Energy"."""
|
||||
row = run(SHAMPOO_ROW, brand="Dove")
|
||||
|
||||
assert not row.get("nutrients")
|
||||
|
||||
|
||||
def test_a_shampoo_records_not_applicable_rather_than_a_silent_blank():
|
||||
"""A gap nobody can explain is the one somebody eventually fills with
|
||||
invented data. The coverage report reads this to exclude the row from the
|
||||
nutrition denominator instead of reporting it missing forever."""
|
||||
row = run(SHAMPOO_ROW, brand="Dove")
|
||||
|
||||
assert row["field_sources"]["nutrients"]["method"] == "not_applicable"
|
||||
|
||||
|
||||
def test_a_shampoo_still_gets_highlights():
|
||||
"""Non-consumable rules out nutrition, not description. A shampoo has
|
||||
perfectly good highlights."""
|
||||
row = run(SHAMPOO_ROW, brand="Dove")
|
||||
|
||||
assert row["highlights"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# It fills blanks only
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_values_the_sheet_supplied_are_never_overwritten():
|
||||
"""The pipeline's first rule: the store's own data is authoritative."""
|
||||
row = run({**FOOD_ROW,
|
||||
"highlights": ["Fairtrade cocoa"],
|
||||
"nutrients": ["Protein 7.3 g per 100 g"]})
|
||||
|
||||
assert row["highlights"] == ["Fairtrade cocoa"]
|
||||
assert row["nutrients"] == ["Protein 7.3 g per 100 g"]
|
||||
|
||||
|
||||
def test_a_column_of_empty_strings_counts_as_blank():
|
||||
"""The spreadsheet parser produces [''] from a column that exists with no
|
||||
value in it. Treating that as "already filled" keeps the row [''] forever."""
|
||||
assert _has_entries([""]) is False
|
||||
assert _has_entries(["", " "]) is False
|
||||
assert _has_entries(["Real"]) is True
|
||||
assert _has_entries([]) is False
|
||||
|
||||
row = run({**FOOD_ROW, "highlights": [""]})
|
||||
assert row["highlights"] != [""]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The stage contract
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_provenance_from_an_earlier_stage_survives():
|
||||
row = run({**FOOD_ROW,
|
||||
"field_sources": {"gtin": {"method": "derived"}}})
|
||||
|
||||
assert row["field_sources"]["gtin"]["method"] == "derived"
|
||||
assert row["field_sources"]["highlights"]["method"] == "derived"
|
||||
|
||||
|
||||
def test_the_stage_never_raises_on_a_malformed_row():
|
||||
for product in ({}, {"product_name": None}, {"category": 123},
|
||||
{"title": "", "category": None, "highlights": "not a list"}):
|
||||
assert isinstance(run(product), dict)
|
||||
177
tests/test_no_fabricated_identifiers.py
Normal file
177
tests/test_no_fabricated_identifiers.py
Normal file
@@ -0,0 +1,177 @@
|
||||
"""Defaults that invent a regulatory identifier or a commercial claim.
|
||||
|
||||
THE FAILURE THIS FILE EXISTS FOR
|
||||
--------------------------------
|
||||
`user_products._build_product_dict` filled two blanks with constants:
|
||||
|
||||
fssai_license = req.fssai_license or sample.get("fssai_license") or "10012042000244"
|
||||
providers = req.providers or list(sample.get("providers") or
|
||||
["Amazon", "Flipkart", "BigBasket", "Jiomart", "Blinkit", "Zepto"])
|
||||
|
||||
The first constant is not a placeholder. `10012042000244` is Lion Dates' real,
|
||||
registered FSSAI licence - it is still in `brand_registry.FSSAI_LICENSES` under
|
||||
that brand. Every product uploaded for a brand with no existing row was stamped
|
||||
with it, which attributes legal responsibility for that food to a business that
|
||||
never made it. It reached production at least once:
|
||||
`scripts/merge_haldiram.py` exists specifically to strip it back off
|
||||
`brand_haldirams`.
|
||||
|
||||
The second asserts a product is stocked by six named marketplaces on the basis
|
||||
of nothing at all.
|
||||
|
||||
Both are now sourced from the brand's own rows, or left empty. The tests below
|
||||
pin three things:
|
||||
|
||||
1. The literal constants are gone from the module.
|
||||
2. An unknown brand gets NO licence rather than someone else's.
|
||||
3. Consensus refuses to answer when a brand's own rows disagree - because at
|
||||
that point one of them is already wrong and a tie-break would just be
|
||||
picking which product to mislabel.
|
||||
|
||||
No database: `consensus_value` and `fssai_for_brand` take the rows as an
|
||||
argument, which is what makes them testable at all.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
import inspect
|
||||
|
||||
from app.api.routers import user_products
|
||||
from app.services.enrichment.catalog_consensus import consensus_value, fssai_for_brand
|
||||
|
||||
|
||||
LION_DATES_LICENCE = "10012042000244"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. The constants are gone
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _executable_string_literals(module) -> list:
|
||||
"""Every string constant the module can actually evaluate.
|
||||
|
||||
Docstrings and comments are excluded deliberately: the fix's own comment
|
||||
has to be free to name the constant it removed, or the explanation of why
|
||||
the bug mattered cannot be written down next to the code that had it.
|
||||
"""
|
||||
tree = ast.parse(inspect.getsource(module))
|
||||
|
||||
docstrings = set()
|
||||
for node in ast.walk(tree):
|
||||
if isinstance(node, (ast.Module, ast.ClassDef, ast.FunctionDef, ast.AsyncFunctionDef)):
|
||||
body = getattr(node, "body", [])
|
||||
if (body and isinstance(body[0], ast.Expr)
|
||||
and isinstance(body[0].value, ast.Constant)
|
||||
and isinstance(body[0].value.value, str)):
|
||||
docstrings.add(id(body[0].value))
|
||||
|
||||
return [n.value for n in ast.walk(tree)
|
||||
if isinstance(n, ast.Constant) and isinstance(n.value, str)
|
||||
and id(n) not in docstrings]
|
||||
|
||||
|
||||
def test_lion_dates_licence_is_not_a_fallback_anywhere_in_the_upload_path():
|
||||
literals = _executable_string_literals(user_products)
|
||||
|
||||
assert LION_DATES_LICENCE not in literals, (
|
||||
"A real registered FSSAI licence must never appear as a default. "
|
||||
"It belongs to Lion Dates and to no other brand."
|
||||
)
|
||||
|
||||
|
||||
def test_the_six_marketplace_default_is_gone():
|
||||
literals = _executable_string_literals(user_products)
|
||||
|
||||
for marketplace in ("Blinkit", "BigBasket", "Jiomart", "Zepto"):
|
||||
assert marketplace not in literals, (
|
||||
f"{marketplace} appears as an evaluable literal. Listing "
|
||||
f"marketplaces nobody verified is a false availability claim."
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. An unknown brand gets nothing
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_an_unknown_brand_with_no_rows_gets_no_licence():
|
||||
value, source = fssai_for_brand("Entirely Unknown Brand", rows=[])
|
||||
|
||||
assert value is None
|
||||
assert source == "unknown"
|
||||
|
||||
|
||||
def test_a_registry_brand_still_gets_its_mapped_licence():
|
||||
"""The curated map remains the first and best source."""
|
||||
value, source = fssai_for_brand("Lion Dates", rows=[])
|
||||
|
||||
assert value == LION_DATES_LICENCE
|
||||
assert source == "brand_registry"
|
||||
|
||||
|
||||
def test_an_unmapped_brand_inherits_from_its_own_agreeing_rows():
|
||||
"""400 Britannia rows carrying one licence is good evidence for the 401st -
|
||||
unlike a constant, this value genuinely belongs to the brand."""
|
||||
rows = [{"fssai_license": "11223344556677"} for _ in range(20)]
|
||||
|
||||
value, source = fssai_for_brand("Some Unmapped Brand", rows=rows)
|
||||
|
||||
assert value == "11223344556677"
|
||||
assert source == "catalog_consensus"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. Consensus declines rather than guesses
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_conflicting_licences_propagate_nothing():
|
||||
"""Two different licences on one brand means one is already wrong. Picking
|
||||
the more common one would just spread whichever error is ahead."""
|
||||
rows = ([{"fssai_license": "11111111111111"}] * 10 +
|
||||
[{"fssai_license": "22222222222222"}] * 8)
|
||||
|
||||
value, source = fssai_for_brand("Conflicted Brand", rows=rows)
|
||||
|
||||
assert value is None
|
||||
assert source == "unknown"
|
||||
|
||||
|
||||
def test_a_single_dissenting_row_does_not_veto_a_clear_majority():
|
||||
"""One bad row among many should not block the other 19 from being useful."""
|
||||
rows = [{"fssai_license": "11111111111111"}] * 19 + [{"fssai_license": "99999999999999"}]
|
||||
|
||||
value, _ = consensus_value("fssai_license", rows)
|
||||
|
||||
assert value == "11111111111111"
|
||||
|
||||
|
||||
def test_too_few_rows_is_not_a_consensus():
|
||||
"""Two rows agreeing proves nothing about a third."""
|
||||
value, why = consensus_value("fssai_license", [{"fssai_license": "1"}, {"fssai_license": "1"}])
|
||||
|
||||
assert value is None
|
||||
assert why["reason"] == "too few populated rows"
|
||||
|
||||
|
||||
def test_blank_values_are_not_counted_as_agreement():
|
||||
"""A column that is empty on every row must not come back as a consensus of
|
||||
empties - that would read as "the brand agrees there is no licence"."""
|
||||
value, _ = consensus_value("fssai_license", [{"fssai_license": None}] * 20)
|
||||
|
||||
assert value is None
|
||||
|
||||
|
||||
def test_list_valued_columns_reach_consensus_too():
|
||||
"""`providers` is a TEXT[]; lists are unhashable, so the modal calculation
|
||||
has to key them as tuples or it raises."""
|
||||
rows = [{"providers": ["Amazon", "Flipkart"]}] * 10
|
||||
|
||||
value, _ = consensus_value("providers", rows)
|
||||
|
||||
assert value == ["Amazon", "Flipkart"]
|
||||
|
||||
|
||||
def test_consensus_never_raises_on_unusable_rows():
|
||||
value, why = consensus_value("providers", [{"providers": {"unhashable": ["dict"]}}] * 5)
|
||||
|
||||
assert value is None
|
||||
assert "reason" in why
|
||||
@@ -184,15 +184,36 @@ def test_a_matching_record_with_no_usable_values_is_unavailable(off_returns):
|
||||
assert facts["data_status"] == "unavailable"
|
||||
|
||||
|
||||
def test_the_floor_can_be_relaxed_per_call(off_returns):
|
||||
"""The threshold is a judgement, not a constant, so it is a parameter.
|
||||
def test_a_shorter_off_name_no_longer_needs_the_floor_relaxed(off_returns):
|
||||
""""Aachi Biryani Masala" against our "Aachi Biryani Masala 50 g" is plainly
|
||||
the same product and scores 0.716 - below the conservative 0.78 default.
|
||||
|
||||
"Aachi Biryani Masala" against our "Aachi Biryani Masala 50 g" is plainly
|
||||
the same product and scores 0.716 - below the conservative default. The
|
||||
backfill script exposes this as --min-similarity for exactly this reason.
|
||||
It used to need `min_name_similarity=0.70` to get through. It no longer
|
||||
does: this path passes `barcode_is_identity=True`, and the candidate's name
|
||||
is ours minus the pack size, so the containment rule accepts it while the
|
||||
floor stays where the measured yield table put it. See
|
||||
tests/test_barcode_name_containment.py - 149 of 300 barcoded rows were
|
||||
being refused this way.
|
||||
"""
|
||||
off_returns["body"] = {"status": 1, "product": _product(
|
||||
"Aachi Biryani Masala", "Aachi", "50 g")}
|
||||
|
||||
facts = nds.fetch_verified_nutrition_by_barcode(
|
||||
"8906021120272", "Aachi", "Aachi Biryani Masala 50 g", "50 g")
|
||||
|
||||
assert facts["data_status"] == "verified"
|
||||
|
||||
|
||||
def test_the_floor_can_still_be_relaxed_per_call(off_returns):
|
||||
"""The threshold is a judgement, not a constant, so it is a parameter, and
|
||||
the backfill script exposes it as --min-similarity.
|
||||
|
||||
"Aachi Biryani Masala Mix" carries a token ours does not, so containment
|
||||
does NOT rescue it - it is exactly the shape the floor exists to judge. It
|
||||
scores 0.703: refused at the 0.78 default, accepted at 0.70.
|
||||
"""
|
||||
off_returns["body"] = {"status": 1, "product": _product(
|
||||
"Aachi Biryani Masala Mix", "Aachi", "50 g")}
|
||||
args = ("8906021120272", "Aachi", "Aachi Biryani Masala 50 g", "50 g")
|
||||
|
||||
assert nds.fetch_verified_nutrition_by_barcode(*args)["data_status"] == "unavailable"
|
||||
|
||||
Reference in New Issue
Block a user