Valid Barcode Generation
This commit is contained in:
318
tests/test_off_bulk_backfill.py
Normal file
318
tests/test_off_bulk_backfill.py
Normal file
@@ -0,0 +1,318 @@
|
||||
"""Guards on the Open Food Facts bulk barcode backfill.
|
||||
|
||||
Every product name and barcode below is real - taken from the live catalogue
|
||||
and from what `search.openfoodfacts.org` actually returns for those brands. The
|
||||
three rules being locked down here each cost a measurable number of wrong
|
||||
barcodes when they were absent from a dry run over the real data:
|
||||
|
||||
1. THE ASYMMETRY
|
||||
`matching.name_similarity` divides the token overlap by the TARGET's token
|
||||
count only, so an OFF name that is a superset of ours scores near-perfectly.
|
||||
Uncorrected it accepted "Butter milk amul" for our "Amul Butter" at 0.882 -
|
||||
buttermilk sold as butter - and "Dairy Milk Silk Minis" for "Dairy Milk
|
||||
Silk" at 0.933. Scoring both directions and keeping the minimum drops those
|
||||
to 0.418 and 0.783 while true matches stay at 1.000, because a real match is
|
||||
symmetric by construction.
|
||||
|
||||
2. THE WRONG MARKET
|
||||
Cadbury's OFF entries are dominated by Mondelez EU codes (7622...). Eight of
|
||||
them were accepted at 1.000 - "Bournvita", "Gems", "Celebrations",
|
||||
"Bournville" - all the right product line in the wrong market, with a
|
||||
different pack and a different code. Requiring the Indian GS1 prefix 890
|
||||
replaced every one with its correct 8901233... Indian code.
|
||||
|
||||
3. THE ANONYMOUS ROW
|
||||
The catalogue holds rows titled only by brand and size ("Amul 90g", "Amul
|
||||
1kg") and OFF holds an entry named just "Amul". Normalising both to "amul"
|
||||
matched them at 1.000 and stamped a real barcode onto six products with no
|
||||
identity in common. A title that is nothing but its brand has no product
|
||||
identity, and must yield no candidates at all.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from app.services.enrichment.barcode.sources.off_bulk import (
|
||||
accept_barcode,
|
||||
brand_tokens,
|
||||
has_extra_variant_conflict,
|
||||
normalize_for_match,
|
||||
score_candidates,
|
||||
strip_sizes,
|
||||
symmetric_similarity,
|
||||
)
|
||||
|
||||
AMUL = brand_tokens("amul")
|
||||
CADBURY = brand_tokens("cadbury")
|
||||
BRITANNIA = brand_tokens("britannia")
|
||||
HUL = brand_tokens("hindustan unilever")
|
||||
|
||||
|
||||
def _hit(code, name, quantity=None):
|
||||
return {"code": code, "product_name": name, "quantity": quantity}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Normalisation
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("raw, expected", [
|
||||
("Britannia Marie Gold 250g", "Britannia Marie Gold"),
|
||||
("Amul Butter 1L", "Amul Butter"),
|
||||
("Cadbury 5 Star 9.8 g", "Cadbury 5 Star"),
|
||||
("Brit 50-50 maska chaska 40g (20)", "Brit 50-50 maska chaska"),
|
||||
("Amul Milk Powder 200mL", "Amul Milk Powder"),
|
||||
])
|
||||
def test_strip_sizes_removes_pack_sizes_and_pack_counts(raw, expected):
|
||||
assert strip_sizes(raw) == expected
|
||||
|
||||
|
||||
def test_brand_prefix_is_stripped_from_both_sides():
|
||||
"""Our names carry the brand, OFF's frequently do not - the comparison key
|
||||
has to survive either."""
|
||||
assert normalize_for_match("Britannia Marie Gold 250g", BRITANNIA) == "marie gold"
|
||||
assert normalize_for_match("Britannia Marie Gold", BRITANNIA) == "marie gold"
|
||||
assert normalize_for_match("Milk Bikis", BRITANNIA) == "milk bikis"
|
||||
|
||||
|
||||
def test_hul_noise_token_is_recovered_from_aliases():
|
||||
"""Every Hindustan Unilever alias is "hul <subbrand>", so "hul" is brand
|
||||
noise while "lux" appears once and is a real product name. Without this our
|
||||
mangled "Hindustan Unilever Hul Lux" could never reach OFF's "Lux"."""
|
||||
assert "hul" in HUL
|
||||
assert "lux" not in HUL
|
||||
assert normalize_for_match("Hindustan Unilever Hul Lux 500g", HUL) == "lux"
|
||||
assert normalize_for_match("Lux", HUL) == "lux"
|
||||
|
||||
|
||||
def test_tata_does_not_inherit_hindustan_unilever_tokens():
|
||||
"""Tata's catalogue is stored in brand_hindustan_unilever, so
|
||||
resolve_parent_brand("tata") is "hindustan unilever". Inheriting that
|
||||
parent's alias tokens would strip real words from Tata product names."""
|
||||
tata = brand_tokens("tata")
|
||||
assert "tata" in tata
|
||||
assert "hul" not in tata
|
||||
assert "unilever" not in tata
|
||||
|
||||
|
||||
def test_brand_only_title_has_no_product_identity():
|
||||
"""Rule 3. "Amul 90g" is a brand and a size, nothing else."""
|
||||
assert normalize_for_match("Amul 90g", AMUL) == ""
|
||||
assert normalize_for_match("Amul", AMUL) == ""
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Scoring
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("off_name, our_title, tokens, ceiling", [
|
||||
("Butter milk amul", "Amul Butter", AMUL, 0.70),
|
||||
("Amul Masala Buttermilk", "Amul Butter", AMUL, 0.70),
|
||||
("Dairy Milk Silk Minis", "Cadbury Dairy Milk Silk", CADBURY, 0.88),
|
||||
("dairy milk silk bubbly", "Cadbury Dairy Milk Silk", CADBURY, 0.88),
|
||||
("Amul Pro Plus", "Amul Pro", AMUL, 0.70),
|
||||
])
|
||||
def test_superset_names_are_not_auto_applied(off_name, our_title, tokens, ceiling):
|
||||
"""Rule 1: each of these scored high enough to be written before symmetric
|
||||
scoring, and each is a different retail product."""
|
||||
score = symmetric_similarity(
|
||||
normalize_for_match(off_name, tokens),
|
||||
normalize_for_match(our_title, tokens),
|
||||
)
|
||||
assert score < ceiling
|
||||
|
||||
|
||||
@pytest.mark.parametrize("off_name, our_title, tokens", [
|
||||
("Britannia Marie Gold", "Britannia Marie Gold 250g", BRITANNIA),
|
||||
("Amul ghee", "Amul Ghee 500g", AMUL),
|
||||
("amul masti dahi", "Amul Masti Dahi 1kg", AMUL),
|
||||
("Malai Paneer", "Amul Malai Paneer 200g", AMUL),
|
||||
("Cadbury 5 star", "Cadbury 5 Star 18g", CADBURY),
|
||||
])
|
||||
def test_genuine_matches_still_score_perfectly(off_name, our_title, tokens):
|
||||
"""Symmetric scoring must not cost a single true positive: a real match is
|
||||
symmetric, so both directions agree."""
|
||||
score = symmetric_similarity(
|
||||
normalize_for_match(off_name, tokens),
|
||||
normalize_for_match(our_title, tokens),
|
||||
)
|
||||
assert score == pytest.approx(1.0)
|
||||
|
||||
|
||||
def test_extra_variant_terms_are_conflicts():
|
||||
assert has_extra_variant_conflict("Dairy Milk Silk Minis", "Dairy Milk Silk")
|
||||
assert has_extra_variant_conflict("Amul Pro Plus", "Amul Pro")
|
||||
assert not has_extra_variant_conflict("Amul Lite Bread Spread", "Amul Lite Bread Spread")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Barcode acceptance - "valid 8 or 13 digit code"
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_ean13_and_gtin8_are_accepted_as_is():
|
||||
assert accept_barcode("8901063014206") == ("8901063014206", "EAN-13", None)
|
||||
assert accept_barcode("32220520") == ("32220520", "GTIN-8", None)
|
||||
|
||||
|
||||
def test_upc_a_is_widened_to_ean13_keeping_the_original():
|
||||
"""A leading zero contributes nothing to the GTIN checksum, so the padded
|
||||
code is still valid; the 12-digit original is kept for the `upc` field."""
|
||||
assert accept_barcode("036000291452") == ("0036000291452", "EAN-13", "036000291452")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("raw", [
|
||||
"12345678901234", # valid GTIN-14: a shipping carton, not a retail pack
|
||||
"8901063014207", # EAN-13 body with a wrong check digit
|
||||
"890106301420", # 12 digits, checksum fails
|
||||
"1234567", # too short
|
||||
"8900000000000.0", # the Excel-float junk already sitting in some rows
|
||||
"",
|
||||
None,
|
||||
"not-a-barcode",
|
||||
])
|
||||
def test_invalid_or_out_of_scope_codes_are_rejected(raw):
|
||||
assert accept_barcode(raw) is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# End-to-end candidate selection
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_marie_gold_matches_despite_off_having_no_quantity():
|
||||
"""The case this backfill exists for. `matching.is_match` cannot express it
|
||||
because its size gate rejects a blank size outright, and 57 of 146 Amul
|
||||
hits carry quantity: null."""
|
||||
corpus = [_hit("8901063014206", "Britannia Marie Gold", None)]
|
||||
got = score_candidates(corpus, "Britannia Marie Gold", ["250g"], BRITANNIA, 0.70)
|
||||
assert len(got) == 1
|
||||
assert got[0].barcode == "8901063014206"
|
||||
assert got[0].score == pytest.approx(1.0)
|
||||
assert got[0].size_bonus == 0
|
||||
|
||||
|
||||
def test_matching_quantity_outranks_a_better_name_score():
|
||||
"""Size never vetoes, but among acceptable candidates it decides."""
|
||||
corpus = [
|
||||
_hit("8901063014206", "Britannia Marie Gold", None),
|
||||
_hit("8901063014213", "Britannia Marie Gold", "250g"),
|
||||
]
|
||||
got = score_candidates(corpus, "Britannia Marie Gold", ["250g"], BRITANNIA, 0.70)
|
||||
assert got[0].barcode == "8901063014213"
|
||||
assert got[0].size_bonus == 1
|
||||
|
||||
|
||||
def test_foreign_market_codes_are_rejected_by_default():
|
||||
"""Rule 2. The Mondelez EU Bournvita is the right line, wrong market."""
|
||||
corpus = [_hit("7622201766269", "Bournvita", "2 kg")]
|
||||
assert score_candidates(corpus, "Cadbury Bournvita", ["500g"], CADBURY, 0.70) == []
|
||||
|
||||
allowed = score_candidates(corpus, "Cadbury Bournvita", ["500g"], CADBURY, 0.70,
|
||||
require_india_prefix=False)
|
||||
assert allowed and allowed[0].barcode == "7622201766269"
|
||||
|
||||
|
||||
def test_anonymous_brand_only_group_yields_nothing():
|
||||
"""Rule 3: "Amul 90g" against OFF's "Amul" must not produce a candidate."""
|
||||
corpus = [_hit("8901262010320", "Amul", "200gm")]
|
||||
assert score_candidates(corpus, "Amul", ["90g"], AMUL, 0.70) == []
|
||||
|
||||
|
||||
def test_buttermilk_is_never_offered_for_butter():
|
||||
corpus = [
|
||||
_hit("8901262200004", "Butter milk amul", "500ml"),
|
||||
_hit("8901262201995", "Amul Buttermilk", "310ml"),
|
||||
]
|
||||
assert score_candidates(corpus, "Amul Butter", ["500ml"], AMUL, 0.70) == []
|
||||
|
||||
|
||||
def test_ranking_is_deterministic_for_equal_scores():
|
||||
"""A rerun over an unchanged corpus must pick the same product, so the
|
||||
barcode breaks the tie."""
|
||||
corpus = [
|
||||
_hit("8901262031060", "Amul ghee", None),
|
||||
_hit("8901262031059", "Amul ghee", None),
|
||||
]
|
||||
first = score_candidates(corpus, "Amul Ghee", ["500g"], AMUL, 0.70)
|
||||
second = score_candidates(list(reversed(corpus)), "Amul Ghee", ["500g"], AMUL, 0.70)
|
||||
assert first[0].barcode == second[0].barcode == "8901262031059"
|
||||
|
||||
|
||||
def test_nameless_hits_are_ignored():
|
||||
corpus = [_hit("8901262031059", None), _hit("8901262031059", "")]
|
||||
assert score_candidates(corpus, "Amul Ghee", ["500g"], AMUL, 0.70) == []
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The JSON writer - regression on the export_brand_to_seed_file trap
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_json_writer_preserves_every_other_key(tmp_path):
|
||||
"""`brand_sync.export_brand_to_seed_file` rebuilds each product from
|
||||
EXPORT_COLUMNS and `upsert_products_into_catalog_file` then replaces the
|
||||
whole dict, which would delete `embedding`, `size`, `variant_key`,
|
||||
`gst_percent` and the seven JSON-only barcode keys. This writer mutates in
|
||||
place; that difference is the whole reason it exists.
|
||||
"""
|
||||
from scripts.backfill_barcodes_from_off import BARCODE_KEYS, write_json
|
||||
|
||||
product = {
|
||||
"image_id": "amul_amul_ghee_500g",
|
||||
"product_name": "Amul Ghee 500g",
|
||||
"title": "Amul Ghee",
|
||||
"size": "500g",
|
||||
"variant_key": "amul_amul_ghee_500g",
|
||||
"gst_percent": 12,
|
||||
"cost_price": 240.0,
|
||||
"embedding": [0.1, 0.2, 0.3],
|
||||
"barcode": None,
|
||||
}
|
||||
original = json.loads(json.dumps(product))
|
||||
path = tmp_path / "brand_catalog_amul.json"
|
||||
data = {"brand": "amul", "total_products": 1, "products": [product]}
|
||||
path.write_text(json.dumps(data), encoding="utf-8")
|
||||
|
||||
changed = write_json(path, data, {
|
||||
"amul_amul_ghee_500g": {
|
||||
"barcode": "8901262031059", "barcode_type": "EAN-13",
|
||||
"gtin": "8901262031059", "ean13": "8901262031059", "upc": None,
|
||||
"barcode_source": "openfoodfacts_bulk", "barcode_verified": False,
|
||||
"barcode_lookup_status": "name_matched", "barcode_last_updated": 1.0,
|
||||
},
|
||||
})
|
||||
assert changed == 1
|
||||
|
||||
written = json.loads(path.read_text(encoding="utf-8"))["products"][0]
|
||||
assert written["barcode"] == "8901262031059"
|
||||
for key, value in original.items():
|
||||
if key not in BARCODE_KEYS:
|
||||
assert written[key] == value, f"{key} was altered"
|
||||
assert set(written) == set(original) | set(BARCODE_KEYS)
|
||||
assert json.loads(path.read_text(encoding="utf-8"))["total_products"] == 1
|
||||
|
||||
|
||||
def test_json_writer_leaves_unmatched_products_alone(tmp_path):
|
||||
from scripts.backfill_barcodes_from_off import write_json
|
||||
|
||||
path = tmp_path / "brand_catalog_amul.json"
|
||||
data = {"brand": "amul", "products": [{"image_id": "other", "barcode": None}]}
|
||||
payload = json.dumps(data)
|
||||
path.write_text(payload, encoding="utf-8")
|
||||
|
||||
assert write_json(path, data, {"not_present": {"barcode": "8901262031059"}}) == 0
|
||||
assert path.read_text(encoding="utf-8") == payload
|
||||
|
||||
|
||||
def test_existing_barcodes_are_never_regrouped(tmp_path):
|
||||
"""Tata has 23 manually sourced barcodes; a product that already has one is
|
||||
excluded from matching entirely rather than re-looked-up."""
|
||||
from scripts.backfill_barcodes_from_off import group_by_title
|
||||
|
||||
products = [
|
||||
{"image_id": "a", "title": "Tata Salt", "barcode": "8904043901015"},
|
||||
{"image_id": "b", "title": "Tata Salt", "barcode": ""},
|
||||
{"image_id": "c", "title": "Tata Salt"},
|
||||
]
|
||||
groups = group_by_title(products)
|
||||
assert [p["image_id"] for p in groups["Tata Salt"]] == ["b", "c"]
|
||||
Reference in New Issue
Block a user