Valid Barcode Generation

This commit is contained in:
sriram
2026-09-07 15:45:25 +05:30
parent c2d7f0ca1d
commit 75dd3eb3ce
40 changed files with 11893 additions and 557 deletions

View File

@@ -0,0 +1,318 @@
"""Guards on the Open Food Facts bulk barcode backfill.
Every product name and barcode below is real - taken from the live catalogue
and from what `search.openfoodfacts.org` actually returns for those brands. The
three rules being locked down here each cost a measurable number of wrong
barcodes when they were absent from a dry run over the real data:
1. THE ASYMMETRY
`matching.name_similarity` divides the token overlap by the TARGET's token
count only, so an OFF name that is a superset of ours scores near-perfectly.
Uncorrected it accepted "Butter milk amul" for our "Amul Butter" at 0.882 -
buttermilk sold as butter - and "Dairy Milk Silk Minis" for "Dairy Milk
Silk" at 0.933. Scoring both directions and keeping the minimum drops those
to 0.418 and 0.783 while true matches stay at 1.000, because a real match is
symmetric by construction.
2. THE WRONG MARKET
Cadbury's OFF entries are dominated by Mondelez EU codes (7622...). Eight of
them were accepted at 1.000 - "Bournvita", "Gems", "Celebrations",
"Bournville" - all the right product line in the wrong market, with a
different pack and a different code. Requiring the Indian GS1 prefix 890
replaced every one with its correct 8901233... Indian code.
3. THE ANONYMOUS ROW
The catalogue holds rows titled only by brand and size ("Amul 90g", "Amul
1kg") and OFF holds an entry named just "Amul". Normalising both to "amul"
matched them at 1.000 and stamped a real barcode onto six products with no
identity in common. A title that is nothing but its brand has no product
identity, and must yield no candidates at all.
"""
from __future__ import annotations
import json
import pytest
from app.services.enrichment.barcode.sources.off_bulk import (
accept_barcode,
brand_tokens,
has_extra_variant_conflict,
normalize_for_match,
score_candidates,
strip_sizes,
symmetric_similarity,
)
AMUL = brand_tokens("amul")
CADBURY = brand_tokens("cadbury")
BRITANNIA = brand_tokens("britannia")
HUL = brand_tokens("hindustan unilever")
def _hit(code, name, quantity=None):
return {"code": code, "product_name": name, "quantity": quantity}
# ---------------------------------------------------------------------------
# Normalisation
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("raw, expected", [
("Britannia Marie Gold 250g", "Britannia Marie Gold"),
("Amul Butter 1L", "Amul Butter"),
("Cadbury 5 Star 9.8 g", "Cadbury 5 Star"),
("Brit 50-50 maska chaska 40g (20)", "Brit 50-50 maska chaska"),
("Amul Milk Powder 200mL", "Amul Milk Powder"),
])
def test_strip_sizes_removes_pack_sizes_and_pack_counts(raw, expected):
assert strip_sizes(raw) == expected
def test_brand_prefix_is_stripped_from_both_sides():
"""Our names carry the brand, OFF's frequently do not - the comparison key
has to survive either."""
assert normalize_for_match("Britannia Marie Gold 250g", BRITANNIA) == "marie gold"
assert normalize_for_match("Britannia Marie Gold", BRITANNIA) == "marie gold"
assert normalize_for_match("Milk Bikis", BRITANNIA) == "milk bikis"
def test_hul_noise_token_is_recovered_from_aliases():
"""Every Hindustan Unilever alias is "hul <subbrand>", so "hul" is brand
noise while "lux" appears once and is a real product name. Without this our
mangled "Hindustan Unilever Hul Lux" could never reach OFF's "Lux"."""
assert "hul" in HUL
assert "lux" not in HUL
assert normalize_for_match("Hindustan Unilever Hul Lux 500g", HUL) == "lux"
assert normalize_for_match("Lux", HUL) == "lux"
def test_tata_does_not_inherit_hindustan_unilever_tokens():
"""Tata's catalogue is stored in brand_hindustan_unilever, so
resolve_parent_brand("tata") is "hindustan unilever". Inheriting that
parent's alias tokens would strip real words from Tata product names."""
tata = brand_tokens("tata")
assert "tata" in tata
assert "hul" not in tata
assert "unilever" not in tata
def test_brand_only_title_has_no_product_identity():
"""Rule 3. "Amul 90g" is a brand and a size, nothing else."""
assert normalize_for_match("Amul 90g", AMUL) == ""
assert normalize_for_match("Amul", AMUL) == ""
# ---------------------------------------------------------------------------
# Scoring
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("off_name, our_title, tokens, ceiling", [
("Butter milk amul", "Amul Butter", AMUL, 0.70),
("Amul Masala Buttermilk", "Amul Butter", AMUL, 0.70),
("Dairy Milk Silk Minis", "Cadbury Dairy Milk Silk", CADBURY, 0.88),
("dairy milk silk bubbly", "Cadbury Dairy Milk Silk", CADBURY, 0.88),
("Amul Pro Plus", "Amul Pro", AMUL, 0.70),
])
def test_superset_names_are_not_auto_applied(off_name, our_title, tokens, ceiling):
"""Rule 1: each of these scored high enough to be written before symmetric
scoring, and each is a different retail product."""
score = symmetric_similarity(
normalize_for_match(off_name, tokens),
normalize_for_match(our_title, tokens),
)
assert score < ceiling
@pytest.mark.parametrize("off_name, our_title, tokens", [
("Britannia Marie Gold", "Britannia Marie Gold 250g", BRITANNIA),
("Amul ghee", "Amul Ghee 500g", AMUL),
("amul masti dahi", "Amul Masti Dahi 1kg", AMUL),
("Malai Paneer", "Amul Malai Paneer 200g", AMUL),
("Cadbury 5 star", "Cadbury 5 Star 18g", CADBURY),
])
def test_genuine_matches_still_score_perfectly(off_name, our_title, tokens):
"""Symmetric scoring must not cost a single true positive: a real match is
symmetric, so both directions agree."""
score = symmetric_similarity(
normalize_for_match(off_name, tokens),
normalize_for_match(our_title, tokens),
)
assert score == pytest.approx(1.0)
def test_extra_variant_terms_are_conflicts():
assert has_extra_variant_conflict("Dairy Milk Silk Minis", "Dairy Milk Silk")
assert has_extra_variant_conflict("Amul Pro Plus", "Amul Pro")
assert not has_extra_variant_conflict("Amul Lite Bread Spread", "Amul Lite Bread Spread")
# ---------------------------------------------------------------------------
# Barcode acceptance - "valid 8 or 13 digit code"
# ---------------------------------------------------------------------------
def test_ean13_and_gtin8_are_accepted_as_is():
assert accept_barcode("8901063014206") == ("8901063014206", "EAN-13", None)
assert accept_barcode("32220520") == ("32220520", "GTIN-8", None)
def test_upc_a_is_widened_to_ean13_keeping_the_original():
"""A leading zero contributes nothing to the GTIN checksum, so the padded
code is still valid; the 12-digit original is kept for the `upc` field."""
assert accept_barcode("036000291452") == ("0036000291452", "EAN-13", "036000291452")
@pytest.mark.parametrize("raw", [
"12345678901234", # valid GTIN-14: a shipping carton, not a retail pack
"8901063014207", # EAN-13 body with a wrong check digit
"890106301420", # 12 digits, checksum fails
"1234567", # too short
"8900000000000.0", # the Excel-float junk already sitting in some rows
"",
None,
"not-a-barcode",
])
def test_invalid_or_out_of_scope_codes_are_rejected(raw):
assert accept_barcode(raw) is None
# ---------------------------------------------------------------------------
# End-to-end candidate selection
# ---------------------------------------------------------------------------
def test_marie_gold_matches_despite_off_having_no_quantity():
"""The case this backfill exists for. `matching.is_match` cannot express it
because its size gate rejects a blank size outright, and 57 of 146 Amul
hits carry quantity: null."""
corpus = [_hit("8901063014206", "Britannia Marie Gold", None)]
got = score_candidates(corpus, "Britannia Marie Gold", ["250g"], BRITANNIA, 0.70)
assert len(got) == 1
assert got[0].barcode == "8901063014206"
assert got[0].score == pytest.approx(1.0)
assert got[0].size_bonus == 0
def test_matching_quantity_outranks_a_better_name_score():
"""Size never vetoes, but among acceptable candidates it decides."""
corpus = [
_hit("8901063014206", "Britannia Marie Gold", None),
_hit("8901063014213", "Britannia Marie Gold", "250g"),
]
got = score_candidates(corpus, "Britannia Marie Gold", ["250g"], BRITANNIA, 0.70)
assert got[0].barcode == "8901063014213"
assert got[0].size_bonus == 1
def test_foreign_market_codes_are_rejected_by_default():
"""Rule 2. The Mondelez EU Bournvita is the right line, wrong market."""
corpus = [_hit("7622201766269", "Bournvita", "2 kg")]
assert score_candidates(corpus, "Cadbury Bournvita", ["500g"], CADBURY, 0.70) == []
allowed = score_candidates(corpus, "Cadbury Bournvita", ["500g"], CADBURY, 0.70,
require_india_prefix=False)
assert allowed and allowed[0].barcode == "7622201766269"
def test_anonymous_brand_only_group_yields_nothing():
"""Rule 3: "Amul 90g" against OFF's "Amul" must not produce a candidate."""
corpus = [_hit("8901262010320", "Amul", "200gm")]
assert score_candidates(corpus, "Amul", ["90g"], AMUL, 0.70) == []
def test_buttermilk_is_never_offered_for_butter():
corpus = [
_hit("8901262200004", "Butter milk amul", "500ml"),
_hit("8901262201995", "Amul Buttermilk", "310ml"),
]
assert score_candidates(corpus, "Amul Butter", ["500ml"], AMUL, 0.70) == []
def test_ranking_is_deterministic_for_equal_scores():
"""A rerun over an unchanged corpus must pick the same product, so the
barcode breaks the tie."""
corpus = [
_hit("8901262031060", "Amul ghee", None),
_hit("8901262031059", "Amul ghee", None),
]
first = score_candidates(corpus, "Amul Ghee", ["500g"], AMUL, 0.70)
second = score_candidates(list(reversed(corpus)), "Amul Ghee", ["500g"], AMUL, 0.70)
assert first[0].barcode == second[0].barcode == "8901262031059"
def test_nameless_hits_are_ignored():
corpus = [_hit("8901262031059", None), _hit("8901262031059", "")]
assert score_candidates(corpus, "Amul Ghee", ["500g"], AMUL, 0.70) == []
# ---------------------------------------------------------------------------
# The JSON writer - regression on the export_brand_to_seed_file trap
# ---------------------------------------------------------------------------
def test_json_writer_preserves_every_other_key(tmp_path):
"""`brand_sync.export_brand_to_seed_file` rebuilds each product from
EXPORT_COLUMNS and `upsert_products_into_catalog_file` then replaces the
whole dict, which would delete `embedding`, `size`, `variant_key`,
`gst_percent` and the seven JSON-only barcode keys. This writer mutates in
place; that difference is the whole reason it exists.
"""
from scripts.backfill_barcodes_from_off import BARCODE_KEYS, write_json
product = {
"image_id": "amul_amul_ghee_500g",
"product_name": "Amul Ghee 500g",
"title": "Amul Ghee",
"size": "500g",
"variant_key": "amul_amul_ghee_500g",
"gst_percent": 12,
"cost_price": 240.0,
"embedding": [0.1, 0.2, 0.3],
"barcode": None,
}
original = json.loads(json.dumps(product))
path = tmp_path / "brand_catalog_amul.json"
data = {"brand": "amul", "total_products": 1, "products": [product]}
path.write_text(json.dumps(data), encoding="utf-8")
changed = write_json(path, data, {
"amul_amul_ghee_500g": {
"barcode": "8901262031059", "barcode_type": "EAN-13",
"gtin": "8901262031059", "ean13": "8901262031059", "upc": None,
"barcode_source": "openfoodfacts_bulk", "barcode_verified": False,
"barcode_lookup_status": "name_matched", "barcode_last_updated": 1.0,
},
})
assert changed == 1
written = json.loads(path.read_text(encoding="utf-8"))["products"][0]
assert written["barcode"] == "8901262031059"
for key, value in original.items():
if key not in BARCODE_KEYS:
assert written[key] == value, f"{key} was altered"
assert set(written) == set(original) | set(BARCODE_KEYS)
assert json.loads(path.read_text(encoding="utf-8"))["total_products"] == 1
def test_json_writer_leaves_unmatched_products_alone(tmp_path):
from scripts.backfill_barcodes_from_off import write_json
path = tmp_path / "brand_catalog_amul.json"
data = {"brand": "amul", "products": [{"image_id": "other", "barcode": None}]}
payload = json.dumps(data)
path.write_text(payload, encoding="utf-8")
assert write_json(path, data, {"not_present": {"barcode": "8901262031059"}}) == 0
assert path.read_text(encoding="utf-8") == payload
def test_existing_barcodes_are_never_regrouped(tmp_path):
"""Tata has 23 manually sourced barcodes; a product that already has one is
excluded from matching entirely rather than re-looked-up."""
from scripts.backfill_barcodes_from_off import group_by_title
products = [
{"image_id": "a", "title": "Tata Salt", "barcode": "8904043901015"},
{"image_id": "b", "title": "Tata Salt", "barcode": ""},
{"image_id": "c", "title": "Tata Salt"},
]
groups = group_by_title(products)
assert [p["image_id"] for p in groups["Tata Salt"]] == ["b", "c"]