Catalog feature updates on column fields

This commit is contained in:
sriram
2026-09-08 15:18:29 +05:30
parent 2749bee1a3
commit 10b24c6348
60 changed files with 9224 additions and 31 deletions

View File

@@ -0,0 +1,117 @@
"""Derives the rest of a product's barcode identity from the barcode itself.
WHY THIS IS A SEPARATE STAGE FROM BarcodeEnrichmentStage
--------------------------------------------------------
That stage FINDS a barcode, needs the network, and is off by default
(`ENABLE_BARCODE_LOOKUP`, see settings.py:420-423 for why). This one FINDS
NOTHING. It takes a barcode the row already has - typed into the merchant's
spreadsheet, seeded from a catalog, or just located by the cascade - and fills
in the fields that are pure arithmetic on those digits:
barcode_type from the length (classify_barcode_type)
gtin the validated digits (a GTIN is what a barcode encodes)
ean13 zero-padded UPC-A (to_ean13)
upc the digits, for UPC-A only
There is no lookup, no host, no rate limit and no failure mode beyond "these
digits are not a valid GTIN", so it needs no settings flag and costs nothing.
THE FAILURE IT ADDRESSES
Measured against production on 2026-09-08: `upc` was 0.0% filled, `ean13`
6.4%, `gtin` 8.7% - against `barcode` at 18.4%. Every one of those could
have been computed from the barcode already sitting in the same row. They
were not, because the only code that produced them was inside the disabled
network cascade, and the writer dropped them anyway.
WHY IT RUNS AFTER THE LOOKUP STAGE
So it also normalises whatever the cascade just found. The cascade already
validates, but a sheet-supplied barcode never passes through
`validate_barcode` at all today - it goes straight from the spreadsheet to
the database. This stage is the first thing that checks those digits.
WHAT IT WILL NOT DO
It will not correct, reformat or delete `barcode`. If the digits fail
checksum validation the stage returns NOTHING, leaving the merchant's value
exactly as typed - `enrichment/base.py`'s merge guard would refuse to blank
it anyway, and silently "fixing" a barcode a shop supplied would be worse
than leaving it visibly wrong. The failure is recorded in `field_sources`
so the coverage report can surface it.
"""
from __future__ import annotations
import logging
import time
from typing import Any, Dict
from app.services.enrichment.base import EnrichmentStage, StageOutcome
from app.services.enrichment.barcode.models import BarcodeType
from app.services.enrichment.barcode.validators import (
classify_barcode_type,
normalize_barcode,
to_ean13,
validate_barcode,
)
logger = logging.getLogger(__name__)
class BarcodeIdentityStage(EnrichmentStage):
"""Offline, deterministic, additive. Never raises, never erases."""
name = "barcode_identity"
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
raw = product.get("barcode")
if not str(raw or "").strip():
return StageOutcome(stage_name=self.name, fields={})
code = validate_barcode(raw)
if not code:
# Not a GTIN. Say so in the provenance rather than in the data, and
# leave `barcode` untouched.
digits = normalize_barcode(raw)
reason = (f"{len(digits)} digits is not a GTIN-8/12/13/14 length"
if digits else "no digits in the value")
return StageOutcome(
stage_name=self.name,
fields={"field_sources": {"barcode": {
"method": "unvalidated",
"source": product.get("barcode_source") or "sheet",
"note": f"failed checksum/format validation: {reason}",
}}},
error=f"barcode {raw!r} failed validation: {reason}",
)
barcode_type = classify_barcode_type(code)
fields: Dict[str, Any] = {
"barcode": code, # normalised digits, same value
"barcode_type": barcode_type.value,
"gtin": code,
"ean13": to_ean13(code), # None for GTIN-8, which is not a short EAN-13
"upc": code if barcode_type is BarcodeType.UPC_A else None,
}
# Only claim provenance we can stand behind. A barcode that arrived on
# the sheet is the merchant's assertion, not ours, and is emphatically
# not "verified" - that word is reserved for the cascade's
# brand+size+name-matched result.
if not str(product.get("barcode_source") or "").strip():
fields["barcode_source"] = "sheet"
fields["barcode_lookup_status"] = "sheet_validated"
fields["barcode_verified"] = False
fields["barcode_last_updated"] = time.time()
fields["field_sources"] = {
"barcode": {
"method": "sourced" if product.get("barcode_verified") else "asserted",
"source": product.get("barcode_source") or "sheet",
},
# These four are arithmetic on the barcode, never a lookup. Calling
# them "sourced" would overstate them.
"gtin": {"method": "derived", "source": "validators.validate_barcode"},
"ean13": {"method": "derived", "source": "validators.to_ean13"},
"upc": {"method": "derived", "source": "validators.classify_barcode_type"},
"barcode_type": {"method": "derived", "source": "validators.classify_barcode_type"},
}
return StageOutcome(stage_name=self.name, fields=fields)

View File

@@ -114,9 +114,74 @@ def name_similarity(candidate_title: str, target_title: str) -> float:
return round((overlap * 0.6) + (seq_ratio * 0.4), 3)
def name_is_contained(candidate_title: str, target_title: str,
target_brand: str = "") -> bool:
"""True when the candidate's name is our name with only brand/size removed.
WHY THIS EXISTS - measured, not theoretical
Open Food Facts stores short product names. We store long ones. Running
`backfill_nutrition_from_barcodes` over the catalog on 2026-09-08, 149
of 300 barcoded rows were rejected as "found, wrong product" when the
barcode had resolved perfectly:
"Nestle Munch 8.9g" -> OFF "Munch" similarity 0.332
"Coca-Cola Maaza 750ml" -> OFF "Maaza" similarity 0.304
"Cadbury Perk 22 g" -> OFF "Perk" similarity 0.302
`name_similarity` divides the token overlap by the TARGET's token
count, so a one-token candidate against a three-token target cannot
exceed ~0.33 however right it is.
WHY NOT JUST LOWER THE THRESHOLD
Because the same run also correctly rejected:
"Pepsico Lays 1kg" -> OFF "Spanish tomato tango" 0.133
"Coca-Cola Fanta 750ml" -> OFF "Orange" 0.089
"Lion Dates Powder 100g" -> OFF "PEPER NOTEN" 0.097
Those sit BELOW the containment cases but a threshold low enough to
admit 0.30 also admits them. The measured yield table at
settings.py:449-477 raised this floor to 0.78 for exactly that reason.
Containment separates the two groups on structure rather than on a
number: "Munch" is every token of our name minus brand and size;
"Orange" is not a subset of "Coca-Cola Fanta 750ml" at all.
THE RULE
Every token of the candidate's name must appear in the target's, once
brand tokens and size tokens are discounted, and the candidate must
carry at least one token that is not the brand. A bare brand name
("Colgate", "godrej" - both real OFF titles) therefore does NOT match,
which matters because those would otherwise attach to every product of
that brand.
"""
cand_tokens = _tokens(candidate_title)
target_tokens = _tokens(target_title)
if not cand_tokens or not target_tokens:
return False
brand_tokens = _tokens(target_brand)
# A candidate that is only the brand identifies a brand, not a product.
if not (cand_tokens - brand_tokens):
return False
# Size tokens are not identity: our title carries the pack size, OFF's
# usually does not, and `size_matches` has already checked the size
# separately by the time this is consulted.
def _meaningful(tokens):
return {t for t in tokens if not _SIZE_TOKEN_RE.fullmatch(t)}
return _meaningful(cand_tokens) <= _meaningful(target_tokens | brand_tokens)
# A token that is purely a quantity ("750ml", "8", "9g", "1kg"). Size is
# compared by `size_matches`, so it must not also decide name identity.
_SIZE_TOKEN_RE = re.compile(r"\d+(?:\.\d+)?(?:g|kg|ml|l|mg|cl|oz|gm|ltr|pcs|n)?", re.I)
def is_match(candidate: BarcodeCandidate, target_brand: str, target_title: str, target_size: str,
brand_aliases: Optional[Iterable[str]] = None,
min_name_similarity: float = 0.45) -> tuple[bool, float]:
min_name_similarity: float = 0.45,
barcode_is_identity: bool = False) -> tuple[bool, float]:
"""The combined gate a candidate must pass to be accepted:
1. Brand matches (or overlaps a known alias).
2. Pack size matches within a tight tolerance.
@@ -126,16 +191,52 @@ def is_match(candidate: BarcodeCandidate, target_brand: str, target_title: str,
product line from the same brand.
Returns (matched, confidence) - confidence is diagnostic only, stored
on the result for audit/QA but never used to override rule 1-3.
`barcode_is_identity` says the caller already knows WHICH product this is,
because it looked the candidate up BY its GTIN rather than by searching.
That changes what rules 2 and 4 are for: they stop being evidence of
identity and become sanity checks against our barcode being on the wrong
row. A sanity check cannot fail on information the source does not have, so
under this flag:
* rule 2 (size) - a BLANK candidate size no longer vetoes. Open Food
Facts leaves `quantity` null on a large share of records (57 of 146
Amul hits), and `size_matches` returns False whenever either side is
blank. A record with no quantity does not disagree with our pack size;
it says nothing about it. A quantity that is PRESENT and different
still vetoes - that is our barcode pointing at the wrong pack.
* rule 4 (name) - see `name_is_contained`.
Rules 1 and 3 are unaffected: a different brand, or a "sugar free" the
target does not have, still means a different product.
It defaults to False because every relaxation here is unsafe on the SEARCH
path, where many candidates compete and name and size are the only things
telling them apart - "Munch" with no size would match every Nestle product
containing that word. Pass True only where a single candidate was fetched
by barcode. Today that is `fetch_verified_nutrition_by_barcode` and
`scripts/backfill_nutrition_from_barcodes`, and nothing else.
Measured on 2026-09-08: of 300 barcoded catalog rows, 149 were refused as
"found, wrong product" with the barcode resolving perfectly. The name gate
was the visible symptom, but the SIZE gate rejected most of them first.
"""
if not brand_matches(candidate.candidate_brand, target_brand, brand_aliases):
return False, 0.0
if not size_matches(candidate.candidate_size, target_size):
# A blank candidate size is missing information, not a disagreement - but
# only when the barcode already established identity. On the search path a
# sizeless candidate is genuinely unidentifiable and must still be refused.
size_unknown = barcode_is_identity and not str(candidate.candidate_size or "").strip()
if not size_unknown and not size_matches(candidate.candidate_size, target_size):
return False, 0.0
if has_conflicting_variant_terms(candidate.candidate_title, target_title):
return False, 0.0
similarity = name_similarity(candidate.candidate_title, target_title)
if similarity < min_name_similarity:
if barcode_is_identity and name_is_contained(
candidate.candidate_title, target_title, target_brand):
return True, similarity
return False, similarity
return True, similarity

View File

@@ -88,6 +88,25 @@ class EnrichmentStage(ABC):
# HsnGstEnrichmentStage and BarcodeEnrichmentStage.
if outcome.fields:
for key, value in outcome.fields.items():
# `field_sources` ACCUMULATES; every other key is assigned.
#
# It is a map keyed by column name, and each stage knows the
# provenance of only the columns it filled. Assigning it like
# anything else would mean the last stage to run erases what
# every earlier stage recorded - so the barcode stage's
# provenance would vanish the moment the HSN stage ran, and
# the coverage report would show values with no origin.
#
# A shallow merge is the right depth: each key's value is one
# flat record about one column. This mirrors the `||` in
# vector_store's ON CONFLICT clause, so the in-memory merge
# and the database merge agree.
if key == "field_sources" and isinstance(value, dict):
merged = dict(product.get("field_sources") or {})
merged.update(value)
product["field_sources"] = merged
continue
blank_incoming = value is None or (isinstance(value, str) and not value.strip())
existing = product.get(key)
held = existing is not None and not (isinstance(existing, str) and not existing.strip())

View File

@@ -0,0 +1,142 @@
"""Fills a blank field from what the brand's OWN rows already agree on.
WHY THIS EXISTS
`fssai_license` was 69.3% filled on 2026-09-08, sourced entirely from a
hardcoded 34-brand map (`brand_registry.FSSAI_LICENSES`). A brand outside
that map got nothing - except on one path, which got something far worse.
`user_products._build_product_dict` read:
fssai_license = req.fssai_license or sample_existing.get(...) or "10012042000244"
That constant is LION DATES' real, registered FSSAI licence. Any brand with
no sample row was stamped with it. This is not a cosmetic default: an FSSAI
number identifies the food business legally answerable for the product, and
inventing one attributes a stranger's regulatory liability to a product they
never made. `scripts/merge_haldiram.py:36` exists because this already
reached production once, on `brand_haldirams`.
The honest source for a blank licence is the brand's own catalog: 400
Britannia rows carrying one licence is good evidence for the 401st. That is
what this module reads.
THE RULE IT ENFORCES
Propagate only from UNAMBIGUOUS agreement. If a brand's rows carry two
different licences, one of them is already wrong and this module returns
None rather than picking. A blank field is a gap; a confidently wrong
regulatory identifier is a liability.
Nothing here invents a value. Every result is a value already present on a
row of the same brand, which is why the provenance method is
`catalog_consensus` and never `sourced`.
"""
from __future__ import annotations
import logging
from collections import Counter
from typing import Any, Dict, List, Optional, Tuple
logger = logging.getLogger(__name__)
# A single dissenting row should not veto 400 agreeing ones, but a genuine
# split must. Set so that "399 of 400 agree" propagates and "60/40" does not.
_MIN_AGREEMENT = 0.85
# Below this many populated rows there is no consensus to speak of, only a
# coincidence. Two rows agreeing proves nothing about a third.
_MIN_ROWS = 3
def _modal(values: List[Any], min_agreement: float = _MIN_AGREEMENT,
min_rows: int = _MIN_ROWS) -> Tuple[Optional[Any], Dict[str, Any]]:
"""The one value the population agrees on, or None with the reason why."""
populated = [v for v in values if v not in (None, "", [], {})]
if len(populated) < min_rows:
return None, {"reason": "too few populated rows", "rows": len(populated)}
# Lists (providers) are unhashable; compare them as ordered tuples.
keyed = [tuple(v) if isinstance(v, list) else v for v in populated]
counts = Counter(keyed)
winner, hits = counts.most_common(1)[0]
agreement = hits / len(keyed)
if agreement < min_agreement:
return None, {"reason": "no clear majority", "agreement": round(agreement, 3),
"distinct": len(counts)}
return (list(winner) if isinstance(winner, tuple) else winner), {
"agreement": round(agreement, 3), "rows": len(keyed)}
def consensus_value(column: str, rows: List[Dict[str, Any]],
min_agreement: float = _MIN_AGREEMENT
) -> Tuple[Optional[Any], Dict[str, Any]]:
"""The agreed value of `column` across `rows`, plus why it was or was not
reached. Never raises: an unreadable row set yields (None, reason)."""
try:
return _modal([r.get(column) for r in rows], min_agreement=min_agreement)
except Exception as e: # pragma: no cover - defensive
logger.debug("consensus for %s failed: %s", column, e)
return None, {"reason": f"error: {e}"}
def consensus_rows(brand: str, columns: List[str], limit: int = 300) -> List[Dict[str, Any]]:
"""Read only the columns consensus needs, for a brand's rows.
Deliberately NOT `get_products_by_brand`, which is `SELECT *` and therefore
carries the 384-dimension embedding on every row. Measured against
production: 244 Hindustan Unilever rows cost 3.0 MB that way, 4.7 KB of the
7.2 KB per row being an embedding string nothing here looks at.
Two named columns bring the same read down to roughly 50 KB. On a backend
container capped at 2560 MB that difference is not dangerous either way -
it is just the difference between reading what is needed and reading
everything, once per brand per upload.
Probes `information_schema` first, because the column set genuinely differs
between brand tables and a missing column would otherwise raise.
"""
from app.services.vector_store import _connect, _sanitize_name
conn = _connect()
if conn is None:
return []
table = f"brand_{_sanitize_name(brand)}"
try:
with conn.cursor() as cur:
cur.execute(
"SELECT column_name FROM information_schema.columns "
"WHERE table_schema = 'public' AND table_name = %s",
(table,),
)
present = {r[0] for r in cur.fetchall()}
wanted = [c for c in columns if c in present]
if not wanted:
return []
select = ", ".join(f'"{c}"' for c in wanted)
cur.execute(f'SELECT {select} FROM "{table}" LIMIT %s', (limit,))
return [dict(zip(wanted, row)) for row in cur.fetchall()]
except Exception as e: # noqa: BLE001 - defaults are a nicety, not the write
logger.debug("consensus read failed for %s: %s", brand, e)
return []
finally:
conn.close()
def fssai_for_brand(brand: str, rows: Optional[List[Dict[str, Any]]] = None) -> Tuple[Optional[str], str]:
"""The licence to use for a new product of `brand`, and where it came from.
Order: the curated registry map, then the brand's own rows. Never a
constant, never another brand's number.
"""
from app.services.brand_registry import get_fssai_license
mapped = get_fssai_license(brand)
if mapped:
return mapped, "brand_registry"
if rows:
value, _why = consensus_value("fssai_license", rows)
if value:
return str(value), "catalog_consensus"
return None, "unknown"

View File

@@ -0,0 +1 @@
"""Offline content enrichment - the display columns the store pipeline left blank."""

View File

@@ -0,0 +1,112 @@
"""Fills `highlights` and `nutrients` for rows the store pipeline leaves empty.
THE FAILURE THIS ADDRESSES
`catalog_engine.generate_product_highlights` and `generate_nutrients_info`
have existed for a long time and `brand_discovery._build_product` calls
both. The store-catalog pipeline never did: `_to_storage_row` simply passed
whatever the sheet had through, so a colleague's upload - which carries
neither column - landed `highlights=[]` and `nutrients=[]` on every row.
That is the whole reason those two columns look healthy in aggregate
(95.3% / 69.7% on 2026-09-08) while being empty for exactly the rows this
work is about.
WHAT IT WRITES, AND HOW HONESTLY
`highlights` is marketing copy derived from fields we already hold - the
category, the pack size, the brand. It is `derived`, never `sourced`.
`nutrients` is the display list. Where real per-100g figures exist,
`nutrition_score_sync.sync_nutrients_to_brand_tables` renders them from
`nutrition_facts` and overwrites whatever this stage wrote - that mirror is
the better source and runs later. This stage only supplies the
category-keyword fallback, flagged `estimated`, so a row is not blank while
it waits for a nutrition lookup that may never succeed.
THE CONSUMABILITY GATE
`generate_nutrients_info` works off category keywords, so a Hair Care row
whose category or description happens to contain a matching word acquires
entries like "Vitamin B Complex - Energy". Shampoo has no nutrients. This
stage refuses to write the column at all for a non-consumable, which is the
same gate `nutrition_data_service` applies on the lookup path and the same
reason `purge_non_consumable_nutrition.py` had to exist.
"""
from __future__ import annotations
import logging
from typing import Any, Dict, List
from app.services.enrichment.base import EnrichmentStage, StageOutcome
logger = logging.getLogger(__name__)
class ContentEnrichmentStage(EnrichmentStage):
"""Offline, deterministic, fills blanks only. Never raises, never erases."""
name = "content"
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
# Imported lazily: catalog_engine pulls in the image and LLM services,
# and this stage runs inside ingestion where those are already loaded
# but the enrichment package on its own should not require them.
from app.core.catalog_engine import (
generate_nutrients_info,
generate_product_highlights,
)
from app.services.consumability import is_non_consumable
fields: Dict[str, Any] = {}
sources: Dict[str, Any] = {}
title = product.get("title") or product.get("product_name") or ""
category = product.get("category") or ""
if not _has_entries(product.get("highlights")):
try:
highlights = generate_product_highlights(product, brand)
except Exception as e: # never abort a row
logger.debug("highlight generation failed for %r: %s", title, e)
highlights = []
if highlights:
fields["highlights"] = highlights
sources["highlights"] = {"method": "derived",
"source": "catalog_engine.generate_product_highlights"}
if not _has_entries(product.get("nutrients")):
if is_non_consumable(category, title):
# Not a gap - a column that cannot apply. Recording it stops
# the coverage report counting shampoo as missing nutrition
# forever, which is what makes someone eventually fabricate it.
sources["nutrients"] = {"method": "not_applicable",
"source": "non_consumable_product"}
else:
try:
nutrients = generate_nutrients_info(product, brand)
except Exception as e:
logger.debug("nutrient generation failed for %r: %s", title, e)
nutrients = []
if nutrients:
fields["nutrients"] = nutrients
sources["nutrients"] = {
"method": "estimated",
"source": "catalog_engine.generate_nutrients_info",
"note": "category keywords; replaced by real per-100g "
"figures when a nutrition lookup succeeds",
}
if sources:
fields["field_sources"] = sources
return StageOutcome(stage_name=self.name, fields=fields)
def _has_entries(value: Any) -> bool:
"""True when the column already carries something worth keeping.
A list of empty strings counts as empty: the spreadsheet parser produces
those from a column that exists but has no value in it, and treating one as
"already filled" is how a row keeps `['']` forever.
"""
if not isinstance(value, (list, tuple)):
return bool(value)
return any(str(v).strip() for v in value)

View File

@@ -60,6 +60,17 @@ def _build_default_stages() -> List[EnrichmentStage]:
except Exception as e:
logger.error(f"Barcode enrichment stage unavailable: {e}")
# Runs AFTER the lookup so it normalises whatever that found, and runs at
# all even when the lookup is disabled - which is the point. It derives
# barcode_type/gtin/ean13/upc from a barcode the row already has, offline
# and for free, so a sheet-supplied barcode finally gets validated and
# expanded instead of going straight to the database unchecked.
try:
from app.services.enrichment.barcode.identity_stage import BarcodeIdentityStage
stages.append(BarcodeIdentityStage())
except Exception as e:
logger.error(f"Barcode identity stage unavailable: {e}")
# HSN / GST & pricing enrichment (see app/services/enrichment/hsn_gst/) -
# deterministic, offline, pure-additive. Runs AFTER the barcode stage so
# every stored/exported row carries both sets of fields; a failure here
@@ -70,9 +81,15 @@ def _build_default_stages() -> List[EnrichmentStage]:
except Exception as e:
logger.error(f"HSN/GST enrichment stage unavailable: {e}")
# Future stages register here, e.g.:
# from app.services.enrichment.nutrition.stage import NutritionEnrichmentStage
# stages.append(NutritionEnrichmentStage())
# Offline display columns. Registered last so the nutrients fallback it
# writes is the lowest-priority source: the real per-100g figures mirrored
# by nutrition_score_sync overwrite it whenever a lookup succeeds.
try:
from app.services.enrichment.content.stage import ContentEnrichmentStage
stages.append(ContentEnrichmentStage())
except Exception as e:
logger.error(f"Content enrichment stage unavailable: {e}")
return stages

View File

@@ -0,0 +1,254 @@
"""Finds barcodes for freshly-ingested rows, in bulk, before nutrition runs.
WHY THIS RUNS BEFORE THE NUTRITION JOB, NOT ALONGSIDE IT
--------------------------------------------------------
Ordering here is a correctness property, not a preference.
`fetch_verified_nutrition_by_barcode` matches on the GTIN and returns at
confidence 0.95. The name search it falls back to accepts at a minimum of 0.32.
The nutrition job runs with `skip_if_verified=True`, so whichever path lands
first WINS PERMANENTLY - a 0.32 name match blocks the 0.95 barcode match from
ever being attempted. Two jobs racing would produce exactly that, silently, and
the catalog would end up with the worse of two available answers.
So this is a phase inside the same job, ahead of the nutrition phases.
WHY BULK, NOT THE PER-PRODUCT CASCADE
-------------------------------------
The per-product search endpoint Open Food Facts exposes is capped at 10
requests per minute. A 200-row upload is twenty minutes of waiting, which is
why `ENABLE_BARCODE_LOOKUP` defaults false (settings.py:420-423) and why the
inline stage stays off.
`off_bulk.fetch_brand_corpus` fetches a brand's ENTIRE Open Food Facts
catalogue in about five requests and matches offline against it. A brand is a
brand whether it has 3 rows or 300, so the cost is per brand, not per product.
That is what makes barcode enrichment affordable on the shared host at all.
It also uses `off_bulk.score_candidates` rather than the live matcher, because
that scorer already fixes two measured flaws: `matching.name_similarity` is
asymmetric ("Butter milk amul" vs "Amul Butter" scores 0.882 one way and 0.418
the other), and `matching.size_matches` vetoes any candidate with a blank size
when 57 of 146 Amul OFF records have `quantity: null`.
THE THRESHOLD IS 0.88, NOT 0.78
-------------------------------
`BARCODE_MIN_NAME_SIMILARITY` (0.78) is the floor for the REVERSE direction,
where a barcode has already established identity and the name is a sanity
check. This is the forward direction: many candidates compete and the name
carries the whole decision. `scripts/backfill_barcodes_from_off.py` measured
0.88 as the safe auto-apply point and 0.70-0.88 as review-only, and this reuses
that number rather than inventing one.
WHAT IT WRITES
barcode, barcode_type, gtin, ean13, barcode_source, barcode_verified=False,
barcode_lookup_status='name_matched', barcode_last_updated, and the
field_sources record - via targeted UPDATEs that pin the row's current
value, never via upsert_brand_products.
WHY NOT THE UPSERT
`get_products_by_brand` is `SELECT *`, so a row's `embedding` comes back as
a pgvector string, and `upsert_brand_products` only accepts a list - it
would write NULL and destroy the embedding. Targeted UPDATEs also cannot
clobber a concurrent write.
WHAT IT WILL NOT DO
It never overwrites a barcode a row already holds. A merchant typing one in
is holding the pack; nothing found by name similarity outranks that.
"""
from __future__ import annotations
import logging
import time
from typing import Any, Dict, Iterable, List, Optional, Tuple
from psycopg.types.json import Json
from app.services.enrichment.barcode.validators import (
classify_barcode_type,
to_ean13,
validate_barcode,
)
from app.services.vector_store import _connect, _sanitize_name
logger = logging.getLogger(__name__)
# Matches scripts/backfill_barcodes_from_off.py, which measured it.
DEFAULT_MIN_SIMILARITY = 0.88
BARCODE_SOURCE = "openfoodfacts_bulk (search.openfoodfacts.org)"
def _rows_needing_a_barcode(cur, table: str) -> List[Dict[str, Any]]:
"""Rows with no usable barcode. Probes the column list first, because a
table written before the schema migration may still lack the newer ones."""
cur.execute(
"SELECT column_name FROM information_schema.columns "
"WHERE table_schema = 'public' AND table_name = %s",
(table,),
)
present = {r[0] for r in cur.fetchall()}
if not {"id", "product_name", "barcode"} <= present:
return []
size = "size" if "size" in present else "NULL AS size"
cur.execute(
f'SELECT id, product_name, title, {size}, category, barcode '
f'FROM "{table}" '
f"WHERE barcode IS NULL OR btrim(barcode) = ''"
)
return [{"id": r[0], "product_name": r[1], "title": r[2], "size": r[3],
"category": r[4], "barcode": r[5]} for r in cur.fetchall()]
def enrich_brand_barcodes(brand: str, *, min_similarity: float = DEFAULT_MIN_SIMILARITY,
dry_run: bool = False,
progress_cb=None) -> Dict[str, int]:
"""Fill blank barcodes for one brand from its Open Food Facts corpus.
Never raises: a brand whose corpus cannot be fetched reports zero and the
caller moves to the next one. Enrichment is best-effort by contract.
"""
from app.services.enrichment.barcode.sources.off_bulk import (
brand_tokens,
fetch_brand_corpus,
score_candidates,
)
stats = {"candidates": 0, "matched": 0, "written": 0, "rejected": 0}
# The same refusal `store_catalog_pipeline.stages_8_9_enrichment` makes for
# the inline stages, for the same reason: a barcode identifies a
# manufactured article and the Own Products bucket is loose produce - an
# apple, a bunch of coriander. There is no GTIN to find, and a name match
# against some packaged product's corpus could only attach the wrong one.
from app.services.generic_products import OWN_PRODUCTS_BRAND
if brand == OWN_PRODUCTS_BRAND:
return stats
table = f"brand_{_sanitize_name(brand)}"
conn = _connect()
if conn is None:
return stats
try:
with conn.cursor() as cur:
rows = _rows_needing_a_barcode(cur, table)
if not rows:
return stats
stats["candidates"] = len(rows)
try:
# Returns the hit LIST directly (the on-disk cache file wraps it in
# a "hits" key; the function unwraps it). About five requests for a
# whole brand, then served from disk on later runs.
hits = fetch_brand_corpus(brand) or []
except Exception as e: # noqa: BLE001
logger.warning("OFF corpus unavailable for %s: %s", brand, e)
return stats
if not hits:
logger.info("Open Food Facts holds no India catalogue for %s", brand)
return stats
# Stripped from both sides before names are compared, so "Hindustan
# Unilever Hul Lux" reduces to "lux" on our side and matches OFF's
# "Lux". Computed once per brand, not once per row.
drop = brand_tokens(brand)
for index, row in enumerate(rows):
if progress_cb:
progress_cb(index, len(rows))
title = row.get("title") or row.get("product_name") or ""
try:
scored = score_candidates(
hits, title, [row.get("size") or ""], drop,
review_min=min_similarity,
)
except Exception as e: # noqa: BLE001
logger.debug("scoring failed for %r: %s", title, e)
continue
# score_candidates already drops anything under review_min, so the
# first entry is the best acceptable one. The explicit re-check is
# kept because the ordering contract is "best first", not "all
# above the floor" - relying on the filter alone would silently
# break if that ever changed.
best = scored[0] if scored else None
if not best or best.score < min_similarity:
stats["rejected"] += 1
continue
code = validate_barcode(getattr(best, "barcode", None))
if not code:
stats["rejected"] += 1
continue
stats["matched"] += 1
if dry_run:
continue
kind = classify_barcode_type(code)
sources = {
"barcode": {"method": "sourced", "source": BARCODE_SOURCE,
"confidence": round(float(best.score), 3),
"note": "matched on name against the brand's OFF "
"catalogue; not verified against the pack"},
"gtin": {"method": "derived", "source": "validators.validate_barcode"},
"ean13": {"method": "derived", "source": "validators.to_ean13"},
}
try:
with conn.cursor() as cur:
cur.execute(
f'UPDATE "{table}" SET barcode = %s, barcode_type = %s, '
f"gtin = %s, ean13 = %s, barcode_source = %s, "
f"barcode_verified = FALSE, barcode_lookup_status = %s, "
f"barcode_last_updated = NOW(), "
f"field_sources = COALESCE(field_sources, '{{}}'::jsonb) "
f" || %s::jsonb, "
f"updated_at = CURRENT_TIMESTAMP "
f"WHERE id = %s "
f" AND (barcode IS NULL OR btrim(barcode) = '')",
(code, kind.value, code, to_ean13(code), BARCODE_SOURCE,
"name_matched", Json(sources), row["id"]),
)
stats["written"] += cur.rowcount
conn.commit()
except Exception as e: # noqa: BLE001
conn.rollback()
logger.warning("barcode write failed for %s id=%s: %s",
table, row["id"], e)
finally:
conn.close()
return stats
def enrich_barcodes_for_brands(brands: Iterable[str], *,
min_similarity: float = DEFAULT_MIN_SIMILARITY,
dry_run: bool = False,
progress_cb=None) -> Dict[str, Dict[str, int]]:
"""Run `enrich_brand_barcodes` over several brands, one at a time.
Deliberately sequential. `EnrichmentPipeline`'s five-way concurrency is for
per-row work against a local corpus; firing five brand-corpus fetches at
Open Food Facts at once is how a shared host earns a rate limit.
"""
out: Dict[str, Dict[str, int]] = {}
names = [b.strip() for b in brands if b and b.strip()]
for i, brand in enumerate(names):
if progress_cb:
progress_cb(i, len(names))
try:
stats = enrich_brand_barcodes(brand, min_similarity=min_similarity,
dry_run=dry_run)
except Exception as e: # noqa: BLE001
logger.warning("barcode enrichment failed for %s: %s", brand, e)
continue
if stats.get("candidates"):
out[brand] = stats
logger.info("%s: %d without a barcode, %d matched, %d written",
brand, stats["candidates"], stats["matched"], stats["written"])
return out