Catalog feature updates on column fields
This commit is contained in:
117
app/services/enrichment/barcode/identity_stage.py
Normal file
117
app/services/enrichment/barcode/identity_stage.py
Normal file
@@ -0,0 +1,117 @@
|
||||
"""Derives the rest of a product's barcode identity from the barcode itself.
|
||||
|
||||
WHY THIS IS A SEPARATE STAGE FROM BarcodeEnrichmentStage
|
||||
--------------------------------------------------------
|
||||
That stage FINDS a barcode, needs the network, and is off by default
|
||||
(`ENABLE_BARCODE_LOOKUP`, see settings.py:420-423 for why). This one FINDS
|
||||
NOTHING. It takes a barcode the row already has - typed into the merchant's
|
||||
spreadsheet, seeded from a catalog, or just located by the cascade - and fills
|
||||
in the fields that are pure arithmetic on those digits:
|
||||
|
||||
barcode_type from the length (classify_barcode_type)
|
||||
gtin the validated digits (a GTIN is what a barcode encodes)
|
||||
ean13 zero-padded UPC-A (to_ean13)
|
||||
upc the digits, for UPC-A only
|
||||
|
||||
There is no lookup, no host, no rate limit and no failure mode beyond "these
|
||||
digits are not a valid GTIN", so it needs no settings flag and costs nothing.
|
||||
|
||||
THE FAILURE IT ADDRESSES
|
||||
Measured against production on 2026-09-08: `upc` was 0.0% filled, `ean13`
|
||||
6.4%, `gtin` 8.7% - against `barcode` at 18.4%. Every one of those could
|
||||
have been computed from the barcode already sitting in the same row. They
|
||||
were not, because the only code that produced them was inside the disabled
|
||||
network cascade, and the writer dropped them anyway.
|
||||
|
||||
WHY IT RUNS AFTER THE LOOKUP STAGE
|
||||
So it also normalises whatever the cascade just found. The cascade already
|
||||
validates, but a sheet-supplied barcode never passes through
|
||||
`validate_barcode` at all today - it goes straight from the spreadsheet to
|
||||
the database. This stage is the first thing that checks those digits.
|
||||
|
||||
WHAT IT WILL NOT DO
|
||||
It will not correct, reformat or delete `barcode`. If the digits fail
|
||||
checksum validation the stage returns NOTHING, leaving the merchant's value
|
||||
exactly as typed - `enrichment/base.py`'s merge guard would refuse to blank
|
||||
it anyway, and silently "fixing" a barcode a shop supplied would be worse
|
||||
than leaving it visibly wrong. The failure is recorded in `field_sources`
|
||||
so the coverage report can surface it.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import time
|
||||
from typing import Any, Dict
|
||||
|
||||
from app.services.enrichment.base import EnrichmentStage, StageOutcome
|
||||
from app.services.enrichment.barcode.models import BarcodeType
|
||||
from app.services.enrichment.barcode.validators import (
|
||||
classify_barcode_type,
|
||||
normalize_barcode,
|
||||
to_ean13,
|
||||
validate_barcode,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class BarcodeIdentityStage(EnrichmentStage):
|
||||
"""Offline, deterministic, additive. Never raises, never erases."""
|
||||
|
||||
name = "barcode_identity"
|
||||
|
||||
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
|
||||
raw = product.get("barcode")
|
||||
if not str(raw or "").strip():
|
||||
return StageOutcome(stage_name=self.name, fields={})
|
||||
|
||||
code = validate_barcode(raw)
|
||||
if not code:
|
||||
# Not a GTIN. Say so in the provenance rather than in the data, and
|
||||
# leave `barcode` untouched.
|
||||
digits = normalize_barcode(raw)
|
||||
reason = (f"{len(digits)} digits is not a GTIN-8/12/13/14 length"
|
||||
if digits else "no digits in the value")
|
||||
return StageOutcome(
|
||||
stage_name=self.name,
|
||||
fields={"field_sources": {"barcode": {
|
||||
"method": "unvalidated",
|
||||
"source": product.get("barcode_source") or "sheet",
|
||||
"note": f"failed checksum/format validation: {reason}",
|
||||
}}},
|
||||
error=f"barcode {raw!r} failed validation: {reason}",
|
||||
)
|
||||
|
||||
barcode_type = classify_barcode_type(code)
|
||||
fields: Dict[str, Any] = {
|
||||
"barcode": code, # normalised digits, same value
|
||||
"barcode_type": barcode_type.value,
|
||||
"gtin": code,
|
||||
"ean13": to_ean13(code), # None for GTIN-8, which is not a short EAN-13
|
||||
"upc": code if barcode_type is BarcodeType.UPC_A else None,
|
||||
}
|
||||
|
||||
# Only claim provenance we can stand behind. A barcode that arrived on
|
||||
# the sheet is the merchant's assertion, not ours, and is emphatically
|
||||
# not "verified" - that word is reserved for the cascade's
|
||||
# brand+size+name-matched result.
|
||||
if not str(product.get("barcode_source") or "").strip():
|
||||
fields["barcode_source"] = "sheet"
|
||||
fields["barcode_lookup_status"] = "sheet_validated"
|
||||
fields["barcode_verified"] = False
|
||||
fields["barcode_last_updated"] = time.time()
|
||||
|
||||
fields["field_sources"] = {
|
||||
"barcode": {
|
||||
"method": "sourced" if product.get("barcode_verified") else "asserted",
|
||||
"source": product.get("barcode_source") or "sheet",
|
||||
},
|
||||
# These four are arithmetic on the barcode, never a lookup. Calling
|
||||
# them "sourced" would overstate them.
|
||||
"gtin": {"method": "derived", "source": "validators.validate_barcode"},
|
||||
"ean13": {"method": "derived", "source": "validators.to_ean13"},
|
||||
"upc": {"method": "derived", "source": "validators.classify_barcode_type"},
|
||||
"barcode_type": {"method": "derived", "source": "validators.classify_barcode_type"},
|
||||
}
|
||||
|
||||
return StageOutcome(stage_name=self.name, fields=fields)
|
||||
@@ -114,9 +114,74 @@ def name_similarity(candidate_title: str, target_title: str) -> float:
|
||||
return round((overlap * 0.6) + (seq_ratio * 0.4), 3)
|
||||
|
||||
|
||||
def name_is_contained(candidate_title: str, target_title: str,
|
||||
target_brand: str = "") -> bool:
|
||||
"""True when the candidate's name is our name with only brand/size removed.
|
||||
|
||||
WHY THIS EXISTS - measured, not theoretical
|
||||
Open Food Facts stores short product names. We store long ones. Running
|
||||
`backfill_nutrition_from_barcodes` over the catalog on 2026-09-08, 149
|
||||
of 300 barcoded rows were rejected as "found, wrong product" when the
|
||||
barcode had resolved perfectly:
|
||||
|
||||
"Nestle Munch 8.9g" -> OFF "Munch" similarity 0.332
|
||||
"Coca-Cola Maaza 750ml" -> OFF "Maaza" similarity 0.304
|
||||
"Cadbury Perk 22 g" -> OFF "Perk" similarity 0.302
|
||||
|
||||
`name_similarity` divides the token overlap by the TARGET's token
|
||||
count, so a one-token candidate against a three-token target cannot
|
||||
exceed ~0.33 however right it is.
|
||||
|
||||
WHY NOT JUST LOWER THE THRESHOLD
|
||||
Because the same run also correctly rejected:
|
||||
|
||||
"Pepsico Lays 1kg" -> OFF "Spanish tomato tango" 0.133
|
||||
"Coca-Cola Fanta 750ml" -> OFF "Orange" 0.089
|
||||
"Lion Dates Powder 100g" -> OFF "PEPER NOTEN" 0.097
|
||||
|
||||
Those sit BELOW the containment cases but a threshold low enough to
|
||||
admit 0.30 also admits them. The measured yield table at
|
||||
settings.py:449-477 raised this floor to 0.78 for exactly that reason.
|
||||
Containment separates the two groups on structure rather than on a
|
||||
number: "Munch" is every token of our name minus brand and size;
|
||||
"Orange" is not a subset of "Coca-Cola Fanta 750ml" at all.
|
||||
|
||||
THE RULE
|
||||
Every token of the candidate's name must appear in the target's, once
|
||||
brand tokens and size tokens are discounted, and the candidate must
|
||||
carry at least one token that is not the brand. A bare brand name
|
||||
("Colgate", "godrej" - both real OFF titles) therefore does NOT match,
|
||||
which matters because those would otherwise attach to every product of
|
||||
that brand.
|
||||
"""
|
||||
cand_tokens = _tokens(candidate_title)
|
||||
target_tokens = _tokens(target_title)
|
||||
if not cand_tokens or not target_tokens:
|
||||
return False
|
||||
|
||||
brand_tokens = _tokens(target_brand)
|
||||
# A candidate that is only the brand identifies a brand, not a product.
|
||||
if not (cand_tokens - brand_tokens):
|
||||
return False
|
||||
|
||||
# Size tokens are not identity: our title carries the pack size, OFF's
|
||||
# usually does not, and `size_matches` has already checked the size
|
||||
# separately by the time this is consulted.
|
||||
def _meaningful(tokens):
|
||||
return {t for t in tokens if not _SIZE_TOKEN_RE.fullmatch(t)}
|
||||
|
||||
return _meaningful(cand_tokens) <= _meaningful(target_tokens | brand_tokens)
|
||||
|
||||
|
||||
# A token that is purely a quantity ("750ml", "8", "9g", "1kg"). Size is
|
||||
# compared by `size_matches`, so it must not also decide name identity.
|
||||
_SIZE_TOKEN_RE = re.compile(r"\d+(?:\.\d+)?(?:g|kg|ml|l|mg|cl|oz|gm|ltr|pcs|n)?", re.I)
|
||||
|
||||
|
||||
def is_match(candidate: BarcodeCandidate, target_brand: str, target_title: str, target_size: str,
|
||||
brand_aliases: Optional[Iterable[str]] = None,
|
||||
min_name_similarity: float = 0.45) -> tuple[bool, float]:
|
||||
min_name_similarity: float = 0.45,
|
||||
barcode_is_identity: bool = False) -> tuple[bool, float]:
|
||||
"""The combined gate a candidate must pass to be accepted:
|
||||
1. Brand matches (or overlaps a known alias).
|
||||
2. Pack size matches within a tight tolerance.
|
||||
@@ -126,16 +191,52 @@ def is_match(candidate: BarcodeCandidate, target_brand: str, target_title: str,
|
||||
product line from the same brand.
|
||||
Returns (matched, confidence) - confidence is diagnostic only, stored
|
||||
on the result for audit/QA but never used to override rule 1-3.
|
||||
|
||||
`barcode_is_identity` says the caller already knows WHICH product this is,
|
||||
because it looked the candidate up BY its GTIN rather than by searching.
|
||||
That changes what rules 2 and 4 are for: they stop being evidence of
|
||||
identity and become sanity checks against our barcode being on the wrong
|
||||
row. A sanity check cannot fail on information the source does not have, so
|
||||
under this flag:
|
||||
|
||||
* rule 2 (size) - a BLANK candidate size no longer vetoes. Open Food
|
||||
Facts leaves `quantity` null on a large share of records (57 of 146
|
||||
Amul hits), and `size_matches` returns False whenever either side is
|
||||
blank. A record with no quantity does not disagree with our pack size;
|
||||
it says nothing about it. A quantity that is PRESENT and different
|
||||
still vetoes - that is our barcode pointing at the wrong pack.
|
||||
* rule 4 (name) - see `name_is_contained`.
|
||||
|
||||
Rules 1 and 3 are unaffected: a different brand, or a "sugar free" the
|
||||
target does not have, still means a different product.
|
||||
|
||||
It defaults to False because every relaxation here is unsafe on the SEARCH
|
||||
path, where many candidates compete and name and size are the only things
|
||||
telling them apart - "Munch" with no size would match every Nestle product
|
||||
containing that word. Pass True only where a single candidate was fetched
|
||||
by barcode. Today that is `fetch_verified_nutrition_by_barcode` and
|
||||
`scripts/backfill_nutrition_from_barcodes`, and nothing else.
|
||||
|
||||
Measured on 2026-09-08: of 300 barcoded catalog rows, 149 were refused as
|
||||
"found, wrong product" with the barcode resolving perfectly. The name gate
|
||||
was the visible symptom, but the SIZE gate rejected most of them first.
|
||||
"""
|
||||
if not brand_matches(candidate.candidate_brand, target_brand, brand_aliases):
|
||||
return False, 0.0
|
||||
if not size_matches(candidate.candidate_size, target_size):
|
||||
# A blank candidate size is missing information, not a disagreement - but
|
||||
# only when the barcode already established identity. On the search path a
|
||||
# sizeless candidate is genuinely unidentifiable and must still be refused.
|
||||
size_unknown = barcode_is_identity and not str(candidate.candidate_size or "").strip()
|
||||
if not size_unknown and not size_matches(candidate.candidate_size, target_size):
|
||||
return False, 0.0
|
||||
if has_conflicting_variant_terms(candidate.candidate_title, target_title):
|
||||
return False, 0.0
|
||||
|
||||
similarity = name_similarity(candidate.candidate_title, target_title)
|
||||
if similarity < min_name_similarity:
|
||||
if barcode_is_identity and name_is_contained(
|
||||
candidate.candidate_title, target_title, target_brand):
|
||||
return True, similarity
|
||||
return False, similarity
|
||||
|
||||
return True, similarity
|
||||
|
||||
@@ -88,6 +88,25 @@ class EnrichmentStage(ABC):
|
||||
# HsnGstEnrichmentStage and BarcodeEnrichmentStage.
|
||||
if outcome.fields:
|
||||
for key, value in outcome.fields.items():
|
||||
# `field_sources` ACCUMULATES; every other key is assigned.
|
||||
#
|
||||
# It is a map keyed by column name, and each stage knows the
|
||||
# provenance of only the columns it filled. Assigning it like
|
||||
# anything else would mean the last stage to run erases what
|
||||
# every earlier stage recorded - so the barcode stage's
|
||||
# provenance would vanish the moment the HSN stage ran, and
|
||||
# the coverage report would show values with no origin.
|
||||
#
|
||||
# A shallow merge is the right depth: each key's value is one
|
||||
# flat record about one column. This mirrors the `||` in
|
||||
# vector_store's ON CONFLICT clause, so the in-memory merge
|
||||
# and the database merge agree.
|
||||
if key == "field_sources" and isinstance(value, dict):
|
||||
merged = dict(product.get("field_sources") or {})
|
||||
merged.update(value)
|
||||
product["field_sources"] = merged
|
||||
continue
|
||||
|
||||
blank_incoming = value is None or (isinstance(value, str) and not value.strip())
|
||||
existing = product.get(key)
|
||||
held = existing is not None and not (isinstance(existing, str) and not existing.strip())
|
||||
|
||||
142
app/services/enrichment/catalog_consensus.py
Normal file
142
app/services/enrichment/catalog_consensus.py
Normal file
@@ -0,0 +1,142 @@
|
||||
"""Fills a blank field from what the brand's OWN rows already agree on.
|
||||
|
||||
WHY THIS EXISTS
|
||||
`fssai_license` was 69.3% filled on 2026-09-08, sourced entirely from a
|
||||
hardcoded 34-brand map (`brand_registry.FSSAI_LICENSES`). A brand outside
|
||||
that map got nothing - except on one path, which got something far worse.
|
||||
|
||||
`user_products._build_product_dict` read:
|
||||
|
||||
fssai_license = req.fssai_license or sample_existing.get(...) or "10012042000244"
|
||||
|
||||
That constant is LION DATES' real, registered FSSAI licence. Any brand with
|
||||
no sample row was stamped with it. This is not a cosmetic default: an FSSAI
|
||||
number identifies the food business legally answerable for the product, and
|
||||
inventing one attributes a stranger's regulatory liability to a product they
|
||||
never made. `scripts/merge_haldiram.py:36` exists because this already
|
||||
reached production once, on `brand_haldirams`.
|
||||
|
||||
The honest source for a blank licence is the brand's own catalog: 400
|
||||
Britannia rows carrying one licence is good evidence for the 401st. That is
|
||||
what this module reads.
|
||||
|
||||
THE RULE IT ENFORCES
|
||||
Propagate only from UNAMBIGUOUS agreement. If a brand's rows carry two
|
||||
different licences, one of them is already wrong and this module returns
|
||||
None rather than picking. A blank field is a gap; a confidently wrong
|
||||
regulatory identifier is a liability.
|
||||
|
||||
Nothing here invents a value. Every result is a value already present on a
|
||||
row of the same brand, which is why the provenance method is
|
||||
`catalog_consensus` and never `sourced`.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from collections import Counter
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# A single dissenting row should not veto 400 agreeing ones, but a genuine
|
||||
# split must. Set so that "399 of 400 agree" propagates and "60/40" does not.
|
||||
_MIN_AGREEMENT = 0.85
|
||||
|
||||
# Below this many populated rows there is no consensus to speak of, only a
|
||||
# coincidence. Two rows agreeing proves nothing about a third.
|
||||
_MIN_ROWS = 3
|
||||
|
||||
|
||||
def _modal(values: List[Any], min_agreement: float = _MIN_AGREEMENT,
|
||||
min_rows: int = _MIN_ROWS) -> Tuple[Optional[Any], Dict[str, Any]]:
|
||||
"""The one value the population agrees on, or None with the reason why."""
|
||||
populated = [v for v in values if v not in (None, "", [], {})]
|
||||
if len(populated) < min_rows:
|
||||
return None, {"reason": "too few populated rows", "rows": len(populated)}
|
||||
|
||||
# Lists (providers) are unhashable; compare them as ordered tuples.
|
||||
keyed = [tuple(v) if isinstance(v, list) else v for v in populated]
|
||||
counts = Counter(keyed)
|
||||
winner, hits = counts.most_common(1)[0]
|
||||
agreement = hits / len(keyed)
|
||||
if agreement < min_agreement:
|
||||
return None, {"reason": "no clear majority", "agreement": round(agreement, 3),
|
||||
"distinct": len(counts)}
|
||||
|
||||
return (list(winner) if isinstance(winner, tuple) else winner), {
|
||||
"agreement": round(agreement, 3), "rows": len(keyed)}
|
||||
|
||||
|
||||
def consensus_value(column: str, rows: List[Dict[str, Any]],
|
||||
min_agreement: float = _MIN_AGREEMENT
|
||||
) -> Tuple[Optional[Any], Dict[str, Any]]:
|
||||
"""The agreed value of `column` across `rows`, plus why it was or was not
|
||||
reached. Never raises: an unreadable row set yields (None, reason)."""
|
||||
try:
|
||||
return _modal([r.get(column) for r in rows], min_agreement=min_agreement)
|
||||
except Exception as e: # pragma: no cover - defensive
|
||||
logger.debug("consensus for %s failed: %s", column, e)
|
||||
return None, {"reason": f"error: {e}"}
|
||||
|
||||
|
||||
def consensus_rows(brand: str, columns: List[str], limit: int = 300) -> List[Dict[str, Any]]:
|
||||
"""Read only the columns consensus needs, for a brand's rows.
|
||||
|
||||
Deliberately NOT `get_products_by_brand`, which is `SELECT *` and therefore
|
||||
carries the 384-dimension embedding on every row. Measured against
|
||||
production: 244 Hindustan Unilever rows cost 3.0 MB that way, 4.7 KB of the
|
||||
7.2 KB per row being an embedding string nothing here looks at.
|
||||
|
||||
Two named columns bring the same read down to roughly 50 KB. On a backend
|
||||
container capped at 2560 MB that difference is not dangerous either way -
|
||||
it is just the difference between reading what is needed and reading
|
||||
everything, once per brand per upload.
|
||||
|
||||
Probes `information_schema` first, because the column set genuinely differs
|
||||
between brand tables and a missing column would otherwise raise.
|
||||
"""
|
||||
from app.services.vector_store import _connect, _sanitize_name
|
||||
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
return []
|
||||
table = f"brand_{_sanitize_name(brand)}"
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
cur.execute(
|
||||
"SELECT column_name FROM information_schema.columns "
|
||||
"WHERE table_schema = 'public' AND table_name = %s",
|
||||
(table,),
|
||||
)
|
||||
present = {r[0] for r in cur.fetchall()}
|
||||
wanted = [c for c in columns if c in present]
|
||||
if not wanted:
|
||||
return []
|
||||
select = ", ".join(f'"{c}"' for c in wanted)
|
||||
cur.execute(f'SELECT {select} FROM "{table}" LIMIT %s', (limit,))
|
||||
return [dict(zip(wanted, row)) for row in cur.fetchall()]
|
||||
except Exception as e: # noqa: BLE001 - defaults are a nicety, not the write
|
||||
logger.debug("consensus read failed for %s: %s", brand, e)
|
||||
return []
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def fssai_for_brand(brand: str, rows: Optional[List[Dict[str, Any]]] = None) -> Tuple[Optional[str], str]:
|
||||
"""The licence to use for a new product of `brand`, and where it came from.
|
||||
|
||||
Order: the curated registry map, then the brand's own rows. Never a
|
||||
constant, never another brand's number.
|
||||
"""
|
||||
from app.services.brand_registry import get_fssai_license
|
||||
|
||||
mapped = get_fssai_license(brand)
|
||||
if mapped:
|
||||
return mapped, "brand_registry"
|
||||
|
||||
if rows:
|
||||
value, _why = consensus_value("fssai_license", rows)
|
||||
if value:
|
||||
return str(value), "catalog_consensus"
|
||||
|
||||
return None, "unknown"
|
||||
1
app/services/enrichment/content/__init__.py
Normal file
1
app/services/enrichment/content/__init__.py
Normal file
@@ -0,0 +1 @@
|
||||
"""Offline content enrichment - the display columns the store pipeline left blank."""
|
||||
112
app/services/enrichment/content/stage.py
Normal file
112
app/services/enrichment/content/stage.py
Normal file
@@ -0,0 +1,112 @@
|
||||
"""Fills `highlights` and `nutrients` for rows the store pipeline leaves empty.
|
||||
|
||||
THE FAILURE THIS ADDRESSES
|
||||
`catalog_engine.generate_product_highlights` and `generate_nutrients_info`
|
||||
have existed for a long time and `brand_discovery._build_product` calls
|
||||
both. The store-catalog pipeline never did: `_to_storage_row` simply passed
|
||||
whatever the sheet had through, so a colleague's upload - which carries
|
||||
neither column - landed `highlights=[]` and `nutrients=[]` on every row.
|
||||
|
||||
That is the whole reason those two columns look healthy in aggregate
|
||||
(95.3% / 69.7% on 2026-09-08) while being empty for exactly the rows this
|
||||
work is about.
|
||||
|
||||
WHAT IT WRITES, AND HOW HONESTLY
|
||||
`highlights` is marketing copy derived from fields we already hold - the
|
||||
category, the pack size, the brand. It is `derived`, never `sourced`.
|
||||
|
||||
`nutrients` is the display list. Where real per-100g figures exist,
|
||||
`nutrition_score_sync.sync_nutrients_to_brand_tables` renders them from
|
||||
`nutrition_facts` and overwrites whatever this stage wrote - that mirror is
|
||||
the better source and runs later. This stage only supplies the
|
||||
category-keyword fallback, flagged `estimated`, so a row is not blank while
|
||||
it waits for a nutrition lookup that may never succeed.
|
||||
|
||||
THE CONSUMABILITY GATE
|
||||
`generate_nutrients_info` works off category keywords, so a Hair Care row
|
||||
whose category or description happens to contain a matching word acquires
|
||||
entries like "Vitamin B Complex - Energy". Shampoo has no nutrients. This
|
||||
stage refuses to write the column at all for a non-consumable, which is the
|
||||
same gate `nutrition_data_service` applies on the lookup path and the same
|
||||
reason `purge_non_consumable_nutrition.py` had to exist.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Any, Dict, List
|
||||
|
||||
from app.services.enrichment.base import EnrichmentStage, StageOutcome
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class ContentEnrichmentStage(EnrichmentStage):
|
||||
"""Offline, deterministic, fills blanks only. Never raises, never erases."""
|
||||
|
||||
name = "content"
|
||||
|
||||
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
|
||||
# Imported lazily: catalog_engine pulls in the image and LLM services,
|
||||
# and this stage runs inside ingestion where those are already loaded
|
||||
# but the enrichment package on its own should not require them.
|
||||
from app.core.catalog_engine import (
|
||||
generate_nutrients_info,
|
||||
generate_product_highlights,
|
||||
)
|
||||
from app.services.consumability import is_non_consumable
|
||||
|
||||
fields: Dict[str, Any] = {}
|
||||
sources: Dict[str, Any] = {}
|
||||
|
||||
title = product.get("title") or product.get("product_name") or ""
|
||||
category = product.get("category") or ""
|
||||
|
||||
if not _has_entries(product.get("highlights")):
|
||||
try:
|
||||
highlights = generate_product_highlights(product, brand)
|
||||
except Exception as e: # never abort a row
|
||||
logger.debug("highlight generation failed for %r: %s", title, e)
|
||||
highlights = []
|
||||
if highlights:
|
||||
fields["highlights"] = highlights
|
||||
sources["highlights"] = {"method": "derived",
|
||||
"source": "catalog_engine.generate_product_highlights"}
|
||||
|
||||
if not _has_entries(product.get("nutrients")):
|
||||
if is_non_consumable(category, title):
|
||||
# Not a gap - a column that cannot apply. Recording it stops
|
||||
# the coverage report counting shampoo as missing nutrition
|
||||
# forever, which is what makes someone eventually fabricate it.
|
||||
sources["nutrients"] = {"method": "not_applicable",
|
||||
"source": "non_consumable_product"}
|
||||
else:
|
||||
try:
|
||||
nutrients = generate_nutrients_info(product, brand)
|
||||
except Exception as e:
|
||||
logger.debug("nutrient generation failed for %r: %s", title, e)
|
||||
nutrients = []
|
||||
if nutrients:
|
||||
fields["nutrients"] = nutrients
|
||||
sources["nutrients"] = {
|
||||
"method": "estimated",
|
||||
"source": "catalog_engine.generate_nutrients_info",
|
||||
"note": "category keywords; replaced by real per-100g "
|
||||
"figures when a nutrition lookup succeeds",
|
||||
}
|
||||
|
||||
if sources:
|
||||
fields["field_sources"] = sources
|
||||
|
||||
return StageOutcome(stage_name=self.name, fields=fields)
|
||||
|
||||
|
||||
def _has_entries(value: Any) -> bool:
|
||||
"""True when the column already carries something worth keeping.
|
||||
|
||||
A list of empty strings counts as empty: the spreadsheet parser produces
|
||||
those from a column that exists but has no value in it, and treating one as
|
||||
"already filled" is how a row keeps `['']` forever.
|
||||
"""
|
||||
if not isinstance(value, (list, tuple)):
|
||||
return bool(value)
|
||||
return any(str(v).strip() for v in value)
|
||||
@@ -60,6 +60,17 @@ def _build_default_stages() -> List[EnrichmentStage]:
|
||||
except Exception as e:
|
||||
logger.error(f"Barcode enrichment stage unavailable: {e}")
|
||||
|
||||
# Runs AFTER the lookup so it normalises whatever that found, and runs at
|
||||
# all even when the lookup is disabled - which is the point. It derives
|
||||
# barcode_type/gtin/ean13/upc from a barcode the row already has, offline
|
||||
# and for free, so a sheet-supplied barcode finally gets validated and
|
||||
# expanded instead of going straight to the database unchecked.
|
||||
try:
|
||||
from app.services.enrichment.barcode.identity_stage import BarcodeIdentityStage
|
||||
stages.append(BarcodeIdentityStage())
|
||||
except Exception as e:
|
||||
logger.error(f"Barcode identity stage unavailable: {e}")
|
||||
|
||||
# HSN / GST & pricing enrichment (see app/services/enrichment/hsn_gst/) -
|
||||
# deterministic, offline, pure-additive. Runs AFTER the barcode stage so
|
||||
# every stored/exported row carries both sets of fields; a failure here
|
||||
@@ -70,9 +81,15 @@ def _build_default_stages() -> List[EnrichmentStage]:
|
||||
except Exception as e:
|
||||
logger.error(f"HSN/GST enrichment stage unavailable: {e}")
|
||||
|
||||
# Future stages register here, e.g.:
|
||||
# from app.services.enrichment.nutrition.stage import NutritionEnrichmentStage
|
||||
# stages.append(NutritionEnrichmentStage())
|
||||
# Offline display columns. Registered last so the nutrients fallback it
|
||||
# writes is the lowest-priority source: the real per-100g figures mirrored
|
||||
# by nutrition_score_sync overwrite it whenever a lookup succeeds.
|
||||
try:
|
||||
from app.services.enrichment.content.stage import ContentEnrichmentStage
|
||||
stages.append(ContentEnrichmentStage())
|
||||
except Exception as e:
|
||||
logger.error(f"Content enrichment stage unavailable: {e}")
|
||||
|
||||
return stages
|
||||
|
||||
|
||||
|
||||
254
app/services/enrichment/post_ingest_barcodes.py
Normal file
254
app/services/enrichment/post_ingest_barcodes.py
Normal file
@@ -0,0 +1,254 @@
|
||||
"""Finds barcodes for freshly-ingested rows, in bulk, before nutrition runs.
|
||||
|
||||
WHY THIS RUNS BEFORE THE NUTRITION JOB, NOT ALONGSIDE IT
|
||||
--------------------------------------------------------
|
||||
Ordering here is a correctness property, not a preference.
|
||||
|
||||
`fetch_verified_nutrition_by_barcode` matches on the GTIN and returns at
|
||||
confidence 0.95. The name search it falls back to accepts at a minimum of 0.32.
|
||||
The nutrition job runs with `skip_if_verified=True`, so whichever path lands
|
||||
first WINS PERMANENTLY - a 0.32 name match blocks the 0.95 barcode match from
|
||||
ever being attempted. Two jobs racing would produce exactly that, silently, and
|
||||
the catalog would end up with the worse of two available answers.
|
||||
|
||||
So this is a phase inside the same job, ahead of the nutrition phases.
|
||||
|
||||
WHY BULK, NOT THE PER-PRODUCT CASCADE
|
||||
-------------------------------------
|
||||
The per-product search endpoint Open Food Facts exposes is capped at 10
|
||||
requests per minute. A 200-row upload is twenty minutes of waiting, which is
|
||||
why `ENABLE_BARCODE_LOOKUP` defaults false (settings.py:420-423) and why the
|
||||
inline stage stays off.
|
||||
|
||||
`off_bulk.fetch_brand_corpus` fetches a brand's ENTIRE Open Food Facts
|
||||
catalogue in about five requests and matches offline against it. A brand is a
|
||||
brand whether it has 3 rows or 300, so the cost is per brand, not per product.
|
||||
That is what makes barcode enrichment affordable on the shared host at all.
|
||||
|
||||
It also uses `off_bulk.score_candidates` rather than the live matcher, because
|
||||
that scorer already fixes two measured flaws: `matching.name_similarity` is
|
||||
asymmetric ("Butter milk amul" vs "Amul Butter" scores 0.882 one way and 0.418
|
||||
the other), and `matching.size_matches` vetoes any candidate with a blank size
|
||||
when 57 of 146 Amul OFF records have `quantity: null`.
|
||||
|
||||
THE THRESHOLD IS 0.88, NOT 0.78
|
||||
-------------------------------
|
||||
`BARCODE_MIN_NAME_SIMILARITY` (0.78) is the floor for the REVERSE direction,
|
||||
where a barcode has already established identity and the name is a sanity
|
||||
check. This is the forward direction: many candidates compete and the name
|
||||
carries the whole decision. `scripts/backfill_barcodes_from_off.py` measured
|
||||
0.88 as the safe auto-apply point and 0.70-0.88 as review-only, and this reuses
|
||||
that number rather than inventing one.
|
||||
|
||||
WHAT IT WRITES
|
||||
barcode, barcode_type, gtin, ean13, barcode_source, barcode_verified=False,
|
||||
barcode_lookup_status='name_matched', barcode_last_updated, and the
|
||||
field_sources record - via targeted UPDATEs that pin the row's current
|
||||
value, never via upsert_brand_products.
|
||||
|
||||
WHY NOT THE UPSERT
|
||||
`get_products_by_brand` is `SELECT *`, so a row's `embedding` comes back as
|
||||
a pgvector string, and `upsert_brand_products` only accepts a list - it
|
||||
would write NULL and destroy the embedding. Targeted UPDATEs also cannot
|
||||
clobber a concurrent write.
|
||||
|
||||
WHAT IT WILL NOT DO
|
||||
It never overwrites a barcode a row already holds. A merchant typing one in
|
||||
is holding the pack; nothing found by name similarity outranks that.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import time
|
||||
from typing import Any, Dict, Iterable, List, Optional, Tuple
|
||||
|
||||
from psycopg.types.json import Json
|
||||
|
||||
from app.services.enrichment.barcode.validators import (
|
||||
classify_barcode_type,
|
||||
to_ean13,
|
||||
validate_barcode,
|
||||
)
|
||||
from app.services.vector_store import _connect, _sanitize_name
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Matches scripts/backfill_barcodes_from_off.py, which measured it.
|
||||
DEFAULT_MIN_SIMILARITY = 0.88
|
||||
|
||||
BARCODE_SOURCE = "openfoodfacts_bulk (search.openfoodfacts.org)"
|
||||
|
||||
|
||||
def _rows_needing_a_barcode(cur, table: str) -> List[Dict[str, Any]]:
|
||||
"""Rows with no usable barcode. Probes the column list first, because a
|
||||
table written before the schema migration may still lack the newer ones."""
|
||||
cur.execute(
|
||||
"SELECT column_name FROM information_schema.columns "
|
||||
"WHERE table_schema = 'public' AND table_name = %s",
|
||||
(table,),
|
||||
)
|
||||
present = {r[0] for r in cur.fetchall()}
|
||||
if not {"id", "product_name", "barcode"} <= present:
|
||||
return []
|
||||
|
||||
size = "size" if "size" in present else "NULL AS size"
|
||||
cur.execute(
|
||||
f'SELECT id, product_name, title, {size}, category, barcode '
|
||||
f'FROM "{table}" '
|
||||
f"WHERE barcode IS NULL OR btrim(barcode) = ''"
|
||||
)
|
||||
return [{"id": r[0], "product_name": r[1], "title": r[2], "size": r[3],
|
||||
"category": r[4], "barcode": r[5]} for r in cur.fetchall()]
|
||||
|
||||
|
||||
def enrich_brand_barcodes(brand: str, *, min_similarity: float = DEFAULT_MIN_SIMILARITY,
|
||||
dry_run: bool = False,
|
||||
progress_cb=None) -> Dict[str, int]:
|
||||
"""Fill blank barcodes for one brand from its Open Food Facts corpus.
|
||||
|
||||
Never raises: a brand whose corpus cannot be fetched reports zero and the
|
||||
caller moves to the next one. Enrichment is best-effort by contract.
|
||||
"""
|
||||
from app.services.enrichment.barcode.sources.off_bulk import (
|
||||
brand_tokens,
|
||||
fetch_brand_corpus,
|
||||
score_candidates,
|
||||
)
|
||||
|
||||
stats = {"candidates": 0, "matched": 0, "written": 0, "rejected": 0}
|
||||
|
||||
# The same refusal `store_catalog_pipeline.stages_8_9_enrichment` makes for
|
||||
# the inline stages, for the same reason: a barcode identifies a
|
||||
# manufactured article and the Own Products bucket is loose produce - an
|
||||
# apple, a bunch of coriander. There is no GTIN to find, and a name match
|
||||
# against some packaged product's corpus could only attach the wrong one.
|
||||
from app.services.generic_products import OWN_PRODUCTS_BRAND
|
||||
if brand == OWN_PRODUCTS_BRAND:
|
||||
return stats
|
||||
|
||||
table = f"brand_{_sanitize_name(brand)}"
|
||||
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
return stats
|
||||
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
rows = _rows_needing_a_barcode(cur, table)
|
||||
if not rows:
|
||||
return stats
|
||||
stats["candidates"] = len(rows)
|
||||
|
||||
try:
|
||||
# Returns the hit LIST directly (the on-disk cache file wraps it in
|
||||
# a "hits" key; the function unwraps it). About five requests for a
|
||||
# whole brand, then served from disk on later runs.
|
||||
hits = fetch_brand_corpus(brand) or []
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.warning("OFF corpus unavailable for %s: %s", brand, e)
|
||||
return stats
|
||||
|
||||
if not hits:
|
||||
logger.info("Open Food Facts holds no India catalogue for %s", brand)
|
||||
return stats
|
||||
|
||||
# Stripped from both sides before names are compared, so "Hindustan
|
||||
# Unilever Hul Lux" reduces to "lux" on our side and matches OFF's
|
||||
# "Lux". Computed once per brand, not once per row.
|
||||
drop = brand_tokens(brand)
|
||||
|
||||
for index, row in enumerate(rows):
|
||||
if progress_cb:
|
||||
progress_cb(index, len(rows))
|
||||
|
||||
title = row.get("title") or row.get("product_name") or ""
|
||||
try:
|
||||
scored = score_candidates(
|
||||
hits, title, [row.get("size") or ""], drop,
|
||||
review_min=min_similarity,
|
||||
)
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.debug("scoring failed for %r: %s", title, e)
|
||||
continue
|
||||
|
||||
# score_candidates already drops anything under review_min, so the
|
||||
# first entry is the best acceptable one. The explicit re-check is
|
||||
# kept because the ordering contract is "best first", not "all
|
||||
# above the floor" - relying on the filter alone would silently
|
||||
# break if that ever changed.
|
||||
best = scored[0] if scored else None
|
||||
if not best or best.score < min_similarity:
|
||||
stats["rejected"] += 1
|
||||
continue
|
||||
|
||||
code = validate_barcode(getattr(best, "barcode", None))
|
||||
if not code:
|
||||
stats["rejected"] += 1
|
||||
continue
|
||||
|
||||
stats["matched"] += 1
|
||||
if dry_run:
|
||||
continue
|
||||
|
||||
kind = classify_barcode_type(code)
|
||||
sources = {
|
||||
"barcode": {"method": "sourced", "source": BARCODE_SOURCE,
|
||||
"confidence": round(float(best.score), 3),
|
||||
"note": "matched on name against the brand's OFF "
|
||||
"catalogue; not verified against the pack"},
|
||||
"gtin": {"method": "derived", "source": "validators.validate_barcode"},
|
||||
"ean13": {"method": "derived", "source": "validators.to_ean13"},
|
||||
}
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
cur.execute(
|
||||
f'UPDATE "{table}" SET barcode = %s, barcode_type = %s, '
|
||||
f"gtin = %s, ean13 = %s, barcode_source = %s, "
|
||||
f"barcode_verified = FALSE, barcode_lookup_status = %s, "
|
||||
f"barcode_last_updated = NOW(), "
|
||||
f"field_sources = COALESCE(field_sources, '{{}}'::jsonb) "
|
||||
f" || %s::jsonb, "
|
||||
f"updated_at = CURRENT_TIMESTAMP "
|
||||
f"WHERE id = %s "
|
||||
f" AND (barcode IS NULL OR btrim(barcode) = '')",
|
||||
(code, kind.value, code, to_ean13(code), BARCODE_SOURCE,
|
||||
"name_matched", Json(sources), row["id"]),
|
||||
)
|
||||
stats["written"] += cur.rowcount
|
||||
conn.commit()
|
||||
except Exception as e: # noqa: BLE001
|
||||
conn.rollback()
|
||||
logger.warning("barcode write failed for %s id=%s: %s",
|
||||
table, row["id"], e)
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
return stats
|
||||
|
||||
|
||||
def enrich_barcodes_for_brands(brands: Iterable[str], *,
|
||||
min_similarity: float = DEFAULT_MIN_SIMILARITY,
|
||||
dry_run: bool = False,
|
||||
progress_cb=None) -> Dict[str, Dict[str, int]]:
|
||||
"""Run `enrich_brand_barcodes` over several brands, one at a time.
|
||||
|
||||
Deliberately sequential. `EnrichmentPipeline`'s five-way concurrency is for
|
||||
per-row work against a local corpus; firing five brand-corpus fetches at
|
||||
Open Food Facts at once is how a shared host earns a rate limit.
|
||||
"""
|
||||
out: Dict[str, Dict[str, int]] = {}
|
||||
names = [b.strip() for b in brands if b and b.strip()]
|
||||
for i, brand in enumerate(names):
|
||||
if progress_cb:
|
||||
progress_cb(i, len(names))
|
||||
try:
|
||||
stats = enrich_brand_barcodes(brand, min_similarity=min_similarity,
|
||||
dry_run=dry_run)
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.warning("barcode enrichment failed for %s: %s", brand, e)
|
||||
continue
|
||||
if stats.get("candidates"):
|
||||
out[brand] = stats
|
||||
logger.info("%s: %d without a barcode, %d matched, %d written",
|
||||
brand, stats["candidates"], stats["matched"], stats["written"])
|
||||
return out
|
||||
Reference in New Issue
Block a user