backend stores data file enrichment pipeline

This commit is contained in:
sriram
2026-08-18 16:58:36 +05:30
parent 691efef880
commit 7a4583372f
36 changed files with 4599 additions and 0 deletions

View File

@@ -0,0 +1,50 @@
"""
Product Enrichment Service
===========================
WHY THIS PACKAGE EXISTS
------------------------
Everything upstream of this package (ollama_service.py, product_validator.py,
sku_service.py, image_search.py, ...) either GENERATES a product row or
JUDGES whether an already-generated row is plausible. None of it ever goes
out and fetches a piece of ground-truth data from an authoritative external
source and attaches it to the row. That is what this package is for.
The first concrete enricher is barcode lookup (see `barcode/`): resolving a
real, checksum-valid EAN-13/UPC/GTIN for a catalog row from trusted external
sources, never from the LLM. The package is deliberately structured so this
is ONE stage among what will eventually be several independent ones (HSN
code, GST rate, nutrition, allergens, manufacturer details, ...) - see
`base.py` for the `EnrichmentStage` contract every future stage implements,
and `pipeline.py` for the orchestrator that runs them in sequence.
DESIGN PRINCIPLES (apply to every enrichment stage, present and future)
-------------------------------------------------------------------------
- NEVER FABRICATE. If a stage cannot find authentic, verifiable data, it
writes NULL/None + a "not_found" status - never a guessed or LLM-
generated value. This mirrors the whole point of `product_validator.py`
and `sku_service.py`'s "real ID or clearly-labelled internal fallback"
design, taken one step further: for barcodes there IS no safe synthetic
fallback (a fabricated barcode is actively harmful - it can collide with
a real product), so the fallback is always NULL, never a generated value.
- NEVER RAISE. A single product's enrichment failure (timeout, malformed
response, source outage) must never abort the batch or the pipeline run.
Every stage catches its own errors and degrades to "not found" for that
one row.
- ADDITIVE. This package has no dependents before this change and does not
modify the behaviour of any existing module; it is only ever imported
from new call sites (see `app/core/catalog_engine.py`'s Step 2.6 and
`app/services/vector_store.py`'s additive barcode columns).
- INDEPENDENT STAGES. Each stage owns its own external calls, matching
rules, caching, and DB columns. A future HSN/GST/nutrition stage does not
need to know barcode lookup exists, and vice versa - see `pipeline.py`.
"""
from app.services.enrichment.base import EnrichmentStage, StageOutcome
from app.services.enrichment.pipeline import EnrichmentPipeline, run_default_pipeline
__all__ = [
"EnrichmentStage",
"StageOutcome",
"EnrichmentPipeline",
"run_default_pipeline",
]

View File

@@ -0,0 +1,47 @@
"""
Barcode Retrieval & Product Enrichment module.
Public entry points:
lookup_barcode(brand, product_title, size, category="") -> BarcodeResult
Synchronous single-product cascading lookup. Safe to call from a
script, notebook, or the Streamlit UI's "look up a barcode" action.
batch_lookup_barcodes(products, brand) -> list[BarcodeResult]
Async batch lookup used by the pipeline stage (see `stage.py`) and
available directly for a CLI/manual batch job.
See the package's module docstrings for the full architecture:
models.py - BarcodeCandidate / BarcodeResult / enums
validators.py - EAN-13/GTIN/UPC checksum + format validation
matching.py - exact product matching rules
cache.py - local SQLite lookup cache
retry.py - tenacity-based retry/backoff
sources/ - the 4-tier cascading source registry
service.py - orchestrates cache -> sources -> validate -> match
stage.py - adapts the service to the generic EnrichmentStage
contract used by app/services/enrichment/pipeline.py
"""
from typing import Any, Dict, List
from app.services.enrichment.barcode.models import BarcodeCandidate, BarcodeResult, BarcodeType, LookupStatus
from app.services.enrichment.barcode.service import BarcodeLookupService, get_default_service
def lookup_barcode(brand: str, product_title: str, size: str, category: str = "") -> BarcodeResult:
return get_default_service().lookup_one(brand, product_title, size, category)
async def batch_lookup_barcodes(products: List[Dict[str, Any]], brand: str) -> List[BarcodeResult]:
return await get_default_service().batch_lookup(products, brand)
__all__ = [
"BarcodeCandidate",
"BarcodeResult",
"BarcodeType",
"LookupStatus",
"BarcodeLookupService",
"get_default_service",
"lookup_barcode",
"batch_lookup_barcodes",
]

View File

@@ -0,0 +1,127 @@
"""
Local lookup cache for barcode results.
WHY A SEPARATE SQLITE FILE (data/cache/barcode_lookup.db) INSTEAD OF
POSTGRES
---------------------------------------------------------------------
This cache exists purely to avoid repeating the SAME external-API search
(GS1/Open Food Facts/UPCItemDB/manufacturer site) twice for the same
(brand, product, size) - see "Cache barcode lookups to avoid repeated API
calls" / "Prevent duplicate barcode searches" (Performance Requirements).
It is deliberately NOT the authoritative store (that is Postgres, via
`vector_store.py`'s additive barcode columns) - it needs to work even when
Postgres is unreachable/USE_PGVECTOR=false, and it needs to be cheap and
local on an 8GB-RAM/CPU-only machine, so plain stdlib `sqlite3` (no new
dependency, no server process) is the right tool here, following the same
"file-backed, atomic-write" spirit as `sku_service.py`'s SKU sequence
counter.
A row's cache key is a normalized (brand, product_title, size) triple, not
the barcode itself (we're caching "what did we already look up", not "what
maps to what").
"""
from __future__ import annotations
import json
import logging
import re
import sqlite3
import threading
import time
from pathlib import Path
from typing import Optional
from app.services.enrichment.barcode.models import BarcodeResult
logger = logging.getLogger(__name__)
_DB_PATH = Path("data") / "cache" / "barcode_lookup.db"
_lock = threading.Lock()
_initialized = False
def _normalize_key_part(text: Optional[str]) -> str:
return re.sub(r"\s+", " ", (text or "").strip().lower())
def cache_key(brand: str, product_title: str, size: str) -> str:
return f"{_normalize_key_part(brand)}|{_normalize_key_part(product_title)}|{_normalize_key_part(size)}"
def _connect() -> sqlite3.Connection:
_DB_PATH.parent.mkdir(parents=True, exist_ok=True)
conn = sqlite3.connect(str(_DB_PATH), timeout=10)
conn.execute("PRAGMA journal_mode=WAL")
return conn
def _ensure_schema(conn: sqlite3.Connection) -> None:
global _initialized
if _initialized:
return
conn.execute(
"""
CREATE TABLE IF NOT EXISTS barcode_lookup_cache (
cache_key TEXT PRIMARY KEY,
brand TEXT,
product_title TEXT,
size TEXT,
result_json TEXT NOT NULL,
created_at REAL NOT NULL
)
"""
)
conn.commit()
_initialized = True
def get_cached(brand: str, product_title: str, size: str, ttl_seconds: float) -> Optional[BarcodeResult]:
"""Returns a cached BarcodeResult if present and not older than
`ttl_seconds`, else None. Never raises - a cache read failure is
treated exactly like a cache miss."""
key = cache_key(brand, product_title, size)
try:
with _lock, _connect() as conn:
_ensure_schema(conn)
row = conn.execute(
"SELECT result_json, created_at FROM barcode_lookup_cache WHERE cache_key = ?",
(key,),
).fetchone()
except Exception as e:
logger.debug(f"Barcode cache read failed for '{key}': {e}")
return None
if not row:
return None
result_json, created_at = row
if ttl_seconds > 0 and (time.time() - created_at) > ttl_seconds:
return None
try:
data = json.loads(result_json)
return BarcodeResult(**data)
except Exception as e:
logger.debug(f"Barcode cache entry for '{key}' unreadable, treating as miss: {e}")
return None
def set_cached(brand: str, product_title: str, size: str, result: BarcodeResult) -> None:
"""Best-effort write - a failure here never blocks the lookup itself,
it just means this row won't benefit from caching next time."""
key = cache_key(brand, product_title, size)
try:
payload = json.dumps(result.__dict__)
with _lock, _connect() as conn:
_ensure_schema(conn)
conn.execute(
"""
INSERT INTO barcode_lookup_cache (cache_key, brand, product_title, size, result_json, created_at)
VALUES (?, ?, ?, ?, ?, ?)
ON CONFLICT(cache_key) DO UPDATE SET
result_json = excluded.result_json,
created_at = excluded.created_at
""",
(key, brand, product_title, size, payload, time.time()),
)
conn.commit()
except Exception as e:
logger.debug(f"Barcode cache write failed for '{key}': {e}")

View File

@@ -0,0 +1,141 @@
"""
Exact-product matching rules for barcode candidates.
A source returning SOME EAN-13 for a product with a similar name is not
good enough - the "Product Matching Rules" requirement is explicit that
brand, product name, variant, and size must all match, and that a
different pack size (or a differently-branded variant like "Sugar-Free")
must be REJECTED rather than accepted as "close enough". This module is
deliberately conservative: a candidate only passes if every check agrees;
`is_valid_ean13`-style tricks are not enough to earn a false positive here
the way they might in a fuzzy search feature.
Deliberately stdlib-only (`difflib`, already in the standard library) -
this project explicitly removed `fuzzywuzzy`/`python-Levenshtein` as dead
weight (see requirements.txt), so no new fuzzy-matching dependency is
introduced here either.
"""
from __future__ import annotations
import re
from difflib import SequenceMatcher
from typing import Iterable, Optional
from app.services.enrichment.barcode.models import BarcodeCandidate
# quantity_utils rather than image_search: this repo's image_search.py has no
# quantity helpers, and it is a working file the store-catalog work leaves alone.
from app.services.quantity_utils import quantities_match
# Words that mark a genuinely DIFFERENT retail variant from the plain/base
# product - if the candidate's title contains one of these and the
# target product's own title/product_name does NOT, the candidate is
# rejected outright regardless of how well the rest of the name matches
# (this is exactly the "Marie Gold 200g" vs "Marie Gold Sugar-Free"
# example from the spec). Kept as a small, explicit, reviewable list
# rather than a fuzzy heuristic - false negatives (missing a real match)
# are far cheaper here than false positives (storing the wrong product's
# barcode).
VARIANT_DISTINGUISHING_TERMS = {
"sugar free", "sugarfree", "sugar-free", "no sugar", "zero sugar",
"diet", "family pack", "value pack", "jumbo pack", "combo pack",
"combo", "gift pack", "gift box", "mini pack", "party pack",
"refill pack", "refill", "pouch pack", "twin pack", "saver pack",
"economy pack", "jar", "tin", "pet jar",
}
_STOPWORDS = {"the", "and", "of", "with", "for", "a", "an", "new", "pack", "india"}
def _normalize(text: Optional[str]) -> str:
return re.sub(r"[^a-z0-9\s]", " ", (text or "").lower()).strip()
def _tokens(text: Optional[str]) -> set:
return {t for t in _normalize(text).split() if t and t not in _STOPWORDS}
def brand_matches(candidate_brand: str, target_brand: str, brand_aliases: Optional[Iterable[str]] = None) -> bool:
"""True if the candidate's reported brand plausibly refers to the same
brand as the target. Accepts an exact/substring match on the target
brand name itself, or a match against any known alias (e.g. "HUL" for
"Hindustan Unilever") supplied by the caller via
`brand_registry.get_brand_alias_set()`."""
cand = _normalize(candidate_brand)
target = _normalize(target_brand)
if not cand or not target:
return False
if target in cand or cand in target:
return True
for alias in (brand_aliases or []):
alias_n = _normalize(alias)
if alias_n and (alias_n in cand or cand in alias_n):
return True
return False
def size_matches(candidate_size: str, target_size: str, tolerance: float = 0.03) -> bool:
"""Tight tolerance (3%, vs. the 15% used for image matching elsewhere
in this project) - a barcode belongs to exactly one pack size, so
"close enough" is not an acceptable bar the way it is for reusing a
product photo. Falls back to a plain normalized-string equality check
when neither string parses as a numeric quantity (e.g. count-based
sizes like "10 tablets"), rather than silently treating unparsable
sizes as a match.
"""
if not candidate_size or not target_size:
return False
if quantities_match(candidate_size, target_size, tolerance=tolerance):
return True
return _normalize(candidate_size) == _normalize(target_size)
def has_conflicting_variant_terms(candidate_title: str, target_title: str) -> bool:
"""True if the candidate's title names a distinguishing variant
(sugar-free, family pack, ...) that the target product does NOT -
which means the candidate is a real but DIFFERENT product, not the one
being looked up."""
cand_n = _normalize(candidate_title)
target_n = _normalize(target_title)
for term in VARIANT_DISTINGUISHING_TERMS:
if term in cand_n and term not in target_n:
return True
return False
def name_similarity(candidate_title: str, target_title: str) -> float:
"""0.0-1.0 token-overlap-weighted similarity. Used only as a secondary
signal / diagnostic (`match_confidence`) - never as the sole gate for
acceptance; see `is_match()`."""
cand_tokens, target_tokens = _tokens(candidate_title), _tokens(target_title)
if not cand_tokens or not target_tokens:
return 0.0
overlap = len(cand_tokens & target_tokens) / len(target_tokens)
seq_ratio = SequenceMatcher(None, _normalize(candidate_title), _normalize(target_title)).ratio()
return round((overlap * 0.6) + (seq_ratio * 0.4), 3)
def is_match(candidate: BarcodeCandidate, target_brand: str, target_title: str, target_size: str,
brand_aliases: Optional[Iterable[str]] = None,
min_name_similarity: float = 0.45) -> tuple[bool, float]:
"""The combined gate a candidate must pass to be accepted:
1. Brand matches (or overlaps a known alias).
2. Pack size matches within a tight tolerance.
3. No conflicting variant terms (family pack / sugar-free / ...).
4. Product-name similarity clears a floor - catches the case where
brand+size coincidentally match but it's a completely different
product line from the same brand.
Returns (matched, confidence) - confidence is diagnostic only, stored
on the result for audit/QA but never used to override rule 1-3.
"""
if not brand_matches(candidate.candidate_brand, target_brand, brand_aliases):
return False, 0.0
if not size_matches(candidate.candidate_size, target_size):
return False, 0.0
if has_conflicting_variant_terms(candidate.candidate_title, target_title):
return False, 0.0
similarity = name_similarity(candidate.candidate_title, target_title)
if similarity < min_name_similarity:
return False, similarity
return True, similarity

View File

@@ -0,0 +1,82 @@
"""Data shapes shared across the barcode lookup module."""
from __future__ import annotations
import time
from dataclasses import dataclass, field
from enum import Enum
from typing import Optional
class BarcodeType(str, Enum):
EAN13 = "EAN-13"
UPC_A = "UPC-A"
GTIN14 = "GTIN-14"
GTIN8 = "GTIN-8"
UNKNOWN = "UNKNOWN"
class LookupStatus(str, Enum):
"""Mirrors the `barcode_lookup_status` DB column."""
VERIFIED = "verified" # found + checksum-valid + matched product
NOT_FOUND = "not_found" # every source exhausted, nothing matched
INVALID_CANDIDATE = "invalid_candidate" # a candidate was found but failed
# checksum/format validation or product matching - rejected, not stored
ERROR = "error" # a source/network error prevented a full search
DISABLED = "disabled" # barcode lookup turned off via settings
CACHED = "cached" # served from the local lookup cache
@dataclass
class BarcodeCandidate:
"""A raw, not-yet-validated candidate returned by one source."""
barcode: str
source_name: str # e.g. "Open Food Facts"
candidate_title: str = "" # the source's own product title, for matching
candidate_brand: str = ""
candidate_size: str = ""
candidate_countries: str = "" # raw country tag/string from the source, if any
@dataclass
class BarcodeResult:
"""Final, caller-facing result. Field names match the DB columns in
`vector_store.py` and the JSON export shape requested for the
catalog (`barcode`, `barcode_type`, `barcode_source`, `barcode_verified`,
...) 1:1, so `service.py` -> product dict -> DB row -> JSON export is a
straight field copy with no renaming at any layer.
"""
barcode: Optional[str] = None
barcode_type: Optional[str] = None
gtin: Optional[str] = None
ean13: Optional[str] = None
upc: Optional[str] = None
barcode_source: Optional[str] = None
barcode_verified: bool = False
barcode_lookup_status: str = LookupStatus.NOT_FOUND.value
barcode_last_updated: float = field(default_factory=time.time)
match_confidence: float = 0.0
@classmethod
def null_result(cls, status: LookupStatus = LookupStatus.NOT_FOUND) -> "BarcodeResult":
"""The required NULL shape: barcode=NULL, barcode_verified=false,
barcode_source=NULL - used for every path where no authentic,
matching barcode could be confirmed. Never construct a
BarcodeResult with a barcode value outside of `service.py`'s
validated-and-matched path.
"""
return cls(barcode_lookup_status=status.value)
def as_product_fields(self) -> dict:
"""Flat dict merged directly into the catalog row - see
`stage.py`."""
return {
"barcode": self.barcode,
"barcode_type": self.barcode_type,
"gtin": self.gtin,
"ean13": self.ean13,
"upc": self.upc,
"barcode_source": self.barcode_source,
"barcode_verified": self.barcode_verified,
"barcode_lookup_status": self.barcode_lookup_status,
"barcode_last_updated": self.barcode_last_updated,
}

View File

@@ -0,0 +1,47 @@
"""
Configurable retry-with-exponential-backoff for barcode source calls.
Built on `tenacity`, already a project dependency (see requirements.txt -
used elsewhere for httpx-based retries). Only retries on genuinely
transient failures (network/timeout errors); a malformed response or a
"no results" outcome is not retried, since retrying those wastes the
source's rate-limit budget for no benefit (relevant for UPCItemDB's
trial-tier daily cap in particular).
"""
from __future__ import annotations
import logging
import requests
from tenacity import (
retry,
retry_if_exception_type,
stop_after_attempt,
wait_exponential,
before_sleep_log,
)
logger = logging.getLogger(__name__)
_RETRYABLE_EXCEPTIONS = (
requests.exceptions.ConnectionError,
requests.exceptions.Timeout,
requests.exceptions.ChunkedEncodingError,
)
def with_retry(max_attempts: int = 3, min_wait: float = 1.0, max_wait: float = 8.0):
"""Decorator factory: `max_attempts` total tries, exponential backoff
between `min_wait` and `max_wait` seconds. Applied per-source (see
`sources/*.py`), not globally, so one slow/unreliable source retrying
doesn't compound delay across the whole cascade - a source that keeps
failing simply falls through to the next tier faster than a shared
global retry budget would allow.
"""
return retry(
reraise=True,
stop=stop_after_attempt(max_attempts),
wait=wait_exponential(multiplier=min_wait, max=max_wait),
retry=retry_if_exception_type(_RETRYABLE_EXCEPTIONS),
before_sleep=before_sleep_log(logger, logging.DEBUG),
)

View File

@@ -0,0 +1,170 @@
"""
BarcodeLookupService - the cascading lookup orchestrator.
This is the ONE place that decides "is this barcode good enough to store".
Every candidate from every source, regardless of tier, passes through the
exact same two gates before it can become a `BarcodeResult`:
1. `validators.validate_barcode()` - checksum + format.
2. `matching.is_match()` - brand + size + variant + name.
A source being "trusted" (e.g. GS1 India) does not skip either gate -
trust only affects ORDER (which tier is tried first), never whether
validation is required.
Cascade behaviour (see module docstring in `sources/__init__.py` for the
tier list): tiers are tried in order; the first tier that produces at
least one validated, matching candidate wins and the search stops - "The
system should stop searching as soon as a verified barcode is found."
Every tier is independently wrapped so a source outage never prevents
falling through to the next one, and if every tier is exhausted with
nothing matching, the result is the required NULL shape
(`BarcodeResult.null_result()`), never a guess.
"""
from __future__ import annotations
import asyncio
import logging
from typing import Any, Dict, List, Optional
from app.infrastructure.settings import (
ENABLE_BARCODE_LOOKUP,
BARCODE_LOOKUP_CACHE_TTL_SECONDS,
BARCODE_LOOKUP_MAX_CONCURRENCY,
)
from app.services.enrichment.barcode import cache
from app.services.enrichment.barcode.matching import is_match
from app.services.enrichment.barcode.models import BarcodeCandidate, BarcodeResult, BarcodeType, LookupStatus
from app.services.enrichment.barcode.sources import get_default_sources
from app.services.enrichment.barcode.validators import classify_barcode_type, to_ean13, validate_barcode
logger = logging.getLogger(__name__)
try:
from app.services.brand_registry import get_brand_alias_set
except Exception: # pragma: no cover - keeps the service importable/testable
# in isolation even if brand_registry ever fails to import.
def get_brand_alias_set(_brand: str) -> set:
return set()
def _build_result(barcode: str, source_name: str, confidence: float) -> BarcodeResult:
btype = classify_barcode_type(barcode)
ean13 = to_ean13(barcode) if btype in (BarcodeType.UPC_A, BarcodeType.EAN13) else None
return BarcodeResult(
barcode=barcode,
barcode_type=btype.value,
gtin=barcode,
ean13=ean13,
upc=barcode if btype == BarcodeType.UPC_A else None,
barcode_source=source_name,
barcode_verified=True,
barcode_lookup_status=LookupStatus.VERIFIED.value,
match_confidence=confidence,
)
class BarcodeLookupService:
def __init__(self, sources=None):
self.sources = sources if sources is not None else get_default_sources()
def lookup_one(self, brand: str, product_title: str, size: str, category: str = "",
use_cache: bool = True) -> BarcodeResult:
"""Synchronous cascading lookup for a single (brand, product,
size). Never raises - every failure path returns a NULL-shaped
`BarcodeResult` with a descriptive `barcode_lookup_status`."""
if not ENABLE_BARCODE_LOOKUP:
return BarcodeResult.null_result(LookupStatus.DISABLED)
if not brand or not product_title or not size:
logger.debug(f"Barcode lookup skipped - missing brand/title/size ('{brand}', '{product_title}', '{size}')")
return BarcodeResult.null_result(LookupStatus.ERROR)
if use_cache:
cached = cache.get_cached(brand, product_title, size, BARCODE_LOOKUP_CACHE_TTL_SECONDS)
if cached is not None:
result = cached
result.barcode_lookup_status = (
LookupStatus.CACHED.value if result.barcode_verified else result.barcode_lookup_status
)
return result
brand_aliases = self._safe_brand_aliases(brand)
had_source_error = False
for source in self.sources:
if not source.available:
continue
try:
raw_candidates = source.search(brand, product_title, size, category)
except Exception as e:
# Sources are documented to never raise, but this is the
# pipeline-wide safety net referenced in base.py.
logger.warning(f"[{source.name}] barcode search raised unexpectedly: {e}")
had_source_error = True
continue
result = self._first_validated_match(raw_candidates, brand, product_title, size, brand_aliases)
if result is not None:
if use_cache:
cache.set_cached(brand, product_title, size, result)
return result
status = LookupStatus.ERROR if had_source_error else LookupStatus.NOT_FOUND
result = BarcodeResult.null_result(status)
if use_cache:
# Cache negative results too (Performance Requirements: "Prevent
# duplicate barcode searches") - a short TTL still applies, so a
# transient "not found" doesn't permanently block a later re-check.
cache.set_cached(brand, product_title, size, result)
return result
@staticmethod
def _safe_brand_aliases(brand: str) -> set:
try:
return get_brand_alias_set(brand)
except Exception:
return set()
@staticmethod
def _first_validated_match(raw_candidates: List[BarcodeCandidate], brand: str, product_title: str,
size: str, brand_aliases: set) -> Optional[BarcodeResult]:
for candidate in raw_candidates:
clean_barcode = validate_barcode(candidate.barcode)
if not clean_barcode:
continue # invalid checksum/length/format - never stored, not even flagged
matched, confidence = is_match(candidate, brand, product_title, size, brand_aliases)
if not matched:
continue
return _build_result(clean_barcode, candidate.source_name, confidence)
return None
async def batch_lookup(self, products: List[Dict[str, Any]], brand: str) -> List[BarcodeResult]:
"""Async batch entry point used by `stage.py`. Bounded concurrency
(`BARCODE_LOOKUP_MAX_CONCURRENCY`) so a large brand catalog doesn't
fire dozens of simultaneous requests at any one source, and results
are returned in the SAME ORDER as `products` regardless of which
finished first, so callers can zip() them back together safely."""
semaphore = asyncio.Semaphore(max(1, BARCODE_LOOKUP_MAX_CONCURRENCY))
async def _one(product: Dict[str, Any]) -> BarcodeResult:
async with semaphore:
return await asyncio.to_thread(
self.lookup_one,
brand,
product.get("title") or product.get("product_name") or "",
product.get("size") or "",
product.get("category") or "",
)
return await asyncio.gather(*(_one(p) for p in products))
# Module-level singleton - sources are stateless, so one shared instance is
# fine for both the sync CLI/test path and the async pipeline stage.
_default_service: Optional[BarcodeLookupService] = None
def get_default_service() -> BarcodeLookupService:
global _default_service
if _default_service is None:
_default_service = BarcodeLookupService()
return _default_service

View File

@@ -0,0 +1,35 @@
"""
Tier-ordered source registry - this list IS the cascade order described in
the spec: GS1 India -> Open Food Facts -> trusted barcode databases ->
manufacturer website -> (exhausted -> NULL). `service.py` iterates this
list in order and stops at the first source that produces a validated,
matching candidate.
"""
from app.services.enrichment.barcode.sources.base import BarcodeSource
from app.services.enrichment.barcode.sources.gs1_india import GS1IndiaSource
from app.services.enrichment.barcode.sources.open_food_facts import OpenFoodFactsSource
from app.services.enrichment.barcode.sources.upc_database import UPCDatabaseSource
from app.services.enrichment.barcode.sources.manufacturer_site import ManufacturerSiteSource
def get_default_sources() -> list[BarcodeSource]:
"""Fresh instances each call - sources are stateless/cheap to
construct, and this avoids any shared-mutable-state surprises across
concurrent batch lookups."""
sources = [
GS1IndiaSource(),
OpenFoodFactsSource(),
UPCDatabaseSource(),
ManufacturerSiteSource(),
]
return sorted(sources, key=lambda s: s.tier)
__all__ = [
"BarcodeSource",
"GS1IndiaSource",
"OpenFoodFactsSource",
"UPCDatabaseSource",
"ManufacturerSiteSource",
"get_default_sources",
]

View File

@@ -0,0 +1,46 @@
"""
Pluggable barcode source contract, following the same adapter-registry
pattern already used for marketplace SKU resolution
(app/services/sku_service.py's `_MARKETPLACE_PATTERNS`) and the Product
Resolver's adapter registry (see the Catalog_Project's `resolve_parent_brand`
family) - every source is a self-contained class with one job: given a
brand/title/size, return zero or more raw candidates. It does NOT decide
whether a candidate is a match (that's `matching.py`) or whether its
barcode is valid (that's `validators.py`) - keeping those concerns
separate is what lets `service.py` apply the SAME validation/matching gate
uniformly regardless of which tier produced the candidate.
"""
from __future__ import annotations
from abc import ABC, abstractmethod
from typing import List
from app.services.enrichment.barcode.models import BarcodeCandidate
class BarcodeSource(ABC):
#: Human-readable label stored in `barcode_source` on a successful
#: match, e.g. "Open Food Facts", "GS1 India".
name: str = "Unknown Source"
#: Lower = tried first in the cascade (see sources/__init__.py).
tier: int = 99
@property
def available(self) -> bool:
"""False when the source is unusable for reasons unrelated to any
one query (missing API key, disabled via settings, dependency not
installed, ...). The cascade skips unavailable sources entirely
rather than querying and failing every time."""
return True
@abstractmethod
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
"""Best-effort search. MUST NOT raise - catch internally and
return [] on any failure (network error, malformed response,
nothing found). `service.py` treats an empty list and an
exception identically (fall through to the next tier), so
swallowing the error here vs. letting it propagate makes no
behavioural difference except robustness - always swallow it.
"""
raise NotImplementedError

View File

@@ -0,0 +1,91 @@
"""
Tier 1: GS1 India / GEPIR (the official Indian GTIN registry).
HONESTY NOTE - READ BEFORE WIRING UP CREDENTIALS
--------------------------------------------------
GS1 India does not currently publish a free, public, no-registration JSON
API for GTIN lookup (unlike Open Food Facts). Their real product-data
registry is accessed either through GEPIR (https://gepir.gs1.org/ - a
human web UI with no documented public API) or through GS1 India's paid
"Verified by GS1" data-as-a-service product, which requires a commercial
account and an API key.
Rather than scrape GEPIR's web UI (fragile, likely against its terms, and
exactly the kind of "pretend this is a real integration" shortcut this
project's whole hallucination-prevention philosophy exists to avoid - see
product_validator.py's module docstring), this adapter is a REAL,
WORKING integration point that:
- does nothing (returns [], `available=False`) when no credentials are
configured, so the cascade cleanly falls through to Open Food Facts
(Tier 2) - exactly the "if not found -> next step" behaviour the
spec asks for, just starting from Tier 2 in practice until GS1 India
API access is provisioned;
- immediately becomes live the moment `GS1_INDIA_API_BASE_URL` and
`GS1_INDIA_API_KEY` are set (see .env.example), with no code changes
needed elsewhere - `service.py` already treats every tier uniformly.
If/when real GS1 India API access is provisioned, fill in the request
shape in `search()` below to match that API's actual contract (endpoint
path, auth header, response schema) - the surrounding plumbing
(validation, matching, caching, retry) does not need to change.
"""
from __future__ import annotations
import logging
from typing import List
import requests
from app.infrastructure.settings import GS1_INDIA_API_BASE_URL, GS1_INDIA_API_KEY, BARCODE_LOOKUP_TIMEOUT_SECONDS
from app.services.enrichment.barcode.models import BarcodeCandidate
from app.services.enrichment.barcode.retry import with_retry
from app.services.enrichment.barcode.sources.base import BarcodeSource
logger = logging.getLogger(__name__)
class GS1IndiaSource(BarcodeSource):
name = "GS1 India"
tier = 1
@property
def available(self) -> bool:
return bool(GS1_INDIA_API_BASE_URL and GS1_INDIA_API_KEY)
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
if not self.available:
return []
@with_retry(max_attempts=3)
def _call():
return requests.get(
f"{GS1_INDIA_API_BASE_URL.rstrip('/')}/search",
params={"brand": brand, "product": product_title, "size": size, "country": "India"},
headers={"Authorization": f"Bearer {GS1_INDIA_API_KEY}"},
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
)
try:
resp = _call()
if resp.status_code != 200:
logger.debug(f"GS1 India lookup non-200 ({resp.status_code}) for '{brand} {product_title} {size}'")
return []
data = resp.json()
except Exception as e:
logger.debug(f"GS1 India lookup failed for '{brand} {product_title} {size}': {e}")
return []
candidates: List[BarcodeCandidate] = []
for item in (data.get("results") or data.get("items") or []):
gtin = item.get("gtin") or item.get("barcode") or item.get("code")
if not gtin:
continue
candidates.append(BarcodeCandidate(
barcode=str(gtin),
source_name=self.name,
candidate_title=item.get("productName") or item.get("title") or "",
candidate_brand=item.get("brandName") or item.get("brand") or "",
candidate_size=item.get("netContent") or item.get("size") or "",
candidate_countries="in",
))
return candidates

View File

@@ -0,0 +1,108 @@
"""
Tier 4: manufacturer / official product page - last resort.
Follows the exact same "DuckDuckGo search -> read structure straight off
the result, no page-scraping HTML parser" spirit as
`app/services/sku_service.py`'s `find_website_product_id()`, except here
we DO need to fetch the page text (a barcode isn't embedded in the result
URL the way an Amazon ASIN is), so this is the one tier that performs a
capped number of lightweight page fetches, each wrapped in its own
try/except so one slow/broken page can never block the others.
This tier is intentionally the lowest-trust one: a number that merely sits
near the word "barcode"/"EAN"/"UPC"/"GTIN" on a webpage is only a
CANDIDATE. It still has to pass `validators.validate_barcode()` (checksum)
and `matching.is_match()` (brand/size/name) in `service.py` exactly like
every other tier's candidates - nothing here is trusted just because it
came from what looks like an official page.
"""
from __future__ import annotations
import logging
import re
from typing import List
import requests
from app.infrastructure.settings import BARCODE_LOOKUP_TIMEOUT_SECONDS, ENABLE_MANUFACTURER_SITE_LOOKUP
from app.services.enrichment.barcode.models import BarcodeCandidate
from app.services.enrichment.barcode.retry import with_retry
from app.services.enrichment.barcode.sources.base import BarcodeSource
logger = logging.getLogger(__name__)
_BROWSER_UA = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
)
# A barcode label followed, within a short distance, by an 8/12/13/14-digit
# run. Intentionally permissive on the label (source pages phrase this
# differently) and tight on the digit run (only plausible GTIN lengths) -
# every hit is still just a CANDIDATE, checksum-validated afterwards.
_BARCODE_NEAR_LABEL_RE = re.compile(
r"(?:barcode|ean|upc|gtin)\D{0,15}(\d{8}|\d{12,14})",
re.IGNORECASE,
)
_MAX_PAGES = 3
class ManufacturerSiteSource(BarcodeSource):
name = "Manufacturer Website"
tier = 4
@property
def available(self) -> bool:
if not ENABLE_MANUFACTURER_SITE_LOOKUP:
return False
try:
import ddgs # noqa: F401
return True
except ImportError:
logger.debug("Manufacturer-site barcode lookup unavailable: 'ddgs' package not installed")
return False
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
if not self.available:
return []
query = f"{brand} {product_title} {size} barcode EAN".strip()
try:
from ddgs import DDGS
with DDGS(timeout=10) as ddgs:
results = ddgs.text(query, region="in-en", safesearch="off", max_results=_MAX_PAGES)
except Exception as e:
logger.debug(f"Manufacturer-site search failed for '{query}': {e}")
return []
candidates: List[BarcodeCandidate] = []
for r in (results or [])[:_MAX_PAGES]:
url = str(r.get("href") or r.get("url") or "")
if not url.startswith("http"):
continue
for code in self._extract_barcodes(url):
candidates.append(BarcodeCandidate(
barcode=code,
source_name=f"{self.name} ({url})",
candidate_title=r.get("title") or product_title,
candidate_brand=brand,
candidate_size=size,
candidate_countries="",
))
return candidates
def _extract_barcodes(self, url: str) -> List[str]:
@with_retry(max_attempts=2)
def _call():
return requests.get(url, headers={"User-Agent": _BROWSER_UA}, timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS)
try:
resp = _call()
if resp.status_code != 200:
return []
text = resp.text[:200_000] # cap - this is a text scan, not a full-page render
except Exception as e:
logger.debug(f"Manufacturer page fetch failed for '{url}': {e}")
return []
return [m.group(1) for m in _BARCODE_NEAR_LABEL_RE.finditer(text)]

View File

@@ -0,0 +1,118 @@
"""
Tier 2: Open Food Facts / Open Beauty Facts / Open Products Facts.
Real, free, no-API-key public search API - the same family of open,
community-maintained product databases already used as the project's
PRIMARY image source (see `app/services/image_search.py`'s
`OPEN_FACTS_HOSTS` / `_query_openfacts`). Every entry is keyed by its real
barcode (the `code` field IS the GTIN/EAN/UPC printed on the physical
pack), which is exactly the ground-truth this module needs - unlike
`image_search.py`, which only reads the image URLs off each entry, this
adapter reads the `code` field itself.
Deliberately a separate, self-contained query function rather than
importing `image_search._query_openfacts` (which is a private, underscore-
prefixed helper): this adapter needs different response fields (`code`,
`countries_tags`) and India-specific filtering that the image-search
helper has no reason to carry. `quantities_match`/`parse_quantity_grams`
ARE imported from `image_search.py` (public functions) for size
comparison, so the size-matching logic itself is not duplicated - see
`matching.py`.
"""
from __future__ import annotations
import logging
from typing import List
import requests
from app.infrastructure.settings import BARCODE_LOOKUP_TIMEOUT_SECONDS, BARCODE_COUNTRY_TAG
from app.services.enrichment.barcode.models import BarcodeCandidate
from app.services.enrichment.barcode.retry import with_retry
from app.services.enrichment.barcode.sources.base import BarcodeSource
logger = logging.getLogger(__name__)
_HOSTS = [
"world.openfoodfacts.org",
"world.openbeautyfacts.org",
"world.openproductsfacts.org",
]
_BROWSER_UA = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
)
_FIELDS = "code,product_name,brands,brands_tags,quantity,countries_tags"
class OpenFoodFactsSource(BarcodeSource):
name = "Open Food Facts"
tier = 2
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
query = f"{brand or ''} {product_title or ''}".strip()
if not query:
return []
products = self._query(query)
if not products and brand and product_title:
# Combined query too specific (common for regional brand-name
# variants) - retry title-only, same fallback image_search.py uses.
products = self._query(product_title)
candidates: List[BarcodeCandidate] = []
for item in products:
code = str(item.get("code") or "").strip()
if not code:
continue
countries = ",".join(item.get("countries_tags") or [])
candidates.append(BarcodeCandidate(
barcode=code,
source_name=self.name,
candidate_title=item.get("product_name") or "",
candidate_brand=item.get("brands") or "",
candidate_size=item.get("quantity") or "",
candidate_countries=countries,
))
# Soft India-market prioritisation: candidates whose countries_tags
# mention the target market are tried first, but non-tagged/other-
# market candidates are kept (not dropped) since many genuine
# Indian FMCG entries simply have this field blank upstream - the
# brand/size/name gate in matching.py is what actually decides
# correctness, this only affects which validated match is found
# (and therefore stops the cascade) first.
if BARCODE_COUNTRY_TAG:
tag = BARCODE_COUNTRY_TAG.lower()
candidates.sort(key=lambda c: 0 if tag in c.candidate_countries.lower() else 1)
return candidates
def _query(self, query: str) -> list:
for host in _HOSTS:
@with_retry(max_attempts=2)
def _call(host=host):
return requests.get(
f"https://{host}/cgi/search.pl",
params={
"search_terms": query,
"search_simple": 1,
"action": "process",
"json": 1,
"page_size": 20,
"fields": _FIELDS,
},
headers={"User-Agent": _BROWSER_UA},
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
)
try:
resp = _call()
if resp.status_code != 200:
continue
products = resp.json().get("products", [])
if products:
return products
except Exception as e:
logger.debug(f"Open*Facts barcode lookup failed on {host} for '{query}': {e}")
continue
return []

View File

@@ -0,0 +1,81 @@
"""
Tier 3: "trusted barcode databases" - UPCItemDB.
Real, free (trial-tier, no API key required, rate-limited) reverse lookup:
search by product name/brand keywords and get back candidate items with
their own `upc`/`ean` fields. This is the general "trusted barcode
database" tier the spec asks for as a fallback below GS1 India / Open
Food Facts.
RATE LIMIT: UPCItemDB's free trial endpoint is capped (documented as
~100 requests/day, ~1 request/second) - this is exactly why this tier
sits BELOW Open Food Facts (unlimited, no key) in the cascade, and why
`service.py` only calls a lower tier at all when every higher tier has
already failed to produce a validated match, plus why the local cache
(`cache.py`) matters most for this specific source. If a paid UPCItemDB
key is available, set `UPC_DATABASE_API_KEY` (see .env.example) to switch
to the production endpoint with a higher quota - the request shape below
already supports both.
"""
from __future__ import annotations
import logging
from typing import List
import requests
from app.infrastructure.settings import BARCODE_LOOKUP_TIMEOUT_SECONDS, UPC_DATABASE_API_KEY
from app.services.enrichment.barcode.models import BarcodeCandidate
from app.services.enrichment.barcode.retry import with_retry
from app.services.enrichment.barcode.sources.base import BarcodeSource
logger = logging.getLogger(__name__)
_TRIAL_URL = "https://api.upcitemdb.com/prod/trial/search"
_PROD_URL = "https://api.upcitemdb.com/prod/v1/search"
class UPCDatabaseSource(BarcodeSource):
name = "UPCItemDB"
tier = 3
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
query = f"{brand or ''} {product_title or ''} {size or ''}".strip()
if not query:
return []
url = _PROD_URL if UPC_DATABASE_API_KEY else _TRIAL_URL
headers = {"Accept": "application/json"}
if UPC_DATABASE_API_KEY:
headers["user_key"] = UPC_DATABASE_API_KEY
headers["key_type"] = "3scale"
@with_retry(max_attempts=2)
def _call():
return requests.get(url, params={"s": query, "type": "product"}, headers=headers,
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS)
try:
resp = _call()
if resp.status_code != 200:
logger.debug(f"UPCItemDB non-200 ({resp.status_code}) for '{query}'")
return []
data = resp.json()
except Exception as e:
logger.debug(f"UPCItemDB lookup failed for '{query}': {e}")
return []
candidates: List[BarcodeCandidate] = []
for item in data.get("items", []):
code = item.get("ean") or item.get("upc")
if not code:
continue
candidates.append(BarcodeCandidate(
barcode=str(code),
source_name=self.name,
candidate_title=item.get("title") or "",
candidate_brand=item.get("brand") or "",
candidate_size=item.get("size") or "",
candidate_countries="",
))
return candidates

View File

@@ -0,0 +1,37 @@
"""Adapts BarcodeLookupService to the EnrichmentStage contract so it can be
registered in app/services/enrichment/pipeline.py's default pipeline."""
from __future__ import annotations
import logging
from typing import Any, Dict
from app.infrastructure.settings import ENABLE_BARCODE_LOOKUP
from app.services.enrichment.base import EnrichmentStage, StageOutcome
from app.services.enrichment.barcode.service import get_default_service
logger = logging.getLogger(__name__)
class BarcodeEnrichmentStage(EnrichmentStage):
name = "barcode_lookup"
@property
def enabled(self) -> bool:
return ENABLE_BARCODE_LOOKUP
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
service = get_default_service()
title = product.get("title") or product.get("product_name") or ""
size = product.get("size") or ""
category = product.get("category") or ""
try:
import asyncio
result = await asyncio.to_thread(service.lookup_one, brand, title, size, category)
except Exception as e:
logger.warning(f"Barcode enrichment failed for '{title}' {size}: {e}")
from app.services.enrichment.barcode.models import BarcodeResult, LookupStatus
result = BarcodeResult.null_result(LookupStatus.ERROR)
return StageOutcome(stage_name=self.name, fields=result.as_product_fields(),
error=None if result.barcode_verified else result.barcode_lookup_status)

View File

@@ -0,0 +1,99 @@
"""
Barcode format + checksum validation.
Deliberately the LAST gate a candidate passes through before it's allowed
into a `BarcodeResult` (see `service.py`), regardless of which source
produced it or how much we otherwise trust that source - a source telling
us "this is the barcode" is never sufficient on its own; the digits have to
actually check out mathematically. This is what "Validate EAN-13 checksum"
/ "Validate GTIN format" / "Reject invalid barcode lengths" / "Reject
malformed barcode values" (Barcode Validation requirements) means in code.
No third-party dependency - GTIN/EAN/UPC-A all share one checksum
algorithm (the classic "alternating 3/1 weights counted from the rightmost
digit before the check digit"), so GTIN-8/12/13/14 are all validated by the
same function; only the accepted LENGTH differs per barcode type.
"""
from __future__ import annotations
import re
from typing import Optional
from app.services.enrichment.barcode.models import BarcodeType
_VALID_LENGTHS = {8: BarcodeType.GTIN8, 12: BarcodeType.UPC_A, 13: BarcodeType.EAN13, 14: BarcodeType.GTIN14}
# Digits only, no separators - callers are expected to call normalize_barcode()
# first if the raw source string may contain spaces/hyphens.
_DIGITS_ONLY_RE = re.compile(r"^\d+$")
def normalize_barcode(raw: Optional[str]) -> Optional[str]:
"""Strip everything but digits (spaces, hyphens, a stray 'EAN:' label,
etc.). Returns None for empty/unusable input - never raises."""
if not raw:
return None
digits = re.sub(r"\D", "", str(raw))
return digits or None
def gtin_check_digit(digits_without_check: str) -> int:
"""Compute the correct check digit for a GTIN-8/12/13/14 payload
(i.e. every digit EXCEPT the check digit itself), using the standard
alternating 3/1 weighting counted from the rightmost digit."""
total = 0
for i, ch in enumerate(reversed(digits_without_check)):
weight = 3 if i % 2 == 0 else 1
total += int(ch) * weight
return (10 - (total % 10)) % 10
def has_valid_checksum(code: str) -> bool:
"""True if `code`'s own last digit matches the checksum computed over
the rest of it. `code` must already be digits-only."""
if not code or not _DIGITS_ONLY_RE.match(code):
return False
body, check_digit = code[:-1], code[-1]
try:
expected = gtin_check_digit(body)
except (ValueError, IndexError):
return False
return str(expected) == check_digit
def classify_barcode_type(code: str) -> BarcodeType:
"""Barcode type purely from its (already checksum-validated) length.
UPC-A (12 digits) is the one ambiguous case worth calling out: it is
numerically a GTIN-13 with a leading zero, but is reported as UPC-A
here since that's the label the requesting spec/UI expects for a
12-digit code."""
return _VALID_LENGTHS.get(len(code), BarcodeType.UNKNOWN)
def validate_barcode(raw: Optional[str]) -> Optional[str]:
"""Single entry point: normalize, check length, check checksum.
Returns the clean digits-only barcode string if and only if it is a
genuinely valid GTIN-8/12/13/14, else None. This is the ONLY function
other modules should call to decide "is this barcode good enough to
store" - never inline a length/regex check elsewhere.
"""
code = normalize_barcode(raw)
if not code:
return None
if len(code) not in _VALID_LENGTHS:
return None
if not has_valid_checksum(code):
return None
return code
def to_ean13(code: str) -> Optional[str]:
"""Zero-pad a valid UPC-A (12 digits) up to its equivalent EAN-13
representation. GTIN-8 is intentionally NOT padded (an 8-digit GTIN is
its own distinct symbology, not a truncated EAN-13) - returns None for
anything that isn't 12 or 13 digits already."""
if len(code) == 13:
return code
if len(code) == 12:
return "0" + code
return None

View File

@@ -0,0 +1,80 @@
"""
Base contract every enrichment stage implements.
A stage takes ONE already-validated catalog row (a plain dict, the same
`enhanced_product` shape `catalog_engine.py` builds) plus the brand name,
and returns the SAME dict with additional fields merged in - it never
removes or renames a key it didn't add itself, and it never raises: any
internal failure is caught and reported via `StageOutcome.error` instead.
To add a new enrichment stage later (HSN, GST, nutrition, allergens, ...):
1. Subclass `EnrichmentStage`.
2. Implement `async def enrich_one(product, brand) -> StageOutcome`.
3. Register an instance in `pipeline.run_default_pipeline()` (or build
a custom `EnrichmentPipeline([...])` for a one-off run).
No other file needs to change - `catalog_engine.py`'s call site and
`vector_store.py`'s upsert already iterate whatever keys are present on the
product dict via `.get(...)`, so a new stage's fields flow through to the
database/JSON export automatically the same way barcode fields do.
"""
from __future__ import annotations
import logging
from abc import ABC, abstractmethod
from dataclasses import dataclass, field
from typing import Any, Dict, Optional
logger = logging.getLogger(__name__)
@dataclass
class StageOutcome:
"""Result of running one stage on one product row."""
stage_name: str
fields: Dict[str, Any] = field(default_factory=dict)
error: Optional[str] = None
@property
def ok(self) -> bool:
return self.error is None
class EnrichmentStage(ABC):
"""One independent, pluggable enrichment step.
`name` is used for logging/metrics only. `enabled` lets a stage report
itself as switched off (e.g. via a settings flag) without the pipeline
orchestrator needing to know why - it's simply skipped and every
product passes through untouched.
"""
name: str = "unnamed_stage"
@property
def enabled(self) -> bool:
return True
@abstractmethod
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
"""Enrich a single product row. MUST NOT raise - catch internally
and return a StageOutcome with `error` set instead."""
raise NotImplementedError
async def apply(self, product: Dict[str, Any], brand: str) -> Dict[str, Any]:
"""Run this stage on `product` and merge the result in place.
Never raises - a stage bug degrades to "no fields added", logged,
rather than aborting the whole catalog row.
"""
try:
outcome = await self.enrich_one(product, brand)
except Exception as e: # last-resort safety net - stages should
# already catch their own errors, but a pipeline-wide guarantee
# of "never raises" is worth the redundancy here.
logger.error(f"[{self.name}] unhandled exception enriching '{product.get('product_name')}': {e}")
return product
if outcome.fields:
product.update(outcome.fields)
if not outcome.ok:
logger.debug(f"[{self.name}] {product.get('product_name')}: {outcome.error}")
return product

View File

@@ -0,0 +1,32 @@
"""
HSN / GST & Pricing enrichment module.
Public entry points:
resolve_hsn_gst(category, product_title="") -> HsnGstInfo
Deterministic, offline category -> (HSN code, GST %, review flag)
resolution, mirroring the shape already present in the coca-cola
and milky_mist catalog exports (hsn_code / gst_percent /
hsn_gst_needs_review / selling_price / tax_amount /
final_selling_price / cost_price / profit_before_tax /
profit_after_tax).
enrich_pricing_fields(product) -> dict
Computes selling_price / tax_amount / final_selling_price (plus the
always-null cost/profit fields) from a catalog row's price_range,
using the same convention as the existing enriched brands: the
selling price is the upper bound of the row's retail price range.
See the package's module docstrings for the architecture:
models.py - HsnGstInfo result type + curated category -> HSN/GST table
stage.py - adapts the resolver to the generic EnrichmentStage
contract used by app/services/enrichment/pipeline.py
"""
from app.services.enrichment.hsn_gst.models import HsnGstInfo, resolve_hsn_gst, enrich_pricing_fields
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
__all__ = [
"HsnGstInfo",
"resolve_hsn_gst",
"enrich_pricing_fields",
"HsnGstEnrichmentStage",
]

View File

@@ -0,0 +1,259 @@
"""
Curated, deterministic HSN / GST lookup for catalog product categories.
WHY THIS IS NOT "MADE UP"
--------------------------
HSN (Harmonized System of Nomenclature) is the 4-8 digit goods-classification
code used on every Indian tax invoice, and GST% is the Indian Goods &
Services Tax rate applied to that HSN chapter. Both are public, legal,
category-level facts - not per-SKU secrets. Every entry in `HSN_GST_TABLE`
below is a *typical* classification for the product category as sold at
retail (matching the values already present in the coca-cola / milky_mist
catalog exports where those overlap, e.g. Beverages -> 2009, Dairy -> 0401,
Dairy - Desserts -> 2105, Tea & Coffee -> 0902).
The catch: an HSN chapter can cover several GST rates depending on the exact
item and how it is packaged, so a category-level guess can be wrong for a
specific product. That is precisely what `HSN_GST_NEEDS_REVIEW` is for - it is
True whenever the mapping is approximate (the default), and False only for the
few categories whose retail classification is unambiguous. The flag exists so
a human reviewer / the Streamlit UI can spot-check exactly these rows.
DESIGN PRINCIPLES
------------------
- DETERMINISTIC & OFFLINE. No LLM, no network, no randomness. The same
category always yields the same (HSN, GST%, review) tuple, so re-runs and
backfills are stable and diffable.
- ADDITIVE. Only ever attaches new keys to a product row; never removes or
rewrites an existing key (the stage skips rows that already carry these
fields, so re-running the pipeline over an already-enriched catalog is a
no-op).
- SAFE DEFAULTS. Unknown/unrecognised categories get hsn_code=None and
gst_percent=None with needs_review=True, rather than a fabricated code -
a wrong HSN on a real tax document is worse than no HSN at all.
"""
from __future__ import annotations
import re
from dataclasses import dataclass
from typing import Dict, Optional, Tuple
# (HSN code, GST %, needs_review). Review=True whenever the category can map
# to more than one GST rate / HSN chapter at retail; False for unambiguous ones.
HSN_GST_TABLE: Dict[str, Tuple[str, int, bool]] = {
# ---- Food & beverage (matches the existing coca-cola / milky_mist exports) ----
"Dairy": ("0401", 5, False),
"Cheese": ("0406", 12, True),
"Dairy - Desserts": ("2105", 5, False),
"Ice Cream": ("2105", 18, True),
"Beverages": ("2009", 5, False),
"Tea & Coffee": ("0902", 5, False),
"Food & Beverages": ("2106", 18, True),
"Health Drinks": ("2202", 18, True),
"Health Foods": ("2106", 18, True),
"Breakfast Cereal": ("1904", 18, False),
"Chocolates": ("1806", 18, False),
"Candy & Confectionery": ("1704", 18, False),
"Biscuits & Cookies": ("1905", 18, True),
"Crackers": ("1905", 18, True),
"Rusk": ("1905", 5, True),
"Cakes & Muffins": ("1905", 18, True),
"Bakery & Breads": ("1905", 5, False),
"Snacks": ("1905", 18, True),
"Noodles & Instant Food": ("1902", 18, False),
"Pasta & Noodles": ("1902", 18, False),
"Atta & Staples": ("1101", 5, False),
"Salt & Staples": ("2501", 5, False),
"Pulses, Grains & Spices": ("0713", 5, True),
"Spices & Masalas": ("0910", 5, False),
"Cooking Oils": ("1517", 5, False),
"Pickles & Chutneys": ("2001", 12, True),
"Dry Fruits & Nuts": ("0801", 12, True),
"Food - Spreads": ("2007", 12, True),
"Food - Mixes": ("2106", 18, True),
"Food - Soups & Sauces": ("2103", 12, False),
# ---- Personal care & household ----
"Hair Care": ("3305", 18, False),
"Skin Care": ("3304", 18, False),
"Skin & Bath Care": ("3307", 18, False),
"Bath Soap": ("3401", 18, False),
"Beauty Care": ("3304", 18, False),
"Oral Care": ("3306", 18, False),
"Fragrance & Deodorants": ("3303", 18, False),
"Men's Grooming": ("8212", 18, True),
"Baby Care": ("3304", 18, True),
"Feminine Hygiene": ("9619", 12, False),
"Personal Care": ("3307", 18, True),
"Detergents & Fabric Care": ("3402", 18, False),
"Dishwash": ("3402", 18, False),
"Household Cleaning": ("3402", 18, True),
"Household - Air Freshener": ("3307", 18, True),
"Personal Care - Mosquito Repellent": ("3808", 18, True),
# ---- Health care ----
"Health Care - Cold & Cough": ("3004", 12, True),
"Health Care - Antiseptic": ("3808", 18, True),
"Health Care - Ayurvedic": ("3003", 12, True),
"Health Care - Digestive": ("3004", 12, True),
"Health Care - First Aid": ("3005", 12, True),
}
# Keyword fallbacks so a product whose category label is slightly different
# from the table (or missing entirely) still resolves to a sensible chapter.
_KEYWORD_FALLBACKS: Tuple[Tuple[str, str, int, bool], ...] = (
("toothpaste", "3306", 18, False),
("shampoo", "3305", 18, False),
("soap", "3401", 18, False),
("detergent", "3402", 18, False),
("biscuit", "1905", 18, True),
("cookie", "1905", 18, True),
("chocolate", "1806", 18, False),
("candy", "1704", 18, False),
("toffee", "1704", 18, False),
("milk", "0401", 5, False),
("paneer", "0401", 5, False),
("curd", "0401", 5, False),
("cheese", "0406", 12, True),
("ice cream", "2105", 18, True),
("juice", "2009", 5, False),
("tea", "0902", 5, False),
("coffee", "0902", 5, False),
("health drink", "2202", 18, True),
("chips", "1905", 18, True),
("namkeen", "1905", 18, True),
("bread", "1905", 5, False),
("atta", "1101", 5, False),
("flour", "1101", 5, False),
("masala", "0910", 5, False),
("spice", "0910", 5, False),
("pasta", "1902", 18, False),
("noodle", "1902", 18, False),
("cooking oil", "1517", 5, False),
("oil", "1517", 5, True),
("pickle", "2001", 12, True),
("jam", "2007", 12, True),
("honey", "0409", 5, False),
("raisin", "0806", 5, True),
("dry fruit", "0801", 12, True),
("cereal", "1904", 18, False),
("deodorant", "3303", 18, False),
("perfume", "3303", 18, False),
("lipstick", "3304", 18, False),
("makeup", "3304", 18, False),
("face wash", "3307", 18, False),
("lotion", "3307", 18, False),
("sunscreen", "3304", 18, False),
("razor", "8212", 18, True),
("diaper", "9619", 12, False),
("mosquito", "3808", 18, True),
("repellent", "3808", 18, True),
("air freshener", "3307", 18, True),
("dishwash", "3402", 18, False),
("floor cleaner", "3402", 18, True),
("hand wash", "3401", 18, False),
)
@dataclass(frozen=True)
class HsnGstInfo:
"""HSN / GST resolution for one product category, mirroring the fields
stored on enriched catalog rows (coca-cola / milky_mist exports)."""
hsn_code: Optional[str] = None
gst_percent: Optional[int] = None
hsn_gst_needs_review: bool = True
def as_fields(self) -> Dict[str, object]:
return {
"hsn_code": self.hsn_code,
"gst_percent": self.gst_percent,
"hsn_gst_needs_review": self.hsn_gst_needs_review,
}
_UNKNOWN = HsnGstInfo(hsn_code=None, gst_percent=None, hsn_gst_needs_review=True)
def resolve_hsn_gst(category: Optional[str], product_title: str = "") -> HsnGstInfo:
"""Resolve (HSN code, GST %, needs_review) for a product category.
Exact category matches in `HSN_GST_TABLE` win; otherwise the product
title/description is scanned against `_KEYWORD_FALLBACKS`; otherwise
returns the safe unknown shape (all-None, needs_review=True).
"""
cat = (category or "").strip()
if cat:
exact = HSN_GST_TABLE.get(cat)
if exact:
return HsnGstInfo(*exact)
haystack = f"{product_title or ''} {cat}".lower()
for kw, hsn, gst, review in _KEYWORD_FALLBACKS:
if kw in haystack:
return HsnGstInfo(hsn_code=hsn, gst_percent=gst, hsn_gst_needs_review=review)
return _UNKNOWN
# Matches a price range string of the form "₹12-14" (also tolerates
# "Rs 12-14", "12 - 14", a single "₹12", or corrupted rupee symbols).
_RANGE_RE = re.compile(r"₹?\s*(\d+(?:\.\d+)?)\s*[-–—]\s*(\d+(?:\.\d+)?)")
_SINGLE_RE = re.compile(r"₹?\s*(\d+(?:\.\d+)?)")
def _extract_selling_price(price_range: object) -> Optional[float]:
"""Return the upper bound of a row's price range (the convention used by
the existing coca-cola / milky_mist exports for `selling_price`), or the
single value when the row only carries one price."""
if price_range is None:
return None
text = str(price_range).replace(",", "").strip()
if not text:
return None
m = _RANGE_RE.search(text)
if m:
try:
return float(m.group(2))
except ValueError:
return None
m = _SINGLE_RE.search(text)
if m:
try:
return float(m.group(1))
except ValueError:
return None
return None
def _round2(value: Optional[float]) -> Optional[float]:
if value is None:
return None
return round(value, 2)
def enrich_pricing_fields(product: dict) -> Dict[str, object]:
"""Compute the pricing fields added by this stage for one catalog row.
Convention (matches the coca-cola / milky_mist exports):
selling_price = upper bound of the row's `price_range`
tax_amount = selling_price * gst_percent / 100
final_selling_price = selling_price + tax_amount
`cost_price`, `profit_before_tax` and `profit_after_tax` are not
derivable from the data this pipeline generates, so they stay None
(exactly as the existing enriched exports store them).
"""
selling_price = _extract_selling_price(product.get("price_range"))
gst = product.get("gst_percent")
tax_amount = None
final_selling_price = None
if selling_price is not None and isinstance(gst, (int, float)) and gst and gst > 0:
tax_amount = _round2(selling_price * float(gst) / 100.0)
final_selling_price = _round2(selling_price + tax_amount)
return {
"selling_price": _round2(selling_price),
"cost_price": None,
"tax_amount": tax_amount,
"final_selling_price": final_selling_price,
"profit_before_tax": None,
"profit_after_tax": None,
}

View File

@@ -0,0 +1,48 @@
"""Adapts the HSN/GST resolver to the EnrichmentStage contract so it can be
registered in app/services/enrichment/pipeline.py's default pipeline.
Unlike the barcode stage this needs no network access at all - HSN/GST are
deterministic, category-level facts - so it is synchronous and effectively
free. It is pure ADDITIVE: rows that already carry the hsn_code/gst_percent
keys (e.g. re-running the pipeline over a previously-enriched catalog) are
left untouched.
"""
from __future__ import annotations
import logging
from typing import Any, Dict
from app.infrastructure.settings import ENABLE_HSN_GST_ENRICHMENT
from app.services.enrichment.base import EnrichmentStage, StageOutcome
from app.services.enrichment.hsn_gst.models import enrich_pricing_fields, resolve_hsn_gst
logger = logging.getLogger(__name__)
class HsnGstEnrichmentStage(EnrichmentStage):
name = "hsn_gst"
@property
def enabled(self) -> bool:
return ENABLE_HSN_GST_ENRICHMENT
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
# Already enriched (previous run / manual backfill) - never overwrite.
if product.get("hsn_code") is not None or product.get("gst_percent") is not None:
return StageOutcome(stage_name=self.name, fields={})
title = product.get("title") or product.get("product_name") or ""
category = product.get("category") or ""
hsn_info = resolve_hsn_gst(category, title)
fields: Dict[str, Any] = hsn_info.as_fields()
# Pricing fields depend on gst_percent, which was just resolved
# above, so hand the resolved rate to the pricing helper.
fields.update(enrich_pricing_fields({**product, "gst_percent": fields["gst_percent"]}))
needs_review = fields["hsn_gst_needs_review"]
return StageOutcome(
stage_name=self.name,
fields=fields,
error=None if not needs_review else "category-level HSN/GST estimate - verify against the physical pack"
)

View File

@@ -0,0 +1,89 @@
"""
Orchestrates a list of independent `EnrichmentStage`s over a batch of
catalog rows, running each stage's per-row work concurrently (bounded by
`max_concurrency`) and NEVER letting one row's failure affect any other
row or stage.
`catalog_engine.py` calls `run_default_pipeline()` once, between the
deterministic-validation step (product_validator.py) and catalog assembly
- see that file's "Step 2.6" for the call site and
docs/BARCODE_ENRICHMENT.md for the full pipeline-position rationale.
"""
from __future__ import annotations
import asyncio
import logging
from typing import Any, Dict, List, Optional, Sequence
from app.services.enrichment.base import EnrichmentStage
logger = logging.getLogger(__name__)
class EnrichmentPipeline:
def __init__(self, stages: Sequence[EnrichmentStage], max_concurrency: int = 5):
self.stages = [s for s in stages if s.enabled]
self.max_concurrency = max(1, max_concurrency)
async def run(self, products: List[Dict[str, Any]], brand: str) -> List[Dict[str, Any]]:
if not self.stages or not products:
return products
semaphore = asyncio.Semaphore(self.max_concurrency)
async def _run_row(product: Dict[str, Any]) -> Dict[str, Any]:
async with semaphore:
for stage in self.stages:
product = await stage.apply(product, brand)
return product
results = await asyncio.gather(*(_run_row(p) for p in products), return_exceptions=True)
final: List[Dict[str, Any]] = []
for original, result in zip(products, results):
if isinstance(result, Exception):
logger.error(f"Enrichment pipeline failed for '{original.get('product_name')}': {result}")
final.append(original)
else:
final.append(result)
return final
def _build_default_stages() -> List[EnrichmentStage]:
"""Import stages lazily so importing this module never pulls in a
stage's own dependencies (network clients, DB drivers, ...) unless a
default pipeline is actually requested."""
stages: List[EnrichmentStage] = []
try:
from app.services.enrichment.barcode.stage import BarcodeEnrichmentStage
stages.append(BarcodeEnrichmentStage())
except Exception as e:
logger.error(f"Barcode enrichment stage unavailable: {e}")
# HSN / GST & pricing enrichment (see app/services/enrichment/hsn_gst/) -
# deterministic, offline, pure-additive. Runs AFTER the barcode stage so
# every stored/exported row carries both sets of fields; a failure here
# degrades a row to "no HSN/GST attached", never drops the row.
try:
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
stages.append(HsnGstEnrichmentStage())
except Exception as e:
logger.error(f"HSN/GST enrichment stage unavailable: {e}")
# Future stages register here, e.g.:
# from app.services.enrichment.nutrition.stage import NutritionEnrichmentStage
# stages.append(NutritionEnrichmentStage())
return stages
_default_pipeline: Optional[EnrichmentPipeline] = None
async def run_default_pipeline(products: List[Dict[str, Any]], brand: str,
max_concurrency: int = 5) -> List[Dict[str, Any]]:
"""Convenience entry point used by catalog_engine.py: runs every
currently-registered, enabled enrichment stage over `products`."""
global _default_pipeline
if _default_pipeline is None:
_default_pipeline = EnrichmentPipeline(_build_default_stages(), max_concurrency=max_concurrency)
return await _default_pipeline.run(products, brand)