backend stores data file enrichment pipeline

This commit is contained in:
sriram
2026-08-18 16:58:36 +05:30
parent 691efef880
commit 7a4583372f
36 changed files with 4599 additions and 0 deletions

View File

@@ -0,0 +1,183 @@
"""Admin endpoints for turning a store spreadsheet into brand-catalog rows.
Three endpoints, matching how the other long-running admin jobs in this
project are exposed (see `nutrition_admin.py`):
POST /api/admin/store-catalog/preview - parse only, show the column map
POST /api/admin/store-catalog/ingest - 202 + job_id, runs in background
GET /api/admin/store-catalog/jobs/{id} - poll stage + row progress
`preview` exists because the mapping from a store's headers onto catalog
fields is a guess. Running an 11-stage scrape over 2000 rows only to discover
that "Item" was read as the description is expensive; showing the operator the
mapping first costs one parse.
"""
from __future__ import annotations
import logging
from typing import Optional
from fastapi import APIRouter, Depends, File, HTTPException, UploadFile, status
from pydantic import BaseModel
from app.api.background import run_in_background
from app.api.deps import require_admin
from app.api.store_catalog_job_store import store_catalog_job_store
from app.core import store_catalog_pipeline as pipeline
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/admin/store-catalog", tags=["admin", "catalog"])
# Same ceilings as the user-products upload path, for the same reasons.
MAX_UPLOAD_BYTES = 10 * 1024 * 1024
MAX_UPLOAD_ROWS = 2000
PREVIEW_ROWS = 10
class StoreCatalogJobOut(BaseModel):
job_id: str
filename: str
status: str
detail: Optional[str] = None
stage_index: int = 0
stage_name: str = ""
total_stages: int = pipeline.TOTAL_STAGES
rows_done: int = 0
rows_total: int = 0
result: Optional[dict] = None
async def _read_upload(file: UploadFile) -> bytes:
contents = await file.read()
if not contents:
raise HTTPException(status_code=400, detail="The uploaded file is empty.")
if len(contents) > MAX_UPLOAD_BYTES:
raise HTTPException(
status_code=413,
detail=f"File is larger than the {MAX_UPLOAD_BYTES // (1024 * 1024)}MB limit.",
)
return contents
@router.post("/preview", dependencies=[Depends(require_admin)])
async def preview_store_catalog(file: UploadFile = File(...)) -> dict:
"""Parse the sheet and report how its columns were understood."""
contents = await _read_upload(file)
try:
df, mapping = pipeline.parse_spreadsheet(file.filename or "upload.xlsx", contents)
except HTTPException:
raise
except ValueError as exc:
raise HTTPException(status_code=400, detail=str(exc)) from exc
except Exception as exc: # noqa: BLE001 - unreadable file is a user error
raise HTTPException(status_code=400, detail=f"Could not parse the file: {exc}") from exc
if df.empty:
raise HTTPException(status_code=400, detail="The file has no data rows.")
if len(df) > MAX_UPLOAD_ROWS:
raise HTTPException(
status_code=413,
detail=f"{len(df)} rows exceeds the {MAX_UPLOAD_ROWS}-row limit for one upload.",
)
sample = df.head(PREVIEW_ROWS).fillna("").astype(str).to_dict(orient="records")
return {
"filename": file.filename,
"rows_total": int(len(df)),
"recognised_columns": {f: str(c) for f, c in mapping.columns.items()},
"ignored_columns": mapping.ignored,
"unrecognised_columns": mapping.unrecognised,
"brand_column_present": "brand" in mapping.columns,
"preview": sample,
"stages": list(pipeline.STAGE_NAMES),
}
def _run_ingest_job(job_id: str, filename: str, contents: bytes,
use_llm: bool, fetch_images: bool) -> None:
store_catalog_job_store.update(job_id, status="running", detail="Starting pipeline")
def progress(stage_index: int, stage_name: str, done: int, total: int) -> None:
store_catalog_job_store.update(
job_id, stage_index=stage_index, stage_name=stage_name,
rows_done=done, rows_total=total,
)
try:
result = pipeline.run_pipeline(
filename, contents, progress=progress,
use_llm=use_llm, fetch_images=fetch_images,
)
body = result.as_dict()
if body.get("storage_error"):
# Rows were built but none reached the database. Reporting this as
# success would leave the operator believing the catalog changed.
store_catalog_job_store.update(
job_id, status="failed", result=body,
detail=f"Built {body['products_built']} row(s) but storing them failed: "
f"{body['storage_error']}",
)
return
store_catalog_job_store.update(
job_id, status="done", result=body,
detail=(f"{body['inserted']} inserted, {body['backfilled']} backfilled, "
f"{body['skipped_existing']} unchanged, {body['rejected']} rejected"),
)
except Exception as exc: # noqa: BLE001 - a daemon thread must not die silently
logger.exception("Store-catalog ingestion job %s failed", job_id)
store_catalog_job_store.update(job_id, status="failed", detail=str(exc))
@router.post("/ingest", status_code=status.HTTP_202_ACCEPTED,
dependencies=[Depends(require_admin)])
async def ingest_store_catalog(
file: UploadFile = File(...),
use_llm: bool = True,
fetch_images: bool = True,
) -> StoreCatalogJobOut:
"""Kick off the 11-stage pipeline and return a job id to poll.
The file is read here rather than in the worker: `UploadFile` is backed by
a temporary file tied to the request, so it is gone by the time a
background thread would get to it.
"""
contents = await _read_upload(file)
filename = file.filename or "upload.xlsx"
# Fail fast on an unparseable file so the caller gets a 400 now rather than
# a job that transitions straight to "failed".
try:
df, _mapping = pipeline.parse_spreadsheet(filename, contents)
except Exception as exc: # noqa: BLE001
raise HTTPException(status_code=400, detail=f"Could not parse the file: {exc}") from exc
if df.empty:
raise HTTPException(status_code=400, detail="The file has no data rows.")
if len(df) > MAX_UPLOAD_ROWS:
raise HTTPException(
status_code=413,
detail=f"{len(df)} rows exceeds the {MAX_UPLOAD_ROWS}-row limit for one upload.",
)
job = store_catalog_job_store.create(filename)
store_catalog_job_store.update(job.job_id, rows_total=int(len(df)))
run_in_background(
lambda: _run_ingest_job(job.job_id, filename, contents, use_llm, fetch_images),
name=f"store-catalog-{job.job_id[:8]}",
)
return StoreCatalogJobOut(
job_id=job.job_id, filename=filename, status=job.status,
rows_total=int(len(df)),
)
@router.get("/jobs/{job_id}", dependencies=[Depends(require_admin)])
def get_store_catalog_job(job_id: str) -> StoreCatalogJobOut:
job = store_catalog_job_store.get(job_id)
if not job:
raise HTTPException(status_code=404, detail="Job not found")
return StoreCatalogJobOut(
job_id=job.job_id, filename=job.filename, status=job.status, detail=job.detail,
stage_index=job.stage_index, stage_name=job.stage_name,
total_stages=job.total_stages, rows_done=job.rows_done,
rows_total=job.rows_total, result=job.result,
)

View File

@@ -0,0 +1,93 @@
"""Job tracking for store-spreadsheet catalog ingestion.
Same pattern and the same documented trade-offs as `job_store.py`,
`store_job_store.py` and `nutrition_job_store.py`: a process-local dict behind
a lock, lost on restart, not shared across uvicorn workers. Adding a broker for
this would be operational weight the project has already decided against (see
`job_store.py`).
It is its own module rather than a reuse of `nutrition_job_store` because this
job reports a different shape of progress: an 11-stage pipeline needs to say
*which stage* it is on as well as how many rows it has finished, so the UI can
show "Stage 6/11 - Image Search, 48/120 rows".
"""
from __future__ import annotations
import threading
import time
import uuid
from dataclasses import dataclass, field
from typing import Dict, Optional
@dataclass
class StoreCatalogJob:
job_id: str
filename: str
status: str = "pending" # pending -> running -> done | failed
detail: Optional[str] = None
result: Optional[dict] = None
stage_index: int = 0 # 1-based; 0 while still pending
stage_name: str = ""
total_stages: int = 11
rows_done: int = 0
rows_total: int = 0
created_at: float = field(default_factory=time.time)
updated_at: float = field(default_factory=time.time)
class StoreCatalogJobStore:
def __init__(self) -> None:
self._jobs: Dict[str, StoreCatalogJob] = {}
self._lock = threading.Lock()
def create(self, filename: str) -> StoreCatalogJob:
job = StoreCatalogJob(job_id=str(uuid.uuid4()), filename=filename)
with self._lock:
self._jobs[job.job_id] = job
return job
def update(
self,
job_id: str,
status: Optional[str] = None,
detail: Optional[str] = None,
result: Optional[dict] = None,
stage_index: Optional[int] = None,
stage_name: Optional[str] = None,
rows_done: Optional[int] = None,
rows_total: Optional[int] = None,
) -> None:
"""Every field is optional and None means "leave alone".
That matters: `job_store.update` sets `detail` unconditionally, so
marking a job running there wipes whatever detail it already had. A
progress callback firing many times per second must not erase state it
was not asked to change.
"""
with self._lock:
job = self._jobs.get(job_id)
if not job:
return
if status is not None:
job.status = status
if detail is not None:
job.detail = detail
if result is not None:
job.result = result
if stage_index is not None:
job.stage_index = stage_index
if stage_name is not None:
job.stage_name = stage_name
if rows_done is not None:
job.rows_done = rows_done
if rows_total is not None:
job.rows_total = rows_total
job.updated_at = time.time()
def get(self, job_id: str) -> Optional[StoreCatalogJob]:
with self._lock:
return self._jobs.get(job_id)
store_catalog_job_store = StoreCatalogJobStore()

View File

@@ -0,0 +1,687 @@
"""
Store-spreadsheet -> 11-stage enrichment -> brand table.
A store sends a product list as Excel/CSV. The columns roughly resemble the
brand-catalog schema but are named inconsistently and are always incomplete -
typically product_name, description, category and little else. This module
turns such a file into proper catalog rows: it scrapes/derives everything the
sheet did not supply, validates the result, and writes it to the brand table
the product belongs to (Britannia 50-50 -> brand_britannia).
Relationship to the existing brand pipeline
-------------------------------------------
`app/core/catalog_engine.py` starts from a BRAND NAME and asks an LLM to
enumerate its products. This module starts from a SPREADSHEET ROW, so the
discovery stage is replaced by row intake and everything downstream is
gap-filling. catalog_engine is not modified or called for orchestration; only
its `_select_best_images` contamination filter is reused.
Stage order (the numbering the operator sees in the UI):
1. Brand Resolution & FSSAI Licence Mapping
2. Row Intake & Normalisation (replaces AI product discovery)
3. Title & Category Consistency Guard
4. Pack-Size Variant Explosion & Unit Safety
5. Pricing Band Estimation
6. Image Search & Contamination Filtering
7. Marketplace & Internal SKU Resolution
8. Barcode Retrieval & Enrichment
9. HSN / GST Tax Enrichment
10. Deterministic Product Validation Gate
11. Vector Embedding & Storage
Two rules hold throughout:
* **Fill only blanks.** The store's own data is authoritative; a stage that
finds a value already present leaves it alone. This is what separates the
pipeline from `user_products._build_product_dict`, which fills the same
gaps with static defaults ("Rs.100-250", a canonical S3 URL). Here the
defaults are replaced by real lookups.
* **A stage never raises.** Enrichment is best-effort; a barcode lookup
timing out must not lose the row. Failures are recorded per row and the
row continues with that field empty.
"""
from __future__ import annotations
import asyncio
import logging
import re
from dataclasses import dataclass, field
from typing import Any, Callable, Dict, List, Optional, Tuple
from app.infrastructure.settings import (
ENABLE_BARCODE_LOOKUP,
ENABLE_HSN_GST_ENRICHMENT,
ENABLE_PRODUCT_VALIDATION,
ENABLE_SKU_WEB_LOOKUP,
MAX_VARIANTS_PER_PRODUCT,
USE_EMBEDDINGS,
)
# The spreadsheet machinery already exists and is well tested; importing it
# from the router module is deliberate. Extracting it into a service would be
# tidier, but user_products.py is a working file this feature does not touch,
# and tests/test_user_products_upload.py monkeypatches these names on that
# module object.
from app.api.routers.user_products import (
AddProductRequest,
_slugify,
_text,
map_spreadsheet_columns,
read_products_dataframe,
row_to_request,
)
from app.services import price_estimator
from app.services.brand_registry import (
BRAND_ALIASES,
get_fssai_license,
resolve_parent_brand,
)
from app.services.category_registry import (
detect_category_from_text,
sanitize_category_language,
)
from app.services.category_units import fix_or_reject_size
from app.services.embeddings_service import embed_texts
from app.services.enrichment.barcode.stage import BarcodeEnrichmentStage
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
from app.services.enrichment.pipeline import EnrichmentPipeline
from app.services.product_validator import validate_catalog
from app.services.sku_service import resolve_product_sku
from app.services.title_validator import validate_and_fix_title
from app.services.vector_store import (
_sanitize_name,
display_name_for_suffix,
get_products_by_brand,
upsert_brand_products,
)
logger = logging.getLogger(__name__)
STAGE_NAMES: Tuple[str, ...] = (
"Brand Resolution & FSSAI Mapping",
"Row Intake & Normalisation",
"Title & Category Consistency Guard",
"Pack-Size Variant Explosion & Unit Safety",
"Pricing Band Estimation",
"Image Search & Contamination Filtering",
"Marketplace & Internal SKU Resolution",
"Barcode Retrieval & Enrichment",
"HSN / GST Tax Enrichment",
"Deterministic Product Validation Gate",
"Vector Embedding & Storage",
)
TOTAL_STAGES = len(STAGE_NAMES)
# Columns a store row may legitimately leave blank and the pipeline fills in.
# Anything not listed here is copied through untouched.
_ENRICHABLE = (
"category", "description", "fssai_license", "price_range", "product_sku",
"sku_source", "hsn_code", "barcode", "barcode_type", "final_selling_price",
"selling_price", "image_url", "image_urls", "size_variants",
)
_SIZE_IN_TITLE = re.compile(
r"\b(\d+(?:[.,]\d+)?)\s*(kg|kgs|g|gm|gms|gram|grams|ml|l|ltr|litre|liters?|pack|pcs|n)\b",
re.I,
)
def _blank(value: Any) -> bool:
"""True if a field carries no usable value."""
if value is None:
return True
if isinstance(value, str):
return not value.strip()
if isinstance(value, (list, tuple, dict, set)):
return len(value) == 0
return False
@dataclass
class RowError:
row: int
product_name: str
error: str
@dataclass
class PipelineResult:
"""Everything the UI needs to explain what happened to an upload."""
rows_total: int = 0
products_built: int = 0
inserted: int = 0
backfilled: int = 0
skipped_existing: int = 0
rejected: int = 0
brands: List[str] = field(default_factory=list)
rejections: List[Dict[str, Any]] = field(default_factory=list)
errors: List[RowError] = field(default_factory=list)
warnings: List[str] = field(default_factory=list)
recognised_columns: Dict[str, str] = field(default_factory=dict)
unrecognised_columns: List[str] = field(default_factory=list)
storage_error: Optional[str] = None
def as_dict(self) -> Dict[str, Any]:
return {
"rows_total": self.rows_total,
"products_built": self.products_built,
"inserted": self.inserted,
"backfilled": self.backfilled,
"skipped_existing": self.skipped_existing,
"rejected": self.rejected,
"brands": self.brands,
"rejections": self.rejections[:50],
"errors": [vars(e) for e in self.errors[:50]],
"error_count": len(self.errors),
"warnings": self.warnings,
"recognised_columns": self.recognised_columns,
"unrecognised_columns": self.unrecognised_columns,
"storage_error": self.storage_error,
}
ProgressFn = Callable[[int, str, int, int], None]
"""Called as (stage_index_1_based, stage_name, rows_done, rows_total)."""
def _noop_progress(stage: int, name: str, done: int, total: int) -> None:
return None
# ---------------------------------------------------------------------------
# Stage 1 - Brand resolution & FSSAI
# ---------------------------------------------------------------------------
_INFERRED_BRAND_COL = "__inferred_brand__"
def infer_brand(product_name: str) -> Optional[str]:
"""Work out which brand a product name belongs to.
Store files very often have no brand column at all - the brand is simply
the first word or two of the product name. Matches the longest known alias
appearing anywhere in the name, so "britannia 50-50" resolves; longest-first
matters, because "cadbury dairy milk" must win over "cadbury".
Runs before `row_to_request`, which rejects a brand-less row outright.
"""
haystack = f" {(product_name or '').lower().strip()} "
best: Optional[str] = None
for alias in BRAND_ALIASES:
if re.search(rf"(?<![a-z0-9]){re.escape(alias)}(?![a-z0-9])", haystack):
if best is None or len(alias) > len(best):
best = alias
if best:
return best
# Last resort: assume the leading word names the brand. Better than
# dropping the row, and the validation gate will catch nonsense later.
first = (product_name or "").strip().split()
return first[0] if first else None
def stage_1_brand_and_fssai(row: Dict[str, Any]) -> Dict[str, Any]:
brand = row.get("brand") or ""
parent = resolve_parent_brand(brand)
row["brand"] = parent
row["brand_name"] = parent
row["_table"] = f"brand_{_sanitize_name(parent)}"
if _blank(row.get("fssai_license")):
# None means "not a food brand" as much as "unknown", so an empty
# column is a legitimate outcome, never an error.
row["fssai_license"] = get_fssai_license(parent)
return row
# ---------------------------------------------------------------------------
# Stage 2 - Row intake (replaces AI discovery)
# ---------------------------------------------------------------------------
def stage_2_row_intake(row: Dict[str, Any], *, use_llm: bool = True) -> Dict[str, Any]:
"""The spreadsheet row *is* the product. Only fill a missing description.
The LLM call is best-effort and optional: with Ollama unreachable the row
keeps an empty description and later stages still work.
"""
if not _blank(row.get("description")) or not use_llm:
return row
try:
from app.services.ollama_service import fetch_product_details
details = fetch_product_details(row.get("brand", ""), row.get("product_name", ""))
if details and details.get("description"):
row["description"] = str(details["description"]).strip()
if _blank(row.get("size_variants")) and details and details.get("size_variants"):
row["size_variants"] = list(details["size_variants"])
except Exception as exc: # noqa: BLE001 - enrichment is best-effort
logger.debug("LLM enrichment skipped for %r: %s", row.get("product_name"), exc)
return row
# ---------------------------------------------------------------------------
# Stage 3 - Title & category consistency
# ---------------------------------------------------------------------------
def stage_3_title_category(row: Dict[str, Any]) -> Dict[str, Any]:
title = row.get("title") or row.get("product_name") or ""
category = row.get("category")
if _blank(category):
category = detect_category_from_text(f"{title} {row.get('description') or ''}")
row["_category_deterministic"] = bool(category)
if not category:
# "General" rather than None: the column is not nullable in
# practice and _to_storage_row already assumed this default. The
# row is still reported as un-categorised in the job warnings, so
# the signal is preserved rather than swallowed.
category = "General"
row.setdefault("_notes", []).append("category could not be detected; defaulted to General")
row["category"] = category
else:
row["_category_deterministic"] = True
if category:
fixed, changed, removed = validate_and_fix_title(
title, category, brand=row.get("brand")
)
row["title"] = fixed
if changed:
row.setdefault("_notes", []).append(
f"title corrected against category '{category}' (removed {removed})"
)
if not _blank(row.get("description")):
row["description"] = sanitize_category_language(row["description"], category)
else:
row["title"] = title
return row
# ---------------------------------------------------------------------------
# Stage 4 - Pack-size explosion & unit safety
# ---------------------------------------------------------------------------
def _sizes_for(row: Dict[str, Any]) -> List[str]:
sizes = [str(s).strip() for s in (row.get("size_variants") or []) if str(s).strip()]
if sizes:
return sizes
match = _SIZE_IN_TITLE.search(row.get("product_name") or "")
if match:
return [match.group(0).strip()]
return list(price_estimator.default_size_variants(
row.get("category") or "", row.get("product_name") or ""
))
def stage_4_explode_sizes(row: Dict[str, Any]) -> Tuple[List[Dict[str, Any]], List[str]]:
"""One input row becomes one row per pack size.
Barcodes and SKUs identify a specific pack, so they are only meaningful
once sizes are separate rows. Sizes whose unit makes no sense for the
category (a biscuit measured in cm) are dropped with a reason.
"""
out: List[Dict[str, Any]] = []
dropped: List[str] = []
category = row.get("category") or ""
for size in _sizes_for(row)[:MAX_VARIANTS_PER_PRODUCT]:
fixed, rejected, reason = fix_or_reject_size(size, category, row.get("product_name") or "")
if rejected or not fixed:
dropped.append(f"{size}: {reason or 'invalid pack size'}")
continue
variant = dict(row)
variant["size"] = fixed
variant["size_variants"] = [fixed]
out.append(variant)
if not out and not dropped:
variant = dict(row)
variant["size"] = "Standard"
variant["size_variants"] = ["Standard"]
out.append(variant)
return out, dropped
# ---------------------------------------------------------------------------
# Stage 5 - Pricing bands
# ---------------------------------------------------------------------------
def stage_5_pricing(row: Dict[str, Any]) -> Dict[str, Any]:
size = row.get("size") or "Standard"
if _blank(row.get("price_range")):
if not _blank(row.get("final_selling_price")):
price = float(row["final_selling_price"])
lo, hi = int(round(price * 0.95)), int(round(price * 1.05))
else:
lo, hi = price_estimator.estimate_price_range_for_size(
size, row.get("product_name") or "", row.get("brand") or "",
row.get("category") or "",
)
row["price_range"] = f"₹{lo}-{hi}"
return row
# ---------------------------------------------------------------------------
# Stage 6 - Images
# ---------------------------------------------------------------------------
def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, Any]:
if not _blank(row.get("image_urls")) or not enabled:
return row
try:
from app.core.catalog_engine import catalog_engine
from app.services.image_search import find_all_image_urls
candidates = find_all_image_urls(
row.get("product_name") or "", brand=row.get("brand"), max_results=24
)
if candidates:
best = catalog_engine._select_best_images(
candidates, row.get("product_name") or "", row.get("brand") or "", max_images=10
)
if best:
row["image_urls"] = list(best)
row["image_url"] = best[0]
except Exception as exc: # noqa: BLE001 - an image is not worth the row
logger.debug("Image search skipped for %r: %s", row.get("product_name"), exc)
return row
# ---------------------------------------------------------------------------
# Stage 7 - SKU
# ---------------------------------------------------------------------------
def stage_7_sku(row: Dict[str, Any]) -> Dict[str, Any]:
if not _blank(row.get("product_sku")):
return row
try:
resolved = resolve_product_sku(
row.get("brand") or "", row.get("product_name") or "", row.get("size") or ""
)
row["product_sku"] = resolved.get("product_sku")
row["sku_source"] = resolved.get("sku_source")
except Exception as exc: # noqa: BLE001
logger.debug("SKU resolution failed for %r: %s", row.get("product_name"), exc)
return row
# ---------------------------------------------------------------------------
# Stages 8 & 9 - barcode and HSN/GST, via the ported enrichment framework
# ---------------------------------------------------------------------------
async def stages_8_9_enrichment(rows: List[Dict[str, Any]], brand: str) -> List[Dict[str, Any]]:
"""Both stages are `EnrichmentStage`s, so the ported EnrichmentPipeline
runs them with its own concurrency limit and never-raises guarantee.
Each disables itself via its settings flag, so this is a no-op when both
are off.
"""
stages = []
if ENABLE_BARCODE_LOOKUP:
stages.append(BarcodeEnrichmentStage())
if ENABLE_HSN_GST_ENRICHMENT:
stages.append(HsnGstEnrichmentStage())
if not stages:
return rows
return await EnrichmentPipeline(stages).run(rows, brand)
# ---------------------------------------------------------------------------
# Stage 10 - validation gate
# ---------------------------------------------------------------------------
def stage_10_validate(rows: List[Dict[str, Any]], brand: str, *, images_checked: bool = True):
if not ENABLE_PRODUCT_VALIDATION:
return rows, [], {}
known = [bool(r.get("_category_deterministic")) for r in rows]
return validate_catalog(rows, brand, known_category_flags=known,
images_checked=images_checked)
# ---------------------------------------------------------------------------
# Stage 11 - embed & store
# ---------------------------------------------------------------------------
def build_image_id(brand: str, product_name: str, size: str) -> str:
"""Deterministic, and unique per pack size.
Matches `user_products.py`'s brand_slug + product_slug form so the two
upload paths agree, with the size folded in - without it every size of a
product collapses onto one row under ON CONFLICT (image_id). Deliberately
NOT s3_service.generate_image_id, which appends a random uuid4 and so can
never match an existing row.
"""
name = (product_name or "").strip()
# Store sheets usually carry the pack size inside the product name already
# ("Cadbury Dairy Milk 100g"). Appending it again would produce
# ..._100g_100g and, worse, a different id than the same product uploaded
# through the user-products path.
suffix = "" if _slugify(size) and _slugify(size) in _slugify(name) else f" {size}"
slug = _slugify(f"{name}{suffix}".strip())
return f"{_sanitize_name(brand)}_{slug}"
def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
"""Project an enriched row onto the brand-table columns."""
name = row.get("product_name") or row.get("title") or ""
size = row.get("size") or "Standard"
display = name if size.lower() in name.lower() else f"{name} {size}".strip()
category = row.get("category") or "General"
description = row.get("description") or (
f"{display} from {row.get('brand')}."
)
return {
"product_name": display,
"title": row.get("title") or name,
"description": description,
"category": category,
"image_id": build_image_id(row.get("brand") or "", name, size),
"image_url": row.get("image_url"),
"image_urls": list(row.get("image_urls") or []),
"price_range": row.get("price_range"),
"size_variants": [size],
"providers": list(row.get("providers") or []),
"fssai_license": row.get("fssai_license"),
"product_sku": row.get("product_sku"),
"sku_source": row.get("sku_source"),
"hsn_code": row.get("hsn_code"),
"final_selling_price": row.get("final_selling_price"),
"selling_price": row.get("selling_price"),
"barcode": row.get("barcode"),
"barcode_type": row.get("barcode_type"),
"highlights": list(row.get("highlights") or []),
"nutrients": list(row.get("nutrients") or []),
"search_query": f"{row.get('brand')} {display} {category} {description}",
}
def _merge_with_existing(new: Dict[str, Any], existing: Dict[str, Any]) -> Tuple[Dict[str, Any], bool]:
"""Keep the stored row, filling only the columns it left empty.
Returns (row_to_write, changed). `changed` is False when the stored row
was already complete, which is what makes re-uploading the same file a
no-op.
"""
merged = dict(existing)
changed = False
for key, value in new.items():
if key in ("image_id",):
continue
if _blank(existing.get(key)) and not _blank(value):
merged[key] = value
changed = True
merged["image_id"] = new["image_id"]
return merged, changed
def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResult) -> None:
"""Embed and upsert, splitting inserts from backfills.
`cleanup=False` is load-bearing: cleanup=True deletes every row in the
table that is not in this batch, which for a 3-row store file would wipe
the brand's entire catalog.
"""
if not rows:
return
try:
existing_by_id = {
r.get("image_id"): r
for r in (get_products_by_brand(brand) or [])
if r.get("image_id")
}
except Exception as exc: # noqa: BLE001 - treat as "nothing stored yet"
logger.warning("Could not read existing products for %s: %s", brand, exc)
existing_by_id = {}
to_write: List[Dict[str, Any]] = []
for row in rows:
prior = existing_by_id.get(row["image_id"])
if prior is None:
to_write.append(row)
result.inserted += 1
continue
merged, changed = _merge_with_existing(row, prior)
if changed:
to_write.append(merged)
result.backfilled += 1
else:
result.skipped_existing += 1
if not to_write:
return
if USE_EMBEDDINGS:
try:
vectors = embed_texts([r.get("search_query") or r["product_name"] for r in to_write])
for row, vector in zip(to_write, vectors):
row["embedding"] = vector
except Exception as exc: # noqa: BLE001 - a row without a vector is
# still a usable catalog row; it just won't match semantic search.
logger.warning("Embedding failed for %s: %s", brand, exc)
try:
upsert_brand_products(brand, to_write, cleanup=False)
except Exception as exc: # noqa: BLE001
result.storage_error = str(exc)
result.inserted = 0
result.backfilled = 0
logger.exception("Storing store-catalog rows failed for %s", brand)
# ---------------------------------------------------------------------------
# Orchestration
# ---------------------------------------------------------------------------
def parse_spreadsheet(filename: str, content: bytes):
"""Parse an upload into (DataFrame, column mapping). Raises ValueError."""
df = read_products_dataframe(filename, content)
mapping = map_spreadsheet_columns(df.columns)
return df, mapping
def run_pipeline(
filename: str,
content: bytes,
*,
progress: ProgressFn = _noop_progress,
use_llm: bool = True,
fetch_images: bool = True,
) -> PipelineResult:
"""Run all 11 stages over an uploaded store spreadsheet.
Synchronous by design - it is called on a background daemon thread by the
router, matching how every other long job in this project runs.
"""
result = PipelineResult()
df, mapping = parse_spreadsheet(filename, content)
result.recognised_columns = {f: str(c) for f, c in mapping.columns.items()}
result.unrecognised_columns = list(mapping.unrecognised)
records = df.to_dict(orient="records")
result.rows_total = len(records)
# A store sheet frequently has no brand column; the brand lives inside the
# product name. Rather than fork row_to_request (a working, tested function
# this feature does not touch), advertise a synthetic brand column and fill
# it per row from the inferred value.
brand_column_supplied = "brand" in mapping.columns
if not brand_column_supplied:
mapping.columns["brand"] = _INFERRED_BRAND_COL
# ---- stages 1-3, per uploaded row --------------------------------------
prepared: List[Dict[str, Any]] = []
for position, record in enumerate(records):
row_no = position + 2 # header occupies row 1
raw_name = _text(record, mapping, "product_name") or _text(record, mapping, "title") or ""
if not brand_column_supplied:
record[_INFERRED_BRAND_COL] = infer_brand(raw_name) or ""
try:
req = row_to_request(record, mapping)
except ValueError as exc:
result.errors.append(RowError(row_no, raw_name, str(exc)))
continue
row = req.model_dump()
row["_row"] = row_no
prepared.append(row)
progress(1, STAGE_NAMES[0], 0, len(prepared))
prepared = [stage_1_brand_and_fssai(r) for r in prepared]
for index, row in enumerate(prepared, start=1):
stage_2_row_intake(row, use_llm=use_llm)
progress(2, STAGE_NAMES[1], index, len(prepared))
progress(3, STAGE_NAMES[2], 0, len(prepared))
prepared = [stage_3_title_category(r) for r in prepared]
# ---- stage 4, one row becomes many -------------------------------------
exploded: List[Dict[str, Any]] = []
for row in prepared:
variants, dropped = stage_4_explode_sizes(row)
exploded.extend(variants)
for reason in dropped:
result.warnings.append(f"row {row.get('_row')}: dropped pack size {reason}")
for note in row.get("_notes") or []:
result.warnings.append(f"row {row.get('_row')}: {note}")
progress(4, STAGE_NAMES[3], len(exploded), len(exploded))
result.products_built = len(exploded)
# ---- stages 5-7, per exploded row --------------------------------------
for index, row in enumerate(exploded, start=1):
stage_5_pricing(row)
progress(5, STAGE_NAMES[4], index, len(exploded))
for index, row in enumerate(exploded, start=1):
stage_6_images(row, enabled=fetch_images)
progress(6, STAGE_NAMES[5], index, len(exploded))
for index, row in enumerate(exploded, start=1):
stage_7_sku(row)
progress(7, STAGE_NAMES[6], index, len(exploded))
# ---- stages 8-11, grouped by destination brand -------------------------
by_brand: Dict[str, List[Dict[str, Any]]] = {}
for row in exploded:
by_brand.setdefault(row["brand"], []).append(row)
result.brands = sorted(display_name_for_suffix(_sanitize_name(b)) for b in by_brand)
for brand, rows in by_brand.items():
progress(8, STAGE_NAMES[7], 0, len(rows))
try:
rows = asyncio.run(stages_8_9_enrichment(rows, brand))
except Exception as exc: # noqa: BLE001 - enrichment must not lose rows
logger.warning("Barcode/HSN enrichment failed for %s: %s", brand, exc)
progress(9, STAGE_NAMES[8], len(rows), len(rows))
kept, rejected, _summary = stage_10_validate(rows, brand, images_checked=fetch_images)
result.rejected += len(rejected)
for bad in rejected:
result.rejections.append({
"product_name": bad.get("product_name") or bad.get("title"),
"size": bad.get("size"),
"reason": "; ".join(
i.get("message", "") if isinstance(i, dict) else str(i)
for i in (bad.get("validation_issues") or [])
) or "failed the validation gate",
})
progress(10, STAGE_NAMES[9], len(kept), len(rows))
storage_rows = [_to_storage_row(r) for r in kept]
# A single sheet can name the same pack twice; last one wins, so the
# batch never presents two rows with the same image_id to the upsert.
deduped: Dict[str, Dict[str, Any]] = {r["image_id"]: r for r in storage_rows}
stage_11_store(list(deduped.values()), brand, result)
progress(11, STAGE_NAMES[10], len(deduped), len(deduped))
return result

View File

@@ -185,6 +185,49 @@ USE_PLAYWRIGHT_FALLBACK = _bool("USE_PLAYWRIGHT_FALLBACK", "true")
# out 1x1 tracking pixels / broken placeholder images)
MIN_IMAGE_BYTES = int(os.getenv("MIN_IMAGE_BYTES", "3000"))
# ---------------------------------------------------------------------------
# Store-catalog enrichment pipeline (app/core/store_catalog_pipeline.py)
# ---------------------------------------------------------------------------
# Stages 7-10 of the store-Excel ingestion pipeline. Ported from the sibling
# Universal_Catalog_Barcode_Enrichment project along with the code that reads
# them; the names are kept identical so the ported modules need no edits.
#
# The three network-touching flags default to FALSE here, unlike in the
# sibling. This backend serves an interactive API on a shared 8GB host, and a
# 2000-row upload with web lookups on would fire thousands of outbound
# requests. Turn them on deliberately, per environment.
ENABLE_SKU_WEB_LOOKUP = _bool("ENABLE_SKU_WEB_LOOKUP", "false")
ENABLE_BARCODE_LOOKUP = _bool("ENABLE_BARCODE_LOOKUP", "false")
ENABLE_MANUFACTURER_SITE_LOOKUP = _bool("ENABLE_MANUFACTURER_SITE_LOOKUP", "false")
# Offline/deterministic stages - safe to leave on.
ENABLE_HSN_GST_ENRICHMENT = _bool("ENABLE_HSN_GST_ENRICHMENT", "true")
ENABLE_PRODUCT_VALIDATION = _bool("ENABLE_PRODUCT_VALIDATION", "true")
# Validation gate thresholds: below REJECT the row is dropped, below REVIEW it
# is stored but flagged `validation_status="review"`.
VALIDATION_REJECT_THRESHOLD = float(os.getenv("VALIDATION_REJECT_THRESHOLD", "0.35"))
VALIDATION_REVIEW_THRESHOLD = float(os.getenv("VALIDATION_REVIEW_THRESHOLD", "0.70"))
# Pack-size explosion: how many size rows one uploaded product may become.
MAX_VARIANTS_PER_PRODUCT = int(os.getenv("MAX_VARIANTS_PER_PRODUCT", "6"))
ENABLE_PER_VARIANT_IMAGES = _bool("ENABLE_PER_VARIANT_IMAGES", "false")
PER_VARIANT_IMAGE_MAX_RESULTS = int(os.getenv("PER_VARIANT_IMAGE_MAX_RESULTS", "10"))
# Barcode lookup tuning. The cache TTL is long (30 days) because a GTIN for a
# given pack size does not change, and negative results are cached too.
BARCODE_LOOKUP_TIMEOUT_SECONDS = float(os.getenv("BARCODE_LOOKUP_TIMEOUT_SECONDS", "10"))
BARCODE_LOOKUP_CACHE_TTL_SECONDS = float(os.getenv("BARCODE_LOOKUP_CACHE_TTL_SECONDS", str(30 * 24 * 3600)))
BARCODE_LOOKUP_MAX_CONCURRENCY = int(os.getenv("BARCODE_LOOKUP_MAX_CONCURRENCY", "5"))
BARCODE_COUNTRY_TAG = os.getenv("BARCODE_COUNTRY_TAG", "india")
# Optional barcode source credentials. Each source disables itself when its
# key is blank, so leaving these unset simply narrows the lookup cascade.
GS1_INDIA_API_BASE_URL = os.getenv("GS1_INDIA_API_BASE_URL", "")
GS1_INDIA_API_KEY = os.getenv("GS1_INDIA_API_KEY", "")
UPC_DATABASE_API_KEY = os.getenv("UPC_DATABASE_API_KEY", "")
# ---------------------------------------------------------------------------
# HTTP client defaults
# ---------------------------------------------------------------------------

View File

@@ -25,6 +25,7 @@ from app.api.routers import health, brands, search, chat, catalog, system
from app.api.routers import stores, discounts, analytics as store_analytics, trending, recommendations, store_admin
from app.api.routers import nutrition, nutrition_admin, upload
from app.api.routers import auth, user_products, admin_train, mcp_info
from app.api.routers import store_catalog
from app.services.store_db import ensure_store_intelligence_schema
from app.services.nutrition_db import ensure_nutrition_schema
@@ -203,6 +204,7 @@ app.include_router(store_admin.router, prefix="/api")
app.include_router(nutrition.router, prefix="/api")
app.include_router(nutrition_admin.router, prefix="/api")
app.include_router(upload.router, prefix="/api")
app.include_router(store_catalog.router, prefix="/api")
app.include_router(mcp_info.router, prefix="/api")
# MCP lives outside /api on purpose: it is a protocol endpoint for AI clients,

View File

@@ -1,5 +1,6 @@
import re
from functools import lru_cache
from typing import Dict, Optional
BRAND_ALIASES = {
# Cadbury family
@@ -313,3 +314,59 @@ def get_known_sub_brands(brand: str) -> list[str]:
known.append(rest)
seen.add(rest)
return known
# ---------------------------------------------------------------------------
# FSSAI licences (stage 1 of the store-catalog pipeline)
# ---------------------------------------------------------------------------
# Ported from the sibling Universal_Catalog_Barcode_Enrichment project. Purely
# additive: BRAND_ALIASES and resolve_parent_brand are untouched.
#
# Food brands only. A miss means "not a food brand" (P&G, Colgate-Palmolive,
# J&J, Reckitt, Godrej) just as much as it means "unknown", so callers must
# treat None as "leave the column empty", never as an error.
FSSAI_LICENSES: Dict[str, str] = {
"britannia": "10012022000103",
"pepsico": "10012031000047",
"amul": "10012021000243",
"cadbury": "10014022002711",
"hindustan unilever": "10012022000217",
"nestle": "10012011000168",
"itc": "10018042000305",
"coca-cola": "10012042000424",
"tata": "12414003000511",
"parle": "10012022000046",
"marico": "10012022000258",
"dabur": "10012011000084",
"hatsun": "10012042000071",
"milky mist": "10017042003191",
"aachi": "10014042000577",
"sakthi": "10012042000300",
"kaleesuwari": "10012042000302",
"idhayam": "10012042000109",
"cavinkare": "10013042000366",
"naga": "10012042000192",
"manna": "10014042000169",
"grb": "10012042000148",
"anil": "10012042000213",
"lion dates": "10012042000244",
"brooke bond": "10013022001897",
"mother dairy": "10012011000015",
"haldiram": "10012011000140",
"fortune": "10012021000071",
"paper boat": "10012043000083",
"bisk farm": "10012031000012",
"mtr": "10012043000058",
"everest": "10012022000526",
"mdh": "10012011000062",
}
def get_fssai_license(brand: str) -> Optional[str]:
"""Return the 14-digit FSSAI licence number for a food brand, or None.
Resolves to the canonical parent first, so sub-brands (e.g. 'hul lux')
inherit the parent's licence and every row in a brand table agrees.
"""
canonical = resolve_parent_brand(brand).lower().strip()
return FSSAI_LICENSES.get(canonical)

View File

@@ -0,0 +1,338 @@
"""
Category -> Allowed Pack-Size-Unit Rules
==========================================
WHY THIS FILE EXISTS
---------------------
Reported bug: brand="Britannia" produced catalog rows like
"Tiger 10cm", "Nutrichoice 12cm", "Marie Gold 15cm" and
"Britannia Milk Bikis - 200ml, 500ml, 1L". The small local LLM
(qwen2.5:1.5b) occasionally predicts a *dimension* unit (cm/inch/m) or the
*wrong physical-state* unit (ml/L for a solid biscuit) instead of a
plausible FMCG pack-size unit, because nothing anywhere in the pipeline
ever told it - or checked afterwards - which units are even legal for a
given product category.
Two existing point-fixes come close but don't cover this:
- `price_estimator._parse_size_to_grams()` parses a size string into a
gram/ml-equivalent float for pricing math. It has an explicit fallback
for *unrecognised* unit tokens (case: "cm" is not in its litre/kg/base
unit sets): "leave the numeric value as-is rather than guessing". That
is the correct, conservative choice for a *pricing* function - but it
means "10cm" silently becomes 10.0 (treated as if it were 10 grams),
which is exactly wrong for a *validation* function: a bogus unit must
never be treated as an implicitly-valid one.
- `price_estimator.normalize_size_unit()` swaps ml<->g labels for
categories it already knows are exclusively solid or exclusively liquid
- but it only recognises `ml/l` <-> `g/kg` confusion. It has no concept
of "cm" at all, so a dimension unit passes through completely
unexamined.
This module is the missing, single source of truth for "what pack-size
units are even legal for this product category", used in two places:
1. GENERATION TIME (app/services/ollama_service.py): to tell the LLM, as
part of the prompt, exactly which units are allowed for the category it
is currently enumerating products for - so the wrong unit is far less
likely to be generated in the first place.
2. VALIDATION TIME (app/core/catalog_engine.py, app/services/
product_validator.py): to reject/replace a generated size string whose
unit contradicts its resolved category, deterministically and without
an LLM call, as a defence-in-depth backstop for whatever still gets
through the prompt-level fix (e.g. the `fetch_brand_catalog_with_gemini`
fallback path, or a cached/registry-sourced product).
Per the project's own stated requirement: package size must be determined
from the PRODUCT CATEGORY, and dimension units (cm/inch/m) must never be
generated for an FMCG consumable unless the category genuinely represents
a physical dimension (none currently in this catalog do).
"""
from __future__ import annotations
import re
from typing import Optional
# ---------------------------------------------------------------------------
# Unit vocabularies
# ---------------------------------------------------------------------------
WEIGHT_UNITS: set[str] = {
"g", "gm", "gms", "gram", "grams",
"kg", "kgs", "kilo", "kilos", "kilogram", "kilograms",
}
VOLUME_UNITS: set[str] = {
"ml", "mls", "millilitre", "millilitres", "milliliter", "milliliters",
"l", "lt", "ltr", "ltrs", "litre", "litres", "liter", "liters",
}
# Count-based packs (stationery, tablets, diapers, agarbatti sticks, etc.) -
# legitimate for a handful of non-food categories this catalog also covers.
COUNT_UNITS: set[str] = {
"pcs", "pc", "piece", "pieces", "unit", "units", "tablet", "tablets",
"capsule", "capsules", "strip", "strips", "sheet", "sheets", "roll",
"rolls", "stick", "sticks", "count", "ct", "pack", "packs",
}
# Physical-dimension units. These are NEVER a valid FMCG/personal-care pack
# size - no biscuit, chocolate, milk, tea, coffee, or ice-cream product is
# ever sold "by the centimetre". Kept as an explicit set (rather than "any
# unit we don't recognise") so the reject reason can name the exact problem.
DIMENSION_UNITS: set[str] = {
"cm", "centimetre", "centimetres", "centimeter", "centimeters",
"mm", "millimetre", "millimetres", "millimeter", "millimeters",
"m", "metre", "metres", "meter", "meters",
"inch", "inches", "in",
"ft", "feet", "foot",
}
_KNOWN_UNIT_TOKENS: set[str] = WEIGHT_UNITS | VOLUME_UNITS | COUNT_UNITS | DIMENSION_UNITS
# ---------------------------------------------------------------------------
# Category -> allowed unit TYPE ("weight" / "volume" / "weight_or_volume" /
# "count" / "unknown"). Keyed by lowercase, human-readable category strings -
# the same vocabulary produced by app.services.brand_registry.
# SUB_BRAND_CATEGORY and used as `known_category`/`category_value` throughout
# catalog_engine.py. This is deliberately the AUTHORITATIVE per-category
# rulebook the user specified: Biscuits -> g/kg, Milk/Juices -> ml/L,
# Chocolate -> g, Tea -> g, Coffee -> g, Ice Cream -> ml/L - extended to
# every other category already present in this project so every catalog row
# gets a consistent check, not just the categories in the bug report.
# ---------------------------------------------------------------------------
CATEGORY_UNIT_TYPE: dict[str, str] = {
# --- Weight-only (solid FMCG) ---
"biscuits & cookies": "weight",
"crackers": "weight",
"rusk": "weight",
"cakes & muffins": "weight",
"bakery & breads": "weight",
"snacks": "weight",
"chocolates": "weight",
"candy & confectionery": "weight",
"atta & staples": "weight",
"spices & masalas": "weight",
"pasta & noodles": "weight",
"noodles & instant food": "weight",
"breakfast cereal": "weight",
"dry fruits & nuts": "weight",
"health foods": "weight",
"millets": "weight",
"salt & staples": "weight",
"tea & coffee": "weight",
"tea": "weight",
"coffee": "weight",
"health drinks": "weight", # malted/powder drinks: Horlicks, Bournvita, Boost, Complan
"bath soap": "weight",
"stationery": "count",
"household - agarbatti": "count",
"feminine hygiene": "count",
"baby care": "weight_or_volume",
# --- Volume-only (liquid FMCG) ---
"beverages": "volume",
"juices": "volume",
"food - soups & sauces": "volume",
"household cleaning": "volume",
"dishwash": "volume",
"ice cream": "volume",
"cooking oils": "volume",
"wellness oils": "volume",
"household - lamp oil": "volume",
# --- Mixed weight-or-volume (category legitimately spans both solid and
# liquid sub-products, e.g. "Dairy" covers milk (ml) AND
# cheese/paneer/butter/curd (g)) ---
"dairy": "weight_or_volume",
"dairy - desserts": "weight_or_volume",
"hair care": "weight_or_volume", # shampoo/conditioner (ml) vs hair oil/cream (g/ml)
"skin & bath care": "weight_or_volume",
"skin care": "weight_or_volume",
"beauty care": "weight_or_volume",
"fragrance & deodorants": "weight_or_volume",
"men's grooming": "weight_or_volume",
"detergents & fabric care": "weight_or_volume", # powder (kg) vs liquid (L)
"oral care": "weight_or_volume", # toothpaste (g) vs mouthwash (ml)
"personal care": "weight_or_volume",
"personal care - mosquito repellent": "weight_or_volume",
"mosquito repellent": "weight_or_volume",
"air freshener": "weight_or_volume",
"health care - cold & cough": "weight_or_volume",
"health care - digestive": "weight_or_volume",
"health care - ayurvedic": "weight_or_volume",
"health care - antiseptic": "weight_or_volume",
"health care - first aid": "count",
"pickles & chutneys": "weight",
"food - spreads": "weight",
"food - mixes": "weight",
"rice & pulses": "weight",
"sweets": "weight",
"ready to eat": "weight_or_volume",
}
# No category in this FMCG/personal-care catalog is legitimately sold "by
# the centimetre" - kept as an explicit, extensible override point (e.g. a
# future "Home Textiles" or "Furnishings" category) rather than a hardcoded
# blanket rule, per the requirement that dimension units are only allowed
# "if the product category genuinely represents physical dimensions".
DIMENSION_OK_CATEGORIES: set[str] = set()
def get_unit_type(category: Optional[str]) -> str:
"""Return the allowed unit TYPE for `category` ('weight', 'volume',
'weight_or_volume', 'count', or 'unknown' if the category isn't in our
rulebook - callers should treat 'unknown' permissively, not as a
rejection, since it just means we have no opinion yet, not that
anything is wrong)."""
if not category:
return "unknown"
return CATEGORY_UNIT_TYPE.get(category.strip().lower(), "unknown")
def get_allowed_units(category: Optional[str]) -> set[str]:
"""Concrete set of unit tokens allowed for `category`."""
unit_type = get_unit_type(category)
if unit_type == "weight":
return set(WEIGHT_UNITS)
if unit_type == "volume":
return set(VOLUME_UNITS)
if unit_type == "weight_or_volume":
return WEIGHT_UNITS | VOLUME_UNITS
if unit_type == "count":
return set(COUNT_UNITS)
# unknown category: permissive - anything except a dimension unit
return WEIGHT_UNITS | VOLUME_UNITS | COUNT_UNITS
def allowed_units_hint(category: Optional[str]) -> str:
"""Short, human-readable allowed-units phrase for `category`, suitable
for embedding directly into an LLM prompt, e.g. 'grams (g) or
kilograms (kg)'. Used by ollama_service.py to tell the model, per
category, exactly which units it must use."""
unit_type = get_unit_type(category)
return {
"weight": "grams (g) or kilograms (kg) ONLY",
"volume": "millilitres (ml) or litres (L) ONLY",
"weight_or_volume": "grams/kilograms (g/kg) for solid items or millilitres/litres (ml/L) for liquid items",
"count": "a piece/unit count (e.g. '10 pcs', '1 pack')",
"unknown": "grams (g), kilograms (kg), millilitres (ml), or litres (L) as appropriate",
}[unit_type]
def is_forbidden_dimension_unit(unit: Optional[str], category: Optional[str] = None) -> bool:
"""True if `unit` is a physical-dimension unit (cm/inch/m/...) that is
never valid for `category` (see DIMENSION_OK_CATEGORIES)."""
if not unit:
return False
u = unit.strip().lower()
if u not in DIMENSION_UNITS:
return False
if category and category.strip().lower() in DIMENSION_OK_CATEGORIES:
return False
return True
def parse_unit(size: Optional[str]) -> tuple[Optional[float], Optional[str]]:
"""Extract (numeric_value, unit_token) from a size string like '200g',
'1.5 L', '10cm'. Returns (None, None) if no leading number is found."""
if not size:
return None, None
s = size.strip().lower()
match = re.search(r"(\d+(?:\.\d+)?)\s*([a-z]*)", s)
if not match or not match.group(1):
return None, None
value = float(match.group(1))
unit = match.group(2).strip() or None
return value, unit
def validate_unit_for_category(size: Optional[str], category: Optional[str]) -> tuple[bool, Optional[str]]:
"""Check whether `size`'s unit is legal for `category`.
Returns (is_valid, reason). `reason` is populated whenever is_valid is
False, explaining exactly what was wrong (dimension unit vs.
wrong-physical-state unit vs. unrecognised token) so callers can log or
surface it in an audit trail.
"""
if not size or not size.strip():
return False, "size is missing"
value, unit = parse_unit(size)
if value is None:
# No parseable number at all (e.g. "Family Pack") - not this
# function's concern; price_estimator's own fallback handles it.
return True, None
if not unit:
# Bare number, no unit token (e.g. "200") - ambiguous but not a
# unit-category contradiction per se; leave to size-plausibility
# checks elsewhere.
return True, None
if is_forbidden_dimension_unit(unit, category):
return False, (
f"unit '{unit}' is a physical dimension, not a valid FMCG pack-size "
f"unit for category '{category}'"
)
if unit not in _KNOWN_UNIT_TOKENS:
# Unrecognised token (e.g. "pcs" variants we don't track, "x6",
# brand-specific packaging words) - not a contradiction we can
# prove, so don't reject.
return True, None
allowed = get_allowed_units(category)
if unit not in allowed:
unit_type = get_unit_type(category)
return False, (
f"unit '{unit}' is not valid for category '{category}' "
f"(expects {unit_type.replace('_', ' ')} units: {allowed_units_hint(category)})"
)
return True, None
def fix_or_reject_size(
size: Optional[str], category: Optional[str], product_title: str = ""
) -> tuple[Optional[str], bool, Optional[str]]:
"""Authoritative size-unit correction, called at both generation time
and validation time.
Returns (corrected_size_or_None, was_changed, reason):
- If `size`'s unit is already valid for `category`: (size, False, None).
- If `size` uses the WRONG PHYSICAL STATE for `category` (e.g. '200ml'
for a strictly-solid category) but the number itself is plausible: the
unit label is swapped (numeric value preserved), same behaviour as
price_estimator.normalize_size_unit(), since a mislabelled-but-
plausible quantity is safely recoverable.
- If `size` uses a DIMENSION unit (cm/inch/m) or the unit can't be
meaningfully reinterpreted for the category: there is no safe
conversion (10cm has no defensible gram-equivalent), so this returns
(None, True, reason) - the caller MUST NOT keep the original value and
should substitute a category-appropriate default size instead (see
price_estimator.default_size_variants()), never silently reuse the
rejected number under a new label.
"""
if not size or not size.strip():
return size, False, None
value, unit = parse_unit(size)
if value is None or not unit:
return size, False, None
ok, reason = validate_unit_for_category(size, category)
if ok:
return size, False, None
if is_forbidden_dimension_unit(unit, category):
# No defensible conversion exists - reject outright.
return None, True, reason
# Wrong-physical-state-but-plausible-number case: swap the label,
# preserving the numeric value, mirroring the existing
# price_estimator.normalize_size_unit() behaviour for ml<->g.
unit_type = get_unit_type(category)
value_str = (
str(int(value)) if float(value).is_integer() else str(value)
)
if unit_type == "weight" and unit in VOLUME_UNITS:
corrected = f"{value_str}g" if unit in {"ml", "mls"} else f"{value_str}kg"
return corrected, True, reason
if unit_type == "volume" and unit in WEIGHT_UNITS:
corrected = f"{value_str}ml" if unit in {"g", "gm", "gms", "gram", "grams"} else f"{value_str}L"
return corrected, True, reason
# Anything else we can't confidently repair (e.g. a count-unit category
# given a weight unit) - reject rather than guess.
return None, True, reason

View File

@@ -0,0 +1,50 @@
"""
Product Enrichment Service
===========================
WHY THIS PACKAGE EXISTS
------------------------
Everything upstream of this package (ollama_service.py, product_validator.py,
sku_service.py, image_search.py, ...) either GENERATES a product row or
JUDGES whether an already-generated row is plausible. None of it ever goes
out and fetches a piece of ground-truth data from an authoritative external
source and attaches it to the row. That is what this package is for.
The first concrete enricher is barcode lookup (see `barcode/`): resolving a
real, checksum-valid EAN-13/UPC/GTIN for a catalog row from trusted external
sources, never from the LLM. The package is deliberately structured so this
is ONE stage among what will eventually be several independent ones (HSN
code, GST rate, nutrition, allergens, manufacturer details, ...) - see
`base.py` for the `EnrichmentStage` contract every future stage implements,
and `pipeline.py` for the orchestrator that runs them in sequence.
DESIGN PRINCIPLES (apply to every enrichment stage, present and future)
-------------------------------------------------------------------------
- NEVER FABRICATE. If a stage cannot find authentic, verifiable data, it
writes NULL/None + a "not_found" status - never a guessed or LLM-
generated value. This mirrors the whole point of `product_validator.py`
and `sku_service.py`'s "real ID or clearly-labelled internal fallback"
design, taken one step further: for barcodes there IS no safe synthetic
fallback (a fabricated barcode is actively harmful - it can collide with
a real product), so the fallback is always NULL, never a generated value.
- NEVER RAISE. A single product's enrichment failure (timeout, malformed
response, source outage) must never abort the batch or the pipeline run.
Every stage catches its own errors and degrades to "not found" for that
one row.
- ADDITIVE. This package has no dependents before this change and does not
modify the behaviour of any existing module; it is only ever imported
from new call sites (see `app/core/catalog_engine.py`'s Step 2.6 and
`app/services/vector_store.py`'s additive barcode columns).
- INDEPENDENT STAGES. Each stage owns its own external calls, matching
rules, caching, and DB columns. A future HSN/GST/nutrition stage does not
need to know barcode lookup exists, and vice versa - see `pipeline.py`.
"""
from app.services.enrichment.base import EnrichmentStage, StageOutcome
from app.services.enrichment.pipeline import EnrichmentPipeline, run_default_pipeline
__all__ = [
"EnrichmentStage",
"StageOutcome",
"EnrichmentPipeline",
"run_default_pipeline",
]

View File

@@ -0,0 +1,47 @@
"""
Barcode Retrieval & Product Enrichment module.
Public entry points:
lookup_barcode(brand, product_title, size, category="") -> BarcodeResult
Synchronous single-product cascading lookup. Safe to call from a
script, notebook, or the Streamlit UI's "look up a barcode" action.
batch_lookup_barcodes(products, brand) -> list[BarcodeResult]
Async batch lookup used by the pipeline stage (see `stage.py`) and
available directly for a CLI/manual batch job.
See the package's module docstrings for the full architecture:
models.py - BarcodeCandidate / BarcodeResult / enums
validators.py - EAN-13/GTIN/UPC checksum + format validation
matching.py - exact product matching rules
cache.py - local SQLite lookup cache
retry.py - tenacity-based retry/backoff
sources/ - the 4-tier cascading source registry
service.py - orchestrates cache -> sources -> validate -> match
stage.py - adapts the service to the generic EnrichmentStage
contract used by app/services/enrichment/pipeline.py
"""
from typing import Any, Dict, List
from app.services.enrichment.barcode.models import BarcodeCandidate, BarcodeResult, BarcodeType, LookupStatus
from app.services.enrichment.barcode.service import BarcodeLookupService, get_default_service
def lookup_barcode(brand: str, product_title: str, size: str, category: str = "") -> BarcodeResult:
return get_default_service().lookup_one(brand, product_title, size, category)
async def batch_lookup_barcodes(products: List[Dict[str, Any]], brand: str) -> List[BarcodeResult]:
return await get_default_service().batch_lookup(products, brand)
__all__ = [
"BarcodeCandidate",
"BarcodeResult",
"BarcodeType",
"LookupStatus",
"BarcodeLookupService",
"get_default_service",
"lookup_barcode",
"batch_lookup_barcodes",
]

View File

@@ -0,0 +1,127 @@
"""
Local lookup cache for barcode results.
WHY A SEPARATE SQLITE FILE (data/cache/barcode_lookup.db) INSTEAD OF
POSTGRES
---------------------------------------------------------------------
This cache exists purely to avoid repeating the SAME external-API search
(GS1/Open Food Facts/UPCItemDB/manufacturer site) twice for the same
(brand, product, size) - see "Cache barcode lookups to avoid repeated API
calls" / "Prevent duplicate barcode searches" (Performance Requirements).
It is deliberately NOT the authoritative store (that is Postgres, via
`vector_store.py`'s additive barcode columns) - it needs to work even when
Postgres is unreachable/USE_PGVECTOR=false, and it needs to be cheap and
local on an 8GB-RAM/CPU-only machine, so plain stdlib `sqlite3` (no new
dependency, no server process) is the right tool here, following the same
"file-backed, atomic-write" spirit as `sku_service.py`'s SKU sequence
counter.
A row's cache key is a normalized (brand, product_title, size) triple, not
the barcode itself (we're caching "what did we already look up", not "what
maps to what").
"""
from __future__ import annotations
import json
import logging
import re
import sqlite3
import threading
import time
from pathlib import Path
from typing import Optional
from app.services.enrichment.barcode.models import BarcodeResult
logger = logging.getLogger(__name__)
_DB_PATH = Path("data") / "cache" / "barcode_lookup.db"
_lock = threading.Lock()
_initialized = False
def _normalize_key_part(text: Optional[str]) -> str:
return re.sub(r"\s+", " ", (text or "").strip().lower())
def cache_key(brand: str, product_title: str, size: str) -> str:
return f"{_normalize_key_part(brand)}|{_normalize_key_part(product_title)}|{_normalize_key_part(size)}"
def _connect() -> sqlite3.Connection:
_DB_PATH.parent.mkdir(parents=True, exist_ok=True)
conn = sqlite3.connect(str(_DB_PATH), timeout=10)
conn.execute("PRAGMA journal_mode=WAL")
return conn
def _ensure_schema(conn: sqlite3.Connection) -> None:
global _initialized
if _initialized:
return
conn.execute(
"""
CREATE TABLE IF NOT EXISTS barcode_lookup_cache (
cache_key TEXT PRIMARY KEY,
brand TEXT,
product_title TEXT,
size TEXT,
result_json TEXT NOT NULL,
created_at REAL NOT NULL
)
"""
)
conn.commit()
_initialized = True
def get_cached(brand: str, product_title: str, size: str, ttl_seconds: float) -> Optional[BarcodeResult]:
"""Returns a cached BarcodeResult if present and not older than
`ttl_seconds`, else None. Never raises - a cache read failure is
treated exactly like a cache miss."""
key = cache_key(brand, product_title, size)
try:
with _lock, _connect() as conn:
_ensure_schema(conn)
row = conn.execute(
"SELECT result_json, created_at FROM barcode_lookup_cache WHERE cache_key = ?",
(key,),
).fetchone()
except Exception as e:
logger.debug(f"Barcode cache read failed for '{key}': {e}")
return None
if not row:
return None
result_json, created_at = row
if ttl_seconds > 0 and (time.time() - created_at) > ttl_seconds:
return None
try:
data = json.loads(result_json)
return BarcodeResult(**data)
except Exception as e:
logger.debug(f"Barcode cache entry for '{key}' unreadable, treating as miss: {e}")
return None
def set_cached(brand: str, product_title: str, size: str, result: BarcodeResult) -> None:
"""Best-effort write - a failure here never blocks the lookup itself,
it just means this row won't benefit from caching next time."""
key = cache_key(brand, product_title, size)
try:
payload = json.dumps(result.__dict__)
with _lock, _connect() as conn:
_ensure_schema(conn)
conn.execute(
"""
INSERT INTO barcode_lookup_cache (cache_key, brand, product_title, size, result_json, created_at)
VALUES (?, ?, ?, ?, ?, ?)
ON CONFLICT(cache_key) DO UPDATE SET
result_json = excluded.result_json,
created_at = excluded.created_at
""",
(key, brand, product_title, size, payload, time.time()),
)
conn.commit()
except Exception as e:
logger.debug(f"Barcode cache write failed for '{key}': {e}")

View File

@@ -0,0 +1,141 @@
"""
Exact-product matching rules for barcode candidates.
A source returning SOME EAN-13 for a product with a similar name is not
good enough - the "Product Matching Rules" requirement is explicit that
brand, product name, variant, and size must all match, and that a
different pack size (or a differently-branded variant like "Sugar-Free")
must be REJECTED rather than accepted as "close enough". This module is
deliberately conservative: a candidate only passes if every check agrees;
`is_valid_ean13`-style tricks are not enough to earn a false positive here
the way they might in a fuzzy search feature.
Deliberately stdlib-only (`difflib`, already in the standard library) -
this project explicitly removed `fuzzywuzzy`/`python-Levenshtein` as dead
weight (see requirements.txt), so no new fuzzy-matching dependency is
introduced here either.
"""
from __future__ import annotations
import re
from difflib import SequenceMatcher
from typing import Iterable, Optional
from app.services.enrichment.barcode.models import BarcodeCandidate
# quantity_utils rather than image_search: this repo's image_search.py has no
# quantity helpers, and it is a working file the store-catalog work leaves alone.
from app.services.quantity_utils import quantities_match
# Words that mark a genuinely DIFFERENT retail variant from the plain/base
# product - if the candidate's title contains one of these and the
# target product's own title/product_name does NOT, the candidate is
# rejected outright regardless of how well the rest of the name matches
# (this is exactly the "Marie Gold 200g" vs "Marie Gold Sugar-Free"
# example from the spec). Kept as a small, explicit, reviewable list
# rather than a fuzzy heuristic - false negatives (missing a real match)
# are far cheaper here than false positives (storing the wrong product's
# barcode).
VARIANT_DISTINGUISHING_TERMS = {
"sugar free", "sugarfree", "sugar-free", "no sugar", "zero sugar",
"diet", "family pack", "value pack", "jumbo pack", "combo pack",
"combo", "gift pack", "gift box", "mini pack", "party pack",
"refill pack", "refill", "pouch pack", "twin pack", "saver pack",
"economy pack", "jar", "tin", "pet jar",
}
_STOPWORDS = {"the", "and", "of", "with", "for", "a", "an", "new", "pack", "india"}
def _normalize(text: Optional[str]) -> str:
return re.sub(r"[^a-z0-9\s]", " ", (text or "").lower()).strip()
def _tokens(text: Optional[str]) -> set:
return {t for t in _normalize(text).split() if t and t not in _STOPWORDS}
def brand_matches(candidate_brand: str, target_brand: str, brand_aliases: Optional[Iterable[str]] = None) -> bool:
"""True if the candidate's reported brand plausibly refers to the same
brand as the target. Accepts an exact/substring match on the target
brand name itself, or a match against any known alias (e.g. "HUL" for
"Hindustan Unilever") supplied by the caller via
`brand_registry.get_brand_alias_set()`."""
cand = _normalize(candidate_brand)
target = _normalize(target_brand)
if not cand or not target:
return False
if target in cand or cand in target:
return True
for alias in (brand_aliases or []):
alias_n = _normalize(alias)
if alias_n and (alias_n in cand or cand in alias_n):
return True
return False
def size_matches(candidate_size: str, target_size: str, tolerance: float = 0.03) -> bool:
"""Tight tolerance (3%, vs. the 15% used for image matching elsewhere
in this project) - a barcode belongs to exactly one pack size, so
"close enough" is not an acceptable bar the way it is for reusing a
product photo. Falls back to a plain normalized-string equality check
when neither string parses as a numeric quantity (e.g. count-based
sizes like "10 tablets"), rather than silently treating unparsable
sizes as a match.
"""
if not candidate_size or not target_size:
return False
if quantities_match(candidate_size, target_size, tolerance=tolerance):
return True
return _normalize(candidate_size) == _normalize(target_size)
def has_conflicting_variant_terms(candidate_title: str, target_title: str) -> bool:
"""True if the candidate's title names a distinguishing variant
(sugar-free, family pack, ...) that the target product does NOT -
which means the candidate is a real but DIFFERENT product, not the one
being looked up."""
cand_n = _normalize(candidate_title)
target_n = _normalize(target_title)
for term in VARIANT_DISTINGUISHING_TERMS:
if term in cand_n and term not in target_n:
return True
return False
def name_similarity(candidate_title: str, target_title: str) -> float:
"""0.0-1.0 token-overlap-weighted similarity. Used only as a secondary
signal / diagnostic (`match_confidence`) - never as the sole gate for
acceptance; see `is_match()`."""
cand_tokens, target_tokens = _tokens(candidate_title), _tokens(target_title)
if not cand_tokens or not target_tokens:
return 0.0
overlap = len(cand_tokens & target_tokens) / len(target_tokens)
seq_ratio = SequenceMatcher(None, _normalize(candidate_title), _normalize(target_title)).ratio()
return round((overlap * 0.6) + (seq_ratio * 0.4), 3)
def is_match(candidate: BarcodeCandidate, target_brand: str, target_title: str, target_size: str,
brand_aliases: Optional[Iterable[str]] = None,
min_name_similarity: float = 0.45) -> tuple[bool, float]:
"""The combined gate a candidate must pass to be accepted:
1. Brand matches (or overlaps a known alias).
2. Pack size matches within a tight tolerance.
3. No conflicting variant terms (family pack / sugar-free / ...).
4. Product-name similarity clears a floor - catches the case where
brand+size coincidentally match but it's a completely different
product line from the same brand.
Returns (matched, confidence) - confidence is diagnostic only, stored
on the result for audit/QA but never used to override rule 1-3.
"""
if not brand_matches(candidate.candidate_brand, target_brand, brand_aliases):
return False, 0.0
if not size_matches(candidate.candidate_size, target_size):
return False, 0.0
if has_conflicting_variant_terms(candidate.candidate_title, target_title):
return False, 0.0
similarity = name_similarity(candidate.candidate_title, target_title)
if similarity < min_name_similarity:
return False, similarity
return True, similarity

View File

@@ -0,0 +1,82 @@
"""Data shapes shared across the barcode lookup module."""
from __future__ import annotations
import time
from dataclasses import dataclass, field
from enum import Enum
from typing import Optional
class BarcodeType(str, Enum):
EAN13 = "EAN-13"
UPC_A = "UPC-A"
GTIN14 = "GTIN-14"
GTIN8 = "GTIN-8"
UNKNOWN = "UNKNOWN"
class LookupStatus(str, Enum):
"""Mirrors the `barcode_lookup_status` DB column."""
VERIFIED = "verified" # found + checksum-valid + matched product
NOT_FOUND = "not_found" # every source exhausted, nothing matched
INVALID_CANDIDATE = "invalid_candidate" # a candidate was found but failed
# checksum/format validation or product matching - rejected, not stored
ERROR = "error" # a source/network error prevented a full search
DISABLED = "disabled" # barcode lookup turned off via settings
CACHED = "cached" # served from the local lookup cache
@dataclass
class BarcodeCandidate:
"""A raw, not-yet-validated candidate returned by one source."""
barcode: str
source_name: str # e.g. "Open Food Facts"
candidate_title: str = "" # the source's own product title, for matching
candidate_brand: str = ""
candidate_size: str = ""
candidate_countries: str = "" # raw country tag/string from the source, if any
@dataclass
class BarcodeResult:
"""Final, caller-facing result. Field names match the DB columns in
`vector_store.py` and the JSON export shape requested for the
catalog (`barcode`, `barcode_type`, `barcode_source`, `barcode_verified`,
...) 1:1, so `service.py` -> product dict -> DB row -> JSON export is a
straight field copy with no renaming at any layer.
"""
barcode: Optional[str] = None
barcode_type: Optional[str] = None
gtin: Optional[str] = None
ean13: Optional[str] = None
upc: Optional[str] = None
barcode_source: Optional[str] = None
barcode_verified: bool = False
barcode_lookup_status: str = LookupStatus.NOT_FOUND.value
barcode_last_updated: float = field(default_factory=time.time)
match_confidence: float = 0.0
@classmethod
def null_result(cls, status: LookupStatus = LookupStatus.NOT_FOUND) -> "BarcodeResult":
"""The required NULL shape: barcode=NULL, barcode_verified=false,
barcode_source=NULL - used for every path where no authentic,
matching barcode could be confirmed. Never construct a
BarcodeResult with a barcode value outside of `service.py`'s
validated-and-matched path.
"""
return cls(barcode_lookup_status=status.value)
def as_product_fields(self) -> dict:
"""Flat dict merged directly into the catalog row - see
`stage.py`."""
return {
"barcode": self.barcode,
"barcode_type": self.barcode_type,
"gtin": self.gtin,
"ean13": self.ean13,
"upc": self.upc,
"barcode_source": self.barcode_source,
"barcode_verified": self.barcode_verified,
"barcode_lookup_status": self.barcode_lookup_status,
"barcode_last_updated": self.barcode_last_updated,
}

View File

@@ -0,0 +1,47 @@
"""
Configurable retry-with-exponential-backoff for barcode source calls.
Built on `tenacity`, already a project dependency (see requirements.txt -
used elsewhere for httpx-based retries). Only retries on genuinely
transient failures (network/timeout errors); a malformed response or a
"no results" outcome is not retried, since retrying those wastes the
source's rate-limit budget for no benefit (relevant for UPCItemDB's
trial-tier daily cap in particular).
"""
from __future__ import annotations
import logging
import requests
from tenacity import (
retry,
retry_if_exception_type,
stop_after_attempt,
wait_exponential,
before_sleep_log,
)
logger = logging.getLogger(__name__)
_RETRYABLE_EXCEPTIONS = (
requests.exceptions.ConnectionError,
requests.exceptions.Timeout,
requests.exceptions.ChunkedEncodingError,
)
def with_retry(max_attempts: int = 3, min_wait: float = 1.0, max_wait: float = 8.0):
"""Decorator factory: `max_attempts` total tries, exponential backoff
between `min_wait` and `max_wait` seconds. Applied per-source (see
`sources/*.py`), not globally, so one slow/unreliable source retrying
doesn't compound delay across the whole cascade - a source that keeps
failing simply falls through to the next tier faster than a shared
global retry budget would allow.
"""
return retry(
reraise=True,
stop=stop_after_attempt(max_attempts),
wait=wait_exponential(multiplier=min_wait, max=max_wait),
retry=retry_if_exception_type(_RETRYABLE_EXCEPTIONS),
before_sleep=before_sleep_log(logger, logging.DEBUG),
)

View File

@@ -0,0 +1,170 @@
"""
BarcodeLookupService - the cascading lookup orchestrator.
This is the ONE place that decides "is this barcode good enough to store".
Every candidate from every source, regardless of tier, passes through the
exact same two gates before it can become a `BarcodeResult`:
1. `validators.validate_barcode()` - checksum + format.
2. `matching.is_match()` - brand + size + variant + name.
A source being "trusted" (e.g. GS1 India) does not skip either gate -
trust only affects ORDER (which tier is tried first), never whether
validation is required.
Cascade behaviour (see module docstring in `sources/__init__.py` for the
tier list): tiers are tried in order; the first tier that produces at
least one validated, matching candidate wins and the search stops - "The
system should stop searching as soon as a verified barcode is found."
Every tier is independently wrapped so a source outage never prevents
falling through to the next one, and if every tier is exhausted with
nothing matching, the result is the required NULL shape
(`BarcodeResult.null_result()`), never a guess.
"""
from __future__ import annotations
import asyncio
import logging
from typing import Any, Dict, List, Optional
from app.infrastructure.settings import (
ENABLE_BARCODE_LOOKUP,
BARCODE_LOOKUP_CACHE_TTL_SECONDS,
BARCODE_LOOKUP_MAX_CONCURRENCY,
)
from app.services.enrichment.barcode import cache
from app.services.enrichment.barcode.matching import is_match
from app.services.enrichment.barcode.models import BarcodeCandidate, BarcodeResult, BarcodeType, LookupStatus
from app.services.enrichment.barcode.sources import get_default_sources
from app.services.enrichment.barcode.validators import classify_barcode_type, to_ean13, validate_barcode
logger = logging.getLogger(__name__)
try:
from app.services.brand_registry import get_brand_alias_set
except Exception: # pragma: no cover - keeps the service importable/testable
# in isolation even if brand_registry ever fails to import.
def get_brand_alias_set(_brand: str) -> set:
return set()
def _build_result(barcode: str, source_name: str, confidence: float) -> BarcodeResult:
btype = classify_barcode_type(barcode)
ean13 = to_ean13(barcode) if btype in (BarcodeType.UPC_A, BarcodeType.EAN13) else None
return BarcodeResult(
barcode=barcode,
barcode_type=btype.value,
gtin=barcode,
ean13=ean13,
upc=barcode if btype == BarcodeType.UPC_A else None,
barcode_source=source_name,
barcode_verified=True,
barcode_lookup_status=LookupStatus.VERIFIED.value,
match_confidence=confidence,
)
class BarcodeLookupService:
def __init__(self, sources=None):
self.sources = sources if sources is not None else get_default_sources()
def lookup_one(self, brand: str, product_title: str, size: str, category: str = "",
use_cache: bool = True) -> BarcodeResult:
"""Synchronous cascading lookup for a single (brand, product,
size). Never raises - every failure path returns a NULL-shaped
`BarcodeResult` with a descriptive `barcode_lookup_status`."""
if not ENABLE_BARCODE_LOOKUP:
return BarcodeResult.null_result(LookupStatus.DISABLED)
if not brand or not product_title or not size:
logger.debug(f"Barcode lookup skipped - missing brand/title/size ('{brand}', '{product_title}', '{size}')")
return BarcodeResult.null_result(LookupStatus.ERROR)
if use_cache:
cached = cache.get_cached(brand, product_title, size, BARCODE_LOOKUP_CACHE_TTL_SECONDS)
if cached is not None:
result = cached
result.barcode_lookup_status = (
LookupStatus.CACHED.value if result.barcode_verified else result.barcode_lookup_status
)
return result
brand_aliases = self._safe_brand_aliases(brand)
had_source_error = False
for source in self.sources:
if not source.available:
continue
try:
raw_candidates = source.search(brand, product_title, size, category)
except Exception as e:
# Sources are documented to never raise, but this is the
# pipeline-wide safety net referenced in base.py.
logger.warning(f"[{source.name}] barcode search raised unexpectedly: {e}")
had_source_error = True
continue
result = self._first_validated_match(raw_candidates, brand, product_title, size, brand_aliases)
if result is not None:
if use_cache:
cache.set_cached(brand, product_title, size, result)
return result
status = LookupStatus.ERROR if had_source_error else LookupStatus.NOT_FOUND
result = BarcodeResult.null_result(status)
if use_cache:
# Cache negative results too (Performance Requirements: "Prevent
# duplicate barcode searches") - a short TTL still applies, so a
# transient "not found" doesn't permanently block a later re-check.
cache.set_cached(brand, product_title, size, result)
return result
@staticmethod
def _safe_brand_aliases(brand: str) -> set:
try:
return get_brand_alias_set(brand)
except Exception:
return set()
@staticmethod
def _first_validated_match(raw_candidates: List[BarcodeCandidate], brand: str, product_title: str,
size: str, brand_aliases: set) -> Optional[BarcodeResult]:
for candidate in raw_candidates:
clean_barcode = validate_barcode(candidate.barcode)
if not clean_barcode:
continue # invalid checksum/length/format - never stored, not even flagged
matched, confidence = is_match(candidate, brand, product_title, size, brand_aliases)
if not matched:
continue
return _build_result(clean_barcode, candidate.source_name, confidence)
return None
async def batch_lookup(self, products: List[Dict[str, Any]], brand: str) -> List[BarcodeResult]:
"""Async batch entry point used by `stage.py`. Bounded concurrency
(`BARCODE_LOOKUP_MAX_CONCURRENCY`) so a large brand catalog doesn't
fire dozens of simultaneous requests at any one source, and results
are returned in the SAME ORDER as `products` regardless of which
finished first, so callers can zip() them back together safely."""
semaphore = asyncio.Semaphore(max(1, BARCODE_LOOKUP_MAX_CONCURRENCY))
async def _one(product: Dict[str, Any]) -> BarcodeResult:
async with semaphore:
return await asyncio.to_thread(
self.lookup_one,
brand,
product.get("title") or product.get("product_name") or "",
product.get("size") or "",
product.get("category") or "",
)
return await asyncio.gather(*(_one(p) for p in products))
# Module-level singleton - sources are stateless, so one shared instance is
# fine for both the sync CLI/test path and the async pipeline stage.
_default_service: Optional[BarcodeLookupService] = None
def get_default_service() -> BarcodeLookupService:
global _default_service
if _default_service is None:
_default_service = BarcodeLookupService()
return _default_service

View File

@@ -0,0 +1,35 @@
"""
Tier-ordered source registry - this list IS the cascade order described in
the spec: GS1 India -> Open Food Facts -> trusted barcode databases ->
manufacturer website -> (exhausted -> NULL). `service.py` iterates this
list in order and stops at the first source that produces a validated,
matching candidate.
"""
from app.services.enrichment.barcode.sources.base import BarcodeSource
from app.services.enrichment.barcode.sources.gs1_india import GS1IndiaSource
from app.services.enrichment.barcode.sources.open_food_facts import OpenFoodFactsSource
from app.services.enrichment.barcode.sources.upc_database import UPCDatabaseSource
from app.services.enrichment.barcode.sources.manufacturer_site import ManufacturerSiteSource
def get_default_sources() -> list[BarcodeSource]:
"""Fresh instances each call - sources are stateless/cheap to
construct, and this avoids any shared-mutable-state surprises across
concurrent batch lookups."""
sources = [
GS1IndiaSource(),
OpenFoodFactsSource(),
UPCDatabaseSource(),
ManufacturerSiteSource(),
]
return sorted(sources, key=lambda s: s.tier)
__all__ = [
"BarcodeSource",
"GS1IndiaSource",
"OpenFoodFactsSource",
"UPCDatabaseSource",
"ManufacturerSiteSource",
"get_default_sources",
]

View File

@@ -0,0 +1,46 @@
"""
Pluggable barcode source contract, following the same adapter-registry
pattern already used for marketplace SKU resolution
(app/services/sku_service.py's `_MARKETPLACE_PATTERNS`) and the Product
Resolver's adapter registry (see the Catalog_Project's `resolve_parent_brand`
family) - every source is a self-contained class with one job: given a
brand/title/size, return zero or more raw candidates. It does NOT decide
whether a candidate is a match (that's `matching.py`) or whether its
barcode is valid (that's `validators.py`) - keeping those concerns
separate is what lets `service.py` apply the SAME validation/matching gate
uniformly regardless of which tier produced the candidate.
"""
from __future__ import annotations
from abc import ABC, abstractmethod
from typing import List
from app.services.enrichment.barcode.models import BarcodeCandidate
class BarcodeSource(ABC):
#: Human-readable label stored in `barcode_source` on a successful
#: match, e.g. "Open Food Facts", "GS1 India".
name: str = "Unknown Source"
#: Lower = tried first in the cascade (see sources/__init__.py).
tier: int = 99
@property
def available(self) -> bool:
"""False when the source is unusable for reasons unrelated to any
one query (missing API key, disabled via settings, dependency not
installed, ...). The cascade skips unavailable sources entirely
rather than querying and failing every time."""
return True
@abstractmethod
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
"""Best-effort search. MUST NOT raise - catch internally and
return [] on any failure (network error, malformed response,
nothing found). `service.py` treats an empty list and an
exception identically (fall through to the next tier), so
swallowing the error here vs. letting it propagate makes no
behavioural difference except robustness - always swallow it.
"""
raise NotImplementedError

View File

@@ -0,0 +1,91 @@
"""
Tier 1: GS1 India / GEPIR (the official Indian GTIN registry).
HONESTY NOTE - READ BEFORE WIRING UP CREDENTIALS
--------------------------------------------------
GS1 India does not currently publish a free, public, no-registration JSON
API for GTIN lookup (unlike Open Food Facts). Their real product-data
registry is accessed either through GEPIR (https://gepir.gs1.org/ - a
human web UI with no documented public API) or through GS1 India's paid
"Verified by GS1" data-as-a-service product, which requires a commercial
account and an API key.
Rather than scrape GEPIR's web UI (fragile, likely against its terms, and
exactly the kind of "pretend this is a real integration" shortcut this
project's whole hallucination-prevention philosophy exists to avoid - see
product_validator.py's module docstring), this adapter is a REAL,
WORKING integration point that:
- does nothing (returns [], `available=False`) when no credentials are
configured, so the cascade cleanly falls through to Open Food Facts
(Tier 2) - exactly the "if not found -> next step" behaviour the
spec asks for, just starting from Tier 2 in practice until GS1 India
API access is provisioned;
- immediately becomes live the moment `GS1_INDIA_API_BASE_URL` and
`GS1_INDIA_API_KEY` are set (see .env.example), with no code changes
needed elsewhere - `service.py` already treats every tier uniformly.
If/when real GS1 India API access is provisioned, fill in the request
shape in `search()` below to match that API's actual contract (endpoint
path, auth header, response schema) - the surrounding plumbing
(validation, matching, caching, retry) does not need to change.
"""
from __future__ import annotations
import logging
from typing import List
import requests
from app.infrastructure.settings import GS1_INDIA_API_BASE_URL, GS1_INDIA_API_KEY, BARCODE_LOOKUP_TIMEOUT_SECONDS
from app.services.enrichment.barcode.models import BarcodeCandidate
from app.services.enrichment.barcode.retry import with_retry
from app.services.enrichment.barcode.sources.base import BarcodeSource
logger = logging.getLogger(__name__)
class GS1IndiaSource(BarcodeSource):
name = "GS1 India"
tier = 1
@property
def available(self) -> bool:
return bool(GS1_INDIA_API_BASE_URL and GS1_INDIA_API_KEY)
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
if not self.available:
return []
@with_retry(max_attempts=3)
def _call():
return requests.get(
f"{GS1_INDIA_API_BASE_URL.rstrip('/')}/search",
params={"brand": brand, "product": product_title, "size": size, "country": "India"},
headers={"Authorization": f"Bearer {GS1_INDIA_API_KEY}"},
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
)
try:
resp = _call()
if resp.status_code != 200:
logger.debug(f"GS1 India lookup non-200 ({resp.status_code}) for '{brand} {product_title} {size}'")
return []
data = resp.json()
except Exception as e:
logger.debug(f"GS1 India lookup failed for '{brand} {product_title} {size}': {e}")
return []
candidates: List[BarcodeCandidate] = []
for item in (data.get("results") or data.get("items") or []):
gtin = item.get("gtin") or item.get("barcode") or item.get("code")
if not gtin:
continue
candidates.append(BarcodeCandidate(
barcode=str(gtin),
source_name=self.name,
candidate_title=item.get("productName") or item.get("title") or "",
candidate_brand=item.get("brandName") or item.get("brand") or "",
candidate_size=item.get("netContent") or item.get("size") or "",
candidate_countries="in",
))
return candidates

View File

@@ -0,0 +1,108 @@
"""
Tier 4: manufacturer / official product page - last resort.
Follows the exact same "DuckDuckGo search -> read structure straight off
the result, no page-scraping HTML parser" spirit as
`app/services/sku_service.py`'s `find_website_product_id()`, except here
we DO need to fetch the page text (a barcode isn't embedded in the result
URL the way an Amazon ASIN is), so this is the one tier that performs a
capped number of lightweight page fetches, each wrapped in its own
try/except so one slow/broken page can never block the others.
This tier is intentionally the lowest-trust one: a number that merely sits
near the word "barcode"/"EAN"/"UPC"/"GTIN" on a webpage is only a
CANDIDATE. It still has to pass `validators.validate_barcode()` (checksum)
and `matching.is_match()` (brand/size/name) in `service.py` exactly like
every other tier's candidates - nothing here is trusted just because it
came from what looks like an official page.
"""
from __future__ import annotations
import logging
import re
from typing import List
import requests
from app.infrastructure.settings import BARCODE_LOOKUP_TIMEOUT_SECONDS, ENABLE_MANUFACTURER_SITE_LOOKUP
from app.services.enrichment.barcode.models import BarcodeCandidate
from app.services.enrichment.barcode.retry import with_retry
from app.services.enrichment.barcode.sources.base import BarcodeSource
logger = logging.getLogger(__name__)
_BROWSER_UA = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
)
# A barcode label followed, within a short distance, by an 8/12/13/14-digit
# run. Intentionally permissive on the label (source pages phrase this
# differently) and tight on the digit run (only plausible GTIN lengths) -
# every hit is still just a CANDIDATE, checksum-validated afterwards.
_BARCODE_NEAR_LABEL_RE = re.compile(
r"(?:barcode|ean|upc|gtin)\D{0,15}(\d{8}|\d{12,14})",
re.IGNORECASE,
)
_MAX_PAGES = 3
class ManufacturerSiteSource(BarcodeSource):
name = "Manufacturer Website"
tier = 4
@property
def available(self) -> bool:
if not ENABLE_MANUFACTURER_SITE_LOOKUP:
return False
try:
import ddgs # noqa: F401
return True
except ImportError:
logger.debug("Manufacturer-site barcode lookup unavailable: 'ddgs' package not installed")
return False
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
if not self.available:
return []
query = f"{brand} {product_title} {size} barcode EAN".strip()
try:
from ddgs import DDGS
with DDGS(timeout=10) as ddgs:
results = ddgs.text(query, region="in-en", safesearch="off", max_results=_MAX_PAGES)
except Exception as e:
logger.debug(f"Manufacturer-site search failed for '{query}': {e}")
return []
candidates: List[BarcodeCandidate] = []
for r in (results or [])[:_MAX_PAGES]:
url = str(r.get("href") or r.get("url") or "")
if not url.startswith("http"):
continue
for code in self._extract_barcodes(url):
candidates.append(BarcodeCandidate(
barcode=code,
source_name=f"{self.name} ({url})",
candidate_title=r.get("title") or product_title,
candidate_brand=brand,
candidate_size=size,
candidate_countries="",
))
return candidates
def _extract_barcodes(self, url: str) -> List[str]:
@with_retry(max_attempts=2)
def _call():
return requests.get(url, headers={"User-Agent": _BROWSER_UA}, timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS)
try:
resp = _call()
if resp.status_code != 200:
return []
text = resp.text[:200_000] # cap - this is a text scan, not a full-page render
except Exception as e:
logger.debug(f"Manufacturer page fetch failed for '{url}': {e}")
return []
return [m.group(1) for m in _BARCODE_NEAR_LABEL_RE.finditer(text)]

View File

@@ -0,0 +1,118 @@
"""
Tier 2: Open Food Facts / Open Beauty Facts / Open Products Facts.
Real, free, no-API-key public search API - the same family of open,
community-maintained product databases already used as the project's
PRIMARY image source (see `app/services/image_search.py`'s
`OPEN_FACTS_HOSTS` / `_query_openfacts`). Every entry is keyed by its real
barcode (the `code` field IS the GTIN/EAN/UPC printed on the physical
pack), which is exactly the ground-truth this module needs - unlike
`image_search.py`, which only reads the image URLs off each entry, this
adapter reads the `code` field itself.
Deliberately a separate, self-contained query function rather than
importing `image_search._query_openfacts` (which is a private, underscore-
prefixed helper): this adapter needs different response fields (`code`,
`countries_tags`) and India-specific filtering that the image-search
helper has no reason to carry. `quantities_match`/`parse_quantity_grams`
ARE imported from `image_search.py` (public functions) for size
comparison, so the size-matching logic itself is not duplicated - see
`matching.py`.
"""
from __future__ import annotations
import logging
from typing import List
import requests
from app.infrastructure.settings import BARCODE_LOOKUP_TIMEOUT_SECONDS, BARCODE_COUNTRY_TAG
from app.services.enrichment.barcode.models import BarcodeCandidate
from app.services.enrichment.barcode.retry import with_retry
from app.services.enrichment.barcode.sources.base import BarcodeSource
logger = logging.getLogger(__name__)
_HOSTS = [
"world.openfoodfacts.org",
"world.openbeautyfacts.org",
"world.openproductsfacts.org",
]
_BROWSER_UA = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
)
_FIELDS = "code,product_name,brands,brands_tags,quantity,countries_tags"
class OpenFoodFactsSource(BarcodeSource):
name = "Open Food Facts"
tier = 2
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
query = f"{brand or ''} {product_title or ''}".strip()
if not query:
return []
products = self._query(query)
if not products and brand and product_title:
# Combined query too specific (common for regional brand-name
# variants) - retry title-only, same fallback image_search.py uses.
products = self._query(product_title)
candidates: List[BarcodeCandidate] = []
for item in products:
code = str(item.get("code") or "").strip()
if not code:
continue
countries = ",".join(item.get("countries_tags") or [])
candidates.append(BarcodeCandidate(
barcode=code,
source_name=self.name,
candidate_title=item.get("product_name") or "",
candidate_brand=item.get("brands") or "",
candidate_size=item.get("quantity") or "",
candidate_countries=countries,
))
# Soft India-market prioritisation: candidates whose countries_tags
# mention the target market are tried first, but non-tagged/other-
# market candidates are kept (not dropped) since many genuine
# Indian FMCG entries simply have this field blank upstream - the
# brand/size/name gate in matching.py is what actually decides
# correctness, this only affects which validated match is found
# (and therefore stops the cascade) first.
if BARCODE_COUNTRY_TAG:
tag = BARCODE_COUNTRY_TAG.lower()
candidates.sort(key=lambda c: 0 if tag in c.candidate_countries.lower() else 1)
return candidates
def _query(self, query: str) -> list:
for host in _HOSTS:
@with_retry(max_attempts=2)
def _call(host=host):
return requests.get(
f"https://{host}/cgi/search.pl",
params={
"search_terms": query,
"search_simple": 1,
"action": "process",
"json": 1,
"page_size": 20,
"fields": _FIELDS,
},
headers={"User-Agent": _BROWSER_UA},
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
)
try:
resp = _call()
if resp.status_code != 200:
continue
products = resp.json().get("products", [])
if products:
return products
except Exception as e:
logger.debug(f"Open*Facts barcode lookup failed on {host} for '{query}': {e}")
continue
return []

View File

@@ -0,0 +1,81 @@
"""
Tier 3: "trusted barcode databases" - UPCItemDB.
Real, free (trial-tier, no API key required, rate-limited) reverse lookup:
search by product name/brand keywords and get back candidate items with
their own `upc`/`ean` fields. This is the general "trusted barcode
database" tier the spec asks for as a fallback below GS1 India / Open
Food Facts.
RATE LIMIT: UPCItemDB's free trial endpoint is capped (documented as
~100 requests/day, ~1 request/second) - this is exactly why this tier
sits BELOW Open Food Facts (unlimited, no key) in the cascade, and why
`service.py` only calls a lower tier at all when every higher tier has
already failed to produce a validated match, plus why the local cache
(`cache.py`) matters most for this specific source. If a paid UPCItemDB
key is available, set `UPC_DATABASE_API_KEY` (see .env.example) to switch
to the production endpoint with a higher quota - the request shape below
already supports both.
"""
from __future__ import annotations
import logging
from typing import List
import requests
from app.infrastructure.settings import BARCODE_LOOKUP_TIMEOUT_SECONDS, UPC_DATABASE_API_KEY
from app.services.enrichment.barcode.models import BarcodeCandidate
from app.services.enrichment.barcode.retry import with_retry
from app.services.enrichment.barcode.sources.base import BarcodeSource
logger = logging.getLogger(__name__)
_TRIAL_URL = "https://api.upcitemdb.com/prod/trial/search"
_PROD_URL = "https://api.upcitemdb.com/prod/v1/search"
class UPCDatabaseSource(BarcodeSource):
name = "UPCItemDB"
tier = 3
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
query = f"{brand or ''} {product_title or ''} {size or ''}".strip()
if not query:
return []
url = _PROD_URL if UPC_DATABASE_API_KEY else _TRIAL_URL
headers = {"Accept": "application/json"}
if UPC_DATABASE_API_KEY:
headers["user_key"] = UPC_DATABASE_API_KEY
headers["key_type"] = "3scale"
@with_retry(max_attempts=2)
def _call():
return requests.get(url, params={"s": query, "type": "product"}, headers=headers,
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS)
try:
resp = _call()
if resp.status_code != 200:
logger.debug(f"UPCItemDB non-200 ({resp.status_code}) for '{query}'")
return []
data = resp.json()
except Exception as e:
logger.debug(f"UPCItemDB lookup failed for '{query}': {e}")
return []
candidates: List[BarcodeCandidate] = []
for item in data.get("items", []):
code = item.get("ean") or item.get("upc")
if not code:
continue
candidates.append(BarcodeCandidate(
barcode=str(code),
source_name=self.name,
candidate_title=item.get("title") or "",
candidate_brand=item.get("brand") or "",
candidate_size=item.get("size") or "",
candidate_countries="",
))
return candidates

View File

@@ -0,0 +1,37 @@
"""Adapts BarcodeLookupService to the EnrichmentStage contract so it can be
registered in app/services/enrichment/pipeline.py's default pipeline."""
from __future__ import annotations
import logging
from typing import Any, Dict
from app.infrastructure.settings import ENABLE_BARCODE_LOOKUP
from app.services.enrichment.base import EnrichmentStage, StageOutcome
from app.services.enrichment.barcode.service import get_default_service
logger = logging.getLogger(__name__)
class BarcodeEnrichmentStage(EnrichmentStage):
name = "barcode_lookup"
@property
def enabled(self) -> bool:
return ENABLE_BARCODE_LOOKUP
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
service = get_default_service()
title = product.get("title") or product.get("product_name") or ""
size = product.get("size") or ""
category = product.get("category") or ""
try:
import asyncio
result = await asyncio.to_thread(service.lookup_one, brand, title, size, category)
except Exception as e:
logger.warning(f"Barcode enrichment failed for '{title}' {size}: {e}")
from app.services.enrichment.barcode.models import BarcodeResult, LookupStatus
result = BarcodeResult.null_result(LookupStatus.ERROR)
return StageOutcome(stage_name=self.name, fields=result.as_product_fields(),
error=None if result.barcode_verified else result.barcode_lookup_status)

View File

@@ -0,0 +1,99 @@
"""
Barcode format + checksum validation.
Deliberately the LAST gate a candidate passes through before it's allowed
into a `BarcodeResult` (see `service.py`), regardless of which source
produced it or how much we otherwise trust that source - a source telling
us "this is the barcode" is never sufficient on its own; the digits have to
actually check out mathematically. This is what "Validate EAN-13 checksum"
/ "Validate GTIN format" / "Reject invalid barcode lengths" / "Reject
malformed barcode values" (Barcode Validation requirements) means in code.
No third-party dependency - GTIN/EAN/UPC-A all share one checksum
algorithm (the classic "alternating 3/1 weights counted from the rightmost
digit before the check digit"), so GTIN-8/12/13/14 are all validated by the
same function; only the accepted LENGTH differs per barcode type.
"""
from __future__ import annotations
import re
from typing import Optional
from app.services.enrichment.barcode.models import BarcodeType
_VALID_LENGTHS = {8: BarcodeType.GTIN8, 12: BarcodeType.UPC_A, 13: BarcodeType.EAN13, 14: BarcodeType.GTIN14}
# Digits only, no separators - callers are expected to call normalize_barcode()
# first if the raw source string may contain spaces/hyphens.
_DIGITS_ONLY_RE = re.compile(r"^\d+$")
def normalize_barcode(raw: Optional[str]) -> Optional[str]:
"""Strip everything but digits (spaces, hyphens, a stray 'EAN:' label,
etc.). Returns None for empty/unusable input - never raises."""
if not raw:
return None
digits = re.sub(r"\D", "", str(raw))
return digits or None
def gtin_check_digit(digits_without_check: str) -> int:
"""Compute the correct check digit for a GTIN-8/12/13/14 payload
(i.e. every digit EXCEPT the check digit itself), using the standard
alternating 3/1 weighting counted from the rightmost digit."""
total = 0
for i, ch in enumerate(reversed(digits_without_check)):
weight = 3 if i % 2 == 0 else 1
total += int(ch) * weight
return (10 - (total % 10)) % 10
def has_valid_checksum(code: str) -> bool:
"""True if `code`'s own last digit matches the checksum computed over
the rest of it. `code` must already be digits-only."""
if not code or not _DIGITS_ONLY_RE.match(code):
return False
body, check_digit = code[:-1], code[-1]
try:
expected = gtin_check_digit(body)
except (ValueError, IndexError):
return False
return str(expected) == check_digit
def classify_barcode_type(code: str) -> BarcodeType:
"""Barcode type purely from its (already checksum-validated) length.
UPC-A (12 digits) is the one ambiguous case worth calling out: it is
numerically a GTIN-13 with a leading zero, but is reported as UPC-A
here since that's the label the requesting spec/UI expects for a
12-digit code."""
return _VALID_LENGTHS.get(len(code), BarcodeType.UNKNOWN)
def validate_barcode(raw: Optional[str]) -> Optional[str]:
"""Single entry point: normalize, check length, check checksum.
Returns the clean digits-only barcode string if and only if it is a
genuinely valid GTIN-8/12/13/14, else None. This is the ONLY function
other modules should call to decide "is this barcode good enough to
store" - never inline a length/regex check elsewhere.
"""
code = normalize_barcode(raw)
if not code:
return None
if len(code) not in _VALID_LENGTHS:
return None
if not has_valid_checksum(code):
return None
return code
def to_ean13(code: str) -> Optional[str]:
"""Zero-pad a valid UPC-A (12 digits) up to its equivalent EAN-13
representation. GTIN-8 is intentionally NOT padded (an 8-digit GTIN is
its own distinct symbology, not a truncated EAN-13) - returns None for
anything that isn't 12 or 13 digits already."""
if len(code) == 13:
return code
if len(code) == 12:
return "0" + code
return None

View File

@@ -0,0 +1,80 @@
"""
Base contract every enrichment stage implements.
A stage takes ONE already-validated catalog row (a plain dict, the same
`enhanced_product` shape `catalog_engine.py` builds) plus the brand name,
and returns the SAME dict with additional fields merged in - it never
removes or renames a key it didn't add itself, and it never raises: any
internal failure is caught and reported via `StageOutcome.error` instead.
To add a new enrichment stage later (HSN, GST, nutrition, allergens, ...):
1. Subclass `EnrichmentStage`.
2. Implement `async def enrich_one(product, brand) -> StageOutcome`.
3. Register an instance in `pipeline.run_default_pipeline()` (or build
a custom `EnrichmentPipeline([...])` for a one-off run).
No other file needs to change - `catalog_engine.py`'s call site and
`vector_store.py`'s upsert already iterate whatever keys are present on the
product dict via `.get(...)`, so a new stage's fields flow through to the
database/JSON export automatically the same way barcode fields do.
"""
from __future__ import annotations
import logging
from abc import ABC, abstractmethod
from dataclasses import dataclass, field
from typing import Any, Dict, Optional
logger = logging.getLogger(__name__)
@dataclass
class StageOutcome:
"""Result of running one stage on one product row."""
stage_name: str
fields: Dict[str, Any] = field(default_factory=dict)
error: Optional[str] = None
@property
def ok(self) -> bool:
return self.error is None
class EnrichmentStage(ABC):
"""One independent, pluggable enrichment step.
`name` is used for logging/metrics only. `enabled` lets a stage report
itself as switched off (e.g. via a settings flag) without the pipeline
orchestrator needing to know why - it's simply skipped and every
product passes through untouched.
"""
name: str = "unnamed_stage"
@property
def enabled(self) -> bool:
return True
@abstractmethod
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
"""Enrich a single product row. MUST NOT raise - catch internally
and return a StageOutcome with `error` set instead."""
raise NotImplementedError
async def apply(self, product: Dict[str, Any], brand: str) -> Dict[str, Any]:
"""Run this stage on `product` and merge the result in place.
Never raises - a stage bug degrades to "no fields added", logged,
rather than aborting the whole catalog row.
"""
try:
outcome = await self.enrich_one(product, brand)
except Exception as e: # last-resort safety net - stages should
# already catch their own errors, but a pipeline-wide guarantee
# of "never raises" is worth the redundancy here.
logger.error(f"[{self.name}] unhandled exception enriching '{product.get('product_name')}': {e}")
return product
if outcome.fields:
product.update(outcome.fields)
if not outcome.ok:
logger.debug(f"[{self.name}] {product.get('product_name')}: {outcome.error}")
return product

View File

@@ -0,0 +1,32 @@
"""
HSN / GST & Pricing enrichment module.
Public entry points:
resolve_hsn_gst(category, product_title="") -> HsnGstInfo
Deterministic, offline category -> (HSN code, GST %, review flag)
resolution, mirroring the shape already present in the coca-cola
and milky_mist catalog exports (hsn_code / gst_percent /
hsn_gst_needs_review / selling_price / tax_amount /
final_selling_price / cost_price / profit_before_tax /
profit_after_tax).
enrich_pricing_fields(product) -> dict
Computes selling_price / tax_amount / final_selling_price (plus the
always-null cost/profit fields) from a catalog row's price_range,
using the same convention as the existing enriched brands: the
selling price is the upper bound of the row's retail price range.
See the package's module docstrings for the architecture:
models.py - HsnGstInfo result type + curated category -> HSN/GST table
stage.py - adapts the resolver to the generic EnrichmentStage
contract used by app/services/enrichment/pipeline.py
"""
from app.services.enrichment.hsn_gst.models import HsnGstInfo, resolve_hsn_gst, enrich_pricing_fields
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
__all__ = [
"HsnGstInfo",
"resolve_hsn_gst",
"enrich_pricing_fields",
"HsnGstEnrichmentStage",
]

View File

@@ -0,0 +1,259 @@
"""
Curated, deterministic HSN / GST lookup for catalog product categories.
WHY THIS IS NOT "MADE UP"
--------------------------
HSN (Harmonized System of Nomenclature) is the 4-8 digit goods-classification
code used on every Indian tax invoice, and GST% is the Indian Goods &
Services Tax rate applied to that HSN chapter. Both are public, legal,
category-level facts - not per-SKU secrets. Every entry in `HSN_GST_TABLE`
below is a *typical* classification for the product category as sold at
retail (matching the values already present in the coca-cola / milky_mist
catalog exports where those overlap, e.g. Beverages -> 2009, Dairy -> 0401,
Dairy - Desserts -> 2105, Tea & Coffee -> 0902).
The catch: an HSN chapter can cover several GST rates depending on the exact
item and how it is packaged, so a category-level guess can be wrong for a
specific product. That is precisely what `HSN_GST_NEEDS_REVIEW` is for - it is
True whenever the mapping is approximate (the default), and False only for the
few categories whose retail classification is unambiguous. The flag exists so
a human reviewer / the Streamlit UI can spot-check exactly these rows.
DESIGN PRINCIPLES
------------------
- DETERMINISTIC & OFFLINE. No LLM, no network, no randomness. The same
category always yields the same (HSN, GST%, review) tuple, so re-runs and
backfills are stable and diffable.
- ADDITIVE. Only ever attaches new keys to a product row; never removes or
rewrites an existing key (the stage skips rows that already carry these
fields, so re-running the pipeline over an already-enriched catalog is a
no-op).
- SAFE DEFAULTS. Unknown/unrecognised categories get hsn_code=None and
gst_percent=None with needs_review=True, rather than a fabricated code -
a wrong HSN on a real tax document is worse than no HSN at all.
"""
from __future__ import annotations
import re
from dataclasses import dataclass
from typing import Dict, Optional, Tuple
# (HSN code, GST %, needs_review). Review=True whenever the category can map
# to more than one GST rate / HSN chapter at retail; False for unambiguous ones.
HSN_GST_TABLE: Dict[str, Tuple[str, int, bool]] = {
# ---- Food & beverage (matches the existing coca-cola / milky_mist exports) ----
"Dairy": ("0401", 5, False),
"Cheese": ("0406", 12, True),
"Dairy - Desserts": ("2105", 5, False),
"Ice Cream": ("2105", 18, True),
"Beverages": ("2009", 5, False),
"Tea & Coffee": ("0902", 5, False),
"Food & Beverages": ("2106", 18, True),
"Health Drinks": ("2202", 18, True),
"Health Foods": ("2106", 18, True),
"Breakfast Cereal": ("1904", 18, False),
"Chocolates": ("1806", 18, False),
"Candy & Confectionery": ("1704", 18, False),
"Biscuits & Cookies": ("1905", 18, True),
"Crackers": ("1905", 18, True),
"Rusk": ("1905", 5, True),
"Cakes & Muffins": ("1905", 18, True),
"Bakery & Breads": ("1905", 5, False),
"Snacks": ("1905", 18, True),
"Noodles & Instant Food": ("1902", 18, False),
"Pasta & Noodles": ("1902", 18, False),
"Atta & Staples": ("1101", 5, False),
"Salt & Staples": ("2501", 5, False),
"Pulses, Grains & Spices": ("0713", 5, True),
"Spices & Masalas": ("0910", 5, False),
"Cooking Oils": ("1517", 5, False),
"Pickles & Chutneys": ("2001", 12, True),
"Dry Fruits & Nuts": ("0801", 12, True),
"Food - Spreads": ("2007", 12, True),
"Food - Mixes": ("2106", 18, True),
"Food - Soups & Sauces": ("2103", 12, False),
# ---- Personal care & household ----
"Hair Care": ("3305", 18, False),
"Skin Care": ("3304", 18, False),
"Skin & Bath Care": ("3307", 18, False),
"Bath Soap": ("3401", 18, False),
"Beauty Care": ("3304", 18, False),
"Oral Care": ("3306", 18, False),
"Fragrance & Deodorants": ("3303", 18, False),
"Men's Grooming": ("8212", 18, True),
"Baby Care": ("3304", 18, True),
"Feminine Hygiene": ("9619", 12, False),
"Personal Care": ("3307", 18, True),
"Detergents & Fabric Care": ("3402", 18, False),
"Dishwash": ("3402", 18, False),
"Household Cleaning": ("3402", 18, True),
"Household - Air Freshener": ("3307", 18, True),
"Personal Care - Mosquito Repellent": ("3808", 18, True),
# ---- Health care ----
"Health Care - Cold & Cough": ("3004", 12, True),
"Health Care - Antiseptic": ("3808", 18, True),
"Health Care - Ayurvedic": ("3003", 12, True),
"Health Care - Digestive": ("3004", 12, True),
"Health Care - First Aid": ("3005", 12, True),
}
# Keyword fallbacks so a product whose category label is slightly different
# from the table (or missing entirely) still resolves to a sensible chapter.
_KEYWORD_FALLBACKS: Tuple[Tuple[str, str, int, bool], ...] = (
("toothpaste", "3306", 18, False),
("shampoo", "3305", 18, False),
("soap", "3401", 18, False),
("detergent", "3402", 18, False),
("biscuit", "1905", 18, True),
("cookie", "1905", 18, True),
("chocolate", "1806", 18, False),
("candy", "1704", 18, False),
("toffee", "1704", 18, False),
("milk", "0401", 5, False),
("paneer", "0401", 5, False),
("curd", "0401", 5, False),
("cheese", "0406", 12, True),
("ice cream", "2105", 18, True),
("juice", "2009", 5, False),
("tea", "0902", 5, False),
("coffee", "0902", 5, False),
("health drink", "2202", 18, True),
("chips", "1905", 18, True),
("namkeen", "1905", 18, True),
("bread", "1905", 5, False),
("atta", "1101", 5, False),
("flour", "1101", 5, False),
("masala", "0910", 5, False),
("spice", "0910", 5, False),
("pasta", "1902", 18, False),
("noodle", "1902", 18, False),
("cooking oil", "1517", 5, False),
("oil", "1517", 5, True),
("pickle", "2001", 12, True),
("jam", "2007", 12, True),
("honey", "0409", 5, False),
("raisin", "0806", 5, True),
("dry fruit", "0801", 12, True),
("cereal", "1904", 18, False),
("deodorant", "3303", 18, False),
("perfume", "3303", 18, False),
("lipstick", "3304", 18, False),
("makeup", "3304", 18, False),
("face wash", "3307", 18, False),
("lotion", "3307", 18, False),
("sunscreen", "3304", 18, False),
("razor", "8212", 18, True),
("diaper", "9619", 12, False),
("mosquito", "3808", 18, True),
("repellent", "3808", 18, True),
("air freshener", "3307", 18, True),
("dishwash", "3402", 18, False),
("floor cleaner", "3402", 18, True),
("hand wash", "3401", 18, False),
)
@dataclass(frozen=True)
class HsnGstInfo:
"""HSN / GST resolution for one product category, mirroring the fields
stored on enriched catalog rows (coca-cola / milky_mist exports)."""
hsn_code: Optional[str] = None
gst_percent: Optional[int] = None
hsn_gst_needs_review: bool = True
def as_fields(self) -> Dict[str, object]:
return {
"hsn_code": self.hsn_code,
"gst_percent": self.gst_percent,
"hsn_gst_needs_review": self.hsn_gst_needs_review,
}
_UNKNOWN = HsnGstInfo(hsn_code=None, gst_percent=None, hsn_gst_needs_review=True)
def resolve_hsn_gst(category: Optional[str], product_title: str = "") -> HsnGstInfo:
"""Resolve (HSN code, GST %, needs_review) for a product category.
Exact category matches in `HSN_GST_TABLE` win; otherwise the product
title/description is scanned against `_KEYWORD_FALLBACKS`; otherwise
returns the safe unknown shape (all-None, needs_review=True).
"""
cat = (category or "").strip()
if cat:
exact = HSN_GST_TABLE.get(cat)
if exact:
return HsnGstInfo(*exact)
haystack = f"{product_title or ''} {cat}".lower()
for kw, hsn, gst, review in _KEYWORD_FALLBACKS:
if kw in haystack:
return HsnGstInfo(hsn_code=hsn, gst_percent=gst, hsn_gst_needs_review=review)
return _UNKNOWN
# Matches a price range string of the form "₹12-14" (also tolerates
# "Rs 12-14", "12 - 14", a single "₹12", or corrupted rupee symbols).
_RANGE_RE = re.compile(r"₹?\s*(\d+(?:\.\d+)?)\s*[-–—]\s*(\d+(?:\.\d+)?)")
_SINGLE_RE = re.compile(r"₹?\s*(\d+(?:\.\d+)?)")
def _extract_selling_price(price_range: object) -> Optional[float]:
"""Return the upper bound of a row's price range (the convention used by
the existing coca-cola / milky_mist exports for `selling_price`), or the
single value when the row only carries one price."""
if price_range is None:
return None
text = str(price_range).replace(",", "").strip()
if not text:
return None
m = _RANGE_RE.search(text)
if m:
try:
return float(m.group(2))
except ValueError:
return None
m = _SINGLE_RE.search(text)
if m:
try:
return float(m.group(1))
except ValueError:
return None
return None
def _round2(value: Optional[float]) -> Optional[float]:
if value is None:
return None
return round(value, 2)
def enrich_pricing_fields(product: dict) -> Dict[str, object]:
"""Compute the pricing fields added by this stage for one catalog row.
Convention (matches the coca-cola / milky_mist exports):
selling_price = upper bound of the row's `price_range`
tax_amount = selling_price * gst_percent / 100
final_selling_price = selling_price + tax_amount
`cost_price`, `profit_before_tax` and `profit_after_tax` are not
derivable from the data this pipeline generates, so they stay None
(exactly as the existing enriched exports store them).
"""
selling_price = _extract_selling_price(product.get("price_range"))
gst = product.get("gst_percent")
tax_amount = None
final_selling_price = None
if selling_price is not None and isinstance(gst, (int, float)) and gst and gst > 0:
tax_amount = _round2(selling_price * float(gst) / 100.0)
final_selling_price = _round2(selling_price + tax_amount)
return {
"selling_price": _round2(selling_price),
"cost_price": None,
"tax_amount": tax_amount,
"final_selling_price": final_selling_price,
"profit_before_tax": None,
"profit_after_tax": None,
}

View File

@@ -0,0 +1,48 @@
"""Adapts the HSN/GST resolver to the EnrichmentStage contract so it can be
registered in app/services/enrichment/pipeline.py's default pipeline.
Unlike the barcode stage this needs no network access at all - HSN/GST are
deterministic, category-level facts - so it is synchronous and effectively
free. It is pure ADDITIVE: rows that already carry the hsn_code/gst_percent
keys (e.g. re-running the pipeline over a previously-enriched catalog) are
left untouched.
"""
from __future__ import annotations
import logging
from typing import Any, Dict
from app.infrastructure.settings import ENABLE_HSN_GST_ENRICHMENT
from app.services.enrichment.base import EnrichmentStage, StageOutcome
from app.services.enrichment.hsn_gst.models import enrich_pricing_fields, resolve_hsn_gst
logger = logging.getLogger(__name__)
class HsnGstEnrichmentStage(EnrichmentStage):
name = "hsn_gst"
@property
def enabled(self) -> bool:
return ENABLE_HSN_GST_ENRICHMENT
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
# Already enriched (previous run / manual backfill) - never overwrite.
if product.get("hsn_code") is not None or product.get("gst_percent") is not None:
return StageOutcome(stage_name=self.name, fields={})
title = product.get("title") or product.get("product_name") or ""
category = product.get("category") or ""
hsn_info = resolve_hsn_gst(category, title)
fields: Dict[str, Any] = hsn_info.as_fields()
# Pricing fields depend on gst_percent, which was just resolved
# above, so hand the resolved rate to the pricing helper.
fields.update(enrich_pricing_fields({**product, "gst_percent": fields["gst_percent"]}))
needs_review = fields["hsn_gst_needs_review"]
return StageOutcome(
stage_name=self.name,
fields=fields,
error=None if not needs_review else "category-level HSN/GST estimate - verify against the physical pack"
)

View File

@@ -0,0 +1,89 @@
"""
Orchestrates a list of independent `EnrichmentStage`s over a batch of
catalog rows, running each stage's per-row work concurrently (bounded by
`max_concurrency`) and NEVER letting one row's failure affect any other
row or stage.
`catalog_engine.py` calls `run_default_pipeline()` once, between the
deterministic-validation step (product_validator.py) and catalog assembly
- see that file's "Step 2.6" for the call site and
docs/BARCODE_ENRICHMENT.md for the full pipeline-position rationale.
"""
from __future__ import annotations
import asyncio
import logging
from typing import Any, Dict, List, Optional, Sequence
from app.services.enrichment.base import EnrichmentStage
logger = logging.getLogger(__name__)
class EnrichmentPipeline:
def __init__(self, stages: Sequence[EnrichmentStage], max_concurrency: int = 5):
self.stages = [s for s in stages if s.enabled]
self.max_concurrency = max(1, max_concurrency)
async def run(self, products: List[Dict[str, Any]], brand: str) -> List[Dict[str, Any]]:
if not self.stages or not products:
return products
semaphore = asyncio.Semaphore(self.max_concurrency)
async def _run_row(product: Dict[str, Any]) -> Dict[str, Any]:
async with semaphore:
for stage in self.stages:
product = await stage.apply(product, brand)
return product
results = await asyncio.gather(*(_run_row(p) for p in products), return_exceptions=True)
final: List[Dict[str, Any]] = []
for original, result in zip(products, results):
if isinstance(result, Exception):
logger.error(f"Enrichment pipeline failed for '{original.get('product_name')}': {result}")
final.append(original)
else:
final.append(result)
return final
def _build_default_stages() -> List[EnrichmentStage]:
"""Import stages lazily so importing this module never pulls in a
stage's own dependencies (network clients, DB drivers, ...) unless a
default pipeline is actually requested."""
stages: List[EnrichmentStage] = []
try:
from app.services.enrichment.barcode.stage import BarcodeEnrichmentStage
stages.append(BarcodeEnrichmentStage())
except Exception as e:
logger.error(f"Barcode enrichment stage unavailable: {e}")
# HSN / GST & pricing enrichment (see app/services/enrichment/hsn_gst/) -
# deterministic, offline, pure-additive. Runs AFTER the barcode stage so
# every stored/exported row carries both sets of fields; a failure here
# degrades a row to "no HSN/GST attached", never drops the row.
try:
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
stages.append(HsnGstEnrichmentStage())
except Exception as e:
logger.error(f"HSN/GST enrichment stage unavailable: {e}")
# Future stages register here, e.g.:
# from app.services.enrichment.nutrition.stage import NutritionEnrichmentStage
# stages.append(NutritionEnrichmentStage())
return stages
_default_pipeline: Optional[EnrichmentPipeline] = None
async def run_default_pipeline(products: List[Dict[str, Any]], brand: str,
max_concurrency: int = 5) -> List[Dict[str, Any]]:
"""Convenience entry point used by catalog_engine.py: runs every
currently-registered, enabled enrichment stage over `products`."""
global _default_pipeline
if _default_pipeline is None:
_default_pipeline = EnrichmentPipeline(_build_default_stages(), max_concurrency=max_concurrency)
return await _default_pipeline.run(products, brand)

View File

@@ -353,3 +353,32 @@ def parse_price_string(text: str) -> Optional[float]:
if len(numbers) >= 2:
return (numbers[0] + numbers[1]) / 2.0
return numbers[0]
def estimate_price_range_for_size(
size: str,
product_title: str = "",
brand: str = "",
category_hint: str = "",
) -> Tuple[int, int]:
"""Return a (min, max) price range for a SINGLE pack size.
Ported from the sibling Universal_Catalog_Barcode_Enrichment project, where
`product_validator.validate_price_range` uses it as the anchor to judge
whether a scraped price is plausible for the pack size. Additive: no
existing function in this module changes.
The range is centred on `estimate_price` with a deterministic +-5%-15%
jitter keyed on product+size, so the same product gets the same range on
every run while still representing genuine retail spread rather than a
single fixed MRP.
"""
price = estimate_price(size, product_title, brand, category_hint)
seed = f"{brand}|{category_hint}|{product_title}|{size}|range"
jitter = _deterministic_unit(seed)
spread = 0.05 + jitter * 0.10
lo = max(1, int(round(price * (1 - spread))))
hi = int(round(price * (1 + spread)))
if hi <= lo:
hi = lo + 1
return (lo, hi)

View File

@@ -0,0 +1,445 @@
"""
Deterministic Product Validation & Confidence Scoring
=======================================================
WHY THIS FILE EXISTS
---------------------
Every stage upstream of this module (ollama_service.py, title_validator.py,
price_estimator.py, sku_service.py) already applies point-fixes for
*specific, previously-reported* hallucination patterns (category-inconsistent
titles, unrealistic prices, non-existent brands, malformed JSON, ...). None
of them, however, ever made a final, holistic pass over a fully-assembled
catalog row and asked "taken together, does this look like a REAL product
worth keeping?" That gap is why individually-plausible-looking rows with
compounding small defects (a slightly-off price + a missing image + a vague
size) could still reach the database.
This module is that final gate. It is intentionally:
- DETERMINISTIC: no LLM call, no randomness. The exact same product dict
always produces the exact same report. That is what makes this a
hallucination *filter* rather than another hallucination *source*.
- CONSERVATIVE: every check only fires on strong, explainable evidence
(reuses title_validator's category-conflict detector, price_estimator's
category-anchored price bands, and sku_service's own SKU format) rather
than guessing. False rejections are treated as seriously as false
acceptances - see docs/VALIDATION_PIPELINE.md for the tuning rationale.
- ADDITIVE: this module has zero dependents before this change and does not
modify any existing function's behaviour; it is only ever imported and
called from new code (see app/core/catalog_engine.py's call site).
USAGE
-----
Call `validate_product(product_dict, brand, ...)` once per fully-assembled
catalog row, AFTER pricing/SKU/image resolution but BEFORE the row is
appended to the catalog output / sent to upsert_brand_products(). The
returned ValidationReport carries a 0.0-1.0 confidence score and a
status of "verified" / "needs_review" / "rejected" that the caller uses to
decide whether to keep, flag, or drop the row - see
ProductCatalogEngine.generate_catalog() in catalog_engine.py.
"""
from __future__ import annotations
import logging
import re
from typing import Any, Optional
from pydantic import BaseModel, Field
from app.services.title_validator import find_category_conflicts
from app.services import price_estimator
from app.services import category_units as cu
logger = logging.getLogger(__name__)
# ---------------------------------------------------------------------------
# Categories that are never legitimate for any FMCG / personal-care brand.
# Deliberately duplicated (rather than imported) from
# app.services.ollama_service._IRRELEVANT_CATEGORIES: ollama_service already
# imports from brand_registry, and this module is imported BY
# catalog_engine.py alongside ollama_service, so importing a private
# (underscore-prefixed) name across modules would create a fragile coupling
# for a list that changes maybe once a year. Keep both lists in sync if you
# add a new one.
# ---------------------------------------------------------------------------
IRRELEVANT_CATEGORIES: set[str] = {
"pet supplies", "pet food", "pet care", "automotive", "auto parts",
"automotive parts", "car accessories", "motorcycle", "bike",
"electronics", "gadgets", "computers", "mobile phones",
"furniture", "home decor", "furnishings", "appliances",
"clothing", "footwear", "fashion", "jewelry", "accessories",
"office supplies", "stationery", "books", "toys", "games",
"sports equipment", "sports gear", "gardening", "garden supplies",
"hardware", "tools", "building materials",
"musical instruments", "party supplies", "craft supplies",
}
# Title strings that are structurally a placeholder/fabrication artefact
# rather than a real product name, independent of any specific brand.
_PLACEHOLDER_TITLE_PATTERNS: list[re.Pattern] = [
# Bare "Product 3" / "Brand Product 3", or any single leading token
# (typically the brand name) immediately followed by "Product N" - the
# exact shape of the hardcoded placeholder data in the unused
# agentic_pipeline.py scaffold (see that file's module docstring) and
# the generic fallback a hallucinating LLM sometimes produces when it
# runs out of real products to enumerate.
re.compile(r"^\s*(\S+\s+)?product\s*#?\s*\d+\s*$", re.IGNORECASE),
re.compile(r"^\s*(unknown|n/?a|unnamed|untitled|sample|placeholder|tbd|todo|test\s*product)\s*$", re.IGNORECASE),
re.compile(r"^https?://", re.IGNORECASE),
# "5 Star (Soft Drink Candy) - Soft Drink Candy" style category echo -
# same pattern catalog_engine.py already filters at discovery time; kept
# here too as defence-in-depth for rows that reach this stage some other
# way (e.g. a future direct-insert code path that skips catalog_engine).
re.compile(r"\(([^)]{2,40})\)\s*-\s*\1", re.IGNORECASE),
]
# Internal SKUs are always formatted BRAND-PRODUCT-SIZE-SEQ by
# sku_service.generate_internal_sku(), e.g. "AACHI-SAM-500-001".
_INTERNAL_SKU_RE = re.compile(r"^[A-Z0-9]{2,15}-[A-Z0-9]{2,15}-[A-Z0-9]{1,10}-\d{3,}$")
_PRICE_RANGE_RE = re.compile(r"₹\s*(\d+(?:\.\d+)?)\s*-\s*(\d+(?:\.\d+)?)")
# Plausible bounds for a single retail FMCG pack, in gram/ml equivalent.
# Below 1: parsing failure. Above 30000 (30kg): almost certainly a
# hallucinated bulk size no e-commerce FMCG listing would use (real bulk
# packs top out around 25kg rice/atta/detergent sacks).
_MIN_PACK_GRAMS = 1.0
_MAX_PACK_GRAMS = 30_000.0
class ValidationIssue(BaseModel):
"""A single, explainable reason a confidence score moved."""
field: str
severity: str = Field(description="'error' (strong evidence of a problem) or 'warning' (weaker signal)")
message: str
penalty: float = 0.0
class ValidationConfig(BaseModel):
"""Tunable thresholds - see docs/VALIDATION_PIPELINE.md for defaults
and how to change them via settings.py without editing this file."""
reject_below: float = 0.35
review_below: float = 0.70
class ValidationReport(BaseModel):
"""Result of validating one fully-assembled catalog row."""
product_name: str
status: str = "verified" # "verified" | "needs_review" | "rejected"
confidence: float = 1.0
grounded: bool = False
issues: list[ValidationIssue] = Field(default_factory=list)
@property
def is_rejected(self) -> bool:
return self.status == "rejected"
@property
def needs_review(self) -> bool:
return self.status == "needs_review"
def summary(self) -> str:
if not self.issues:
return f"{self.product_name}: OK (confidence={self.confidence:.2f})"
reasons = "; ".join(f"{i.field}: {i.message}" for i in self.issues)
return f"{self.product_name}: {self.status} (confidence={self.confidence:.2f}) - {reasons}"
def _looks_like_placeholder(title: str) -> bool:
if not title or not title.strip():
return True
return any(p.search(title.strip()) for p in _PLACEHOLDER_TITLE_PATTERNS)
def validate_size(size: str) -> tuple[bool, Optional[str]]:
"""Check that `size` parses to a plausible FMCG retail-pack quantity.
Reuses price_estimator's own size parser so "what counts as a valid
size" can never silently drift out of sync with what pricing actually
uses to compute cost.
"""
if not size or not size.strip():
return False, "size/pack is missing"
grams = price_estimator._parse_size_to_grams(size)
if grams <= 0:
return False, f"could not parse a usable quantity from size '{size}'"
if grams < _MIN_PACK_GRAMS or grams > _MAX_PACK_GRAMS:
return False, f"parsed quantity ({grams:g}) from size '{size}' is outside a plausible retail-pack range"
return True, None
def validate_price_range(
price_range: str, size: str, product_title: str, brand: str, category: str,
) -> tuple[bool, Optional[str]]:
"""Check that `price_range` is well-formed and not a gross outlier
against the category+size price anchor computed by price_estimator.
The tolerance band (0.3x-3x of the anchor) is deliberately wide - this
check exists to catch gross hallucinations (a ₹5,000 biscuit packet, a
₹2 detergent bottle), not to second-guess normal retail price spread,
which price_estimator's own range already accounts for.
"""
if not price_range or not price_range.strip():
return False, "price_range is missing"
m = _PRICE_RANGE_RE.search(price_range)
if not m:
return False, f"price_range '{price_range}' is not a well-formed ₹lo-hi string"
lo, hi = float(m.group(1)), float(m.group(2))
if lo <= 0 or hi <= 0 or hi < lo:
return False, f"price_range '{price_range}' has non-positive or inverted bounds"
try:
anchor_lo, anchor_hi = price_estimator.estimate_price_range_for_size(
size, product_title, brand, category
)
except Exception:
return True, None # anchor unavailable (e.g. blank size) - don't fail the row on that alone
if hi < anchor_lo * 0.3 or lo > anchor_hi * 3.0:
return False, (
f"price_range '{price_range}' is implausible for a {size or 'unspecified-size'} "
f"{category or 'product'} (expected roughly ₹{anchor_lo}-{anchor_hi})"
)
return True, None
def validate_sku(sku: str, sku_source: str) -> tuple[bool, Optional[str]]:
"""Check `product_sku` is present and, for internally-generated SKUs,
matches sku_service's own BRAND-PRODUCT-SIZE-SEQ format. Real
marketplace IDs (sku_source != "Internal") are trusted as-is since they
were found via an actual web lookup, not generated - their format
legitimately varies by platform (ASIN vs Flipkart PID vs ...).
"""
if not sku or not sku.strip():
return False, "product_sku is missing"
if sku_source and sku_source != "Internal":
return True, None
if not _INTERNAL_SKU_RE.match(sku.strip()):
return False, f"internal SKU '{sku}' does not match the expected BRAND-PRODUCT-SIZE-SEQ pattern"
return True, None
def validate_product(
product: dict[str, Any],
brand: str,
*,
category_resolved_deterministically: bool = False,
grounded: bool = False,
images_checked: bool = True,
config: Optional[ValidationConfig] = None,
) -> ValidationReport:
"""Validate one fully-assembled catalog row and return a confidence-scored report.
Parameters
----------
product:
The enhanced product dict as built by catalog_engine.py, expected to
contain (at minimum) title/product_name, category, size, price_range,
product_sku, sku_source, image_urls.
brand:
The brand this row is being catalogued under.
category_resolved_deterministically:
True when the category came from brand_registry's curated sub-brand
registry (ground truth) rather than a keyword heuristic - a positive
evidence signal, not a validity requirement.
grounded:
True when an independent, external signal corroborates this exact
product (e.g. an Open Food Facts catalog match, or the row already
existing in the database from a prior validated run). See
catalog_engine.py's call site for exactly what sets this.
config:
Optional threshold override; defaults to ValidationConfig() (which
itself defaults to the values in app.infrastructure.settings).
"""
cfg = config or ValidationConfig()
title = str(product.get("title") or product.get("product_name") or "").strip()
category = str(product.get("category") or "").strip()
size = str(product.get("size") or "").strip()
price_range = str(product.get("price_range") or "").strip()
sku = str(product.get("product_sku") or "").strip()
sku_source = str(product.get("sku_source") or "").strip()
images = product.get("image_urls") or []
report = ValidationReport(product_name=title or "(untitled)", grounded=grounded)
score = 0.55 # neutral baseline - moves up/down based on evidence below
# 1. Title sanity -----------------------------------------------------
if _looks_like_placeholder(title):
report.issues.append(ValidationIssue(
field="title", severity="error",
message=f"title '{title}' looks like a placeholder/fabricated entry, not a real product name",
penalty=0.50,
))
score -= 0.50
elif len(title) < 2:
report.issues.append(ValidationIssue(
field="title", severity="error",
message="title is too short to be a real product name", penalty=0.40,
))
score -= 0.40
# 2. Category sanity ----------------------------------------------------
if not category or category == "Uncategorized":
report.issues.append(ValidationIssue(
field="category", severity="warning",
message="category could not be resolved", penalty=0.10,
))
score -= 0.10
elif category.lower() in IRRELEVANT_CATEGORIES:
report.issues.append(ValidationIssue(
field="category", severity="error",
message=f"category '{category}' is not a valid FMCG/personal-care category", penalty=0.60,
))
score -= 0.60
# 3. Title <-> category consistency (reuses title_validator's own
# conflict detector so the two modules can never disagree about what
# counts as a contradiction) --------------------------------------
if category and category not in ("Uncategorized", "General"):
conflicts = find_category_conflicts(title, category)
if conflicts:
report.issues.append(ValidationIssue(
field="title", severity="error",
message=f"title contains word(s) inconsistent with category '{category}': {', '.join(conflicts)}",
penalty=0.35,
))
score -= 0.35
# 4. Size / pack --------------------------------------------------------
ok, msg = validate_size(size)
if not ok:
report.issues.append(ValidationIssue(field="size", severity="error", message=msg, penalty=0.25))
score -= 0.25
# 4b. Size unit <-> category compatibility (see app/services/
# category_units.py). This is a last-resort backstop: by the time a
# row reaches this final validator, its size should already have
# been through the generation-time gate (app/services/
# generation_verifier.py, called from ollama_service.py) AND
# catalog_engine.py's own dimension-unit defence-in-depth pass, both
# of which reject/repair a bad unit long before enrichment. This
# check exists purely to catch anything that reached validate_product
# some other way (e.g. a pre-existing DB row from before this fix, or
# a future caller that doesn't route through catalog_engine).
if size:
ok, msg = cu.validate_unit_for_category(size, category)
if not ok:
report.issues.append(ValidationIssue(field="size", severity="error", message=msg, penalty=0.15))
score -= 0.15
# 5. Price range ----------------------------------------------------------
ok, msg = validate_price_range(price_range, size, title, brand, category)
if not ok:
report.issues.append(ValidationIssue(field="price_range", severity="error", message=msg, penalty=0.30))
score -= 0.30
# 6. SKU --------------------------------------------------------------------
ok, msg = validate_sku(sku, sku_source)
if not ok:
report.issues.append(ValidationIssue(field="product_sku", severity="warning", message=msg, penalty=0.10))
score -= 0.10
# 7. Image presence -----------------------------------------------------
# Only meaningful if image search actually ran. When the operator disables
# it (the store-catalog pipeline's offline mode), an empty image list says
# nothing about whether the product is real, and penalising it here would
# reject every row in a perfectly good upload.
if not images and images_checked:
report.issues.append(ValidationIssue(
field="image_urls", severity="warning",
message="no images were found for this product", penalty=0.15,
))
score -= 0.15
# 8. Positive evidence --------------------------------------------------
if category_resolved_deterministically:
score += 0.15
if grounded:
score += 0.15
if images:
score += 0.05
score = max(0.0, min(1.0, score))
report.confidence = round(score, 3)
if score < cfg.reject_below:
report.status = "rejected"
elif score < cfg.review_below:
report.status = "needs_review"
else:
report.status = "verified"
return report
def validate_catalog(
products: list[dict[str, Any]],
brand: str,
*,
known_category_flags: Optional[list[bool]] = None,
grounded_flags: Optional[list[bool]] = None,
images_checked: bool = True,
config: Optional[ValidationConfig] = None,
) -> tuple[list[dict[str, Any]], list[dict[str, Any]], dict[str, Any]]:
"""Validate an entire list of assembled catalog rows in one pass.
Returns (kept_products, rejected_products, summary) where `kept_products`
includes both "verified" and "needs_review" rows (each annotated with
`validation_status` / `confidence_score` / `validation_issues` so a
human-in-the-loop reviewer or a downstream consumer can filter further),
and `rejected_products` holds the rows dropped from the catalog along
with why, for audit/QA purposes - see docs/VALIDATION_PIPELINE.md.
"""
cfg = config or ValidationConfig()
known_category_flags = known_category_flags or [False] * len(products)
grounded_flags = grounded_flags or [False] * len(products)
kept: list[dict[str, Any]] = []
rejected: list[dict[str, Any]] = []
status_counts = {"verified": 0, "needs_review": 0, "rejected": 0}
for i, p in enumerate(products):
det = known_category_flags[i] if i < len(known_category_flags) else False
grd = grounded_flags[i] if i < len(grounded_flags) else False
report = validate_product(
p, brand,
category_resolved_deterministically=det,
grounded=grd,
images_checked=images_checked,
config=cfg,
)
status_counts[report.status] = status_counts.get(report.status, 0) + 1
annotated = {
**p,
"validation_status": report.status,
"confidence_score": report.confidence,
"validation_issues": [i.message for i in report.issues],
}
if report.is_rejected:
logger.warning(f"🚫 Rejected (validation): {report.summary()}")
rejected.append(annotated)
else:
if report.needs_review:
logger.info(f"⚠️ Needs review: {report.summary()}")
else:
logger.info(f"✅ Verified: {report.product_name} (confidence={report.confidence:.2f})")
kept.append(annotated)
summary = {
"total_evaluated": len(products),
"verified": status_counts["verified"],
"needs_review": status_counts["needs_review"],
"rejected": status_counts["rejected"],
"reject_threshold": cfg.reject_below,
"review_threshold": cfg.review_below,
}
logger.info(
f"🔍 Validation summary for '{brand}': {summary['verified']} verified, "
f"{summary['needs_review']} needs_review, {summary['rejected']} rejected "
f"(of {summary['total_evaluated']} rows)"
)
return kept, rejected, summary

View File

@@ -0,0 +1,82 @@
"""
Pack-size / quantity parsing helpers.
Ported from the sibling `Universal_Catalog_Barcode_Enrichment` project, where
these lived inside `image_search.py`. They are a standalone module here on
purpose: the barcode matcher needs them, and this repo's `image_search.py` is
a working file that the store-catalog work is not supposed to touch. Extracting
rather than appending keeps that guarantee.
Deliberately self-contained (no import from price_estimator.py) - these are
only used for approximate "same pack size?" comparisons, never for pricing
math. Grams and millilitres are treated as interchangeable because this only
ever compares two size LABELS for a plausible match, not physics.
"""
from __future__ import annotations
import re
from typing import List, Optional
_QTY_RE = re.compile(
r"(\d+(?:[.,]\d+)?)\s*[-_]?\s*"
r"(kilograms?|kilos?|kgs?|grams?|gms?|litres?|liters?|ltrs?|lts?|mls?|ml|g|kg|l)\b"
)
_QTY_LITRE_UNITS = {"l", "lt", "lts", "ltr", "ltrs", "litre", "litres", "liter", "liters"}
_QTY_KG_UNITS = {"kg", "kgs", "kilo", "kilos", "kilogram", "kilograms"}
def parse_quantity_grams(text: Optional[str]) -> Optional[float]:
"""Parse the first weight/volume mentioned in `text` (e.g. '200 g',
'1kg', '500ml', a URL fragment like 'good-day-100-g') into a
normalised grams-or-millilitres-equivalent float. Returns None if no
recognisable quantity is found.
"""
if not text:
return None
match = _QTY_RE.search(str(text).lower())
if not match:
return None
try:
value = float(match.group(1).replace(",", "."))
except ValueError:
return None
unit = match.group(2)
if unit in _QTY_LITRE_UNITS or unit in _QTY_KG_UNITS:
value *= 1000
return value
def quantities_match(a: Optional[str], b: Optional[str], tolerance: float = 0.15) -> bool:
"""True if size labels `a` and `b` plausibly refer to the same pack size,
allowing for rounding/formatting differences between sources (e.g. '200 g'
vs '210g' vs '0.2 kg'). Used to match a requested pack-size variant against
Open Food Facts' real per-barcode `quantity` field.
"""
qa, qb = parse_quantity_grams(a), parse_quantity_grams(b)
if qa is None or qb is None or qa <= 0:
return False
return abs(qa - qb) / qa <= tolerance
def find_quantity_mentions(text: Optional[str]) -> List[float]:
"""Return every distinct weight/volume value mentioned anywhere in `text`
(e.g. a full image/product URL), normalised to a grams-equivalent float.
Unlike parse_quantity_grams() (first match only), this scans the whole
string - used to detect when a candidate image's URL names a DIFFERENT
sibling pack size than the one being searched for (e.g. a "...-500-g..."
URL turning up while resolving images for the 100g variant).
"""
if not text:
return []
out: List[float] = []
for m in _QTY_RE.finditer(str(text).lower()):
try:
value = float(m.group(1).replace(",", "."))
except ValueError:
continue
unit = m.group(2)
if unit in _QTY_LITRE_UNITS or unit in _QTY_KG_UNITS:
value *= 1000
out.append(value)
return out

275
app/services/sku_service.py Normal file
View File

@@ -0,0 +1,275 @@
"""
Product SKU / website product-ID resolution service.
For every catalog record (one exact brand + product/variety + pack size)
this module resolves a `product_sku` in two stages:
1. REAL MARKETPLACE ID (best-effort, network-dependent)
A DuckDuckGo web search (via the `ddgs` package - already a project
dependency, see app/services/image_search.py for the same pattern)
restricted to known Indian e-commerce domains. If a result URL is on
one of those domains, the real product ID is extracted directly from
the URL itself (Amazon ASIN, Flipkart `pid`, BigBasket/JioMart/
Blinkit/Nykaa numeric product IDs) - no page scraping/HTML parsing
required, so this is fast and doesn't trip anti-bot defenses.
2. INTERNAL SKU (deterministic fallback)
When no real product ID can be found (common - the LLM-discovered
product may not exist verbatim on any marketplace, or the search
genuinely turns up nothing usable), a human-readable internal SKU is
generated instead, e.g. "AACHI-SAM-500-001" for
"Aachi Sambar Powder" at size "500g":
AACHI - brand code (first brand word, up to 6 chars)
SAM - product code (first meaningful word of the title,
first 3 letters; brand name and generic
filler words like "Original"/"Classic"
are skipped)
500 - size code (leading numeric portion of the pack size)
001 - sequence (persisted per BRAND-PRODUCT-SIZE prefix in
brand_catalog_{brand}.json under a
``_sku_sequences`` key so repeat pipeline
runs hand out fresh, non-colliding numbers
instead of restarting at 001 every time)
Both paths populate the same two catalog fields, so downstream code
(JSON export, DB storage, UI) never needs to know which path was taken:
product_sku - the ID/SKU string itself
sku_source - where it came from, e.g. "Amazon (ASIN)",
"Flipkart (PID)", or "Internal"
"""
from __future__ import annotations
import json
import logging
import os
import re
import threading
from pathlib import Path
from typing import Any, Dict, Optional, Tuple
from app.infrastructure.settings import DATA_DIR, ENABLE_SKU_WEB_LOOKUP
from app.services.brand_registry import resolve_parent_brand
logger = logging.getLogger(__name__)
_seq_lock = threading.Lock()
# Deviation from the sibling project, deliberate on both counts.
#
# 1. Absolute, not relative. The sibling used Path("data"), which resolves
# against the current working directory - fine for a CLI run from the repo
# root, wrong for a uvicorn process started from anywhere else.
# 2. Its own directory, not the brand catalog files. The sibling stored the
# counter inside data/brand_catalog_<brand>.json. In this repo those files
# are real seed catalogs under data/seed_catalogs/, owned by brand_sync;
# export_brand_to_seed_file() rewrites them wholesale, which would silently
# drop the counters and start handing out duplicate SKUs. Keeping the
# counter in its own file leaves the seed catalogs untouched.
_data_dir = DATA_DIR / "sku_sequences"
def _brand_seq_file(brand: str) -> Path:
storage_brand = resolve_parent_brand(brand)
safe_brand = storage_brand.strip().replace(" & ", " and ").replace("&", "_and_").replace(" ", "_").replace("-", "_")
return _data_dir / f"{safe_brand}.json"
# Words that shouldn't be used as the "meaningful" word when building the
# product code portion of an internal SKU.
_STOPWORDS = {"the", "and", "of", "with", "for", "in", "a", "an", "new", "pack", "combo", "pouch"}
_GENERIC_FILLER = {"original", "regular", "classic", "plain"}
# Domain -> (regex to pull the ID out of the URL, human-readable source label).
# Matching happens directly against the result URL - none of these require
# fetching/parsing the actual product page.
_MARKETPLACE_PATTERNS: list[Tuple[str, "re.Pattern[str]", str]] = [
("amazon.in", re.compile(r"/dp/([A-Z0-9]{10})", re.I), "Amazon (ASIN)"),
("amazon.com", re.compile(r"/dp/([A-Z0-9]{10})", re.I), "Amazon (ASIN)"),
("flipkart.com", re.compile(r"[?&]pid=([A-Z0-9]+)", re.I), "Flipkart (PID)"),
("bigbasket.com", re.compile(r"/pd/(\d+)/?"), "BigBasket (Product ID)"),
("jiomart.com", re.compile(r"/p/[^/?]+/(\d+)"), "JioMart (Product ID)"),
("blinkit.com", re.compile(r"/prn/[^/]+/prid/(\d+)"), "Blinkit (Product ID)"),
("nykaa.com", re.compile(r"/p/(\d+)"), "Nykaa (Product ID)"),
]
# Domains searched for a real product ID - kept in sync with the patterns above.
_SEARCH_DOMAINS = [d for d, _pat, _label in _MARKETPLACE_PATTERNS]
def _brand_code(brand: str) -> str:
if not brand:
return "GEN"
first = re.split(r"\s+", brand.strip())[0]
code = re.sub(r"[^A-Za-z0-9]", "", first).upper()
return code[:6] or "GEN"
def _product_code(title: str, brand: str = "") -> str:
"""First meaningful word of the (brand-stripped) title -> first 3 letters.
Uses only the base product name, not any " - Variety" suffix added by
the variety-explosion step, so different flavours of the same product
share the same product code (they're differentiated by the sequence
number instead).
"""
clean = title or ""
if brand and clean.lower().startswith(brand.strip().lower()):
clean = clean[len(brand.strip()):].strip()
base = clean.split(" - ")[0]
words = re.findall(r"[A-Za-z0-9]+", base)
for w in words:
wl = w.lower()
if wl in _STOPWORDS or wl in _GENERIC_FILLER:
continue
code = re.sub(r"[^A-Za-z0-9]", "", w).upper()
if code:
return code[:3]
return "PRD"
def _size_code(size: str) -> str:
if not size:
return "STD"
m = re.search(r"(\d+(?:\.\d+)?)", size)
if not m:
return "STD"
return m.group(1).replace(".", "")
def _atomic_write_json(path: Path, data: Dict[str, Any]) -> None:
"""Write JSON to `path` atomically: write to a temp file in the same
directory, flush+fsync it, then os.replace() it onto the destination.
os.replace() is atomic on both POSIX and Windows - the destination
file is guaranteed to either be the OLD complete content or the NEW
complete content, NEVER a half-written/truncated hybrid, no matter
when the process is killed. This is what prevents an accidental
termination (Ctrl+C, crash, killed terminal, power loss) from
corrupting data/brand_catalog_{brand}.json - which previously used a
plain write_text()/open(...,'w') that truncates the file in place
before writing the new content, so a kill mid-write left a half-empty
or invalid-JSON file that the *next* run would either silently reset
(SKU counters restarting at 0, causing duplicate SKUs) or, worse,
silently treat as "no existing catalog" and discard every previously
saved product for that brand (see save_catalog_to_data_folder() in
streamlit_app.py for the matching fix on the read/merge side).
"""
tmp_path = path.with_suffix(path.suffix + f".tmp{os.getpid()}")
text = json.dumps(data, indent=2, ensure_ascii=False)
with open(tmp_path, "w", encoding="utf-8") as f:
f.write(text)
f.flush()
os.fsync(f.fileno())
os.replace(tmp_path, path) # atomic on POSIX and Windows
def _next_sequence(brand: str, prefix: str) -> int:
"""File-backed counter stored inside the brand catalog JSON file
(data/brand_catalog_{brand}.json) under a reserved ``_sku_sequences``
key, so re-running the pipeline for the same brand keeps handing out
fresh sequence numbers for a given BRAND-PRODUCT-SIZE prefix instead
of restarting at 001.
"""
seq_file = _brand_seq_file(brand)
with _seq_lock:
data: Dict[str, Any] = {}
counters: Dict[str, int] = {}
if seq_file.exists():
try:
data = json.loads(seq_file.read_text(encoding="utf-8"))
counters = data.get("_sku_sequences", {})
except Exception as e:
# The file exists but isn't valid JSON - almost always the
# result of a previous run being killed mid-write. Preserve
# the corrupt file under a .corrupt name instead of silently
# overwriting it, so no evidence/data is lost, and log
# loudly rather than quietly resetting counters to 0 (which
# would otherwise produce duplicate/colliding SKUs).
logger.warning(
f"SKU sequence file {seq_file} is corrupted ({e}); "
f"resetting counters for this brand. The unreadable "
f"file has been preserved for inspection."
)
try:
seq_file.rename(seq_file.with_suffix(seq_file.suffix + ".corrupt"))
except Exception:
pass
data = {}
counters = {}
next_val = int(counters.get(prefix, 0)) + 1
counters[prefix] = next_val
data["_sku_sequences"] = counters
try:
_data_dir.mkdir(parents=True, exist_ok=True)
_atomic_write_json(seq_file, data)
except Exception as e:
logger.warning(f"Could not persist SKU sequence counter: {e}")
return next_val
def generate_internal_sku(brand: str, product_title: str, size: str) -> str:
"""Deterministic-format internal SKU, e.g. 'AACHI-SAM-500-001'."""
prefix = f"{_brand_code(brand)}-{_product_code(product_title, brand)}-{_size_code(size)}"
seq = _next_sequence(brand, prefix)
return f"{prefix}-{seq:03d}"
def find_website_product_id(brand: str, product_title: str, size: str) -> Optional[Tuple[str, str]]:
"""Best-effort search for a real marketplace product ID.
Returns (product_id, source_label) or None. Never raises - any
network/parsing failure just means "no real ID found", which the
caller falls back to an internal SKU for.
"""
if not ENABLE_SKU_WEB_LOOKUP:
return None
query = f"{brand} {product_title} {size}".strip()
if not query:
return None
try:
from ddgs import DDGS
except ImportError:
logger.debug("SKU web lookup unavailable: 'ddgs' package not installed")
return None
try:
with DDGS(timeout=10) as ddgs:
results = ddgs.text(query, region="in-en", safesearch="off", max_results=8)
except Exception as e:
logger.debug(f"SKU web lookup failed for '{query}': {e}")
return None
for r in results or []:
url = str(r.get("href") or r.get("url") or "")
if not url:
continue
url_lower = url.lower()
for domain, pattern, label in _MARKETPLACE_PATTERNS:
if domain not in url_lower:
continue
m = pattern.search(url)
if m:
return m.group(1).upper(), label
return None
def resolve_product_sku(brand: str, product_title: str, size: str) -> Dict[str, str]:
"""Main entry point used by the catalog engine for each exploded
(brand, product/variety, pack size) catalog row. Tries a real
marketplace product ID first, falls back to a deterministic internal
SKU when none can be found.
"""
found = None
try:
found = find_website_product_id(brand, product_title, size)
except Exception as e:
logger.debug(f"SKU lookup error for '{brand} {product_title} {size}': {e}")
if found:
product_id, source = found
return {"product_sku": product_id, "sku_source": source}
internal_sku = generate_internal_sku(brand, product_title, size)
return {"product_sku": internal_sku, "sku_source": "Internal"}

View File

@@ -0,0 +1,276 @@
"""
Title <-> Category consistency guard.
WHY THIS FILE EXISTS
---------------------
Reported bug: brand="ITC" produced a catalog row titled
"ITC Engage Biscuit Box - Classic Salted". The CATEGORY came back correct
("Fragrance & Deodorants") because app.services.brand_registry.
get_sub_brand_category() matched the known ITC sub-brand "Engage" inside
the title and returned its real, curated category. The IMAGES came back
correct too, because image search is keyed off that same "Engage" match
plus the (correct) category. But nothing ever checked whether the TITLE
TEXT ITSELF was consistent with that resolved category - so the small
local LLM's hallucinated "Biscuit Box" (mixed in because it was asked for
ITC's "Biscuits & Cookies" category products and free-associated a real
ITC sub-brand name into the wrong product line) sailed straight through
into the stored catalog record unmodified.
This module is the missing check. It is intentionally centralised in one
place rather than duplicated, and intentionally conservative: it only
strips a word/phrase from a title when that word/phrase is a *strong,
unambiguous* signal for a category OTHER than the one the product has
already been resolved to (deterministically, via the sub-brand registry,
or heuristically). Ambiguous words (e.g. "milk", which is valid across
several categories) are only flagged when NONE of their valid categories
match - never guessed away speculatively.
USAGE
-----
Call `validate_and_fix_title(title, category, sub_brand_hint=...)` right
after a product's category has been resolved (deterministically or via
heuristic) and BEFORE that title is used for image search, S3 storage,
DB storage, or JSON export - see app/core/catalog_engine.py, which is the
canonical caller. It returns a (possibly corrected) title plus metadata
about what was changed, so the caller can log/audit every correction
instead of it happening silently.
"""
from __future__ import annotations
import re
import logging
from typing import Iterable, Optional
logger = logging.getLogger(__name__)
# ---------------------------------------------------------------------------
# Canonical word/phrase -> valid-categories map.
#
# Deliberately generous for ambiguous words (mapped to every category they
# could legitimately belong to) so this NEVER strips a word just because a
# product could conceivably be miscategorised - it only strips a word when
# the product's *actual* resolved category isn't anywhere in that word's
# allowed set, i.e. the word is a hard contradiction, not merely unusual.
#
# Multi-word phrases are matched first (longest-phrase-wins) so e.g. "body
# wash" is recognised as a single Skin & Bath Care signal rather than being
# torn apart into two separately-ambiguous words ("body", "wash").
# ---------------------------------------------------------------------------
#
# NOTE ON WHAT IS DELIBERATELY *NOT* IN THIS MAP:
# Ingredient / flavour-descriptor words - "milk", "butter", "cheese",
# "chocolate", "honey", "cashew", "almond", "salt", etc. - are intentionally
# EXCLUDED even though they're each strongly associated with one category
# (Dairy, Chocolates, Dry Fruits & Nuts...). That's because they are also
# extremely common, completely legitimate FLAVOUR/variant modifiers inside
# OTHER categories' real product names - "Good Day Butter Cookies", "Milk
# Bikis", "Cashew Cookies", "Chocolate Cake", "Honey Oats" are all real
# products, not hallucinations. Flagging these words caused false-positive
# corrections that damaged genuine titles, so this map only contains STRONG
# signals: words that denote WHAT THE PRODUCT ITSELF IS (a head noun, not a
# flavour), which essentially never legitimately appear in another
# category's product name. This is what makes it safe to auto-strip them.
CATEGORY_TYPE_WORDS: dict[str, set[str]] = {
# --- Biscuits, bakery, snacks ---
"biscuit": {"Biscuits & Cookies"}, "biscuits": {"Biscuits & Cookies"},
"cookie": {"Biscuits & Cookies"}, "cookies": {"Biscuits & Cookies"},
"wafer": {"Biscuits & Cookies"}, "wafers": {"Biscuits & Cookies"},
"cracker": {"Crackers"}, "crackers": {"Crackers"}, "saltine": {"Crackers"},
"rusk": {"Rusk"}, "rusks": {"Rusk"},
"cake": {"Cakes & Muffins"}, "cakes": {"Cakes & Muffins"},
"muffin": {"Cakes & Muffins"}, "muffins": {"Cakes & Muffins"},
"bread": {"Bakery & Breads"}, "bun": {"Bakery & Breads"}, "buns": {"Bakery & Breads"}, "loaf": {"Bakery & Breads"},
"snack": {"Snacks"}, "snacks": {"Snacks"}, "chips": {"Snacks"}, "namkeen": {"Snacks"},
"appalam": {"Snacks"}, "appalams": {"Snacks"}, "papad": {"Snacks"}, "bhujia": {"Snacks"},
"soan papdi": {"Sweets"}, "gulab jamun": {"Sweets"}, "rasgulla": {"Sweets"},
# --- Staples, spices, grains ---
"atta": {"Atta & Staples"}, "maida": {"Atta & Staples"}, "sooji": {"Atta & Staples"},
"besan": {"Atta & Staples"}, "basmati rice": {"Rice & Pulses", "Atta & Staples"},
"multigrain flour": {"Atta & Staples"},
"masala": {"Spices & Masalas"}, "masalas": {"Spices & Masalas"},
"curry powder": {"Spices & Masalas"}, "garam masala": {"Spices & Masalas"},
"chilli powder": {"Spices & Masalas"}, "turmeric powder": {"Spices & Masalas"},
"deggi mirch": {"Spices & Masalas"}, "tikhalal": {"Spices & Masalas"}, "kashmiri lal": {"Spices & Masalas"},
"pasta": {"Pasta & Noodles"}, "vermicelli": {"Pasta & Noodles"}, "semiya": {"Pasta & Noodles"},
"macaroni": {"Pasta & Noodles"}, "spaghetti": {"Pasta & Noodles"},
"noodle": {"Pasta & Noodles", "Noodles & Instant Food"}, "noodles": {"Pasta & Noodles", "Noodles & Instant Food"},
# --- Desserts (head-noun only, not ingredient words) ---
"ice cream": {"Ice Cream"}, "icecream": {"Ice Cream"}, "kulfi": {"Ice Cream"}, "frozen dessert": {"Ice Cream"},
"shrikhand": {"Dairy - Desserts"}, "basundi": {"Dairy - Desserts"}, "mishti doi": {"Dairy - Desserts"},
# --- Oils ---
"cooking oil": {"Cooking Oils"}, "sunflower oil": {"Cooking Oils"}, "groundnut oil": {"Cooking Oils"},
"gingelly oil": {"Cooking Oils"}, "olive oil": {"Cooking Oils"}, "mustard oil": {"Cooking Oils"},
"vegetable oil": {"Cooking Oils"}, "refined oil": {"Cooking Oils"}, "lamp oil": {"Household - Lamp Oil"},
# --- Beverages ---
"tea": {"Tea & Coffee", "Beverages", "Health Drinks"}, "coffee": {"Tea & Coffee", "Beverages"},
"juice": {"Beverages"}, "juices": {"Beverages"}, "squash": {"Beverages"},
"aamras": {"Beverages"}, "jaljeera": {"Beverages"},
"lassi": {"Dairy", "Beverages"}, "buttermilk": {"Beverages"}, "chaas": {"Beverages"},
"milkshake": {"Beverages"}, "milkshakes": {"Beverages"}, "cold coffee": {"Beverages"}, "smoothie": {"Beverages"},
# --- Confectionery / spreads ---
"candy": {"Candy & Confectionery"}, "toffee": {"Candy & Confectionery"},
"lollipop": {"Candy & Confectionery"}, "eclair": {"Candy & Confectionery"}, "eclairs": {"Candy & Confectionery"},
"pickle": {"Pickles & Chutneys"}, "pickles": {"Pickles & Chutneys"}, "achar": {"Pickles & Chutneys"},
"chutney": {"Pickles & Chutneys"}, "murabba": {"Pickles & Chutneys"},
"health mix": {"Health Foods"}, "sathumaavu": {"Health Foods"}, "ragi malt": {"Health Foods"},
"millet": {"Health Foods"}, "millets": {"Health Foods"},
# --- Health / malted drinks ---
"horlicks": {"Health Drinks"}, "bournvita": {"Health Drinks"}, "complan": {"Health Drinks"},
"maltova": {"Health Drinks"}, "protinex": {"Health Drinks"}, "malted drink": {"Health Drinks"},
"glucon": {"Health Drinks"},
# --- Hair care ---
"shampoo": {"Hair Care"}, "conditioner": {"Hair Care"}, "hair oil": {"Hair Care"},
"hair serum": {"Hair Care"}, "hair cream": {"Hair Care"}, "hair color": {"Hair Care"},
"hair dye": {"Hair Care"}, "anti-dandruff": {"Hair Care"}, "hair fall": {"Hair Care"},
# --- Bath / skin / beauty ---
"soap": {"Bath Soap", "Skin & Bath Care", "Personal Care"},
"body wash": {"Skin & Bath Care", "Personal Care"}, "shower gel": {"Skin & Bath Care"},
"hand wash": {"Skin & Bath Care"}, "hand cream": {"Skin & Bath Care"},
"face wash": {"Skin Care", "Beauty Care"}, "face cream": {"Skin & Bath Care", "Skin Care"},
"face powder": {"Beauty Care"}, "moisturiser": {"Skin Care", "Beauty Care"},
"moisturizer": {"Skin Care", "Beauty Care"}, "sunscreen": {"Skin Care", "Beauty Care"}, "sunblock": {"Skin Care", "Beauty Care"},
"foundation": {"Beauty Care"}, "lipstick": {"Beauty Care"}, "lip balm": {"Beauty Care"},
"makeup": {"Beauty Care"}, "cosmetic": {"Beauty Care"}, "eyeliner": {"Beauty Care"},
"kajal": {"Beauty Care"}, "mascara": {"Beauty Care"}, "nail polish": {"Beauty Care"},
"fairness": {"Beauty Care"},
# --- Oral care ---
"toothpaste": {"Oral Care"}, "toothbrush": {"Oral Care"}, "mouthwash": {"Oral Care"}, "dental": {"Oral Care"},
# --- Fragrance ---
"deodorant": {"Fragrance & Deodorants"}, "deodorants": {"Fragrance & Deodorants"}, "deo": {"Fragrance & Deodorants"},
"perfume": {"Fragrance & Deodorants"}, "perfumes": {"Fragrance & Deodorants"}, "fragrance": {"Fragrance & Deodorants"},
"cologne": {"Fragrance & Deodorants"}, "body spray": {"Fragrance & Deodorants"}, "roll-on": {"Fragrance & Deodorants"},
# --- Men's grooming ---
"shaving cream": {"Men's Grooming"}, "shaving foam": {"Men's Grooming"}, "after shave": {"Men's Grooming"},
"razor": {"Men's Grooming"}, "blade": {"Men's Grooming"}, "trimmer": {"Men's Grooming"}, "beard": {"Men's Grooming"},
# --- Home care / laundry ---
"detergent": {"Detergents & Fabric Care"}, "washing powder": {"Detergents & Fabric Care"},
"fabric wash": {"Detergents & Fabric Care"}, "fabric care": {"Detergents & Fabric Care"},
"dishwash": {"Household Cleaning"}, "floor cleaner": {"Household Cleaning"}, "toilet cleaner": {"Household Cleaning"},
"fabric softener": {"Household Cleaning"}, "stain remover": {"Household Cleaning"}, "handwash": {"Household Cleaning"},
"mosquito repellent": {"Mosquito Repellent"}, "air freshener": {"Air Freshener"},
# --- Baby care ---
"baby powder": {"Baby Care"}, "baby lotion": {"Baby Care"}, "baby oil": {"Baby Care"}, "baby shampoo": {"Baby Care"},
}
# Sort once, longest-phrase-first, so multi-word phrases are matched
# before their component single words could be (see module docstring).
_SORTED_TYPE_WORDS = sorted(CATEGORY_TYPE_WORDS.keys(), key=len, reverse=True)
# Connector/filler tokens that shouldn't be left dangling after a
# conflicting word is stripped out (e.g. "Engage - Classic Salted" is
# fine; "Engage - -" or "Engage with" is not).
_DANGLING_TOKENS = {"-", "&", "and", "with", "for", "the", "a", "an", "of"}
def find_category_conflicts(title: str, category: str) -> list[str]:
"""Return the list of phrases in `title` that are strong signals for a
category OTHER than `category`. Empty list = no detected conflict
(does not mean the title is *proven* correct, just that nothing in it
contradicts the given category)."""
if not title or not category:
return []
cat = category.strip()
text = " " + title.lower() + " "
conflicts: list[str] = []
consumed = set()
for phrase in _SORTED_TYPE_WORDS:
pattern = r"(?<![a-z0-9])" + re.escape(phrase) + r"(?![a-z0-9])"
m = re.search(pattern, text)
if not m:
continue
span = (m.start(), m.end())
if any(not (span[1] <= s or span[0] >= e) for s, e in consumed):
continue # already covered by a longer phrase match
allowed = CATEGORY_TYPE_WORDS[phrase]
if cat not in allowed:
conflicts.append(phrase)
consumed.add(span)
return conflicts
def sanitize_title_for_category(title: str, category: str) -> str:
"""Strip any word/phrase from `title` that contradicts `category`,
then tidy up leftover punctuation/connector words. Pure text
transform - does not know about brands or sub-brands."""
if not title or not category:
return title
cat = category.strip()
result = title
for phrase in _SORTED_TYPE_WORDS:
allowed = CATEGORY_TYPE_WORDS[phrase]
if cat in allowed:
continue
pattern = re.compile(r"(?<![a-zA-Z0-9])" + re.escape(phrase) + r"(?![a-zA-Z0-9])", re.IGNORECASE)
result = pattern.sub(" ", result)
# Tidy: collapse whitespace, strip dangling connector tokens/punctuation
# left behind at the edges or next to now-empty gaps.
tokens = [t for t in re.split(r"\s+", result.strip()) if t]
cleaned_tokens: list[str] = []
for t in tokens:
bare = t.strip(".,!?;:'\"()").lower()
if bare in _DANGLING_TOKENS and (not cleaned_tokens or cleaned_tokens[-1].strip(".,!?;:'\"()").lower() in _DANGLING_TOKENS):
continue
cleaned_tokens.append(t)
while cleaned_tokens and cleaned_tokens[-1].strip(".,!?;:'\"()").lower() in _DANGLING_TOKENS:
cleaned_tokens.pop()
while cleaned_tokens and cleaned_tokens[0].strip(".,!?;:'\"()").lower() in _DANGLING_TOKENS:
cleaned_tokens.pop(0)
cleaned = " ".join(cleaned_tokens).strip()
return cleaned
def validate_and_fix_title(
title: str,
category: Optional[str],
sub_brand_hint: Optional[str] = None,
brand: Optional[str] = None,
) -> tuple[str, bool, list[str]]:
"""Authoritative title/category consistency check + repair.
Returns (fixed_title, was_changed, removed_phrases).
`category` should be the ALREADY-RESOLVED, trusted category for this
product (e.g. from brand_registry.get_sub_brand_category(), or the
pipeline's heuristic classification) - this function treats it as
ground truth and corrects the title to match it, never the reverse.
`sub_brand_hint` (e.g. "engage") is used ONLY as a last-resort anchor
if stripping conflicting words leaves nothing usable behind - it is
never invented, only ever a value the caller already resolved
deterministically (e.g. via brand_registry.get_sub_brand_match()).
"""
if not title or not category or category in ("Uncategorized", "General"):
return title, False, []
conflicts = find_category_conflicts(title, category)
if not conflicts:
return title, False, []
fixed = sanitize_title_for_category(title, category)
# If sanitizing collapsed the title to nothing (or to something too
# short/generic to stand alone, e.g. just leftover flavour words with
# no anchor), fall back to the known sub-brand or brand name so we
# never emit an empty/blank title.
if not fixed or len(fixed.strip()) < 2:
anchor = (sub_brand_hint or brand or "").strip()
# Preserve a trailing " - Variant" suffix from the original title
# if present (flavour/variant naming is independent of the
# category-word bug this function targets).
variant_suffix = ""
m = re.search(r"\s-\s(.+)$", title.strip())
if m and not any(w in m.group(1).lower() for w in conflicts):
variant_suffix = f" - {m.group(1).strip()}"
fixed = (anchor.title() if anchor else category) + variant_suffix
fixed = fixed.strip()
changed = fixed.strip().lower() != title.strip().lower()
if changed:
logger.warning(
f"⚠️ Title/category mismatch corrected: '{title}' -> '{fixed}' "
f"(category='{category}', removed={conflicts})"
)
return fixed, changed, conflicts

View File

@@ -0,0 +1,5 @@
{
"_sku_sequences": {
"AACHI-SAM-100": 1
}
}

View File

@@ -0,0 +1,9 @@
{
"_sku_sequences": {
"BRITAN-50-200": 11,
"BRITAN-50-100": 4,
"BRITAN-50-250": 2,
"BRITAN-50-500": 3,
"BRITAN-GOO-200": 3
}
}

View File

@@ -0,0 +1,5 @@
{
"_sku_sequences": {
"CADBUR-DAI-100": 1
}
}

View File

@@ -0,0 +1,283 @@
"""Tests for the store-spreadsheet -> 11-stage pipeline -> brand table flow.
Follows the pattern established by test_user_products_upload.py: monkeypatch
the storage/network boundary *on the pipeline module object* (it imports those
names directly), build real .xlsx fixtures in memory, and assert on what was
captured rather than on the HTTP status alone.
Every test runs with use_llm=False and fetch_images=False so nothing here
touches Ollama or the open web.
"""
from __future__ import annotations
import io
import pytest
from app.core import store_catalog_pipeline as pipeline
openpyxl = pytest.importorskip("openpyxl")
def _sheet(headers, rows) -> bytes:
wb = openpyxl.Workbook()
ws = wb.active
ws.append(headers)
for row in rows:
ws.append(row)
buf = io.BytesIO()
wb.save(buf)
return buf.getvalue()
@pytest.fixture
def store(monkeypatch):
"""A fake brand table. Returns the dict of image_id -> stored row."""
table: dict = {}
def fake_upsert(brand, rows, cleanup=False):
assert cleanup is False, "cleanup=True would delete the brand's existing catalog"
for row in rows:
table[row["image_id"]] = dict(row)
return len(rows)
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: list(table.values()))
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
return table
def _run(content, **kw):
return pipeline.run_pipeline("store.xlsx", content, use_llm=False, fetch_images=False, **kw)
# ---------------------------------------------------------------------------
# Column mapping / brand resolution
# ---------------------------------------------------------------------------
def test_messy_headers_map_onto_catalog_fields(store):
content = _sheet(
["Item Name", "Product Description", "Segment", "Net Weight", "Random Column"],
[["Britannia 50-50", "Salty biscuit", "Biscuits", "200g", "junk"]],
)
result = _run(content)
assert result.recognised_columns["product_name"] == "Item Name"
assert result.recognised_columns["description"] == "Product Description"
assert result.recognised_columns["category"] == "Segment"
assert result.recognised_columns["size_variants"] == "Net Weight"
assert result.unrecognised_columns == ["Random Column"]
def test_brand_is_inferred_when_the_sheet_has_no_brand_column(store):
"""Store files routinely omit the brand; it lives in the product name."""
content = _sheet(["Item Name"], [["Britannia 50-50"], ["Cadbury Dairy Milk 100g"]])
result = _run(content)
assert result.errors == []
assert result.brands == ["Britannia", "Cadbury"]
def test_longest_alias_wins_when_inferring_a_brand():
"""'cadbury dairy milk' must beat the shorter 'cadbury' substring."""
assert pipeline.infer_brand("Cadbury Dairy Milk Silk 100g") == "cadbury dairy milk"
def test_rows_land_in_the_brand_table_the_product_belongs_to(store):
content = _sheet(
["Item Name", "Net Weight"],
[["Britannia 50-50", "200g"], ["Aachi Sambar Powder", "100g"]],
)
_run(content)
assert sorted(store) == ["aachi_aachi_sambar_powder_100g", "britannia_britannia_50_50_200g"]
# ---------------------------------------------------------------------------
# Gap filling
# ---------------------------------------------------------------------------
def test_missing_fields_are_filled_by_the_pipeline(store):
"""A sheet with only a product name still produces a complete row."""
content = _sheet(["Item Name"], [["Britannia Good Day Biscuits 200g"]])
_run(content)
row = next(iter(store.values()))
assert row["fssai_license"] == "10012022000103" # stage 1
assert row["category"] # stage 3
assert row["price_range"].startswith("₹") # stage 5
assert row["product_sku"] # stage 7
assert row["hsn_code"] # stage 9
assert row["search_query"] # stage 11
def test_values_supplied_by_the_store_are_never_overwritten(store):
content = _sheet(
["Item Name", "Net Weight", "MRP Range", "HSN Code"],
[["Britannia 50-50", "200g", "₹111-222", "9999"]],
)
_run(content)
row = next(iter(store.values()))
assert row["price_range"] == "₹111-222"
assert row["hsn_code"] == "9999"
def test_an_uncategorised_row_gets_no_hsn_code(store):
"""HSN/GST is a regulatory value keyed on category. When the category
cannot be determined the column is deliberately left empty rather than
guessed - a wrong tax code is worse than a missing one."""
content = _sheet(["Item Name"], [["Britannia 50-50"]])
result = _run(content)
row = next(iter(store.values()))
assert row["category"] == "General"
assert row["hsn_code"] is None
assert any("category could not be detected" in w for w in result.warnings)
def test_non_food_brands_get_no_fssai_licence(store):
"""A miss means 'not a food brand', not an error - the column stays empty."""
assert pipeline.get_fssai_license("Colgate") is None
# ---------------------------------------------------------------------------
# Stage 4 - pack-size explosion and unit safety
# ---------------------------------------------------------------------------
def test_one_row_explodes_into_one_row_per_pack_size(store):
content = _sheet(["Item Name", "Pack Size"], [["Britannia 50-50", "100g, 200g, 500g"]])
_run(content)
assert len(store) == 3
assert {r["size_variants"][0] for r in store.values()} == {"100g", "200g", "500g"}
def test_each_pack_size_gets_its_own_image_id(store):
"""Without the size in the id every variant collapses onto one row under
ON CONFLICT (image_id)."""
content = _sheet(["Item Name", "Pack Size"], [["Britannia 50-50", "100g, 200g"]])
_run(content)
assert len(set(store)) == 2
def test_the_size_is_not_duplicated_when_already_in_the_product_name():
assert pipeline.build_image_id("Cadbury", "Cadbury Dairy Milk 100g", "100g") == \
"cadbury_cadbury_dairy_milk_100g"
def test_a_pack_size_with_a_nonsense_unit_is_dropped_with_a_reason(store):
"""A biscuit measured in centimetres is a data error, not a pack size."""
content = _sheet(["Item Name", "Segment", "Pack Size"], [["Britannia 50-50", "Biscuits", "15cm"]])
result = _run(content)
assert any("15cm" in w for w in result.warnings)
# ---------------------------------------------------------------------------
# Stage 11 - storage semantics
# ---------------------------------------------------------------------------
def test_reingesting_the_same_file_changes_nothing(store):
content = _sheet(["Item Name", "Net Weight"], [["Britannia 50-50", "200g"]])
first = _run(content)
assert (first.inserted, first.skipped_existing) == (1, 0)
second = _run(content)
assert (second.inserted, second.backfilled, second.skipped_existing) == (0, 0, 1)
assert len(store) == 1
def test_an_existing_row_has_only_its_blank_columns_backfilled(store):
content = _sheet(["Item Name", "Net Weight"], [["Britannia Good Day Biscuits", "200g"]])
_run(content)
key = next(iter(store))
store[key]["hsn_code"] = None
store[key]["price_range"] = "₹999-1000" # a value already held
result = _run(content)
assert result.backfilled == 1
assert store[key]["hsn_code"] == "1905" # blank -> filled
assert store[key]["price_range"] == "₹999-1000" # held value untouched
def test_a_storage_failure_is_reported_rather_than_counted_as_success(store, monkeypatch):
def boom(brand, rows, cleanup=False):
raise RuntimeError("pgvector is down")
monkeypatch.setattr(pipeline, "upsert_brand_products", boom)
result = _run(_sheet(["Item Name"], [["Britannia 50-50 200g"]]))
assert result.storage_error and "pgvector is down" in result.storage_error
assert result.inserted == 0
# ---------------------------------------------------------------------------
# Row-level error handling
# ---------------------------------------------------------------------------
def test_a_row_with_no_usable_product_name_is_reported_not_fatal(store):
content = _sheet(["Item Name", "Net Weight"], [["", "200g"], ["Britannia 50-50", "200g"]])
result = _run(content)
assert len(result.errors) == 1
assert result.errors[0].row == 2 # header is row 1
assert result.inserted == 1 # the good row still landed
def test_the_same_pack_listed_twice_is_written_once(store):
content = _sheet(
["Item Name", "Net Weight"],
[["Britannia 50-50", "200g"], ["Britannia 50-50", "200g"]],
)
_run(content)
assert len(store) == 1
# ---------------------------------------------------------------------------
# HTTP surface
# ---------------------------------------------------------------------------
def test_preview_reports_how_the_columns_were_understood(client, admin_headers):
content = _sheet(["Item Name", "Segment", "Mystery"], [["Britannia 50-50", "Biscuits", "?"]])
resp = client.post(
"/api/admin/store-catalog/preview",
files={"file": ("store.xlsx", content)},
headers=admin_headers,
)
assert resp.status_code == 200
body = resp.json()
assert body["recognised_columns"]["product_name"] == "Item Name"
assert body["unrecognised_columns"] == ["Mystery"]
assert body["brand_column_present"] is False
assert len(body["stages"]) == 11
def test_preview_requires_admin(client):
content = _sheet(["Item Name"], [["Britannia 50-50"]])
resp = client.post("/api/admin/store-catalog/preview", files={"file": ("s.xlsx", content)})
assert resp.status_code == 401
def test_an_unparseable_file_is_rejected_before_a_job_is_created(client, admin_headers):
resp = client.post(
"/api/admin/store-catalog/ingest",
files={"file": ("notes.txt", b"this is not a spreadsheet")},
headers=admin_headers,
)
assert resp.status_code == 400
def test_ingest_returns_a_job_id_that_can_be_polled(client, admin_headers, monkeypatch):
# Keep the worker off the network and out of the database.
monkeypatch.setattr(pipeline, "upsert_brand_products", lambda b, r, cleanup=False: len(r))
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: [])
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
content = _sheet(["Item Name", "Segment"], [["Britannia 50-50", "Biscuits"]])
resp = client.post(
"/api/admin/store-catalog/ingest",
files={"file": ("store.xlsx", content)},
headers=admin_headers,
)
assert resp.status_code == 202
job_id = resp.json()["job_id"]
assert resp.json()["rows_total"] == 1
poll = client.get(f"/api/admin/store-catalog/jobs/{job_id}", headers=admin_headers)
assert poll.status_code == 200
body = poll.json()
assert body["status"] in {"pending", "running", "done", "failed"}
assert body["total_stages"] == 11
def test_polling_an_unknown_job_is_a_404(client, admin_headers):
resp = client.get("/api/admin/store-catalog/jobs/does-not-exist", headers=admin_headers)
assert resp.status_code == 404