From 7a4583372f2282cefce3db13d1396ea1f7656631 Mon Sep 17 00:00:00 2001 From: sriram Date: Tue, 18 Aug 2026 16:58:36 +0530 Subject: [PATCH] backend stores data file enrichment pipeline --- app/api/routers/store_catalog.py | 183 +++++ app/api/store_catalog_job_store.py | 93 +++ app/core/store_catalog_pipeline.py | 687 ++++++++++++++++++ app/infrastructure/settings.py | 43 ++ app/main.py | 2 + app/services/brand_registry.py | 57 ++ app/services/category_units.py | 338 +++++++++ app/services/enrichment/__init__.py | 50 ++ app/services/enrichment/barcode/__init__.py | 47 ++ app/services/enrichment/barcode/cache.py | 127 ++++ app/services/enrichment/barcode/matching.py | 141 ++++ app/services/enrichment/barcode/models.py | 82 +++ app/services/enrichment/barcode/retry.py | 47 ++ app/services/enrichment/barcode/service.py | 170 +++++ .../enrichment/barcode/sources/__init__.py | 35 + .../enrichment/barcode/sources/base.py | 46 ++ .../enrichment/barcode/sources/gs1_india.py | 91 +++ .../barcode/sources/manufacturer_site.py | 108 +++ .../barcode/sources/open_food_facts.py | 118 +++ .../barcode/sources/upc_database.py | 81 +++ app/services/enrichment/barcode/stage.py | 37 + app/services/enrichment/barcode/validators.py | 99 +++ app/services/enrichment/base.py | 80 ++ app/services/enrichment/hsn_gst/__init__.py | 32 + app/services/enrichment/hsn_gst/models.py | 259 +++++++ app/services/enrichment/hsn_gst/stage.py | 48 ++ app/services/enrichment/pipeline.py | 89 +++ app/services/price_estimator.py | 29 + app/services/product_validator.py | 445 ++++++++++++ app/services/quantity_utils.py | 82 +++ app/services/sku_service.py | 275 +++++++ app/services/title_validator.py | 276 +++++++ data/sku_sequences/Aachi.json | 5 + data/sku_sequences/britannia.json | 9 + data/sku_sequences/cadbury.json | 5 + tests/test_store_catalog_pipeline.py | 283 ++++++++ 36 files changed, 4599 insertions(+) create mode 100644 app/api/routers/store_catalog.py create mode 100644 app/api/store_catalog_job_store.py create mode 100644 app/core/store_catalog_pipeline.py create mode 100644 app/services/category_units.py create mode 100644 app/services/enrichment/__init__.py create mode 100644 app/services/enrichment/barcode/__init__.py create mode 100644 app/services/enrichment/barcode/cache.py create mode 100644 app/services/enrichment/barcode/matching.py create mode 100644 app/services/enrichment/barcode/models.py create mode 100644 app/services/enrichment/barcode/retry.py create mode 100644 app/services/enrichment/barcode/service.py create mode 100644 app/services/enrichment/barcode/sources/__init__.py create mode 100644 app/services/enrichment/barcode/sources/base.py create mode 100644 app/services/enrichment/barcode/sources/gs1_india.py create mode 100644 app/services/enrichment/barcode/sources/manufacturer_site.py create mode 100644 app/services/enrichment/barcode/sources/open_food_facts.py create mode 100644 app/services/enrichment/barcode/sources/upc_database.py create mode 100644 app/services/enrichment/barcode/stage.py create mode 100644 app/services/enrichment/barcode/validators.py create mode 100644 app/services/enrichment/base.py create mode 100644 app/services/enrichment/hsn_gst/__init__.py create mode 100644 app/services/enrichment/hsn_gst/models.py create mode 100644 app/services/enrichment/hsn_gst/stage.py create mode 100644 app/services/enrichment/pipeline.py create mode 100644 app/services/product_validator.py create mode 100644 app/services/quantity_utils.py create mode 100644 app/services/sku_service.py create mode 100644 app/services/title_validator.py create mode 100644 data/sku_sequences/Aachi.json create mode 100644 data/sku_sequences/britannia.json create mode 100644 data/sku_sequences/cadbury.json create mode 100644 tests/test_store_catalog_pipeline.py diff --git a/app/api/routers/store_catalog.py b/app/api/routers/store_catalog.py new file mode 100644 index 0000000..e80e3ae --- /dev/null +++ b/app/api/routers/store_catalog.py @@ -0,0 +1,183 @@ +"""Admin endpoints for turning a store spreadsheet into brand-catalog rows. + +Three endpoints, matching how the other long-running admin jobs in this +project are exposed (see `nutrition_admin.py`): + + POST /api/admin/store-catalog/preview - parse only, show the column map + POST /api/admin/store-catalog/ingest - 202 + job_id, runs in background + GET /api/admin/store-catalog/jobs/{id} - poll stage + row progress + +`preview` exists because the mapping from a store's headers onto catalog +fields is a guess. Running an 11-stage scrape over 2000 rows only to discover +that "Item" was read as the description is expensive; showing the operator the +mapping first costs one parse. +""" +from __future__ import annotations + +import logging +from typing import Optional + +from fastapi import APIRouter, Depends, File, HTTPException, UploadFile, status +from pydantic import BaseModel + +from app.api.background import run_in_background +from app.api.deps import require_admin +from app.api.store_catalog_job_store import store_catalog_job_store +from app.core import store_catalog_pipeline as pipeline + +logger = logging.getLogger(__name__) +router = APIRouter(prefix="/admin/store-catalog", tags=["admin", "catalog"]) + +# Same ceilings as the user-products upload path, for the same reasons. +MAX_UPLOAD_BYTES = 10 * 1024 * 1024 +MAX_UPLOAD_ROWS = 2000 +PREVIEW_ROWS = 10 + + +class StoreCatalogJobOut(BaseModel): + job_id: str + filename: str + status: str + detail: Optional[str] = None + stage_index: int = 0 + stage_name: str = "" + total_stages: int = pipeline.TOTAL_STAGES + rows_done: int = 0 + rows_total: int = 0 + result: Optional[dict] = None + + +async def _read_upload(file: UploadFile) -> bytes: + contents = await file.read() + if not contents: + raise HTTPException(status_code=400, detail="The uploaded file is empty.") + if len(contents) > MAX_UPLOAD_BYTES: + raise HTTPException( + status_code=413, + detail=f"File is larger than the {MAX_UPLOAD_BYTES // (1024 * 1024)}MB limit.", + ) + return contents + + +@router.post("/preview", dependencies=[Depends(require_admin)]) +async def preview_store_catalog(file: UploadFile = File(...)) -> dict: + """Parse the sheet and report how its columns were understood.""" + contents = await _read_upload(file) + try: + df, mapping = pipeline.parse_spreadsheet(file.filename or "upload.xlsx", contents) + except HTTPException: + raise + except ValueError as exc: + raise HTTPException(status_code=400, detail=str(exc)) from exc + except Exception as exc: # noqa: BLE001 - unreadable file is a user error + raise HTTPException(status_code=400, detail=f"Could not parse the file: {exc}") from exc + + if df.empty: + raise HTTPException(status_code=400, detail="The file has no data rows.") + if len(df) > MAX_UPLOAD_ROWS: + raise HTTPException( + status_code=413, + detail=f"{len(df)} rows exceeds the {MAX_UPLOAD_ROWS}-row limit for one upload.", + ) + + sample = df.head(PREVIEW_ROWS).fillna("").astype(str).to_dict(orient="records") + return { + "filename": file.filename, + "rows_total": int(len(df)), + "recognised_columns": {f: str(c) for f, c in mapping.columns.items()}, + "ignored_columns": mapping.ignored, + "unrecognised_columns": mapping.unrecognised, + "brand_column_present": "brand" in mapping.columns, + "preview": sample, + "stages": list(pipeline.STAGE_NAMES), + } + + +def _run_ingest_job(job_id: str, filename: str, contents: bytes, + use_llm: bool, fetch_images: bool) -> None: + store_catalog_job_store.update(job_id, status="running", detail="Starting pipeline") + + def progress(stage_index: int, stage_name: str, done: int, total: int) -> None: + store_catalog_job_store.update( + job_id, stage_index=stage_index, stage_name=stage_name, + rows_done=done, rows_total=total, + ) + + try: + result = pipeline.run_pipeline( + filename, contents, progress=progress, + use_llm=use_llm, fetch_images=fetch_images, + ) + body = result.as_dict() + if body.get("storage_error"): + # Rows were built but none reached the database. Reporting this as + # success would leave the operator believing the catalog changed. + store_catalog_job_store.update( + job_id, status="failed", result=body, + detail=f"Built {body['products_built']} row(s) but storing them failed: " + f"{body['storage_error']}", + ) + return + store_catalog_job_store.update( + job_id, status="done", result=body, + detail=(f"{body['inserted']} inserted, {body['backfilled']} backfilled, " + f"{body['skipped_existing']} unchanged, {body['rejected']} rejected"), + ) + except Exception as exc: # noqa: BLE001 - a daemon thread must not die silently + logger.exception("Store-catalog ingestion job %s failed", job_id) + store_catalog_job_store.update(job_id, status="failed", detail=str(exc)) + + +@router.post("/ingest", status_code=status.HTTP_202_ACCEPTED, + dependencies=[Depends(require_admin)]) +async def ingest_store_catalog( + file: UploadFile = File(...), + use_llm: bool = True, + fetch_images: bool = True, +) -> StoreCatalogJobOut: + """Kick off the 11-stage pipeline and return a job id to poll. + + The file is read here rather than in the worker: `UploadFile` is backed by + a temporary file tied to the request, so it is gone by the time a + background thread would get to it. + """ + contents = await _read_upload(file) + filename = file.filename or "upload.xlsx" + + # Fail fast on an unparseable file so the caller gets a 400 now rather than + # a job that transitions straight to "failed". + try: + df, _mapping = pipeline.parse_spreadsheet(filename, contents) + except Exception as exc: # noqa: BLE001 + raise HTTPException(status_code=400, detail=f"Could not parse the file: {exc}") from exc + if df.empty: + raise HTTPException(status_code=400, detail="The file has no data rows.") + if len(df) > MAX_UPLOAD_ROWS: + raise HTTPException( + status_code=413, + detail=f"{len(df)} rows exceeds the {MAX_UPLOAD_ROWS}-row limit for one upload.", + ) + + job = store_catalog_job_store.create(filename) + store_catalog_job_store.update(job.job_id, rows_total=int(len(df))) + run_in_background( + lambda: _run_ingest_job(job.job_id, filename, contents, use_llm, fetch_images), + name=f"store-catalog-{job.job_id[:8]}", + ) + return StoreCatalogJobOut( + job_id=job.job_id, filename=filename, status=job.status, + rows_total=int(len(df)), + ) + + +@router.get("/jobs/{job_id}", dependencies=[Depends(require_admin)]) +def get_store_catalog_job(job_id: str) -> StoreCatalogJobOut: + job = store_catalog_job_store.get(job_id) + if not job: + raise HTTPException(status_code=404, detail="Job not found") + return StoreCatalogJobOut( + job_id=job.job_id, filename=job.filename, status=job.status, detail=job.detail, + stage_index=job.stage_index, stage_name=job.stage_name, + total_stages=job.total_stages, rows_done=job.rows_done, + rows_total=job.rows_total, result=job.result, + ) diff --git a/app/api/store_catalog_job_store.py b/app/api/store_catalog_job_store.py new file mode 100644 index 0000000..b4dca46 --- /dev/null +++ b/app/api/store_catalog_job_store.py @@ -0,0 +1,93 @@ +"""Job tracking for store-spreadsheet catalog ingestion. + +Same pattern and the same documented trade-offs as `job_store.py`, +`store_job_store.py` and `nutrition_job_store.py`: a process-local dict behind +a lock, lost on restart, not shared across uvicorn workers. Adding a broker for +this would be operational weight the project has already decided against (see +`job_store.py`). + +It is its own module rather than a reuse of `nutrition_job_store` because this +job reports a different shape of progress: an 11-stage pipeline needs to say +*which stage* it is on as well as how many rows it has finished, so the UI can +show "Stage 6/11 - Image Search, 48/120 rows". +""" +from __future__ import annotations + +import threading +import time +import uuid +from dataclasses import dataclass, field +from typing import Dict, Optional + + +@dataclass +class StoreCatalogJob: + job_id: str + filename: str + status: str = "pending" # pending -> running -> done | failed + detail: Optional[str] = None + result: Optional[dict] = None + stage_index: int = 0 # 1-based; 0 while still pending + stage_name: str = "" + total_stages: int = 11 + rows_done: int = 0 + rows_total: int = 0 + created_at: float = field(default_factory=time.time) + updated_at: float = field(default_factory=time.time) + + +class StoreCatalogJobStore: + def __init__(self) -> None: + self._jobs: Dict[str, StoreCatalogJob] = {} + self._lock = threading.Lock() + + def create(self, filename: str) -> StoreCatalogJob: + job = StoreCatalogJob(job_id=str(uuid.uuid4()), filename=filename) + with self._lock: + self._jobs[job.job_id] = job + return job + + def update( + self, + job_id: str, + status: Optional[str] = None, + detail: Optional[str] = None, + result: Optional[dict] = None, + stage_index: Optional[int] = None, + stage_name: Optional[str] = None, + rows_done: Optional[int] = None, + rows_total: Optional[int] = None, + ) -> None: + """Every field is optional and None means "leave alone". + + That matters: `job_store.update` sets `detail` unconditionally, so + marking a job running there wipes whatever detail it already had. A + progress callback firing many times per second must not erase state it + was not asked to change. + """ + with self._lock: + job = self._jobs.get(job_id) + if not job: + return + if status is not None: + job.status = status + if detail is not None: + job.detail = detail + if result is not None: + job.result = result + if stage_index is not None: + job.stage_index = stage_index + if stage_name is not None: + job.stage_name = stage_name + if rows_done is not None: + job.rows_done = rows_done + if rows_total is not None: + job.rows_total = rows_total + job.updated_at = time.time() + + def get(self, job_id: str) -> Optional[StoreCatalogJob]: + with self._lock: + return self._jobs.get(job_id) + + +store_catalog_job_store = StoreCatalogJobStore() diff --git a/app/core/store_catalog_pipeline.py b/app/core/store_catalog_pipeline.py new file mode 100644 index 0000000..99e4959 --- /dev/null +++ b/app/core/store_catalog_pipeline.py @@ -0,0 +1,687 @@ +""" +Store-spreadsheet -> 11-stage enrichment -> brand table. + +A store sends a product list as Excel/CSV. The columns roughly resemble the +brand-catalog schema but are named inconsistently and are always incomplete - +typically product_name, description, category and little else. This module +turns such a file into proper catalog rows: it scrapes/derives everything the +sheet did not supply, validates the result, and writes it to the brand table +the product belongs to (Britannia 50-50 -> brand_britannia). + +Relationship to the existing brand pipeline +------------------------------------------- +`app/core/catalog_engine.py` starts from a BRAND NAME and asks an LLM to +enumerate its products. This module starts from a SPREADSHEET ROW, so the +discovery stage is replaced by row intake and everything downstream is +gap-filling. catalog_engine is not modified or called for orchestration; only +its `_select_best_images` contamination filter is reused. + +Stage order (the numbering the operator sees in the UI): + + 1. Brand Resolution & FSSAI Licence Mapping + 2. Row Intake & Normalisation (replaces AI product discovery) + 3. Title & Category Consistency Guard + 4. Pack-Size Variant Explosion & Unit Safety + 5. Pricing Band Estimation + 6. Image Search & Contamination Filtering + 7. Marketplace & Internal SKU Resolution + 8. Barcode Retrieval & Enrichment + 9. HSN / GST Tax Enrichment + 10. Deterministic Product Validation Gate + 11. Vector Embedding & Storage + +Two rules hold throughout: + +* **Fill only blanks.** The store's own data is authoritative; a stage that + finds a value already present leaves it alone. This is what separates the + pipeline from `user_products._build_product_dict`, which fills the same + gaps with static defaults ("Rs.100-250", a canonical S3 URL). Here the + defaults are replaced by real lookups. +* **A stage never raises.** Enrichment is best-effort; a barcode lookup + timing out must not lose the row. Failures are recorded per row and the + row continues with that field empty. +""" +from __future__ import annotations + +import asyncio +import logging +import re +from dataclasses import dataclass, field +from typing import Any, Callable, Dict, List, Optional, Tuple + +from app.infrastructure.settings import ( + ENABLE_BARCODE_LOOKUP, + ENABLE_HSN_GST_ENRICHMENT, + ENABLE_PRODUCT_VALIDATION, + ENABLE_SKU_WEB_LOOKUP, + MAX_VARIANTS_PER_PRODUCT, + USE_EMBEDDINGS, +) + +# The spreadsheet machinery already exists and is well tested; importing it +# from the router module is deliberate. Extracting it into a service would be +# tidier, but user_products.py is a working file this feature does not touch, +# and tests/test_user_products_upload.py monkeypatches these names on that +# module object. +from app.api.routers.user_products import ( + AddProductRequest, + _slugify, + _text, + map_spreadsheet_columns, + read_products_dataframe, + row_to_request, +) +from app.services import price_estimator +from app.services.brand_registry import ( + BRAND_ALIASES, + get_fssai_license, + resolve_parent_brand, +) +from app.services.category_registry import ( + detect_category_from_text, + sanitize_category_language, +) +from app.services.category_units import fix_or_reject_size +from app.services.embeddings_service import embed_texts +from app.services.enrichment.barcode.stage import BarcodeEnrichmentStage +from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage +from app.services.enrichment.pipeline import EnrichmentPipeline +from app.services.product_validator import validate_catalog +from app.services.sku_service import resolve_product_sku +from app.services.title_validator import validate_and_fix_title +from app.services.vector_store import ( + _sanitize_name, + display_name_for_suffix, + get_products_by_brand, + upsert_brand_products, +) + +logger = logging.getLogger(__name__) + +STAGE_NAMES: Tuple[str, ...] = ( + "Brand Resolution & FSSAI Mapping", + "Row Intake & Normalisation", + "Title & Category Consistency Guard", + "Pack-Size Variant Explosion & Unit Safety", + "Pricing Band Estimation", + "Image Search & Contamination Filtering", + "Marketplace & Internal SKU Resolution", + "Barcode Retrieval & Enrichment", + "HSN / GST Tax Enrichment", + "Deterministic Product Validation Gate", + "Vector Embedding & Storage", +) + +TOTAL_STAGES = len(STAGE_NAMES) + +# Columns a store row may legitimately leave blank and the pipeline fills in. +# Anything not listed here is copied through untouched. +_ENRICHABLE = ( + "category", "description", "fssai_license", "price_range", "product_sku", + "sku_source", "hsn_code", "barcode", "barcode_type", "final_selling_price", + "selling_price", "image_url", "image_urls", "size_variants", +) + +_SIZE_IN_TITLE = re.compile( + r"\b(\d+(?:[.,]\d+)?)\s*(kg|kgs|g|gm|gms|gram|grams|ml|l|ltr|litre|liters?|pack|pcs|n)\b", + re.I, +) + + +def _blank(value: Any) -> bool: + """True if a field carries no usable value.""" + if value is None: + return True + if isinstance(value, str): + return not value.strip() + if isinstance(value, (list, tuple, dict, set)): + return len(value) == 0 + return False + + +@dataclass +class RowError: + row: int + product_name: str + error: str + + +@dataclass +class PipelineResult: + """Everything the UI needs to explain what happened to an upload.""" + + rows_total: int = 0 + products_built: int = 0 + inserted: int = 0 + backfilled: int = 0 + skipped_existing: int = 0 + rejected: int = 0 + brands: List[str] = field(default_factory=list) + rejections: List[Dict[str, Any]] = field(default_factory=list) + errors: List[RowError] = field(default_factory=list) + warnings: List[str] = field(default_factory=list) + recognised_columns: Dict[str, str] = field(default_factory=dict) + unrecognised_columns: List[str] = field(default_factory=list) + storage_error: Optional[str] = None + + def as_dict(self) -> Dict[str, Any]: + return { + "rows_total": self.rows_total, + "products_built": self.products_built, + "inserted": self.inserted, + "backfilled": self.backfilled, + "skipped_existing": self.skipped_existing, + "rejected": self.rejected, + "brands": self.brands, + "rejections": self.rejections[:50], + "errors": [vars(e) for e in self.errors[:50]], + "error_count": len(self.errors), + "warnings": self.warnings, + "recognised_columns": self.recognised_columns, + "unrecognised_columns": self.unrecognised_columns, + "storage_error": self.storage_error, + } + + +ProgressFn = Callable[[int, str, int, int], None] +"""Called as (stage_index_1_based, stage_name, rows_done, rows_total).""" + + +def _noop_progress(stage: int, name: str, done: int, total: int) -> None: + return None + + +# --------------------------------------------------------------------------- +# Stage 1 - Brand resolution & FSSAI +# --------------------------------------------------------------------------- +_INFERRED_BRAND_COL = "__inferred_brand__" + + +def infer_brand(product_name: str) -> Optional[str]: + """Work out which brand a product name belongs to. + + Store files very often have no brand column at all - the brand is simply + the first word or two of the product name. Matches the longest known alias + appearing anywhere in the name, so "britannia 50-50" resolves; longest-first + matters, because "cadbury dairy milk" must win over "cadbury". + + Runs before `row_to_request`, which rejects a brand-less row outright. + """ + haystack = f" {(product_name or '').lower().strip()} " + best: Optional[str] = None + for alias in BRAND_ALIASES: + if re.search(rf"(? len(best): + best = alias + if best: + return best + + # Last resort: assume the leading word names the brand. Better than + # dropping the row, and the validation gate will catch nonsense later. + first = (product_name or "").strip().split() + return first[0] if first else None + + +def stage_1_brand_and_fssai(row: Dict[str, Any]) -> Dict[str, Any]: + brand = row.get("brand") or "" + parent = resolve_parent_brand(brand) + row["brand"] = parent + row["brand_name"] = parent + row["_table"] = f"brand_{_sanitize_name(parent)}" + if _blank(row.get("fssai_license")): + # None means "not a food brand" as much as "unknown", so an empty + # column is a legitimate outcome, never an error. + row["fssai_license"] = get_fssai_license(parent) + return row + + +# --------------------------------------------------------------------------- +# Stage 2 - Row intake (replaces AI discovery) +# --------------------------------------------------------------------------- +def stage_2_row_intake(row: Dict[str, Any], *, use_llm: bool = True) -> Dict[str, Any]: + """The spreadsheet row *is* the product. Only fill a missing description. + + The LLM call is best-effort and optional: with Ollama unreachable the row + keeps an empty description and later stages still work. + """ + if not _blank(row.get("description")) or not use_llm: + return row + try: + from app.services.ollama_service import fetch_product_details + + details = fetch_product_details(row.get("brand", ""), row.get("product_name", "")) + if details and details.get("description"): + row["description"] = str(details["description"]).strip() + if _blank(row.get("size_variants")) and details and details.get("size_variants"): + row["size_variants"] = list(details["size_variants"]) + except Exception as exc: # noqa: BLE001 - enrichment is best-effort + logger.debug("LLM enrichment skipped for %r: %s", row.get("product_name"), exc) + return row + + +# --------------------------------------------------------------------------- +# Stage 3 - Title & category consistency +# --------------------------------------------------------------------------- +def stage_3_title_category(row: Dict[str, Any]) -> Dict[str, Any]: + title = row.get("title") or row.get("product_name") or "" + category = row.get("category") + + if _blank(category): + category = detect_category_from_text(f"{title} {row.get('description') or ''}") + row["_category_deterministic"] = bool(category) + if not category: + # "General" rather than None: the column is not nullable in + # practice and _to_storage_row already assumed this default. The + # row is still reported as un-categorised in the job warnings, so + # the signal is preserved rather than swallowed. + category = "General" + row.setdefault("_notes", []).append("category could not be detected; defaulted to General") + row["category"] = category + else: + row["_category_deterministic"] = True + + if category: + fixed, changed, removed = validate_and_fix_title( + title, category, brand=row.get("brand") + ) + row["title"] = fixed + if changed: + row.setdefault("_notes", []).append( + f"title corrected against category '{category}' (removed {removed})" + ) + if not _blank(row.get("description")): + row["description"] = sanitize_category_language(row["description"], category) + else: + row["title"] = title + return row + + +# --------------------------------------------------------------------------- +# Stage 4 - Pack-size explosion & unit safety +# --------------------------------------------------------------------------- +def _sizes_for(row: Dict[str, Any]) -> List[str]: + sizes = [str(s).strip() for s in (row.get("size_variants") or []) if str(s).strip()] + if sizes: + return sizes + match = _SIZE_IN_TITLE.search(row.get("product_name") or "") + if match: + return [match.group(0).strip()] + return list(price_estimator.default_size_variants( + row.get("category") or "", row.get("product_name") or "" + )) + + +def stage_4_explode_sizes(row: Dict[str, Any]) -> Tuple[List[Dict[str, Any]], List[str]]: + """One input row becomes one row per pack size. + + Barcodes and SKUs identify a specific pack, so they are only meaningful + once sizes are separate rows. Sizes whose unit makes no sense for the + category (a biscuit measured in cm) are dropped with a reason. + """ + out: List[Dict[str, Any]] = [] + dropped: List[str] = [] + category = row.get("category") or "" + + for size in _sizes_for(row)[:MAX_VARIANTS_PER_PRODUCT]: + fixed, rejected, reason = fix_or_reject_size(size, category, row.get("product_name") or "") + if rejected or not fixed: + dropped.append(f"{size}: {reason or 'invalid pack size'}") + continue + variant = dict(row) + variant["size"] = fixed + variant["size_variants"] = [fixed] + out.append(variant) + + if not out and not dropped: + variant = dict(row) + variant["size"] = "Standard" + variant["size_variants"] = ["Standard"] + out.append(variant) + return out, dropped + + +# --------------------------------------------------------------------------- +# Stage 5 - Pricing bands +# --------------------------------------------------------------------------- +def stage_5_pricing(row: Dict[str, Any]) -> Dict[str, Any]: + size = row.get("size") or "Standard" + if _blank(row.get("price_range")): + if not _blank(row.get("final_selling_price")): + price = float(row["final_selling_price"]) + lo, hi = int(round(price * 0.95)), int(round(price * 1.05)) + else: + lo, hi = price_estimator.estimate_price_range_for_size( + size, row.get("product_name") or "", row.get("brand") or "", + row.get("category") or "", + ) + row["price_range"] = f"₹{lo}-{hi}" + return row + + +# --------------------------------------------------------------------------- +# Stage 6 - Images +# --------------------------------------------------------------------------- +def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, Any]: + if not _blank(row.get("image_urls")) or not enabled: + return row + try: + from app.core.catalog_engine import catalog_engine + from app.services.image_search import find_all_image_urls + + candidates = find_all_image_urls( + row.get("product_name") or "", brand=row.get("brand"), max_results=24 + ) + if candidates: + best = catalog_engine._select_best_images( + candidates, row.get("product_name") or "", row.get("brand") or "", max_images=10 + ) + if best: + row["image_urls"] = list(best) + row["image_url"] = best[0] + except Exception as exc: # noqa: BLE001 - an image is not worth the row + logger.debug("Image search skipped for %r: %s", row.get("product_name"), exc) + return row + + +# --------------------------------------------------------------------------- +# Stage 7 - SKU +# --------------------------------------------------------------------------- +def stage_7_sku(row: Dict[str, Any]) -> Dict[str, Any]: + if not _blank(row.get("product_sku")): + return row + try: + resolved = resolve_product_sku( + row.get("brand") or "", row.get("product_name") or "", row.get("size") or "" + ) + row["product_sku"] = resolved.get("product_sku") + row["sku_source"] = resolved.get("sku_source") + except Exception as exc: # noqa: BLE001 + logger.debug("SKU resolution failed for %r: %s", row.get("product_name"), exc) + return row + + +# --------------------------------------------------------------------------- +# Stages 8 & 9 - barcode and HSN/GST, via the ported enrichment framework +# --------------------------------------------------------------------------- +async def stages_8_9_enrichment(rows: List[Dict[str, Any]], brand: str) -> List[Dict[str, Any]]: + """Both stages are `EnrichmentStage`s, so the ported EnrichmentPipeline + runs them with its own concurrency limit and never-raises guarantee. + Each disables itself via its settings flag, so this is a no-op when both + are off. + """ + stages = [] + if ENABLE_BARCODE_LOOKUP: + stages.append(BarcodeEnrichmentStage()) + if ENABLE_HSN_GST_ENRICHMENT: + stages.append(HsnGstEnrichmentStage()) + if not stages: + return rows + return await EnrichmentPipeline(stages).run(rows, brand) + + +# --------------------------------------------------------------------------- +# Stage 10 - validation gate +# --------------------------------------------------------------------------- +def stage_10_validate(rows: List[Dict[str, Any]], brand: str, *, images_checked: bool = True): + if not ENABLE_PRODUCT_VALIDATION: + return rows, [], {} + known = [bool(r.get("_category_deterministic")) for r in rows] + return validate_catalog(rows, brand, known_category_flags=known, + images_checked=images_checked) + + +# --------------------------------------------------------------------------- +# Stage 11 - embed & store +# --------------------------------------------------------------------------- +def build_image_id(brand: str, product_name: str, size: str) -> str: + """Deterministic, and unique per pack size. + + Matches `user_products.py`'s brand_slug + product_slug form so the two + upload paths agree, with the size folded in - without it every size of a + product collapses onto one row under ON CONFLICT (image_id). Deliberately + NOT s3_service.generate_image_id, which appends a random uuid4 and so can + never match an existing row. + """ + name = (product_name or "").strip() + # Store sheets usually carry the pack size inside the product name already + # ("Cadbury Dairy Milk 100g"). Appending it again would produce + # ..._100g_100g and, worse, a different id than the same product uploaded + # through the user-products path. + suffix = "" if _slugify(size) and _slugify(size) in _slugify(name) else f" {size}" + slug = _slugify(f"{name}{suffix}".strip()) + return f"{_sanitize_name(brand)}_{slug}" + + +def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]: + """Project an enriched row onto the brand-table columns.""" + name = row.get("product_name") or row.get("title") or "" + size = row.get("size") or "Standard" + display = name if size.lower() in name.lower() else f"{name} {size}".strip() + category = row.get("category") or "General" + description = row.get("description") or ( + f"{display} from {row.get('brand')}." + ) + return { + "product_name": display, + "title": row.get("title") or name, + "description": description, + "category": category, + "image_id": build_image_id(row.get("brand") or "", name, size), + "image_url": row.get("image_url"), + "image_urls": list(row.get("image_urls") or []), + "price_range": row.get("price_range"), + "size_variants": [size], + "providers": list(row.get("providers") or []), + "fssai_license": row.get("fssai_license"), + "product_sku": row.get("product_sku"), + "sku_source": row.get("sku_source"), + "hsn_code": row.get("hsn_code"), + "final_selling_price": row.get("final_selling_price"), + "selling_price": row.get("selling_price"), + "barcode": row.get("barcode"), + "barcode_type": row.get("barcode_type"), + "highlights": list(row.get("highlights") or []), + "nutrients": list(row.get("nutrients") or []), + "search_query": f"{row.get('brand')} {display} {category} {description}", + } + + +def _merge_with_existing(new: Dict[str, Any], existing: Dict[str, Any]) -> Tuple[Dict[str, Any], bool]: + """Keep the stored row, filling only the columns it left empty. + + Returns (row_to_write, changed). `changed` is False when the stored row + was already complete, which is what makes re-uploading the same file a + no-op. + """ + merged = dict(existing) + changed = False + for key, value in new.items(): + if key in ("image_id",): + continue + if _blank(existing.get(key)) and not _blank(value): + merged[key] = value + changed = True + merged["image_id"] = new["image_id"] + return merged, changed + + +def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResult) -> None: + """Embed and upsert, splitting inserts from backfills. + + `cleanup=False` is load-bearing: cleanup=True deletes every row in the + table that is not in this batch, which for a 3-row store file would wipe + the brand's entire catalog. + """ + if not rows: + return + + try: + existing_by_id = { + r.get("image_id"): r + for r in (get_products_by_brand(brand) or []) + if r.get("image_id") + } + except Exception as exc: # noqa: BLE001 - treat as "nothing stored yet" + logger.warning("Could not read existing products for %s: %s", brand, exc) + existing_by_id = {} + + to_write: List[Dict[str, Any]] = [] + for row in rows: + prior = existing_by_id.get(row["image_id"]) + if prior is None: + to_write.append(row) + result.inserted += 1 + continue + merged, changed = _merge_with_existing(row, prior) + if changed: + to_write.append(merged) + result.backfilled += 1 + else: + result.skipped_existing += 1 + + if not to_write: + return + + if USE_EMBEDDINGS: + try: + vectors = embed_texts([r.get("search_query") or r["product_name"] for r in to_write]) + for row, vector in zip(to_write, vectors): + row["embedding"] = vector + except Exception as exc: # noqa: BLE001 - a row without a vector is + # still a usable catalog row; it just won't match semantic search. + logger.warning("Embedding failed for %s: %s", brand, exc) + + try: + upsert_brand_products(brand, to_write, cleanup=False) + except Exception as exc: # noqa: BLE001 + result.storage_error = str(exc) + result.inserted = 0 + result.backfilled = 0 + logger.exception("Storing store-catalog rows failed for %s", brand) + + +# --------------------------------------------------------------------------- +# Orchestration +# --------------------------------------------------------------------------- +def parse_spreadsheet(filename: str, content: bytes): + """Parse an upload into (DataFrame, column mapping). Raises ValueError.""" + df = read_products_dataframe(filename, content) + mapping = map_spreadsheet_columns(df.columns) + return df, mapping + + +def run_pipeline( + filename: str, + content: bytes, + *, + progress: ProgressFn = _noop_progress, + use_llm: bool = True, + fetch_images: bool = True, +) -> PipelineResult: + """Run all 11 stages over an uploaded store spreadsheet. + + Synchronous by design - it is called on a background daemon thread by the + router, matching how every other long job in this project runs. + """ + result = PipelineResult() + df, mapping = parse_spreadsheet(filename, content) + result.recognised_columns = {f: str(c) for f, c in mapping.columns.items()} + result.unrecognised_columns = list(mapping.unrecognised) + records = df.to_dict(orient="records") + result.rows_total = len(records) + + # A store sheet frequently has no brand column; the brand lives inside the + # product name. Rather than fork row_to_request (a working, tested function + # this feature does not touch), advertise a synthetic brand column and fill + # it per row from the inferred value. + brand_column_supplied = "brand" in mapping.columns + if not brand_column_supplied: + mapping.columns["brand"] = _INFERRED_BRAND_COL + + # ---- stages 1-3, per uploaded row -------------------------------------- + prepared: List[Dict[str, Any]] = [] + for position, record in enumerate(records): + row_no = position + 2 # header occupies row 1 + + raw_name = _text(record, mapping, "product_name") or _text(record, mapping, "title") or "" + if not brand_column_supplied: + record[_INFERRED_BRAND_COL] = infer_brand(raw_name) or "" + + try: + req = row_to_request(record, mapping) + except ValueError as exc: + result.errors.append(RowError(row_no, raw_name, str(exc))) + continue + + row = req.model_dump() + row["_row"] = row_no + prepared.append(row) + + progress(1, STAGE_NAMES[0], 0, len(prepared)) + prepared = [stage_1_brand_and_fssai(r) for r in prepared] + + for index, row in enumerate(prepared, start=1): + stage_2_row_intake(row, use_llm=use_llm) + progress(2, STAGE_NAMES[1], index, len(prepared)) + + progress(3, STAGE_NAMES[2], 0, len(prepared)) + prepared = [stage_3_title_category(r) for r in prepared] + + # ---- stage 4, one row becomes many ------------------------------------- + exploded: List[Dict[str, Any]] = [] + for row in prepared: + variants, dropped = stage_4_explode_sizes(row) + exploded.extend(variants) + for reason in dropped: + result.warnings.append(f"row {row.get('_row')}: dropped pack size {reason}") + for note in row.get("_notes") or []: + result.warnings.append(f"row {row.get('_row')}: {note}") + progress(4, STAGE_NAMES[3], len(exploded), len(exploded)) + result.products_built = len(exploded) + + # ---- stages 5-7, per exploded row -------------------------------------- + for index, row in enumerate(exploded, start=1): + stage_5_pricing(row) + progress(5, STAGE_NAMES[4], index, len(exploded)) + for index, row in enumerate(exploded, start=1): + stage_6_images(row, enabled=fetch_images) + progress(6, STAGE_NAMES[5], index, len(exploded)) + for index, row in enumerate(exploded, start=1): + stage_7_sku(row) + progress(7, STAGE_NAMES[6], index, len(exploded)) + + # ---- stages 8-11, grouped by destination brand ------------------------- + by_brand: Dict[str, List[Dict[str, Any]]] = {} + for row in exploded: + by_brand.setdefault(row["brand"], []).append(row) + result.brands = sorted(display_name_for_suffix(_sanitize_name(b)) for b in by_brand) + + for brand, rows in by_brand.items(): + progress(8, STAGE_NAMES[7], 0, len(rows)) + try: + rows = asyncio.run(stages_8_9_enrichment(rows, brand)) + except Exception as exc: # noqa: BLE001 - enrichment must not lose rows + logger.warning("Barcode/HSN enrichment failed for %s: %s", brand, exc) + progress(9, STAGE_NAMES[8], len(rows), len(rows)) + + kept, rejected, _summary = stage_10_validate(rows, brand, images_checked=fetch_images) + result.rejected += len(rejected) + for bad in rejected: + result.rejections.append({ + "product_name": bad.get("product_name") or bad.get("title"), + "size": bad.get("size"), + "reason": "; ".join( + i.get("message", "") if isinstance(i, dict) else str(i) + for i in (bad.get("validation_issues") or []) + ) or "failed the validation gate", + }) + progress(10, STAGE_NAMES[9], len(kept), len(rows)) + + storage_rows = [_to_storage_row(r) for r in kept] + # A single sheet can name the same pack twice; last one wins, so the + # batch never presents two rows with the same image_id to the upsert. + deduped: Dict[str, Dict[str, Any]] = {r["image_id"]: r for r in storage_rows} + stage_11_store(list(deduped.values()), brand, result) + progress(11, STAGE_NAMES[10], len(deduped), len(deduped)) + + return result diff --git a/app/infrastructure/settings.py b/app/infrastructure/settings.py index ba54b37..7c6f0a7 100644 --- a/app/infrastructure/settings.py +++ b/app/infrastructure/settings.py @@ -185,6 +185,49 @@ USE_PLAYWRIGHT_FALLBACK = _bool("USE_PLAYWRIGHT_FALLBACK", "true") # out 1x1 tracking pixels / broken placeholder images) MIN_IMAGE_BYTES = int(os.getenv("MIN_IMAGE_BYTES", "3000")) +# --------------------------------------------------------------------------- +# Store-catalog enrichment pipeline (app/core/store_catalog_pipeline.py) +# --------------------------------------------------------------------------- +# Stages 7-10 of the store-Excel ingestion pipeline. Ported from the sibling +# Universal_Catalog_Barcode_Enrichment project along with the code that reads +# them; the names are kept identical so the ported modules need no edits. +# +# The three network-touching flags default to FALSE here, unlike in the +# sibling. This backend serves an interactive API on a shared 8GB host, and a +# 2000-row upload with web lookups on would fire thousands of outbound +# requests. Turn them on deliberately, per environment. +ENABLE_SKU_WEB_LOOKUP = _bool("ENABLE_SKU_WEB_LOOKUP", "false") +ENABLE_BARCODE_LOOKUP = _bool("ENABLE_BARCODE_LOOKUP", "false") +ENABLE_MANUFACTURER_SITE_LOOKUP = _bool("ENABLE_MANUFACTURER_SITE_LOOKUP", "false") + +# Offline/deterministic stages - safe to leave on. +ENABLE_HSN_GST_ENRICHMENT = _bool("ENABLE_HSN_GST_ENRICHMENT", "true") +ENABLE_PRODUCT_VALIDATION = _bool("ENABLE_PRODUCT_VALIDATION", "true") + +# Validation gate thresholds: below REJECT the row is dropped, below REVIEW it +# is stored but flagged `validation_status="review"`. +VALIDATION_REJECT_THRESHOLD = float(os.getenv("VALIDATION_REJECT_THRESHOLD", "0.35")) +VALIDATION_REVIEW_THRESHOLD = float(os.getenv("VALIDATION_REVIEW_THRESHOLD", "0.70")) + +# Pack-size explosion: how many size rows one uploaded product may become. +MAX_VARIANTS_PER_PRODUCT = int(os.getenv("MAX_VARIANTS_PER_PRODUCT", "6")) +ENABLE_PER_VARIANT_IMAGES = _bool("ENABLE_PER_VARIANT_IMAGES", "false") +PER_VARIANT_IMAGE_MAX_RESULTS = int(os.getenv("PER_VARIANT_IMAGE_MAX_RESULTS", "10")) + +# Barcode lookup tuning. The cache TTL is long (30 days) because a GTIN for a +# given pack size does not change, and negative results are cached too. +BARCODE_LOOKUP_TIMEOUT_SECONDS = float(os.getenv("BARCODE_LOOKUP_TIMEOUT_SECONDS", "10")) +BARCODE_LOOKUP_CACHE_TTL_SECONDS = float(os.getenv("BARCODE_LOOKUP_CACHE_TTL_SECONDS", str(30 * 24 * 3600))) +BARCODE_LOOKUP_MAX_CONCURRENCY = int(os.getenv("BARCODE_LOOKUP_MAX_CONCURRENCY", "5")) +BARCODE_COUNTRY_TAG = os.getenv("BARCODE_COUNTRY_TAG", "india") + +# Optional barcode source credentials. Each source disables itself when its +# key is blank, so leaving these unset simply narrows the lookup cascade. +GS1_INDIA_API_BASE_URL = os.getenv("GS1_INDIA_API_BASE_URL", "") +GS1_INDIA_API_KEY = os.getenv("GS1_INDIA_API_KEY", "") +UPC_DATABASE_API_KEY = os.getenv("UPC_DATABASE_API_KEY", "") + + # --------------------------------------------------------------------------- # HTTP client defaults # --------------------------------------------------------------------------- diff --git a/app/main.py b/app/main.py index d39e5dd..43b5822 100644 --- a/app/main.py +++ b/app/main.py @@ -25,6 +25,7 @@ from app.api.routers import health, brands, search, chat, catalog, system from app.api.routers import stores, discounts, analytics as store_analytics, trending, recommendations, store_admin from app.api.routers import nutrition, nutrition_admin, upload from app.api.routers import auth, user_products, admin_train, mcp_info +from app.api.routers import store_catalog from app.services.store_db import ensure_store_intelligence_schema from app.services.nutrition_db import ensure_nutrition_schema @@ -203,6 +204,7 @@ app.include_router(store_admin.router, prefix="/api") app.include_router(nutrition.router, prefix="/api") app.include_router(nutrition_admin.router, prefix="/api") app.include_router(upload.router, prefix="/api") +app.include_router(store_catalog.router, prefix="/api") app.include_router(mcp_info.router, prefix="/api") # MCP lives outside /api on purpose: it is a protocol endpoint for AI clients, diff --git a/app/services/brand_registry.py b/app/services/brand_registry.py index 4e16595..6b5bc59 100644 --- a/app/services/brand_registry.py +++ b/app/services/brand_registry.py @@ -1,5 +1,6 @@ import re from functools import lru_cache +from typing import Dict, Optional BRAND_ALIASES = { # Cadbury family @@ -313,3 +314,59 @@ def get_known_sub_brands(brand: str) -> list[str]: known.append(rest) seen.add(rest) return known + + +# --------------------------------------------------------------------------- +# FSSAI licences (stage 1 of the store-catalog pipeline) +# --------------------------------------------------------------------------- +# Ported from the sibling Universal_Catalog_Barcode_Enrichment project. Purely +# additive: BRAND_ALIASES and resolve_parent_brand are untouched. +# +# Food brands only. A miss means "not a food brand" (P&G, Colgate-Palmolive, +# J&J, Reckitt, Godrej) just as much as it means "unknown", so callers must +# treat None as "leave the column empty", never as an error. +FSSAI_LICENSES: Dict[str, str] = { + "britannia": "10012022000103", + "pepsico": "10012031000047", + "amul": "10012021000243", + "cadbury": "10014022002711", + "hindustan unilever": "10012022000217", + "nestle": "10012011000168", + "itc": "10018042000305", + "coca-cola": "10012042000424", + "tata": "12414003000511", + "parle": "10012022000046", + "marico": "10012022000258", + "dabur": "10012011000084", + "hatsun": "10012042000071", + "milky mist": "10017042003191", + "aachi": "10014042000577", + "sakthi": "10012042000300", + "kaleesuwari": "10012042000302", + "idhayam": "10012042000109", + "cavinkare": "10013042000366", + "naga": "10012042000192", + "manna": "10014042000169", + "grb": "10012042000148", + "anil": "10012042000213", + "lion dates": "10012042000244", + "brooke bond": "10013022001897", + "mother dairy": "10012011000015", + "haldiram": "10012011000140", + "fortune": "10012021000071", + "paper boat": "10012043000083", + "bisk farm": "10012031000012", + "mtr": "10012043000058", + "everest": "10012022000526", + "mdh": "10012011000062", +} + + +def get_fssai_license(brand: str) -> Optional[str]: + """Return the 14-digit FSSAI licence number for a food brand, or None. + + Resolves to the canonical parent first, so sub-brands (e.g. 'hul lux') + inherit the parent's licence and every row in a brand table agrees. + """ + canonical = resolve_parent_brand(brand).lower().strip() + return FSSAI_LICENSES.get(canonical) diff --git a/app/services/category_units.py b/app/services/category_units.py new file mode 100644 index 0000000..f4ce3ef --- /dev/null +++ b/app/services/category_units.py @@ -0,0 +1,338 @@ +""" +Category -> Allowed Pack-Size-Unit Rules +========================================== + +WHY THIS FILE EXISTS +--------------------- +Reported bug: brand="Britannia" produced catalog rows like +"Tiger 10cm", "Nutrichoice 12cm", "Marie Gold 15cm" and +"Britannia Milk Bikis - 200ml, 500ml, 1L". The small local LLM +(qwen2.5:1.5b) occasionally predicts a *dimension* unit (cm/inch/m) or the +*wrong physical-state* unit (ml/L for a solid biscuit) instead of a +plausible FMCG pack-size unit, because nothing anywhere in the pipeline +ever told it - or checked afterwards - which units are even legal for a +given product category. + +Two existing point-fixes come close but don't cover this: + +- `price_estimator._parse_size_to_grams()` parses a size string into a + gram/ml-equivalent float for pricing math. It has an explicit fallback + for *unrecognised* unit tokens (case: "cm" is not in its litre/kg/base + unit sets): "leave the numeric value as-is rather than guessing". That + is the correct, conservative choice for a *pricing* function - but it + means "10cm" silently becomes 10.0 (treated as if it were 10 grams), + which is exactly wrong for a *validation* function: a bogus unit must + never be treated as an implicitly-valid one. +- `price_estimator.normalize_size_unit()` swaps ml<->g labels for + categories it already knows are exclusively solid or exclusively liquid + - but it only recognises `ml/l` <-> `g/kg` confusion. It has no concept + of "cm" at all, so a dimension unit passes through completely + unexamined. + +This module is the missing, single source of truth for "what pack-size +units are even legal for this product category", used in two places: + +1. GENERATION TIME (app/services/ollama_service.py): to tell the LLM, as + part of the prompt, exactly which units are allowed for the category it + is currently enumerating products for - so the wrong unit is far less + likely to be generated in the first place. +2. VALIDATION TIME (app/core/catalog_engine.py, app/services/ + product_validator.py): to reject/replace a generated size string whose + unit contradicts its resolved category, deterministically and without + an LLM call, as a defence-in-depth backstop for whatever still gets + through the prompt-level fix (e.g. the `fetch_brand_catalog_with_gemini` + fallback path, or a cached/registry-sourced product). + +Per the project's own stated requirement: package size must be determined +from the PRODUCT CATEGORY, and dimension units (cm/inch/m) must never be +generated for an FMCG consumable unless the category genuinely represents +a physical dimension (none currently in this catalog do). +""" +from __future__ import annotations + +import re +from typing import Optional + +# --------------------------------------------------------------------------- +# Unit vocabularies +# --------------------------------------------------------------------------- +WEIGHT_UNITS: set[str] = { + "g", "gm", "gms", "gram", "grams", + "kg", "kgs", "kilo", "kilos", "kilogram", "kilograms", +} +VOLUME_UNITS: set[str] = { + "ml", "mls", "millilitre", "millilitres", "milliliter", "milliliters", + "l", "lt", "ltr", "ltrs", "litre", "litres", "liter", "liters", +} +# Count-based packs (stationery, tablets, diapers, agarbatti sticks, etc.) - +# legitimate for a handful of non-food categories this catalog also covers. +COUNT_UNITS: set[str] = { + "pcs", "pc", "piece", "pieces", "unit", "units", "tablet", "tablets", + "capsule", "capsules", "strip", "strips", "sheet", "sheets", "roll", + "rolls", "stick", "sticks", "count", "ct", "pack", "packs", +} +# Physical-dimension units. These are NEVER a valid FMCG/personal-care pack +# size - no biscuit, chocolate, milk, tea, coffee, or ice-cream product is +# ever sold "by the centimetre". Kept as an explicit set (rather than "any +# unit we don't recognise") so the reject reason can name the exact problem. +DIMENSION_UNITS: set[str] = { + "cm", "centimetre", "centimetres", "centimeter", "centimeters", + "mm", "millimetre", "millimetres", "millimeter", "millimeters", + "m", "metre", "metres", "meter", "meters", + "inch", "inches", "in", + "ft", "feet", "foot", +} + +_KNOWN_UNIT_TOKENS: set[str] = WEIGHT_UNITS | VOLUME_UNITS | COUNT_UNITS | DIMENSION_UNITS + +# --------------------------------------------------------------------------- +# Category -> allowed unit TYPE ("weight" / "volume" / "weight_or_volume" / +# "count" / "unknown"). Keyed by lowercase, human-readable category strings - +# the same vocabulary produced by app.services.brand_registry. +# SUB_BRAND_CATEGORY and used as `known_category`/`category_value` throughout +# catalog_engine.py. This is deliberately the AUTHORITATIVE per-category +# rulebook the user specified: Biscuits -> g/kg, Milk/Juices -> ml/L, +# Chocolate -> g, Tea -> g, Coffee -> g, Ice Cream -> ml/L - extended to +# every other category already present in this project so every catalog row +# gets a consistent check, not just the categories in the bug report. +# --------------------------------------------------------------------------- +CATEGORY_UNIT_TYPE: dict[str, str] = { + # --- Weight-only (solid FMCG) --- + "biscuits & cookies": "weight", + "crackers": "weight", + "rusk": "weight", + "cakes & muffins": "weight", + "bakery & breads": "weight", + "snacks": "weight", + "chocolates": "weight", + "candy & confectionery": "weight", + "atta & staples": "weight", + "spices & masalas": "weight", + "pasta & noodles": "weight", + "noodles & instant food": "weight", + "breakfast cereal": "weight", + "dry fruits & nuts": "weight", + "health foods": "weight", + "millets": "weight", + "salt & staples": "weight", + "tea & coffee": "weight", + "tea": "weight", + "coffee": "weight", + "health drinks": "weight", # malted/powder drinks: Horlicks, Bournvita, Boost, Complan + "bath soap": "weight", + "stationery": "count", + "household - agarbatti": "count", + "feminine hygiene": "count", + "baby care": "weight_or_volume", + # --- Volume-only (liquid FMCG) --- + "beverages": "volume", + "juices": "volume", + "food - soups & sauces": "volume", + "household cleaning": "volume", + "dishwash": "volume", + "ice cream": "volume", + "cooking oils": "volume", + "wellness oils": "volume", + "household - lamp oil": "volume", + # --- Mixed weight-or-volume (category legitimately spans both solid and + # liquid sub-products, e.g. "Dairy" covers milk (ml) AND + # cheese/paneer/butter/curd (g)) --- + "dairy": "weight_or_volume", + "dairy - desserts": "weight_or_volume", + "hair care": "weight_or_volume", # shampoo/conditioner (ml) vs hair oil/cream (g/ml) + "skin & bath care": "weight_or_volume", + "skin care": "weight_or_volume", + "beauty care": "weight_or_volume", + "fragrance & deodorants": "weight_or_volume", + "men's grooming": "weight_or_volume", + "detergents & fabric care": "weight_or_volume", # powder (kg) vs liquid (L) + "oral care": "weight_or_volume", # toothpaste (g) vs mouthwash (ml) + "personal care": "weight_or_volume", + "personal care - mosquito repellent": "weight_or_volume", + "mosquito repellent": "weight_or_volume", + "air freshener": "weight_or_volume", + "health care - cold & cough": "weight_or_volume", + "health care - digestive": "weight_or_volume", + "health care - ayurvedic": "weight_or_volume", + "health care - antiseptic": "weight_or_volume", + "health care - first aid": "count", + "pickles & chutneys": "weight", + "food - spreads": "weight", + "food - mixes": "weight", + "rice & pulses": "weight", + "sweets": "weight", + "ready to eat": "weight_or_volume", +} + +# No category in this FMCG/personal-care catalog is legitimately sold "by +# the centimetre" - kept as an explicit, extensible override point (e.g. a +# future "Home Textiles" or "Furnishings" category) rather than a hardcoded +# blanket rule, per the requirement that dimension units are only allowed +# "if the product category genuinely represents physical dimensions". +DIMENSION_OK_CATEGORIES: set[str] = set() + + +def get_unit_type(category: Optional[str]) -> str: + """Return the allowed unit TYPE for `category` ('weight', 'volume', + 'weight_or_volume', 'count', or 'unknown' if the category isn't in our + rulebook - callers should treat 'unknown' permissively, not as a + rejection, since it just means we have no opinion yet, not that + anything is wrong).""" + if not category: + return "unknown" + return CATEGORY_UNIT_TYPE.get(category.strip().lower(), "unknown") + + +def get_allowed_units(category: Optional[str]) -> set[str]: + """Concrete set of unit tokens allowed for `category`.""" + unit_type = get_unit_type(category) + if unit_type == "weight": + return set(WEIGHT_UNITS) + if unit_type == "volume": + return set(VOLUME_UNITS) + if unit_type == "weight_or_volume": + return WEIGHT_UNITS | VOLUME_UNITS + if unit_type == "count": + return set(COUNT_UNITS) + # unknown category: permissive - anything except a dimension unit + return WEIGHT_UNITS | VOLUME_UNITS | COUNT_UNITS + + +def allowed_units_hint(category: Optional[str]) -> str: + """Short, human-readable allowed-units phrase for `category`, suitable + for embedding directly into an LLM prompt, e.g. 'grams (g) or + kilograms (kg)'. Used by ollama_service.py to tell the model, per + category, exactly which units it must use.""" + unit_type = get_unit_type(category) + return { + "weight": "grams (g) or kilograms (kg) ONLY", + "volume": "millilitres (ml) or litres (L) ONLY", + "weight_or_volume": "grams/kilograms (g/kg) for solid items or millilitres/litres (ml/L) for liquid items", + "count": "a piece/unit count (e.g. '10 pcs', '1 pack')", + "unknown": "grams (g), kilograms (kg), millilitres (ml), or litres (L) as appropriate", + }[unit_type] + + +def is_forbidden_dimension_unit(unit: Optional[str], category: Optional[str] = None) -> bool: + """True if `unit` is a physical-dimension unit (cm/inch/m/...) that is + never valid for `category` (see DIMENSION_OK_CATEGORIES).""" + if not unit: + return False + u = unit.strip().lower() + if u not in DIMENSION_UNITS: + return False + if category and category.strip().lower() in DIMENSION_OK_CATEGORIES: + return False + return True + + +def parse_unit(size: Optional[str]) -> tuple[Optional[float], Optional[str]]: + """Extract (numeric_value, unit_token) from a size string like '200g', + '1.5 L', '10cm'. Returns (None, None) if no leading number is found.""" + if not size: + return None, None + s = size.strip().lower() + match = re.search(r"(\d+(?:\.\d+)?)\s*([a-z]*)", s) + if not match or not match.group(1): + return None, None + value = float(match.group(1)) + unit = match.group(2).strip() or None + return value, unit + + +def validate_unit_for_category(size: Optional[str], category: Optional[str]) -> tuple[bool, Optional[str]]: + """Check whether `size`'s unit is legal for `category`. + + Returns (is_valid, reason). `reason` is populated whenever is_valid is + False, explaining exactly what was wrong (dimension unit vs. + wrong-physical-state unit vs. unrecognised token) so callers can log or + surface it in an audit trail. + """ + if not size or not size.strip(): + return False, "size is missing" + value, unit = parse_unit(size) + if value is None: + # No parseable number at all (e.g. "Family Pack") - not this + # function's concern; price_estimator's own fallback handles it. + return True, None + if not unit: + # Bare number, no unit token (e.g. "200") - ambiguous but not a + # unit-category contradiction per se; leave to size-plausibility + # checks elsewhere. + return True, None + + if is_forbidden_dimension_unit(unit, category): + return False, ( + f"unit '{unit}' is a physical dimension, not a valid FMCG pack-size " + f"unit for category '{category}'" + ) + + if unit not in _KNOWN_UNIT_TOKENS: + # Unrecognised token (e.g. "pcs" variants we don't track, "x6", + # brand-specific packaging words) - not a contradiction we can + # prove, so don't reject. + return True, None + + allowed = get_allowed_units(category) + if unit not in allowed: + unit_type = get_unit_type(category) + return False, ( + f"unit '{unit}' is not valid for category '{category}' " + f"(expects {unit_type.replace('_', ' ')} units: {allowed_units_hint(category)})" + ) + return True, None + + +def fix_or_reject_size( + size: Optional[str], category: Optional[str], product_title: str = "" +) -> tuple[Optional[str], bool, Optional[str]]: + """Authoritative size-unit correction, called at both generation time + and validation time. + + Returns (corrected_size_or_None, was_changed, reason): + - If `size`'s unit is already valid for `category`: (size, False, None). + - If `size` uses the WRONG PHYSICAL STATE for `category` (e.g. '200ml' + for a strictly-solid category) but the number itself is plausible: the + unit label is swapped (numeric value preserved), same behaviour as + price_estimator.normalize_size_unit(), since a mislabelled-but- + plausible quantity is safely recoverable. + - If `size` uses a DIMENSION unit (cm/inch/m) or the unit can't be + meaningfully reinterpreted for the category: there is no safe + conversion (10cm has no defensible gram-equivalent), so this returns + (None, True, reason) - the caller MUST NOT keep the original value and + should substitute a category-appropriate default size instead (see + price_estimator.default_size_variants()), never silently reuse the + rejected number under a new label. + """ + if not size or not size.strip(): + return size, False, None + + value, unit = parse_unit(size) + if value is None or not unit: + return size, False, None + + ok, reason = validate_unit_for_category(size, category) + if ok: + return size, False, None + + if is_forbidden_dimension_unit(unit, category): + # No defensible conversion exists - reject outright. + return None, True, reason + + # Wrong-physical-state-but-plausible-number case: swap the label, + # preserving the numeric value, mirroring the existing + # price_estimator.normalize_size_unit() behaviour for ml<->g. + unit_type = get_unit_type(category) + value_str = ( + str(int(value)) if float(value).is_integer() else str(value) + ) + if unit_type == "weight" and unit in VOLUME_UNITS: + corrected = f"{value_str}g" if unit in {"ml", "mls"} else f"{value_str}kg" + return corrected, True, reason + if unit_type == "volume" and unit in WEIGHT_UNITS: + corrected = f"{value_str}ml" if unit in {"g", "gm", "gms", "gram", "grams"} else f"{value_str}L" + return corrected, True, reason + + # Anything else we can't confidently repair (e.g. a count-unit category + # given a weight unit) - reject rather than guess. + return None, True, reason diff --git a/app/services/enrichment/__init__.py b/app/services/enrichment/__init__.py new file mode 100644 index 0000000..98c1c3f --- /dev/null +++ b/app/services/enrichment/__init__.py @@ -0,0 +1,50 @@ +""" +Product Enrichment Service +=========================== + +WHY THIS PACKAGE EXISTS +------------------------ +Everything upstream of this package (ollama_service.py, product_validator.py, +sku_service.py, image_search.py, ...) either GENERATES a product row or +JUDGES whether an already-generated row is plausible. None of it ever goes +out and fetches a piece of ground-truth data from an authoritative external +source and attaches it to the row. That is what this package is for. + +The first concrete enricher is barcode lookup (see `barcode/`): resolving a +real, checksum-valid EAN-13/UPC/GTIN for a catalog row from trusted external +sources, never from the LLM. The package is deliberately structured so this +is ONE stage among what will eventually be several independent ones (HSN +code, GST rate, nutrition, allergens, manufacturer details, ...) - see +`base.py` for the `EnrichmentStage` contract every future stage implements, +and `pipeline.py` for the orchestrator that runs them in sequence. + +DESIGN PRINCIPLES (apply to every enrichment stage, present and future) +------------------------------------------------------------------------- +- NEVER FABRICATE. If a stage cannot find authentic, verifiable data, it + writes NULL/None + a "not_found" status - never a guessed or LLM- + generated value. This mirrors the whole point of `product_validator.py` + and `sku_service.py`'s "real ID or clearly-labelled internal fallback" + design, taken one step further: for barcodes there IS no safe synthetic + fallback (a fabricated barcode is actively harmful - it can collide with + a real product), so the fallback is always NULL, never a generated value. +- NEVER RAISE. A single product's enrichment failure (timeout, malformed + response, source outage) must never abort the batch or the pipeline run. + Every stage catches its own errors and degrades to "not found" for that + one row. +- ADDITIVE. This package has no dependents before this change and does not + modify the behaviour of any existing module; it is only ever imported + from new call sites (see `app/core/catalog_engine.py`'s Step 2.6 and + `app/services/vector_store.py`'s additive barcode columns). +- INDEPENDENT STAGES. Each stage owns its own external calls, matching + rules, caching, and DB columns. A future HSN/GST/nutrition stage does not + need to know barcode lookup exists, and vice versa - see `pipeline.py`. +""" +from app.services.enrichment.base import EnrichmentStage, StageOutcome +from app.services.enrichment.pipeline import EnrichmentPipeline, run_default_pipeline + +__all__ = [ + "EnrichmentStage", + "StageOutcome", + "EnrichmentPipeline", + "run_default_pipeline", +] diff --git a/app/services/enrichment/barcode/__init__.py b/app/services/enrichment/barcode/__init__.py new file mode 100644 index 0000000..71f2989 --- /dev/null +++ b/app/services/enrichment/barcode/__init__.py @@ -0,0 +1,47 @@ +""" +Barcode Retrieval & Product Enrichment module. + +Public entry points: + lookup_barcode(brand, product_title, size, category="") -> BarcodeResult + Synchronous single-product cascading lookup. Safe to call from a + script, notebook, or the Streamlit UI's "look up a barcode" action. + + batch_lookup_barcodes(products, brand) -> list[BarcodeResult] + Async batch lookup used by the pipeline stage (see `stage.py`) and + available directly for a CLI/manual batch job. + +See the package's module docstrings for the full architecture: + models.py - BarcodeCandidate / BarcodeResult / enums + validators.py - EAN-13/GTIN/UPC checksum + format validation + matching.py - exact product matching rules + cache.py - local SQLite lookup cache + retry.py - tenacity-based retry/backoff + sources/ - the 4-tier cascading source registry + service.py - orchestrates cache -> sources -> validate -> match + stage.py - adapts the service to the generic EnrichmentStage + contract used by app/services/enrichment/pipeline.py +""" +from typing import Any, Dict, List + +from app.services.enrichment.barcode.models import BarcodeCandidate, BarcodeResult, BarcodeType, LookupStatus +from app.services.enrichment.barcode.service import BarcodeLookupService, get_default_service + + +def lookup_barcode(brand: str, product_title: str, size: str, category: str = "") -> BarcodeResult: + return get_default_service().lookup_one(brand, product_title, size, category) + + +async def batch_lookup_barcodes(products: List[Dict[str, Any]], brand: str) -> List[BarcodeResult]: + return await get_default_service().batch_lookup(products, brand) + + +__all__ = [ + "BarcodeCandidate", + "BarcodeResult", + "BarcodeType", + "LookupStatus", + "BarcodeLookupService", + "get_default_service", + "lookup_barcode", + "batch_lookup_barcodes", +] diff --git a/app/services/enrichment/barcode/cache.py b/app/services/enrichment/barcode/cache.py new file mode 100644 index 0000000..69f40bf --- /dev/null +++ b/app/services/enrichment/barcode/cache.py @@ -0,0 +1,127 @@ +""" +Local lookup cache for barcode results. + +WHY A SEPARATE SQLITE FILE (data/cache/barcode_lookup.db) INSTEAD OF +POSTGRES +--------------------------------------------------------------------- +This cache exists purely to avoid repeating the SAME external-API search +(GS1/Open Food Facts/UPCItemDB/manufacturer site) twice for the same +(brand, product, size) - see "Cache barcode lookups to avoid repeated API +calls" / "Prevent duplicate barcode searches" (Performance Requirements). +It is deliberately NOT the authoritative store (that is Postgres, via +`vector_store.py`'s additive barcode columns) - it needs to work even when +Postgres is unreachable/USE_PGVECTOR=false, and it needs to be cheap and +local on an 8GB-RAM/CPU-only machine, so plain stdlib `sqlite3` (no new +dependency, no server process) is the right tool here, following the same +"file-backed, atomic-write" spirit as `sku_service.py`'s SKU sequence +counter. + +A row's cache key is a normalized (brand, product_title, size) triple, not +the barcode itself (we're caching "what did we already look up", not "what +maps to what"). +""" +from __future__ import annotations + +import json +import logging +import re +import sqlite3 +import threading +import time +from pathlib import Path +from typing import Optional + +from app.services.enrichment.barcode.models import BarcodeResult + +logger = logging.getLogger(__name__) + +_DB_PATH = Path("data") / "cache" / "barcode_lookup.db" +_lock = threading.Lock() +_initialized = False + + +def _normalize_key_part(text: Optional[str]) -> str: + return re.sub(r"\s+", " ", (text or "").strip().lower()) + + +def cache_key(brand: str, product_title: str, size: str) -> str: + return f"{_normalize_key_part(brand)}|{_normalize_key_part(product_title)}|{_normalize_key_part(size)}" + + +def _connect() -> sqlite3.Connection: + _DB_PATH.parent.mkdir(parents=True, exist_ok=True) + conn = sqlite3.connect(str(_DB_PATH), timeout=10) + conn.execute("PRAGMA journal_mode=WAL") + return conn + + +def _ensure_schema(conn: sqlite3.Connection) -> None: + global _initialized + if _initialized: + return + conn.execute( + """ + CREATE TABLE IF NOT EXISTS barcode_lookup_cache ( + cache_key TEXT PRIMARY KEY, + brand TEXT, + product_title TEXT, + size TEXT, + result_json TEXT NOT NULL, + created_at REAL NOT NULL + ) + """ + ) + conn.commit() + _initialized = True + + +def get_cached(brand: str, product_title: str, size: str, ttl_seconds: float) -> Optional[BarcodeResult]: + """Returns a cached BarcodeResult if present and not older than + `ttl_seconds`, else None. Never raises - a cache read failure is + treated exactly like a cache miss.""" + key = cache_key(brand, product_title, size) + try: + with _lock, _connect() as conn: + _ensure_schema(conn) + row = conn.execute( + "SELECT result_json, created_at FROM barcode_lookup_cache WHERE cache_key = ?", + (key,), + ).fetchone() + except Exception as e: + logger.debug(f"Barcode cache read failed for '{key}': {e}") + return None + + if not row: + return None + result_json, created_at = row + if ttl_seconds > 0 and (time.time() - created_at) > ttl_seconds: + return None + try: + data = json.loads(result_json) + return BarcodeResult(**data) + except Exception as e: + logger.debug(f"Barcode cache entry for '{key}' unreadable, treating as miss: {e}") + return None + + +def set_cached(brand: str, product_title: str, size: str, result: BarcodeResult) -> None: + """Best-effort write - a failure here never blocks the lookup itself, + it just means this row won't benefit from caching next time.""" + key = cache_key(brand, product_title, size) + try: + payload = json.dumps(result.__dict__) + with _lock, _connect() as conn: + _ensure_schema(conn) + conn.execute( + """ + INSERT INTO barcode_lookup_cache (cache_key, brand, product_title, size, result_json, created_at) + VALUES (?, ?, ?, ?, ?, ?) + ON CONFLICT(cache_key) DO UPDATE SET + result_json = excluded.result_json, + created_at = excluded.created_at + """, + (key, brand, product_title, size, payload, time.time()), + ) + conn.commit() + except Exception as e: + logger.debug(f"Barcode cache write failed for '{key}': {e}") diff --git a/app/services/enrichment/barcode/matching.py b/app/services/enrichment/barcode/matching.py new file mode 100644 index 0000000..1ed3a62 --- /dev/null +++ b/app/services/enrichment/barcode/matching.py @@ -0,0 +1,141 @@ +""" +Exact-product matching rules for barcode candidates. + +A source returning SOME EAN-13 for a product with a similar name is not +good enough - the "Product Matching Rules" requirement is explicit that +brand, product name, variant, and size must all match, and that a +different pack size (or a differently-branded variant like "Sugar-Free") +must be REJECTED rather than accepted as "close enough". This module is +deliberately conservative: a candidate only passes if every check agrees; +`is_valid_ean13`-style tricks are not enough to earn a false positive here +the way they might in a fuzzy search feature. + +Deliberately stdlib-only (`difflib`, already in the standard library) - +this project explicitly removed `fuzzywuzzy`/`python-Levenshtein` as dead +weight (see requirements.txt), so no new fuzzy-matching dependency is +introduced here either. +""" +from __future__ import annotations + +import re +from difflib import SequenceMatcher +from typing import Iterable, Optional + +from app.services.enrichment.barcode.models import BarcodeCandidate +# quantity_utils rather than image_search: this repo's image_search.py has no +# quantity helpers, and it is a working file the store-catalog work leaves alone. +from app.services.quantity_utils import quantities_match + +# Words that mark a genuinely DIFFERENT retail variant from the plain/base +# product - if the candidate's title contains one of these and the +# target product's own title/product_name does NOT, the candidate is +# rejected outright regardless of how well the rest of the name matches +# (this is exactly the "Marie Gold 200g" vs "Marie Gold Sugar-Free" +# example from the spec). Kept as a small, explicit, reviewable list +# rather than a fuzzy heuristic - false negatives (missing a real match) +# are far cheaper here than false positives (storing the wrong product's +# barcode). +VARIANT_DISTINGUISHING_TERMS = { + "sugar free", "sugarfree", "sugar-free", "no sugar", "zero sugar", + "diet", "family pack", "value pack", "jumbo pack", "combo pack", + "combo", "gift pack", "gift box", "mini pack", "party pack", + "refill pack", "refill", "pouch pack", "twin pack", "saver pack", + "economy pack", "jar", "tin", "pet jar", +} + +_STOPWORDS = {"the", "and", "of", "with", "for", "a", "an", "new", "pack", "india"} + + +def _normalize(text: Optional[str]) -> str: + return re.sub(r"[^a-z0-9\s]", " ", (text or "").lower()).strip() + + +def _tokens(text: Optional[str]) -> set: + return {t for t in _normalize(text).split() if t and t not in _STOPWORDS} + + +def brand_matches(candidate_brand: str, target_brand: str, brand_aliases: Optional[Iterable[str]] = None) -> bool: + """True if the candidate's reported brand plausibly refers to the same + brand as the target. Accepts an exact/substring match on the target + brand name itself, or a match against any known alias (e.g. "HUL" for + "Hindustan Unilever") supplied by the caller via + `brand_registry.get_brand_alias_set()`.""" + cand = _normalize(candidate_brand) + target = _normalize(target_brand) + if not cand or not target: + return False + if target in cand or cand in target: + return True + for alias in (brand_aliases or []): + alias_n = _normalize(alias) + if alias_n and (alias_n in cand or cand in alias_n): + return True + return False + + +def size_matches(candidate_size: str, target_size: str, tolerance: float = 0.03) -> bool: + """Tight tolerance (3%, vs. the 15% used for image matching elsewhere + in this project) - a barcode belongs to exactly one pack size, so + "close enough" is not an acceptable bar the way it is for reusing a + product photo. Falls back to a plain normalized-string equality check + when neither string parses as a numeric quantity (e.g. count-based + sizes like "10 tablets"), rather than silently treating unparsable + sizes as a match. + """ + if not candidate_size or not target_size: + return False + if quantities_match(candidate_size, target_size, tolerance=tolerance): + return True + return _normalize(candidate_size) == _normalize(target_size) + + +def has_conflicting_variant_terms(candidate_title: str, target_title: str) -> bool: + """True if the candidate's title names a distinguishing variant + (sugar-free, family pack, ...) that the target product does NOT - + which means the candidate is a real but DIFFERENT product, not the one + being looked up.""" + cand_n = _normalize(candidate_title) + target_n = _normalize(target_title) + for term in VARIANT_DISTINGUISHING_TERMS: + if term in cand_n and term not in target_n: + return True + return False + + +def name_similarity(candidate_title: str, target_title: str) -> float: + """0.0-1.0 token-overlap-weighted similarity. Used only as a secondary + signal / diagnostic (`match_confidence`) - never as the sole gate for + acceptance; see `is_match()`.""" + cand_tokens, target_tokens = _tokens(candidate_title), _tokens(target_title) + if not cand_tokens or not target_tokens: + return 0.0 + overlap = len(cand_tokens & target_tokens) / len(target_tokens) + seq_ratio = SequenceMatcher(None, _normalize(candidate_title), _normalize(target_title)).ratio() + return round((overlap * 0.6) + (seq_ratio * 0.4), 3) + + +def is_match(candidate: BarcodeCandidate, target_brand: str, target_title: str, target_size: str, + brand_aliases: Optional[Iterable[str]] = None, + min_name_similarity: float = 0.45) -> tuple[bool, float]: + """The combined gate a candidate must pass to be accepted: + 1. Brand matches (or overlaps a known alias). + 2. Pack size matches within a tight tolerance. + 3. No conflicting variant terms (family pack / sugar-free / ...). + 4. Product-name similarity clears a floor - catches the case where + brand+size coincidentally match but it's a completely different + product line from the same brand. + Returns (matched, confidence) - confidence is diagnostic only, stored + on the result for audit/QA but never used to override rule 1-3. + """ + if not brand_matches(candidate.candidate_brand, target_brand, brand_aliases): + return False, 0.0 + if not size_matches(candidate.candidate_size, target_size): + return False, 0.0 + if has_conflicting_variant_terms(candidate.candidate_title, target_title): + return False, 0.0 + + similarity = name_similarity(candidate.candidate_title, target_title) + if similarity < min_name_similarity: + return False, similarity + + return True, similarity diff --git a/app/services/enrichment/barcode/models.py b/app/services/enrichment/barcode/models.py new file mode 100644 index 0000000..28f38b3 --- /dev/null +++ b/app/services/enrichment/barcode/models.py @@ -0,0 +1,82 @@ +"""Data shapes shared across the barcode lookup module.""" +from __future__ import annotations + +import time +from dataclasses import dataclass, field +from enum import Enum +from typing import Optional + + +class BarcodeType(str, Enum): + EAN13 = "EAN-13" + UPC_A = "UPC-A" + GTIN14 = "GTIN-14" + GTIN8 = "GTIN-8" + UNKNOWN = "UNKNOWN" + + +class LookupStatus(str, Enum): + """Mirrors the `barcode_lookup_status` DB column.""" + VERIFIED = "verified" # found + checksum-valid + matched product + NOT_FOUND = "not_found" # every source exhausted, nothing matched + INVALID_CANDIDATE = "invalid_candidate" # a candidate was found but failed + # checksum/format validation or product matching - rejected, not stored + ERROR = "error" # a source/network error prevented a full search + DISABLED = "disabled" # barcode lookup turned off via settings + CACHED = "cached" # served from the local lookup cache + + +@dataclass +class BarcodeCandidate: + """A raw, not-yet-validated candidate returned by one source.""" + barcode: str + source_name: str # e.g. "Open Food Facts" + candidate_title: str = "" # the source's own product title, for matching + candidate_brand: str = "" + candidate_size: str = "" + candidate_countries: str = "" # raw country tag/string from the source, if any + + +@dataclass +class BarcodeResult: + """Final, caller-facing result. Field names match the DB columns in + `vector_store.py` and the JSON export shape requested for the + catalog (`barcode`, `barcode_type`, `barcode_source`, `barcode_verified`, + ...) 1:1, so `service.py` -> product dict -> DB row -> JSON export is a + straight field copy with no renaming at any layer. + """ + barcode: Optional[str] = None + barcode_type: Optional[str] = None + gtin: Optional[str] = None + ean13: Optional[str] = None + upc: Optional[str] = None + barcode_source: Optional[str] = None + barcode_verified: bool = False + barcode_lookup_status: str = LookupStatus.NOT_FOUND.value + barcode_last_updated: float = field(default_factory=time.time) + match_confidence: float = 0.0 + + @classmethod + def null_result(cls, status: LookupStatus = LookupStatus.NOT_FOUND) -> "BarcodeResult": + """The required NULL shape: barcode=NULL, barcode_verified=false, + barcode_source=NULL - used for every path where no authentic, + matching barcode could be confirmed. Never construct a + BarcodeResult with a barcode value outside of `service.py`'s + validated-and-matched path. + """ + return cls(barcode_lookup_status=status.value) + + def as_product_fields(self) -> dict: + """Flat dict merged directly into the catalog row - see + `stage.py`.""" + return { + "barcode": self.barcode, + "barcode_type": self.barcode_type, + "gtin": self.gtin, + "ean13": self.ean13, + "upc": self.upc, + "barcode_source": self.barcode_source, + "barcode_verified": self.barcode_verified, + "barcode_lookup_status": self.barcode_lookup_status, + "barcode_last_updated": self.barcode_last_updated, + } diff --git a/app/services/enrichment/barcode/retry.py b/app/services/enrichment/barcode/retry.py new file mode 100644 index 0000000..ddf2515 --- /dev/null +++ b/app/services/enrichment/barcode/retry.py @@ -0,0 +1,47 @@ +""" +Configurable retry-with-exponential-backoff for barcode source calls. + +Built on `tenacity`, already a project dependency (see requirements.txt - +used elsewhere for httpx-based retries). Only retries on genuinely +transient failures (network/timeout errors); a malformed response or a +"no results" outcome is not retried, since retrying those wastes the +source's rate-limit budget for no benefit (relevant for UPCItemDB's +trial-tier daily cap in particular). +""" +from __future__ import annotations + +import logging + +import requests +from tenacity import ( + retry, + retry_if_exception_type, + stop_after_attempt, + wait_exponential, + before_sleep_log, +) + +logger = logging.getLogger(__name__) + +_RETRYABLE_EXCEPTIONS = ( + requests.exceptions.ConnectionError, + requests.exceptions.Timeout, + requests.exceptions.ChunkedEncodingError, +) + + +def with_retry(max_attempts: int = 3, min_wait: float = 1.0, max_wait: float = 8.0): + """Decorator factory: `max_attempts` total tries, exponential backoff + between `min_wait` and `max_wait` seconds. Applied per-source (see + `sources/*.py`), not globally, so one slow/unreliable source retrying + doesn't compound delay across the whole cascade - a source that keeps + failing simply falls through to the next tier faster than a shared + global retry budget would allow. + """ + return retry( + reraise=True, + stop=stop_after_attempt(max_attempts), + wait=wait_exponential(multiplier=min_wait, max=max_wait), + retry=retry_if_exception_type(_RETRYABLE_EXCEPTIONS), + before_sleep=before_sleep_log(logger, logging.DEBUG), + ) diff --git a/app/services/enrichment/barcode/service.py b/app/services/enrichment/barcode/service.py new file mode 100644 index 0000000..19a2e7f --- /dev/null +++ b/app/services/enrichment/barcode/service.py @@ -0,0 +1,170 @@ +""" +BarcodeLookupService - the cascading lookup orchestrator. + +This is the ONE place that decides "is this barcode good enough to store". +Every candidate from every source, regardless of tier, passes through the +exact same two gates before it can become a `BarcodeResult`: + 1. `validators.validate_barcode()` - checksum + format. + 2. `matching.is_match()` - brand + size + variant + name. +A source being "trusted" (e.g. GS1 India) does not skip either gate - +trust only affects ORDER (which tier is tried first), never whether +validation is required. + +Cascade behaviour (see module docstring in `sources/__init__.py` for the +tier list): tiers are tried in order; the first tier that produces at +least one validated, matching candidate wins and the search stops - "The +system should stop searching as soon as a verified barcode is found." +Every tier is independently wrapped so a source outage never prevents +falling through to the next one, and if every tier is exhausted with +nothing matching, the result is the required NULL shape +(`BarcodeResult.null_result()`), never a guess. +""" +from __future__ import annotations + +import asyncio +import logging +from typing import Any, Dict, List, Optional + +from app.infrastructure.settings import ( + ENABLE_BARCODE_LOOKUP, + BARCODE_LOOKUP_CACHE_TTL_SECONDS, + BARCODE_LOOKUP_MAX_CONCURRENCY, +) +from app.services.enrichment.barcode import cache +from app.services.enrichment.barcode.matching import is_match +from app.services.enrichment.barcode.models import BarcodeCandidate, BarcodeResult, BarcodeType, LookupStatus +from app.services.enrichment.barcode.sources import get_default_sources +from app.services.enrichment.barcode.validators import classify_barcode_type, to_ean13, validate_barcode + +logger = logging.getLogger(__name__) + +try: + from app.services.brand_registry import get_brand_alias_set +except Exception: # pragma: no cover - keeps the service importable/testable + # in isolation even if brand_registry ever fails to import. + def get_brand_alias_set(_brand: str) -> set: + return set() + + +def _build_result(barcode: str, source_name: str, confidence: float) -> BarcodeResult: + btype = classify_barcode_type(barcode) + ean13 = to_ean13(barcode) if btype in (BarcodeType.UPC_A, BarcodeType.EAN13) else None + return BarcodeResult( + barcode=barcode, + barcode_type=btype.value, + gtin=barcode, + ean13=ean13, + upc=barcode if btype == BarcodeType.UPC_A else None, + barcode_source=source_name, + barcode_verified=True, + barcode_lookup_status=LookupStatus.VERIFIED.value, + match_confidence=confidence, + ) + + +class BarcodeLookupService: + def __init__(self, sources=None): + self.sources = sources if sources is not None else get_default_sources() + + def lookup_one(self, brand: str, product_title: str, size: str, category: str = "", + use_cache: bool = True) -> BarcodeResult: + """Synchronous cascading lookup for a single (brand, product, + size). Never raises - every failure path returns a NULL-shaped + `BarcodeResult` with a descriptive `barcode_lookup_status`.""" + if not ENABLE_BARCODE_LOOKUP: + return BarcodeResult.null_result(LookupStatus.DISABLED) + + if not brand or not product_title or not size: + logger.debug(f"Barcode lookup skipped - missing brand/title/size ('{brand}', '{product_title}', '{size}')") + return BarcodeResult.null_result(LookupStatus.ERROR) + + if use_cache: + cached = cache.get_cached(brand, product_title, size, BARCODE_LOOKUP_CACHE_TTL_SECONDS) + if cached is not None: + result = cached + result.barcode_lookup_status = ( + LookupStatus.CACHED.value if result.barcode_verified else result.barcode_lookup_status + ) + return result + + brand_aliases = self._safe_brand_aliases(brand) + had_source_error = False + + for source in self.sources: + if not source.available: + continue + try: + raw_candidates = source.search(brand, product_title, size, category) + except Exception as e: + # Sources are documented to never raise, but this is the + # pipeline-wide safety net referenced in base.py. + logger.warning(f"[{source.name}] barcode search raised unexpectedly: {e}") + had_source_error = True + continue + + result = self._first_validated_match(raw_candidates, brand, product_title, size, brand_aliases) + if result is not None: + if use_cache: + cache.set_cached(brand, product_title, size, result) + return result + + status = LookupStatus.ERROR if had_source_error else LookupStatus.NOT_FOUND + result = BarcodeResult.null_result(status) + if use_cache: + # Cache negative results too (Performance Requirements: "Prevent + # duplicate barcode searches") - a short TTL still applies, so a + # transient "not found" doesn't permanently block a later re-check. + cache.set_cached(brand, product_title, size, result) + return result + + @staticmethod + def _safe_brand_aliases(brand: str) -> set: + try: + return get_brand_alias_set(brand) + except Exception: + return set() + + @staticmethod + def _first_validated_match(raw_candidates: List[BarcodeCandidate], brand: str, product_title: str, + size: str, brand_aliases: set) -> Optional[BarcodeResult]: + for candidate in raw_candidates: + clean_barcode = validate_barcode(candidate.barcode) + if not clean_barcode: + continue # invalid checksum/length/format - never stored, not even flagged + matched, confidence = is_match(candidate, brand, product_title, size, brand_aliases) + if not matched: + continue + return _build_result(clean_barcode, candidate.source_name, confidence) + return None + + async def batch_lookup(self, products: List[Dict[str, Any]], brand: str) -> List[BarcodeResult]: + """Async batch entry point used by `stage.py`. Bounded concurrency + (`BARCODE_LOOKUP_MAX_CONCURRENCY`) so a large brand catalog doesn't + fire dozens of simultaneous requests at any one source, and results + are returned in the SAME ORDER as `products` regardless of which + finished first, so callers can zip() them back together safely.""" + semaphore = asyncio.Semaphore(max(1, BARCODE_LOOKUP_MAX_CONCURRENCY)) + + async def _one(product: Dict[str, Any]) -> BarcodeResult: + async with semaphore: + return await asyncio.to_thread( + self.lookup_one, + brand, + product.get("title") or product.get("product_name") or "", + product.get("size") or "", + product.get("category") or "", + ) + + return await asyncio.gather(*(_one(p) for p in products)) + + +# Module-level singleton - sources are stateless, so one shared instance is +# fine for both the sync CLI/test path and the async pipeline stage. +_default_service: Optional[BarcodeLookupService] = None + + +def get_default_service() -> BarcodeLookupService: + global _default_service + if _default_service is None: + _default_service = BarcodeLookupService() + return _default_service diff --git a/app/services/enrichment/barcode/sources/__init__.py b/app/services/enrichment/barcode/sources/__init__.py new file mode 100644 index 0000000..81a010f --- /dev/null +++ b/app/services/enrichment/barcode/sources/__init__.py @@ -0,0 +1,35 @@ +""" +Tier-ordered source registry - this list IS the cascade order described in +the spec: GS1 India -> Open Food Facts -> trusted barcode databases -> +manufacturer website -> (exhausted -> NULL). `service.py` iterates this +list in order and stops at the first source that produces a validated, +matching candidate. +""" +from app.services.enrichment.barcode.sources.base import BarcodeSource +from app.services.enrichment.barcode.sources.gs1_india import GS1IndiaSource +from app.services.enrichment.barcode.sources.open_food_facts import OpenFoodFactsSource +from app.services.enrichment.barcode.sources.upc_database import UPCDatabaseSource +from app.services.enrichment.barcode.sources.manufacturer_site import ManufacturerSiteSource + + +def get_default_sources() -> list[BarcodeSource]: + """Fresh instances each call - sources are stateless/cheap to + construct, and this avoids any shared-mutable-state surprises across + concurrent batch lookups.""" + sources = [ + GS1IndiaSource(), + OpenFoodFactsSource(), + UPCDatabaseSource(), + ManufacturerSiteSource(), + ] + return sorted(sources, key=lambda s: s.tier) + + +__all__ = [ + "BarcodeSource", + "GS1IndiaSource", + "OpenFoodFactsSource", + "UPCDatabaseSource", + "ManufacturerSiteSource", + "get_default_sources", +] diff --git a/app/services/enrichment/barcode/sources/base.py b/app/services/enrichment/barcode/sources/base.py new file mode 100644 index 0000000..cc878fd --- /dev/null +++ b/app/services/enrichment/barcode/sources/base.py @@ -0,0 +1,46 @@ +""" +Pluggable barcode source contract, following the same adapter-registry +pattern already used for marketplace SKU resolution +(app/services/sku_service.py's `_MARKETPLACE_PATTERNS`) and the Product +Resolver's adapter registry (see the Catalog_Project's `resolve_parent_brand` +family) - every source is a self-contained class with one job: given a +brand/title/size, return zero or more raw candidates. It does NOT decide +whether a candidate is a match (that's `matching.py`) or whether its +barcode is valid (that's `validators.py`) - keeping those concerns +separate is what lets `service.py` apply the SAME validation/matching gate +uniformly regardless of which tier produced the candidate. +""" +from __future__ import annotations + +from abc import ABC, abstractmethod +from typing import List + +from app.services.enrichment.barcode.models import BarcodeCandidate + + +class BarcodeSource(ABC): + #: Human-readable label stored in `barcode_source` on a successful + #: match, e.g. "Open Food Facts", "GS1 India". + name: str = "Unknown Source" + + #: Lower = tried first in the cascade (see sources/__init__.py). + tier: int = 99 + + @property + def available(self) -> bool: + """False when the source is unusable for reasons unrelated to any + one query (missing API key, disabled via settings, dependency not + installed, ...). The cascade skips unavailable sources entirely + rather than querying and failing every time.""" + return True + + @abstractmethod + def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]: + """Best-effort search. MUST NOT raise - catch internally and + return [] on any failure (network error, malformed response, + nothing found). `service.py` treats an empty list and an + exception identically (fall through to the next tier), so + swallowing the error here vs. letting it propagate makes no + behavioural difference except robustness - always swallow it. + """ + raise NotImplementedError diff --git a/app/services/enrichment/barcode/sources/gs1_india.py b/app/services/enrichment/barcode/sources/gs1_india.py new file mode 100644 index 0000000..8d1fa75 --- /dev/null +++ b/app/services/enrichment/barcode/sources/gs1_india.py @@ -0,0 +1,91 @@ +""" +Tier 1: GS1 India / GEPIR (the official Indian GTIN registry). + +HONESTY NOTE - READ BEFORE WIRING UP CREDENTIALS +-------------------------------------------------- +GS1 India does not currently publish a free, public, no-registration JSON +API for GTIN lookup (unlike Open Food Facts). Their real product-data +registry is accessed either through GEPIR (https://gepir.gs1.org/ - a +human web UI with no documented public API) or through GS1 India's paid +"Verified by GS1" data-as-a-service product, which requires a commercial +account and an API key. + +Rather than scrape GEPIR's web UI (fragile, likely against its terms, and +exactly the kind of "pretend this is a real integration" shortcut this +project's whole hallucination-prevention philosophy exists to avoid - see +product_validator.py's module docstring), this adapter is a REAL, +WORKING integration point that: + - does nothing (returns [], `available=False`) when no credentials are + configured, so the cascade cleanly falls through to Open Food Facts + (Tier 2) - exactly the "if not found -> next step" behaviour the + spec asks for, just starting from Tier 2 in practice until GS1 India + API access is provisioned; + - immediately becomes live the moment `GS1_INDIA_API_BASE_URL` and + `GS1_INDIA_API_KEY` are set (see .env.example), with no code changes + needed elsewhere - `service.py` already treats every tier uniformly. + +If/when real GS1 India API access is provisioned, fill in the request +shape in `search()` below to match that API's actual contract (endpoint +path, auth header, response schema) - the surrounding plumbing +(validation, matching, caching, retry) does not need to change. +""" +from __future__ import annotations + +import logging +from typing import List + +import requests + +from app.infrastructure.settings import GS1_INDIA_API_BASE_URL, GS1_INDIA_API_KEY, BARCODE_LOOKUP_TIMEOUT_SECONDS +from app.services.enrichment.barcode.models import BarcodeCandidate +from app.services.enrichment.barcode.retry import with_retry +from app.services.enrichment.barcode.sources.base import BarcodeSource + +logger = logging.getLogger(__name__) + + +class GS1IndiaSource(BarcodeSource): + name = "GS1 India" + tier = 1 + + @property + def available(self) -> bool: + return bool(GS1_INDIA_API_BASE_URL and GS1_INDIA_API_KEY) + + def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]: + if not self.available: + return [] + + @with_retry(max_attempts=3) + def _call(): + return requests.get( + f"{GS1_INDIA_API_BASE_URL.rstrip('/')}/search", + params={"brand": brand, "product": product_title, "size": size, "country": "India"}, + headers={"Authorization": f"Bearer {GS1_INDIA_API_KEY}"}, + timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS, + ) + + try: + resp = _call() + if resp.status_code != 200: + logger.debug(f"GS1 India lookup non-200 ({resp.status_code}) for '{brand} {product_title} {size}'") + return [] + data = resp.json() + except Exception as e: + logger.debug(f"GS1 India lookup failed for '{brand} {product_title} {size}': {e}") + return [] + + candidates: List[BarcodeCandidate] = [] + for item in (data.get("results") or data.get("items") or []): + gtin = item.get("gtin") or item.get("barcode") or item.get("code") + if not gtin: + continue + candidates.append(BarcodeCandidate( + barcode=str(gtin), + source_name=self.name, + candidate_title=item.get("productName") or item.get("title") or "", + candidate_brand=item.get("brandName") or item.get("brand") or "", + candidate_size=item.get("netContent") or item.get("size") or "", + candidate_countries="in", + )) + return candidates diff --git a/app/services/enrichment/barcode/sources/manufacturer_site.py b/app/services/enrichment/barcode/sources/manufacturer_site.py new file mode 100644 index 0000000..b5f47d6 --- /dev/null +++ b/app/services/enrichment/barcode/sources/manufacturer_site.py @@ -0,0 +1,108 @@ +""" +Tier 4: manufacturer / official product page - last resort. + +Follows the exact same "DuckDuckGo search -> read structure straight off +the result, no page-scraping HTML parser" spirit as +`app/services/sku_service.py`'s `find_website_product_id()`, except here +we DO need to fetch the page text (a barcode isn't embedded in the result +URL the way an Amazon ASIN is), so this is the one tier that performs a +capped number of lightweight page fetches, each wrapped in its own +try/except so one slow/broken page can never block the others. + +This tier is intentionally the lowest-trust one: a number that merely sits +near the word "barcode"/"EAN"/"UPC"/"GTIN" on a webpage is only a +CANDIDATE. It still has to pass `validators.validate_barcode()` (checksum) +and `matching.is_match()` (brand/size/name) in `service.py` exactly like +every other tier's candidates - nothing here is trusted just because it +came from what looks like an official page. +""" +from __future__ import annotations + +import logging +import re +from typing import List + +import requests + +from app.infrastructure.settings import BARCODE_LOOKUP_TIMEOUT_SECONDS, ENABLE_MANUFACTURER_SITE_LOOKUP +from app.services.enrichment.barcode.models import BarcodeCandidate +from app.services.enrichment.barcode.retry import with_retry +from app.services.enrichment.barcode.sources.base import BarcodeSource + +logger = logging.getLogger(__name__) + +_BROWSER_UA = ( + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" +) + +# A barcode label followed, within a short distance, by an 8/12/13/14-digit +# run. Intentionally permissive on the label (source pages phrase this +# differently) and tight on the digit run (only plausible GTIN lengths) - +# every hit is still just a CANDIDATE, checksum-validated afterwards. +_BARCODE_NEAR_LABEL_RE = re.compile( + r"(?:barcode|ean|upc|gtin)\D{0,15}(\d{8}|\d{12,14})", + re.IGNORECASE, +) +_MAX_PAGES = 3 + + +class ManufacturerSiteSource(BarcodeSource): + name = "Manufacturer Website" + tier = 4 + + @property + def available(self) -> bool: + if not ENABLE_MANUFACTURER_SITE_LOOKUP: + return False + try: + import ddgs # noqa: F401 + return True + except ImportError: + logger.debug("Manufacturer-site barcode lookup unavailable: 'ddgs' package not installed") + return False + + def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]: + if not self.available: + return [] + + query = f"{brand} {product_title} {size} barcode EAN".strip() + try: + from ddgs import DDGS + with DDGS(timeout=10) as ddgs: + results = ddgs.text(query, region="in-en", safesearch="off", max_results=_MAX_PAGES) + except Exception as e: + logger.debug(f"Manufacturer-site search failed for '{query}': {e}") + return [] + + candidates: List[BarcodeCandidate] = [] + for r in (results or [])[:_MAX_PAGES]: + url = str(r.get("href") or r.get("url") or "") + if not url.startswith("http"): + continue + for code in self._extract_barcodes(url): + candidates.append(BarcodeCandidate( + barcode=code, + source_name=f"{self.name} ({url})", + candidate_title=r.get("title") or product_title, + candidate_brand=brand, + candidate_size=size, + candidate_countries="", + )) + return candidates + + def _extract_barcodes(self, url: str) -> List[str]: + @with_retry(max_attempts=2) + def _call(): + return requests.get(url, headers={"User-Agent": _BROWSER_UA}, timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS) + + try: + resp = _call() + if resp.status_code != 200: + return [] + text = resp.text[:200_000] # cap - this is a text scan, not a full-page render + except Exception as e: + logger.debug(f"Manufacturer page fetch failed for '{url}': {e}") + return [] + + return [m.group(1) for m in _BARCODE_NEAR_LABEL_RE.finditer(text)] diff --git a/app/services/enrichment/barcode/sources/open_food_facts.py b/app/services/enrichment/barcode/sources/open_food_facts.py new file mode 100644 index 0000000..2821870 --- /dev/null +++ b/app/services/enrichment/barcode/sources/open_food_facts.py @@ -0,0 +1,118 @@ +""" +Tier 2: Open Food Facts / Open Beauty Facts / Open Products Facts. + +Real, free, no-API-key public search API - the same family of open, +community-maintained product databases already used as the project's +PRIMARY image source (see `app/services/image_search.py`'s +`OPEN_FACTS_HOSTS` / `_query_openfacts`). Every entry is keyed by its real +barcode (the `code` field IS the GTIN/EAN/UPC printed on the physical +pack), which is exactly the ground-truth this module needs - unlike +`image_search.py`, which only reads the image URLs off each entry, this +adapter reads the `code` field itself. + +Deliberately a separate, self-contained query function rather than +importing `image_search._query_openfacts` (which is a private, underscore- +prefixed helper): this adapter needs different response fields (`code`, +`countries_tags`) and India-specific filtering that the image-search +helper has no reason to carry. `quantities_match`/`parse_quantity_grams` +ARE imported from `image_search.py` (public functions) for size +comparison, so the size-matching logic itself is not duplicated - see +`matching.py`. +""" +from __future__ import annotations + +import logging +from typing import List + +import requests + +from app.infrastructure.settings import BARCODE_LOOKUP_TIMEOUT_SECONDS, BARCODE_COUNTRY_TAG +from app.services.enrichment.barcode.models import BarcodeCandidate +from app.services.enrichment.barcode.retry import with_retry +from app.services.enrichment.barcode.sources.base import BarcodeSource + +logger = logging.getLogger(__name__) + +_HOSTS = [ + "world.openfoodfacts.org", + "world.openbeautyfacts.org", + "world.openproductsfacts.org", +] +_BROWSER_UA = ( + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" +) +_FIELDS = "code,product_name,brands,brands_tags,quantity,countries_tags" + + +class OpenFoodFactsSource(BarcodeSource): + name = "Open Food Facts" + tier = 2 + + def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]: + query = f"{brand or ''} {product_title or ''}".strip() + if not query: + return [] + + products = self._query(query) + if not products and brand and product_title: + # Combined query too specific (common for regional brand-name + # variants) - retry title-only, same fallback image_search.py uses. + products = self._query(product_title) + + candidates: List[BarcodeCandidate] = [] + for item in products: + code = str(item.get("code") or "").strip() + if not code: + continue + countries = ",".join(item.get("countries_tags") or []) + candidates.append(BarcodeCandidate( + barcode=code, + source_name=self.name, + candidate_title=item.get("product_name") or "", + candidate_brand=item.get("brands") or "", + candidate_size=item.get("quantity") or "", + candidate_countries=countries, + )) + + # Soft India-market prioritisation: candidates whose countries_tags + # mention the target market are tried first, but non-tagged/other- + # market candidates are kept (not dropped) since many genuine + # Indian FMCG entries simply have this field blank upstream - the + # brand/size/name gate in matching.py is what actually decides + # correctness, this only affects which validated match is found + # (and therefore stops the cascade) first. + if BARCODE_COUNTRY_TAG: + tag = BARCODE_COUNTRY_TAG.lower() + candidates.sort(key=lambda c: 0 if tag in c.candidate_countries.lower() else 1) + return candidates + + def _query(self, query: str) -> list: + for host in _HOSTS: + @with_retry(max_attempts=2) + def _call(host=host): + return requests.get( + f"https://{host}/cgi/search.pl", + params={ + "search_terms": query, + "search_simple": 1, + "action": "process", + "json": 1, + "page_size": 20, + "fields": _FIELDS, + }, + headers={"User-Agent": _BROWSER_UA}, + timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS, + ) + + try: + resp = _call() + if resp.status_code != 200: + continue + products = resp.json().get("products", []) + if products: + return products + except Exception as e: + logger.debug(f"Open*Facts barcode lookup failed on {host} for '{query}': {e}") + continue + return [] diff --git a/app/services/enrichment/barcode/sources/upc_database.py b/app/services/enrichment/barcode/sources/upc_database.py new file mode 100644 index 0000000..8c841d0 --- /dev/null +++ b/app/services/enrichment/barcode/sources/upc_database.py @@ -0,0 +1,81 @@ +""" +Tier 3: "trusted barcode databases" - UPCItemDB. + +Real, free (trial-tier, no API key required, rate-limited) reverse lookup: +search by product name/brand keywords and get back candidate items with +their own `upc`/`ean` fields. This is the general "trusted barcode +database" tier the spec asks for as a fallback below GS1 India / Open +Food Facts. + +RATE LIMIT: UPCItemDB's free trial endpoint is capped (documented as +~100 requests/day, ~1 request/second) - this is exactly why this tier +sits BELOW Open Food Facts (unlimited, no key) in the cascade, and why +`service.py` only calls a lower tier at all when every higher tier has +already failed to produce a validated match, plus why the local cache +(`cache.py`) matters most for this specific source. If a paid UPCItemDB +key is available, set `UPC_DATABASE_API_KEY` (see .env.example) to switch +to the production endpoint with a higher quota - the request shape below +already supports both. +""" +from __future__ import annotations + +import logging +from typing import List + +import requests + +from app.infrastructure.settings import BARCODE_LOOKUP_TIMEOUT_SECONDS, UPC_DATABASE_API_KEY +from app.services.enrichment.barcode.models import BarcodeCandidate +from app.services.enrichment.barcode.retry import with_retry +from app.services.enrichment.barcode.sources.base import BarcodeSource + +logger = logging.getLogger(__name__) + +_TRIAL_URL = "https://api.upcitemdb.com/prod/trial/search" +_PROD_URL = "https://api.upcitemdb.com/prod/v1/search" + + +class UPCDatabaseSource(BarcodeSource): + name = "UPCItemDB" + tier = 3 + + def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]: + query = f"{brand or ''} {product_title or ''} {size or ''}".strip() + if not query: + return [] + + url = _PROD_URL if UPC_DATABASE_API_KEY else _TRIAL_URL + headers = {"Accept": "application/json"} + if UPC_DATABASE_API_KEY: + headers["user_key"] = UPC_DATABASE_API_KEY + headers["key_type"] = "3scale" + + @with_retry(max_attempts=2) + def _call(): + return requests.get(url, params={"s": query, "type": "product"}, headers=headers, + timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS) + + try: + resp = _call() + if resp.status_code != 200: + logger.debug(f"UPCItemDB non-200 ({resp.status_code}) for '{query}'") + return [] + data = resp.json() + except Exception as e: + logger.debug(f"UPCItemDB lookup failed for '{query}': {e}") + return [] + + candidates: List[BarcodeCandidate] = [] + for item in data.get("items", []): + code = item.get("ean") or item.get("upc") + if not code: + continue + candidates.append(BarcodeCandidate( + barcode=str(code), + source_name=self.name, + candidate_title=item.get("title") or "", + candidate_brand=item.get("brand") or "", + candidate_size=item.get("size") or "", + candidate_countries="", + )) + return candidates diff --git a/app/services/enrichment/barcode/stage.py b/app/services/enrichment/barcode/stage.py new file mode 100644 index 0000000..3de812f --- /dev/null +++ b/app/services/enrichment/barcode/stage.py @@ -0,0 +1,37 @@ +"""Adapts BarcodeLookupService to the EnrichmentStage contract so it can be +registered in app/services/enrichment/pipeline.py's default pipeline.""" +from __future__ import annotations + +import logging +from typing import Any, Dict + +from app.infrastructure.settings import ENABLE_BARCODE_LOOKUP +from app.services.enrichment.base import EnrichmentStage, StageOutcome +from app.services.enrichment.barcode.service import get_default_service + +logger = logging.getLogger(__name__) + + +class BarcodeEnrichmentStage(EnrichmentStage): + name = "barcode_lookup" + + @property + def enabled(self) -> bool: + return ENABLE_BARCODE_LOOKUP + + async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome: + service = get_default_service() + title = product.get("title") or product.get("product_name") or "" + size = product.get("size") or "" + category = product.get("category") or "" + + try: + import asyncio + result = await asyncio.to_thread(service.lookup_one, brand, title, size, category) + except Exception as e: + logger.warning(f"Barcode enrichment failed for '{title}' {size}: {e}") + from app.services.enrichment.barcode.models import BarcodeResult, LookupStatus + result = BarcodeResult.null_result(LookupStatus.ERROR) + + return StageOutcome(stage_name=self.name, fields=result.as_product_fields(), + error=None if result.barcode_verified else result.barcode_lookup_status) diff --git a/app/services/enrichment/barcode/validators.py b/app/services/enrichment/barcode/validators.py new file mode 100644 index 0000000..6e92862 --- /dev/null +++ b/app/services/enrichment/barcode/validators.py @@ -0,0 +1,99 @@ +""" +Barcode format + checksum validation. + +Deliberately the LAST gate a candidate passes through before it's allowed +into a `BarcodeResult` (see `service.py`), regardless of which source +produced it or how much we otherwise trust that source - a source telling +us "this is the barcode" is never sufficient on its own; the digits have to +actually check out mathematically. This is what "Validate EAN-13 checksum" +/ "Validate GTIN format" / "Reject invalid barcode lengths" / "Reject +malformed barcode values" (Barcode Validation requirements) means in code. + +No third-party dependency - GTIN/EAN/UPC-A all share one checksum +algorithm (the classic "alternating 3/1 weights counted from the rightmost +digit before the check digit"), so GTIN-8/12/13/14 are all validated by the +same function; only the accepted LENGTH differs per barcode type. +""" +from __future__ import annotations + +import re +from typing import Optional + +from app.services.enrichment.barcode.models import BarcodeType + +_VALID_LENGTHS = {8: BarcodeType.GTIN8, 12: BarcodeType.UPC_A, 13: BarcodeType.EAN13, 14: BarcodeType.GTIN14} + +# Digits only, no separators - callers are expected to call normalize_barcode() +# first if the raw source string may contain spaces/hyphens. +_DIGITS_ONLY_RE = re.compile(r"^\d+$") + + +def normalize_barcode(raw: Optional[str]) -> Optional[str]: + """Strip everything but digits (spaces, hyphens, a stray 'EAN:' label, + etc.). Returns None for empty/unusable input - never raises.""" + if not raw: + return None + digits = re.sub(r"\D", "", str(raw)) + return digits or None + + +def gtin_check_digit(digits_without_check: str) -> int: + """Compute the correct check digit for a GTIN-8/12/13/14 payload + (i.e. every digit EXCEPT the check digit itself), using the standard + alternating 3/1 weighting counted from the rightmost digit.""" + total = 0 + for i, ch in enumerate(reversed(digits_without_check)): + weight = 3 if i % 2 == 0 else 1 + total += int(ch) * weight + return (10 - (total % 10)) % 10 + + +def has_valid_checksum(code: str) -> bool: + """True if `code`'s own last digit matches the checksum computed over + the rest of it. `code` must already be digits-only.""" + if not code or not _DIGITS_ONLY_RE.match(code): + return False + body, check_digit = code[:-1], code[-1] + try: + expected = gtin_check_digit(body) + except (ValueError, IndexError): + return False + return str(expected) == check_digit + + +def classify_barcode_type(code: str) -> BarcodeType: + """Barcode type purely from its (already checksum-validated) length. + UPC-A (12 digits) is the one ambiguous case worth calling out: it is + numerically a GTIN-13 with a leading zero, but is reported as UPC-A + here since that's the label the requesting spec/UI expects for a + 12-digit code.""" + return _VALID_LENGTHS.get(len(code), BarcodeType.UNKNOWN) + + +def validate_barcode(raw: Optional[str]) -> Optional[str]: + """Single entry point: normalize, check length, check checksum. + Returns the clean digits-only barcode string if and only if it is a + genuinely valid GTIN-8/12/13/14, else None. This is the ONLY function + other modules should call to decide "is this barcode good enough to + store" - never inline a length/regex check elsewhere. + """ + code = normalize_barcode(raw) + if not code: + return None + if len(code) not in _VALID_LENGTHS: + return None + if not has_valid_checksum(code): + return None + return code + + +def to_ean13(code: str) -> Optional[str]: + """Zero-pad a valid UPC-A (12 digits) up to its equivalent EAN-13 + representation. GTIN-8 is intentionally NOT padded (an 8-digit GTIN is + its own distinct symbology, not a truncated EAN-13) - returns None for + anything that isn't 12 or 13 digits already.""" + if len(code) == 13: + return code + if len(code) == 12: + return "0" + code + return None diff --git a/app/services/enrichment/base.py b/app/services/enrichment/base.py new file mode 100644 index 0000000..3e84f89 --- /dev/null +++ b/app/services/enrichment/base.py @@ -0,0 +1,80 @@ +""" +Base contract every enrichment stage implements. + +A stage takes ONE already-validated catalog row (a plain dict, the same +`enhanced_product` shape `catalog_engine.py` builds) plus the brand name, +and returns the SAME dict with additional fields merged in - it never +removes or renames a key it didn't add itself, and it never raises: any +internal failure is caught and reported via `StageOutcome.error` instead. + +To add a new enrichment stage later (HSN, GST, nutrition, allergens, ...): + 1. Subclass `EnrichmentStage`. + 2. Implement `async def enrich_one(product, brand) -> StageOutcome`. + 3. Register an instance in `pipeline.run_default_pipeline()` (or build + a custom `EnrichmentPipeline([...])` for a one-off run). +No other file needs to change - `catalog_engine.py`'s call site and +`vector_store.py`'s upsert already iterate whatever keys are present on the +product dict via `.get(...)`, so a new stage's fields flow through to the +database/JSON export automatically the same way barcode fields do. +""" +from __future__ import annotations + +import logging +from abc import ABC, abstractmethod +from dataclasses import dataclass, field +from typing import Any, Dict, Optional + +logger = logging.getLogger(__name__) + + +@dataclass +class StageOutcome: + """Result of running one stage on one product row.""" + stage_name: str + fields: Dict[str, Any] = field(default_factory=dict) + error: Optional[str] = None + + @property + def ok(self) -> bool: + return self.error is None + + +class EnrichmentStage(ABC): + """One independent, pluggable enrichment step. + + `name` is used for logging/metrics only. `enabled` lets a stage report + itself as switched off (e.g. via a settings flag) without the pipeline + orchestrator needing to know why - it's simply skipped and every + product passes through untouched. + """ + + name: str = "unnamed_stage" + + @property + def enabled(self) -> bool: + return True + + @abstractmethod + async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome: + """Enrich a single product row. MUST NOT raise - catch internally + and return a StageOutcome with `error` set instead.""" + raise NotImplementedError + + async def apply(self, product: Dict[str, Any], brand: str) -> Dict[str, Any]: + """Run this stage on `product` and merge the result in place. + Never raises - a stage bug degrades to "no fields added", logged, + rather than aborting the whole catalog row. + """ + try: + outcome = await self.enrich_one(product, brand) + except Exception as e: # last-resort safety net - stages should + # already catch their own errors, but a pipeline-wide guarantee + # of "never raises" is worth the redundancy here. + logger.error(f"[{self.name}] unhandled exception enriching '{product.get('product_name')}': {e}") + return product + + if outcome.fields: + product.update(outcome.fields) + if not outcome.ok: + logger.debug(f"[{self.name}] {product.get('product_name')}: {outcome.error}") + return product diff --git a/app/services/enrichment/hsn_gst/__init__.py b/app/services/enrichment/hsn_gst/__init__.py new file mode 100644 index 0000000..e233247 --- /dev/null +++ b/app/services/enrichment/hsn_gst/__init__.py @@ -0,0 +1,32 @@ +""" +HSN / GST & Pricing enrichment module. + +Public entry points: + resolve_hsn_gst(category, product_title="") -> HsnGstInfo + Deterministic, offline category -> (HSN code, GST %, review flag) + resolution, mirroring the shape already present in the coca-cola + and milky_mist catalog exports (hsn_code / gst_percent / + hsn_gst_needs_review / selling_price / tax_amount / + final_selling_price / cost_price / profit_before_tax / + profit_after_tax). + + enrich_pricing_fields(product) -> dict + Computes selling_price / tax_amount / final_selling_price (plus the + always-null cost/profit fields) from a catalog row's price_range, + using the same convention as the existing enriched brands: the + selling price is the upper bound of the row's retail price range. + +See the package's module docstrings for the architecture: + models.py - HsnGstInfo result type + curated category -> HSN/GST table + stage.py - adapts the resolver to the generic EnrichmentStage + contract used by app/services/enrichment/pipeline.py +""" +from app.services.enrichment.hsn_gst.models import HsnGstInfo, resolve_hsn_gst, enrich_pricing_fields +from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage + +__all__ = [ + "HsnGstInfo", + "resolve_hsn_gst", + "enrich_pricing_fields", + "HsnGstEnrichmentStage", +] diff --git a/app/services/enrichment/hsn_gst/models.py b/app/services/enrichment/hsn_gst/models.py new file mode 100644 index 0000000..a7c0246 --- /dev/null +++ b/app/services/enrichment/hsn_gst/models.py @@ -0,0 +1,259 @@ +""" +Curated, deterministic HSN / GST lookup for catalog product categories. + +WHY THIS IS NOT "MADE UP" +-------------------------- +HSN (Harmonized System of Nomenclature) is the 4-8 digit goods-classification +code used on every Indian tax invoice, and GST% is the Indian Goods & +Services Tax rate applied to that HSN chapter. Both are public, legal, +category-level facts - not per-SKU secrets. Every entry in `HSN_GST_TABLE` +below is a *typical* classification for the product category as sold at +retail (matching the values already present in the coca-cola / milky_mist +catalog exports where those overlap, e.g. Beverages -> 2009, Dairy -> 0401, +Dairy - Desserts -> 2105, Tea & Coffee -> 0902). + +The catch: an HSN chapter can cover several GST rates depending on the exact +item and how it is packaged, so a category-level guess can be wrong for a +specific product. That is precisely what `HSN_GST_NEEDS_REVIEW` is for - it is +True whenever the mapping is approximate (the default), and False only for the +few categories whose retail classification is unambiguous. The flag exists so +a human reviewer / the Streamlit UI can spot-check exactly these rows. + +DESIGN PRINCIPLES +------------------ +- DETERMINISTIC & OFFLINE. No LLM, no network, no randomness. The same + category always yields the same (HSN, GST%, review) tuple, so re-runs and + backfills are stable and diffable. +- ADDITIVE. Only ever attaches new keys to a product row; never removes or + rewrites an existing key (the stage skips rows that already carry these + fields, so re-running the pipeline over an already-enriched catalog is a + no-op). +- SAFE DEFAULTS. Unknown/unrecognised categories get hsn_code=None and + gst_percent=None with needs_review=True, rather than a fabricated code - + a wrong HSN on a real tax document is worse than no HSN at all. +""" +from __future__ import annotations + +import re +from dataclasses import dataclass +from typing import Dict, Optional, Tuple + + +# (HSN code, GST %, needs_review). Review=True whenever the category can map +# to more than one GST rate / HSN chapter at retail; False for unambiguous ones. +HSN_GST_TABLE: Dict[str, Tuple[str, int, bool]] = { + # ---- Food & beverage (matches the existing coca-cola / milky_mist exports) ---- + "Dairy": ("0401", 5, False), + "Cheese": ("0406", 12, True), + "Dairy - Desserts": ("2105", 5, False), + "Ice Cream": ("2105", 18, True), + "Beverages": ("2009", 5, False), + "Tea & Coffee": ("0902", 5, False), + "Food & Beverages": ("2106", 18, True), + "Health Drinks": ("2202", 18, True), + "Health Foods": ("2106", 18, True), + "Breakfast Cereal": ("1904", 18, False), + "Chocolates": ("1806", 18, False), + "Candy & Confectionery": ("1704", 18, False), + "Biscuits & Cookies": ("1905", 18, True), + "Crackers": ("1905", 18, True), + "Rusk": ("1905", 5, True), + "Cakes & Muffins": ("1905", 18, True), + "Bakery & Breads": ("1905", 5, False), + "Snacks": ("1905", 18, True), + "Noodles & Instant Food": ("1902", 18, False), + "Pasta & Noodles": ("1902", 18, False), + "Atta & Staples": ("1101", 5, False), + "Salt & Staples": ("2501", 5, False), + "Pulses, Grains & Spices": ("0713", 5, True), + "Spices & Masalas": ("0910", 5, False), + "Cooking Oils": ("1517", 5, False), + "Pickles & Chutneys": ("2001", 12, True), + "Dry Fruits & Nuts": ("0801", 12, True), + "Food - Spreads": ("2007", 12, True), + "Food - Mixes": ("2106", 18, True), + "Food - Soups & Sauces": ("2103", 12, False), + # ---- Personal care & household ---- + "Hair Care": ("3305", 18, False), + "Skin Care": ("3304", 18, False), + "Skin & Bath Care": ("3307", 18, False), + "Bath Soap": ("3401", 18, False), + "Beauty Care": ("3304", 18, False), + "Oral Care": ("3306", 18, False), + "Fragrance & Deodorants": ("3303", 18, False), + "Men's Grooming": ("8212", 18, True), + "Baby Care": ("3304", 18, True), + "Feminine Hygiene": ("9619", 12, False), + "Personal Care": ("3307", 18, True), + "Detergents & Fabric Care": ("3402", 18, False), + "Dishwash": ("3402", 18, False), + "Household Cleaning": ("3402", 18, True), + "Household - Air Freshener": ("3307", 18, True), + "Personal Care - Mosquito Repellent": ("3808", 18, True), + # ---- Health care ---- + "Health Care - Cold & Cough": ("3004", 12, True), + "Health Care - Antiseptic": ("3808", 18, True), + "Health Care - Ayurvedic": ("3003", 12, True), + "Health Care - Digestive": ("3004", 12, True), + "Health Care - First Aid": ("3005", 12, True), +} + +# Keyword fallbacks so a product whose category label is slightly different +# from the table (or missing entirely) still resolves to a sensible chapter. +_KEYWORD_FALLBACKS: Tuple[Tuple[str, str, int, bool], ...] = ( + ("toothpaste", "3306", 18, False), + ("shampoo", "3305", 18, False), + ("soap", "3401", 18, False), + ("detergent", "3402", 18, False), + ("biscuit", "1905", 18, True), + ("cookie", "1905", 18, True), + ("chocolate", "1806", 18, False), + ("candy", "1704", 18, False), + ("toffee", "1704", 18, False), + ("milk", "0401", 5, False), + ("paneer", "0401", 5, False), + ("curd", "0401", 5, False), + ("cheese", "0406", 12, True), + ("ice cream", "2105", 18, True), + ("juice", "2009", 5, False), + ("tea", "0902", 5, False), + ("coffee", "0902", 5, False), + ("health drink", "2202", 18, True), + ("chips", "1905", 18, True), + ("namkeen", "1905", 18, True), + ("bread", "1905", 5, False), + ("atta", "1101", 5, False), + ("flour", "1101", 5, False), + ("masala", "0910", 5, False), + ("spice", "0910", 5, False), + ("pasta", "1902", 18, False), + ("noodle", "1902", 18, False), + ("cooking oil", "1517", 5, False), + ("oil", "1517", 5, True), + ("pickle", "2001", 12, True), + ("jam", "2007", 12, True), + ("honey", "0409", 5, False), + ("raisin", "0806", 5, True), + ("dry fruit", "0801", 12, True), + ("cereal", "1904", 18, False), + ("deodorant", "3303", 18, False), + ("perfume", "3303", 18, False), + ("lipstick", "3304", 18, False), + ("makeup", "3304", 18, False), + ("face wash", "3307", 18, False), + ("lotion", "3307", 18, False), + ("sunscreen", "3304", 18, False), + ("razor", "8212", 18, True), + ("diaper", "9619", 12, False), + ("mosquito", "3808", 18, True), + ("repellent", "3808", 18, True), + ("air freshener", "3307", 18, True), + ("dishwash", "3402", 18, False), + ("floor cleaner", "3402", 18, True), + ("hand wash", "3401", 18, False), +) + + +@dataclass(frozen=True) +class HsnGstInfo: + """HSN / GST resolution for one product category, mirroring the fields + stored on enriched catalog rows (coca-cola / milky_mist exports).""" + hsn_code: Optional[str] = None + gst_percent: Optional[int] = None + hsn_gst_needs_review: bool = True + + def as_fields(self) -> Dict[str, object]: + return { + "hsn_code": self.hsn_code, + "gst_percent": self.gst_percent, + "hsn_gst_needs_review": self.hsn_gst_needs_review, + } + + +_UNKNOWN = HsnGstInfo(hsn_code=None, gst_percent=None, hsn_gst_needs_review=True) + + +def resolve_hsn_gst(category: Optional[str], product_title: str = "") -> HsnGstInfo: + """Resolve (HSN code, GST %, needs_review) for a product category. + + Exact category matches in `HSN_GST_TABLE` win; otherwise the product + title/description is scanned against `_KEYWORD_FALLBACKS`; otherwise + returns the safe unknown shape (all-None, needs_review=True). + """ + cat = (category or "").strip() + if cat: + exact = HSN_GST_TABLE.get(cat) + if exact: + return HsnGstInfo(*exact) + + haystack = f"{product_title or ''} {cat}".lower() + for kw, hsn, gst, review in _KEYWORD_FALLBACKS: + if kw in haystack: + return HsnGstInfo(hsn_code=hsn, gst_percent=gst, hsn_gst_needs_review=review) + + return _UNKNOWN + + +# Matches a price range string of the form "₹12-14" (also tolerates +# "Rs 12-14", "12 - 14", a single "₹12", or corrupted rupee symbols). +_RANGE_RE = re.compile(r"₹?\s*(\d+(?:\.\d+)?)\s*[-–—]\s*(\d+(?:\.\d+)?)") +_SINGLE_RE = re.compile(r"₹?\s*(\d+(?:\.\d+)?)") + + +def _extract_selling_price(price_range: object) -> Optional[float]: + """Return the upper bound of a row's price range (the convention used by + the existing coca-cola / milky_mist exports for `selling_price`), or the + single value when the row only carries one price.""" + if price_range is None: + return None + text = str(price_range).replace(",", "").strip() + if not text: + return None + m = _RANGE_RE.search(text) + if m: + try: + return float(m.group(2)) + except ValueError: + return None + m = _SINGLE_RE.search(text) + if m: + try: + return float(m.group(1)) + except ValueError: + return None + return None + + +def _round2(value: Optional[float]) -> Optional[float]: + if value is None: + return None + return round(value, 2) + + +def enrich_pricing_fields(product: dict) -> Dict[str, object]: + """Compute the pricing fields added by this stage for one catalog row. + + Convention (matches the coca-cola / milky_mist exports): + selling_price = upper bound of the row's `price_range` + tax_amount = selling_price * gst_percent / 100 + final_selling_price = selling_price + tax_amount + `cost_price`, `profit_before_tax` and `profit_after_tax` are not + derivable from the data this pipeline generates, so they stay None + (exactly as the existing enriched exports store them). + """ + selling_price = _extract_selling_price(product.get("price_range")) + gst = product.get("gst_percent") + tax_amount = None + final_selling_price = None + if selling_price is not None and isinstance(gst, (int, float)) and gst and gst > 0: + tax_amount = _round2(selling_price * float(gst) / 100.0) + final_selling_price = _round2(selling_price + tax_amount) + + return { + "selling_price": _round2(selling_price), + "cost_price": None, + "tax_amount": tax_amount, + "final_selling_price": final_selling_price, + "profit_before_tax": None, + "profit_after_tax": None, + } diff --git a/app/services/enrichment/hsn_gst/stage.py b/app/services/enrichment/hsn_gst/stage.py new file mode 100644 index 0000000..d7e2890 --- /dev/null +++ b/app/services/enrichment/hsn_gst/stage.py @@ -0,0 +1,48 @@ +"""Adapts the HSN/GST resolver to the EnrichmentStage contract so it can be +registered in app/services/enrichment/pipeline.py's default pipeline. + +Unlike the barcode stage this needs no network access at all - HSN/GST are +deterministic, category-level facts - so it is synchronous and effectively +free. It is pure ADDITIVE: rows that already carry the hsn_code/gst_percent +keys (e.g. re-running the pipeline over a previously-enriched catalog) are +left untouched. +""" +from __future__ import annotations + +import logging +from typing import Any, Dict + +from app.infrastructure.settings import ENABLE_HSN_GST_ENRICHMENT +from app.services.enrichment.base import EnrichmentStage, StageOutcome +from app.services.enrichment.hsn_gst.models import enrich_pricing_fields, resolve_hsn_gst + +logger = logging.getLogger(__name__) + + +class HsnGstEnrichmentStage(EnrichmentStage): + name = "hsn_gst" + + @property + def enabled(self) -> bool: + return ENABLE_HSN_GST_ENRICHMENT + + async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome: + # Already enriched (previous run / manual backfill) - never overwrite. + if product.get("hsn_code") is not None or product.get("gst_percent") is not None: + return StageOutcome(stage_name=self.name, fields={}) + + title = product.get("title") or product.get("product_name") or "" + category = product.get("category") or "" + + hsn_info = resolve_hsn_gst(category, title) + fields: Dict[str, Any] = hsn_info.as_fields() + # Pricing fields depend on gst_percent, which was just resolved + # above, so hand the resolved rate to the pricing helper. + fields.update(enrich_pricing_fields({**product, "gst_percent": fields["gst_percent"]})) + + needs_review = fields["hsn_gst_needs_review"] + return StageOutcome( + stage_name=self.name, + fields=fields, + error=None if not needs_review else "category-level HSN/GST estimate - verify against the physical pack" + ) diff --git a/app/services/enrichment/pipeline.py b/app/services/enrichment/pipeline.py new file mode 100644 index 0000000..8971578 --- /dev/null +++ b/app/services/enrichment/pipeline.py @@ -0,0 +1,89 @@ +""" +Orchestrates a list of independent `EnrichmentStage`s over a batch of +catalog rows, running each stage's per-row work concurrently (bounded by +`max_concurrency`) and NEVER letting one row's failure affect any other +row or stage. + +`catalog_engine.py` calls `run_default_pipeline()` once, between the +deterministic-validation step (product_validator.py) and catalog assembly +- see that file's "Step 2.6" for the call site and +docs/BARCODE_ENRICHMENT.md for the full pipeline-position rationale. +""" +from __future__ import annotations + +import asyncio +import logging +from typing import Any, Dict, List, Optional, Sequence + +from app.services.enrichment.base import EnrichmentStage + +logger = logging.getLogger(__name__) + + +class EnrichmentPipeline: + def __init__(self, stages: Sequence[EnrichmentStage], max_concurrency: int = 5): + self.stages = [s for s in stages if s.enabled] + self.max_concurrency = max(1, max_concurrency) + + async def run(self, products: List[Dict[str, Any]], brand: str) -> List[Dict[str, Any]]: + if not self.stages or not products: + return products + + semaphore = asyncio.Semaphore(self.max_concurrency) + + async def _run_row(product: Dict[str, Any]) -> Dict[str, Any]: + async with semaphore: + for stage in self.stages: + product = await stage.apply(product, brand) + return product + + results = await asyncio.gather(*(_run_row(p) for p in products), return_exceptions=True) + + final: List[Dict[str, Any]] = [] + for original, result in zip(products, results): + if isinstance(result, Exception): + logger.error(f"Enrichment pipeline failed for '{original.get('product_name')}': {result}") + final.append(original) + else: + final.append(result) + return final + + +def _build_default_stages() -> List[EnrichmentStage]: + """Import stages lazily so importing this module never pulls in a + stage's own dependencies (network clients, DB drivers, ...) unless a + default pipeline is actually requested.""" + stages: List[EnrichmentStage] = [] + try: + from app.services.enrichment.barcode.stage import BarcodeEnrichmentStage + stages.append(BarcodeEnrichmentStage()) + except Exception as e: + logger.error(f"Barcode enrichment stage unavailable: {e}") + + # HSN / GST & pricing enrichment (see app/services/enrichment/hsn_gst/) - + # deterministic, offline, pure-additive. Runs AFTER the barcode stage so + # every stored/exported row carries both sets of fields; a failure here + # degrades a row to "no HSN/GST attached", never drops the row. + try: + from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage + stages.append(HsnGstEnrichmentStage()) + except Exception as e: + logger.error(f"HSN/GST enrichment stage unavailable: {e}") + + # Future stages register here, e.g.: + # from app.services.enrichment.nutrition.stage import NutritionEnrichmentStage + # stages.append(NutritionEnrichmentStage()) + return stages + + +_default_pipeline: Optional[EnrichmentPipeline] = None + + +async def run_default_pipeline(products: List[Dict[str, Any]], brand: str, + max_concurrency: int = 5) -> List[Dict[str, Any]]: + """Convenience entry point used by catalog_engine.py: runs every + currently-registered, enabled enrichment stage over `products`.""" + global _default_pipeline + if _default_pipeline is None: + _default_pipeline = EnrichmentPipeline(_build_default_stages(), max_concurrency=max_concurrency) + return await _default_pipeline.run(products, brand) diff --git a/app/services/price_estimator.py b/app/services/price_estimator.py index bebafd6..1dd77c5 100644 --- a/app/services/price_estimator.py +++ b/app/services/price_estimator.py @@ -353,3 +353,32 @@ def parse_price_string(text: str) -> Optional[float]: if len(numbers) >= 2: return (numbers[0] + numbers[1]) / 2.0 return numbers[0] + + +def estimate_price_range_for_size( + size: str, + product_title: str = "", + brand: str = "", + category_hint: str = "", +) -> Tuple[int, int]: + """Return a (min, max) price range for a SINGLE pack size. + + Ported from the sibling Universal_Catalog_Barcode_Enrichment project, where + `product_validator.validate_price_range` uses it as the anchor to judge + whether a scraped price is plausible for the pack size. Additive: no + existing function in this module changes. + + The range is centred on `estimate_price` with a deterministic +-5%-15% + jitter keyed on product+size, so the same product gets the same range on + every run while still representing genuine retail spread rather than a + single fixed MRP. + """ + price = estimate_price(size, product_title, brand, category_hint) + seed = f"{brand}|{category_hint}|{product_title}|{size}|range" + jitter = _deterministic_unit(seed) + spread = 0.05 + jitter * 0.10 + lo = max(1, int(round(price * (1 - spread)))) + hi = int(round(price * (1 + spread))) + if hi <= lo: + hi = lo + 1 + return (lo, hi) diff --git a/app/services/product_validator.py b/app/services/product_validator.py new file mode 100644 index 0000000..9cdda23 --- /dev/null +++ b/app/services/product_validator.py @@ -0,0 +1,445 @@ +""" +Deterministic Product Validation & Confidence Scoring +======================================================= + +WHY THIS FILE EXISTS +--------------------- +Every stage upstream of this module (ollama_service.py, title_validator.py, +price_estimator.py, sku_service.py) already applies point-fixes for +*specific, previously-reported* hallucination patterns (category-inconsistent +titles, unrealistic prices, non-existent brands, malformed JSON, ...). None +of them, however, ever made a final, holistic pass over a fully-assembled +catalog row and asked "taken together, does this look like a REAL product +worth keeping?" That gap is why individually-plausible-looking rows with +compounding small defects (a slightly-off price + a missing image + a vague +size) could still reach the database. + +This module is that final gate. It is intentionally: + +- DETERMINISTIC: no LLM call, no randomness. The exact same product dict + always produces the exact same report. That is what makes this a + hallucination *filter* rather than another hallucination *source*. +- CONSERVATIVE: every check only fires on strong, explainable evidence + (reuses title_validator's category-conflict detector, price_estimator's + category-anchored price bands, and sku_service's own SKU format) rather + than guessing. False rejections are treated as seriously as false + acceptances - see docs/VALIDATION_PIPELINE.md for the tuning rationale. +- ADDITIVE: this module has zero dependents before this change and does not + modify any existing function's behaviour; it is only ever imported and + called from new code (see app/core/catalog_engine.py's call site). + +USAGE +----- +Call `validate_product(product_dict, brand, ...)` once per fully-assembled +catalog row, AFTER pricing/SKU/image resolution but BEFORE the row is +appended to the catalog output / sent to upsert_brand_products(). The +returned ValidationReport carries a 0.0-1.0 confidence score and a +status of "verified" / "needs_review" / "rejected" that the caller uses to +decide whether to keep, flag, or drop the row - see +ProductCatalogEngine.generate_catalog() in catalog_engine.py. +""" +from __future__ import annotations + +import logging +import re +from typing import Any, Optional + +from pydantic import BaseModel, Field + +from app.services.title_validator import find_category_conflicts +from app.services import price_estimator +from app.services import category_units as cu + +logger = logging.getLogger(__name__) + + +# --------------------------------------------------------------------------- +# Categories that are never legitimate for any FMCG / personal-care brand. +# Deliberately duplicated (rather than imported) from +# app.services.ollama_service._IRRELEVANT_CATEGORIES: ollama_service already +# imports from brand_registry, and this module is imported BY +# catalog_engine.py alongside ollama_service, so importing a private +# (underscore-prefixed) name across modules would create a fragile coupling +# for a list that changes maybe once a year. Keep both lists in sync if you +# add a new one. +# --------------------------------------------------------------------------- +IRRELEVANT_CATEGORIES: set[str] = { + "pet supplies", "pet food", "pet care", "automotive", "auto parts", + "automotive parts", "car accessories", "motorcycle", "bike", + "electronics", "gadgets", "computers", "mobile phones", + "furniture", "home decor", "furnishings", "appliances", + "clothing", "footwear", "fashion", "jewelry", "accessories", + "office supplies", "stationery", "books", "toys", "games", + "sports equipment", "sports gear", "gardening", "garden supplies", + "hardware", "tools", "building materials", + "musical instruments", "party supplies", "craft supplies", +} + +# Title strings that are structurally a placeholder/fabrication artefact +# rather than a real product name, independent of any specific brand. +_PLACEHOLDER_TITLE_PATTERNS: list[re.Pattern] = [ + # Bare "Product 3" / "Brand Product 3", or any single leading token + # (typically the brand name) immediately followed by "Product N" - the + # exact shape of the hardcoded placeholder data in the unused + # agentic_pipeline.py scaffold (see that file's module docstring) and + # the generic fallback a hallucinating LLM sometimes produces when it + # runs out of real products to enumerate. + re.compile(r"^\s*(\S+\s+)?product\s*#?\s*\d+\s*$", re.IGNORECASE), + re.compile(r"^\s*(unknown|n/?a|unnamed|untitled|sample|placeholder|tbd|todo|test\s*product)\s*$", re.IGNORECASE), + re.compile(r"^https?://", re.IGNORECASE), + # "5 Star (Soft Drink Candy) - Soft Drink Candy" style category echo - + # same pattern catalog_engine.py already filters at discovery time; kept + # here too as defence-in-depth for rows that reach this stage some other + # way (e.g. a future direct-insert code path that skips catalog_engine). + re.compile(r"\(([^)]{2,40})\)\s*-\s*\1", re.IGNORECASE), +] + +# Internal SKUs are always formatted BRAND-PRODUCT-SIZE-SEQ by +# sku_service.generate_internal_sku(), e.g. "AACHI-SAM-500-001". +_INTERNAL_SKU_RE = re.compile(r"^[A-Z0-9]{2,15}-[A-Z0-9]{2,15}-[A-Z0-9]{1,10}-\d{3,}$") + +_PRICE_RANGE_RE = re.compile(r"₹\s*(\d+(?:\.\d+)?)\s*-\s*(\d+(?:\.\d+)?)") + +# Plausible bounds for a single retail FMCG pack, in gram/ml equivalent. +# Below 1: parsing failure. Above 30000 (30kg): almost certainly a +# hallucinated bulk size no e-commerce FMCG listing would use (real bulk +# packs top out around 25kg rice/atta/detergent sacks). +_MIN_PACK_GRAMS = 1.0 +_MAX_PACK_GRAMS = 30_000.0 + + +class ValidationIssue(BaseModel): + """A single, explainable reason a confidence score moved.""" + + field: str + severity: str = Field(description="'error' (strong evidence of a problem) or 'warning' (weaker signal)") + message: str + penalty: float = 0.0 + + +class ValidationConfig(BaseModel): + """Tunable thresholds - see docs/VALIDATION_PIPELINE.md for defaults + and how to change them via settings.py without editing this file.""" + + reject_below: float = 0.35 + review_below: float = 0.70 + + +class ValidationReport(BaseModel): + """Result of validating one fully-assembled catalog row.""" + + product_name: str + status: str = "verified" # "verified" | "needs_review" | "rejected" + confidence: float = 1.0 + grounded: bool = False + issues: list[ValidationIssue] = Field(default_factory=list) + + @property + def is_rejected(self) -> bool: + return self.status == "rejected" + + @property + def needs_review(self) -> bool: + return self.status == "needs_review" + + def summary(self) -> str: + if not self.issues: + return f"{self.product_name}: OK (confidence={self.confidence:.2f})" + reasons = "; ".join(f"{i.field}: {i.message}" for i in self.issues) + return f"{self.product_name}: {self.status} (confidence={self.confidence:.2f}) - {reasons}" + + +def _looks_like_placeholder(title: str) -> bool: + if not title or not title.strip(): + return True + return any(p.search(title.strip()) for p in _PLACEHOLDER_TITLE_PATTERNS) + + +def validate_size(size: str) -> tuple[bool, Optional[str]]: + """Check that `size` parses to a plausible FMCG retail-pack quantity. + + Reuses price_estimator's own size parser so "what counts as a valid + size" can never silently drift out of sync with what pricing actually + uses to compute cost. + """ + if not size or not size.strip(): + return False, "size/pack is missing" + grams = price_estimator._parse_size_to_grams(size) + if grams <= 0: + return False, f"could not parse a usable quantity from size '{size}'" + if grams < _MIN_PACK_GRAMS or grams > _MAX_PACK_GRAMS: + return False, f"parsed quantity ({grams:g}) from size '{size}' is outside a plausible retail-pack range" + return True, None + + +def validate_price_range( + price_range: str, size: str, product_title: str, brand: str, category: str, +) -> tuple[bool, Optional[str]]: + """Check that `price_range` is well-formed and not a gross outlier + against the category+size price anchor computed by price_estimator. + + The tolerance band (0.3x-3x of the anchor) is deliberately wide - this + check exists to catch gross hallucinations (a ₹5,000 biscuit packet, a + ₹2 detergent bottle), not to second-guess normal retail price spread, + which price_estimator's own range already accounts for. + """ + if not price_range or not price_range.strip(): + return False, "price_range is missing" + m = _PRICE_RANGE_RE.search(price_range) + if not m: + return False, f"price_range '{price_range}' is not a well-formed ₹lo-hi string" + lo, hi = float(m.group(1)), float(m.group(2)) + if lo <= 0 or hi <= 0 or hi < lo: + return False, f"price_range '{price_range}' has non-positive or inverted bounds" + try: + anchor_lo, anchor_hi = price_estimator.estimate_price_range_for_size( + size, product_title, brand, category + ) + except Exception: + return True, None # anchor unavailable (e.g. blank size) - don't fail the row on that alone + if hi < anchor_lo * 0.3 or lo > anchor_hi * 3.0: + return False, ( + f"price_range '{price_range}' is implausible for a {size or 'unspecified-size'} " + f"{category or 'product'} (expected roughly ₹{anchor_lo}-{anchor_hi})" + ) + return True, None + + +def validate_sku(sku: str, sku_source: str) -> tuple[bool, Optional[str]]: + """Check `product_sku` is present and, for internally-generated SKUs, + matches sku_service's own BRAND-PRODUCT-SIZE-SEQ format. Real + marketplace IDs (sku_source != "Internal") are trusted as-is since they + were found via an actual web lookup, not generated - their format + legitimately varies by platform (ASIN vs Flipkart PID vs ...). + """ + if not sku or not sku.strip(): + return False, "product_sku is missing" + if sku_source and sku_source != "Internal": + return True, None + if not _INTERNAL_SKU_RE.match(sku.strip()): + return False, f"internal SKU '{sku}' does not match the expected BRAND-PRODUCT-SIZE-SEQ pattern" + return True, None + + +def validate_product( + product: dict[str, Any], + brand: str, + *, + category_resolved_deterministically: bool = False, + grounded: bool = False, + images_checked: bool = True, + config: Optional[ValidationConfig] = None, +) -> ValidationReport: + """Validate one fully-assembled catalog row and return a confidence-scored report. + + Parameters + ---------- + product: + The enhanced product dict as built by catalog_engine.py, expected to + contain (at minimum) title/product_name, category, size, price_range, + product_sku, sku_source, image_urls. + brand: + The brand this row is being catalogued under. + category_resolved_deterministically: + True when the category came from brand_registry's curated sub-brand + registry (ground truth) rather than a keyword heuristic - a positive + evidence signal, not a validity requirement. + grounded: + True when an independent, external signal corroborates this exact + product (e.g. an Open Food Facts catalog match, or the row already + existing in the database from a prior validated run). See + catalog_engine.py's call site for exactly what sets this. + config: + Optional threshold override; defaults to ValidationConfig() (which + itself defaults to the values in app.infrastructure.settings). + """ + cfg = config or ValidationConfig() + title = str(product.get("title") or product.get("product_name") or "").strip() + category = str(product.get("category") or "").strip() + size = str(product.get("size") or "").strip() + price_range = str(product.get("price_range") or "").strip() + sku = str(product.get("product_sku") or "").strip() + sku_source = str(product.get("sku_source") or "").strip() + images = product.get("image_urls") or [] + + report = ValidationReport(product_name=title or "(untitled)", grounded=grounded) + score = 0.55 # neutral baseline - moves up/down based on evidence below + + # 1. Title sanity ----------------------------------------------------- + if _looks_like_placeholder(title): + report.issues.append(ValidationIssue( + field="title", severity="error", + message=f"title '{title}' looks like a placeholder/fabricated entry, not a real product name", + penalty=0.50, + )) + score -= 0.50 + elif len(title) < 2: + report.issues.append(ValidationIssue( + field="title", severity="error", + message="title is too short to be a real product name", penalty=0.40, + )) + score -= 0.40 + + # 2. Category sanity ---------------------------------------------------- + if not category or category == "Uncategorized": + report.issues.append(ValidationIssue( + field="category", severity="warning", + message="category could not be resolved", penalty=0.10, + )) + score -= 0.10 + elif category.lower() in IRRELEVANT_CATEGORIES: + report.issues.append(ValidationIssue( + field="category", severity="error", + message=f"category '{category}' is not a valid FMCG/personal-care category", penalty=0.60, + )) + score -= 0.60 + + # 3. Title <-> category consistency (reuses title_validator's own + # conflict detector so the two modules can never disagree about what + # counts as a contradiction) -------------------------------------- + if category and category not in ("Uncategorized", "General"): + conflicts = find_category_conflicts(title, category) + if conflicts: + report.issues.append(ValidationIssue( + field="title", severity="error", + message=f"title contains word(s) inconsistent with category '{category}': {', '.join(conflicts)}", + penalty=0.35, + )) + score -= 0.35 + + # 4. Size / pack -------------------------------------------------------- + ok, msg = validate_size(size) + if not ok: + report.issues.append(ValidationIssue(field="size", severity="error", message=msg, penalty=0.25)) + score -= 0.25 + + # 4b. Size unit <-> category compatibility (see app/services/ + # category_units.py). This is a last-resort backstop: by the time a + # row reaches this final validator, its size should already have + # been through the generation-time gate (app/services/ + # generation_verifier.py, called from ollama_service.py) AND + # catalog_engine.py's own dimension-unit defence-in-depth pass, both + # of which reject/repair a bad unit long before enrichment. This + # check exists purely to catch anything that reached validate_product + # some other way (e.g. a pre-existing DB row from before this fix, or + # a future caller that doesn't route through catalog_engine). + if size: + ok, msg = cu.validate_unit_for_category(size, category) + if not ok: + report.issues.append(ValidationIssue(field="size", severity="error", message=msg, penalty=0.15)) + score -= 0.15 + + # 5. Price range ---------------------------------------------------------- + ok, msg = validate_price_range(price_range, size, title, brand, category) + if not ok: + report.issues.append(ValidationIssue(field="price_range", severity="error", message=msg, penalty=0.30)) + score -= 0.30 + + # 6. SKU -------------------------------------------------------------------- + ok, msg = validate_sku(sku, sku_source) + if not ok: + report.issues.append(ValidationIssue(field="product_sku", severity="warning", message=msg, penalty=0.10)) + score -= 0.10 + + # 7. Image presence ----------------------------------------------------- + # Only meaningful if image search actually ran. When the operator disables + # it (the store-catalog pipeline's offline mode), an empty image list says + # nothing about whether the product is real, and penalising it here would + # reject every row in a perfectly good upload. + if not images and images_checked: + report.issues.append(ValidationIssue( + field="image_urls", severity="warning", + message="no images were found for this product", penalty=0.15, + )) + score -= 0.15 + + # 8. Positive evidence -------------------------------------------------- + if category_resolved_deterministically: + score += 0.15 + if grounded: + score += 0.15 + if images: + score += 0.05 + + score = max(0.0, min(1.0, score)) + report.confidence = round(score, 3) + + if score < cfg.reject_below: + report.status = "rejected" + elif score < cfg.review_below: + report.status = "needs_review" + else: + report.status = "verified" + + return report + + +def validate_catalog( + products: list[dict[str, Any]], + brand: str, + *, + known_category_flags: Optional[list[bool]] = None, + grounded_flags: Optional[list[bool]] = None, + images_checked: bool = True, + config: Optional[ValidationConfig] = None, +) -> tuple[list[dict[str, Any]], list[dict[str, Any]], dict[str, Any]]: + """Validate an entire list of assembled catalog rows in one pass. + + Returns (kept_products, rejected_products, summary) where `kept_products` + includes both "verified" and "needs_review" rows (each annotated with + `validation_status` / `confidence_score` / `validation_issues` so a + human-in-the-loop reviewer or a downstream consumer can filter further), + and `rejected_products` holds the rows dropped from the catalog along + with why, for audit/QA purposes - see docs/VALIDATION_PIPELINE.md. + """ + cfg = config or ValidationConfig() + known_category_flags = known_category_flags or [False] * len(products) + grounded_flags = grounded_flags or [False] * len(products) + + kept: list[dict[str, Any]] = [] + rejected: list[dict[str, Any]] = [] + status_counts = {"verified": 0, "needs_review": 0, "rejected": 0} + + for i, p in enumerate(products): + det = known_category_flags[i] if i < len(known_category_flags) else False + grd = grounded_flags[i] if i < len(grounded_flags) else False + report = validate_product( + p, brand, + category_resolved_deterministically=det, + grounded=grd, + images_checked=images_checked, + config=cfg, + ) + status_counts[report.status] = status_counts.get(report.status, 0) + 1 + + annotated = { + **p, + "validation_status": report.status, + "confidence_score": report.confidence, + "validation_issues": [i.message for i in report.issues], + } + + if report.is_rejected: + logger.warning(f"🚫 Rejected (validation): {report.summary()}") + rejected.append(annotated) + else: + if report.needs_review: + logger.info(f"⚠️ Needs review: {report.summary()}") + else: + logger.info(f"✅ Verified: {report.product_name} (confidence={report.confidence:.2f})") + kept.append(annotated) + + summary = { + "total_evaluated": len(products), + "verified": status_counts["verified"], + "needs_review": status_counts["needs_review"], + "rejected": status_counts["rejected"], + "reject_threshold": cfg.reject_below, + "review_threshold": cfg.review_below, + } + logger.info( + f"🔍 Validation summary for '{brand}': {summary['verified']} verified, " + f"{summary['needs_review']} needs_review, {summary['rejected']} rejected " + f"(of {summary['total_evaluated']} rows)" + ) + return kept, rejected, summary diff --git a/app/services/quantity_utils.py b/app/services/quantity_utils.py new file mode 100644 index 0000000..f2a44df --- /dev/null +++ b/app/services/quantity_utils.py @@ -0,0 +1,82 @@ +""" +Pack-size / quantity parsing helpers. + +Ported from the sibling `Universal_Catalog_Barcode_Enrichment` project, where +these lived inside `image_search.py`. They are a standalone module here on +purpose: the barcode matcher needs them, and this repo's `image_search.py` is +a working file that the store-catalog work is not supposed to touch. Extracting +rather than appending keeps that guarantee. + +Deliberately self-contained (no import from price_estimator.py) - these are +only used for approximate "same pack size?" comparisons, never for pricing +math. Grams and millilitres are treated as interchangeable because this only +ever compares two size LABELS for a plausible match, not physics. +""" +from __future__ import annotations + +import re +from typing import List, Optional + +_QTY_RE = re.compile( + r"(\d+(?:[.,]\d+)?)\s*[-_]?\s*" + r"(kilograms?|kilos?|kgs?|grams?|gms?|litres?|liters?|ltrs?|lts?|mls?|ml|g|kg|l)\b" +) +_QTY_LITRE_UNITS = {"l", "lt", "lts", "ltr", "ltrs", "litre", "litres", "liter", "liters"} +_QTY_KG_UNITS = {"kg", "kgs", "kilo", "kilos", "kilogram", "kilograms"} + + +def parse_quantity_grams(text: Optional[str]) -> Optional[float]: + """Parse the first weight/volume mentioned in `text` (e.g. '200 g', + '1kg', '500ml', a URL fragment like 'good-day-100-g') into a + normalised grams-or-millilitres-equivalent float. Returns None if no + recognisable quantity is found. + """ + if not text: + return None + match = _QTY_RE.search(str(text).lower()) + if not match: + return None + try: + value = float(match.group(1).replace(",", ".")) + except ValueError: + return None + unit = match.group(2) + if unit in _QTY_LITRE_UNITS or unit in _QTY_KG_UNITS: + value *= 1000 + return value + + +def quantities_match(a: Optional[str], b: Optional[str], tolerance: float = 0.15) -> bool: + """True if size labels `a` and `b` plausibly refer to the same pack size, + allowing for rounding/formatting differences between sources (e.g. '200 g' + vs '210g' vs '0.2 kg'). Used to match a requested pack-size variant against + Open Food Facts' real per-barcode `quantity` field. + """ + qa, qb = parse_quantity_grams(a), parse_quantity_grams(b) + if qa is None or qb is None or qa <= 0: + return False + return abs(qa - qb) / qa <= tolerance + + +def find_quantity_mentions(text: Optional[str]) -> List[float]: + """Return every distinct weight/volume value mentioned anywhere in `text` + (e.g. a full image/product URL), normalised to a grams-equivalent float. + + Unlike parse_quantity_grams() (first match only), this scans the whole + string - used to detect when a candidate image's URL names a DIFFERENT + sibling pack size than the one being searched for (e.g. a "...-500-g..." + URL turning up while resolving images for the 100g variant). + """ + if not text: + return [] + out: List[float] = [] + for m in _QTY_RE.finditer(str(text).lower()): + try: + value = float(m.group(1).replace(",", ".")) + except ValueError: + continue + unit = m.group(2) + if unit in _QTY_LITRE_UNITS or unit in _QTY_KG_UNITS: + value *= 1000 + out.append(value) + return out diff --git a/app/services/sku_service.py b/app/services/sku_service.py new file mode 100644 index 0000000..088d2ef --- /dev/null +++ b/app/services/sku_service.py @@ -0,0 +1,275 @@ +""" +Product SKU / website product-ID resolution service. + +For every catalog record (one exact brand + product/variety + pack size) +this module resolves a `product_sku` in two stages: + + 1. REAL MARKETPLACE ID (best-effort, network-dependent) + A DuckDuckGo web search (via the `ddgs` package - already a project + dependency, see app/services/image_search.py for the same pattern) + restricted to known Indian e-commerce domains. If a result URL is on + one of those domains, the real product ID is extracted directly from + the URL itself (Amazon ASIN, Flipkart `pid`, BigBasket/JioMart/ + Blinkit/Nykaa numeric product IDs) - no page scraping/HTML parsing + required, so this is fast and doesn't trip anti-bot defenses. + + 2. INTERNAL SKU (deterministic fallback) + When no real product ID can be found (common - the LLM-discovered + product may not exist verbatim on any marketplace, or the search + genuinely turns up nothing usable), a human-readable internal SKU is + generated instead, e.g. "AACHI-SAM-500-001" for + "Aachi Sambar Powder" at size "500g": + AACHI - brand code (first brand word, up to 6 chars) + SAM - product code (first meaningful word of the title, + first 3 letters; brand name and generic + filler words like "Original"/"Classic" + are skipped) + 500 - size code (leading numeric portion of the pack size) + 001 - sequence (persisted per BRAND-PRODUCT-SIZE prefix in + brand_catalog_{brand}.json under a + ``_sku_sequences`` key so repeat pipeline + runs hand out fresh, non-colliding numbers + instead of restarting at 001 every time) + +Both paths populate the same two catalog fields, so downstream code +(JSON export, DB storage, UI) never needs to know which path was taken: + product_sku - the ID/SKU string itself + sku_source - where it came from, e.g. "Amazon (ASIN)", + "Flipkart (PID)", or "Internal" +""" +from __future__ import annotations + +import json +import logging +import os +import re +import threading +from pathlib import Path +from typing import Any, Dict, Optional, Tuple + +from app.infrastructure.settings import DATA_DIR, ENABLE_SKU_WEB_LOOKUP +from app.services.brand_registry import resolve_parent_brand + +logger = logging.getLogger(__name__) + +_seq_lock = threading.Lock() + +# Deviation from the sibling project, deliberate on both counts. +# +# 1. Absolute, not relative. The sibling used Path("data"), which resolves +# against the current working directory - fine for a CLI run from the repo +# root, wrong for a uvicorn process started from anywhere else. +# 2. Its own directory, not the brand catalog files. The sibling stored the +# counter inside data/brand_catalog_.json. In this repo those files +# are real seed catalogs under data/seed_catalogs/, owned by brand_sync; +# export_brand_to_seed_file() rewrites them wholesale, which would silently +# drop the counters and start handing out duplicate SKUs. Keeping the +# counter in its own file leaves the seed catalogs untouched. +_data_dir = DATA_DIR / "sku_sequences" + + +def _brand_seq_file(brand: str) -> Path: + storage_brand = resolve_parent_brand(brand) + safe_brand = storage_brand.strip().replace(" & ", " and ").replace("&", "_and_").replace(" ", "_").replace("-", "_") + return _data_dir / f"{safe_brand}.json" + +# Words that shouldn't be used as the "meaningful" word when building the +# product code portion of an internal SKU. +_STOPWORDS = {"the", "and", "of", "with", "for", "in", "a", "an", "new", "pack", "combo", "pouch"} +_GENERIC_FILLER = {"original", "regular", "classic", "plain"} + +# Domain -> (regex to pull the ID out of the URL, human-readable source label). +# Matching happens directly against the result URL - none of these require +# fetching/parsing the actual product page. +_MARKETPLACE_PATTERNS: list[Tuple[str, "re.Pattern[str]", str]] = [ + ("amazon.in", re.compile(r"/dp/([A-Z0-9]{10})", re.I), "Amazon (ASIN)"), + ("amazon.com", re.compile(r"/dp/([A-Z0-9]{10})", re.I), "Amazon (ASIN)"), + ("flipkart.com", re.compile(r"[?&]pid=([A-Z0-9]+)", re.I), "Flipkart (PID)"), + ("bigbasket.com", re.compile(r"/pd/(\d+)/?"), "BigBasket (Product ID)"), + ("jiomart.com", re.compile(r"/p/[^/?]+/(\d+)"), "JioMart (Product ID)"), + ("blinkit.com", re.compile(r"/prn/[^/]+/prid/(\d+)"), "Blinkit (Product ID)"), + ("nykaa.com", re.compile(r"/p/(\d+)"), "Nykaa (Product ID)"), +] + +# Domains searched for a real product ID - kept in sync with the patterns above. +_SEARCH_DOMAINS = [d for d, _pat, _label in _MARKETPLACE_PATTERNS] + + +def _brand_code(brand: str) -> str: + if not brand: + return "GEN" + first = re.split(r"\s+", brand.strip())[0] + code = re.sub(r"[^A-Za-z0-9]", "", first).upper() + return code[:6] or "GEN" + + +def _product_code(title: str, brand: str = "") -> str: + """First meaningful word of the (brand-stripped) title -> first 3 letters. + + Uses only the base product name, not any " - Variety" suffix added by + the variety-explosion step, so different flavours of the same product + share the same product code (they're differentiated by the sequence + number instead). + """ + clean = title or "" + if brand and clean.lower().startswith(brand.strip().lower()): + clean = clean[len(brand.strip()):].strip() + base = clean.split(" - ")[0] + words = re.findall(r"[A-Za-z0-9]+", base) + for w in words: + wl = w.lower() + if wl in _STOPWORDS or wl in _GENERIC_FILLER: + continue + code = re.sub(r"[^A-Za-z0-9]", "", w).upper() + if code: + return code[:3] + return "PRD" + + +def _size_code(size: str) -> str: + if not size: + return "STD" + m = re.search(r"(\d+(?:\.\d+)?)", size) + if not m: + return "STD" + return m.group(1).replace(".", "") + + +def _atomic_write_json(path: Path, data: Dict[str, Any]) -> None: + """Write JSON to `path` atomically: write to a temp file in the same + directory, flush+fsync it, then os.replace() it onto the destination. + + os.replace() is atomic on both POSIX and Windows - the destination + file is guaranteed to either be the OLD complete content or the NEW + complete content, NEVER a half-written/truncated hybrid, no matter + when the process is killed. This is what prevents an accidental + termination (Ctrl+C, crash, killed terminal, power loss) from + corrupting data/brand_catalog_{brand}.json - which previously used a + plain write_text()/open(...,'w') that truncates the file in place + before writing the new content, so a kill mid-write left a half-empty + or invalid-JSON file that the *next* run would either silently reset + (SKU counters restarting at 0, causing duplicate SKUs) or, worse, + silently treat as "no existing catalog" and discard every previously + saved product for that brand (see save_catalog_to_data_folder() in + streamlit_app.py for the matching fix on the read/merge side). + """ + tmp_path = path.with_suffix(path.suffix + f".tmp{os.getpid()}") + text = json.dumps(data, indent=2, ensure_ascii=False) + with open(tmp_path, "w", encoding="utf-8") as f: + f.write(text) + f.flush() + os.fsync(f.fileno()) + os.replace(tmp_path, path) # atomic on POSIX and Windows + + +def _next_sequence(brand: str, prefix: str) -> int: + """File-backed counter stored inside the brand catalog JSON file + (data/brand_catalog_{brand}.json) under a reserved ``_sku_sequences`` + key, so re-running the pipeline for the same brand keeps handing out + fresh sequence numbers for a given BRAND-PRODUCT-SIZE prefix instead + of restarting at 001. + """ + seq_file = _brand_seq_file(brand) + with _seq_lock: + data: Dict[str, Any] = {} + counters: Dict[str, int] = {} + if seq_file.exists(): + try: + data = json.loads(seq_file.read_text(encoding="utf-8")) + counters = data.get("_sku_sequences", {}) + except Exception as e: + # The file exists but isn't valid JSON - almost always the + # result of a previous run being killed mid-write. Preserve + # the corrupt file under a .corrupt name instead of silently + # overwriting it, so no evidence/data is lost, and log + # loudly rather than quietly resetting counters to 0 (which + # would otherwise produce duplicate/colliding SKUs). + logger.warning( + f"SKU sequence file {seq_file} is corrupted ({e}); " + f"resetting counters for this brand. The unreadable " + f"file has been preserved for inspection." + ) + try: + seq_file.rename(seq_file.with_suffix(seq_file.suffix + ".corrupt")) + except Exception: + pass + data = {} + counters = {} + next_val = int(counters.get(prefix, 0)) + 1 + counters[prefix] = next_val + data["_sku_sequences"] = counters + try: + _data_dir.mkdir(parents=True, exist_ok=True) + _atomic_write_json(seq_file, data) + except Exception as e: + logger.warning(f"Could not persist SKU sequence counter: {e}") + return next_val + + +def generate_internal_sku(brand: str, product_title: str, size: str) -> str: + """Deterministic-format internal SKU, e.g. 'AACHI-SAM-500-001'.""" + prefix = f"{_brand_code(brand)}-{_product_code(product_title, brand)}-{_size_code(size)}" + seq = _next_sequence(brand, prefix) + return f"{prefix}-{seq:03d}" + + +def find_website_product_id(brand: str, product_title: str, size: str) -> Optional[Tuple[str, str]]: + """Best-effort search for a real marketplace product ID. + + Returns (product_id, source_label) or None. Never raises - any + network/parsing failure just means "no real ID found", which the + caller falls back to an internal SKU for. + """ + if not ENABLE_SKU_WEB_LOOKUP: + return None + + query = f"{brand} {product_title} {size}".strip() + if not query: + return None + + try: + from ddgs import DDGS + except ImportError: + logger.debug("SKU web lookup unavailable: 'ddgs' package not installed") + return None + + try: + with DDGS(timeout=10) as ddgs: + results = ddgs.text(query, region="in-en", safesearch="off", max_results=8) + except Exception as e: + logger.debug(f"SKU web lookup failed for '{query}': {e}") + return None + + for r in results or []: + url = str(r.get("href") or r.get("url") or "") + if not url: + continue + url_lower = url.lower() + for domain, pattern, label in _MARKETPLACE_PATTERNS: + if domain not in url_lower: + continue + m = pattern.search(url) + if m: + return m.group(1).upper(), label + + return None + + +def resolve_product_sku(brand: str, product_title: str, size: str) -> Dict[str, str]: + """Main entry point used by the catalog engine for each exploded + (brand, product/variety, pack size) catalog row. Tries a real + marketplace product ID first, falls back to a deterministic internal + SKU when none can be found. + """ + found = None + try: + found = find_website_product_id(brand, product_title, size) + except Exception as e: + logger.debug(f"SKU lookup error for '{brand} {product_title} {size}': {e}") + + if found: + product_id, source = found + return {"product_sku": product_id, "sku_source": source} + + internal_sku = generate_internal_sku(brand, product_title, size) + return {"product_sku": internal_sku, "sku_source": "Internal"} diff --git a/app/services/title_validator.py b/app/services/title_validator.py new file mode 100644 index 0000000..063f0fc --- /dev/null +++ b/app/services/title_validator.py @@ -0,0 +1,276 @@ +""" +Title <-> Category consistency guard. + +WHY THIS FILE EXISTS +--------------------- +Reported bug: brand="ITC" produced a catalog row titled +"ITC Engage Biscuit Box - Classic Salted". The CATEGORY came back correct +("Fragrance & Deodorants") because app.services.brand_registry. +get_sub_brand_category() matched the known ITC sub-brand "Engage" inside +the title and returned its real, curated category. The IMAGES came back +correct too, because image search is keyed off that same "Engage" match +plus the (correct) category. But nothing ever checked whether the TITLE +TEXT ITSELF was consistent with that resolved category - so the small +local LLM's hallucinated "Biscuit Box" (mixed in because it was asked for +ITC's "Biscuits & Cookies" category products and free-associated a real +ITC sub-brand name into the wrong product line) sailed straight through +into the stored catalog record unmodified. + +This module is the missing check. It is intentionally centralised in one +place rather than duplicated, and intentionally conservative: it only +strips a word/phrase from a title when that word/phrase is a *strong, +unambiguous* signal for a category OTHER than the one the product has +already been resolved to (deterministically, via the sub-brand registry, +or heuristically). Ambiguous words (e.g. "milk", which is valid across +several categories) are only flagged when NONE of their valid categories +match - never guessed away speculatively. + +USAGE +----- +Call `validate_and_fix_title(title, category, sub_brand_hint=...)` right +after a product's category has been resolved (deterministically or via +heuristic) and BEFORE that title is used for image search, S3 storage, +DB storage, or JSON export - see app/core/catalog_engine.py, which is the +canonical caller. It returns a (possibly corrected) title plus metadata +about what was changed, so the caller can log/audit every correction +instead of it happening silently. +""" +from __future__ import annotations + +import re +import logging +from typing import Iterable, Optional + +logger = logging.getLogger(__name__) + + +# --------------------------------------------------------------------------- +# Canonical word/phrase -> valid-categories map. +# +# Deliberately generous for ambiguous words (mapped to every category they +# could legitimately belong to) so this NEVER strips a word just because a +# product could conceivably be miscategorised - it only strips a word when +# the product's *actual* resolved category isn't anywhere in that word's +# allowed set, i.e. the word is a hard contradiction, not merely unusual. +# +# Multi-word phrases are matched first (longest-phrase-wins) so e.g. "body +# wash" is recognised as a single Skin & Bath Care signal rather than being +# torn apart into two separately-ambiguous words ("body", "wash"). +# --------------------------------------------------------------------------- +# +# NOTE ON WHAT IS DELIBERATELY *NOT* IN THIS MAP: +# Ingredient / flavour-descriptor words - "milk", "butter", "cheese", +# "chocolate", "honey", "cashew", "almond", "salt", etc. - are intentionally +# EXCLUDED even though they're each strongly associated with one category +# (Dairy, Chocolates, Dry Fruits & Nuts...). That's because they are also +# extremely common, completely legitimate FLAVOUR/variant modifiers inside +# OTHER categories' real product names - "Good Day Butter Cookies", "Milk +# Bikis", "Cashew Cookies", "Chocolate Cake", "Honey Oats" are all real +# products, not hallucinations. Flagging these words caused false-positive +# corrections that damaged genuine titles, so this map only contains STRONG +# signals: words that denote WHAT THE PRODUCT ITSELF IS (a head noun, not a +# flavour), which essentially never legitimately appear in another +# category's product name. This is what makes it safe to auto-strip them. +CATEGORY_TYPE_WORDS: dict[str, set[str]] = { + # --- Biscuits, bakery, snacks --- + "biscuit": {"Biscuits & Cookies"}, "biscuits": {"Biscuits & Cookies"}, + "cookie": {"Biscuits & Cookies"}, "cookies": {"Biscuits & Cookies"}, + "wafer": {"Biscuits & Cookies"}, "wafers": {"Biscuits & Cookies"}, + "cracker": {"Crackers"}, "crackers": {"Crackers"}, "saltine": {"Crackers"}, + "rusk": {"Rusk"}, "rusks": {"Rusk"}, + "cake": {"Cakes & Muffins"}, "cakes": {"Cakes & Muffins"}, + "muffin": {"Cakes & Muffins"}, "muffins": {"Cakes & Muffins"}, + "bread": {"Bakery & Breads"}, "bun": {"Bakery & Breads"}, "buns": {"Bakery & Breads"}, "loaf": {"Bakery & Breads"}, + "snack": {"Snacks"}, "snacks": {"Snacks"}, "chips": {"Snacks"}, "namkeen": {"Snacks"}, + "appalam": {"Snacks"}, "appalams": {"Snacks"}, "papad": {"Snacks"}, "bhujia": {"Snacks"}, + "soan papdi": {"Sweets"}, "gulab jamun": {"Sweets"}, "rasgulla": {"Sweets"}, + # --- Staples, spices, grains --- + "atta": {"Atta & Staples"}, "maida": {"Atta & Staples"}, "sooji": {"Atta & Staples"}, + "besan": {"Atta & Staples"}, "basmati rice": {"Rice & Pulses", "Atta & Staples"}, + "multigrain flour": {"Atta & Staples"}, + "masala": {"Spices & Masalas"}, "masalas": {"Spices & Masalas"}, + "curry powder": {"Spices & Masalas"}, "garam masala": {"Spices & Masalas"}, + "chilli powder": {"Spices & Masalas"}, "turmeric powder": {"Spices & Masalas"}, + "deggi mirch": {"Spices & Masalas"}, "tikhalal": {"Spices & Masalas"}, "kashmiri lal": {"Spices & Masalas"}, + "pasta": {"Pasta & Noodles"}, "vermicelli": {"Pasta & Noodles"}, "semiya": {"Pasta & Noodles"}, + "macaroni": {"Pasta & Noodles"}, "spaghetti": {"Pasta & Noodles"}, + "noodle": {"Pasta & Noodles", "Noodles & Instant Food"}, "noodles": {"Pasta & Noodles", "Noodles & Instant Food"}, + # --- Desserts (head-noun only, not ingredient words) --- + "ice cream": {"Ice Cream"}, "icecream": {"Ice Cream"}, "kulfi": {"Ice Cream"}, "frozen dessert": {"Ice Cream"}, + "shrikhand": {"Dairy - Desserts"}, "basundi": {"Dairy - Desserts"}, "mishti doi": {"Dairy - Desserts"}, + # --- Oils --- + "cooking oil": {"Cooking Oils"}, "sunflower oil": {"Cooking Oils"}, "groundnut oil": {"Cooking Oils"}, + "gingelly oil": {"Cooking Oils"}, "olive oil": {"Cooking Oils"}, "mustard oil": {"Cooking Oils"}, + "vegetable oil": {"Cooking Oils"}, "refined oil": {"Cooking Oils"}, "lamp oil": {"Household - Lamp Oil"}, + # --- Beverages --- + "tea": {"Tea & Coffee", "Beverages", "Health Drinks"}, "coffee": {"Tea & Coffee", "Beverages"}, + "juice": {"Beverages"}, "juices": {"Beverages"}, "squash": {"Beverages"}, + "aamras": {"Beverages"}, "jaljeera": {"Beverages"}, + "lassi": {"Dairy", "Beverages"}, "buttermilk": {"Beverages"}, "chaas": {"Beverages"}, + "milkshake": {"Beverages"}, "milkshakes": {"Beverages"}, "cold coffee": {"Beverages"}, "smoothie": {"Beverages"}, + # --- Confectionery / spreads --- + "candy": {"Candy & Confectionery"}, "toffee": {"Candy & Confectionery"}, + "lollipop": {"Candy & Confectionery"}, "eclair": {"Candy & Confectionery"}, "eclairs": {"Candy & Confectionery"}, + "pickle": {"Pickles & Chutneys"}, "pickles": {"Pickles & Chutneys"}, "achar": {"Pickles & Chutneys"}, + "chutney": {"Pickles & Chutneys"}, "murabba": {"Pickles & Chutneys"}, + "health mix": {"Health Foods"}, "sathumaavu": {"Health Foods"}, "ragi malt": {"Health Foods"}, + "millet": {"Health Foods"}, "millets": {"Health Foods"}, + # --- Health / malted drinks --- + "horlicks": {"Health Drinks"}, "bournvita": {"Health Drinks"}, "complan": {"Health Drinks"}, + "maltova": {"Health Drinks"}, "protinex": {"Health Drinks"}, "malted drink": {"Health Drinks"}, + "glucon": {"Health Drinks"}, + # --- Hair care --- + "shampoo": {"Hair Care"}, "conditioner": {"Hair Care"}, "hair oil": {"Hair Care"}, + "hair serum": {"Hair Care"}, "hair cream": {"Hair Care"}, "hair color": {"Hair Care"}, + "hair dye": {"Hair Care"}, "anti-dandruff": {"Hair Care"}, "hair fall": {"Hair Care"}, + # --- Bath / skin / beauty --- + "soap": {"Bath Soap", "Skin & Bath Care", "Personal Care"}, + "body wash": {"Skin & Bath Care", "Personal Care"}, "shower gel": {"Skin & Bath Care"}, + "hand wash": {"Skin & Bath Care"}, "hand cream": {"Skin & Bath Care"}, + "face wash": {"Skin Care", "Beauty Care"}, "face cream": {"Skin & Bath Care", "Skin Care"}, + "face powder": {"Beauty Care"}, "moisturiser": {"Skin Care", "Beauty Care"}, + "moisturizer": {"Skin Care", "Beauty Care"}, "sunscreen": {"Skin Care", "Beauty Care"}, "sunblock": {"Skin Care", "Beauty Care"}, + "foundation": {"Beauty Care"}, "lipstick": {"Beauty Care"}, "lip balm": {"Beauty Care"}, + "makeup": {"Beauty Care"}, "cosmetic": {"Beauty Care"}, "eyeliner": {"Beauty Care"}, + "kajal": {"Beauty Care"}, "mascara": {"Beauty Care"}, "nail polish": {"Beauty Care"}, + "fairness": {"Beauty Care"}, + # --- Oral care --- + "toothpaste": {"Oral Care"}, "toothbrush": {"Oral Care"}, "mouthwash": {"Oral Care"}, "dental": {"Oral Care"}, + # --- Fragrance --- + "deodorant": {"Fragrance & Deodorants"}, "deodorants": {"Fragrance & Deodorants"}, "deo": {"Fragrance & Deodorants"}, + "perfume": {"Fragrance & Deodorants"}, "perfumes": {"Fragrance & Deodorants"}, "fragrance": {"Fragrance & Deodorants"}, + "cologne": {"Fragrance & Deodorants"}, "body spray": {"Fragrance & Deodorants"}, "roll-on": {"Fragrance & Deodorants"}, + # --- Men's grooming --- + "shaving cream": {"Men's Grooming"}, "shaving foam": {"Men's Grooming"}, "after shave": {"Men's Grooming"}, + "razor": {"Men's Grooming"}, "blade": {"Men's Grooming"}, "trimmer": {"Men's Grooming"}, "beard": {"Men's Grooming"}, + # --- Home care / laundry --- + "detergent": {"Detergents & Fabric Care"}, "washing powder": {"Detergents & Fabric Care"}, + "fabric wash": {"Detergents & Fabric Care"}, "fabric care": {"Detergents & Fabric Care"}, + "dishwash": {"Household Cleaning"}, "floor cleaner": {"Household Cleaning"}, "toilet cleaner": {"Household Cleaning"}, + "fabric softener": {"Household Cleaning"}, "stain remover": {"Household Cleaning"}, "handwash": {"Household Cleaning"}, + "mosquito repellent": {"Mosquito Repellent"}, "air freshener": {"Air Freshener"}, + # --- Baby care --- + "baby powder": {"Baby Care"}, "baby lotion": {"Baby Care"}, "baby oil": {"Baby Care"}, "baby shampoo": {"Baby Care"}, +} + +# Sort once, longest-phrase-first, so multi-word phrases are matched +# before their component single words could be (see module docstring). +_SORTED_TYPE_WORDS = sorted(CATEGORY_TYPE_WORDS.keys(), key=len, reverse=True) + +# Connector/filler tokens that shouldn't be left dangling after a +# conflicting word is stripped out (e.g. "Engage - Classic Salted" is +# fine; "Engage - -" or "Engage with" is not). +_DANGLING_TOKENS = {"-", "&", "and", "with", "for", "the", "a", "an", "of"} + + +def find_category_conflicts(title: str, category: str) -> list[str]: + """Return the list of phrases in `title` that are strong signals for a + category OTHER than `category`. Empty list = no detected conflict + (does not mean the title is *proven* correct, just that nothing in it + contradicts the given category).""" + if not title or not category: + return [] + cat = category.strip() + text = " " + title.lower() + " " + conflicts: list[str] = [] + consumed = set() + for phrase in _SORTED_TYPE_WORDS: + pattern = r"(?= e) for s, e in consumed): + continue # already covered by a longer phrase match + allowed = CATEGORY_TYPE_WORDS[phrase] + if cat not in allowed: + conflicts.append(phrase) + consumed.add(span) + return conflicts + + +def sanitize_title_for_category(title: str, category: str) -> str: + """Strip any word/phrase from `title` that contradicts `category`, + then tidy up leftover punctuation/connector words. Pure text + transform - does not know about brands or sub-brands.""" + if not title or not category: + return title + cat = category.strip() + result = title + for phrase in _SORTED_TYPE_WORDS: + allowed = CATEGORY_TYPE_WORDS[phrase] + if cat in allowed: + continue + pattern = re.compile(r"(? tuple[str, bool, list[str]]: + """Authoritative title/category consistency check + repair. + + Returns (fixed_title, was_changed, removed_phrases). + + `category` should be the ALREADY-RESOLVED, trusted category for this + product (e.g. from brand_registry.get_sub_brand_category(), or the + pipeline's heuristic classification) - this function treats it as + ground truth and corrects the title to match it, never the reverse. + + `sub_brand_hint` (e.g. "engage") is used ONLY as a last-resort anchor + if stripping conflicting words leaves nothing usable behind - it is + never invented, only ever a value the caller already resolved + deterministically (e.g. via brand_registry.get_sub_brand_match()). + """ + if not title or not category or category in ("Uncategorized", "General"): + return title, False, [] + + conflicts = find_category_conflicts(title, category) + if not conflicts: + return title, False, [] + + fixed = sanitize_title_for_category(title, category) + + # If sanitizing collapsed the title to nothing (or to something too + # short/generic to stand alone, e.g. just leftover flavour words with + # no anchor), fall back to the known sub-brand or brand name so we + # never emit an empty/blank title. + if not fixed or len(fixed.strip()) < 2: + anchor = (sub_brand_hint or brand or "").strip() + # Preserve a trailing " - Variant" suffix from the original title + # if present (flavour/variant naming is independent of the + # category-word bug this function targets). + variant_suffix = "" + m = re.search(r"\s-\s(.+)$", title.strip()) + if m and not any(w in m.group(1).lower() for w in conflicts): + variant_suffix = f" - {m.group(1).strip()}" + fixed = (anchor.title() if anchor else category) + variant_suffix + fixed = fixed.strip() + + changed = fixed.strip().lower() != title.strip().lower() + if changed: + logger.warning( + f"⚠️ Title/category mismatch corrected: '{title}' -> '{fixed}' " + f"(category='{category}', removed={conflicts})" + ) + return fixed, changed, conflicts diff --git a/data/sku_sequences/Aachi.json b/data/sku_sequences/Aachi.json new file mode 100644 index 0000000..2b7b086 --- /dev/null +++ b/data/sku_sequences/Aachi.json @@ -0,0 +1,5 @@ +{ + "_sku_sequences": { + "AACHI-SAM-100": 1 + } +} \ No newline at end of file diff --git a/data/sku_sequences/britannia.json b/data/sku_sequences/britannia.json new file mode 100644 index 0000000..21aeda8 --- /dev/null +++ b/data/sku_sequences/britannia.json @@ -0,0 +1,9 @@ +{ + "_sku_sequences": { + "BRITAN-50-200": 11, + "BRITAN-50-100": 4, + "BRITAN-50-250": 2, + "BRITAN-50-500": 3, + "BRITAN-GOO-200": 3 + } +} \ No newline at end of file diff --git a/data/sku_sequences/cadbury.json b/data/sku_sequences/cadbury.json new file mode 100644 index 0000000..c3085d0 --- /dev/null +++ b/data/sku_sequences/cadbury.json @@ -0,0 +1,5 @@ +{ + "_sku_sequences": { + "CADBUR-DAI-100": 1 + } +} \ No newline at end of file diff --git a/tests/test_store_catalog_pipeline.py b/tests/test_store_catalog_pipeline.py new file mode 100644 index 0000000..801c196 --- /dev/null +++ b/tests/test_store_catalog_pipeline.py @@ -0,0 +1,283 @@ +"""Tests for the store-spreadsheet -> 11-stage pipeline -> brand table flow. + +Follows the pattern established by test_user_products_upload.py: monkeypatch +the storage/network boundary *on the pipeline module object* (it imports those +names directly), build real .xlsx fixtures in memory, and assert on what was +captured rather than on the HTTP status alone. + +Every test runs with use_llm=False and fetch_images=False so nothing here +touches Ollama or the open web. +""" +from __future__ import annotations + +import io + +import pytest + +from app.core import store_catalog_pipeline as pipeline + +openpyxl = pytest.importorskip("openpyxl") + + +def _sheet(headers, rows) -> bytes: + wb = openpyxl.Workbook() + ws = wb.active + ws.append(headers) + for row in rows: + ws.append(row) + buf = io.BytesIO() + wb.save(buf) + return buf.getvalue() + + +@pytest.fixture +def store(monkeypatch): + """A fake brand table. Returns the dict of image_id -> stored row.""" + table: dict = {} + + def fake_upsert(brand, rows, cleanup=False): + assert cleanup is False, "cleanup=True would delete the brand's existing catalog" + for row in rows: + table[row["image_id"]] = dict(row) + return len(rows) + + monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert) + monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: list(table.values())) + monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts]) + return table + + +def _run(content, **kw): + return pipeline.run_pipeline("store.xlsx", content, use_llm=False, fetch_images=False, **kw) + + +# --------------------------------------------------------------------------- +# Column mapping / brand resolution +# --------------------------------------------------------------------------- +def test_messy_headers_map_onto_catalog_fields(store): + content = _sheet( + ["Item Name", "Product Description", "Segment", "Net Weight", "Random Column"], + [["Britannia 50-50", "Salty biscuit", "Biscuits", "200g", "junk"]], + ) + result = _run(content) + assert result.recognised_columns["product_name"] == "Item Name" + assert result.recognised_columns["description"] == "Product Description" + assert result.recognised_columns["category"] == "Segment" + assert result.recognised_columns["size_variants"] == "Net Weight" + assert result.unrecognised_columns == ["Random Column"] + + +def test_brand_is_inferred_when_the_sheet_has_no_brand_column(store): + """Store files routinely omit the brand; it lives in the product name.""" + content = _sheet(["Item Name"], [["Britannia 50-50"], ["Cadbury Dairy Milk 100g"]]) + result = _run(content) + assert result.errors == [] + assert result.brands == ["Britannia", "Cadbury"] + + +def test_longest_alias_wins_when_inferring_a_brand(): + """'cadbury dairy milk' must beat the shorter 'cadbury' substring.""" + assert pipeline.infer_brand("Cadbury Dairy Milk Silk 100g") == "cadbury dairy milk" + + +def test_rows_land_in_the_brand_table_the_product_belongs_to(store): + content = _sheet( + ["Item Name", "Net Weight"], + [["Britannia 50-50", "200g"], ["Aachi Sambar Powder", "100g"]], + ) + _run(content) + assert sorted(store) == ["aachi_aachi_sambar_powder_100g", "britannia_britannia_50_50_200g"] + + +# --------------------------------------------------------------------------- +# Gap filling +# --------------------------------------------------------------------------- +def test_missing_fields_are_filled_by_the_pipeline(store): + """A sheet with only a product name still produces a complete row.""" + content = _sheet(["Item Name"], [["Britannia Good Day Biscuits 200g"]]) + _run(content) + + row = next(iter(store.values())) + assert row["fssai_license"] == "10012022000103" # stage 1 + assert row["category"] # stage 3 + assert row["price_range"].startswith("₹") # stage 5 + assert row["product_sku"] # stage 7 + assert row["hsn_code"] # stage 9 + assert row["search_query"] # stage 11 + + +def test_values_supplied_by_the_store_are_never_overwritten(store): + content = _sheet( + ["Item Name", "Net Weight", "MRP Range", "HSN Code"], + [["Britannia 50-50", "200g", "₹111-222", "9999"]], + ) + _run(content) + row = next(iter(store.values())) + assert row["price_range"] == "₹111-222" + assert row["hsn_code"] == "9999" + + +def test_an_uncategorised_row_gets_no_hsn_code(store): + """HSN/GST is a regulatory value keyed on category. When the category + cannot be determined the column is deliberately left empty rather than + guessed - a wrong tax code is worse than a missing one.""" + content = _sheet(["Item Name"], [["Britannia 50-50"]]) + result = _run(content) + + row = next(iter(store.values())) + assert row["category"] == "General" + assert row["hsn_code"] is None + assert any("category could not be detected" in w for w in result.warnings) + + +def test_non_food_brands_get_no_fssai_licence(store): + """A miss means 'not a food brand', not an error - the column stays empty.""" + assert pipeline.get_fssai_license("Colgate") is None + + +# --------------------------------------------------------------------------- +# Stage 4 - pack-size explosion and unit safety +# --------------------------------------------------------------------------- +def test_one_row_explodes_into_one_row_per_pack_size(store): + content = _sheet(["Item Name", "Pack Size"], [["Britannia 50-50", "100g, 200g, 500g"]]) + _run(content) + assert len(store) == 3 + assert {r["size_variants"][0] for r in store.values()} == {"100g", "200g", "500g"} + + +def test_each_pack_size_gets_its_own_image_id(store): + """Without the size in the id every variant collapses onto one row under + ON CONFLICT (image_id).""" + content = _sheet(["Item Name", "Pack Size"], [["Britannia 50-50", "100g, 200g"]]) + _run(content) + assert len(set(store)) == 2 + + +def test_the_size_is_not_duplicated_when_already_in_the_product_name(): + assert pipeline.build_image_id("Cadbury", "Cadbury Dairy Milk 100g", "100g") == \ + "cadbury_cadbury_dairy_milk_100g" + + +def test_a_pack_size_with_a_nonsense_unit_is_dropped_with_a_reason(store): + """A biscuit measured in centimetres is a data error, not a pack size.""" + content = _sheet(["Item Name", "Segment", "Pack Size"], [["Britannia 50-50", "Biscuits", "15cm"]]) + result = _run(content) + assert any("15cm" in w for w in result.warnings) + + +# --------------------------------------------------------------------------- +# Stage 11 - storage semantics +# --------------------------------------------------------------------------- +def test_reingesting_the_same_file_changes_nothing(store): + content = _sheet(["Item Name", "Net Weight"], [["Britannia 50-50", "200g"]]) + first = _run(content) + assert (first.inserted, first.skipped_existing) == (1, 0) + + second = _run(content) + assert (second.inserted, second.backfilled, second.skipped_existing) == (0, 0, 1) + assert len(store) == 1 + + +def test_an_existing_row_has_only_its_blank_columns_backfilled(store): + content = _sheet(["Item Name", "Net Weight"], [["Britannia Good Day Biscuits", "200g"]]) + _run(content) + + key = next(iter(store)) + store[key]["hsn_code"] = None + store[key]["price_range"] = "₹999-1000" # a value already held + + result = _run(content) + assert result.backfilled == 1 + assert store[key]["hsn_code"] == "1905" # blank -> filled + assert store[key]["price_range"] == "₹999-1000" # held value untouched + + +def test_a_storage_failure_is_reported_rather_than_counted_as_success(store, monkeypatch): + def boom(brand, rows, cleanup=False): + raise RuntimeError("pgvector is down") + + monkeypatch.setattr(pipeline, "upsert_brand_products", boom) + result = _run(_sheet(["Item Name"], [["Britannia 50-50 200g"]])) + assert result.storage_error and "pgvector is down" in result.storage_error + assert result.inserted == 0 + + +# --------------------------------------------------------------------------- +# Row-level error handling +# --------------------------------------------------------------------------- +def test_a_row_with_no_usable_product_name_is_reported_not_fatal(store): + content = _sheet(["Item Name", "Net Weight"], [["", "200g"], ["Britannia 50-50", "200g"]]) + result = _run(content) + assert len(result.errors) == 1 + assert result.errors[0].row == 2 # header is row 1 + assert result.inserted == 1 # the good row still landed + + +def test_the_same_pack_listed_twice_is_written_once(store): + content = _sheet( + ["Item Name", "Net Weight"], + [["Britannia 50-50", "200g"], ["Britannia 50-50", "200g"]], + ) + _run(content) + assert len(store) == 1 + + +# --------------------------------------------------------------------------- +# HTTP surface +# --------------------------------------------------------------------------- +def test_preview_reports_how_the_columns_were_understood(client, admin_headers): + content = _sheet(["Item Name", "Segment", "Mystery"], [["Britannia 50-50", "Biscuits", "?"]]) + resp = client.post( + "/api/admin/store-catalog/preview", + files={"file": ("store.xlsx", content)}, + headers=admin_headers, + ) + assert resp.status_code == 200 + body = resp.json() + assert body["recognised_columns"]["product_name"] == "Item Name" + assert body["unrecognised_columns"] == ["Mystery"] + assert body["brand_column_present"] is False + assert len(body["stages"]) == 11 + + +def test_preview_requires_admin(client): + content = _sheet(["Item Name"], [["Britannia 50-50"]]) + resp = client.post("/api/admin/store-catalog/preview", files={"file": ("s.xlsx", content)}) + assert resp.status_code == 401 + + +def test_an_unparseable_file_is_rejected_before_a_job_is_created(client, admin_headers): + resp = client.post( + "/api/admin/store-catalog/ingest", + files={"file": ("notes.txt", b"this is not a spreadsheet")}, + headers=admin_headers, + ) + assert resp.status_code == 400 + + +def test_ingest_returns_a_job_id_that_can_be_polled(client, admin_headers, monkeypatch): + # Keep the worker off the network and out of the database. + monkeypatch.setattr(pipeline, "upsert_brand_products", lambda b, r, cleanup=False: len(r)) + monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: []) + monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts]) + + content = _sheet(["Item Name", "Segment"], [["Britannia 50-50", "Biscuits"]]) + resp = client.post( + "/api/admin/store-catalog/ingest", + files={"file": ("store.xlsx", content)}, + headers=admin_headers, + ) + assert resp.status_code == 202 + job_id = resp.json()["job_id"] + assert resp.json()["rows_total"] == 1 + + poll = client.get(f"/api/admin/store-catalog/jobs/{job_id}", headers=admin_headers) + assert poll.status_code == 200 + body = poll.json() + assert body["status"] in {"pending", "running", "done", "failed"} + assert body["total_stages"] == 11 + + +def test_polling_an_unknown_job_is_a_404(client, admin_headers): + resp = client.get("/api/admin/store-catalog/jobs/does-not-exist", headers=admin_headers) + assert resp.status_code == 404