backend stores data file enrichment pipeline
This commit is contained in:
183
app/api/routers/store_catalog.py
Normal file
183
app/api/routers/store_catalog.py
Normal file
@@ -0,0 +1,183 @@
|
||||
"""Admin endpoints for turning a store spreadsheet into brand-catalog rows.
|
||||
|
||||
Three endpoints, matching how the other long-running admin jobs in this
|
||||
project are exposed (see `nutrition_admin.py`):
|
||||
|
||||
POST /api/admin/store-catalog/preview - parse only, show the column map
|
||||
POST /api/admin/store-catalog/ingest - 202 + job_id, runs in background
|
||||
GET /api/admin/store-catalog/jobs/{id} - poll stage + row progress
|
||||
|
||||
`preview` exists because the mapping from a store's headers onto catalog
|
||||
fields is a guess. Running an 11-stage scrape over 2000 rows only to discover
|
||||
that "Item" was read as the description is expensive; showing the operator the
|
||||
mapping first costs one parse.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Optional
|
||||
|
||||
from fastapi import APIRouter, Depends, File, HTTPException, UploadFile, status
|
||||
from pydantic import BaseModel
|
||||
|
||||
from app.api.background import run_in_background
|
||||
from app.api.deps import require_admin
|
||||
from app.api.store_catalog_job_store import store_catalog_job_store
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
router = APIRouter(prefix="/admin/store-catalog", tags=["admin", "catalog"])
|
||||
|
||||
# Same ceilings as the user-products upload path, for the same reasons.
|
||||
MAX_UPLOAD_BYTES = 10 * 1024 * 1024
|
||||
MAX_UPLOAD_ROWS = 2000
|
||||
PREVIEW_ROWS = 10
|
||||
|
||||
|
||||
class StoreCatalogJobOut(BaseModel):
|
||||
job_id: str
|
||||
filename: str
|
||||
status: str
|
||||
detail: Optional[str] = None
|
||||
stage_index: int = 0
|
||||
stage_name: str = ""
|
||||
total_stages: int = pipeline.TOTAL_STAGES
|
||||
rows_done: int = 0
|
||||
rows_total: int = 0
|
||||
result: Optional[dict] = None
|
||||
|
||||
|
||||
async def _read_upload(file: UploadFile) -> bytes:
|
||||
contents = await file.read()
|
||||
if not contents:
|
||||
raise HTTPException(status_code=400, detail="The uploaded file is empty.")
|
||||
if len(contents) > MAX_UPLOAD_BYTES:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=f"File is larger than the {MAX_UPLOAD_BYTES // (1024 * 1024)}MB limit.",
|
||||
)
|
||||
return contents
|
||||
|
||||
|
||||
@router.post("/preview", dependencies=[Depends(require_admin)])
|
||||
async def preview_store_catalog(file: UploadFile = File(...)) -> dict:
|
||||
"""Parse the sheet and report how its columns were understood."""
|
||||
contents = await _read_upload(file)
|
||||
try:
|
||||
df, mapping = pipeline.parse_spreadsheet(file.filename or "upload.xlsx", contents)
|
||||
except HTTPException:
|
||||
raise
|
||||
except ValueError as exc:
|
||||
raise HTTPException(status_code=400, detail=str(exc)) from exc
|
||||
except Exception as exc: # noqa: BLE001 - unreadable file is a user error
|
||||
raise HTTPException(status_code=400, detail=f"Could not parse the file: {exc}") from exc
|
||||
|
||||
if df.empty:
|
||||
raise HTTPException(status_code=400, detail="The file has no data rows.")
|
||||
if len(df) > MAX_UPLOAD_ROWS:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=f"{len(df)} rows exceeds the {MAX_UPLOAD_ROWS}-row limit for one upload.",
|
||||
)
|
||||
|
||||
sample = df.head(PREVIEW_ROWS).fillna("").astype(str).to_dict(orient="records")
|
||||
return {
|
||||
"filename": file.filename,
|
||||
"rows_total": int(len(df)),
|
||||
"recognised_columns": {f: str(c) for f, c in mapping.columns.items()},
|
||||
"ignored_columns": mapping.ignored,
|
||||
"unrecognised_columns": mapping.unrecognised,
|
||||
"brand_column_present": "brand" in mapping.columns,
|
||||
"preview": sample,
|
||||
"stages": list(pipeline.STAGE_NAMES),
|
||||
}
|
||||
|
||||
|
||||
def _run_ingest_job(job_id: str, filename: str, contents: bytes,
|
||||
use_llm: bool, fetch_images: bool) -> None:
|
||||
store_catalog_job_store.update(job_id, status="running", detail="Starting pipeline")
|
||||
|
||||
def progress(stage_index: int, stage_name: str, done: int, total: int) -> None:
|
||||
store_catalog_job_store.update(
|
||||
job_id, stage_index=stage_index, stage_name=stage_name,
|
||||
rows_done=done, rows_total=total,
|
||||
)
|
||||
|
||||
try:
|
||||
result = pipeline.run_pipeline(
|
||||
filename, contents, progress=progress,
|
||||
use_llm=use_llm, fetch_images=fetch_images,
|
||||
)
|
||||
body = result.as_dict()
|
||||
if body.get("storage_error"):
|
||||
# Rows were built but none reached the database. Reporting this as
|
||||
# success would leave the operator believing the catalog changed.
|
||||
store_catalog_job_store.update(
|
||||
job_id, status="failed", result=body,
|
||||
detail=f"Built {body['products_built']} row(s) but storing them failed: "
|
||||
f"{body['storage_error']}",
|
||||
)
|
||||
return
|
||||
store_catalog_job_store.update(
|
||||
job_id, status="done", result=body,
|
||||
detail=(f"{body['inserted']} inserted, {body['backfilled']} backfilled, "
|
||||
f"{body['skipped_existing']} unchanged, {body['rejected']} rejected"),
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 - a daemon thread must not die silently
|
||||
logger.exception("Store-catalog ingestion job %s failed", job_id)
|
||||
store_catalog_job_store.update(job_id, status="failed", detail=str(exc))
|
||||
|
||||
|
||||
@router.post("/ingest", status_code=status.HTTP_202_ACCEPTED,
|
||||
dependencies=[Depends(require_admin)])
|
||||
async def ingest_store_catalog(
|
||||
file: UploadFile = File(...),
|
||||
use_llm: bool = True,
|
||||
fetch_images: bool = True,
|
||||
) -> StoreCatalogJobOut:
|
||||
"""Kick off the 11-stage pipeline and return a job id to poll.
|
||||
|
||||
The file is read here rather than in the worker: `UploadFile` is backed by
|
||||
a temporary file tied to the request, so it is gone by the time a
|
||||
background thread would get to it.
|
||||
"""
|
||||
contents = await _read_upload(file)
|
||||
filename = file.filename or "upload.xlsx"
|
||||
|
||||
# Fail fast on an unparseable file so the caller gets a 400 now rather than
|
||||
# a job that transitions straight to "failed".
|
||||
try:
|
||||
df, _mapping = pipeline.parse_spreadsheet(filename, contents)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
raise HTTPException(status_code=400, detail=f"Could not parse the file: {exc}") from exc
|
||||
if df.empty:
|
||||
raise HTTPException(status_code=400, detail="The file has no data rows.")
|
||||
if len(df) > MAX_UPLOAD_ROWS:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=f"{len(df)} rows exceeds the {MAX_UPLOAD_ROWS}-row limit for one upload.",
|
||||
)
|
||||
|
||||
job = store_catalog_job_store.create(filename)
|
||||
store_catalog_job_store.update(job.job_id, rows_total=int(len(df)))
|
||||
run_in_background(
|
||||
lambda: _run_ingest_job(job.job_id, filename, contents, use_llm, fetch_images),
|
||||
name=f"store-catalog-{job.job_id[:8]}",
|
||||
)
|
||||
return StoreCatalogJobOut(
|
||||
job_id=job.job_id, filename=filename, status=job.status,
|
||||
rows_total=int(len(df)),
|
||||
)
|
||||
|
||||
|
||||
@router.get("/jobs/{job_id}", dependencies=[Depends(require_admin)])
|
||||
def get_store_catalog_job(job_id: str) -> StoreCatalogJobOut:
|
||||
job = store_catalog_job_store.get(job_id)
|
||||
if not job:
|
||||
raise HTTPException(status_code=404, detail="Job not found")
|
||||
return StoreCatalogJobOut(
|
||||
job_id=job.job_id, filename=job.filename, status=job.status, detail=job.detail,
|
||||
stage_index=job.stage_index, stage_name=job.stage_name,
|
||||
total_stages=job.total_stages, rows_done=job.rows_done,
|
||||
rows_total=job.rows_total, result=job.result,
|
||||
)
|
||||
93
app/api/store_catalog_job_store.py
Normal file
93
app/api/store_catalog_job_store.py
Normal file
@@ -0,0 +1,93 @@
|
||||
"""Job tracking for store-spreadsheet catalog ingestion.
|
||||
|
||||
Same pattern and the same documented trade-offs as `job_store.py`,
|
||||
`store_job_store.py` and `nutrition_job_store.py`: a process-local dict behind
|
||||
a lock, lost on restart, not shared across uvicorn workers. Adding a broker for
|
||||
this would be operational weight the project has already decided against (see
|
||||
`job_store.py`).
|
||||
|
||||
It is its own module rather than a reuse of `nutrition_job_store` because this
|
||||
job reports a different shape of progress: an 11-stage pipeline needs to say
|
||||
*which stage* it is on as well as how many rows it has finished, so the UI can
|
||||
show "Stage 6/11 - Image Search, 48/120 rows".
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
import time
|
||||
import uuid
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Dict, Optional
|
||||
|
||||
|
||||
@dataclass
|
||||
class StoreCatalogJob:
|
||||
job_id: str
|
||||
filename: str
|
||||
status: str = "pending" # pending -> running -> done | failed
|
||||
detail: Optional[str] = None
|
||||
result: Optional[dict] = None
|
||||
stage_index: int = 0 # 1-based; 0 while still pending
|
||||
stage_name: str = ""
|
||||
total_stages: int = 11
|
||||
rows_done: int = 0
|
||||
rows_total: int = 0
|
||||
created_at: float = field(default_factory=time.time)
|
||||
updated_at: float = field(default_factory=time.time)
|
||||
|
||||
|
||||
class StoreCatalogJobStore:
|
||||
def __init__(self) -> None:
|
||||
self._jobs: Dict[str, StoreCatalogJob] = {}
|
||||
self._lock = threading.Lock()
|
||||
|
||||
def create(self, filename: str) -> StoreCatalogJob:
|
||||
job = StoreCatalogJob(job_id=str(uuid.uuid4()), filename=filename)
|
||||
with self._lock:
|
||||
self._jobs[job.job_id] = job
|
||||
return job
|
||||
|
||||
def update(
|
||||
self,
|
||||
job_id: str,
|
||||
status: Optional[str] = None,
|
||||
detail: Optional[str] = None,
|
||||
result: Optional[dict] = None,
|
||||
stage_index: Optional[int] = None,
|
||||
stage_name: Optional[str] = None,
|
||||
rows_done: Optional[int] = None,
|
||||
rows_total: Optional[int] = None,
|
||||
) -> None:
|
||||
"""Every field is optional and None means "leave alone".
|
||||
|
||||
That matters: `job_store.update` sets `detail` unconditionally, so
|
||||
marking a job running there wipes whatever detail it already had. A
|
||||
progress callback firing many times per second must not erase state it
|
||||
was not asked to change.
|
||||
"""
|
||||
with self._lock:
|
||||
job = self._jobs.get(job_id)
|
||||
if not job:
|
||||
return
|
||||
if status is not None:
|
||||
job.status = status
|
||||
if detail is not None:
|
||||
job.detail = detail
|
||||
if result is not None:
|
||||
job.result = result
|
||||
if stage_index is not None:
|
||||
job.stage_index = stage_index
|
||||
if stage_name is not None:
|
||||
job.stage_name = stage_name
|
||||
if rows_done is not None:
|
||||
job.rows_done = rows_done
|
||||
if rows_total is not None:
|
||||
job.rows_total = rows_total
|
||||
job.updated_at = time.time()
|
||||
|
||||
def get(self, job_id: str) -> Optional[StoreCatalogJob]:
|
||||
with self._lock:
|
||||
return self._jobs.get(job_id)
|
||||
|
||||
|
||||
store_catalog_job_store = StoreCatalogJobStore()
|
||||
687
app/core/store_catalog_pipeline.py
Normal file
687
app/core/store_catalog_pipeline.py
Normal file
@@ -0,0 +1,687 @@
|
||||
"""
|
||||
Store-spreadsheet -> 11-stage enrichment -> brand table.
|
||||
|
||||
A store sends a product list as Excel/CSV. The columns roughly resemble the
|
||||
brand-catalog schema but are named inconsistently and are always incomplete -
|
||||
typically product_name, description, category and little else. This module
|
||||
turns such a file into proper catalog rows: it scrapes/derives everything the
|
||||
sheet did not supply, validates the result, and writes it to the brand table
|
||||
the product belongs to (Britannia 50-50 -> brand_britannia).
|
||||
|
||||
Relationship to the existing brand pipeline
|
||||
-------------------------------------------
|
||||
`app/core/catalog_engine.py` starts from a BRAND NAME and asks an LLM to
|
||||
enumerate its products. This module starts from a SPREADSHEET ROW, so the
|
||||
discovery stage is replaced by row intake and everything downstream is
|
||||
gap-filling. catalog_engine is not modified or called for orchestration; only
|
||||
its `_select_best_images` contamination filter is reused.
|
||||
|
||||
Stage order (the numbering the operator sees in the UI):
|
||||
|
||||
1. Brand Resolution & FSSAI Licence Mapping
|
||||
2. Row Intake & Normalisation (replaces AI product discovery)
|
||||
3. Title & Category Consistency Guard
|
||||
4. Pack-Size Variant Explosion & Unit Safety
|
||||
5. Pricing Band Estimation
|
||||
6. Image Search & Contamination Filtering
|
||||
7. Marketplace & Internal SKU Resolution
|
||||
8. Barcode Retrieval & Enrichment
|
||||
9. HSN / GST Tax Enrichment
|
||||
10. Deterministic Product Validation Gate
|
||||
11. Vector Embedding & Storage
|
||||
|
||||
Two rules hold throughout:
|
||||
|
||||
* **Fill only blanks.** The store's own data is authoritative; a stage that
|
||||
finds a value already present leaves it alone. This is what separates the
|
||||
pipeline from `user_products._build_product_dict`, which fills the same
|
||||
gaps with static defaults ("Rs.100-250", a canonical S3 URL). Here the
|
||||
defaults are replaced by real lookups.
|
||||
* **A stage never raises.** Enrichment is best-effort; a barcode lookup
|
||||
timing out must not lose the row. Failures are recorded per row and the
|
||||
row continues with that field empty.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Callable, Dict, List, Optional, Tuple
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
ENABLE_BARCODE_LOOKUP,
|
||||
ENABLE_HSN_GST_ENRICHMENT,
|
||||
ENABLE_PRODUCT_VALIDATION,
|
||||
ENABLE_SKU_WEB_LOOKUP,
|
||||
MAX_VARIANTS_PER_PRODUCT,
|
||||
USE_EMBEDDINGS,
|
||||
)
|
||||
|
||||
# The spreadsheet machinery already exists and is well tested; importing it
|
||||
# from the router module is deliberate. Extracting it into a service would be
|
||||
# tidier, but user_products.py is a working file this feature does not touch,
|
||||
# and tests/test_user_products_upload.py monkeypatches these names on that
|
||||
# module object.
|
||||
from app.api.routers.user_products import (
|
||||
AddProductRequest,
|
||||
_slugify,
|
||||
_text,
|
||||
map_spreadsheet_columns,
|
||||
read_products_dataframe,
|
||||
row_to_request,
|
||||
)
|
||||
from app.services import price_estimator
|
||||
from app.services.brand_registry import (
|
||||
BRAND_ALIASES,
|
||||
get_fssai_license,
|
||||
resolve_parent_brand,
|
||||
)
|
||||
from app.services.category_registry import (
|
||||
detect_category_from_text,
|
||||
sanitize_category_language,
|
||||
)
|
||||
from app.services.category_units import fix_or_reject_size
|
||||
from app.services.embeddings_service import embed_texts
|
||||
from app.services.enrichment.barcode.stage import BarcodeEnrichmentStage
|
||||
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
|
||||
from app.services.enrichment.pipeline import EnrichmentPipeline
|
||||
from app.services.product_validator import validate_catalog
|
||||
from app.services.sku_service import resolve_product_sku
|
||||
from app.services.title_validator import validate_and_fix_title
|
||||
from app.services.vector_store import (
|
||||
_sanitize_name,
|
||||
display_name_for_suffix,
|
||||
get_products_by_brand,
|
||||
upsert_brand_products,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
STAGE_NAMES: Tuple[str, ...] = (
|
||||
"Brand Resolution & FSSAI Mapping",
|
||||
"Row Intake & Normalisation",
|
||||
"Title & Category Consistency Guard",
|
||||
"Pack-Size Variant Explosion & Unit Safety",
|
||||
"Pricing Band Estimation",
|
||||
"Image Search & Contamination Filtering",
|
||||
"Marketplace & Internal SKU Resolution",
|
||||
"Barcode Retrieval & Enrichment",
|
||||
"HSN / GST Tax Enrichment",
|
||||
"Deterministic Product Validation Gate",
|
||||
"Vector Embedding & Storage",
|
||||
)
|
||||
|
||||
TOTAL_STAGES = len(STAGE_NAMES)
|
||||
|
||||
# Columns a store row may legitimately leave blank and the pipeline fills in.
|
||||
# Anything not listed here is copied through untouched.
|
||||
_ENRICHABLE = (
|
||||
"category", "description", "fssai_license", "price_range", "product_sku",
|
||||
"sku_source", "hsn_code", "barcode", "barcode_type", "final_selling_price",
|
||||
"selling_price", "image_url", "image_urls", "size_variants",
|
||||
)
|
||||
|
||||
_SIZE_IN_TITLE = re.compile(
|
||||
r"\b(\d+(?:[.,]\d+)?)\s*(kg|kgs|g|gm|gms|gram|grams|ml|l|ltr|litre|liters?|pack|pcs|n)\b",
|
||||
re.I,
|
||||
)
|
||||
|
||||
|
||||
def _blank(value: Any) -> bool:
|
||||
"""True if a field carries no usable value."""
|
||||
if value is None:
|
||||
return True
|
||||
if isinstance(value, str):
|
||||
return not value.strip()
|
||||
if isinstance(value, (list, tuple, dict, set)):
|
||||
return len(value) == 0
|
||||
return False
|
||||
|
||||
|
||||
@dataclass
|
||||
class RowError:
|
||||
row: int
|
||||
product_name: str
|
||||
error: str
|
||||
|
||||
|
||||
@dataclass
|
||||
class PipelineResult:
|
||||
"""Everything the UI needs to explain what happened to an upload."""
|
||||
|
||||
rows_total: int = 0
|
||||
products_built: int = 0
|
||||
inserted: int = 0
|
||||
backfilled: int = 0
|
||||
skipped_existing: int = 0
|
||||
rejected: int = 0
|
||||
brands: List[str] = field(default_factory=list)
|
||||
rejections: List[Dict[str, Any]] = field(default_factory=list)
|
||||
errors: List[RowError] = field(default_factory=list)
|
||||
warnings: List[str] = field(default_factory=list)
|
||||
recognised_columns: Dict[str, str] = field(default_factory=dict)
|
||||
unrecognised_columns: List[str] = field(default_factory=list)
|
||||
storage_error: Optional[str] = None
|
||||
|
||||
def as_dict(self) -> Dict[str, Any]:
|
||||
return {
|
||||
"rows_total": self.rows_total,
|
||||
"products_built": self.products_built,
|
||||
"inserted": self.inserted,
|
||||
"backfilled": self.backfilled,
|
||||
"skipped_existing": self.skipped_existing,
|
||||
"rejected": self.rejected,
|
||||
"brands": self.brands,
|
||||
"rejections": self.rejections[:50],
|
||||
"errors": [vars(e) for e in self.errors[:50]],
|
||||
"error_count": len(self.errors),
|
||||
"warnings": self.warnings,
|
||||
"recognised_columns": self.recognised_columns,
|
||||
"unrecognised_columns": self.unrecognised_columns,
|
||||
"storage_error": self.storage_error,
|
||||
}
|
||||
|
||||
|
||||
ProgressFn = Callable[[int, str, int, int], None]
|
||||
"""Called as (stage_index_1_based, stage_name, rows_done, rows_total)."""
|
||||
|
||||
|
||||
def _noop_progress(stage: int, name: str, done: int, total: int) -> None:
|
||||
return None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 1 - Brand resolution & FSSAI
|
||||
# ---------------------------------------------------------------------------
|
||||
_INFERRED_BRAND_COL = "__inferred_brand__"
|
||||
|
||||
|
||||
def infer_brand(product_name: str) -> Optional[str]:
|
||||
"""Work out which brand a product name belongs to.
|
||||
|
||||
Store files very often have no brand column at all - the brand is simply
|
||||
the first word or two of the product name. Matches the longest known alias
|
||||
appearing anywhere in the name, so "britannia 50-50" resolves; longest-first
|
||||
matters, because "cadbury dairy milk" must win over "cadbury".
|
||||
|
||||
Runs before `row_to_request`, which rejects a brand-less row outright.
|
||||
"""
|
||||
haystack = f" {(product_name or '').lower().strip()} "
|
||||
best: Optional[str] = None
|
||||
for alias in BRAND_ALIASES:
|
||||
if re.search(rf"(?<![a-z0-9]){re.escape(alias)}(?![a-z0-9])", haystack):
|
||||
if best is None or len(alias) > len(best):
|
||||
best = alias
|
||||
if best:
|
||||
return best
|
||||
|
||||
# Last resort: assume the leading word names the brand. Better than
|
||||
# dropping the row, and the validation gate will catch nonsense later.
|
||||
first = (product_name or "").strip().split()
|
||||
return first[0] if first else None
|
||||
|
||||
|
||||
def stage_1_brand_and_fssai(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
brand = row.get("brand") or ""
|
||||
parent = resolve_parent_brand(brand)
|
||||
row["brand"] = parent
|
||||
row["brand_name"] = parent
|
||||
row["_table"] = f"brand_{_sanitize_name(parent)}"
|
||||
if _blank(row.get("fssai_license")):
|
||||
# None means "not a food brand" as much as "unknown", so an empty
|
||||
# column is a legitimate outcome, never an error.
|
||||
row["fssai_license"] = get_fssai_license(parent)
|
||||
return row
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 2 - Row intake (replaces AI discovery)
|
||||
# ---------------------------------------------------------------------------
|
||||
def stage_2_row_intake(row: Dict[str, Any], *, use_llm: bool = True) -> Dict[str, Any]:
|
||||
"""The spreadsheet row *is* the product. Only fill a missing description.
|
||||
|
||||
The LLM call is best-effort and optional: with Ollama unreachable the row
|
||||
keeps an empty description and later stages still work.
|
||||
"""
|
||||
if not _blank(row.get("description")) or not use_llm:
|
||||
return row
|
||||
try:
|
||||
from app.services.ollama_service import fetch_product_details
|
||||
|
||||
details = fetch_product_details(row.get("brand", ""), row.get("product_name", ""))
|
||||
if details and details.get("description"):
|
||||
row["description"] = str(details["description"]).strip()
|
||||
if _blank(row.get("size_variants")) and details and details.get("size_variants"):
|
||||
row["size_variants"] = list(details["size_variants"])
|
||||
except Exception as exc: # noqa: BLE001 - enrichment is best-effort
|
||||
logger.debug("LLM enrichment skipped for %r: %s", row.get("product_name"), exc)
|
||||
return row
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 3 - Title & category consistency
|
||||
# ---------------------------------------------------------------------------
|
||||
def stage_3_title_category(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
title = row.get("title") or row.get("product_name") or ""
|
||||
category = row.get("category")
|
||||
|
||||
if _blank(category):
|
||||
category = detect_category_from_text(f"{title} {row.get('description') or ''}")
|
||||
row["_category_deterministic"] = bool(category)
|
||||
if not category:
|
||||
# "General" rather than None: the column is not nullable in
|
||||
# practice and _to_storage_row already assumed this default. The
|
||||
# row is still reported as un-categorised in the job warnings, so
|
||||
# the signal is preserved rather than swallowed.
|
||||
category = "General"
|
||||
row.setdefault("_notes", []).append("category could not be detected; defaulted to General")
|
||||
row["category"] = category
|
||||
else:
|
||||
row["_category_deterministic"] = True
|
||||
|
||||
if category:
|
||||
fixed, changed, removed = validate_and_fix_title(
|
||||
title, category, brand=row.get("brand")
|
||||
)
|
||||
row["title"] = fixed
|
||||
if changed:
|
||||
row.setdefault("_notes", []).append(
|
||||
f"title corrected against category '{category}' (removed {removed})"
|
||||
)
|
||||
if not _blank(row.get("description")):
|
||||
row["description"] = sanitize_category_language(row["description"], category)
|
||||
else:
|
||||
row["title"] = title
|
||||
return row
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 4 - Pack-size explosion & unit safety
|
||||
# ---------------------------------------------------------------------------
|
||||
def _sizes_for(row: Dict[str, Any]) -> List[str]:
|
||||
sizes = [str(s).strip() for s in (row.get("size_variants") or []) if str(s).strip()]
|
||||
if sizes:
|
||||
return sizes
|
||||
match = _SIZE_IN_TITLE.search(row.get("product_name") or "")
|
||||
if match:
|
||||
return [match.group(0).strip()]
|
||||
return list(price_estimator.default_size_variants(
|
||||
row.get("category") or "", row.get("product_name") or ""
|
||||
))
|
||||
|
||||
|
||||
def stage_4_explode_sizes(row: Dict[str, Any]) -> Tuple[List[Dict[str, Any]], List[str]]:
|
||||
"""One input row becomes one row per pack size.
|
||||
|
||||
Barcodes and SKUs identify a specific pack, so they are only meaningful
|
||||
once sizes are separate rows. Sizes whose unit makes no sense for the
|
||||
category (a biscuit measured in cm) are dropped with a reason.
|
||||
"""
|
||||
out: List[Dict[str, Any]] = []
|
||||
dropped: List[str] = []
|
||||
category = row.get("category") or ""
|
||||
|
||||
for size in _sizes_for(row)[:MAX_VARIANTS_PER_PRODUCT]:
|
||||
fixed, rejected, reason = fix_or_reject_size(size, category, row.get("product_name") or "")
|
||||
if rejected or not fixed:
|
||||
dropped.append(f"{size}: {reason or 'invalid pack size'}")
|
||||
continue
|
||||
variant = dict(row)
|
||||
variant["size"] = fixed
|
||||
variant["size_variants"] = [fixed]
|
||||
out.append(variant)
|
||||
|
||||
if not out and not dropped:
|
||||
variant = dict(row)
|
||||
variant["size"] = "Standard"
|
||||
variant["size_variants"] = ["Standard"]
|
||||
out.append(variant)
|
||||
return out, dropped
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 5 - Pricing bands
|
||||
# ---------------------------------------------------------------------------
|
||||
def stage_5_pricing(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
size = row.get("size") or "Standard"
|
||||
if _blank(row.get("price_range")):
|
||||
if not _blank(row.get("final_selling_price")):
|
||||
price = float(row["final_selling_price"])
|
||||
lo, hi = int(round(price * 0.95)), int(round(price * 1.05))
|
||||
else:
|
||||
lo, hi = price_estimator.estimate_price_range_for_size(
|
||||
size, row.get("product_name") or "", row.get("brand") or "",
|
||||
row.get("category") or "",
|
||||
)
|
||||
row["price_range"] = f"₹{lo}-{hi}"
|
||||
return row
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 6 - Images
|
||||
# ---------------------------------------------------------------------------
|
||||
def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, Any]:
|
||||
if not _blank(row.get("image_urls")) or not enabled:
|
||||
return row
|
||||
try:
|
||||
from app.core.catalog_engine import catalog_engine
|
||||
from app.services.image_search import find_all_image_urls
|
||||
|
||||
candidates = find_all_image_urls(
|
||||
row.get("product_name") or "", brand=row.get("brand"), max_results=24
|
||||
)
|
||||
if candidates:
|
||||
best = catalog_engine._select_best_images(
|
||||
candidates, row.get("product_name") or "", row.get("brand") or "", max_images=10
|
||||
)
|
||||
if best:
|
||||
row["image_urls"] = list(best)
|
||||
row["image_url"] = best[0]
|
||||
except Exception as exc: # noqa: BLE001 - an image is not worth the row
|
||||
logger.debug("Image search skipped for %r: %s", row.get("product_name"), exc)
|
||||
return row
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 7 - SKU
|
||||
# ---------------------------------------------------------------------------
|
||||
def stage_7_sku(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
if not _blank(row.get("product_sku")):
|
||||
return row
|
||||
try:
|
||||
resolved = resolve_product_sku(
|
||||
row.get("brand") or "", row.get("product_name") or "", row.get("size") or ""
|
||||
)
|
||||
row["product_sku"] = resolved.get("product_sku")
|
||||
row["sku_source"] = resolved.get("sku_source")
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.debug("SKU resolution failed for %r: %s", row.get("product_name"), exc)
|
||||
return row
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stages 8 & 9 - barcode and HSN/GST, via the ported enrichment framework
|
||||
# ---------------------------------------------------------------------------
|
||||
async def stages_8_9_enrichment(rows: List[Dict[str, Any]], brand: str) -> List[Dict[str, Any]]:
|
||||
"""Both stages are `EnrichmentStage`s, so the ported EnrichmentPipeline
|
||||
runs them with its own concurrency limit and never-raises guarantee.
|
||||
Each disables itself via its settings flag, so this is a no-op when both
|
||||
are off.
|
||||
"""
|
||||
stages = []
|
||||
if ENABLE_BARCODE_LOOKUP:
|
||||
stages.append(BarcodeEnrichmentStage())
|
||||
if ENABLE_HSN_GST_ENRICHMENT:
|
||||
stages.append(HsnGstEnrichmentStage())
|
||||
if not stages:
|
||||
return rows
|
||||
return await EnrichmentPipeline(stages).run(rows, brand)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 10 - validation gate
|
||||
# ---------------------------------------------------------------------------
|
||||
def stage_10_validate(rows: List[Dict[str, Any]], brand: str, *, images_checked: bool = True):
|
||||
if not ENABLE_PRODUCT_VALIDATION:
|
||||
return rows, [], {}
|
||||
known = [bool(r.get("_category_deterministic")) for r in rows]
|
||||
return validate_catalog(rows, brand, known_category_flags=known,
|
||||
images_checked=images_checked)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 11 - embed & store
|
||||
# ---------------------------------------------------------------------------
|
||||
def build_image_id(brand: str, product_name: str, size: str) -> str:
|
||||
"""Deterministic, and unique per pack size.
|
||||
|
||||
Matches `user_products.py`'s brand_slug + product_slug form so the two
|
||||
upload paths agree, with the size folded in - without it every size of a
|
||||
product collapses onto one row under ON CONFLICT (image_id). Deliberately
|
||||
NOT s3_service.generate_image_id, which appends a random uuid4 and so can
|
||||
never match an existing row.
|
||||
"""
|
||||
name = (product_name or "").strip()
|
||||
# Store sheets usually carry the pack size inside the product name already
|
||||
# ("Cadbury Dairy Milk 100g"). Appending it again would produce
|
||||
# ..._100g_100g and, worse, a different id than the same product uploaded
|
||||
# through the user-products path.
|
||||
suffix = "" if _slugify(size) and _slugify(size) in _slugify(name) else f" {size}"
|
||||
slug = _slugify(f"{name}{suffix}".strip())
|
||||
return f"{_sanitize_name(brand)}_{slug}"
|
||||
|
||||
|
||||
def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Project an enriched row onto the brand-table columns."""
|
||||
name = row.get("product_name") or row.get("title") or ""
|
||||
size = row.get("size") or "Standard"
|
||||
display = name if size.lower() in name.lower() else f"{name} {size}".strip()
|
||||
category = row.get("category") or "General"
|
||||
description = row.get("description") or (
|
||||
f"{display} from {row.get('brand')}."
|
||||
)
|
||||
return {
|
||||
"product_name": display,
|
||||
"title": row.get("title") or name,
|
||||
"description": description,
|
||||
"category": category,
|
||||
"image_id": build_image_id(row.get("brand") or "", name, size),
|
||||
"image_url": row.get("image_url"),
|
||||
"image_urls": list(row.get("image_urls") or []),
|
||||
"price_range": row.get("price_range"),
|
||||
"size_variants": [size],
|
||||
"providers": list(row.get("providers") or []),
|
||||
"fssai_license": row.get("fssai_license"),
|
||||
"product_sku": row.get("product_sku"),
|
||||
"sku_source": row.get("sku_source"),
|
||||
"hsn_code": row.get("hsn_code"),
|
||||
"final_selling_price": row.get("final_selling_price"),
|
||||
"selling_price": row.get("selling_price"),
|
||||
"barcode": row.get("barcode"),
|
||||
"barcode_type": row.get("barcode_type"),
|
||||
"highlights": list(row.get("highlights") or []),
|
||||
"nutrients": list(row.get("nutrients") or []),
|
||||
"search_query": f"{row.get('brand')} {display} {category} {description}",
|
||||
}
|
||||
|
||||
|
||||
def _merge_with_existing(new: Dict[str, Any], existing: Dict[str, Any]) -> Tuple[Dict[str, Any], bool]:
|
||||
"""Keep the stored row, filling only the columns it left empty.
|
||||
|
||||
Returns (row_to_write, changed). `changed` is False when the stored row
|
||||
was already complete, which is what makes re-uploading the same file a
|
||||
no-op.
|
||||
"""
|
||||
merged = dict(existing)
|
||||
changed = False
|
||||
for key, value in new.items():
|
||||
if key in ("image_id",):
|
||||
continue
|
||||
if _blank(existing.get(key)) and not _blank(value):
|
||||
merged[key] = value
|
||||
changed = True
|
||||
merged["image_id"] = new["image_id"]
|
||||
return merged, changed
|
||||
|
||||
|
||||
def stage_11_store(rows: List[Dict[str, Any]], brand: str, result: PipelineResult) -> None:
|
||||
"""Embed and upsert, splitting inserts from backfills.
|
||||
|
||||
`cleanup=False` is load-bearing: cleanup=True deletes every row in the
|
||||
table that is not in this batch, which for a 3-row store file would wipe
|
||||
the brand's entire catalog.
|
||||
"""
|
||||
if not rows:
|
||||
return
|
||||
|
||||
try:
|
||||
existing_by_id = {
|
||||
r.get("image_id"): r
|
||||
for r in (get_products_by_brand(brand) or [])
|
||||
if r.get("image_id")
|
||||
}
|
||||
except Exception as exc: # noqa: BLE001 - treat as "nothing stored yet"
|
||||
logger.warning("Could not read existing products for %s: %s", brand, exc)
|
||||
existing_by_id = {}
|
||||
|
||||
to_write: List[Dict[str, Any]] = []
|
||||
for row in rows:
|
||||
prior = existing_by_id.get(row["image_id"])
|
||||
if prior is None:
|
||||
to_write.append(row)
|
||||
result.inserted += 1
|
||||
continue
|
||||
merged, changed = _merge_with_existing(row, prior)
|
||||
if changed:
|
||||
to_write.append(merged)
|
||||
result.backfilled += 1
|
||||
else:
|
||||
result.skipped_existing += 1
|
||||
|
||||
if not to_write:
|
||||
return
|
||||
|
||||
if USE_EMBEDDINGS:
|
||||
try:
|
||||
vectors = embed_texts([r.get("search_query") or r["product_name"] for r in to_write])
|
||||
for row, vector in zip(to_write, vectors):
|
||||
row["embedding"] = vector
|
||||
except Exception as exc: # noqa: BLE001 - a row without a vector is
|
||||
# still a usable catalog row; it just won't match semantic search.
|
||||
logger.warning("Embedding failed for %s: %s", brand, exc)
|
||||
|
||||
try:
|
||||
upsert_brand_products(brand, to_write, cleanup=False)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
result.storage_error = str(exc)
|
||||
result.inserted = 0
|
||||
result.backfilled = 0
|
||||
logger.exception("Storing store-catalog rows failed for %s", brand)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Orchestration
|
||||
# ---------------------------------------------------------------------------
|
||||
def parse_spreadsheet(filename: str, content: bytes):
|
||||
"""Parse an upload into (DataFrame, column mapping). Raises ValueError."""
|
||||
df = read_products_dataframe(filename, content)
|
||||
mapping = map_spreadsheet_columns(df.columns)
|
||||
return df, mapping
|
||||
|
||||
|
||||
def run_pipeline(
|
||||
filename: str,
|
||||
content: bytes,
|
||||
*,
|
||||
progress: ProgressFn = _noop_progress,
|
||||
use_llm: bool = True,
|
||||
fetch_images: bool = True,
|
||||
) -> PipelineResult:
|
||||
"""Run all 11 stages over an uploaded store spreadsheet.
|
||||
|
||||
Synchronous by design - it is called on a background daemon thread by the
|
||||
router, matching how every other long job in this project runs.
|
||||
"""
|
||||
result = PipelineResult()
|
||||
df, mapping = parse_spreadsheet(filename, content)
|
||||
result.recognised_columns = {f: str(c) for f, c in mapping.columns.items()}
|
||||
result.unrecognised_columns = list(mapping.unrecognised)
|
||||
records = df.to_dict(orient="records")
|
||||
result.rows_total = len(records)
|
||||
|
||||
# A store sheet frequently has no brand column; the brand lives inside the
|
||||
# product name. Rather than fork row_to_request (a working, tested function
|
||||
# this feature does not touch), advertise a synthetic brand column and fill
|
||||
# it per row from the inferred value.
|
||||
brand_column_supplied = "brand" in mapping.columns
|
||||
if not brand_column_supplied:
|
||||
mapping.columns["brand"] = _INFERRED_BRAND_COL
|
||||
|
||||
# ---- stages 1-3, per uploaded row --------------------------------------
|
||||
prepared: List[Dict[str, Any]] = []
|
||||
for position, record in enumerate(records):
|
||||
row_no = position + 2 # header occupies row 1
|
||||
|
||||
raw_name = _text(record, mapping, "product_name") or _text(record, mapping, "title") or ""
|
||||
if not brand_column_supplied:
|
||||
record[_INFERRED_BRAND_COL] = infer_brand(raw_name) or ""
|
||||
|
||||
try:
|
||||
req = row_to_request(record, mapping)
|
||||
except ValueError as exc:
|
||||
result.errors.append(RowError(row_no, raw_name, str(exc)))
|
||||
continue
|
||||
|
||||
row = req.model_dump()
|
||||
row["_row"] = row_no
|
||||
prepared.append(row)
|
||||
|
||||
progress(1, STAGE_NAMES[0], 0, len(prepared))
|
||||
prepared = [stage_1_brand_and_fssai(r) for r in prepared]
|
||||
|
||||
for index, row in enumerate(prepared, start=1):
|
||||
stage_2_row_intake(row, use_llm=use_llm)
|
||||
progress(2, STAGE_NAMES[1], index, len(prepared))
|
||||
|
||||
progress(3, STAGE_NAMES[2], 0, len(prepared))
|
||||
prepared = [stage_3_title_category(r) for r in prepared]
|
||||
|
||||
# ---- stage 4, one row becomes many -------------------------------------
|
||||
exploded: List[Dict[str, Any]] = []
|
||||
for row in prepared:
|
||||
variants, dropped = stage_4_explode_sizes(row)
|
||||
exploded.extend(variants)
|
||||
for reason in dropped:
|
||||
result.warnings.append(f"row {row.get('_row')}: dropped pack size {reason}")
|
||||
for note in row.get("_notes") or []:
|
||||
result.warnings.append(f"row {row.get('_row')}: {note}")
|
||||
progress(4, STAGE_NAMES[3], len(exploded), len(exploded))
|
||||
result.products_built = len(exploded)
|
||||
|
||||
# ---- stages 5-7, per exploded row --------------------------------------
|
||||
for index, row in enumerate(exploded, start=1):
|
||||
stage_5_pricing(row)
|
||||
progress(5, STAGE_NAMES[4], index, len(exploded))
|
||||
for index, row in enumerate(exploded, start=1):
|
||||
stage_6_images(row, enabled=fetch_images)
|
||||
progress(6, STAGE_NAMES[5], index, len(exploded))
|
||||
for index, row in enumerate(exploded, start=1):
|
||||
stage_7_sku(row)
|
||||
progress(7, STAGE_NAMES[6], index, len(exploded))
|
||||
|
||||
# ---- stages 8-11, grouped by destination brand -------------------------
|
||||
by_brand: Dict[str, List[Dict[str, Any]]] = {}
|
||||
for row in exploded:
|
||||
by_brand.setdefault(row["brand"], []).append(row)
|
||||
result.brands = sorted(display_name_for_suffix(_sanitize_name(b)) for b in by_brand)
|
||||
|
||||
for brand, rows in by_brand.items():
|
||||
progress(8, STAGE_NAMES[7], 0, len(rows))
|
||||
try:
|
||||
rows = asyncio.run(stages_8_9_enrichment(rows, brand))
|
||||
except Exception as exc: # noqa: BLE001 - enrichment must not lose rows
|
||||
logger.warning("Barcode/HSN enrichment failed for %s: %s", brand, exc)
|
||||
progress(9, STAGE_NAMES[8], len(rows), len(rows))
|
||||
|
||||
kept, rejected, _summary = stage_10_validate(rows, brand, images_checked=fetch_images)
|
||||
result.rejected += len(rejected)
|
||||
for bad in rejected:
|
||||
result.rejections.append({
|
||||
"product_name": bad.get("product_name") or bad.get("title"),
|
||||
"size": bad.get("size"),
|
||||
"reason": "; ".join(
|
||||
i.get("message", "") if isinstance(i, dict) else str(i)
|
||||
for i in (bad.get("validation_issues") or [])
|
||||
) or "failed the validation gate",
|
||||
})
|
||||
progress(10, STAGE_NAMES[9], len(kept), len(rows))
|
||||
|
||||
storage_rows = [_to_storage_row(r) for r in kept]
|
||||
# A single sheet can name the same pack twice; last one wins, so the
|
||||
# batch never presents two rows with the same image_id to the upsert.
|
||||
deduped: Dict[str, Dict[str, Any]] = {r["image_id"]: r for r in storage_rows}
|
||||
stage_11_store(list(deduped.values()), brand, result)
|
||||
progress(11, STAGE_NAMES[10], len(deduped), len(deduped))
|
||||
|
||||
return result
|
||||
@@ -185,6 +185,49 @@ USE_PLAYWRIGHT_FALLBACK = _bool("USE_PLAYWRIGHT_FALLBACK", "true")
|
||||
# out 1x1 tracking pixels / broken placeholder images)
|
||||
MIN_IMAGE_BYTES = int(os.getenv("MIN_IMAGE_BYTES", "3000"))
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Store-catalog enrichment pipeline (app/core/store_catalog_pipeline.py)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stages 7-10 of the store-Excel ingestion pipeline. Ported from the sibling
|
||||
# Universal_Catalog_Barcode_Enrichment project along with the code that reads
|
||||
# them; the names are kept identical so the ported modules need no edits.
|
||||
#
|
||||
# The three network-touching flags default to FALSE here, unlike in the
|
||||
# sibling. This backend serves an interactive API on a shared 8GB host, and a
|
||||
# 2000-row upload with web lookups on would fire thousands of outbound
|
||||
# requests. Turn them on deliberately, per environment.
|
||||
ENABLE_SKU_WEB_LOOKUP = _bool("ENABLE_SKU_WEB_LOOKUP", "false")
|
||||
ENABLE_BARCODE_LOOKUP = _bool("ENABLE_BARCODE_LOOKUP", "false")
|
||||
ENABLE_MANUFACTURER_SITE_LOOKUP = _bool("ENABLE_MANUFACTURER_SITE_LOOKUP", "false")
|
||||
|
||||
# Offline/deterministic stages - safe to leave on.
|
||||
ENABLE_HSN_GST_ENRICHMENT = _bool("ENABLE_HSN_GST_ENRICHMENT", "true")
|
||||
ENABLE_PRODUCT_VALIDATION = _bool("ENABLE_PRODUCT_VALIDATION", "true")
|
||||
|
||||
# Validation gate thresholds: below REJECT the row is dropped, below REVIEW it
|
||||
# is stored but flagged `validation_status="review"`.
|
||||
VALIDATION_REJECT_THRESHOLD = float(os.getenv("VALIDATION_REJECT_THRESHOLD", "0.35"))
|
||||
VALIDATION_REVIEW_THRESHOLD = float(os.getenv("VALIDATION_REVIEW_THRESHOLD", "0.70"))
|
||||
|
||||
# Pack-size explosion: how many size rows one uploaded product may become.
|
||||
MAX_VARIANTS_PER_PRODUCT = int(os.getenv("MAX_VARIANTS_PER_PRODUCT", "6"))
|
||||
ENABLE_PER_VARIANT_IMAGES = _bool("ENABLE_PER_VARIANT_IMAGES", "false")
|
||||
PER_VARIANT_IMAGE_MAX_RESULTS = int(os.getenv("PER_VARIANT_IMAGE_MAX_RESULTS", "10"))
|
||||
|
||||
# Barcode lookup tuning. The cache TTL is long (30 days) because a GTIN for a
|
||||
# given pack size does not change, and negative results are cached too.
|
||||
BARCODE_LOOKUP_TIMEOUT_SECONDS = float(os.getenv("BARCODE_LOOKUP_TIMEOUT_SECONDS", "10"))
|
||||
BARCODE_LOOKUP_CACHE_TTL_SECONDS = float(os.getenv("BARCODE_LOOKUP_CACHE_TTL_SECONDS", str(30 * 24 * 3600)))
|
||||
BARCODE_LOOKUP_MAX_CONCURRENCY = int(os.getenv("BARCODE_LOOKUP_MAX_CONCURRENCY", "5"))
|
||||
BARCODE_COUNTRY_TAG = os.getenv("BARCODE_COUNTRY_TAG", "india")
|
||||
|
||||
# Optional barcode source credentials. Each source disables itself when its
|
||||
# key is blank, so leaving these unset simply narrows the lookup cascade.
|
||||
GS1_INDIA_API_BASE_URL = os.getenv("GS1_INDIA_API_BASE_URL", "")
|
||||
GS1_INDIA_API_KEY = os.getenv("GS1_INDIA_API_KEY", "")
|
||||
UPC_DATABASE_API_KEY = os.getenv("UPC_DATABASE_API_KEY", "")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# HTTP client defaults
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -25,6 +25,7 @@ from app.api.routers import health, brands, search, chat, catalog, system
|
||||
from app.api.routers import stores, discounts, analytics as store_analytics, trending, recommendations, store_admin
|
||||
from app.api.routers import nutrition, nutrition_admin, upload
|
||||
from app.api.routers import auth, user_products, admin_train, mcp_info
|
||||
from app.api.routers import store_catalog
|
||||
from app.services.store_db import ensure_store_intelligence_schema
|
||||
from app.services.nutrition_db import ensure_nutrition_schema
|
||||
|
||||
@@ -203,6 +204,7 @@ app.include_router(store_admin.router, prefix="/api")
|
||||
app.include_router(nutrition.router, prefix="/api")
|
||||
app.include_router(nutrition_admin.router, prefix="/api")
|
||||
app.include_router(upload.router, prefix="/api")
|
||||
app.include_router(store_catalog.router, prefix="/api")
|
||||
app.include_router(mcp_info.router, prefix="/api")
|
||||
|
||||
# MCP lives outside /api on purpose: it is a protocol endpoint for AI clients,
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import re
|
||||
from functools import lru_cache
|
||||
from typing import Dict, Optional
|
||||
|
||||
BRAND_ALIASES = {
|
||||
# Cadbury family
|
||||
@@ -313,3 +314,59 @@ def get_known_sub_brands(brand: str) -> list[str]:
|
||||
known.append(rest)
|
||||
seen.add(rest)
|
||||
return known
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# FSSAI licences (stage 1 of the store-catalog pipeline)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Ported from the sibling Universal_Catalog_Barcode_Enrichment project. Purely
|
||||
# additive: BRAND_ALIASES and resolve_parent_brand are untouched.
|
||||
#
|
||||
# Food brands only. A miss means "not a food brand" (P&G, Colgate-Palmolive,
|
||||
# J&J, Reckitt, Godrej) just as much as it means "unknown", so callers must
|
||||
# treat None as "leave the column empty", never as an error.
|
||||
FSSAI_LICENSES: Dict[str, str] = {
|
||||
"britannia": "10012022000103",
|
||||
"pepsico": "10012031000047",
|
||||
"amul": "10012021000243",
|
||||
"cadbury": "10014022002711",
|
||||
"hindustan unilever": "10012022000217",
|
||||
"nestle": "10012011000168",
|
||||
"itc": "10018042000305",
|
||||
"coca-cola": "10012042000424",
|
||||
"tata": "12414003000511",
|
||||
"parle": "10012022000046",
|
||||
"marico": "10012022000258",
|
||||
"dabur": "10012011000084",
|
||||
"hatsun": "10012042000071",
|
||||
"milky mist": "10017042003191",
|
||||
"aachi": "10014042000577",
|
||||
"sakthi": "10012042000300",
|
||||
"kaleesuwari": "10012042000302",
|
||||
"idhayam": "10012042000109",
|
||||
"cavinkare": "10013042000366",
|
||||
"naga": "10012042000192",
|
||||
"manna": "10014042000169",
|
||||
"grb": "10012042000148",
|
||||
"anil": "10012042000213",
|
||||
"lion dates": "10012042000244",
|
||||
"brooke bond": "10013022001897",
|
||||
"mother dairy": "10012011000015",
|
||||
"haldiram": "10012011000140",
|
||||
"fortune": "10012021000071",
|
||||
"paper boat": "10012043000083",
|
||||
"bisk farm": "10012031000012",
|
||||
"mtr": "10012043000058",
|
||||
"everest": "10012022000526",
|
||||
"mdh": "10012011000062",
|
||||
}
|
||||
|
||||
|
||||
def get_fssai_license(brand: str) -> Optional[str]:
|
||||
"""Return the 14-digit FSSAI licence number for a food brand, or None.
|
||||
|
||||
Resolves to the canonical parent first, so sub-brands (e.g. 'hul lux')
|
||||
inherit the parent's licence and every row in a brand table agrees.
|
||||
"""
|
||||
canonical = resolve_parent_brand(brand).lower().strip()
|
||||
return FSSAI_LICENSES.get(canonical)
|
||||
|
||||
338
app/services/category_units.py
Normal file
338
app/services/category_units.py
Normal file
@@ -0,0 +1,338 @@
|
||||
"""
|
||||
Category -> Allowed Pack-Size-Unit Rules
|
||||
==========================================
|
||||
|
||||
WHY THIS FILE EXISTS
|
||||
---------------------
|
||||
Reported bug: brand="Britannia" produced catalog rows like
|
||||
"Tiger 10cm", "Nutrichoice 12cm", "Marie Gold 15cm" and
|
||||
"Britannia Milk Bikis - 200ml, 500ml, 1L". The small local LLM
|
||||
(qwen2.5:1.5b) occasionally predicts a *dimension* unit (cm/inch/m) or the
|
||||
*wrong physical-state* unit (ml/L for a solid biscuit) instead of a
|
||||
plausible FMCG pack-size unit, because nothing anywhere in the pipeline
|
||||
ever told it - or checked afterwards - which units are even legal for a
|
||||
given product category.
|
||||
|
||||
Two existing point-fixes come close but don't cover this:
|
||||
|
||||
- `price_estimator._parse_size_to_grams()` parses a size string into a
|
||||
gram/ml-equivalent float for pricing math. It has an explicit fallback
|
||||
for *unrecognised* unit tokens (case: "cm" is not in its litre/kg/base
|
||||
unit sets): "leave the numeric value as-is rather than guessing". That
|
||||
is the correct, conservative choice for a *pricing* function - but it
|
||||
means "10cm" silently becomes 10.0 (treated as if it were 10 grams),
|
||||
which is exactly wrong for a *validation* function: a bogus unit must
|
||||
never be treated as an implicitly-valid one.
|
||||
- `price_estimator.normalize_size_unit()` swaps ml<->g labels for
|
||||
categories it already knows are exclusively solid or exclusively liquid
|
||||
- but it only recognises `ml/l` <-> `g/kg` confusion. It has no concept
|
||||
of "cm" at all, so a dimension unit passes through completely
|
||||
unexamined.
|
||||
|
||||
This module is the missing, single source of truth for "what pack-size
|
||||
units are even legal for this product category", used in two places:
|
||||
|
||||
1. GENERATION TIME (app/services/ollama_service.py): to tell the LLM, as
|
||||
part of the prompt, exactly which units are allowed for the category it
|
||||
is currently enumerating products for - so the wrong unit is far less
|
||||
likely to be generated in the first place.
|
||||
2. VALIDATION TIME (app/core/catalog_engine.py, app/services/
|
||||
product_validator.py): to reject/replace a generated size string whose
|
||||
unit contradicts its resolved category, deterministically and without
|
||||
an LLM call, as a defence-in-depth backstop for whatever still gets
|
||||
through the prompt-level fix (e.g. the `fetch_brand_catalog_with_gemini`
|
||||
fallback path, or a cached/registry-sourced product).
|
||||
|
||||
Per the project's own stated requirement: package size must be determined
|
||||
from the PRODUCT CATEGORY, and dimension units (cm/inch/m) must never be
|
||||
generated for an FMCG consumable unless the category genuinely represents
|
||||
a physical dimension (none currently in this catalog do).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Optional
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Unit vocabularies
|
||||
# ---------------------------------------------------------------------------
|
||||
WEIGHT_UNITS: set[str] = {
|
||||
"g", "gm", "gms", "gram", "grams",
|
||||
"kg", "kgs", "kilo", "kilos", "kilogram", "kilograms",
|
||||
}
|
||||
VOLUME_UNITS: set[str] = {
|
||||
"ml", "mls", "millilitre", "millilitres", "milliliter", "milliliters",
|
||||
"l", "lt", "ltr", "ltrs", "litre", "litres", "liter", "liters",
|
||||
}
|
||||
# Count-based packs (stationery, tablets, diapers, agarbatti sticks, etc.) -
|
||||
# legitimate for a handful of non-food categories this catalog also covers.
|
||||
COUNT_UNITS: set[str] = {
|
||||
"pcs", "pc", "piece", "pieces", "unit", "units", "tablet", "tablets",
|
||||
"capsule", "capsules", "strip", "strips", "sheet", "sheets", "roll",
|
||||
"rolls", "stick", "sticks", "count", "ct", "pack", "packs",
|
||||
}
|
||||
# Physical-dimension units. These are NEVER a valid FMCG/personal-care pack
|
||||
# size - no biscuit, chocolate, milk, tea, coffee, or ice-cream product is
|
||||
# ever sold "by the centimetre". Kept as an explicit set (rather than "any
|
||||
# unit we don't recognise") so the reject reason can name the exact problem.
|
||||
DIMENSION_UNITS: set[str] = {
|
||||
"cm", "centimetre", "centimetres", "centimeter", "centimeters",
|
||||
"mm", "millimetre", "millimetres", "millimeter", "millimeters",
|
||||
"m", "metre", "metres", "meter", "meters",
|
||||
"inch", "inches", "in",
|
||||
"ft", "feet", "foot",
|
||||
}
|
||||
|
||||
_KNOWN_UNIT_TOKENS: set[str] = WEIGHT_UNITS | VOLUME_UNITS | COUNT_UNITS | DIMENSION_UNITS
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Category -> allowed unit TYPE ("weight" / "volume" / "weight_or_volume" /
|
||||
# "count" / "unknown"). Keyed by lowercase, human-readable category strings -
|
||||
# the same vocabulary produced by app.services.brand_registry.
|
||||
# SUB_BRAND_CATEGORY and used as `known_category`/`category_value` throughout
|
||||
# catalog_engine.py. This is deliberately the AUTHORITATIVE per-category
|
||||
# rulebook the user specified: Biscuits -> g/kg, Milk/Juices -> ml/L,
|
||||
# Chocolate -> g, Tea -> g, Coffee -> g, Ice Cream -> ml/L - extended to
|
||||
# every other category already present in this project so every catalog row
|
||||
# gets a consistent check, not just the categories in the bug report.
|
||||
# ---------------------------------------------------------------------------
|
||||
CATEGORY_UNIT_TYPE: dict[str, str] = {
|
||||
# --- Weight-only (solid FMCG) ---
|
||||
"biscuits & cookies": "weight",
|
||||
"crackers": "weight",
|
||||
"rusk": "weight",
|
||||
"cakes & muffins": "weight",
|
||||
"bakery & breads": "weight",
|
||||
"snacks": "weight",
|
||||
"chocolates": "weight",
|
||||
"candy & confectionery": "weight",
|
||||
"atta & staples": "weight",
|
||||
"spices & masalas": "weight",
|
||||
"pasta & noodles": "weight",
|
||||
"noodles & instant food": "weight",
|
||||
"breakfast cereal": "weight",
|
||||
"dry fruits & nuts": "weight",
|
||||
"health foods": "weight",
|
||||
"millets": "weight",
|
||||
"salt & staples": "weight",
|
||||
"tea & coffee": "weight",
|
||||
"tea": "weight",
|
||||
"coffee": "weight",
|
||||
"health drinks": "weight", # malted/powder drinks: Horlicks, Bournvita, Boost, Complan
|
||||
"bath soap": "weight",
|
||||
"stationery": "count",
|
||||
"household - agarbatti": "count",
|
||||
"feminine hygiene": "count",
|
||||
"baby care": "weight_or_volume",
|
||||
# --- Volume-only (liquid FMCG) ---
|
||||
"beverages": "volume",
|
||||
"juices": "volume",
|
||||
"food - soups & sauces": "volume",
|
||||
"household cleaning": "volume",
|
||||
"dishwash": "volume",
|
||||
"ice cream": "volume",
|
||||
"cooking oils": "volume",
|
||||
"wellness oils": "volume",
|
||||
"household - lamp oil": "volume",
|
||||
# --- Mixed weight-or-volume (category legitimately spans both solid and
|
||||
# liquid sub-products, e.g. "Dairy" covers milk (ml) AND
|
||||
# cheese/paneer/butter/curd (g)) ---
|
||||
"dairy": "weight_or_volume",
|
||||
"dairy - desserts": "weight_or_volume",
|
||||
"hair care": "weight_or_volume", # shampoo/conditioner (ml) vs hair oil/cream (g/ml)
|
||||
"skin & bath care": "weight_or_volume",
|
||||
"skin care": "weight_or_volume",
|
||||
"beauty care": "weight_or_volume",
|
||||
"fragrance & deodorants": "weight_or_volume",
|
||||
"men's grooming": "weight_or_volume",
|
||||
"detergents & fabric care": "weight_or_volume", # powder (kg) vs liquid (L)
|
||||
"oral care": "weight_or_volume", # toothpaste (g) vs mouthwash (ml)
|
||||
"personal care": "weight_or_volume",
|
||||
"personal care - mosquito repellent": "weight_or_volume",
|
||||
"mosquito repellent": "weight_or_volume",
|
||||
"air freshener": "weight_or_volume",
|
||||
"health care - cold & cough": "weight_or_volume",
|
||||
"health care - digestive": "weight_or_volume",
|
||||
"health care - ayurvedic": "weight_or_volume",
|
||||
"health care - antiseptic": "weight_or_volume",
|
||||
"health care - first aid": "count",
|
||||
"pickles & chutneys": "weight",
|
||||
"food - spreads": "weight",
|
||||
"food - mixes": "weight",
|
||||
"rice & pulses": "weight",
|
||||
"sweets": "weight",
|
||||
"ready to eat": "weight_or_volume",
|
||||
}
|
||||
|
||||
# No category in this FMCG/personal-care catalog is legitimately sold "by
|
||||
# the centimetre" - kept as an explicit, extensible override point (e.g. a
|
||||
# future "Home Textiles" or "Furnishings" category) rather than a hardcoded
|
||||
# blanket rule, per the requirement that dimension units are only allowed
|
||||
# "if the product category genuinely represents physical dimensions".
|
||||
DIMENSION_OK_CATEGORIES: set[str] = set()
|
||||
|
||||
|
||||
def get_unit_type(category: Optional[str]) -> str:
|
||||
"""Return the allowed unit TYPE for `category` ('weight', 'volume',
|
||||
'weight_or_volume', 'count', or 'unknown' if the category isn't in our
|
||||
rulebook - callers should treat 'unknown' permissively, not as a
|
||||
rejection, since it just means we have no opinion yet, not that
|
||||
anything is wrong)."""
|
||||
if not category:
|
||||
return "unknown"
|
||||
return CATEGORY_UNIT_TYPE.get(category.strip().lower(), "unknown")
|
||||
|
||||
|
||||
def get_allowed_units(category: Optional[str]) -> set[str]:
|
||||
"""Concrete set of unit tokens allowed for `category`."""
|
||||
unit_type = get_unit_type(category)
|
||||
if unit_type == "weight":
|
||||
return set(WEIGHT_UNITS)
|
||||
if unit_type == "volume":
|
||||
return set(VOLUME_UNITS)
|
||||
if unit_type == "weight_or_volume":
|
||||
return WEIGHT_UNITS | VOLUME_UNITS
|
||||
if unit_type == "count":
|
||||
return set(COUNT_UNITS)
|
||||
# unknown category: permissive - anything except a dimension unit
|
||||
return WEIGHT_UNITS | VOLUME_UNITS | COUNT_UNITS
|
||||
|
||||
|
||||
def allowed_units_hint(category: Optional[str]) -> str:
|
||||
"""Short, human-readable allowed-units phrase for `category`, suitable
|
||||
for embedding directly into an LLM prompt, e.g. 'grams (g) or
|
||||
kilograms (kg)'. Used by ollama_service.py to tell the model, per
|
||||
category, exactly which units it must use."""
|
||||
unit_type = get_unit_type(category)
|
||||
return {
|
||||
"weight": "grams (g) or kilograms (kg) ONLY",
|
||||
"volume": "millilitres (ml) or litres (L) ONLY",
|
||||
"weight_or_volume": "grams/kilograms (g/kg) for solid items or millilitres/litres (ml/L) for liquid items",
|
||||
"count": "a piece/unit count (e.g. '10 pcs', '1 pack')",
|
||||
"unknown": "grams (g), kilograms (kg), millilitres (ml), or litres (L) as appropriate",
|
||||
}[unit_type]
|
||||
|
||||
|
||||
def is_forbidden_dimension_unit(unit: Optional[str], category: Optional[str] = None) -> bool:
|
||||
"""True if `unit` is a physical-dimension unit (cm/inch/m/...) that is
|
||||
never valid for `category` (see DIMENSION_OK_CATEGORIES)."""
|
||||
if not unit:
|
||||
return False
|
||||
u = unit.strip().lower()
|
||||
if u not in DIMENSION_UNITS:
|
||||
return False
|
||||
if category and category.strip().lower() in DIMENSION_OK_CATEGORIES:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def parse_unit(size: Optional[str]) -> tuple[Optional[float], Optional[str]]:
|
||||
"""Extract (numeric_value, unit_token) from a size string like '200g',
|
||||
'1.5 L', '10cm'. Returns (None, None) if no leading number is found."""
|
||||
if not size:
|
||||
return None, None
|
||||
s = size.strip().lower()
|
||||
match = re.search(r"(\d+(?:\.\d+)?)\s*([a-z]*)", s)
|
||||
if not match or not match.group(1):
|
||||
return None, None
|
||||
value = float(match.group(1))
|
||||
unit = match.group(2).strip() or None
|
||||
return value, unit
|
||||
|
||||
|
||||
def validate_unit_for_category(size: Optional[str], category: Optional[str]) -> tuple[bool, Optional[str]]:
|
||||
"""Check whether `size`'s unit is legal for `category`.
|
||||
|
||||
Returns (is_valid, reason). `reason` is populated whenever is_valid is
|
||||
False, explaining exactly what was wrong (dimension unit vs.
|
||||
wrong-physical-state unit vs. unrecognised token) so callers can log or
|
||||
surface it in an audit trail.
|
||||
"""
|
||||
if not size or not size.strip():
|
||||
return False, "size is missing"
|
||||
value, unit = parse_unit(size)
|
||||
if value is None:
|
||||
# No parseable number at all (e.g. "Family Pack") - not this
|
||||
# function's concern; price_estimator's own fallback handles it.
|
||||
return True, None
|
||||
if not unit:
|
||||
# Bare number, no unit token (e.g. "200") - ambiguous but not a
|
||||
# unit-category contradiction per se; leave to size-plausibility
|
||||
# checks elsewhere.
|
||||
return True, None
|
||||
|
||||
if is_forbidden_dimension_unit(unit, category):
|
||||
return False, (
|
||||
f"unit '{unit}' is a physical dimension, not a valid FMCG pack-size "
|
||||
f"unit for category '{category}'"
|
||||
)
|
||||
|
||||
if unit not in _KNOWN_UNIT_TOKENS:
|
||||
# Unrecognised token (e.g. "pcs" variants we don't track, "x6",
|
||||
# brand-specific packaging words) - not a contradiction we can
|
||||
# prove, so don't reject.
|
||||
return True, None
|
||||
|
||||
allowed = get_allowed_units(category)
|
||||
if unit not in allowed:
|
||||
unit_type = get_unit_type(category)
|
||||
return False, (
|
||||
f"unit '{unit}' is not valid for category '{category}' "
|
||||
f"(expects {unit_type.replace('_', ' ')} units: {allowed_units_hint(category)})"
|
||||
)
|
||||
return True, None
|
||||
|
||||
|
||||
def fix_or_reject_size(
|
||||
size: Optional[str], category: Optional[str], product_title: str = ""
|
||||
) -> tuple[Optional[str], bool, Optional[str]]:
|
||||
"""Authoritative size-unit correction, called at both generation time
|
||||
and validation time.
|
||||
|
||||
Returns (corrected_size_or_None, was_changed, reason):
|
||||
- If `size`'s unit is already valid for `category`: (size, False, None).
|
||||
- If `size` uses the WRONG PHYSICAL STATE for `category` (e.g. '200ml'
|
||||
for a strictly-solid category) but the number itself is plausible: the
|
||||
unit label is swapped (numeric value preserved), same behaviour as
|
||||
price_estimator.normalize_size_unit(), since a mislabelled-but-
|
||||
plausible quantity is safely recoverable.
|
||||
- If `size` uses a DIMENSION unit (cm/inch/m) or the unit can't be
|
||||
meaningfully reinterpreted for the category: there is no safe
|
||||
conversion (10cm has no defensible gram-equivalent), so this returns
|
||||
(None, True, reason) - the caller MUST NOT keep the original value and
|
||||
should substitute a category-appropriate default size instead (see
|
||||
price_estimator.default_size_variants()), never silently reuse the
|
||||
rejected number under a new label.
|
||||
"""
|
||||
if not size or not size.strip():
|
||||
return size, False, None
|
||||
|
||||
value, unit = parse_unit(size)
|
||||
if value is None or not unit:
|
||||
return size, False, None
|
||||
|
||||
ok, reason = validate_unit_for_category(size, category)
|
||||
if ok:
|
||||
return size, False, None
|
||||
|
||||
if is_forbidden_dimension_unit(unit, category):
|
||||
# No defensible conversion exists - reject outright.
|
||||
return None, True, reason
|
||||
|
||||
# Wrong-physical-state-but-plausible-number case: swap the label,
|
||||
# preserving the numeric value, mirroring the existing
|
||||
# price_estimator.normalize_size_unit() behaviour for ml<->g.
|
||||
unit_type = get_unit_type(category)
|
||||
value_str = (
|
||||
str(int(value)) if float(value).is_integer() else str(value)
|
||||
)
|
||||
if unit_type == "weight" and unit in VOLUME_UNITS:
|
||||
corrected = f"{value_str}g" if unit in {"ml", "mls"} else f"{value_str}kg"
|
||||
return corrected, True, reason
|
||||
if unit_type == "volume" and unit in WEIGHT_UNITS:
|
||||
corrected = f"{value_str}ml" if unit in {"g", "gm", "gms", "gram", "grams"} else f"{value_str}L"
|
||||
return corrected, True, reason
|
||||
|
||||
# Anything else we can't confidently repair (e.g. a count-unit category
|
||||
# given a weight unit) - reject rather than guess.
|
||||
return None, True, reason
|
||||
50
app/services/enrichment/__init__.py
Normal file
50
app/services/enrichment/__init__.py
Normal file
@@ -0,0 +1,50 @@
|
||||
"""
|
||||
Product Enrichment Service
|
||||
===========================
|
||||
|
||||
WHY THIS PACKAGE EXISTS
|
||||
------------------------
|
||||
Everything upstream of this package (ollama_service.py, product_validator.py,
|
||||
sku_service.py, image_search.py, ...) either GENERATES a product row or
|
||||
JUDGES whether an already-generated row is plausible. None of it ever goes
|
||||
out and fetches a piece of ground-truth data from an authoritative external
|
||||
source and attaches it to the row. That is what this package is for.
|
||||
|
||||
The first concrete enricher is barcode lookup (see `barcode/`): resolving a
|
||||
real, checksum-valid EAN-13/UPC/GTIN for a catalog row from trusted external
|
||||
sources, never from the LLM. The package is deliberately structured so this
|
||||
is ONE stage among what will eventually be several independent ones (HSN
|
||||
code, GST rate, nutrition, allergens, manufacturer details, ...) - see
|
||||
`base.py` for the `EnrichmentStage` contract every future stage implements,
|
||||
and `pipeline.py` for the orchestrator that runs them in sequence.
|
||||
|
||||
DESIGN PRINCIPLES (apply to every enrichment stage, present and future)
|
||||
-------------------------------------------------------------------------
|
||||
- NEVER FABRICATE. If a stage cannot find authentic, verifiable data, it
|
||||
writes NULL/None + a "not_found" status - never a guessed or LLM-
|
||||
generated value. This mirrors the whole point of `product_validator.py`
|
||||
and `sku_service.py`'s "real ID or clearly-labelled internal fallback"
|
||||
design, taken one step further: for barcodes there IS no safe synthetic
|
||||
fallback (a fabricated barcode is actively harmful - it can collide with
|
||||
a real product), so the fallback is always NULL, never a generated value.
|
||||
- NEVER RAISE. A single product's enrichment failure (timeout, malformed
|
||||
response, source outage) must never abort the batch or the pipeline run.
|
||||
Every stage catches its own errors and degrades to "not found" for that
|
||||
one row.
|
||||
- ADDITIVE. This package has no dependents before this change and does not
|
||||
modify the behaviour of any existing module; it is only ever imported
|
||||
from new call sites (see `app/core/catalog_engine.py`'s Step 2.6 and
|
||||
`app/services/vector_store.py`'s additive barcode columns).
|
||||
- INDEPENDENT STAGES. Each stage owns its own external calls, matching
|
||||
rules, caching, and DB columns. A future HSN/GST/nutrition stage does not
|
||||
need to know barcode lookup exists, and vice versa - see `pipeline.py`.
|
||||
"""
|
||||
from app.services.enrichment.base import EnrichmentStage, StageOutcome
|
||||
from app.services.enrichment.pipeline import EnrichmentPipeline, run_default_pipeline
|
||||
|
||||
__all__ = [
|
||||
"EnrichmentStage",
|
||||
"StageOutcome",
|
||||
"EnrichmentPipeline",
|
||||
"run_default_pipeline",
|
||||
]
|
||||
47
app/services/enrichment/barcode/__init__.py
Normal file
47
app/services/enrichment/barcode/__init__.py
Normal file
@@ -0,0 +1,47 @@
|
||||
"""
|
||||
Barcode Retrieval & Product Enrichment module.
|
||||
|
||||
Public entry points:
|
||||
lookup_barcode(brand, product_title, size, category="") -> BarcodeResult
|
||||
Synchronous single-product cascading lookup. Safe to call from a
|
||||
script, notebook, or the Streamlit UI's "look up a barcode" action.
|
||||
|
||||
batch_lookup_barcodes(products, brand) -> list[BarcodeResult]
|
||||
Async batch lookup used by the pipeline stage (see `stage.py`) and
|
||||
available directly for a CLI/manual batch job.
|
||||
|
||||
See the package's module docstrings for the full architecture:
|
||||
models.py - BarcodeCandidate / BarcodeResult / enums
|
||||
validators.py - EAN-13/GTIN/UPC checksum + format validation
|
||||
matching.py - exact product matching rules
|
||||
cache.py - local SQLite lookup cache
|
||||
retry.py - tenacity-based retry/backoff
|
||||
sources/ - the 4-tier cascading source registry
|
||||
service.py - orchestrates cache -> sources -> validate -> match
|
||||
stage.py - adapts the service to the generic EnrichmentStage
|
||||
contract used by app/services/enrichment/pipeline.py
|
||||
"""
|
||||
from typing import Any, Dict, List
|
||||
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate, BarcodeResult, BarcodeType, LookupStatus
|
||||
from app.services.enrichment.barcode.service import BarcodeLookupService, get_default_service
|
||||
|
||||
|
||||
def lookup_barcode(brand: str, product_title: str, size: str, category: str = "") -> BarcodeResult:
|
||||
return get_default_service().lookup_one(brand, product_title, size, category)
|
||||
|
||||
|
||||
async def batch_lookup_barcodes(products: List[Dict[str, Any]], brand: str) -> List[BarcodeResult]:
|
||||
return await get_default_service().batch_lookup(products, brand)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"BarcodeCandidate",
|
||||
"BarcodeResult",
|
||||
"BarcodeType",
|
||||
"LookupStatus",
|
||||
"BarcodeLookupService",
|
||||
"get_default_service",
|
||||
"lookup_barcode",
|
||||
"batch_lookup_barcodes",
|
||||
]
|
||||
127
app/services/enrichment/barcode/cache.py
Normal file
127
app/services/enrichment/barcode/cache.py
Normal file
@@ -0,0 +1,127 @@
|
||||
"""
|
||||
Local lookup cache for barcode results.
|
||||
|
||||
WHY A SEPARATE SQLITE FILE (data/cache/barcode_lookup.db) INSTEAD OF
|
||||
POSTGRES
|
||||
---------------------------------------------------------------------
|
||||
This cache exists purely to avoid repeating the SAME external-API search
|
||||
(GS1/Open Food Facts/UPCItemDB/manufacturer site) twice for the same
|
||||
(brand, product, size) - see "Cache barcode lookups to avoid repeated API
|
||||
calls" / "Prevent duplicate barcode searches" (Performance Requirements).
|
||||
It is deliberately NOT the authoritative store (that is Postgres, via
|
||||
`vector_store.py`'s additive barcode columns) - it needs to work even when
|
||||
Postgres is unreachable/USE_PGVECTOR=false, and it needs to be cheap and
|
||||
local on an 8GB-RAM/CPU-only machine, so plain stdlib `sqlite3` (no new
|
||||
dependency, no server process) is the right tool here, following the same
|
||||
"file-backed, atomic-write" spirit as `sku_service.py`'s SKU sequence
|
||||
counter.
|
||||
|
||||
A row's cache key is a normalized (brand, product_title, size) triple, not
|
||||
the barcode itself (we're caching "what did we already look up", not "what
|
||||
maps to what").
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
import sqlite3
|
||||
import threading
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
from app.services.enrichment.barcode.models import BarcodeResult
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_DB_PATH = Path("data") / "cache" / "barcode_lookup.db"
|
||||
_lock = threading.Lock()
|
||||
_initialized = False
|
||||
|
||||
|
||||
def _normalize_key_part(text: Optional[str]) -> str:
|
||||
return re.sub(r"\s+", " ", (text or "").strip().lower())
|
||||
|
||||
|
||||
def cache_key(brand: str, product_title: str, size: str) -> str:
|
||||
return f"{_normalize_key_part(brand)}|{_normalize_key_part(product_title)}|{_normalize_key_part(size)}"
|
||||
|
||||
|
||||
def _connect() -> sqlite3.Connection:
|
||||
_DB_PATH.parent.mkdir(parents=True, exist_ok=True)
|
||||
conn = sqlite3.connect(str(_DB_PATH), timeout=10)
|
||||
conn.execute("PRAGMA journal_mode=WAL")
|
||||
return conn
|
||||
|
||||
|
||||
def _ensure_schema(conn: sqlite3.Connection) -> None:
|
||||
global _initialized
|
||||
if _initialized:
|
||||
return
|
||||
conn.execute(
|
||||
"""
|
||||
CREATE TABLE IF NOT EXISTS barcode_lookup_cache (
|
||||
cache_key TEXT PRIMARY KEY,
|
||||
brand TEXT,
|
||||
product_title TEXT,
|
||||
size TEXT,
|
||||
result_json TEXT NOT NULL,
|
||||
created_at REAL NOT NULL
|
||||
)
|
||||
"""
|
||||
)
|
||||
conn.commit()
|
||||
_initialized = True
|
||||
|
||||
|
||||
def get_cached(brand: str, product_title: str, size: str, ttl_seconds: float) -> Optional[BarcodeResult]:
|
||||
"""Returns a cached BarcodeResult if present and not older than
|
||||
`ttl_seconds`, else None. Never raises - a cache read failure is
|
||||
treated exactly like a cache miss."""
|
||||
key = cache_key(brand, product_title, size)
|
||||
try:
|
||||
with _lock, _connect() as conn:
|
||||
_ensure_schema(conn)
|
||||
row = conn.execute(
|
||||
"SELECT result_json, created_at FROM barcode_lookup_cache WHERE cache_key = ?",
|
||||
(key,),
|
||||
).fetchone()
|
||||
except Exception as e:
|
||||
logger.debug(f"Barcode cache read failed for '{key}': {e}")
|
||||
return None
|
||||
|
||||
if not row:
|
||||
return None
|
||||
result_json, created_at = row
|
||||
if ttl_seconds > 0 and (time.time() - created_at) > ttl_seconds:
|
||||
return None
|
||||
try:
|
||||
data = json.loads(result_json)
|
||||
return BarcodeResult(**data)
|
||||
except Exception as e:
|
||||
logger.debug(f"Barcode cache entry for '{key}' unreadable, treating as miss: {e}")
|
||||
return None
|
||||
|
||||
|
||||
def set_cached(brand: str, product_title: str, size: str, result: BarcodeResult) -> None:
|
||||
"""Best-effort write - a failure here never blocks the lookup itself,
|
||||
it just means this row won't benefit from caching next time."""
|
||||
key = cache_key(brand, product_title, size)
|
||||
try:
|
||||
payload = json.dumps(result.__dict__)
|
||||
with _lock, _connect() as conn:
|
||||
_ensure_schema(conn)
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO barcode_lookup_cache (cache_key, brand, product_title, size, result_json, created_at)
|
||||
VALUES (?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT(cache_key) DO UPDATE SET
|
||||
result_json = excluded.result_json,
|
||||
created_at = excluded.created_at
|
||||
""",
|
||||
(key, brand, product_title, size, payload, time.time()),
|
||||
)
|
||||
conn.commit()
|
||||
except Exception as e:
|
||||
logger.debug(f"Barcode cache write failed for '{key}': {e}")
|
||||
141
app/services/enrichment/barcode/matching.py
Normal file
141
app/services/enrichment/barcode/matching.py
Normal file
@@ -0,0 +1,141 @@
|
||||
"""
|
||||
Exact-product matching rules for barcode candidates.
|
||||
|
||||
A source returning SOME EAN-13 for a product with a similar name is not
|
||||
good enough - the "Product Matching Rules" requirement is explicit that
|
||||
brand, product name, variant, and size must all match, and that a
|
||||
different pack size (or a differently-branded variant like "Sugar-Free")
|
||||
must be REJECTED rather than accepted as "close enough". This module is
|
||||
deliberately conservative: a candidate only passes if every check agrees;
|
||||
`is_valid_ean13`-style tricks are not enough to earn a false positive here
|
||||
the way they might in a fuzzy search feature.
|
||||
|
||||
Deliberately stdlib-only (`difflib`, already in the standard library) -
|
||||
this project explicitly removed `fuzzywuzzy`/`python-Levenshtein` as dead
|
||||
weight (see requirements.txt), so no new fuzzy-matching dependency is
|
||||
introduced here either.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from difflib import SequenceMatcher
|
||||
from typing import Iterable, Optional
|
||||
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate
|
||||
# quantity_utils rather than image_search: this repo's image_search.py has no
|
||||
# quantity helpers, and it is a working file the store-catalog work leaves alone.
|
||||
from app.services.quantity_utils import quantities_match
|
||||
|
||||
# Words that mark a genuinely DIFFERENT retail variant from the plain/base
|
||||
# product - if the candidate's title contains one of these and the
|
||||
# target product's own title/product_name does NOT, the candidate is
|
||||
# rejected outright regardless of how well the rest of the name matches
|
||||
# (this is exactly the "Marie Gold 200g" vs "Marie Gold Sugar-Free"
|
||||
# example from the spec). Kept as a small, explicit, reviewable list
|
||||
# rather than a fuzzy heuristic - false negatives (missing a real match)
|
||||
# are far cheaper here than false positives (storing the wrong product's
|
||||
# barcode).
|
||||
VARIANT_DISTINGUISHING_TERMS = {
|
||||
"sugar free", "sugarfree", "sugar-free", "no sugar", "zero sugar",
|
||||
"diet", "family pack", "value pack", "jumbo pack", "combo pack",
|
||||
"combo", "gift pack", "gift box", "mini pack", "party pack",
|
||||
"refill pack", "refill", "pouch pack", "twin pack", "saver pack",
|
||||
"economy pack", "jar", "tin", "pet jar",
|
||||
}
|
||||
|
||||
_STOPWORDS = {"the", "and", "of", "with", "for", "a", "an", "new", "pack", "india"}
|
||||
|
||||
|
||||
def _normalize(text: Optional[str]) -> str:
|
||||
return re.sub(r"[^a-z0-9\s]", " ", (text or "").lower()).strip()
|
||||
|
||||
|
||||
def _tokens(text: Optional[str]) -> set:
|
||||
return {t for t in _normalize(text).split() if t and t not in _STOPWORDS}
|
||||
|
||||
|
||||
def brand_matches(candidate_brand: str, target_brand: str, brand_aliases: Optional[Iterable[str]] = None) -> bool:
|
||||
"""True if the candidate's reported brand plausibly refers to the same
|
||||
brand as the target. Accepts an exact/substring match on the target
|
||||
brand name itself, or a match against any known alias (e.g. "HUL" for
|
||||
"Hindustan Unilever") supplied by the caller via
|
||||
`brand_registry.get_brand_alias_set()`."""
|
||||
cand = _normalize(candidate_brand)
|
||||
target = _normalize(target_brand)
|
||||
if not cand or not target:
|
||||
return False
|
||||
if target in cand or cand in target:
|
||||
return True
|
||||
for alias in (brand_aliases or []):
|
||||
alias_n = _normalize(alias)
|
||||
if alias_n and (alias_n in cand or cand in alias_n):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def size_matches(candidate_size: str, target_size: str, tolerance: float = 0.03) -> bool:
|
||||
"""Tight tolerance (3%, vs. the 15% used for image matching elsewhere
|
||||
in this project) - a barcode belongs to exactly one pack size, so
|
||||
"close enough" is not an acceptable bar the way it is for reusing a
|
||||
product photo. Falls back to a plain normalized-string equality check
|
||||
when neither string parses as a numeric quantity (e.g. count-based
|
||||
sizes like "10 tablets"), rather than silently treating unparsable
|
||||
sizes as a match.
|
||||
"""
|
||||
if not candidate_size or not target_size:
|
||||
return False
|
||||
if quantities_match(candidate_size, target_size, tolerance=tolerance):
|
||||
return True
|
||||
return _normalize(candidate_size) == _normalize(target_size)
|
||||
|
||||
|
||||
def has_conflicting_variant_terms(candidate_title: str, target_title: str) -> bool:
|
||||
"""True if the candidate's title names a distinguishing variant
|
||||
(sugar-free, family pack, ...) that the target product does NOT -
|
||||
which means the candidate is a real but DIFFERENT product, not the one
|
||||
being looked up."""
|
||||
cand_n = _normalize(candidate_title)
|
||||
target_n = _normalize(target_title)
|
||||
for term in VARIANT_DISTINGUISHING_TERMS:
|
||||
if term in cand_n and term not in target_n:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def name_similarity(candidate_title: str, target_title: str) -> float:
|
||||
"""0.0-1.0 token-overlap-weighted similarity. Used only as a secondary
|
||||
signal / diagnostic (`match_confidence`) - never as the sole gate for
|
||||
acceptance; see `is_match()`."""
|
||||
cand_tokens, target_tokens = _tokens(candidate_title), _tokens(target_title)
|
||||
if not cand_tokens or not target_tokens:
|
||||
return 0.0
|
||||
overlap = len(cand_tokens & target_tokens) / len(target_tokens)
|
||||
seq_ratio = SequenceMatcher(None, _normalize(candidate_title), _normalize(target_title)).ratio()
|
||||
return round((overlap * 0.6) + (seq_ratio * 0.4), 3)
|
||||
|
||||
|
||||
def is_match(candidate: BarcodeCandidate, target_brand: str, target_title: str, target_size: str,
|
||||
brand_aliases: Optional[Iterable[str]] = None,
|
||||
min_name_similarity: float = 0.45) -> tuple[bool, float]:
|
||||
"""The combined gate a candidate must pass to be accepted:
|
||||
1. Brand matches (or overlaps a known alias).
|
||||
2. Pack size matches within a tight tolerance.
|
||||
3. No conflicting variant terms (family pack / sugar-free / ...).
|
||||
4. Product-name similarity clears a floor - catches the case where
|
||||
brand+size coincidentally match but it's a completely different
|
||||
product line from the same brand.
|
||||
Returns (matched, confidence) - confidence is diagnostic only, stored
|
||||
on the result for audit/QA but never used to override rule 1-3.
|
||||
"""
|
||||
if not brand_matches(candidate.candidate_brand, target_brand, brand_aliases):
|
||||
return False, 0.0
|
||||
if not size_matches(candidate.candidate_size, target_size):
|
||||
return False, 0.0
|
||||
if has_conflicting_variant_terms(candidate.candidate_title, target_title):
|
||||
return False, 0.0
|
||||
|
||||
similarity = name_similarity(candidate.candidate_title, target_title)
|
||||
if similarity < min_name_similarity:
|
||||
return False, similarity
|
||||
|
||||
return True, similarity
|
||||
82
app/services/enrichment/barcode/models.py
Normal file
82
app/services/enrichment/barcode/models.py
Normal file
@@ -0,0 +1,82 @@
|
||||
"""Data shapes shared across the barcode lookup module."""
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from enum import Enum
|
||||
from typing import Optional
|
||||
|
||||
|
||||
class BarcodeType(str, Enum):
|
||||
EAN13 = "EAN-13"
|
||||
UPC_A = "UPC-A"
|
||||
GTIN14 = "GTIN-14"
|
||||
GTIN8 = "GTIN-8"
|
||||
UNKNOWN = "UNKNOWN"
|
||||
|
||||
|
||||
class LookupStatus(str, Enum):
|
||||
"""Mirrors the `barcode_lookup_status` DB column."""
|
||||
VERIFIED = "verified" # found + checksum-valid + matched product
|
||||
NOT_FOUND = "not_found" # every source exhausted, nothing matched
|
||||
INVALID_CANDIDATE = "invalid_candidate" # a candidate was found but failed
|
||||
# checksum/format validation or product matching - rejected, not stored
|
||||
ERROR = "error" # a source/network error prevented a full search
|
||||
DISABLED = "disabled" # barcode lookup turned off via settings
|
||||
CACHED = "cached" # served from the local lookup cache
|
||||
|
||||
|
||||
@dataclass
|
||||
class BarcodeCandidate:
|
||||
"""A raw, not-yet-validated candidate returned by one source."""
|
||||
barcode: str
|
||||
source_name: str # e.g. "Open Food Facts"
|
||||
candidate_title: str = "" # the source's own product title, for matching
|
||||
candidate_brand: str = ""
|
||||
candidate_size: str = ""
|
||||
candidate_countries: str = "" # raw country tag/string from the source, if any
|
||||
|
||||
|
||||
@dataclass
|
||||
class BarcodeResult:
|
||||
"""Final, caller-facing result. Field names match the DB columns in
|
||||
`vector_store.py` and the JSON export shape requested for the
|
||||
catalog (`barcode`, `barcode_type`, `barcode_source`, `barcode_verified`,
|
||||
...) 1:1, so `service.py` -> product dict -> DB row -> JSON export is a
|
||||
straight field copy with no renaming at any layer.
|
||||
"""
|
||||
barcode: Optional[str] = None
|
||||
barcode_type: Optional[str] = None
|
||||
gtin: Optional[str] = None
|
||||
ean13: Optional[str] = None
|
||||
upc: Optional[str] = None
|
||||
barcode_source: Optional[str] = None
|
||||
barcode_verified: bool = False
|
||||
barcode_lookup_status: str = LookupStatus.NOT_FOUND.value
|
||||
barcode_last_updated: float = field(default_factory=time.time)
|
||||
match_confidence: float = 0.0
|
||||
|
||||
@classmethod
|
||||
def null_result(cls, status: LookupStatus = LookupStatus.NOT_FOUND) -> "BarcodeResult":
|
||||
"""The required NULL shape: barcode=NULL, barcode_verified=false,
|
||||
barcode_source=NULL - used for every path where no authentic,
|
||||
matching barcode could be confirmed. Never construct a
|
||||
BarcodeResult with a barcode value outside of `service.py`'s
|
||||
validated-and-matched path.
|
||||
"""
|
||||
return cls(barcode_lookup_status=status.value)
|
||||
|
||||
def as_product_fields(self) -> dict:
|
||||
"""Flat dict merged directly into the catalog row - see
|
||||
`stage.py`."""
|
||||
return {
|
||||
"barcode": self.barcode,
|
||||
"barcode_type": self.barcode_type,
|
||||
"gtin": self.gtin,
|
||||
"ean13": self.ean13,
|
||||
"upc": self.upc,
|
||||
"barcode_source": self.barcode_source,
|
||||
"barcode_verified": self.barcode_verified,
|
||||
"barcode_lookup_status": self.barcode_lookup_status,
|
||||
"barcode_last_updated": self.barcode_last_updated,
|
||||
}
|
||||
47
app/services/enrichment/barcode/retry.py
Normal file
47
app/services/enrichment/barcode/retry.py
Normal file
@@ -0,0 +1,47 @@
|
||||
"""
|
||||
Configurable retry-with-exponential-backoff for barcode source calls.
|
||||
|
||||
Built on `tenacity`, already a project dependency (see requirements.txt -
|
||||
used elsewhere for httpx-based retries). Only retries on genuinely
|
||||
transient failures (network/timeout errors); a malformed response or a
|
||||
"no results" outcome is not retried, since retrying those wastes the
|
||||
source's rate-limit budget for no benefit (relevant for UPCItemDB's
|
||||
trial-tier daily cap in particular).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
import requests
|
||||
from tenacity import (
|
||||
retry,
|
||||
retry_if_exception_type,
|
||||
stop_after_attempt,
|
||||
wait_exponential,
|
||||
before_sleep_log,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_RETRYABLE_EXCEPTIONS = (
|
||||
requests.exceptions.ConnectionError,
|
||||
requests.exceptions.Timeout,
|
||||
requests.exceptions.ChunkedEncodingError,
|
||||
)
|
||||
|
||||
|
||||
def with_retry(max_attempts: int = 3, min_wait: float = 1.0, max_wait: float = 8.0):
|
||||
"""Decorator factory: `max_attempts` total tries, exponential backoff
|
||||
between `min_wait` and `max_wait` seconds. Applied per-source (see
|
||||
`sources/*.py`), not globally, so one slow/unreliable source retrying
|
||||
doesn't compound delay across the whole cascade - a source that keeps
|
||||
failing simply falls through to the next tier faster than a shared
|
||||
global retry budget would allow.
|
||||
"""
|
||||
return retry(
|
||||
reraise=True,
|
||||
stop=stop_after_attempt(max_attempts),
|
||||
wait=wait_exponential(multiplier=min_wait, max=max_wait),
|
||||
retry=retry_if_exception_type(_RETRYABLE_EXCEPTIONS),
|
||||
before_sleep=before_sleep_log(logger, logging.DEBUG),
|
||||
)
|
||||
170
app/services/enrichment/barcode/service.py
Normal file
170
app/services/enrichment/barcode/service.py
Normal file
@@ -0,0 +1,170 @@
|
||||
"""
|
||||
BarcodeLookupService - the cascading lookup orchestrator.
|
||||
|
||||
This is the ONE place that decides "is this barcode good enough to store".
|
||||
Every candidate from every source, regardless of tier, passes through the
|
||||
exact same two gates before it can become a `BarcodeResult`:
|
||||
1. `validators.validate_barcode()` - checksum + format.
|
||||
2. `matching.is_match()` - brand + size + variant + name.
|
||||
A source being "trusted" (e.g. GS1 India) does not skip either gate -
|
||||
trust only affects ORDER (which tier is tried first), never whether
|
||||
validation is required.
|
||||
|
||||
Cascade behaviour (see module docstring in `sources/__init__.py` for the
|
||||
tier list): tiers are tried in order; the first tier that produces at
|
||||
least one validated, matching candidate wins and the search stops - "The
|
||||
system should stop searching as soon as a verified barcode is found."
|
||||
Every tier is independently wrapped so a source outage never prevents
|
||||
falling through to the next one, and if every tier is exhausted with
|
||||
nothing matching, the result is the required NULL shape
|
||||
(`BarcodeResult.null_result()`), never a guess.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
ENABLE_BARCODE_LOOKUP,
|
||||
BARCODE_LOOKUP_CACHE_TTL_SECONDS,
|
||||
BARCODE_LOOKUP_MAX_CONCURRENCY,
|
||||
)
|
||||
from app.services.enrichment.barcode import cache
|
||||
from app.services.enrichment.barcode.matching import is_match
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate, BarcodeResult, BarcodeType, LookupStatus
|
||||
from app.services.enrichment.barcode.sources import get_default_sources
|
||||
from app.services.enrichment.barcode.validators import classify_barcode_type, to_ean13, validate_barcode
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
try:
|
||||
from app.services.brand_registry import get_brand_alias_set
|
||||
except Exception: # pragma: no cover - keeps the service importable/testable
|
||||
# in isolation even if brand_registry ever fails to import.
|
||||
def get_brand_alias_set(_brand: str) -> set:
|
||||
return set()
|
||||
|
||||
|
||||
def _build_result(barcode: str, source_name: str, confidence: float) -> BarcodeResult:
|
||||
btype = classify_barcode_type(barcode)
|
||||
ean13 = to_ean13(barcode) if btype in (BarcodeType.UPC_A, BarcodeType.EAN13) else None
|
||||
return BarcodeResult(
|
||||
barcode=barcode,
|
||||
barcode_type=btype.value,
|
||||
gtin=barcode,
|
||||
ean13=ean13,
|
||||
upc=barcode if btype == BarcodeType.UPC_A else None,
|
||||
barcode_source=source_name,
|
||||
barcode_verified=True,
|
||||
barcode_lookup_status=LookupStatus.VERIFIED.value,
|
||||
match_confidence=confidence,
|
||||
)
|
||||
|
||||
|
||||
class BarcodeLookupService:
|
||||
def __init__(self, sources=None):
|
||||
self.sources = sources if sources is not None else get_default_sources()
|
||||
|
||||
def lookup_one(self, brand: str, product_title: str, size: str, category: str = "",
|
||||
use_cache: bool = True) -> BarcodeResult:
|
||||
"""Synchronous cascading lookup for a single (brand, product,
|
||||
size). Never raises - every failure path returns a NULL-shaped
|
||||
`BarcodeResult` with a descriptive `barcode_lookup_status`."""
|
||||
if not ENABLE_BARCODE_LOOKUP:
|
||||
return BarcodeResult.null_result(LookupStatus.DISABLED)
|
||||
|
||||
if not brand or not product_title or not size:
|
||||
logger.debug(f"Barcode lookup skipped - missing brand/title/size ('{brand}', '{product_title}', '{size}')")
|
||||
return BarcodeResult.null_result(LookupStatus.ERROR)
|
||||
|
||||
if use_cache:
|
||||
cached = cache.get_cached(brand, product_title, size, BARCODE_LOOKUP_CACHE_TTL_SECONDS)
|
||||
if cached is not None:
|
||||
result = cached
|
||||
result.barcode_lookup_status = (
|
||||
LookupStatus.CACHED.value if result.barcode_verified else result.barcode_lookup_status
|
||||
)
|
||||
return result
|
||||
|
||||
brand_aliases = self._safe_brand_aliases(brand)
|
||||
had_source_error = False
|
||||
|
||||
for source in self.sources:
|
||||
if not source.available:
|
||||
continue
|
||||
try:
|
||||
raw_candidates = source.search(brand, product_title, size, category)
|
||||
except Exception as e:
|
||||
# Sources are documented to never raise, but this is the
|
||||
# pipeline-wide safety net referenced in base.py.
|
||||
logger.warning(f"[{source.name}] barcode search raised unexpectedly: {e}")
|
||||
had_source_error = True
|
||||
continue
|
||||
|
||||
result = self._first_validated_match(raw_candidates, brand, product_title, size, brand_aliases)
|
||||
if result is not None:
|
||||
if use_cache:
|
||||
cache.set_cached(brand, product_title, size, result)
|
||||
return result
|
||||
|
||||
status = LookupStatus.ERROR if had_source_error else LookupStatus.NOT_FOUND
|
||||
result = BarcodeResult.null_result(status)
|
||||
if use_cache:
|
||||
# Cache negative results too (Performance Requirements: "Prevent
|
||||
# duplicate barcode searches") - a short TTL still applies, so a
|
||||
# transient "not found" doesn't permanently block a later re-check.
|
||||
cache.set_cached(brand, product_title, size, result)
|
||||
return result
|
||||
|
||||
@staticmethod
|
||||
def _safe_brand_aliases(brand: str) -> set:
|
||||
try:
|
||||
return get_brand_alias_set(brand)
|
||||
except Exception:
|
||||
return set()
|
||||
|
||||
@staticmethod
|
||||
def _first_validated_match(raw_candidates: List[BarcodeCandidate], brand: str, product_title: str,
|
||||
size: str, brand_aliases: set) -> Optional[BarcodeResult]:
|
||||
for candidate in raw_candidates:
|
||||
clean_barcode = validate_barcode(candidate.barcode)
|
||||
if not clean_barcode:
|
||||
continue # invalid checksum/length/format - never stored, not even flagged
|
||||
matched, confidence = is_match(candidate, brand, product_title, size, brand_aliases)
|
||||
if not matched:
|
||||
continue
|
||||
return _build_result(clean_barcode, candidate.source_name, confidence)
|
||||
return None
|
||||
|
||||
async def batch_lookup(self, products: List[Dict[str, Any]], brand: str) -> List[BarcodeResult]:
|
||||
"""Async batch entry point used by `stage.py`. Bounded concurrency
|
||||
(`BARCODE_LOOKUP_MAX_CONCURRENCY`) so a large brand catalog doesn't
|
||||
fire dozens of simultaneous requests at any one source, and results
|
||||
are returned in the SAME ORDER as `products` regardless of which
|
||||
finished first, so callers can zip() them back together safely."""
|
||||
semaphore = asyncio.Semaphore(max(1, BARCODE_LOOKUP_MAX_CONCURRENCY))
|
||||
|
||||
async def _one(product: Dict[str, Any]) -> BarcodeResult:
|
||||
async with semaphore:
|
||||
return await asyncio.to_thread(
|
||||
self.lookup_one,
|
||||
brand,
|
||||
product.get("title") or product.get("product_name") or "",
|
||||
product.get("size") or "",
|
||||
product.get("category") or "",
|
||||
)
|
||||
|
||||
return await asyncio.gather(*(_one(p) for p in products))
|
||||
|
||||
|
||||
# Module-level singleton - sources are stateless, so one shared instance is
|
||||
# fine for both the sync CLI/test path and the async pipeline stage.
|
||||
_default_service: Optional[BarcodeLookupService] = None
|
||||
|
||||
|
||||
def get_default_service() -> BarcodeLookupService:
|
||||
global _default_service
|
||||
if _default_service is None:
|
||||
_default_service = BarcodeLookupService()
|
||||
return _default_service
|
||||
35
app/services/enrichment/barcode/sources/__init__.py
Normal file
35
app/services/enrichment/barcode/sources/__init__.py
Normal file
@@ -0,0 +1,35 @@
|
||||
"""
|
||||
Tier-ordered source registry - this list IS the cascade order described in
|
||||
the spec: GS1 India -> Open Food Facts -> trusted barcode databases ->
|
||||
manufacturer website -> (exhausted -> NULL). `service.py` iterates this
|
||||
list in order and stops at the first source that produces a validated,
|
||||
matching candidate.
|
||||
"""
|
||||
from app.services.enrichment.barcode.sources.base import BarcodeSource
|
||||
from app.services.enrichment.barcode.sources.gs1_india import GS1IndiaSource
|
||||
from app.services.enrichment.barcode.sources.open_food_facts import OpenFoodFactsSource
|
||||
from app.services.enrichment.barcode.sources.upc_database import UPCDatabaseSource
|
||||
from app.services.enrichment.barcode.sources.manufacturer_site import ManufacturerSiteSource
|
||||
|
||||
|
||||
def get_default_sources() -> list[BarcodeSource]:
|
||||
"""Fresh instances each call - sources are stateless/cheap to
|
||||
construct, and this avoids any shared-mutable-state surprises across
|
||||
concurrent batch lookups."""
|
||||
sources = [
|
||||
GS1IndiaSource(),
|
||||
OpenFoodFactsSource(),
|
||||
UPCDatabaseSource(),
|
||||
ManufacturerSiteSource(),
|
||||
]
|
||||
return sorted(sources, key=lambda s: s.tier)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"BarcodeSource",
|
||||
"GS1IndiaSource",
|
||||
"OpenFoodFactsSource",
|
||||
"UPCDatabaseSource",
|
||||
"ManufacturerSiteSource",
|
||||
"get_default_sources",
|
||||
]
|
||||
46
app/services/enrichment/barcode/sources/base.py
Normal file
46
app/services/enrichment/barcode/sources/base.py
Normal file
@@ -0,0 +1,46 @@
|
||||
"""
|
||||
Pluggable barcode source contract, following the same adapter-registry
|
||||
pattern already used for marketplace SKU resolution
|
||||
(app/services/sku_service.py's `_MARKETPLACE_PATTERNS`) and the Product
|
||||
Resolver's adapter registry (see the Catalog_Project's `resolve_parent_brand`
|
||||
family) - every source is a self-contained class with one job: given a
|
||||
brand/title/size, return zero or more raw candidates. It does NOT decide
|
||||
whether a candidate is a match (that's `matching.py`) or whether its
|
||||
barcode is valid (that's `validators.py`) - keeping those concerns
|
||||
separate is what lets `service.py` apply the SAME validation/matching gate
|
||||
uniformly regardless of which tier produced the candidate.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from abc import ABC, abstractmethod
|
||||
from typing import List
|
||||
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate
|
||||
|
||||
|
||||
class BarcodeSource(ABC):
|
||||
#: Human-readable label stored in `barcode_source` on a successful
|
||||
#: match, e.g. "Open Food Facts", "GS1 India".
|
||||
name: str = "Unknown Source"
|
||||
|
||||
#: Lower = tried first in the cascade (see sources/__init__.py).
|
||||
tier: int = 99
|
||||
|
||||
@property
|
||||
def available(self) -> bool:
|
||||
"""False when the source is unusable for reasons unrelated to any
|
||||
one query (missing API key, disabled via settings, dependency not
|
||||
installed, ...). The cascade skips unavailable sources entirely
|
||||
rather than querying and failing every time."""
|
||||
return True
|
||||
|
||||
@abstractmethod
|
||||
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
|
||||
"""Best-effort search. MUST NOT raise - catch internally and
|
||||
return [] on any failure (network error, malformed response,
|
||||
nothing found). `service.py` treats an empty list and an
|
||||
exception identically (fall through to the next tier), so
|
||||
swallowing the error here vs. letting it propagate makes no
|
||||
behavioural difference except robustness - always swallow it.
|
||||
"""
|
||||
raise NotImplementedError
|
||||
91
app/services/enrichment/barcode/sources/gs1_india.py
Normal file
91
app/services/enrichment/barcode/sources/gs1_india.py
Normal file
@@ -0,0 +1,91 @@
|
||||
"""
|
||||
Tier 1: GS1 India / GEPIR (the official Indian GTIN registry).
|
||||
|
||||
HONESTY NOTE - READ BEFORE WIRING UP CREDENTIALS
|
||||
--------------------------------------------------
|
||||
GS1 India does not currently publish a free, public, no-registration JSON
|
||||
API for GTIN lookup (unlike Open Food Facts). Their real product-data
|
||||
registry is accessed either through GEPIR (https://gepir.gs1.org/ - a
|
||||
human web UI with no documented public API) or through GS1 India's paid
|
||||
"Verified by GS1" data-as-a-service product, which requires a commercial
|
||||
account and an API key.
|
||||
|
||||
Rather than scrape GEPIR's web UI (fragile, likely against its terms, and
|
||||
exactly the kind of "pretend this is a real integration" shortcut this
|
||||
project's whole hallucination-prevention philosophy exists to avoid - see
|
||||
product_validator.py's module docstring), this adapter is a REAL,
|
||||
WORKING integration point that:
|
||||
- does nothing (returns [], `available=False`) when no credentials are
|
||||
configured, so the cascade cleanly falls through to Open Food Facts
|
||||
(Tier 2) - exactly the "if not found -> next step" behaviour the
|
||||
spec asks for, just starting from Tier 2 in practice until GS1 India
|
||||
API access is provisioned;
|
||||
- immediately becomes live the moment `GS1_INDIA_API_BASE_URL` and
|
||||
`GS1_INDIA_API_KEY` are set (see .env.example), with no code changes
|
||||
needed elsewhere - `service.py` already treats every tier uniformly.
|
||||
|
||||
If/when real GS1 India API access is provisioned, fill in the request
|
||||
shape in `search()` below to match that API's actual contract (endpoint
|
||||
path, auth header, response schema) - the surrounding plumbing
|
||||
(validation, matching, caching, retry) does not need to change.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import List
|
||||
|
||||
import requests
|
||||
|
||||
from app.infrastructure.settings import GS1_INDIA_API_BASE_URL, GS1_INDIA_API_KEY, BARCODE_LOOKUP_TIMEOUT_SECONDS
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate
|
||||
from app.services.enrichment.barcode.retry import with_retry
|
||||
from app.services.enrichment.barcode.sources.base import BarcodeSource
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class GS1IndiaSource(BarcodeSource):
|
||||
name = "GS1 India"
|
||||
tier = 1
|
||||
|
||||
@property
|
||||
def available(self) -> bool:
|
||||
return bool(GS1_INDIA_API_BASE_URL and GS1_INDIA_API_KEY)
|
||||
|
||||
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
|
||||
if not self.available:
|
||||
return []
|
||||
|
||||
@with_retry(max_attempts=3)
|
||||
def _call():
|
||||
return requests.get(
|
||||
f"{GS1_INDIA_API_BASE_URL.rstrip('/')}/search",
|
||||
params={"brand": brand, "product": product_title, "size": size, "country": "India"},
|
||||
headers={"Authorization": f"Bearer {GS1_INDIA_API_KEY}"},
|
||||
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
|
||||
)
|
||||
|
||||
try:
|
||||
resp = _call()
|
||||
if resp.status_code != 200:
|
||||
logger.debug(f"GS1 India lookup non-200 ({resp.status_code}) for '{brand} {product_title} {size}'")
|
||||
return []
|
||||
data = resp.json()
|
||||
except Exception as e:
|
||||
logger.debug(f"GS1 India lookup failed for '{brand} {product_title} {size}': {e}")
|
||||
return []
|
||||
|
||||
candidates: List[BarcodeCandidate] = []
|
||||
for item in (data.get("results") or data.get("items") or []):
|
||||
gtin = item.get("gtin") or item.get("barcode") or item.get("code")
|
||||
if not gtin:
|
||||
continue
|
||||
candidates.append(BarcodeCandidate(
|
||||
barcode=str(gtin),
|
||||
source_name=self.name,
|
||||
candidate_title=item.get("productName") or item.get("title") or "",
|
||||
candidate_brand=item.get("brandName") or item.get("brand") or "",
|
||||
candidate_size=item.get("netContent") or item.get("size") or "",
|
||||
candidate_countries="in",
|
||||
))
|
||||
return candidates
|
||||
108
app/services/enrichment/barcode/sources/manufacturer_site.py
Normal file
108
app/services/enrichment/barcode/sources/manufacturer_site.py
Normal file
@@ -0,0 +1,108 @@
|
||||
"""
|
||||
Tier 4: manufacturer / official product page - last resort.
|
||||
|
||||
Follows the exact same "DuckDuckGo search -> read structure straight off
|
||||
the result, no page-scraping HTML parser" spirit as
|
||||
`app/services/sku_service.py`'s `find_website_product_id()`, except here
|
||||
we DO need to fetch the page text (a barcode isn't embedded in the result
|
||||
URL the way an Amazon ASIN is), so this is the one tier that performs a
|
||||
capped number of lightweight page fetches, each wrapped in its own
|
||||
try/except so one slow/broken page can never block the others.
|
||||
|
||||
This tier is intentionally the lowest-trust one: a number that merely sits
|
||||
near the word "barcode"/"EAN"/"UPC"/"GTIN" on a webpage is only a
|
||||
CANDIDATE. It still has to pass `validators.validate_barcode()` (checksum)
|
||||
and `matching.is_match()` (brand/size/name) in `service.py` exactly like
|
||||
every other tier's candidates - nothing here is trusted just because it
|
||||
came from what looks like an official page.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from typing import List
|
||||
|
||||
import requests
|
||||
|
||||
from app.infrastructure.settings import BARCODE_LOOKUP_TIMEOUT_SECONDS, ENABLE_MANUFACTURER_SITE_LOOKUP
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate
|
||||
from app.services.enrichment.barcode.retry import with_retry
|
||||
from app.services.enrichment.barcode.sources.base import BarcodeSource
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_BROWSER_UA = (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
)
|
||||
|
||||
# A barcode label followed, within a short distance, by an 8/12/13/14-digit
|
||||
# run. Intentionally permissive on the label (source pages phrase this
|
||||
# differently) and tight on the digit run (only plausible GTIN lengths) -
|
||||
# every hit is still just a CANDIDATE, checksum-validated afterwards.
|
||||
_BARCODE_NEAR_LABEL_RE = re.compile(
|
||||
r"(?:barcode|ean|upc|gtin)\D{0,15}(\d{8}|\d{12,14})",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_MAX_PAGES = 3
|
||||
|
||||
|
||||
class ManufacturerSiteSource(BarcodeSource):
|
||||
name = "Manufacturer Website"
|
||||
tier = 4
|
||||
|
||||
@property
|
||||
def available(self) -> bool:
|
||||
if not ENABLE_MANUFACTURER_SITE_LOOKUP:
|
||||
return False
|
||||
try:
|
||||
import ddgs # noqa: F401
|
||||
return True
|
||||
except ImportError:
|
||||
logger.debug("Manufacturer-site barcode lookup unavailable: 'ddgs' package not installed")
|
||||
return False
|
||||
|
||||
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
|
||||
if not self.available:
|
||||
return []
|
||||
|
||||
query = f"{brand} {product_title} {size} barcode EAN".strip()
|
||||
try:
|
||||
from ddgs import DDGS
|
||||
with DDGS(timeout=10) as ddgs:
|
||||
results = ddgs.text(query, region="in-en", safesearch="off", max_results=_MAX_PAGES)
|
||||
except Exception as e:
|
||||
logger.debug(f"Manufacturer-site search failed for '{query}': {e}")
|
||||
return []
|
||||
|
||||
candidates: List[BarcodeCandidate] = []
|
||||
for r in (results or [])[:_MAX_PAGES]:
|
||||
url = str(r.get("href") or r.get("url") or "")
|
||||
if not url.startswith("http"):
|
||||
continue
|
||||
for code in self._extract_barcodes(url):
|
||||
candidates.append(BarcodeCandidate(
|
||||
barcode=code,
|
||||
source_name=f"{self.name} ({url})",
|
||||
candidate_title=r.get("title") or product_title,
|
||||
candidate_brand=brand,
|
||||
candidate_size=size,
|
||||
candidate_countries="",
|
||||
))
|
||||
return candidates
|
||||
|
||||
def _extract_barcodes(self, url: str) -> List[str]:
|
||||
@with_retry(max_attempts=2)
|
||||
def _call():
|
||||
return requests.get(url, headers={"User-Agent": _BROWSER_UA}, timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS)
|
||||
|
||||
try:
|
||||
resp = _call()
|
||||
if resp.status_code != 200:
|
||||
return []
|
||||
text = resp.text[:200_000] # cap - this is a text scan, not a full-page render
|
||||
except Exception as e:
|
||||
logger.debug(f"Manufacturer page fetch failed for '{url}': {e}")
|
||||
return []
|
||||
|
||||
return [m.group(1) for m in _BARCODE_NEAR_LABEL_RE.finditer(text)]
|
||||
118
app/services/enrichment/barcode/sources/open_food_facts.py
Normal file
118
app/services/enrichment/barcode/sources/open_food_facts.py
Normal file
@@ -0,0 +1,118 @@
|
||||
"""
|
||||
Tier 2: Open Food Facts / Open Beauty Facts / Open Products Facts.
|
||||
|
||||
Real, free, no-API-key public search API - the same family of open,
|
||||
community-maintained product databases already used as the project's
|
||||
PRIMARY image source (see `app/services/image_search.py`'s
|
||||
`OPEN_FACTS_HOSTS` / `_query_openfacts`). Every entry is keyed by its real
|
||||
barcode (the `code` field IS the GTIN/EAN/UPC printed on the physical
|
||||
pack), which is exactly the ground-truth this module needs - unlike
|
||||
`image_search.py`, which only reads the image URLs off each entry, this
|
||||
adapter reads the `code` field itself.
|
||||
|
||||
Deliberately a separate, self-contained query function rather than
|
||||
importing `image_search._query_openfacts` (which is a private, underscore-
|
||||
prefixed helper): this adapter needs different response fields (`code`,
|
||||
`countries_tags`) and India-specific filtering that the image-search
|
||||
helper has no reason to carry. `quantities_match`/`parse_quantity_grams`
|
||||
ARE imported from `image_search.py` (public functions) for size
|
||||
comparison, so the size-matching logic itself is not duplicated - see
|
||||
`matching.py`.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import List
|
||||
|
||||
import requests
|
||||
|
||||
from app.infrastructure.settings import BARCODE_LOOKUP_TIMEOUT_SECONDS, BARCODE_COUNTRY_TAG
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate
|
||||
from app.services.enrichment.barcode.retry import with_retry
|
||||
from app.services.enrichment.barcode.sources.base import BarcodeSource
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_HOSTS = [
|
||||
"world.openfoodfacts.org",
|
||||
"world.openbeautyfacts.org",
|
||||
"world.openproductsfacts.org",
|
||||
]
|
||||
_BROWSER_UA = (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
)
|
||||
_FIELDS = "code,product_name,brands,brands_tags,quantity,countries_tags"
|
||||
|
||||
|
||||
class OpenFoodFactsSource(BarcodeSource):
|
||||
name = "Open Food Facts"
|
||||
tier = 2
|
||||
|
||||
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
|
||||
query = f"{brand or ''} {product_title or ''}".strip()
|
||||
if not query:
|
||||
return []
|
||||
|
||||
products = self._query(query)
|
||||
if not products and brand and product_title:
|
||||
# Combined query too specific (common for regional brand-name
|
||||
# variants) - retry title-only, same fallback image_search.py uses.
|
||||
products = self._query(product_title)
|
||||
|
||||
candidates: List[BarcodeCandidate] = []
|
||||
for item in products:
|
||||
code = str(item.get("code") or "").strip()
|
||||
if not code:
|
||||
continue
|
||||
countries = ",".join(item.get("countries_tags") or [])
|
||||
candidates.append(BarcodeCandidate(
|
||||
barcode=code,
|
||||
source_name=self.name,
|
||||
candidate_title=item.get("product_name") or "",
|
||||
candidate_brand=item.get("brands") or "",
|
||||
candidate_size=item.get("quantity") or "",
|
||||
candidate_countries=countries,
|
||||
))
|
||||
|
||||
# Soft India-market prioritisation: candidates whose countries_tags
|
||||
# mention the target market are tried first, but non-tagged/other-
|
||||
# market candidates are kept (not dropped) since many genuine
|
||||
# Indian FMCG entries simply have this field blank upstream - the
|
||||
# brand/size/name gate in matching.py is what actually decides
|
||||
# correctness, this only affects which validated match is found
|
||||
# (and therefore stops the cascade) first.
|
||||
if BARCODE_COUNTRY_TAG:
|
||||
tag = BARCODE_COUNTRY_TAG.lower()
|
||||
candidates.sort(key=lambda c: 0 if tag in c.candidate_countries.lower() else 1)
|
||||
return candidates
|
||||
|
||||
def _query(self, query: str) -> list:
|
||||
for host in _HOSTS:
|
||||
@with_retry(max_attempts=2)
|
||||
def _call(host=host):
|
||||
return requests.get(
|
||||
f"https://{host}/cgi/search.pl",
|
||||
params={
|
||||
"search_terms": query,
|
||||
"search_simple": 1,
|
||||
"action": "process",
|
||||
"json": 1,
|
||||
"page_size": 20,
|
||||
"fields": _FIELDS,
|
||||
},
|
||||
headers={"User-Agent": _BROWSER_UA},
|
||||
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
|
||||
)
|
||||
|
||||
try:
|
||||
resp = _call()
|
||||
if resp.status_code != 200:
|
||||
continue
|
||||
products = resp.json().get("products", [])
|
||||
if products:
|
||||
return products
|
||||
except Exception as e:
|
||||
logger.debug(f"Open*Facts barcode lookup failed on {host} for '{query}': {e}")
|
||||
continue
|
||||
return []
|
||||
81
app/services/enrichment/barcode/sources/upc_database.py
Normal file
81
app/services/enrichment/barcode/sources/upc_database.py
Normal file
@@ -0,0 +1,81 @@
|
||||
"""
|
||||
Tier 3: "trusted barcode databases" - UPCItemDB.
|
||||
|
||||
Real, free (trial-tier, no API key required, rate-limited) reverse lookup:
|
||||
search by product name/brand keywords and get back candidate items with
|
||||
their own `upc`/`ean` fields. This is the general "trusted barcode
|
||||
database" tier the spec asks for as a fallback below GS1 India / Open
|
||||
Food Facts.
|
||||
|
||||
RATE LIMIT: UPCItemDB's free trial endpoint is capped (documented as
|
||||
~100 requests/day, ~1 request/second) - this is exactly why this tier
|
||||
sits BELOW Open Food Facts (unlimited, no key) in the cascade, and why
|
||||
`service.py` only calls a lower tier at all when every higher tier has
|
||||
already failed to produce a validated match, plus why the local cache
|
||||
(`cache.py`) matters most for this specific source. If a paid UPCItemDB
|
||||
key is available, set `UPC_DATABASE_API_KEY` (see .env.example) to switch
|
||||
to the production endpoint with a higher quota - the request shape below
|
||||
already supports both.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import List
|
||||
|
||||
import requests
|
||||
|
||||
from app.infrastructure.settings import BARCODE_LOOKUP_TIMEOUT_SECONDS, UPC_DATABASE_API_KEY
|
||||
from app.services.enrichment.barcode.models import BarcodeCandidate
|
||||
from app.services.enrichment.barcode.retry import with_retry
|
||||
from app.services.enrichment.barcode.sources.base import BarcodeSource
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_TRIAL_URL = "https://api.upcitemdb.com/prod/trial/search"
|
||||
_PROD_URL = "https://api.upcitemdb.com/prod/v1/search"
|
||||
|
||||
|
||||
class UPCDatabaseSource(BarcodeSource):
|
||||
name = "UPCItemDB"
|
||||
tier = 3
|
||||
|
||||
def search(self, brand: str, product_title: str, size: str, category: str = "") -> List[BarcodeCandidate]:
|
||||
query = f"{brand or ''} {product_title or ''} {size or ''}".strip()
|
||||
if not query:
|
||||
return []
|
||||
|
||||
url = _PROD_URL if UPC_DATABASE_API_KEY else _TRIAL_URL
|
||||
headers = {"Accept": "application/json"}
|
||||
if UPC_DATABASE_API_KEY:
|
||||
headers["user_key"] = UPC_DATABASE_API_KEY
|
||||
headers["key_type"] = "3scale"
|
||||
|
||||
@with_retry(max_attempts=2)
|
||||
def _call():
|
||||
return requests.get(url, params={"s": query, "type": "product"}, headers=headers,
|
||||
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS)
|
||||
|
||||
try:
|
||||
resp = _call()
|
||||
if resp.status_code != 200:
|
||||
logger.debug(f"UPCItemDB non-200 ({resp.status_code}) for '{query}'")
|
||||
return []
|
||||
data = resp.json()
|
||||
except Exception as e:
|
||||
logger.debug(f"UPCItemDB lookup failed for '{query}': {e}")
|
||||
return []
|
||||
|
||||
candidates: List[BarcodeCandidate] = []
|
||||
for item in data.get("items", []):
|
||||
code = item.get("ean") or item.get("upc")
|
||||
if not code:
|
||||
continue
|
||||
candidates.append(BarcodeCandidate(
|
||||
barcode=str(code),
|
||||
source_name=self.name,
|
||||
candidate_title=item.get("title") or "",
|
||||
candidate_brand=item.get("brand") or "",
|
||||
candidate_size=item.get("size") or "",
|
||||
candidate_countries="",
|
||||
))
|
||||
return candidates
|
||||
37
app/services/enrichment/barcode/stage.py
Normal file
37
app/services/enrichment/barcode/stage.py
Normal file
@@ -0,0 +1,37 @@
|
||||
"""Adapts BarcodeLookupService to the EnrichmentStage contract so it can be
|
||||
registered in app/services/enrichment/pipeline.py's default pipeline."""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Any, Dict
|
||||
|
||||
from app.infrastructure.settings import ENABLE_BARCODE_LOOKUP
|
||||
from app.services.enrichment.base import EnrichmentStage, StageOutcome
|
||||
from app.services.enrichment.barcode.service import get_default_service
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class BarcodeEnrichmentStage(EnrichmentStage):
|
||||
name = "barcode_lookup"
|
||||
|
||||
@property
|
||||
def enabled(self) -> bool:
|
||||
return ENABLE_BARCODE_LOOKUP
|
||||
|
||||
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
|
||||
service = get_default_service()
|
||||
title = product.get("title") or product.get("product_name") or ""
|
||||
size = product.get("size") or ""
|
||||
category = product.get("category") or ""
|
||||
|
||||
try:
|
||||
import asyncio
|
||||
result = await asyncio.to_thread(service.lookup_one, brand, title, size, category)
|
||||
except Exception as e:
|
||||
logger.warning(f"Barcode enrichment failed for '{title}' {size}: {e}")
|
||||
from app.services.enrichment.barcode.models import BarcodeResult, LookupStatus
|
||||
result = BarcodeResult.null_result(LookupStatus.ERROR)
|
||||
|
||||
return StageOutcome(stage_name=self.name, fields=result.as_product_fields(),
|
||||
error=None if result.barcode_verified else result.barcode_lookup_status)
|
||||
99
app/services/enrichment/barcode/validators.py
Normal file
99
app/services/enrichment/barcode/validators.py
Normal file
@@ -0,0 +1,99 @@
|
||||
"""
|
||||
Barcode format + checksum validation.
|
||||
|
||||
Deliberately the LAST gate a candidate passes through before it's allowed
|
||||
into a `BarcodeResult` (see `service.py`), regardless of which source
|
||||
produced it or how much we otherwise trust that source - a source telling
|
||||
us "this is the barcode" is never sufficient on its own; the digits have to
|
||||
actually check out mathematically. This is what "Validate EAN-13 checksum"
|
||||
/ "Validate GTIN format" / "Reject invalid barcode lengths" / "Reject
|
||||
malformed barcode values" (Barcode Validation requirements) means in code.
|
||||
|
||||
No third-party dependency - GTIN/EAN/UPC-A all share one checksum
|
||||
algorithm (the classic "alternating 3/1 weights counted from the rightmost
|
||||
digit before the check digit"), so GTIN-8/12/13/14 are all validated by the
|
||||
same function; only the accepted LENGTH differs per barcode type.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Optional
|
||||
|
||||
from app.services.enrichment.barcode.models import BarcodeType
|
||||
|
||||
_VALID_LENGTHS = {8: BarcodeType.GTIN8, 12: BarcodeType.UPC_A, 13: BarcodeType.EAN13, 14: BarcodeType.GTIN14}
|
||||
|
||||
# Digits only, no separators - callers are expected to call normalize_barcode()
|
||||
# first if the raw source string may contain spaces/hyphens.
|
||||
_DIGITS_ONLY_RE = re.compile(r"^\d+$")
|
||||
|
||||
|
||||
def normalize_barcode(raw: Optional[str]) -> Optional[str]:
|
||||
"""Strip everything but digits (spaces, hyphens, a stray 'EAN:' label,
|
||||
etc.). Returns None for empty/unusable input - never raises."""
|
||||
if not raw:
|
||||
return None
|
||||
digits = re.sub(r"\D", "", str(raw))
|
||||
return digits or None
|
||||
|
||||
|
||||
def gtin_check_digit(digits_without_check: str) -> int:
|
||||
"""Compute the correct check digit for a GTIN-8/12/13/14 payload
|
||||
(i.e. every digit EXCEPT the check digit itself), using the standard
|
||||
alternating 3/1 weighting counted from the rightmost digit."""
|
||||
total = 0
|
||||
for i, ch in enumerate(reversed(digits_without_check)):
|
||||
weight = 3 if i % 2 == 0 else 1
|
||||
total += int(ch) * weight
|
||||
return (10 - (total % 10)) % 10
|
||||
|
||||
|
||||
def has_valid_checksum(code: str) -> bool:
|
||||
"""True if `code`'s own last digit matches the checksum computed over
|
||||
the rest of it. `code` must already be digits-only."""
|
||||
if not code or not _DIGITS_ONLY_RE.match(code):
|
||||
return False
|
||||
body, check_digit = code[:-1], code[-1]
|
||||
try:
|
||||
expected = gtin_check_digit(body)
|
||||
except (ValueError, IndexError):
|
||||
return False
|
||||
return str(expected) == check_digit
|
||||
|
||||
|
||||
def classify_barcode_type(code: str) -> BarcodeType:
|
||||
"""Barcode type purely from its (already checksum-validated) length.
|
||||
UPC-A (12 digits) is the one ambiguous case worth calling out: it is
|
||||
numerically a GTIN-13 with a leading zero, but is reported as UPC-A
|
||||
here since that's the label the requesting spec/UI expects for a
|
||||
12-digit code."""
|
||||
return _VALID_LENGTHS.get(len(code), BarcodeType.UNKNOWN)
|
||||
|
||||
|
||||
def validate_barcode(raw: Optional[str]) -> Optional[str]:
|
||||
"""Single entry point: normalize, check length, check checksum.
|
||||
Returns the clean digits-only barcode string if and only if it is a
|
||||
genuinely valid GTIN-8/12/13/14, else None. This is the ONLY function
|
||||
other modules should call to decide "is this barcode good enough to
|
||||
store" - never inline a length/regex check elsewhere.
|
||||
"""
|
||||
code = normalize_barcode(raw)
|
||||
if not code:
|
||||
return None
|
||||
if len(code) not in _VALID_LENGTHS:
|
||||
return None
|
||||
if not has_valid_checksum(code):
|
||||
return None
|
||||
return code
|
||||
|
||||
|
||||
def to_ean13(code: str) -> Optional[str]:
|
||||
"""Zero-pad a valid UPC-A (12 digits) up to its equivalent EAN-13
|
||||
representation. GTIN-8 is intentionally NOT padded (an 8-digit GTIN is
|
||||
its own distinct symbology, not a truncated EAN-13) - returns None for
|
||||
anything that isn't 12 or 13 digits already."""
|
||||
if len(code) == 13:
|
||||
return code
|
||||
if len(code) == 12:
|
||||
return "0" + code
|
||||
return None
|
||||
80
app/services/enrichment/base.py
Normal file
80
app/services/enrichment/base.py
Normal file
@@ -0,0 +1,80 @@
|
||||
"""
|
||||
Base contract every enrichment stage implements.
|
||||
|
||||
A stage takes ONE already-validated catalog row (a plain dict, the same
|
||||
`enhanced_product` shape `catalog_engine.py` builds) plus the brand name,
|
||||
and returns the SAME dict with additional fields merged in - it never
|
||||
removes or renames a key it didn't add itself, and it never raises: any
|
||||
internal failure is caught and reported via `StageOutcome.error` instead.
|
||||
|
||||
To add a new enrichment stage later (HSN, GST, nutrition, allergens, ...):
|
||||
1. Subclass `EnrichmentStage`.
|
||||
2. Implement `async def enrich_one(product, brand) -> StageOutcome`.
|
||||
3. Register an instance in `pipeline.run_default_pipeline()` (or build
|
||||
a custom `EnrichmentPipeline([...])` for a one-off run).
|
||||
No other file needs to change - `catalog_engine.py`'s call site and
|
||||
`vector_store.py`'s upsert already iterate whatever keys are present on the
|
||||
product dict via `.get(...)`, so a new stage's fields flow through to the
|
||||
database/JSON export automatically the same way barcode fields do.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from abc import ABC, abstractmethod
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class StageOutcome:
|
||||
"""Result of running one stage on one product row."""
|
||||
stage_name: str
|
||||
fields: Dict[str, Any] = field(default_factory=dict)
|
||||
error: Optional[str] = None
|
||||
|
||||
@property
|
||||
def ok(self) -> bool:
|
||||
return self.error is None
|
||||
|
||||
|
||||
class EnrichmentStage(ABC):
|
||||
"""One independent, pluggable enrichment step.
|
||||
|
||||
`name` is used for logging/metrics only. `enabled` lets a stage report
|
||||
itself as switched off (e.g. via a settings flag) without the pipeline
|
||||
orchestrator needing to know why - it's simply skipped and every
|
||||
product passes through untouched.
|
||||
"""
|
||||
|
||||
name: str = "unnamed_stage"
|
||||
|
||||
@property
|
||||
def enabled(self) -> bool:
|
||||
return True
|
||||
|
||||
@abstractmethod
|
||||
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
|
||||
"""Enrich a single product row. MUST NOT raise - catch internally
|
||||
and return a StageOutcome with `error` set instead."""
|
||||
raise NotImplementedError
|
||||
|
||||
async def apply(self, product: Dict[str, Any], brand: str) -> Dict[str, Any]:
|
||||
"""Run this stage on `product` and merge the result in place.
|
||||
Never raises - a stage bug degrades to "no fields added", logged,
|
||||
rather than aborting the whole catalog row.
|
||||
"""
|
||||
try:
|
||||
outcome = await self.enrich_one(product, brand)
|
||||
except Exception as e: # last-resort safety net - stages should
|
||||
# already catch their own errors, but a pipeline-wide guarantee
|
||||
# of "never raises" is worth the redundancy here.
|
||||
logger.error(f"[{self.name}] unhandled exception enriching '{product.get('product_name')}': {e}")
|
||||
return product
|
||||
|
||||
if outcome.fields:
|
||||
product.update(outcome.fields)
|
||||
if not outcome.ok:
|
||||
logger.debug(f"[{self.name}] {product.get('product_name')}: {outcome.error}")
|
||||
return product
|
||||
32
app/services/enrichment/hsn_gst/__init__.py
Normal file
32
app/services/enrichment/hsn_gst/__init__.py
Normal file
@@ -0,0 +1,32 @@
|
||||
"""
|
||||
HSN / GST & Pricing enrichment module.
|
||||
|
||||
Public entry points:
|
||||
resolve_hsn_gst(category, product_title="") -> HsnGstInfo
|
||||
Deterministic, offline category -> (HSN code, GST %, review flag)
|
||||
resolution, mirroring the shape already present in the coca-cola
|
||||
and milky_mist catalog exports (hsn_code / gst_percent /
|
||||
hsn_gst_needs_review / selling_price / tax_amount /
|
||||
final_selling_price / cost_price / profit_before_tax /
|
||||
profit_after_tax).
|
||||
|
||||
enrich_pricing_fields(product) -> dict
|
||||
Computes selling_price / tax_amount / final_selling_price (plus the
|
||||
always-null cost/profit fields) from a catalog row's price_range,
|
||||
using the same convention as the existing enriched brands: the
|
||||
selling price is the upper bound of the row's retail price range.
|
||||
|
||||
See the package's module docstrings for the architecture:
|
||||
models.py - HsnGstInfo result type + curated category -> HSN/GST table
|
||||
stage.py - adapts the resolver to the generic EnrichmentStage
|
||||
contract used by app/services/enrichment/pipeline.py
|
||||
"""
|
||||
from app.services.enrichment.hsn_gst.models import HsnGstInfo, resolve_hsn_gst, enrich_pricing_fields
|
||||
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
|
||||
|
||||
__all__ = [
|
||||
"HsnGstInfo",
|
||||
"resolve_hsn_gst",
|
||||
"enrich_pricing_fields",
|
||||
"HsnGstEnrichmentStage",
|
||||
]
|
||||
259
app/services/enrichment/hsn_gst/models.py
Normal file
259
app/services/enrichment/hsn_gst/models.py
Normal file
@@ -0,0 +1,259 @@
|
||||
"""
|
||||
Curated, deterministic HSN / GST lookup for catalog product categories.
|
||||
|
||||
WHY THIS IS NOT "MADE UP"
|
||||
--------------------------
|
||||
HSN (Harmonized System of Nomenclature) is the 4-8 digit goods-classification
|
||||
code used on every Indian tax invoice, and GST% is the Indian Goods &
|
||||
Services Tax rate applied to that HSN chapter. Both are public, legal,
|
||||
category-level facts - not per-SKU secrets. Every entry in `HSN_GST_TABLE`
|
||||
below is a *typical* classification for the product category as sold at
|
||||
retail (matching the values already present in the coca-cola / milky_mist
|
||||
catalog exports where those overlap, e.g. Beverages -> 2009, Dairy -> 0401,
|
||||
Dairy - Desserts -> 2105, Tea & Coffee -> 0902).
|
||||
|
||||
The catch: an HSN chapter can cover several GST rates depending on the exact
|
||||
item and how it is packaged, so a category-level guess can be wrong for a
|
||||
specific product. That is precisely what `HSN_GST_NEEDS_REVIEW` is for - it is
|
||||
True whenever the mapping is approximate (the default), and False only for the
|
||||
few categories whose retail classification is unambiguous. The flag exists so
|
||||
a human reviewer / the Streamlit UI can spot-check exactly these rows.
|
||||
|
||||
DESIGN PRINCIPLES
|
||||
------------------
|
||||
- DETERMINISTIC & OFFLINE. No LLM, no network, no randomness. The same
|
||||
category always yields the same (HSN, GST%, review) tuple, so re-runs and
|
||||
backfills are stable and diffable.
|
||||
- ADDITIVE. Only ever attaches new keys to a product row; never removes or
|
||||
rewrites an existing key (the stage skips rows that already carry these
|
||||
fields, so re-running the pipeline over an already-enriched catalog is a
|
||||
no-op).
|
||||
- SAFE DEFAULTS. Unknown/unrecognised categories get hsn_code=None and
|
||||
gst_percent=None with needs_review=True, rather than a fabricated code -
|
||||
a wrong HSN on a real tax document is worse than no HSN at all.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from typing import Dict, Optional, Tuple
|
||||
|
||||
|
||||
# (HSN code, GST %, needs_review). Review=True whenever the category can map
|
||||
# to more than one GST rate / HSN chapter at retail; False for unambiguous ones.
|
||||
HSN_GST_TABLE: Dict[str, Tuple[str, int, bool]] = {
|
||||
# ---- Food & beverage (matches the existing coca-cola / milky_mist exports) ----
|
||||
"Dairy": ("0401", 5, False),
|
||||
"Cheese": ("0406", 12, True),
|
||||
"Dairy - Desserts": ("2105", 5, False),
|
||||
"Ice Cream": ("2105", 18, True),
|
||||
"Beverages": ("2009", 5, False),
|
||||
"Tea & Coffee": ("0902", 5, False),
|
||||
"Food & Beverages": ("2106", 18, True),
|
||||
"Health Drinks": ("2202", 18, True),
|
||||
"Health Foods": ("2106", 18, True),
|
||||
"Breakfast Cereal": ("1904", 18, False),
|
||||
"Chocolates": ("1806", 18, False),
|
||||
"Candy & Confectionery": ("1704", 18, False),
|
||||
"Biscuits & Cookies": ("1905", 18, True),
|
||||
"Crackers": ("1905", 18, True),
|
||||
"Rusk": ("1905", 5, True),
|
||||
"Cakes & Muffins": ("1905", 18, True),
|
||||
"Bakery & Breads": ("1905", 5, False),
|
||||
"Snacks": ("1905", 18, True),
|
||||
"Noodles & Instant Food": ("1902", 18, False),
|
||||
"Pasta & Noodles": ("1902", 18, False),
|
||||
"Atta & Staples": ("1101", 5, False),
|
||||
"Salt & Staples": ("2501", 5, False),
|
||||
"Pulses, Grains & Spices": ("0713", 5, True),
|
||||
"Spices & Masalas": ("0910", 5, False),
|
||||
"Cooking Oils": ("1517", 5, False),
|
||||
"Pickles & Chutneys": ("2001", 12, True),
|
||||
"Dry Fruits & Nuts": ("0801", 12, True),
|
||||
"Food - Spreads": ("2007", 12, True),
|
||||
"Food - Mixes": ("2106", 18, True),
|
||||
"Food - Soups & Sauces": ("2103", 12, False),
|
||||
# ---- Personal care & household ----
|
||||
"Hair Care": ("3305", 18, False),
|
||||
"Skin Care": ("3304", 18, False),
|
||||
"Skin & Bath Care": ("3307", 18, False),
|
||||
"Bath Soap": ("3401", 18, False),
|
||||
"Beauty Care": ("3304", 18, False),
|
||||
"Oral Care": ("3306", 18, False),
|
||||
"Fragrance & Deodorants": ("3303", 18, False),
|
||||
"Men's Grooming": ("8212", 18, True),
|
||||
"Baby Care": ("3304", 18, True),
|
||||
"Feminine Hygiene": ("9619", 12, False),
|
||||
"Personal Care": ("3307", 18, True),
|
||||
"Detergents & Fabric Care": ("3402", 18, False),
|
||||
"Dishwash": ("3402", 18, False),
|
||||
"Household Cleaning": ("3402", 18, True),
|
||||
"Household - Air Freshener": ("3307", 18, True),
|
||||
"Personal Care - Mosquito Repellent": ("3808", 18, True),
|
||||
# ---- Health care ----
|
||||
"Health Care - Cold & Cough": ("3004", 12, True),
|
||||
"Health Care - Antiseptic": ("3808", 18, True),
|
||||
"Health Care - Ayurvedic": ("3003", 12, True),
|
||||
"Health Care - Digestive": ("3004", 12, True),
|
||||
"Health Care - First Aid": ("3005", 12, True),
|
||||
}
|
||||
|
||||
# Keyword fallbacks so a product whose category label is slightly different
|
||||
# from the table (or missing entirely) still resolves to a sensible chapter.
|
||||
_KEYWORD_FALLBACKS: Tuple[Tuple[str, str, int, bool], ...] = (
|
||||
("toothpaste", "3306", 18, False),
|
||||
("shampoo", "3305", 18, False),
|
||||
("soap", "3401", 18, False),
|
||||
("detergent", "3402", 18, False),
|
||||
("biscuit", "1905", 18, True),
|
||||
("cookie", "1905", 18, True),
|
||||
("chocolate", "1806", 18, False),
|
||||
("candy", "1704", 18, False),
|
||||
("toffee", "1704", 18, False),
|
||||
("milk", "0401", 5, False),
|
||||
("paneer", "0401", 5, False),
|
||||
("curd", "0401", 5, False),
|
||||
("cheese", "0406", 12, True),
|
||||
("ice cream", "2105", 18, True),
|
||||
("juice", "2009", 5, False),
|
||||
("tea", "0902", 5, False),
|
||||
("coffee", "0902", 5, False),
|
||||
("health drink", "2202", 18, True),
|
||||
("chips", "1905", 18, True),
|
||||
("namkeen", "1905", 18, True),
|
||||
("bread", "1905", 5, False),
|
||||
("atta", "1101", 5, False),
|
||||
("flour", "1101", 5, False),
|
||||
("masala", "0910", 5, False),
|
||||
("spice", "0910", 5, False),
|
||||
("pasta", "1902", 18, False),
|
||||
("noodle", "1902", 18, False),
|
||||
("cooking oil", "1517", 5, False),
|
||||
("oil", "1517", 5, True),
|
||||
("pickle", "2001", 12, True),
|
||||
("jam", "2007", 12, True),
|
||||
("honey", "0409", 5, False),
|
||||
("raisin", "0806", 5, True),
|
||||
("dry fruit", "0801", 12, True),
|
||||
("cereal", "1904", 18, False),
|
||||
("deodorant", "3303", 18, False),
|
||||
("perfume", "3303", 18, False),
|
||||
("lipstick", "3304", 18, False),
|
||||
("makeup", "3304", 18, False),
|
||||
("face wash", "3307", 18, False),
|
||||
("lotion", "3307", 18, False),
|
||||
("sunscreen", "3304", 18, False),
|
||||
("razor", "8212", 18, True),
|
||||
("diaper", "9619", 12, False),
|
||||
("mosquito", "3808", 18, True),
|
||||
("repellent", "3808", 18, True),
|
||||
("air freshener", "3307", 18, True),
|
||||
("dishwash", "3402", 18, False),
|
||||
("floor cleaner", "3402", 18, True),
|
||||
("hand wash", "3401", 18, False),
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class HsnGstInfo:
|
||||
"""HSN / GST resolution for one product category, mirroring the fields
|
||||
stored on enriched catalog rows (coca-cola / milky_mist exports)."""
|
||||
hsn_code: Optional[str] = None
|
||||
gst_percent: Optional[int] = None
|
||||
hsn_gst_needs_review: bool = True
|
||||
|
||||
def as_fields(self) -> Dict[str, object]:
|
||||
return {
|
||||
"hsn_code": self.hsn_code,
|
||||
"gst_percent": self.gst_percent,
|
||||
"hsn_gst_needs_review": self.hsn_gst_needs_review,
|
||||
}
|
||||
|
||||
|
||||
_UNKNOWN = HsnGstInfo(hsn_code=None, gst_percent=None, hsn_gst_needs_review=True)
|
||||
|
||||
|
||||
def resolve_hsn_gst(category: Optional[str], product_title: str = "") -> HsnGstInfo:
|
||||
"""Resolve (HSN code, GST %, needs_review) for a product category.
|
||||
|
||||
Exact category matches in `HSN_GST_TABLE` win; otherwise the product
|
||||
title/description is scanned against `_KEYWORD_FALLBACKS`; otherwise
|
||||
returns the safe unknown shape (all-None, needs_review=True).
|
||||
"""
|
||||
cat = (category or "").strip()
|
||||
if cat:
|
||||
exact = HSN_GST_TABLE.get(cat)
|
||||
if exact:
|
||||
return HsnGstInfo(*exact)
|
||||
|
||||
haystack = f"{product_title or ''} {cat}".lower()
|
||||
for kw, hsn, gst, review in _KEYWORD_FALLBACKS:
|
||||
if kw in haystack:
|
||||
return HsnGstInfo(hsn_code=hsn, gst_percent=gst, hsn_gst_needs_review=review)
|
||||
|
||||
return _UNKNOWN
|
||||
|
||||
|
||||
# Matches a price range string of the form "₹12-14" (also tolerates
|
||||
# "Rs 12-14", "12 - 14", a single "₹12", or corrupted rupee symbols).
|
||||
_RANGE_RE = re.compile(r"₹?\s*(\d+(?:\.\d+)?)\s*[-–—]\s*(\d+(?:\.\d+)?)")
|
||||
_SINGLE_RE = re.compile(r"₹?\s*(\d+(?:\.\d+)?)")
|
||||
|
||||
|
||||
def _extract_selling_price(price_range: object) -> Optional[float]:
|
||||
"""Return the upper bound of a row's price range (the convention used by
|
||||
the existing coca-cola / milky_mist exports for `selling_price`), or the
|
||||
single value when the row only carries one price."""
|
||||
if price_range is None:
|
||||
return None
|
||||
text = str(price_range).replace(",", "").strip()
|
||||
if not text:
|
||||
return None
|
||||
m = _RANGE_RE.search(text)
|
||||
if m:
|
||||
try:
|
||||
return float(m.group(2))
|
||||
except ValueError:
|
||||
return None
|
||||
m = _SINGLE_RE.search(text)
|
||||
if m:
|
||||
try:
|
||||
return float(m.group(1))
|
||||
except ValueError:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def _round2(value: Optional[float]) -> Optional[float]:
|
||||
if value is None:
|
||||
return None
|
||||
return round(value, 2)
|
||||
|
||||
|
||||
def enrich_pricing_fields(product: dict) -> Dict[str, object]:
|
||||
"""Compute the pricing fields added by this stage for one catalog row.
|
||||
|
||||
Convention (matches the coca-cola / milky_mist exports):
|
||||
selling_price = upper bound of the row's `price_range`
|
||||
tax_amount = selling_price * gst_percent / 100
|
||||
final_selling_price = selling_price + tax_amount
|
||||
`cost_price`, `profit_before_tax` and `profit_after_tax` are not
|
||||
derivable from the data this pipeline generates, so they stay None
|
||||
(exactly as the existing enriched exports store them).
|
||||
"""
|
||||
selling_price = _extract_selling_price(product.get("price_range"))
|
||||
gst = product.get("gst_percent")
|
||||
tax_amount = None
|
||||
final_selling_price = None
|
||||
if selling_price is not None and isinstance(gst, (int, float)) and gst and gst > 0:
|
||||
tax_amount = _round2(selling_price * float(gst) / 100.0)
|
||||
final_selling_price = _round2(selling_price + tax_amount)
|
||||
|
||||
return {
|
||||
"selling_price": _round2(selling_price),
|
||||
"cost_price": None,
|
||||
"tax_amount": tax_amount,
|
||||
"final_selling_price": final_selling_price,
|
||||
"profit_before_tax": None,
|
||||
"profit_after_tax": None,
|
||||
}
|
||||
48
app/services/enrichment/hsn_gst/stage.py
Normal file
48
app/services/enrichment/hsn_gst/stage.py
Normal file
@@ -0,0 +1,48 @@
|
||||
"""Adapts the HSN/GST resolver to the EnrichmentStage contract so it can be
|
||||
registered in app/services/enrichment/pipeline.py's default pipeline.
|
||||
|
||||
Unlike the barcode stage this needs no network access at all - HSN/GST are
|
||||
deterministic, category-level facts - so it is synchronous and effectively
|
||||
free. It is pure ADDITIVE: rows that already carry the hsn_code/gst_percent
|
||||
keys (e.g. re-running the pipeline over a previously-enriched catalog) are
|
||||
left untouched.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Any, Dict
|
||||
|
||||
from app.infrastructure.settings import ENABLE_HSN_GST_ENRICHMENT
|
||||
from app.services.enrichment.base import EnrichmentStage, StageOutcome
|
||||
from app.services.enrichment.hsn_gst.models import enrich_pricing_fields, resolve_hsn_gst
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class HsnGstEnrichmentStage(EnrichmentStage):
|
||||
name = "hsn_gst"
|
||||
|
||||
@property
|
||||
def enabled(self) -> bool:
|
||||
return ENABLE_HSN_GST_ENRICHMENT
|
||||
|
||||
async def enrich_one(self, product: Dict[str, Any], brand: str) -> StageOutcome:
|
||||
# Already enriched (previous run / manual backfill) - never overwrite.
|
||||
if product.get("hsn_code") is not None or product.get("gst_percent") is not None:
|
||||
return StageOutcome(stage_name=self.name, fields={})
|
||||
|
||||
title = product.get("title") or product.get("product_name") or ""
|
||||
category = product.get("category") or ""
|
||||
|
||||
hsn_info = resolve_hsn_gst(category, title)
|
||||
fields: Dict[str, Any] = hsn_info.as_fields()
|
||||
# Pricing fields depend on gst_percent, which was just resolved
|
||||
# above, so hand the resolved rate to the pricing helper.
|
||||
fields.update(enrich_pricing_fields({**product, "gst_percent": fields["gst_percent"]}))
|
||||
|
||||
needs_review = fields["hsn_gst_needs_review"]
|
||||
return StageOutcome(
|
||||
stage_name=self.name,
|
||||
fields=fields,
|
||||
error=None if not needs_review else "category-level HSN/GST estimate - verify against the physical pack"
|
||||
)
|
||||
89
app/services/enrichment/pipeline.py
Normal file
89
app/services/enrichment/pipeline.py
Normal file
@@ -0,0 +1,89 @@
|
||||
"""
|
||||
Orchestrates a list of independent `EnrichmentStage`s over a batch of
|
||||
catalog rows, running each stage's per-row work concurrently (bounded by
|
||||
`max_concurrency`) and NEVER letting one row's failure affect any other
|
||||
row or stage.
|
||||
|
||||
`catalog_engine.py` calls `run_default_pipeline()` once, between the
|
||||
deterministic-validation step (product_validator.py) and catalog assembly
|
||||
- see that file's "Step 2.6" for the call site and
|
||||
docs/BARCODE_ENRICHMENT.md for the full pipeline-position rationale.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional, Sequence
|
||||
|
||||
from app.services.enrichment.base import EnrichmentStage
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class EnrichmentPipeline:
|
||||
def __init__(self, stages: Sequence[EnrichmentStage], max_concurrency: int = 5):
|
||||
self.stages = [s for s in stages if s.enabled]
|
||||
self.max_concurrency = max(1, max_concurrency)
|
||||
|
||||
async def run(self, products: List[Dict[str, Any]], brand: str) -> List[Dict[str, Any]]:
|
||||
if not self.stages or not products:
|
||||
return products
|
||||
|
||||
semaphore = asyncio.Semaphore(self.max_concurrency)
|
||||
|
||||
async def _run_row(product: Dict[str, Any]) -> Dict[str, Any]:
|
||||
async with semaphore:
|
||||
for stage in self.stages:
|
||||
product = await stage.apply(product, brand)
|
||||
return product
|
||||
|
||||
results = await asyncio.gather(*(_run_row(p) for p in products), return_exceptions=True)
|
||||
|
||||
final: List[Dict[str, Any]] = []
|
||||
for original, result in zip(products, results):
|
||||
if isinstance(result, Exception):
|
||||
logger.error(f"Enrichment pipeline failed for '{original.get('product_name')}': {result}")
|
||||
final.append(original)
|
||||
else:
|
||||
final.append(result)
|
||||
return final
|
||||
|
||||
|
||||
def _build_default_stages() -> List[EnrichmentStage]:
|
||||
"""Import stages lazily so importing this module never pulls in a
|
||||
stage's own dependencies (network clients, DB drivers, ...) unless a
|
||||
default pipeline is actually requested."""
|
||||
stages: List[EnrichmentStage] = []
|
||||
try:
|
||||
from app.services.enrichment.barcode.stage import BarcodeEnrichmentStage
|
||||
stages.append(BarcodeEnrichmentStage())
|
||||
except Exception as e:
|
||||
logger.error(f"Barcode enrichment stage unavailable: {e}")
|
||||
|
||||
# HSN / GST & pricing enrichment (see app/services/enrichment/hsn_gst/) -
|
||||
# deterministic, offline, pure-additive. Runs AFTER the barcode stage so
|
||||
# every stored/exported row carries both sets of fields; a failure here
|
||||
# degrades a row to "no HSN/GST attached", never drops the row.
|
||||
try:
|
||||
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
|
||||
stages.append(HsnGstEnrichmentStage())
|
||||
except Exception as e:
|
||||
logger.error(f"HSN/GST enrichment stage unavailable: {e}")
|
||||
|
||||
# Future stages register here, e.g.:
|
||||
# from app.services.enrichment.nutrition.stage import NutritionEnrichmentStage
|
||||
# stages.append(NutritionEnrichmentStage())
|
||||
return stages
|
||||
|
||||
|
||||
_default_pipeline: Optional[EnrichmentPipeline] = None
|
||||
|
||||
|
||||
async def run_default_pipeline(products: List[Dict[str, Any]], brand: str,
|
||||
max_concurrency: int = 5) -> List[Dict[str, Any]]:
|
||||
"""Convenience entry point used by catalog_engine.py: runs every
|
||||
currently-registered, enabled enrichment stage over `products`."""
|
||||
global _default_pipeline
|
||||
if _default_pipeline is None:
|
||||
_default_pipeline = EnrichmentPipeline(_build_default_stages(), max_concurrency=max_concurrency)
|
||||
return await _default_pipeline.run(products, brand)
|
||||
@@ -353,3 +353,32 @@ def parse_price_string(text: str) -> Optional[float]:
|
||||
if len(numbers) >= 2:
|
||||
return (numbers[0] + numbers[1]) / 2.0
|
||||
return numbers[0]
|
||||
|
||||
|
||||
def estimate_price_range_for_size(
|
||||
size: str,
|
||||
product_title: str = "",
|
||||
brand: str = "",
|
||||
category_hint: str = "",
|
||||
) -> Tuple[int, int]:
|
||||
"""Return a (min, max) price range for a SINGLE pack size.
|
||||
|
||||
Ported from the sibling Universal_Catalog_Barcode_Enrichment project, where
|
||||
`product_validator.validate_price_range` uses it as the anchor to judge
|
||||
whether a scraped price is plausible for the pack size. Additive: no
|
||||
existing function in this module changes.
|
||||
|
||||
The range is centred on `estimate_price` with a deterministic +-5%-15%
|
||||
jitter keyed on product+size, so the same product gets the same range on
|
||||
every run while still representing genuine retail spread rather than a
|
||||
single fixed MRP.
|
||||
"""
|
||||
price = estimate_price(size, product_title, brand, category_hint)
|
||||
seed = f"{brand}|{category_hint}|{product_title}|{size}|range"
|
||||
jitter = _deterministic_unit(seed)
|
||||
spread = 0.05 + jitter * 0.10
|
||||
lo = max(1, int(round(price * (1 - spread))))
|
||||
hi = int(round(price * (1 + spread)))
|
||||
if hi <= lo:
|
||||
hi = lo + 1
|
||||
return (lo, hi)
|
||||
|
||||
445
app/services/product_validator.py
Normal file
445
app/services/product_validator.py
Normal file
@@ -0,0 +1,445 @@
|
||||
"""
|
||||
Deterministic Product Validation & Confidence Scoring
|
||||
=======================================================
|
||||
|
||||
WHY THIS FILE EXISTS
|
||||
---------------------
|
||||
Every stage upstream of this module (ollama_service.py, title_validator.py,
|
||||
price_estimator.py, sku_service.py) already applies point-fixes for
|
||||
*specific, previously-reported* hallucination patterns (category-inconsistent
|
||||
titles, unrealistic prices, non-existent brands, malformed JSON, ...). None
|
||||
of them, however, ever made a final, holistic pass over a fully-assembled
|
||||
catalog row and asked "taken together, does this look like a REAL product
|
||||
worth keeping?" That gap is why individually-plausible-looking rows with
|
||||
compounding small defects (a slightly-off price + a missing image + a vague
|
||||
size) could still reach the database.
|
||||
|
||||
This module is that final gate. It is intentionally:
|
||||
|
||||
- DETERMINISTIC: no LLM call, no randomness. The exact same product dict
|
||||
always produces the exact same report. That is what makes this a
|
||||
hallucination *filter* rather than another hallucination *source*.
|
||||
- CONSERVATIVE: every check only fires on strong, explainable evidence
|
||||
(reuses title_validator's category-conflict detector, price_estimator's
|
||||
category-anchored price bands, and sku_service's own SKU format) rather
|
||||
than guessing. False rejections are treated as seriously as false
|
||||
acceptances - see docs/VALIDATION_PIPELINE.md for the tuning rationale.
|
||||
- ADDITIVE: this module has zero dependents before this change and does not
|
||||
modify any existing function's behaviour; it is only ever imported and
|
||||
called from new code (see app/core/catalog_engine.py's call site).
|
||||
|
||||
USAGE
|
||||
-----
|
||||
Call `validate_product(product_dict, brand, ...)` once per fully-assembled
|
||||
catalog row, AFTER pricing/SKU/image resolution but BEFORE the row is
|
||||
appended to the catalog output / sent to upsert_brand_products(). The
|
||||
returned ValidationReport carries a 0.0-1.0 confidence score and a
|
||||
status of "verified" / "needs_review" / "rejected" that the caller uses to
|
||||
decide whether to keep, flag, or drop the row - see
|
||||
ProductCatalogEngine.generate_catalog() in catalog_engine.py.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from typing import Any, Optional
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from app.services.title_validator import find_category_conflicts
|
||||
from app.services import price_estimator
|
||||
from app.services import category_units as cu
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Categories that are never legitimate for any FMCG / personal-care brand.
|
||||
# Deliberately duplicated (rather than imported) from
|
||||
# app.services.ollama_service._IRRELEVANT_CATEGORIES: ollama_service already
|
||||
# imports from brand_registry, and this module is imported BY
|
||||
# catalog_engine.py alongside ollama_service, so importing a private
|
||||
# (underscore-prefixed) name across modules would create a fragile coupling
|
||||
# for a list that changes maybe once a year. Keep both lists in sync if you
|
||||
# add a new one.
|
||||
# ---------------------------------------------------------------------------
|
||||
IRRELEVANT_CATEGORIES: set[str] = {
|
||||
"pet supplies", "pet food", "pet care", "automotive", "auto parts",
|
||||
"automotive parts", "car accessories", "motorcycle", "bike",
|
||||
"electronics", "gadgets", "computers", "mobile phones",
|
||||
"furniture", "home decor", "furnishings", "appliances",
|
||||
"clothing", "footwear", "fashion", "jewelry", "accessories",
|
||||
"office supplies", "stationery", "books", "toys", "games",
|
||||
"sports equipment", "sports gear", "gardening", "garden supplies",
|
||||
"hardware", "tools", "building materials",
|
||||
"musical instruments", "party supplies", "craft supplies",
|
||||
}
|
||||
|
||||
# Title strings that are structurally a placeholder/fabrication artefact
|
||||
# rather than a real product name, independent of any specific brand.
|
||||
_PLACEHOLDER_TITLE_PATTERNS: list[re.Pattern] = [
|
||||
# Bare "Product 3" / "Brand Product 3", or any single leading token
|
||||
# (typically the brand name) immediately followed by "Product N" - the
|
||||
# exact shape of the hardcoded placeholder data in the unused
|
||||
# agentic_pipeline.py scaffold (see that file's module docstring) and
|
||||
# the generic fallback a hallucinating LLM sometimes produces when it
|
||||
# runs out of real products to enumerate.
|
||||
re.compile(r"^\s*(\S+\s+)?product\s*#?\s*\d+\s*$", re.IGNORECASE),
|
||||
re.compile(r"^\s*(unknown|n/?a|unnamed|untitled|sample|placeholder|tbd|todo|test\s*product)\s*$", re.IGNORECASE),
|
||||
re.compile(r"^https?://", re.IGNORECASE),
|
||||
# "5 Star (Soft Drink Candy) - Soft Drink Candy" style category echo -
|
||||
# same pattern catalog_engine.py already filters at discovery time; kept
|
||||
# here too as defence-in-depth for rows that reach this stage some other
|
||||
# way (e.g. a future direct-insert code path that skips catalog_engine).
|
||||
re.compile(r"\(([^)]{2,40})\)\s*-\s*\1", re.IGNORECASE),
|
||||
]
|
||||
|
||||
# Internal SKUs are always formatted BRAND-PRODUCT-SIZE-SEQ by
|
||||
# sku_service.generate_internal_sku(), e.g. "AACHI-SAM-500-001".
|
||||
_INTERNAL_SKU_RE = re.compile(r"^[A-Z0-9]{2,15}-[A-Z0-9]{2,15}-[A-Z0-9]{1,10}-\d{3,}$")
|
||||
|
||||
_PRICE_RANGE_RE = re.compile(r"₹\s*(\d+(?:\.\d+)?)\s*-\s*(\d+(?:\.\d+)?)")
|
||||
|
||||
# Plausible bounds for a single retail FMCG pack, in gram/ml equivalent.
|
||||
# Below 1: parsing failure. Above 30000 (30kg): almost certainly a
|
||||
# hallucinated bulk size no e-commerce FMCG listing would use (real bulk
|
||||
# packs top out around 25kg rice/atta/detergent sacks).
|
||||
_MIN_PACK_GRAMS = 1.0
|
||||
_MAX_PACK_GRAMS = 30_000.0
|
||||
|
||||
|
||||
class ValidationIssue(BaseModel):
|
||||
"""A single, explainable reason a confidence score moved."""
|
||||
|
||||
field: str
|
||||
severity: str = Field(description="'error' (strong evidence of a problem) or 'warning' (weaker signal)")
|
||||
message: str
|
||||
penalty: float = 0.0
|
||||
|
||||
|
||||
class ValidationConfig(BaseModel):
|
||||
"""Tunable thresholds - see docs/VALIDATION_PIPELINE.md for defaults
|
||||
and how to change them via settings.py without editing this file."""
|
||||
|
||||
reject_below: float = 0.35
|
||||
review_below: float = 0.70
|
||||
|
||||
|
||||
class ValidationReport(BaseModel):
|
||||
"""Result of validating one fully-assembled catalog row."""
|
||||
|
||||
product_name: str
|
||||
status: str = "verified" # "verified" | "needs_review" | "rejected"
|
||||
confidence: float = 1.0
|
||||
grounded: bool = False
|
||||
issues: list[ValidationIssue] = Field(default_factory=list)
|
||||
|
||||
@property
|
||||
def is_rejected(self) -> bool:
|
||||
return self.status == "rejected"
|
||||
|
||||
@property
|
||||
def needs_review(self) -> bool:
|
||||
return self.status == "needs_review"
|
||||
|
||||
def summary(self) -> str:
|
||||
if not self.issues:
|
||||
return f"{self.product_name}: OK (confidence={self.confidence:.2f})"
|
||||
reasons = "; ".join(f"{i.field}: {i.message}" for i in self.issues)
|
||||
return f"{self.product_name}: {self.status} (confidence={self.confidence:.2f}) - {reasons}"
|
||||
|
||||
|
||||
def _looks_like_placeholder(title: str) -> bool:
|
||||
if not title or not title.strip():
|
||||
return True
|
||||
return any(p.search(title.strip()) for p in _PLACEHOLDER_TITLE_PATTERNS)
|
||||
|
||||
|
||||
def validate_size(size: str) -> tuple[bool, Optional[str]]:
|
||||
"""Check that `size` parses to a plausible FMCG retail-pack quantity.
|
||||
|
||||
Reuses price_estimator's own size parser so "what counts as a valid
|
||||
size" can never silently drift out of sync with what pricing actually
|
||||
uses to compute cost.
|
||||
"""
|
||||
if not size or not size.strip():
|
||||
return False, "size/pack is missing"
|
||||
grams = price_estimator._parse_size_to_grams(size)
|
||||
if grams <= 0:
|
||||
return False, f"could not parse a usable quantity from size '{size}'"
|
||||
if grams < _MIN_PACK_GRAMS or grams > _MAX_PACK_GRAMS:
|
||||
return False, f"parsed quantity ({grams:g}) from size '{size}' is outside a plausible retail-pack range"
|
||||
return True, None
|
||||
|
||||
|
||||
def validate_price_range(
|
||||
price_range: str, size: str, product_title: str, brand: str, category: str,
|
||||
) -> tuple[bool, Optional[str]]:
|
||||
"""Check that `price_range` is well-formed and not a gross outlier
|
||||
against the category+size price anchor computed by price_estimator.
|
||||
|
||||
The tolerance band (0.3x-3x of the anchor) is deliberately wide - this
|
||||
check exists to catch gross hallucinations (a ₹5,000 biscuit packet, a
|
||||
₹2 detergent bottle), not to second-guess normal retail price spread,
|
||||
which price_estimator's own range already accounts for.
|
||||
"""
|
||||
if not price_range or not price_range.strip():
|
||||
return False, "price_range is missing"
|
||||
m = _PRICE_RANGE_RE.search(price_range)
|
||||
if not m:
|
||||
return False, f"price_range '{price_range}' is not a well-formed ₹lo-hi string"
|
||||
lo, hi = float(m.group(1)), float(m.group(2))
|
||||
if lo <= 0 or hi <= 0 or hi < lo:
|
||||
return False, f"price_range '{price_range}' has non-positive or inverted bounds"
|
||||
try:
|
||||
anchor_lo, anchor_hi = price_estimator.estimate_price_range_for_size(
|
||||
size, product_title, brand, category
|
||||
)
|
||||
except Exception:
|
||||
return True, None # anchor unavailable (e.g. blank size) - don't fail the row on that alone
|
||||
if hi < anchor_lo * 0.3 or lo > anchor_hi * 3.0:
|
||||
return False, (
|
||||
f"price_range '{price_range}' is implausible for a {size or 'unspecified-size'} "
|
||||
f"{category or 'product'} (expected roughly ₹{anchor_lo}-{anchor_hi})"
|
||||
)
|
||||
return True, None
|
||||
|
||||
|
||||
def validate_sku(sku: str, sku_source: str) -> tuple[bool, Optional[str]]:
|
||||
"""Check `product_sku` is present and, for internally-generated SKUs,
|
||||
matches sku_service's own BRAND-PRODUCT-SIZE-SEQ format. Real
|
||||
marketplace IDs (sku_source != "Internal") are trusted as-is since they
|
||||
were found via an actual web lookup, not generated - their format
|
||||
legitimately varies by platform (ASIN vs Flipkart PID vs ...).
|
||||
"""
|
||||
if not sku or not sku.strip():
|
||||
return False, "product_sku is missing"
|
||||
if sku_source and sku_source != "Internal":
|
||||
return True, None
|
||||
if not _INTERNAL_SKU_RE.match(sku.strip()):
|
||||
return False, f"internal SKU '{sku}' does not match the expected BRAND-PRODUCT-SIZE-SEQ pattern"
|
||||
return True, None
|
||||
|
||||
|
||||
def validate_product(
|
||||
product: dict[str, Any],
|
||||
brand: str,
|
||||
*,
|
||||
category_resolved_deterministically: bool = False,
|
||||
grounded: bool = False,
|
||||
images_checked: bool = True,
|
||||
config: Optional[ValidationConfig] = None,
|
||||
) -> ValidationReport:
|
||||
"""Validate one fully-assembled catalog row and return a confidence-scored report.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
product:
|
||||
The enhanced product dict as built by catalog_engine.py, expected to
|
||||
contain (at minimum) title/product_name, category, size, price_range,
|
||||
product_sku, sku_source, image_urls.
|
||||
brand:
|
||||
The brand this row is being catalogued under.
|
||||
category_resolved_deterministically:
|
||||
True when the category came from brand_registry's curated sub-brand
|
||||
registry (ground truth) rather than a keyword heuristic - a positive
|
||||
evidence signal, not a validity requirement.
|
||||
grounded:
|
||||
True when an independent, external signal corroborates this exact
|
||||
product (e.g. an Open Food Facts catalog match, or the row already
|
||||
existing in the database from a prior validated run). See
|
||||
catalog_engine.py's call site for exactly what sets this.
|
||||
config:
|
||||
Optional threshold override; defaults to ValidationConfig() (which
|
||||
itself defaults to the values in app.infrastructure.settings).
|
||||
"""
|
||||
cfg = config or ValidationConfig()
|
||||
title = str(product.get("title") or product.get("product_name") or "").strip()
|
||||
category = str(product.get("category") or "").strip()
|
||||
size = str(product.get("size") or "").strip()
|
||||
price_range = str(product.get("price_range") or "").strip()
|
||||
sku = str(product.get("product_sku") or "").strip()
|
||||
sku_source = str(product.get("sku_source") or "").strip()
|
||||
images = product.get("image_urls") or []
|
||||
|
||||
report = ValidationReport(product_name=title or "(untitled)", grounded=grounded)
|
||||
score = 0.55 # neutral baseline - moves up/down based on evidence below
|
||||
|
||||
# 1. Title sanity -----------------------------------------------------
|
||||
if _looks_like_placeholder(title):
|
||||
report.issues.append(ValidationIssue(
|
||||
field="title", severity="error",
|
||||
message=f"title '{title}' looks like a placeholder/fabricated entry, not a real product name",
|
||||
penalty=0.50,
|
||||
))
|
||||
score -= 0.50
|
||||
elif len(title) < 2:
|
||||
report.issues.append(ValidationIssue(
|
||||
field="title", severity="error",
|
||||
message="title is too short to be a real product name", penalty=0.40,
|
||||
))
|
||||
score -= 0.40
|
||||
|
||||
# 2. Category sanity ----------------------------------------------------
|
||||
if not category or category == "Uncategorized":
|
||||
report.issues.append(ValidationIssue(
|
||||
field="category", severity="warning",
|
||||
message="category could not be resolved", penalty=0.10,
|
||||
))
|
||||
score -= 0.10
|
||||
elif category.lower() in IRRELEVANT_CATEGORIES:
|
||||
report.issues.append(ValidationIssue(
|
||||
field="category", severity="error",
|
||||
message=f"category '{category}' is not a valid FMCG/personal-care category", penalty=0.60,
|
||||
))
|
||||
score -= 0.60
|
||||
|
||||
# 3. Title <-> category consistency (reuses title_validator's own
|
||||
# conflict detector so the two modules can never disagree about what
|
||||
# counts as a contradiction) --------------------------------------
|
||||
if category and category not in ("Uncategorized", "General"):
|
||||
conflicts = find_category_conflicts(title, category)
|
||||
if conflicts:
|
||||
report.issues.append(ValidationIssue(
|
||||
field="title", severity="error",
|
||||
message=f"title contains word(s) inconsistent with category '{category}': {', '.join(conflicts)}",
|
||||
penalty=0.35,
|
||||
))
|
||||
score -= 0.35
|
||||
|
||||
# 4. Size / pack --------------------------------------------------------
|
||||
ok, msg = validate_size(size)
|
||||
if not ok:
|
||||
report.issues.append(ValidationIssue(field="size", severity="error", message=msg, penalty=0.25))
|
||||
score -= 0.25
|
||||
|
||||
# 4b. Size unit <-> category compatibility (see app/services/
|
||||
# category_units.py). This is a last-resort backstop: by the time a
|
||||
# row reaches this final validator, its size should already have
|
||||
# been through the generation-time gate (app/services/
|
||||
# generation_verifier.py, called from ollama_service.py) AND
|
||||
# catalog_engine.py's own dimension-unit defence-in-depth pass, both
|
||||
# of which reject/repair a bad unit long before enrichment. This
|
||||
# check exists purely to catch anything that reached validate_product
|
||||
# some other way (e.g. a pre-existing DB row from before this fix, or
|
||||
# a future caller that doesn't route through catalog_engine).
|
||||
if size:
|
||||
ok, msg = cu.validate_unit_for_category(size, category)
|
||||
if not ok:
|
||||
report.issues.append(ValidationIssue(field="size", severity="error", message=msg, penalty=0.15))
|
||||
score -= 0.15
|
||||
|
||||
# 5. Price range ----------------------------------------------------------
|
||||
ok, msg = validate_price_range(price_range, size, title, brand, category)
|
||||
if not ok:
|
||||
report.issues.append(ValidationIssue(field="price_range", severity="error", message=msg, penalty=0.30))
|
||||
score -= 0.30
|
||||
|
||||
# 6. SKU --------------------------------------------------------------------
|
||||
ok, msg = validate_sku(sku, sku_source)
|
||||
if not ok:
|
||||
report.issues.append(ValidationIssue(field="product_sku", severity="warning", message=msg, penalty=0.10))
|
||||
score -= 0.10
|
||||
|
||||
# 7. Image presence -----------------------------------------------------
|
||||
# Only meaningful if image search actually ran. When the operator disables
|
||||
# it (the store-catalog pipeline's offline mode), an empty image list says
|
||||
# nothing about whether the product is real, and penalising it here would
|
||||
# reject every row in a perfectly good upload.
|
||||
if not images and images_checked:
|
||||
report.issues.append(ValidationIssue(
|
||||
field="image_urls", severity="warning",
|
||||
message="no images were found for this product", penalty=0.15,
|
||||
))
|
||||
score -= 0.15
|
||||
|
||||
# 8. Positive evidence --------------------------------------------------
|
||||
if category_resolved_deterministically:
|
||||
score += 0.15
|
||||
if grounded:
|
||||
score += 0.15
|
||||
if images:
|
||||
score += 0.05
|
||||
|
||||
score = max(0.0, min(1.0, score))
|
||||
report.confidence = round(score, 3)
|
||||
|
||||
if score < cfg.reject_below:
|
||||
report.status = "rejected"
|
||||
elif score < cfg.review_below:
|
||||
report.status = "needs_review"
|
||||
else:
|
||||
report.status = "verified"
|
||||
|
||||
return report
|
||||
|
||||
|
||||
def validate_catalog(
|
||||
products: list[dict[str, Any]],
|
||||
brand: str,
|
||||
*,
|
||||
known_category_flags: Optional[list[bool]] = None,
|
||||
grounded_flags: Optional[list[bool]] = None,
|
||||
images_checked: bool = True,
|
||||
config: Optional[ValidationConfig] = None,
|
||||
) -> tuple[list[dict[str, Any]], list[dict[str, Any]], dict[str, Any]]:
|
||||
"""Validate an entire list of assembled catalog rows in one pass.
|
||||
|
||||
Returns (kept_products, rejected_products, summary) where `kept_products`
|
||||
includes both "verified" and "needs_review" rows (each annotated with
|
||||
`validation_status` / `confidence_score` / `validation_issues` so a
|
||||
human-in-the-loop reviewer or a downstream consumer can filter further),
|
||||
and `rejected_products` holds the rows dropped from the catalog along
|
||||
with why, for audit/QA purposes - see docs/VALIDATION_PIPELINE.md.
|
||||
"""
|
||||
cfg = config or ValidationConfig()
|
||||
known_category_flags = known_category_flags or [False] * len(products)
|
||||
grounded_flags = grounded_flags or [False] * len(products)
|
||||
|
||||
kept: list[dict[str, Any]] = []
|
||||
rejected: list[dict[str, Any]] = []
|
||||
status_counts = {"verified": 0, "needs_review": 0, "rejected": 0}
|
||||
|
||||
for i, p in enumerate(products):
|
||||
det = known_category_flags[i] if i < len(known_category_flags) else False
|
||||
grd = grounded_flags[i] if i < len(grounded_flags) else False
|
||||
report = validate_product(
|
||||
p, brand,
|
||||
category_resolved_deterministically=det,
|
||||
grounded=grd,
|
||||
images_checked=images_checked,
|
||||
config=cfg,
|
||||
)
|
||||
status_counts[report.status] = status_counts.get(report.status, 0) + 1
|
||||
|
||||
annotated = {
|
||||
**p,
|
||||
"validation_status": report.status,
|
||||
"confidence_score": report.confidence,
|
||||
"validation_issues": [i.message for i in report.issues],
|
||||
}
|
||||
|
||||
if report.is_rejected:
|
||||
logger.warning(f"🚫 Rejected (validation): {report.summary()}")
|
||||
rejected.append(annotated)
|
||||
else:
|
||||
if report.needs_review:
|
||||
logger.info(f"⚠️ Needs review: {report.summary()}")
|
||||
else:
|
||||
logger.info(f"✅ Verified: {report.product_name} (confidence={report.confidence:.2f})")
|
||||
kept.append(annotated)
|
||||
|
||||
summary = {
|
||||
"total_evaluated": len(products),
|
||||
"verified": status_counts["verified"],
|
||||
"needs_review": status_counts["needs_review"],
|
||||
"rejected": status_counts["rejected"],
|
||||
"reject_threshold": cfg.reject_below,
|
||||
"review_threshold": cfg.review_below,
|
||||
}
|
||||
logger.info(
|
||||
f"🔍 Validation summary for '{brand}': {summary['verified']} verified, "
|
||||
f"{summary['needs_review']} needs_review, {summary['rejected']} rejected "
|
||||
f"(of {summary['total_evaluated']} rows)"
|
||||
)
|
||||
return kept, rejected, summary
|
||||
82
app/services/quantity_utils.py
Normal file
82
app/services/quantity_utils.py
Normal file
@@ -0,0 +1,82 @@
|
||||
"""
|
||||
Pack-size / quantity parsing helpers.
|
||||
|
||||
Ported from the sibling `Universal_Catalog_Barcode_Enrichment` project, where
|
||||
these lived inside `image_search.py`. They are a standalone module here on
|
||||
purpose: the barcode matcher needs them, and this repo's `image_search.py` is
|
||||
a working file that the store-catalog work is not supposed to touch. Extracting
|
||||
rather than appending keeps that guarantee.
|
||||
|
||||
Deliberately self-contained (no import from price_estimator.py) - these are
|
||||
only used for approximate "same pack size?" comparisons, never for pricing
|
||||
math. Grams and millilitres are treated as interchangeable because this only
|
||||
ever compares two size LABELS for a plausible match, not physics.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import List, Optional
|
||||
|
||||
_QTY_RE = re.compile(
|
||||
r"(\d+(?:[.,]\d+)?)\s*[-_]?\s*"
|
||||
r"(kilograms?|kilos?|kgs?|grams?|gms?|litres?|liters?|ltrs?|lts?|mls?|ml|g|kg|l)\b"
|
||||
)
|
||||
_QTY_LITRE_UNITS = {"l", "lt", "lts", "ltr", "ltrs", "litre", "litres", "liter", "liters"}
|
||||
_QTY_KG_UNITS = {"kg", "kgs", "kilo", "kilos", "kilogram", "kilograms"}
|
||||
|
||||
|
||||
def parse_quantity_grams(text: Optional[str]) -> Optional[float]:
|
||||
"""Parse the first weight/volume mentioned in `text` (e.g. '200 g',
|
||||
'1kg', '500ml', a URL fragment like 'good-day-100-g') into a
|
||||
normalised grams-or-millilitres-equivalent float. Returns None if no
|
||||
recognisable quantity is found.
|
||||
"""
|
||||
if not text:
|
||||
return None
|
||||
match = _QTY_RE.search(str(text).lower())
|
||||
if not match:
|
||||
return None
|
||||
try:
|
||||
value = float(match.group(1).replace(",", "."))
|
||||
except ValueError:
|
||||
return None
|
||||
unit = match.group(2)
|
||||
if unit in _QTY_LITRE_UNITS or unit in _QTY_KG_UNITS:
|
||||
value *= 1000
|
||||
return value
|
||||
|
||||
|
||||
def quantities_match(a: Optional[str], b: Optional[str], tolerance: float = 0.15) -> bool:
|
||||
"""True if size labels `a` and `b` plausibly refer to the same pack size,
|
||||
allowing for rounding/formatting differences between sources (e.g. '200 g'
|
||||
vs '210g' vs '0.2 kg'). Used to match a requested pack-size variant against
|
||||
Open Food Facts' real per-barcode `quantity` field.
|
||||
"""
|
||||
qa, qb = parse_quantity_grams(a), parse_quantity_grams(b)
|
||||
if qa is None or qb is None or qa <= 0:
|
||||
return False
|
||||
return abs(qa - qb) / qa <= tolerance
|
||||
|
||||
|
||||
def find_quantity_mentions(text: Optional[str]) -> List[float]:
|
||||
"""Return every distinct weight/volume value mentioned anywhere in `text`
|
||||
(e.g. a full image/product URL), normalised to a grams-equivalent float.
|
||||
|
||||
Unlike parse_quantity_grams() (first match only), this scans the whole
|
||||
string - used to detect when a candidate image's URL names a DIFFERENT
|
||||
sibling pack size than the one being searched for (e.g. a "...-500-g..."
|
||||
URL turning up while resolving images for the 100g variant).
|
||||
"""
|
||||
if not text:
|
||||
return []
|
||||
out: List[float] = []
|
||||
for m in _QTY_RE.finditer(str(text).lower()):
|
||||
try:
|
||||
value = float(m.group(1).replace(",", "."))
|
||||
except ValueError:
|
||||
continue
|
||||
unit = m.group(2)
|
||||
if unit in _QTY_LITRE_UNITS or unit in _QTY_KG_UNITS:
|
||||
value *= 1000
|
||||
out.append(value)
|
||||
return out
|
||||
275
app/services/sku_service.py
Normal file
275
app/services/sku_service.py
Normal file
@@ -0,0 +1,275 @@
|
||||
"""
|
||||
Product SKU / website product-ID resolution service.
|
||||
|
||||
For every catalog record (one exact brand + product/variety + pack size)
|
||||
this module resolves a `product_sku` in two stages:
|
||||
|
||||
1. REAL MARKETPLACE ID (best-effort, network-dependent)
|
||||
A DuckDuckGo web search (via the `ddgs` package - already a project
|
||||
dependency, see app/services/image_search.py for the same pattern)
|
||||
restricted to known Indian e-commerce domains. If a result URL is on
|
||||
one of those domains, the real product ID is extracted directly from
|
||||
the URL itself (Amazon ASIN, Flipkart `pid`, BigBasket/JioMart/
|
||||
Blinkit/Nykaa numeric product IDs) - no page scraping/HTML parsing
|
||||
required, so this is fast and doesn't trip anti-bot defenses.
|
||||
|
||||
2. INTERNAL SKU (deterministic fallback)
|
||||
When no real product ID can be found (common - the LLM-discovered
|
||||
product may not exist verbatim on any marketplace, or the search
|
||||
genuinely turns up nothing usable), a human-readable internal SKU is
|
||||
generated instead, e.g. "AACHI-SAM-500-001" for
|
||||
"Aachi Sambar Powder" at size "500g":
|
||||
AACHI - brand code (first brand word, up to 6 chars)
|
||||
SAM - product code (first meaningful word of the title,
|
||||
first 3 letters; brand name and generic
|
||||
filler words like "Original"/"Classic"
|
||||
are skipped)
|
||||
500 - size code (leading numeric portion of the pack size)
|
||||
001 - sequence (persisted per BRAND-PRODUCT-SIZE prefix in
|
||||
brand_catalog_{brand}.json under a
|
||||
``_sku_sequences`` key so repeat pipeline
|
||||
runs hand out fresh, non-colliding numbers
|
||||
instead of restarting at 001 every time)
|
||||
|
||||
Both paths populate the same two catalog fields, so downstream code
|
||||
(JSON export, DB storage, UI) never needs to know which path was taken:
|
||||
product_sku - the ID/SKU string itself
|
||||
sku_source - where it came from, e.g. "Amazon (ASIN)",
|
||||
"Flipkart (PID)", or "Internal"
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import threading
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Optional, Tuple
|
||||
|
||||
from app.infrastructure.settings import DATA_DIR, ENABLE_SKU_WEB_LOOKUP
|
||||
from app.services.brand_registry import resolve_parent_brand
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_seq_lock = threading.Lock()
|
||||
|
||||
# Deviation from the sibling project, deliberate on both counts.
|
||||
#
|
||||
# 1. Absolute, not relative. The sibling used Path("data"), which resolves
|
||||
# against the current working directory - fine for a CLI run from the repo
|
||||
# root, wrong for a uvicorn process started from anywhere else.
|
||||
# 2. Its own directory, not the brand catalog files. The sibling stored the
|
||||
# counter inside data/brand_catalog_<brand>.json. In this repo those files
|
||||
# are real seed catalogs under data/seed_catalogs/, owned by brand_sync;
|
||||
# export_brand_to_seed_file() rewrites them wholesale, which would silently
|
||||
# drop the counters and start handing out duplicate SKUs. Keeping the
|
||||
# counter in its own file leaves the seed catalogs untouched.
|
||||
_data_dir = DATA_DIR / "sku_sequences"
|
||||
|
||||
|
||||
def _brand_seq_file(brand: str) -> Path:
|
||||
storage_brand = resolve_parent_brand(brand)
|
||||
safe_brand = storage_brand.strip().replace(" & ", " and ").replace("&", "_and_").replace(" ", "_").replace("-", "_")
|
||||
return _data_dir / f"{safe_brand}.json"
|
||||
|
||||
# Words that shouldn't be used as the "meaningful" word when building the
|
||||
# product code portion of an internal SKU.
|
||||
_STOPWORDS = {"the", "and", "of", "with", "for", "in", "a", "an", "new", "pack", "combo", "pouch"}
|
||||
_GENERIC_FILLER = {"original", "regular", "classic", "plain"}
|
||||
|
||||
# Domain -> (regex to pull the ID out of the URL, human-readable source label).
|
||||
# Matching happens directly against the result URL - none of these require
|
||||
# fetching/parsing the actual product page.
|
||||
_MARKETPLACE_PATTERNS: list[Tuple[str, "re.Pattern[str]", str]] = [
|
||||
("amazon.in", re.compile(r"/dp/([A-Z0-9]{10})", re.I), "Amazon (ASIN)"),
|
||||
("amazon.com", re.compile(r"/dp/([A-Z0-9]{10})", re.I), "Amazon (ASIN)"),
|
||||
("flipkart.com", re.compile(r"[?&]pid=([A-Z0-9]+)", re.I), "Flipkart (PID)"),
|
||||
("bigbasket.com", re.compile(r"/pd/(\d+)/?"), "BigBasket (Product ID)"),
|
||||
("jiomart.com", re.compile(r"/p/[^/?]+/(\d+)"), "JioMart (Product ID)"),
|
||||
("blinkit.com", re.compile(r"/prn/[^/]+/prid/(\d+)"), "Blinkit (Product ID)"),
|
||||
("nykaa.com", re.compile(r"/p/(\d+)"), "Nykaa (Product ID)"),
|
||||
]
|
||||
|
||||
# Domains searched for a real product ID - kept in sync with the patterns above.
|
||||
_SEARCH_DOMAINS = [d for d, _pat, _label in _MARKETPLACE_PATTERNS]
|
||||
|
||||
|
||||
def _brand_code(brand: str) -> str:
|
||||
if not brand:
|
||||
return "GEN"
|
||||
first = re.split(r"\s+", brand.strip())[0]
|
||||
code = re.sub(r"[^A-Za-z0-9]", "", first).upper()
|
||||
return code[:6] or "GEN"
|
||||
|
||||
|
||||
def _product_code(title: str, brand: str = "") -> str:
|
||||
"""First meaningful word of the (brand-stripped) title -> first 3 letters.
|
||||
|
||||
Uses only the base product name, not any " - Variety" suffix added by
|
||||
the variety-explosion step, so different flavours of the same product
|
||||
share the same product code (they're differentiated by the sequence
|
||||
number instead).
|
||||
"""
|
||||
clean = title or ""
|
||||
if brand and clean.lower().startswith(brand.strip().lower()):
|
||||
clean = clean[len(brand.strip()):].strip()
|
||||
base = clean.split(" - ")[0]
|
||||
words = re.findall(r"[A-Za-z0-9]+", base)
|
||||
for w in words:
|
||||
wl = w.lower()
|
||||
if wl in _STOPWORDS or wl in _GENERIC_FILLER:
|
||||
continue
|
||||
code = re.sub(r"[^A-Za-z0-9]", "", w).upper()
|
||||
if code:
|
||||
return code[:3]
|
||||
return "PRD"
|
||||
|
||||
|
||||
def _size_code(size: str) -> str:
|
||||
if not size:
|
||||
return "STD"
|
||||
m = re.search(r"(\d+(?:\.\d+)?)", size)
|
||||
if not m:
|
||||
return "STD"
|
||||
return m.group(1).replace(".", "")
|
||||
|
||||
|
||||
def _atomic_write_json(path: Path, data: Dict[str, Any]) -> None:
|
||||
"""Write JSON to `path` atomically: write to a temp file in the same
|
||||
directory, flush+fsync it, then os.replace() it onto the destination.
|
||||
|
||||
os.replace() is atomic on both POSIX and Windows - the destination
|
||||
file is guaranteed to either be the OLD complete content or the NEW
|
||||
complete content, NEVER a half-written/truncated hybrid, no matter
|
||||
when the process is killed. This is what prevents an accidental
|
||||
termination (Ctrl+C, crash, killed terminal, power loss) from
|
||||
corrupting data/brand_catalog_{brand}.json - which previously used a
|
||||
plain write_text()/open(...,'w') that truncates the file in place
|
||||
before writing the new content, so a kill mid-write left a half-empty
|
||||
or invalid-JSON file that the *next* run would either silently reset
|
||||
(SKU counters restarting at 0, causing duplicate SKUs) or, worse,
|
||||
silently treat as "no existing catalog" and discard every previously
|
||||
saved product for that brand (see save_catalog_to_data_folder() in
|
||||
streamlit_app.py for the matching fix on the read/merge side).
|
||||
"""
|
||||
tmp_path = path.with_suffix(path.suffix + f".tmp{os.getpid()}")
|
||||
text = json.dumps(data, indent=2, ensure_ascii=False)
|
||||
with open(tmp_path, "w", encoding="utf-8") as f:
|
||||
f.write(text)
|
||||
f.flush()
|
||||
os.fsync(f.fileno())
|
||||
os.replace(tmp_path, path) # atomic on POSIX and Windows
|
||||
|
||||
|
||||
def _next_sequence(brand: str, prefix: str) -> int:
|
||||
"""File-backed counter stored inside the brand catalog JSON file
|
||||
(data/brand_catalog_{brand}.json) under a reserved ``_sku_sequences``
|
||||
key, so re-running the pipeline for the same brand keeps handing out
|
||||
fresh sequence numbers for a given BRAND-PRODUCT-SIZE prefix instead
|
||||
of restarting at 001.
|
||||
"""
|
||||
seq_file = _brand_seq_file(brand)
|
||||
with _seq_lock:
|
||||
data: Dict[str, Any] = {}
|
||||
counters: Dict[str, int] = {}
|
||||
if seq_file.exists():
|
||||
try:
|
||||
data = json.loads(seq_file.read_text(encoding="utf-8"))
|
||||
counters = data.get("_sku_sequences", {})
|
||||
except Exception as e:
|
||||
# The file exists but isn't valid JSON - almost always the
|
||||
# result of a previous run being killed mid-write. Preserve
|
||||
# the corrupt file under a .corrupt name instead of silently
|
||||
# overwriting it, so no evidence/data is lost, and log
|
||||
# loudly rather than quietly resetting counters to 0 (which
|
||||
# would otherwise produce duplicate/colliding SKUs).
|
||||
logger.warning(
|
||||
f"SKU sequence file {seq_file} is corrupted ({e}); "
|
||||
f"resetting counters for this brand. The unreadable "
|
||||
f"file has been preserved for inspection."
|
||||
)
|
||||
try:
|
||||
seq_file.rename(seq_file.with_suffix(seq_file.suffix + ".corrupt"))
|
||||
except Exception:
|
||||
pass
|
||||
data = {}
|
||||
counters = {}
|
||||
next_val = int(counters.get(prefix, 0)) + 1
|
||||
counters[prefix] = next_val
|
||||
data["_sku_sequences"] = counters
|
||||
try:
|
||||
_data_dir.mkdir(parents=True, exist_ok=True)
|
||||
_atomic_write_json(seq_file, data)
|
||||
except Exception as e:
|
||||
logger.warning(f"Could not persist SKU sequence counter: {e}")
|
||||
return next_val
|
||||
|
||||
|
||||
def generate_internal_sku(brand: str, product_title: str, size: str) -> str:
|
||||
"""Deterministic-format internal SKU, e.g. 'AACHI-SAM-500-001'."""
|
||||
prefix = f"{_brand_code(brand)}-{_product_code(product_title, brand)}-{_size_code(size)}"
|
||||
seq = _next_sequence(brand, prefix)
|
||||
return f"{prefix}-{seq:03d}"
|
||||
|
||||
|
||||
def find_website_product_id(brand: str, product_title: str, size: str) -> Optional[Tuple[str, str]]:
|
||||
"""Best-effort search for a real marketplace product ID.
|
||||
|
||||
Returns (product_id, source_label) or None. Never raises - any
|
||||
network/parsing failure just means "no real ID found", which the
|
||||
caller falls back to an internal SKU for.
|
||||
"""
|
||||
if not ENABLE_SKU_WEB_LOOKUP:
|
||||
return None
|
||||
|
||||
query = f"{brand} {product_title} {size}".strip()
|
||||
if not query:
|
||||
return None
|
||||
|
||||
try:
|
||||
from ddgs import DDGS
|
||||
except ImportError:
|
||||
logger.debug("SKU web lookup unavailable: 'ddgs' package not installed")
|
||||
return None
|
||||
|
||||
try:
|
||||
with DDGS(timeout=10) as ddgs:
|
||||
results = ddgs.text(query, region="in-en", safesearch="off", max_results=8)
|
||||
except Exception as e:
|
||||
logger.debug(f"SKU web lookup failed for '{query}': {e}")
|
||||
return None
|
||||
|
||||
for r in results or []:
|
||||
url = str(r.get("href") or r.get("url") or "")
|
||||
if not url:
|
||||
continue
|
||||
url_lower = url.lower()
|
||||
for domain, pattern, label in _MARKETPLACE_PATTERNS:
|
||||
if domain not in url_lower:
|
||||
continue
|
||||
m = pattern.search(url)
|
||||
if m:
|
||||
return m.group(1).upper(), label
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def resolve_product_sku(brand: str, product_title: str, size: str) -> Dict[str, str]:
|
||||
"""Main entry point used by the catalog engine for each exploded
|
||||
(brand, product/variety, pack size) catalog row. Tries a real
|
||||
marketplace product ID first, falls back to a deterministic internal
|
||||
SKU when none can be found.
|
||||
"""
|
||||
found = None
|
||||
try:
|
||||
found = find_website_product_id(brand, product_title, size)
|
||||
except Exception as e:
|
||||
logger.debug(f"SKU lookup error for '{brand} {product_title} {size}': {e}")
|
||||
|
||||
if found:
|
||||
product_id, source = found
|
||||
return {"product_sku": product_id, "sku_source": source}
|
||||
|
||||
internal_sku = generate_internal_sku(brand, product_title, size)
|
||||
return {"product_sku": internal_sku, "sku_source": "Internal"}
|
||||
276
app/services/title_validator.py
Normal file
276
app/services/title_validator.py
Normal file
@@ -0,0 +1,276 @@
|
||||
"""
|
||||
Title <-> Category consistency guard.
|
||||
|
||||
WHY THIS FILE EXISTS
|
||||
---------------------
|
||||
Reported bug: brand="ITC" produced a catalog row titled
|
||||
"ITC Engage Biscuit Box - Classic Salted". The CATEGORY came back correct
|
||||
("Fragrance & Deodorants") because app.services.brand_registry.
|
||||
get_sub_brand_category() matched the known ITC sub-brand "Engage" inside
|
||||
the title and returned its real, curated category. The IMAGES came back
|
||||
correct too, because image search is keyed off that same "Engage" match
|
||||
plus the (correct) category. But nothing ever checked whether the TITLE
|
||||
TEXT ITSELF was consistent with that resolved category - so the small
|
||||
local LLM's hallucinated "Biscuit Box" (mixed in because it was asked for
|
||||
ITC's "Biscuits & Cookies" category products and free-associated a real
|
||||
ITC sub-brand name into the wrong product line) sailed straight through
|
||||
into the stored catalog record unmodified.
|
||||
|
||||
This module is the missing check. It is intentionally centralised in one
|
||||
place rather than duplicated, and intentionally conservative: it only
|
||||
strips a word/phrase from a title when that word/phrase is a *strong,
|
||||
unambiguous* signal for a category OTHER than the one the product has
|
||||
already been resolved to (deterministically, via the sub-brand registry,
|
||||
or heuristically). Ambiguous words (e.g. "milk", which is valid across
|
||||
several categories) are only flagged when NONE of their valid categories
|
||||
match - never guessed away speculatively.
|
||||
|
||||
USAGE
|
||||
-----
|
||||
Call `validate_and_fix_title(title, category, sub_brand_hint=...)` right
|
||||
after a product's category has been resolved (deterministically or via
|
||||
heuristic) and BEFORE that title is used for image search, S3 storage,
|
||||
DB storage, or JSON export - see app/core/catalog_engine.py, which is the
|
||||
canonical caller. It returns a (possibly corrected) title plus metadata
|
||||
about what was changed, so the caller can log/audit every correction
|
||||
instead of it happening silently.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import logging
|
||||
from typing import Iterable, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Canonical word/phrase -> valid-categories map.
|
||||
#
|
||||
# Deliberately generous for ambiguous words (mapped to every category they
|
||||
# could legitimately belong to) so this NEVER strips a word just because a
|
||||
# product could conceivably be miscategorised - it only strips a word when
|
||||
# the product's *actual* resolved category isn't anywhere in that word's
|
||||
# allowed set, i.e. the word is a hard contradiction, not merely unusual.
|
||||
#
|
||||
# Multi-word phrases are matched first (longest-phrase-wins) so e.g. "body
|
||||
# wash" is recognised as a single Skin & Bath Care signal rather than being
|
||||
# torn apart into two separately-ambiguous words ("body", "wash").
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# NOTE ON WHAT IS DELIBERATELY *NOT* IN THIS MAP:
|
||||
# Ingredient / flavour-descriptor words - "milk", "butter", "cheese",
|
||||
# "chocolate", "honey", "cashew", "almond", "salt", etc. - are intentionally
|
||||
# EXCLUDED even though they're each strongly associated with one category
|
||||
# (Dairy, Chocolates, Dry Fruits & Nuts...). That's because they are also
|
||||
# extremely common, completely legitimate FLAVOUR/variant modifiers inside
|
||||
# OTHER categories' real product names - "Good Day Butter Cookies", "Milk
|
||||
# Bikis", "Cashew Cookies", "Chocolate Cake", "Honey Oats" are all real
|
||||
# products, not hallucinations. Flagging these words caused false-positive
|
||||
# corrections that damaged genuine titles, so this map only contains STRONG
|
||||
# signals: words that denote WHAT THE PRODUCT ITSELF IS (a head noun, not a
|
||||
# flavour), which essentially never legitimately appear in another
|
||||
# category's product name. This is what makes it safe to auto-strip them.
|
||||
CATEGORY_TYPE_WORDS: dict[str, set[str]] = {
|
||||
# --- Biscuits, bakery, snacks ---
|
||||
"biscuit": {"Biscuits & Cookies"}, "biscuits": {"Biscuits & Cookies"},
|
||||
"cookie": {"Biscuits & Cookies"}, "cookies": {"Biscuits & Cookies"},
|
||||
"wafer": {"Biscuits & Cookies"}, "wafers": {"Biscuits & Cookies"},
|
||||
"cracker": {"Crackers"}, "crackers": {"Crackers"}, "saltine": {"Crackers"},
|
||||
"rusk": {"Rusk"}, "rusks": {"Rusk"},
|
||||
"cake": {"Cakes & Muffins"}, "cakes": {"Cakes & Muffins"},
|
||||
"muffin": {"Cakes & Muffins"}, "muffins": {"Cakes & Muffins"},
|
||||
"bread": {"Bakery & Breads"}, "bun": {"Bakery & Breads"}, "buns": {"Bakery & Breads"}, "loaf": {"Bakery & Breads"},
|
||||
"snack": {"Snacks"}, "snacks": {"Snacks"}, "chips": {"Snacks"}, "namkeen": {"Snacks"},
|
||||
"appalam": {"Snacks"}, "appalams": {"Snacks"}, "papad": {"Snacks"}, "bhujia": {"Snacks"},
|
||||
"soan papdi": {"Sweets"}, "gulab jamun": {"Sweets"}, "rasgulla": {"Sweets"},
|
||||
# --- Staples, spices, grains ---
|
||||
"atta": {"Atta & Staples"}, "maida": {"Atta & Staples"}, "sooji": {"Atta & Staples"},
|
||||
"besan": {"Atta & Staples"}, "basmati rice": {"Rice & Pulses", "Atta & Staples"},
|
||||
"multigrain flour": {"Atta & Staples"},
|
||||
"masala": {"Spices & Masalas"}, "masalas": {"Spices & Masalas"},
|
||||
"curry powder": {"Spices & Masalas"}, "garam masala": {"Spices & Masalas"},
|
||||
"chilli powder": {"Spices & Masalas"}, "turmeric powder": {"Spices & Masalas"},
|
||||
"deggi mirch": {"Spices & Masalas"}, "tikhalal": {"Spices & Masalas"}, "kashmiri lal": {"Spices & Masalas"},
|
||||
"pasta": {"Pasta & Noodles"}, "vermicelli": {"Pasta & Noodles"}, "semiya": {"Pasta & Noodles"},
|
||||
"macaroni": {"Pasta & Noodles"}, "spaghetti": {"Pasta & Noodles"},
|
||||
"noodle": {"Pasta & Noodles", "Noodles & Instant Food"}, "noodles": {"Pasta & Noodles", "Noodles & Instant Food"},
|
||||
# --- Desserts (head-noun only, not ingredient words) ---
|
||||
"ice cream": {"Ice Cream"}, "icecream": {"Ice Cream"}, "kulfi": {"Ice Cream"}, "frozen dessert": {"Ice Cream"},
|
||||
"shrikhand": {"Dairy - Desserts"}, "basundi": {"Dairy - Desserts"}, "mishti doi": {"Dairy - Desserts"},
|
||||
# --- Oils ---
|
||||
"cooking oil": {"Cooking Oils"}, "sunflower oil": {"Cooking Oils"}, "groundnut oil": {"Cooking Oils"},
|
||||
"gingelly oil": {"Cooking Oils"}, "olive oil": {"Cooking Oils"}, "mustard oil": {"Cooking Oils"},
|
||||
"vegetable oil": {"Cooking Oils"}, "refined oil": {"Cooking Oils"}, "lamp oil": {"Household - Lamp Oil"},
|
||||
# --- Beverages ---
|
||||
"tea": {"Tea & Coffee", "Beverages", "Health Drinks"}, "coffee": {"Tea & Coffee", "Beverages"},
|
||||
"juice": {"Beverages"}, "juices": {"Beverages"}, "squash": {"Beverages"},
|
||||
"aamras": {"Beverages"}, "jaljeera": {"Beverages"},
|
||||
"lassi": {"Dairy", "Beverages"}, "buttermilk": {"Beverages"}, "chaas": {"Beverages"},
|
||||
"milkshake": {"Beverages"}, "milkshakes": {"Beverages"}, "cold coffee": {"Beverages"}, "smoothie": {"Beverages"},
|
||||
# --- Confectionery / spreads ---
|
||||
"candy": {"Candy & Confectionery"}, "toffee": {"Candy & Confectionery"},
|
||||
"lollipop": {"Candy & Confectionery"}, "eclair": {"Candy & Confectionery"}, "eclairs": {"Candy & Confectionery"},
|
||||
"pickle": {"Pickles & Chutneys"}, "pickles": {"Pickles & Chutneys"}, "achar": {"Pickles & Chutneys"},
|
||||
"chutney": {"Pickles & Chutneys"}, "murabba": {"Pickles & Chutneys"},
|
||||
"health mix": {"Health Foods"}, "sathumaavu": {"Health Foods"}, "ragi malt": {"Health Foods"},
|
||||
"millet": {"Health Foods"}, "millets": {"Health Foods"},
|
||||
# --- Health / malted drinks ---
|
||||
"horlicks": {"Health Drinks"}, "bournvita": {"Health Drinks"}, "complan": {"Health Drinks"},
|
||||
"maltova": {"Health Drinks"}, "protinex": {"Health Drinks"}, "malted drink": {"Health Drinks"},
|
||||
"glucon": {"Health Drinks"},
|
||||
# --- Hair care ---
|
||||
"shampoo": {"Hair Care"}, "conditioner": {"Hair Care"}, "hair oil": {"Hair Care"},
|
||||
"hair serum": {"Hair Care"}, "hair cream": {"Hair Care"}, "hair color": {"Hair Care"},
|
||||
"hair dye": {"Hair Care"}, "anti-dandruff": {"Hair Care"}, "hair fall": {"Hair Care"},
|
||||
# --- Bath / skin / beauty ---
|
||||
"soap": {"Bath Soap", "Skin & Bath Care", "Personal Care"},
|
||||
"body wash": {"Skin & Bath Care", "Personal Care"}, "shower gel": {"Skin & Bath Care"},
|
||||
"hand wash": {"Skin & Bath Care"}, "hand cream": {"Skin & Bath Care"},
|
||||
"face wash": {"Skin Care", "Beauty Care"}, "face cream": {"Skin & Bath Care", "Skin Care"},
|
||||
"face powder": {"Beauty Care"}, "moisturiser": {"Skin Care", "Beauty Care"},
|
||||
"moisturizer": {"Skin Care", "Beauty Care"}, "sunscreen": {"Skin Care", "Beauty Care"}, "sunblock": {"Skin Care", "Beauty Care"},
|
||||
"foundation": {"Beauty Care"}, "lipstick": {"Beauty Care"}, "lip balm": {"Beauty Care"},
|
||||
"makeup": {"Beauty Care"}, "cosmetic": {"Beauty Care"}, "eyeliner": {"Beauty Care"},
|
||||
"kajal": {"Beauty Care"}, "mascara": {"Beauty Care"}, "nail polish": {"Beauty Care"},
|
||||
"fairness": {"Beauty Care"},
|
||||
# --- Oral care ---
|
||||
"toothpaste": {"Oral Care"}, "toothbrush": {"Oral Care"}, "mouthwash": {"Oral Care"}, "dental": {"Oral Care"},
|
||||
# --- Fragrance ---
|
||||
"deodorant": {"Fragrance & Deodorants"}, "deodorants": {"Fragrance & Deodorants"}, "deo": {"Fragrance & Deodorants"},
|
||||
"perfume": {"Fragrance & Deodorants"}, "perfumes": {"Fragrance & Deodorants"}, "fragrance": {"Fragrance & Deodorants"},
|
||||
"cologne": {"Fragrance & Deodorants"}, "body spray": {"Fragrance & Deodorants"}, "roll-on": {"Fragrance & Deodorants"},
|
||||
# --- Men's grooming ---
|
||||
"shaving cream": {"Men's Grooming"}, "shaving foam": {"Men's Grooming"}, "after shave": {"Men's Grooming"},
|
||||
"razor": {"Men's Grooming"}, "blade": {"Men's Grooming"}, "trimmer": {"Men's Grooming"}, "beard": {"Men's Grooming"},
|
||||
# --- Home care / laundry ---
|
||||
"detergent": {"Detergents & Fabric Care"}, "washing powder": {"Detergents & Fabric Care"},
|
||||
"fabric wash": {"Detergents & Fabric Care"}, "fabric care": {"Detergents & Fabric Care"},
|
||||
"dishwash": {"Household Cleaning"}, "floor cleaner": {"Household Cleaning"}, "toilet cleaner": {"Household Cleaning"},
|
||||
"fabric softener": {"Household Cleaning"}, "stain remover": {"Household Cleaning"}, "handwash": {"Household Cleaning"},
|
||||
"mosquito repellent": {"Mosquito Repellent"}, "air freshener": {"Air Freshener"},
|
||||
# --- Baby care ---
|
||||
"baby powder": {"Baby Care"}, "baby lotion": {"Baby Care"}, "baby oil": {"Baby Care"}, "baby shampoo": {"Baby Care"},
|
||||
}
|
||||
|
||||
# Sort once, longest-phrase-first, so multi-word phrases are matched
|
||||
# before their component single words could be (see module docstring).
|
||||
_SORTED_TYPE_WORDS = sorted(CATEGORY_TYPE_WORDS.keys(), key=len, reverse=True)
|
||||
|
||||
# Connector/filler tokens that shouldn't be left dangling after a
|
||||
# conflicting word is stripped out (e.g. "Engage - Classic Salted" is
|
||||
# fine; "Engage - -" or "Engage with" is not).
|
||||
_DANGLING_TOKENS = {"-", "&", "and", "with", "for", "the", "a", "an", "of"}
|
||||
|
||||
|
||||
def find_category_conflicts(title: str, category: str) -> list[str]:
|
||||
"""Return the list of phrases in `title` that are strong signals for a
|
||||
category OTHER than `category`. Empty list = no detected conflict
|
||||
(does not mean the title is *proven* correct, just that nothing in it
|
||||
contradicts the given category)."""
|
||||
if not title or not category:
|
||||
return []
|
||||
cat = category.strip()
|
||||
text = " " + title.lower() + " "
|
||||
conflicts: list[str] = []
|
||||
consumed = set()
|
||||
for phrase in _SORTED_TYPE_WORDS:
|
||||
pattern = r"(?<![a-z0-9])" + re.escape(phrase) + r"(?![a-z0-9])"
|
||||
m = re.search(pattern, text)
|
||||
if not m:
|
||||
continue
|
||||
span = (m.start(), m.end())
|
||||
if any(not (span[1] <= s or span[0] >= e) for s, e in consumed):
|
||||
continue # already covered by a longer phrase match
|
||||
allowed = CATEGORY_TYPE_WORDS[phrase]
|
||||
if cat not in allowed:
|
||||
conflicts.append(phrase)
|
||||
consumed.add(span)
|
||||
return conflicts
|
||||
|
||||
|
||||
def sanitize_title_for_category(title: str, category: str) -> str:
|
||||
"""Strip any word/phrase from `title` that contradicts `category`,
|
||||
then tidy up leftover punctuation/connector words. Pure text
|
||||
transform - does not know about brands or sub-brands."""
|
||||
if not title or not category:
|
||||
return title
|
||||
cat = category.strip()
|
||||
result = title
|
||||
for phrase in _SORTED_TYPE_WORDS:
|
||||
allowed = CATEGORY_TYPE_WORDS[phrase]
|
||||
if cat in allowed:
|
||||
continue
|
||||
pattern = re.compile(r"(?<![a-zA-Z0-9])" + re.escape(phrase) + r"(?![a-zA-Z0-9])", re.IGNORECASE)
|
||||
result = pattern.sub(" ", result)
|
||||
|
||||
# Tidy: collapse whitespace, strip dangling connector tokens/punctuation
|
||||
# left behind at the edges or next to now-empty gaps.
|
||||
tokens = [t for t in re.split(r"\s+", result.strip()) if t]
|
||||
cleaned_tokens: list[str] = []
|
||||
for t in tokens:
|
||||
bare = t.strip(".,!?;:'\"()").lower()
|
||||
if bare in _DANGLING_TOKENS and (not cleaned_tokens or cleaned_tokens[-1].strip(".,!?;:'\"()").lower() in _DANGLING_TOKENS):
|
||||
continue
|
||||
cleaned_tokens.append(t)
|
||||
while cleaned_tokens and cleaned_tokens[-1].strip(".,!?;:'\"()").lower() in _DANGLING_TOKENS:
|
||||
cleaned_tokens.pop()
|
||||
while cleaned_tokens and cleaned_tokens[0].strip(".,!?;:'\"()").lower() in _DANGLING_TOKENS:
|
||||
cleaned_tokens.pop(0)
|
||||
|
||||
cleaned = " ".join(cleaned_tokens).strip()
|
||||
return cleaned
|
||||
|
||||
|
||||
def validate_and_fix_title(
|
||||
title: str,
|
||||
category: Optional[str],
|
||||
sub_brand_hint: Optional[str] = None,
|
||||
brand: Optional[str] = None,
|
||||
) -> tuple[str, bool, list[str]]:
|
||||
"""Authoritative title/category consistency check + repair.
|
||||
|
||||
Returns (fixed_title, was_changed, removed_phrases).
|
||||
|
||||
`category` should be the ALREADY-RESOLVED, trusted category for this
|
||||
product (e.g. from brand_registry.get_sub_brand_category(), or the
|
||||
pipeline's heuristic classification) - this function treats it as
|
||||
ground truth and corrects the title to match it, never the reverse.
|
||||
|
||||
`sub_brand_hint` (e.g. "engage") is used ONLY as a last-resort anchor
|
||||
if stripping conflicting words leaves nothing usable behind - it is
|
||||
never invented, only ever a value the caller already resolved
|
||||
deterministically (e.g. via brand_registry.get_sub_brand_match()).
|
||||
"""
|
||||
if not title or not category or category in ("Uncategorized", "General"):
|
||||
return title, False, []
|
||||
|
||||
conflicts = find_category_conflicts(title, category)
|
||||
if not conflicts:
|
||||
return title, False, []
|
||||
|
||||
fixed = sanitize_title_for_category(title, category)
|
||||
|
||||
# If sanitizing collapsed the title to nothing (or to something too
|
||||
# short/generic to stand alone, e.g. just leftover flavour words with
|
||||
# no anchor), fall back to the known sub-brand or brand name so we
|
||||
# never emit an empty/blank title.
|
||||
if not fixed or len(fixed.strip()) < 2:
|
||||
anchor = (sub_brand_hint or brand or "").strip()
|
||||
# Preserve a trailing " - Variant" suffix from the original title
|
||||
# if present (flavour/variant naming is independent of the
|
||||
# category-word bug this function targets).
|
||||
variant_suffix = ""
|
||||
m = re.search(r"\s-\s(.+)$", title.strip())
|
||||
if m and not any(w in m.group(1).lower() for w in conflicts):
|
||||
variant_suffix = f" - {m.group(1).strip()}"
|
||||
fixed = (anchor.title() if anchor else category) + variant_suffix
|
||||
fixed = fixed.strip()
|
||||
|
||||
changed = fixed.strip().lower() != title.strip().lower()
|
||||
if changed:
|
||||
logger.warning(
|
||||
f"⚠️ Title/category mismatch corrected: '{title}' -> '{fixed}' "
|
||||
f"(category='{category}', removed={conflicts})"
|
||||
)
|
||||
return fixed, changed, conflicts
|
||||
5
data/sku_sequences/Aachi.json
Normal file
5
data/sku_sequences/Aachi.json
Normal file
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"_sku_sequences": {
|
||||
"AACHI-SAM-100": 1
|
||||
}
|
||||
}
|
||||
9
data/sku_sequences/britannia.json
Normal file
9
data/sku_sequences/britannia.json
Normal file
@@ -0,0 +1,9 @@
|
||||
{
|
||||
"_sku_sequences": {
|
||||
"BRITAN-50-200": 11,
|
||||
"BRITAN-50-100": 4,
|
||||
"BRITAN-50-250": 2,
|
||||
"BRITAN-50-500": 3,
|
||||
"BRITAN-GOO-200": 3
|
||||
}
|
||||
}
|
||||
5
data/sku_sequences/cadbury.json
Normal file
5
data/sku_sequences/cadbury.json
Normal file
@@ -0,0 +1,5 @@
|
||||
{
|
||||
"_sku_sequences": {
|
||||
"CADBUR-DAI-100": 1
|
||||
}
|
||||
}
|
||||
283
tests/test_store_catalog_pipeline.py
Normal file
283
tests/test_store_catalog_pipeline.py
Normal file
@@ -0,0 +1,283 @@
|
||||
"""Tests for the store-spreadsheet -> 11-stage pipeline -> brand table flow.
|
||||
|
||||
Follows the pattern established by test_user_products_upload.py: monkeypatch
|
||||
the storage/network boundary *on the pipeline module object* (it imports those
|
||||
names directly), build real .xlsx fixtures in memory, and assert on what was
|
||||
captured rather than on the HTTP status alone.
|
||||
|
||||
Every test runs with use_llm=False and fetch_images=False so nothing here
|
||||
touches Ollama or the open web.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
|
||||
import pytest
|
||||
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
|
||||
openpyxl = pytest.importorskip("openpyxl")
|
||||
|
||||
|
||||
def _sheet(headers, rows) -> bytes:
|
||||
wb = openpyxl.Workbook()
|
||||
ws = wb.active
|
||||
ws.append(headers)
|
||||
for row in rows:
|
||||
ws.append(row)
|
||||
buf = io.BytesIO()
|
||||
wb.save(buf)
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store(monkeypatch):
|
||||
"""A fake brand table. Returns the dict of image_id -> stored row."""
|
||||
table: dict = {}
|
||||
|
||||
def fake_upsert(brand, rows, cleanup=False):
|
||||
assert cleanup is False, "cleanup=True would delete the brand's existing catalog"
|
||||
for row in rows:
|
||||
table[row["image_id"]] = dict(row)
|
||||
return len(rows)
|
||||
|
||||
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
|
||||
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: list(table.values()))
|
||||
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
|
||||
return table
|
||||
|
||||
|
||||
def _run(content, **kw):
|
||||
return pipeline.run_pipeline("store.xlsx", content, use_llm=False, fetch_images=False, **kw)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Column mapping / brand resolution
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_messy_headers_map_onto_catalog_fields(store):
|
||||
content = _sheet(
|
||||
["Item Name", "Product Description", "Segment", "Net Weight", "Random Column"],
|
||||
[["Britannia 50-50", "Salty biscuit", "Biscuits", "200g", "junk"]],
|
||||
)
|
||||
result = _run(content)
|
||||
assert result.recognised_columns["product_name"] == "Item Name"
|
||||
assert result.recognised_columns["description"] == "Product Description"
|
||||
assert result.recognised_columns["category"] == "Segment"
|
||||
assert result.recognised_columns["size_variants"] == "Net Weight"
|
||||
assert result.unrecognised_columns == ["Random Column"]
|
||||
|
||||
|
||||
def test_brand_is_inferred_when_the_sheet_has_no_brand_column(store):
|
||||
"""Store files routinely omit the brand; it lives in the product name."""
|
||||
content = _sheet(["Item Name"], [["Britannia 50-50"], ["Cadbury Dairy Milk 100g"]])
|
||||
result = _run(content)
|
||||
assert result.errors == []
|
||||
assert result.brands == ["Britannia", "Cadbury"]
|
||||
|
||||
|
||||
def test_longest_alias_wins_when_inferring_a_brand():
|
||||
"""'cadbury dairy milk' must beat the shorter 'cadbury' substring."""
|
||||
assert pipeline.infer_brand("Cadbury Dairy Milk Silk 100g") == "cadbury dairy milk"
|
||||
|
||||
|
||||
def test_rows_land_in_the_brand_table_the_product_belongs_to(store):
|
||||
content = _sheet(
|
||||
["Item Name", "Net Weight"],
|
||||
[["Britannia 50-50", "200g"], ["Aachi Sambar Powder", "100g"]],
|
||||
)
|
||||
_run(content)
|
||||
assert sorted(store) == ["aachi_aachi_sambar_powder_100g", "britannia_britannia_50_50_200g"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Gap filling
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_missing_fields_are_filled_by_the_pipeline(store):
|
||||
"""A sheet with only a product name still produces a complete row."""
|
||||
content = _sheet(["Item Name"], [["Britannia Good Day Biscuits 200g"]])
|
||||
_run(content)
|
||||
|
||||
row = next(iter(store.values()))
|
||||
assert row["fssai_license"] == "10012022000103" # stage 1
|
||||
assert row["category"] # stage 3
|
||||
assert row["price_range"].startswith("₹") # stage 5
|
||||
assert row["product_sku"] # stage 7
|
||||
assert row["hsn_code"] # stage 9
|
||||
assert row["search_query"] # stage 11
|
||||
|
||||
|
||||
def test_values_supplied_by_the_store_are_never_overwritten(store):
|
||||
content = _sheet(
|
||||
["Item Name", "Net Weight", "MRP Range", "HSN Code"],
|
||||
[["Britannia 50-50", "200g", "₹111-222", "9999"]],
|
||||
)
|
||||
_run(content)
|
||||
row = next(iter(store.values()))
|
||||
assert row["price_range"] == "₹111-222"
|
||||
assert row["hsn_code"] == "9999"
|
||||
|
||||
|
||||
def test_an_uncategorised_row_gets_no_hsn_code(store):
|
||||
"""HSN/GST is a regulatory value keyed on category. When the category
|
||||
cannot be determined the column is deliberately left empty rather than
|
||||
guessed - a wrong tax code is worse than a missing one."""
|
||||
content = _sheet(["Item Name"], [["Britannia 50-50"]])
|
||||
result = _run(content)
|
||||
|
||||
row = next(iter(store.values()))
|
||||
assert row["category"] == "General"
|
||||
assert row["hsn_code"] is None
|
||||
assert any("category could not be detected" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_non_food_brands_get_no_fssai_licence(store):
|
||||
"""A miss means 'not a food brand', not an error - the column stays empty."""
|
||||
assert pipeline.get_fssai_license("Colgate") is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 4 - pack-size explosion and unit safety
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_one_row_explodes_into_one_row_per_pack_size(store):
|
||||
content = _sheet(["Item Name", "Pack Size"], [["Britannia 50-50", "100g, 200g, 500g"]])
|
||||
_run(content)
|
||||
assert len(store) == 3
|
||||
assert {r["size_variants"][0] for r in store.values()} == {"100g", "200g", "500g"}
|
||||
|
||||
|
||||
def test_each_pack_size_gets_its_own_image_id(store):
|
||||
"""Without the size in the id every variant collapses onto one row under
|
||||
ON CONFLICT (image_id)."""
|
||||
content = _sheet(["Item Name", "Pack Size"], [["Britannia 50-50", "100g, 200g"]])
|
||||
_run(content)
|
||||
assert len(set(store)) == 2
|
||||
|
||||
|
||||
def test_the_size_is_not_duplicated_when_already_in_the_product_name():
|
||||
assert pipeline.build_image_id("Cadbury", "Cadbury Dairy Milk 100g", "100g") == \
|
||||
"cadbury_cadbury_dairy_milk_100g"
|
||||
|
||||
|
||||
def test_a_pack_size_with_a_nonsense_unit_is_dropped_with_a_reason(store):
|
||||
"""A biscuit measured in centimetres is a data error, not a pack size."""
|
||||
content = _sheet(["Item Name", "Segment", "Pack Size"], [["Britannia 50-50", "Biscuits", "15cm"]])
|
||||
result = _run(content)
|
||||
assert any("15cm" in w for w in result.warnings)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 11 - storage semantics
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_reingesting_the_same_file_changes_nothing(store):
|
||||
content = _sheet(["Item Name", "Net Weight"], [["Britannia 50-50", "200g"]])
|
||||
first = _run(content)
|
||||
assert (first.inserted, first.skipped_existing) == (1, 0)
|
||||
|
||||
second = _run(content)
|
||||
assert (second.inserted, second.backfilled, second.skipped_existing) == (0, 0, 1)
|
||||
assert len(store) == 1
|
||||
|
||||
|
||||
def test_an_existing_row_has_only_its_blank_columns_backfilled(store):
|
||||
content = _sheet(["Item Name", "Net Weight"], [["Britannia Good Day Biscuits", "200g"]])
|
||||
_run(content)
|
||||
|
||||
key = next(iter(store))
|
||||
store[key]["hsn_code"] = None
|
||||
store[key]["price_range"] = "₹999-1000" # a value already held
|
||||
|
||||
result = _run(content)
|
||||
assert result.backfilled == 1
|
||||
assert store[key]["hsn_code"] == "1905" # blank -> filled
|
||||
assert store[key]["price_range"] == "₹999-1000" # held value untouched
|
||||
|
||||
|
||||
def test_a_storage_failure_is_reported_rather_than_counted_as_success(store, monkeypatch):
|
||||
def boom(brand, rows, cleanup=False):
|
||||
raise RuntimeError("pgvector is down")
|
||||
|
||||
monkeypatch.setattr(pipeline, "upsert_brand_products", boom)
|
||||
result = _run(_sheet(["Item Name"], [["Britannia 50-50 200g"]]))
|
||||
assert result.storage_error and "pgvector is down" in result.storage_error
|
||||
assert result.inserted == 0
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Row-level error handling
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_row_with_no_usable_product_name_is_reported_not_fatal(store):
|
||||
content = _sheet(["Item Name", "Net Weight"], [["", "200g"], ["Britannia 50-50", "200g"]])
|
||||
result = _run(content)
|
||||
assert len(result.errors) == 1
|
||||
assert result.errors[0].row == 2 # header is row 1
|
||||
assert result.inserted == 1 # the good row still landed
|
||||
|
||||
|
||||
def test_the_same_pack_listed_twice_is_written_once(store):
|
||||
content = _sheet(
|
||||
["Item Name", "Net Weight"],
|
||||
[["Britannia 50-50", "200g"], ["Britannia 50-50", "200g"]],
|
||||
)
|
||||
_run(content)
|
||||
assert len(store) == 1
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# HTTP surface
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_preview_reports_how_the_columns_were_understood(client, admin_headers):
|
||||
content = _sheet(["Item Name", "Segment", "Mystery"], [["Britannia 50-50", "Biscuits", "?"]])
|
||||
resp = client.post(
|
||||
"/api/admin/store-catalog/preview",
|
||||
files={"file": ("store.xlsx", content)},
|
||||
headers=admin_headers,
|
||||
)
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert body["recognised_columns"]["product_name"] == "Item Name"
|
||||
assert body["unrecognised_columns"] == ["Mystery"]
|
||||
assert body["brand_column_present"] is False
|
||||
assert len(body["stages"]) == 11
|
||||
|
||||
|
||||
def test_preview_requires_admin(client):
|
||||
content = _sheet(["Item Name"], [["Britannia 50-50"]])
|
||||
resp = client.post("/api/admin/store-catalog/preview", files={"file": ("s.xlsx", content)})
|
||||
assert resp.status_code == 401
|
||||
|
||||
|
||||
def test_an_unparseable_file_is_rejected_before_a_job_is_created(client, admin_headers):
|
||||
resp = client.post(
|
||||
"/api/admin/store-catalog/ingest",
|
||||
files={"file": ("notes.txt", b"this is not a spreadsheet")},
|
||||
headers=admin_headers,
|
||||
)
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_ingest_returns_a_job_id_that_can_be_polled(client, admin_headers, monkeypatch):
|
||||
# Keep the worker off the network and out of the database.
|
||||
monkeypatch.setattr(pipeline, "upsert_brand_products", lambda b, r, cleanup=False: len(r))
|
||||
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: [])
|
||||
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
|
||||
|
||||
content = _sheet(["Item Name", "Segment"], [["Britannia 50-50", "Biscuits"]])
|
||||
resp = client.post(
|
||||
"/api/admin/store-catalog/ingest",
|
||||
files={"file": ("store.xlsx", content)},
|
||||
headers=admin_headers,
|
||||
)
|
||||
assert resp.status_code == 202
|
||||
job_id = resp.json()["job_id"]
|
||||
assert resp.json()["rows_total"] == 1
|
||||
|
||||
poll = client.get(f"/api/admin/store-catalog/jobs/{job_id}", headers=admin_headers)
|
||||
assert poll.status_code == 200
|
||||
body = poll.json()
|
||||
assert body["status"] in {"pending", "running", "done", "failed"}
|
||||
assert body["total_stages"] == 11
|
||||
|
||||
|
||||
def test_polling_an_unknown_job_is_a_404(client, admin_headers):
|
||||
resp = client.get("/api/admin/store-catalog/jobs/does-not-exist", headers=admin_headers)
|
||||
assert resp.status_code == 404
|
||||
Reference in New Issue
Block a user