Dagster Orchestration

This commit is contained in:
sriram
2026-08-29 14:49:45 +05:30
parent 27d53fa957
commit 998df898db
28 changed files with 1455 additions and 76 deletions

View File

@@ -28,7 +28,7 @@ from dataclasses import dataclass
from typing import List, Optional, Tuple
from fastapi import HTTPException, UploadFile
from pydantic import BaseModel
from pydantic import BaseModel, Field
from app.api.batch_job_store import batch_job_store
from app.core import batch_ingest, batch_worker
@@ -182,6 +182,22 @@ def parse_all(read: List[Tuple[str, bytes]], limits: UploadLimits):
# ---------------------------------------------------------------------------
# Response shape
# ---------------------------------------------------------------------------
class StageOut(BaseModel):
"""One pipeline stage as it happened to one file.
Kept even in slim list responses: eleven of these per file is a few hundred
bytes, unlike the `products` manifest slim exists to drop.
"""
index: int
name: str
rows_done: int = 0
rows_total: int = 0
started_at: Optional[float] = None
# None while the stage is still running.
finished_at: Optional[float] = None
class BatchFileOut(BaseModel):
index: int
filename: str
@@ -196,7 +212,12 @@ class BatchFileOut(BaseModel):
total_stages: int = pipeline.TOTAL_STAGES
rows_done: int = 0
rows_total: int = 0
# The stages this file has been through, so a FINISHED file can still show
# its timeline. The scalars above only ever say where it is right now.
stages: List[StageOut] = Field(default_factory=list)
size_bytes: int = 0
started_at: Optional[float] = None
finished_at: Optional[float] = None
result: Optional[dict] = None
@@ -213,6 +234,14 @@ class BatchOut(BaseModel):
current_file: Optional[str] = None
use_llm: bool
fetch_images: bool
# Which executor owns this batch - "inprocess" or "dagster". A dagster batch
# sits queued until the orchestrator picks it up, which the UI has to be
# able to say out loud rather than showing a run that looks stuck.
runner: str = batch_ingest.RUNNER_INPROCESS
# The 11 stage names, in order, so a client can draw the whole pipeline
# before a file has entered any of it. Served rather than duplicated in the
# frontend so the two cannot drift when a stage is added.
stage_names: List[str] = Field(default_factory=lambda: list(pipeline.STAGE_NAMES))
totals: dict
brands: List[str]
files: List[BatchFileOut]
@@ -287,6 +316,49 @@ def stage_and_queue(
return manifest, True
def stage_for_orchestrator(
valid: List[Tuple[str, bytes, int]],
invalid: List[Tuple[str, str]],
*,
use_llm: bool,
fetch_images: bool,
submitted_by: Optional[str] = None,
) -> batch_ingest.BatchManifest:
"""Publish a batch for Dagster to claim, and hand it to no one else.
Same staging as `stage_and_queue`, minus the `batch_worker.submit`. The
batch is left `queued` and stamped `runner="dagster"`, which is what
`_pick_batch_id` and `batch_upload_sensor` filter on; the in-process worker
never scans for work, so leaving it unsubmitted is enough to keep the two
executors off each other's batches.
A third function rather than a `runner=` argument on `stage_and_queue`, for
the reason given in `stage_pending`: whether work starts here or somewhere
else is not the kind of decision that should hang off a boolean anyone can
flip later.
The batch sits queued until an orchestrator actually runs - which, if
`dagster dev` is not up, is never. Callers are expected to say so rather
than present it as a run in progress.
"""
manifest = batch_ingest.stage_batch(
[(name, contents) for name, contents, _n in valid],
use_llm=use_llm,
fetch_images=fetch_images,
invalid=invalid,
submitted_by=submitted_by,
)
for entry, (_name, _contents, rows) in zip(manifest.files, valid):
entry.rows_total = rows
manifest.runner = batch_ingest.RUNNER_DAGSTER
manifest.detail = (
"Waiting for the Dagster orchestrator to pick this batch up."
)
batch_ingest.write_manifest(manifest)
batch_job_store.put(manifest)
return manifest
def stage_pending(
valid: List[Tuple[str, bytes, int]],
invalid: List[Tuple[str, str]],

View File

@@ -202,7 +202,13 @@ def get_catalog_batch(batch_id: str) -> BatchOut:
@router.post("/batches/{batch_id}/resume", dependencies=[Depends(require_admin)])
def resume_catalog_batch(batch_id: str) -> BatchOut:
"""Re-queue a batch a restart cut short, or one that was queued behind a full queue."""
"""Re-queue a batch a restart cut short, or one that was queued behind a full queue.
Also the way out of a batch staged for Dagster that no orchestrator ever
came for - the "run it here instead" button. Because this hands the batch to
THIS container's worker, it also takes ownership: the runner is flipped to
`inprocess` so Dagster will not claim a batch that is already running here.
"""
manifest = batch_ingest.read_manifest(batch_id)
if not manifest:
raise HTTPException(status_code=404, detail="Batch not found")
@@ -217,6 +223,7 @@ def resume_catalog_batch(batch_id: str) -> BatchOut:
batch_job_store.clear_cancel(batch_id)
manifest.status = batch_ingest.QUEUED
manifest.detail = None
manifest.runner = batch_ingest.RUNNER_INPROCESS
batch_ingest.write_manifest(manifest)
batch_job_store.put(manifest)
@@ -314,6 +321,10 @@ class InboxStartRequest(InboxSelection):
# belongs to the person who can see what the machine is already doing.
use_llm: bool = False
fetch_images: bool = False
# Who runs it. "inprocess" is this container's worker thread and is the
# default, so an existing client that never sends the field is unaffected.
# "dagster" stages the batch and leaves it for the orchestrator to claim.
runner: str = batch_ingest.RUNNER_INPROCESS
class InboxDismissOut(BaseModel):
@@ -405,6 +416,14 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
The originals are removed afterwards, so the same sheet cannot be started
twice from a stale checkbox in another tab.
"""
if request.runner not in batch_ingest.RUNNERS:
raise HTTPException(
status_code=400,
detail="Unknown runner {!r}. Expected one of: {}.".format(
request.runner, ", ".join(sorted(batch_ingest.RUNNERS))
),
)
grouped = _parse_file_ids(request.file_ids)
if not grouped:
raise HTTPException(status_code=400, detail="No files were selected.")
@@ -447,23 +466,35 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
unique_senders = sorted(set(senders))
submitted_by = ", ".join(unique_senders)[:120] if unique_senders else None
manifest, started = batch_common.stage_and_queue(
picked,
[],
use_llm=request.use_llm,
fetch_images=request.fetch_images,
submitted_by=submitted_by,
)
if not started:
raise HTTPException(
status_code=429,
detail=(
"Too many batches are already queued. These files have been "
"taken out of the inbox and saved as batch "
f"{manifest.batch_id} - press Resume on it once the current "
"batch finishes."
),
if request.runner == batch_ingest.RUNNER_DAGSTER:
# Staged and left alone: Dagster claims it on its next sensor tick, or
# from the Launchpad. Nothing here waits on that, and the batch is
# durable either way.
manifest = batch_common.stage_for_orchestrator(
picked,
[],
use_llm=request.use_llm,
fetch_images=request.fetch_images,
submitted_by=submitted_by,
)
else:
manifest, started = batch_common.stage_and_queue(
picked,
[],
use_llm=request.use_llm,
fetch_images=request.fetch_images,
submitted_by=submitted_by,
)
if not started:
raise HTTPException(
status_code=429,
detail=(
"Too many batches are already queued. These files have been "
"taken out of the inbox and saved as batch "
f"{manifest.batch_id} - press Resume on it once the current "
"batch finishes."
),
)
# Only now, once the bytes are safely staged under a new id. Retiring them
# first would lose the files outright if staging then failed.

View File

@@ -173,7 +173,13 @@ _KEYWORD_RULES: Tuple[Tuple[str, Tuple[str, ...]], ...] = (
("description", ("description", "desc", "detail")),
("category", ("category", "segment")),
("brand", ("brand", "manufacturer", "company")),
("size_variants", ("size", "pack", "weight", "volume", "net qty", "quantity")),
# "net qty" / "net quantity" is the Indian labelling term for a pack size.
# A bare "Quantity" column is not: in a store sheet it is how many units the
# shop has or is ordering, and mapping it here made a case-pack count of 72
# into the pack size, which then got appended to the product name. Worse,
# mapping is first-wins by column position, so a leading "Quantity" column
# also shut out the sheet's real "Pack Size" column.
("size_variants", ("size", "pack", "weight", "volume", "net qty", "net quantity")),
("providers", ("provider", "platform", "marketplace", "available at")),
("highlights", ("highlight", "feature", "benefit")),
("nutrients", ("nutrient", "nutrition")),
@@ -186,12 +192,38 @@ _KEYWORD_RULES: Tuple[Tuple[str, Tuple[str, ...]], ...] = (
_BLANK_VALUES = frozenset({"", "nan", "none", "null", "na", "n/a", "-", "--", "#n/a"})
# Fields whose value is a NAME, and which therefore must not be fed from a
# sheet's id/code column for the same concept. The keyword rules match on
# substring, so "categoryid" satisfies the "category" rule and a column of
# 1/2/3 ends up stored as the product's category. Only these two fields need
# the guard: "hsn code" and "barcode" ARE identifier fields and must keep
# matching their rules.
_NAME_ONLY_FIELDS = frozenset({"category", "brand"})
_IDENTIFIER_SUFFIXES = ("id", "ids", "code", "codes", "no", "num", "number")
def _is_identifier_header(normalized: str) -> bool:
"""True when a header names an id/code column rather than a name column,
covering both "categoryid" and "category id" (the normalizer strips the
underscore in "category_id" to a space, and nothing at all from
"categoryid")."""
return any(
normalized.endswith(suffix) and normalized[: -len(suffix)].strip()
for suffix in _IDENTIFIER_SUFFIXES
)
def _canonical_field(normalized: str) -> Optional[str]:
exact = _EXACT_HEADERS.get(normalized)
if exact:
return exact
for canonical, keywords in _KEYWORD_RULES:
if any(keyword in normalized for keyword in keywords):
if canonical in _NAME_ONLY_FIELDS and _is_identifier_header(normalized):
# An id column for a name field: claim nothing, so the real
# name column (if the sheet has one) is still free to match and
# the id is reported back under `unrecognised` in the preview.
return None
return canonical
return None
@@ -453,14 +485,24 @@ def _build_product_dict(req: AddProductRequest, brand_parent: str,
if s3_urls:
final_image_urls = list(s3_urls)
# 2. Inherit from brand sample
if not final_image_urls and sample_existing.get("image_urls"):
final_image_urls = list(sample_existing.get("image_urls"))
# 3. Canonical S3 fallback URL
# A product's images are NOT inheritable from its brand.
#
# This used to fall back to `sample_existing["image_urls"]` - the
# `_brand_sample()` row, i.e. one arbitrary product of the brand,
# resolved once and reused for every row in the upload. It copied that
# product's photographs verbatim onto every image-less sibling, which is
# why Marie Gold and Milk Bikis both shipped carrying four
# `britannia_..._good_day_cashew_cookies_200g/` URLs while their own
# image_ids were perfectly correct. Another product's photo is never a
# defensible default for this one, so there is no fallback here.
#
# Nor is a URL invented. The old canonical fallback guessed
# `https://nearledaily.s3.ap-south-1.amazonaws.com/...` while the
# configured bucket is DigitalOcean Spaces (see S3_ENDPOINT), so it
# produced a guaranteed 404 that merely looked like an image. Leaving
# the list empty lets ProductCard render its real "no image" state.
if not final_image_urls:
canonical_s3 = f"https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/{brand_slug}/{image_id}/image_000.jpg"
final_image_urls = [canonical_s3]
logger.info("No image found for '%s' (%s)", product_name, image_id)
primary_image_url = final_image_urls[0] if final_image_urls else None
search_text = f"{brand_parent} {product_name} {category} {description} {price_range}"

View File

@@ -92,6 +92,11 @@ RETIRED = "retired"
TERMINAL_BATCH_STATES = {DONE, FAILED, PARTIAL, CANCELLED, RETIRED}
# Who runs a batch. See `BatchManifest.runner`.
RUNNER_INPROCESS = "inprocess"
RUNNER_DAGSTER = "dagster"
RUNNERS = {RUNNER_INPROCESS, RUNNER_DAGSTER}
# `..`, separators and drive letters all stripped. UploadFile.filename is
# attacker-controlled in the general case, and it is used to build a path.
_UNSAFE = re.compile(r"[^A-Za-z0-9._-]+")
@@ -110,6 +115,31 @@ def _safe_name(filename: str) -> str:
return base[:120]
@dataclass
class StageRecord:
"""One of the 11 pipeline stages, as it happened to one file.
The scalar `stage_index`/`stage_name` fields below say where a file is *now*
and are overwritten on every tick, so once a file finishes there is no trace
of what it went through. This keeps that trace: a completed file can still
show its whole timeline, which is the point of the orchestration view.
One record per stage INDEX, not per callback. Stages 8-11 run once per brand
(`store_catalog_pipeline.run_pipeline` loops `for brand, rows in
by_brand.items()`), so a three-brand sheet reports 8,9,10,11 three times
over. Those fold into the same record - earliest start, latest finish,
largest row counts - so a file always has at most eleven of these however
many brands its rows land in.
"""
index: int # 1-based, matching STAGE_NAMES
name: str
rows_done: int = 0
rows_total: int = 0
started_at: Optional[float] = None
finished_at: Optional[float] = None
@dataclass
class BatchFile:
"""One spreadsheet inside a batch, and how far it got."""
@@ -129,6 +159,9 @@ class BatchFile:
total_stages: int = pipeline.TOTAL_STAGES
rows_done: int = 0
rows_total: int = 0
# The stages this file has entered so far, in the order it entered them.
# Empty while queued; eleven entries once the pipeline has run through.
stages: List[StageRecord] = field(default_factory=list)
result: Optional[Dict[str, Any]] = None
started_at: Optional[float] = None
finished_at: Optional[float] = None
@@ -149,6 +182,18 @@ class BatchManifest:
# batch an admin uploaded directly. Carried so the Batch tab can say where
# a run came from instead of leaving it to be guessed from filenames.
submitted_by: Optional[str] = None
# Which executor owns this batch: the API's worker thread, or Dagster.
#
# Both watch the same directory and both can run the same `run_batch`, and
# until this field existed nothing arbitrated between them - enabling
# `batch_upload_sensor` beside a running API meant both claimed every queued
# batch and ingested it twice. The worker only ever runs what is explicitly
# submitted to it, so the field is really a claim check for the Dagster
# side: `_pick_batch_id` and the sensor ignore anything not marked "dagster".
#
# Defaults to INPROCESS so every manifest written before this existed, and
# every batch an admin uploads directly, keeps behaving exactly as it did.
runner: str = RUNNER_INPROCESS
files: List[BatchFile] = field(default_factory=list)
# -- derived, recomputed rather than stored, so they cannot drift ---------
@@ -222,6 +267,7 @@ class BatchManifest:
"updated_at": self.updated_at,
"detail": self.detail,
"submitted_by": self.submitted_by,
"runner": self.runner,
"files_total": self.files_total,
"files_done": self.files_done,
"files_failed": self.files_failed,
@@ -234,10 +280,18 @@ class BatchManifest:
@classmethod
def from_dict(cls, raw: Dict[str, Any]) -> "BatchManifest":
allowed = set(BatchFile.__dataclass_fields__)
files = [
BatchFile(**{k: v for k, v in entry.items() if k in allowed})
for entry in (raw.get("files") or [])
]
stage_fields = set(StageRecord.__dataclass_fields__)
files = []
for entry in raw.get("files") or []:
fields_ = {k: v for k, v in entry.items() if k in allowed}
# `asdict` flattened these to plain dicts on the way out; rebuild
# them so callers get StageRecords whichever direction the manifest
# came from (live object, or re-read off disk after a restart).
fields_["stages"] = [
StageRecord(**{k: v for k, v in stage.items() if k in stage_fields})
for stage in (entry.get("stages") or [])
]
files.append(BatchFile(**fields_))
return cls(
batch_id=raw["batch_id"],
status=raw.get("status", QUEUED),
@@ -247,6 +301,7 @@ class BatchManifest:
updated_at=float(raw.get("updated_at") or time.time()),
detail=raw.get("detail"),
submitted_by=raw.get("submitted_by"),
runner=raw.get("runner") or RUNNER_INPROCESS,
files=files,
)
@@ -457,6 +512,53 @@ def _noop_change(manifest: "BatchManifest") -> None:
_PROGRESS_FLUSH_SECONDS = 5.0
def _record_stage(entry: "BatchFile", index: int, name: str,
done: int, total: int, now: float) -> None:
"""Fold one progress tick into `entry.stages`.
Keyed by stage INDEX rather than appended, because the pipeline visits
stages 8-11 once per brand in the sheet: a three-brand file reports
8,9,10,11 three times over, with `rows_total` reset to that brand's group
size each pass. Appending would produce twenty-three entries for eleven
stages and a UI that appears to run backwards. Folding keeps the first
`started_at`, extends `finished_at`, and takes the high-water mark of both
row counts, so the record reads as "this stage, across the whole file".
Entering a stage closes every record before it. That is deliberate rather
than closing only the immediately-previous one: stage 1 emits a single tick
and stages 8-11 interleave, so "everything with a lower index is done" is
the only rule that leaves no record permanently open.
"""
if index <= 0:
return
for stage in entry.stages:
if stage.index < index and stage.finished_at is None:
stage.finished_at = now
for stage in entry.stages:
if stage.index == index:
stage.rows_done = max(stage.rows_done, done)
stage.rows_total = max(stage.rows_total, total)
stage.finished_at = None if done < total else now
return
entry.stages.append(StageRecord(
index=index,
name=name,
rows_done=done,
rows_total=total,
started_at=now,
finished_at=now if total and done >= total else None,
))
def _close_stages(entry: "BatchFile", now: float) -> None:
"""Mark whatever is still open as finished, once the file itself is done."""
for stage in entry.stages:
if stage.finished_at is None:
stage.finished_at = now
def run_batch(
batch_id: str,
*,
@@ -495,6 +597,9 @@ def run_batch(
entry.started_at = time.time()
entry.stage_index = 0
entry.stage_name = ""
# A re-run of an interrupted file starts its timeline over rather than
# appending to the one from the attempt that died.
entry.stages = []
write_manifest(manifest)
on_change(manifest)
@@ -502,13 +607,14 @@ def run_batch(
def progress(stage_index: int, stage_name: str, done: int, total: int,
_entry: BatchFile = entry) -> None:
now = time.time()
_record_stage(_entry, stage_index, stage_name, done, total, now)
_entry.stage_index = stage_index
_entry.stage_name = stage_name
_entry.rows_done = done
_entry.rows_total = total
manifest.updated_at = time.time()
manifest.updated_at = now
on_change(manifest)
now = time.time()
if now - last_flush[0] >= _PROGRESS_FLUSH_SECONDS:
last_flush[0] = now
write_manifest(manifest)
@@ -548,6 +654,9 @@ def run_batch(
entry.detail = str(exc)
entry.finished_at = time.time()
# The last stage never sees a "next stage" tick to close it, and a file
# that raised leaves whichever stage it died in open.
_close_stages(entry, entry.finished_at)
write_manifest(manifest)
on_change(manifest)
@@ -586,6 +695,9 @@ def scan_interrupted() -> List[str]:
entry.stage_index = 0
entry.stage_name = ""
entry.rows_done = 0
# The timeline described an attempt that no longer counts; the
# re-run builds a fresh one.
entry.stages = []
entry.detail = "Interrupted by a restart; queued again."
manifest.status = INTERRUPTED
manifest.detail = "Interrupted by a restart. Press Resume to continue."

View File

@@ -8,6 +8,7 @@ been removed - see app/services/image_search.py and
app/services/playwright_image_fallback.py for details.
"""
import json
import re
import sys
from pathlib import Path
from typing import Dict, List, Any
@@ -376,10 +377,41 @@ class ProductCatalogEngine:
"""Select the best images from a list of URLs based on quality and relevance"""
if not image_urls:
return []
# Words that distinguish THIS product from its brand-mates. The brand
# itself is removed on purpose: every Britannia URL contains
# "britannia", so it separates nothing - what tells Marie Gold from Good
# Day is "marie"/"gold" vs "good"/"day". Pack sizes and short filler
# words are dropped for the same reason.
_brand_words = {w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) if w}
_distinctive: List[str] = []
for word in re.split(r"[^a-z0-9]+", (product_title or "").lower()):
if (
len(word) > 2
and word not in _brand_words
and word not in _distinctive
and not re.fullmatch(r"\d+(?:kg|g|gm|gms|ml|l|ltr|pcs|n)?", word)
):
_distinctive.append(word)
# Earlier words identify a product more strongly than later ones: a
# title runs brand -> sub-brand -> variant -> size, so in "Coca-Cola
# Sprite Lemon 750ml" the word that separates this product from its
# brand-mates is "sprite", and "lemon" is only a modifier. Without this
# weighting a combo listing that merely shares the modifier
# ("...coca-cola-750-ml-limca-soft-drink-lemon-lime...") ties with the
# real Sprite image and then wins on domain reputation.
_weights = {word: len(_distinctive) - i for i, word in enumerate(_distinctive)}
def _mentions_product(url_lower: str) -> int:
"""How strongly the URL names THIS product, not merely its brand."""
if not _distinctive:
return 1 # nothing to distinguish by; don't penalise anything
return sum(weight for word, weight in _weights.items() if word in url_lower)
# Filter and score images
scored_images = []
for url in image_urls:
if not url or not url.startswith('http'):
continue
@@ -476,14 +508,41 @@ class ProductCatalogEngine:
# Avoid problematic URLs
if any(bad in url_lower for bad in ['encrypted-tbn', 'googleusercontent', 'data:', 'placeholder']):
score -= 5
scored_images.append((score, url))
# Sort by score (highest first) and take top images
scored_images.sort(key=lambda x: x[0], reverse=True)
best_images = [url for score, url in scored_images[:max_images]]
logger.info(f"📸 Selected {len(best_images)} best images for {product_title}")
# Multi-product listings picture several products at once, so even
# when they name this one the photo is not of it alone.
if any(kw in url_lower for kw in ['combo', 'multipack', 'multi-pack', 'pack-of', 'packof', 'assorted', 'variety-pack']):
score -= 8
scored_images.append((_mentions_product(url_lower), score, url))
# Sort by (does the URL name this product, then score).
#
# Relevance has to be a GATE, not another additive term. Domain
# reputation is worth up to +20 here while a matching product word is
# worth +1, so a BigBasket photo of a Limca combo (15 + 2 + 2 + 2 = 21)
# outranked the correct Sprite image on a lesser domain (8 + 2 + 1 + 1 =
# 12) - and index 0 is what becomes the product's `image_url`. That is
# the "top image is not this product" bug. Ranking every URL that names
# the product above every URL that doesn't makes the primary image
# correct, while the existing scoring still orders each group.
#
# Non-matching URLs are kept as a tail rather than dropped: a product
# whose distinctive words never appear in any URL (common for
# CDN-hashed filenames) would otherwise end up with no images at all.
scored_images.sort(key=lambda x: (x[0], x[1]), reverse=True)
best_images = [url for _relevant, _score, url in scored_images[:max_images]]
matched = sum(1 for relevant, _s, _u in scored_images[:max_images] if relevant)
logger.info(
f"📸 Selected {len(best_images)} best images for {product_title} "
f"({matched} naming the product)"
)
if best_images and not matched:
logger.warning(
f"📸 No candidate image URL names {product_title!r} - the primary "
f"image may not be this product"
)
return best_images
def search_with_python(self, query: str, brand: str) -> List[str]:

View File

@@ -79,7 +79,7 @@ from app.services.category_registry import (
detect_category_from_text,
sanitize_category_language,
)
from app.services.category_units import fix_or_reject_size
from app.services.category_units import fix_or_reject_size, parse_unit
from app.services.embeddings_service import embed_texts
from app.services.enrichment.barcode.stage import BarcodeEnrichmentStage
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
@@ -274,10 +274,33 @@ def stage_2_row_intake(row: Dict[str, Any], *, use_llm: bool = True) -> Dict[str
# ---------------------------------------------------------------------------
# Stage 3 - Title & category consistency
# ---------------------------------------------------------------------------
def _is_category_code(category: Any) -> bool:
"""True when `category` is an opaque code rather than a category name.
A sheet with a `categoryid` column (values 1, 2, 3) has that column claimed
by the `category` keyword rule - "category" is a substring of "categoryid" -
so the id lands in the category field. Stored as-is it reaches the user
("Coca-Cola . 1" on the product card, "1" in the sidebar filter) and it
silently disables three downstream systems that are all keyed by category
NAME: the pack-size unit rulebook, HSN/GST enrichment, and category-scoped
search. A number is never a category name, so treat it as not supplied and
let the keyword detector resolve it from the product name instead.
"""
text = str(category or "").strip()
return bool(text) and text.replace(".", "", 1).isdigit()
def stage_3_title_category(row: Dict[str, Any]) -> Dict[str, Any]:
title = row.get("title") or row.get("product_name") or ""
category = row.get("category")
if _is_category_code(category):
row.setdefault("_notes", []).append(
f"category {str(category).strip()!r} is an id, not a name; detecting from the product name"
)
category = None
row["category"] = None
if _blank(category):
category = detect_category_from_text(f"{title} {row.get('description') or ''}")
row["_category_deterministic"] = bool(category)
@@ -311,10 +334,40 @@ def stage_3_title_category(row: Dict[str, Any]) -> Dict[str, Any]:
# ---------------------------------------------------------------------------
# Stage 4 - Pack-size explosion & unit safety
# ---------------------------------------------------------------------------
def _is_unitless_number(size: str) -> bool:
"""True for a bare quantity like "72" - a number carrying no unit.
A store sheet's `Quantity` / `Case Pack` / `Units Per Pack` column is claimed
by the `size_variants` keyword rule in `map_spreadsheet_columns`, so its value
arrives here dressed as a pack size. It is not one: a pack size needs a unit
to mean anything, and keeping the bare number costs twice over. It is
appended to the product name ("Coca-Cola 750ml" becomes "Coca-Cola 750ml 72"),
and it is folded into `image_id`, so the next upload with a different
quantity inserts a duplicate product instead of updating this one.
`fix_or_reject_size` deliberately passes bare numbers through - a number
without a unit contradicts no category - so the check has to happen here,
before the size ever reaches it.
Only a parsed number with no unit token qualifies. A word-only label
("Standard", "Family Pack") parses as (None, None) and is left alone.
"""
value, unit = parse_unit(size)
return value is not None and not unit
def _sizes_for(row: Dict[str, Any]) -> List[str]:
sizes = [str(s).strip() for s in (row.get("size_variants") or []) if str(s).strip()]
declared = [str(s).strip() for s in (row.get("size_variants") or []) if str(s).strip()]
sizes = [s for s in declared if not _is_unitless_number(s)]
for ignored in declared:
if _is_unitless_number(ignored):
row.setdefault("_notes", []).append(
f"ignored pack size {ignored!r}: a number with no unit is a quantity, not a size"
)
if sizes:
return sizes
# Every declared size was a bare quantity (or none were given). Fall through
# to the name, which for a store sheet usually carries the real size already.
match = _SIZE_IN_TITLE.search(row.get("product_name") or "")
if match:
return [match.group(0).strip()]
@@ -391,7 +444,14 @@ def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, An
row["image_urls"] = list(best)
row["image_url"] = best[0]
except Exception as exc: # noqa: BLE001 - an image is not worth the row
logger.debug("Image search skipped for %r: %s", row.get("product_name"), exc)
# `warning`, not `debug`: at the default log level a debug line is
# invisible, so a row that silently lost its images looked identical to
# one that never wanted any. The row still survives - an image is not
# worth failing it - but the operator gets told.
logger.warning("Image search failed for %r: %s", row.get("product_name"), exc)
row.setdefault("_notes", []).append(f"image search failed: {exc}")
if _blank(row.get("image_urls")):
row.setdefault("_notes", []).append("no image found for this product")
return row
@@ -696,6 +756,13 @@ def run_pipeline(
progress(4, STAGE_NAMES[3], len(exploded), len(exploded))
result.products_built = len(exploded)
# Stage 4 copies the parent row into each variant, notes included, and the
# parent's notes have just been reported. Clear them so that the collection
# after stage 7 reports only what stages 5-7 add, once per variant, rather
# than repeating stages 1-4 once per pack size.
for row in exploded:
row["_notes"] = []
# ---- stages 5-7, per exploded row --------------------------------------
for index, row in enumerate(exploded, start=1):
stage_5_pricing(row)
@@ -707,6 +774,12 @@ def run_pipeline(
stage_7_sku(row)
progress(7, STAGE_NAMES[6], index, len(exploded))
# Whatever stages 5-7 recorded per variant - most usefully, that a product
# ended up with no image.
for row in exploded:
for note in row.get("_notes") or []:
result.warnings.append(f"row {row.get('_row')}: {note}")
# ---- stages 8-11, grouped by destination brand -------------------------
by_brand: Dict[str, List[Dict[str, Any]]] = {}
for row in exploded:

View File

@@ -53,7 +53,12 @@ from typing import Dict, List, Optional, Tuple
# this category when sanitizing cross-category language
# out of a generated description.
CATEGORY_REGISTRY: List[Dict[str, object]] = [
{"category": "Biscuits & Cookies", "keywords": ["biscuits", "biscuit", "biscit", "biskut", "cookies", "cookie"], "generic_term": "biscuit"},
# "bikis" is here so that "Britannia Milk Bikis" resolves as the biscuit it
# is. Without it the only keyword in that name is Dairy's "milk", which not
# only mislabels the product but hands stage 4 a volume unit rulebook - the
# exact "Britannia Milk Bikis - 200ml, 500ml, 1L" defect category_units.py
# was written to stop.
{"category": "Biscuits & Cookies", "keywords": ["biscuits", "biscuit", "biscit", "biskut", "cookies", "cookie", "bikis"], "generic_term": "biscuit"},
{"category": "Rusk", "keywords": ["rusks", "rusk"], "generic_term": "rusk"},
{"category": "Crackers", "keywords": ["crackers", "cracker", "saltine"], "generic_term": "cracker"},
{"category": "Cakes & Muffins", "keywords": ["cakes", "cake", "muffins", "muffin"], "generic_term": "bakery item"},
@@ -62,6 +67,13 @@ CATEGORY_REGISTRY: List[Dict[str, object]] = [
{"category": "Candy & Confectionery", "keywords": ["candy", "candies", "toffee", "toffees", "lollipop", "lollipops", "confectionery", "mints", "chewing gum"], "generic_term": "candy"},
{"category": "Snacks", "keywords": ["snacks", "snack", "chips", "namkeen", "wafers", "wafer", "kurkure", "lays"], "generic_term": "snack"},
{"category": "Chocolates", "keywords": ["chocolates", "chocolate", "chocate", "choclate", "cocoa", "cadbury chocolate", "dairy milk"], "generic_term": "chocolate"},
# Listed ahead of "Cooking Oils" so a drink is resolved before that entry's
# very broad bare "oil" keyword gets a chance, and ahead of "Dairy" only in
# keywords it does not share - "milk", "lassi" and "buttermilk" are
# deliberately left to Dairy. Kept free of "soda" (baking soda is a staple,
# not a drink) and of "tea"/"coffee" on their own (those are sold as leaves
# and grounds far more often than as a drink).
{"category": "Beverages", "keywords": ["beverages", "beverage", "soft drink", "soft drinks", "cold drink", "cold drinks", "carbonated", "aerated drink", "cola", "coke", "juice", "juices", "squash", "sharbat", "energy drink", "sports drink", "mineral water", "packaged drinking water", "lemonade", "iced tea", "thums up", "sprite", "fanta", "limca", "maaza", "pepsi", "mirinda"], "generic_term": "beverage"},
{"category": "Cooking Oils", "keywords": ["cooking oil", "edible oil", "sunflower oil", "mustard oil", "vanaspati", "refined oil", "oil", "oils"], "generic_term": "cooking oil"},
{"category": "Atta & Staples", "keywords": ["atta", "wheat flour", "flour", "rice", "dal", "pulses", "staples", "suji", "maida"], "generic_term": "staple product"},
{"category": "Dairy", "keywords": ["milk", "dairy", "cheese", "paneer", "panner", "paner", "paneerr", "curd", "yogurt", "butter", "ghee", "dahi"], "generic_term": "dairy product"},