Dagster Orchestration
This commit is contained in:
@@ -28,7 +28,7 @@ from dataclasses import dataclass
|
||||
from typing import List, Optional, Tuple
|
||||
|
||||
from fastapi import HTTPException, UploadFile
|
||||
from pydantic import BaseModel
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from app.api.batch_job_store import batch_job_store
|
||||
from app.core import batch_ingest, batch_worker
|
||||
@@ -182,6 +182,22 @@ def parse_all(read: List[Tuple[str, bytes]], limits: UploadLimits):
|
||||
# ---------------------------------------------------------------------------
|
||||
# Response shape
|
||||
# ---------------------------------------------------------------------------
|
||||
class StageOut(BaseModel):
|
||||
"""One pipeline stage as it happened to one file.
|
||||
|
||||
Kept even in slim list responses: eleven of these per file is a few hundred
|
||||
bytes, unlike the `products` manifest slim exists to drop.
|
||||
"""
|
||||
|
||||
index: int
|
||||
name: str
|
||||
rows_done: int = 0
|
||||
rows_total: int = 0
|
||||
started_at: Optional[float] = None
|
||||
# None while the stage is still running.
|
||||
finished_at: Optional[float] = None
|
||||
|
||||
|
||||
class BatchFileOut(BaseModel):
|
||||
index: int
|
||||
filename: str
|
||||
@@ -196,7 +212,12 @@ class BatchFileOut(BaseModel):
|
||||
total_stages: int = pipeline.TOTAL_STAGES
|
||||
rows_done: int = 0
|
||||
rows_total: int = 0
|
||||
# The stages this file has been through, so a FINISHED file can still show
|
||||
# its timeline. The scalars above only ever say where it is right now.
|
||||
stages: List[StageOut] = Field(default_factory=list)
|
||||
size_bytes: int = 0
|
||||
started_at: Optional[float] = None
|
||||
finished_at: Optional[float] = None
|
||||
result: Optional[dict] = None
|
||||
|
||||
|
||||
@@ -213,6 +234,14 @@ class BatchOut(BaseModel):
|
||||
current_file: Optional[str] = None
|
||||
use_llm: bool
|
||||
fetch_images: bool
|
||||
# Which executor owns this batch - "inprocess" or "dagster". A dagster batch
|
||||
# sits queued until the orchestrator picks it up, which the UI has to be
|
||||
# able to say out loud rather than showing a run that looks stuck.
|
||||
runner: str = batch_ingest.RUNNER_INPROCESS
|
||||
# The 11 stage names, in order, so a client can draw the whole pipeline
|
||||
# before a file has entered any of it. Served rather than duplicated in the
|
||||
# frontend so the two cannot drift when a stage is added.
|
||||
stage_names: List[str] = Field(default_factory=lambda: list(pipeline.STAGE_NAMES))
|
||||
totals: dict
|
||||
brands: List[str]
|
||||
files: List[BatchFileOut]
|
||||
@@ -287,6 +316,49 @@ def stage_and_queue(
|
||||
return manifest, True
|
||||
|
||||
|
||||
def stage_for_orchestrator(
|
||||
valid: List[Tuple[str, bytes, int]],
|
||||
invalid: List[Tuple[str, str]],
|
||||
*,
|
||||
use_llm: bool,
|
||||
fetch_images: bool,
|
||||
submitted_by: Optional[str] = None,
|
||||
) -> batch_ingest.BatchManifest:
|
||||
"""Publish a batch for Dagster to claim, and hand it to no one else.
|
||||
|
||||
Same staging as `stage_and_queue`, minus the `batch_worker.submit`. The
|
||||
batch is left `queued` and stamped `runner="dagster"`, which is what
|
||||
`_pick_batch_id` and `batch_upload_sensor` filter on; the in-process worker
|
||||
never scans for work, so leaving it unsubmitted is enough to keep the two
|
||||
executors off each other's batches.
|
||||
|
||||
A third function rather than a `runner=` argument on `stage_and_queue`, for
|
||||
the reason given in `stage_pending`: whether work starts here or somewhere
|
||||
else is not the kind of decision that should hang off a boolean anyone can
|
||||
flip later.
|
||||
|
||||
The batch sits queued until an orchestrator actually runs - which, if
|
||||
`dagster dev` is not up, is never. Callers are expected to say so rather
|
||||
than present it as a run in progress.
|
||||
"""
|
||||
manifest = batch_ingest.stage_batch(
|
||||
[(name, contents) for name, contents, _n in valid],
|
||||
use_llm=use_llm,
|
||||
fetch_images=fetch_images,
|
||||
invalid=invalid,
|
||||
submitted_by=submitted_by,
|
||||
)
|
||||
for entry, (_name, _contents, rows) in zip(manifest.files, valid):
|
||||
entry.rows_total = rows
|
||||
manifest.runner = batch_ingest.RUNNER_DAGSTER
|
||||
manifest.detail = (
|
||||
"Waiting for the Dagster orchestrator to pick this batch up."
|
||||
)
|
||||
batch_ingest.write_manifest(manifest)
|
||||
batch_job_store.put(manifest)
|
||||
return manifest
|
||||
|
||||
|
||||
def stage_pending(
|
||||
valid: List[Tuple[str, bytes, int]],
|
||||
invalid: List[Tuple[str, str]],
|
||||
|
||||
@@ -202,7 +202,13 @@ def get_catalog_batch(batch_id: str) -> BatchOut:
|
||||
|
||||
@router.post("/batches/{batch_id}/resume", dependencies=[Depends(require_admin)])
|
||||
def resume_catalog_batch(batch_id: str) -> BatchOut:
|
||||
"""Re-queue a batch a restart cut short, or one that was queued behind a full queue."""
|
||||
"""Re-queue a batch a restart cut short, or one that was queued behind a full queue.
|
||||
|
||||
Also the way out of a batch staged for Dagster that no orchestrator ever
|
||||
came for - the "run it here instead" button. Because this hands the batch to
|
||||
THIS container's worker, it also takes ownership: the runner is flipped to
|
||||
`inprocess` so Dagster will not claim a batch that is already running here.
|
||||
"""
|
||||
manifest = batch_ingest.read_manifest(batch_id)
|
||||
if not manifest:
|
||||
raise HTTPException(status_code=404, detail="Batch not found")
|
||||
@@ -217,6 +223,7 @@ def resume_catalog_batch(batch_id: str) -> BatchOut:
|
||||
batch_job_store.clear_cancel(batch_id)
|
||||
manifest.status = batch_ingest.QUEUED
|
||||
manifest.detail = None
|
||||
manifest.runner = batch_ingest.RUNNER_INPROCESS
|
||||
batch_ingest.write_manifest(manifest)
|
||||
batch_job_store.put(manifest)
|
||||
|
||||
@@ -314,6 +321,10 @@ class InboxStartRequest(InboxSelection):
|
||||
# belongs to the person who can see what the machine is already doing.
|
||||
use_llm: bool = False
|
||||
fetch_images: bool = False
|
||||
# Who runs it. "inprocess" is this container's worker thread and is the
|
||||
# default, so an existing client that never sends the field is unaffected.
|
||||
# "dagster" stages the batch and leaves it for the orchestrator to claim.
|
||||
runner: str = batch_ingest.RUNNER_INPROCESS
|
||||
|
||||
|
||||
class InboxDismissOut(BaseModel):
|
||||
@@ -405,6 +416,14 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
|
||||
The originals are removed afterwards, so the same sheet cannot be started
|
||||
twice from a stale checkbox in another tab.
|
||||
"""
|
||||
if request.runner not in batch_ingest.RUNNERS:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail="Unknown runner {!r}. Expected one of: {}.".format(
|
||||
request.runner, ", ".join(sorted(batch_ingest.RUNNERS))
|
||||
),
|
||||
)
|
||||
|
||||
grouped = _parse_file_ids(request.file_ids)
|
||||
if not grouped:
|
||||
raise HTTPException(status_code=400, detail="No files were selected.")
|
||||
@@ -447,23 +466,35 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
|
||||
unique_senders = sorted(set(senders))
|
||||
submitted_by = ", ".join(unique_senders)[:120] if unique_senders else None
|
||||
|
||||
manifest, started = batch_common.stage_and_queue(
|
||||
picked,
|
||||
[],
|
||||
use_llm=request.use_llm,
|
||||
fetch_images=request.fetch_images,
|
||||
submitted_by=submitted_by,
|
||||
)
|
||||
if not started:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail=(
|
||||
"Too many batches are already queued. These files have been "
|
||||
"taken out of the inbox and saved as batch "
|
||||
f"{manifest.batch_id} - press Resume on it once the current "
|
||||
"batch finishes."
|
||||
),
|
||||
if request.runner == batch_ingest.RUNNER_DAGSTER:
|
||||
# Staged and left alone: Dagster claims it on its next sensor tick, or
|
||||
# from the Launchpad. Nothing here waits on that, and the batch is
|
||||
# durable either way.
|
||||
manifest = batch_common.stage_for_orchestrator(
|
||||
picked,
|
||||
[],
|
||||
use_llm=request.use_llm,
|
||||
fetch_images=request.fetch_images,
|
||||
submitted_by=submitted_by,
|
||||
)
|
||||
else:
|
||||
manifest, started = batch_common.stage_and_queue(
|
||||
picked,
|
||||
[],
|
||||
use_llm=request.use_llm,
|
||||
fetch_images=request.fetch_images,
|
||||
submitted_by=submitted_by,
|
||||
)
|
||||
if not started:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail=(
|
||||
"Too many batches are already queued. These files have been "
|
||||
"taken out of the inbox and saved as batch "
|
||||
f"{manifest.batch_id} - press Resume on it once the current "
|
||||
"batch finishes."
|
||||
),
|
||||
)
|
||||
|
||||
# Only now, once the bytes are safely staged under a new id. Retiring them
|
||||
# first would lose the files outright if staging then failed.
|
||||
|
||||
@@ -173,7 +173,13 @@ _KEYWORD_RULES: Tuple[Tuple[str, Tuple[str, ...]], ...] = (
|
||||
("description", ("description", "desc", "detail")),
|
||||
("category", ("category", "segment")),
|
||||
("brand", ("brand", "manufacturer", "company")),
|
||||
("size_variants", ("size", "pack", "weight", "volume", "net qty", "quantity")),
|
||||
# "net qty" / "net quantity" is the Indian labelling term for a pack size.
|
||||
# A bare "Quantity" column is not: in a store sheet it is how many units the
|
||||
# shop has or is ordering, and mapping it here made a case-pack count of 72
|
||||
# into the pack size, which then got appended to the product name. Worse,
|
||||
# mapping is first-wins by column position, so a leading "Quantity" column
|
||||
# also shut out the sheet's real "Pack Size" column.
|
||||
("size_variants", ("size", "pack", "weight", "volume", "net qty", "net quantity")),
|
||||
("providers", ("provider", "platform", "marketplace", "available at")),
|
||||
("highlights", ("highlight", "feature", "benefit")),
|
||||
("nutrients", ("nutrient", "nutrition")),
|
||||
@@ -186,12 +192,38 @@ _KEYWORD_RULES: Tuple[Tuple[str, Tuple[str, ...]], ...] = (
|
||||
_BLANK_VALUES = frozenset({"", "nan", "none", "null", "na", "n/a", "-", "--", "#n/a"})
|
||||
|
||||
|
||||
# Fields whose value is a NAME, and which therefore must not be fed from a
|
||||
# sheet's id/code column for the same concept. The keyword rules match on
|
||||
# substring, so "categoryid" satisfies the "category" rule and a column of
|
||||
# 1/2/3 ends up stored as the product's category. Only these two fields need
|
||||
# the guard: "hsn code" and "barcode" ARE identifier fields and must keep
|
||||
# matching their rules.
|
||||
_NAME_ONLY_FIELDS = frozenset({"category", "brand"})
|
||||
_IDENTIFIER_SUFFIXES = ("id", "ids", "code", "codes", "no", "num", "number")
|
||||
|
||||
|
||||
def _is_identifier_header(normalized: str) -> bool:
|
||||
"""True when a header names an id/code column rather than a name column,
|
||||
covering both "categoryid" and "category id" (the normalizer strips the
|
||||
underscore in "category_id" to a space, and nothing at all from
|
||||
"categoryid")."""
|
||||
return any(
|
||||
normalized.endswith(suffix) and normalized[: -len(suffix)].strip()
|
||||
for suffix in _IDENTIFIER_SUFFIXES
|
||||
)
|
||||
|
||||
|
||||
def _canonical_field(normalized: str) -> Optional[str]:
|
||||
exact = _EXACT_HEADERS.get(normalized)
|
||||
if exact:
|
||||
return exact
|
||||
for canonical, keywords in _KEYWORD_RULES:
|
||||
if any(keyword in normalized for keyword in keywords):
|
||||
if canonical in _NAME_ONLY_FIELDS and _is_identifier_header(normalized):
|
||||
# An id column for a name field: claim nothing, so the real
|
||||
# name column (if the sheet has one) is still free to match and
|
||||
# the id is reported back under `unrecognised` in the preview.
|
||||
return None
|
||||
return canonical
|
||||
return None
|
||||
|
||||
@@ -453,14 +485,24 @@ def _build_product_dict(req: AddProductRequest, brand_parent: str,
|
||||
if s3_urls:
|
||||
final_image_urls = list(s3_urls)
|
||||
|
||||
# 2. Inherit from brand sample
|
||||
if not final_image_urls and sample_existing.get("image_urls"):
|
||||
final_image_urls = list(sample_existing.get("image_urls"))
|
||||
|
||||
# 3. Canonical S3 fallback URL
|
||||
# A product's images are NOT inheritable from its brand.
|
||||
#
|
||||
# This used to fall back to `sample_existing["image_urls"]` - the
|
||||
# `_brand_sample()` row, i.e. one arbitrary product of the brand,
|
||||
# resolved once and reused for every row in the upload. It copied that
|
||||
# product's photographs verbatim onto every image-less sibling, which is
|
||||
# why Marie Gold and Milk Bikis both shipped carrying four
|
||||
# `britannia_..._good_day_cashew_cookies_200g/` URLs while their own
|
||||
# image_ids were perfectly correct. Another product's photo is never a
|
||||
# defensible default for this one, so there is no fallback here.
|
||||
#
|
||||
# Nor is a URL invented. The old canonical fallback guessed
|
||||
# `https://nearledaily.s3.ap-south-1.amazonaws.com/...` while the
|
||||
# configured bucket is DigitalOcean Spaces (see S3_ENDPOINT), so it
|
||||
# produced a guaranteed 404 that merely looked like an image. Leaving
|
||||
# the list empty lets ProductCard render its real "no image" state.
|
||||
if not final_image_urls:
|
||||
canonical_s3 = f"https://nearledaily.s3.ap-south-1.amazonaws.com/daily/brands/{brand_slug}/{image_id}/image_000.jpg"
|
||||
final_image_urls = [canonical_s3]
|
||||
logger.info("No image found for '%s' (%s)", product_name, image_id)
|
||||
|
||||
primary_image_url = final_image_urls[0] if final_image_urls else None
|
||||
search_text = f"{brand_parent} {product_name} {category} {description} {price_range}"
|
||||
|
||||
@@ -92,6 +92,11 @@ RETIRED = "retired"
|
||||
|
||||
TERMINAL_BATCH_STATES = {DONE, FAILED, PARTIAL, CANCELLED, RETIRED}
|
||||
|
||||
# Who runs a batch. See `BatchManifest.runner`.
|
||||
RUNNER_INPROCESS = "inprocess"
|
||||
RUNNER_DAGSTER = "dagster"
|
||||
RUNNERS = {RUNNER_INPROCESS, RUNNER_DAGSTER}
|
||||
|
||||
# `..`, separators and drive letters all stripped. UploadFile.filename is
|
||||
# attacker-controlled in the general case, and it is used to build a path.
|
||||
_UNSAFE = re.compile(r"[^A-Za-z0-9._-]+")
|
||||
@@ -110,6 +115,31 @@ def _safe_name(filename: str) -> str:
|
||||
return base[:120]
|
||||
|
||||
|
||||
@dataclass
|
||||
class StageRecord:
|
||||
"""One of the 11 pipeline stages, as it happened to one file.
|
||||
|
||||
The scalar `stage_index`/`stage_name` fields below say where a file is *now*
|
||||
and are overwritten on every tick, so once a file finishes there is no trace
|
||||
of what it went through. This keeps that trace: a completed file can still
|
||||
show its whole timeline, which is the point of the orchestration view.
|
||||
|
||||
One record per stage INDEX, not per callback. Stages 8-11 run once per brand
|
||||
(`store_catalog_pipeline.run_pipeline` loops `for brand, rows in
|
||||
by_brand.items()`), so a three-brand sheet reports 8,9,10,11 three times
|
||||
over. Those fold into the same record - earliest start, latest finish,
|
||||
largest row counts - so a file always has at most eleven of these however
|
||||
many brands its rows land in.
|
||||
"""
|
||||
|
||||
index: int # 1-based, matching STAGE_NAMES
|
||||
name: str
|
||||
rows_done: int = 0
|
||||
rows_total: int = 0
|
||||
started_at: Optional[float] = None
|
||||
finished_at: Optional[float] = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class BatchFile:
|
||||
"""One spreadsheet inside a batch, and how far it got."""
|
||||
@@ -129,6 +159,9 @@ class BatchFile:
|
||||
total_stages: int = pipeline.TOTAL_STAGES
|
||||
rows_done: int = 0
|
||||
rows_total: int = 0
|
||||
# The stages this file has entered so far, in the order it entered them.
|
||||
# Empty while queued; eleven entries once the pipeline has run through.
|
||||
stages: List[StageRecord] = field(default_factory=list)
|
||||
result: Optional[Dict[str, Any]] = None
|
||||
started_at: Optional[float] = None
|
||||
finished_at: Optional[float] = None
|
||||
@@ -149,6 +182,18 @@ class BatchManifest:
|
||||
# batch an admin uploaded directly. Carried so the Batch tab can say where
|
||||
# a run came from instead of leaving it to be guessed from filenames.
|
||||
submitted_by: Optional[str] = None
|
||||
# Which executor owns this batch: the API's worker thread, or Dagster.
|
||||
#
|
||||
# Both watch the same directory and both can run the same `run_batch`, and
|
||||
# until this field existed nothing arbitrated between them - enabling
|
||||
# `batch_upload_sensor` beside a running API meant both claimed every queued
|
||||
# batch and ingested it twice. The worker only ever runs what is explicitly
|
||||
# submitted to it, so the field is really a claim check for the Dagster
|
||||
# side: `_pick_batch_id` and the sensor ignore anything not marked "dagster".
|
||||
#
|
||||
# Defaults to INPROCESS so every manifest written before this existed, and
|
||||
# every batch an admin uploads directly, keeps behaving exactly as it did.
|
||||
runner: str = RUNNER_INPROCESS
|
||||
files: List[BatchFile] = field(default_factory=list)
|
||||
|
||||
# -- derived, recomputed rather than stored, so they cannot drift ---------
|
||||
@@ -222,6 +267,7 @@ class BatchManifest:
|
||||
"updated_at": self.updated_at,
|
||||
"detail": self.detail,
|
||||
"submitted_by": self.submitted_by,
|
||||
"runner": self.runner,
|
||||
"files_total": self.files_total,
|
||||
"files_done": self.files_done,
|
||||
"files_failed": self.files_failed,
|
||||
@@ -234,10 +280,18 @@ class BatchManifest:
|
||||
@classmethod
|
||||
def from_dict(cls, raw: Dict[str, Any]) -> "BatchManifest":
|
||||
allowed = set(BatchFile.__dataclass_fields__)
|
||||
files = [
|
||||
BatchFile(**{k: v for k, v in entry.items() if k in allowed})
|
||||
for entry in (raw.get("files") or [])
|
||||
]
|
||||
stage_fields = set(StageRecord.__dataclass_fields__)
|
||||
files = []
|
||||
for entry in raw.get("files") or []:
|
||||
fields_ = {k: v for k, v in entry.items() if k in allowed}
|
||||
# `asdict` flattened these to plain dicts on the way out; rebuild
|
||||
# them so callers get StageRecords whichever direction the manifest
|
||||
# came from (live object, or re-read off disk after a restart).
|
||||
fields_["stages"] = [
|
||||
StageRecord(**{k: v for k, v in stage.items() if k in stage_fields})
|
||||
for stage in (entry.get("stages") or [])
|
||||
]
|
||||
files.append(BatchFile(**fields_))
|
||||
return cls(
|
||||
batch_id=raw["batch_id"],
|
||||
status=raw.get("status", QUEUED),
|
||||
@@ -247,6 +301,7 @@ class BatchManifest:
|
||||
updated_at=float(raw.get("updated_at") or time.time()),
|
||||
detail=raw.get("detail"),
|
||||
submitted_by=raw.get("submitted_by"),
|
||||
runner=raw.get("runner") or RUNNER_INPROCESS,
|
||||
files=files,
|
||||
)
|
||||
|
||||
@@ -457,6 +512,53 @@ def _noop_change(manifest: "BatchManifest") -> None:
|
||||
_PROGRESS_FLUSH_SECONDS = 5.0
|
||||
|
||||
|
||||
def _record_stage(entry: "BatchFile", index: int, name: str,
|
||||
done: int, total: int, now: float) -> None:
|
||||
"""Fold one progress tick into `entry.stages`.
|
||||
|
||||
Keyed by stage INDEX rather than appended, because the pipeline visits
|
||||
stages 8-11 once per brand in the sheet: a three-brand file reports
|
||||
8,9,10,11 three times over, with `rows_total` reset to that brand's group
|
||||
size each pass. Appending would produce twenty-three entries for eleven
|
||||
stages and a UI that appears to run backwards. Folding keeps the first
|
||||
`started_at`, extends `finished_at`, and takes the high-water mark of both
|
||||
row counts, so the record reads as "this stage, across the whole file".
|
||||
|
||||
Entering a stage closes every record before it. That is deliberate rather
|
||||
than closing only the immediately-previous one: stage 1 emits a single tick
|
||||
and stages 8-11 interleave, so "everything with a lower index is done" is
|
||||
the only rule that leaves no record permanently open.
|
||||
"""
|
||||
if index <= 0:
|
||||
return
|
||||
for stage in entry.stages:
|
||||
if stage.index < index and stage.finished_at is None:
|
||||
stage.finished_at = now
|
||||
|
||||
for stage in entry.stages:
|
||||
if stage.index == index:
|
||||
stage.rows_done = max(stage.rows_done, done)
|
||||
stage.rows_total = max(stage.rows_total, total)
|
||||
stage.finished_at = None if done < total else now
|
||||
return
|
||||
|
||||
entry.stages.append(StageRecord(
|
||||
index=index,
|
||||
name=name,
|
||||
rows_done=done,
|
||||
rows_total=total,
|
||||
started_at=now,
|
||||
finished_at=now if total and done >= total else None,
|
||||
))
|
||||
|
||||
|
||||
def _close_stages(entry: "BatchFile", now: float) -> None:
|
||||
"""Mark whatever is still open as finished, once the file itself is done."""
|
||||
for stage in entry.stages:
|
||||
if stage.finished_at is None:
|
||||
stage.finished_at = now
|
||||
|
||||
|
||||
def run_batch(
|
||||
batch_id: str,
|
||||
*,
|
||||
@@ -495,6 +597,9 @@ def run_batch(
|
||||
entry.started_at = time.time()
|
||||
entry.stage_index = 0
|
||||
entry.stage_name = ""
|
||||
# A re-run of an interrupted file starts its timeline over rather than
|
||||
# appending to the one from the attempt that died.
|
||||
entry.stages = []
|
||||
write_manifest(manifest)
|
||||
on_change(manifest)
|
||||
|
||||
@@ -502,13 +607,14 @@ def run_batch(
|
||||
|
||||
def progress(stage_index: int, stage_name: str, done: int, total: int,
|
||||
_entry: BatchFile = entry) -> None:
|
||||
now = time.time()
|
||||
_record_stage(_entry, stage_index, stage_name, done, total, now)
|
||||
_entry.stage_index = stage_index
|
||||
_entry.stage_name = stage_name
|
||||
_entry.rows_done = done
|
||||
_entry.rows_total = total
|
||||
manifest.updated_at = time.time()
|
||||
manifest.updated_at = now
|
||||
on_change(manifest)
|
||||
now = time.time()
|
||||
if now - last_flush[0] >= _PROGRESS_FLUSH_SECONDS:
|
||||
last_flush[0] = now
|
||||
write_manifest(manifest)
|
||||
@@ -548,6 +654,9 @@ def run_batch(
|
||||
entry.detail = str(exc)
|
||||
|
||||
entry.finished_at = time.time()
|
||||
# The last stage never sees a "next stage" tick to close it, and a file
|
||||
# that raised leaves whichever stage it died in open.
|
||||
_close_stages(entry, entry.finished_at)
|
||||
write_manifest(manifest)
|
||||
on_change(manifest)
|
||||
|
||||
@@ -586,6 +695,9 @@ def scan_interrupted() -> List[str]:
|
||||
entry.stage_index = 0
|
||||
entry.stage_name = ""
|
||||
entry.rows_done = 0
|
||||
# The timeline described an attempt that no longer counts; the
|
||||
# re-run builds a fresh one.
|
||||
entry.stages = []
|
||||
entry.detail = "Interrupted by a restart; queued again."
|
||||
manifest.status = INTERRUPTED
|
||||
manifest.detail = "Interrupted by a restart. Press Resume to continue."
|
||||
|
||||
@@ -8,6 +8,7 @@ been removed - see app/services/image_search.py and
|
||||
app/services/playwright_image_fallback.py for details.
|
||||
"""
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Dict, List, Any
|
||||
@@ -376,10 +377,41 @@ class ProductCatalogEngine:
|
||||
"""Select the best images from a list of URLs based on quality and relevance"""
|
||||
if not image_urls:
|
||||
return []
|
||||
|
||||
|
||||
# Words that distinguish THIS product from its brand-mates. The brand
|
||||
# itself is removed on purpose: every Britannia URL contains
|
||||
# "britannia", so it separates nothing - what tells Marie Gold from Good
|
||||
# Day is "marie"/"gold" vs "good"/"day". Pack sizes and short filler
|
||||
# words are dropped for the same reason.
|
||||
_brand_words = {w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) if w}
|
||||
_distinctive: List[str] = []
|
||||
for word in re.split(r"[^a-z0-9]+", (product_title or "").lower()):
|
||||
if (
|
||||
len(word) > 2
|
||||
and word not in _brand_words
|
||||
and word not in _distinctive
|
||||
and not re.fullmatch(r"\d+(?:kg|g|gm|gms|ml|l|ltr|pcs|n)?", word)
|
||||
):
|
||||
_distinctive.append(word)
|
||||
|
||||
# Earlier words identify a product more strongly than later ones: a
|
||||
# title runs brand -> sub-brand -> variant -> size, so in "Coca-Cola
|
||||
# Sprite Lemon 750ml" the word that separates this product from its
|
||||
# brand-mates is "sprite", and "lemon" is only a modifier. Without this
|
||||
# weighting a combo listing that merely shares the modifier
|
||||
# ("...coca-cola-750-ml-limca-soft-drink-lemon-lime...") ties with the
|
||||
# real Sprite image and then wins on domain reputation.
|
||||
_weights = {word: len(_distinctive) - i for i, word in enumerate(_distinctive)}
|
||||
|
||||
def _mentions_product(url_lower: str) -> int:
|
||||
"""How strongly the URL names THIS product, not merely its brand."""
|
||||
if not _distinctive:
|
||||
return 1 # nothing to distinguish by; don't penalise anything
|
||||
return sum(weight for word, weight in _weights.items() if word in url_lower)
|
||||
|
||||
# Filter and score images
|
||||
scored_images = []
|
||||
|
||||
|
||||
for url in image_urls:
|
||||
if not url or not url.startswith('http'):
|
||||
continue
|
||||
@@ -476,14 +508,41 @@ class ProductCatalogEngine:
|
||||
# Avoid problematic URLs
|
||||
if any(bad in url_lower for bad in ['encrypted-tbn', 'googleusercontent', 'data:', 'placeholder']):
|
||||
score -= 5
|
||||
|
||||
scored_images.append((score, url))
|
||||
|
||||
# Sort by score (highest first) and take top images
|
||||
scored_images.sort(key=lambda x: x[0], reverse=True)
|
||||
best_images = [url for score, url in scored_images[:max_images]]
|
||||
|
||||
logger.info(f"📸 Selected {len(best_images)} best images for {product_title}")
|
||||
|
||||
# Multi-product listings picture several products at once, so even
|
||||
# when they name this one the photo is not of it alone.
|
||||
if any(kw in url_lower for kw in ['combo', 'multipack', 'multi-pack', 'pack-of', 'packof', 'assorted', 'variety-pack']):
|
||||
score -= 8
|
||||
|
||||
scored_images.append((_mentions_product(url_lower), score, url))
|
||||
|
||||
# Sort by (does the URL name this product, then score).
|
||||
#
|
||||
# Relevance has to be a GATE, not another additive term. Domain
|
||||
# reputation is worth up to +20 here while a matching product word is
|
||||
# worth +1, so a BigBasket photo of a Limca combo (15 + 2 + 2 + 2 = 21)
|
||||
# outranked the correct Sprite image on a lesser domain (8 + 2 + 1 + 1 =
|
||||
# 12) - and index 0 is what becomes the product's `image_url`. That is
|
||||
# the "top image is not this product" bug. Ranking every URL that names
|
||||
# the product above every URL that doesn't makes the primary image
|
||||
# correct, while the existing scoring still orders each group.
|
||||
#
|
||||
# Non-matching URLs are kept as a tail rather than dropped: a product
|
||||
# whose distinctive words never appear in any URL (common for
|
||||
# CDN-hashed filenames) would otherwise end up with no images at all.
|
||||
scored_images.sort(key=lambda x: (x[0], x[1]), reverse=True)
|
||||
best_images = [url for _relevant, _score, url in scored_images[:max_images]]
|
||||
|
||||
matched = sum(1 for relevant, _s, _u in scored_images[:max_images] if relevant)
|
||||
logger.info(
|
||||
f"📸 Selected {len(best_images)} best images for {product_title} "
|
||||
f"({matched} naming the product)"
|
||||
)
|
||||
if best_images and not matched:
|
||||
logger.warning(
|
||||
f"📸 No candidate image URL names {product_title!r} - the primary "
|
||||
f"image may not be this product"
|
||||
)
|
||||
return best_images
|
||||
|
||||
def search_with_python(self, query: str, brand: str) -> List[str]:
|
||||
|
||||
@@ -79,7 +79,7 @@ from app.services.category_registry import (
|
||||
detect_category_from_text,
|
||||
sanitize_category_language,
|
||||
)
|
||||
from app.services.category_units import fix_or_reject_size
|
||||
from app.services.category_units import fix_or_reject_size, parse_unit
|
||||
from app.services.embeddings_service import embed_texts
|
||||
from app.services.enrichment.barcode.stage import BarcodeEnrichmentStage
|
||||
from app.services.enrichment.hsn_gst.stage import HsnGstEnrichmentStage
|
||||
@@ -274,10 +274,33 @@ def stage_2_row_intake(row: Dict[str, Any], *, use_llm: bool = True) -> Dict[str
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 3 - Title & category consistency
|
||||
# ---------------------------------------------------------------------------
|
||||
def _is_category_code(category: Any) -> bool:
|
||||
"""True when `category` is an opaque code rather than a category name.
|
||||
|
||||
A sheet with a `categoryid` column (values 1, 2, 3) has that column claimed
|
||||
by the `category` keyword rule - "category" is a substring of "categoryid" -
|
||||
so the id lands in the category field. Stored as-is it reaches the user
|
||||
("Coca-Cola . 1" on the product card, "1" in the sidebar filter) and it
|
||||
silently disables three downstream systems that are all keyed by category
|
||||
NAME: the pack-size unit rulebook, HSN/GST enrichment, and category-scoped
|
||||
search. A number is never a category name, so treat it as not supplied and
|
||||
let the keyword detector resolve it from the product name instead.
|
||||
"""
|
||||
text = str(category or "").strip()
|
||||
return bool(text) and text.replace(".", "", 1).isdigit()
|
||||
|
||||
|
||||
def stage_3_title_category(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
title = row.get("title") or row.get("product_name") or ""
|
||||
category = row.get("category")
|
||||
|
||||
if _is_category_code(category):
|
||||
row.setdefault("_notes", []).append(
|
||||
f"category {str(category).strip()!r} is an id, not a name; detecting from the product name"
|
||||
)
|
||||
category = None
|
||||
row["category"] = None
|
||||
|
||||
if _blank(category):
|
||||
category = detect_category_from_text(f"{title} {row.get('description') or ''}")
|
||||
row["_category_deterministic"] = bool(category)
|
||||
@@ -311,10 +334,40 @@ def stage_3_title_category(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 4 - Pack-size explosion & unit safety
|
||||
# ---------------------------------------------------------------------------
|
||||
def _is_unitless_number(size: str) -> bool:
|
||||
"""True for a bare quantity like "72" - a number carrying no unit.
|
||||
|
||||
A store sheet's `Quantity` / `Case Pack` / `Units Per Pack` column is claimed
|
||||
by the `size_variants` keyword rule in `map_spreadsheet_columns`, so its value
|
||||
arrives here dressed as a pack size. It is not one: a pack size needs a unit
|
||||
to mean anything, and keeping the bare number costs twice over. It is
|
||||
appended to the product name ("Coca-Cola 750ml" becomes "Coca-Cola 750ml 72"),
|
||||
and it is folded into `image_id`, so the next upload with a different
|
||||
quantity inserts a duplicate product instead of updating this one.
|
||||
|
||||
`fix_or_reject_size` deliberately passes bare numbers through - a number
|
||||
without a unit contradicts no category - so the check has to happen here,
|
||||
before the size ever reaches it.
|
||||
|
||||
Only a parsed number with no unit token qualifies. A word-only label
|
||||
("Standard", "Family Pack") parses as (None, None) and is left alone.
|
||||
"""
|
||||
value, unit = parse_unit(size)
|
||||
return value is not None and not unit
|
||||
|
||||
|
||||
def _sizes_for(row: Dict[str, Any]) -> List[str]:
|
||||
sizes = [str(s).strip() for s in (row.get("size_variants") or []) if str(s).strip()]
|
||||
declared = [str(s).strip() for s in (row.get("size_variants") or []) if str(s).strip()]
|
||||
sizes = [s for s in declared if not _is_unitless_number(s)]
|
||||
for ignored in declared:
|
||||
if _is_unitless_number(ignored):
|
||||
row.setdefault("_notes", []).append(
|
||||
f"ignored pack size {ignored!r}: a number with no unit is a quantity, not a size"
|
||||
)
|
||||
if sizes:
|
||||
return sizes
|
||||
# Every declared size was a bare quantity (or none were given). Fall through
|
||||
# to the name, which for a store sheet usually carries the real size already.
|
||||
match = _SIZE_IN_TITLE.search(row.get("product_name") or "")
|
||||
if match:
|
||||
return [match.group(0).strip()]
|
||||
@@ -391,7 +444,14 @@ def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, An
|
||||
row["image_urls"] = list(best)
|
||||
row["image_url"] = best[0]
|
||||
except Exception as exc: # noqa: BLE001 - an image is not worth the row
|
||||
logger.debug("Image search skipped for %r: %s", row.get("product_name"), exc)
|
||||
# `warning`, not `debug`: at the default log level a debug line is
|
||||
# invisible, so a row that silently lost its images looked identical to
|
||||
# one that never wanted any. The row still survives - an image is not
|
||||
# worth failing it - but the operator gets told.
|
||||
logger.warning("Image search failed for %r: %s", row.get("product_name"), exc)
|
||||
row.setdefault("_notes", []).append(f"image search failed: {exc}")
|
||||
if _blank(row.get("image_urls")):
|
||||
row.setdefault("_notes", []).append("no image found for this product")
|
||||
return row
|
||||
|
||||
|
||||
@@ -696,6 +756,13 @@ def run_pipeline(
|
||||
progress(4, STAGE_NAMES[3], len(exploded), len(exploded))
|
||||
result.products_built = len(exploded)
|
||||
|
||||
# Stage 4 copies the parent row into each variant, notes included, and the
|
||||
# parent's notes have just been reported. Clear them so that the collection
|
||||
# after stage 7 reports only what stages 5-7 add, once per variant, rather
|
||||
# than repeating stages 1-4 once per pack size.
|
||||
for row in exploded:
|
||||
row["_notes"] = []
|
||||
|
||||
# ---- stages 5-7, per exploded row --------------------------------------
|
||||
for index, row in enumerate(exploded, start=1):
|
||||
stage_5_pricing(row)
|
||||
@@ -707,6 +774,12 @@ def run_pipeline(
|
||||
stage_7_sku(row)
|
||||
progress(7, STAGE_NAMES[6], index, len(exploded))
|
||||
|
||||
# Whatever stages 5-7 recorded per variant - most usefully, that a product
|
||||
# ended up with no image.
|
||||
for row in exploded:
|
||||
for note in row.get("_notes") or []:
|
||||
result.warnings.append(f"row {row.get('_row')}: {note}")
|
||||
|
||||
# ---- stages 8-11, grouped by destination brand -------------------------
|
||||
by_brand: Dict[str, List[Dict[str, Any]]] = {}
|
||||
for row in exploded:
|
||||
|
||||
@@ -53,7 +53,12 @@ from typing import Dict, List, Optional, Tuple
|
||||
# this category when sanitizing cross-category language
|
||||
# out of a generated description.
|
||||
CATEGORY_REGISTRY: List[Dict[str, object]] = [
|
||||
{"category": "Biscuits & Cookies", "keywords": ["biscuits", "biscuit", "biscit", "biskut", "cookies", "cookie"], "generic_term": "biscuit"},
|
||||
# "bikis" is here so that "Britannia Milk Bikis" resolves as the biscuit it
|
||||
# is. Without it the only keyword in that name is Dairy's "milk", which not
|
||||
# only mislabels the product but hands stage 4 a volume unit rulebook - the
|
||||
# exact "Britannia Milk Bikis - 200ml, 500ml, 1L" defect category_units.py
|
||||
# was written to stop.
|
||||
{"category": "Biscuits & Cookies", "keywords": ["biscuits", "biscuit", "biscit", "biskut", "cookies", "cookie", "bikis"], "generic_term": "biscuit"},
|
||||
{"category": "Rusk", "keywords": ["rusks", "rusk"], "generic_term": "rusk"},
|
||||
{"category": "Crackers", "keywords": ["crackers", "cracker", "saltine"], "generic_term": "cracker"},
|
||||
{"category": "Cakes & Muffins", "keywords": ["cakes", "cake", "muffins", "muffin"], "generic_term": "bakery item"},
|
||||
@@ -62,6 +67,13 @@ CATEGORY_REGISTRY: List[Dict[str, object]] = [
|
||||
{"category": "Candy & Confectionery", "keywords": ["candy", "candies", "toffee", "toffees", "lollipop", "lollipops", "confectionery", "mints", "chewing gum"], "generic_term": "candy"},
|
||||
{"category": "Snacks", "keywords": ["snacks", "snack", "chips", "namkeen", "wafers", "wafer", "kurkure", "lays"], "generic_term": "snack"},
|
||||
{"category": "Chocolates", "keywords": ["chocolates", "chocolate", "chocate", "choclate", "cocoa", "cadbury chocolate", "dairy milk"], "generic_term": "chocolate"},
|
||||
# Listed ahead of "Cooking Oils" so a drink is resolved before that entry's
|
||||
# very broad bare "oil" keyword gets a chance, and ahead of "Dairy" only in
|
||||
# keywords it does not share - "milk", "lassi" and "buttermilk" are
|
||||
# deliberately left to Dairy. Kept free of "soda" (baking soda is a staple,
|
||||
# not a drink) and of "tea"/"coffee" on their own (those are sold as leaves
|
||||
# and grounds far more often than as a drink).
|
||||
{"category": "Beverages", "keywords": ["beverages", "beverage", "soft drink", "soft drinks", "cold drink", "cold drinks", "carbonated", "aerated drink", "cola", "coke", "juice", "juices", "squash", "sharbat", "energy drink", "sports drink", "mineral water", "packaged drinking water", "lemonade", "iced tea", "thums up", "sprite", "fanta", "limca", "maaza", "pepsi", "mirinda"], "generic_term": "beverage"},
|
||||
{"category": "Cooking Oils", "keywords": ["cooking oil", "edible oil", "sunflower oil", "mustard oil", "vanaspati", "refined oil", "oil", "oils"], "generic_term": "cooking oil"},
|
||||
{"category": "Atta & Staples", "keywords": ["atta", "wheat flour", "flour", "rice", "dal", "pulses", "staples", "suji", "maida"], "generic_term": "staple product"},
|
||||
{"category": "Dairy", "keywords": ["milk", "dairy", "cheese", "paneer", "panner", "paner", "paneerr", "curd", "yogurt", "butter", "ghee", "dahi"], "generic_term": "dairy product"},
|
||||
|
||||
Reference in New Issue
Block a user