Backend- file ingestion API Updates

This commit is contained in:
sriram
2026-08-28 07:49:56 +05:30
parent f698720ee2
commit a54bd43f8b
17 changed files with 2306 additions and 138 deletions

View File

@@ -0,0 +1,85 @@
"""Live view of the batches this process knows about.
Same pattern and the same documented trade-offs as `store_catalog_job_store.py`
and its siblings: a process-local dict behind a lock, not shared across uvicorn
workers. Adding a broker for this would be operational weight the project has
already decided against (see `job_store.py`).
WHAT IS DIFFERENT HERE, AND WHY IT STILL EARNS ITS PLACE
--------------------------------------------------------
Unlike the other job stores, this one is not the only record. `manifest.json`
on the volume is the durable truth; this is a cache in front of it, and it
exists for one reason: the UI polls every 3 seconds while row-level progress
ticks many times a second. Serving those polls from memory keeps both the disk
writes and the read path off the hot loop. A cache miss is not a 404 - `get()`
falls back to reading the manifest, so a batch from before the last restart is
still visible.
Cancellation lives here too rather than on disk. It is a request about the run
in flight, and the run in flight is in this process.
"""
from __future__ import annotations
import threading
from typing import Dict, List, Optional
from app.core import batch_ingest
class BatchJobStore:
def __init__(self) -> None:
self._batches: Dict[str, batch_ingest.BatchManifest] = {}
self._cancelled: set = set()
self._lock = threading.Lock()
def put(self, manifest: batch_ingest.BatchManifest) -> None:
"""Record (or refresh) a batch. This is the `on_change` callback."""
with self._lock:
self._batches[manifest.batch_id] = manifest
def get(self, batch_id: str) -> Optional[batch_ingest.BatchManifest]:
"""Live state if we have it, otherwise whatever is on disk."""
with self._lock:
cached = self._batches.get(batch_id)
if cached is not None:
return cached
return batch_ingest.read_manifest(batch_id)
def recent(self, limit: int = 20) -> List[batch_ingest.BatchManifest]:
"""Newest first, merging the live view over the on-disk one.
Reading the directory rather than only the cache means a restart does
not make previous batches vanish from the list.
"""
with self._lock:
live = dict(self._batches)
merged: Dict[str, batch_ingest.BatchManifest] = {}
for manifest in batch_ingest.list_manifests():
merged[manifest.batch_id] = live.get(manifest.batch_id, manifest)
for batch_id, manifest in live.items():
merged.setdefault(batch_id, manifest)
ordered = sorted(merged.values(), key=lambda m: m.created_at, reverse=True)
return ordered[: max(1, limit)]
# -- cancellation --------------------------------------------------------
def cancel(self, batch_id: str) -> None:
"""Ask the worker to stop before it picks up the next file.
Nothing interrupts the file already running. Killing a pipeline halfway
would leave some of its rows written and the rest not, with no record of
where it stopped; letting the current file finish is both simpler and
the only version with a defined outcome.
"""
with self._lock:
self._cancelled.add(batch_id)
def is_cancelled(self, batch_id: str) -> bool:
with self._lock:
return batch_id in self._cancelled
def clear_cancel(self, batch_id: str) -> None:
with self._lock:
self._cancelled.discard(batch_id)
batch_job_store = BatchJobStore()

View File

@@ -0,0 +1,385 @@
"""Admin endpoints for ingesting several store spreadsheets as one batch.
POST /api/admin/catalog-batch/preview - parse only, per file
POST /api/admin/catalog-batch/ingest - 202 + batch_id
GET /api/admin/catalog-batch/batches - recent batches
GET /api/admin/catalog-batch/batches/{id} - poll one batch
POST /api/admin/catalog-batch/batches/{id}/resume - after a restart
POST /api/admin/catalog-batch/batches/{id}/cancel - stop the rest
This is the multi-file sibling of `store_catalog.py`, and it deliberately does
not replace it: the single-file endpoints are untouched and still work. What is
different is the unit of work. Five files are one batch with one id, so the
question a colleague actually asks - "did the drop land?" - has one answer
rather than five.
WHY THE NETWORK STAGES DEFAULT OFF HERE
---------------------------------------
The single-file UI sends `use_llm=true, fetch_images=true`. At one file that is
a considered trade. At twenty it is thousands of outbound requests and, for the
image stage, a Playwright subprocess that can burn three minutes on its own -
on a single-vCPU container that is also serving the API. So a batch opts IN to
those stages; it does not opt out. `USE_OLLAMA` is false in production anyway,
which makes `use_llm` a no-op there and the honest default obvious.
"""
from __future__ import annotations
import logging
import queue
from typing import List, Optional
from fastapi import APIRouter, Depends, File, HTTPException, UploadFile, status
from pydantic import BaseModel
from app.api.batch_job_store import batch_job_store
from app.api.deps import require_admin
from app.core import batch_ingest, batch_worker
from app.core import store_catalog_pipeline as pipeline
from app.infrastructure.settings import (
BATCH_MAX_FILES,
BATCH_MAX_TOTAL_BYTES,
BATCH_MAX_TOTAL_ROWS,
)
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/admin/catalog-batch", tags=["admin", "catalog"])
# Per-file ceilings match store_catalog.py exactly. A file that is too big for
# the single-file endpoint is not somehow acceptable because it arrived with
# four friends.
MAX_UPLOAD_BYTES = 10 * 1024 * 1024
MAX_UPLOAD_ROWS = 2000
PREVIEW_ROWS = 10
# The worker cannot import the API layer without a cycle, so the wiring is done
# here, at import, once.
batch_worker.configure(
on_change=batch_job_store.put,
should_cancel=batch_job_store.is_cancelled,
)
class BatchFileOut(BaseModel):
index: int
filename: str
status: str
detail: Optional[str] = None
stage_index: int = 0
stage_name: str = ""
total_stages: int = pipeline.TOTAL_STAGES
rows_done: int = 0
rows_total: int = 0
size_bytes: int = 0
result: Optional[dict] = None
class BatchOut(BaseModel):
batch_id: str
status: str
detail: Optional[str] = None
created_at: float
updated_at: float
files_total: int
files_done: int
files_failed: int
current_file: Optional[str] = None
use_llm: bool
fetch_images: bool
totals: dict
brands: List[str]
files: List[BatchFileOut]
def _to_out(manifest: batch_ingest.BatchManifest) -> BatchOut:
body = manifest.to_dict()
body["files"] = [BatchFileOut(**{
key: entry[key] for key in BatchFileOut.model_fields if key in entry
}) for entry in body["files"]]
return BatchOut(**{k: v for k, v in body.items() if k in BatchOut.model_fields})
async def _read_uploads(files: List[UploadFile]) -> List[tuple]:
"""Read every upload into memory, enforcing the count and size ceilings.
Read here rather than in the worker for the same reason `store_catalog.py`
gives: `UploadFile` is backed by a temporary file tied to the request, and
it is gone before a background thread would reach it.
"""
if not files:
raise HTTPException(status_code=400, detail="No files were uploaded.")
if len(files) > BATCH_MAX_FILES:
raise HTTPException(
status_code=413,
detail=(
f"{len(files)} files exceeds the {BATCH_MAX_FILES}-file limit for one "
f"batch. Split the drop and upload it in two batches."
),
)
read: List[tuple] = []
total = 0
for upload in files:
contents = await upload.read()
name = upload.filename or "upload.xlsx"
if not contents:
# Recorded rather than raised - an empty file among nine good ones
# is a fact about that file, not a reason to reject the drop.
read.append((name, b""))
continue
if len(contents) > MAX_UPLOAD_BYTES:
raise HTTPException(
status_code=413,
detail=(
f"'{name}' is larger than the "
f"{MAX_UPLOAD_BYTES // (1024 * 1024)}MB per-file limit."
),
)
total += len(contents)
if total > BATCH_MAX_TOTAL_BYTES:
raise HTTPException(
status_code=413,
detail=(
f"The batch is larger than the "
f"{BATCH_MAX_TOTAL_BYTES // (1024 * 1024)}MB total limit."
),
)
read.append((name, contents))
return read
def _parse_all(read: List[tuple]):
"""Split the uploads into (valid, invalid) by trying to parse each one.
Failing fast here is what stops a batch transitioning straight to "failed"
a second after it started - the same reasoning as `store_catalog.py:147`,
applied per file so that one bad sheet does not condemn the others.
"""
valid: List[tuple] = []
invalid: List[tuple] = []
rows_total = 0
for name, contents in read:
if not contents:
invalid.append((name, "The file is empty."))
continue
try:
df, _mapping = pipeline.parse_spreadsheet(name, contents)
except HTTPException as exc:
# read_products_dataframe raises HTTPException for an unsupported
# extension or a missing Excel reader; its message already names
# the file and what to do about it.
invalid.append((name, str(exc.detail)))
continue
except Exception as exc: # noqa: BLE001 - an unreadable sheet is user error
invalid.append((name, f"Could not parse the file: {exc}"))
continue
if df.empty:
invalid.append((name, "The file has no data rows."))
continue
if len(df) > MAX_UPLOAD_ROWS:
invalid.append((
name,
f"{len(df)} rows exceeds the {MAX_UPLOAD_ROWS}-row per-file limit.",
))
continue
rows_total += int(len(df))
if rows_total > BATCH_MAX_TOTAL_ROWS:
raise HTTPException(
status_code=413,
detail=(
f"The batch totals more than {BATCH_MAX_TOTAL_ROWS} rows. "
f"Split it and upload in two batches."
),
)
valid.append((name, contents, int(len(df))))
return valid, invalid, rows_total
@router.post("/preview", dependencies=[Depends(require_admin)])
async def preview_catalog_batch(files: List[UploadFile] = File(...)) -> dict:
"""Parse every file and report how its columns were understood.
Nothing is staged and no batch is created. The mapping from a store's own
headers onto catalog fields is a guess, and finding out that "Item" was read
as the description after twenty files have been scraped is expensive.
"""
read = await _read_uploads(files)
out = []
for name, contents in read:
if not contents:
out.append({"filename": name, "ok": False, "error": "The file is empty."})
continue
try:
df, mapping = pipeline.parse_spreadsheet(name, contents)
except HTTPException as exc:
out.append({"filename": name, "ok": False, "error": str(exc.detail)})
continue
except Exception as exc: # noqa: BLE001
out.append({"filename": name, "ok": False,
"error": f"Could not parse the file: {exc}"})
continue
if df.empty:
out.append({"filename": name, "ok": False, "error": "The file has no data rows."})
continue
out.append({
"filename": name,
"ok": True,
"rows_total": int(len(df)),
"over_row_limit": bool(len(df) > MAX_UPLOAD_ROWS),
"recognised_columns": {f: str(c) for f, c in mapping.columns.items()},
"unrecognised_columns": mapping.unrecognised,
"brand_column_present": "brand" in mapping.columns,
"preview": df.head(PREVIEW_ROWS).fillna("").astype(str).to_dict(orient="records"),
})
return {
"files": out,
"files_total": len(out),
"files_ok": sum(1 for f in out if f.get("ok")),
"rows_total": sum(int(f.get("rows_total") or 0) for f in out if f.get("ok")),
"stages": list(pipeline.STAGE_NAMES),
"limits": {
"max_files": BATCH_MAX_FILES,
"max_rows_per_file": MAX_UPLOAD_ROWS,
"max_rows_total": BATCH_MAX_TOTAL_ROWS,
"max_bytes_per_file": MAX_UPLOAD_BYTES,
"max_bytes_total": BATCH_MAX_TOTAL_BYTES,
},
}
@router.post("/ingest", status_code=status.HTTP_202_ACCEPTED,
dependencies=[Depends(require_admin)])
async def ingest_catalog_batch(
files: List[UploadFile] = File(...),
use_llm: bool = False,
fetch_images: bool = False,
) -> BatchOut:
"""Stage the files, queue the batch, and return an id to poll.
Returns immediately. In production the browser reaches this through Traefik
on a different host to the frontend, so anything that sat on the request
path would be racing an idle timeout nobody here controls.
"""
read = await _read_uploads(files)
valid, invalid, _rows = _parse_all(read)
if not valid:
detail = "; ".join(f"{name}: {reason}" for name, reason in invalid)
raise HTTPException(
status_code=400,
detail=f"None of the uploaded files could be ingested. {detail}",
)
manifest = batch_ingest.stage_batch(
[(name, contents) for name, contents, _n in valid],
use_llm=use_llm,
fetch_images=fetch_images,
invalid=invalid,
)
for entry, (_name, _contents, rows) in zip(manifest.files, valid):
entry.rows_total = rows
batch_ingest.write_manifest(manifest)
batch_job_store.put(manifest)
try:
batch_worker.submit(manifest.batch_id)
except queue.Full:
manifest.status = batch_ingest.QUEUED
manifest.detail = (
"The ingestion queue is full. This batch is staged and can be started "
"with Resume once the running batches finish."
)
batch_ingest.write_manifest(manifest)
batch_job_store.put(manifest)
raise HTTPException(
status_code=429,
detail=(
"Too many batches are already queued. This one has been saved - "
"press Resume on it once the current batch finishes."
),
)
return _to_out(manifest)
@router.get("/batches", dependencies=[Depends(require_admin)])
def list_catalog_batches(limit: int = 20) -> dict:
limit = max(1, min(limit, 100))
return {"batches": [_to_out(m) for m in batch_job_store.recent(limit)]}
@router.get("/batches/{batch_id}", dependencies=[Depends(require_admin)])
def get_catalog_batch(batch_id: str) -> BatchOut:
manifest = batch_job_store.get(batch_id)
if not manifest:
raise HTTPException(status_code=404, detail="Batch not found")
return _to_out(manifest)
@router.post("/batches/{batch_id}/resume", dependencies=[Depends(require_admin)])
def resume_catalog_batch(batch_id: str) -> BatchOut:
"""Re-queue a batch a restart cut short, or one that was queued behind a full queue."""
manifest = batch_ingest.read_manifest(batch_id)
if not manifest:
raise HTTPException(status_code=404, detail="Batch not found")
pending = [f for f in manifest.files if f.status == batch_ingest.QUEUED]
if not pending:
raise HTTPException(
status_code=409,
detail=f"Nothing left to run in this batch (status: {manifest.status}).",
)
batch_job_store.clear_cancel(batch_id)
manifest.status = batch_ingest.QUEUED
manifest.detail = None
batch_ingest.write_manifest(manifest)
batch_job_store.put(manifest)
try:
batch_worker.submit(batch_id)
except queue.Full:
raise HTTPException(
status_code=429,
detail="Too many batches are already queued. Try again shortly.",
)
return _to_out(manifest)
@router.post("/batches/{batch_id}/cancel", dependencies=[Depends(require_admin)])
def cancel_catalog_batch(batch_id: str) -> BatchOut:
"""Stop before the next file. The file already running is allowed to finish.
Interrupting a pipeline mid-file would leave some of its rows written and
the rest not, with nothing recording where it stopped. Letting the current
file complete is the only version of "cancel" with a defined outcome.
"""
manifest = batch_job_store.get(batch_id)
if not manifest:
raise HTTPException(status_code=404, detail="Batch not found")
batch_job_store.cancel(batch_id)
# A batch that has not started yet has no worker to notice the flag, so
# cancel it here and be done.
on_disk = batch_ingest.read_manifest(batch_id)
if on_disk and on_disk.status in {batch_ingest.QUEUED, batch_ingest.INTERRUPTED}:
for entry in on_disk.files:
if entry.status == batch_ingest.QUEUED:
entry.status = batch_ingest.CANCELLED
entry.detail = "Cancelled before this file started."
on_disk.settle()
on_disk.detail = "Cancelled."
batch_ingest.write_manifest(on_disk)
batch_job_store.put(on_disk)
return _to_out(on_disk)
manifest.detail = "Cancelling - the file currently running will finish first."
batch_job_store.put(manifest)
return _to_out(manifest)

520
app/core/batch_ingest.py Normal file
View File

@@ -0,0 +1,520 @@
"""
Multi-file store spreadsheets -> the same 11-stage pipeline -> brand tables.
WHAT THIS ADDS OVER `store_catalog_pipeline`
--------------------------------------------
Nothing about the pipeline itself. `run_pipeline()` already turns one whole
spreadsheet into catalog rows, and this module calls it unchanged, once per
file. What is new is everything *around* a file:
* several files are one unit of work with one id, so "did the whole drop
land?" has an answer;
* the uploads are written to disk before any work starts, so a restart
mid-batch loses nothing but time;
* one file failing does not take the others with it.
WHY THE FILES GO TO DISK
------------------------
The single-file path reads the upload into memory and hands the bytes to a
daemon thread (store_catalog.py). That is fine for one file and one operator
watching it: if the process dies, they re-upload. A five-file batch is a
different proposition - the colleague who sent them is not sitting there, and
silently losing the drop is worse than any amount of extra code. So the bytes
are staged under BATCH_UPLOAD_DIR, which is on the container's declared volume,
and a manifest records what state each file reached.
THIS MODULE IS THE SHARED CORE
------------------------------
Two callers, one implementation:
app/api/routers/batch_catalog.py -> production (worker thread)
orchestration/assets/batch_catalog.py -> Dagster (development)
That is the same arrangement `orchestration/assets/catalog.py` already
describes for the brand pipeline: "Wrapping rather than reimplementing is what
stops the two paths drifting: a fix to a stage fixes both." Dagster is not
deployed here and is not on any request path; it wraps these functions so the
graph is inspectable and re-runnable locally, and production executes the very
same code without paying for a daemon.
"""
from __future__ import annotations
import json
import logging
import os
import re
import shutil
import time
import uuid
from dataclasses import asdict, dataclass, field
from pathlib import Path
from typing import Any, Callable, Dict, List, Optional, Tuple
from app.core import store_catalog_pipeline as pipeline
from app.infrastructure.settings import (
BATCH_AUTO_RESUME,
BATCH_RETENTION_DAYS,
BATCH_UPLOAD_DIR,
)
logger = logging.getLogger(__name__)
MANIFEST_NAME = "manifest.json"
# File states. `cancelled` only ever applies to files that had not started.
QUEUED = "queued"
RUNNING = "running"
DONE = "done"
FAILED = "failed"
CANCELLED = "cancelled"
# Batch states. `partial` is not cosmetic: a batch where four of five files
# landed must not read as a flat success, or nobody goes looking for the fifth.
INTERRUPTED = "interrupted"
PARTIAL = "partial"
TERMINAL_BATCH_STATES = {DONE, FAILED, PARTIAL, CANCELLED}
# `..`, separators and drive letters all stripped. UploadFile.filename is
# attacker-controlled in the general case, and it is used to build a path.
_UNSAFE = re.compile(r"[^A-Za-z0-9._-]+")
def _safe_name(filename: str) -> str:
"""A filename that cannot escape the batch directory.
Path components are discarded rather than escaped: nothing downstream needs
the original directory, and `os.path.basename` alone is not enough here,
because a Windows-authored name like "..\\evil.csv" keeps its backslash on
a Linux container and basename leaves it untouched.
"""
base = str(filename or "upload.xlsx").replace("\\", "/").rsplit("/", 1)[-1]
base = _UNSAFE.sub("_", base).lstrip(".") or "upload.xlsx"
return base[:120]
@dataclass
class BatchFile:
"""One spreadsheet inside a batch, and how far it got."""
index: int
filename: str # what the colleague called it, for display
stored_name: str # what it is called on disk
size_bytes: int = 0
status: str = QUEUED
detail: Optional[str] = None
stage_index: int = 0 # 1-based; 0 while queued
stage_name: str = ""
total_stages: int = pipeline.TOTAL_STAGES
rows_done: int = 0
rows_total: int = 0
result: Optional[Dict[str, Any]] = None
started_at: Optional[float] = None
finished_at: Optional[float] = None
@dataclass
class BatchManifest:
"""The whole batch. Serialised to manifest.json verbatim."""
batch_id: str
status: str = QUEUED
use_llm: bool = False
fetch_images: bool = False
created_at: float = field(default_factory=time.time)
updated_at: float = field(default_factory=time.time)
detail: Optional[str] = None
files: List[BatchFile] = field(default_factory=list)
# -- derived, recomputed rather than stored, so they cannot drift ---------
@property
def files_total(self) -> int:
return len(self.files)
@property
def files_done(self) -> int:
return sum(1 for f in self.files if f.status == DONE)
@property
def files_failed(self) -> int:
return sum(1 for f in self.files if f.status == FAILED)
@property
def current_file(self) -> Optional[str]:
for entry in self.files:
if entry.status == RUNNING:
return entry.filename
return None
def totals(self) -> Dict[str, int]:
"""Summed across every file that produced a result."""
keys = ("rows_total", "products_built", "inserted", "backfilled",
"skipped_existing", "rejected", "error_count")
out = {key: 0 for key in keys}
for entry in self.files:
for key in keys:
out[key] += int((entry.result or {}).get(key) or 0)
return out
def brands(self) -> List[str]:
seen = set()
for entry in self.files:
seen.update((entry.result or {}).get("brands") or [])
return sorted(seen)
def settle(self) -> str:
"""Recompute the batch status from its files. Returns the new status."""
states = {f.status for f in self.files}
if states & {QUEUED, RUNNING}:
self.status = RUNNING if RUNNING in states else QUEUED
elif self.files_done and self.files_failed:
self.status = PARTIAL
elif self.files_done:
self.status = DONE
elif self.files_failed:
self.status = FAILED
else:
self.status = CANCELLED
self.updated_at = time.time()
return self.status
# -- serialisation -------------------------------------------------------
def to_dict(self) -> Dict[str, Any]:
return {
"batch_id": self.batch_id,
"status": self.status,
"use_llm": self.use_llm,
"fetch_images": self.fetch_images,
"created_at": self.created_at,
"updated_at": self.updated_at,
"detail": self.detail,
"files_total": self.files_total,
"files_done": self.files_done,
"files_failed": self.files_failed,
"current_file": self.current_file,
"totals": self.totals(),
"brands": self.brands(),
"files": [asdict(f) for f in self.files],
}
@classmethod
def from_dict(cls, raw: Dict[str, Any]) -> "BatchManifest":
allowed = set(BatchFile.__dataclass_fields__)
files = [
BatchFile(**{k: v for k, v in entry.items() if k in allowed})
for entry in (raw.get("files") or [])
]
return cls(
batch_id=raw["batch_id"],
status=raw.get("status", QUEUED),
use_llm=bool(raw.get("use_llm", False)),
fetch_images=bool(raw.get("fetch_images", False)),
created_at=float(raw.get("created_at") or time.time()),
updated_at=float(raw.get("updated_at") or time.time()),
detail=raw.get("detail"),
files=files,
)
# ---------------------------------------------------------------------------
# Disk layout
# ---------------------------------------------------------------------------
def batch_root() -> Path:
"""Read at call time, not import time, so tests can repoint the directory."""
return Path(BATCH_UPLOAD_DIR)
def batch_dir(batch_id: str) -> Path:
# The id is generated here (uuid4), never taken from a request, but this is
# still the function that turns it into a path - so it validates.
if not re.fullmatch(r"[A-Za-z0-9_-]{1,64}", batch_id or ""):
raise ValueError("Invalid batch id: {!r}".format(batch_id))
return batch_root() / batch_id
def manifest_path(batch_id: str) -> Path:
return batch_dir(batch_id) / MANIFEST_NAME
def write_manifest(manifest: BatchManifest) -> None:
"""Write via a temp file and os.replace.
A half-written manifest.json is indistinguishable from a corrupt one on the
next boot, and the recovery path reads every manifest it finds.
"""
target = manifest_path(manifest.batch_id)
target.parent.mkdir(parents=True, exist_ok=True)
tmp = target.with_name(MANIFEST_NAME + ".tmp")
tmp.write_text(json.dumps(manifest.to_dict(), indent=2), encoding="utf-8")
os.replace(tmp, target)
def read_manifest(batch_id: str) -> Optional[BatchManifest]:
try:
path = manifest_path(batch_id)
except ValueError:
return None
if not path.exists():
return None
try:
return BatchManifest.from_dict(json.loads(path.read_text(encoding="utf-8")))
except Exception as exc: # noqa: BLE001 - a bad manifest must not break a listing
logger.warning("Ignoring unreadable manifest %s: %s", path, exc)
return None
def list_manifests() -> List[BatchManifest]:
"""Every readable batch on disk, newest first."""
root = batch_root()
if not root.exists():
return []
found = []
for child in sorted(root.iterdir()):
if not child.is_dir():
continue
manifest = read_manifest(child.name)
if manifest:
found.append(manifest)
return sorted(found, key=lambda m: m.created_at, reverse=True)
# ---------------------------------------------------------------------------
# Staging
# ---------------------------------------------------------------------------
def stage_batch(
uploads: List[Tuple[str, bytes]],
*,
use_llm: bool = False,
fetch_images: bool = False,
invalid: Optional[List[Tuple[str, str]]] = None,
) -> BatchManifest:
"""Write the uploads to disk and return the manifest describing them.
`invalid` carries files the caller already rejected (unparseable, empty).
They are recorded as failed members of the batch rather than dropped: an
operator who selected six files and sees five must be told what happened to
the sixth, and the batch page is the only place they will look.
"""
batch_id = uuid.uuid4().hex
directory = batch_dir(batch_id)
directory.mkdir(parents=True, exist_ok=True)
manifest = BatchManifest(batch_id=batch_id, use_llm=use_llm, fetch_images=fetch_images)
position = 0
for filename, content in uploads:
stored = "{:02d}_{}".format(position, _safe_name(filename))
(directory / stored).write_bytes(content)
manifest.files.append(
BatchFile(
index=position,
filename=filename or stored,
stored_name=stored,
size_bytes=len(content),
)
)
position += 1
for filename, reason in invalid or []:
manifest.files.append(
BatchFile(
index=position,
filename=filename or "(unnamed)",
stored_name="",
status=FAILED,
detail=reason,
finished_at=time.time(),
)
)
position += 1
write_manifest(manifest)
return manifest
# ---------------------------------------------------------------------------
# Execution
# ---------------------------------------------------------------------------
OnChange = Callable[["BatchManifest"], None]
"""Called after every file-level transition, and on row progress.
The callback is what keeps the in-memory view the API serves in step with the
run. It must be cheap - it fires on every progress tick - which is why the
manifest is NOT written to disk from inside it.
"""
def _noop_change(manifest: "BatchManifest") -> None:
return None
# Row-level progress arrives many times a second. Persisting each one would be
# hundreds of small writes per file, to record something nobody reads back from
# disk anyway - the API answers polls from memory. Disk is for surviving a
# restart, and a restart only needs to know which FILE was in flight.
_PROGRESS_FLUSH_SECONDS = 5.0
def run_batch(
batch_id: str,
*,
on_change: OnChange = _noop_change,
should_cancel: Optional[Callable[[], bool]] = None,
) -> BatchManifest:
"""Run every queued file in the batch, in order, one at a time.
Files are independent. A file that raises is marked failed with the reason
and the loop moves to the next one - the pipeline's own "a stage never
raises" rule protects rows within a file, and this is the same idea one
level up.
"""
manifest = read_manifest(batch_id)
if manifest is None:
raise FileNotFoundError("No manifest for batch {}".format(batch_id))
directory = batch_dir(batch_id)
manifest.status = RUNNING
manifest.detail = None
manifest.updated_at = time.time()
write_manifest(manifest)
on_change(manifest)
for entry in manifest.files:
if entry.status != QUEUED:
continue
if should_cancel is not None and should_cancel():
entry.status = CANCELLED
entry.detail = "Cancelled before this file started."
entry.finished_at = time.time()
continue
entry.status = RUNNING
entry.started_at = time.time()
entry.stage_index = 0
entry.stage_name = ""
write_manifest(manifest)
on_change(manifest)
last_flush = [time.time()]
def progress(stage_index: int, stage_name: str, done: int, total: int,
_entry: BatchFile = entry) -> None:
_entry.stage_index = stage_index
_entry.stage_name = stage_name
_entry.rows_done = done
_entry.rows_total = total
manifest.updated_at = time.time()
on_change(manifest)
now = time.time()
if now - last_flush[0] >= _PROGRESS_FLUSH_SECONDS:
last_flush[0] = now
write_manifest(manifest)
try:
content = (directory / entry.stored_name).read_bytes()
result = pipeline.run_pipeline(
entry.filename,
content,
progress=progress,
use_llm=manifest.use_llm,
fetch_images=manifest.fetch_images,
)
body = result.as_dict()
entry.result = body
if body.get("storage_error"):
# Same judgement as the single-file path: rows were built but
# none reached the database, and calling that success would
# leave the operator believing the catalog changed.
entry.status = FAILED
entry.detail = (
"Built {} row(s) but storing them failed: {}".format(
body["products_built"], body["storage_error"]
)
)
else:
entry.status = DONE
entry.detail = (
"{} inserted, {} backfilled, {} unchanged, {} rejected".format(
body["inserted"], body["backfilled"],
body["skipped_existing"], body["rejected"],
)
)
except Exception as exc: # noqa: BLE001 - one bad file must not end the batch
logger.exception("Batch %s: file %s failed", batch_id, entry.filename)
entry.status = FAILED
entry.detail = str(exc)
entry.finished_at = time.time()
write_manifest(manifest)
on_change(manifest)
manifest.settle()
write_manifest(manifest)
on_change(manifest)
return manifest
# ---------------------------------------------------------------------------
# Recovery and retention
# ---------------------------------------------------------------------------
def scan_interrupted() -> List[str]:
"""Mark batches that a restart cut short. Returns the ids resumable now.
DELIBERATELY DOES NOT RE-RUN ANYTHING by default. A container caught in a
restart loop would otherwise re-enter the heaviest work in the application
on every boot, turning a slow start into an unrecoverable one. The files and
the manifest are on the volume, so nothing is lost by waiting for a human to
press Resume - and BATCH_AUTO_RESUME=true is there for a deployment that has
earned the trust.
"""
resumable: List[str] = []
for manifest in list_manifests():
if manifest.status in TERMINAL_BATCH_STATES or manifest.status == INTERRUPTED:
continue
for entry in manifest.files:
if entry.status == RUNNING:
entry.status = QUEUED
entry.stage_index = 0
entry.stage_name = ""
entry.rows_done = 0
entry.detail = "Interrupted by a restart; queued again."
manifest.status = INTERRUPTED
manifest.detail = "Interrupted by a restart. Press Resume to continue."
manifest.updated_at = time.time()
write_manifest(manifest)
resumable.append(manifest.batch_id)
if resumable and not BATCH_AUTO_RESUME:
logger.info(
"Found %d interrupted batch(es); leaving them for a manual resume "
"(BATCH_AUTO_RESUME is false).", len(resumable),
)
return resumable
def purge_expired(now: Optional[float] = None) -> List[str]:
"""Delete staged files for batches older than BATCH_RETENTION_DAYS.
Called when the worker goes idle, never on a request path - deleting a few
hundred megabytes should not be something an operator waits on.
"""
if BATCH_RETENTION_DAYS <= 0:
return []
cutoff = (now if now is not None else time.time()) - BATCH_RETENTION_DAYS * 86400
removed: List[str] = []
for manifest in list_manifests():
if manifest.created_at >= cutoff:
continue
if manifest.status not in TERMINAL_BATCH_STATES:
# An old batch still queued is a bug somewhere, but deleting the
# only copy of its input is not the way to find out.
continue
try:
shutil.rmtree(batch_dir(manifest.batch_id))
removed.append(manifest.batch_id)
except OSError as exc:
logger.warning("Could not purge batch %s: %s", manifest.batch_id, exc)
if removed:
logger.info("Purged %d expired batch upload(s).", len(removed))
return removed

122
app/core/batch_worker.py Normal file
View File

@@ -0,0 +1,122 @@
"""One worker thread for every batch this process will ever run.
WHY A SINGLE BOUNDED WORKER, AND NOT `run_in_background`
---------------------------------------------------------
`app/api/background.py` starts a fresh daemon thread per job and returns. That
is right for the jobs it serves - they are started by hand, one at a time, by
an operator watching the result. It is the wrong shape for this feature.
A batch is up to twenty spreadsheets of two thousand rows. The deployment this
runs on is a single container with one vCPU and no CPU limit, so nothing above
stops two batches from interleaving; they would simply both run, at half speed
each, while the API tries to answer requests and the health probe tries to get
a socket. Two operators uploading at the same time is not an unusual event, it
is a Tuesday.
So: one thread, one batch at a time, and a bounded queue in front. A second
batch waits its turn instead of competing, and beyond `BATCH_QUEUE_MAX` the
endpoint says 429 rather than accepting work it has no intention of starting.
THE THREAD IS STARTED LAZILY
----------------------------
Not at import, not in the lifespan startup. A container that never receives a
batch pays nothing for this module beyond the import, which is why `submit()`
is the only thing that can bring the worker to life. Boot cost on this host is
already ~21s of imports and is the thing most worth not adding to.
"""
from __future__ import annotations
import logging
import queue
import threading
from typing import Callable, Optional
from app.core import batch_ingest
from app.infrastructure.settings import BATCH_QUEUE_MAX
logger = logging.getLogger(__name__)
QueueFull = queue.Full
_queue: "queue.Queue[str]" = queue.Queue(maxsize=max(1, BATCH_QUEUE_MAX))
_worker: Optional[threading.Thread] = None
_lock = threading.Lock()
# Set by the router at import so this module does not import the API layer -
# app.api.batch_job_store already imports app.core.batch_ingest, and closing
# that loop the other way would be a circular import at boot.
_on_change: Optional[Callable[[batch_ingest.BatchManifest], None]] = None
_should_cancel: Optional[Callable[[str], bool]] = None
def configure(
*,
on_change: Callable[[batch_ingest.BatchManifest], None],
should_cancel: Callable[[str], bool],
) -> None:
"""Wire the worker to the job store. Called once, by the router module."""
global _on_change, _should_cancel
_on_change = on_change
_should_cancel = should_cancel
def queue_depth() -> int:
return _queue.qsize()
def is_running() -> bool:
return _worker is not None and _worker.is_alive()
def submit(batch_id: str) -> None:
"""Enqueue a batch and make sure the worker exists. Raises `queue.Full`.
`put_nowait` rather than `put`: blocking here would block the request
handler, which is the one thing an endpoint that returns 202 must never do.
"""
_queue.put_nowait(batch_id)
_ensure_worker()
def _ensure_worker() -> None:
global _worker
with _lock:
if _worker is not None and _worker.is_alive():
return
_worker = threading.Thread(target=_loop, name="catalog-batch-worker", daemon=True)
_worker.start()
def _loop() -> None:
"""Drain the queue forever.
Every iteration is wrapped, because a worker that dies on one bad batch
would leave every future batch queued behind a thread that is not there -
a failure that looks, from the UI, exactly like a batch that is merely slow.
"""
while True:
batch_id = _queue.get()
try:
_run_one(batch_id)
except Exception: # noqa: BLE001 - see docstring
logger.exception("Batch worker: unhandled error on batch %s", batch_id)
finally:
_queue.task_done()
if _queue.empty():
# Retention runs when there is nothing waiting, so deleting old
# uploads never delays a batch and never sits on a request path.
try:
batch_ingest.purge_expired()
except Exception: # noqa: BLE001 - housekeeping must not kill the worker
logger.exception("Batch worker: purge failed")
def _run_one(batch_id: str) -> None:
on_change = _on_change or (lambda manifest: None)
cancelled = _should_cancel or (lambda _id: False)
batch_ingest.run_batch(
batch_id,
on_change=on_change,
should_cancel=lambda: cancelled(batch_id),
)

View File

@@ -134,6 +134,41 @@ MODEL_ARTIFACTS_DIR = _dir(
"MODEL_ARTIFACTS_DIR", _BACKEND_ROOT / "app" / "intelligence" / "artifacts"
)
# ---------------------------------------------------------------------------
# Batch catalog ingestion (multi-file upload -> the 11-stage pipeline)
# ---------------------------------------------------------------------------
# Staged uploads live under DATA_DIR because that path is already a declared
# volume (backend/Dockerfile). A batch that survives a container restart is the
# whole point of writing the files down instead of holding them in the worker
# thread the way the single-file path does.
#
# Every ceiling below is enforced in the application, not at the proxy. In
# production the browser calls mcp.nearle.ai.in directly, so neither nginx's
# client_max_body_size nor Caddy's request_body cap is in front of these
# endpoints - whatever Traefik defaults to is, and it is not ours to rely on.
BATCH_UPLOAD_DIR = _dir("BATCH_UPLOAD_DIR", DATA_DIR / "batch_uploads")
# Per-file limits stay at the single-upload values (10MB / 2000 rows, see
# app/api/routers/store_catalog.py); these bound the BATCH on top of that.
BATCH_MAX_FILES = int(os.getenv("BATCH_MAX_FILES", "20"))
BATCH_MAX_TOTAL_BYTES = int(os.getenv("BATCH_MAX_TOTAL_BYTES", str(50 * 1024 * 1024)))
BATCH_MAX_TOTAL_ROWS = int(os.getenv("BATCH_MAX_TOTAL_ROWS", "20000"))
# Batches waiting behind the one running. Past this the endpoint returns 429
# rather than accepting work it has no intention of starting soon.
BATCH_QUEUE_MAX = int(os.getenv("BATCH_QUEUE_MAX", "4"))
# Staged files are deleted this many days after the batch was created. Without
# this the upload directory only grows, on a host whose disk is the scarcest
# resource it has.
BATCH_RETENTION_DAYS = int(os.getenv("BATCH_RETENTION_DAYS", "7"))
# Deliberately false. A batch interrupted by a restart is marked "interrupted"
# and waits for someone to press Resume. Auto-resuming would mean a container
# stuck in a restart loop re-runs the heaviest work in the app on every boot,
# which is precisely how a slow start turns into an unrecoverable spiral.
BATCH_AUTO_RESUME = _bool("BATCH_AUTO_RESUME", "false")
# Pristine copies of the bundled seed catalogs and pre-trained models, placed
# here by the Dockerfile at a path that is never itself mounted over.
#

View File

@@ -22,6 +22,7 @@ from fastapi.responses import FileResponse
from app.infrastructure.persistence import restore_bundled_assets
from app.infrastructure.settings import (
API_CORS_ORIGINS,
BATCH_AUTO_RESUME,
BRAND_SYNC_INTERVAL_SECONDS,
cleaned_env_names,
)
@@ -30,7 +31,7 @@ from app.api.routers import health, brands, search, suggest, chat, catalog, syst
from app.api.routers import stores, discounts, analytics as store_analytics, trending, recommendations, store_admin
from app.api.routers import nutrition, nutrition_admin, upload
from app.api.routers import auth, user_products, admin_train, mcp_info
from app.api.routers import store_catalog
from app.api.routers import store_catalog, batch_catalog
from app.services.store_db import ensure_store_intelligence_schema
from app.services.nutrition_db import ensure_nutrition_schema
@@ -82,6 +83,31 @@ async def lifespan(_app: FastAPI):
except Exception as e:
logger.warning("Could not restore bundled assets: %s", e)
# Cheap, and it must happen before anyone can press Resume: a batch that a
# restart cut short is still marked "running" on disk, and until it is
# reconciled the UI shows it as in flight with nothing behind it. This only
# rewrites manifests - it deliberately starts no work. See
# batch_ingest.scan_interrupted() for why auto-resume is not the default.
try:
from app.core.batch_ingest import scan_interrupted
interrupted = scan_interrupted()
if interrupted:
logger.info("Marked %d interrupted catalog batch(es).", len(interrupted))
if interrupted and BATCH_AUTO_RESUME:
# Opt-in only. Importing the worker here rather than at module
# scope keeps the queue and its thread out of a boot that never
# needs them.
from app.core import batch_worker
for batch_id in interrupted:
try:
batch_worker.submit(batch_id)
except Exception as exc: # noqa: BLE001 - a full queue is not fatal
logger.warning("Could not auto-resume batch %s: %s", batch_id, exc)
except Exception as e:
logger.warning("Could not reconcile interrupted catalog batches: %s", e)
def _async_init():
try:
ensure_store_intelligence_schema()
@@ -266,6 +292,7 @@ app.include_router(nutrition.router, prefix="/api")
app.include_router(nutrition_admin.router, prefix="/api")
app.include_router(upload.router, prefix="/api")
app.include_router(store_catalog.router, prefix="/api")
app.include_router(batch_catalog.router, prefix="/api")
app.include_router(mcp_info.router, prefix="/api")
# MCP lives outside /api on purpose: it is a protocol endpoint for AI clients,