Backend- file ingestion API Updates
This commit is contained in:
85
app/api/batch_job_store.py
Normal file
85
app/api/batch_job_store.py
Normal file
@@ -0,0 +1,85 @@
|
||||
"""Live view of the batches this process knows about.
|
||||
|
||||
Same pattern and the same documented trade-offs as `store_catalog_job_store.py`
|
||||
and its siblings: a process-local dict behind a lock, not shared across uvicorn
|
||||
workers. Adding a broker for this would be operational weight the project has
|
||||
already decided against (see `job_store.py`).
|
||||
|
||||
WHAT IS DIFFERENT HERE, AND WHY IT STILL EARNS ITS PLACE
|
||||
--------------------------------------------------------
|
||||
Unlike the other job stores, this one is not the only record. `manifest.json`
|
||||
on the volume is the durable truth; this is a cache in front of it, and it
|
||||
exists for one reason: the UI polls every 3 seconds while row-level progress
|
||||
ticks many times a second. Serving those polls from memory keeps both the disk
|
||||
writes and the read path off the hot loop. A cache miss is not a 404 - `get()`
|
||||
falls back to reading the manifest, so a batch from before the last restart is
|
||||
still visible.
|
||||
|
||||
Cancellation lives here too rather than on disk. It is a request about the run
|
||||
in flight, and the run in flight is in this process.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
from app.core import batch_ingest
|
||||
|
||||
|
||||
class BatchJobStore:
|
||||
def __init__(self) -> None:
|
||||
self._batches: Dict[str, batch_ingest.BatchManifest] = {}
|
||||
self._cancelled: set = set()
|
||||
self._lock = threading.Lock()
|
||||
|
||||
def put(self, manifest: batch_ingest.BatchManifest) -> None:
|
||||
"""Record (or refresh) a batch. This is the `on_change` callback."""
|
||||
with self._lock:
|
||||
self._batches[manifest.batch_id] = manifest
|
||||
|
||||
def get(self, batch_id: str) -> Optional[batch_ingest.BatchManifest]:
|
||||
"""Live state if we have it, otherwise whatever is on disk."""
|
||||
with self._lock:
|
||||
cached = self._batches.get(batch_id)
|
||||
if cached is not None:
|
||||
return cached
|
||||
return batch_ingest.read_manifest(batch_id)
|
||||
|
||||
def recent(self, limit: int = 20) -> List[batch_ingest.BatchManifest]:
|
||||
"""Newest first, merging the live view over the on-disk one.
|
||||
|
||||
Reading the directory rather than only the cache means a restart does
|
||||
not make previous batches vanish from the list.
|
||||
"""
|
||||
with self._lock:
|
||||
live = dict(self._batches)
|
||||
merged: Dict[str, batch_ingest.BatchManifest] = {}
|
||||
for manifest in batch_ingest.list_manifests():
|
||||
merged[manifest.batch_id] = live.get(manifest.batch_id, manifest)
|
||||
for batch_id, manifest in live.items():
|
||||
merged.setdefault(batch_id, manifest)
|
||||
ordered = sorted(merged.values(), key=lambda m: m.created_at, reverse=True)
|
||||
return ordered[: max(1, limit)]
|
||||
|
||||
# -- cancellation --------------------------------------------------------
|
||||
def cancel(self, batch_id: str) -> None:
|
||||
"""Ask the worker to stop before it picks up the next file.
|
||||
|
||||
Nothing interrupts the file already running. Killing a pipeline halfway
|
||||
would leave some of its rows written and the rest not, with no record of
|
||||
where it stopped; letting the current file finish is both simpler and
|
||||
the only version with a defined outcome.
|
||||
"""
|
||||
with self._lock:
|
||||
self._cancelled.add(batch_id)
|
||||
|
||||
def is_cancelled(self, batch_id: str) -> bool:
|
||||
with self._lock:
|
||||
return batch_id in self._cancelled
|
||||
|
||||
def clear_cancel(self, batch_id: str) -> None:
|
||||
with self._lock:
|
||||
self._cancelled.discard(batch_id)
|
||||
|
||||
|
||||
batch_job_store = BatchJobStore()
|
||||
385
app/api/routers/batch_catalog.py
Normal file
385
app/api/routers/batch_catalog.py
Normal file
@@ -0,0 +1,385 @@
|
||||
"""Admin endpoints for ingesting several store spreadsheets as one batch.
|
||||
|
||||
POST /api/admin/catalog-batch/preview - parse only, per file
|
||||
POST /api/admin/catalog-batch/ingest - 202 + batch_id
|
||||
GET /api/admin/catalog-batch/batches - recent batches
|
||||
GET /api/admin/catalog-batch/batches/{id} - poll one batch
|
||||
POST /api/admin/catalog-batch/batches/{id}/resume - after a restart
|
||||
POST /api/admin/catalog-batch/batches/{id}/cancel - stop the rest
|
||||
|
||||
This is the multi-file sibling of `store_catalog.py`, and it deliberately does
|
||||
not replace it: the single-file endpoints are untouched and still work. What is
|
||||
different is the unit of work. Five files are one batch with one id, so the
|
||||
question a colleague actually asks - "did the drop land?" - has one answer
|
||||
rather than five.
|
||||
|
||||
WHY THE NETWORK STAGES DEFAULT OFF HERE
|
||||
---------------------------------------
|
||||
The single-file UI sends `use_llm=true, fetch_images=true`. At one file that is
|
||||
a considered trade. At twenty it is thousands of outbound requests and, for the
|
||||
image stage, a Playwright subprocess that can burn three minutes on its own -
|
||||
on a single-vCPU container that is also serving the API. So a batch opts IN to
|
||||
those stages; it does not opt out. `USE_OLLAMA` is false in production anyway,
|
||||
which makes `use_llm` a no-op there and the honest default obvious.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import queue
|
||||
from typing import List, Optional
|
||||
|
||||
from fastapi import APIRouter, Depends, File, HTTPException, UploadFile, status
|
||||
from pydantic import BaseModel
|
||||
|
||||
from app.api.batch_job_store import batch_job_store
|
||||
from app.api.deps import require_admin
|
||||
from app.core import batch_ingest, batch_worker
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.infrastructure.settings import (
|
||||
BATCH_MAX_FILES,
|
||||
BATCH_MAX_TOTAL_BYTES,
|
||||
BATCH_MAX_TOTAL_ROWS,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
router = APIRouter(prefix="/admin/catalog-batch", tags=["admin", "catalog"])
|
||||
|
||||
# Per-file ceilings match store_catalog.py exactly. A file that is too big for
|
||||
# the single-file endpoint is not somehow acceptable because it arrived with
|
||||
# four friends.
|
||||
MAX_UPLOAD_BYTES = 10 * 1024 * 1024
|
||||
MAX_UPLOAD_ROWS = 2000
|
||||
PREVIEW_ROWS = 10
|
||||
|
||||
# The worker cannot import the API layer without a cycle, so the wiring is done
|
||||
# here, at import, once.
|
||||
batch_worker.configure(
|
||||
on_change=batch_job_store.put,
|
||||
should_cancel=batch_job_store.is_cancelled,
|
||||
)
|
||||
|
||||
|
||||
class BatchFileOut(BaseModel):
|
||||
index: int
|
||||
filename: str
|
||||
status: str
|
||||
detail: Optional[str] = None
|
||||
stage_index: int = 0
|
||||
stage_name: str = ""
|
||||
total_stages: int = pipeline.TOTAL_STAGES
|
||||
rows_done: int = 0
|
||||
rows_total: int = 0
|
||||
size_bytes: int = 0
|
||||
result: Optional[dict] = None
|
||||
|
||||
|
||||
class BatchOut(BaseModel):
|
||||
batch_id: str
|
||||
status: str
|
||||
detail: Optional[str] = None
|
||||
created_at: float
|
||||
updated_at: float
|
||||
files_total: int
|
||||
files_done: int
|
||||
files_failed: int
|
||||
current_file: Optional[str] = None
|
||||
use_llm: bool
|
||||
fetch_images: bool
|
||||
totals: dict
|
||||
brands: List[str]
|
||||
files: List[BatchFileOut]
|
||||
|
||||
|
||||
def _to_out(manifest: batch_ingest.BatchManifest) -> BatchOut:
|
||||
body = manifest.to_dict()
|
||||
body["files"] = [BatchFileOut(**{
|
||||
key: entry[key] for key in BatchFileOut.model_fields if key in entry
|
||||
}) for entry in body["files"]]
|
||||
return BatchOut(**{k: v for k, v in body.items() if k in BatchOut.model_fields})
|
||||
|
||||
|
||||
async def _read_uploads(files: List[UploadFile]) -> List[tuple]:
|
||||
"""Read every upload into memory, enforcing the count and size ceilings.
|
||||
|
||||
Read here rather than in the worker for the same reason `store_catalog.py`
|
||||
gives: `UploadFile` is backed by a temporary file tied to the request, and
|
||||
it is gone before a background thread would reach it.
|
||||
"""
|
||||
if not files:
|
||||
raise HTTPException(status_code=400, detail="No files were uploaded.")
|
||||
if len(files) > BATCH_MAX_FILES:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=(
|
||||
f"{len(files)} files exceeds the {BATCH_MAX_FILES}-file limit for one "
|
||||
f"batch. Split the drop and upload it in two batches."
|
||||
),
|
||||
)
|
||||
|
||||
read: List[tuple] = []
|
||||
total = 0
|
||||
for upload in files:
|
||||
contents = await upload.read()
|
||||
name = upload.filename or "upload.xlsx"
|
||||
if not contents:
|
||||
# Recorded rather than raised - an empty file among nine good ones
|
||||
# is a fact about that file, not a reason to reject the drop.
|
||||
read.append((name, b""))
|
||||
continue
|
||||
if len(contents) > MAX_UPLOAD_BYTES:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=(
|
||||
f"'{name}' is larger than the "
|
||||
f"{MAX_UPLOAD_BYTES // (1024 * 1024)}MB per-file limit."
|
||||
),
|
||||
)
|
||||
total += len(contents)
|
||||
if total > BATCH_MAX_TOTAL_BYTES:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=(
|
||||
f"The batch is larger than the "
|
||||
f"{BATCH_MAX_TOTAL_BYTES // (1024 * 1024)}MB total limit."
|
||||
),
|
||||
)
|
||||
read.append((name, contents))
|
||||
return read
|
||||
|
||||
|
||||
def _parse_all(read: List[tuple]):
|
||||
"""Split the uploads into (valid, invalid) by trying to parse each one.
|
||||
|
||||
Failing fast here is what stops a batch transitioning straight to "failed"
|
||||
a second after it started - the same reasoning as `store_catalog.py:147`,
|
||||
applied per file so that one bad sheet does not condemn the others.
|
||||
"""
|
||||
valid: List[tuple] = []
|
||||
invalid: List[tuple] = []
|
||||
rows_total = 0
|
||||
|
||||
for name, contents in read:
|
||||
if not contents:
|
||||
invalid.append((name, "The file is empty."))
|
||||
continue
|
||||
try:
|
||||
df, _mapping = pipeline.parse_spreadsheet(name, contents)
|
||||
except HTTPException as exc:
|
||||
# read_products_dataframe raises HTTPException for an unsupported
|
||||
# extension or a missing Excel reader; its message already names
|
||||
# the file and what to do about it.
|
||||
invalid.append((name, str(exc.detail)))
|
||||
continue
|
||||
except Exception as exc: # noqa: BLE001 - an unreadable sheet is user error
|
||||
invalid.append((name, f"Could not parse the file: {exc}"))
|
||||
continue
|
||||
|
||||
if df.empty:
|
||||
invalid.append((name, "The file has no data rows."))
|
||||
continue
|
||||
if len(df) > MAX_UPLOAD_ROWS:
|
||||
invalid.append((
|
||||
name,
|
||||
f"{len(df)} rows exceeds the {MAX_UPLOAD_ROWS}-row per-file limit.",
|
||||
))
|
||||
continue
|
||||
|
||||
rows_total += int(len(df))
|
||||
if rows_total > BATCH_MAX_TOTAL_ROWS:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=(
|
||||
f"The batch totals more than {BATCH_MAX_TOTAL_ROWS} rows. "
|
||||
f"Split it and upload in two batches."
|
||||
),
|
||||
)
|
||||
valid.append((name, contents, int(len(df))))
|
||||
|
||||
return valid, invalid, rows_total
|
||||
|
||||
|
||||
@router.post("/preview", dependencies=[Depends(require_admin)])
|
||||
async def preview_catalog_batch(files: List[UploadFile] = File(...)) -> dict:
|
||||
"""Parse every file and report how its columns were understood.
|
||||
|
||||
Nothing is staged and no batch is created. The mapping from a store's own
|
||||
headers onto catalog fields is a guess, and finding out that "Item" was read
|
||||
as the description after twenty files have been scraped is expensive.
|
||||
"""
|
||||
read = await _read_uploads(files)
|
||||
out = []
|
||||
for name, contents in read:
|
||||
if not contents:
|
||||
out.append({"filename": name, "ok": False, "error": "The file is empty."})
|
||||
continue
|
||||
try:
|
||||
df, mapping = pipeline.parse_spreadsheet(name, contents)
|
||||
except HTTPException as exc:
|
||||
out.append({"filename": name, "ok": False, "error": str(exc.detail)})
|
||||
continue
|
||||
except Exception as exc: # noqa: BLE001
|
||||
out.append({"filename": name, "ok": False,
|
||||
"error": f"Could not parse the file: {exc}"})
|
||||
continue
|
||||
|
||||
if df.empty:
|
||||
out.append({"filename": name, "ok": False, "error": "The file has no data rows."})
|
||||
continue
|
||||
|
||||
out.append({
|
||||
"filename": name,
|
||||
"ok": True,
|
||||
"rows_total": int(len(df)),
|
||||
"over_row_limit": bool(len(df) > MAX_UPLOAD_ROWS),
|
||||
"recognised_columns": {f: str(c) for f, c in mapping.columns.items()},
|
||||
"unrecognised_columns": mapping.unrecognised,
|
||||
"brand_column_present": "brand" in mapping.columns,
|
||||
"preview": df.head(PREVIEW_ROWS).fillna("").astype(str).to_dict(orient="records"),
|
||||
})
|
||||
|
||||
return {
|
||||
"files": out,
|
||||
"files_total": len(out),
|
||||
"files_ok": sum(1 for f in out if f.get("ok")),
|
||||
"rows_total": sum(int(f.get("rows_total") or 0) for f in out if f.get("ok")),
|
||||
"stages": list(pipeline.STAGE_NAMES),
|
||||
"limits": {
|
||||
"max_files": BATCH_MAX_FILES,
|
||||
"max_rows_per_file": MAX_UPLOAD_ROWS,
|
||||
"max_rows_total": BATCH_MAX_TOTAL_ROWS,
|
||||
"max_bytes_per_file": MAX_UPLOAD_BYTES,
|
||||
"max_bytes_total": BATCH_MAX_TOTAL_BYTES,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
@router.post("/ingest", status_code=status.HTTP_202_ACCEPTED,
|
||||
dependencies=[Depends(require_admin)])
|
||||
async def ingest_catalog_batch(
|
||||
files: List[UploadFile] = File(...),
|
||||
use_llm: bool = False,
|
||||
fetch_images: bool = False,
|
||||
) -> BatchOut:
|
||||
"""Stage the files, queue the batch, and return an id to poll.
|
||||
|
||||
Returns immediately. In production the browser reaches this through Traefik
|
||||
on a different host to the frontend, so anything that sat on the request
|
||||
path would be racing an idle timeout nobody here controls.
|
||||
"""
|
||||
read = await _read_uploads(files)
|
||||
valid, invalid, _rows = _parse_all(read)
|
||||
|
||||
if not valid:
|
||||
detail = "; ".join(f"{name}: {reason}" for name, reason in invalid)
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"None of the uploaded files could be ingested. {detail}",
|
||||
)
|
||||
|
||||
manifest = batch_ingest.stage_batch(
|
||||
[(name, contents) for name, contents, _n in valid],
|
||||
use_llm=use_llm,
|
||||
fetch_images=fetch_images,
|
||||
invalid=invalid,
|
||||
)
|
||||
for entry, (_name, _contents, rows) in zip(manifest.files, valid):
|
||||
entry.rows_total = rows
|
||||
batch_ingest.write_manifest(manifest)
|
||||
batch_job_store.put(manifest)
|
||||
|
||||
try:
|
||||
batch_worker.submit(manifest.batch_id)
|
||||
except queue.Full:
|
||||
manifest.status = batch_ingest.QUEUED
|
||||
manifest.detail = (
|
||||
"The ingestion queue is full. This batch is staged and can be started "
|
||||
"with Resume once the running batches finish."
|
||||
)
|
||||
batch_ingest.write_manifest(manifest)
|
||||
batch_job_store.put(manifest)
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail=(
|
||||
"Too many batches are already queued. This one has been saved - "
|
||||
"press Resume on it once the current batch finishes."
|
||||
),
|
||||
)
|
||||
|
||||
return _to_out(manifest)
|
||||
|
||||
|
||||
@router.get("/batches", dependencies=[Depends(require_admin)])
|
||||
def list_catalog_batches(limit: int = 20) -> dict:
|
||||
limit = max(1, min(limit, 100))
|
||||
return {"batches": [_to_out(m) for m in batch_job_store.recent(limit)]}
|
||||
|
||||
|
||||
@router.get("/batches/{batch_id}", dependencies=[Depends(require_admin)])
|
||||
def get_catalog_batch(batch_id: str) -> BatchOut:
|
||||
manifest = batch_job_store.get(batch_id)
|
||||
if not manifest:
|
||||
raise HTTPException(status_code=404, detail="Batch not found")
|
||||
return _to_out(manifest)
|
||||
|
||||
|
||||
@router.post("/batches/{batch_id}/resume", dependencies=[Depends(require_admin)])
|
||||
def resume_catalog_batch(batch_id: str) -> BatchOut:
|
||||
"""Re-queue a batch a restart cut short, or one that was queued behind a full queue."""
|
||||
manifest = batch_ingest.read_manifest(batch_id)
|
||||
if not manifest:
|
||||
raise HTTPException(status_code=404, detail="Batch not found")
|
||||
|
||||
pending = [f for f in manifest.files if f.status == batch_ingest.QUEUED]
|
||||
if not pending:
|
||||
raise HTTPException(
|
||||
status_code=409,
|
||||
detail=f"Nothing left to run in this batch (status: {manifest.status}).",
|
||||
)
|
||||
|
||||
batch_job_store.clear_cancel(batch_id)
|
||||
manifest.status = batch_ingest.QUEUED
|
||||
manifest.detail = None
|
||||
batch_ingest.write_manifest(manifest)
|
||||
batch_job_store.put(manifest)
|
||||
|
||||
try:
|
||||
batch_worker.submit(batch_id)
|
||||
except queue.Full:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail="Too many batches are already queued. Try again shortly.",
|
||||
)
|
||||
return _to_out(manifest)
|
||||
|
||||
|
||||
@router.post("/batches/{batch_id}/cancel", dependencies=[Depends(require_admin)])
|
||||
def cancel_catalog_batch(batch_id: str) -> BatchOut:
|
||||
"""Stop before the next file. The file already running is allowed to finish.
|
||||
|
||||
Interrupting a pipeline mid-file would leave some of its rows written and
|
||||
the rest not, with nothing recording where it stopped. Letting the current
|
||||
file complete is the only version of "cancel" with a defined outcome.
|
||||
"""
|
||||
manifest = batch_job_store.get(batch_id)
|
||||
if not manifest:
|
||||
raise HTTPException(status_code=404, detail="Batch not found")
|
||||
|
||||
batch_job_store.cancel(batch_id)
|
||||
|
||||
# A batch that has not started yet has no worker to notice the flag, so
|
||||
# cancel it here and be done.
|
||||
on_disk = batch_ingest.read_manifest(batch_id)
|
||||
if on_disk and on_disk.status in {batch_ingest.QUEUED, batch_ingest.INTERRUPTED}:
|
||||
for entry in on_disk.files:
|
||||
if entry.status == batch_ingest.QUEUED:
|
||||
entry.status = batch_ingest.CANCELLED
|
||||
entry.detail = "Cancelled before this file started."
|
||||
on_disk.settle()
|
||||
on_disk.detail = "Cancelled."
|
||||
batch_ingest.write_manifest(on_disk)
|
||||
batch_job_store.put(on_disk)
|
||||
return _to_out(on_disk)
|
||||
|
||||
manifest.detail = "Cancelling - the file currently running will finish first."
|
||||
batch_job_store.put(manifest)
|
||||
return _to_out(manifest)
|
||||
520
app/core/batch_ingest.py
Normal file
520
app/core/batch_ingest.py
Normal file
@@ -0,0 +1,520 @@
|
||||
"""
|
||||
Multi-file store spreadsheets -> the same 11-stage pipeline -> brand tables.
|
||||
|
||||
WHAT THIS ADDS OVER `store_catalog_pipeline`
|
||||
--------------------------------------------
|
||||
Nothing about the pipeline itself. `run_pipeline()` already turns one whole
|
||||
spreadsheet into catalog rows, and this module calls it unchanged, once per
|
||||
file. What is new is everything *around* a file:
|
||||
|
||||
* several files are one unit of work with one id, so "did the whole drop
|
||||
land?" has an answer;
|
||||
* the uploads are written to disk before any work starts, so a restart
|
||||
mid-batch loses nothing but time;
|
||||
* one file failing does not take the others with it.
|
||||
|
||||
WHY THE FILES GO TO DISK
|
||||
------------------------
|
||||
The single-file path reads the upload into memory and hands the bytes to a
|
||||
daemon thread (store_catalog.py). That is fine for one file and one operator
|
||||
watching it: if the process dies, they re-upload. A five-file batch is a
|
||||
different proposition - the colleague who sent them is not sitting there, and
|
||||
silently losing the drop is worse than any amount of extra code. So the bytes
|
||||
are staged under BATCH_UPLOAD_DIR, which is on the container's declared volume,
|
||||
and a manifest records what state each file reached.
|
||||
|
||||
THIS MODULE IS THE SHARED CORE
|
||||
------------------------------
|
||||
Two callers, one implementation:
|
||||
|
||||
app/api/routers/batch_catalog.py -> production (worker thread)
|
||||
orchestration/assets/batch_catalog.py -> Dagster (development)
|
||||
|
||||
That is the same arrangement `orchestration/assets/catalog.py` already
|
||||
describes for the brand pipeline: "Wrapping rather than reimplementing is what
|
||||
stops the two paths drifting: a fix to a stage fixes both." Dagster is not
|
||||
deployed here and is not on any request path; it wraps these functions so the
|
||||
graph is inspectable and re-runnable locally, and production executes the very
|
||||
same code without paying for a daemon.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import time
|
||||
import uuid
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable, Dict, List, Optional, Tuple
|
||||
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.infrastructure.settings import (
|
||||
BATCH_AUTO_RESUME,
|
||||
BATCH_RETENTION_DAYS,
|
||||
BATCH_UPLOAD_DIR,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
MANIFEST_NAME = "manifest.json"
|
||||
|
||||
# File states. `cancelled` only ever applies to files that had not started.
|
||||
QUEUED = "queued"
|
||||
RUNNING = "running"
|
||||
DONE = "done"
|
||||
FAILED = "failed"
|
||||
CANCELLED = "cancelled"
|
||||
|
||||
# Batch states. `partial` is not cosmetic: a batch where four of five files
|
||||
# landed must not read as a flat success, or nobody goes looking for the fifth.
|
||||
INTERRUPTED = "interrupted"
|
||||
PARTIAL = "partial"
|
||||
|
||||
TERMINAL_BATCH_STATES = {DONE, FAILED, PARTIAL, CANCELLED}
|
||||
|
||||
# `..`, separators and drive letters all stripped. UploadFile.filename is
|
||||
# attacker-controlled in the general case, and it is used to build a path.
|
||||
_UNSAFE = re.compile(r"[^A-Za-z0-9._-]+")
|
||||
|
||||
|
||||
def _safe_name(filename: str) -> str:
|
||||
"""A filename that cannot escape the batch directory.
|
||||
|
||||
Path components are discarded rather than escaped: nothing downstream needs
|
||||
the original directory, and `os.path.basename` alone is not enough here,
|
||||
because a Windows-authored name like "..\\evil.csv" keeps its backslash on
|
||||
a Linux container and basename leaves it untouched.
|
||||
"""
|
||||
base = str(filename or "upload.xlsx").replace("\\", "/").rsplit("/", 1)[-1]
|
||||
base = _UNSAFE.sub("_", base).lstrip(".") or "upload.xlsx"
|
||||
return base[:120]
|
||||
|
||||
|
||||
@dataclass
|
||||
class BatchFile:
|
||||
"""One spreadsheet inside a batch, and how far it got."""
|
||||
|
||||
index: int
|
||||
filename: str # what the colleague called it, for display
|
||||
stored_name: str # what it is called on disk
|
||||
size_bytes: int = 0
|
||||
status: str = QUEUED
|
||||
detail: Optional[str] = None
|
||||
stage_index: int = 0 # 1-based; 0 while queued
|
||||
stage_name: str = ""
|
||||
total_stages: int = pipeline.TOTAL_STAGES
|
||||
rows_done: int = 0
|
||||
rows_total: int = 0
|
||||
result: Optional[Dict[str, Any]] = None
|
||||
started_at: Optional[float] = None
|
||||
finished_at: Optional[float] = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class BatchManifest:
|
||||
"""The whole batch. Serialised to manifest.json verbatim."""
|
||||
|
||||
batch_id: str
|
||||
status: str = QUEUED
|
||||
use_llm: bool = False
|
||||
fetch_images: bool = False
|
||||
created_at: float = field(default_factory=time.time)
|
||||
updated_at: float = field(default_factory=time.time)
|
||||
detail: Optional[str] = None
|
||||
files: List[BatchFile] = field(default_factory=list)
|
||||
|
||||
# -- derived, recomputed rather than stored, so they cannot drift ---------
|
||||
@property
|
||||
def files_total(self) -> int:
|
||||
return len(self.files)
|
||||
|
||||
@property
|
||||
def files_done(self) -> int:
|
||||
return sum(1 for f in self.files if f.status == DONE)
|
||||
|
||||
@property
|
||||
def files_failed(self) -> int:
|
||||
return sum(1 for f in self.files if f.status == FAILED)
|
||||
|
||||
@property
|
||||
def current_file(self) -> Optional[str]:
|
||||
for entry in self.files:
|
||||
if entry.status == RUNNING:
|
||||
return entry.filename
|
||||
return None
|
||||
|
||||
def totals(self) -> Dict[str, int]:
|
||||
"""Summed across every file that produced a result."""
|
||||
keys = ("rows_total", "products_built", "inserted", "backfilled",
|
||||
"skipped_existing", "rejected", "error_count")
|
||||
out = {key: 0 for key in keys}
|
||||
for entry in self.files:
|
||||
for key in keys:
|
||||
out[key] += int((entry.result or {}).get(key) or 0)
|
||||
return out
|
||||
|
||||
def brands(self) -> List[str]:
|
||||
seen = set()
|
||||
for entry in self.files:
|
||||
seen.update((entry.result or {}).get("brands") or [])
|
||||
return sorted(seen)
|
||||
|
||||
def settle(self) -> str:
|
||||
"""Recompute the batch status from its files. Returns the new status."""
|
||||
states = {f.status for f in self.files}
|
||||
if states & {QUEUED, RUNNING}:
|
||||
self.status = RUNNING if RUNNING in states else QUEUED
|
||||
elif self.files_done and self.files_failed:
|
||||
self.status = PARTIAL
|
||||
elif self.files_done:
|
||||
self.status = DONE
|
||||
elif self.files_failed:
|
||||
self.status = FAILED
|
||||
else:
|
||||
self.status = CANCELLED
|
||||
self.updated_at = time.time()
|
||||
return self.status
|
||||
|
||||
# -- serialisation -------------------------------------------------------
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {
|
||||
"batch_id": self.batch_id,
|
||||
"status": self.status,
|
||||
"use_llm": self.use_llm,
|
||||
"fetch_images": self.fetch_images,
|
||||
"created_at": self.created_at,
|
||||
"updated_at": self.updated_at,
|
||||
"detail": self.detail,
|
||||
"files_total": self.files_total,
|
||||
"files_done": self.files_done,
|
||||
"files_failed": self.files_failed,
|
||||
"current_file": self.current_file,
|
||||
"totals": self.totals(),
|
||||
"brands": self.brands(),
|
||||
"files": [asdict(f) for f in self.files],
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, raw: Dict[str, Any]) -> "BatchManifest":
|
||||
allowed = set(BatchFile.__dataclass_fields__)
|
||||
files = [
|
||||
BatchFile(**{k: v for k, v in entry.items() if k in allowed})
|
||||
for entry in (raw.get("files") or [])
|
||||
]
|
||||
return cls(
|
||||
batch_id=raw["batch_id"],
|
||||
status=raw.get("status", QUEUED),
|
||||
use_llm=bool(raw.get("use_llm", False)),
|
||||
fetch_images=bool(raw.get("fetch_images", False)),
|
||||
created_at=float(raw.get("created_at") or time.time()),
|
||||
updated_at=float(raw.get("updated_at") or time.time()),
|
||||
detail=raw.get("detail"),
|
||||
files=files,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Disk layout
|
||||
# ---------------------------------------------------------------------------
|
||||
def batch_root() -> Path:
|
||||
"""Read at call time, not import time, so tests can repoint the directory."""
|
||||
return Path(BATCH_UPLOAD_DIR)
|
||||
|
||||
|
||||
def batch_dir(batch_id: str) -> Path:
|
||||
# The id is generated here (uuid4), never taken from a request, but this is
|
||||
# still the function that turns it into a path - so it validates.
|
||||
if not re.fullmatch(r"[A-Za-z0-9_-]{1,64}", batch_id or ""):
|
||||
raise ValueError("Invalid batch id: {!r}".format(batch_id))
|
||||
return batch_root() / batch_id
|
||||
|
||||
|
||||
def manifest_path(batch_id: str) -> Path:
|
||||
return batch_dir(batch_id) / MANIFEST_NAME
|
||||
|
||||
|
||||
def write_manifest(manifest: BatchManifest) -> None:
|
||||
"""Write via a temp file and os.replace.
|
||||
|
||||
A half-written manifest.json is indistinguishable from a corrupt one on the
|
||||
next boot, and the recovery path reads every manifest it finds.
|
||||
"""
|
||||
target = manifest_path(manifest.batch_id)
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = target.with_name(MANIFEST_NAME + ".tmp")
|
||||
tmp.write_text(json.dumps(manifest.to_dict(), indent=2), encoding="utf-8")
|
||||
os.replace(tmp, target)
|
||||
|
||||
|
||||
def read_manifest(batch_id: str) -> Optional[BatchManifest]:
|
||||
try:
|
||||
path = manifest_path(batch_id)
|
||||
except ValueError:
|
||||
return None
|
||||
if not path.exists():
|
||||
return None
|
||||
try:
|
||||
return BatchManifest.from_dict(json.loads(path.read_text(encoding="utf-8")))
|
||||
except Exception as exc: # noqa: BLE001 - a bad manifest must not break a listing
|
||||
logger.warning("Ignoring unreadable manifest %s: %s", path, exc)
|
||||
return None
|
||||
|
||||
|
||||
def list_manifests() -> List[BatchManifest]:
|
||||
"""Every readable batch on disk, newest first."""
|
||||
root = batch_root()
|
||||
if not root.exists():
|
||||
return []
|
||||
found = []
|
||||
for child in sorted(root.iterdir()):
|
||||
if not child.is_dir():
|
||||
continue
|
||||
manifest = read_manifest(child.name)
|
||||
if manifest:
|
||||
found.append(manifest)
|
||||
return sorted(found, key=lambda m: m.created_at, reverse=True)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Staging
|
||||
# ---------------------------------------------------------------------------
|
||||
def stage_batch(
|
||||
uploads: List[Tuple[str, bytes]],
|
||||
*,
|
||||
use_llm: bool = False,
|
||||
fetch_images: bool = False,
|
||||
invalid: Optional[List[Tuple[str, str]]] = None,
|
||||
) -> BatchManifest:
|
||||
"""Write the uploads to disk and return the manifest describing them.
|
||||
|
||||
`invalid` carries files the caller already rejected (unparseable, empty).
|
||||
They are recorded as failed members of the batch rather than dropped: an
|
||||
operator who selected six files and sees five must be told what happened to
|
||||
the sixth, and the batch page is the only place they will look.
|
||||
"""
|
||||
batch_id = uuid.uuid4().hex
|
||||
directory = batch_dir(batch_id)
|
||||
directory.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
manifest = BatchManifest(batch_id=batch_id, use_llm=use_llm, fetch_images=fetch_images)
|
||||
|
||||
position = 0
|
||||
for filename, content in uploads:
|
||||
stored = "{:02d}_{}".format(position, _safe_name(filename))
|
||||
(directory / stored).write_bytes(content)
|
||||
manifest.files.append(
|
||||
BatchFile(
|
||||
index=position,
|
||||
filename=filename or stored,
|
||||
stored_name=stored,
|
||||
size_bytes=len(content),
|
||||
)
|
||||
)
|
||||
position += 1
|
||||
|
||||
for filename, reason in invalid or []:
|
||||
manifest.files.append(
|
||||
BatchFile(
|
||||
index=position,
|
||||
filename=filename or "(unnamed)",
|
||||
stored_name="",
|
||||
status=FAILED,
|
||||
detail=reason,
|
||||
finished_at=time.time(),
|
||||
)
|
||||
)
|
||||
position += 1
|
||||
|
||||
write_manifest(manifest)
|
||||
return manifest
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Execution
|
||||
# ---------------------------------------------------------------------------
|
||||
OnChange = Callable[["BatchManifest"], None]
|
||||
"""Called after every file-level transition, and on row progress.
|
||||
|
||||
The callback is what keeps the in-memory view the API serves in step with the
|
||||
run. It must be cheap - it fires on every progress tick - which is why the
|
||||
manifest is NOT written to disk from inside it.
|
||||
"""
|
||||
|
||||
|
||||
def _noop_change(manifest: "BatchManifest") -> None:
|
||||
return None
|
||||
|
||||
|
||||
# Row-level progress arrives many times a second. Persisting each one would be
|
||||
# hundreds of small writes per file, to record something nobody reads back from
|
||||
# disk anyway - the API answers polls from memory. Disk is for surviving a
|
||||
# restart, and a restart only needs to know which FILE was in flight.
|
||||
_PROGRESS_FLUSH_SECONDS = 5.0
|
||||
|
||||
|
||||
def run_batch(
|
||||
batch_id: str,
|
||||
*,
|
||||
on_change: OnChange = _noop_change,
|
||||
should_cancel: Optional[Callable[[], bool]] = None,
|
||||
) -> BatchManifest:
|
||||
"""Run every queued file in the batch, in order, one at a time.
|
||||
|
||||
Files are independent. A file that raises is marked failed with the reason
|
||||
and the loop moves to the next one - the pipeline's own "a stage never
|
||||
raises" rule protects rows within a file, and this is the same idea one
|
||||
level up.
|
||||
"""
|
||||
manifest = read_manifest(batch_id)
|
||||
if manifest is None:
|
||||
raise FileNotFoundError("No manifest for batch {}".format(batch_id))
|
||||
|
||||
directory = batch_dir(batch_id)
|
||||
manifest.status = RUNNING
|
||||
manifest.detail = None
|
||||
manifest.updated_at = time.time()
|
||||
write_manifest(manifest)
|
||||
on_change(manifest)
|
||||
|
||||
for entry in manifest.files:
|
||||
if entry.status != QUEUED:
|
||||
continue
|
||||
|
||||
if should_cancel is not None and should_cancel():
|
||||
entry.status = CANCELLED
|
||||
entry.detail = "Cancelled before this file started."
|
||||
entry.finished_at = time.time()
|
||||
continue
|
||||
|
||||
entry.status = RUNNING
|
||||
entry.started_at = time.time()
|
||||
entry.stage_index = 0
|
||||
entry.stage_name = ""
|
||||
write_manifest(manifest)
|
||||
on_change(manifest)
|
||||
|
||||
last_flush = [time.time()]
|
||||
|
||||
def progress(stage_index: int, stage_name: str, done: int, total: int,
|
||||
_entry: BatchFile = entry) -> None:
|
||||
_entry.stage_index = stage_index
|
||||
_entry.stage_name = stage_name
|
||||
_entry.rows_done = done
|
||||
_entry.rows_total = total
|
||||
manifest.updated_at = time.time()
|
||||
on_change(manifest)
|
||||
now = time.time()
|
||||
if now - last_flush[0] >= _PROGRESS_FLUSH_SECONDS:
|
||||
last_flush[0] = now
|
||||
write_manifest(manifest)
|
||||
|
||||
try:
|
||||
content = (directory / entry.stored_name).read_bytes()
|
||||
result = pipeline.run_pipeline(
|
||||
entry.filename,
|
||||
content,
|
||||
progress=progress,
|
||||
use_llm=manifest.use_llm,
|
||||
fetch_images=manifest.fetch_images,
|
||||
)
|
||||
body = result.as_dict()
|
||||
entry.result = body
|
||||
if body.get("storage_error"):
|
||||
# Same judgement as the single-file path: rows were built but
|
||||
# none reached the database, and calling that success would
|
||||
# leave the operator believing the catalog changed.
|
||||
entry.status = FAILED
|
||||
entry.detail = (
|
||||
"Built {} row(s) but storing them failed: {}".format(
|
||||
body["products_built"], body["storage_error"]
|
||||
)
|
||||
)
|
||||
else:
|
||||
entry.status = DONE
|
||||
entry.detail = (
|
||||
"{} inserted, {} backfilled, {} unchanged, {} rejected".format(
|
||||
body["inserted"], body["backfilled"],
|
||||
body["skipped_existing"], body["rejected"],
|
||||
)
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 - one bad file must not end the batch
|
||||
logger.exception("Batch %s: file %s failed", batch_id, entry.filename)
|
||||
entry.status = FAILED
|
||||
entry.detail = str(exc)
|
||||
|
||||
entry.finished_at = time.time()
|
||||
write_manifest(manifest)
|
||||
on_change(manifest)
|
||||
|
||||
manifest.settle()
|
||||
write_manifest(manifest)
|
||||
on_change(manifest)
|
||||
return manifest
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Recovery and retention
|
||||
# ---------------------------------------------------------------------------
|
||||
def scan_interrupted() -> List[str]:
|
||||
"""Mark batches that a restart cut short. Returns the ids resumable now.
|
||||
|
||||
DELIBERATELY DOES NOT RE-RUN ANYTHING by default. A container caught in a
|
||||
restart loop would otherwise re-enter the heaviest work in the application
|
||||
on every boot, turning a slow start into an unrecoverable one. The files and
|
||||
the manifest are on the volume, so nothing is lost by waiting for a human to
|
||||
press Resume - and BATCH_AUTO_RESUME=true is there for a deployment that has
|
||||
earned the trust.
|
||||
"""
|
||||
resumable: List[str] = []
|
||||
for manifest in list_manifests():
|
||||
if manifest.status in TERMINAL_BATCH_STATES or manifest.status == INTERRUPTED:
|
||||
continue
|
||||
for entry in manifest.files:
|
||||
if entry.status == RUNNING:
|
||||
entry.status = QUEUED
|
||||
entry.stage_index = 0
|
||||
entry.stage_name = ""
|
||||
entry.rows_done = 0
|
||||
entry.detail = "Interrupted by a restart; queued again."
|
||||
manifest.status = INTERRUPTED
|
||||
manifest.detail = "Interrupted by a restart. Press Resume to continue."
|
||||
manifest.updated_at = time.time()
|
||||
write_manifest(manifest)
|
||||
resumable.append(manifest.batch_id)
|
||||
|
||||
if resumable and not BATCH_AUTO_RESUME:
|
||||
logger.info(
|
||||
"Found %d interrupted batch(es); leaving them for a manual resume "
|
||||
"(BATCH_AUTO_RESUME is false).", len(resumable),
|
||||
)
|
||||
return resumable
|
||||
|
||||
|
||||
def purge_expired(now: Optional[float] = None) -> List[str]:
|
||||
"""Delete staged files for batches older than BATCH_RETENTION_DAYS.
|
||||
|
||||
Called when the worker goes idle, never on a request path - deleting a few
|
||||
hundred megabytes should not be something an operator waits on.
|
||||
"""
|
||||
if BATCH_RETENTION_DAYS <= 0:
|
||||
return []
|
||||
cutoff = (now if now is not None else time.time()) - BATCH_RETENTION_DAYS * 86400
|
||||
removed: List[str] = []
|
||||
for manifest in list_manifests():
|
||||
if manifest.created_at >= cutoff:
|
||||
continue
|
||||
if manifest.status not in TERMINAL_BATCH_STATES:
|
||||
# An old batch still queued is a bug somewhere, but deleting the
|
||||
# only copy of its input is not the way to find out.
|
||||
continue
|
||||
try:
|
||||
shutil.rmtree(batch_dir(manifest.batch_id))
|
||||
removed.append(manifest.batch_id)
|
||||
except OSError as exc:
|
||||
logger.warning("Could not purge batch %s: %s", manifest.batch_id, exc)
|
||||
if removed:
|
||||
logger.info("Purged %d expired batch upload(s).", len(removed))
|
||||
return removed
|
||||
122
app/core/batch_worker.py
Normal file
122
app/core/batch_worker.py
Normal file
@@ -0,0 +1,122 @@
|
||||
"""One worker thread for every batch this process will ever run.
|
||||
|
||||
WHY A SINGLE BOUNDED WORKER, AND NOT `run_in_background`
|
||||
---------------------------------------------------------
|
||||
`app/api/background.py` starts a fresh daemon thread per job and returns. That
|
||||
is right for the jobs it serves - they are started by hand, one at a time, by
|
||||
an operator watching the result. It is the wrong shape for this feature.
|
||||
|
||||
A batch is up to twenty spreadsheets of two thousand rows. The deployment this
|
||||
runs on is a single container with one vCPU and no CPU limit, so nothing above
|
||||
stops two batches from interleaving; they would simply both run, at half speed
|
||||
each, while the API tries to answer requests and the health probe tries to get
|
||||
a socket. Two operators uploading at the same time is not an unusual event, it
|
||||
is a Tuesday.
|
||||
|
||||
So: one thread, one batch at a time, and a bounded queue in front. A second
|
||||
batch waits its turn instead of competing, and beyond `BATCH_QUEUE_MAX` the
|
||||
endpoint says 429 rather than accepting work it has no intention of starting.
|
||||
|
||||
THE THREAD IS STARTED LAZILY
|
||||
----------------------------
|
||||
Not at import, not in the lifespan startup. A container that never receives a
|
||||
batch pays nothing for this module beyond the import, which is why `submit()`
|
||||
is the only thing that can bring the worker to life. Boot cost on this host is
|
||||
already ~21s of imports and is the thing most worth not adding to.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import queue
|
||||
import threading
|
||||
from typing import Callable, Optional
|
||||
|
||||
from app.core import batch_ingest
|
||||
from app.infrastructure.settings import BATCH_QUEUE_MAX
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
QueueFull = queue.Full
|
||||
|
||||
_queue: "queue.Queue[str]" = queue.Queue(maxsize=max(1, BATCH_QUEUE_MAX))
|
||||
_worker: Optional[threading.Thread] = None
|
||||
_lock = threading.Lock()
|
||||
|
||||
# Set by the router at import so this module does not import the API layer -
|
||||
# app.api.batch_job_store already imports app.core.batch_ingest, and closing
|
||||
# that loop the other way would be a circular import at boot.
|
||||
_on_change: Optional[Callable[[batch_ingest.BatchManifest], None]] = None
|
||||
_should_cancel: Optional[Callable[[str], bool]] = None
|
||||
|
||||
|
||||
def configure(
|
||||
*,
|
||||
on_change: Callable[[batch_ingest.BatchManifest], None],
|
||||
should_cancel: Callable[[str], bool],
|
||||
) -> None:
|
||||
"""Wire the worker to the job store. Called once, by the router module."""
|
||||
global _on_change, _should_cancel
|
||||
_on_change = on_change
|
||||
_should_cancel = should_cancel
|
||||
|
||||
|
||||
def queue_depth() -> int:
|
||||
return _queue.qsize()
|
||||
|
||||
|
||||
def is_running() -> bool:
|
||||
return _worker is not None and _worker.is_alive()
|
||||
|
||||
|
||||
def submit(batch_id: str) -> None:
|
||||
"""Enqueue a batch and make sure the worker exists. Raises `queue.Full`.
|
||||
|
||||
`put_nowait` rather than `put`: blocking here would block the request
|
||||
handler, which is the one thing an endpoint that returns 202 must never do.
|
||||
"""
|
||||
_queue.put_nowait(batch_id)
|
||||
_ensure_worker()
|
||||
|
||||
|
||||
def _ensure_worker() -> None:
|
||||
global _worker
|
||||
with _lock:
|
||||
if _worker is not None and _worker.is_alive():
|
||||
return
|
||||
_worker = threading.Thread(target=_loop, name="catalog-batch-worker", daemon=True)
|
||||
_worker.start()
|
||||
|
||||
|
||||
def _loop() -> None:
|
||||
"""Drain the queue forever.
|
||||
|
||||
Every iteration is wrapped, because a worker that dies on one bad batch
|
||||
would leave every future batch queued behind a thread that is not there -
|
||||
a failure that looks, from the UI, exactly like a batch that is merely slow.
|
||||
"""
|
||||
while True:
|
||||
batch_id = _queue.get()
|
||||
try:
|
||||
_run_one(batch_id)
|
||||
except Exception: # noqa: BLE001 - see docstring
|
||||
logger.exception("Batch worker: unhandled error on batch %s", batch_id)
|
||||
finally:
|
||||
_queue.task_done()
|
||||
|
||||
if _queue.empty():
|
||||
# Retention runs when there is nothing waiting, so deleting old
|
||||
# uploads never delays a batch and never sits on a request path.
|
||||
try:
|
||||
batch_ingest.purge_expired()
|
||||
except Exception: # noqa: BLE001 - housekeeping must not kill the worker
|
||||
logger.exception("Batch worker: purge failed")
|
||||
|
||||
|
||||
def _run_one(batch_id: str) -> None:
|
||||
on_change = _on_change or (lambda manifest: None)
|
||||
cancelled = _should_cancel or (lambda _id: False)
|
||||
batch_ingest.run_batch(
|
||||
batch_id,
|
||||
on_change=on_change,
|
||||
should_cancel=lambda: cancelled(batch_id),
|
||||
)
|
||||
@@ -134,6 +134,41 @@ MODEL_ARTIFACTS_DIR = _dir(
|
||||
"MODEL_ARTIFACTS_DIR", _BACKEND_ROOT / "app" / "intelligence" / "artifacts"
|
||||
)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Batch catalog ingestion (multi-file upload -> the 11-stage pipeline)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Staged uploads live under DATA_DIR because that path is already a declared
|
||||
# volume (backend/Dockerfile). A batch that survives a container restart is the
|
||||
# whole point of writing the files down instead of holding them in the worker
|
||||
# thread the way the single-file path does.
|
||||
#
|
||||
# Every ceiling below is enforced in the application, not at the proxy. In
|
||||
# production the browser calls mcp.nearle.ai.in directly, so neither nginx's
|
||||
# client_max_body_size nor Caddy's request_body cap is in front of these
|
||||
# endpoints - whatever Traefik defaults to is, and it is not ours to rely on.
|
||||
BATCH_UPLOAD_DIR = _dir("BATCH_UPLOAD_DIR", DATA_DIR / "batch_uploads")
|
||||
|
||||
# Per-file limits stay at the single-upload values (10MB / 2000 rows, see
|
||||
# app/api/routers/store_catalog.py); these bound the BATCH on top of that.
|
||||
BATCH_MAX_FILES = int(os.getenv("BATCH_MAX_FILES", "20"))
|
||||
BATCH_MAX_TOTAL_BYTES = int(os.getenv("BATCH_MAX_TOTAL_BYTES", str(50 * 1024 * 1024)))
|
||||
BATCH_MAX_TOTAL_ROWS = int(os.getenv("BATCH_MAX_TOTAL_ROWS", "20000"))
|
||||
|
||||
# Batches waiting behind the one running. Past this the endpoint returns 429
|
||||
# rather than accepting work it has no intention of starting soon.
|
||||
BATCH_QUEUE_MAX = int(os.getenv("BATCH_QUEUE_MAX", "4"))
|
||||
|
||||
# Staged files are deleted this many days after the batch was created. Without
|
||||
# this the upload directory only grows, on a host whose disk is the scarcest
|
||||
# resource it has.
|
||||
BATCH_RETENTION_DAYS = int(os.getenv("BATCH_RETENTION_DAYS", "7"))
|
||||
|
||||
# Deliberately false. A batch interrupted by a restart is marked "interrupted"
|
||||
# and waits for someone to press Resume. Auto-resuming would mean a container
|
||||
# stuck in a restart loop re-runs the heaviest work in the app on every boot,
|
||||
# which is precisely how a slow start turns into an unrecoverable spiral.
|
||||
BATCH_AUTO_RESUME = _bool("BATCH_AUTO_RESUME", "false")
|
||||
|
||||
# Pristine copies of the bundled seed catalogs and pre-trained models, placed
|
||||
# here by the Dockerfile at a path that is never itself mounted over.
|
||||
#
|
||||
|
||||
29
app/main.py
29
app/main.py
@@ -22,6 +22,7 @@ from fastapi.responses import FileResponse
|
||||
from app.infrastructure.persistence import restore_bundled_assets
|
||||
from app.infrastructure.settings import (
|
||||
API_CORS_ORIGINS,
|
||||
BATCH_AUTO_RESUME,
|
||||
BRAND_SYNC_INTERVAL_SECONDS,
|
||||
cleaned_env_names,
|
||||
)
|
||||
@@ -30,7 +31,7 @@ from app.api.routers import health, brands, search, suggest, chat, catalog, syst
|
||||
from app.api.routers import stores, discounts, analytics as store_analytics, trending, recommendations, store_admin
|
||||
from app.api.routers import nutrition, nutrition_admin, upload
|
||||
from app.api.routers import auth, user_products, admin_train, mcp_info
|
||||
from app.api.routers import store_catalog
|
||||
from app.api.routers import store_catalog, batch_catalog
|
||||
from app.services.store_db import ensure_store_intelligence_schema
|
||||
from app.services.nutrition_db import ensure_nutrition_schema
|
||||
|
||||
@@ -82,6 +83,31 @@ async def lifespan(_app: FastAPI):
|
||||
except Exception as e:
|
||||
logger.warning("Could not restore bundled assets: %s", e)
|
||||
|
||||
# Cheap, and it must happen before anyone can press Resume: a batch that a
|
||||
# restart cut short is still marked "running" on disk, and until it is
|
||||
# reconciled the UI shows it as in flight with nothing behind it. This only
|
||||
# rewrites manifests - it deliberately starts no work. See
|
||||
# batch_ingest.scan_interrupted() for why auto-resume is not the default.
|
||||
try:
|
||||
from app.core.batch_ingest import scan_interrupted
|
||||
|
||||
interrupted = scan_interrupted()
|
||||
if interrupted:
|
||||
logger.info("Marked %d interrupted catalog batch(es).", len(interrupted))
|
||||
if interrupted and BATCH_AUTO_RESUME:
|
||||
# Opt-in only. Importing the worker here rather than at module
|
||||
# scope keeps the queue and its thread out of a boot that never
|
||||
# needs them.
|
||||
from app.core import batch_worker
|
||||
|
||||
for batch_id in interrupted:
|
||||
try:
|
||||
batch_worker.submit(batch_id)
|
||||
except Exception as exc: # noqa: BLE001 - a full queue is not fatal
|
||||
logger.warning("Could not auto-resume batch %s: %s", batch_id, exc)
|
||||
except Exception as e:
|
||||
logger.warning("Could not reconcile interrupted catalog batches: %s", e)
|
||||
|
||||
def _async_init():
|
||||
try:
|
||||
ensure_store_intelligence_schema()
|
||||
@@ -266,6 +292,7 @@ app.include_router(nutrition.router, prefix="/api")
|
||||
app.include_router(nutrition_admin.router, prefix="/api")
|
||||
app.include_router(upload.router, prefix="/api")
|
||||
app.include_router(store_catalog.router, prefix="/api")
|
||||
app.include_router(batch_catalog.router, prefix="/api")
|
||||
app.include_router(mcp_info.router, prefix="/api")
|
||||
|
||||
# MCP lives outside /api on purpose: it is a protocol endpoint for AI clients,
|
||||
|
||||
Reference in New Issue
Block a user