Backend- file ingestion API Updates
This commit is contained in:
85
app/api/batch_job_store.py
Normal file
85
app/api/batch_job_store.py
Normal file
@@ -0,0 +1,85 @@
|
||||
"""Live view of the batches this process knows about.
|
||||
|
||||
Same pattern and the same documented trade-offs as `store_catalog_job_store.py`
|
||||
and its siblings: a process-local dict behind a lock, not shared across uvicorn
|
||||
workers. Adding a broker for this would be operational weight the project has
|
||||
already decided against (see `job_store.py`).
|
||||
|
||||
WHAT IS DIFFERENT HERE, AND WHY IT STILL EARNS ITS PLACE
|
||||
--------------------------------------------------------
|
||||
Unlike the other job stores, this one is not the only record. `manifest.json`
|
||||
on the volume is the durable truth; this is a cache in front of it, and it
|
||||
exists for one reason: the UI polls every 3 seconds while row-level progress
|
||||
ticks many times a second. Serving those polls from memory keeps both the disk
|
||||
writes and the read path off the hot loop. A cache miss is not a 404 - `get()`
|
||||
falls back to reading the manifest, so a batch from before the last restart is
|
||||
still visible.
|
||||
|
||||
Cancellation lives here too rather than on disk. It is a request about the run
|
||||
in flight, and the run in flight is in this process.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
from app.core import batch_ingest
|
||||
|
||||
|
||||
class BatchJobStore:
|
||||
def __init__(self) -> None:
|
||||
self._batches: Dict[str, batch_ingest.BatchManifest] = {}
|
||||
self._cancelled: set = set()
|
||||
self._lock = threading.Lock()
|
||||
|
||||
def put(self, manifest: batch_ingest.BatchManifest) -> None:
|
||||
"""Record (or refresh) a batch. This is the `on_change` callback."""
|
||||
with self._lock:
|
||||
self._batches[manifest.batch_id] = manifest
|
||||
|
||||
def get(self, batch_id: str) -> Optional[batch_ingest.BatchManifest]:
|
||||
"""Live state if we have it, otherwise whatever is on disk."""
|
||||
with self._lock:
|
||||
cached = self._batches.get(batch_id)
|
||||
if cached is not None:
|
||||
return cached
|
||||
return batch_ingest.read_manifest(batch_id)
|
||||
|
||||
def recent(self, limit: int = 20) -> List[batch_ingest.BatchManifest]:
|
||||
"""Newest first, merging the live view over the on-disk one.
|
||||
|
||||
Reading the directory rather than only the cache means a restart does
|
||||
not make previous batches vanish from the list.
|
||||
"""
|
||||
with self._lock:
|
||||
live = dict(self._batches)
|
||||
merged: Dict[str, batch_ingest.BatchManifest] = {}
|
||||
for manifest in batch_ingest.list_manifests():
|
||||
merged[manifest.batch_id] = live.get(manifest.batch_id, manifest)
|
||||
for batch_id, manifest in live.items():
|
||||
merged.setdefault(batch_id, manifest)
|
||||
ordered = sorted(merged.values(), key=lambda m: m.created_at, reverse=True)
|
||||
return ordered[: max(1, limit)]
|
||||
|
||||
# -- cancellation --------------------------------------------------------
|
||||
def cancel(self, batch_id: str) -> None:
|
||||
"""Ask the worker to stop before it picks up the next file.
|
||||
|
||||
Nothing interrupts the file already running. Killing a pipeline halfway
|
||||
would leave some of its rows written and the rest not, with no record of
|
||||
where it stopped; letting the current file finish is both simpler and
|
||||
the only version with a defined outcome.
|
||||
"""
|
||||
with self._lock:
|
||||
self._cancelled.add(batch_id)
|
||||
|
||||
def is_cancelled(self, batch_id: str) -> bool:
|
||||
with self._lock:
|
||||
return batch_id in self._cancelled
|
||||
|
||||
def clear_cancel(self, batch_id: str) -> None:
|
||||
with self._lock:
|
||||
self._cancelled.discard(batch_id)
|
||||
|
||||
|
||||
batch_job_store = BatchJobStore()
|
||||
385
app/api/routers/batch_catalog.py
Normal file
385
app/api/routers/batch_catalog.py
Normal file
@@ -0,0 +1,385 @@
|
||||
"""Admin endpoints for ingesting several store spreadsheets as one batch.
|
||||
|
||||
POST /api/admin/catalog-batch/preview - parse only, per file
|
||||
POST /api/admin/catalog-batch/ingest - 202 + batch_id
|
||||
GET /api/admin/catalog-batch/batches - recent batches
|
||||
GET /api/admin/catalog-batch/batches/{id} - poll one batch
|
||||
POST /api/admin/catalog-batch/batches/{id}/resume - after a restart
|
||||
POST /api/admin/catalog-batch/batches/{id}/cancel - stop the rest
|
||||
|
||||
This is the multi-file sibling of `store_catalog.py`, and it deliberately does
|
||||
not replace it: the single-file endpoints are untouched and still work. What is
|
||||
different is the unit of work. Five files are one batch with one id, so the
|
||||
question a colleague actually asks - "did the drop land?" - has one answer
|
||||
rather than five.
|
||||
|
||||
WHY THE NETWORK STAGES DEFAULT OFF HERE
|
||||
---------------------------------------
|
||||
The single-file UI sends `use_llm=true, fetch_images=true`. At one file that is
|
||||
a considered trade. At twenty it is thousands of outbound requests and, for the
|
||||
image stage, a Playwright subprocess that can burn three minutes on its own -
|
||||
on a single-vCPU container that is also serving the API. So a batch opts IN to
|
||||
those stages; it does not opt out. `USE_OLLAMA` is false in production anyway,
|
||||
which makes `use_llm` a no-op there and the honest default obvious.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import queue
|
||||
from typing import List, Optional
|
||||
|
||||
from fastapi import APIRouter, Depends, File, HTTPException, UploadFile, status
|
||||
from pydantic import BaseModel
|
||||
|
||||
from app.api.batch_job_store import batch_job_store
|
||||
from app.api.deps import require_admin
|
||||
from app.core import batch_ingest, batch_worker
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.infrastructure.settings import (
|
||||
BATCH_MAX_FILES,
|
||||
BATCH_MAX_TOTAL_BYTES,
|
||||
BATCH_MAX_TOTAL_ROWS,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
router = APIRouter(prefix="/admin/catalog-batch", tags=["admin", "catalog"])
|
||||
|
||||
# Per-file ceilings match store_catalog.py exactly. A file that is too big for
|
||||
# the single-file endpoint is not somehow acceptable because it arrived with
|
||||
# four friends.
|
||||
MAX_UPLOAD_BYTES = 10 * 1024 * 1024
|
||||
MAX_UPLOAD_ROWS = 2000
|
||||
PREVIEW_ROWS = 10
|
||||
|
||||
# The worker cannot import the API layer without a cycle, so the wiring is done
|
||||
# here, at import, once.
|
||||
batch_worker.configure(
|
||||
on_change=batch_job_store.put,
|
||||
should_cancel=batch_job_store.is_cancelled,
|
||||
)
|
||||
|
||||
|
||||
class BatchFileOut(BaseModel):
|
||||
index: int
|
||||
filename: str
|
||||
status: str
|
||||
detail: Optional[str] = None
|
||||
stage_index: int = 0
|
||||
stage_name: str = ""
|
||||
total_stages: int = pipeline.TOTAL_STAGES
|
||||
rows_done: int = 0
|
||||
rows_total: int = 0
|
||||
size_bytes: int = 0
|
||||
result: Optional[dict] = None
|
||||
|
||||
|
||||
class BatchOut(BaseModel):
|
||||
batch_id: str
|
||||
status: str
|
||||
detail: Optional[str] = None
|
||||
created_at: float
|
||||
updated_at: float
|
||||
files_total: int
|
||||
files_done: int
|
||||
files_failed: int
|
||||
current_file: Optional[str] = None
|
||||
use_llm: bool
|
||||
fetch_images: bool
|
||||
totals: dict
|
||||
brands: List[str]
|
||||
files: List[BatchFileOut]
|
||||
|
||||
|
||||
def _to_out(manifest: batch_ingest.BatchManifest) -> BatchOut:
|
||||
body = manifest.to_dict()
|
||||
body["files"] = [BatchFileOut(**{
|
||||
key: entry[key] for key in BatchFileOut.model_fields if key in entry
|
||||
}) for entry in body["files"]]
|
||||
return BatchOut(**{k: v for k, v in body.items() if k in BatchOut.model_fields})
|
||||
|
||||
|
||||
async def _read_uploads(files: List[UploadFile]) -> List[tuple]:
|
||||
"""Read every upload into memory, enforcing the count and size ceilings.
|
||||
|
||||
Read here rather than in the worker for the same reason `store_catalog.py`
|
||||
gives: `UploadFile` is backed by a temporary file tied to the request, and
|
||||
it is gone before a background thread would reach it.
|
||||
"""
|
||||
if not files:
|
||||
raise HTTPException(status_code=400, detail="No files were uploaded.")
|
||||
if len(files) > BATCH_MAX_FILES:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=(
|
||||
f"{len(files)} files exceeds the {BATCH_MAX_FILES}-file limit for one "
|
||||
f"batch. Split the drop and upload it in two batches."
|
||||
),
|
||||
)
|
||||
|
||||
read: List[tuple] = []
|
||||
total = 0
|
||||
for upload in files:
|
||||
contents = await upload.read()
|
||||
name = upload.filename or "upload.xlsx"
|
||||
if not contents:
|
||||
# Recorded rather than raised - an empty file among nine good ones
|
||||
# is a fact about that file, not a reason to reject the drop.
|
||||
read.append((name, b""))
|
||||
continue
|
||||
if len(contents) > MAX_UPLOAD_BYTES:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=(
|
||||
f"'{name}' is larger than the "
|
||||
f"{MAX_UPLOAD_BYTES // (1024 * 1024)}MB per-file limit."
|
||||
),
|
||||
)
|
||||
total += len(contents)
|
||||
if total > BATCH_MAX_TOTAL_BYTES:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=(
|
||||
f"The batch is larger than the "
|
||||
f"{BATCH_MAX_TOTAL_BYTES // (1024 * 1024)}MB total limit."
|
||||
),
|
||||
)
|
||||
read.append((name, contents))
|
||||
return read
|
||||
|
||||
|
||||
def _parse_all(read: List[tuple]):
|
||||
"""Split the uploads into (valid, invalid) by trying to parse each one.
|
||||
|
||||
Failing fast here is what stops a batch transitioning straight to "failed"
|
||||
a second after it started - the same reasoning as `store_catalog.py:147`,
|
||||
applied per file so that one bad sheet does not condemn the others.
|
||||
"""
|
||||
valid: List[tuple] = []
|
||||
invalid: List[tuple] = []
|
||||
rows_total = 0
|
||||
|
||||
for name, contents in read:
|
||||
if not contents:
|
||||
invalid.append((name, "The file is empty."))
|
||||
continue
|
||||
try:
|
||||
df, _mapping = pipeline.parse_spreadsheet(name, contents)
|
||||
except HTTPException as exc:
|
||||
# read_products_dataframe raises HTTPException for an unsupported
|
||||
# extension or a missing Excel reader; its message already names
|
||||
# the file and what to do about it.
|
||||
invalid.append((name, str(exc.detail)))
|
||||
continue
|
||||
except Exception as exc: # noqa: BLE001 - an unreadable sheet is user error
|
||||
invalid.append((name, f"Could not parse the file: {exc}"))
|
||||
continue
|
||||
|
||||
if df.empty:
|
||||
invalid.append((name, "The file has no data rows."))
|
||||
continue
|
||||
if len(df) > MAX_UPLOAD_ROWS:
|
||||
invalid.append((
|
||||
name,
|
||||
f"{len(df)} rows exceeds the {MAX_UPLOAD_ROWS}-row per-file limit.",
|
||||
))
|
||||
continue
|
||||
|
||||
rows_total += int(len(df))
|
||||
if rows_total > BATCH_MAX_TOTAL_ROWS:
|
||||
raise HTTPException(
|
||||
status_code=413,
|
||||
detail=(
|
||||
f"The batch totals more than {BATCH_MAX_TOTAL_ROWS} rows. "
|
||||
f"Split it and upload in two batches."
|
||||
),
|
||||
)
|
||||
valid.append((name, contents, int(len(df))))
|
||||
|
||||
return valid, invalid, rows_total
|
||||
|
||||
|
||||
@router.post("/preview", dependencies=[Depends(require_admin)])
|
||||
async def preview_catalog_batch(files: List[UploadFile] = File(...)) -> dict:
|
||||
"""Parse every file and report how its columns were understood.
|
||||
|
||||
Nothing is staged and no batch is created. The mapping from a store's own
|
||||
headers onto catalog fields is a guess, and finding out that "Item" was read
|
||||
as the description after twenty files have been scraped is expensive.
|
||||
"""
|
||||
read = await _read_uploads(files)
|
||||
out = []
|
||||
for name, contents in read:
|
||||
if not contents:
|
||||
out.append({"filename": name, "ok": False, "error": "The file is empty."})
|
||||
continue
|
||||
try:
|
||||
df, mapping = pipeline.parse_spreadsheet(name, contents)
|
||||
except HTTPException as exc:
|
||||
out.append({"filename": name, "ok": False, "error": str(exc.detail)})
|
||||
continue
|
||||
except Exception as exc: # noqa: BLE001
|
||||
out.append({"filename": name, "ok": False,
|
||||
"error": f"Could not parse the file: {exc}"})
|
||||
continue
|
||||
|
||||
if df.empty:
|
||||
out.append({"filename": name, "ok": False, "error": "The file has no data rows."})
|
||||
continue
|
||||
|
||||
out.append({
|
||||
"filename": name,
|
||||
"ok": True,
|
||||
"rows_total": int(len(df)),
|
||||
"over_row_limit": bool(len(df) > MAX_UPLOAD_ROWS),
|
||||
"recognised_columns": {f: str(c) for f, c in mapping.columns.items()},
|
||||
"unrecognised_columns": mapping.unrecognised,
|
||||
"brand_column_present": "brand" in mapping.columns,
|
||||
"preview": df.head(PREVIEW_ROWS).fillna("").astype(str).to_dict(orient="records"),
|
||||
})
|
||||
|
||||
return {
|
||||
"files": out,
|
||||
"files_total": len(out),
|
||||
"files_ok": sum(1 for f in out if f.get("ok")),
|
||||
"rows_total": sum(int(f.get("rows_total") or 0) for f in out if f.get("ok")),
|
||||
"stages": list(pipeline.STAGE_NAMES),
|
||||
"limits": {
|
||||
"max_files": BATCH_MAX_FILES,
|
||||
"max_rows_per_file": MAX_UPLOAD_ROWS,
|
||||
"max_rows_total": BATCH_MAX_TOTAL_ROWS,
|
||||
"max_bytes_per_file": MAX_UPLOAD_BYTES,
|
||||
"max_bytes_total": BATCH_MAX_TOTAL_BYTES,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
@router.post("/ingest", status_code=status.HTTP_202_ACCEPTED,
|
||||
dependencies=[Depends(require_admin)])
|
||||
async def ingest_catalog_batch(
|
||||
files: List[UploadFile] = File(...),
|
||||
use_llm: bool = False,
|
||||
fetch_images: bool = False,
|
||||
) -> BatchOut:
|
||||
"""Stage the files, queue the batch, and return an id to poll.
|
||||
|
||||
Returns immediately. In production the browser reaches this through Traefik
|
||||
on a different host to the frontend, so anything that sat on the request
|
||||
path would be racing an idle timeout nobody here controls.
|
||||
"""
|
||||
read = await _read_uploads(files)
|
||||
valid, invalid, _rows = _parse_all(read)
|
||||
|
||||
if not valid:
|
||||
detail = "; ".join(f"{name}: {reason}" for name, reason in invalid)
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"None of the uploaded files could be ingested. {detail}",
|
||||
)
|
||||
|
||||
manifest = batch_ingest.stage_batch(
|
||||
[(name, contents) for name, contents, _n in valid],
|
||||
use_llm=use_llm,
|
||||
fetch_images=fetch_images,
|
||||
invalid=invalid,
|
||||
)
|
||||
for entry, (_name, _contents, rows) in zip(manifest.files, valid):
|
||||
entry.rows_total = rows
|
||||
batch_ingest.write_manifest(manifest)
|
||||
batch_job_store.put(manifest)
|
||||
|
||||
try:
|
||||
batch_worker.submit(manifest.batch_id)
|
||||
except queue.Full:
|
||||
manifest.status = batch_ingest.QUEUED
|
||||
manifest.detail = (
|
||||
"The ingestion queue is full. This batch is staged and can be started "
|
||||
"with Resume once the running batches finish."
|
||||
)
|
||||
batch_ingest.write_manifest(manifest)
|
||||
batch_job_store.put(manifest)
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail=(
|
||||
"Too many batches are already queued. This one has been saved - "
|
||||
"press Resume on it once the current batch finishes."
|
||||
),
|
||||
)
|
||||
|
||||
return _to_out(manifest)
|
||||
|
||||
|
||||
@router.get("/batches", dependencies=[Depends(require_admin)])
|
||||
def list_catalog_batches(limit: int = 20) -> dict:
|
||||
limit = max(1, min(limit, 100))
|
||||
return {"batches": [_to_out(m) for m in batch_job_store.recent(limit)]}
|
||||
|
||||
|
||||
@router.get("/batches/{batch_id}", dependencies=[Depends(require_admin)])
|
||||
def get_catalog_batch(batch_id: str) -> BatchOut:
|
||||
manifest = batch_job_store.get(batch_id)
|
||||
if not manifest:
|
||||
raise HTTPException(status_code=404, detail="Batch not found")
|
||||
return _to_out(manifest)
|
||||
|
||||
|
||||
@router.post("/batches/{batch_id}/resume", dependencies=[Depends(require_admin)])
|
||||
def resume_catalog_batch(batch_id: str) -> BatchOut:
|
||||
"""Re-queue a batch a restart cut short, or one that was queued behind a full queue."""
|
||||
manifest = batch_ingest.read_manifest(batch_id)
|
||||
if not manifest:
|
||||
raise HTTPException(status_code=404, detail="Batch not found")
|
||||
|
||||
pending = [f for f in manifest.files if f.status == batch_ingest.QUEUED]
|
||||
if not pending:
|
||||
raise HTTPException(
|
||||
status_code=409,
|
||||
detail=f"Nothing left to run in this batch (status: {manifest.status}).",
|
||||
)
|
||||
|
||||
batch_job_store.clear_cancel(batch_id)
|
||||
manifest.status = batch_ingest.QUEUED
|
||||
manifest.detail = None
|
||||
batch_ingest.write_manifest(manifest)
|
||||
batch_job_store.put(manifest)
|
||||
|
||||
try:
|
||||
batch_worker.submit(batch_id)
|
||||
except queue.Full:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail="Too many batches are already queued. Try again shortly.",
|
||||
)
|
||||
return _to_out(manifest)
|
||||
|
||||
|
||||
@router.post("/batches/{batch_id}/cancel", dependencies=[Depends(require_admin)])
|
||||
def cancel_catalog_batch(batch_id: str) -> BatchOut:
|
||||
"""Stop before the next file. The file already running is allowed to finish.
|
||||
|
||||
Interrupting a pipeline mid-file would leave some of its rows written and
|
||||
the rest not, with nothing recording where it stopped. Letting the current
|
||||
file complete is the only version of "cancel" with a defined outcome.
|
||||
"""
|
||||
manifest = batch_job_store.get(batch_id)
|
||||
if not manifest:
|
||||
raise HTTPException(status_code=404, detail="Batch not found")
|
||||
|
||||
batch_job_store.cancel(batch_id)
|
||||
|
||||
# A batch that has not started yet has no worker to notice the flag, so
|
||||
# cancel it here and be done.
|
||||
on_disk = batch_ingest.read_manifest(batch_id)
|
||||
if on_disk and on_disk.status in {batch_ingest.QUEUED, batch_ingest.INTERRUPTED}:
|
||||
for entry in on_disk.files:
|
||||
if entry.status == batch_ingest.QUEUED:
|
||||
entry.status = batch_ingest.CANCELLED
|
||||
entry.detail = "Cancelled before this file started."
|
||||
on_disk.settle()
|
||||
on_disk.detail = "Cancelled."
|
||||
batch_ingest.write_manifest(on_disk)
|
||||
batch_job_store.put(on_disk)
|
||||
return _to_out(on_disk)
|
||||
|
||||
manifest.detail = "Cancelling - the file currently running will finish first."
|
||||
batch_job_store.put(manifest)
|
||||
return _to_out(manifest)
|
||||
Reference in New Issue
Block a user