Backend- file ingestion API Updates

This commit is contained in:
sriram
2026-08-28 07:49:56 +05:30
parent f698720ee2
commit a54bd43f8b
17 changed files with 2306 additions and 138 deletions

View File

@@ -0,0 +1,385 @@
"""Admin endpoints for ingesting several store spreadsheets as one batch.
POST /api/admin/catalog-batch/preview - parse only, per file
POST /api/admin/catalog-batch/ingest - 202 + batch_id
GET /api/admin/catalog-batch/batches - recent batches
GET /api/admin/catalog-batch/batches/{id} - poll one batch
POST /api/admin/catalog-batch/batches/{id}/resume - after a restart
POST /api/admin/catalog-batch/batches/{id}/cancel - stop the rest
This is the multi-file sibling of `store_catalog.py`, and it deliberately does
not replace it: the single-file endpoints are untouched and still work. What is
different is the unit of work. Five files are one batch with one id, so the
question a colleague actually asks - "did the drop land?" - has one answer
rather than five.
WHY THE NETWORK STAGES DEFAULT OFF HERE
---------------------------------------
The single-file UI sends `use_llm=true, fetch_images=true`. At one file that is
a considered trade. At twenty it is thousands of outbound requests and, for the
image stage, a Playwright subprocess that can burn three minutes on its own -
on a single-vCPU container that is also serving the API. So a batch opts IN to
those stages; it does not opt out. `USE_OLLAMA` is false in production anyway,
which makes `use_llm` a no-op there and the honest default obvious.
"""
from __future__ import annotations
import logging
import queue
from typing import List, Optional
from fastapi import APIRouter, Depends, File, HTTPException, UploadFile, status
from pydantic import BaseModel
from app.api.batch_job_store import batch_job_store
from app.api.deps import require_admin
from app.core import batch_ingest, batch_worker
from app.core import store_catalog_pipeline as pipeline
from app.infrastructure.settings import (
BATCH_MAX_FILES,
BATCH_MAX_TOTAL_BYTES,
BATCH_MAX_TOTAL_ROWS,
)
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/admin/catalog-batch", tags=["admin", "catalog"])
# Per-file ceilings match store_catalog.py exactly. A file that is too big for
# the single-file endpoint is not somehow acceptable because it arrived with
# four friends.
MAX_UPLOAD_BYTES = 10 * 1024 * 1024
MAX_UPLOAD_ROWS = 2000
PREVIEW_ROWS = 10
# The worker cannot import the API layer without a cycle, so the wiring is done
# here, at import, once.
batch_worker.configure(
on_change=batch_job_store.put,
should_cancel=batch_job_store.is_cancelled,
)
class BatchFileOut(BaseModel):
index: int
filename: str
status: str
detail: Optional[str] = None
stage_index: int = 0
stage_name: str = ""
total_stages: int = pipeline.TOTAL_STAGES
rows_done: int = 0
rows_total: int = 0
size_bytes: int = 0
result: Optional[dict] = None
class BatchOut(BaseModel):
batch_id: str
status: str
detail: Optional[str] = None
created_at: float
updated_at: float
files_total: int
files_done: int
files_failed: int
current_file: Optional[str] = None
use_llm: bool
fetch_images: bool
totals: dict
brands: List[str]
files: List[BatchFileOut]
def _to_out(manifest: batch_ingest.BatchManifest) -> BatchOut:
body = manifest.to_dict()
body["files"] = [BatchFileOut(**{
key: entry[key] for key in BatchFileOut.model_fields if key in entry
}) for entry in body["files"]]
return BatchOut(**{k: v for k, v in body.items() if k in BatchOut.model_fields})
async def _read_uploads(files: List[UploadFile]) -> List[tuple]:
"""Read every upload into memory, enforcing the count and size ceilings.
Read here rather than in the worker for the same reason `store_catalog.py`
gives: `UploadFile` is backed by a temporary file tied to the request, and
it is gone before a background thread would reach it.
"""
if not files:
raise HTTPException(status_code=400, detail="No files were uploaded.")
if len(files) > BATCH_MAX_FILES:
raise HTTPException(
status_code=413,
detail=(
f"{len(files)} files exceeds the {BATCH_MAX_FILES}-file limit for one "
f"batch. Split the drop and upload it in two batches."
),
)
read: List[tuple] = []
total = 0
for upload in files:
contents = await upload.read()
name = upload.filename or "upload.xlsx"
if not contents:
# Recorded rather than raised - an empty file among nine good ones
# is a fact about that file, not a reason to reject the drop.
read.append((name, b""))
continue
if len(contents) > MAX_UPLOAD_BYTES:
raise HTTPException(
status_code=413,
detail=(
f"'{name}' is larger than the "
f"{MAX_UPLOAD_BYTES // (1024 * 1024)}MB per-file limit."
),
)
total += len(contents)
if total > BATCH_MAX_TOTAL_BYTES:
raise HTTPException(
status_code=413,
detail=(
f"The batch is larger than the "
f"{BATCH_MAX_TOTAL_BYTES // (1024 * 1024)}MB total limit."
),
)
read.append((name, contents))
return read
def _parse_all(read: List[tuple]):
"""Split the uploads into (valid, invalid) by trying to parse each one.
Failing fast here is what stops a batch transitioning straight to "failed"
a second after it started - the same reasoning as `store_catalog.py:147`,
applied per file so that one bad sheet does not condemn the others.
"""
valid: List[tuple] = []
invalid: List[tuple] = []
rows_total = 0
for name, contents in read:
if not contents:
invalid.append((name, "The file is empty."))
continue
try:
df, _mapping = pipeline.parse_spreadsheet(name, contents)
except HTTPException as exc:
# read_products_dataframe raises HTTPException for an unsupported
# extension or a missing Excel reader; its message already names
# the file and what to do about it.
invalid.append((name, str(exc.detail)))
continue
except Exception as exc: # noqa: BLE001 - an unreadable sheet is user error
invalid.append((name, f"Could not parse the file: {exc}"))
continue
if df.empty:
invalid.append((name, "The file has no data rows."))
continue
if len(df) > MAX_UPLOAD_ROWS:
invalid.append((
name,
f"{len(df)} rows exceeds the {MAX_UPLOAD_ROWS}-row per-file limit.",
))
continue
rows_total += int(len(df))
if rows_total > BATCH_MAX_TOTAL_ROWS:
raise HTTPException(
status_code=413,
detail=(
f"The batch totals more than {BATCH_MAX_TOTAL_ROWS} rows. "
f"Split it and upload in two batches."
),
)
valid.append((name, contents, int(len(df))))
return valid, invalid, rows_total
@router.post("/preview", dependencies=[Depends(require_admin)])
async def preview_catalog_batch(files: List[UploadFile] = File(...)) -> dict:
"""Parse every file and report how its columns were understood.
Nothing is staged and no batch is created. The mapping from a store's own
headers onto catalog fields is a guess, and finding out that "Item" was read
as the description after twenty files have been scraped is expensive.
"""
read = await _read_uploads(files)
out = []
for name, contents in read:
if not contents:
out.append({"filename": name, "ok": False, "error": "The file is empty."})
continue
try:
df, mapping = pipeline.parse_spreadsheet(name, contents)
except HTTPException as exc:
out.append({"filename": name, "ok": False, "error": str(exc.detail)})
continue
except Exception as exc: # noqa: BLE001
out.append({"filename": name, "ok": False,
"error": f"Could not parse the file: {exc}"})
continue
if df.empty:
out.append({"filename": name, "ok": False, "error": "The file has no data rows."})
continue
out.append({
"filename": name,
"ok": True,
"rows_total": int(len(df)),
"over_row_limit": bool(len(df) > MAX_UPLOAD_ROWS),
"recognised_columns": {f: str(c) for f, c in mapping.columns.items()},
"unrecognised_columns": mapping.unrecognised,
"brand_column_present": "brand" in mapping.columns,
"preview": df.head(PREVIEW_ROWS).fillna("").astype(str).to_dict(orient="records"),
})
return {
"files": out,
"files_total": len(out),
"files_ok": sum(1 for f in out if f.get("ok")),
"rows_total": sum(int(f.get("rows_total") or 0) for f in out if f.get("ok")),
"stages": list(pipeline.STAGE_NAMES),
"limits": {
"max_files": BATCH_MAX_FILES,
"max_rows_per_file": MAX_UPLOAD_ROWS,
"max_rows_total": BATCH_MAX_TOTAL_ROWS,
"max_bytes_per_file": MAX_UPLOAD_BYTES,
"max_bytes_total": BATCH_MAX_TOTAL_BYTES,
},
}
@router.post("/ingest", status_code=status.HTTP_202_ACCEPTED,
dependencies=[Depends(require_admin)])
async def ingest_catalog_batch(
files: List[UploadFile] = File(...),
use_llm: bool = False,
fetch_images: bool = False,
) -> BatchOut:
"""Stage the files, queue the batch, and return an id to poll.
Returns immediately. In production the browser reaches this through Traefik
on a different host to the frontend, so anything that sat on the request
path would be racing an idle timeout nobody here controls.
"""
read = await _read_uploads(files)
valid, invalid, _rows = _parse_all(read)
if not valid:
detail = "; ".join(f"{name}: {reason}" for name, reason in invalid)
raise HTTPException(
status_code=400,
detail=f"None of the uploaded files could be ingested. {detail}",
)
manifest = batch_ingest.stage_batch(
[(name, contents) for name, contents, _n in valid],
use_llm=use_llm,
fetch_images=fetch_images,
invalid=invalid,
)
for entry, (_name, _contents, rows) in zip(manifest.files, valid):
entry.rows_total = rows
batch_ingest.write_manifest(manifest)
batch_job_store.put(manifest)
try:
batch_worker.submit(manifest.batch_id)
except queue.Full:
manifest.status = batch_ingest.QUEUED
manifest.detail = (
"The ingestion queue is full. This batch is staged and can be started "
"with Resume once the running batches finish."
)
batch_ingest.write_manifest(manifest)
batch_job_store.put(manifest)
raise HTTPException(
status_code=429,
detail=(
"Too many batches are already queued. This one has been saved - "
"press Resume on it once the current batch finishes."
),
)
return _to_out(manifest)
@router.get("/batches", dependencies=[Depends(require_admin)])
def list_catalog_batches(limit: int = 20) -> dict:
limit = max(1, min(limit, 100))
return {"batches": [_to_out(m) for m in batch_job_store.recent(limit)]}
@router.get("/batches/{batch_id}", dependencies=[Depends(require_admin)])
def get_catalog_batch(batch_id: str) -> BatchOut:
manifest = batch_job_store.get(batch_id)
if not manifest:
raise HTTPException(status_code=404, detail="Batch not found")
return _to_out(manifest)
@router.post("/batches/{batch_id}/resume", dependencies=[Depends(require_admin)])
def resume_catalog_batch(batch_id: str) -> BatchOut:
"""Re-queue a batch a restart cut short, or one that was queued behind a full queue."""
manifest = batch_ingest.read_manifest(batch_id)
if not manifest:
raise HTTPException(status_code=404, detail="Batch not found")
pending = [f for f in manifest.files if f.status == batch_ingest.QUEUED]
if not pending:
raise HTTPException(
status_code=409,
detail=f"Nothing left to run in this batch (status: {manifest.status}).",
)
batch_job_store.clear_cancel(batch_id)
manifest.status = batch_ingest.QUEUED
manifest.detail = None
batch_ingest.write_manifest(manifest)
batch_job_store.put(manifest)
try:
batch_worker.submit(batch_id)
except queue.Full:
raise HTTPException(
status_code=429,
detail="Too many batches are already queued. Try again shortly.",
)
return _to_out(manifest)
@router.post("/batches/{batch_id}/cancel", dependencies=[Depends(require_admin)])
def cancel_catalog_batch(batch_id: str) -> BatchOut:
"""Stop before the next file. The file already running is allowed to finish.
Interrupting a pipeline mid-file would leave some of its rows written and
the rest not, with nothing recording where it stopped. Letting the current
file complete is the only version of "cancel" with a defined outcome.
"""
manifest = batch_job_store.get(batch_id)
if not manifest:
raise HTTPException(status_code=404, detail="Batch not found")
batch_job_store.cancel(batch_id)
# A batch that has not started yet has no worker to notice the flag, so
# cancel it here and be done.
on_disk = batch_ingest.read_manifest(batch_id)
if on_disk and on_disk.status in {batch_ingest.QUEUED, batch_ingest.INTERRUPTED}:
for entry in on_disk.files:
if entry.status == batch_ingest.QUEUED:
entry.status = batch_ingest.CANCELLED
entry.detail = "Cancelled before this file started."
on_disk.settle()
on_disk.detail = "Cancelled."
batch_ingest.write_manifest(on_disk)
batch_job_store.put(on_disk)
return _to_out(on_disk)
manifest.detail = "Cancelling - the file currently running will finish first."
batch_job_store.put(manifest)
return _to_out(manifest)