ingestion updates

This commit is contained in:
sriram
2026-08-28 08:57:10 +05:30
parent a54bd43f8b
commit 0e75d32f61
9 changed files with 1147 additions and 4 deletions

View File

@@ -33,7 +33,7 @@ from pydantic import BaseModel
from app.api.batch_job_store import batch_job_store
from app.api.deps import require_admin
from app.core import batch_ingest, batch_worker
from app.core import batch_ingest, batch_worker, inbox
from app.core import store_catalog_pipeline as pipeline
from app.infrastructure.settings import (
BATCH_MAX_FILES,
@@ -77,6 +77,7 @@ class BatchOut(BaseModel):
batch_id: str
status: str
detail: Optional[str] = None
submitted_by: Optional[str] = None
created_at: float
updated_at: float
files_total: int
@@ -308,6 +309,139 @@ async def ingest_catalog_batch(
return _to_out(manifest)
# ---------------------------------------------------------------------------
# The review inbox: files a colleague dropped, waiting for a decision
# ---------------------------------------------------------------------------
# These are the admin half of the two-actor flow. The uploader half lives in
# app/api/routers/uploads.py and can reach none of this.
class InboxSelection(BaseModel):
file_ids: List[str]
use_llm: bool = False
fetch_images: bool = False
class InboxDismissal(BaseModel):
file_ids: List[str]
@router.get("/inbox", dependencies=[Depends(require_admin)])
def list_inbox() -> dict:
"""Files awaiting review, grouped by the drop they arrived in.
`pending_count` is what the badge renders, and it is computed here rather
than by summing the response client-side so the two can never disagree.
"""
submissions = inbox.list_pending()
return {
"pending_count": sum(len(s.pending_files) for s in submissions),
"submissions": [
{
"submission_id": s.submission_id,
"submitted_by": s.submitted_by,
"created_at": s.created_at,
"files": [
{
"file_id": f.file_id,
"filename": f.filename,
"rows_total": f.rows_total,
"size_bytes": f.size_bytes,
}
for f in s.pending_files
],
}
for s in submissions
],
}
@router.post("/from-inbox", status_code=status.HTTP_202_ACCEPTED,
dependencies=[Depends(require_admin)])
def start_batch_from_inbox(selection: InboxSelection) -> BatchOut:
"""Compose a batch out of the selected inbox files and start it.
The files may come from different drops on different days; that is the
point of selecting per file rather than per submission. From here on this
is an ordinary batch and every existing path - progress, resume, cancel -
applies unchanged.
"""
if not selection.file_ids:
raise HTTPException(status_code=400, detail="No files were selected.")
try:
uploads, submitters = inbox.collect_for_batch(selection.file_ids)
except KeyError as exc:
raise HTTPException(
status_code=404, detail=f"No such file in the inbox: {exc.args[0]}"
) from exc
except ValueError as exc:
# Two admin tabs open on the same inbox. Tell the second one what
# happened rather than silently running the file a second time.
raise HTTPException(status_code=409, detail=str(exc)) from exc
# Re-parse rather than trusting the row counts recorded at upload: the
# ceilings are a property of the batch about to run, not of the drops it
# was assembled from, and a selection can span any number of drops.
read = [(name, contents) for name, contents in uploads]
valid, invalid, _rows = _parse_all(read)
if not valid:
detail = "; ".join(f"{name}: {reason}" for name, reason in invalid)
raise HTTPException(
status_code=400,
detail=f"None of the selected files could be ingested. {detail}",
)
manifest = batch_ingest.stage_batch(
[(name, contents) for name, contents, _n in valid],
use_llm=selection.use_llm,
fetch_images=selection.fetch_images,
invalid=invalid,
)
for entry, (_name, _contents, rows) in zip(manifest.files, valid):
entry.rows_total = rows
manifest.submitted_by = ", ".join(submitters) or None
batch_ingest.write_manifest(manifest)
batch_job_store.put(manifest)
try:
batch_worker.submit(manifest.batch_id)
except queue.Full:
manifest.detail = (
"The ingestion queue is full. This batch is staged and can be started "
"with Resume once the running batches finish."
)
batch_ingest.write_manifest(manifest)
batch_job_store.put(manifest)
raise HTTPException(
status_code=429,
detail=(
"Too many batches are already queued. This selection has been staged - "
"press Resume on it once the current batch finishes."
),
)
# Only now, once the batch exists AND is queued. Marking first would drop
# the files out of the inbox with nothing left to retry from if staging
# had then failed.
inbox.mark_consumed(selection.file_ids, manifest.batch_id)
return _to_out(manifest)
@router.post("/inbox/dismiss", dependencies=[Depends(require_admin)])
def dismiss_inbox_files(dismissal: InboxDismissal) -> dict:
"""Mark files as never-to-run, so the badge can reach zero."""
if not dismissal.file_ids:
raise HTTPException(status_code=400, detail="No files were selected.")
changed = inbox.dismiss(dismissal.file_ids)
if not changed:
raise HTTPException(
status_code=409,
detail="None of those files were still awaiting review.",
)
return {"dismissed": changed, "pending_count": inbox.pending_count()}
@router.get("/batches", dependencies=[Depends(require_admin)])
def list_catalog_batches(limit: int = 20) -> dict:
limit = max(1, min(limit, 100))

212
app/api/routers/uploads.py Normal file
View File

@@ -0,0 +1,212 @@
"""The one endpoint an outside contributor may call.
POST /api/uploads/catalog - drop spreadsheets into the review inbox
This is deliberately its own router, with its own prefix and its own guard, so
that the difference between it and everything else in the app is visible in one
screen rather than inferred from a decorator halfway down a 400-line admin
module. Everything else that touches catalog data is `require_admin`; this is
`require_permission("upload_catalog")`, and that single line is the whole
security boundary of the feature.
WHAT THIS ENDPOINT CANNOT DO
----------------------------
Start work. Files land in the inbox and wait for an admin to select them
(`app/core/inbox.py`). No worker is touched, no queue is entered, no thread is
started. That is what makes it safe to hand a credential to someone outside the
team: the worst a leaked uploader key costs is bounded disk, never CPU on a
one-vCPU host that is also serving the API.
It also cannot read anything. There is no GET here on purpose - the decision was
that this is a one-way drop, so the credential grants no visibility into the
catalog, into other submissions, or even into the submitter's own past uploads.
WHY A BAD SHEET IS REJECTED HERE AND NOT LATER
----------------------------------------------
The file is parsed during the request, while the colleague is still watching.
Telling them "row 1 has no product name column" in the 202 is worth far more
than discovering it days later in an admin panel they cannot see, with no way to
ask them for a corrected file except out of band.
"""
from __future__ import annotations
import logging
from typing import List, Optional
from fastapi import APIRouter, Depends, File, HTTPException, UploadFile, status
from pydantic import BaseModel
from app.api.deps import require_permission
from app.core import inbox
from app.core import store_catalog_pipeline as pipeline
from app.infrastructure.security import Principal
from app.infrastructure.settings import (
BATCH_MAX_FILES,
BATCH_MAX_TOTAL_BYTES,
BATCH_MAX_TOTAL_ROWS,
INBOX_MAX_PENDING_FILES,
)
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/uploads", tags=["uploads"])
# Identical to the admin batch path. A file that is too large for an admin to
# upload is not somehow acceptable because a colleague sent it.
MAX_UPLOAD_BYTES = 10 * 1024 * 1024
MAX_UPLOAD_ROWS = 2000
class SubmittedFileOut(BaseModel):
filename: str
accepted: bool
rows_total: int = 0
error: Optional[str] = None
class SubmissionOut(BaseModel):
submission_id: Optional[str]
submitted_by: str
files_accepted: int
files_rejected: int
files: List[SubmittedFileOut]
message: str
@router.post("/catalog", status_code=status.HTTP_202_ACCEPTED)
async def submit_catalog_files(
files: List[UploadFile] = File(...),
principal: Principal = Depends(require_permission("upload_catalog")),
) -> SubmissionOut:
"""Accept spreadsheets into the review inbox. Starts nothing.
Returns 202 with a per-file verdict. A drop where some files parse and some
do not is a partial success, not a failure: the good ones are kept and the
caller is told precisely which sheet to fix.
"""
if not files:
raise HTTPException(status_code=400, detail="No files were uploaded.")
if len(files) > BATCH_MAX_FILES:
raise HTTPException(
status_code=413,
detail=(
f"{len(files)} files exceeds the {BATCH_MAX_FILES}-file limit for one "
f"upload. Send them in smaller drops."
),
)
# Refuse before reading a byte if the inbox is already backed up. This is
# the ceiling that stops an unattended key filling the disk one perfectly
# valid file at a time.
already_waiting = inbox.pending_count()
if already_waiting + len(files) > INBOX_MAX_PENDING_FILES:
raise HTTPException(
status_code=429,
detail=(
f"The review inbox already holds {already_waiting} file(s) awaiting "
f"review, and the limit is {INBOX_MAX_PENDING_FILES}. Please wait until "
f"some have been processed."
),
)
accepted: list = []
rejected: list = []
total_bytes = 0
total_rows = 0
for upload in files:
name = upload.filename or "upload.xlsx"
contents = await upload.read()
if not contents:
rejected.append((name, "The file is empty."))
continue
if len(contents) > MAX_UPLOAD_BYTES:
rejected.append((
name,
f"Larger than the {MAX_UPLOAD_BYTES // (1024 * 1024)}MB per-file limit.",
))
continue
total_bytes += len(contents)
if total_bytes > BATCH_MAX_TOTAL_BYTES:
raise HTTPException(
status_code=413,
detail=(
f"This drop is larger than the "
f"{BATCH_MAX_TOTAL_BYTES // (1024 * 1024)}MB total limit."
),
)
# Parse now, while the sender is still here to be told.
try:
frame, _mapping = pipeline.parse_spreadsheet(name, contents)
except HTTPException as exc:
# read_products_dataframe raises this for an unsupported extension
# or a missing Excel reader; its message already names the file and
# says what to do.
rejected.append((name, str(exc.detail)))
continue
except Exception as exc: # noqa: BLE001 - an unreadable sheet is caller error
rejected.append((name, f"Could not parse the file: {exc}"))
continue
if frame.empty:
rejected.append((name, "The file has no data rows."))
continue
rows = int(len(frame))
if rows > MAX_UPLOAD_ROWS:
rejected.append((
name, f"{rows} rows exceeds the {MAX_UPLOAD_ROWS}-row per-file limit."
))
continue
total_rows += rows
if total_rows > BATCH_MAX_TOTAL_ROWS:
raise HTTPException(
status_code=413,
detail=(
f"This drop totals more than {BATCH_MAX_TOTAL_ROWS} rows. "
f"Send it in two smaller drops."
),
)
accepted.append((name, contents, rows))
submission = None
if accepted:
submission = inbox.stage_submission(
accepted,
# The key's NAME, never its secret. Principal.username is the name
# half of the API_KEYS entry (see principal_for_api_key).
submitted_by=principal.username,
)
logger.info(
"Inbox: %d file(s) from %s awaiting review (submission %s)",
len(accepted), principal.username, submission.submission_id,
)
out = [
SubmittedFileOut(filename=n, accepted=True, rows_total=r)
for n, _c, r in accepted
] + [
SubmittedFileOut(filename=n, accepted=False, error=e) for n, e in rejected
]
if accepted and rejected:
message = (
f"{len(accepted)} file(s) received and awaiting review. "
f"{len(rejected)} could not be read - see the errors below and resend those."
)
elif accepted:
message = f"{len(accepted)} file(s) received and awaiting review."
else:
message = "No file could be read. Nothing was received - see the errors below."
return SubmissionOut(
submission_id=submission.submission_id if submission else None,
submitted_by=principal.username,
files_accepted=len(accepted),
files_rejected=len(rejected),
files=out,
message=message,
)