upload files without API-Key

This commit is contained in:
sriram
2026-08-28 17:13:31 +05:30
parent 52f5d3be1d
commit 5aa2669f7d
9 changed files with 1396 additions and 90 deletions

View File

@@ -266,3 +266,50 @@ def stage_and_queue(
return manifest, False
return manifest, True
def stage_pending(
valid: List[Tuple[str, bytes, int]],
invalid: List[Tuple[str, str]],
*,
submitted_by: Optional[str] = None,
) -> batch_ingest.BatchManifest:
"""Write the files down and stop. The review inbox half of the flow.
Deliberately a separate function rather than a `queue=False` flag on
`stage_and_queue`: the admin routes must queue unconditionally, and a shared
boolean is exactly the kind of default that gets inverted in a later edit and
silently starts running work nobody approved.
`use_llm` / `fetch_images` are not taken here. They are run-time choices, and
the person who makes them is the admin pressing Start in the inbox - not the
colleague who dropped the file. They are supplied to `stage_and_queue` when
the selected files become a real batch.
"""
manifest = batch_ingest.stage_batch(
[(name, contents) for name, contents, _n in valid],
invalid=invalid,
submitted_by=submitted_by,
)
# stage_batch records size but not row counts; it never parsed the files.
for entry, (_name, _contents, rows) in zip(manifest.files, valid):
entry.rows_total = rows
manifest.status = batch_ingest.PENDING
manifest.detail = "Waiting for review. Nothing runs until an admin starts it."
batch_ingest.write_manifest(manifest)
batch_job_store.put(manifest)
return manifest
def pending_submissions() -> List[batch_ingest.BatchManifest]:
"""Every drop awaiting review, newest first.
Read from disk rather than the in-memory store: the store is populated by
whatever this process has seen, and an inbox that empties itself on restart
would look exactly like a colleague's files having been processed.
"""
return [
m for m in batch_ingest.list_manifests()
if m.status == batch_ingest.PENDING and m.files
]

View File

@@ -6,6 +6,9 @@
GET /api/admin/catalog-batch/batches/{id} - poll one batch
POST /api/admin/catalog-batch/batches/{id}/resume - after a restart
POST /api/admin/catalog-batch/batches/{id}/cancel - stop the rest
GET /api/admin/catalog-batch/inbox - drops awaiting review
POST /api/admin/catalog-batch/from-inbox - run selected files
POST /api/admin/catalog-batch/inbox/dismiss - discard selected files
This is the multi-file sibling of `store_catalog.py`, and it deliberately does
not replace it: the single-file endpoints are untouched and still work. What is
@@ -33,9 +36,10 @@ difference is that its reads are filtered to the caller's own submissions.
from __future__ import annotations
import queue
from typing import List
from typing import Dict, List, Optional
from fastapi import APIRouter, Depends, File, HTTPException, UploadFile, status
from pydantic import BaseModel, Field
from app.api import batch_common
from app.api.batch_job_store import batch_job_store
@@ -172,8 +176,20 @@ async def ingest_catalog_batch(
@router.get("/batches", dependencies=[Depends(require_admin)])
def list_catalog_batches(limit: int = 20) -> dict:
"""Runs, newest first. Inbox drops are NOT runs and are excluded.
A `pending` submission has never been near the pipeline. Listing it here
would put a row with no progress and no result in the Batch tab, next to
real runs, and the Resume button beside it would be a lie.
"""
limit = max(1, min(limit, 100))
return {"batches": [_to_out(m) for m in batch_job_store.recent(limit)]}
# Over-fetch, then drop the pending ones, so filtering cannot return fewer
# than `limit` runs just because the inbox happens to be busy.
runs = [
m for m in batch_job_store.recent(limit * 10)
if m.status != batch_ingest.PENDING
][:limit]
return {"batches": [_to_out(m) for m in runs]}
@router.get("/batches/{batch_id}", dependencies=[Depends(require_admin)])
@@ -245,3 +261,216 @@ def cancel_catalog_batch(batch_id: str) -> BatchOut:
manifest.detail = "Cancelling - the file currently running will finish first."
batch_job_store.put(manifest)
return _to_out(manifest)
# ---------------------------------------------------------------------------
# The review inbox
# ---------------------------------------------------------------------------
# The other half of the flow in app/api/routers/uploads.py: a colleague posts
# spreadsheets there with no credential, they land as `pending`, and nothing
# runs until somebody here says so. These three routes are what the Inbox tab
# in the admin UI calls (frontend/src/pages/InboxPanel.jsx).
#
# Selection is per FILE and crosses submissions on purpose. "Two sheets from
# Monday's drop plus one from today, as one batch" is the request an operator
# actually has, and it has no expression in a model where the unit is the drop.
class InboxFileOut(BaseModel):
"""One spreadsheet waiting for review."""
# "{batch_id}:{index}". Addressed by a compound id rather than a bare index
# because the UI holds one flat selection set spanning every submission, and
# an index alone is not unique across two of them.
file_id: str
filename: str
rows_total: int = 0
size_bytes: int = 0
class InboxSubmissionOut(BaseModel):
"""One drop: the files that arrived together, and who sent them."""
submission_id: str
submitted_by: Optional[str] = None
created_at: float
files: List[InboxFileOut]
class InboxOut(BaseModel):
pending_count: int
submissions: List[InboxSubmissionOut]
class InboxSelection(BaseModel):
"""The files an admin ticked."""
file_ids: List[str] = Field(default_factory=list)
class InboxStartRequest(InboxSelection):
# Chosen HERE, not by the sender - see the note on the POST handler in
# uploads.py. These commit the host to outbound work, so the decision
# belongs to the person who can see what the machine is already doing.
use_llm: bool = False
fetch_images: bool = False
class InboxDismissOut(BaseModel):
dismissed: int
def _parse_file_ids(file_ids: List[str]) -> Dict[str, List[int]]:
"""Group "{batch_id}:{index}" into {batch_id: [index, ...]}.
Malformed ids are dropped rather than raising: the UI polls every five
seconds and prunes its selection against what came back, so a tick can
legitimately refer to a file another tab started a moment ago. Failing the
whole request would let one stale checkbox block the rest.
"""
grouped: Dict[str, List[int]] = {}
for raw in file_ids or []:
batch_id, _, index = str(raw).partition(":")
if not batch_id or not index.isdigit():
continue
grouped.setdefault(batch_id, []).append(int(index))
return grouped
@router.get("/inbox", dependencies=[Depends(require_admin)])
def list_inbox() -> InboxOut:
"""Every drop awaiting review, newest first."""
submissions = []
pending_count = 0
for manifest in batch_common.pending_submissions():
files = [
InboxFileOut(
file_id=f"{manifest.batch_id}:{entry.index}",
filename=entry.filename,
rows_total=entry.rows_total,
size_bytes=entry.size_bytes,
)
# A file the sender's own upload already rejected is recorded on the
# manifest so they can be told about it, but it has no bytes on disk
# and cannot be run - so it is not offered for selection.
for entry in manifest.files
if entry.status != batch_ingest.FAILED and entry.stored_name
]
if not files:
continue
pending_count += len(files)
submissions.append(
InboxSubmissionOut(
submission_id=manifest.batch_id,
submitted_by=manifest.submitted_by,
created_at=manifest.created_at,
files=files,
)
)
return InboxOut(pending_count=pending_count, submissions=submissions)
@router.post("/from-inbox", status_code=status.HTTP_202_ACCEPTED,
dependencies=[Depends(require_admin)])
def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
"""Take the selected files out of the inbox and run them as one batch.
The bytes are COPIED into a fresh batch rather than the pending manifest
being promoted in place. Two reasons: a selection can span submissions, and
there is no such thing as promoting two manifests into one; and the run gets
its own id, so it has an identity distinct from the drop it came from -
which is what the Batch tab lists and what Resume acts on.
The originals are removed afterwards, so the same sheet cannot be started
twice from a stale checkbox in another tab.
"""
grouped = _parse_file_ids(request.file_ids)
if not grouped:
raise HTTPException(status_code=400, detail="No files were selected.")
# (filename, bytes, rows) - the shape stage_and_queue takes, which is also
# what parse_all returns, so nothing needs reparsing here.
picked: List[tuple] = []
senders: List[str] = []
for batch_id, indices in grouped.items():
manifest = batch_ingest.read_manifest(batch_id)
if not manifest or manifest.status != batch_ingest.PENDING:
continue
directory = batch_ingest.batch_dir(batch_id)
for entry in manifest.files:
if entry.index not in indices or not entry.stored_name:
continue
try:
contents = (directory / entry.stored_name).read_bytes()
except OSError:
# Retention or a concurrent dismiss got there first. Skipping is
# right: the file is genuinely gone, and the poll that follows
# will show it has left the inbox.
continue
picked.append((entry.filename, contents, entry.rows_total))
if manifest.submitted_by:
senders.append(manifest.submitted_by)
if not picked:
raise HTTPException(
status_code=409,
detail=(
"None of those files are still waiting - they may have been "
"started or dismissed already. Refresh the inbox."
),
)
# Preserved so the Batch tab can say where a run came from: one name when a
# drop came from one colleague, a joined list when a batch was assembled
# from several, which is exactly when the question gets asked.
unique_senders = sorted(set(senders))
submitted_by = ", ".join(unique_senders)[:120] if unique_senders else None
manifest, started = batch_common.stage_and_queue(
picked,
[],
use_llm=request.use_llm,
fetch_images=request.fetch_images,
submitted_by=submitted_by,
)
if not started:
raise HTTPException(
status_code=429,
detail=(
"Too many batches are already queued. These files have been "
"taken out of the inbox and saved as batch "
f"{manifest.batch_id} - press Resume on it once the current "
"batch finishes."
),
)
# Only now, once the bytes are safely staged under a new id. Removing them
# first would lose the files outright if staging then failed.
for batch_id, indices in grouped.items():
batch_ingest.remove_files(batch_id, indices)
return _to_out(manifest)
@router.post("/inbox/dismiss", dependencies=[Depends(require_admin)])
def dismiss_inbox_files(request: InboxSelection) -> InboxDismissOut:
"""Discard the selected files. The bytes go with them.
Deliberately irreversible and deliberately unceremonious: this is the
disposal path for a drop nobody wants, and on an endpoint anyone can post to
it is the control that keeps the volume from filling with rejected sheets.
"""
grouped = _parse_file_ids(request.file_ids)
if not grouped:
raise HTTPException(status_code=400, detail="No files were selected.")
dismissed = 0
for batch_id, indices in grouped.items():
manifest = batch_ingest.read_manifest(batch_id)
# Only ever the inbox. Without this check a crafted id would delete
# files out of a batch that was mid-run.
if not manifest or manifest.status != batch_ingest.PENDING:
continue
dismissed += batch_ingest.remove_files(batch_id, indices)
return InboxDismissOut(dismissed=dismissed)

View File

@@ -47,19 +47,21 @@ delete; those stay on the admin router.
from __future__ import annotations
import logging
from typing import List
from typing import List, Optional
from fastapi import APIRouter, Depends, File, HTTPException, UploadFile, status
from fastapi import APIRouter, Depends, File, Form, HTTPException, UploadFile, status
from app.api import batch_common
from app.api.batch_job_store import batch_job_store
from app.api.deps import require_permission
from app.api.deps import get_optional_principal, require_permission
from app.core import batch_ingest
from app.infrastructure.security import Principal
from app.infrastructure.settings import (
BATCH_MAX_FILES,
BATCH_MAX_TOTAL_BYTES,
BATCH_MAX_TOTAL_ROWS,
INBOX_MAX_PENDING_BYTES,
INBOX_MAX_PENDING_FILES,
)
logger = logging.getLogger(__name__)
@@ -101,25 +103,71 @@ def _owns(manifest: batch_ingest.BatchManifest, principal: Principal) -> bool:
return bool(manifest.submitted_by) and manifest.submitted_by == principal.username
def _inbox_capacity_or_429(incoming_files: int, incoming_bytes: int) -> None:
"""Refuse a drop that would push the review inbox past its ceiling.
This endpoint takes files from anyone, and it queues nothing, so
BATCH_QUEUE_MAX - the bound that made an authenticated uploader safe - does
not apply here. Unreviewed submissions accumulate on the volume until
somebody acts on them, which on this host is the scarcest resource there is.
Counted over drops still awaiting review only, so starting or dismissing one
frees its share at once.
"""
pending = batch_common.pending_submissions()
files_now = sum(len(m.files) for m in pending)
bytes_now = sum(f.size_bytes for m in pending for f in m.files)
if files_now + incoming_files > INBOX_MAX_PENDING_FILES:
raise HTTPException(
status_code=429,
detail=(
f"The review inbox is full ({files_now} file(s) awaiting review, "
f"limit {INBOX_MAX_PENDING_FILES}). Nothing was stored. Ask an "
f"admin to clear the inbox, then resend."
),
)
if bytes_now + incoming_bytes > INBOX_MAX_PENDING_BYTES:
raise HTTPException(
status_code=429,
detail=(
f"The review inbox is full "
f"({bytes_now // (1024 * 1024)}MB awaiting review, limit "
f"{INBOX_MAX_PENDING_BYTES // (1024 * 1024)}MB). Nothing was "
f"stored. Ask an admin to clear the inbox, then resend."
),
)
@router.post("/catalog", status_code=status.HTTP_202_ACCEPTED)
async def ingest_catalog_files(
files: List[UploadFile] = File(...),
use_llm: bool = False,
fetch_images: bool = False,
principal: Principal = Depends(require_permission("upload_catalog")),
sender: Optional[str] = Form(None),
principal: Optional[Principal] = Depends(get_optional_principal),
) -> CatalogUploadOut:
"""Accept spreadsheets and run the catalog pipeline over them.
"""Accept spreadsheets into the review inbox. No credential required.
Returns 202 and a `batch_id` to poll. A drop where some files parse and some
do not is a partial success, not a failure: the good ones are ingested and
the bad ones come back in `files` as `status: "failed"` with the reason, so
the caller knows exactly which sheet to fix and resend.
do not is a partial success, not a failure: the good ones are stored and the
bad ones come back in `files` as `status: "failed"` with the reason, so the
caller knows exactly which sheet to fix and resend.
`use_llm` and `fetch_images` default OFF, the opposite of the single-file
admin route. Both are network stages, and image search in particular spawns
a Playwright subprocess that can spend minutes per batch on a host with one
vCPU. An API client that genuinely wants them can ask; an API client that
does not think about it gets the cheap, predictable path.
NOTHING SENT HERE RUNS ON ARRIVAL.
The files are staged and the batch is left `pending`. An admin sees it in the
review inbox, ticks the sheets they want and presses Start; only then does
anything reach the pipeline or the catalog. That gate is what makes an
endpoint anybody can post to acceptable: the cost of an unwanted drop is
disk until someone declines it, not products in the live catalog.
`use_llm` and `fetch_images` are NOT accepted here, though they used to be.
They decide how a run behaves, and the person who decides that is now the
admin pressing Start - not the sender. Leaving them on this endpoint would
let an anonymous caller commit the host to Playwright image search.
`sender` is a free-text label, not identity - it is whatever the caller
typed. It exists because the inbox groups drops by who sent them, and three
colleagues all showing as "anonymous" is an inbox nobody can triage. A real
credential, if one is presented, wins over it.
"""
limits = _limits()
read = await batch_common.read_uploads(files, limits)
@@ -132,45 +180,43 @@ async def ingest_catalog_files(
detail=f"None of the uploaded files could be ingested. {detail}",
)
manifest, started = batch_common.stage_and_queue(
valid,
invalid,
use_llm=use_llm,
fetch_images=fetch_images,
# The key's NAME, never its secret. Principal.username is the name half
# of the API_KEYS entry (see principal_for_api_key).
submitted_by=principal.username,
_inbox_capacity_or_429(
incoming_files=len(valid),
incoming_bytes=sum(len(contents) for _n, contents, _r in valid),
)
if not started:
# The batch is staged and durable, but this caller has no Resume button
# - that lives on the admin router - so the honest instruction is to
# send it again shortly. The id is included so an admin can find and
# resume this one instead if the caller reports it.
raise HTTPException(
status_code=429,
detail=(
f"Too many batches are already queued. Batch {manifest.batch_id} has "
f"been saved but not started; retry this upload shortly."
),
)
# A presented credential still names the sender - `get_optional_principal`
# returns None only when NO credential was sent, and still raises on one
# that is present and wrong. `sender` is trusted for a label and nothing
# else; it is truncated because it is rendered in the admin UI.
submitted_by = (
principal.username if principal
else ((sender or "").strip()[:60] or "anonymous")
)
manifest = batch_common.stage_pending(
valid,
invalid,
submitted_by=submitted_by,
)
logger.info(
"Catalog batch %s queued: %d file(s), %d rejected, from %s",
manifest.batch_id, len(valid), len(invalid), principal.username,
"Catalog drop %s received for review: %d file(s), %d rejected, from %s%s",
manifest.batch_id, len(valid), len(invalid), submitted_by,
"" if principal else " (no credential)",
)
body = batch_common.to_out(manifest).model_dump()
message = (
f"{len(valid)} file(s) received and waiting for review. Nothing runs "
f"until an admin starts them. "
f"Poll GET /api/uploads/catalog/{manifest.batch_id} for status."
)
if invalid:
message = (
f"{len(valid)} file(s) accepted and queued for ingestion. "
f"{len(invalid)} could not be read - see 'files' for the reason on each, "
f"and resend those."
)
else:
message = (
f"{len(valid)} file(s) accepted and queued for ingestion. "
f"Poll GET /api/uploads/catalog/{manifest.batch_id} for progress."
f"{len(valid)} file(s) received and waiting for review. "
f"{len(invalid)} could not be read - see 'files' for the reason on "
f"each, and resend those."
)
return CatalogUploadOut(**body, message=message)
@@ -180,7 +226,12 @@ def list_my_catalog_batches(
limit: int = 20,
principal: Principal = Depends(require_permission("upload_catalog")),
) -> dict:
"""The batches this credential has sent, newest first."""
"""The batches this credential has sent, newest first.
Still credentialed, unlike the single-batch read below. An anonymous LIST
would hand any caller every other sender's drops in one request, which is
a different thing entirely from letting someone check the id they hold.
"""
limit = max(1, min(limit, 100))
# Over-fetch before filtering: `recent` orders by creation across every
# caller, so taking `limit` first would return fewer than `limit` of this
@@ -194,15 +245,26 @@ def list_my_catalog_batches(
@router.get("/catalog/{batch_id}")
def get_my_catalog_batch(
batch_id: str,
principal: Principal = Depends(require_permission("upload_catalog")),
principal: Optional[Principal] = Depends(get_optional_principal),
) -> batch_common.BatchOut:
"""Progress and result for one batch this credential sent.
"""Progress and result for one drop, addressed by its id.
A batch belonging to someone else is a 404, not a 403: whether a given id
exists is not this caller's business, and the two answers must therefore be
indistinguishable.
THE ID IS THE CREDENTIAL HERE, and it has to be: the sender needed no
credential to post, so requiring one to read the result would leave them
unable to find out what happened to their own file. `batch_id` is a
`uuid4().hex` handed only to whoever submitted the drop - 128 bits, not
enumerable - so holding it is the proof of having sent it.
A caller who DID present a credential is held to it, and sees only their own
submissions. That is stricter than anonymous access to the same row, which
is the right way round: a named key should not become a way to browse.
Either way an id you may not see is a 404, never a 403 - whether it exists
is not the caller's business, so the two answers must be indistinguishable.
"""
manifest = batch_job_store.get(batch_id)
if not manifest or not _owns(manifest, principal):
if not manifest:
raise HTTPException(status_code=404, detail="Batch not found")
if principal is not None and not _owns(manifest, principal):
raise HTTPException(status_code=404, detail="Batch not found")
return batch_common.to_out(manifest)