upload-catalog-integration

This commit is contained in:
sriram
2026-08-29 11:21:11 +05:30
parent 5aa2669f7d
commit 27d53fa957
8 changed files with 584 additions and 48 deletions

View File

@@ -187,6 +187,10 @@ class BatchFileOut(BaseModel):
filename: str
status: str
detail: Optional[str] = None
# Set only on a file that was released out of the review inbox: the id of
# the run that took it, which is how a sender gets from the drop id they
# hold to the batch that carries their results.
released_to: Optional[str] = None
stage_index: int = 0
stage_name: str = ""
total_stages: int = pipeline.TOTAL_STAGES
@@ -214,11 +218,26 @@ class BatchOut(BaseModel):
files: List[BatchFileOut]
def to_out(manifest: batch_ingest.BatchManifest) -> BatchOut:
def to_out(manifest: batch_ingest.BatchManifest, *, slim: bool = False) -> BatchOut:
"""Render a manifest for the API.
`slim=True` drops the per-file `products` manifest, which is for LIST
responses. That list can run to thousands of rows per file
(store_catalog_pipeline.MAX_REPORTED_PRODUCTS), so twenty batches rendered
in full is a multi-megabyte response to a request that only wanted to know
what ran lately. The single-batch read keeps it - that is where a caller
goes to reconcile a specific run.
"""
body = manifest.to_dict()
body["files"] = [BatchFileOut(**{
key: entry[key] for key in BatchFileOut.model_fields if key in entry
}) for entry in body["files"]]
files = []
for entry in body["files"]:
rendered = {key: entry[key] for key in BatchFileOut.model_fields if key in entry}
if slim and isinstance(rendered.get("result"), dict):
rendered["result"] = {
k: v for k, v in rendered["result"].items() if k != "products"
}
files.append(BatchFileOut(**rendered))
body["files"] = files
return BatchOut(**{k: v for k, v in body.items() if k in BatchOut.model_fields})

View File

@@ -189,7 +189,7 @@ def list_catalog_batches(limit: int = 20) -> dict:
m for m in batch_job_store.recent(limit * 10)
if m.status != batch_ingest.PENDING
][:limit]
return {"batches": [_to_out(m) for m in runs]}
return {"batches": [_to_out(m, slim=True) for m in runs]}
@router.get("/batches/{batch_id}", dependencies=[Depends(require_admin)])
@@ -320,6 +320,27 @@ class InboxDismissOut(BaseModel):
dismissed: int
def _retire(batch_id: str, indices: List[int], *, state: str,
released_to: Optional[str] = None) -> int:
"""Retire files and refresh the cached manifest. Returns how many.
The refresh is the part that is easy to forget and impossible to see:
`batch_job_store.get` prefers its in-memory copy over the disk, so a drop
retired without this keeps reporting `queued` to the sender polling it -
the exact question this whole record exists to answer. batch_ingest does
not know about the store (the store imports IT), so the refresh belongs
here, where both are already in hand.
"""
retired = batch_ingest.retire_files(
batch_id, indices, state=state, released_to=released_to,
)
if retired:
updated = batch_ingest.read_manifest(batch_id)
if updated:
batch_job_store.put(updated)
return retired
def _parse_file_ids(file_ids: List[str]) -> Dict[str, List[int]]:
"""Group "{batch_id}:{index}" into {batch_id: [index, ...]}.
@@ -444,10 +465,15 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
),
)
# Only now, once the bytes are safely staged under a new id. Removing them
# Only now, once the bytes are safely staged under a new id. Retiring them
# first would lose the files outright if staging then failed.
#
# `released_to` is the whole point: the sender polls the drop id they were
# given, and this is how they learn which run took their sheet and where to
# follow it. Without it a release is indistinguishable from a deletion.
for batch_id, indices in grouped.items():
batch_ingest.remove_files(batch_id, indices)
_retire(batch_id, indices, state=batch_ingest.RELEASED,
released_to=manifest.batch_id)
return _to_out(manifest)
@@ -459,6 +485,10 @@ def dismiss_inbox_files(request: InboxSelection) -> InboxDismissOut:
Deliberately irreversible and deliberately unceremonious: this is the
disposal path for a drop nobody wants, and on an endpoint anyone can post to
it is the control that keeps the volume from filling with rejected sheets.
The file entry survives as a record reading `dismissed`, so the sender who
polls their drop id is told they were declined rather than left staring at a
404. Only the bytes are gone.
"""
grouped = _parse_file_ids(request.file_ids)
if not grouped:
@@ -471,6 +501,6 @@ def dismiss_inbox_files(request: InboxSelection) -> InboxDismissOut:
# files out of a batch that was mid-run.
if not manifest or manifest.status != batch_ingest.PENDING:
continue
dismissed += batch_ingest.remove_files(batch_id, indices)
dismissed += _retire(batch_id, indices, state=batch_ingest.DISMISSED)
return InboxDismissOut(dismissed=dismissed)

View File

@@ -239,7 +239,7 @@ def list_my_catalog_batches(
mine = [
m for m in batch_job_store.recent(limit * 10) if _owns(m, principal)
][:limit]
return {"batches": [batch_common.to_out(m) for m in mine]}
return {"batches": [batch_common.to_out(m, slim=True) for m in mine]}
@router.get("/catalog/{batch_id}")