upload-catalog-integration
This commit is contained in:
@@ -187,6 +187,10 @@ class BatchFileOut(BaseModel):
|
||||
filename: str
|
||||
status: str
|
||||
detail: Optional[str] = None
|
||||
# Set only on a file that was released out of the review inbox: the id of
|
||||
# the run that took it, which is how a sender gets from the drop id they
|
||||
# hold to the batch that carries their results.
|
||||
released_to: Optional[str] = None
|
||||
stage_index: int = 0
|
||||
stage_name: str = ""
|
||||
total_stages: int = pipeline.TOTAL_STAGES
|
||||
@@ -214,11 +218,26 @@ class BatchOut(BaseModel):
|
||||
files: List[BatchFileOut]
|
||||
|
||||
|
||||
def to_out(manifest: batch_ingest.BatchManifest) -> BatchOut:
|
||||
def to_out(manifest: batch_ingest.BatchManifest, *, slim: bool = False) -> BatchOut:
|
||||
"""Render a manifest for the API.
|
||||
|
||||
`slim=True` drops the per-file `products` manifest, which is for LIST
|
||||
responses. That list can run to thousands of rows per file
|
||||
(store_catalog_pipeline.MAX_REPORTED_PRODUCTS), so twenty batches rendered
|
||||
in full is a multi-megabyte response to a request that only wanted to know
|
||||
what ran lately. The single-batch read keeps it - that is where a caller
|
||||
goes to reconcile a specific run.
|
||||
"""
|
||||
body = manifest.to_dict()
|
||||
body["files"] = [BatchFileOut(**{
|
||||
key: entry[key] for key in BatchFileOut.model_fields if key in entry
|
||||
}) for entry in body["files"]]
|
||||
files = []
|
||||
for entry in body["files"]:
|
||||
rendered = {key: entry[key] for key in BatchFileOut.model_fields if key in entry}
|
||||
if slim and isinstance(rendered.get("result"), dict):
|
||||
rendered["result"] = {
|
||||
k: v for k, v in rendered["result"].items() if k != "products"
|
||||
}
|
||||
files.append(BatchFileOut(**rendered))
|
||||
body["files"] = files
|
||||
return BatchOut(**{k: v for k, v in body.items() if k in BatchOut.model_fields})
|
||||
|
||||
|
||||
|
||||
@@ -189,7 +189,7 @@ def list_catalog_batches(limit: int = 20) -> dict:
|
||||
m for m in batch_job_store.recent(limit * 10)
|
||||
if m.status != batch_ingest.PENDING
|
||||
][:limit]
|
||||
return {"batches": [_to_out(m) for m in runs]}
|
||||
return {"batches": [_to_out(m, slim=True) for m in runs]}
|
||||
|
||||
|
||||
@router.get("/batches/{batch_id}", dependencies=[Depends(require_admin)])
|
||||
@@ -320,6 +320,27 @@ class InboxDismissOut(BaseModel):
|
||||
dismissed: int
|
||||
|
||||
|
||||
def _retire(batch_id: str, indices: List[int], *, state: str,
|
||||
released_to: Optional[str] = None) -> int:
|
||||
"""Retire files and refresh the cached manifest. Returns how many.
|
||||
|
||||
The refresh is the part that is easy to forget and impossible to see:
|
||||
`batch_job_store.get` prefers its in-memory copy over the disk, so a drop
|
||||
retired without this keeps reporting `queued` to the sender polling it -
|
||||
the exact question this whole record exists to answer. batch_ingest does
|
||||
not know about the store (the store imports IT), so the refresh belongs
|
||||
here, where both are already in hand.
|
||||
"""
|
||||
retired = batch_ingest.retire_files(
|
||||
batch_id, indices, state=state, released_to=released_to,
|
||||
)
|
||||
if retired:
|
||||
updated = batch_ingest.read_manifest(batch_id)
|
||||
if updated:
|
||||
batch_job_store.put(updated)
|
||||
return retired
|
||||
|
||||
|
||||
def _parse_file_ids(file_ids: List[str]) -> Dict[str, List[int]]:
|
||||
"""Group "{batch_id}:{index}" into {batch_id: [index, ...]}.
|
||||
|
||||
@@ -444,10 +465,15 @@ def start_batch_from_inbox(request: InboxStartRequest) -> BatchOut:
|
||||
),
|
||||
)
|
||||
|
||||
# Only now, once the bytes are safely staged under a new id. Removing them
|
||||
# Only now, once the bytes are safely staged under a new id. Retiring them
|
||||
# first would lose the files outright if staging then failed.
|
||||
#
|
||||
# `released_to` is the whole point: the sender polls the drop id they were
|
||||
# given, and this is how they learn which run took their sheet and where to
|
||||
# follow it. Without it a release is indistinguishable from a deletion.
|
||||
for batch_id, indices in grouped.items():
|
||||
batch_ingest.remove_files(batch_id, indices)
|
||||
_retire(batch_id, indices, state=batch_ingest.RELEASED,
|
||||
released_to=manifest.batch_id)
|
||||
|
||||
return _to_out(manifest)
|
||||
|
||||
@@ -459,6 +485,10 @@ def dismiss_inbox_files(request: InboxSelection) -> InboxDismissOut:
|
||||
Deliberately irreversible and deliberately unceremonious: this is the
|
||||
disposal path for a drop nobody wants, and on an endpoint anyone can post to
|
||||
it is the control that keeps the volume from filling with rejected sheets.
|
||||
|
||||
The file entry survives as a record reading `dismissed`, so the sender who
|
||||
polls their drop id is told they were declined rather than left staring at a
|
||||
404. Only the bytes are gone.
|
||||
"""
|
||||
grouped = _parse_file_ids(request.file_ids)
|
||||
if not grouped:
|
||||
@@ -471,6 +501,6 @@ def dismiss_inbox_files(request: InboxSelection) -> InboxDismissOut:
|
||||
# files out of a batch that was mid-run.
|
||||
if not manifest or manifest.status != batch_ingest.PENDING:
|
||||
continue
|
||||
dismissed += batch_ingest.remove_files(batch_id, indices)
|
||||
dismissed += _retire(batch_id, indices, state=batch_ingest.DISMISSED)
|
||||
|
||||
return InboxDismissOut(dismissed=dismissed)
|
||||
|
||||
@@ -239,7 +239,7 @@ def list_my_catalog_batches(
|
||||
mine = [
|
||||
m for m in batch_job_store.recent(limit * 10) if _owns(m, principal)
|
||||
][:limit]
|
||||
return {"batches": [batch_common.to_out(m) for m in mine]}
|
||||
return {"batches": [batch_common.to_out(m, slim=True) for m in mine]}
|
||||
|
||||
|
||||
@router.get("/catalog/{batch_id}")
|
||||
|
||||
Reference in New Issue
Block a user