Dagster Orchestration

This commit is contained in:
sriram
2026-08-29 14:49:45 +05:30
parent 27d53fa957
commit 998df898db
28 changed files with 1455 additions and 76 deletions

View File

@@ -65,18 +65,30 @@ def _pick_batch_id(config: BatchConfig) -> str:
if config.batch_id:
return config.batch_id
# `runner` is the claim check. The API's worker thread and this asset both
# read the same directory and both call the same run_batch, so without it
# every queued batch was fair game to both and enabling batch_upload_sensor
# beside a running API ingested each batch twice. An admin choosing
# "Dagster" in the orchestration tab is what stamps a batch for this side.
#
# An explicit config.batch_id above bypasses this on purpose: that is a
# person naming a batch in the Launchpad, which is an instruction, not a
# poll.
waiting = [
m for m in batch_ingest.list_manifests()
if m.status in (batch_ingest.QUEUED, batch_ingest.INTERRUPTED)
and m.runner == batch_ingest.RUNNER_DAGSTER
]
if not waiting:
raise Failure(
description=(
"No batch_id given and no staged batch is waiting to run.\n\n"
"Stage one through POST /api/admin/catalog-batch/ingest (or "
"POST /api/uploads/catalog), or pass {\"batch_id\": \"...\"} in "
"the Launchpad. Staged batches live under BATCH_UPLOAD_DIR "
f"({batch_ingest.batch_root()})."
"No batch_id given and no staged batch is waiting for Dagster.\n\n"
"Stage one from the Admin -> Dagster Orchestration tab (tick "
"files, Start selected), or pass {\"batch_id\": \"...\"} in the "
"Launchpad to run a specific batch regardless of its runner. "
"Batches staged for this container's own worker are "
"deliberately not picked up here. Staged batches live under "
f"BATCH_UPLOAD_DIR ({batch_ingest.batch_root()})."
),
metadata={"staged_batches": len(batch_ingest.list_manifests())},
)
@@ -208,9 +220,16 @@ def batch_catalog_rows(
the write has already happened - checking there would be checking after the
fact. backend/.env points at production; see orchestration/config.py.
Per-file progress is logged rather than pushed anywhere. In production the
same callback feeds the polling endpoint; in Dagster the run log IS the
progress view, and writing to both would be two sources of truth.
Stage progress is logged here, one line per stage per file, so the run log
is a readable progress view rather than a wall of silence between "Ingesting
x.xlsx" and the final metadata.
It is not pushed anywhere from this callback, and does not need to be:
`run_batch` already flushes the manifest to disk every few seconds, and the
API reads batches from that same manifest. So the Admin orchestration tab
follows a Dagster run through the ordinary batch endpoints, with no channel
between the two processes beyond the file both already use - which is what
keeps that tab working in production, where Dagster does not exist.
"""
from app.core import batch_ingest
@@ -225,13 +244,32 @@ def batch_catalog_rows(
manifest.fetch_images = bool(config.fetch_images or manifest.fetch_images)
batch_ingest.write_manifest(manifest)
seen = {"file": None}
# (filename, stage_index) of the last line written, so a callback that
# fires many times a second produces one line per stage rather than per row.
seen = {"at": None}
def on_change(current):
name = current.current_file
if name and name != seen["file"]:
seen["file"] = name
if not name:
return
entry = next(
(f for f in current.files if f.filename == name and f.status == "running"),
None,
)
if entry is None:
return
at = (name, entry.stage_index)
if at == seen["at"]:
return
seen["at"] = at
if entry.stage_index <= 0:
context.log.info("Ingesting %s", name)
else:
context.log.info(
"%s - stage %d/%d %s (%d/%d rows)",
name, entry.stage_index, entry.total_stages, entry.stage_name,
entry.rows_done, entry.rows_total,
)
result = batch_ingest.run_batch(batch_id, on_change=on_change)
totals = result.totals()