Dagster Orchestration

This commit is contained in:
sriram
2026-08-29 14:49:45 +05:30
parent 27d53fa957
commit 998df898db
28 changed files with 1455 additions and 76 deletions

View File

@@ -260,8 +260,16 @@ process - see `app/core/batch_worker.py` for why one worker rather than a
thread per upload. The job here exists so the graph is inspectable and a batch
can be re-run from the Launchpad on a development machine.
Do not turn `batch_upload_sensor` on next to a running API: both would pick up
the same staged batch. That is why it ships STOPPED like the rest.
`batch_upload_sensor` is safe to run next to a live API. Each batch carries a
`runner` field naming its owner, and the sensor (and `_pick_batch_id`) claim
only `runner == "dagster"` — the batches an admin sent here from **Admin →
Dagster Orchestration**. Anything staged for the API's own worker is left
alone, so the two no longer race for the same files. It still ships STOPPED
like the rest: a development machine should not begin ingesting just because a
directory has something in it.
Passing an explicit `{"batch_id": "..."}` in the Launchpad bypasses the runner
filter — that is a person naming a batch, not a poll.
Port 3030, not Dagster's default 3000 - `serve.py` binds `PORTS=3000,8000` and
the frontend nginx also listens on 3000.

View File

@@ -65,18 +65,30 @@ def _pick_batch_id(config: BatchConfig) -> str:
if config.batch_id:
return config.batch_id
# `runner` is the claim check. The API's worker thread and this asset both
# read the same directory and both call the same run_batch, so without it
# every queued batch was fair game to both and enabling batch_upload_sensor
# beside a running API ingested each batch twice. An admin choosing
# "Dagster" in the orchestration tab is what stamps a batch for this side.
#
# An explicit config.batch_id above bypasses this on purpose: that is a
# person naming a batch in the Launchpad, which is an instruction, not a
# poll.
waiting = [
m for m in batch_ingest.list_manifests()
if m.status in (batch_ingest.QUEUED, batch_ingest.INTERRUPTED)
and m.runner == batch_ingest.RUNNER_DAGSTER
]
if not waiting:
raise Failure(
description=(
"No batch_id given and no staged batch is waiting to run.\n\n"
"Stage one through POST /api/admin/catalog-batch/ingest (or "
"POST /api/uploads/catalog), or pass {\"batch_id\": \"...\"} in "
"the Launchpad. Staged batches live under BATCH_UPLOAD_DIR "
f"({batch_ingest.batch_root()})."
"No batch_id given and no staged batch is waiting for Dagster.\n\n"
"Stage one from the Admin -> Dagster Orchestration tab (tick "
"files, Start selected), or pass {\"batch_id\": \"...\"} in the "
"Launchpad to run a specific batch regardless of its runner. "
"Batches staged for this container's own worker are "
"deliberately not picked up here. Staged batches live under "
f"BATCH_UPLOAD_DIR ({batch_ingest.batch_root()})."
),
metadata={"staged_batches": len(batch_ingest.list_manifests())},
)
@@ -208,9 +220,16 @@ def batch_catalog_rows(
the write has already happened - checking there would be checking after the
fact. backend/.env points at production; see orchestration/config.py.
Per-file progress is logged rather than pushed anywhere. In production the
same callback feeds the polling endpoint; in Dagster the run log IS the
progress view, and writing to both would be two sources of truth.
Stage progress is logged here, one line per stage per file, so the run log
is a readable progress view rather than a wall of silence between "Ingesting
x.xlsx" and the final metadata.
It is not pushed anywhere from this callback, and does not need to be:
`run_batch` already flushes the manifest to disk every few seconds, and the
API reads batches from that same manifest. So the Admin orchestration tab
follows a Dagster run through the ordinary batch endpoints, with no channel
between the two processes beyond the file both already use - which is what
keeps that tab working in production, where Dagster does not exist.
"""
from app.core import batch_ingest
@@ -225,13 +244,32 @@ def batch_catalog_rows(
manifest.fetch_images = bool(config.fetch_images or manifest.fetch_images)
batch_ingest.write_manifest(manifest)
seen = {"file": None}
# (filename, stage_index) of the last line written, so a callback that
# fires many times a second produces one line per stage rather than per row.
seen = {"at": None}
def on_change(current):
name = current.current_file
if name and name != seen["file"]:
seen["file"] = name
if not name:
return
entry = next(
(f for f in current.files if f.filename == name and f.status == "running"),
None,
)
if entry is None:
return
at = (name, entry.stage_index)
if at == seen["at"]:
return
seen["at"] = at
if entry.stage_index <= 0:
context.log.info("Ingesting %s", name)
else:
context.log.info(
"%s - stage %d/%d %s (%d/%d rows)",
name, entry.stage_index, entry.total_stages, entry.stage_name,
entry.rows_done, entry.rows_total,
)
result = batch_ingest.run_batch(batch_id, on_change=on_change)
totals = result.totals()

View File

@@ -172,8 +172,13 @@ def batch_upload_sensor(context: SensorEvaluationContext):
NOTE ON THE NORMAL PRODUCTION PATH: nothing here is involved. The API runs
a staged batch itself, on a bounded worker thread, and this sensor exists
for the development machine where Dagster is the thing driving the work.
Turning it on alongside a running API would mean both trying to ingest the
same batch, which is why it ships STOPPED.
It is now safe to run alongside a live API. Both executors read the same
directory, but a batch carries a `runner` naming which one owns it, and
this sensor claims only `runner == "dagster"` - the batches an admin sent
here from the orchestration tab. It still ships STOPPED, because a
development machine should not start ingesting because a directory
happened to have something in it.
"""
from app.core import batch_ingest
@@ -181,12 +186,15 @@ def batch_upload_sensor(context: SensorEvaluationContext):
if not root.exists():
return SkipReason("Batch upload directory {} does not exist.".format(root))
# Only batches an admin explicitly handed to Dagster. See the same filter,
# and the reason for it, in assets/batch_catalog.py:_pick_batch_id.
waiting = [
m for m in batch_ingest.list_manifests()
if m.status in (batch_ingest.QUEUED, batch_ingest.INTERRUPTED)
and m.runner == batch_ingest.RUNNER_DAGSTER
]
if not waiting:
return SkipReason("No staged batch is waiting to run.")
return SkipReason("No staged batch is waiting for Dagster.")
already = set(filter(None, (context.cursor or "").split(",")))
fresh = [m for m in waiting if m.batch_id not in already]