Dagster Orchestration
This commit is contained in:
@@ -260,8 +260,16 @@ process - see `app/core/batch_worker.py` for why one worker rather than a
|
||||
thread per upload. The job here exists so the graph is inspectable and a batch
|
||||
can be re-run from the Launchpad on a development machine.
|
||||
|
||||
Do not turn `batch_upload_sensor` on next to a running API: both would pick up
|
||||
the same staged batch. That is why it ships STOPPED like the rest.
|
||||
`batch_upload_sensor` is safe to run next to a live API. Each batch carries a
|
||||
`runner` field naming its owner, and the sensor (and `_pick_batch_id`) claim
|
||||
only `runner == "dagster"` — the batches an admin sent here from **Admin →
|
||||
Dagster Orchestration**. Anything staged for the API's own worker is left
|
||||
alone, so the two no longer race for the same files. It still ships STOPPED
|
||||
like the rest: a development machine should not begin ingesting just because a
|
||||
directory has something in it.
|
||||
|
||||
Passing an explicit `{"batch_id": "..."}` in the Launchpad bypasses the runner
|
||||
filter — that is a person naming a batch, not a poll.
|
||||
|
||||
Port 3030, not Dagster's default 3000 - `serve.py` binds `PORTS=3000,8000` and
|
||||
the frontend nginx also listens on 3000.
|
||||
|
||||
@@ -65,18 +65,30 @@ def _pick_batch_id(config: BatchConfig) -> str:
|
||||
if config.batch_id:
|
||||
return config.batch_id
|
||||
|
||||
# `runner` is the claim check. The API's worker thread and this asset both
|
||||
# read the same directory and both call the same run_batch, so without it
|
||||
# every queued batch was fair game to both and enabling batch_upload_sensor
|
||||
# beside a running API ingested each batch twice. An admin choosing
|
||||
# "Dagster" in the orchestration tab is what stamps a batch for this side.
|
||||
#
|
||||
# An explicit config.batch_id above bypasses this on purpose: that is a
|
||||
# person naming a batch in the Launchpad, which is an instruction, not a
|
||||
# poll.
|
||||
waiting = [
|
||||
m for m in batch_ingest.list_manifests()
|
||||
if m.status in (batch_ingest.QUEUED, batch_ingest.INTERRUPTED)
|
||||
and m.runner == batch_ingest.RUNNER_DAGSTER
|
||||
]
|
||||
if not waiting:
|
||||
raise Failure(
|
||||
description=(
|
||||
"No batch_id given and no staged batch is waiting to run.\n\n"
|
||||
"Stage one through POST /api/admin/catalog-batch/ingest (or "
|
||||
"POST /api/uploads/catalog), or pass {\"batch_id\": \"...\"} in "
|
||||
"the Launchpad. Staged batches live under BATCH_UPLOAD_DIR "
|
||||
f"({batch_ingest.batch_root()})."
|
||||
"No batch_id given and no staged batch is waiting for Dagster.\n\n"
|
||||
"Stage one from the Admin -> Dagster Orchestration tab (tick "
|
||||
"files, Start selected), or pass {\"batch_id\": \"...\"} in the "
|
||||
"Launchpad to run a specific batch regardless of its runner. "
|
||||
"Batches staged for this container's own worker are "
|
||||
"deliberately not picked up here. Staged batches live under "
|
||||
f"BATCH_UPLOAD_DIR ({batch_ingest.batch_root()})."
|
||||
),
|
||||
metadata={"staged_batches": len(batch_ingest.list_manifests())},
|
||||
)
|
||||
@@ -208,9 +220,16 @@ def batch_catalog_rows(
|
||||
the write has already happened - checking there would be checking after the
|
||||
fact. backend/.env points at production; see orchestration/config.py.
|
||||
|
||||
Per-file progress is logged rather than pushed anywhere. In production the
|
||||
same callback feeds the polling endpoint; in Dagster the run log IS the
|
||||
progress view, and writing to both would be two sources of truth.
|
||||
Stage progress is logged here, one line per stage per file, so the run log
|
||||
is a readable progress view rather than a wall of silence between "Ingesting
|
||||
x.xlsx" and the final metadata.
|
||||
|
||||
It is not pushed anywhere from this callback, and does not need to be:
|
||||
`run_batch` already flushes the manifest to disk every few seconds, and the
|
||||
API reads batches from that same manifest. So the Admin orchestration tab
|
||||
follows a Dagster run through the ordinary batch endpoints, with no channel
|
||||
between the two processes beyond the file both already use - which is what
|
||||
keeps that tab working in production, where Dagster does not exist.
|
||||
"""
|
||||
from app.core import batch_ingest
|
||||
|
||||
@@ -225,13 +244,32 @@ def batch_catalog_rows(
|
||||
manifest.fetch_images = bool(config.fetch_images or manifest.fetch_images)
|
||||
batch_ingest.write_manifest(manifest)
|
||||
|
||||
seen = {"file": None}
|
||||
# (filename, stage_index) of the last line written, so a callback that
|
||||
# fires many times a second produces one line per stage rather than per row.
|
||||
seen = {"at": None}
|
||||
|
||||
def on_change(current):
|
||||
name = current.current_file
|
||||
if name and name != seen["file"]:
|
||||
seen["file"] = name
|
||||
if not name:
|
||||
return
|
||||
entry = next(
|
||||
(f for f in current.files if f.filename == name and f.status == "running"),
|
||||
None,
|
||||
)
|
||||
if entry is None:
|
||||
return
|
||||
at = (name, entry.stage_index)
|
||||
if at == seen["at"]:
|
||||
return
|
||||
seen["at"] = at
|
||||
if entry.stage_index <= 0:
|
||||
context.log.info("Ingesting %s", name)
|
||||
else:
|
||||
context.log.info(
|
||||
"%s - stage %d/%d %s (%d/%d rows)",
|
||||
name, entry.stage_index, entry.total_stages, entry.stage_name,
|
||||
entry.rows_done, entry.rows_total,
|
||||
)
|
||||
|
||||
result = batch_ingest.run_batch(batch_id, on_change=on_change)
|
||||
totals = result.totals()
|
||||
|
||||
@@ -172,8 +172,13 @@ def batch_upload_sensor(context: SensorEvaluationContext):
|
||||
NOTE ON THE NORMAL PRODUCTION PATH: nothing here is involved. The API runs
|
||||
a staged batch itself, on a bounded worker thread, and this sensor exists
|
||||
for the development machine where Dagster is the thing driving the work.
|
||||
Turning it on alongside a running API would mean both trying to ingest the
|
||||
same batch, which is why it ships STOPPED.
|
||||
|
||||
It is now safe to run alongside a live API. Both executors read the same
|
||||
directory, but a batch carries a `runner` naming which one owns it, and
|
||||
this sensor claims only `runner == "dagster"` - the batches an admin sent
|
||||
here from the orchestration tab. It still ships STOPPED, because a
|
||||
development machine should not start ingesting because a directory
|
||||
happened to have something in it.
|
||||
"""
|
||||
from app.core import batch_ingest
|
||||
|
||||
@@ -181,12 +186,15 @@ def batch_upload_sensor(context: SensorEvaluationContext):
|
||||
if not root.exists():
|
||||
return SkipReason("Batch upload directory {} does not exist.".format(root))
|
||||
|
||||
# Only batches an admin explicitly handed to Dagster. See the same filter,
|
||||
# and the reason for it, in assets/batch_catalog.py:_pick_batch_id.
|
||||
waiting = [
|
||||
m for m in batch_ingest.list_manifests()
|
||||
if m.status in (batch_ingest.QUEUED, batch_ingest.INTERRUPTED)
|
||||
and m.runner == batch_ingest.RUNNER_DAGSTER
|
||||
]
|
||||
if not waiting:
|
||||
return SkipReason("No staged batch is waiting to run.")
|
||||
return SkipReason("No staged batch is waiting for Dagster.")
|
||||
|
||||
already = set(filter(None, (context.cursor or "").split(",")))
|
||||
fresh = [m for m in waiting if m.batch_id not in already]
|
||||
|
||||
Reference in New Issue
Block a user