Backend- file ingestion API Updates
This commit is contained in:
@@ -31,6 +31,7 @@ from dagster import (
|
||||
)
|
||||
|
||||
from orchestration.jobs import (
|
||||
batch_ingestion_job,
|
||||
catalog_ingestion_job,
|
||||
embedding_refresh_job,
|
||||
ml_training_job,
|
||||
@@ -149,4 +150,66 @@ def seed_catalog_sensor(context: SensorEvaluationContext):
|
||||
]
|
||||
|
||||
|
||||
ALL_SENSORS = [seed_catalog_sensor]
|
||||
@sensor(
|
||||
job=batch_ingestion_job,
|
||||
minimum_interval_seconds=60,
|
||||
default_status=DefaultSensorStatus.STOPPED,
|
||||
description="Run a batch of uploaded spreadsheets once one is staged and waiting.",
|
||||
)
|
||||
def batch_upload_sensor(context: SensorEvaluationContext):
|
||||
"""Watch BATCH_UPLOAD_DIR for batches the API staged but did not run.
|
||||
|
||||
There is NO schedule for batch ingestion, deliberately. A batch exists
|
||||
because a person uploaded files; there is nothing to do on a cron, and a
|
||||
schedule that woke up hourly to find nothing would be pure cost on a
|
||||
machine that is already sharing itself with Postgres, the API and Vite.
|
||||
|
||||
Like seed_catalog_sensor this is as simple as it can be - a directory
|
||||
listing once a minute, no watcher process, no broker. The cursor is the set
|
||||
of batch ids already requested, so a batch is launched once even though it
|
||||
stays on disk afterwards.
|
||||
|
||||
NOTE ON THE NORMAL PRODUCTION PATH: nothing here is involved. The API runs
|
||||
a staged batch itself, on a bounded worker thread, and this sensor exists
|
||||
for the development machine where Dagster is the thing driving the work.
|
||||
Turning it on alongside a running API would mean both trying to ingest the
|
||||
same batch, which is why it ships STOPPED.
|
||||
"""
|
||||
from app.core import batch_ingest
|
||||
|
||||
root = batch_ingest.batch_root()
|
||||
if not root.exists():
|
||||
return SkipReason("Batch upload directory {} does not exist.".format(root))
|
||||
|
||||
waiting = [
|
||||
m for m in batch_ingest.list_manifests()
|
||||
if m.status in (batch_ingest.QUEUED, batch_ingest.INTERRUPTED)
|
||||
]
|
||||
if not waiting:
|
||||
return SkipReason("No staged batch is waiting to run.")
|
||||
|
||||
already = set(filter(None, (context.cursor or "").split(",")))
|
||||
fresh = [m for m in waiting if m.batch_id not in already]
|
||||
if not fresh:
|
||||
return SkipReason(
|
||||
"{} staged batch(es), all already requested.".format(len(waiting))
|
||||
)
|
||||
|
||||
# Keep the cursor bounded - it is a string in the Dagster instance, not a
|
||||
# log, and only the recent tail is ever consulted.
|
||||
context.update_cursor(",".join(list(already | {m.batch_id for m in fresh})[-200:]))
|
||||
|
||||
context.log.info(
|
||||
"Requesting runs for staged batch(es): %s",
|
||||
", ".join(m.batch_id for m in fresh),
|
||||
)
|
||||
return [
|
||||
RunRequest(
|
||||
run_key=m.batch_id,
|
||||
run_config={"ops": {"batch_manifest": {"config": {"batch_id": m.batch_id}}}},
|
||||
)
|
||||
for m in fresh
|
||||
]
|
||||
|
||||
|
||||
ALL_SENSORS = [seed_catalog_sensor, batch_upload_sensor]
|
||||
|
||||
Reference in New Issue
Block a user