Backend- file ingestion API Updates

This commit is contained in:
sriram
2026-08-28 07:49:56 +05:30
parent f698720ee2
commit a54bd43f8b
17 changed files with 2306 additions and 138 deletions

View File

@@ -31,6 +31,7 @@ from dagster import (
)
from orchestration.jobs import (
batch_ingestion_job,
catalog_ingestion_job,
embedding_refresh_job,
ml_training_job,
@@ -149,4 +150,66 @@ def seed_catalog_sensor(context: SensorEvaluationContext):
]
ALL_SENSORS = [seed_catalog_sensor]
@sensor(
job=batch_ingestion_job,
minimum_interval_seconds=60,
default_status=DefaultSensorStatus.STOPPED,
description="Run a batch of uploaded spreadsheets once one is staged and waiting.",
)
def batch_upload_sensor(context: SensorEvaluationContext):
"""Watch BATCH_UPLOAD_DIR for batches the API staged but did not run.
There is NO schedule for batch ingestion, deliberately. A batch exists
because a person uploaded files; there is nothing to do on a cron, and a
schedule that woke up hourly to find nothing would be pure cost on a
machine that is already sharing itself with Postgres, the API and Vite.
Like seed_catalog_sensor this is as simple as it can be - a directory
listing once a minute, no watcher process, no broker. The cursor is the set
of batch ids already requested, so a batch is launched once even though it
stays on disk afterwards.
NOTE ON THE NORMAL PRODUCTION PATH: nothing here is involved. The API runs
a staged batch itself, on a bounded worker thread, and this sensor exists
for the development machine where Dagster is the thing driving the work.
Turning it on alongside a running API would mean both trying to ingest the
same batch, which is why it ships STOPPED.
"""
from app.core import batch_ingest
root = batch_ingest.batch_root()
if not root.exists():
return SkipReason("Batch upload directory {} does not exist.".format(root))
waiting = [
m for m in batch_ingest.list_manifests()
if m.status in (batch_ingest.QUEUED, batch_ingest.INTERRUPTED)
]
if not waiting:
return SkipReason("No staged batch is waiting to run.")
already = set(filter(None, (context.cursor or "").split(",")))
fresh = [m for m in waiting if m.batch_id not in already]
if not fresh:
return SkipReason(
"{} staged batch(es), all already requested.".format(len(waiting))
)
# Keep the cursor bounded - it is a string in the Dagster instance, not a
# log, and only the recent tail is ever consulted.
context.update_cursor(",".join(list(already | {m.batch_id for m in fresh})[-200:]))
context.log.info(
"Requesting runs for staged batch(es): %s",
", ".join(m.batch_id for m in fresh),
)
return [
RunRequest(
run_key=m.batch_id,
run_config={"ops": {"batch_manifest": {"config": {"batch_id": m.batch_id}}}},
)
for m in fresh
]
ALL_SENSORS = [seed_catalog_sensor, batch_upload_sensor]