Backend- file ingestion API Updates

This commit is contained in:
sriram
2026-08-28 07:49:56 +05:30
parent f698720ee2
commit a54bd43f8b
17 changed files with 2306 additions and 138 deletions

View File

@@ -134,6 +134,41 @@ MODEL_ARTIFACTS_DIR = _dir(
"MODEL_ARTIFACTS_DIR", _BACKEND_ROOT / "app" / "intelligence" / "artifacts"
)
# ---------------------------------------------------------------------------
# Batch catalog ingestion (multi-file upload -> the 11-stage pipeline)
# ---------------------------------------------------------------------------
# Staged uploads live under DATA_DIR because that path is already a declared
# volume (backend/Dockerfile). A batch that survives a container restart is the
# whole point of writing the files down instead of holding them in the worker
# thread the way the single-file path does.
#
# Every ceiling below is enforced in the application, not at the proxy. In
# production the browser calls mcp.nearle.ai.in directly, so neither nginx's
# client_max_body_size nor Caddy's request_body cap is in front of these
# endpoints - whatever Traefik defaults to is, and it is not ours to rely on.
BATCH_UPLOAD_DIR = _dir("BATCH_UPLOAD_DIR", DATA_DIR / "batch_uploads")
# Per-file limits stay at the single-upload values (10MB / 2000 rows, see
# app/api/routers/store_catalog.py); these bound the BATCH on top of that.
BATCH_MAX_FILES = int(os.getenv("BATCH_MAX_FILES", "20"))
BATCH_MAX_TOTAL_BYTES = int(os.getenv("BATCH_MAX_TOTAL_BYTES", str(50 * 1024 * 1024)))
BATCH_MAX_TOTAL_ROWS = int(os.getenv("BATCH_MAX_TOTAL_ROWS", "20000"))
# Batches waiting behind the one running. Past this the endpoint returns 429
# rather than accepting work it has no intention of starting soon.
BATCH_QUEUE_MAX = int(os.getenv("BATCH_QUEUE_MAX", "4"))
# Staged files are deleted this many days after the batch was created. Without
# this the upload directory only grows, on a host whose disk is the scarcest
# resource it has.
BATCH_RETENTION_DAYS = int(os.getenv("BATCH_RETENTION_DAYS", "7"))
# Deliberately false. A batch interrupted by a restart is marked "interrupted"
# and waits for someone to press Resume. Auto-resuming would mean a container
# stuck in a restart loop re-runs the heaviest work in the app on every boot,
# which is precisely how a slow start turns into an unrecoverable spiral.
BATCH_AUTO_RESUME = _bool("BATCH_AUTO_RESUME", "false")
# Pristine copies of the bundled seed catalogs and pre-trained models, placed
# here by the Dockerfile at a path that is never itself mounted over.
#