Backend- file ingestion API Updates

This commit is contained in:
sriram
2026-08-28 07:49:56 +05:30
parent f698720ee2
commit a54bd43f8b
17 changed files with 2306 additions and 138 deletions

View File

@@ -1,3 +1,3 @@
from orchestration.assets import catalog, ml, nutrition
from orchestration.assets import batch_catalog, catalog, ml, nutrition
__all__ = ["catalog", "nutrition", "ml"]
__all__ = ["catalog", "nutrition", "ml", "batch_catalog"]

View File

@@ -0,0 +1,315 @@
"""Batch ingestion assets: a staged upload -> parsed -> ingested -> reported.
Every asset delegates to `app.core.batch_ingest`, which is the same module the
production endpoint calls. What Dagster adds here is the lineage between the
steps, per-run metadata that survives a restart, retries on the step that
reaches the network, and a launchpad for re-running a batch that went wrong -
none of which the worker thread and its in-memory store can give.
WHY THESE ASSETS ARE NOT PARTITIONED
------------------------------------
The brand assets in `catalog.py` are partitioned per brand, because a brand is
a stable, enumerable thing and one partition per brand buys per-brand retry and
bounded memory. A batch is neither stable nor enumerable in advance - it comes
into existence when somebody uploads files. Modelling it as a partition would
mean adding a dynamic partition per upload and never removing it, so the
partition set grows forever and the UI fills with dead keys. The batch id lives
in run config instead, which is what run config is for.
WHERE THIS RUNS
---------------
`dagster dev` on a developer's machine. Dagster is not deployed: it is absent
from docker-compose.prod.yml, its dependencies are in a separate requirements
file the API image does not install, and it is not on any request path. The
production path executes the identical functions on a bounded worker thread
(app/core/batch_worker.py). That is deliberate - a webserver plus daemon costs
400-600MB resident, and the host this ships to does not have it to give.
"""
# NOTE: deliberately no `from __future__ import annotations` here.
# Dagster resolves the decorated function signatures at definition time to
# validate the `context` parameter and to infer asset input types. Under
# PEP 563/649 the annotations arrive as strings and that validation fails
# with "Cannot annotate `context` parameter with type AssetExecutionContext".
# Local Python is 3.14, which defers annotations by default, so this is not
# hypothetical.
import time
from typing import Any, Dict, List
from dagster import (
AssetExecutionContext,
Backoff,
Failure,
Jitter,
MetadataValue,
RetryPolicy,
asset,
)
from orchestration.config import BatchConfig, database_target, require_local_database
# Same policy and same reasoning as catalog.py: only the step that leaves the
# machine is retried. A file that fails to parse will fail identically on a
# second attempt, and retrying it only delays the red run somebody has to read.
NETWORK_RETRY = RetryPolicy(
max_retries=2, delay=5, backoff=Backoff.EXPONENTIAL, jitter=Jitter.PLUS_MINUS
)
GROUP = "batch_catalog"
def _pick_batch_id(config: BatchConfig) -> str:
"""Run config wins; otherwise the oldest batch still waiting to run."""
from app.core import batch_ingest
if config.batch_id:
return config.batch_id
waiting = [
m for m in batch_ingest.list_manifests()
if m.status in (batch_ingest.QUEUED, batch_ingest.INTERRUPTED)
]
if not waiting:
raise Failure(
description=(
"No batch_id given and no staged batch is waiting to run.\n\n"
"Upload one through Admin -> Batch Catalog Ingestion on the site, "
"or pass {\"batch_id\": \"...\"} in the Launchpad. Staged batches "
"live under BATCH_UPLOAD_DIR ({}).".format(batch_ingest.batch_root())
),
metadata={"staged_batches": len(batch_ingest.list_manifests())},
)
return sorted(waiting, key=lambda m: m.created_at)[0].batch_id
@asset(
group_name=GROUP,
description="The staged upload batch this run will ingest.",
)
def batch_manifest(context: AssetExecutionContext, config: BatchConfig) -> Dict[str, Any]:
"""Root of the lineage: resolve which batch, and prove it is readable.
Resolving here rather than inside each downstream asset means the run page
shows which files a run actually touched, up front, instead of it being
discovered halfway through.
"""
from app.core import batch_ingest
batch_id = _pick_batch_id(config)
manifest = batch_ingest.read_manifest(batch_id)
if manifest is None:
raise Failure(
description="No manifest for batch '{}' under {}.".format(
batch_id, batch_ingest.batch_root()
),
metadata={"batch_id": batch_id},
)
pending = [f for f in manifest.files if f.status == batch_ingest.QUEUED]
context.add_output_metadata(
{
"batch_id": batch_id,
"status": manifest.status,
"files_total": manifest.files_total,
"files_pending": len(pending),
"use_llm": config.use_llm or manifest.use_llm,
"fetch_images": config.fetch_images or manifest.fetch_images,
"database": database_target(),
"files": MetadataValue.md(
"\n".join(
"- `{}` - {}".format(f.filename, f.status) for f in manifest.files
) or "_none_"
),
}
)
return {"batch_id": batch_id, "files_pending": len(pending)}
@asset(
group_name=GROUP,
description="Column mapping and row count for each staged file.",
)
def batch_parsed_files(
context: AssetExecutionContext, batch_manifest: Dict[str, Any]
) -> List[Dict[str, Any]]:
"""Parse every staged file without ingesting anything.
This is the Dagster equivalent of the UI's Preview step, and it exists for
the same reason: the mapping from a store's headers onto catalog fields is
a guess, and it is much cheaper to see it now than after eleven stages have
run over two thousand rows.
A file that will not parse is reported, not raised. One unreadable sheet
among ten is a data problem worth seeing, not a reason to block the nine.
"""
from app.core import batch_ingest
from app.core import store_catalog_pipeline as pipeline
batch_id = batch_manifest["batch_id"]
manifest = batch_ingest.read_manifest(batch_id)
directory = batch_ingest.batch_dir(batch_id)
parsed: List[Dict[str, Any]] = []
lines = []
for entry in manifest.files:
if not entry.stored_name:
parsed.append({"filename": entry.filename, "ok": False,
"error": entry.detail or "not staged"})
lines.append("- `{}` - **not staged**".format(entry.filename))
continue
try:
content = (directory / entry.stored_name).read_bytes()
frame, mapping = pipeline.parse_spreadsheet(entry.filename, content)
except Exception as exc: # noqa: BLE001 - see docstring
parsed.append({"filename": entry.filename, "ok": False, "error": str(exc)})
lines.append("- `{}` - **unreadable**: {}".format(entry.filename, exc))
continue
recognised = {f: str(c) for f, c in mapping.columns.items()}
parsed.append({
"filename": entry.filename,
"ok": True,
"rows": int(len(frame)),
"recognised_columns": recognised,
"unrecognised_columns": list(mapping.unrecognised),
})
lines.append("- `{}` - {} rows, {} mapped column(s)".format(
entry.filename, len(frame), len(recognised)
))
context.add_output_metadata(
{
"batch_id": batch_id,
"files": len(parsed),
"files_readable": sum(1 for p in parsed if p["ok"]),
"rows_total": sum(int(p.get("rows") or 0) for p in parsed),
"detail": MetadataValue.md("\n".join(lines) or "_none_"),
}
)
return parsed
@asset(
group_name=GROUP,
retry_policy=NETWORK_RETRY,
description="The 11-stage pipeline run over every pending file in the batch.",
)
def batch_catalog_rows(
context: AssetExecutionContext,
config: BatchConfig,
batch_manifest: Dict[str, Any],
batch_parsed_files: List[Dict[str, Any]],
) -> Dict[str, Any]:
"""Stages 1-11, per file, via the shared `batch_ingest.run_batch`.
THE WRITE GUARD IS CALLED HERE, not only in the reporting asset downstream.
`run_batch` reaches stage 11 and upserts, so by the time the next asset runs
the write has already happened - checking there would be checking after the
fact. backend/.env points at production; see orchestration/config.py.
Per-file progress is logged rather than pushed anywhere. In production the
same callback feeds the polling endpoint; in Dagster the run log IS the
progress view, and writing to both would be two sources of truth.
"""
from app.core import batch_ingest
target = require_local_database("batch_catalog_rows")
batch_id = batch_manifest["batch_id"]
started = time.time()
# Run config overrides what the upload asked for, so a batch uploaded with
# images off can be re-run with them on without re-uploading.
manifest = batch_ingest.read_manifest(batch_id)
manifest.use_llm = bool(config.use_llm or manifest.use_llm)
manifest.fetch_images = bool(config.fetch_images or manifest.fetch_images)
batch_ingest.write_manifest(manifest)
seen = {"file": None}
def on_change(current):
name = current.current_file
if name and name != seen["file"]:
seen["file"] = name
context.log.info("Ingesting %s", name)
result = batch_ingest.run_batch(batch_id, on_change=on_change)
totals = result.totals()
context.add_output_metadata(
{
"batch_id": batch_id,
"status": result.status,
"files_total": result.files_total,
"files_done": result.files_done,
"files_failed": result.files_failed,
"rows_ingested": totals["rows_total"],
"products_built": totals["products_built"],
"inserted": totals["inserted"],
"backfilled": totals["backfilled"],
"unchanged": totals["skipped_existing"],
"rejected": totals["rejected"],
"brands": ", ".join(result.brands()) or "none",
"database": target,
"cleanup": "False (never deletes rows outside this batch)",
"duration_s": round(time.time() - started, 2),
}
)
return result.to_dict()
@asset(
group_name=GROUP,
description="Per-file outcome of the batch, and a red run if none of it landed.",
)
def batch_ingest_report(
context: AssetExecutionContext, batch_catalog_rows: Dict[str, Any]
) -> Dict[str, Any]:
"""Turn the batch result into a readable report, and decide run colour.
A batch where every file failed materializes GREEN without this: nothing
raised, so from Dagster's point of view the work completed. That is the
worst outcome available here - the run looks fine and the catalog did not
change - so it is made loud, in the same spirit as `raw_products` failing on
an empty brand.
A partial batch stays green. Four files landing out of five is a data
problem to read about, not a pipeline failure, and failing the run would
make the four look like they had not happened.
"""
files = batch_catalog_rows.get("files") or []
done = [f for f in files if f.get("status") == "done"]
failed = [f for f in files if f.get("status") == "failed"]
lines = []
for entry in files:
lines.append("- `{}` - **{}** - {}".format(
entry.get("filename"), entry.get("status"), entry.get("detail") or ""
))
context.add_output_metadata(
{
"batch_id": batch_catalog_rows.get("batch_id"),
"status": batch_catalog_rows.get("status"),
"files_done": len(done),
"files_failed": len(failed),
"totals": MetadataValue.json(batch_catalog_rows.get("totals") or {}),
"report": MetadataValue.md("\n".join(lines) or "_no files_"),
}
)
if files and not done:
raise Failure(
description=(
"Every file in batch '{}' failed; nothing reached the catalog.".format(
batch_catalog_rows.get("batch_id")
)
),
metadata={"files_failed": len(failed)},
)
return {
"batch_id": batch_catalog_rows.get("batch_id"),
"status": batch_catalog_rows.get("status"),
"files_done": len(done),
"files_failed": len(failed),
}