Backend- file ingestion API Updates
This commit is contained in:
@@ -1,3 +1,3 @@
|
||||
from orchestration.assets import catalog, ml, nutrition
|
||||
from orchestration.assets import batch_catalog, catalog, ml, nutrition
|
||||
|
||||
__all__ = ["catalog", "nutrition", "ml"]
|
||||
__all__ = ["catalog", "nutrition", "ml", "batch_catalog"]
|
||||
|
||||
315
orchestration/assets/batch_catalog.py
Normal file
315
orchestration/assets/batch_catalog.py
Normal file
@@ -0,0 +1,315 @@
|
||||
"""Batch ingestion assets: a staged upload -> parsed -> ingested -> reported.
|
||||
|
||||
Every asset delegates to `app.core.batch_ingest`, which is the same module the
|
||||
production endpoint calls. What Dagster adds here is the lineage between the
|
||||
steps, per-run metadata that survives a restart, retries on the step that
|
||||
reaches the network, and a launchpad for re-running a batch that went wrong -
|
||||
none of which the worker thread and its in-memory store can give.
|
||||
|
||||
WHY THESE ASSETS ARE NOT PARTITIONED
|
||||
------------------------------------
|
||||
The brand assets in `catalog.py` are partitioned per brand, because a brand is
|
||||
a stable, enumerable thing and one partition per brand buys per-brand retry and
|
||||
bounded memory. A batch is neither stable nor enumerable in advance - it comes
|
||||
into existence when somebody uploads files. Modelling it as a partition would
|
||||
mean adding a dynamic partition per upload and never removing it, so the
|
||||
partition set grows forever and the UI fills with dead keys. The batch id lives
|
||||
in run config instead, which is what run config is for.
|
||||
|
||||
WHERE THIS RUNS
|
||||
---------------
|
||||
`dagster dev` on a developer's machine. Dagster is not deployed: it is absent
|
||||
from docker-compose.prod.yml, its dependencies are in a separate requirements
|
||||
file the API image does not install, and it is not on any request path. The
|
||||
production path executes the identical functions on a bounded worker thread
|
||||
(app/core/batch_worker.py). That is deliberate - a webserver plus daemon costs
|
||||
400-600MB resident, and the host this ships to does not have it to give.
|
||||
"""
|
||||
# NOTE: deliberately no `from __future__ import annotations` here.
|
||||
# Dagster resolves the decorated function signatures at definition time to
|
||||
# validate the `context` parameter and to infer asset input types. Under
|
||||
# PEP 563/649 the annotations arrive as strings and that validation fails
|
||||
# with "Cannot annotate `context` parameter with type AssetExecutionContext".
|
||||
# Local Python is 3.14, which defers annotations by default, so this is not
|
||||
# hypothetical.
|
||||
|
||||
import time
|
||||
from typing import Any, Dict, List
|
||||
|
||||
from dagster import (
|
||||
AssetExecutionContext,
|
||||
Backoff,
|
||||
Failure,
|
||||
Jitter,
|
||||
MetadataValue,
|
||||
RetryPolicy,
|
||||
asset,
|
||||
)
|
||||
|
||||
from orchestration.config import BatchConfig, database_target, require_local_database
|
||||
|
||||
# Same policy and same reasoning as catalog.py: only the step that leaves the
|
||||
# machine is retried. A file that fails to parse will fail identically on a
|
||||
# second attempt, and retrying it only delays the red run somebody has to read.
|
||||
NETWORK_RETRY = RetryPolicy(
|
||||
max_retries=2, delay=5, backoff=Backoff.EXPONENTIAL, jitter=Jitter.PLUS_MINUS
|
||||
)
|
||||
|
||||
GROUP = "batch_catalog"
|
||||
|
||||
|
||||
def _pick_batch_id(config: BatchConfig) -> str:
|
||||
"""Run config wins; otherwise the oldest batch still waiting to run."""
|
||||
from app.core import batch_ingest
|
||||
|
||||
if config.batch_id:
|
||||
return config.batch_id
|
||||
|
||||
waiting = [
|
||||
m for m in batch_ingest.list_manifests()
|
||||
if m.status in (batch_ingest.QUEUED, batch_ingest.INTERRUPTED)
|
||||
]
|
||||
if not waiting:
|
||||
raise Failure(
|
||||
description=(
|
||||
"No batch_id given and no staged batch is waiting to run.\n\n"
|
||||
"Upload one through Admin -> Batch Catalog Ingestion on the site, "
|
||||
"or pass {\"batch_id\": \"...\"} in the Launchpad. Staged batches "
|
||||
"live under BATCH_UPLOAD_DIR ({}).".format(batch_ingest.batch_root())
|
||||
),
|
||||
metadata={"staged_batches": len(batch_ingest.list_manifests())},
|
||||
)
|
||||
return sorted(waiting, key=lambda m: m.created_at)[0].batch_id
|
||||
|
||||
|
||||
@asset(
|
||||
group_name=GROUP,
|
||||
description="The staged upload batch this run will ingest.",
|
||||
)
|
||||
def batch_manifest(context: AssetExecutionContext, config: BatchConfig) -> Dict[str, Any]:
|
||||
"""Root of the lineage: resolve which batch, and prove it is readable.
|
||||
|
||||
Resolving here rather than inside each downstream asset means the run page
|
||||
shows which files a run actually touched, up front, instead of it being
|
||||
discovered halfway through.
|
||||
"""
|
||||
from app.core import batch_ingest
|
||||
|
||||
batch_id = _pick_batch_id(config)
|
||||
manifest = batch_ingest.read_manifest(batch_id)
|
||||
if manifest is None:
|
||||
raise Failure(
|
||||
description="No manifest for batch '{}' under {}.".format(
|
||||
batch_id, batch_ingest.batch_root()
|
||||
),
|
||||
metadata={"batch_id": batch_id},
|
||||
)
|
||||
|
||||
pending = [f for f in manifest.files if f.status == batch_ingest.QUEUED]
|
||||
context.add_output_metadata(
|
||||
{
|
||||
"batch_id": batch_id,
|
||||
"status": manifest.status,
|
||||
"files_total": manifest.files_total,
|
||||
"files_pending": len(pending),
|
||||
"use_llm": config.use_llm or manifest.use_llm,
|
||||
"fetch_images": config.fetch_images or manifest.fetch_images,
|
||||
"database": database_target(),
|
||||
"files": MetadataValue.md(
|
||||
"\n".join(
|
||||
"- `{}` - {}".format(f.filename, f.status) for f in manifest.files
|
||||
) or "_none_"
|
||||
),
|
||||
}
|
||||
)
|
||||
return {"batch_id": batch_id, "files_pending": len(pending)}
|
||||
|
||||
|
||||
@asset(
|
||||
group_name=GROUP,
|
||||
description="Column mapping and row count for each staged file.",
|
||||
)
|
||||
def batch_parsed_files(
|
||||
context: AssetExecutionContext, batch_manifest: Dict[str, Any]
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Parse every staged file without ingesting anything.
|
||||
|
||||
This is the Dagster equivalent of the UI's Preview step, and it exists for
|
||||
the same reason: the mapping from a store's headers onto catalog fields is
|
||||
a guess, and it is much cheaper to see it now than after eleven stages have
|
||||
run over two thousand rows.
|
||||
|
||||
A file that will not parse is reported, not raised. One unreadable sheet
|
||||
among ten is a data problem worth seeing, not a reason to block the nine.
|
||||
"""
|
||||
from app.core import batch_ingest
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
|
||||
batch_id = batch_manifest["batch_id"]
|
||||
manifest = batch_ingest.read_manifest(batch_id)
|
||||
directory = batch_ingest.batch_dir(batch_id)
|
||||
|
||||
parsed: List[Dict[str, Any]] = []
|
||||
lines = []
|
||||
for entry in manifest.files:
|
||||
if not entry.stored_name:
|
||||
parsed.append({"filename": entry.filename, "ok": False,
|
||||
"error": entry.detail or "not staged"})
|
||||
lines.append("- `{}` - **not staged**".format(entry.filename))
|
||||
continue
|
||||
try:
|
||||
content = (directory / entry.stored_name).read_bytes()
|
||||
frame, mapping = pipeline.parse_spreadsheet(entry.filename, content)
|
||||
except Exception as exc: # noqa: BLE001 - see docstring
|
||||
parsed.append({"filename": entry.filename, "ok": False, "error": str(exc)})
|
||||
lines.append("- `{}` - **unreadable**: {}".format(entry.filename, exc))
|
||||
continue
|
||||
|
||||
recognised = {f: str(c) for f, c in mapping.columns.items()}
|
||||
parsed.append({
|
||||
"filename": entry.filename,
|
||||
"ok": True,
|
||||
"rows": int(len(frame)),
|
||||
"recognised_columns": recognised,
|
||||
"unrecognised_columns": list(mapping.unrecognised),
|
||||
})
|
||||
lines.append("- `{}` - {} rows, {} mapped column(s)".format(
|
||||
entry.filename, len(frame), len(recognised)
|
||||
))
|
||||
|
||||
context.add_output_metadata(
|
||||
{
|
||||
"batch_id": batch_id,
|
||||
"files": len(parsed),
|
||||
"files_readable": sum(1 for p in parsed if p["ok"]),
|
||||
"rows_total": sum(int(p.get("rows") or 0) for p in parsed),
|
||||
"detail": MetadataValue.md("\n".join(lines) or "_none_"),
|
||||
}
|
||||
)
|
||||
return parsed
|
||||
|
||||
|
||||
@asset(
|
||||
group_name=GROUP,
|
||||
retry_policy=NETWORK_RETRY,
|
||||
description="The 11-stage pipeline run over every pending file in the batch.",
|
||||
)
|
||||
def batch_catalog_rows(
|
||||
context: AssetExecutionContext,
|
||||
config: BatchConfig,
|
||||
batch_manifest: Dict[str, Any],
|
||||
batch_parsed_files: List[Dict[str, Any]],
|
||||
) -> Dict[str, Any]:
|
||||
"""Stages 1-11, per file, via the shared `batch_ingest.run_batch`.
|
||||
|
||||
THE WRITE GUARD IS CALLED HERE, not only in the reporting asset downstream.
|
||||
`run_batch` reaches stage 11 and upserts, so by the time the next asset runs
|
||||
the write has already happened - checking there would be checking after the
|
||||
fact. backend/.env points at production; see orchestration/config.py.
|
||||
|
||||
Per-file progress is logged rather than pushed anywhere. In production the
|
||||
same callback feeds the polling endpoint; in Dagster the run log IS the
|
||||
progress view, and writing to both would be two sources of truth.
|
||||
"""
|
||||
from app.core import batch_ingest
|
||||
|
||||
target = require_local_database("batch_catalog_rows")
|
||||
batch_id = batch_manifest["batch_id"]
|
||||
started = time.time()
|
||||
|
||||
# Run config overrides what the upload asked for, so a batch uploaded with
|
||||
# images off can be re-run with them on without re-uploading.
|
||||
manifest = batch_ingest.read_manifest(batch_id)
|
||||
manifest.use_llm = bool(config.use_llm or manifest.use_llm)
|
||||
manifest.fetch_images = bool(config.fetch_images or manifest.fetch_images)
|
||||
batch_ingest.write_manifest(manifest)
|
||||
|
||||
seen = {"file": None}
|
||||
|
||||
def on_change(current):
|
||||
name = current.current_file
|
||||
if name and name != seen["file"]:
|
||||
seen["file"] = name
|
||||
context.log.info("Ingesting %s", name)
|
||||
|
||||
result = batch_ingest.run_batch(batch_id, on_change=on_change)
|
||||
totals = result.totals()
|
||||
|
||||
context.add_output_metadata(
|
||||
{
|
||||
"batch_id": batch_id,
|
||||
"status": result.status,
|
||||
"files_total": result.files_total,
|
||||
"files_done": result.files_done,
|
||||
"files_failed": result.files_failed,
|
||||
"rows_ingested": totals["rows_total"],
|
||||
"products_built": totals["products_built"],
|
||||
"inserted": totals["inserted"],
|
||||
"backfilled": totals["backfilled"],
|
||||
"unchanged": totals["skipped_existing"],
|
||||
"rejected": totals["rejected"],
|
||||
"brands": ", ".join(result.brands()) or "none",
|
||||
"database": target,
|
||||
"cleanup": "False (never deletes rows outside this batch)",
|
||||
"duration_s": round(time.time() - started, 2),
|
||||
}
|
||||
)
|
||||
return result.to_dict()
|
||||
|
||||
|
||||
@asset(
|
||||
group_name=GROUP,
|
||||
description="Per-file outcome of the batch, and a red run if none of it landed.",
|
||||
)
|
||||
def batch_ingest_report(
|
||||
context: AssetExecutionContext, batch_catalog_rows: Dict[str, Any]
|
||||
) -> Dict[str, Any]:
|
||||
"""Turn the batch result into a readable report, and decide run colour.
|
||||
|
||||
A batch where every file failed materializes GREEN without this: nothing
|
||||
raised, so from Dagster's point of view the work completed. That is the
|
||||
worst outcome available here - the run looks fine and the catalog did not
|
||||
change - so it is made loud, in the same spirit as `raw_products` failing on
|
||||
an empty brand.
|
||||
|
||||
A partial batch stays green. Four files landing out of five is a data
|
||||
problem to read about, not a pipeline failure, and failing the run would
|
||||
make the four look like they had not happened.
|
||||
"""
|
||||
files = batch_catalog_rows.get("files") or []
|
||||
done = [f for f in files if f.get("status") == "done"]
|
||||
failed = [f for f in files if f.get("status") == "failed"]
|
||||
|
||||
lines = []
|
||||
for entry in files:
|
||||
lines.append("- `{}` - **{}** - {}".format(
|
||||
entry.get("filename"), entry.get("status"), entry.get("detail") or ""
|
||||
))
|
||||
|
||||
context.add_output_metadata(
|
||||
{
|
||||
"batch_id": batch_catalog_rows.get("batch_id"),
|
||||
"status": batch_catalog_rows.get("status"),
|
||||
"files_done": len(done),
|
||||
"files_failed": len(failed),
|
||||
"totals": MetadataValue.json(batch_catalog_rows.get("totals") or {}),
|
||||
"report": MetadataValue.md("\n".join(lines) or "_no files_"),
|
||||
}
|
||||
)
|
||||
|
||||
if files and not done:
|
||||
raise Failure(
|
||||
description=(
|
||||
"Every file in batch '{}' failed; nothing reached the catalog.".format(
|
||||
batch_catalog_rows.get("batch_id")
|
||||
)
|
||||
),
|
||||
metadata={"files_failed": len(failed)},
|
||||
)
|
||||
|
||||
return {
|
||||
"batch_id": batch_catalog_rows.get("batch_id"),
|
||||
"status": batch_catalog_rows.get("status"),
|
||||
"files_done": len(done),
|
||||
"files_failed": len(failed),
|
||||
}
|
||||
Reference in New Issue
Block a user