upload-catalog-integration

This commit is contained in:
sriram
2026-08-29 11:21:11 +05:30
parent 5aa2669f7d
commit 27d53fa957
8 changed files with 584 additions and 48 deletions

View File

@@ -0,0 +1,320 @@
"""What a sender can learn about their own drop, and what the result tells them.
Three integrator questions produced this file, and each is pinned here rather
than only answered in prose:
1. How does a held batch get released, and how does the sender find out? The
drop id they were handed has to stay meaningful for the whole lifecycle -
the first version of the inbox deleted the drop on release, which left them
polling a 404 with no way to tell "running" from "declined" from "lost".
2. Does the result carry the identifiers needed to reconcile a sheet against
the catalog? `image_id` is the join key; matching on product_name is what
silently creates duplicates.
3. Is a SKU the sheet supplied preserved, or replaced by a minted one? The
sheet wins, and `sku_source` has to say so without the caller diffing
against the file they sent.
"""
from __future__ import annotations
import io
import pytest
from app.api.batch_job_store import batch_job_store
from app.core import batch_ingest
from app.core import store_catalog_pipeline as pipeline
UPLOAD = "/api/uploads/catalog"
INBOX = "/api/admin/catalog-batch/inbox"
FROM_INBOX = "/api/admin/catalog-batch/from-inbox"
DISMISS = "/api/admin/catalog-batch/inbox/dismiss"
HEADERS = ["Product Name", "Category", "Brand"]
ROWS = [["Amul Butter 100g", "Butter", "Amul"]]
def _csv(headers=HEADERS, rows=ROWS) -> bytes:
return ("\n".join([",".join(headers)] + [",".join(r) for r in rows])).encode()
def _files(*pairs):
return [("files", (n, io.BytesIO(c), "application/octet-stream"))
for n, c in pairs]
@pytest.fixture(autouse=True)
def _isolate_sku_counter(tmp_path, monkeypatch):
from app.services import sku_service
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
@pytest.fixture(autouse=True)
def batch_root(tmp_path, monkeypatch):
root = tmp_path / "batch_uploads"
monkeypatch.setattr(batch_ingest, "BATCH_UPLOAD_DIR", root)
return root
@pytest.fixture(autouse=True)
def submitted(monkeypatch):
from app.core import batch_worker
seen: list = []
monkeypatch.setattr(batch_worker, "submit", seen.append)
return seen
@pytest.fixture(autouse=True)
def _clean_job_store():
batch_job_store._batches.clear()
batch_job_store._cancelled.clear()
yield
batch_job_store._batches.clear()
batch_job_store._cancelled.clear()
@pytest.fixture
def store(monkeypatch):
"""Stand in for the catalog table, keyed the way the real one is.
Keyed on image_id because that is what upsert_brand_products deduplicates
on - which is the whole claim question 2 rests on.
"""
table: dict = {}
def _get(brand):
return list(table.get((brand or "").lower(), {}).values())
def _upsert(brand, products, cleanup=False):
rows = table.setdefault((brand or "").lower(), {})
for product in products:
rows[product["image_id"]] = dict(product)
return len(rows)
monkeypatch.setattr(pipeline, "get_products_by_brand", _get)
monkeypatch.setattr(pipeline, "upsert_brand_products", _upsert)
monkeypatch.setattr(pipeline, "USE_EMBEDDINGS", False)
return table
def _drop(client, sender="priya", *names):
response = client.post(
UPLOAD,
files=_files(*[(n, _csv()) for n in (names or ("a.csv",))]),
data={"sender": sender},
)
assert response.status_code == 202, response.text
return response.json()["batch_id"]
# ---------------------------------------------------------------------------
# 1. Releasing a held drop, as the sender sees it
# ---------------------------------------------------------------------------
def test_a_released_drop_points_the_sender_at_the_run(client, admin_headers):
"""THE answer to "how does a held batch get released".
The sender polls the one id they were given. After an admin starts their
file, that id must still resolve AND carry the id of the run that took it -
otherwise there is no path from the drop to the result.
"""
drop_id = _drop(client, "priya", "catalog.csv")
started = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
headers=admin_headers)
run_id = started.json()["batch_id"]
polled = client.get(f"{UPLOAD}/{drop_id}")
assert polled.status_code == 200, polled.text
entry = polled.json()["files"][0]
assert entry["status"] == "released"
assert entry["released_to"] == run_id
# And that id is readable with no credential either, so the sender can
# follow it without one being issued to them.
assert client.get(f"{UPLOAD}/{run_id}").status_code == 200
def test_a_dismissed_drop_says_so_rather_than_vanishing(client, admin_headers):
drop_id = _drop(client, "priya")
client.post(DISMISS, json={"file_ids": [f"{drop_id}:0"]}, headers=admin_headers)
entry = client.get(f"{UPLOAD}/{drop_id}").json()["files"][0]
assert entry["status"] == "dismissed"
assert entry["released_to"] is None
def test_a_half_released_drop_keeps_the_rest_waiting(client, admin_headers):
drop_id = _drop(client, "priya", "one.csv", "two.csv")
client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]}, headers=admin_headers)
files = client.get(f"{UPLOAD}/{drop_id}").json()["files"]
by_name = {f["filename"]: f for f in files}
assert by_name["one.csv"]["status"] == "released"
assert by_name["two.csv"]["status"] == "queued"
# Still in the inbox, so the drop as a whole has not been settled.
assert client.get(INBOX, headers=admin_headers).json()["pending_count"] == 1
def test_a_retired_drop_leaves_the_inbox_and_frees_its_disk(client, admin_headers,
batch_root):
drop_id = _drop(client, "priya")
client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]}, headers=admin_headers)
assert client.get(INBOX, headers=admin_headers).json()["pending_count"] == 0
# The record survives; the bytes do not. Keeping the manifest is what makes
# the sender's poll work - keeping the sheet would defeat the disk control.
assert not list((batch_root / drop_id).glob("*.csv"))
assert batch_ingest.read_manifest(drop_id).status == "retired"
def test_a_retired_drop_is_purged_on_retention(client, admin_headers):
"""The tombstone is small, but it is not permanent."""
drop_id = _drop(client, "priya")
client.post(DISMISS, json={"file_ids": [f"{drop_id}:0"]}, headers=admin_headers)
import time
purged = batch_ingest.purge_expired(
now=time.time() + (batch_ingest.BATCH_RETENTION_DAYS + 1) * 86400
)
assert drop_id in purged
def test_a_released_drop_is_not_resurrected_by_a_restart(client, admin_headers):
"""`retired` is terminal, so the interrupt sweep must leave it alone."""
drop_id = _drop(client, "priya")
client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]}, headers=admin_headers)
assert drop_id not in batch_ingest.scan_interrupted()
assert batch_ingest.read_manifest(drop_id).status == "retired"
# ---------------------------------------------------------------------------
# 2. Identifiers in the result
# ---------------------------------------------------------------------------
def test_the_result_names_every_product_it_created(client, admin_headers, store):
drop_id = _drop(client, "priya", "catalog.csv")
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
headers=admin_headers).json()["batch_id"]
manifest = batch_ingest.run_batch(run_id)
products = manifest.files[0].result["products"]
assert len(products) == 1
product = products[0]
assert product["brand"] == "amul"
assert product["image_id"]
assert product["product_name"]
assert product["product_sku"]
assert product["disposition"] == "inserted"
def test_resending_a_sheet_reports_unchanged_not_an_empty_list(client, admin_headers,
store):
"""The case that decides whether a caller can reconcile at all.
An unchanged row writes nothing. Reporting only what was written would hand
back an empty list for a completely successful re-send, which reads as total
failure - and the ids are exactly what the caller needs to confirm the rows
are already there.
"""
def _run_one():
drop_id = _drop(client, "priya")
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
headers=admin_headers).json()["batch_id"]
return batch_ingest.run_batch(run_id).files[0].result["products"]
first = _run_one()
second = _run_one()
assert [p["disposition"] for p in first] == ["inserted"]
assert [p["disposition"] for p in second] == ["unchanged"]
# The join key is stable across both, which is what makes it a join key.
assert first[0]["image_id"] == second[0]["image_id"]
assert first[0]["product_sku"] == second[0]["product_sku"]
def test_the_product_manifest_is_absent_from_list_responses(client, admin_headers,
store):
"""Thousands of rows per file is a single-batch payload, not a list one."""
drop_id = _drop(client, "priya")
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
headers=admin_headers).json()["batch_id"]
# on_change is what batch_worker passes in production; without it the job
# store keeps the pre-run copy and the list below would read a stale result.
batch_ingest.run_batch(run_id, on_change=batch_job_store.put)
listed = client.get("/api/admin/catalog-batch/batches",
headers=admin_headers).json()["batches"]
listed_result = [b for b in listed if b["batch_id"] == run_id][0]["files"][0]["result"]
assert "products" not in listed_result
# The counts a list view actually needs are still there.
assert listed_result["inserted"] == 1
single = client.get(f"{UPLOAD}/{run_id}").json()["files"][0]["result"]
assert len(single["products"]) == 1
def test_the_product_manifest_is_capped(client, admin_headers, store, monkeypatch):
monkeypatch.setattr(pipeline, "MAX_REPORTED_PRODUCTS", 1)
rows = [["Amul Butter 100g", "Butter", "Amul"], ["Amul Ghee 1L", "Ghee", "Amul"]]
response = client.post(UPLOAD, files=_files(("a.csv", _csv(rows=rows))))
drop_id = response.json()["batch_id"]
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
headers=admin_headers).json()["batch_id"]
result = batch_ingest.run_batch(run_id).files[0].result
assert len(result["products"]) == 1
assert result["products_truncated"] is True
# Truncating the REPORT must not truncate the work.
assert result["inserted"] == 2
# ---------------------------------------------------------------------------
# 3. Whose SKU wins
# ---------------------------------------------------------------------------
def test_a_sheet_sku_is_preserved_and_labelled():
row = {"product_sku": "MY-SKU-1", "brand": "amul", "product_name": "Butter"}
out = pipeline.stage_7_sku(dict(row))
assert out["product_sku"] == "MY-SKU-1"
assert out["sku_source"] == "sheet"
def test_a_sheets_own_sku_source_is_not_overwritten():
row = {"product_sku": "MY-SKU-1", "sku_source": "erp-export"}
assert pipeline.stage_7_sku(dict(row))["sku_source"] == "erp-export"
def test_a_blank_sku_is_minted_internally():
"""ENABLE_SKU_WEB_LOOKUP is false by default and unset in production, so
the marketplace branch is dead there and this is the only other outcome."""
row = {"product_sku": "", "brand": "Amul", "product_name": "Butter", "size": "100g"}
out = pipeline.stage_7_sku(dict(row))
assert out["product_sku"]
assert out["sku_source"] == "Internal"
def test_a_reupload_does_not_renumber_an_existing_products_sku(client, admin_headers,
store):
"""The sequence counter hands out a fresh number every run, so the stability
of a minted SKU rests entirely on _merge_with_existing keeping the stored
value. Pinned here because nothing else would notice it changing."""
def _run_one():
drop_id = _drop(client, "priya")
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
headers=admin_headers).json()["batch_id"]
return batch_ingest.run_batch(run_id).files[0].result["products"][0]
first = _run_one()
second = _run_one()
assert first["product_sku"] == second["product_sku"]

View File

@@ -283,9 +283,17 @@ def test_dismissing_deletes_the_file_and_its_bytes(client, admin_headers, batch_
assert result.status_code == 200, result.text
assert result.json() == {"dismissed": 1}
assert client.get(INBOX, headers=admin_headers).json()["pending_count"] == 0
# The disposal path on an endpoint anyone can post to. If the bytes survived
# a dismiss, the volume would fill with sheets that were already refused.
assert not (batch_root / batch_id).exists()
assert not list((batch_root / batch_id).glob("*.csv"))
# The RECORD survives, though, and that is deliberate: the sender polls the
# only id they were ever given, and must be told they were declined rather
# than left unable to tell a refusal from a lost file.
polled = client.get(f"{UPLOAD}/{batch_id}")
assert polled.status_code == 200, polled.text
assert polled.json()["files"][0]["status"] == "dismissed"
def test_dismissing_cannot_touch_a_running_batch(client, admin_headers, monkeypatch):