upload-catalog-integration
This commit is contained in:
320
tests/test_drop_lifecycle.py
Normal file
320
tests/test_drop_lifecycle.py
Normal file
@@ -0,0 +1,320 @@
|
||||
"""What a sender can learn about their own drop, and what the result tells them.
|
||||
|
||||
Three integrator questions produced this file, and each is pinned here rather
|
||||
than only answered in prose:
|
||||
|
||||
1. How does a held batch get released, and how does the sender find out? The
|
||||
drop id they were handed has to stay meaningful for the whole lifecycle -
|
||||
the first version of the inbox deleted the drop on release, which left them
|
||||
polling a 404 with no way to tell "running" from "declined" from "lost".
|
||||
2. Does the result carry the identifiers needed to reconcile a sheet against
|
||||
the catalog? `image_id` is the join key; matching on product_name is what
|
||||
silently creates duplicates.
|
||||
3. Is a SKU the sheet supplied preserved, or replaced by a minted one? The
|
||||
sheet wins, and `sku_source` has to say so without the caller diffing
|
||||
against the file they sent.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
|
||||
import pytest
|
||||
|
||||
from app.api.batch_job_store import batch_job_store
|
||||
from app.core import batch_ingest
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
|
||||
UPLOAD = "/api/uploads/catalog"
|
||||
INBOX = "/api/admin/catalog-batch/inbox"
|
||||
FROM_INBOX = "/api/admin/catalog-batch/from-inbox"
|
||||
DISMISS = "/api/admin/catalog-batch/inbox/dismiss"
|
||||
|
||||
HEADERS = ["Product Name", "Category", "Brand"]
|
||||
ROWS = [["Amul Butter 100g", "Butter", "Amul"]]
|
||||
|
||||
|
||||
def _csv(headers=HEADERS, rows=ROWS) -> bytes:
|
||||
return ("\n".join([",".join(headers)] + [",".join(r) for r in rows])).encode()
|
||||
|
||||
|
||||
def _files(*pairs):
|
||||
return [("files", (n, io.BytesIO(c), "application/octet-stream"))
|
||||
for n, c in pairs]
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolate_sku_counter(tmp_path, monkeypatch):
|
||||
from app.services import sku_service
|
||||
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def batch_root(tmp_path, monkeypatch):
|
||||
root = tmp_path / "batch_uploads"
|
||||
monkeypatch.setattr(batch_ingest, "BATCH_UPLOAD_DIR", root)
|
||||
return root
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def submitted(monkeypatch):
|
||||
from app.core import batch_worker
|
||||
|
||||
seen: list = []
|
||||
monkeypatch.setattr(batch_worker, "submit", seen.append)
|
||||
return seen
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _clean_job_store():
|
||||
batch_job_store._batches.clear()
|
||||
batch_job_store._cancelled.clear()
|
||||
yield
|
||||
batch_job_store._batches.clear()
|
||||
batch_job_store._cancelled.clear()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store(monkeypatch):
|
||||
"""Stand in for the catalog table, keyed the way the real one is.
|
||||
|
||||
Keyed on image_id because that is what upsert_brand_products deduplicates
|
||||
on - which is the whole claim question 2 rests on.
|
||||
"""
|
||||
table: dict = {}
|
||||
|
||||
def _get(brand):
|
||||
return list(table.get((brand or "").lower(), {}).values())
|
||||
|
||||
def _upsert(brand, products, cleanup=False):
|
||||
rows = table.setdefault((brand or "").lower(), {})
|
||||
for product in products:
|
||||
rows[product["image_id"]] = dict(product)
|
||||
return len(rows)
|
||||
|
||||
monkeypatch.setattr(pipeline, "get_products_by_brand", _get)
|
||||
monkeypatch.setattr(pipeline, "upsert_brand_products", _upsert)
|
||||
monkeypatch.setattr(pipeline, "USE_EMBEDDINGS", False)
|
||||
return table
|
||||
|
||||
|
||||
def _drop(client, sender="priya", *names):
|
||||
response = client.post(
|
||||
UPLOAD,
|
||||
files=_files(*[(n, _csv()) for n in (names or ("a.csv",))]),
|
||||
data={"sender": sender},
|
||||
)
|
||||
assert response.status_code == 202, response.text
|
||||
return response.json()["batch_id"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. Releasing a held drop, as the sender sees it
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_released_drop_points_the_sender_at_the_run(client, admin_headers):
|
||||
"""THE answer to "how does a held batch get released".
|
||||
|
||||
The sender polls the one id they were given. After an admin starts their
|
||||
file, that id must still resolve AND carry the id of the run that took it -
|
||||
otherwise there is no path from the drop to the result.
|
||||
"""
|
||||
drop_id = _drop(client, "priya", "catalog.csv")
|
||||
|
||||
started = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
|
||||
headers=admin_headers)
|
||||
run_id = started.json()["batch_id"]
|
||||
|
||||
polled = client.get(f"{UPLOAD}/{drop_id}")
|
||||
assert polled.status_code == 200, polled.text
|
||||
entry = polled.json()["files"][0]
|
||||
assert entry["status"] == "released"
|
||||
assert entry["released_to"] == run_id
|
||||
|
||||
# And that id is readable with no credential either, so the sender can
|
||||
# follow it without one being issued to them.
|
||||
assert client.get(f"{UPLOAD}/{run_id}").status_code == 200
|
||||
|
||||
|
||||
def test_a_dismissed_drop_says_so_rather_than_vanishing(client, admin_headers):
|
||||
drop_id = _drop(client, "priya")
|
||||
|
||||
client.post(DISMISS, json={"file_ids": [f"{drop_id}:0"]}, headers=admin_headers)
|
||||
|
||||
entry = client.get(f"{UPLOAD}/{drop_id}").json()["files"][0]
|
||||
assert entry["status"] == "dismissed"
|
||||
assert entry["released_to"] is None
|
||||
|
||||
|
||||
def test_a_half_released_drop_keeps_the_rest_waiting(client, admin_headers):
|
||||
drop_id = _drop(client, "priya", "one.csv", "two.csv")
|
||||
|
||||
client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]}, headers=admin_headers)
|
||||
|
||||
files = client.get(f"{UPLOAD}/{drop_id}").json()["files"]
|
||||
by_name = {f["filename"]: f for f in files}
|
||||
assert by_name["one.csv"]["status"] == "released"
|
||||
assert by_name["two.csv"]["status"] == "queued"
|
||||
# Still in the inbox, so the drop as a whole has not been settled.
|
||||
assert client.get(INBOX, headers=admin_headers).json()["pending_count"] == 1
|
||||
|
||||
|
||||
def test_a_retired_drop_leaves_the_inbox_and_frees_its_disk(client, admin_headers,
|
||||
batch_root):
|
||||
drop_id = _drop(client, "priya")
|
||||
|
||||
client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]}, headers=admin_headers)
|
||||
|
||||
assert client.get(INBOX, headers=admin_headers).json()["pending_count"] == 0
|
||||
# The record survives; the bytes do not. Keeping the manifest is what makes
|
||||
# the sender's poll work - keeping the sheet would defeat the disk control.
|
||||
assert not list((batch_root / drop_id).glob("*.csv"))
|
||||
assert batch_ingest.read_manifest(drop_id).status == "retired"
|
||||
|
||||
|
||||
def test_a_retired_drop_is_purged_on_retention(client, admin_headers):
|
||||
"""The tombstone is small, but it is not permanent."""
|
||||
drop_id = _drop(client, "priya")
|
||||
client.post(DISMISS, json={"file_ids": [f"{drop_id}:0"]}, headers=admin_headers)
|
||||
|
||||
import time
|
||||
purged = batch_ingest.purge_expired(
|
||||
now=time.time() + (batch_ingest.BATCH_RETENTION_DAYS + 1) * 86400
|
||||
)
|
||||
|
||||
assert drop_id in purged
|
||||
|
||||
|
||||
def test_a_released_drop_is_not_resurrected_by_a_restart(client, admin_headers):
|
||||
"""`retired` is terminal, so the interrupt sweep must leave it alone."""
|
||||
drop_id = _drop(client, "priya")
|
||||
client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]}, headers=admin_headers)
|
||||
|
||||
assert drop_id not in batch_ingest.scan_interrupted()
|
||||
assert batch_ingest.read_manifest(drop_id).status == "retired"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Identifiers in the result
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_the_result_names_every_product_it_created(client, admin_headers, store):
|
||||
drop_id = _drop(client, "priya", "catalog.csv")
|
||||
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
|
||||
headers=admin_headers).json()["batch_id"]
|
||||
|
||||
manifest = batch_ingest.run_batch(run_id)
|
||||
|
||||
products = manifest.files[0].result["products"]
|
||||
assert len(products) == 1
|
||||
product = products[0]
|
||||
assert product["brand"] == "amul"
|
||||
assert product["image_id"]
|
||||
assert product["product_name"]
|
||||
assert product["product_sku"]
|
||||
assert product["disposition"] == "inserted"
|
||||
|
||||
|
||||
def test_resending_a_sheet_reports_unchanged_not_an_empty_list(client, admin_headers,
|
||||
store):
|
||||
"""The case that decides whether a caller can reconcile at all.
|
||||
|
||||
An unchanged row writes nothing. Reporting only what was written would hand
|
||||
back an empty list for a completely successful re-send, which reads as total
|
||||
failure - and the ids are exactly what the caller needs to confirm the rows
|
||||
are already there.
|
||||
"""
|
||||
def _run_one():
|
||||
drop_id = _drop(client, "priya")
|
||||
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
|
||||
headers=admin_headers).json()["batch_id"]
|
||||
return batch_ingest.run_batch(run_id).files[0].result["products"]
|
||||
|
||||
first = _run_one()
|
||||
second = _run_one()
|
||||
|
||||
assert [p["disposition"] for p in first] == ["inserted"]
|
||||
assert [p["disposition"] for p in second] == ["unchanged"]
|
||||
# The join key is stable across both, which is what makes it a join key.
|
||||
assert first[0]["image_id"] == second[0]["image_id"]
|
||||
assert first[0]["product_sku"] == second[0]["product_sku"]
|
||||
|
||||
|
||||
def test_the_product_manifest_is_absent_from_list_responses(client, admin_headers,
|
||||
store):
|
||||
"""Thousands of rows per file is a single-batch payload, not a list one."""
|
||||
drop_id = _drop(client, "priya")
|
||||
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
|
||||
headers=admin_headers).json()["batch_id"]
|
||||
# on_change is what batch_worker passes in production; without it the job
|
||||
# store keeps the pre-run copy and the list below would read a stale result.
|
||||
batch_ingest.run_batch(run_id, on_change=batch_job_store.put)
|
||||
|
||||
listed = client.get("/api/admin/catalog-batch/batches",
|
||||
headers=admin_headers).json()["batches"]
|
||||
listed_result = [b for b in listed if b["batch_id"] == run_id][0]["files"][0]["result"]
|
||||
assert "products" not in listed_result
|
||||
# The counts a list view actually needs are still there.
|
||||
assert listed_result["inserted"] == 1
|
||||
|
||||
single = client.get(f"{UPLOAD}/{run_id}").json()["files"][0]["result"]
|
||||
assert len(single["products"]) == 1
|
||||
|
||||
|
||||
def test_the_product_manifest_is_capped(client, admin_headers, store, monkeypatch):
|
||||
monkeypatch.setattr(pipeline, "MAX_REPORTED_PRODUCTS", 1)
|
||||
rows = [["Amul Butter 100g", "Butter", "Amul"], ["Amul Ghee 1L", "Ghee", "Amul"]]
|
||||
response = client.post(UPLOAD, files=_files(("a.csv", _csv(rows=rows))))
|
||||
drop_id = response.json()["batch_id"]
|
||||
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
|
||||
headers=admin_headers).json()["batch_id"]
|
||||
|
||||
result = batch_ingest.run_batch(run_id).files[0].result
|
||||
|
||||
assert len(result["products"]) == 1
|
||||
assert result["products_truncated"] is True
|
||||
# Truncating the REPORT must not truncate the work.
|
||||
assert result["inserted"] == 2
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. Whose SKU wins
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_sheet_sku_is_preserved_and_labelled():
|
||||
row = {"product_sku": "MY-SKU-1", "brand": "amul", "product_name": "Butter"}
|
||||
|
||||
out = pipeline.stage_7_sku(dict(row))
|
||||
|
||||
assert out["product_sku"] == "MY-SKU-1"
|
||||
assert out["sku_source"] == "sheet"
|
||||
|
||||
|
||||
def test_a_sheets_own_sku_source_is_not_overwritten():
|
||||
row = {"product_sku": "MY-SKU-1", "sku_source": "erp-export"}
|
||||
|
||||
assert pipeline.stage_7_sku(dict(row))["sku_source"] == "erp-export"
|
||||
|
||||
|
||||
def test_a_blank_sku_is_minted_internally():
|
||||
"""ENABLE_SKU_WEB_LOOKUP is false by default and unset in production, so
|
||||
the marketplace branch is dead there and this is the only other outcome."""
|
||||
row = {"product_sku": "", "brand": "Amul", "product_name": "Butter", "size": "100g"}
|
||||
|
||||
out = pipeline.stage_7_sku(dict(row))
|
||||
|
||||
assert out["product_sku"]
|
||||
assert out["sku_source"] == "Internal"
|
||||
|
||||
|
||||
def test_a_reupload_does_not_renumber_an_existing_products_sku(client, admin_headers,
|
||||
store):
|
||||
"""The sequence counter hands out a fresh number every run, so the stability
|
||||
of a minted SKU rests entirely on _merge_with_existing keeping the stored
|
||||
value. Pinned here because nothing else would notice it changing."""
|
||||
def _run_one():
|
||||
drop_id = _drop(client, "priya")
|
||||
run_id = client.post(FROM_INBOX, json={"file_ids": [f"{drop_id}:0"]},
|
||||
headers=admin_headers).json()["batch_id"]
|
||||
return batch_ingest.run_batch(run_id).files[0].result["products"][0]
|
||||
|
||||
first = _run_one()
|
||||
second = _run_one()
|
||||
|
||||
assert first["product_sku"] == second["product_sku"]
|
||||
@@ -283,9 +283,17 @@ def test_dismissing_deletes_the_file_and_its_bytes(client, admin_headers, batch_
|
||||
assert result.status_code == 200, result.text
|
||||
assert result.json() == {"dismissed": 1}
|
||||
assert client.get(INBOX, headers=admin_headers).json()["pending_count"] == 0
|
||||
|
||||
# The disposal path on an endpoint anyone can post to. If the bytes survived
|
||||
# a dismiss, the volume would fill with sheets that were already refused.
|
||||
assert not (batch_root / batch_id).exists()
|
||||
assert not list((batch_root / batch_id).glob("*.csv"))
|
||||
|
||||
# The RECORD survives, though, and that is deliberate: the sender polls the
|
||||
# only id they were ever given, and must be told they were declined rather
|
||||
# than left unable to tell a refusal from a lost file.
|
||||
polled = client.get(f"{UPLOAD}/{batch_id}")
|
||||
assert polled.status_code == 200, polled.text
|
||||
assert polled.json()["files"][0]["status"] == "dismissed"
|
||||
|
||||
|
||||
def test_dismissing_cannot_touch_a_running_batch(client, admin_headers, monkeypatch):
|
||||
|
||||
Reference in New Issue
Block a user