ingestion updates

This commit is contained in:
sriram
2026-08-28 08:57:10 +05:30
parent a54bd43f8b
commit 0e75d32f61
9 changed files with 1147 additions and 4 deletions

390
tests/test_inbox_upload.py Normal file
View File

@@ -0,0 +1,390 @@
"""Two-actor flow: a colleague drops files, an admin decides what runs.
Follows test_batch_catalog_ingest.py: point the directories at tmp_path,
monkeypatch the storage/embedding boundary, and stub the background worker by
default. Nothing here loads sentence-transformers or torch, nothing reaches the
network, and nothing writes into the repository's data directory.
The most important tests in this file are the ones asserting what an `uploader`
credential CANNOT do. That role exists specifically so an outside contributor
does not get the `user` role's catalog write access, and an assertion is the
only thing that keeps it true as endpoints are added.
"""
from __future__ import annotations
import io
import pytest
from app.core import batch_ingest, inbox
from app.core import store_catalog_pipeline as pipeline
openpyxl = pytest.importorskip("openpyxl")
HEADERS = ["Product Name", "Category", "Brand"]
ROWS = [["Amul Butter 100g", "Butter", "Amul"]]
UPLOAD_KEY = "k" * 43
UPLOAD = "/api/uploads/catalog"
INBOX = "/api/admin/catalog-batch/inbox"
FROM_INBOX = "/api/admin/catalog-batch/from-inbox"
DISMISS = "/api/admin/catalog-batch/inbox/dismiss"
def _csv(rows=ROWS) -> bytes:
return ("\n".join([",".join(HEADERS)] + [",".join(r) for r in rows])).encode()
def _sheet(rows=ROWS) -> bytes:
wb = openpyxl.Workbook()
ws = wb.active
ws.append(HEADERS)
for row in rows:
ws.append(row)
buf = io.BytesIO()
wb.save(buf)
return buf.getvalue()
def _files(*pairs):
return [("files", (n, io.BytesIO(c), "application/octet-stream"))
for n, c in pairs]
@pytest.fixture(autouse=True)
def _isolate_sku_counter(tmp_path, monkeypatch):
from app.services import sku_service
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
@pytest.fixture(autouse=True)
def dirs(tmp_path, monkeypatch):
"""Both staging directories under tmp_path. Read at call time by design."""
monkeypatch.setattr(inbox, "INBOX_UPLOAD_DIR", tmp_path / "inbox")
monkeypatch.setattr(batch_ingest, "BATCH_UPLOAD_DIR", tmp_path / "batch_uploads")
return tmp_path
@pytest.fixture(autouse=True)
def no_background_worker(monkeypatch):
"""Same trap as in test_batch_catalog_ingest: a worker still running after
teardown resolves the directory setting again and writes into the real
data/ directory. Stub it; the tests here are about routing and state."""
from app.core import batch_worker
submitted: list = []
monkeypatch.setattr(batch_worker, "submit", submitted.append)
return submitted
@pytest.fixture(autouse=True)
def upload_key(monkeypatch):
"""One uploader API key.
Patched on `security`, not on `settings`: security.py does
`from ...settings import API_KEYS` at import, binding the dict object, so
rebinding the name in `settings` leaves `principal_for_api_key` still
looking at the original and every request comes back 401.
"""
from app.infrastructure import security
monkeypatch.setattr(
security, "API_KEYS", {UPLOAD_KEY: ("catalog-drop", "uploader")}
)
return {"X-API-Key": UPLOAD_KEY}
@pytest.fixture
def store(monkeypatch):
table: dict = {}
def fake_upsert(brand, rows, cleanup=False):
assert cleanup is False
for row in rows:
table[row["image_id"]] = dict(row)
return len(rows)
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: list(table.values()))
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
return table
def _submit(client, upload_key, *pairs):
return client.post(UPLOAD, files=_files(*pairs), headers=upload_key)
# ---------------------------------------------------------------------------
# The security boundary - the reason the `uploader` role exists
# ---------------------------------------------------------------------------
def test_uploader_key_can_submit(client, upload_key):
r = _submit(client, upload_key, ("a.csv", _csv()))
assert r.status_code == 202
body = r.json()
assert body["files_accepted"] == 1
assert body["submitted_by"] == "catalog-drop"
@pytest.mark.parametrize("method,path", [
("get", "/api/admin/catalog-batch/inbox"),
("post", "/api/admin/catalog-batch/from-inbox"),
("post", "/api/admin/catalog-batch/inbox/dismiss"),
("post", "/api/admin/catalog-batch/preview"),
("post", "/api/admin/catalog-batch/ingest"),
("get", "/api/admin/catalog-batch/batches"),
("get", "/api/admin/catalog-batch/batches/abc"),
("post", "/api/admin/catalog-batch/batches/abc/resume"),
("post", "/api/admin/catalog-batch/batches/abc/cancel"),
("post", "/api/admin/store-catalog/preview"),
("post", "/api/admin/store-catalog/ingest"),
("get", "/api/admin/store-catalog/jobs/abc"),
])
def test_uploader_key_is_refused_on_every_admin_route(client, upload_key, method, path):
"""An uploader credential must reach nothing but its own endpoint."""
r = getattr(client, method)(path, headers=upload_key)
assert r.status_code == 403, f"{method.upper()} {path} returned {r.status_code}"
@pytest.mark.parametrize("path", [
"/api/user/products/add",
"/api/user/products/upload-file",
"/api/upload/stores",
])
def test_uploader_key_has_none_of_the_user_roles_write_access(client, upload_key, path):
"""This is the specific over-grant avoided by NOT reusing the `user` role.
A `user` key would carry add_product / upload_batch_products /
upload_store_inventory - real catalog writes - to solve a problem that
needed one verb.
"""
r = client.post(path, headers=upload_key)
assert r.status_code == 403, f"{path} returned {r.status_code}"
def test_upload_endpoint_rejects_anonymous(client):
r = client.post(UPLOAD, files=_files(("a.csv", _csv())))
assert r.status_code == 401
def test_admin_can_still_use_the_upload_endpoint(client, admin_headers):
"""admin bypasses every permission check (Principal.has_permission)."""
r = client.post(UPLOAD, files=_files(("a.csv", _csv())), headers=admin_headers)
assert r.status_code == 202
def test_role_allow_lists_stay_in_step(client):
"""settings._parse_api_keys duplicates the role names from security.py
because importing back would be a cycle. Pin the duplication."""
from app.infrastructure import settings
from app.infrastructure.security import VALID_ROLES
for role in VALID_ROLES:
parsed = settings._parse_api_keys(f"n:{role}:{'x' * 43}")
assert parsed[("x" * 43)] == ("n", role)
with pytest.raises(RuntimeError):
settings._parse_api_keys(f"n:wizard:{'x' * 43}")
# ---------------------------------------------------------------------------
# Intake
# ---------------------------------------------------------------------------
def test_a_broken_sheet_is_refused_at_upload_and_never_enters_the_inbox(client, upload_key):
r = _submit(client, upload_key, ("good.csv", _csv()), ("broken.pdf", b"%PDF-1.4"))
assert r.status_code == 202
body = r.json()
assert body["files_accepted"] == 1
assert body["files_rejected"] == 1
bad = [f for f in body["files"] if f["filename"] == "broken.pdf"][0]
assert bad["accepted"] is False and bad["error"]
# Only the good one is waiting.
assert inbox.pending_count() == 1
def test_a_drop_where_nothing_parses_stages_no_submission(client, upload_key):
r = _submit(client, upload_key, ("broken.pdf", b"%PDF-1.4"))
assert r.status_code == 202
assert r.json()["submission_id"] is None
assert inbox.pending_count() == 0
assert inbox.list_submissions() == []
def test_uploaded_filenames_cannot_escape_the_inbox(client, upload_key, dirs):
r = _submit(client, upload_key, ("../../evil.csv", _csv()))
assert r.status_code == 202
submission = inbox.list_submissions()[0]
directory = inbox.submission_dir(submission.submission_id)
for entry in submission.files:
written = directory / entry.stored_name
assert written.resolve().parent == directory.resolve()
assert not (dirs / "evil.csv").exists()
def test_a_full_inbox_refuses_further_uploads(client, upload_key, monkeypatch):
from app.api.routers import uploads
monkeypatch.setattr(uploads, "INBOX_MAX_PENDING_FILES", 1)
assert _submit(client, upload_key, ("a.csv", _csv())).status_code == 202
r = _submit(client, upload_key, ("b.csv", _csv()))
assert r.status_code == 429
assert "awaiting review" in r.json()["detail"]
def test_too_many_files_in_one_drop_is_413(client, upload_key, monkeypatch):
from app.api.routers import uploads
monkeypatch.setattr(uploads, "BATCH_MAX_FILES", 2)
r = _submit(client, upload_key,
("a.csv", _csv()), ("b.csv", _csv()), ("c.csv", _csv()))
assert r.status_code == 413
def test_uploading_starts_no_work(client, upload_key, no_background_worker):
"""The whole safety property: an uploader can spend disk, never CPU."""
_submit(client, upload_key, ("a.csv", _csv()))
assert no_background_worker == []
# ---------------------------------------------------------------------------
# The admin side
# ---------------------------------------------------------------------------
def test_inbox_lists_pending_grouped_by_drop(client, upload_key, admin_headers):
_submit(client, upload_key, ("a.csv", _csv()), ("b.csv", _csv()))
_submit(client, upload_key, ("c.csv", _csv()))
r = client.get(INBOX, headers=admin_headers)
assert r.status_code == 200
body = r.json()
assert body["pending_count"] == 3
assert len(body["submissions"]) == 2
assert all(s["submitted_by"] == "catalog-drop" for s in body["submissions"])
assert body["submissions"][0]["files"][0]["rows_total"] == 1
def test_selecting_files_across_two_drops_makes_one_batch(
client, upload_key, admin_headers, store, no_background_worker
):
"""The requirement that ruled out reusing batch-resume: files from
different drops, composed into a single run."""
_submit(client, upload_key, ("a.csv", _csv()), ("b.csv", _csv()))
_submit(client, upload_key, ("c.csv", _csv()))
listing = client.get(INBOX, headers=admin_headers).json()
first = listing["submissions"][0]["files"][0]["file_id"]
second = listing["submissions"][1]["files"][0]["file_id"]
r = client.post(FROM_INBOX, json={"file_ids": [first, second]}, headers=admin_headers)
assert r.status_code == 202
batch = r.json()
assert batch["files_total"] == 2
assert batch["submitted_by"] == "catalog-drop"
assert no_background_worker == [batch["batch_id"]]
# The unselected file is untouched.
after = client.get(INBOX, headers=admin_headers).json()
assert after["pending_count"] == 1
def test_a_started_file_cannot_be_started_twice(client, upload_key, admin_headers, store):
"""Two admin tabs on the same inbox must not double-run a file."""
_submit(client, upload_key, ("a.csv", _csv()))
file_id = client.get(INBOX, headers=admin_headers).json()["submissions"][0]["files"][0]["file_id"]
assert client.post(FROM_INBOX, json={"file_ids": [file_id]}, headers=admin_headers).status_code == 202
again = client.post(FROM_INBOX, json={"file_ids": [file_id]}, headers=admin_headers)
assert again.status_code == 409
assert "already consumed" in again.json()["detail"]
def test_selecting_an_unknown_file_is_404(client, admin_headers):
r = client.post(FROM_INBOX, json={"file_ids": ["nope"]}, headers=admin_headers)
assert r.status_code == 404
def test_selecting_nothing_is_400(client, admin_headers):
assert client.post(FROM_INBOX, json={"file_ids": []}, headers=admin_headers).status_code == 400
def test_dismiss_clears_the_badge_without_running_anything(
client, upload_key, admin_headers, no_background_worker
):
_submit(client, upload_key, ("a.csv", _csv()))
file_id = client.get(INBOX, headers=admin_headers).json()["submissions"][0]["files"][0]["file_id"]
r = client.post(DISMISS, json={"file_ids": [file_id]}, headers=admin_headers)
assert r.status_code == 200
assert r.json() == {"dismissed": 1, "pending_count": 0}
assert no_background_worker == [], "dismiss must not start work"
assert client.get(INBOX, headers=admin_headers).json()["pending_count"] == 0
def test_dismissing_an_already_decided_file_is_409(client, upload_key, admin_headers):
_submit(client, upload_key, ("a.csv", _csv()))
file_id = client.get(INBOX, headers=admin_headers).json()["submissions"][0]["files"][0]["file_id"]
client.post(DISMISS, json={"file_ids": [file_id]}, headers=admin_headers)
again = client.post(DISMISS, json={"file_ids": [file_id]}, headers=admin_headers)
assert again.status_code == 409
# ---------------------------------------------------------------------------
# End to end, and retention
# ---------------------------------------------------------------------------
def test_selected_files_actually_reach_the_catalog(client, upload_key, admin_headers, store):
"""Run the composed batch for real (worker stubbed, pipeline not)."""
_submit(client, upload_key, ("a.csv", _csv()), ("b.xlsx", _sheet()))
listing = client.get(INBOX, headers=admin_headers).json()
ids = [f["file_id"] for f in listing["submissions"][0]["files"]]
started = client.post(FROM_INBOX, json={"file_ids": ids}, headers=admin_headers).json()
result = batch_ingest.run_batch(started["batch_id"])
assert result.status == "done"
assert result.files_done == 2
assert result.submitted_by == "catalog-drop"
assert store, "no rows reached the fake brand table"
def test_purge_removes_decided_submissions_and_spares_pending(client, upload_key, admin_headers, monkeypatch):
import time
monkeypatch.setattr(inbox, "INBOX_RETENTION_DAYS", 7)
_submit(client, upload_key, ("old.csv", _csv()))
old = inbox.list_submissions()[0]
file_id = old.files[0].file_id
client.post(DISMISS, json={"file_ids": [file_id]}, headers=admin_headers)
aged = inbox.read_record(old.submission_id)
aged.created_at = time.time() - 30 * 86400
inbox.write_record(aged)
_submit(client, upload_key, ("pending.csv", _csv()))
still_pending = [s for s in inbox.list_submissions() if s.pending_files][0]
stale_but_unreviewed = inbox.read_record(still_pending.submission_id)
stale_but_unreviewed.created_at = time.time() - 30 * 86400
inbox.write_record(stale_but_unreviewed)
removed = inbox.purge_expired()
assert removed == [old.submission_id]
assert not inbox.submission_dir(old.submission_id).exists()
assert inbox.submission_dir(still_pending.submission_id).exists(), \
"a file nobody has reviewed must never be purged"
def test_purge_is_disabled_when_retention_is_zero(client, upload_key, monkeypatch):
monkeypatch.setattr(inbox, "INBOX_RETENTION_DAYS", 0)
_submit(client, upload_key, ("a.csv", _csv()))
assert inbox.purge_expired() == []
def test_unreadable_record_is_skipped_not_raised(client, upload_key):
_submit(client, upload_key, ("a.csv", _csv()))
sid = inbox.list_submissions()[0].submission_id
inbox.record_path(sid).write_text("{ not json", encoding="utf-8")
assert inbox.read_record(sid) is None
assert inbox.list_submissions() == []
def test_submission_dir_rejects_a_traversing_id():
with pytest.raises(ValueError):
inbox.submission_dir("../escape")