Dagster Orchestration

This commit is contained in:
sriram
2026-08-29 14:49:45 +05:30
parent 27d53fa957
commit 998df898db
28 changed files with 1455 additions and 76 deletions

View File

@@ -0,0 +1,177 @@
"""Tests for the per-file stage timeline and for batch runner ownership.
Two features, tested together because they exist for the same screen: the Admin
"Dagster Orchestration" tab, which shows every file's progress through the 11
named pipeline stages and has to say which executor owns the batch it is
watching.
Fixture conventions follow test_batch_catalog_ingest.py: BATCH_UPLOAD_DIR points
at tmp_path, the storage and embedding boundary is patched on the pipeline
module object, and nothing here touches the network or a real database.
"""
from __future__ import annotations
import io
import pytest
from app.core import batch_ingest
from app.core import store_catalog_pipeline as pipeline
openpyxl = pytest.importorskip("openpyxl")
# Three brands on purpose: stages 8-11 run once per brand, so this is the sheet
# shape that used to produce a jittering stage index and duplicate records.
THREE_BRANDS = [
["Amul Butter 100g", "Dairy", "Amul"],
["Britannia Marie Gold 250g", "Biscuits", "Britannia"],
["Cadbury Dairy Milk 150g", "Chocolate", "Cadbury"],
]
def _sheet(rows=THREE_BRANDS) -> bytes:
wb = openpyxl.Workbook()
ws = wb.active
ws.append(["Product Name", "Category", "Brand"])
for row in rows:
ws.append(row)
buf = io.BytesIO()
wb.save(buf)
return buf.getvalue()
@pytest.fixture(autouse=True)
def _isolate_sku_counter(tmp_path, monkeypatch):
from app.services import sku_service
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
@pytest.fixture(autouse=True)
def batch_root(tmp_path, monkeypatch):
root = tmp_path / "batch_uploads"
monkeypatch.setattr(batch_ingest, "BATCH_UPLOAD_DIR", root)
return root
@pytest.fixture(autouse=True)
def store(monkeypatch):
table: dict = {}
def fake_upsert(brand, rows, cleanup=False):
assert cleanup is False, "cleanup=True would delete the brand's existing catalog"
for row in rows:
table[row["image_id"]] = dict(row)
return len(rows)
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: list(table.values()))
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
return table
def _run_one_file() -> batch_ingest.BatchFile:
manifest = batch_ingest.stage_batch([("catalog.xlsx", _sheet())])
result = batch_ingest.run_batch(manifest.batch_id)
return result.files[0]
# ---------------------------------------------------------------------------
# The stage timeline
# ---------------------------------------------------------------------------
def test_a_finished_file_records_every_stage_exactly_once():
"""Stages 8-11 run once per brand. Three brands must still yield eleven
records, not twenty-three."""
entry = _run_one_file()
assert entry.status == batch_ingest.DONE
assert [s.index for s in entry.stages] == list(range(1, pipeline.TOTAL_STAGES + 1))
def test_the_timeline_names_match_the_pipeline():
entry = _run_one_file()
assert [s.name for s in entry.stages] == list(pipeline.STAGE_NAMES)
def test_every_stage_is_closed_once_the_file_finishes():
"""The last stage never sees a following tick to close it, and a file that
raised leaves whichever stage it died in open."""
entry = _run_one_file()
assert all(s.finished_at is not None for s in entry.stages)
assert all(s.started_at <= s.finished_at for s in entry.stages)
def test_the_timeline_survives_a_reread_from_disk():
"""A Dagster run is a different process; the API only ever sees the
manifest that run flushed to disk."""
manifest = batch_ingest.stage_batch([("catalog.xlsx", _sheet())])
batch_ingest.run_batch(manifest.batch_id)
reloaded = batch_ingest.read_manifest(manifest.batch_id)
stages = reloaded.files[0].stages
assert len(stages) == pipeline.TOTAL_STAGES
assert isinstance(stages[0], batch_ingest.StageRecord)
def test_a_queued_file_has_no_timeline_yet():
manifest = batch_ingest.stage_batch([("catalog.xlsx", _sheet())])
assert manifest.files[0].stages == []
def test_a_restart_clears_a_half_finished_timeline():
"""It described an attempt that no longer counts; the re-run builds a new
one rather than appending to it."""
manifest = batch_ingest.stage_batch([("catalog.xlsx", _sheet())])
entry = manifest.files[0]
entry.status = batch_ingest.RUNNING
entry.stages = [batch_ingest.StageRecord(index=1, name="Brand Resolution", started_at=1.0)]
batch_ingest.write_manifest(manifest)
batch_ingest.scan_interrupted()
after = batch_ingest.read_manifest(manifest.batch_id).files[0]
assert after.status == batch_ingest.QUEUED
assert after.stages == []
def test_repeated_ticks_for_one_stage_fold_into_one_record():
"""Directly, so the folding rule is pinned independently of a real run."""
entry = batch_ingest.BatchFile(index=0, filename="x.xlsx", stored_name="x")
batch_ingest._record_stage(entry, 8, "Barcode", 0, 5, 100.0)
batch_ingest._record_stage(entry, 9, "HSN", 5, 5, 101.0)
batch_ingest._record_stage(entry, 8, "Barcode", 0, 9, 102.0) # next brand
batch_ingest._record_stage(entry, 9, "HSN", 9, 9, 103.0)
assert [s.index for s in entry.stages] == [8, 9]
# High-water mark across brands, and the first sighting keeps its start.
assert entry.stages[0].rows_total == 9
assert entry.stages[0].started_at == 100.0
# ---------------------------------------------------------------------------
# Runner ownership
# ---------------------------------------------------------------------------
def test_a_batch_defaults_to_the_in_process_runner():
"""Every manifest written before `runner` existed must keep behaving as it
did, which means the default has to be the API's own worker."""
manifest = batch_ingest.stage_batch([("catalog.xlsx", _sheet())])
assert manifest.runner == batch_ingest.RUNNER_INPROCESS
def test_the_runner_survives_a_reread_from_disk():
manifest = batch_ingest.stage_batch([("catalog.xlsx", _sheet())])
manifest.runner = batch_ingest.RUNNER_DAGSTER
batch_ingest.write_manifest(manifest)
assert batch_ingest.read_manifest(manifest.batch_id).runner == batch_ingest.RUNNER_DAGSTER
def test_a_manifest_without_a_runner_field_reads_as_in_process():
"""Manifests staged before this field existed are still on disk."""
manifest = batch_ingest.stage_batch([("catalog.xlsx", _sheet())])
raw = manifest.to_dict()
raw.pop("runner")
assert batch_ingest.BatchManifest.from_dict(raw).runner == batch_ingest.RUNNER_INPROCESS

View File

@@ -0,0 +1,96 @@
"""Tests for image selection - which candidate becomes a product's primary image.
`_select_best_images` returns a ranked list and index 0 becomes the row's
`image_url`, which is what a ProductCard shows. Its scoring is built almost
entirely from URL-string heuristics, and domain reputation is worth up to +20
while a matching product word used to be worth +1. That let a well-known shop's
photo of a *different* product outrank the correct image, which is how a
Coca-Cola product ended up fronted by a Limca combo shot and how every Britannia
biscuit ended up showing Good Day.
These tests pin the ordering rule: a URL that names THIS product outranks one
that only names its brand, however reputable the brand URL's domain.
"""
from __future__ import annotations
from app.core.catalog_engine import catalog_engine
def _top(urls, title, brand):
return catalog_engine._select_best_images(urls, title, brand, max_images=10)[0]
def test_a_product_matching_url_beats_a_higher_domain_score():
"""media.britannia.co.in scores +20 as an official brand domain; the
correct image sits on an anonymous CDN worth +8. Relevance must still win."""
good_day_on_official_domain = "https://media.britannia.co.in/good_day_cashew.jpg"
marie_gold_on_a_cdn = "https://cdn.example.com/marie_gold_biscuit.jpg"
assert _top(
[good_day_on_official_domain, marie_gold_on_a_cdn],
"Britannia Marie Gold 250g",
"Britannia",
) == marie_gold_on_a_cdn
def test_britannia_siblings_each_get_their_own_image():
"""The reported bug: Good Day, Marie Gold and Milk Bikis all showed Good Day."""
candidates = [
"https://media.britannia.co.in/britannia_good_day_cashew_cookies.jpg",
"https://media.britannia.co.in/britannia_marie_gold_biscuits.jpg",
"https://media.britannia.co.in/britannia_milk_bikis_pack.jpg",
]
assert "good_day" in _top(candidates, "Britannia Good Day Cashew Cookies 200g", "Britannia")
assert "marie_gold" in _top(candidates, "Britannia Marie Gold 250g", "Britannia")
assert "milk_bikis" in _top(candidates, "Britannia Milk Bikis 150g", "Britannia")
def test_the_earlier_words_of_a_title_identify_it_more_strongly():
"""A title runs brand -> sub-brand -> variant -> size. A combo listing that
shares only the trailing modifier ("lemon") must not outrank the image that
names the product itself ("sprite")."""
combo = (
"https://www.bigbasket.com/media/uploads/p/xl/1212306_1-bb-combo-coca-cola-"
"soft-drink-original-taste-750-ml-limca-soft-drink-lemon-lime-750-ml.jpg"
)
sprite = "https://cdn.example.com/750ml-sprite-soft-drink-1000x1000.jpg"
assert _top([combo, sprite], "Coca-Cola Sprite Lemon 750ml", "Coca-Cola") == sprite
def test_a_brand_only_url_is_kept_rather_than_discarded():
"""Relevance is a ranking gate, not a filter: a product whose words appear
in no URL (CDN hashes, opaque filenames) must still end up with images."""
opaque = "https://cdn.example.com/a1b2c3d4e5.jpg"
assert catalog_engine._select_best_images(
[opaque], "Britannia Tiger 100g", "Britannia"
) == [opaque]
def test_the_brand_name_alone_does_not_count_as_relevance():
"""Every Britannia URL contains "britannia", so it separates nothing."""
brand_only = "https://media.britannia.co.in/britannia.jpg"
named = "https://cdn.example.com/tiger_glucose.jpg"
assert _top([brand_only, named], "Britannia Tiger Glucose 100g", "Britannia") == named
def test_a_pack_size_alone_does_not_count_as_relevance():
""""750ml" is shared by every drink in the range."""
wrong_product = "https://media.britannia.co.in/thums-up-750ml.jpg"
right_product = "https://cdn.example.com/fanta-orange.jpg"
assert _top(
[wrong_product, right_product], "Coca-Cola Fanta 750ml", "Coca-Cola"
) == right_product
# ---------------------------------------------------------------------------
# Category detection feeding the same products
# ---------------------------------------------------------------------------
def test_milk_bikis_is_a_biscuit_not_a_dairy_product():
""""milk" was the only keyword in the name, so it resolved to Dairy - which
then licenses ml/L pack sizes for a biscuit."""
from app.services.category_registry import detect_category_from_text
assert detect_category_from_text("Britannia Milk Bikis 150g") == "Biscuits & Cookies"
# ...without stealing genuinely dairy products from Dairy.
assert detect_category_from_text("Amul Milk 500ml") == "Dairy"
assert detect_category_from_text("Amul Butter Pasteurised") == "Dairy"

View File

@@ -249,3 +249,58 @@ def test_resolve_brands_prefers_explicit_run_config(monkeypatch):
from orchestration.config import resolve_brands
assert resolve_brands(["Britannia"]) == ["Britannia"]
# ---------------------------------------------------------------------------
# Runner ownership - which batches this side is allowed to claim
# ---------------------------------------------------------------------------
@pytest.fixture
def staged(tmp_path, monkeypatch):
"""A batch directory holding one manifest per runner."""
from app.core import batch_ingest
monkeypatch.setattr(batch_ingest, "BATCH_UPLOAD_DIR", tmp_path / "batch_uploads")
def stage(runner):
manifest = batch_ingest.stage_batch([("catalog.csv", b"Product Name\nAmul Butter 100g\n")])
manifest.runner = runner
manifest.status = batch_ingest.QUEUED
batch_ingest.write_manifest(manifest)
return manifest.batch_id
return stage
def test_dagster_ignores_a_batch_owned_by_the_api_worker(staged):
"""Both executors watch the same directory. Before `runner` existed, running
the sensor beside a live API meant both claimed every queued batch and
ingested it twice."""
from dagster import Failure
from orchestration.assets.batch_catalog import _pick_batch_id
from orchestration.config import BatchConfig
staged("inprocess")
with pytest.raises(Failure):
_pick_batch_id(BatchConfig())
def test_dagster_claims_a_batch_staged_for_it(staged):
from orchestration.assets.batch_catalog import _pick_batch_id
from orchestration.config import BatchConfig
staged("inprocess")
mine = staged("dagster")
assert _pick_batch_id(BatchConfig()) == mine
def test_an_explicit_batch_id_bypasses_the_runner_filter(staged):
"""A person naming a batch in the Launchpad is an instruction, not a poll."""
from orchestration.assets.batch_catalog import _pick_batch_id
from orchestration.config import BatchConfig
theirs = staged("inprocess")
assert _pick_batch_id(BatchConfig(batch_id=theirs)) == theirs

View File

@@ -0,0 +1,95 @@
"""Tests for the row-level repairs in scripts/repair_catalog.py.
The three repair functions are pure, so they are tested here without a database.
What matters most is what they DON'T touch: this script rewrites live catalog
rows, so a false positive silently destroys a correct product name or a correct
set of images.
"""
from __future__ import annotations
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from scripts.repair_catalog import repair_category, repair_images, repair_product_name
# ---------------------------------------------------------------------------
# Names
# ---------------------------------------------------------------------------
def test_a_trailing_bare_number_is_stripped():
assert repair_product_name("Coca-Cola 750ml 72") == "Coca-Cola 750ml"
def test_a_real_pack_size_suffix_is_kept():
"""The size suffix is correct behaviour for a multi-pack-size row."""
assert repair_product_name("Britannia Good Day Cashew Cookies 200g") is None
def test_a_clean_name_is_left_alone():
assert repair_product_name("Coca-Cola 750ml") is None
assert repair_product_name("Britannia Marie Gold") is None
def test_a_word_suffix_is_not_a_number():
assert repair_product_name("Britannia 50-50 Family Pack") is None
def test_a_name_that_is_only_a_number_is_not_gutted():
"""Nothing with letters would remain, so there is no safe repair."""
assert repair_product_name("750 72") is None
def test_a_single_token_name_is_untouched():
assert repair_product_name("72") is None
# ---------------------------------------------------------------------------
# Categories
# ---------------------------------------------------------------------------
def test_a_numeric_category_is_resolved_from_the_name():
assert repair_category("1", "Coca-Cola 750ml") == "Beverages"
def test_a_known_category_is_left_alone():
assert repair_category("Biscuits & Cookies", "Britannia Marie Gold") is None
def test_general_is_a_decision_not_corruption():
"""`General` is what the pipeline stores when detection genuinely failed."""
assert repair_category("General", "Something Unclassifiable Xyz") is None
def test_an_unresolvable_numeric_category_becomes_general():
assert repair_category("3", "Mystery Item Xyz") == "General"
# ---------------------------------------------------------------------------
# Images
# ---------------------------------------------------------------------------
_MINE = "https://cdn.example.com/daily/brands/britannia/britannia_marie_gold_250g/image_000.jpg"
_THEIRS = "https://cdn.example.com/daily/brands/britannia/britannia_good_day_200g/image_000.jpg"
def test_images_filed_under_another_product_are_cleared():
assert repair_images("britannia_marie_gold_250g", _THEIRS, [_THEIRS]) == (None, [])
def test_a_products_own_images_are_kept():
assert repair_images("britannia_marie_gold_250g", _MINE, [_MINE]) is None
def test_a_plain_cdn_url_is_not_judged():
"""No image_id folder in the path means no evidence either way."""
plain = "https://cdn.example.com/some/photo.jpg"
assert repair_images("britannia_marie_gold_250g", plain, [plain]) is None
def test_a_row_with_no_images_needs_no_repair():
assert repair_images("britannia_marie_gold_250g", None, []) is None
def test_one_foreign_url_among_several_condemns_the_set():
"""A bled set is not partially trustworthy - it was copied wholesale."""
assert repair_images("britannia_marie_gold_250g", _MINE, [_MINE, _THEIRS]) == (None, [])

View File

@@ -345,3 +345,74 @@ def test_the_inbox_is_closed_to_anonymous(client):
assert client.get(INBOX).status_code == 401
assert client.post(FROM_INBOX, json={"file_ids": []}).status_code == 401
assert client.post(DISMISS, json={"file_ids": []}).status_code == 401
# ---------------------------------------------------------------------------
# Choosing who runs the batch
# ---------------------------------------------------------------------------
def test_starting_a_file_defaults_to_this_containers_worker(client, admin_headers, submitted):
"""An existing client that never sends `runner` must be unaffected."""
batch_id = _drop(client, "priya", "catalog.csv")
started = client.post(FROM_INBOX, json={"file_ids": [f"{batch_id}:0"]},
headers=admin_headers)
assert started.json()["runner"] == batch_ingest.RUNNER_INPROCESS
assert submitted == [started.json()["batch_id"]]
def test_a_dagster_batch_is_staged_but_handed_to_no_worker(client, admin_headers, submitted):
"""The whole point of the runner field: Dagster claims this one, so the
in-process worker must never be given it."""
batch_id = _drop(client, "priya", "catalog.csv")
started = client.post(
FROM_INBOX,
json={"file_ids": [f"{batch_id}:0"], "runner": "dagster"},
headers=admin_headers,
)
assert started.status_code == 202, started.text
run = started.json()
assert run["runner"] == batch_ingest.RUNNER_DAGSTER
assert run["status"] == batch_ingest.QUEUED
assert submitted == [], "a dagster batch must not reach the in-process worker"
# And it still left the inbox, so it cannot be started twice.
assert client.get(INBOX, headers=admin_headers).json()["pending_count"] == 0
def test_an_unknown_runner_is_refused(client, admin_headers, submitted):
batch_id = _drop(client, "priya", "catalog.csv")
started = client.post(
FROM_INBOX,
json={"file_ids": [f"{batch_id}:0"], "runner": "kubernetes"},
headers=admin_headers,
)
assert started.status_code == 400
assert "kubernetes" in started.json()["detail"]
assert submitted == []
# Refused before anything was taken out of the inbox.
assert client.get(INBOX, headers=admin_headers).json()["pending_count"] == 1
def test_running_a_stranded_dagster_batch_here_takes_ownership(client, admin_headers, submitted):
"""Resume is the "run it here instead" button for a batch no orchestrator
came for. It must also claim the batch, or Dagster could still pick up
something already running in this process."""
batch_id = _drop(client, "priya", "catalog.csv")
run = client.post(
FROM_INBOX,
json={"file_ids": [f"{batch_id}:0"], "runner": "dagster"},
headers=admin_headers,
).json()
resumed = client.post(
f"/api/admin/catalog-batch/batches/{run['batch_id']}/resume",
headers=admin_headers,
)
assert resumed.status_code == 200, resumed.text
assert resumed.json()["runner"] == batch_ingest.RUNNER_INPROCESS
assert submitted == [run["batch_id"]]

View File

@@ -143,6 +143,43 @@ def test_an_uncategorised_row_gets_no_hsn_code(store):
assert any("category could not be detected" in w for w in result.warnings)
def test_a_categoryid_column_is_not_read_as_the_category(store):
""""category" is a substring of "categoryid", so the keyword rule used to
claim the id column and store a bare 1/2/3 as the product's category."""
content = _sheet(["Item Name", "categoryid"], [["Coca-Cola 750ml", "1"]])
result = _run(content)
assert "category" not in result.recognised_columns
assert "categoryid" in result.unrecognised_columns
def test_a_numeric_category_is_resolved_from_the_product_name(store):
"""Defence in depth: even when a numeric code reaches the category field
it must not be stored as one - a number is never a category name."""
content = _sheet(["Item Name", "Category"], [["Coca-Cola 750ml", "1"]])
result = _run(content)
row = next(iter(store.values()))
assert row["category"] == "Beverages"
assert any("is an id, not a name" in w for w in result.warnings)
def test_a_soft_drink_resolves_to_beverages(store):
"""`Beverages` was missing from the canonical taxonomy entirely, so a cola
fell through to General while category_units and the HSN table both knew
the name."""
content = _sheet(["Item Name"], [["Coca-Cola 750ml"]])
_run(content)
assert next(iter(store.values()))["category"] == "Beverages"
def test_a_real_category_name_is_still_trusted(store):
"""Only numeric codes are re-detected; a name the store supplied wins."""
content = _sheet(["Item Name", "Segment"], [["Britannia 50-50", "Biscuits"]])
_run(content)
assert next(iter(store.values()))["category"] == "Biscuits"
def test_non_food_brands_get_no_fssai_licence(store):
"""A miss means 'not a food brand', not an error - the column stays empty."""
assert pipeline.get_fssai_license("Colgate") is None
@@ -178,6 +215,59 @@ def test_a_pack_size_with_a_nonsense_unit_is_dropped_with_a_reason(store):
assert any("15cm" in w for w in result.warnings)
def test_a_quantity_column_never_reaches_the_product_name(store):
"""The reported bug: a case-pack count of 72 became part of the name
("Coca-Cola 750ml 72") and was folded into the image_id, so the next upload
inserted a duplicate instead of updating the row."""
content = _sheet(["Item Name", "Quantity"], [["Coca-Cola 750ml", "72"]])
result = _run(content)
stored = list(store.values())
assert [r["product_name"] for r in stored] == ["Coca-Cola 750ml"]
assert "72" not in stored[0]["image_id"]
assert "Quantity" in result.unrecognised_columns
def test_a_unitless_number_in_a_size_column_is_rejected(store):
"""Defence in depth for the headers that legitimately DO map to
size_variants: "Net Weight" is a pack size column, but a bare 72 in it is
still not a pack size."""
content = _sheet(["Item Name", "Net Weight"], [["Coca-Cola 750ml", "72"]])
result = _run(content)
stored = list(store.values())
assert [r["product_name"] for r in stored] == ["Coca-Cola 750ml"]
assert any("72" in w for w in result.warnings)
def test_the_size_in_the_name_is_used_when_the_sheet_supplies_no_real_one(store):
"""Having rejected the bare number, the pipeline falls back to the name."""
content = _sheet(["Item Name", "Net Weight"], [["Coca-Cola 750ml", "72"]])
_run(content)
assert [r["size_variants"] for r in store.values()] == [["750ml"]]
def test_a_quantity_column_does_not_shut_out_the_real_pack_size(store):
"""Column mapping is first-wins by position, so a leading `Quantity` column
used to claim size_variants and discard the sheet's actual `Pack Size`."""
content = _sheet(
["Item Name", "Quantity", "Pack Size"],
[["Britannia Marie Gold Biscuits", "24", "250g"]],
)
result = _run(content)
assert result.recognised_columns["size_variants"] == "Pack Size"
assert "Quantity" in result.unrecognised_columns
assert [r["size_variants"] for r in store.values()] == [["250g"]]
def test_a_word_only_pack_size_is_still_accepted(store):
"""The unitless-number filter must not swallow "Family Pack"."""
content = _sheet(["Item Name", "Pack Size"], [["Britannia 50-50", "Family Pack"]])
_run(content)
assert [r["size_variants"] for r in store.values()] == [["Family Pack"]]
# ---------------------------------------------------------------------------
# Stage 11 - storage semantics
# ---------------------------------------------------------------------------

View File

@@ -368,3 +368,64 @@ def test_the_seed_catalog_is_only_written_for_rows_the_database_took(
assert resp.status_code == 503
assert catalog_writes == [], "the JSON catalog must not gain products the database refused"
# ---------------------------------------------------------------------------
# Brand-level defaults must not include images
# ---------------------------------------------------------------------------
def test_a_product_does_not_inherit_another_products_images(monkeypatch):
"""The reported bug: every image-less Britannia product showed Good Day.
`_brand_sample()` returns ONE arbitrary product of the brand and is resolved
once per upload, so copying its `image_urls` onto each image-less sibling
stamped that product's photographs across the whole brand. A photograph is
never a brand-level default.
"""
monkeypatch.setattr(user_products.s3_service, "enabled", False)
good_day_images = [
"https://cdn.example.com/britannia_good_day_cashew_cookies_200g/image_000.jpg",
"https://cdn.example.com/britannia_good_day_cashew_cookies_200g/image_001.jpg",
]
sample = {"image_urls": good_day_images, "category": "Biscuits & Cookies"}
built = user_products._build_product_dict(
user_products.AddProductRequest(product_name="Britannia Marie Gold 250g", brand="Britannia"),
"Britannia",
sample,
)
assert built["image_urls"] == []
assert built["image_url"] is None
def test_a_product_with_no_image_gets_no_invented_url(monkeypatch):
"""The old fallback guessed an S3 URL on a bucket that is not the configured
one, so it 404'd while still looking like an image to the frontend."""
monkeypatch.setattr(user_products.s3_service, "enabled", False)
built = user_products._build_product_dict(
user_products.AddProductRequest(product_name="Britannia Tiger 100g", brand="Britannia"),
"Britannia",
{},
)
assert built["image_urls"] == []
assert built["image_url"] is None
def test_an_explicit_image_url_is_still_honoured(monkeypatch):
"""Removing the fallbacks must not drop an image the user actually gave."""
monkeypatch.setattr(user_products.s3_service, "enabled", False)
built = user_products._build_product_dict(
user_products.AddProductRequest(
product_name="Britannia Tiger 100g",
brand="Britannia",
image_url="https://cdn.example.com/tiger.jpg",
),
"Britannia",
{"image_urls": ["https://cdn.example.com/good_day.jpg"]},
)
assert built["image_url"] == "https://cdn.example.com/tiger.jpg"