image repair on existing products

This commit is contained in:
sriram
2026-09-04 13:23:43 +05:30
parent 0c3e23fac4
commit 8059120ce3
4 changed files with 347 additions and 10 deletions

View File

@@ -161,3 +161,237 @@ def test_the_bucket_is_recognised_under_either_spelling():
assert _is_own_products("own_products") is True
assert _is_own_products("Own Products") is True
assert _is_own_products("Amul") is False
# ---------------------------------------------------------------------------
# The curated table reaches the two write paths
# ---------------------------------------------------------------------------
# Building 159 reviewed images and then not wiring them in is the failure this
# section exists to prevent. Until these tests existed, `produce_reference.
# image_url` had NO runtime consumer at all: the repair script re-searched every
# row, and the ingestion pipeline never even passed `produce=True`. So the
# contact sheet showed one set of images and the code would have written another.
CURATED_COMMODITY = "Apple"
UNCURATED_COMMODITY = "Avocado" # genuinely absent from produce_reference.json
BUCKET = "own_products" # repair() passes the TABLE SUFFIX, not a name
@pytest.fixture
def repair_module(monkeypatch):
"""The repair script with its process-lifetime search cache emptied.
`_SEARCH_CACHE` is module-level and never cleared, so without this the first
test to run would answer for every test after it.
"""
from scripts import repair_brand_images
monkeypatch.setattr(repair_brand_images, "_SEARCH_CACHE", {})
monkeypatch.setattr(repair_brand_images, "_url_is_live", lambda url, timeout: True)
return repair_brand_images
@pytest.fixture
def no_search(monkeypatch):
"""Record calls to the search, and fail loudly if one was not expected."""
calls = []
def _fake(title, brand=None, country_hint=None, validate=True,
max_results=24, produce=False):
calls.append({"title": title, "brand": brand, "produce": produce})
return []
monkeypatch.setattr(image_search, "find_all_image_urls", _fake)
return calls
def test_the_repair_returns_the_reviewed_image(repair_module, no_search):
from app.services import produce_reference
found = repair_module._search_replacement(CURATED_COMMODITY, BUCKET, 8)
assert found == [produce_reference.image_url(CURATED_COMMODITY)]
def test_the_repair_does_not_search_when_it_already_has_the_answer(
repair_module, no_search):
"""The short-circuit is the behaviour, not an optimisation.
A curated URL that went through the rest of `_search_replacement` would be
thrown away by the `_names_product` corroboration gate: the reviewed Commons
photo for Apple is `Honeycrisp.jpg`, and a Commons filename names the
cultivar, not the catalogue word.
"""
repair_module._search_replacement(CURATED_COMMODITY, BUCKET, 8)
assert no_search == [], "the curated branch must return before searching"
def test_a_pack_size_still_finds_the_commodity(repair_module, no_search):
from app.services import produce_reference
assert repair_module._search_replacement("Apple 1kg", BUCKET, 8) == [
produce_reference.image_url(CURATED_COMMODITY)]
assert no_search == []
def test_a_variety_shares_its_commodity_image(repair_module, no_search):
"""`Mango Totapuri` and `Mango Alphonso` are the same fruit."""
from app.services import produce_reference
assert repair_module._search_replacement("Mango Totapuri 1kg", BUCKET, 8) == [
produce_reference.image_url("Mango")]
def test_an_uncurated_commodity_falls_through_to_a_produce_search(
repair_module, no_search):
repair_module._search_replacement(UNCURATED_COMMODITY, BUCKET, 8)
assert len(no_search) == 1
assert no_search[0]["produce"] is True
assert no_search[0]["brand"] == "", "the bucket name must not enter the query"
def test_a_branded_product_never_consults_the_curated_table(
repair_module, no_search):
"""`lookup` falls back through shorter leading prefixes, so asking it about
every brand's rows would put a photo of fruit on Apple-branded anything."""
repair_module._search_replacement("Britannia Good Day 200g", "britannia", 8)
assert len(no_search) == 1
assert no_search[0]["produce"] is False
def test_a_curated_url_that_fails_its_probe_is_used_anyway(
repair_module, no_search, monkeypatch, caplog):
"""upload.wikimedia.org throttles a fast sweep and the probe treats 429
exactly like 404. Falling back to a search here would let a transient
rate-limit quietly replace reviewed photos with searched ones."""
from app.services import produce_reference
monkeypatch.setattr(repair_module, "_url_is_live", lambda url, timeout: False)
with caplog.at_level("WARNING"):
found = repair_module._search_replacement(CURATED_COMMODITY, BUCKET, 8)
assert found == [produce_reference.image_url(CURATED_COMMODITY)]
assert no_search == [], "a failed probe must not trigger a search"
assert "did not verify" in caplog.text
# ---------------------------------------------------------------------------
# The derived-product guard
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("name", [
"Coconut Oil", "Apple Cider Vinegar", "Tomato Ketchup", "Corn Flour",
])
def test_a_derived_product_does_not_inherit_the_raw_commodity_photo(name):
"""The prefix fallback that correctly resolves `Mango Totapuri` to `Mango`
also resolves `Coconut Oil` to `Coconut`. A bottle of oil is not a variety
of coconut, and showing the raw fruit for it is the apple-juice error
pointing the other way."""
from app.services import produce_reference
assert produce_reference.image_url(name) is None
def test_the_guard_does_not_touch_nutrition_matching():
"""`usda_fdc_id` keeps the plain fallback on purpose - the guard is about
not showing a misleading picture, and nutrition has its own dispatch."""
from app.services import produce_reference
assert produce_reference.usda_fdc_id("Apple 1kg") == 171688
assert produce_reference.usda_fdc_id("Coconut Oil") is not None
# ---------------------------------------------------------------------------
# Stage 6 of the ingestion pipeline
# ---------------------------------------------------------------------------
@pytest.fixture
def stage_6(monkeypatch):
"""`stage_6_images` plus a recorder on the ranking it must not reach."""
from app.core import catalog_engine as catalog_engine_module
from app.core import store_catalog_pipeline
ranked = []
def _fake_rank(urls, title, brand, max_images=20):
ranked.append(title)
return list(urls)
monkeypatch.setattr(catalog_engine_module.catalog_engine,
"_select_best_images", _fake_rank)
return store_catalog_pipeline, ranked
def _produce_row(name):
from app.services.generic_products import OWN_PRODUCTS_BRAND
return {"product_name": name, "brand": OWN_PRODUCTS_BRAND}
def test_an_uploaded_commodity_gets_the_reviewed_image(stage_6, no_search):
from app.services import produce_reference
pipeline, ranked = stage_6
row = pipeline.stage_6_images(_produce_row(CURATED_COMMODITY))
expected = produce_reference.image_url(CURATED_COMMODITY)
assert row["image_url"] == expected
assert row["image_urls"] == [expected]
assert no_search == []
def test_the_reviewed_image_is_not_put_through_the_ranking(stage_6, no_search):
"""`_select_best_images` scores a URL by whether it names the product, so
`Honeycrisp.jpg` scores zero and any searched URL containing "apple" sorts
above it. Index 0 is what becomes `image_url`, so merging rather than
short-circuiting would leave the wrong image on the card."""
pipeline, ranked = stage_6
pipeline.stage_6_images(_produce_row(CURATED_COMMODITY))
assert ranked == []
def test_uploaded_produce_with_no_curated_entry_still_gets_produce_mode(
stage_6, no_search):
"""The fix that was missing entirely: this stage never passed `produce=`, so
every new commodity was searched against the packaged-goods databases with
the "-plant -tree -fish" exclusion block that fights produce queries."""
pipeline, ranked = stage_6
pipeline.stage_6_images(_produce_row(UNCURATED_COMMODITY))
assert len(no_search) == 1
assert no_search[0]["produce"] is True
assert no_search[0]["brand"] is None, "the bucket must not enter the query"
def test_an_uploaded_branded_product_is_unchanged(stage_6, no_search):
pipeline, ranked = stage_6
pipeline.stage_6_images({"product_name": "Britannia Good Day 200g",
"brand": "Britannia"})
assert len(no_search) == 1
assert no_search[0]["produce"] is False
assert no_search[0]["brand"] == "Britannia"
def test_a_row_that_already_has_images_is_left_alone(stage_6, no_search):
pipeline, ranked = stage_6
row = dict(_produce_row(CURATED_COMMODITY),
image_urls=["https://example.com/already.jpg"])
assert pipeline.stage_6_images(row)["image_urls"] == [
"https://example.com/already.jpg"]
assert no_search == []
def test_images_can_still_be_switched_off(stage_6, no_search):
pipeline, ranked = stage_6
row = pipeline.stage_6_images(_produce_row(CURATED_COMMODITY), enabled=False)
assert "image_url" not in row
assert no_search == []