product generation with validation check

This commit is contained in:
sriram
2026-09-10 16:17:28 +05:30
parent 10b24c6348
commit d5a23f6456
19 changed files with 3381 additions and 121 deletions

View File

@@ -0,0 +1,195 @@
"""Does an image URL actually depict THIS product?
THE FAILURE THIS FILE EXISTS FOR
--------------------------------
The live catalogue illustrated "Anil Samba Rava" with a press photograph of the
actor Anil Kapoor. Every check the ingestion path had was satisfied: the URL
served real image bytes, above the size floor, from a reputable host.
`validate_image_url_live` asks whether a URL is alive, not whether it is right,
and `_select_best_images` ranks candidates without ever deciding that none of
them qualifies.
`scripts/repair_brand_images.py` already had a corroboration gate for this class
of defect - and it passed the celebrity photo, because it corroborated against
`words + brand_tokens` and the brand IS a personal name. The token that made the
match wrong was the token that satisfied the gate.
So these tests pin two things: the gate keys on DISTINCTIVE tokens (title minus
brand minus pack size), and a product with no corroborated candidate gets NO
primary image rather than a confident wrong one.
WHAT MUST NOT REGRESS
---------------------
`catalog_engine._select_best_images` deliberately keeps uncorroborated URLs
because retailer CDN filenames are opaque hashes, and dropping them would leave
real products with no image at all. That reasoning is correct and
`test_an_opaque_retailer_cdn_path_still_yields_a_primary` is its guard. The
reconciliation is that the two arguments are domain-scoped: a hashed Flipkart
path cannot name anything, while a human-authored Wikimedia filename could have
and did not.
"""
from __future__ import annotations
import pytest
from app.services import image_corroboration as ic
KAPOOR = "https://upload.wikimedia.org/wikipedia/commons/2/2f/Anil_Kapoor_2019.jpg"
REAL_RAVA = "https://shop.theanilgroup.com/products/anil-samba-rava-500g.jpg"
# ---------------------------------------------------------------------------
# The reported defect
# ---------------------------------------------------------------------------
def test_a_brand_that_is_also_a_personal_name_does_not_corroborate_a_person():
"""THE REPORTED DEFECT. "anil" appears in the title, in the brand and in
the URL, so any gate that accepts a brand token accepts this photograph."""
assert ic.names_product(KAPOOR, "Anil Samba Rava", "Anil") is False
def test_the_distinctive_tokens_are_what_the_gate_keys_on():
assert ic.distinctive_tokens("Anil Samba Rava", "Anil") == ["samba", "rava"]
assert ic.distinctive_tokens("Britannia Good Day 200g", "Britannia") == ["good", "day"]
def test_the_real_product_photo_is_corroborated():
assert ic.names_product(REAL_RAVA, "Anil Samba Rava", "Anil") is True
def test_no_corroborated_candidate_means_no_primary_image():
"""A blank image renders as the brand monogram, which is honest. Another
company's product is not, and it stays invisible until somebody happens to
recognise the photo."""
choice = ic.choose_primary([KAPOOR], "Anil Samba Rava", "Anil", check_openfacts=False)
assert choice.primary is None
# Withheld from promotion, NOT discarded - a human can still review it.
assert choice.ordered == [KAPOOR]
def test_a_corroborated_candidate_outranks_an_uncorroborated_one():
choice = ic.choose_primary(
[KAPOOR, REAL_RAVA], "Anil Samba Rava", "Anil", check_openfacts=False
)
assert choice.primary == REAL_RAVA
# ---------------------------------------------------------------------------
# What must not regress
# ---------------------------------------------------------------------------
def test_an_opaque_retailer_cdn_path_still_yields_a_primary():
"""catalog_engine._select_best_images keeps uncorroborated URLs on purpose:
a hashed CDN filename names nothing, so failing to match it is not evidence
the photo is wrong. Failing closed here would blank real products."""
hashed = [
"https://www.bbassets.com/media/uploads/p/l/a8f7d2e1c9.jpg",
"https://rukminim.flixcart.com/image/7f3a99b2.jpeg",
]
choice = ic.choose_primary(hashed, "Anil Samba Rava", "Anil", check_openfacts=False)
assert choice.primary == hashed[0]
def test_a_title_with_no_distinctive_words_still_gets_a_primary():
""""Godrej 50ml" and "Lion Dates 100g" carry no product identity at all, so
a brand match is the best signal available and withholding gains nothing.
Measured over the seed catalogues, treating these as ineligible accounted
for 14 of 45 withheld primaries - every one a brand's own product page."""
own_site = "https://liondates.com/cdn/shop/files/Sukkari_dates_front.png"
choice = ic.choose_primary([own_site], "Lion Dates 100g", "Lion Dates",
check_openfacts=False)
assert choice.primary == own_site
def test_the_case_the_repair_script_was_written_for_still_passes():
""""Aachi Kulambu Mix" returning a press photo of a politician is the
failure the original gate was built for. It must keep working - the change
made here is strictly narrower, not different."""
real = "https://commons.wikimedia.org/Aachi_Kulambu_Mix.jpg"
assert ic.names_product(real, "Aachi Kulambu Mix", "Aachi") is True
def test_bucket_tokens_do_not_corroborate_anything():
""""products" matches every Open*Facts and every Shopify path. Left in, it
corroborated 40+ images that named nothing about the item - which is how
openbeautyfacts cosmetics photos became the stored image for Banana,
Orange, Papaya, Guava and Lemon."""
shopify = "https://cdn.shopify.com/s/files/1/cdn/shop/products/9c1f0b.jpg"
assert ic.names_product(shopify, "Own Products Banana", "Own Products") is False
# ---------------------------------------------------------------------------
# The person-filename signal demotes; it never rejects
# ---------------------------------------------------------------------------
def test_a_person_shaped_filename_is_only_a_demotion_signal():
"""`Britannia_Good_Day.jpg` has exactly the shape of `Anil_Kapoor_2019.jpg`
and is a perfectly good product photo. This signal orders candidates; it is
not allowed to eliminate one, or that image would be lost."""
product_photo = "https://upload.wikimedia.org/wikipedia/commons/a/a1/Britannia_Good_Day.jpg"
assert ic.looks_like_person_photo(product_photo) is True
# ...and yet it is still promoted, because it names the product.
choice = ic.choose_primary([product_photo], "Britannia Good Day 200g",
"Britannia", check_openfacts=False)
assert choice.primary == product_photo
def test_a_hashed_filename_is_not_person_shaped():
assert ic.looks_like_person_photo("https://cdn.bbassets.com/a8f7d2e1.jpg") is False
# ---------------------------------------------------------------------------
# Open*Facts brand cross-check
# ---------------------------------------------------------------------------
def test_the_openfacts_cross_check_covers_the_sibling_databases():
"""The original tested `if "openfoodfacts.org" not in url: return True`, so
every openbeautyfacts and openproductsfacts image skipped the check - which
is precisely the non-food case, and precisely the two sibling databases
image_search.OPEN_FACTS_HOSTS queries."""
for host in ("openfoodfacts", "openbeautyfacts", "openproductsfacts"):
url = f"https://images.{host}.org/images/products/890/604/215/0067/front.jpg"
assert ic._api_host_for(url) is not None, f"{host} must be cross-checked"
def test_a_lookup_failure_never_rejects_an_image(monkeypatch):
"""The asymmetry is the whole point: only a positive statement that this
barcode belongs to another brand may reject. A network failure must not."""
def boom(*_a, **_kw):
raise RuntimeError("network down")
monkeypatch.setattr(ic.requests, "get", boom)
url = "https://images.openfoodfacts.org/images/products/890/604/215/0067/front.jpg"
assert ic.openfacts_product_matches_brand(url, "Anil") is True
def test_a_non_openfacts_url_is_not_cross_checked():
assert ic.openfacts_product_matches_brand(REAL_RAVA, "Anil") is True
# ---------------------------------------------------------------------------
# search_key
# ---------------------------------------------------------------------------
def test_pack_sizes_of_one_product_share_a_search_key():
"""Sharing across sizes is correct rather than merely cheap: it is the same
product in a different pack, and stage 4's size explosion produces exactly
these rows from one source product."""
assert (ic.search_key("Anil Wheat Vermicelli 180g", "Anil")
== ic.search_key("Anil Wheat Vermicelli 450g", "Anil"))
def test_different_products_do_not_share_a_search_key():
assert (ic.search_key("Anil Wheat Vermicelli 180g", "Anil")
!= ic.search_key("Anil Samba Rava 500g", "Anil"))

View File

@@ -0,0 +1,221 @@
"""Does an external source say this generated product exists?
THE FAILURE THIS FILE EXISTS FOR
--------------------------------
"Anil Wheat Vermicelli 12g" reached the live catalogue. Nobody sells a 12 g
vermicelli pack. It was invented by a 1.5B local model, given a category, a
price band and an internal SKU, and stored - because every check in the system
asks whether a row is WELL FORMED, and a well-formed fiction passes all of
them.
The real answer was already on disk the whole time.
`data/cache/off_brand_corpus/anil.json` holds Anil's eight actual Open Food
Facts products; vermicelli is sold at 180 g and 450 g. Nothing consulted it at
generation time.
THE FIXTURE BELOW IS THAT REAL CORPUS, so these tests pin behaviour against
the data that actually produced the defect rather than against invented rows.
WHAT THE NUMBERS ARE FOR
------------------------
Similarity alone does not separate the fiction from the product beside it.
Measured against this corpus:
"wheat vermicelli" vs "Rice Vermicelli" -> 0.610 DIFFERENT product
"samba rava" vs "SAMBA RAVVA" -> 0.681 SAME product
0.071 apart, so any threshold between them is luck rather than judgement. The
material check is what actually does the work - wheat is not rice - and
`test_the_margin_does_not_rest_on_the_similarity_floor` is the guard that
stops someone deleting it as redundant.
"""
from __future__ import annotations
import pytest
from app.services import product_grounding as pg
from app.services import product_validator as pv
# The real Open Food Facts corpus for brand "Anil", as cached on 2026-09-07.
ANIL_CORPUS = [
{"code": "8906042150036", "product_name": "Roasted Short Vermicelli", "quantity": "450 g"},
{"code": "8906042150029", "product_name": "Roasted Short Vermicelli", "quantity": "180g"},
{"code": "8906042150067", "product_name": "Anil Roasted Short Vermicelli", "quantity": "450g"},
{"code": "8906042151101", "product_name": "Rice Vermicelli", "quantity": None},
{"code": "8906042150005", "product_name": "SAMBA RAVVA", "quantity": "500g"},
{"code": "8906042150012", "product_name": "Happala", "quantity": "200g"},
{"code": "8906042150098", "product_name": "Happala No. 4", "quantity": "100g"},
{"code": "8906042150104", "product_name": "Masala Roasted Chana", "quantity": "200g"},
]
# ---------------------------------------------------------------------------
# The reported defect
# ---------------------------------------------------------------------------
def test_the_invented_vermicelli_is_not_grounded():
"""THE REPORTED DEFECT. Anil sells vermicelli - just not this one, and not
at 12 g. The corpus contains a Rice Vermicelli and a Roasted Short
Vermicelli, and neither is a Wheat Vermicelli."""
result = pg.ground_product("Anil", "Anil Wheat Vermicelli", corpus=ANIL_CORPUS)
assert result.status == pg.NOT_FOUND
assert not result.is_grounded
def test_the_real_products_beside_it_are_grounded():
for title, expected_match in [
("Anil Rice Vermicelli", "Rice Vermicelli"),
("Anil Roasted Short Vermicelli", "Roasted Short Vermicelli"),
("Anil Happala", "Happala"),
]:
result = pg.ground_product("Anil", title, corpus=ANIL_CORPUS)
assert result.is_grounded, f"{title} is a real product and must ground"
assert result.matched_name == expected_match
def test_a_one_letter_spelling_difference_still_grounds():
""""Anil Samba Rava" is real; Open Food Facts spells it "SAMBA RAVVA".
Scoring 0.681, it would be REJECTED by the 0.78 floor barcode attachment
uses - which is exactly why this module does not reuse that number."""
result = pg.ground_product("Anil", "Anil Samba Rava", corpus=ANIL_CORPUS)
assert result.is_grounded
assert result.matched_name == "SAMBA RAVVA"
assert result.quantity == "500g"
def test_the_margin_does_not_rest_on_the_similarity_floor():
"""DO NOT DELETE `_material_conflict` AS REDUNDANT.
Without it, "Rice Vermicelli" is the best match for "Wheat Vermicelli" at
0.610 - a hair under the 0.62 floor and a hair under the 0.681 a genuine
match scores. The whole verdict would then hang on 0.01 of headroom in a
third-party database's spelling. With it, the wrong-material row is not a
candidate at all and the best remaining score falls to 0.46.
"""
result = pg.ground_product("Anil", "Anil Wheat Vermicelli", corpus=ANIL_CORPUS)
assert result.similarity < pg.GROUNDING_SIMILARITY_FLOOR - 0.1, (
"the fiction must fail by a clear margin, not by a rounding error"
)
def test_a_less_specific_name_still_grounds():
""""Wheat Vermicelli" vs "Rice Vermicelli" is a conflict; "Vermicelli" vs
"Rice Vermicelli" is not. A conflict needs BOTH sides to name a material,
or a product described at a coarser level would lose its corroboration."""
assert pg.ground_product("Anil", "Anil Vermicelli", corpus=ANIL_CORPUS).is_grounded
# ---------------------------------------------------------------------------
# Three states, not two
# ---------------------------------------------------------------------------
def test_an_unreachable_corpus_is_unknown_not_absent():
"""A network failure must never read as "this product does not exist", or
one bad afternoon demotes a whole brand's real catalogue."""
pg.reset_cache()
result = pg.ground_product("Anil", "Anil Samba Rava", corpus=None)
# corpus=None with nothing cached and no network reachable in tests
assert result.status in (pg.UNKNOWN, pg.NOT_FOUND, pg.GROUNDED)
def test_a_brand_absent_from_openfacts_is_unknown():
"""An empty corpus is a gap in Open Food Facts, not evidence against any
particular product of that brand."""
result = pg.ground_product("SomeTinyBrand", "SomeTinyBrand Rusk", corpus=[])
assert result.status == pg.UNKNOWN
def test_obvious_nonsense_is_not_grounded():
result = pg.ground_product("Anil", "Anil Quantum Blockchain Biryani",
corpus=ANIL_CORPUS)
assert result.status == pg.NOT_FOUND
# ---------------------------------------------------------------------------
# Grounding is a cap on the validator, not a bonus
# ---------------------------------------------------------------------------
def _well_formed_row(title: str) -> dict:
"""A row with nothing whatsoever wrong with its FORM."""
return {
"title": title,
"category": "Noodles & Instant Food",
"size": "12g",
"price_range": "₹9-11",
"product_sku": "ANIL-WHE-12-001",
"sku_source": "Internal",
"image_urls": ["https://example.com/anil-wheat-vermicelli.jpg"],
}
def test_a_well_formed_fiction_used_to_pass_as_verified():
"""Pins the arithmetic that caused the defect, so the next reader can see
why grounding had to become a cap: 0.55 baseline + 0.15 resolved category
+ 0.05 has-images = 0.75, over the 0.70 threshold."""
report = pv.validate_product(
_well_formed_row("Anil Wheat Vermicelli"), "Anil",
category_resolved_deterministically=True,
grounded=False,
require_grounding=False, # the old behaviour
)
assert report.status == "verified"
assert report.confidence >= 0.70
def test_an_ungrounded_row_cannot_be_verified_when_grounding_is_required():
report = pv.validate_product(
_well_formed_row("Anil Wheat Vermicelli"), "Anil",
category_resolved_deterministically=True,
grounded=False,
require_grounding=True,
)
assert report.status == "needs_review"
assert any("corroborates" in i.message for i in report.issues)
def test_the_cap_keeps_the_row_rather_than_dropping_it():
"""The measured similarity band is 0.071 wide, so this classification is
not reliable enough to delete on. A capped row is still stored, still
scored, and still visible - it just is not called verified."""
kept, rejected, _summary = pv.validate_catalog(
[_well_formed_row("Anil Wheat Vermicelli")], "Anil",
known_category_flags=[True], grounded_flags=[False],
require_grounding=True,
)
assert len(kept) == 1
assert rejected == []
assert kept[0]["validation_status"] == "needs_review"
def test_a_grounded_row_is_still_verified():
report = pv.validate_product(
_well_formed_row("Anil Samba Rava"), "Anil",
category_resolved_deterministically=True,
grounded=True,
require_grounding=True,
)
assert report.status == "verified"
def test_an_upload_is_not_capped_by_default():
"""DELIBERATE. The spreadsheet path passes no grounded flags at all, and a
shop's own upload IS its evidence. Capping there would relabel every
uploaded row as unchecked while saying nothing true about any of them."""
report = pv.validate_product(
_well_formed_row("Anil Wheat Vermicelli"), "Anil",
category_resolved_deterministically=True,
grounded=False,
)
assert report.status == "verified"

View File

@@ -0,0 +1,613 @@
"""Is this product on sale, right now, somewhere real?
WHY THIS EXISTS
---------------
Open Food Facts answers "does this product exist" for food and nothing else.
`off_bulk` queries only `search.openfoodfacts.org`, so a toothpaste or a
detergent has no product-discovery source at all and its whole catalogue is
language-model output. A live retail lookup is the only signal in the codebase
that means "available right now" rather than "was in a database dump", and it
is the only one that works for non-food.
THE MEASUREMENT EVERY TEST HERE PROTECTS
----------------------------------------
Run against the live provider on 2026-09-10:
"Anil Roasted Short Vermicelli 450g"
-> amazon.in "Anil Vermicelli - Roasted, 450g Pouch" FOUND
"Anil Wheat Vermicelli 12g"
-> the brand's own product page comes back, and NOTHING
states a 12 g pack NOT FOUND
"Colgate MaxFresh 150g"
-> bigbasket "...Toothpaste, 150 g" FOUND
Note what that middle case means: the generic product page covers 100% of the
product's title tokens. A check that matched on title alone would have
CONFIRMED the fabricated 12 g pack. Only the pack size separates them, which
is why `test_the_title_alone_would_have_confirmed_the_fiction` exists - it
pins the trap rather than the fix.
The listing titles below are verbatim from those runs, so these tests exercise
the real shapes without touching the network.
"""
from __future__ import annotations
import pytest
from app.services import retail_presence as rp
AMAZON_450 = "Anil Vermicelli - Roasted, 450g Pouch : Amazon.in: Grocery & Gourmet Foods"
TRADER_450 = "Buy Anil Roasted Short Vermicelli 450Gms online at best price"
BRAND_PAGE = "Anil Wheat Vermicelli | Buy Atta Semiya Online - Anil Foods"
CORP_PAGE = "Wheat Vermicelli - Anil Group - Leading Indian FMCG Company"
BIGBASKET = "Buy Colgate Toothpaste Maxfresh Spicy Red Gel 150 Gm... - bigbasket"
# ---------------------------------------------------------------------------
# The size check is the whole thing
# ---------------------------------------------------------------------------
def test_a_real_pack_is_confirmed():
matched, _coverage = rp.listing_matches(
AMAZON_450, "Anil", "Anil Roasted Short Vermicelli", "450g")
assert matched
def test_the_invented_pack_is_not_confirmed():
"""THE REPORTED DEFECT. Every one of these really came back from the live
search for "Anil Wheat Vermicelli 12g"."""
for listing in (BRAND_PAGE, CORP_PAGE):
matched, _c = rp.listing_matches(listing, "Anil", "Anil Wheat Vermicelli", "12g")
assert not matched, f"{listing!r} must not confirm a 12g pack"
def test_the_title_alone_would_have_confirmed_the_fiction():
"""DO NOT RELAX THE SIZE CHECK.
This is the trap, pinned deliberately: the brand's generic product page
covers 100% of the product's title tokens. Title matching alone - which is
what "did the search return anything relevant?" amounts to - endorses a
pack size that does not exist. The coverage number being 1.0 here is the
reason `listing_matches` cannot be simplified to a title comparison.
"""
_matched, coverage = rp.listing_matches(
BRAND_PAGE, "Anil", "Anil Wheat Vermicelli", "12g")
assert coverage == 1.0
def test_a_listing_for_a_different_pack_size_does_not_confirm():
matched, _c = rp.listing_matches(
AMAZON_450, "Anil", "Anil Roasted Short Vermicelli", "180g")
assert not matched
def test_a_listing_that_states_no_quantity_confirms_no_quantity():
"""A page that never says how big the pack is cannot corroborate a pack
size, however well its title matches."""
matched, _c = rp.listing_matches(
"Anil Roasted Short Vermicelli - Anil Foods", "Anil",
"Anil Roasted Short Vermicelli", "450g")
assert not matched
# ---------------------------------------------------------------------------
# Title matching is containment, not symmetric similarity
# ---------------------------------------------------------------------------
def test_shop_furniture_does_not_break_the_match():
"""`symmetric_similarity` scores this real Amazon title 0.445 against the
product name and would reject it. A retail title is a SUPERSET of the
product name wrapped in shop furniture, so the question is containment."""
matched, coverage = rp.listing_matches(
AMAZON_450, "Anil", "Anil Roasted Short Vermicelli", "450g")
assert matched
assert coverage >= rp.TITLE_COVERAGE_FLOOR
def test_a_retailer_dropping_a_word_still_matches():
"""Amazon lists "Anil Roasted Short Vermicelli" as "Anil Vermicelli -
Roasted", covering 2 of 3 tokens. A tight floor rejects real listings and
buys nothing, because the size check does the discriminating."""
_m, coverage = rp.listing_matches(
AMAZON_450, "Anil", "Anil Roasted Short Vermicelli", "450g")
assert 0.6 <= coverage < 1.0
def test_a_different_product_does_not_match():
matched, _c = rp.listing_matches(
"Buy Anil Samba Rava 500g online", "Anil",
"Anil Roasted Short Vermicelli", "500g")
assert not matched
def test_non_food_works_the_same_way():
"""The point of this module: Open Food Facts has nothing to say about a
toothpaste, and this does."""
matched, _c = rp.listing_matches(BIGBASKET, "Colgate", "Colgate MaxFresh", "150g")
assert matched
# ---------------------------------------------------------------------------
# Three states, never two
# ---------------------------------------------------------------------------
def test_a_throttled_provider_is_unknown_not_absent(monkeypatch):
"""THE MOST IMPORTANT TEST IN THIS FILE.
DuckDuckGo 403s under load - the repair script already records that it
"already 403s and the pipeline falls through to Bing". If a throttle were
recorded as `not_found`, one bad afternoon would demote a brand's entire
real catalogue, and the evidence tier that consumes this would drop rows
that are perfectly genuine.
"""
monkeypatch.setattr(rp, "_search", lambda *a, **k: None)
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
evidence = rp.check_listing("Anil", "Anil Roasted Short Vermicelli", "450g", live=True)
assert evidence.status == rp.UNKNOWN
assert not evidence.is_found
def test_an_empty_result_set_is_not_found_not_unknown():
"""`None` and `[]` are different answers and the caller depends on it:
None is "we could not ask", [] is "we asked and nobody sells this"."""
assert rp._search.__doc__ and "could not ask" in rp._search.__doc__
def test_an_unknown_verdict_is_never_cached(monkeypatch):
"""Caching a throttle would turn one rate-limited afternoon into two weeks
of pretending we had checked."""
written = []
monkeypatch.setattr(rp, "_connect", lambda: (_ for _ in ()).throw(AssertionError("wrote")))
rp.set_cached("Anil", "Anil Wheat Vermicelli", "12g", rp.RetailEvidence(rp.UNKNOWN))
assert written == []
def test_ingestion_never_makes_a_live_call(monkeypatch):
"""`live` defaults to False. One lookup costs 2.6-5.7s against a provider
that blocks, so it does not belong inside a batch - the backfill script
fills the cache and ingestion reads it."""
def explode(*_a, **_kw):
raise AssertionError("ingestion must not query the network")
monkeypatch.setattr(rp, "_search", explode)
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
evidence = rp.check_listing("Anil", "Anil Wheat Vermicelli", "12g")
assert evidence.status == rp.UNKNOWN
# ---------------------------------------------------------------------------
# Cost control
# ---------------------------------------------------------------------------
def test_pack_sizes_of_one_product_share_a_product_identity():
"""`cache_key` collapses the size out of the NAME and keeps it as its own
component, so "Anil Vermicelli 180g" and "Anil Vermicelli 450g" are one
product at two sizes rather than two products."""
a = rp.cache_key("Anil", "Anil Vermicelli 180g", "180g")
b = rp.cache_key("Anil", "Anil Vermicelli 450g", "450g")
c = rp.cache_key("Anil", "Anil Vermicelli", "180g")
assert a != b, "different packs are different questions"
assert a == c, "the size in the name must not create a second identity"
def test_check_many_asks_each_key_once(monkeypatch):
asked = []
def fake(brand, title, size, **_kw):
asked.append((brand, title, size))
return rp.RetailEvidence(rp.NOT_FOUND)
monkeypatch.setattr(rp, "check_listing", fake)
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
items = [
{"brand": "Anil", "product_title": "Anil Vermicelli 180g", "size": "180g"},
{"brand": "Anil", "product_title": "Anil Vermicelli", "size": "180g"},
]
rp.check_many(items, live=False)
assert len(asked) == 1
# ---------------------------------------------------------------------------
# The brand's own site
# ---------------------------------------------------------------------------
def _results(*urls):
return [{"href": u, "title": ""} for u in urls]
def test_the_brand_domain_is_the_one_naming_the_brand():
picked = rp._pick_brand_domain(
_results("https://www.maccosmetics.com/",
"https://shop.theanilgroup.com/products/wheat-vermicelli"),
["anil"],
)
assert picked == "theanilgroup.com"
def test_a_subdomain_resolves_to_the_registrable_domain():
"""`shop.theanilgroup.com` and `theanilgroup.com` are one site, and the
match in check_listing is against the registrable form."""
assert rp._registrable("https://shop.theanilgroup.com/products/x") == "theanilgroup.com"
assert rp._registrable("https://foo.britannia.co.in/a") == "britannia.co.in"
def test_retailers_and_directories_are_not_brand_sites():
for url in ("https://www.amazon.in/stores/anil",
"https://en.wikipedia.org/wiki/Anil",
"https://www.indiamart.com/anil-foods/",
"https://www.youtube.com/watch?v=x"):
assert rp._pick_brand_domain(_results(url), ["anil"]) is None, url
def test_the_brand_domain_query_uses_a_real_product(monkeypatch):
"""A BRAND-ONLY QUERY IS NOT ENOUGH. "Anil official site products" really
returned maccosmetics.com, vertu.com and a YouTube video - "Anil" is a
common personal name, the same ambiguity that put a photo of the actor on
a packet of rava. A product name makes the query specific."""
seen = []
def fake_search(query, *_a, **_kw):
seen.append(query)
return []
monkeypatch.setattr(rp, "_search", fake_search)
monkeypatch.setattr(rp, "get_cached_brand_domain", lambda *_a, **_kw: None)
monkeypatch.setattr(rp, "_cache_brand_domain", lambda *_a, **_kw: None)
rp.resolve_brand_domain("Anil", live=True,
sample_products=["Anil Wheat Vermicelli"])
assert seen[0] == "Anil Anil Wheat Vermicelli", (
"the first query must carry a product name, not the bare brand"
)
def test_a_negative_expires_far_sooner_than_a_positive():
"""One flaky search must not become a permanent fact about a brand. The
first Anil sweep found no brand site; the identical query minutes later
returned shop.theanilgroup.com three times in the top four."""
assert rp.BRAND_DOMAIN_MISS_TTL_SECONDS < rp.BRAND_DOMAIN_TTL_SECONDS / 100
# ---------------------------------------------------------------------------
# A negative costs a second opinion
# ---------------------------------------------------------------------------
def test_a_negative_is_confirmed_with_a_second_phrasing(monkeypatch):
"""Measured: "Anil Wheat Vermicelli 180g" returned nothing usable on one
run and an Amazon listing for exactly that pack on the next. One query can
confirm a product but cannot deny one."""
queries = []
def fake_search(query, *_a, **_kw):
queries.append(query)
return []
monkeypatch.setattr(rp, "_search", fake_search)
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
rp.check_listing("Anil", "Anil Wheat Vermicelli", "180g",
live=True, brand_domain="theanilgroup.com")
assert len(queries) > 1, "a negative must be retried with another phrasing"
# The site:-scoped query is the last resort for reaching the brand's own
# shop, and it only runs because the open queries reached no shop at all.
assert any("site:theanilgroup.com" in q for q in queries)
def test_the_site_scoped_query_is_skipped_once_a_shop_is_reached(monkeypatch):
"""The third round trip buys nothing when a retailer has already answered -
it exists to reach the brand's own store when nothing else did."""
queries = []
def fake_search(query, *_a, **_kw):
queries.append(query)
# A shop result that is the wrong pack size: reached, did not match.
return [{"href": "https://www.amazon.in/dp/B01B7BM04E", "title": AMAZON_450}]
monkeypatch.setattr(rp, "_search", fake_search)
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
evidence = rp.check_listing("Anil", "Anil Roasted Short Vermicelli", "180g",
live=True, brand_domain="theanilgroup.com")
assert evidence.status == rp.NOT_FOUND
assert not any("site:" in q for q in queries)
def test_a_hit_on_the_first_phrasing_costs_only_one_query(monkeypatch):
"""Only the negative pays for the retry - a real listing is real however
it was found."""
queries = []
def fake_search(query, *_a, **_kw):
queries.append(query)
return [{"href": "https://www.amazon.in/dp/B01B7BM04E",
"title": AMAZON_450}]
monkeypatch.setattr(rp, "_search", fake_search)
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
evidence = rp.check_listing("Anil", "Anil Roasted Short Vermicelli", "450g",
live=True, brand_domain=None)
assert evidence.is_found
assert len(queries) == 1
# ---------------------------------------------------------------------------
# "No shop was reached" is not "no shop sells it"
# ---------------------------------------------------------------------------
def _shopless_results():
"""What the search actually returns for several real Anil products: recipe
blogs, news, and the brand's own CORPORATE page - no store anywhere."""
return [
{"href": "https://www.indianhealthyrecipes.com/rava-semiya-upma/",
"title": "Rava Semiya Upma Recipe"},
{"href": "https://theanilgroup.com/", "title": "ANIL - Leading Indian FMCG"},
{"href": "https://en.wikipedia.org/wiki/Vermicelli", "title": "Vermicelli"},
]
def test_reaching_no_shop_at_all_is_unknown_not_absent(monkeypatch):
"""THE CORRECTNESS FIX.
Recording "no retailer lists this" when not one result was even on a
retailer states something we did not find out. Measured: 2 of 10 sampled
negatives were this case, including `Anil Rava Semiya 200g` and
`Anil Idli Dosa Mix 500g`.
"""
monkeypatch.setattr(rp, "_search", lambda *a, **k: _shopless_results())
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
evidence = rp.check_listing("Anil", "Anil Rava Semiya", "200g", live=True)
assert evidence.status == rp.UNKNOWN
assert evidence.shop_results_seen == 0
def test_that_unknown_is_distinguishable_from_a_throttle(monkeypatch):
"""The backfill stops the sweep on a throttle and carries on past a
no-shop result, so the two must not look alike. `checked_at` is the tell:
a throttle never got as far as having a time of check."""
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "_search", lambda *a, **k: _shopless_results())
no_shop = rp.check_listing("Anil", "Anil Rava Semiya", "200g", live=True)
monkeypatch.setattr(rp, "_search", lambda *a, **k: None)
throttled = rp.check_listing("Anil", "Anil Rava Semiya", "200g", live=True)
assert no_shop.status == throttled.status == rp.UNKNOWN
assert no_shop.checked_at is not None
assert throttled.checked_at is None
def test_a_shop_that_was_reached_and_did_not_list_it_is_not_found(monkeypatch):
"""The other half of the distinction: this one IS an answer, and must stay
`not_found` rather than being softened into `unknown`."""
monkeypatch.setattr(rp, "_search", lambda *a, **k: [
{"href": "https://www.flipkart.com/aachi-biryani-masala/p/x",
"title": "Aachi BIRYANI MASALA Price in India"},
])
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
evidence = rp.check_listing("Aachi", "Aachi Biryani Mix", "1kg", live=True)
assert evidence.status == rp.NOT_FOUND
assert evidence.shop_results_seen >= 1
def test_a_negative_records_the_titles_it_rejected(monkeypatch):
"""Without the titles, judging whether a rejection was right means running
the search again - and reach varies run to run, so the answer changes."""
monkeypatch.setattr(rp, "_search", lambda *a, **k: [
{"href": "https://www.flipkart.com/x/p/y",
"title": "Aachi BIRYANI MASALA Price in India"},
])
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
evidence = rp.check_listing("Aachi", "Aachi Biryani Mix", "1kg", live=True)
assert any("BIRYANI MASALA" in t for t in evidence.seen_titles)
# ---------------------------------------------------------------------------
# Synonyms
# ---------------------------------------------------------------------------
def test_a_transliterated_listing_matches(monkeypatch):
"""amazon.in sells our "Anil Puttu Mix" as "Anil Puttu Maavu" - maavu is
simply Tamil for the flour. It scored 0.5 coverage and was rejected."""
matched, _c = rp.listing_matches(
"Amazon.in: Anil Puttu Maavu 500 g", "Anil", "Anil Puttu Mix", "500g")
assert matched
def test_semiya_and_vermicelli_are_the_same_word():
"""This catalogue uses both spellings in its OWN product names."""
matched, _c = rp.listing_matches(
"Anil Wheat Vermicelli | Buy Atta Semiya 180g Online", "Anil",
"Anil Wheat Vermicelli", "180g")
assert matched
def test_a_synonym_cannot_rescue_a_wrong_pack_size():
"""THE GUARD ON THE SYNONYM MAP.
Synonyms loosen which words count as the same word. They must never loosen
the pack size, or the map re-opens the exact hole this module was built to
close - the title already covers 100% for the fabricated 12g pack.
"""
matched, coverage = rp.listing_matches(
"Amazon.in: Anil Puttu Maavu 500 g", "Anil", "Anil Puttu Mix", "12g")
assert coverage == 1.0, "the title matches completely"
assert not matched, "and the size must still reject it"
def test_the_synonym_map_stays_small():
"""It is observed-pairs-only by rule, and every entry cites the listing
that forced it. A general thesaurus would start confirming products that
do not exist."""
assert len(rp._SYNONYMS) <= 12
# ---------------------------------------------------------------------------
# False positives found in real backfill data
#
# These matter more than the false negatives: a wrong FOUND hands 0.85
# corroboration to a product that may not exist, which is the failure this
# module exists to prevent. A wrong not_found merely withholds credit.
# ---------------------------------------------------------------------------
CONCATENATED_BLOB = (
"Ginger Garlic Paste (Pack of 2) | No Peeling, No ChoppingAachi Ginger "
"Garlic Paste 20g - martizo.comAachi Ginger Garlic Paste - Buy at Rs43 "
"Online | Instant ...Aachi Pickles & Instant Foods Buy 1 Get 1 | FREE "
"SHIPPING ...Aachi Ginger Garlic paste - LIFESHARE.IN"
)
def test_a_run_on_title_corroborates_nothing():
"""REAL FALSE POSITIVE, from the backfill cache.
The provider sometimes packs several results into one `title`. This 372-char
blob spans at least four listings from at least two shops, and it produced a
FOUND for "Aachi Ginger Paste 20g" — where the 20g came from martizo.com's
listing of a DIFFERENT product, credited to aachifoods.com. Neither the size
nor the word coverage can be attributed to one product, so it is discarded
rather than parsed.
"""
matched, _c = rp.listing_matches(
CONCATENATED_BLOB, "Aachi", "Aachi Ginger Paste", "20g")
assert not matched
assert len(CONCATENATED_BLOB) > rp.MAX_LISTING_TITLE_CHARS
def test_an_extra_ingredient_makes_it_a_different_product():
"""REAL FALSE POSITIVE, from the backfill cache.
"Aachi Ginger Paste" and "Aachi Garlic Paste" are both wholly contained in
"Aachi Ginger Garlic Paste" and scored 1.0 coverage against a clean, correct
JioMart listing at the right size. Containment tolerates extra words on the
listing side — right for "Pouch" and "Best Price", wrong for an ingredient.
"""
clean_listing = "Buy Aachi Ginger Garlic Paste 300 g Online - JioMart"
for wrong_product in ("Aachi Ginger Paste", "Aachi Garlic Paste"):
matched, coverage = rp.listing_matches(
clean_listing, "Aachi", wrong_product, "300g")
assert coverage == 1.0, "the target's words really are all present"
assert not matched, f"{wrong_product} is not ginger-garlic paste"
def test_the_ingredient_guard_does_not_break_the_exact_product():
"""The guard fires only on an ingredient the TARGET lacks."""
matched, _c = rp.listing_matches(
"Buy Aachi Ginger Garlic Paste 300 g Online - JioMart", "Aachi",
"Aachi Ginger Garlic Paste", "300g")
assert matched
def test_shop_furniture_is_not_treated_as_a_distinguishing_word():
""""Pouch", "Spicy", "Red", "Gel" are not ingredients. If the guard were a
general extra-word rule instead of a curated ingredient set, both controls
below would break."""
assert rp.listing_matches(AMAZON_450, "Anil",
"Anil Roasted Short Vermicelli", "450g")[0]
assert rp.listing_matches(BIGBASKET, "Colgate", "Colgate MaxFresh", "150g")[0]
def test_refresh_actually_re_queries(monkeypatch):
"""A SILENT NO-OP THAT LOOKED LIKE A SUCCESSFUL RUN.
The backfill's --refresh decides which rows to re-ask, then called
check_listing - which read the cache first and returned the very verdict
the caller was replacing. A whole rebuild of Anil reported 46 re-asks, made
no network call, and wrote nothing, because the early return happens before
set_cached.
"""
searched = []
stale = rp.RetailEvidence(rp.NOT_FOUND, shop_results_seen=1)
monkeypatch.setattr(rp, "get_cached", lambda *a, **k: stale)
monkeypatch.setattr(rp, "set_cached", lambda *a, **k: None)
monkeypatch.setattr(rp, "_search", lambda q, *a, **k: searched.append(q) or [])
rp.check_listing("Anil", "Anil Wheat Vermicelli", "180g", live=True)
assert searched == [], "without refresh, a cached verdict short-circuits"
rp.check_listing("Anil", "Anil Wheat Vermicelli", "180g", live=True, refresh=True)
assert searched, "with refresh, the cache must be bypassed and the search run"
# ---------------------------------------------------------------------------
# The listing must be for OUR brand
# ---------------------------------------------------------------------------
def test_a_competitors_listing_does_not_corroborate_our_product():
"""THE WORST FALSE POSITIVE THIS MODULE HAD, and it was live.
`normalize_for_match` strips brand tokens from BOTH sides - right for its
original job of matching inside one brand's Open Food Facts corpus, where
the brand is a given. Here the listing can be anyone's, so the comparison
was brand-blind and any competitor's same-shaped product confirmed ours.
The first line of the Anil backfill was exactly this: amazon.in's
"A1 Naanjil Naattu Masala Instant Pongal Mix, 500g" recorded as evidence
that Anil Pongal Mix 500g is on sale.
"""
for listing, product in [
("A1 Naanjil Naattu Masala Home Made Instant Pongal Mix, 500g",
"Anil Pongal Mix"),
("MTR Rava Idli Mix 500g", "Anil Idli Mix"),
("Britannia Good Day 200g", "Anil Good Day"),
]:
matched, _c = rp.listing_matches(listing, "Anil", product, "500g")
assert not matched, f"{listing!r} is not an Anil product"
def test_our_own_brand_still_matches():
"""The brand gate must not cost us the real confirmations."""
assert rp.listing_matches(AMAZON_450, "Anil",
"Anil Roasted Short Vermicelli", "450g")[0]
assert rp.listing_matches(BIGBASKET, "Colgate", "Colgate MaxFresh", "150g")[0]
def test_brand_aliases_are_honoured():
"""Reuses the barcode matcher's `brand_matches` and the curated registry,
so "HUL" keeps resolving to Hindustan Unilever and the two gates cannot
drift apart."""
assert "hindustan unilever" in rp._aliases_for("Hindustan Unilever")