image repair on existing products

This commit is contained in:
sriram
2026-09-04 13:23:43 +05:30
parent 0c3e23fac4
commit 8059120ce3
4 changed files with 347 additions and 10 deletions

View File

@@ -539,6 +539,7 @@ def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, An
return row
try:
from app.core.catalog_engine import catalog_engine
from app.services import produce_reference
from app.services.image_search import find_all_image_urls
# "Own Products" is a bucket, not a brand, so it must not enter the
@@ -547,15 +548,41 @@ def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, An
# name, and _select_best_images derives its distinctive tokens from the
# whole name ({toor, dal}), which is exactly right for a commodity.
brand = row.get("brand") or ""
if brand == OWN_PRODUCTS_BRAND:
is_produce = brand == OWN_PRODUCTS_BRAND
if is_produce:
brand = ""
product_name = row.get("product_name") or ""
# A reviewed photo beats anything a search can find, so take it and stop.
#
# NOT merged into the candidate list below: _select_best_images ranks by
# whether the URL text names the product, and a Commons filename names
# the cultivar ("Honeycrisp.jpg" for Apple). The curated image would
# score zero relevance, sort below any searched URL containing "apple",
# and index 0 - the one that becomes image_url - would still be wrong.
#
# Only the bucket may ask. `produce_reference.lookup` falls back through
# shorter leading prefixes, so "Apple Cider Vinegar" would resolve to
# "apple" and be handed a photo of fruit.
if is_produce:
curated = produce_reference.image_url(product_name)
if curated:
row["image_url"] = curated
row["image_urls"] = [curated]
return row
# produce=True suppresses the commodity-hint table (which turns "Apple"
# into "Apple fruit juice"), skips the packaged-goods databases, and
# drops the "-plant -tree -fish -botanical" Wikimedia exclusion block
# that fights every produce query. Without it this stage recreates the
# exact defect the Own Products image repair exists to undo.
candidates = find_all_image_urls(
row.get("product_name") or "", brand=brand or None, max_results=24
product_name, brand=brand or None, max_results=24, produce=is_produce
)
if candidates:
best = catalog_engine._select_best_images(
candidates, row.get("product_name") or "", brand, max_images=10
candidates, product_name, brand, max_images=10
)
if best:
row["image_urls"] = list(best)

View File

@@ -126,9 +126,46 @@ def usda_fdc_id(product_name: str) -> Optional[int]:
return entry.get("usda_fdc_id") if entry else None
# Words that make a name a DIFFERENT PRODUCT from the commodity it starts with,
# rather than a variety of it.
#
# `lookup` falls back through shorter leading prefixes, which is what lets
# "Mango Totapuri" and "Banana Robusta" find their base commodity - varieties of
# the same thing, correctly sharing one photo. The same fallback turns "Coconut
# Oil" into "Coconut" and "Apple Cider Vinegar" into "Apple", and a bottle of
# oil is not a variety of coconut. Handing it a photo of the raw fruit is the
# same class of error as the apple-juice image this table was built to fix, just
# pointing the other way.
#
# No row in the catalogue trips this today - all 159 names are exact keys - so
# this guards the case a future upload introduces, which is precisely when
# nobody would be looking.
_DERIVED_PRODUCT_WORDS = frozenset({
"oil", "vinegar", "juice", "squash", "syrup", "jam", "jelly", "sauce",
"ketchup", "puree", "paste", "pickle", "powder", "flour", "atta", "rava",
"flakes", "chips", "crisps", "candy", "extract", "essence", "milkshake",
"smoothie", "dried", "fried", "roasted", "pickled", "canned", "frozen",
})
def image_url(product_name: str) -> Optional[str]:
"""The reviewed photo for a commodity, or None.
Stricter than `lookup` on purpose, and only here: `usda_fdc_id` keeps the
plain prefix fallback, because this guard is about not showing a misleading
PICTURE. Nutrition for a derived product is refused further upstream by the
category dispatch in `nutrition_data_service`.
"""
entry = lookup(product_name)
return entry.get("image_url") if entry else None
if not entry:
return None
matched = _normalize(entry.get("product_name") or "")
leftover = set(_normalize(product_name).split()) - set(matched.split())
if leftover & _DERIVED_PRODUCT_WORDS:
return None
return entry.get("image_url")
def all_entries() -> Dict[str, Dict[str, Any]]: