Health score updates in backend

This commit is contained in:
sriram
2026-09-04 12:06:40 +05:30
parent 1300d4c678
commit 0c3e23fac4
33 changed files with 59703 additions and 94 deletions

View File

@@ -0,0 +1,428 @@
#!/usr/bin/env python3
"""
Render every Own Products row as a before/after card, for review before writing.
WHY A REVIEW STEP AND NOT A CONFIDENCE THRESHOLD
------------------------------------------------
No automated filter can tell that `Zunaid_Ahmed_Palak.jpg` is a photograph of a
politician rather than a bunch of spinach, or that `Bitter_Gourd_Curry.jpg` is a
cooked dish rather than the vegetable. Both were the stored image for a live
product, both resolved to real image bytes, and both passed every check the
pipeline had. The only thing that catches them is a person looking.
So this script writes nothing to the database. It renders what WOULD be written
and stops. `repair_brand_images.py --brands "Own Products" --apply` is the step
that writes, and it should only be run once these cards have been looked at.
WHAT EACH CARD SHOWS
--------------------
* the image currently stored, and whether it still resolves
* the curated replacement from `produce_reference`
* where that replacement came from, and for the 15 hand-picked ones, why
* the nutrition that will be attached, with its USDA source, or an explicit
note that there is none and why
USAGE
-----
python -m scripts.build_produce_contact_sheet
python -m scripts.build_produce_contact_sheet --out sheet.html --no-probe
`--no-probe` skips checking whether the stored URLs still resolve, which is the
slow part; without it the script makes one ranged GET per distinct stored URL.
`--embed` downloads every image and inlines it as a small data: URI, which is
required if the page is going to be published as an Artifact - that viewer's
content security policy blocks third-party image hosts, so a page of hotlinked
photographs renders as 318 empty boxes.
Read-only in every mode: it opens no write transaction.
"""
from __future__ import annotations
import argparse
import html
import logging
import sys
import time
from datetime import datetime
from pathlib import Path
from typing import Any, Dict, List
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.services import nutrition_scoring, produce_reference # noqa: E402
from app.services.consumability import classify_edibility # noqa: E402
from app.services.generic_products import OWN_PRODUCTS_BRAND # noqa: E402
from app.services.nutrition_usda_service import ( # noqa: E402
fetch_verified_nutrition_usda,
)
from app.services.vector_store import _connect, _sanitize_name # noqa: E402
logging.basicConfig(level=logging.INFO, format="%(message)s")
logger = logging.getLogger("contact_sheet")
TABLE = f"brand_{_sanitize_name(OWN_PRODUCTS_BRAND)}"
def _rows() -> List[Dict[str, Any]]:
conn = _connect()
if conn is None:
logger.error("No database connection.")
return []
try:
with conn.cursor() as cur:
cur.execute(
f"SELECT product_name, category, image_url FROM {TABLE} "
f"ORDER BY category, product_name"
)
return [{"product_name": r[0], "category": r[1], "image_url": r[2]}
for r in cur.fetchall()]
finally:
conn.close()
# Thumbnails, not photographs. 318 full-size images would be ~40 MB and the
# Artifact ceiling is 16 MB; at 168px they are legible side by side, which is
# all this page asks of them, and the whole set fits in about 2 MB.
_THUMB_PX = 168
_THUMB_QUALITY = 62
def _thumb(url: str, cache: Dict[str, str]) -> str:
"""`url` fetched, downscaled and returned as a data: URI, or "" on failure."""
if not url:
return ""
if url in cache:
return cache[url]
import base64
from io import BytesIO
import requests
from PIL import Image
data = ""
try:
resp = requests.get(url, headers={"User-Agent": _UA}, timeout=20)
if resp.status_code == 200 and resp.content:
img = Image.open(BytesIO(resp.content))
img = img.convert("RGB")
img.thumbnail((_THUMB_PX, _THUMB_PX), Image.LANCZOS)
buf = BytesIO()
img.save(buf, format="JPEG", quality=_THUMB_QUALITY, optimize=True)
data = "data:image/jpeg;base64," + base64.b64encode(buf.getvalue()).decode()
except Exception: # noqa: BLE001 - a broken image is the thing being reported
data = ""
cache[url] = data
time.sleep(0.12)
return data
_UA = "nearle-catalogue/1.0 (produce image review sheet)"
def _probe(url: str, cache: Dict[str, bool]) -> bool:
if not url:
return False
if url in cache:
return cache[url]
from app.services.image_search import validate_image_url_live
try:
ok = bool(validate_image_url_live(url, timeout=8))
except Exception: # noqa: BLE001
ok = False
cache[url] = ok
time.sleep(0.1)
return ok
_CSS = """
/* Palette is the greengrocer's, not the default warm-cream: a cool paper with a
faint green cast, deep leaf-black ink, and aubergine as the one accent - kept
clear of the green/amber/red that carry MEANING on this page. */
:root{
--paper:#eef1ec; --card:#ffffff; --line:#d8dfd6; --line-soft:#e8ede6;
--ink:#131c17; --ink-2:#3d4b43; --dim:#6a7a70;
--accent:#6b3a58; /* aubergine */
--ok:#1a6b45; --ok-bg:#e3f1e9;
--warn:#8a5600; --warn-bg:#fbeed7;
--crit:#9c2b2b; --crit-bg:#fbe6e6;
--shadow:0 1px 2px rgba(19,28,23,.05);
}
@media (prefers-color-scheme:dark){
:root:not([data-theme="light"]){
--paper:#11150f; --card:#1a1f19; --line:#2e352c; --line-soft:#232a22;
--ink:#e8ece6; --ink-2:#b9c3ba; --dim:#8b968c;
--accent:#d99bc4;
--ok:#71d3a1; --ok-bg:#12301f;
--warn:#e8b761; --warn-bg:#33260d;
--crit:#f09393; --crit-bg:#331616;
--shadow:0 1px 2px rgba(0,0,0,.35);
}
}
:root[data-theme="dark"]{
--paper:#11150f; --card:#1a1f19; --line:#2e352c; --line-soft:#232a22;
--ink:#e8ece6; --ink-2:#b9c3ba; --dim:#8b968c;
--accent:#d99bc4;
--ok:#71d3a1; --ok-bg:#12301f;
--warn:#e8b761; --warn-bg:#33260d;
--crit:#f09393; --crit-bg:#331616;
--shadow:0 1px 2px rgba(0,0,0,.35);
}
*{box-sizing:border-box}
body{
margin:0; background:var(--paper); color:var(--ink);
font:400 15px/1.55 "IBM Plex Sans","Segoe UI",system-ui,-apple-system,sans-serif;
-webkit-font-smoothing:antialiased;
}
.wrap{max-width:1500px;margin:0 auto}
header{padding:38px 26px 26px;border-bottom:1px solid var(--line)}
h1{
margin:0 0 8px; font-family:"Bricolage Grotesque","IBM Plex Sans",sans-serif;
font-weight:600; font-size:clamp(26px,3.4vw,38px); letter-spacing:-.025em;
text-wrap:balance; line-height:1.08;
}
.sub{color:var(--ink-2);font-size:14.5px;max-width:66ch;margin:0}
.sub strong{color:var(--ink);font-weight:600}
.tally{display:flex;flex-wrap:wrap;gap:0;margin-top:22px;
border:1px solid var(--line);border-radius:10px;overflow:hidden;background:var(--card)}
.tally .t{padding:11px 18px;border-right:1px solid var(--line);flex:1 1 auto;min-width:132px}
.tally .t:last-child{border-right:0}
.tally b{display:block;font-family:"IBM Plex Mono",ui-monospace,monospace;
font-size:23px;font-weight:600;letter-spacing:-.02em;font-variant-numeric:tabular-nums}
.tally span{font-size:11.5px;color:var(--dim);text-transform:uppercase;letter-spacing:.07em}
.tally .t.crit b{color:var(--crit)} .tally .t.ok b{color:var(--ok)}
h2{
margin:38px 26px 13px; font-size:12.5px; font-weight:600; color:var(--dim);
text-transform:uppercase; letter-spacing:.13em;
display:flex; align-items:baseline; gap:10px;
}
h2::after{content:"";flex:1;height:1px;background:var(--line)}
h2 em{font-style:normal;font-family:"IBM Plex Mono",monospace;color:var(--dim);
font-size:12px;font-variant-numeric:tabular-nums}
.grid{display:grid;grid-template-columns:repeat(auto-fill,minmax(322px,1fr));
gap:13px;padding:0 26px}
/* The left rail encodes the row's actual state, so severity reads before any
text does: dead stored image, wrong-source image, or already fine. */
.card{background:var(--card);border:1px solid var(--line);border-radius:11px;
overflow:hidden;box-shadow:var(--shadow);border-left:3px solid var(--line)}
.card.crit{border-left-color:var(--crit)}
.card.warn{border-left-color:var(--warn)}
.card.ok{border-left-color:var(--ok)}
.name{padding:12px 14px 10px;font-weight:600;font-size:15.5px;letter-spacing:-.01em}
.pair{display:grid;grid-template-columns:1fr 1fr;gap:1px;background:var(--line-soft)}
.cell{background:var(--card);padding:10px}
.lbl{font-size:10px;text-transform:uppercase;letter-spacing:.09em;color:var(--dim);
margin-bottom:7px;display:flex;align-items:center;gap:6px;min-height:16px}
.ph{width:100%;aspect-ratio:1;object-fit:cover;border-radius:7px;
background:var(--line-soft);display:block}
.miss{width:100%;aspect-ratio:1;border-radius:7px;background:var(--line-soft);
display:flex;align-items:center;justify-content:center;color:var(--dim);
font-size:11.5px;text-align:center;padding:10px;line-height:1.4}
.meta{padding:11px 14px 13px;font-size:12.5px;color:var(--ink-2);
border-top:1px solid var(--line-soft);display:flex;flex-direction:column;gap:5px}
.tag{display:inline-block;border-radius:5px;padding:2px 7px;font-size:10.5px;
font-weight:600;text-transform:uppercase;letter-spacing:.05em;white-space:nowrap}
.t-ok{background:var(--ok-bg);color:var(--ok)}
.t-crit{background:var(--crit-bg);color:var(--crit)}
.t-warn{background:var(--warn-bg);color:var(--warn)}
.nut{font-family:"IBM Plex Mono",ui-monospace,monospace;font-size:12px;
font-variant-numeric:tabular-nums;display:flex;align-items:center;gap:7px;flex-wrap:wrap}
code{font-family:"IBM Plex Mono",ui-monospace,monospace;font-size:11px;
background:var(--line-soft);padding:1.5px 5px;border-radius:4px;word-break:break-all;
color:var(--ink-2)}
footer{margin-top:46px;padding:24px 26px 40px;border-top:1px solid var(--line);
color:var(--dim);font-size:13px;line-height:1.6}
footer b{color:var(--ink-2);font-weight:600}
a{color:var(--accent)}
@media (prefers-reduced-motion:reduce){*{animation:none!important;transition:none!important}}
"""
def _card(row: Dict[str, Any], probe_cache: Dict[str, bool], probe: bool,
thumbs: Dict[str, str] | None = None) -> str:
name = row["product_name"] or ""
entry = produce_reference.lookup(name) or {}
verdict = classify_edibility(row["category"], name)
consumable = verdict.edibility.value == "consumable"
stored = row.get("image_url") or ""
stored_ok = _probe(stored, probe_cache) if (probe and stored) else None
proposed = entry.get("image_url")
# The stored URL's HOST is itself a verdict. Open Beauty Facts is a
# cosmetics database and Open Products Facts a packaged-goods one; a raw
# banana photographed front-of-pack is the wrong SUBJECT even when the URL
# resolves perfectly, which no liveness probe can detect.
wrong_source = any(h in stored.lower() for h in
("openbeautyfacts", "openproductsfacts", "openfoodfacts"))
if thumbs is not None and stored and not _thumb(stored, thumbs):
stored_ok = False
if not stored or stored_ok is False:
severity, verdict_tag = "crit", '<span class="tag t-crit">dead</span>'
elif wrong_source:
severity, verdict_tag = "warn", '<span class="tag t-warn">wrong source</span>'
elif stored_ok:
severity, verdict_tag = "ok", '<span class="tag t-ok">resolves</span>'
else:
severity, verdict_tag = "", ""
def _img(url: str, empty: str) -> str:
if not url:
return f'<div class="miss">{empty}</div>'
src = _thumb(url, thumbs) if thumbs is not None else url
if not src:
return '<div class="miss">image did not load</div>'
return f'<img class="ph" src="{html.escape(src)}" loading="lazy" alt="">'
left = _img(stored, "no image stored")
right = _img(proposed, "no curated image")
bits = []
if entry.get("image_source") == "wikimedia_commons":
bits.append('<div><span class="tag t-warn">hand-picked</span> '
"article lead image was a botanical plate</div>")
if entry.get("image_credit"):
bits.append(f'<div><code>{html.escape(str(entry["image_credit"]))}</code></div>')
if not consumable:
bits.append('<div><span class="tag t-warn">not food</span> '
"no nutrition, by design</div>")
elif entry.get("usda_fdc_id"):
facts = fetch_verified_nutrition_usda(name, row["category"])
scores = nutrition_scoring.compute_scores(facts) or {}
if facts.get("data_status") != "unavailable":
bits.append(
f'<div class="nut"><span class="tag t-ok">health '
f'{scores.get("health_score", "-")}</span>'
f'<span>{facts.get("calories_kcal")} kcal</span>'
f'<span>{facts.get("protein_g")} g protein</span>'
f'<span>{facts.get("dietary_fiber_g")} g fibre</span></div>')
bits.append(f'<div>USDA <code>{facts["source_ref"]}</code> '
f'{html.escape(str(entry.get("usda_description", ""))[:52])}</div>')
else:
bits.append('<div><span class="tag t-warn">no nutrition</span> '
"USDA has no entry for this species</div>")
return f"""<article class="card {severity}">
<div class="name">{html.escape(name)}</div>
<div class="pair">
<div class="cell"><div class="lbl">stored now {verdict_tag}</div>{left}</div>
<div class="cell"><div class="lbl">proposed</div>{right}</div>
</div>
<div class="meta">{"".join(bits)}</div>
</article>"""
def build(rows: List[Dict[str, Any]], probe: bool, embed: bool = False) -> str:
cache: Dict[str, bool] = {}
thumbs: Dict[str, str] | None = {} if embed else None
by_category: Dict[str, List[Dict[str, Any]]] = {}
for row in rows:
by_category.setdefault(row["category"] or "(uncategorised)", []).append(row)
sections = []
for category in sorted(by_category):
cards = [_card(row, cache, probe, thumbs) for row in by_category[category]]
logger.info(" %-26s %3d card(s)", category, len(by_category[category]))
sections.append(
f'<h2>{html.escape(category)}<em>{len(by_category[category])}</em></h2>'
f'<div class="grid">{"".join(cards)}</div>')
n_dead = n_wrong = n_food = n_scored = 0
for row in rows:
entry = produce_reference.lookup(row["product_name"] or "") or {}
stored = row.get("image_url") or ""
if classify_edibility(row["category"], row["product_name"] or "") \
.edibility.value == "consumable":
n_food += 1
if entry.get("usda_fdc_id"):
n_scored += 1
if not stored:
n_dead += 1
elif thumbs is not None and not thumbs.get(stored):
# --embed already fetched every URL; a thumbnail that came back
# empty is the same evidence a liveness probe would have gathered.
n_dead += 1
elif probe and not cache.get(stored):
n_dead += 1
elif any(h in stored.lower() for h in
("openbeautyfacts", "openproductsfacts", "openfoodfacts")):
n_wrong += 1
tally = [("t crit", n_dead, "dead images"),
("t crit", n_wrong, "wrong subject"),
("t ok", n_scored, "get a health score"),
("t", len(rows) - n_food, "not food, unscored"),
("t", len(rows), "products in all")]
tiles = "".join(f'<div class="{c}"><b>{v}</b><span>{lbl}</span></div>'
for c, v, lbl in tally)
return f"""<title>Fresh Aisle Audit</title>
<link rel="preconnect" href="https://fonts.googleapis.com">
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
<link rel="stylesheet" href="https://fonts.googleapis.com/css2?\
family=Bricolage+Grotesque:opsz,wght@12..96,500;12..96,600&\
family=IBM+Plex+Mono:wght@400;600&\
family=IBM+Plex+Sans:wght@400;500;600&display=swap">
<style>{_CSS}</style>
<div class="wrap">
<header>
<h1>Every loose thing the shop sells, and the picture it currently shows</h1>
<p class="sub">The {len(rows)} unbranded rows in <code>brand_own_products</code>,
each showing what is stored today beside what would replace it.
<strong>Nothing has been written to the database.</strong> The repair script
runs only after these have been looked at &mdash; because no automated check
can tell that the stored photo for <em>Palak</em> is a politician, or that
<em>Bitter Gourd</em> is a plate of curry.</p>
<div class="tally">{tiles}</div>
</header>
{"".join(sections)}
<footer>
<b>Images</b> &mdash; English Wikipedia article lead images, resolved by
botanical name wherever the common name is also a personal name, plus 15
hand-picked from Wikimedia Commons where the article's own lead image was a
19th-century botanical plate rather than a photograph.<br>
<b>Nutrition</b> &mdash; USDA FoodData Central, Foundation Foods and SR
Legacy, pinned from the open bulk download. Every value is per 100&nbsp;g of
edible portion and links back to its FDC record; commodities USDA has no
entry for get no numbers rather than a near neighbour's.<br>
Generated {datetime.now():%d %B %Y, %H:%M}.
</footer>
</div>"""
def main() -> int:
parser = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--out", default="produce_contact_sheet.html")
parser.add_argument("--no-probe", action="store_true",
help="skip checking whether the stored URLs still resolve")
parser.add_argument("--embed", action="store_true",
help="inline every image as a downscaled data: URI. Required "
"for publishing as an Artifact, whose CSP blocks "
"third-party image hosts.")
args = parser.parse_args()
rows = _rows()
if not rows:
logger.error("No rows read from %s", TABLE)
return 1
logger.info("Rendering %d rows from %s", len(rows), TABLE)
page = build(rows, probe=not args.no_probe, embed=args.embed)
Path(args.out).write_text(page, encoding="utf-8")
logger.info("Wrote %s (%.1f MB)", args.out, len(page.encode("utf-8")) / 1e6)
logger.info("Nothing was written to the database.")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -8,6 +8,16 @@ Usage:
python scripts/enrich_nutrition.py --max-products 200 --no-narrative
python scripts/enrich_nutrition.py --force # re-fetch even already-verified products
# Catalogue-wide, past ACTIVE_BRANDS. Without this the run cannot reach
# Britannia, Parle or ITC at all, because enrich_all_products reads through
# vector_store.list_available_brands(), which honours that filter.
python scripts/enrich_nutrition.py --include-inactive
python scripts/enrich_nutrition.py --brands "Britannia,Parle"
python scripts/enrich_nutrition.py --categories "Fruits & Vegetables,Eggs"
Non-consumables (soap, shampoo, mosquito repellent, flowers) are skipped and
reported separately - see app/services/consumability.py.
Equivalent to POST /api/admin/nutrition-intelligence/enrich.
"""
from __future__ import annotations
@@ -36,6 +46,13 @@ def main() -> None:
parser.add_argument("--force", action="store_true", help="Re-fetch even products already marked 'verified'")
parser.add_argument("--no-narrative", action="store_true", help="Skip the LLM narrative step (facts/scores/tags only)")
parser.add_argument("--max-products", type=int, default=None, help="Cap the number of products processed (for a quick test run)")
parser.add_argument("--brands", default=None,
help="Comma-separated brand display names to enrich (default: whatever ACTIVE_BRANDS allows)")
parser.add_argument("--include-inactive", action="store_true",
help="Enrich every brand table, ignoring ACTIVE_BRANDS. Needed to reach "
"brands the app does not currently display.")
parser.add_argument("--categories", default=None,
help="Comma-separated categories to restrict the run to, e.g. 'Fruits & Vegetables,Eggs'")
args = parser.parse_args()
ensure_nutrition_schema()
@@ -44,12 +61,16 @@ def main() -> None:
generate_narrative=not args.no_narrative,
progress_cb=_progress,
max_products=args.max_products,
brands=[b for b in (args.brands or "").split(",") if b.strip()] or None,
include_inactive=args.include_inactive,
categories=[c for c in (args.categories or "").split(",") if c.strip()] or None,
)
logger.info(
f"Done in {result.duration_seconds}s - "
f"{result.verified} verified, {result.partial} partial, {result.unavailable} unavailable "
f"of {result.total_products} total products"
f"of {result.total_products} total products "
f"({result.skipped_non_consumable} non-consumable skipped)"
)
if result.errors:
logger.warning(f"{len(result.errors)} errors (showing up to 10): {result.errors[:10]}")

View File

@@ -0,0 +1,491 @@
#!/usr/bin/env python3
"""
Remove nutrition rows belonging to products that are not food.
WHY THIS EXISTS
---------------
`app/services/consumability.py` now stops these rows being created. It cannot
remove the ones already there. Measured on the live catalogue:
Colgate-Palmolive Palmolive Naturals Bath Soap health score present
Cavinkare Nyle, Cavinkare Nature's Hair Care health score present
P&G Pantene Hair Care health score present
Godrej Hit Spray Mosquito 291 kcal, health score present
A shopper reading "health score 62" on a mosquito repellent has no way to know
the number is meaningless, and `nutrition_db.query_products` will happily rank
it against actual food.
WHAT IT WILL NOT DELETE
-----------------------
* Anything `consumability` cannot positively identify as non-food. The
classifier is three-state on purpose and this script uses the POSITIVE test
`is_non_consumable`, never `not is_consumable`. An unrecognised product is
reported as UNDECIDED and left exactly where it is - a wrong deletion here is
unrecoverable except from the backup, a wrong retention is visible and
fixable.
* Orphans - nutrition rows whose (brand, image_id) is in no brand table. That
is a different defect (a deleted product, or a brand-name mismatch since
enrichment) and deleting them here would hide it.
WHY IT RE-READS THE CATEGORY FROM THE BRAND TABLE
-------------------------------------------------
`nutrition_facts.category` is a copy taken at enrichment time
(`enrich_one_product`), so it can be stale relative to the product. The verdict
is taken against the brand table's current category and product name, which is
what the catalogue actually claims today.
ORDER OF DELETION
-----------------
Children first - `nutrition_similar_products` and
`nutrition_healthy_alternatives` before `nutrition_insights` and
`nutrition_facts` - so an interrupted run never leaves a similarity edge
pointing at a fact row that no longer exists.
Both cache tables are cleaned in BOTH DIRECTIONS. The inbound direction is the
one that gets forgotten: a shampoo can be some real food's nearest neighbour,
and that edge survives a naive purge, after which `get_similar_products` serves
a dangling reference.
USAGE
-----
python -m scripts.purge_non_consumable_nutrition --audit
python -m scripts.purge_non_consumable_nutrition --audit --json
python -m scripts.purge_non_consumable_nutrition --purge # dry run
python -m scripts.purge_non_consumable_nutrition --purge --apply
python -m scripts.purge_non_consumable_nutrition --restore data/nutrition_purge_backup_X.json --apply
`--audit` is read-only by construction. `--purge` is a dry run until `--apply`,
and `--apply` writes a full backup of every affected row first.
AFTERWARDS
----------
The KNN index and KMeans clusters in `data/artifacts/*.joblib` were fitted with
the deleted rows still in them. Re-run the training step or the similar-products
cache will keep serving keys that no longer exist. The script prints this.
"""
from __future__ import annotations
import argparse
import json
import logging
import sys
from dataclasses import dataclass, field
from datetime import datetime
from decimal import Decimal
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.infrastructure.settings import DB_HOST, DB_NAME # noqa: E402
from app.services.consumability import ( # noqa: E402
Edibility, classify_edibility, is_junk_category,
)
from app.services.vector_store import _connect, display_name_for_suffix # noqa: E402
logging.basicConfig(level=logging.INFO, format="%(message)s")
logger = logging.getLogger("purge_non_consumable_nutrition")
DATA_DIR = Path(__file__).resolve().parents[1] / "data"
# Children first. `nutrition_facts` last, so a crash mid-run leaves referential
# integrity intact rather than half-broken.
CACHE_TABLES = (
("nutrition_similar_products", ("brand", "image_id"), ("similar_brand", "similar_image_id")),
("nutrition_healthy_alternatives", ("brand", "image_id"), ("alt_brand", "alt_image_id")),
)
CORE_TABLES = ("nutrition_insights", "nutrition_facts")
Key = Tuple[str, str]
# ---------------------------------------------------------------------------
# Planning - pure, no database, so it is testable
# ---------------------------------------------------------------------------
@dataclass
class PurgePlan:
delete: List[Dict[str, Any]] = field(default_factory=list)
keep: List[Dict[str, Any]] = field(default_factory=list)
undecided: List[Dict[str, Any]] = field(default_factory=list)
orphans: List[Dict[str, Any]] = field(default_factory=list)
junk_categories: List[Dict[str, Any]] = field(default_factory=list)
@property
def delete_keys(self) -> List[Key]:
return [(r["brand"], r["image_id"]) for r in self.delete]
def plan_purge(
nutrition_rows: List[Dict[str, Any]],
catalogue: Dict[Key, Dict[str, Any]],
) -> PurgePlan:
"""Decide the fate of every `nutrition_facts` row.
`nutrition_rows` needs only brand, image_id, product_name, category and
data_status. `catalogue` is (brand, image_id) -> the brand table's current
product_name and category, which is the authority - see the module
docstring on why the fact row's own copy is not trusted.
"""
plan = PurgePlan()
for row in nutrition_rows:
key = (row.get("brand") or "", row.get("image_id") or "")
source = catalogue.get(key)
if source is None:
plan.orphans.append({**row, "reason": "no matching row in any brand table"})
continue
category = source.get("category") or ""
name = source.get("product_name") or row.get("product_name") or ""
if is_junk_category(category):
plan.junk_categories.append({**row, "category": category, "product_name": name})
verdict = classify_edibility(category, name)
entry = {
**row,
"category": category,
"product_name": name,
"reason": verdict.reason,
"signal": verdict.signal,
}
if verdict.edibility is Edibility.NON_CONSUMABLE:
plan.delete.append(entry)
elif verdict.edibility is Edibility.CONSUMABLE:
plan.keep.append(entry)
else:
plan.undecided.append(entry)
return plan
# ---------------------------------------------------------------------------
# Reading
# ---------------------------------------------------------------------------
def _catalogue_index(cur) -> Dict[Key, Dict[str, Any]]:
"""(brand, image_id) -> current product_name and category.
The brand key is `display_name_for_suffix(...)`, which is the string
`enrich_one_product` was called with, so it is the correct join key.
"""
cur.execute(
"SELECT table_name FROM information_schema.tables "
"WHERE table_schema='public' AND table_name LIKE 'brand_%' ORDER BY table_name"
)
index: Dict[Key, Dict[str, Any]] = {}
for (table,) in cur.fetchall():
cur.execute(
"SELECT column_name FROM information_schema.columns WHERE table_name=%s",
(table,),
)
cols = {r[0] for r in cur.fetchall()}
if "image_id" not in cols:
continue
select = "image_id" + (",product_name" if "product_name" in cols else "")
select += ",category" if "category" in cols else ""
cur.execute(f"SELECT {select} FROM {table} WHERE image_id IS NOT NULL")
brand = display_name_for_suffix(table[len("brand_"):])
for record in cur.fetchall():
row = dict(zip(select.split(","), record))
index[(brand, row["image_id"])] = {
"product_name": row.get("product_name") or "",
"category": row.get("category") or "",
"table": table,
}
return index
def _nutrition_rows(cur) -> List[Dict[str, Any]]:
cur.execute(
"SELECT brand, image_id, product_name, category, data_status "
"FROM nutrition_facts"
)
cols = ("brand", "image_id", "product_name", "category", "data_status")
return [dict(zip(cols, r)) for r in cur.fetchall()]
# ---------------------------------------------------------------------------
# Reporting
# ---------------------------------------------------------------------------
def _report(plan: PurgePlan, as_json: bool) -> Dict[str, Any]:
scored = [r for r in plan.delete if (r.get("data_status") or "") != "unavailable"]
by_category: Dict[str, int] = {}
for row in plan.delete:
by_category[row["category"]] = by_category.get(row["category"], 0) + 1
summary = {
"nutrition_rows": len(plan.delete) + len(plan.keep) + len(plan.undecided) + len(plan.orphans),
"to_delete": len(plan.delete),
"to_delete_holding_real_data": len(scored),
"keep": len(plan.keep),
"undecided": len(plan.undecided),
"orphans": len(plan.orphans),
"junk_categories": len(plan.junk_categories),
"by_category": dict(sorted(by_category.items(), key=lambda kv: -kv[1])),
}
if as_json:
print(json.dumps({"summary": summary, "delete": plan.delete,
"undecided": plan.undecided, "orphans": plan.orphans}, indent=2))
return summary
logger.info("")
logger.info("Target database: %s / %s", DB_HOST, DB_NAME)
logger.info("")
logger.info("%d nutrition_facts row(s)", summary["nutrition_rows"])
logger.info(" %6d non-consumable <-- would be deleted", summary["to_delete"])
logger.info(" %6d of those actually hold nutrition data (not 'unavailable')",
summary["to_delete_holding_real_data"])
logger.info(" %6d consumable, kept", summary["keep"])
logger.info(" %6d undecided, kept - extend the map or fix the category",
summary["undecided"])
logger.info(" %6d orphaned, kept - product missing from every brand table",
summary["orphans"])
if by_category:
logger.info("")
logger.info("Non-consumable rows by category:")
for category, count in sorted(by_category.items(), key=lambda kv: -kv[1]):
logger.info(" %-40s %5d", category or "(blank)", count)
if scored:
logger.info("")
logger.info("Rows holding real nutrition for a non-food product "
"- this is what justifies the change:")
for row in scored[:25]:
logger.info(" %-34s %-26s %s",
(row["product_name"] or "")[:33], row["category"][:25],
row["data_status"])
if len(scored) > 25:
logger.info(" ... and %d more", len(scored) - 25)
if plan.undecided:
logger.info("")
logger.info("UNDECIDED - kept, because guessing here would delete real data:")
for row in plan.undecided[:20]:
logger.info(" %-34s %-22s %s",
(row["product_name"] or "")[:33],
(row["category"] or "(blank)")[:21], row["reason"][:60])
if len(plan.undecided) > 20:
logger.info(" ... and %d more", len(plan.undecided) - 20)
if plan.junk_categories:
logger.info("")
logger.info("%d row(s) carry a bare number as their category. That is import "
"damage and wants a catalogue repair, not a rule here.",
len(plan.junk_categories))
if plan.orphans:
logger.info("")
logger.info("ORPHANS - a nutrition row with no product. Kept and reported; "
"deleting them here would hide the underlying defect:")
for row in plan.orphans[:10]:
logger.info(" %s / %s", row["brand"], row["image_id"])
if len(plan.orphans) > 10:
logger.info(" ... and %d more", len(plan.orphans) - 10)
return summary
# ---------------------------------------------------------------------------
# Writing
# ---------------------------------------------------------------------------
def _plain(value: Any) -> Any:
"""JSON-safe. nutrition_facts has NUMERIC and TIMESTAMP columns, so this
hits both traps that merge_haldiram hit for real."""
if isinstance(value, Decimal):
return float(value)
if isinstance(value, datetime):
return value.isoformat()
return value
def _backup(cur, keys: List[Key]) -> Path:
"""Every row about to be touched, across all four tables, in both
directions for the two cache tables."""
DATA_DIR.mkdir(parents=True, exist_ok=True)
tables: Dict[str, List[Dict[str, Any]]] = {}
for table in CORE_TABLES:
cur.execute(f"SELECT * FROM {table} WHERE (brand, image_id) IN %s",
(tuple(keys),))
cols = [d[0] for d in cur.description]
tables[table] = [{c: _plain(v) for c, v in zip(cols, r)} for r in cur.fetchall()]
for table, own, inbound in CACHE_TABLES:
cur.execute(
f"SELECT * FROM {table} WHERE ({own[0]}, {own[1]}) IN %s "
f"OR ({inbound[0]}, {inbound[1]}) IN %s",
(tuple(keys), tuple(keys)),
)
cols = [d[0] for d in cur.description]
tables[table] = [{c: _plain(v) for c, v in zip(cols, r)} for r in cur.fetchall()]
path = DATA_DIR / f"nutrition_purge_backup_{datetime.now():%Y%m%d_%H%M%S}.json"
path.write_text(json.dumps({
"created_at": datetime.now().isoformat(),
"db_host": DB_HOST, "db_name": DB_NAME,
"keys": [list(k) for k in keys],
"tables": tables,
}, indent=2), encoding="utf-8")
return path
def _delete(cur, keys: List[Key]) -> Dict[str, int]:
deleted: Dict[str, int] = {}
for table, own, inbound in CACHE_TABLES:
cur.execute(
f"DELETE FROM {table} WHERE ({own[0]}, {own[1]}) IN %s "
f"OR ({inbound[0]}, {inbound[1]}) IN %s",
(tuple(keys), tuple(keys)),
)
deleted[table] = cur.rowcount
for table in CORE_TABLES:
cur.execute(f"DELETE FROM {table} WHERE (brand, image_id) IN %s", (tuple(keys),))
deleted[table] = cur.rowcount
return deleted
def _backfill_edibility(cur, plan: PurgePlan) -> Dict[str, int]:
"""Stamp the verdict this run already computed onto the rows it is keeping.
WHY HERE AND NOT IN A MIGRATION: `plan_purge` has just classified every one
of these rows against the catalogue's own category, using the same
classifier and the same inputs `enrich_one_product` uses. Writing the answer
down costs one UPDATE; computing it a second time somewhere else would be a
second chance to compute it differently.
WHY IT MATTERS: `GET /api/nutrition/health-scores` filters on
`nutrition_facts.edibility`, and every row written before that column
existed carries NULL, which the endpoint treats as unconfirmed and excludes.
Without this the endpoint returns nothing until a full re-enrichment run.
Orphans are deliberately left NULL: they have no catalogue row, so there is
no category to classify them from, and guessing is the thing this whole
script exists to stop.
"""
# Idempotent, and the same statement `ensure_nutrition_schema()` runs. Here
# so the script works against a database the API has not started against
# yet, rather than failing on an unknown column.
cur.execute("ALTER TABLE nutrition_facts ADD COLUMN IF NOT EXISTS edibility TEXT")
written: Dict[str, int] = {}
for label, rows in (("consumable", plan.keep), ("unknown", plan.undecided)):
keys = [(r["brand"], r["image_id"]) for r in rows]
if not keys:
written[label] = 0
continue
cur.execute(
"UPDATE nutrition_facts SET edibility = %s WHERE (brand, image_id) IN %s",
(label, tuple(keys)),
)
written[label] = cur.rowcount
return written
def restore(cur, conn, path: Path, apply: bool) -> int:
payload = json.loads(path.read_text(encoding="utf-8"))
if payload.get("db_host") != DB_HOST:
logger.error("Backup was taken from %s but DB_HOST is %s. Refusing.",
payload.get("db_host"), DB_HOST)
return 1
total = 0
for table, rows in payload["tables"].items():
for row in rows:
total += 1
if not apply:
continue
cols = list(row)
cur.execute(
f"INSERT INTO {table} ({', '.join(cols)}) "
f"VALUES ({', '.join('%s' for _ in cols)}) ON CONFLICT DO NOTHING",
[row[c] for c in cols],
)
if apply:
conn.commit()
logger.info("%s %d row(s) from %s", "Restored" if apply else "Would restore",
total, path.name)
return 0
# ---------------------------------------------------------------------------
def main() -> int:
parser = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
mode = parser.add_mutually_exclusive_group(required=True)
mode.add_argument("--audit", action="store_true",
help="read-only report; writes nothing")
mode.add_argument("--purge", action="store_true",
help="delete non-consumable nutrition rows")
mode.add_argument("--restore", metavar="PATH", help="restore a backup file")
parser.add_argument("--apply", action="store_true",
help="commit the changes (default is a dry run)")
parser.add_argument("--dry-run", action="store_true",
help="explicit no-op; this is already the default")
parser.add_argument("--json", action="store_true", help="audit output as JSON")
args = parser.parse_args()
apply = args.apply and not args.dry_run
if args.audit and args.apply:
parser.error("--audit is read-only; it cannot be combined with --apply")
conn = _connect()
if conn is None:
logger.error("No database connection.")
return 1
try:
with conn.cursor() as cur:
if args.restore:
return restore(cur, conn, Path(args.restore), apply)
plan = plan_purge(_nutrition_rows(cur), _catalogue_index(cur))
summary = _report(plan, args.json)
if args.audit:
return 0
logger.info("")
logger.info("Mode: %s", "APPLY - this writes" if apply
else "DRY RUN - nothing is written")
labelled = len(plan.keep) + len(plan.undecided)
if not apply:
logger.info("Re-run with --apply to delete %d row(s) and label %d.",
len(plan.delete), labelled)
return 0
# Labelling runs first and unconditionally: it is additive, it is
# correct even when there is nothing to delete, and it is what makes
# the health-score endpoint able to tell food from not-food in SQL.
written = _backfill_edibility(cur, plan)
conn.commit()
logger.info("")
for label, count in written.items():
logger.info(" labelled %5d row(s) edibility=%s", count, label)
if not plan.delete:
logger.info("")
logger.info("Nothing to delete.")
return 0
path = _backup(cur, plan.delete_keys)
logger.info("Backup written: %s", path)
deleted = _delete(cur, plan.delete_keys)
conn.commit()
logger.info("")
for table, count in deleted.items():
logger.info(" deleted %5d row(s) from %s", count, table)
logger.info("")
logger.info("The similarity and clustering models were fitted with these "
"rows still in them.")
logger.info("Retrain before trusting /similar or /alternatives:")
logger.info(" python -m scripts.train_nutrition_models")
_ = summary
finally:
conn.close()
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,207 @@
#!/usr/bin/env python3
"""
Rebuild the pinned USDA nutrient snapshot from the open bulk download.
WHY THE BULK DOWNLOAD AND NOT THE API
-------------------------------------
The FoodData Central API is the obvious source and it cannot do this job. Its
unauthenticated DEMO_KEY allows TEN requests per hour - measured, with a
Retry-After of thirteen hours once exhausted - and the curated table references
101 distinct foods. Even a personal key would make rebuilding the snapshot a
rate-limit exercise.
The same data is published as a plain zip with no key and no quota:
FoodData_Central_foundation_food_json_*.zip ~0.5 MB
FoodData_Central_sr_legacy_food_json_*.zip ~12 MB
So the snapshot is built from those, and `USDA_FDC_API_KEY` stays optional -
needed only to look up an id the snapshot does not already hold.
WHAT IT STORES, AND WHY RAW
---------------------------
The `foodNutrients` list exactly as USDA publishes it, trimmed to the fields
this codebase reads. NOT pre-mapped to our columns: if the snapshot held
finished per-100g values, the offline tests would exercise a different code
path from the online one, and the unit-checking rules in
`nutrition_usda_service` - the ones that stop a sodium figure being silently
multiplied by a thousand - would go untested.
USAGE
-----
python -m scripts.refresh_usda_snapshot # dry run, shows the diff
python -m scripts.refresh_usda_snapshot --apply
python -m scripts.refresh_usda_snapshot --apply --keep-download
A dry run reports every nutrient whose value moved, so a USDA revision is read
and accepted rather than silently absorbed.
"""
from __future__ import annotations
import argparse
import json
import logging
import shutil
import sys
import tempfile
import zipfile
from pathlib import Path
from typing import Any, Dict, List
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
import requests # noqa: E402
from app.services import produce_reference # noqa: E402
logging.basicConfig(level=logging.INFO, format="%(message)s")
logger = logging.getLogger("refresh_usda_snapshot")
SNAPSHOT = Path(__file__).resolve().parents[1] / "app" / "services" / "data" / "usda_snapshot.json"
# Pinned filenames rather than "latest": a dataset that silently became a newer
# release between two runs would move numbers with no diff to read.
DATASETS = (
("FoundationFoods",
"https://fdc.nal.usda.gov/fdc-datasets/FoodData_Central_foundation_food_json_2025-04-24.zip"),
("SRLegacyFoods",
"https://fdc.nal.usda.gov/fdc-datasets/FoodData_Central_sr_legacy_food_json_2021-10-28.zip"),
)
def _download(url: str, into: Path) -> Path:
name = url.rsplit("/", 1)[-1]
target = into / name
if target.exists():
return target
logger.info(" downloading %s", name)
with requests.get(url, stream=True, timeout=300) as resp:
resp.raise_for_status()
with target.open("wb") as fh:
for chunk in resp.iter_content(1 << 20):
fh.write(chunk)
return target
def _load(zip_path: Path, key: str) -> List[Dict[str, Any]]:
with zipfile.ZipFile(zip_path) as zf:
name = zf.namelist()[0]
with zf.open(name) as fh:
return json.load(fh)[key]
def _trim(food: Dict[str, Any]) -> Dict[str, Any]:
nutrients = []
for fn in food.get("foodNutrients", []):
n = fn.get("nutrient") or {}
if fn.get("amount") is None or not n.get("id"):
continue
nutrients.append({"id": n["id"], "name": n.get("name"),
"unit": n.get("unitName"), "amount": fn["amount"]})
portions = []
for p in (food.get("foodPortions") or [])[:3]:
grams = p.get("gramWeight")
if not grams:
continue
portions.append({
"gram_weight": grams,
"description": (p.get("portionDescription")
or (p.get("measureUnit") or {}).get("name") or ""),
})
return {"fdc_id": food["fdcId"], "description": food.get("description"),
"data_type": food.get("dataType"),
"publication_date": food.get("publicationDate"),
"nutrients": nutrients, "portions": portions}
def _diff(old: Dict[str, Any], new: Dict[str, Any]) -> List[str]:
lines: List[str] = []
for fid, food in sorted(new.items()):
before = old.get(fid)
if not before:
lines.append(f" NEW {fid} {food['description'][:56]}")
continue
if before.get("description") != food.get("description"):
lines.append(f" RENAME {fid} {before['description'][:34]} -> "
f"{food['description'][:34]}")
old_n = {n["id"]: n["amount"] for n in before.get("nutrients", [])}
new_n = {n["id"]: n["amount"] for n in food.get("nutrients", [])}
for nid, amount in sorted(new_n.items()):
if nid in old_n and old_n[nid] != amount:
lines.append(f" VALUE {fid} nutrient {nid}: "
f"{old_n[nid]} -> {amount} ({food['description'][:30]})")
for nid in sorted(set(old_n) - set(new_n)):
lines.append(f" GONE {fid} nutrient {nid} ({food['description'][:30]})")
for fid in sorted(set(old) - set(new)):
lines.append(f" LOST {fid} {old[fid].get('description', '')[:56]}")
return lines
def main() -> int:
parser = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--apply", action="store_true",
help="write the snapshot (default is a dry run)")
parser.add_argument("--keep-download", action="store_true",
help="keep the downloaded zips instead of using a temp dir")
args = parser.parse_args()
wanted = {e["usda_fdc_id"] for e in produce_reference.all_entries().values()
if e.get("usda_fdc_id")}
logger.info("%d FDC id(s) referenced by the curated table", len(wanted))
workdir = Path("usda_bulk") if args.keep_download else Path(tempfile.mkdtemp())
workdir.mkdir(parents=True, exist_ok=True)
snapshot: Dict[str, Any] = {}
try:
for key, url in DATASETS:
path = _download(url, workdir)
for food in _load(path, key):
fid = food.get("fdcId")
if fid in wanted and str(fid) not in snapshot:
snapshot[str(fid)] = _trim(food)
finally:
if not args.keep_download:
shutil.rmtree(workdir, ignore_errors=True)
missing = sorted(wanted - {int(k) for k in snapshot})
if missing:
logger.warning("NOT FOUND in either dataset: %s", missing)
old = {}
if SNAPSHOT.exists():
old = json.loads(SNAPSHOT.read_text(encoding="utf-8")).get("foods", {})
changes = _diff(old, snapshot)
if changes:
logger.info("")
logger.info("%d change(s) against the current snapshot:", len(changes))
for line in changes[:60]:
logger.info("%s", line)
if len(changes) > 60:
logger.info(" ... and %d more", len(changes) - 60)
else:
logger.info("No change against the current snapshot.")
if not args.apply:
logger.info("")
logger.info("Dry run - nothing written. Re-run with --apply.")
return 0
SNAPSHOT.parent.mkdir(parents=True, exist_ok=True)
SNAPSHOT.write_text(json.dumps(
{"source": "USDA FoodData Central bulk download",
"datasets": [u.rsplit("/", 1)[-1].replace(".zip", "") for _, u in DATASETS],
"foods": snapshot},
indent=1, sort_keys=True), encoding="utf-8")
logger.info("")
logger.info("Wrote %s (%d food(s), %.1f MB)", SNAPSHOT, len(snapshot),
SNAPSHOT.stat().st_size / 1e6)
logger.info("Run the test suite: tests/test_nutrition_usda.py asserts every "
"curated id is pinned and every description still matches.")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -75,6 +75,7 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.infrastructure.settings import DB_HOST, DB_NAME # noqa: E402
from app.services.brand_registry import resolve_parent_brand # noqa: E402
from app.services.generic_products import OWN_PRODUCTS_BRAND # noqa: E402
from app.services.vector_store import ( # noqa: E402
_connect,
_list_brand_table_suffixes,
@@ -358,8 +359,18 @@ _OFF_CACHE: Dict[str, bool] = {}
_OFF_BARCODE = re.compile(r"/images/products/((?:\d+/)+)")
# "own" and "products" are the BUCKET's name, not a brand's. Left in, the token
# "products" corroborates any URL containing /images/products/ or
# /cdn/shop/products/ - which is every Open*Facts and every Shopify path - so
# `_names_product` waved through 40+ images that named nothing about the item.
# That is how the openbeautyfacts cosmetics photos became the stored image for
# Banana, Orange, Papaya, Guava, Lemon and twenty more.
_BUCKET_TOKENS = frozenset({"own", "products", "product"})
def _brand_tokens(brand: str) -> List[str]:
return [w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) if len(w) > 2]
return [w for w in re.split(r"[^a-z0-9]+", (brand or "").lower())
if len(w) > 2 and w not in _BUCKET_TOKENS]
def _off_product_matches_brand(url: str, brand: str) -> bool:
@@ -420,6 +431,15 @@ def _search_key(product_name: str, brand: str) -> Tuple[str, str]:
return (brand.lower(), (base or product_name or "").lower())
def _is_own_products(brand: str) -> bool:
"""True for the unbranded-commodity bucket, under either spelling.
`repair()` passes the TABLE SUFFIX ("own_products"), while the pipeline and
the UI use the display name ("Own Products").
"""
return _sanitize_name(brand or "") == _sanitize_name(OWN_PRODUCTS_BRAND)
def _search_replacement(product_name: str, brand: str, max_results: int) -> List[str]:
"""Validated image URLs for one product, best first.
@@ -433,7 +453,17 @@ def _search_replacement(product_name: str, brand: str, max_results: int) -> List
return _SEARCH_CACHE[key]
from app.services.image_search import find_all_image_urls
urls = find_all_image_urls(product_name, brand=brand, validate=True, max_results=max_results)
# `stage_6_images` already blanks the brand for this bucket
# (store_catalog_pipeline.py) and this path never did, so a repair searched
# for "own_products Apple fruit juice" and got exactly what it asked for.
# Produce mode additionally skips the packaged-goods databases and the
# commodity-hint table - see image_search._AMBIQUITY_HINTS.
produce = _is_own_products(brand)
urls = find_all_image_urls(
product_name, brand="" if produce else brand,
validate=True, max_results=max_results, produce=produce,
)
if not urls:
_SEARCH_CACHE[key] = []
return []
@@ -451,14 +481,15 @@ def _search_replacement(product_name: str, brand: str, max_results: int) -> List
# image would have been written as the primary image of four MTR products.
# Let an error here raise: unranked output is not safe to store.
from app.core.catalog_engine import ProductCatalogEngine
rank_brand = "" if produce else brand
urls = ProductCatalogEngine._select_best_images(
ProductCatalogEngine, urls, product_name, brand, max_images=MAX_STORED_IMAGES
ProductCatalogEngine, urls, product_name, rank_brand, max_images=MAX_STORED_IMAGES
)
if not urls:
_SEARCH_CACHE[key] = []
return []
# Drop other brands' products before anything else looks at them.
urls = [u for u in urls if _off_product_matches_brand(u, brand)]
urls = [u for u in urls if produce or _off_product_matches_brand(u, brand)]
if not urls:
_SEARCH_CACHE[key] = []
return []
@@ -480,7 +511,7 @@ def _search_replacement(product_name: str, brand: str, max_results: int) -> List
# (Zepto, Flipkart, Pinterest). That is the intended trade - 94% of rows
# still find a corroborated URL, and the rest are reported for hand
# curation rather than filled with a guess.
ranked = [u for u in urls if _names_product(u, product_name, brand)]
ranked = [u for u in urls if _names_product(u, product_name, rank_brand)]
ranked.sort(key=lambda u: 0 if u.startswith("https://") else 1)
_SEARCH_CACHE[key] = ranked
return ranked