Health score updates in backend
This commit is contained in:
428
scripts/build_produce_contact_sheet.py
Normal file
428
scripts/build_produce_contact_sheet.py
Normal file
@@ -0,0 +1,428 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Render every Own Products row as a before/after card, for review before writing.
|
||||
|
||||
WHY A REVIEW STEP AND NOT A CONFIDENCE THRESHOLD
|
||||
------------------------------------------------
|
||||
No automated filter can tell that `Zunaid_Ahmed_Palak.jpg` is a photograph of a
|
||||
politician rather than a bunch of spinach, or that `Bitter_Gourd_Curry.jpg` is a
|
||||
cooked dish rather than the vegetable. Both were the stored image for a live
|
||||
product, both resolved to real image bytes, and both passed every check the
|
||||
pipeline had. The only thing that catches them is a person looking.
|
||||
|
||||
So this script writes nothing to the database. It renders what WOULD be written
|
||||
and stops. `repair_brand_images.py --brands "Own Products" --apply` is the step
|
||||
that writes, and it should only be run once these cards have been looked at.
|
||||
|
||||
WHAT EACH CARD SHOWS
|
||||
--------------------
|
||||
* the image currently stored, and whether it still resolves
|
||||
* the curated replacement from `produce_reference`
|
||||
* where that replacement came from, and for the 15 hand-picked ones, why
|
||||
* the nutrition that will be attached, with its USDA source, or an explicit
|
||||
note that there is none and why
|
||||
|
||||
USAGE
|
||||
-----
|
||||
python -m scripts.build_produce_contact_sheet
|
||||
python -m scripts.build_produce_contact_sheet --out sheet.html --no-probe
|
||||
|
||||
`--no-probe` skips checking whether the stored URLs still resolve, which is the
|
||||
slow part; without it the script makes one ranged GET per distinct stored URL.
|
||||
`--embed` downloads every image and inlines it as a small data: URI, which is
|
||||
required if the page is going to be published as an Artifact - that viewer's
|
||||
content security policy blocks third-party image hosts, so a page of hotlinked
|
||||
photographs renders as 318 empty boxes.
|
||||
Read-only in every mode: it opens no write transaction.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import html
|
||||
import logging
|
||||
import sys
|
||||
import time
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from app.services import nutrition_scoring, produce_reference # noqa: E402
|
||||
from app.services.consumability import classify_edibility # noqa: E402
|
||||
from app.services.generic_products import OWN_PRODUCTS_BRAND # noqa: E402
|
||||
from app.services.nutrition_usda_service import ( # noqa: E402
|
||||
fetch_verified_nutrition_usda,
|
||||
)
|
||||
from app.services.vector_store import _connect, _sanitize_name # noqa: E402
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
||||
logger = logging.getLogger("contact_sheet")
|
||||
|
||||
TABLE = f"brand_{_sanitize_name(OWN_PRODUCTS_BRAND)}"
|
||||
|
||||
|
||||
def _rows() -> List[Dict[str, Any]]:
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
logger.error("No database connection.")
|
||||
return []
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
cur.execute(
|
||||
f"SELECT product_name, category, image_url FROM {TABLE} "
|
||||
f"ORDER BY category, product_name"
|
||||
)
|
||||
return [{"product_name": r[0], "category": r[1], "image_url": r[2]}
|
||||
for r in cur.fetchall()]
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
# Thumbnails, not photographs. 318 full-size images would be ~40 MB and the
|
||||
# Artifact ceiling is 16 MB; at 168px they are legible side by side, which is
|
||||
# all this page asks of them, and the whole set fits in about 2 MB.
|
||||
_THUMB_PX = 168
|
||||
_THUMB_QUALITY = 62
|
||||
|
||||
|
||||
def _thumb(url: str, cache: Dict[str, str]) -> str:
|
||||
"""`url` fetched, downscaled and returned as a data: URI, or "" on failure."""
|
||||
if not url:
|
||||
return ""
|
||||
if url in cache:
|
||||
return cache[url]
|
||||
|
||||
import base64
|
||||
from io import BytesIO
|
||||
import requests
|
||||
from PIL import Image
|
||||
|
||||
data = ""
|
||||
try:
|
||||
resp = requests.get(url, headers={"User-Agent": _UA}, timeout=20)
|
||||
if resp.status_code == 200 and resp.content:
|
||||
img = Image.open(BytesIO(resp.content))
|
||||
img = img.convert("RGB")
|
||||
img.thumbnail((_THUMB_PX, _THUMB_PX), Image.LANCZOS)
|
||||
buf = BytesIO()
|
||||
img.save(buf, format="JPEG", quality=_THUMB_QUALITY, optimize=True)
|
||||
data = "data:image/jpeg;base64," + base64.b64encode(buf.getvalue()).decode()
|
||||
except Exception: # noqa: BLE001 - a broken image is the thing being reported
|
||||
data = ""
|
||||
cache[url] = data
|
||||
time.sleep(0.12)
|
||||
return data
|
||||
|
||||
|
||||
_UA = "nearle-catalogue/1.0 (produce image review sheet)"
|
||||
|
||||
|
||||
def _probe(url: str, cache: Dict[str, bool]) -> bool:
|
||||
if not url:
|
||||
return False
|
||||
if url in cache:
|
||||
return cache[url]
|
||||
from app.services.image_search import validate_image_url_live
|
||||
try:
|
||||
ok = bool(validate_image_url_live(url, timeout=8))
|
||||
except Exception: # noqa: BLE001
|
||||
ok = False
|
||||
cache[url] = ok
|
||||
time.sleep(0.1)
|
||||
return ok
|
||||
|
||||
|
||||
_CSS = """
|
||||
/* Palette is the greengrocer's, not the default warm-cream: a cool paper with a
|
||||
faint green cast, deep leaf-black ink, and aubergine as the one accent - kept
|
||||
clear of the green/amber/red that carry MEANING on this page. */
|
||||
:root{
|
||||
--paper:#eef1ec; --card:#ffffff; --line:#d8dfd6; --line-soft:#e8ede6;
|
||||
--ink:#131c17; --ink-2:#3d4b43; --dim:#6a7a70;
|
||||
--accent:#6b3a58; /* aubergine */
|
||||
--ok:#1a6b45; --ok-bg:#e3f1e9;
|
||||
--warn:#8a5600; --warn-bg:#fbeed7;
|
||||
--crit:#9c2b2b; --crit-bg:#fbe6e6;
|
||||
--shadow:0 1px 2px rgba(19,28,23,.05);
|
||||
}
|
||||
@media (prefers-color-scheme:dark){
|
||||
:root:not([data-theme="light"]){
|
||||
--paper:#11150f; --card:#1a1f19; --line:#2e352c; --line-soft:#232a22;
|
||||
--ink:#e8ece6; --ink-2:#b9c3ba; --dim:#8b968c;
|
||||
--accent:#d99bc4;
|
||||
--ok:#71d3a1; --ok-bg:#12301f;
|
||||
--warn:#e8b761; --warn-bg:#33260d;
|
||||
--crit:#f09393; --crit-bg:#331616;
|
||||
--shadow:0 1px 2px rgba(0,0,0,.35);
|
||||
}
|
||||
}
|
||||
:root[data-theme="dark"]{
|
||||
--paper:#11150f; --card:#1a1f19; --line:#2e352c; --line-soft:#232a22;
|
||||
--ink:#e8ece6; --ink-2:#b9c3ba; --dim:#8b968c;
|
||||
--accent:#d99bc4;
|
||||
--ok:#71d3a1; --ok-bg:#12301f;
|
||||
--warn:#e8b761; --warn-bg:#33260d;
|
||||
--crit:#f09393; --crit-bg:#331616;
|
||||
--shadow:0 1px 2px rgba(0,0,0,.35);
|
||||
}
|
||||
|
||||
*{box-sizing:border-box}
|
||||
body{
|
||||
margin:0; background:var(--paper); color:var(--ink);
|
||||
font:400 15px/1.55 "IBM Plex Sans","Segoe UI",system-ui,-apple-system,sans-serif;
|
||||
-webkit-font-smoothing:antialiased;
|
||||
}
|
||||
.wrap{max-width:1500px;margin:0 auto}
|
||||
|
||||
header{padding:38px 26px 26px;border-bottom:1px solid var(--line)}
|
||||
h1{
|
||||
margin:0 0 8px; font-family:"Bricolage Grotesque","IBM Plex Sans",sans-serif;
|
||||
font-weight:600; font-size:clamp(26px,3.4vw,38px); letter-spacing:-.025em;
|
||||
text-wrap:balance; line-height:1.08;
|
||||
}
|
||||
.sub{color:var(--ink-2);font-size:14.5px;max-width:66ch;margin:0}
|
||||
.sub strong{color:var(--ink);font-weight:600}
|
||||
|
||||
.tally{display:flex;flex-wrap:wrap;gap:0;margin-top:22px;
|
||||
border:1px solid var(--line);border-radius:10px;overflow:hidden;background:var(--card)}
|
||||
.tally .t{padding:11px 18px;border-right:1px solid var(--line);flex:1 1 auto;min-width:132px}
|
||||
.tally .t:last-child{border-right:0}
|
||||
.tally b{display:block;font-family:"IBM Plex Mono",ui-monospace,monospace;
|
||||
font-size:23px;font-weight:600;letter-spacing:-.02em;font-variant-numeric:tabular-nums}
|
||||
.tally span{font-size:11.5px;color:var(--dim);text-transform:uppercase;letter-spacing:.07em}
|
||||
.tally .t.crit b{color:var(--crit)} .tally .t.ok b{color:var(--ok)}
|
||||
|
||||
h2{
|
||||
margin:38px 26px 13px; font-size:12.5px; font-weight:600; color:var(--dim);
|
||||
text-transform:uppercase; letter-spacing:.13em;
|
||||
display:flex; align-items:baseline; gap:10px;
|
||||
}
|
||||
h2::after{content:"";flex:1;height:1px;background:var(--line)}
|
||||
h2 em{font-style:normal;font-family:"IBM Plex Mono",monospace;color:var(--dim);
|
||||
font-size:12px;font-variant-numeric:tabular-nums}
|
||||
|
||||
.grid{display:grid;grid-template-columns:repeat(auto-fill,minmax(322px,1fr));
|
||||
gap:13px;padding:0 26px}
|
||||
|
||||
/* The left rail encodes the row's actual state, so severity reads before any
|
||||
text does: dead stored image, wrong-source image, or already fine. */
|
||||
.card{background:var(--card);border:1px solid var(--line);border-radius:11px;
|
||||
overflow:hidden;box-shadow:var(--shadow);border-left:3px solid var(--line)}
|
||||
.card.crit{border-left-color:var(--crit)}
|
||||
.card.warn{border-left-color:var(--warn)}
|
||||
.card.ok{border-left-color:var(--ok)}
|
||||
|
||||
.name{padding:12px 14px 10px;font-weight:600;font-size:15.5px;letter-spacing:-.01em}
|
||||
.pair{display:grid;grid-template-columns:1fr 1fr;gap:1px;background:var(--line-soft)}
|
||||
.cell{background:var(--card);padding:10px}
|
||||
.lbl{font-size:10px;text-transform:uppercase;letter-spacing:.09em;color:var(--dim);
|
||||
margin-bottom:7px;display:flex;align-items:center;gap:6px;min-height:16px}
|
||||
.ph{width:100%;aspect-ratio:1;object-fit:cover;border-radius:7px;
|
||||
background:var(--line-soft);display:block}
|
||||
.miss{width:100%;aspect-ratio:1;border-radius:7px;background:var(--line-soft);
|
||||
display:flex;align-items:center;justify-content:center;color:var(--dim);
|
||||
font-size:11.5px;text-align:center;padding:10px;line-height:1.4}
|
||||
|
||||
.meta{padding:11px 14px 13px;font-size:12.5px;color:var(--ink-2);
|
||||
border-top:1px solid var(--line-soft);display:flex;flex-direction:column;gap:5px}
|
||||
.tag{display:inline-block;border-radius:5px;padding:2px 7px;font-size:10.5px;
|
||||
font-weight:600;text-transform:uppercase;letter-spacing:.05em;white-space:nowrap}
|
||||
.t-ok{background:var(--ok-bg);color:var(--ok)}
|
||||
.t-crit{background:var(--crit-bg);color:var(--crit)}
|
||||
.t-warn{background:var(--warn-bg);color:var(--warn)}
|
||||
.nut{font-family:"IBM Plex Mono",ui-monospace,monospace;font-size:12px;
|
||||
font-variant-numeric:tabular-nums;display:flex;align-items:center;gap:7px;flex-wrap:wrap}
|
||||
code{font-family:"IBM Plex Mono",ui-monospace,monospace;font-size:11px;
|
||||
background:var(--line-soft);padding:1.5px 5px;border-radius:4px;word-break:break-all;
|
||||
color:var(--ink-2)}
|
||||
|
||||
footer{margin-top:46px;padding:24px 26px 40px;border-top:1px solid var(--line);
|
||||
color:var(--dim);font-size:13px;line-height:1.6}
|
||||
footer b{color:var(--ink-2);font-weight:600}
|
||||
a{color:var(--accent)}
|
||||
@media (prefers-reduced-motion:reduce){*{animation:none!important;transition:none!important}}
|
||||
"""
|
||||
|
||||
|
||||
def _card(row: Dict[str, Any], probe_cache: Dict[str, bool], probe: bool,
|
||||
thumbs: Dict[str, str] | None = None) -> str:
|
||||
name = row["product_name"] or ""
|
||||
entry = produce_reference.lookup(name) or {}
|
||||
verdict = classify_edibility(row["category"], name)
|
||||
consumable = verdict.edibility.value == "consumable"
|
||||
|
||||
stored = row.get("image_url") or ""
|
||||
stored_ok = _probe(stored, probe_cache) if (probe and stored) else None
|
||||
proposed = entry.get("image_url")
|
||||
|
||||
# The stored URL's HOST is itself a verdict. Open Beauty Facts is a
|
||||
# cosmetics database and Open Products Facts a packaged-goods one; a raw
|
||||
# banana photographed front-of-pack is the wrong SUBJECT even when the URL
|
||||
# resolves perfectly, which no liveness probe can detect.
|
||||
wrong_source = any(h in stored.lower() for h in
|
||||
("openbeautyfacts", "openproductsfacts", "openfoodfacts"))
|
||||
|
||||
if thumbs is not None and stored and not _thumb(stored, thumbs):
|
||||
stored_ok = False
|
||||
if not stored or stored_ok is False:
|
||||
severity, verdict_tag = "crit", '<span class="tag t-crit">dead</span>'
|
||||
elif wrong_source:
|
||||
severity, verdict_tag = "warn", '<span class="tag t-warn">wrong source</span>'
|
||||
elif stored_ok:
|
||||
severity, verdict_tag = "ok", '<span class="tag t-ok">resolves</span>'
|
||||
else:
|
||||
severity, verdict_tag = "", ""
|
||||
|
||||
def _img(url: str, empty: str) -> str:
|
||||
if not url:
|
||||
return f'<div class="miss">{empty}</div>'
|
||||
src = _thumb(url, thumbs) if thumbs is not None else url
|
||||
if not src:
|
||||
return '<div class="miss">image did not load</div>'
|
||||
return f'<img class="ph" src="{html.escape(src)}" loading="lazy" alt="">'
|
||||
|
||||
left = _img(stored, "no image stored")
|
||||
right = _img(proposed, "no curated image")
|
||||
|
||||
bits = []
|
||||
if entry.get("image_source") == "wikimedia_commons":
|
||||
bits.append('<div><span class="tag t-warn">hand-picked</span> '
|
||||
"article lead image was a botanical plate</div>")
|
||||
if entry.get("image_credit"):
|
||||
bits.append(f'<div><code>{html.escape(str(entry["image_credit"]))}</code></div>')
|
||||
|
||||
if not consumable:
|
||||
bits.append('<div><span class="tag t-warn">not food</span> '
|
||||
"no nutrition, by design</div>")
|
||||
elif entry.get("usda_fdc_id"):
|
||||
facts = fetch_verified_nutrition_usda(name, row["category"])
|
||||
scores = nutrition_scoring.compute_scores(facts) or {}
|
||||
if facts.get("data_status") != "unavailable":
|
||||
bits.append(
|
||||
f'<div class="nut"><span class="tag t-ok">health '
|
||||
f'{scores.get("health_score", "-")}</span>'
|
||||
f'<span>{facts.get("calories_kcal")} kcal</span>'
|
||||
f'<span>{facts.get("protein_g")} g protein</span>'
|
||||
f'<span>{facts.get("dietary_fiber_g")} g fibre</span></div>')
|
||||
bits.append(f'<div>USDA <code>{facts["source_ref"]}</code> '
|
||||
f'{html.escape(str(entry.get("usda_description", ""))[:52])}</div>')
|
||||
else:
|
||||
bits.append('<div><span class="tag t-warn">no nutrition</span> '
|
||||
"USDA has no entry for this species</div>")
|
||||
|
||||
return f"""<article class="card {severity}">
|
||||
<div class="name">{html.escape(name)}</div>
|
||||
<div class="pair">
|
||||
<div class="cell"><div class="lbl">stored now {verdict_tag}</div>{left}</div>
|
||||
<div class="cell"><div class="lbl">proposed</div>{right}</div>
|
||||
</div>
|
||||
<div class="meta">{"".join(bits)}</div>
|
||||
</article>"""
|
||||
|
||||
|
||||
def build(rows: List[Dict[str, Any]], probe: bool, embed: bool = False) -> str:
|
||||
cache: Dict[str, bool] = {}
|
||||
thumbs: Dict[str, str] | None = {} if embed else None
|
||||
by_category: Dict[str, List[Dict[str, Any]]] = {}
|
||||
for row in rows:
|
||||
by_category.setdefault(row["category"] or "(uncategorised)", []).append(row)
|
||||
|
||||
sections = []
|
||||
for category in sorted(by_category):
|
||||
cards = [_card(row, cache, probe, thumbs) for row in by_category[category]]
|
||||
logger.info(" %-26s %3d card(s)", category, len(by_category[category]))
|
||||
sections.append(
|
||||
f'<h2>{html.escape(category)}<em>{len(by_category[category])}</em></h2>'
|
||||
f'<div class="grid">{"".join(cards)}</div>')
|
||||
|
||||
n_dead = n_wrong = n_food = n_scored = 0
|
||||
for row in rows:
|
||||
entry = produce_reference.lookup(row["product_name"] or "") or {}
|
||||
stored = row.get("image_url") or ""
|
||||
if classify_edibility(row["category"], row["product_name"] or "") \
|
||||
.edibility.value == "consumable":
|
||||
n_food += 1
|
||||
if entry.get("usda_fdc_id"):
|
||||
n_scored += 1
|
||||
if not stored:
|
||||
n_dead += 1
|
||||
elif thumbs is not None and not thumbs.get(stored):
|
||||
# --embed already fetched every URL; a thumbnail that came back
|
||||
# empty is the same evidence a liveness probe would have gathered.
|
||||
n_dead += 1
|
||||
elif probe and not cache.get(stored):
|
||||
n_dead += 1
|
||||
elif any(h in stored.lower() for h in
|
||||
("openbeautyfacts", "openproductsfacts", "openfoodfacts")):
|
||||
n_wrong += 1
|
||||
|
||||
tally = [("t crit", n_dead, "dead images"),
|
||||
("t crit", n_wrong, "wrong subject"),
|
||||
("t ok", n_scored, "get a health score"),
|
||||
("t", len(rows) - n_food, "not food, unscored"),
|
||||
("t", len(rows), "products in all")]
|
||||
tiles = "".join(f'<div class="{c}"><b>{v}</b><span>{lbl}</span></div>'
|
||||
for c, v, lbl in tally)
|
||||
|
||||
return f"""<title>Fresh Aisle Audit</title>
|
||||
<link rel="preconnect" href="https://fonts.googleapis.com">
|
||||
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
||||
<link rel="stylesheet" href="https://fonts.googleapis.com/css2?\
|
||||
family=Bricolage+Grotesque:opsz,wght@12..96,500;12..96,600&\
|
||||
family=IBM+Plex+Mono:wght@400;600&\
|
||||
family=IBM+Plex+Sans:wght@400;500;600&display=swap">
|
||||
<style>{_CSS}</style>
|
||||
<div class="wrap">
|
||||
<header>
|
||||
<h1>Every loose thing the shop sells, and the picture it currently shows</h1>
|
||||
<p class="sub">The {len(rows)} unbranded rows in <code>brand_own_products</code>,
|
||||
each showing what is stored today beside what would replace it.
|
||||
<strong>Nothing has been written to the database.</strong> The repair script
|
||||
runs only after these have been looked at — because no automated check
|
||||
can tell that the stored photo for <em>Palak</em> is a politician, or that
|
||||
<em>Bitter Gourd</em> is a plate of curry.</p>
|
||||
<div class="tally">{tiles}</div>
|
||||
</header>
|
||||
{"".join(sections)}
|
||||
<footer>
|
||||
<b>Images</b> — English Wikipedia article lead images, resolved by
|
||||
botanical name wherever the common name is also a personal name, plus 15
|
||||
hand-picked from Wikimedia Commons where the article's own lead image was a
|
||||
19th-century botanical plate rather than a photograph.<br>
|
||||
<b>Nutrition</b> — USDA FoodData Central, Foundation Foods and SR
|
||||
Legacy, pinned from the open bulk download. Every value is per 100 g of
|
||||
edible portion and links back to its FDC record; commodities USDA has no
|
||||
entry for get no numbers rather than a near neighbour's.<br>
|
||||
Generated {datetime.now():%d %B %Y, %H:%M}.
|
||||
</footer>
|
||||
</div>"""
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(
|
||||
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
parser.add_argument("--out", default="produce_contact_sheet.html")
|
||||
parser.add_argument("--no-probe", action="store_true",
|
||||
help="skip checking whether the stored URLs still resolve")
|
||||
parser.add_argument("--embed", action="store_true",
|
||||
help="inline every image as a downscaled data: URI. Required "
|
||||
"for publishing as an Artifact, whose CSP blocks "
|
||||
"third-party image hosts.")
|
||||
args = parser.parse_args()
|
||||
|
||||
rows = _rows()
|
||||
if not rows:
|
||||
logger.error("No rows read from %s", TABLE)
|
||||
return 1
|
||||
logger.info("Rendering %d rows from %s", len(rows), TABLE)
|
||||
|
||||
page = build(rows, probe=not args.no_probe, embed=args.embed)
|
||||
Path(args.out).write_text(page, encoding="utf-8")
|
||||
logger.info("Wrote %s (%.1f MB)", args.out, len(page.encode("utf-8")) / 1e6)
|
||||
logger.info("Nothing was written to the database.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -8,6 +8,16 @@ Usage:
|
||||
python scripts/enrich_nutrition.py --max-products 200 --no-narrative
|
||||
python scripts/enrich_nutrition.py --force # re-fetch even already-verified products
|
||||
|
||||
# Catalogue-wide, past ACTIVE_BRANDS. Without this the run cannot reach
|
||||
# Britannia, Parle or ITC at all, because enrich_all_products reads through
|
||||
# vector_store.list_available_brands(), which honours that filter.
|
||||
python scripts/enrich_nutrition.py --include-inactive
|
||||
python scripts/enrich_nutrition.py --brands "Britannia,Parle"
|
||||
python scripts/enrich_nutrition.py --categories "Fruits & Vegetables,Eggs"
|
||||
|
||||
Non-consumables (soap, shampoo, mosquito repellent, flowers) are skipped and
|
||||
reported separately - see app/services/consumability.py.
|
||||
|
||||
Equivalent to POST /api/admin/nutrition-intelligence/enrich.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
@@ -36,6 +46,13 @@ def main() -> None:
|
||||
parser.add_argument("--force", action="store_true", help="Re-fetch even products already marked 'verified'")
|
||||
parser.add_argument("--no-narrative", action="store_true", help="Skip the LLM narrative step (facts/scores/tags only)")
|
||||
parser.add_argument("--max-products", type=int, default=None, help="Cap the number of products processed (for a quick test run)")
|
||||
parser.add_argument("--brands", default=None,
|
||||
help="Comma-separated brand display names to enrich (default: whatever ACTIVE_BRANDS allows)")
|
||||
parser.add_argument("--include-inactive", action="store_true",
|
||||
help="Enrich every brand table, ignoring ACTIVE_BRANDS. Needed to reach "
|
||||
"brands the app does not currently display.")
|
||||
parser.add_argument("--categories", default=None,
|
||||
help="Comma-separated categories to restrict the run to, e.g. 'Fruits & Vegetables,Eggs'")
|
||||
args = parser.parse_args()
|
||||
|
||||
ensure_nutrition_schema()
|
||||
@@ -44,12 +61,16 @@ def main() -> None:
|
||||
generate_narrative=not args.no_narrative,
|
||||
progress_cb=_progress,
|
||||
max_products=args.max_products,
|
||||
brands=[b for b in (args.brands or "").split(",") if b.strip()] or None,
|
||||
include_inactive=args.include_inactive,
|
||||
categories=[c for c in (args.categories or "").split(",") if c.strip()] or None,
|
||||
)
|
||||
|
||||
logger.info(
|
||||
f"Done in {result.duration_seconds}s - "
|
||||
f"{result.verified} verified, {result.partial} partial, {result.unavailable} unavailable "
|
||||
f"of {result.total_products} total products"
|
||||
f"of {result.total_products} total products "
|
||||
f"({result.skipped_non_consumable} non-consumable skipped)"
|
||||
)
|
||||
if result.errors:
|
||||
logger.warning(f"{len(result.errors)} errors (showing up to 10): {result.errors[:10]}")
|
||||
|
||||
491
scripts/purge_non_consumable_nutrition.py
Normal file
491
scripts/purge_non_consumable_nutrition.py
Normal file
@@ -0,0 +1,491 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Remove nutrition rows belonging to products that are not food.
|
||||
|
||||
WHY THIS EXISTS
|
||||
---------------
|
||||
`app/services/consumability.py` now stops these rows being created. It cannot
|
||||
remove the ones already there. Measured on the live catalogue:
|
||||
|
||||
Colgate-Palmolive Palmolive Naturals Bath Soap health score present
|
||||
Cavinkare Nyle, Cavinkare Nature's Hair Care health score present
|
||||
P&G Pantene Hair Care health score present
|
||||
Godrej Hit Spray Mosquito 291 kcal, health score present
|
||||
|
||||
A shopper reading "health score 62" on a mosquito repellent has no way to know
|
||||
the number is meaningless, and `nutrition_db.query_products` will happily rank
|
||||
it against actual food.
|
||||
|
||||
WHAT IT WILL NOT DELETE
|
||||
-----------------------
|
||||
* Anything `consumability` cannot positively identify as non-food. The
|
||||
classifier is three-state on purpose and this script uses the POSITIVE test
|
||||
`is_non_consumable`, never `not is_consumable`. An unrecognised product is
|
||||
reported as UNDECIDED and left exactly where it is - a wrong deletion here is
|
||||
unrecoverable except from the backup, a wrong retention is visible and
|
||||
fixable.
|
||||
* Orphans - nutrition rows whose (brand, image_id) is in no brand table. That
|
||||
is a different defect (a deleted product, or a brand-name mismatch since
|
||||
enrichment) and deleting them here would hide it.
|
||||
|
||||
WHY IT RE-READS THE CATEGORY FROM THE BRAND TABLE
|
||||
-------------------------------------------------
|
||||
`nutrition_facts.category` is a copy taken at enrichment time
|
||||
(`enrich_one_product`), so it can be stale relative to the product. The verdict
|
||||
is taken against the brand table's current category and product name, which is
|
||||
what the catalogue actually claims today.
|
||||
|
||||
ORDER OF DELETION
|
||||
-----------------
|
||||
Children first - `nutrition_similar_products` and
|
||||
`nutrition_healthy_alternatives` before `nutrition_insights` and
|
||||
`nutrition_facts` - so an interrupted run never leaves a similarity edge
|
||||
pointing at a fact row that no longer exists.
|
||||
|
||||
Both cache tables are cleaned in BOTH DIRECTIONS. The inbound direction is the
|
||||
one that gets forgotten: a shampoo can be some real food's nearest neighbour,
|
||||
and that edge survives a naive purge, after which `get_similar_products` serves
|
||||
a dangling reference.
|
||||
|
||||
USAGE
|
||||
-----
|
||||
python -m scripts.purge_non_consumable_nutrition --audit
|
||||
python -m scripts.purge_non_consumable_nutrition --audit --json
|
||||
|
||||
python -m scripts.purge_non_consumable_nutrition --purge # dry run
|
||||
python -m scripts.purge_non_consumable_nutrition --purge --apply
|
||||
python -m scripts.purge_non_consumable_nutrition --restore data/nutrition_purge_backup_X.json --apply
|
||||
|
||||
`--audit` is read-only by construction. `--purge` is a dry run until `--apply`,
|
||||
and `--apply` writes a full backup of every affected row first.
|
||||
|
||||
AFTERWARDS
|
||||
----------
|
||||
The KNN index and KMeans clusters in `data/artifacts/*.joblib` were fitted with
|
||||
the deleted rows still in them. Re-run the training step or the similar-products
|
||||
cache will keep serving keys that no longer exist. The script prints this.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import sys
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime
|
||||
from decimal import Decimal
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from app.infrastructure.settings import DB_HOST, DB_NAME # noqa: E402
|
||||
from app.services.consumability import ( # noqa: E402
|
||||
Edibility, classify_edibility, is_junk_category,
|
||||
)
|
||||
from app.services.vector_store import _connect, display_name_for_suffix # noqa: E402
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
||||
logger = logging.getLogger("purge_non_consumable_nutrition")
|
||||
|
||||
DATA_DIR = Path(__file__).resolve().parents[1] / "data"
|
||||
|
||||
# Children first. `nutrition_facts` last, so a crash mid-run leaves referential
|
||||
# integrity intact rather than half-broken.
|
||||
CACHE_TABLES = (
|
||||
("nutrition_similar_products", ("brand", "image_id"), ("similar_brand", "similar_image_id")),
|
||||
("nutrition_healthy_alternatives", ("brand", "image_id"), ("alt_brand", "alt_image_id")),
|
||||
)
|
||||
CORE_TABLES = ("nutrition_insights", "nutrition_facts")
|
||||
|
||||
Key = Tuple[str, str]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Planning - pure, no database, so it is testable
|
||||
# ---------------------------------------------------------------------------
|
||||
@dataclass
|
||||
class PurgePlan:
|
||||
delete: List[Dict[str, Any]] = field(default_factory=list)
|
||||
keep: List[Dict[str, Any]] = field(default_factory=list)
|
||||
undecided: List[Dict[str, Any]] = field(default_factory=list)
|
||||
orphans: List[Dict[str, Any]] = field(default_factory=list)
|
||||
junk_categories: List[Dict[str, Any]] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def delete_keys(self) -> List[Key]:
|
||||
return [(r["brand"], r["image_id"]) for r in self.delete]
|
||||
|
||||
|
||||
def plan_purge(
|
||||
nutrition_rows: List[Dict[str, Any]],
|
||||
catalogue: Dict[Key, Dict[str, Any]],
|
||||
) -> PurgePlan:
|
||||
"""Decide the fate of every `nutrition_facts` row.
|
||||
|
||||
`nutrition_rows` needs only brand, image_id, product_name, category and
|
||||
data_status. `catalogue` is (brand, image_id) -> the brand table's current
|
||||
product_name and category, which is the authority - see the module
|
||||
docstring on why the fact row's own copy is not trusted.
|
||||
"""
|
||||
plan = PurgePlan()
|
||||
for row in nutrition_rows:
|
||||
key = (row.get("brand") or "", row.get("image_id") or "")
|
||||
source = catalogue.get(key)
|
||||
if source is None:
|
||||
plan.orphans.append({**row, "reason": "no matching row in any brand table"})
|
||||
continue
|
||||
|
||||
category = source.get("category") or ""
|
||||
name = source.get("product_name") or row.get("product_name") or ""
|
||||
if is_junk_category(category):
|
||||
plan.junk_categories.append({**row, "category": category, "product_name": name})
|
||||
|
||||
verdict = classify_edibility(category, name)
|
||||
entry = {
|
||||
**row,
|
||||
"category": category,
|
||||
"product_name": name,
|
||||
"reason": verdict.reason,
|
||||
"signal": verdict.signal,
|
||||
}
|
||||
if verdict.edibility is Edibility.NON_CONSUMABLE:
|
||||
plan.delete.append(entry)
|
||||
elif verdict.edibility is Edibility.CONSUMABLE:
|
||||
plan.keep.append(entry)
|
||||
else:
|
||||
plan.undecided.append(entry)
|
||||
return plan
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Reading
|
||||
# ---------------------------------------------------------------------------
|
||||
def _catalogue_index(cur) -> Dict[Key, Dict[str, Any]]:
|
||||
"""(brand, image_id) -> current product_name and category.
|
||||
|
||||
The brand key is `display_name_for_suffix(...)`, which is the string
|
||||
`enrich_one_product` was called with, so it is the correct join key.
|
||||
"""
|
||||
cur.execute(
|
||||
"SELECT table_name FROM information_schema.tables "
|
||||
"WHERE table_schema='public' AND table_name LIKE 'brand_%' ORDER BY table_name"
|
||||
)
|
||||
index: Dict[Key, Dict[str, Any]] = {}
|
||||
for (table,) in cur.fetchall():
|
||||
cur.execute(
|
||||
"SELECT column_name FROM information_schema.columns WHERE table_name=%s",
|
||||
(table,),
|
||||
)
|
||||
cols = {r[0] for r in cur.fetchall()}
|
||||
if "image_id" not in cols:
|
||||
continue
|
||||
select = "image_id" + (",product_name" if "product_name" in cols else "")
|
||||
select += ",category" if "category" in cols else ""
|
||||
cur.execute(f"SELECT {select} FROM {table} WHERE image_id IS NOT NULL")
|
||||
brand = display_name_for_suffix(table[len("brand_"):])
|
||||
for record in cur.fetchall():
|
||||
row = dict(zip(select.split(","), record))
|
||||
index[(brand, row["image_id"])] = {
|
||||
"product_name": row.get("product_name") or "",
|
||||
"category": row.get("category") or "",
|
||||
"table": table,
|
||||
}
|
||||
return index
|
||||
|
||||
|
||||
def _nutrition_rows(cur) -> List[Dict[str, Any]]:
|
||||
cur.execute(
|
||||
"SELECT brand, image_id, product_name, category, data_status "
|
||||
"FROM nutrition_facts"
|
||||
)
|
||||
cols = ("brand", "image_id", "product_name", "category", "data_status")
|
||||
return [dict(zip(cols, r)) for r in cur.fetchall()]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Reporting
|
||||
# ---------------------------------------------------------------------------
|
||||
def _report(plan: PurgePlan, as_json: bool) -> Dict[str, Any]:
|
||||
scored = [r for r in plan.delete if (r.get("data_status") or "") != "unavailable"]
|
||||
by_category: Dict[str, int] = {}
|
||||
for row in plan.delete:
|
||||
by_category[row["category"]] = by_category.get(row["category"], 0) + 1
|
||||
|
||||
summary = {
|
||||
"nutrition_rows": len(plan.delete) + len(plan.keep) + len(plan.undecided) + len(plan.orphans),
|
||||
"to_delete": len(plan.delete),
|
||||
"to_delete_holding_real_data": len(scored),
|
||||
"keep": len(plan.keep),
|
||||
"undecided": len(plan.undecided),
|
||||
"orphans": len(plan.orphans),
|
||||
"junk_categories": len(plan.junk_categories),
|
||||
"by_category": dict(sorted(by_category.items(), key=lambda kv: -kv[1])),
|
||||
}
|
||||
|
||||
if as_json:
|
||||
print(json.dumps({"summary": summary, "delete": plan.delete,
|
||||
"undecided": plan.undecided, "orphans": plan.orphans}, indent=2))
|
||||
return summary
|
||||
|
||||
logger.info("")
|
||||
logger.info("Target database: %s / %s", DB_HOST, DB_NAME)
|
||||
logger.info("")
|
||||
logger.info("%d nutrition_facts row(s)", summary["nutrition_rows"])
|
||||
logger.info(" %6d non-consumable <-- would be deleted", summary["to_delete"])
|
||||
logger.info(" %6d of those actually hold nutrition data (not 'unavailable')",
|
||||
summary["to_delete_holding_real_data"])
|
||||
logger.info(" %6d consumable, kept", summary["keep"])
|
||||
logger.info(" %6d undecided, kept - extend the map or fix the category",
|
||||
summary["undecided"])
|
||||
logger.info(" %6d orphaned, kept - product missing from every brand table",
|
||||
summary["orphans"])
|
||||
|
||||
if by_category:
|
||||
logger.info("")
|
||||
logger.info("Non-consumable rows by category:")
|
||||
for category, count in sorted(by_category.items(), key=lambda kv: -kv[1]):
|
||||
logger.info(" %-40s %5d", category or "(blank)", count)
|
||||
|
||||
if scored:
|
||||
logger.info("")
|
||||
logger.info("Rows holding real nutrition for a non-food product "
|
||||
"- this is what justifies the change:")
|
||||
for row in scored[:25]:
|
||||
logger.info(" %-34s %-26s %s",
|
||||
(row["product_name"] or "")[:33], row["category"][:25],
|
||||
row["data_status"])
|
||||
if len(scored) > 25:
|
||||
logger.info(" ... and %d more", len(scored) - 25)
|
||||
|
||||
if plan.undecided:
|
||||
logger.info("")
|
||||
logger.info("UNDECIDED - kept, because guessing here would delete real data:")
|
||||
for row in plan.undecided[:20]:
|
||||
logger.info(" %-34s %-22s %s",
|
||||
(row["product_name"] or "")[:33],
|
||||
(row["category"] or "(blank)")[:21], row["reason"][:60])
|
||||
if len(plan.undecided) > 20:
|
||||
logger.info(" ... and %d more", len(plan.undecided) - 20)
|
||||
|
||||
if plan.junk_categories:
|
||||
logger.info("")
|
||||
logger.info("%d row(s) carry a bare number as their category. That is import "
|
||||
"damage and wants a catalogue repair, not a rule here.",
|
||||
len(plan.junk_categories))
|
||||
|
||||
if plan.orphans:
|
||||
logger.info("")
|
||||
logger.info("ORPHANS - a nutrition row with no product. Kept and reported; "
|
||||
"deleting them here would hide the underlying defect:")
|
||||
for row in plan.orphans[:10]:
|
||||
logger.info(" %s / %s", row["brand"], row["image_id"])
|
||||
if len(plan.orphans) > 10:
|
||||
logger.info(" ... and %d more", len(plan.orphans) - 10)
|
||||
|
||||
return summary
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Writing
|
||||
# ---------------------------------------------------------------------------
|
||||
def _plain(value: Any) -> Any:
|
||||
"""JSON-safe. nutrition_facts has NUMERIC and TIMESTAMP columns, so this
|
||||
hits both traps that merge_haldiram hit for real."""
|
||||
if isinstance(value, Decimal):
|
||||
return float(value)
|
||||
if isinstance(value, datetime):
|
||||
return value.isoformat()
|
||||
return value
|
||||
|
||||
|
||||
def _backup(cur, keys: List[Key]) -> Path:
|
||||
"""Every row about to be touched, across all four tables, in both
|
||||
directions for the two cache tables."""
|
||||
DATA_DIR.mkdir(parents=True, exist_ok=True)
|
||||
tables: Dict[str, List[Dict[str, Any]]] = {}
|
||||
|
||||
for table in CORE_TABLES:
|
||||
cur.execute(f"SELECT * FROM {table} WHERE (brand, image_id) IN %s",
|
||||
(tuple(keys),))
|
||||
cols = [d[0] for d in cur.description]
|
||||
tables[table] = [{c: _plain(v) for c, v in zip(cols, r)} for r in cur.fetchall()]
|
||||
|
||||
for table, own, inbound in CACHE_TABLES:
|
||||
cur.execute(
|
||||
f"SELECT * FROM {table} WHERE ({own[0]}, {own[1]}) IN %s "
|
||||
f"OR ({inbound[0]}, {inbound[1]}) IN %s",
|
||||
(tuple(keys), tuple(keys)),
|
||||
)
|
||||
cols = [d[0] for d in cur.description]
|
||||
tables[table] = [{c: _plain(v) for c, v in zip(cols, r)} for r in cur.fetchall()]
|
||||
|
||||
path = DATA_DIR / f"nutrition_purge_backup_{datetime.now():%Y%m%d_%H%M%S}.json"
|
||||
path.write_text(json.dumps({
|
||||
"created_at": datetime.now().isoformat(),
|
||||
"db_host": DB_HOST, "db_name": DB_NAME,
|
||||
"keys": [list(k) for k in keys],
|
||||
"tables": tables,
|
||||
}, indent=2), encoding="utf-8")
|
||||
return path
|
||||
|
||||
|
||||
def _delete(cur, keys: List[Key]) -> Dict[str, int]:
|
||||
deleted: Dict[str, int] = {}
|
||||
for table, own, inbound in CACHE_TABLES:
|
||||
cur.execute(
|
||||
f"DELETE FROM {table} WHERE ({own[0]}, {own[1]}) IN %s "
|
||||
f"OR ({inbound[0]}, {inbound[1]}) IN %s",
|
||||
(tuple(keys), tuple(keys)),
|
||||
)
|
||||
deleted[table] = cur.rowcount
|
||||
for table in CORE_TABLES:
|
||||
cur.execute(f"DELETE FROM {table} WHERE (brand, image_id) IN %s", (tuple(keys),))
|
||||
deleted[table] = cur.rowcount
|
||||
return deleted
|
||||
|
||||
|
||||
def _backfill_edibility(cur, plan: PurgePlan) -> Dict[str, int]:
|
||||
"""Stamp the verdict this run already computed onto the rows it is keeping.
|
||||
|
||||
WHY HERE AND NOT IN A MIGRATION: `plan_purge` has just classified every one
|
||||
of these rows against the catalogue's own category, using the same
|
||||
classifier and the same inputs `enrich_one_product` uses. Writing the answer
|
||||
down costs one UPDATE; computing it a second time somewhere else would be a
|
||||
second chance to compute it differently.
|
||||
|
||||
WHY IT MATTERS: `GET /api/nutrition/health-scores` filters on
|
||||
`nutrition_facts.edibility`, and every row written before that column
|
||||
existed carries NULL, which the endpoint treats as unconfirmed and excludes.
|
||||
Without this the endpoint returns nothing until a full re-enrichment run.
|
||||
|
||||
Orphans are deliberately left NULL: they have no catalogue row, so there is
|
||||
no category to classify them from, and guessing is the thing this whole
|
||||
script exists to stop.
|
||||
"""
|
||||
# Idempotent, and the same statement `ensure_nutrition_schema()` runs. Here
|
||||
# so the script works against a database the API has not started against
|
||||
# yet, rather than failing on an unknown column.
|
||||
cur.execute("ALTER TABLE nutrition_facts ADD COLUMN IF NOT EXISTS edibility TEXT")
|
||||
|
||||
written: Dict[str, int] = {}
|
||||
for label, rows in (("consumable", plan.keep), ("unknown", plan.undecided)):
|
||||
keys = [(r["brand"], r["image_id"]) for r in rows]
|
||||
if not keys:
|
||||
written[label] = 0
|
||||
continue
|
||||
cur.execute(
|
||||
"UPDATE nutrition_facts SET edibility = %s WHERE (brand, image_id) IN %s",
|
||||
(label, tuple(keys)),
|
||||
)
|
||||
written[label] = cur.rowcount
|
||||
return written
|
||||
|
||||
|
||||
def restore(cur, conn, path: Path, apply: bool) -> int:
|
||||
payload = json.loads(path.read_text(encoding="utf-8"))
|
||||
if payload.get("db_host") != DB_HOST:
|
||||
logger.error("Backup was taken from %s but DB_HOST is %s. Refusing.",
|
||||
payload.get("db_host"), DB_HOST)
|
||||
return 1
|
||||
total = 0
|
||||
for table, rows in payload["tables"].items():
|
||||
for row in rows:
|
||||
total += 1
|
||||
if not apply:
|
||||
continue
|
||||
cols = list(row)
|
||||
cur.execute(
|
||||
f"INSERT INTO {table} ({', '.join(cols)}) "
|
||||
f"VALUES ({', '.join('%s' for _ in cols)}) ON CONFLICT DO NOTHING",
|
||||
[row[c] for c in cols],
|
||||
)
|
||||
if apply:
|
||||
conn.commit()
|
||||
logger.info("%s %d row(s) from %s", "Restored" if apply else "Would restore",
|
||||
total, path.name)
|
||||
return 0
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(
|
||||
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
mode = parser.add_mutually_exclusive_group(required=True)
|
||||
mode.add_argument("--audit", action="store_true",
|
||||
help="read-only report; writes nothing")
|
||||
mode.add_argument("--purge", action="store_true",
|
||||
help="delete non-consumable nutrition rows")
|
||||
mode.add_argument("--restore", metavar="PATH", help="restore a backup file")
|
||||
|
||||
parser.add_argument("--apply", action="store_true",
|
||||
help="commit the changes (default is a dry run)")
|
||||
parser.add_argument("--dry-run", action="store_true",
|
||||
help="explicit no-op; this is already the default")
|
||||
parser.add_argument("--json", action="store_true", help="audit output as JSON")
|
||||
args = parser.parse_args()
|
||||
|
||||
apply = args.apply and not args.dry_run
|
||||
if args.audit and args.apply:
|
||||
parser.error("--audit is read-only; it cannot be combined with --apply")
|
||||
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
logger.error("No database connection.")
|
||||
return 1
|
||||
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
if args.restore:
|
||||
return restore(cur, conn, Path(args.restore), apply)
|
||||
|
||||
plan = plan_purge(_nutrition_rows(cur), _catalogue_index(cur))
|
||||
summary = _report(plan, args.json)
|
||||
|
||||
if args.audit:
|
||||
return 0
|
||||
|
||||
logger.info("")
|
||||
logger.info("Mode: %s", "APPLY - this writes" if apply
|
||||
else "DRY RUN - nothing is written")
|
||||
labelled = len(plan.keep) + len(plan.undecided)
|
||||
if not apply:
|
||||
logger.info("Re-run with --apply to delete %d row(s) and label %d.",
|
||||
len(plan.delete), labelled)
|
||||
return 0
|
||||
|
||||
# Labelling runs first and unconditionally: it is additive, it is
|
||||
# correct even when there is nothing to delete, and it is what makes
|
||||
# the health-score endpoint able to tell food from not-food in SQL.
|
||||
written = _backfill_edibility(cur, plan)
|
||||
conn.commit()
|
||||
logger.info("")
|
||||
for label, count in written.items():
|
||||
logger.info(" labelled %5d row(s) edibility=%s", count, label)
|
||||
|
||||
if not plan.delete:
|
||||
logger.info("")
|
||||
logger.info("Nothing to delete.")
|
||||
return 0
|
||||
|
||||
path = _backup(cur, plan.delete_keys)
|
||||
logger.info("Backup written: %s", path)
|
||||
deleted = _delete(cur, plan.delete_keys)
|
||||
conn.commit()
|
||||
|
||||
logger.info("")
|
||||
for table, count in deleted.items():
|
||||
logger.info(" deleted %5d row(s) from %s", count, table)
|
||||
logger.info("")
|
||||
logger.info("The similarity and clustering models were fitted with these "
|
||||
"rows still in them.")
|
||||
logger.info("Retrain before trusting /similar or /alternatives:")
|
||||
logger.info(" python -m scripts.train_nutrition_models")
|
||||
_ = summary
|
||||
finally:
|
||||
conn.close()
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
207
scripts/refresh_usda_snapshot.py
Normal file
207
scripts/refresh_usda_snapshot.py
Normal file
@@ -0,0 +1,207 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Rebuild the pinned USDA nutrient snapshot from the open bulk download.
|
||||
|
||||
WHY THE BULK DOWNLOAD AND NOT THE API
|
||||
-------------------------------------
|
||||
The FoodData Central API is the obvious source and it cannot do this job. Its
|
||||
unauthenticated DEMO_KEY allows TEN requests per hour - measured, with a
|
||||
Retry-After of thirteen hours once exhausted - and the curated table references
|
||||
101 distinct foods. Even a personal key would make rebuilding the snapshot a
|
||||
rate-limit exercise.
|
||||
|
||||
The same data is published as a plain zip with no key and no quota:
|
||||
|
||||
FoodData_Central_foundation_food_json_*.zip ~0.5 MB
|
||||
FoodData_Central_sr_legacy_food_json_*.zip ~12 MB
|
||||
|
||||
So the snapshot is built from those, and `USDA_FDC_API_KEY` stays optional -
|
||||
needed only to look up an id the snapshot does not already hold.
|
||||
|
||||
WHAT IT STORES, AND WHY RAW
|
||||
---------------------------
|
||||
The `foodNutrients` list exactly as USDA publishes it, trimmed to the fields
|
||||
this codebase reads. NOT pre-mapped to our columns: if the snapshot held
|
||||
finished per-100g values, the offline tests would exercise a different code
|
||||
path from the online one, and the unit-checking rules in
|
||||
`nutrition_usda_service` - the ones that stop a sodium figure being silently
|
||||
multiplied by a thousand - would go untested.
|
||||
|
||||
USAGE
|
||||
-----
|
||||
python -m scripts.refresh_usda_snapshot # dry run, shows the diff
|
||||
python -m scripts.refresh_usda_snapshot --apply
|
||||
python -m scripts.refresh_usda_snapshot --apply --keep-download
|
||||
|
||||
A dry run reports every nutrient whose value moved, so a USDA revision is read
|
||||
and accepted rather than silently absorbed.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import logging
|
||||
import shutil
|
||||
import sys
|
||||
import tempfile
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
import requests # noqa: E402
|
||||
|
||||
from app.services import produce_reference # noqa: E402
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
||||
logger = logging.getLogger("refresh_usda_snapshot")
|
||||
|
||||
SNAPSHOT = Path(__file__).resolve().parents[1] / "app" / "services" / "data" / "usda_snapshot.json"
|
||||
|
||||
# Pinned filenames rather than "latest": a dataset that silently became a newer
|
||||
# release between two runs would move numbers with no diff to read.
|
||||
DATASETS = (
|
||||
("FoundationFoods",
|
||||
"https://fdc.nal.usda.gov/fdc-datasets/FoodData_Central_foundation_food_json_2025-04-24.zip"),
|
||||
("SRLegacyFoods",
|
||||
"https://fdc.nal.usda.gov/fdc-datasets/FoodData_Central_sr_legacy_food_json_2021-10-28.zip"),
|
||||
)
|
||||
|
||||
|
||||
def _download(url: str, into: Path) -> Path:
|
||||
name = url.rsplit("/", 1)[-1]
|
||||
target = into / name
|
||||
if target.exists():
|
||||
return target
|
||||
logger.info(" downloading %s", name)
|
||||
with requests.get(url, stream=True, timeout=300) as resp:
|
||||
resp.raise_for_status()
|
||||
with target.open("wb") as fh:
|
||||
for chunk in resp.iter_content(1 << 20):
|
||||
fh.write(chunk)
|
||||
return target
|
||||
|
||||
|
||||
def _load(zip_path: Path, key: str) -> List[Dict[str, Any]]:
|
||||
with zipfile.ZipFile(zip_path) as zf:
|
||||
name = zf.namelist()[0]
|
||||
with zf.open(name) as fh:
|
||||
return json.load(fh)[key]
|
||||
|
||||
|
||||
def _trim(food: Dict[str, Any]) -> Dict[str, Any]:
|
||||
nutrients = []
|
||||
for fn in food.get("foodNutrients", []):
|
||||
n = fn.get("nutrient") or {}
|
||||
if fn.get("amount") is None or not n.get("id"):
|
||||
continue
|
||||
nutrients.append({"id": n["id"], "name": n.get("name"),
|
||||
"unit": n.get("unitName"), "amount": fn["amount"]})
|
||||
portions = []
|
||||
for p in (food.get("foodPortions") or [])[:3]:
|
||||
grams = p.get("gramWeight")
|
||||
if not grams:
|
||||
continue
|
||||
portions.append({
|
||||
"gram_weight": grams,
|
||||
"description": (p.get("portionDescription")
|
||||
or (p.get("measureUnit") or {}).get("name") or ""),
|
||||
})
|
||||
return {"fdc_id": food["fdcId"], "description": food.get("description"),
|
||||
"data_type": food.get("dataType"),
|
||||
"publication_date": food.get("publicationDate"),
|
||||
"nutrients": nutrients, "portions": portions}
|
||||
|
||||
|
||||
def _diff(old: Dict[str, Any], new: Dict[str, Any]) -> List[str]:
|
||||
lines: List[str] = []
|
||||
for fid, food in sorted(new.items()):
|
||||
before = old.get(fid)
|
||||
if not before:
|
||||
lines.append(f" NEW {fid} {food['description'][:56]}")
|
||||
continue
|
||||
if before.get("description") != food.get("description"):
|
||||
lines.append(f" RENAME {fid} {before['description'][:34]} -> "
|
||||
f"{food['description'][:34]}")
|
||||
old_n = {n["id"]: n["amount"] for n in before.get("nutrients", [])}
|
||||
new_n = {n["id"]: n["amount"] for n in food.get("nutrients", [])}
|
||||
for nid, amount in sorted(new_n.items()):
|
||||
if nid in old_n and old_n[nid] != amount:
|
||||
lines.append(f" VALUE {fid} nutrient {nid}: "
|
||||
f"{old_n[nid]} -> {amount} ({food['description'][:30]})")
|
||||
for nid in sorted(set(old_n) - set(new_n)):
|
||||
lines.append(f" GONE {fid} nutrient {nid} ({food['description'][:30]})")
|
||||
for fid in sorted(set(old) - set(new)):
|
||||
lines.append(f" LOST {fid} {old[fid].get('description', '')[:56]}")
|
||||
return lines
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(
|
||||
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
parser.add_argument("--apply", action="store_true",
|
||||
help="write the snapshot (default is a dry run)")
|
||||
parser.add_argument("--keep-download", action="store_true",
|
||||
help="keep the downloaded zips instead of using a temp dir")
|
||||
args = parser.parse_args()
|
||||
|
||||
wanted = {e["usda_fdc_id"] for e in produce_reference.all_entries().values()
|
||||
if e.get("usda_fdc_id")}
|
||||
logger.info("%d FDC id(s) referenced by the curated table", len(wanted))
|
||||
|
||||
workdir = Path("usda_bulk") if args.keep_download else Path(tempfile.mkdtemp())
|
||||
workdir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
snapshot: Dict[str, Any] = {}
|
||||
try:
|
||||
for key, url in DATASETS:
|
||||
path = _download(url, workdir)
|
||||
for food in _load(path, key):
|
||||
fid = food.get("fdcId")
|
||||
if fid in wanted and str(fid) not in snapshot:
|
||||
snapshot[str(fid)] = _trim(food)
|
||||
finally:
|
||||
if not args.keep_download:
|
||||
shutil.rmtree(workdir, ignore_errors=True)
|
||||
|
||||
missing = sorted(wanted - {int(k) for k in snapshot})
|
||||
if missing:
|
||||
logger.warning("NOT FOUND in either dataset: %s", missing)
|
||||
|
||||
old = {}
|
||||
if SNAPSHOT.exists():
|
||||
old = json.loads(SNAPSHOT.read_text(encoding="utf-8")).get("foods", {})
|
||||
|
||||
changes = _diff(old, snapshot)
|
||||
if changes:
|
||||
logger.info("")
|
||||
logger.info("%d change(s) against the current snapshot:", len(changes))
|
||||
for line in changes[:60]:
|
||||
logger.info("%s", line)
|
||||
if len(changes) > 60:
|
||||
logger.info(" ... and %d more", len(changes) - 60)
|
||||
else:
|
||||
logger.info("No change against the current snapshot.")
|
||||
|
||||
if not args.apply:
|
||||
logger.info("")
|
||||
logger.info("Dry run - nothing written. Re-run with --apply.")
|
||||
return 0
|
||||
|
||||
SNAPSHOT.parent.mkdir(parents=True, exist_ok=True)
|
||||
SNAPSHOT.write_text(json.dumps(
|
||||
{"source": "USDA FoodData Central bulk download",
|
||||
"datasets": [u.rsplit("/", 1)[-1].replace(".zip", "") for _, u in DATASETS],
|
||||
"foods": snapshot},
|
||||
indent=1, sort_keys=True), encoding="utf-8")
|
||||
logger.info("")
|
||||
logger.info("Wrote %s (%d food(s), %.1f MB)", SNAPSHOT, len(snapshot),
|
||||
SNAPSHOT.stat().st_size / 1e6)
|
||||
logger.info("Run the test suite: tests/test_nutrition_usda.py asserts every "
|
||||
"curated id is pinned and every description still matches.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -75,6 +75,7 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from app.infrastructure.settings import DB_HOST, DB_NAME # noqa: E402
|
||||
from app.services.brand_registry import resolve_parent_brand # noqa: E402
|
||||
from app.services.generic_products import OWN_PRODUCTS_BRAND # noqa: E402
|
||||
from app.services.vector_store import ( # noqa: E402
|
||||
_connect,
|
||||
_list_brand_table_suffixes,
|
||||
@@ -358,8 +359,18 @@ _OFF_CACHE: Dict[str, bool] = {}
|
||||
_OFF_BARCODE = re.compile(r"/images/products/((?:\d+/)+)")
|
||||
|
||||
|
||||
# "own" and "products" are the BUCKET's name, not a brand's. Left in, the token
|
||||
# "products" corroborates any URL containing /images/products/ or
|
||||
# /cdn/shop/products/ - which is every Open*Facts and every Shopify path - so
|
||||
# `_names_product` waved through 40+ images that named nothing about the item.
|
||||
# That is how the openbeautyfacts cosmetics photos became the stored image for
|
||||
# Banana, Orange, Papaya, Guava, Lemon and twenty more.
|
||||
_BUCKET_TOKENS = frozenset({"own", "products", "product"})
|
||||
|
||||
|
||||
def _brand_tokens(brand: str) -> List[str]:
|
||||
return [w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) if len(w) > 2]
|
||||
return [w for w in re.split(r"[^a-z0-9]+", (brand or "").lower())
|
||||
if len(w) > 2 and w not in _BUCKET_TOKENS]
|
||||
|
||||
|
||||
def _off_product_matches_brand(url: str, brand: str) -> bool:
|
||||
@@ -420,6 +431,15 @@ def _search_key(product_name: str, brand: str) -> Tuple[str, str]:
|
||||
return (brand.lower(), (base or product_name or "").lower())
|
||||
|
||||
|
||||
def _is_own_products(brand: str) -> bool:
|
||||
"""True for the unbranded-commodity bucket, under either spelling.
|
||||
|
||||
`repair()` passes the TABLE SUFFIX ("own_products"), while the pipeline and
|
||||
the UI use the display name ("Own Products").
|
||||
"""
|
||||
return _sanitize_name(brand or "") == _sanitize_name(OWN_PRODUCTS_BRAND)
|
||||
|
||||
|
||||
def _search_replacement(product_name: str, brand: str, max_results: int) -> List[str]:
|
||||
"""Validated image URLs for one product, best first.
|
||||
|
||||
@@ -433,7 +453,17 @@ def _search_replacement(product_name: str, brand: str, max_results: int) -> List
|
||||
return _SEARCH_CACHE[key]
|
||||
|
||||
from app.services.image_search import find_all_image_urls
|
||||
urls = find_all_image_urls(product_name, brand=brand, validate=True, max_results=max_results)
|
||||
|
||||
# `stage_6_images` already blanks the brand for this bucket
|
||||
# (store_catalog_pipeline.py) and this path never did, so a repair searched
|
||||
# for "own_products Apple fruit juice" and got exactly what it asked for.
|
||||
# Produce mode additionally skips the packaged-goods databases and the
|
||||
# commodity-hint table - see image_search._AMBIQUITY_HINTS.
|
||||
produce = _is_own_products(brand)
|
||||
urls = find_all_image_urls(
|
||||
product_name, brand="" if produce else brand,
|
||||
validate=True, max_results=max_results, produce=produce,
|
||||
)
|
||||
if not urls:
|
||||
_SEARCH_CACHE[key] = []
|
||||
return []
|
||||
@@ -451,14 +481,15 @@ def _search_replacement(product_name: str, brand: str, max_results: int) -> List
|
||||
# image would have been written as the primary image of four MTR products.
|
||||
# Let an error here raise: unranked output is not safe to store.
|
||||
from app.core.catalog_engine import ProductCatalogEngine
|
||||
rank_brand = "" if produce else brand
|
||||
urls = ProductCatalogEngine._select_best_images(
|
||||
ProductCatalogEngine, urls, product_name, brand, max_images=MAX_STORED_IMAGES
|
||||
ProductCatalogEngine, urls, product_name, rank_brand, max_images=MAX_STORED_IMAGES
|
||||
)
|
||||
if not urls:
|
||||
_SEARCH_CACHE[key] = []
|
||||
return []
|
||||
# Drop other brands' products before anything else looks at them.
|
||||
urls = [u for u in urls if _off_product_matches_brand(u, brand)]
|
||||
urls = [u for u in urls if produce or _off_product_matches_brand(u, brand)]
|
||||
if not urls:
|
||||
_SEARCH_CACHE[key] = []
|
||||
return []
|
||||
@@ -480,7 +511,7 @@ def _search_replacement(product_name: str, brand: str, max_results: int) -> List
|
||||
# (Zepto, Flipkart, Pinterest). That is the intended trade - 94% of rows
|
||||
# still find a corroborated URL, and the rest are reported for hand
|
||||
# curation rather than filled with a guess.
|
||||
ranked = [u for u in urls if _names_product(u, product_name, brand)]
|
||||
ranked = [u for u in urls if _names_product(u, product_name, rank_brand)]
|
||||
ranked.sort(key=lambda u: 0 if u.startswith("https://") else 1)
|
||||
_SEARCH_CACHE[key] = ranked
|
||||
return ranked
|
||||
|
||||
Reference in New Issue
Block a user