Health score updates in backend
This commit is contained in:
@@ -245,6 +245,14 @@ class BatchOut(BaseModel):
|
||||
# sits queued until the orchestrator picks it up, which the UI has to be
|
||||
# able to say out loud rather than showing a run that looks stuck.
|
||||
runner: str = batch_ingest.RUNNER_INPROCESS
|
||||
# The nutrition-enrichment job queued when this batch finished, pollable at
|
||||
# GET /api/admin/nutrition-intelligence/jobs/{job_id}. Surfaced here so the
|
||||
# poll a client is already doing for the ingestion also reveals the scoring
|
||||
# that follows it, instead of leaving the client to guess that it happened.
|
||||
#
|
||||
# None while the batch is still running, when AUTO_ENRICH_ON_UPLOAD is off,
|
||||
# and for every batch that finished before this existed.
|
||||
nutrition_job_id: Optional[str] = None
|
||||
# The 11 stage names, in order, so a client can draw the whole pipeline
|
||||
# before a file has entered any of it. Served rather than duplicated in the
|
||||
# frontend so the two cannot drift when a stage is added.
|
||||
|
||||
@@ -128,3 +128,56 @@ class NutritionEnrichmentJobOut(BaseModel):
|
||||
unavailable: Optional[int] = None
|
||||
duration_seconds: Optional[float] = None
|
||||
error: Optional[str] = None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Health-score listing (GET /api/nutrition/health-scores)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class HealthScoreItemOut(BaseModel):
|
||||
"""One scored consumable product.
|
||||
|
||||
`health_score` is REQUIRED here, against this module's all-Optional house
|
||||
style, and that is the point: the query filters on `health_score IS NOT
|
||||
NULL`, so a null arriving in this model means the query lost its guarantee.
|
||||
Declaring it required is what turns that into a loud failure instead of a
|
||||
blank cell in whatever dashboard is reading this.
|
||||
"""
|
||||
brand: str
|
||||
image_id: str
|
||||
product_name: Optional[str] = None
|
||||
category: Optional[str] = None
|
||||
|
||||
health_score: float
|
||||
nutrition_score: Optional[float] = None
|
||||
health_band: str # derived from health_score; see _health_band()
|
||||
scoring_version: Optional[str] = None
|
||||
|
||||
# Provenance travels with the number. A score is only as trustworthy as the
|
||||
# source it was computed from, and the consumer should be able to show that.
|
||||
data_status: Optional[str] = None
|
||||
data_source: Optional[str] = None
|
||||
source_url: Optional[str] = None
|
||||
|
||||
calories_kcal: Optional[float] = None
|
||||
protein_g: Optional[float] = None
|
||||
dietary_fiber_g: Optional[float] = None
|
||||
total_sugar_g: Optional[float] = None
|
||||
sodium_mg: Optional[float] = None
|
||||
|
||||
diet_tags: Optional[List[str]] = None
|
||||
allergens: Optional[List[str]] = None
|
||||
|
||||
model_config = {"extra": "ignore"}
|
||||
|
||||
|
||||
class HealthScoreListOut(BaseModel):
|
||||
"""Envelope matching the catalogue convention in `schemas.ProductListOut`
|
||||
rather than the bare-array convention of the older nutrition list
|
||||
endpoints - `total` is the whole reason this endpoint exists, since without
|
||||
it a client paging a thousand products cannot tell when it has finished."""
|
||||
total: int
|
||||
limit: int
|
||||
offset: int
|
||||
generated_at: str
|
||||
items: List[HealthScoreItemOut] = []
|
||||
|
||||
@@ -1,12 +1,17 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import List, Optional
|
||||
import csv
|
||||
import io
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Dict, Iterator, List, Optional
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Query
|
||||
from fastapi.responses import StreamingResponse
|
||||
|
||||
from app.api.nutrition_schemas import (
|
||||
FullNutritionOut, HealthyAlternativeOut, NutritionInsightsOut,
|
||||
PersonalizedRecommendationOut, ProductListItemOut, SimilarProductOut,
|
||||
FullNutritionOut, HealthScoreItemOut, HealthScoreListOut,
|
||||
HealthyAlternativeOut, NutritionInsightsOut, PersonalizedRecommendationOut,
|
||||
ProductListItemOut, SimilarProductOut,
|
||||
)
|
||||
from app.intelligence import nutrition_recommendation, nutrition_similarity
|
||||
from app.services import nutrition_alternatives_service, nutrition_analytics_service, nutrition_db
|
||||
@@ -100,6 +105,139 @@ def filter_products(
|
||||
return [ProductListItemOut(**r) for r in results]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# GET Health Scores - every scored consumable product
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# Distinct from `/products` above, which exists to answer "show me the ten
|
||||
# highest-protein snacks". This one exists to answer "give me the health score
|
||||
# of every consumable product in the catalogue", and that needs three things
|
||||
# `/products` does not offer: a total so a client knows when it has finished
|
||||
# paging, a brand filter, and a one-request export.
|
||||
|
||||
# Derived from health_score alone, never stored - a stored band would be one
|
||||
# more thing that can fall out of step with the score beside it.
|
||||
#
|
||||
# The 60 boundary is not a fresh invention: `nutrition_db.store_healthy_
|
||||
# distribution()` already counts `health_score >= 60` as a "healthy product".
|
||||
# Picking a different number here would leave two parts of the same API
|
||||
# disagreeing about what healthy means.
|
||||
_HEALTH_BANDS = ((80.0, "excellent"), (60.0, "good"), (40.0, "fair"))
|
||||
|
||||
|
||||
def _health_band(score: float) -> str:
|
||||
for threshold, label in _HEALTH_BANDS:
|
||||
if score >= threshold:
|
||||
return label
|
||||
return "poor"
|
||||
|
||||
|
||||
def _to_item(row: Dict[str, Any]) -> HealthScoreItemOut:
|
||||
return HealthScoreItemOut(**row, health_band=_health_band(row["health_score"]))
|
||||
|
||||
|
||||
_CSV_COLUMNS = (
|
||||
"brand", "image_id", "product_name", "category",
|
||||
"health_score", "health_band", "nutrition_score", "scoring_version",
|
||||
"data_status", "data_source", "source_url",
|
||||
"calories_kcal", "protein_g", "dietary_fiber_g", "total_sugar_g", "sodium_mg",
|
||||
"diet_tags", "allergens",
|
||||
)
|
||||
|
||||
|
||||
def _csv_rows(rows: Iterator[Dict[str, Any]]) -> Iterator[str]:
|
||||
"""Yields the export a row at a time so neither the full result set nor the
|
||||
full response body is ever held in memory at once."""
|
||||
buffer = io.StringIO()
|
||||
writer = csv.writer(buffer, lineterminator=chr(10))
|
||||
|
||||
def flush() -> str:
|
||||
value = buffer.getvalue()
|
||||
buffer.seek(0)
|
||||
buffer.truncate(0)
|
||||
return value
|
||||
|
||||
writer.writerow(_CSV_COLUMNS)
|
||||
yield flush()
|
||||
|
||||
for row in rows:
|
||||
row = dict(row, health_band=_health_band(row["health_score"]))
|
||||
writer.writerow([
|
||||
# A Postgres TEXT[] would render as "['Vegan', 'Gluten Free']" via
|
||||
# str(); pipe-joining keeps the cell readable in a spreadsheet.
|
||||
"|".join(row[c] or []) if c in ("diet_tags", "allergens") else
|
||||
("" if row.get(c) is None else row[c])
|
||||
for c in _CSV_COLUMNS
|
||||
])
|
||||
yield flush()
|
||||
|
||||
|
||||
@router.get(
|
||||
"/health-scores",
|
||||
response_model=None,
|
||||
responses={200: {
|
||||
"model": HealthScoreListOut,
|
||||
"content": {"application/json": {}, "text/csv": {}},
|
||||
"description": "Paged JSON envelope, or the whole list as CSV with format=csv.",
|
||||
}},
|
||||
)
|
||||
def list_health_scores(
|
||||
brand: Optional[str] = Query(None, description="Exact brand match, e.g. 'Own Products'"),
|
||||
category: Optional[str] = Query(None, description="Substring match, case-insensitive"),
|
||||
min_score: Optional[float] = Query(None, ge=0, le=100),
|
||||
max_score: Optional[float] = Query(None, ge=0, le=100),
|
||||
include_unknown: bool = Query(
|
||||
False,
|
||||
description="Also return products whose edibility was never confirmed. "
|
||||
"Non-consumables are never returned either way.",
|
||||
),
|
||||
sort_by: str = Query("health_score", description="health_score|nutrition_score|protein|fiber|sugar|sodium|calcium|iron|vitamin_c|calories|product_name|brand|category"),
|
||||
order: str = Query("desc", pattern="^(asc|desc)$"),
|
||||
limit: int = Query(100, ge=1, le=500),
|
||||
offset: int = Query(0, ge=0),
|
||||
fmt: str = Query("json", alias="format", pattern="^(json|csv)$"),
|
||||
):
|
||||
"""Every consumable product that has a health score.
|
||||
|
||||
Only scored products are returned: a consumable whose nutrition could not be
|
||||
matched to a verified source has no score, and is absent rather than present
|
||||
with a null - `compute_scores` refuses to invent one, and this endpoint
|
||||
refuses to imply one.
|
||||
|
||||
Non-consumables (soap, shampoo, mosquito repellent) are excluded by the
|
||||
stored `edibility` verdict, as are products the classifier could not decide
|
||||
on. `include_unknown=true` opts the undecided ones back in; nothing opts a
|
||||
confirmed non-consumable in.
|
||||
|
||||
`format=csv` streams the entire matching set in one response and ignores
|
||||
`limit`/`offset`, which is the point of it.
|
||||
"""
|
||||
filters = {
|
||||
"brand": brand, "category": category,
|
||||
"min_score": min_score, "max_score": max_score,
|
||||
"include_unknown": include_unknown,
|
||||
}
|
||||
|
||||
if fmt == "csv":
|
||||
stamp = datetime.now(timezone.utc).strftime("%Y%m%d")
|
||||
return StreamingResponse(
|
||||
_csv_rows(nutrition_db.iter_health_scores(sort_by=sort_by, order=order, **filters)),
|
||||
media_type="text/csv; charset=utf-8",
|
||||
headers={"Content-Disposition": f'attachment; filename="health_scores_{stamp}.csv"'},
|
||||
)
|
||||
|
||||
rows = nutrition_db.query_health_scores(
|
||||
sort_by=sort_by, order=order, limit=limit, offset=offset, **filters)
|
||||
return HealthScoreListOut(
|
||||
# Counted with the SAME filters that produced `rows`, so the two cannot
|
||||
# describe different populations.
|
||||
total=nutrition_db.count_health_scores(**filters),
|
||||
limit=limit, offset=offset,
|
||||
generated_at=datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
||||
items=[_to_item(r) for r in rows],
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# GET Nutrition Analytics (Feature 9)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -9,6 +9,7 @@ from app.api.background import run_in_background
|
||||
from app.api.deps import require_admin
|
||||
from app.api.nutrition_job_store import nutrition_job_store
|
||||
from app.services import nutrition_enrichment_service
|
||||
from app.services.nutrition_autoenrich import run_enrich_job
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
router = APIRouter(prefix="/admin/nutrition-intelligence", tags=["admin", "nutrition"])
|
||||
@@ -19,30 +20,13 @@ class EnrichRequest(BaseModel):
|
||||
generate_narrative: bool = True
|
||||
max_products: int | None = None
|
||||
|
||||
|
||||
def _run_enrich_job(job_id: str, skip_if_verified: bool, generate_narrative: bool, max_products: int | None) -> None:
|
||||
nutrition_job_store.update(job_id, status="running")
|
||||
|
||||
def progress_cb(done: int, total: int) -> None:
|
||||
nutrition_job_store.update(job_id, processed=done, total=total)
|
||||
|
||||
try:
|
||||
result = nutrition_enrichment_service.enrich_all_products(
|
||||
skip_if_verified=skip_if_verified, generate_narrative=generate_narrative,
|
||||
progress_cb=progress_cb, max_products=max_products,
|
||||
)
|
||||
nutrition_job_store.update(
|
||||
job_id, status="done", detail="Enrichment complete",
|
||||
result={
|
||||
"total_products": result.total_products, "verified": result.verified,
|
||||
"partial": result.partial, "unavailable": result.unavailable,
|
||||
"duration_seconds": result.duration_seconds, "error_count": len(result.errors),
|
||||
"errors": result.errors[:20],
|
||||
},
|
||||
)
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.exception("Nutrition enrichment job %s failed", job_id)
|
||||
nutrition_job_store.update(job_id, status="failed", detail=str(e))
|
||||
# `enrich_all_products` has accepted these three for a while; the API had no
|
||||
# way to pass them, so the endpoint could not do what `scripts/
|
||||
# enrich_nutrition.py` could. In particular there was no way to widen past
|
||||
# ACTIVE_BRANDS, which is what a catalogue-wide run needs.
|
||||
brands: list[str] | None = None
|
||||
include_inactive: bool = False
|
||||
categories: list[str] | None = None
|
||||
|
||||
|
||||
def _run_train_job(job_id: str) -> None:
|
||||
@@ -64,7 +48,15 @@ def enrich_nutrition(payload: EnrichRequest) -> dict:
|
||||
Equivalent to `python scripts/enrich_nutrition.py`."""
|
||||
job = nutrition_job_store.create("enrich")
|
||||
run_in_background(
|
||||
lambda: _run_enrich_job(job.job_id, payload.skip_if_verified, payload.generate_narrative, payload.max_products),
|
||||
lambda: run_enrich_job(
|
||||
job.job_id,
|
||||
skip_if_verified=payload.skip_if_verified,
|
||||
generate_narrative=payload.generate_narrative,
|
||||
max_products=payload.max_products,
|
||||
brands=payload.brands,
|
||||
include_inactive=payload.include_inactive,
|
||||
categories=payload.categories,
|
||||
),
|
||||
name=f"nutrition-enrich-{job.job_id[:8]}",
|
||||
)
|
||||
return {"job_id": job.job_id, "status": job.status}
|
||||
|
||||
@@ -49,7 +49,8 @@ from fastapi.responses import PlainTextResponse
|
||||
|
||||
from app.api.deps import require_permission
|
||||
from app.services.vector_store import _connect
|
||||
from app.services import store_db, nutrition_db
|
||||
from app.services import store_db, nutrition_db, nutrition_scoring
|
||||
from app.services.consumability import Edibility, classify_edibility
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
router = APIRouter(prefix="/upload", tags=["upload"])
|
||||
@@ -566,6 +567,7 @@ async def upload_nutrition_file(file: UploadFile = File(...)) -> Dict[str, Any]:
|
||||
imported_count = 0
|
||||
errors: List[Dict[str, Any]] = []
|
||||
skipped = 0
|
||||
skipped_non_consumable = 0
|
||||
status_counts = {"verified": 0, "partial": 0, "unavailable": 0}
|
||||
|
||||
try:
|
||||
@@ -592,6 +594,20 @@ async def upload_nutrition_file(file: UploadFile = File(...)) -> Dict[str, Any]:
|
||||
image_id = image_id or _slug_id(brand, product_name)
|
||||
category = _opt_str(row, ['category', 'cat'])
|
||||
|
||||
# THE GATE THIS ENDPOINT NEVER HAD. Every other write path into
|
||||
# nutrition_facts refuses non-food; a spreadsheet could walk
|
||||
# straight past them and put a health score on a shampoo, which
|
||||
# then appeared in the public listing endpoints and the
|
||||
# analytics leaderboards. A refused row is reported, not
|
||||
# silently dropped, so the uploader can see it was not imported.
|
||||
verdict = classify_edibility(category or "", product_name or "")
|
||||
if verdict.edibility is Edibility.NON_CONSUMABLE:
|
||||
skipped += 1
|
||||
skipped_non_consumable += 1
|
||||
_record(errors, index,
|
||||
"not a food or drink, so no nutrition was stored ({})".format(verdict.reason))
|
||||
continue
|
||||
|
||||
values = {
|
||||
"calories_kcal": _opt_float(row, ['calories', 'calories_kcal', 'energy']),
|
||||
"protein_g": _opt_float(row, ['protein', 'protein_g']),
|
||||
@@ -617,14 +633,15 @@ async def upload_nutrition_file(file: UploadFile = File(...)) -> Dict[str, Any]:
|
||||
cur.execute(
|
||||
"""
|
||||
INSERT INTO nutrition_facts
|
||||
(brand, image_id, product_name, category, data_status, data_source,
|
||||
(brand, image_id, product_name, category, edibility, data_status, data_source,
|
||||
calories_kcal, protein_g, carbohydrates_g, total_sugar_g, dietary_fiber_g,
|
||||
total_fat_g, sodium_mg, calcium_mg, iron_mg, vitamin_c_mg, updated_at)
|
||||
VALUES (%s, %s, %s, %s, %s, 'excel_upload',
|
||||
VALUES (%s, %s, %s, %s, %s, %s, 'excel_upload',
|
||||
%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, CURRENT_TIMESTAMP)
|
||||
ON CONFLICT (brand, image_id) DO UPDATE SET
|
||||
product_name = COALESCE(EXCLUDED.product_name, nutrition_facts.product_name),
|
||||
category = COALESCE(EXCLUDED.category, nutrition_facts.category),
|
||||
edibility = EXCLUDED.edibility,
|
||||
data_source = 'excel_upload',
|
||||
-- Never downgrade a row a trusted source already verified.
|
||||
data_status = CASE WHEN nutrition_facts.data_status = 'verified'
|
||||
@@ -641,7 +658,7 @@ async def upload_nutrition_file(file: UploadFile = File(...)) -> Dict[str, Any]:
|
||||
vitamin_c_mg = COALESCE(EXCLUDED.vitamin_c_mg, nutrition_facts.vitamin_c_mg),
|
||||
updated_at = CURRENT_TIMESTAMP
|
||||
""",
|
||||
(brand, image_id, product_name, category, data_status,
|
||||
(brand, image_id, product_name, category, verdict.edibility.value, data_status,
|
||||
values["calories_kcal"], values["protein_g"], values["carbohydrates_g"],
|
||||
values["total_sugar_g"], values["dietary_fiber_g"], values["total_fat_g"],
|
||||
values["sodium_mg"], values["calcium_mg"], values["iron_mg"],
|
||||
@@ -652,7 +669,7 @@ async def upload_nutrition_file(file: UploadFile = File(...)) -> Dict[str, Any]:
|
||||
# Only ever built from values this row actually carried. A row
|
||||
# that carried none produces no insight row at all, rather than
|
||||
# a confident-looking one full of defaults.
|
||||
health_score = _opt_float(row, ['health_score', 'nutrition_score', 'score'])
|
||||
sheet_score = _opt_float(row, ['health_score', 'nutrition_score', 'score'])
|
||||
diet_tags_raw = _opt_str(row, ['diet_tags', 'tags', 'diet'])
|
||||
allergens_raw = _opt_str(row, ['allergens', 'allergen'])
|
||||
|
||||
@@ -664,14 +681,47 @@ async def upload_nutrition_file(file: UploadFile = File(...)) -> Dict[str, Any]:
|
||||
# told", and it is what stops an empty list reading as "none".
|
||||
allergen_source = "upload" if allergens is not None else "unavailable"
|
||||
|
||||
positives = []
|
||||
if values["protein_g"] is not None:
|
||||
positives.append(f"Contains {values['protein_g']}g protein per 100g")
|
||||
if values["dietary_fiber_g"] is not None:
|
||||
positives.append(f"Provides {values['dietary_fiber_g']}g dietary fiber")
|
||||
cautions = []
|
||||
if values["total_sugar_g"] is not None:
|
||||
cautions.append(f"{values['total_sugar_g']}g sugar per 100g")
|
||||
# THE SCORE IS COMPUTED HERE, NOT READ OFF THE SHEET.
|
||||
#
|
||||
# This endpoint used to store whatever was in a `health_score`
|
||||
# column. That number shares a column - and every endpoint that
|
||||
# sorts, ranks and filters on it - with scores produced by
|
||||
# `nutrition_scoring.compute_scores` from USDA and Open Food
|
||||
# Facts data. Two scales in one column means the ordering means
|
||||
# nothing: an uploaded 95 outranks a computed 80 without either
|
||||
# number having been measured the same way.
|
||||
#
|
||||
# So the same function scores every row in the system, and the
|
||||
# same rules generate every insight. The uploaded nutrients are
|
||||
# the input; the score is a conclusion drawn from them.
|
||||
facts = dict(values, data_status=data_status,
|
||||
allergens=allergens or [])
|
||||
scores = nutrition_scoring.compute_scores(facts)
|
||||
|
||||
if scores:
|
||||
nutrition_score = scores["nutrition_score"]
|
||||
health_score = scores["health_score"]
|
||||
scoring_version = scores["scoring_version"]
|
||||
else:
|
||||
# Too few nutrients to score fairly. The operator's own
|
||||
# number is kept rather than discarded - they may have it
|
||||
# from a source this sheet does not carry - but
|
||||
# `scoring_version` says where it came from, so nobody
|
||||
# later mistakes it for one of ours.
|
||||
nutrition_score = None
|
||||
health_score = sheet_score
|
||||
scoring_version = "excel_upload"
|
||||
|
||||
positives = nutrition_scoring.generate_positive_insights(facts)
|
||||
cautions = nutrition_scoring.generate_cautions(facts)
|
||||
# Union, not replacement: the rules derive what the numbers
|
||||
# support ("High Protein"), while the sheet may carry what they
|
||||
# cannot ("Vegan" is an ingredient fact, not a nutrient one).
|
||||
derived_tags = nutrition_scoring.classify_diet_tags(facts)
|
||||
if derived_tags or diet_tags:
|
||||
diet_tags = list(dict.fromkeys((diet_tags or []) + derived_tags))
|
||||
if allergens:
|
||||
allergens = nutrition_scoring.normalize_allergens(facts)
|
||||
|
||||
if health_score is not None or diet_tags or allergens or positives or cautions:
|
||||
cur.execute(
|
||||
@@ -680,13 +730,13 @@ async def upload_nutrition_file(file: UploadFile = File(...)) -> Dict[str, Any]:
|
||||
(brand, image_id, data_status, nutrition_score, health_score,
|
||||
scoring_version, positive_insights, nutritional_cautions,
|
||||
diet_tags, allergens, allergen_source, generated_at)
|
||||
VALUES (%s, %s, %s, %s, %s, 'excel_upload', %s, %s, %s, %s, %s, CURRENT_TIMESTAMP)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, CURRENT_TIMESTAMP)
|
||||
ON CONFLICT (brand, image_id) DO UPDATE SET
|
||||
data_status = CASE WHEN nutrition_insights.data_status = 'verified'
|
||||
THEN 'verified' ELSE EXCLUDED.data_status END,
|
||||
nutrition_score = COALESCE(EXCLUDED.nutrition_score, nutrition_insights.nutrition_score),
|
||||
health_score = COALESCE(EXCLUDED.health_score, nutrition_insights.health_score),
|
||||
scoring_version = 'excel_upload',
|
||||
scoring_version = EXCLUDED.scoring_version,
|
||||
positive_insights = COALESCE(EXCLUDED.positive_insights, nutrition_insights.positive_insights),
|
||||
nutritional_cautions = COALESCE(EXCLUDED.nutritional_cautions, nutrition_insights.nutritional_cautions),
|
||||
diet_tags = COALESCE(EXCLUDED.diet_tags, nutrition_insights.diet_tags),
|
||||
@@ -696,8 +746,8 @@ async def upload_nutrition_file(file: UploadFile = File(...)) -> Dict[str, Any]:
|
||||
ELSE EXCLUDED.allergen_source END,
|
||||
generated_at = CURRENT_TIMESTAMP
|
||||
""",
|
||||
(brand, image_id, data_status, health_score, health_score,
|
||||
positives or None, cautions or None,
|
||||
(brand, image_id, data_status, nutrition_score, health_score,
|
||||
scoring_version, positives or None, cautions or None,
|
||||
diet_tags, allergens, allergen_source)
|
||||
)
|
||||
|
||||
@@ -715,6 +765,11 @@ async def upload_nutrition_file(file: UploadFile = File(...)) -> Dict[str, Any]:
|
||||
file.filename, len(df), imported_count, errors, skipped,
|
||||
"nutritional intelligence items",
|
||||
data_status_counts=status_counts,
|
||||
# Reported separately from `rows_skipped`, which otherwise reads as "the
|
||||
# sheet was malformed". These rows were well-formed and deliberately not
|
||||
# imported. Named to match `EnrichmentResult.skipped_non_consumable`, so
|
||||
# both write paths report a refusal the same way.
|
||||
skipped_non_consumable=skipped_non_consumable,
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -918,6 +918,16 @@ def _process_upload(filename: str, content: bytes) -> Dict[str, Any]:
|
||||
|
||||
logger.info("Upload '%s': saved %d, failed %d, skipped %d blank row(s)",
|
||||
filename, body["added_count"], body["error_count"], skipped_blank_rows)
|
||||
|
||||
# Score the consumables that were just stored. `outcome.brands` is already
|
||||
# the set of brands actually written, so nothing new is tracked to answer
|
||||
# this. Returns None (and logs) rather than raising if scoring cannot be
|
||||
# queued - the products are saved either way, and this response has already
|
||||
# been shaped to say so.
|
||||
from app.services.nutrition_autoenrich import submit_enrichment_for_brands
|
||||
|
||||
body["nutrition_job_id"] = submit_enrichment_for_brands(
|
||||
outcome.brands, source="upload of '{}'".format(filename))
|
||||
return body
|
||||
|
||||
|
||||
|
||||
@@ -208,6 +208,12 @@ class BatchManifest:
|
||||
# Defaults to INPROCESS so every manifest written before this existed, and
|
||||
# every batch an admin uploads directly, keeps behaving exactly as it did.
|
||||
runner: str = RUNNER_INPROCESS
|
||||
# The nutrition-enrichment job queued when this batch finished, pollable at
|
||||
# GET /api/admin/nutrition-intelligence/jobs/{job_id}. None when scoring is
|
||||
# switched off, when the batch produced no brands, or - importantly - for
|
||||
# every manifest written before this field existed, which is why `from_dict`
|
||||
# reads it with a default instead of requiring it.
|
||||
nutrition_job_id: Optional[str] = None
|
||||
files: List[BatchFile] = field(default_factory=list)
|
||||
|
||||
# -- derived, recomputed rather than stored, so they cannot drift ---------
|
||||
@@ -282,6 +288,7 @@ class BatchManifest:
|
||||
"detail": self.detail,
|
||||
"submitted_by": self.submitted_by,
|
||||
"runner": self.runner,
|
||||
"nutrition_job_id": self.nutrition_job_id,
|
||||
"files_total": self.files_total,
|
||||
"files_done": self.files_done,
|
||||
"files_failed": self.files_failed,
|
||||
@@ -316,6 +323,7 @@ class BatchManifest:
|
||||
detail=raw.get("detail"),
|
||||
submitted_by=raw.get("submitted_by"),
|
||||
runner=raw.get("runner") or RUNNER_INPROCESS,
|
||||
nutrition_job_id=raw.get("nutrition_job_id"),
|
||||
files=files,
|
||||
)
|
||||
|
||||
@@ -675,6 +683,25 @@ def run_batch(
|
||||
on_change(manifest)
|
||||
|
||||
manifest.settle()
|
||||
|
||||
# Score what was just ingested. Submitted here, after the last file has
|
||||
# settled, rather than per file: one job for the whole batch means one
|
||||
# `skip_if_verified` sweep per brand instead of one per spreadsheet, and a
|
||||
# batch of twenty files for the same brand does not queue twenty jobs.
|
||||
#
|
||||
# This runs on the batch worker's thread, so it must not do the enrichment
|
||||
# itself - `submit_enrichment_for_brands` only starts a separate thread and
|
||||
# returns. It also swallows its own failures: the products ARE ingested by
|
||||
# this point, and a scoring step that could not start is not a reason to
|
||||
# tell the operator their catalogue import failed.
|
||||
try:
|
||||
from app.services.nutrition_autoenrich import submit_enrichment_for_brands
|
||||
|
||||
manifest.nutrition_job_id = submit_enrichment_for_brands(
|
||||
manifest.brands(), source="batch {}".format(batch_id))
|
||||
except Exception: # noqa: BLE001
|
||||
logger.exception("Batch %s: could not queue nutrition enrichment", batch_id)
|
||||
|
||||
write_manifest(manifest)
|
||||
on_change(manifest)
|
||||
return manifest
|
||||
|
||||
@@ -349,6 +349,36 @@ USE_PLAYWRIGHT_FALLBACK = _bool("USE_PLAYWRIGHT_FALLBACK", "true")
|
||||
# out 1x1 tracking pixels / broken placeholder images)
|
||||
MIN_IMAGE_BYTES = int(os.getenv("MIN_IMAGE_BYTES", "3000"))
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# USDA FoodData Central - nutrition for loose, unbranded commodities
|
||||
# ---------------------------------------------------------------------------
|
||||
# Open Food Facts catalogues packaged products and has no entry for a raw
|
||||
# apple, which is why every fresh-produce row had no nutrition at all. USDA's
|
||||
# Foundation Foods and SR Legacy datasets are laboratory analyses of raw
|
||||
# commodities, published per 100 g of edible portion.
|
||||
#
|
||||
# NO KEY IS REQUIRED for normal operation: `nutrition_usda_service` reads a
|
||||
# snapshot built from USDA's open bulk download, so an enrichment run makes zero
|
||||
# outbound USDA calls. A key is only needed to look up an id the snapshot lacks,
|
||||
# or to rebuild the snapshot from the live API.
|
||||
#
|
||||
# Plain os.getenv rather than `_require(..., feature_flag=...)` on purpose: a
|
||||
# fresh checkout with no key must still import, or the whole test suite fails at
|
||||
# collection time.
|
||||
USE_USDA_FDC = _bool("USE_USDA_FDC", "true")
|
||||
USDA_FDC_API_KEY = os.getenv("USDA_FDC_API_KEY", "").strip()
|
||||
|
||||
# Score newly uploaded products automatically, instead of waiting for somebody
|
||||
# to remember to POST /api/admin/nutrition-intelligence/enrich. Runs as a
|
||||
# background job AFTER the ingestion batch finishes, never inside it - see the
|
||||
# comment at the submission site in `app/core/batch_ingest.py`.
|
||||
#
|
||||
# The off switch exists because this is the one part of ingestion that makes
|
||||
# outbound calls per product: an operator loading a very large catalogue on a
|
||||
# metered connection, or re-running an import they intend to score later in one
|
||||
# controlled pass, needs a way to say "not now" without a code change.
|
||||
AUTO_ENRICH_ON_UPLOAD = _bool("AUTO_ENRICH_ON_UPLOAD", "true")
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Store-catalog enrichment pipeline (app/core/store_catalog_pipeline.py)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -137,9 +137,25 @@ QUERY_STOP_WORDS = {
|
||||
}
|
||||
|
||||
|
||||
def _find_matches(text: str) -> List[Tuple[str, str, int]]:
|
||||
def _find_matches(text: str, exact_only: bool = False) -> List[Tuple[str, str, int]]:
|
||||
"""Return (category, matched_keyword, keyword_length) for every keyword
|
||||
found as a whole word/phrase in `text` (case-insensitive)."""
|
||||
found as a whole word/phrase in `text` (case-insensitive).
|
||||
|
||||
`exact_only` suppresses the fuzzy fallback below. It exists because that
|
||||
fallback is tuned for USER QUERIES, where a near-miss is a typo worth
|
||||
recovering, and is actively harmful when the text is a PRODUCT TITLE, where
|
||||
a near-miss is a different product. The case that forced it:
|
||||
|
||||
detect_category_from_text("Colgate-Palmolive Palmolive Naturals")
|
||||
-> "Chocolates"
|
||||
|
||||
because "colgate" scores 0.8 against the misspelling keyword "choclate".
|
||||
A toothpaste brand classified as chocolate is harmless in a search box and
|
||||
is not harmless in `consumability`, which would then hand a bar of soap a
|
||||
health score. Callers deciding something about a specific product pass
|
||||
exact_only=True; the RAG/query path keeps the fuzzy behaviour it was
|
||||
written for.
|
||||
"""
|
||||
if not text:
|
||||
return []
|
||||
lower = text.lower()
|
||||
@@ -152,7 +168,7 @@ def _find_matches(text: str) -> List[Tuple[str, str, int]]:
|
||||
matches.append((category, kw, len(kw)))
|
||||
|
||||
# Fuzzy matching fallback if exact word search found nothing
|
||||
if not matches:
|
||||
if not matches and not exact_only:
|
||||
import difflib
|
||||
words = re.findall(r"\b[a-z]{4,}\b", lower)
|
||||
for entry in CATEGORY_REGISTRY:
|
||||
@@ -169,15 +185,19 @@ def _find_matches(text: str) -> List[Tuple[str, str, int]]:
|
||||
return matches
|
||||
|
||||
|
||||
def detect_category_from_text(text: str) -> Optional[str]:
|
||||
def detect_category_from_text(text: str, exact_only: bool = False) -> Optional[str]:
|
||||
"""Infer a single canonical category from free text (typically a user
|
||||
query), or None if no category-identifying keyword is present.
|
||||
|
||||
When multiple categories match, the one listed earliest in
|
||||
`CATEGORY_REGISTRY` wins (see module docstring); ties within that are
|
||||
broken by the longest matched keyword.
|
||||
|
||||
Pass `exact_only=True` when `text` is a product title rather than a user
|
||||
query - see `_find_matches` for the Colgate/chocolate case that makes the
|
||||
distinction matter.
|
||||
"""
|
||||
matches = _find_matches(text)
|
||||
matches = _find_matches(text, exact_only=exact_only)
|
||||
if not matches:
|
||||
return None
|
||||
|
||||
|
||||
382
app/services/consumability.py
Normal file
382
app/services/consumability.py
Normal file
@@ -0,0 +1,382 @@
|
||||
"""
|
||||
Is this product something a person eats or drinks?
|
||||
|
||||
WHY THIS FILE EXISTS
|
||||
--------------------
|
||||
The nutrition module's contract is that every number it publishes traces to a
|
||||
verified source (see `nutrition_data_service`'s docstring). That contract says
|
||||
nothing about products which have no nutrition panel *at all*, and the only
|
||||
guard that existed was a seventeen-word substring list:
|
||||
|
||||
nutrition_data_service.NON_FOOD_KEYWORDS = ("soap", "detergent", "shampoo", ...)
|
||||
|
||||
matched against `f"{title} {category}".lower()`. It missed every category name it
|
||||
was not literally spelled with. Measured against the live catalogue, these rows
|
||||
carry a health score today:
|
||||
|
||||
Colgate-Palmolive Palmolive Naturals General score present
|
||||
Cavinkare Nyle / Nature's Hair Care score present
|
||||
P&G Pantene Hair Care score present
|
||||
Godrej Hit Spray Personal Care - Mosquito 291 kcal (!)
|
||||
|
||||
A mosquito repellent with a calorie count is not a cosmetic defect. It is the
|
||||
nutrition module asserting a fact about a product it has no business describing,
|
||||
and a shopper has no way to tell that the number is meaningless.
|
||||
|
||||
Being a substring test, that list also has the opposite failure: "soap" matches
|
||||
*soapnut* (reetha), a real commodity. Matching here is whole-word.
|
||||
|
||||
WHY THREE STATES AND NOT A BOOLEAN
|
||||
----------------------------------
|
||||
`is_consumable` and `is_non_consumable` are NOT inverses, and that is the single
|
||||
most important thing in this file.
|
||||
|
||||
Two callers ask opposite questions of the same fact:
|
||||
|
||||
the WRITE gate - "may I attach nutrition to this?" unknown => NO
|
||||
the DELETE gate - "may I destroy this row?" unknown => NO
|
||||
|
||||
Collapsing them into one boolean makes whichever caller loses the coin-toss act
|
||||
destructively on a guess: a single `not is_consumable()` in the purge script
|
||||
would delete every row the classifier merely failed to recognise. So the engine
|
||||
returns `CONSUMABLE | NON_CONSUMABLE | UNKNOWN`, and both booleans are positive
|
||||
tests that return False for UNKNOWN.
|
||||
|
||||
WHY A LAYERED DECISION AND NOT ONE KEYWORD LIST
|
||||
-----------------------------------------------
|
||||
The catalogue's category strings are not one vocabulary. Four maps have drifted
|
||||
apart - CATEGORY_REGISTRY (31 names), HSN_GST_TABLE (60), CATEGORY_UNIT_TYPE
|
||||
(~70) and CATEGORY_TYPE_WORDS - and the live tables hold 66 distinct values
|
||||
including "General" (40 rows) and the bare strings "1".."5" (17 rows) left by a
|
||||
bad import. Any single list is stale the moment somebody adds a category.
|
||||
|
||||
1. An explicit per-category verdict, hand-set, for every string the live
|
||||
catalogue actually contains.
|
||||
2. Failing that, the HSN chapter the category resolves to. The tariff's own
|
||||
classification, already maintained here for tax purposes.
|
||||
3. Failing that, the product title, through the same commodity lexicon and
|
||||
category detector the ingestion pipeline uses.
|
||||
4. Failing that, UNKNOWN - which enriches nothing and deletes nothing.
|
||||
|
||||
WHY THE HSN RANGE IS NOT SIMPLY 01-24
|
||||
--------------------------------------
|
||||
"Chapters 1 to 24 are the food chapters" is the obvious rule and it is wrong at
|
||||
exactly the case this project cares about. Chapter 06 is live plants and cut
|
||||
flowers, and `Flowers` is a live category here with 14 rows that reach the same
|
||||
Own Products table as the vegetables. A naive range would score a jasmine
|
||||
garland. Excluded for the same reason: 05 (inedible animal products), 14
|
||||
(vegetable plaiting materials), 23 (animal feed) and 24 (tobacco - consumed, but
|
||||
it publishes no nutrition panel).
|
||||
|
||||
EDGE CASES, AND WHY THEY WENT THE WAY THEY DID
|
||||
----------------------------------------------
|
||||
Flowers False. Sold loose beside the vegetables and routed by the same
|
||||
produce lexicon, but a garland is not food.
|
||||
Oral Care False. Toothpaste goes in the mouth and is spat out; it carries
|
||||
no nutrition panel, and it is Open *Beauty* Facts that matches it.
|
||||
Health Care - False. A cough syrup is ingested but is regulated as a drug and
|
||||
Cold & Cough / publishes dosage, not nutrition. Scoring it would be the most
|
||||
Digestive dangerous error available here.
|
||||
Baby Care MIXED, so it defers to the title. The audit found 21 live rows
|
||||
Health Care - that are Nestle Cerelac, Nan Pro and Lactogen sitting beside a
|
||||
Ayurvedic bottle of baby oil; Dabur Chyawanprash sits beside cough syrup.
|
||||
Infant formula is among the most nutrition-labelled food sold in
|
||||
India, so a blanket "not food" here would have been a worse
|
||||
defect than the one this module fixes.
|
||||
Health Drinks True. Horlicks, Boost and Complan publish a real nutrition
|
||||
panel, notwithstanding the word "Health".
|
||||
Household - False, and listed explicitly so it can never be swept in by
|
||||
Lamp Oil "Cooking Oils" being True. See `_hsn_verdict` for the second
|
||||
guard on the same trap.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from enum import Enum
|
||||
from typing import Dict, Optional, Tuple
|
||||
|
||||
from app.services.category_registry import _normalize, detect_category_from_text
|
||||
|
||||
|
||||
class Edibility(str, Enum):
|
||||
CONSUMABLE = "consumable"
|
||||
NON_CONSUMABLE = "non_consumable"
|
||||
UNKNOWN = "unknown"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EdibilityVerdict:
|
||||
edibility: Edibility
|
||||
reason: str # human-readable; the purge audit prints this
|
||||
signal: str # category_map | hsn_chapter | title_lexicon | title_keyword | none
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. Explicit verdicts
|
||||
# ---------------------------------------------------------------------------
|
||||
# Keyed on the NORMALIZED category ("Pulses, Grains & Spices" -> "pulses grains
|
||||
# and spices") so a stored string differing only in punctuation or case still
|
||||
# lands here rather than falling through to the HSN guess. `_normalize` is
|
||||
# imported rather than reimplemented: a fifth normalizer that disagreed with the
|
||||
# other four is exactly how this area got into trouble.
|
||||
_CONSUMABLE_CATEGORIES: Tuple[str, ...] = (
|
||||
# Dairy and dairy-adjacent
|
||||
"Dairy", "Cheese", "Dairy - Desserts", "Ice Cream",
|
||||
# Drinks
|
||||
"Beverages", "Tea & Coffee", "Health Drinks", "Food & Beverages",
|
||||
# Fresh, loose goods sold by weight or by the piece
|
||||
"Fruits & Vegetables", "Fresh Herbs & Greens", "Fish & Seafood", "Eggs",
|
||||
# Bakery and biscuit
|
||||
"Biscuits & Cookies", "Biscuits", "Crackers", "Rusk", "Cakes & Muffins",
|
||||
"Bakery & Breads", "Breakfast Cereal",
|
||||
# Confectionery
|
||||
"Chocolates", "Candy & Confectionery",
|
||||
# Savoury
|
||||
"Snacks", "Namkeen", "Ready to Eat", "Noodles & Instant Food",
|
||||
"Pasta & Noodles",
|
||||
# Staples, pulses, spices
|
||||
"Atta & Staples", "Staples", "Flour & Grains", "Salt & Staples",
|
||||
"Sugar & Jaggery", "Pulses, Grains & Spices", "Spices & Masalas",
|
||||
"Cooking Oils", "Dry Fruits & Nuts", "Millets", "Rice & Pulses",
|
||||
# Prepared foods
|
||||
"Food - Mixes", "Food - Spreads", "Food - Soups & Sauces",
|
||||
"Pickles & Chutneys", "Health Foods", "Sweets",
|
||||
)
|
||||
|
||||
_NON_CONSUMABLE_CATEGORIES: Tuple[str, ...] = (
|
||||
# Personal care
|
||||
"Hair Care", "Skin Care", "Skin & Bath Care", "Bath Soap", "Beauty Care",
|
||||
"Oral Care", "Fragrance & Deodorants", "Men's Grooming",
|
||||
"Feminine Hygiene", "Personal Care",
|
||||
# Household
|
||||
"Detergents & Fabric Care", "Dishwash", "Household Cleaning",
|
||||
"Household - Agarbatti", "Household - Lamp Oil", "Household - Air Freshener",
|
||||
"Personal Care - Mosquito Repellent",
|
||||
# Ingested, but regulated as medicines: they publish dosage, not nutrition
|
||||
"Health Care - Cold & Cough", "Health Care - Digestive",
|
||||
"Health Care - Antiseptic", "Health Care - First Aid",
|
||||
# Not food, despite arriving through the produce lexicon
|
||||
"Flowers",
|
||||
)
|
||||
|
||||
# Categories holding BOTH food and non-food, where a single verdict is simply
|
||||
# wrong. Deferred to the title, exactly like an uninformative category.
|
||||
#
|
||||
# Found by running the purge audit before deleting anything - which is what that
|
||||
# dry run is for. "Baby Care" held 21 live rows, and they are Nestle Cerelac,
|
||||
# Nan Pro and Lactogen alongside a bottle of baby oil. Infant formula and baby
|
||||
# cereal are among the most heavily nutrition-labelled products sold in India;
|
||||
# refusing them a score would have been a worse defect than the one being fixed.
|
||||
#
|
||||
# "Health Care - Ayurvedic" is the same shape: Dabur Chyawanprash is eaten by
|
||||
# the spoonful and carries a nutrition panel, while a cough syrup does not.
|
||||
_MIXED_CATEGORIES = frozenset(
|
||||
{_normalize("Baby Care"), _normalize("Health Care - Ayurvedic")}
|
||||
)
|
||||
|
||||
CATEGORY_VERDICTS: Dict[str, Edibility] = {
|
||||
**{_normalize(c): Edibility.CONSUMABLE for c in _CONSUMABLE_CATEGORIES},
|
||||
**{_normalize(c): Edibility.NON_CONSUMABLE for c in _NON_CONSUMABLE_CATEGORIES},
|
||||
}
|
||||
|
||||
# Category strings carrying no information. Treated as MISSING (ask the title),
|
||||
# not as UNKNOWN (refuse) - 40 live rows sit in "General" and many are real food.
|
||||
# The bare numerics "1".."5" are import damage on 17 rows; repairing those is a
|
||||
# catalogue fix, not a consumability rule, so they are recognised and reported
|
||||
# rather than accommodated.
|
||||
_UNINFORMATIVE_CATEGORIES = frozenset(
|
||||
{"", "general", "uncategorized", "uncategorised", "other", "others",
|
||||
"misc", "miscellaneous", "unknown", "na", "none"}
|
||||
)
|
||||
|
||||
_JUNK_CATEGORY_RE = re.compile(r"^\d+$")
|
||||
|
||||
# HSN chapters that are food and drink, per the customs tariff. Deliberately NOT
|
||||
# `range(1, 25)` - see the module docstring for why 05, 06, 14, 23 and 24 are out.
|
||||
_HSN_FOOD_CHAPTERS = frozenset(
|
||||
{1, 2, 3, 4, 7, 8, 9, 10, 11, 12, 13, 15, 16, 17, 18, 19, 20, 21, 22}
|
||||
)
|
||||
# Cosmetics, soap, pharma, insecticide, razors, hygiene articles.
|
||||
_HSN_NON_FOOD_CHAPTERS = frozenset({5, 6, 14, 23, 24, 28, 29, 30, 33, 34, 38, 82, 96})
|
||||
|
||||
# Title words that settle an uninformative category on their own. Deliberately
|
||||
# short: the commodity lexicon and the category detector do the real work, and a
|
||||
# longer list here would start overriding them.
|
||||
_NON_FOOD_TITLE_WORDS = frozenset({
|
||||
"soap", "detergent", "shampoo", "conditioner", "toothpaste", "toothbrush",
|
||||
"mouthwash", "deodorant", "perfume", "cosmetic", "lipstick", "kajal",
|
||||
"talc", "lotion", "moisturizer", "moisturiser", "sunscreen", "facewash",
|
||||
"handwash", "sanitizer", "sanitiser", "diaper", "sanitary", "napkin",
|
||||
"razor", "shaving", "cleaner", "disinfectant", "phenyl", "bleach",
|
||||
"repellent", "mosquito", "agarbatti", "incense", "camphor", "matchbox",
|
||||
"battery", "bulb", "candle", "polish", "freshener", "dishwash",
|
||||
})
|
||||
|
||||
# Title words that positively identify FOOD, consulted before the non-food list.
|
||||
# These exist for the mixed categories above: a product line name is the only
|
||||
# signal a title like "Nestle Cerelac 125g" carries, since the commodity lexicon
|
||||
# knows nothing of it. Naming specific ranges is consistent with how this
|
||||
# codebase already resolves ambiguity (BRAND_ALIASES, and the "bikis" keyword
|
||||
# added to Biscuits & Cookies so Britannia Milk Bikis is not filed as Dairy).
|
||||
_FOOD_TITLE_WORDS = frozenset({
|
||||
# infant and toddler nutrition
|
||||
"cerelac", "lactogen", "nangrow", "nan", "farex", "dexolac", "nusobee",
|
||||
"formula", "infant", "weaning", "porridge", "cereal", "cereals",
|
||||
# ayurvedic preparations eaten as food
|
||||
"chyawanprash", "chyavanprash", "honey", "malt",
|
||||
})
|
||||
|
||||
_WORD_RE = re.compile(r"[a-z0-9]+")
|
||||
|
||||
|
||||
def _words(text: str) -> frozenset:
|
||||
"""Whole words, lowercased. Whole-word matching is the point: the old
|
||||
substring gate classified *soapnut* (reetha) as a soap."""
|
||||
return frozenset(_WORD_RE.findall((text or "").lower()))
|
||||
|
||||
|
||||
def _hsn_verdict(category: str) -> Optional[Edibility]:
|
||||
"""The verdict implied by the HSN chapter this category maps to.
|
||||
|
||||
Reads `HSN_GST_TABLE` directly rather than calling `resolve_hsn_gst`, and
|
||||
that is deliberate. `resolve_hsn_gst` falls through to `_KEYWORD_FALLBACKS`,
|
||||
which scans `f"{product_title} {category}"` and contains a greedy
|
||||
`("oil", "1517")` entry - enough to classify "Household - Lamp Oil" as an
|
||||
edible oil. Reading the table means only an exact category name can match,
|
||||
so the fallbacks can never fire here at all.
|
||||
|
||||
Imported lazily: the hsn_gst package pulls in the enrichment machinery, and
|
||||
the nutrition path should not pay for that import on every call.
|
||||
"""
|
||||
from app.services.enrichment.hsn_gst.models import HSN_GST_TABLE
|
||||
|
||||
target = _normalize(category)
|
||||
for name, entry in HSN_GST_TABLE.items():
|
||||
if _normalize(name) != target:
|
||||
continue
|
||||
try:
|
||||
chapter = int(str(entry[0])[:2])
|
||||
except (TypeError, ValueError, IndexError):
|
||||
return None
|
||||
if chapter in _HSN_FOOD_CHAPTERS:
|
||||
return Edibility.CONSUMABLE
|
||||
if chapter in _HSN_NON_FOOD_CHAPTERS:
|
||||
return Edibility.NON_CONSUMABLE
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def classify_edibility(category: Optional[str], title: str = "") -> EdibilityVerdict:
|
||||
"""The full verdict, with the reason that produced it.
|
||||
|
||||
The reason string is what the purge audit prints, so a row's fate can be
|
||||
argued with rather than taken on trust.
|
||||
"""
|
||||
normalized = _normalize(category)
|
||||
informative = (
|
||||
normalized
|
||||
and normalized not in _UNINFORMATIVE_CATEGORIES
|
||||
and normalized not in _MIXED_CATEGORIES
|
||||
and not _JUNK_CATEGORY_RE.match(normalized)
|
||||
)
|
||||
|
||||
if informative:
|
||||
verdict = CATEGORY_VERDICTS.get(normalized)
|
||||
if verdict is not None:
|
||||
return EdibilityVerdict(
|
||||
verdict,
|
||||
f"category {category!r} is listed as {verdict.value}",
|
||||
"category_map",
|
||||
)
|
||||
|
||||
hsn = _hsn_verdict(category or "")
|
||||
if hsn is not None:
|
||||
return EdibilityVerdict(
|
||||
hsn,
|
||||
f"category {category!r} maps to an HSN chapter that is "
|
||||
+ ("food/beverage" if hsn is Edibility.CONSUMABLE else "not food"),
|
||||
"hsn_chapter",
|
||||
)
|
||||
|
||||
# The category told us nothing usable. Ask the title, through the same two
|
||||
# resolvers the ingestion pipeline uses, so a row classified here agrees
|
||||
# with the row the pipeline would have written.
|
||||
text = (title or "").strip()
|
||||
if text:
|
||||
words = _words(text)
|
||||
|
||||
food_hit = words & _FOOD_TITLE_WORDS
|
||||
if food_hit:
|
||||
return EdibilityVerdict(
|
||||
Edibility.CONSUMABLE,
|
||||
f"title {title!r} names a food product ({sorted(food_hit)[0]!r})",
|
||||
"title_keyword",
|
||||
)
|
||||
|
||||
hit = words & _NON_FOOD_TITLE_WORDS
|
||||
if hit:
|
||||
return EdibilityVerdict(
|
||||
Edibility.NON_CONSUMABLE,
|
||||
f"title {title!r} names a non-food article ({sorted(hit)[0]!r})",
|
||||
"title_keyword",
|
||||
)
|
||||
|
||||
from app.services.generic_products import canonical_category
|
||||
|
||||
# `detect_category_from_text` is pinned to exact_only. Its fuzzy
|
||||
# fallback scores "colgate" at 0.8 against the misspelling keyword
|
||||
# "choclate", so without this a tube of toothpaste reads as Chocolates
|
||||
# and earns a health score. Verified: that is the live behaviour for
|
||||
# "Colgate-Palmolive Palmolive Naturals", whose category is "General".
|
||||
for resolver in (
|
||||
canonical_category,
|
||||
lambda t: detect_category_from_text(t, exact_only=True),
|
||||
):
|
||||
detected = resolver(text)
|
||||
if not detected:
|
||||
continue
|
||||
verdict = CATEGORY_VERDICTS.get(_normalize(detected))
|
||||
if verdict is not None:
|
||||
return EdibilityVerdict(
|
||||
verdict,
|
||||
f"title {title!r} reads as {detected!r}, which is {verdict.value}",
|
||||
"title_lexicon",
|
||||
)
|
||||
|
||||
return EdibilityVerdict(
|
||||
Edibility.UNKNOWN,
|
||||
f"neither category {category!r} nor title {title!r} identifies this "
|
||||
"product; refusing to guess",
|
||||
"none",
|
||||
)
|
||||
|
||||
|
||||
def is_consumable(category: Optional[str], title: str = "") -> bool:
|
||||
"""True only when this is positively something a person eats or drinks.
|
||||
|
||||
The WRITE gate. False for UNKNOWN, so an unrecognised product gets no
|
||||
nutrition row and no health score - the same asymmetry
|
||||
`nutrition_data_service` already chose when it returns
|
||||
`data_status="unavailable"` rather than zeroes.
|
||||
"""
|
||||
return classify_edibility(category, title).edibility is Edibility.CONSUMABLE
|
||||
|
||||
|
||||
def is_non_consumable(category: Optional[str], title: str = "") -> bool:
|
||||
"""True only when this is positively NOT food.
|
||||
|
||||
The DELETE gate, and deliberately not `not is_consumable(...)`. False for
|
||||
UNKNOWN, so the purge script can never destroy a row it merely failed to
|
||||
recognise. See the module docstring.
|
||||
"""
|
||||
return classify_edibility(category, title).edibility is Edibility.NON_CONSUMABLE
|
||||
|
||||
|
||||
def is_junk_category(category: Optional[str]) -> bool:
|
||||
"""True for the bare numeric category strings left by a bad import.
|
||||
|
||||
Reported by the purge audit so the 17 affected rows get repaired in the
|
||||
catalogue rather than worked around here.
|
||||
"""
|
||||
return bool(_JUNK_CATEGORY_RE.match(_normalize(category)))
|
||||
1221
app/services/data/produce_reference.json
Normal file
1221
app/services/data/produce_reference.json
Normal file
File diff suppressed because it is too large
Load Diff
53605
app/services/data/usda_snapshot.json
Normal file
53605
app/services/data/usda_snapshot.json
Normal file
File diff suppressed because it is too large
Load Diff
@@ -48,6 +48,7 @@ from typing import Optional, List
|
||||
import requests
|
||||
from urllib.parse import urlparse
|
||||
import json
|
||||
import re
|
||||
import logging
|
||||
import subprocess
|
||||
import sys
|
||||
@@ -162,7 +163,8 @@ def find_product_quantity_openfacts(title: str, brand: Optional[str] = None) ->
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Wikimedia Commons - free/open media repository, no API key
|
||||
# ---------------------------------------------------------------------------
|
||||
def find_images_wikimedia(title: str, brand: Optional[str] = None, max_results: int = 20) -> list:
|
||||
def find_images_wikimedia(title: str, brand: Optional[str] = None, max_results: int = 20,
|
||||
produce: bool = False) -> list:
|
||||
"""Search Wikimedia Commons (the open media library behind Wikipedia)
|
||||
for product/packaging photos. Good secondary source for established
|
||||
brands; complements Open*Facts which leans more food/grocery."""
|
||||
@@ -180,6 +182,14 @@ def find_images_wikimedia(title: str, brand: Optional[str] = None, max_results:
|
||||
# MediaWiki's standard search syntax including the minus (-) prefix
|
||||
# for terms to exclude.
|
||||
exclusion_terms = "-car -vehicle -automotive -motorsport -racing -motorcycle -bike -people -person -portrait -animal -pet -dog -cat -bird -fish -landscape -nature -tour -travel -building -architecture -sport -game -flower -rose -floral -petal -bouquet -botanical -plant -garden -tree -herb"
|
||||
if produce:
|
||||
# Every one of those exclusions describes what a fruit, a vegetable
|
||||
# or a fish ACTUALLY IS. Searching Commons for a tomato while
|
||||
# excluding "-plant -garden -nature", or for mackerel while
|
||||
# excluding "-fish -animal", asks the index to rule out the answer.
|
||||
# The exclusions exist to keep a BRANDED search off unrelated
|
||||
# subjects, which is not the problem here.
|
||||
exclusion_terms = "-logo -advertisement -packaging"
|
||||
search_query = f"{query} {exclusion_terms} filetype:bitmap"
|
||||
resp = requests.get(
|
||||
"https://commons.wikimedia.org/w/api.php",
|
||||
@@ -402,6 +412,23 @@ def _dedupe(urls: list) -> list:
|
||||
# that image search APIs tend to confuse with non-product content.
|
||||
# Keyed by the ambiguous word (lowercase); value is the context term
|
||||
# to append to the search query.
|
||||
#
|
||||
# EVERY KEY HERE ASSUMES THE WORD IS A PACKAGED BRAND, NOT A COMMODITY.
|
||||
# The table was written for Cadbury Perk and Britannia Tiger, where "perk" and
|
||||
# "tiger" really do need steering away from perks and big cats. For the
|
||||
# Own Products bucket the word IS the literal item, so the hint inverts the
|
||||
# search: `"apple" -> "fruit juice"` is why a search for the fruit returned
|
||||
# juice cartons and milkshakes, and why the stored image for `Apple` was a
|
||||
# 2-litre juice bottle.
|
||||
#
|
||||
# Callers handling loose produce pass `produce=True`, which skips this table
|
||||
# entirely. Do not "fix" an entry by making it more specific - the whole table
|
||||
# is wrong in that context, not any one row of it.
|
||||
# Kept as a named constant rather than an inline literal: this pattern has to
|
||||
# survive being edited by tooling that mangles backslash escapes, and a silent
|
||||
# corruption here turns whole-word matching back into no matching at all.
|
||||
_WORD_BOUNDARY = chr(92) + "b"
|
||||
|
||||
_AMBIQUITY_HINTS: dict[str, str] = {
|
||||
"perk": "chocolate wafer",
|
||||
"crunch": "chocolate wafer",
|
||||
@@ -438,7 +465,8 @@ _AMBIQUITY_HINTS: dict[str, str] = {
|
||||
}
|
||||
|
||||
|
||||
def _clean_search_title(title: str, brand: str | None = None) -> str:
|
||||
def _clean_search_title(title: str, brand: str | None = None,
|
||||
produce: bool = False) -> str:
|
||||
"""Strip redundant brand prefix from title and add context hints for
|
||||
ambiguous product names that confuse image search APIs.
|
||||
|
||||
@@ -448,6 +476,9 @@ def _clean_search_title(title: str, brand: str | None = None) -> str:
|
||||
|
||||
_clean_search_title("Britannia Good Day Biscuits", "Britannia")
|
||||
-> "Good Day Biscuits"
|
||||
|
||||
`produce=True` suppresses the hint table entirely - see `_AMBIQUITY_HINTS`
|
||||
for why it is exactly backwards for a raw commodity.
|
||||
"""
|
||||
clean = (title or "").strip()
|
||||
if brand:
|
||||
@@ -458,10 +489,25 @@ def _clean_search_title(title: str, brand: str | None = None) -> str:
|
||||
if not clean:
|
||||
clean = (title or "").strip()
|
||||
|
||||
# Append context hint for any ambiguous word in the cleaned title
|
||||
if produce:
|
||||
return clean
|
||||
|
||||
# Append a context hint for an ambiguous word in the cleaned title.
|
||||
#
|
||||
# MATCHED AS A WHOLE WORD, not as a substring. The substring version pulled
|
||||
# hints into every title that merely CONTAINED a keyword:
|
||||
#
|
||||
# Pineapple -> "Pineapple fruit juice" (apple)
|
||||
# Custard Apple -> "Custard Apple fruit juice" (apple)
|
||||
# Buttermilk -> "Buttermilk chocolate dairy" (milk)
|
||||
# Baby Corn -> "Baby Corn snack flakes" (corn)
|
||||
# Eggs White -> "Eggs White toothpaste dental" (white)
|
||||
#
|
||||
# Multi-word keys ("good day", "5 star") still need a phrase search, so the
|
||||
# boundary is applied around the whole key rather than per token.
|
||||
clean_lower = clean.lower()
|
||||
for word, hint in _AMBIQUITY_HINTS.items():
|
||||
if word in clean_lower:
|
||||
if re.search(_WORD_BOUNDARY + re.escape(word) + _WORD_BOUNDARY, clean_lower):
|
||||
clean = f"{clean} {hint}"
|
||||
break
|
||||
|
||||
@@ -472,14 +518,16 @@ def _clean_search_title(title: str, brand: str | None = None) -> str:
|
||||
# Public API (same signatures as before, so catalog_engine.py and
|
||||
# downstream callers don't need to change)
|
||||
# ---------------------------------------------------------------------------
|
||||
def find_image_url(title: str, brand: Optional[str] = None, country_hint: Optional[str] = None) -> Optional[str]:
|
||||
def find_image_url(title: str, brand: Optional[str] = None, country_hint: Optional[str] = None,
|
||||
produce: bool = False) -> Optional[str]:
|
||||
"""Get a single working, validated image URL."""
|
||||
urls = find_all_image_urls(title, brand, country_hint)
|
||||
urls = find_all_image_urls(title, brand, country_hint, produce=produce)
|
||||
return urls[0] if urls else None
|
||||
|
||||
|
||||
def find_all_image_urls(title: str, brand: Optional[str] = None, country_hint: Optional[str] = None,
|
||||
validate: bool = True, max_results: int = 24) -> list:
|
||||
validate: bool = True, max_results: int = 24,
|
||||
produce: bool = False) -> list:
|
||||
"""Get working image URLs by trying multiple open-source sources in
|
||||
priority order (cheapest/most-reliable first), merging and validating
|
||||
results. Only escalates to the Playwright browser fallback if every
|
||||
@@ -488,11 +536,18 @@ def find_all_image_urls(title: str, brand: Optional[str] = None, country_hint: O
|
||||
|
||||
# Strip redundant brand prefix from title to avoid repetitive queries
|
||||
# like "Cadbury Perk Cadbury Perk Crunch".
|
||||
search_title = _clean_search_title(title, brand)
|
||||
search_title = _clean_search_title(title, brand, produce=produce)
|
||||
|
||||
candidates.extend(find_images_openfacts(search_title, brand, max_results))
|
||||
# Open*Facts is a PACKAGED-GOODS database - food, beauty and household
|
||||
# products photographed front-of-pack. Asked about loose produce it answers
|
||||
# with whatever carton or bottle mentions the word, which is how the live
|
||||
# catalogue ended up serving openbeautyfacts cosmetics photos for Banana,
|
||||
# Orange, Papaya, Guava, Lemon, Pear and twenty more. Skipped for produce.
|
||||
if not produce:
|
||||
candidates.extend(find_images_openfacts(search_title, brand, max_results))
|
||||
if len(candidates) < max_results:
|
||||
candidates.extend(find_images_wikimedia(search_title, brand, max_results))
|
||||
candidates.extend(find_images_wikimedia(search_title, brand, max_results,
|
||||
produce=produce))
|
||||
if len(candidates) < max_results:
|
||||
candidates.extend(find_all_image_urls_ddg(search_title, brand, max_results))
|
||||
if len(candidates) < max_results:
|
||||
|
||||
119
app/services/nutrition_autoenrich.py
Normal file
119
app/services/nutrition_autoenrich.py
Normal file
@@ -0,0 +1,119 @@
|
||||
"""Scoring newly uploaded products, without making the uploader wait for it.
|
||||
|
||||
WHY THIS IS NOT A PIPELINE STAGE
|
||||
--------------------------------
|
||||
The obvious place is a twelfth stage in `store_catalog_pipeline`, and it is the
|
||||
wrong one. That pipeline runs on `batch_worker`'s single thread draining a
|
||||
bounded queue: one batch at a time, by design, because a batch is up to twenty
|
||||
spreadsheets of two thousand rows. Nutrition enrichment is minutes of outbound
|
||||
network I/O per hundred products, so putting it inside the pipeline would stall
|
||||
every batch queued behind it, and a network hiccup at Open Food Facts would show
|
||||
up to an operator as a catalogue import that hung.
|
||||
|
||||
So enrichment is submitted AFTER the batch settles, onto its own thread, with
|
||||
its own job record - the same shape `POST /api/admin/nutrition-intelligence/
|
||||
enrich` already uses, and pollable at the same `/jobs/{job_id}` endpoint.
|
||||
|
||||
WHY IT NEEDS NO "WHICH ROWS ARE NEW" BOOKKEEPING
|
||||
------------------------------------------------
|
||||
`skip_if_verified=True` makes `enrich_all_products` re-fetch only products that
|
||||
lack verified data. Re-uploading a brand that is already scored therefore costs
|
||||
nothing but a cheap existence check per row, and the products that do get looked
|
||||
up are exactly the ones the upload added. Tracking new image_ids separately
|
||||
would be a second source of truth for the same question.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Any, Iterable, List, Optional
|
||||
|
||||
from app.api.background import run_in_background
|
||||
from app.api.nutrition_job_store import nutrition_job_store
|
||||
from app.infrastructure import settings
|
||||
from app.services import nutrition_enrichment_service
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def run_enrich_job(job_id: str, *, label: str = "", **enrich_kwargs: Any) -> None:
|
||||
"""Runs `enrich_all_products` under a job record.
|
||||
|
||||
Shared by the admin endpoint and the post-upload hook so both report the
|
||||
same result keys - a dashboard reading one should not have to special-case
|
||||
the other.
|
||||
|
||||
`label` is carried into the completion detail rather than only being set at
|
||||
queue time. A finished job whose detail reads "Enrichment complete" tells an
|
||||
operator nothing about which upload it belonged to, and the queued text that
|
||||
did say so has been overwritten by then.
|
||||
"""
|
||||
nutrition_job_store.update(job_id, status="running")
|
||||
|
||||
def progress_cb(done: int, total: int) -> None:
|
||||
nutrition_job_store.update(job_id, processed=done, total=total)
|
||||
|
||||
try:
|
||||
result = nutrition_enrichment_service.enrich_all_products(
|
||||
progress_cb=progress_cb, **enrich_kwargs)
|
||||
nutrition_job_store.update(
|
||||
job_id, status="done",
|
||||
detail="Enrichment complete" + (" ({})".format(label) if label else ""),
|
||||
result={
|
||||
"total_products": result.total_products, "verified": result.verified,
|
||||
"partial": result.partial, "unavailable": result.unavailable,
|
||||
"skipped_non_consumable": result.skipped_non_consumable,
|
||||
"duration_seconds": result.duration_seconds,
|
||||
"error_count": len(result.errors), "errors": result.errors[:20],
|
||||
},
|
||||
)
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.exception("Nutrition enrichment job %s failed", job_id)
|
||||
nutrition_job_store.update(job_id, status="failed", detail=str(e))
|
||||
|
||||
|
||||
def submit_enrichment_for_brands(brands: Iterable[str], *, source: str) -> Optional[str]:
|
||||
"""Queue scoring for freshly ingested brands. Returns the job id, or None.
|
||||
|
||||
Returns None rather than raising for every "nothing to do" case, because
|
||||
every caller is an upload handler finishing its work: an ingestion that
|
||||
genuinely succeeded must not be reported as failed because the optional
|
||||
scoring step could not start.
|
||||
"""
|
||||
if not settings.AUTO_ENRICH_ON_UPLOAD:
|
||||
logger.info("Auto-enrichment disabled (AUTO_ENRICH_ON_UPLOAD=false); "
|
||||
"%s ingested products are unscored until /enrich is run", source)
|
||||
return None
|
||||
|
||||
# dict.fromkeys, not set(): the brand list ends up in a job detail string an
|
||||
# operator reads, and a set would reorder it differently on every run.
|
||||
unique: List[str] = [b for b in dict.fromkeys(b.strip() for b in brands if b) if b]
|
||||
if not unique:
|
||||
return None
|
||||
|
||||
label = "{}: {}".format(source, ", ".join(unique))
|
||||
try:
|
||||
job = nutrition_job_store.create("enrich")
|
||||
nutrition_job_store.update(job.job_id, detail="Queued after " + label)
|
||||
run_in_background(
|
||||
lambda: run_enrich_job(
|
||||
job.job_id,
|
||||
label=label,
|
||||
brands=unique,
|
||||
# WITHOUT THIS, NOTHING IS SCORED. `_brands_to_enrich` honours
|
||||
# ACTIVE_BRANDS by default, and a brand that was just uploaded
|
||||
# is almost never in it - the job would report success over an
|
||||
# empty work list.
|
||||
include_inactive=True,
|
||||
skip_if_verified=True,
|
||||
# The LLM summary is the slow, token-costing step and adds no
|
||||
# health score. The admin /enrich job fills narratives in later.
|
||||
generate_narrative=False,
|
||||
),
|
||||
name="nutrition-autoenrich-{}".format(job.job_id[:8]),
|
||||
)
|
||||
logger.info("Queued nutrition enrichment %s for %s after %s",
|
||||
job.job_id, unique, source)
|
||||
return job.job_id
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.exception("Could not queue nutrition enrichment after %s: %s", source, e)
|
||||
return None
|
||||
@@ -52,6 +52,14 @@ MIN_MATCH_CONFIDENCE = 0.32
|
||||
# Categories in this FMCG catalog that are never food (no nutrition
|
||||
# panel exists), so we skip the network call entirely instead of
|
||||
# generating noisy near-misses.
|
||||
#
|
||||
# SUPERSEDED by app.services.consumability. Kept as a name because it reads
|
||||
# like public API, but it is no longer consulted: as a *substring* test over
|
||||
# seventeen words it both under- and over-matched. It missed every category it
|
||||
# was not literally spelled with - "Hair Care", "Skin Care", "Baby Care",
|
||||
# "Personal Care", "Flowers" - which is how Pantene, Nyle and a Godrej mosquito
|
||||
# spray ended up with health scores; and it matched "soap" inside *soapnut*
|
||||
# (reetha), which is a real commodity.
|
||||
NON_FOOD_KEYWORDS = (
|
||||
"soap", "detergent", "shampoo", "toothpaste", "cosmetic", "deodorant",
|
||||
"diaper", "sanitary", "cleaner", "disinfectant", "battery", "stationery",
|
||||
@@ -126,9 +134,44 @@ def _clean_title_for_search(title: str) -> str:
|
||||
return ' '.join(t.split())
|
||||
|
||||
|
||||
# Categories whose rows are loose commodities rather than packaged products.
|
||||
# These are the canonical names from `category_registry.ALL_CATEGORIES`, so a
|
||||
# row filed by the ingestion pipeline lands here without translation.
|
||||
USDA_FIRST_CATEGORIES = frozenset({
|
||||
"fruits and vegetables", "fresh herbs and greens", "fish and seafood",
|
||||
"eggs", "dry fruits and nuts",
|
||||
})
|
||||
|
||||
|
||||
def _prefers_usda(title: str, category: str) -> bool:
|
||||
"""True when this row is a loose commodity USDA is the better source for.
|
||||
|
||||
Category first, because that is what the pipeline actually assigns. Falling
|
||||
back to the commodity lexicon catches produce misfiled under "General" or
|
||||
one of the bare-numeric categories left by a bad import.
|
||||
"""
|
||||
from app.services.category_registry import _normalize
|
||||
from app.services.generic_products import canonical_category
|
||||
|
||||
if _normalize(category) in USDA_FIRST_CATEGORIES:
|
||||
return True
|
||||
detected = canonical_category(title or "")
|
||||
return bool(detected) and _normalize(detected) in USDA_FIRST_CATEGORIES
|
||||
|
||||
|
||||
def _looks_non_food(title: str, category: str) -> bool:
|
||||
text = f"{title} {category}".lower()
|
||||
return any(kw in text for kw in NON_FOOD_KEYWORDS)
|
||||
"""True only when the product is POSITIVELY not food.
|
||||
|
||||
Delegates to `consumability.is_non_consumable`, which is a three-state
|
||||
classifier: a product it cannot place returns False here, so an unknown
|
||||
product is still looked up rather than silently skipped. That is the
|
||||
opposite of what a `not is_consumable()` would do, and it is deliberate -
|
||||
this gate decides whether to make a network call, not whether to delete
|
||||
anything.
|
||||
"""
|
||||
from app.services.consumability import is_non_consumable
|
||||
|
||||
return is_non_consumable(category, title)
|
||||
|
||||
|
||||
def _search_openfoodfacts(query: str, brand: str = "", category: str = "", max_results: int = 5) -> List[dict]:
|
||||
@@ -308,11 +351,33 @@ def fetch_verified_nutrition(brand: str, title: str, category: str = "") -> Dict
|
||||
now_iso = datetime.now(timezone.utc).isoformat()
|
||||
unavailable = {"data_status": "unavailable", "fetched_at": now_iso}
|
||||
|
||||
if not USE_OPEN_FACTS:
|
||||
return unavailable
|
||||
if _looks_non_food(title, category):
|
||||
return unavailable
|
||||
|
||||
# Loose commodities go to USDA first. Open Food Facts catalogues PACKAGED
|
||||
# products; asked about a raw banana it returns whatever carton mentions
|
||||
# one, and `_match_confidence` then attaches that product's numbers at a
|
||||
# score above the 0.32 floor. USDA publishes laboratory analyses of the raw
|
||||
# commodity itself, per 100 g of edible portion.
|
||||
#
|
||||
# Routed by CATEGORY, not by brand. `Own Products` is the bucket for every
|
||||
# unbranded row, and it holds "Toor Dhal 1kg", "Sugar 1kg" and "Salt 1kg"
|
||||
# alongside the produce - OFF has far better Indian-market data for those,
|
||||
# so a brand test would send them to the wrong source.
|
||||
if _prefers_usda(title, category):
|
||||
from app.services.nutrition_usda_service import fetch_verified_nutrition_usda
|
||||
|
||||
usda = fetch_verified_nutrition_usda(title, category)
|
||||
if usda.get("data_status") != "unavailable":
|
||||
return usda
|
||||
# Falls through to OFF, which occasionally does know a commodity USDA
|
||||
# does not. The reverse is never done: a branded packaged product must
|
||||
# never be handed a raw-commodity number, so `_prefers_usda` is the only
|
||||
# door into the USDA path.
|
||||
|
||||
if not USE_OPEN_FACTS:
|
||||
return unavailable
|
||||
|
||||
candidates = _search_openfoodfacts(query=title, brand=brand, category=category)
|
||||
if not candidates:
|
||||
candidates = _search_openfoodfacts(query=title)
|
||||
@@ -450,6 +515,14 @@ def fetch_verified_nutrition_by_barcode(
|
||||
if not USE_OPEN_FACTS or not (barcode or "").strip():
|
||||
return unavailable
|
||||
|
||||
# The edibility gate the forward path has always had, and this one never
|
||||
# did. A shampoo carries a barcode like anything else, and Open Beauty
|
||||
# Facts knows plenty of them, so without this a personal-care product
|
||||
# acquires a full nutrition row at match_confidence 0.95 - a HIGHER
|
||||
# confidence than anything the name search can produce.
|
||||
if _looks_non_food(title, category):
|
||||
return unavailable
|
||||
|
||||
# Imported here rather than at module scope: this module is reached during
|
||||
# nutrition enrichment, and the barcode package pulls in tenacity plus the
|
||||
# whole source cascade. A local import keeps that off the path of the
|
||||
|
||||
@@ -175,6 +175,26 @@ CREATE TABLE IF NOT EXISTS nutrition_healthy_alternatives (
|
||||
computed_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
|
||||
);
|
||||
CREATE INDEX IF NOT EXISTS idx_nutrition_alt_source ON nutrition_healthy_alternatives(brand, image_id, rank);
|
||||
|
||||
-- The consumable/non-consumable verdict, persisted at the point
|
||||
-- `consumability.classify_edibility` already runs during enrichment.
|
||||
--
|
||||
-- WHY A COLUMN AND NOT A PYTHON FILTER: `GET /api/nutrition/health-scores`
|
||||
-- claims to list "consumable food products", and that claim has to be made in
|
||||
-- the WHERE clause or the `total` it returns will not match the rows it
|
||||
-- returns. Filtering a fetched page in Python would break both the count and
|
||||
-- the paging.
|
||||
--
|
||||
-- WHY NOT JUST TRUST THE WRITE GATE: the gate refuses NON_CONSUMABLE, but
|
||||
-- UNKNOWN is neither refused nor confirmed - the purge audit found 33 such
|
||||
-- rows. NULL (written before this column existed) and 'unknown' are therefore
|
||||
-- both excluded by default: an unclassified row is not evidence of food.
|
||||
--
|
||||
-- These two statements, not the CREATE TABLE above, are what actually add the
|
||||
-- column to a database where nutrition_facts already exists. IF NOT EXISTS on
|
||||
-- both keeps `ensure_nutrition_schema()` idempotent on every API startup.
|
||||
ALTER TABLE nutrition_facts ADD COLUMN IF NOT EXISTS edibility TEXT;
|
||||
CREATE INDEX IF NOT EXISTS idx_nutrition_facts_edibility ON nutrition_facts(edibility);
|
||||
"""
|
||||
|
||||
|
||||
@@ -240,6 +260,7 @@ _FACT_COLUMNS = [
|
||||
"sodium_mg", "potassium_mg", "calcium_mg", "iron_mg", "magnesium_mg", "zinc_mg",
|
||||
"vitamin_a_mcg", "vitamin_c_mg", "vitamin_d_mcg", "vitamin_e_mg", "omega_3_g", "omega_6_g",
|
||||
"extended_nutrients", "per_serving", "ingredients_text", "off_nutriscore", "fetched_at",
|
||||
"edibility",
|
||||
]
|
||||
|
||||
|
||||
@@ -480,6 +501,175 @@ def query_products(
|
||||
conn.close()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Health-score listing (GET /api/nutrition/health-scores)
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# Deliberately NOT built on `query_products` above. That function backs five
|
||||
# live endpoints plus the analytics and recommendation payloads, several of
|
||||
# which splat its rows into raw dicts - widening its SELECT would leak new keys
|
||||
# into responses nobody asked to change. The two share nothing but the tables.
|
||||
#
|
||||
# The listing and the count are built from the SAME `_health_score_filters`,
|
||||
# because a `total` that disagrees with the rows beside it is worse than no
|
||||
# total at all.
|
||||
|
||||
CONSUMABLE = "consumable"
|
||||
NON_CONSUMABLE = "non_consumable"
|
||||
|
||||
# Table-qualified, unlike `_SORTABLE_COLUMNS`. `data_status` exists on both
|
||||
# tables, so leaving the sort column bare is one added option away from an
|
||||
# ambiguous-column error at runtime.
|
||||
_HEALTH_SORT_COLUMNS = {
|
||||
"health_score": "i.health_score", "nutrition_score": "i.nutrition_score",
|
||||
"protein": "f.protein_g", "fiber": "f.dietary_fiber_g", "sugar": "f.total_sugar_g",
|
||||
"sodium": "f.sodium_mg", "calcium": "f.calcium_mg", "iron": "f.iron_mg",
|
||||
"vitamin_c": "f.vitamin_c_mg", "calories": "f.calories_kcal",
|
||||
"product_name": "f.product_name", "brand": "f.brand", "category": "f.category",
|
||||
}
|
||||
|
||||
_HEALTH_SCORE_PROJECTION = """
|
||||
f.brand, f.image_id, f.product_name, f.category, f.edibility,
|
||||
f.data_status, f.data_source, f.source_url,
|
||||
f.calories_kcal, f.protein_g, f.dietary_fiber_g, f.total_sugar_g, f.sodium_mg,
|
||||
i.nutrition_score, i.health_score, i.scoring_version, i.diet_tags, i.allergens
|
||||
"""
|
||||
|
||||
|
||||
def _health_score_filters(
|
||||
brand: Optional[str] = None,
|
||||
category: Optional[str] = None,
|
||||
min_score: Optional[float] = None,
|
||||
max_score: Optional[float] = None,
|
||||
include_unknown: bool = False,
|
||||
) -> tuple:
|
||||
"""The single source of the WHERE clause for both the listing and its count."""
|
||||
# `i.health_score IS NOT NULL` is stated OUTRIGHT, not derived from the sort
|
||||
# column the way `query_products` derives it. The endpoint promises "only
|
||||
# scored products"; that promise must not quietly change when a caller sorts
|
||||
# by protein instead.
|
||||
where = ["i.health_score IS NOT NULL", "f.data_status != 'unavailable'"]
|
||||
params: List[Any] = []
|
||||
|
||||
if include_unknown:
|
||||
# Still never non-consumable. IS DISTINCT FROM (not !=) so NULL, which
|
||||
# is what every row written before the edibility column carries, counts
|
||||
# as unknown rather than dropping out of a NULL comparison.
|
||||
where.append("f.edibility IS DISTINCT FROM %s")
|
||||
params.append(NON_CONSUMABLE)
|
||||
else:
|
||||
where.append("f.edibility = %s")
|
||||
params.append(CONSUMABLE)
|
||||
|
||||
if brand:
|
||||
where.append("f.brand = %s")
|
||||
params.append(brand)
|
||||
if category:
|
||||
where.append("f.category ILIKE %s")
|
||||
params.append(f"%{category}%")
|
||||
if min_score is not None:
|
||||
where.append("i.health_score >= %s")
|
||||
params.append(min_score)
|
||||
if max_score is not None:
|
||||
where.append("i.health_score <= %s")
|
||||
params.append(max_score)
|
||||
|
||||
return " AND ".join(where), params
|
||||
|
||||
|
||||
def count_health_scores(**filters: Any) -> int:
|
||||
"""Total matching rows, for the pagination envelope."""
|
||||
conn = _connect()
|
||||
if not conn:
|
||||
return 0
|
||||
where_clause, params = _health_score_filters(**filters)
|
||||
try:
|
||||
with conn, conn.cursor() as cur:
|
||||
cur.execute(
|
||||
f"""
|
||||
SELECT COUNT(*)
|
||||
FROM nutrition_facts f
|
||||
JOIN nutrition_insights i ON i.brand = f.brand AND i.image_id = f.image_id
|
||||
WHERE {where_clause}
|
||||
""",
|
||||
params,
|
||||
)
|
||||
row = cur.fetchone()
|
||||
return int(row[0]) if row else 0
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.error(f"count_health_scores failed: {e}")
|
||||
return 0
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def query_health_scores(
|
||||
sort_by: str = "health_score",
|
||||
order: str = "desc",
|
||||
limit: int = 100,
|
||||
offset: int = 0,
|
||||
**filters: Any,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""One page of scored consumable products, newest scoring first by default."""
|
||||
conn = _connect()
|
||||
if not conn:
|
||||
return []
|
||||
where_clause, params = _health_score_filters(**filters)
|
||||
col = _HEALTH_SORT_COLUMNS.get(sort_by, "i.health_score")
|
||||
direction = "ASC" if order == "asc" else "DESC"
|
||||
|
||||
# The (brand, image_id) tiebreaker is load-bearing, not cosmetic: it is the
|
||||
# primary key, so it makes the ordering total. Without it Postgres is free
|
||||
# to return tied health_scores in a different order on each call, and OFFSET
|
||||
# paging then hands the caller the same product twice and skips another.
|
||||
sql = f"""
|
||||
SELECT {_HEALTH_SCORE_PROJECTION}
|
||||
FROM nutrition_facts f
|
||||
JOIN nutrition_insights i ON i.brand = f.brand AND i.image_id = f.image_id
|
||||
WHERE {where_clause}
|
||||
ORDER BY {col} {direction} NULLS LAST, f.brand ASC, f.image_id ASC
|
||||
LIMIT %s OFFSET %s
|
||||
"""
|
||||
try:
|
||||
with conn, _dict_cursor(conn) as cur:
|
||||
cur.execute(sql, params + [limit, offset])
|
||||
rows = [dict(r) for r in cur.fetchall()]
|
||||
return [_row_numeric(r, NUMERIC_FACT_COLUMNS + NUMERIC_INSIGHT_COLUMNS) for r in rows]
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.error(f"query_health_scores failed: {e}")
|
||||
return []
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
HEALTH_SCORE_EXPORT_PAGE = 1000
|
||||
|
||||
|
||||
def iter_health_scores(sort_by: str = "health_score", order: str = "desc", **filters: Any):
|
||||
"""Every matching row, a page at a time - the CSV export path.
|
||||
|
||||
Pages rather than materialising the whole result so memory stays flat as the
|
||||
catalogue grows, and so no database connection is held open for the length
|
||||
of the HTTP response. The total ordering in `query_health_scores` is what
|
||||
makes paging safe here; a row inserted mid-export can still shift a later
|
||||
page, which is acceptable for an export and is why the payload carries a
|
||||
`generated_at`.
|
||||
"""
|
||||
offset = 0
|
||||
while True:
|
||||
page = query_health_scores(
|
||||
sort_by=sort_by, order=order,
|
||||
limit=HEALTH_SCORE_EXPORT_PAGE, offset=offset, **filters,
|
||||
)
|
||||
if not page:
|
||||
return
|
||||
for row in page:
|
||||
yield row
|
||||
if len(page) < HEALTH_SCORE_EXPORT_PAGE:
|
||||
return
|
||||
offset += HEALTH_SCORE_EXPORT_PAGE
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Similarity / alternatives caches
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -23,18 +23,28 @@ from dataclasses import dataclass, field
|
||||
from typing import Any, Callable, Dict, List, Optional
|
||||
|
||||
from app.services import nutrition_data_service, nutrition_db, nutrition_scoring
|
||||
from app.services.consumability import Edibility, classify_edibility, is_non_consumable
|
||||
from app.services.nutrition_narrative_service import generate_summary
|
||||
from app.services.vector_store import get_products_by_brand, list_available_brands
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
# Returned by `enrich_one_product` for a product that is not food. Deliberately
|
||||
# NOT "unavailable": that means "we looked and found nothing", and a caller
|
||||
# retrying the unavailable rows should not keep re-asking about soap. The
|
||||
# distinction is the same one `backfill_nutrition_from_barcodes` draws between
|
||||
# a thin record and a rejected one.
|
||||
SKIPPED_NON_CONSUMABLE = "skipped_non_consumable"
|
||||
|
||||
|
||||
@dataclass
|
||||
class EnrichmentResult:
|
||||
total_products: int = 0
|
||||
verified: int = 0
|
||||
partial: int = 0
|
||||
unavailable: int = 0
|
||||
skipped_non_consumable: int = 0
|
||||
errors: List[str] = field(default_factory=list)
|
||||
duration_seconds: float = 0.0
|
||||
|
||||
@@ -42,7 +52,17 @@ class EnrichmentResult:
|
||||
def enrich_one_product(brand: str, image_id: str, product_name: str, category: str,
|
||||
skip_if_verified: bool = False, generate_narrative: bool = True) -> str:
|
||||
"""Runs the full pipeline for a single product. Returns the resulting
|
||||
`data_status` ('verified' | 'partial' | 'unavailable')."""
|
||||
`data_status` ('verified' | 'partial' | 'unavailable' |
|
||||
'skipped_non_consumable')."""
|
||||
# The authoritative gate. `fetch_verified_nutrition` has one too, but this
|
||||
# is the only place that knows (brand, image_id) and can therefore refuse to
|
||||
# WRITE A ROW at all - which is what stops a shampoo appearing in
|
||||
# `query_products`, the analytics leaderboards and the healthy-alternatives
|
||||
# cache with a null score attached.
|
||||
verdict = classify_edibility(category, product_name)
|
||||
if verdict.edibility is Edibility.NON_CONSUMABLE:
|
||||
return SKIPPED_NON_CONSUMABLE
|
||||
|
||||
if skip_if_verified:
|
||||
existing = nutrition_db.get_nutrition_facts(brand, image_id)
|
||||
if existing and existing.get("data_status") == "verified":
|
||||
@@ -53,6 +73,13 @@ def enrich_one_product(brand: str, image_id: str, product_name: str, category: s
|
||||
facts["image_id"] = image_id
|
||||
facts["product_name"] = product_name
|
||||
facts["category"] = category
|
||||
# Persisted so the verdict is available to SQL. `GET /api/nutrition/
|
||||
# health-scores` has to filter on "is this food?" inside the query - a
|
||||
# Python filter applied to a fetched page would make its `total` and its
|
||||
# rows describe different populations. UNKNOWN is stored as such rather
|
||||
# than rounded up to consumable: the classifier not deciding is a fact
|
||||
# worth keeping, not a fact worth guessing.
|
||||
facts["edibility"] = verdict.edibility.value
|
||||
nutrition_db.upsert_nutrition_facts(facts)
|
||||
|
||||
scores = nutrition_scoring.compute_scores(facts)
|
||||
@@ -72,7 +99,10 @@ def enrich_one_product(brand: str, image_id: str, product_name: str, category: s
|
||||
"brand": brand, "image_id": image_id,
|
||||
"positive_insights": positive, "nutritional_cautions": cautions,
|
||||
"ai_summary": ai_summary, "diet_tags": diet_tags, "allergens": allergens,
|
||||
"allergen_source": "openfoodfacts" if allergens else "unavailable",
|
||||
# Derived from where the facts actually came from. Hardcoding
|
||||
# "openfoodfacts" was right when OFF was the only source; with USDA in
|
||||
# the cascade it would credit the wrong database.
|
||||
"allergen_source": (facts.get("data_source") or "unavailable") if allergens else "unavailable",
|
||||
"data_status": facts["data_status"],
|
||||
}
|
||||
if scores:
|
||||
@@ -82,27 +112,76 @@ def enrich_one_product(brand: str, image_id: str, product_name: str, category: s
|
||||
return facts["data_status"]
|
||||
|
||||
|
||||
def _brands_to_enrich(brands: Optional[List[str]], include_inactive: bool) -> List[str]:
|
||||
"""Which brand display names this run covers.
|
||||
|
||||
`list_available_brands()` honours ACTIVE_BRANDS, which is right for
|
||||
everything the app serves and wrong for a catalogue-wide backfill: with
|
||||
ACTIVE_BRANDS set to four brands, enrichment physically cannot reach
|
||||
Britannia or Parle, and their products stay permanently unscored.
|
||||
|
||||
Widening is therefore an explicit ARGUMENT, never an environment mutation -
|
||||
the running API keeps its own scoping either way.
|
||||
"""
|
||||
if brands:
|
||||
return [b.strip() for b in brands if b and b.strip()]
|
||||
if not include_inactive:
|
||||
return list_available_brands()
|
||||
|
||||
# Same escape hatch repair_brand_images._brand_tables uses.
|
||||
from app.services.vector_store import (
|
||||
_connect, _list_brand_table_suffixes, display_name_for_suffix,
|
||||
)
|
||||
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
logger.warning("No database connection; falling back to active brands only")
|
||||
return list_available_brands()
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
suffixes = sorted(set(_list_brand_table_suffixes(cur, include_inactive=True)))
|
||||
return [display_name_for_suffix(s) for s in suffixes]
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def enrich_all_products(
|
||||
skip_if_verified: bool = True,
|
||||
generate_narrative: bool = True,
|
||||
progress_cb: Optional[Callable[[int, int], None]] = None,
|
||||
max_products: Optional[int] = None,
|
||||
brands: Optional[List[str]] = None,
|
||||
include_inactive: bool = False,
|
||||
categories: Optional[List[str]] = None,
|
||||
) -> EnrichmentResult:
|
||||
"""Iterates every product across every brand table (via
|
||||
"""Iterates products across brand tables (via
|
||||
`vector_store.list_available_brands` / `get_products_by_brand` -
|
||||
the exact same source of truth the catalog UI reads from) and runs
|
||||
the pipeline on each."""
|
||||
the pipeline on each.
|
||||
|
||||
Non-consumables are dropped while the work list is built, not inside the
|
||||
loop, so `total_products` and the progress percentage describe the work
|
||||
actually being done. `enrich_one_product` re-checks anyway: Dagster calls it
|
||||
directly and never comes through here.
|
||||
"""
|
||||
start = time.time()
|
||||
result = EnrichmentResult()
|
||||
|
||||
brands = list_available_brands()
|
||||
wanted_categories = {c.strip().lower() for c in categories} if categories else None
|
||||
|
||||
all_products: List[Dict[str, Any]] = []
|
||||
for brand in brands:
|
||||
for brand in _brands_to_enrich(brands, include_inactive):
|
||||
for p in get_products_by_brand(brand):
|
||||
category = p.get("category")
|
||||
name = p.get("title") or p.get("product_name")
|
||||
if wanted_categories and (category or "").strip().lower() not in wanted_categories:
|
||||
continue
|
||||
if is_non_consumable(category, name or ""):
|
||||
result.skipped_non_consumable += 1
|
||||
continue
|
||||
all_products.append({
|
||||
"brand": brand, "image_id": p.get("image_id"),
|
||||
"product_name": p.get("title") or p.get("product_name"),
|
||||
"category": p.get("category"),
|
||||
"product_name": name, "category": category,
|
||||
})
|
||||
if max_products:
|
||||
all_products = all_products[:max_products]
|
||||
@@ -121,6 +200,11 @@ def enrich_all_products(
|
||||
result.verified += 1
|
||||
elif status == "partial":
|
||||
result.partial += 1
|
||||
elif status == SKIPPED_NON_CONSUMABLE:
|
||||
# Only reachable if the pre-filter and the per-product gate
|
||||
# disagree; counted honestly rather than folded into
|
||||
# "unavailable", which would read as "we looked and found none".
|
||||
result.skipped_non_consumable += 1
|
||||
else:
|
||||
result.unavailable += 1
|
||||
except Exception as e: # noqa: BLE001
|
||||
@@ -132,7 +216,8 @@ def enrich_all_products(
|
||||
result.duration_seconds = round(time.time() - start, 1)
|
||||
logger.info(
|
||||
f"Nutrition enrichment complete: {result.verified} verified, {result.partial} partial, "
|
||||
f"{result.unavailable} unavailable of {result.total_products} products in {result.duration_seconds}s"
|
||||
f"{result.unavailable} unavailable of {result.total_products} products "
|
||||
f"({result.skipped_non_consumable} non-consumable skipped) in {result.duration_seconds}s"
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
308
app/services/nutrition_usda_service.py
Normal file
308
app/services/nutrition_usda_service.py
Normal file
@@ -0,0 +1,308 @@
|
||||
"""
|
||||
Verified nutrition for fresh produce, from USDA FoodData Central.
|
||||
|
||||
WHY OPEN FOOD FACTS COULD NOT DO THIS
|
||||
-------------------------------------
|
||||
`nutrition_data_service` is built on Open Food Facts, which catalogues PACKAGED
|
||||
products photographed front-of-pack. It has no entry for a raw apple. Asked for
|
||||
one it answers with whatever carton mentions the word, and the fuzzy name match
|
||||
then attaches that product's numbers. Measured on the live catalogue: all 159
|
||||
rows of `brand_own_products` - every fruit, vegetable, green, fish and egg the
|
||||
shop sells loose - had NO nutrition at all, no benefits and no health score.
|
||||
|
||||
USDA FoodData Central is the other half of the problem. Its Foundation Foods
|
||||
and SR Legacy datasets are laboratory analyses of raw commodities, published
|
||||
per 100 g of edible portion, which is exactly the shape this schema wants.
|
||||
|
||||
WHAT THIS MODULE GUARANTEES
|
||||
---------------------------
|
||||
It returns THE SAME DICT SHAPE as `fetch_verified_nutrition`, deliberately, so
|
||||
`nutrition_db.upsert_nutrition_facts` writes it unchanged and the acquisition
|
||||
paths cannot drift apart. There is a test asserting the key sets match, the
|
||||
same one that already guards the barcode path.
|
||||
|
||||
It never estimates. A nutrient USDA does not report stays `None` all the way to
|
||||
the API response - not zero, not a category average. A commodity absent from
|
||||
`produce_reference` returns the `unavailable` shell.
|
||||
|
||||
THREE RULES IN THE MAPPING LAYER, EACH FOR A REASON
|
||||
---------------------------------------------------
|
||||
1. THE UNIT IS ASSERTED, NEVER ASSUMED. USDA keys nutrients by integer id, and
|
||||
the same id has been served in different units across dataset revisions and
|
||||
between the bulk download (which writes "µg") and the live API (which writes
|
||||
"UG"). A sodium value silently read as grams and multiplied by 1000 would be
|
||||
off by a factor of a million and would still look plausible. When the unit is
|
||||
not one this module recognises for that nutrient, the value is dropped and a
|
||||
warning logged.
|
||||
|
||||
2. IU IS REFUSED RATHER THAN CONVERTED. The IU-to-microgram factor for vitamin A
|
||||
depends on the vitamer (retinol vs beta-carotene), so any single factor is an
|
||||
estimate. Estimates are what this module exists not to produce.
|
||||
|
||||
3. BOTH RESPONSE SHAPES ARE HANDLED. `/food/{id}` nests the nutrient as
|
||||
`{"nutrient": {...}, "amount": ...}`; `/foods/search` flattens it to
|
||||
`{"nutrientId": ..., "unitName": ..., "value": ...}`; the pinned snapshot
|
||||
stores a third, trimmed shape. One accessor reads all three.
|
||||
|
||||
OFFLINE BY DEFAULT
|
||||
------------------
|
||||
The snapshot in `data/usda_snapshot.json` holds the raw USDA record for every id
|
||||
the curated table references, taken from the bulk download - no API key, no rate
|
||||
limit. The live API is consulted only for an id the snapshot lacks, and only
|
||||
when `USDA_FDC_API_KEY` is set. So a normal enrichment run makes zero USDA
|
||||
calls, and the tests are hermetic without stubbing anything.
|
||||
|
||||
The snapshot stores the nutrient list RAW rather than pre-mapped, so the mapping
|
||||
rules above are exercised identically online and offline. A snapshot of finished
|
||||
values would mean the offline tests tested nothing.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
from datetime import datetime, timezone
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
import requests
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
REQUEST_TIMEOUT_SECONDS, USDA_FDC_API_KEY, USE_USDA_FDC,
|
||||
)
|
||||
from app.services import produce_reference
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
USDA_HOST = "api.nal.usda.gov"
|
||||
USDA_DATA_SOURCE = "usda_fdc"
|
||||
|
||||
# A curated identifier is not a fuzzy match, and it is not certainty either:
|
||||
# choosing "Tomatoes, red, ripe, raw, year round average" for a grocer's
|
||||
# "Tomato" is a judgement about which preparation is meant. Recorded well clear
|
||||
# of the 0.32 floor the OFF name search can produce, and below the 0.95 a
|
||||
# barcode earns.
|
||||
USDA_MATCH_CONFIDENCE = 0.9
|
||||
|
||||
_SNAPSHOT_FILE = Path(__file__).parent / "data" / "usda_snapshot.json"
|
||||
|
||||
# USDA nutrient id -> (our column, units this module will accept).
|
||||
# Ids are stable across FoodData Central releases; units are not, hence rule 1.
|
||||
_GRAMS = ("G", "g")
|
||||
_MG = ("MG", "mg")
|
||||
_UG = ("UG", "ug", "µg", "MCG_RE", "mcg")
|
||||
_KCAL = ("KCAL", "kcal")
|
||||
|
||||
_NUTRIENT_MAP: Dict[int, tuple] = {
|
||||
1008: ("calories_kcal", _KCAL), # Energy (kcal). 1062 is kJ - different id.
|
||||
1003: ("protein_g", _GRAMS),
|
||||
1005: ("carbohydrates_g", _GRAMS), # by difference
|
||||
2000: ("total_sugar_g", _GRAMS),
|
||||
1235: ("added_sugar_g", _GRAMS),
|
||||
1079: ("dietary_fiber_g", _GRAMS),
|
||||
1004: ("total_fat_g", _GRAMS),
|
||||
1258: ("saturated_fat_g", _GRAMS),
|
||||
1257: ("trans_fat_g", _GRAMS),
|
||||
1253: ("cholesterol_mg", _MG),
|
||||
1093: ("sodium_mg", _MG),
|
||||
1092: ("potassium_mg", _MG),
|
||||
1087: ("calcium_mg", _MG),
|
||||
1089: ("iron_mg", _MG),
|
||||
1090: ("magnesium_mg", _MG),
|
||||
1095: ("zinc_mg", _MG),
|
||||
1106: ("vitamin_a_mcg", _UG), # RAE. The IU form (1104) is refused.
|
||||
1162: ("vitamin_c_mg", _MG),
|
||||
1114: ("vitamin_d_mcg", _UG), # D2 + D3
|
||||
1109: ("vitamin_e_mg", _MG), # alpha-tocopherol
|
||||
1404: ("omega_3_g", _GRAMS), # 18:3 n-3 (ALA)
|
||||
1316: ("omega_6_g", _GRAMS), # 18:2 n-6 (LA)
|
||||
}
|
||||
|
||||
# Nutrients worth keeping that have no flat column, mirroring
|
||||
# `nutrition_data_service._EXTENDED_NUTRIENT_KEYS`.
|
||||
_EXTENDED_MAP: Dict[int, tuple] = {
|
||||
1165: ("Vitamin B1 (Thiamine)", _MG),
|
||||
1166: ("Vitamin B2 (Riboflavin)", _MG),
|
||||
1175: ("Vitamin B6", _MG),
|
||||
1177: ("Vitamin B9 (Folate)", _UG),
|
||||
1178: ("Vitamin B12", _UG),
|
||||
1167: ("Vitamin B3 (Niacin)", _MG),
|
||||
1091: ("Phosphorus", _MG),
|
||||
1100: ("Iodine", _UG),
|
||||
}
|
||||
|
||||
# Ids deliberately ignored: an International Unit cannot be converted to a mass
|
||||
# without knowing the vitamer, so it is dropped rather than guessed.
|
||||
_REFUSED_IU_IDS = frozenset({1104, 1110})
|
||||
|
||||
|
||||
def _unit_label(accepted: tuple) -> str:
|
||||
if accepted is _MG:
|
||||
return "mg"
|
||||
if accepted is _UG:
|
||||
return "mcg"
|
||||
if accepted is _KCAL:
|
||||
return "kcal"
|
||||
return "g"
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def _snapshot() -> Dict[str, Any]:
|
||||
if not _SNAPSHOT_FILE.exists():
|
||||
logger.warning("USDA snapshot missing at %s", _SNAPSHOT_FILE)
|
||||
return {}
|
||||
return json.loads(_SNAPSHOT_FILE.read_text(encoding="utf-8")).get("foods", {})
|
||||
|
||||
|
||||
def _iter_nutrients(food: Dict[str, Any]):
|
||||
"""Yield (nutrient_id, unit_name, amount) from any of the three shapes."""
|
||||
for entry in food.get("nutrients") or food.get("foodNutrients") or []:
|
||||
if "nutrient" in entry: # /food/{id}
|
||||
n = entry["nutrient"] or {}
|
||||
yield n.get("id"), n.get("unitName"), entry.get("amount")
|
||||
elif "nutrientId" in entry: # /foods/search
|
||||
yield entry.get("nutrientId"), entry.get("unitName"), entry.get("value")
|
||||
else: # pinned snapshot
|
||||
yield entry.get("id"), entry.get("unit"), entry.get("amount")
|
||||
|
||||
|
||||
def _build_fields(food: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Flat per-100g columns. USDA already publishes per 100 g of edible
|
||||
portion, so there is no scaling here - only unit checking."""
|
||||
out: Dict[str, Any] = {name: None for name, _ in _NUTRIENT_MAP.values()}
|
||||
for nid, unit, amount in _iter_nutrients(food):
|
||||
if nid not in _NUTRIENT_MAP or amount is None:
|
||||
continue
|
||||
column, accepted = _NUTRIENT_MAP[nid]
|
||||
if unit not in accepted:
|
||||
logger.warning(
|
||||
"USDA nutrient %s for %s came back in %r, expected one of %r - "
|
||||
"dropping rather than rescaling a value that would look plausible",
|
||||
nid, food.get("description"), unit, accepted,
|
||||
)
|
||||
continue
|
||||
try:
|
||||
out[column] = round(float(amount), 3)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
return out
|
||||
|
||||
|
||||
def _build_extended(food: Dict[str, Any]) -> Dict[str, Any]:
|
||||
out: Dict[str, Any] = {}
|
||||
for nid, unit, amount in _iter_nutrients(food):
|
||||
if nid not in _EXTENDED_MAP or amount is None:
|
||||
continue
|
||||
label, accepted = _EXTENDED_MAP[nid]
|
||||
if unit not in accepted:
|
||||
continue
|
||||
try:
|
||||
out[label] = {"value": round(float(amount), 3), "unit": _unit_label(accepted)}
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
return out
|
||||
|
||||
|
||||
def _portion(food: Dict[str, Any]) -> tuple:
|
||||
portions = food.get("portions") or food.get("foodPortions") or []
|
||||
for p in portions:
|
||||
grams = p.get("gram_weight") or p.get("gramWeight")
|
||||
if not grams:
|
||||
continue
|
||||
label = (p.get("description") or p.get("portionDescription")
|
||||
or (p.get("measureUnit") or {}).get("name") or "")
|
||||
try:
|
||||
return round(float(grams), 3), (label or None)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
return None, None
|
||||
|
||||
|
||||
def _fetch_live(fdc_id: int) -> Optional[Dict[str, Any]]:
|
||||
"""The live API, used only for an id the snapshot lacks."""
|
||||
if not USDA_FDC_API_KEY:
|
||||
return None
|
||||
try:
|
||||
resp = requests.get(
|
||||
f"https://{USDA_HOST}/fdc/v1/food/{fdc_id}",
|
||||
params={"api_key": USDA_FDC_API_KEY},
|
||||
timeout=REQUEST_TIMEOUT_SECONDS,
|
||||
)
|
||||
if resp.status_code != 200:
|
||||
logger.debug("USDA live lookup for %s returned %s", fdc_id, resp.status_code)
|
||||
return None
|
||||
return resp.json()
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.debug("USDA live lookup failed for %s: %s", fdc_id, e)
|
||||
return None
|
||||
|
||||
|
||||
def get_food(fdc_id: int) -> Optional[Dict[str, Any]]:
|
||||
"""The raw USDA record: snapshot first, live API only as a fallback."""
|
||||
food = _snapshot().get(str(fdc_id))
|
||||
if food:
|
||||
return food
|
||||
return _fetch_live(fdc_id)
|
||||
|
||||
|
||||
def fetch_verified_nutrition_usda(
|
||||
product_name: str, category: str = "", fdc_id: Optional[int] = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""Nutrition for one loose commodity, keyed by the curated FDC id.
|
||||
|
||||
Returns the same dict shape as
|
||||
`nutrition_data_service.fetch_verified_nutrition`, including the `off_*`
|
||||
keys - as empty rather than absent. USDA has no notion of a Nutri-Score or
|
||||
an ingredient-derived vegan flag, and `[]` says "USDA told us nothing about
|
||||
this", which is true and is what `classify_diet_tags` reads.
|
||||
"""
|
||||
now_iso = datetime.now(timezone.utc).isoformat()
|
||||
unavailable = {"data_status": "unavailable", "fetched_at": now_iso}
|
||||
|
||||
if not USE_USDA_FDC:
|
||||
return unavailable
|
||||
|
||||
if fdc_id is None:
|
||||
fdc_id = produce_reference.usda_fdc_id(product_name)
|
||||
if not fdc_id:
|
||||
return unavailable
|
||||
|
||||
food = get_food(fdc_id)
|
||||
if not food:
|
||||
return unavailable
|
||||
|
||||
per_100g = _build_fields(food)
|
||||
if not any(v is not None for v in per_100g.values()):
|
||||
return unavailable
|
||||
|
||||
serving_g, serving_label = _portion(food)
|
||||
per_serving: Dict[str, Any] = {}
|
||||
if serving_g:
|
||||
# Arithmetic on given values, not an estimate of them: the same
|
||||
# reasoning that lets a line total be computed from a unit price.
|
||||
factor = serving_g / 100.0
|
||||
per_serving = {k: round(v * factor, 3)
|
||||
for k, v in per_100g.items() if v is not None}
|
||||
|
||||
result: Dict[str, Any] = {
|
||||
"data_status": "verified" if per_100g.get("calories_kcal") is not None else "partial",
|
||||
"data_source": USDA_DATA_SOURCE,
|
||||
"source_ref": str(fdc_id),
|
||||
"source_url": f"https://fdc.nal.usda.gov/food-details/{fdc_id}",
|
||||
"match_confidence": USDA_MATCH_CONFIDENCE,
|
||||
"serving_size_g": serving_g,
|
||||
"serving_size_label": serving_label,
|
||||
"extended_nutrients": _build_extended(food),
|
||||
"per_serving": per_serving,
|
||||
# A raw commodity has no ingredient list and no declared allergens.
|
||||
# Empty, not absent, so the dict shape matches the OFF path exactly.
|
||||
"ingredients_text": None,
|
||||
"off_nutriscore": None,
|
||||
"allergens": [],
|
||||
"off_labels_tags": [],
|
||||
"off_ingredients_analysis_tags": [],
|
||||
"off_categories_tags": [],
|
||||
"fetched_at": now_iso,
|
||||
}
|
||||
result.update(per_100g)
|
||||
return result
|
||||
140
app/services/produce_reference.py
Normal file
140
app/services/produce_reference.py
Normal file
@@ -0,0 +1,140 @@
|
||||
"""
|
||||
The curated table behind fresh-produce nutrition and fresh-produce images.
|
||||
|
||||
WHY ONE TABLE FOR BOTH
|
||||
----------------------
|
||||
The two defects it fixes have the same cause: nothing in the system knew what a
|
||||
row of `brand_own_products` actually IS. "Apple" was a string, so image search
|
||||
appended a hint meant for confectionery brands and returned juice cartons, and
|
||||
nutrition search asked a packaged-goods database about a fruit and got nothing.
|
||||
One hand-checked line per commodity answers both questions, so a wrong entry is
|
||||
corrected in one place rather than two.
|
||||
|
||||
WHY THE IDs ARE PINNED AND NOT SEARCHED
|
||||
---------------------------------------
|
||||
Free-text search against USDA repeats the same class of error one layer down.
|
||||
Measured live while building this table:
|
||||
|
||||
"curry leaves" -> Drumstick leaves, raw a different plant
|
||||
"snake gourd" -> Gourd, dishcloth a different species
|
||||
"sweet lime" -> Sweet potatoes, raw not remotely the same thing
|
||||
"mosambi" -> Wasabi, root, raw
|
||||
|
||||
So every `usda_fdc_id` was resolved once against the USDA bulk download, its
|
||||
returned description was read by a person, and the id was written down. Four
|
||||
were wrong on the first pass and were corrected: `Peach` had matched *Pears,
|
||||
raw*; `Sweet Potato` had matched *Sweet Potato puffs, frozen*; `Tender Coconut`
|
||||
had matched coconut meat rather than water; `Apple` had matched the
|
||||
without-skin entry.
|
||||
|
||||
WHY SOME ENTRIES HAVE NO USDA ID
|
||||
--------------------------------
|
||||
Because USDA has no equivalent, and the module's contract is to say
|
||||
"unavailable" rather than substitute a near neighbour. Amla is *Phyllanthus
|
||||
emblica* and USDA's "Gooseberries, raw" is *Ribes*, a different genus with
|
||||
different vitamin C by an order of magnitude. Also absent: mosambi/sweet lime,
|
||||
dragon fruit, ash gourd, snake gourd, ivy gourd, cluster beans, baby corn,
|
||||
curry leaves, methi leaves, thulasi, sorrel, agathi keerai, ponnanganni keerai,
|
||||
pomfret, paneer and khoya. Those rows get an image and no nutrition, which is
|
||||
the honest answer.
|
||||
|
||||
The 14 Flowers rows deliberately carry an image and no USDA id at all: they are
|
||||
sold loose beside the vegetables and reach this table the same way, but
|
||||
`consumability.is_non_consumable("Flowers")` is True and they are never scored.
|
||||
|
||||
IMAGES
|
||||
------
|
||||
`image_source` says where the URL came from and is part of the record, not
|
||||
decoration:
|
||||
|
||||
wikipedia_lead the lead image of the English Wikipedia article, which is
|
||||
a curated choice by that article's editors
|
||||
wikimedia_commons hand-picked, because the article's lead image is a
|
||||
19th-century botanical plate (Koehler, Blanco) rather than
|
||||
a photograph - 15 items, each named in `image_note`
|
||||
|
||||
Search terms were resolved to articles by BOTANICAL name wherever the common
|
||||
name is ambiguous, which is the fix for the two worst rows in the live
|
||||
catalogue: "Palak" had been a photograph of a politician and "Thulasi" a
|
||||
photograph of an actress, because both are also common personal names.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
_DATA_FILE = Path(__file__).parent / "data" / "produce_reference.json"
|
||||
|
||||
# Pack sizes and bare quantities never identify a commodity. Mirrors
|
||||
# `generic_products._SIZE_RE`, bounded to real units for the same reason: an
|
||||
# unbounded `[a-z]*` after a number ate the following word.
|
||||
_UNITS = (
|
||||
"kg|kgs|g|gm|gms|gram|grams|mg|ml|l|ltr|ltrs|litre|litres|liter|liters"
|
||||
"|pc|pcs|piece|pieces|pack|packs|pkt|n|no|nos|x|dozen|bunch|bundle"
|
||||
)
|
||||
_SIZE_RE = re.compile(
|
||||
r"\b\d+(?:[.,]\d+)?\s*(?:" + _UNITS + r")\b|\b\d+(?:[.,]\d+)?\b", re.IGNORECASE
|
||||
)
|
||||
_NON_WORD = re.compile(r"[^a-z0-9]+")
|
||||
|
||||
|
||||
def _normalize(name: str) -> str:
|
||||
text = _SIZE_RE.sub(" ", (name or "").lower())
|
||||
return _NON_WORD.sub(" ", text).strip()
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def _reference() -> Dict[str, Dict[str, Any]]:
|
||||
raw = json.loads(_DATA_FILE.read_text(encoding="utf-8"))
|
||||
return {_normalize(k): {**v, "product_name": k} for k, v in raw.items()}
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def _raw_reference() -> Dict[str, Dict[str, Any]]:
|
||||
return json.loads(_DATA_FILE.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def lookup(product_name: str) -> Optional[Dict[str, Any]]:
|
||||
"""The curated entry for a catalogue product name, or None.
|
||||
|
||||
Falls back from the full name to progressively shorter leading phrases, so
|
||||
a size variant or a varietal suffix still finds its base commodity:
|
||||
|
||||
"Mango Alphonso 1kg" -> "mango alphonso" -> hit
|
||||
"Mango Sindoora" -> "mango sindoora" -> miss -> "mango" -> hit
|
||||
|
||||
Leading rather than trailing, because an Indian product name puts its head
|
||||
noun first - the same reading `generic_products.canonical_category` takes.
|
||||
"""
|
||||
key = _normalize(product_name)
|
||||
if not key:
|
||||
return None
|
||||
ref = _reference()
|
||||
tokens = key.split()
|
||||
for end in range(len(tokens), 0, -1):
|
||||
hit = ref.get(" ".join(tokens[:end]))
|
||||
if hit:
|
||||
return hit
|
||||
return None
|
||||
|
||||
|
||||
def usda_fdc_id(product_name: str) -> Optional[int]:
|
||||
entry = lookup(product_name)
|
||||
return entry.get("usda_fdc_id") if entry else None
|
||||
|
||||
|
||||
def image_url(product_name: str) -> Optional[str]:
|
||||
entry = lookup(product_name)
|
||||
return entry.get("image_url") if entry else None
|
||||
|
||||
|
||||
def all_entries() -> Dict[str, Dict[str, Any]]:
|
||||
"""The table as authored, keyed by the catalogue product name."""
|
||||
return dict(_raw_reference())
|
||||
|
||||
|
||||
def known_product_names() -> List[str]:
|
||||
return sorted(_raw_reference())
|
||||
Reference in New Issue
Block a user