product json updates

This commit is contained in:
sriram
2026-09-05 12:14:16 +05:30
parent 7ac3571b59
commit d3709c1a4f
35 changed files with 71662 additions and 68299 deletions

View File

@@ -73,6 +73,29 @@ SCORE_COLUMNS: Tuple[str, str] = ("nutrition_score", "health_score")
# as a parameter.
_SAFE_TABLE = re.compile(r"^[A-Za-z0-9_]+$")
# ONE insight row per image_id, chosen deterministically.
#
# Folding the brand is necessary (see the module docstring) but it can match
# more than one row: `nutrition_insights` genuinely holds both `Grb` and `GRB`
# for the same eleven products, written by two enrichment runs a month apart,
# with different scores. `UPDATE ... FROM` against a multi-row match picks an
# arbitrary one, so the mirror flip-flopped between values on consecutive runs
# - caught by re-running the sync and seeing rows change again.
#
# The tie-break is "a real score beats no score, then newest wins". A NULL is
# the absence of information, not a measurement, so letting it overwrite a
# known score would lose data for nothing. This is the same instinct as the
# COALESCE in the Excel upload's upsert.
_BEST_INSIGHT = """
SELECT DISTINCT ON (image_id)
image_id, nutrition_score, health_score
FROM nutrition_insights
WHERE lower(brand) = lower(%s)
ORDER BY image_id,
(health_score IS NOT NULL) DESC,
generated_at DESC
"""
# ---------------------------------------------------------------------------
# brand tables
@@ -127,8 +150,7 @@ def sync_scores_to_brand_tables(brands: Optional[List[str]] = None,
cur.execute(
f"""
SELECT count(*) FROM {table} b
JOIN nutrition_insights i
ON i.image_id = b.image_id AND lower(i.brand) = lower(%s)
JOIN ({_BEST_INSIGHT}) i ON i.image_id = b.image_id
WHERE b.nutrition_score IS DISTINCT FROM i.nutrition_score
OR b.health_score IS DISTINCT FROM i.health_score
""",
@@ -153,9 +175,8 @@ def sync_scores_to_brand_tables(brands: Optional[List[str]] = None,
UPDATE {table} b
SET nutrition_score = i.nutrition_score,
health_score = i.health_score
FROM nutrition_insights i
FROM ({_BEST_INSIGHT}) i
WHERE i.image_id = b.image_id
AND lower(i.brand) = lower(%s)
AND (b.nutrition_score IS DISTINCT FROM i.nutrition_score
OR b.health_score IS DISTINCT FROM i.health_score)
""",
@@ -210,13 +231,23 @@ def _score_index() -> Dict[Tuple[str, str], Tuple[Optional[float], Optional[floa
return {}
try:
with conn.cursor() as cur:
# Same tie-break as _BEST_INSIGHT, for the same reason: two rows
# can fold to one key ('Grb' and 'GRB'), and without an ORDER BY
# whichever the scan happened to return last would win, so the
# JSON and the brand tables could disagree. Ordered so the winner
# comes first, then first-wins below.
cur.execute(
"SELECT brand, image_id, nutrition_score, health_score "
"FROM nutrition_insights"
"FROM nutrition_insights "
"ORDER BY lower(brand), image_id, "
" (health_score IS NOT NULL) DESC, generated_at DESC"
)
out: Dict[Tuple[str, str], Tuple[Optional[float], Optional[float]]] = {}
for brand, image_id, ns, hs in cur.fetchall():
out[((brand or "").casefold(), image_id)] = (
key = ((brand or "").casefold(), image_id)
if key in out:
continue
out[key] = (
float(ns) if isinstance(ns, Decimal) else ns,
float(hs) if isinstance(hs, Decimal) else hs,
)