product json updates
This commit is contained in:
@@ -73,6 +73,29 @@ SCORE_COLUMNS: Tuple[str, str] = ("nutrition_score", "health_score")
|
||||
# as a parameter.
|
||||
_SAFE_TABLE = re.compile(r"^[A-Za-z0-9_]+$")
|
||||
|
||||
# ONE insight row per image_id, chosen deterministically.
|
||||
#
|
||||
# Folding the brand is necessary (see the module docstring) but it can match
|
||||
# more than one row: `nutrition_insights` genuinely holds both `Grb` and `GRB`
|
||||
# for the same eleven products, written by two enrichment runs a month apart,
|
||||
# with different scores. `UPDATE ... FROM` against a multi-row match picks an
|
||||
# arbitrary one, so the mirror flip-flopped between values on consecutive runs
|
||||
# - caught by re-running the sync and seeing rows change again.
|
||||
#
|
||||
# The tie-break is "a real score beats no score, then newest wins". A NULL is
|
||||
# the absence of information, not a measurement, so letting it overwrite a
|
||||
# known score would lose data for nothing. This is the same instinct as the
|
||||
# COALESCE in the Excel upload's upsert.
|
||||
_BEST_INSIGHT = """
|
||||
SELECT DISTINCT ON (image_id)
|
||||
image_id, nutrition_score, health_score
|
||||
FROM nutrition_insights
|
||||
WHERE lower(brand) = lower(%s)
|
||||
ORDER BY image_id,
|
||||
(health_score IS NOT NULL) DESC,
|
||||
generated_at DESC
|
||||
"""
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# brand tables
|
||||
@@ -127,8 +150,7 @@ def sync_scores_to_brand_tables(brands: Optional[List[str]] = None,
|
||||
cur.execute(
|
||||
f"""
|
||||
SELECT count(*) FROM {table} b
|
||||
JOIN nutrition_insights i
|
||||
ON i.image_id = b.image_id AND lower(i.brand) = lower(%s)
|
||||
JOIN ({_BEST_INSIGHT}) i ON i.image_id = b.image_id
|
||||
WHERE b.nutrition_score IS DISTINCT FROM i.nutrition_score
|
||||
OR b.health_score IS DISTINCT FROM i.health_score
|
||||
""",
|
||||
@@ -153,9 +175,8 @@ def sync_scores_to_brand_tables(brands: Optional[List[str]] = None,
|
||||
UPDATE {table} b
|
||||
SET nutrition_score = i.nutrition_score,
|
||||
health_score = i.health_score
|
||||
FROM nutrition_insights i
|
||||
FROM ({_BEST_INSIGHT}) i
|
||||
WHERE i.image_id = b.image_id
|
||||
AND lower(i.brand) = lower(%s)
|
||||
AND (b.nutrition_score IS DISTINCT FROM i.nutrition_score
|
||||
OR b.health_score IS DISTINCT FROM i.health_score)
|
||||
""",
|
||||
@@ -210,13 +231,23 @@ def _score_index() -> Dict[Tuple[str, str], Tuple[Optional[float], Optional[floa
|
||||
return {}
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
# Same tie-break as _BEST_INSIGHT, for the same reason: two rows
|
||||
# can fold to one key ('Grb' and 'GRB'), and without an ORDER BY
|
||||
# whichever the scan happened to return last would win, so the
|
||||
# JSON and the brand tables could disagree. Ordered so the winner
|
||||
# comes first, then first-wins below.
|
||||
cur.execute(
|
||||
"SELECT brand, image_id, nutrition_score, health_score "
|
||||
"FROM nutrition_insights"
|
||||
"FROM nutrition_insights "
|
||||
"ORDER BY lower(brand), image_id, "
|
||||
" (health_score IS NOT NULL) DESC, generated_at DESC"
|
||||
)
|
||||
out: Dict[Tuple[str, str], Tuple[Optional[float], Optional[float]]] = {}
|
||||
for brand, image_id, ns, hs in cur.fetchall():
|
||||
out[((brand or "").casefold(), image_id)] = (
|
||||
key = ((brand or "").casefold(), image_id)
|
||||
if key in out:
|
||||
continue
|
||||
out[key] = (
|
||||
float(ns) if isinstance(ns, Decimal) else ns,
|
||||
float(hs) if isinstance(hs, Decimal) else hs,
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user