backend updates on recommendation system
This commit is contained in:
@@ -305,6 +305,96 @@ def product_rating_and_reviews(conn, product_id: int) -> Dict[str, Any]:
|
||||
return {"sources": [dict(s) for s in sources], "reviews": [dict(r) for r in reviews]}
|
||||
|
||||
|
||||
def rating_sources_for(conn, product_ids: List[int]) -> Dict[int, List[dict]]:
|
||||
"""product_rating_and_reviews' per-platform ratings for many products at once."""
|
||||
out: Dict[int, List[dict]] = {pid: [] for pid in product_ids}
|
||||
rows = conn.execute(
|
||||
"SELECT a.product_id, a.site, a.source_url, l.rating, l.review_count, l.rating_breakdown "
|
||||
"FROM elec.v_product_availability a JOIN elec.source_listing l ON l.id = a.listing_id "
|
||||
"WHERE a.product_id = ANY(%s) AND l.rating > 0 ORDER BY l.review_count DESC NULLS LAST, a.site",
|
||||
(list(product_ids),),
|
||||
).fetchall()
|
||||
for r in rows:
|
||||
out[r["product_id"]].append(dict(r))
|
||||
return out
|
||||
|
||||
|
||||
# p is a variant of t when the two differ only in RAM/storage: same brand and
|
||||
# model and the same processor. One laptop model line ("HP 15") spans many
|
||||
# CPUs, so laptops count as variants only when both state the same processor;
|
||||
# phones state none, so for them the model alone decides. Written so it is
|
||||
# never NULL: NOT NULL would drop the product from recommendations too.
|
||||
_SAME_MODEL = """(p.brand_id = t.brand_id AND p.model_norm = t.model_norm
|
||||
AND p.processor IS NOT DISTINCT FROM t.processor
|
||||
AND (p.processor IS NOT NULL
|
||||
OR t.category_id IS DISTINCT FROM (SELECT id FROM elec.category WHERE slug = 'laptops')))"""
|
||||
|
||||
# Products that may be recommended for target t: verified, same category, not
|
||||
# t itself or another RAM/storage variant of the same model, with an in-stock
|
||||
# (or unknown-stock) best price within +/- %(band)s of t's own (no limit when t
|
||||
# has no price).
|
||||
_RECOMMENDABLE = """
|
||||
FROM elec.product t
|
||||
JOIN elec.product p ON p.category_id = t.category_id AND p.id <> t.id
|
||||
AND p.verification_status = 'verified'
|
||||
AND NOT """ + _SAME_MODEL + """
|
||||
JOIN elec.v_best_price bp ON bp.product_id = p.id AND bp.in_stock IS DISTINCT FROM FALSE
|
||||
LEFT JOIN elec.v_best_price tp ON tp.product_id = t.id
|
||||
WHERE t.id = %(pid)s
|
||||
AND (tp.price IS NULL OR bp.price BETWEEN tp.price * (1 - %(band)s) AND tp.price * (1 + %(band)s))
|
||||
"""
|
||||
_TN_ONLY = (" AND EXISTS (SELECT 1 FROM elec.v_product_availability a"
|
||||
" WHERE a.product_id = p.id AND a.site_region = 'TN')")
|
||||
|
||||
|
||||
def similar_products(conn, product_id: int, *, band: float, limit: int = 30, tn_only: bool = False) -> List[dict]:
|
||||
"""Recommendable products closest to this one by embedding (cosine), with
|
||||
their best price. Empty when the product has no embedding yet.
|
||||
|
||||
An exact scan, not the HNSW index: the category/stock filters would make an
|
||||
approximate index search drop matches, and a category is small enough."""
|
||||
sql = ("SELECT p.id AS product_id, 1 - (p.embedding <=> t.embedding) AS similarity, bp.price AS best_price"
|
||||
+ _RECOMMENDABLE + " AND t.embedding IS NOT NULL AND p.embedding IS NOT NULL"
|
||||
+ (_TN_ONLY if tn_only else "")
|
||||
+ " ORDER BY p.embedding <=> t.embedding LIMIT %(limit)s")
|
||||
return [dict(r) for r in conn.execute(sql, {"pid": product_id, "band": band, "limit": limit})]
|
||||
|
||||
|
||||
def rated_products(conn, product_id: int, *, band: float, tn_only: bool = False) -> List[dict]:
|
||||
"""Recommendable products that at least one platform has rated."""
|
||||
sql = ("SELECT p.id AS product_id, bp.price AS best_price" + _RECOMMENDABLE
|
||||
+ " AND EXISTS (SELECT 1 FROM elec.v_product_availability a JOIN elec.source_listing l"
|
||||
" ON l.id = a.listing_id WHERE a.product_id = p.id AND l.rating > 0)"
|
||||
+ (_TN_ONLY if tn_only else ""))
|
||||
return [dict(r) for r in conn.execute(sql, {"pid": product_id, "band": band})]
|
||||
|
||||
|
||||
def review_sentiment_counts(conn, product_ids: List[int]) -> Dict[int, Dict[str, int]]:
|
||||
"""How many stored reviews of each product are positive / neutral / negative."""
|
||||
out: Dict[int, Dict[str, int]] = {pid: {} for pid in product_ids}
|
||||
rows = conn.execute(
|
||||
"SELECT a.product_id, r.sentiment, count(*)::int AS n "
|
||||
"FROM elec.v_product_availability a JOIN elec.listing_review r ON r.listing_id = a.listing_id "
|
||||
"WHERE a.product_id = ANY(%s) AND r.sentiment IS NOT NULL GROUP BY 1, 2",
|
||||
(list(product_ids),),
|
||||
).fetchall()
|
||||
for r in rows:
|
||||
out[r["product_id"]][r["sentiment"]] = r["n"]
|
||||
return out
|
||||
|
||||
|
||||
def other_variants(conn, product_id: int) -> List[dict]:
|
||||
"""The same model's other verified RAM/storage variants (see _SAME_MODEL)."""
|
||||
return [dict(r) for r in conn.execute(
|
||||
"SELECT v.product_id, v.display_name, v.ram_gb, v.storage_gb, v.best_price "
|
||||
"FROM elec.product t JOIN elec.product p ON p.category_id = t.category_id AND p.id <> t.id "
|
||||
"AND " + _SAME_MODEL + " "
|
||||
"JOIN elec.v_brand_catalog v ON v.product_id = p.id "
|
||||
"WHERE t.id = %s ORDER BY v.ram_gb NULLS LAST, v.storage_gb NULLS LAST, v.display_name",
|
||||
(product_id,),
|
||||
)]
|
||||
|
||||
|
||||
def listings_for_review_backfill(category: Optional[str] = None) -> List[dict]:
|
||||
"""Page-read listings of verified products, for re-reading ratings/reviews."""
|
||||
sql = (
|
||||
|
||||
144
backend/app/electronics/recommend.py
Normal file
144
backend/app/electronics/recommend.py
Normal file
@@ -0,0 +1,144 @@
|
||||
"""Which other products to suggest under a product's ratings and reviews
|
||||
(docs/RECOMMENDATIONS.md): Phase 1 "Similar products" and Phase 2 "Better
|
||||
rated alternatives".
|
||||
|
||||
This module only scores and orders candidates; the database queries that find
|
||||
them (repository.similar_products / rated_products) and the endpoint live
|
||||
elsewhere. Nothing here invents a number: a product with no published rating
|
||||
is scored at the pool's average, and one with no price is never a candidate.
|
||||
|
||||
Hard price limit: every candidate's best price is within SIMILAR_PRICE_BAND
|
||||
(similar) or BETTER_PRICE_BAND (better rated) of the product's own, applied in
|
||||
the database query. A product with no price of its own gets no limit.
|
||||
|
||||
Similar products:
|
||||
|
||||
score = 0.60 x similarity + 0.25 x rating quality + 0.15 x price closeness
|
||||
|
||||
similarity cosine similarity of the two product embeddings (0..1)
|
||||
rating quality Bayesian average / 5, so 5.0 from 3 ratings does not beat
|
||||
4.4 from 2,000: each product's rating is pulled towards the
|
||||
pool average as if PRIOR_WEIGHT extra ratings at that average
|
||||
had been given
|
||||
price closeness 1 at the same price, falling to 0 at twice (or zero) the price
|
||||
|
||||
When fewer than MIN_MATCHES similar products qualify, the list is filled with
|
||||
the best-rated products of the same category (in the same price band).
|
||||
|
||||
Better rated alternatives: products rated higher than this one by at least
|
||||
MIN_REVIEWS_BETTER people, best Bayesian rating first. When two round to the
|
||||
same Bayesian rating, the one whose stored reviews are more positive (share
|
||||
of positive minus share of negative) goes first.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
WEIGHT_SIMILARITY, WEIGHT_RATING, WEIGHT_PRICE = 0.60, 0.25, 0.15
|
||||
PRIOR_WEIGHT = 50
|
||||
DEFAULT_PRIOR_MEAN = 4.0 # used only when no product in the pool has a rating
|
||||
MAX_ITEMS = 6
|
||||
MIN_MATCHES = 3
|
||||
SIMILAR_PRICE_BAND = 0.30 # +/-30% of the product's best price
|
||||
BETTER_PRICE_BAND = 0.20 # +/-20%
|
||||
MIN_REVIEWS_BETTER = 5
|
||||
|
||||
|
||||
def prior_mean(candidates: List[Dict[str, Any]]) -> float:
|
||||
"""Average published rating across the pool."""
|
||||
rated = [c["rating"] for c in candidates if c.get("rating") is not None]
|
||||
return sum(rated) / len(rated) if rated else DEFAULT_PRIOR_MEAN
|
||||
|
||||
|
||||
def bayesian_rating(rating: Optional[float], count: Optional[int], mean: float) -> float:
|
||||
if rating is None:
|
||||
return mean
|
||||
n = max(int(count or 0), 1) # a rating with no stated count weighs as one
|
||||
return (PRIOR_WEIGHT * mean + n * float(rating)) / (PRIOR_WEIGHT + n)
|
||||
|
||||
|
||||
def price_closeness(price: Optional[float], target_price: Optional[float]) -> float:
|
||||
if price is None or not target_price:
|
||||
return 0.5 # unknown: neither helps nor hurts
|
||||
return max(0.0, 1.0 - abs(float(price) - float(target_price)) / float(target_price))
|
||||
|
||||
|
||||
def score(candidate: Dict[str, Any], target_price: Optional[float], mean: float) -> float:
|
||||
quality = bayesian_rating(candidate.get("rating"), candidate.get("rating_count"), mean) / 5
|
||||
return (WEIGHT_SIMILARITY * float(candidate.get("similarity") or 0.0)
|
||||
+ WEIGHT_RATING * quality
|
||||
+ WEIGHT_PRICE * price_closeness(candidate.get("best_price"), target_price))
|
||||
|
||||
|
||||
def reason(candidate: Dict[str, Any], target_rating: Optional[float], basis: str) -> str:
|
||||
label = "Similar specs" if basis == "similar" else "Top rated in this category"
|
||||
rating = candidate.get("rating")
|
||||
if rating is None:
|
||||
return label
|
||||
if target_rating is not None and rating > target_rating:
|
||||
return f"{label} · {rating:.1f}★ vs {target_rating:.1f}★"
|
||||
count = candidate.get("rating_count")
|
||||
return f"{label} · {rating:.1f}★" + (f" ({count:,} rating{'' if count == 1 else 's'})" if count else "")
|
||||
|
||||
|
||||
def recommend(
|
||||
target: Dict[str, Any],
|
||||
similar: List[Dict[str, Any]],
|
||||
top_rated: List[Dict[str, Any]],
|
||||
limit: int = MAX_ITEMS,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Order `similar` by score; when fewer than MIN_MATCHES are found, fill up
|
||||
to `limit` from `top_rated` (best Bayesian rating first).
|
||||
|
||||
Each candidate is a dict with product_id, best_price, rating, rating_count
|
||||
and (for `similar`) similarity. `target` has best_price and rating.
|
||||
Returns the chosen candidates with `reason` and `basis` added."""
|
||||
mean = prior_mean(similar + top_rated)
|
||||
target_price, target_rating = target.get("best_price"), target.get("rating")
|
||||
|
||||
ranked = sorted(similar, key=lambda c: score(c, target_price, mean), reverse=True)[:limit]
|
||||
picked = [{**c, "basis": "similar"} for c in ranked]
|
||||
|
||||
if len(picked) < MIN_MATCHES:
|
||||
seen = {c["product_id"] for c in picked}
|
||||
fill = sorted((c for c in top_rated if c["product_id"] not in seen and c.get("rating") is not None),
|
||||
key=lambda c: bayesian_rating(c["rating"], c.get("rating_count"), mean), reverse=True)
|
||||
picked += [{**c, "basis": "top_rated"} for c in fill[: limit - len(picked)]]
|
||||
|
||||
for c in picked:
|
||||
c["reason"] = reason(c, target_rating, c["basis"])
|
||||
return picked
|
||||
|
||||
|
||||
def sentiment_balance(sentiment: Optional[Dict[str, int]]) -> float:
|
||||
"""Share of positive minus share of negative stored reviews (-1..1); 0 with none."""
|
||||
if not sentiment:
|
||||
return 0.0
|
||||
total = sum(sentiment.values())
|
||||
if not total:
|
||||
return 0.0
|
||||
return (sentiment.get("positive", 0) - sentiment.get("negative", 0)) / total
|
||||
|
||||
|
||||
def better_rated(
|
||||
target: Dict[str, Any],
|
||||
rated: List[Dict[str, Any]],
|
||||
limit: int = MAX_ITEMS,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""The products in `rated` rated higher than `target` (any rated product
|
||||
when the target has no rating), each rated by at least MIN_REVIEWS_BETTER
|
||||
people. Each candidate has rating, rating_count and optionally `sentiment`
|
||||
({"positive": n, "neutral": n, "negative": n})."""
|
||||
target_rating = target.get("rating")
|
||||
mean = prior_mean(rated + [target])
|
||||
keep = [c for c in rated
|
||||
if c.get("rating") is not None and (c.get("rating_count") or 0) >= MIN_REVIEWS_BETTER
|
||||
and (target_rating is None or c["rating"] > target_rating)]
|
||||
keep.sort(key=lambda c: (round(bayesian_rating(c["rating"], c["rating_count"], mean), 1),
|
||||
sentiment_balance(c.get("sentiment"))), reverse=True)
|
||||
picked = []
|
||||
for c in keep[:limit]:
|
||||
vs = f" vs {target_rating:.1f}★" if target_rating is not None else ""
|
||||
picked.append({**c, "basis": "better_rated",
|
||||
"reason": f"{c['rating']:.1f}★{vs} · {c['rating_count']:,} ratings"})
|
||||
return picked
|
||||
Reference in New Issue
Block a user