backend updates on recommendation system

This commit is contained in:
sriram
2026-10-07 11:03:42 +05:30
parent 82f5db1250
commit 71dbb2a6e9
6 changed files with 599 additions and 6 deletions

View File

@@ -305,6 +305,96 @@ def product_rating_and_reviews(conn, product_id: int) -> Dict[str, Any]:
return {"sources": [dict(s) for s in sources], "reviews": [dict(r) for r in reviews]}
def rating_sources_for(conn, product_ids: List[int]) -> Dict[int, List[dict]]:
"""product_rating_and_reviews' per-platform ratings for many products at once."""
out: Dict[int, List[dict]] = {pid: [] for pid in product_ids}
rows = conn.execute(
"SELECT a.product_id, a.site, a.source_url, l.rating, l.review_count, l.rating_breakdown "
"FROM elec.v_product_availability a JOIN elec.source_listing l ON l.id = a.listing_id "
"WHERE a.product_id = ANY(%s) AND l.rating > 0 ORDER BY l.review_count DESC NULLS LAST, a.site",
(list(product_ids),),
).fetchall()
for r in rows:
out[r["product_id"]].append(dict(r))
return out
# p is a variant of t when the two differ only in RAM/storage: same brand and
# model and the same processor. One laptop model line ("HP 15") spans many
# CPUs, so laptops count as variants only when both state the same processor;
# phones state none, so for them the model alone decides. Written so it is
# never NULL: NOT NULL would drop the product from recommendations too.
_SAME_MODEL = """(p.brand_id = t.brand_id AND p.model_norm = t.model_norm
AND p.processor IS NOT DISTINCT FROM t.processor
AND (p.processor IS NOT NULL
OR t.category_id IS DISTINCT FROM (SELECT id FROM elec.category WHERE slug = 'laptops')))"""
# Products that may be recommended for target t: verified, same category, not
# t itself or another RAM/storage variant of the same model, with an in-stock
# (or unknown-stock) best price within +/- %(band)s of t's own (no limit when t
# has no price).
_RECOMMENDABLE = """
FROM elec.product t
JOIN elec.product p ON p.category_id = t.category_id AND p.id <> t.id
AND p.verification_status = 'verified'
AND NOT """ + _SAME_MODEL + """
JOIN elec.v_best_price bp ON bp.product_id = p.id AND bp.in_stock IS DISTINCT FROM FALSE
LEFT JOIN elec.v_best_price tp ON tp.product_id = t.id
WHERE t.id = %(pid)s
AND (tp.price IS NULL OR bp.price BETWEEN tp.price * (1 - %(band)s) AND tp.price * (1 + %(band)s))
"""
_TN_ONLY = (" AND EXISTS (SELECT 1 FROM elec.v_product_availability a"
" WHERE a.product_id = p.id AND a.site_region = 'TN')")
def similar_products(conn, product_id: int, *, band: float, limit: int = 30, tn_only: bool = False) -> List[dict]:
"""Recommendable products closest to this one by embedding (cosine), with
their best price. Empty when the product has no embedding yet.
An exact scan, not the HNSW index: the category/stock filters would make an
approximate index search drop matches, and a category is small enough."""
sql = ("SELECT p.id AS product_id, 1 - (p.embedding <=> t.embedding) AS similarity, bp.price AS best_price"
+ _RECOMMENDABLE + " AND t.embedding IS NOT NULL AND p.embedding IS NOT NULL"
+ (_TN_ONLY if tn_only else "")
+ " ORDER BY p.embedding <=> t.embedding LIMIT %(limit)s")
return [dict(r) for r in conn.execute(sql, {"pid": product_id, "band": band, "limit": limit})]
def rated_products(conn, product_id: int, *, band: float, tn_only: bool = False) -> List[dict]:
"""Recommendable products that at least one platform has rated."""
sql = ("SELECT p.id AS product_id, bp.price AS best_price" + _RECOMMENDABLE
+ " AND EXISTS (SELECT 1 FROM elec.v_product_availability a JOIN elec.source_listing l"
" ON l.id = a.listing_id WHERE a.product_id = p.id AND l.rating > 0)"
+ (_TN_ONLY if tn_only else ""))
return [dict(r) for r in conn.execute(sql, {"pid": product_id, "band": band})]
def review_sentiment_counts(conn, product_ids: List[int]) -> Dict[int, Dict[str, int]]:
"""How many stored reviews of each product are positive / neutral / negative."""
out: Dict[int, Dict[str, int]] = {pid: {} for pid in product_ids}
rows = conn.execute(
"SELECT a.product_id, r.sentiment, count(*)::int AS n "
"FROM elec.v_product_availability a JOIN elec.listing_review r ON r.listing_id = a.listing_id "
"WHERE a.product_id = ANY(%s) AND r.sentiment IS NOT NULL GROUP BY 1, 2",
(list(product_ids),),
).fetchall()
for r in rows:
out[r["product_id"]][r["sentiment"]] = r["n"]
return out
def other_variants(conn, product_id: int) -> List[dict]:
"""The same model's other verified RAM/storage variants (see _SAME_MODEL)."""
return [dict(r) for r in conn.execute(
"SELECT v.product_id, v.display_name, v.ram_gb, v.storage_gb, v.best_price "
"FROM elec.product t JOIN elec.product p ON p.category_id = t.category_id AND p.id <> t.id "
"AND " + _SAME_MODEL + " "
"JOIN elec.v_brand_catalog v ON v.product_id = p.id "
"WHERE t.id = %s ORDER BY v.ram_gb NULLS LAST, v.storage_gb NULLS LAST, v.display_name",
(product_id,),
)]
def listings_for_review_backfill(category: Optional[str] = None) -> List[dict]:
"""Page-read listings of verified products, for re-reading ratings/reviews."""
sql = (

View File

@@ -0,0 +1,144 @@
"""Which other products to suggest under a product's ratings and reviews
(docs/RECOMMENDATIONS.md): Phase 1 "Similar products" and Phase 2 "Better
rated alternatives".
This module only scores and orders candidates; the database queries that find
them (repository.similar_products / rated_products) and the endpoint live
elsewhere. Nothing here invents a number: a product with no published rating
is scored at the pool's average, and one with no price is never a candidate.
Hard price limit: every candidate's best price is within SIMILAR_PRICE_BAND
(similar) or BETTER_PRICE_BAND (better rated) of the product's own, applied in
the database query. A product with no price of its own gets no limit.
Similar products:
score = 0.60 x similarity + 0.25 x rating quality + 0.15 x price closeness
similarity cosine similarity of the two product embeddings (0..1)
rating quality Bayesian average / 5, so 5.0 from 3 ratings does not beat
4.4 from 2,000: each product's rating is pulled towards the
pool average as if PRIOR_WEIGHT extra ratings at that average
had been given
price closeness 1 at the same price, falling to 0 at twice (or zero) the price
When fewer than MIN_MATCHES similar products qualify, the list is filled with
the best-rated products of the same category (in the same price band).
Better rated alternatives: products rated higher than this one by at least
MIN_REVIEWS_BETTER people, best Bayesian rating first. When two round to the
same Bayesian rating, the one whose stored reviews are more positive (share
of positive minus share of negative) goes first.
"""
from __future__ import annotations
from typing import Any, Dict, List, Optional
WEIGHT_SIMILARITY, WEIGHT_RATING, WEIGHT_PRICE = 0.60, 0.25, 0.15
PRIOR_WEIGHT = 50
DEFAULT_PRIOR_MEAN = 4.0 # used only when no product in the pool has a rating
MAX_ITEMS = 6
MIN_MATCHES = 3
SIMILAR_PRICE_BAND = 0.30 # +/-30% of the product's best price
BETTER_PRICE_BAND = 0.20 # +/-20%
MIN_REVIEWS_BETTER = 5
def prior_mean(candidates: List[Dict[str, Any]]) -> float:
"""Average published rating across the pool."""
rated = [c["rating"] for c in candidates if c.get("rating") is not None]
return sum(rated) / len(rated) if rated else DEFAULT_PRIOR_MEAN
def bayesian_rating(rating: Optional[float], count: Optional[int], mean: float) -> float:
if rating is None:
return mean
n = max(int(count or 0), 1) # a rating with no stated count weighs as one
return (PRIOR_WEIGHT * mean + n * float(rating)) / (PRIOR_WEIGHT + n)
def price_closeness(price: Optional[float], target_price: Optional[float]) -> float:
if price is None or not target_price:
return 0.5 # unknown: neither helps nor hurts
return max(0.0, 1.0 - abs(float(price) - float(target_price)) / float(target_price))
def score(candidate: Dict[str, Any], target_price: Optional[float], mean: float) -> float:
quality = bayesian_rating(candidate.get("rating"), candidate.get("rating_count"), mean) / 5
return (WEIGHT_SIMILARITY * float(candidate.get("similarity") or 0.0)
+ WEIGHT_RATING * quality
+ WEIGHT_PRICE * price_closeness(candidate.get("best_price"), target_price))
def reason(candidate: Dict[str, Any], target_rating: Optional[float], basis: str) -> str:
label = "Similar specs" if basis == "similar" else "Top rated in this category"
rating = candidate.get("rating")
if rating is None:
return label
if target_rating is not None and rating > target_rating:
return f"{label} · {rating:.1f}★ vs {target_rating:.1f}★"
count = candidate.get("rating_count")
return f"{label} · {rating:.1f}★" + (f" ({count:,} rating{'' if count == 1 else 's'})" if count else "")
def recommend(
target: Dict[str, Any],
similar: List[Dict[str, Any]],
top_rated: List[Dict[str, Any]],
limit: int = MAX_ITEMS,
) -> List[Dict[str, Any]]:
"""Order `similar` by score; when fewer than MIN_MATCHES are found, fill up
to `limit` from `top_rated` (best Bayesian rating first).
Each candidate is a dict with product_id, best_price, rating, rating_count
and (for `similar`) similarity. `target` has best_price and rating.
Returns the chosen candidates with `reason` and `basis` added."""
mean = prior_mean(similar + top_rated)
target_price, target_rating = target.get("best_price"), target.get("rating")
ranked = sorted(similar, key=lambda c: score(c, target_price, mean), reverse=True)[:limit]
picked = [{**c, "basis": "similar"} for c in ranked]
if len(picked) < MIN_MATCHES:
seen = {c["product_id"] for c in picked}
fill = sorted((c for c in top_rated if c["product_id"] not in seen and c.get("rating") is not None),
key=lambda c: bayesian_rating(c["rating"], c.get("rating_count"), mean), reverse=True)
picked += [{**c, "basis": "top_rated"} for c in fill[: limit - len(picked)]]
for c in picked:
c["reason"] = reason(c, target_rating, c["basis"])
return picked
def sentiment_balance(sentiment: Optional[Dict[str, int]]) -> float:
"""Share of positive minus share of negative stored reviews (-1..1); 0 with none."""
if not sentiment:
return 0.0
total = sum(sentiment.values())
if not total:
return 0.0
return (sentiment.get("positive", 0) - sentiment.get("negative", 0)) / total
def better_rated(
target: Dict[str, Any],
rated: List[Dict[str, Any]],
limit: int = MAX_ITEMS,
) -> List[Dict[str, Any]]:
"""The products in `rated` rated higher than `target` (any rated product
when the target has no rating), each rated by at least MIN_REVIEWS_BETTER
people. Each candidate has rating, rating_count and optionally `sentiment`
({"positive": n, "neutral": n, "negative": n})."""
target_rating = target.get("rating")
mean = prior_mean(rated + [target])
keep = [c for c in rated
if c.get("rating") is not None and (c.get("rating_count") or 0) >= MIN_REVIEWS_BETTER
and (target_rating is None or c["rating"] > target_rating)]
keep.sort(key=lambda c: (round(bayesian_rating(c["rating"], c["rating_count"], mean), 1),
sentiment_balance(c.get("sentiment"))), reverse=True)
picked = []
for c in keep[:limit]:
vs = f" vs {target_rating:.1f}★" if target_rating is not None else ""
picked.append({**c, "basis": "better_rated",
"reason": f"{c['rating']:.1f}★{vs} · {c['rating_count']:,} ratings"})
return picked