backend updates on recommendation system

This commit is contained in:
sriram
2026-10-07 11:03:42 +05:30
parent 82f5db1250
commit 71dbb2a6e9
6 changed files with 599 additions and 6 deletions

View File

@@ -0,0 +1,144 @@
"""Which other products to suggest under a product's ratings and reviews
(docs/RECOMMENDATIONS.md): Phase 1 "Similar products" and Phase 2 "Better
rated alternatives".
This module only scores and orders candidates; the database queries that find
them (repository.similar_products / rated_products) and the endpoint live
elsewhere. Nothing here invents a number: a product with no published rating
is scored at the pool's average, and one with no price is never a candidate.
Hard price limit: every candidate's best price is within SIMILAR_PRICE_BAND
(similar) or BETTER_PRICE_BAND (better rated) of the product's own, applied in
the database query. A product with no price of its own gets no limit.
Similar products:
score = 0.60 x similarity + 0.25 x rating quality + 0.15 x price closeness
similarity cosine similarity of the two product embeddings (0..1)
rating quality Bayesian average / 5, so 5.0 from 3 ratings does not beat
4.4 from 2,000: each product's rating is pulled towards the
pool average as if PRIOR_WEIGHT extra ratings at that average
had been given
price closeness 1 at the same price, falling to 0 at twice (or zero) the price
When fewer than MIN_MATCHES similar products qualify, the list is filled with
the best-rated products of the same category (in the same price band).
Better rated alternatives: products rated higher than this one by at least
MIN_REVIEWS_BETTER people, best Bayesian rating first. When two round to the
same Bayesian rating, the one whose stored reviews are more positive (share
of positive minus share of negative) goes first.
"""
from __future__ import annotations
from typing import Any, Dict, List, Optional
WEIGHT_SIMILARITY, WEIGHT_RATING, WEIGHT_PRICE = 0.60, 0.25, 0.15
PRIOR_WEIGHT = 50
DEFAULT_PRIOR_MEAN = 4.0 # used only when no product in the pool has a rating
MAX_ITEMS = 6
MIN_MATCHES = 3
SIMILAR_PRICE_BAND = 0.30 # +/-30% of the product's best price
BETTER_PRICE_BAND = 0.20 # +/-20%
MIN_REVIEWS_BETTER = 5
def prior_mean(candidates: List[Dict[str, Any]]) -> float:
"""Average published rating across the pool."""
rated = [c["rating"] for c in candidates if c.get("rating") is not None]
return sum(rated) / len(rated) if rated else DEFAULT_PRIOR_MEAN
def bayesian_rating(rating: Optional[float], count: Optional[int], mean: float) -> float:
if rating is None:
return mean
n = max(int(count or 0), 1) # a rating with no stated count weighs as one
return (PRIOR_WEIGHT * mean + n * float(rating)) / (PRIOR_WEIGHT + n)
def price_closeness(price: Optional[float], target_price: Optional[float]) -> float:
if price is None or not target_price:
return 0.5 # unknown: neither helps nor hurts
return max(0.0, 1.0 - abs(float(price) - float(target_price)) / float(target_price))
def score(candidate: Dict[str, Any], target_price: Optional[float], mean: float) -> float:
quality = bayesian_rating(candidate.get("rating"), candidate.get("rating_count"), mean) / 5
return (WEIGHT_SIMILARITY * float(candidate.get("similarity") or 0.0)
+ WEIGHT_RATING * quality
+ WEIGHT_PRICE * price_closeness(candidate.get("best_price"), target_price))
def reason(candidate: Dict[str, Any], target_rating: Optional[float], basis: str) -> str:
label = "Similar specs" if basis == "similar" else "Top rated in this category"
rating = candidate.get("rating")
if rating is None:
return label
if target_rating is not None and rating > target_rating:
return f"{label} · {rating:.1f}★ vs {target_rating:.1f}★"
count = candidate.get("rating_count")
return f"{label} · {rating:.1f}★" + (f" ({count:,} rating{'' if count == 1 else 's'})" if count else "")
def recommend(
target: Dict[str, Any],
similar: List[Dict[str, Any]],
top_rated: List[Dict[str, Any]],
limit: int = MAX_ITEMS,
) -> List[Dict[str, Any]]:
"""Order `similar` by score; when fewer than MIN_MATCHES are found, fill up
to `limit` from `top_rated` (best Bayesian rating first).
Each candidate is a dict with product_id, best_price, rating, rating_count
and (for `similar`) similarity. `target` has best_price and rating.
Returns the chosen candidates with `reason` and `basis` added."""
mean = prior_mean(similar + top_rated)
target_price, target_rating = target.get("best_price"), target.get("rating")
ranked = sorted(similar, key=lambda c: score(c, target_price, mean), reverse=True)[:limit]
picked = [{**c, "basis": "similar"} for c in ranked]
if len(picked) < MIN_MATCHES:
seen = {c["product_id"] for c in picked}
fill = sorted((c for c in top_rated if c["product_id"] not in seen and c.get("rating") is not None),
key=lambda c: bayesian_rating(c["rating"], c.get("rating_count"), mean), reverse=True)
picked += [{**c, "basis": "top_rated"} for c in fill[: limit - len(picked)]]
for c in picked:
c["reason"] = reason(c, target_rating, c["basis"])
return picked
def sentiment_balance(sentiment: Optional[Dict[str, int]]) -> float:
"""Share of positive minus share of negative stored reviews (-1..1); 0 with none."""
if not sentiment:
return 0.0
total = sum(sentiment.values())
if not total:
return 0.0
return (sentiment.get("positive", 0) - sentiment.get("negative", 0)) / total
def better_rated(
target: Dict[str, Any],
rated: List[Dict[str, Any]],
limit: int = MAX_ITEMS,
) -> List[Dict[str, Any]]:
"""The products in `rated` rated higher than `target` (any rated product
when the target has no rating), each rated by at least MIN_REVIEWS_BETTER
people. Each candidate has rating, rating_count and optionally `sentiment`
({"positive": n, "neutral": n, "negative": n})."""
target_rating = target.get("rating")
mean = prior_mean(rated + [target])
keep = [c for c in rated
if c.get("rating") is not None and (c.get("rating_count") or 0) >= MIN_REVIEWS_BETTER
and (target_rating is None or c["rating"] > target_rating)]
keep.sort(key=lambda c: (round(bayesian_rating(c["rating"], c["rating_count"], mean), 1),
sentiment_balance(c.get("sentiment"))), reverse=True)
picked = []
for c in keep[:limit]:
vs = f" vs {target_rating:.1f}★" if target_rating is not None else ""
picked.append({**c, "basis": "better_rated",
"reason": f"{c['rating']:.1f}★{vs} · {c['rating_count']:,} ratings"})
return picked