backend updates on recommendation system

This commit is contained in:
sriram
2026-10-07 11:03:42 +05:30
parent 82f5db1250
commit 71dbb2a6e9
6 changed files with 599 additions and 6 deletions

View File

@@ -7,12 +7,20 @@ Money is returned as a decimal string, never a float.
from __future__ import annotations
from decimal import Decimal
from typing import Any, Dict, List, Optional
from typing import Any, Dict, List, Literal, Optional
from fastapi import APIRouter, HTTPException, Query
from app.electronics.db.connection import connect
from app.electronics.db.repository import product_rating_and_reviews
from app.electronics import recommend as rec
from app.electronics.db.repository import (
other_variants,
product_rating_and_reviews,
rated_products,
rating_sources_for,
review_sentiment_counts,
similar_products,
)
from app.electronics.reviews import select_reviews
router = APIRouter(prefix="/elec", tags=["electronics"])
@@ -106,7 +114,13 @@ def products(
f"LIMIT %(limit)s OFFSET %(offset)s",
{**params, "limit": limit, "offset": offset},
).fetchall()
return {"total": total, "products": [_clean(r) for r in rows]}
sources = rating_sources_for(conn, [r["product_id"] for r in rows])
out = []
for r in rows:
rating = _overall_rating(sources[r["product_id"]])
out.append({**_clean(r), "rating": rating["value"] if rating else None,
"rating_count": rating["count"] if rating else None})
return {"total": total, "products": out}
@router.get("/products/{product_id}")
@@ -185,6 +199,63 @@ def _breakdown(sources: List[dict]) -> Optional[List[dict]]:
for k in ("5", "4", "3", "2", "1")]
@router.get("/products/{product_id}/recommendations")
def recommendations(
product_id: int,
kind: Literal["similar", "better_rated"] = Query("similar", alias="type"),
limit: int = Query(rec.MAX_ITEMS, ge=1, le=12),
tn_only: bool = False,
) -> dict:
"""Products to suggest under this one's ratings and reviews (see
app/electronics/recommend.py), plus the same model's other variants.
type=similar: closest specs; type=better_rated: rated higher, similar price."""
with connect() as conn:
target = conn.execute(
"SELECT product_id, best_price FROM elec.v_brand_catalog WHERE product_id = %s", (product_id,)
).fetchone()
if not target:
raise HTTPException(status_code=404, detail="Product not found or not verified")
if kind == "similar":
similar = similar_products(conn, product_id, band=rec.SIMILAR_PRICE_BAND, tn_only=tn_only)
rated = (rated_products(conn, product_id, band=rec.SIMILAR_PRICE_BAND, tn_only=tn_only)
if len(similar) < rec.MIN_MATCHES else [])
else:
similar = []
rated = rated_products(conn, product_id, band=rec.BETTER_PRICE_BAND, tn_only=tn_only)
ids = {product_id} | {c["product_id"] for c in similar + rated}
ratings = {pid: _overall_rating(src) for pid, src in rating_sources_for(conn, list(ids)).items()}
def with_rating(c: dict) -> dict:
r = ratings.get(c["product_id"])
return {**c, "best_price": _float(c.get("best_price")),
"rating": r["value"] if r else None, "rating_count": r["count"] if r else None}
if kind == "similar":
picked = rec.recommend(with_rating(dict(target)), [with_rating(c) for c in similar],
[with_rating(c) for c in rated], limit)
else:
sentiment = review_sentiment_counts(conn, [c["product_id"] for c in rated])
picked = rec.better_rated(
with_rating(dict(target)),
[{**with_rating(c), "sentiment": sentiment.get(c["product_id"])} for c in rated], limit)
cards = {r["product_id"]: r for r in conn.execute(
"SELECT product_id, brand, display_name, ram_gb, storage_gb, image_url, best_price, best_price_site "
"FROM elec.v_brand_catalog WHERE product_id = ANY(%s)", ([c["product_id"] for c in picked],)
)}
variants = other_variants(conn, product_id)
items = [
{**_clean(cards[c["product_id"]]), "rating": c["rating"], "rating_count": c["rating_count"],
"basis": c["basis"], "reason": c["reason"]}
for c in picked if c["product_id"] in cards
]
return {"product_id": product_id, "type": kind, "items": items,
"other_variants": [_clean(v) for v in variants]}
def _float(value: Any) -> Optional[float]:
return None if value is None else float(value)
@router.get("/products/{product_id}/price-history")
def price_history(product_id: int) -> List[dict]:
with connect() as conn:

View File

@@ -305,6 +305,96 @@ def product_rating_and_reviews(conn, product_id: int) -> Dict[str, Any]:
return {"sources": [dict(s) for s in sources], "reviews": [dict(r) for r in reviews]}
def rating_sources_for(conn, product_ids: List[int]) -> Dict[int, List[dict]]:
"""product_rating_and_reviews' per-platform ratings for many products at once."""
out: Dict[int, List[dict]] = {pid: [] for pid in product_ids}
rows = conn.execute(
"SELECT a.product_id, a.site, a.source_url, l.rating, l.review_count, l.rating_breakdown "
"FROM elec.v_product_availability a JOIN elec.source_listing l ON l.id = a.listing_id "
"WHERE a.product_id = ANY(%s) AND l.rating > 0 ORDER BY l.review_count DESC NULLS LAST, a.site",
(list(product_ids),),
).fetchall()
for r in rows:
out[r["product_id"]].append(dict(r))
return out
# p is a variant of t when the two differ only in RAM/storage: same brand and
# model and the same processor. One laptop model line ("HP 15") spans many
# CPUs, so laptops count as variants only when both state the same processor;
# phones state none, so for them the model alone decides. Written so it is
# never NULL: NOT NULL would drop the product from recommendations too.
_SAME_MODEL = """(p.brand_id = t.brand_id AND p.model_norm = t.model_norm
AND p.processor IS NOT DISTINCT FROM t.processor
AND (p.processor IS NOT NULL
OR t.category_id IS DISTINCT FROM (SELECT id FROM elec.category WHERE slug = 'laptops')))"""
# Products that may be recommended for target t: verified, same category, not
# t itself or another RAM/storage variant of the same model, with an in-stock
# (or unknown-stock) best price within +/- %(band)s of t's own (no limit when t
# has no price).
_RECOMMENDABLE = """
FROM elec.product t
JOIN elec.product p ON p.category_id = t.category_id AND p.id <> t.id
AND p.verification_status = 'verified'
AND NOT """ + _SAME_MODEL + """
JOIN elec.v_best_price bp ON bp.product_id = p.id AND bp.in_stock IS DISTINCT FROM FALSE
LEFT JOIN elec.v_best_price tp ON tp.product_id = t.id
WHERE t.id = %(pid)s
AND (tp.price IS NULL OR bp.price BETWEEN tp.price * (1 - %(band)s) AND tp.price * (1 + %(band)s))
"""
_TN_ONLY = (" AND EXISTS (SELECT 1 FROM elec.v_product_availability a"
" WHERE a.product_id = p.id AND a.site_region = 'TN')")
def similar_products(conn, product_id: int, *, band: float, limit: int = 30, tn_only: bool = False) -> List[dict]:
"""Recommendable products closest to this one by embedding (cosine), with
their best price. Empty when the product has no embedding yet.
An exact scan, not the HNSW index: the category/stock filters would make an
approximate index search drop matches, and a category is small enough."""
sql = ("SELECT p.id AS product_id, 1 - (p.embedding <=> t.embedding) AS similarity, bp.price AS best_price"
+ _RECOMMENDABLE + " AND t.embedding IS NOT NULL AND p.embedding IS NOT NULL"
+ (_TN_ONLY if tn_only else "")
+ " ORDER BY p.embedding <=> t.embedding LIMIT %(limit)s")
return [dict(r) for r in conn.execute(sql, {"pid": product_id, "band": band, "limit": limit})]
def rated_products(conn, product_id: int, *, band: float, tn_only: bool = False) -> List[dict]:
"""Recommendable products that at least one platform has rated."""
sql = ("SELECT p.id AS product_id, bp.price AS best_price" + _RECOMMENDABLE
+ " AND EXISTS (SELECT 1 FROM elec.v_product_availability a JOIN elec.source_listing l"
" ON l.id = a.listing_id WHERE a.product_id = p.id AND l.rating > 0)"
+ (_TN_ONLY if tn_only else ""))
return [dict(r) for r in conn.execute(sql, {"pid": product_id, "band": band})]
def review_sentiment_counts(conn, product_ids: List[int]) -> Dict[int, Dict[str, int]]:
"""How many stored reviews of each product are positive / neutral / negative."""
out: Dict[int, Dict[str, int]] = {pid: {} for pid in product_ids}
rows = conn.execute(
"SELECT a.product_id, r.sentiment, count(*)::int AS n "
"FROM elec.v_product_availability a JOIN elec.listing_review r ON r.listing_id = a.listing_id "
"WHERE a.product_id = ANY(%s) AND r.sentiment IS NOT NULL GROUP BY 1, 2",
(list(product_ids),),
).fetchall()
for r in rows:
out[r["product_id"]][r["sentiment"]] = r["n"]
return out
def other_variants(conn, product_id: int) -> List[dict]:
"""The same model's other verified RAM/storage variants (see _SAME_MODEL)."""
return [dict(r) for r in conn.execute(
"SELECT v.product_id, v.display_name, v.ram_gb, v.storage_gb, v.best_price "
"FROM elec.product t JOIN elec.product p ON p.category_id = t.category_id AND p.id <> t.id "
"AND " + _SAME_MODEL + " "
"JOIN elec.v_brand_catalog v ON v.product_id = p.id "
"WHERE t.id = %s ORDER BY v.ram_gb NULLS LAST, v.storage_gb NULLS LAST, v.display_name",
(product_id,),
)]
def listings_for_review_backfill(category: Optional[str] = None) -> List[dict]:
"""Page-read listings of verified products, for re-reading ratings/reviews."""
sql = (

View File

@@ -0,0 +1,144 @@
"""Which other products to suggest under a product's ratings and reviews
(docs/RECOMMENDATIONS.md): Phase 1 "Similar products" and Phase 2 "Better
rated alternatives".
This module only scores and orders candidates; the database queries that find
them (repository.similar_products / rated_products) and the endpoint live
elsewhere. Nothing here invents a number: a product with no published rating
is scored at the pool's average, and one with no price is never a candidate.
Hard price limit: every candidate's best price is within SIMILAR_PRICE_BAND
(similar) or BETTER_PRICE_BAND (better rated) of the product's own, applied in
the database query. A product with no price of its own gets no limit.
Similar products:
score = 0.60 x similarity + 0.25 x rating quality + 0.15 x price closeness
similarity cosine similarity of the two product embeddings (0..1)
rating quality Bayesian average / 5, so 5.0 from 3 ratings does not beat
4.4 from 2,000: each product's rating is pulled towards the
pool average as if PRIOR_WEIGHT extra ratings at that average
had been given
price closeness 1 at the same price, falling to 0 at twice (or zero) the price
When fewer than MIN_MATCHES similar products qualify, the list is filled with
the best-rated products of the same category (in the same price band).
Better rated alternatives: products rated higher than this one by at least
MIN_REVIEWS_BETTER people, best Bayesian rating first. When two round to the
same Bayesian rating, the one whose stored reviews are more positive (share
of positive minus share of negative) goes first.
"""
from __future__ import annotations
from typing import Any, Dict, List, Optional
WEIGHT_SIMILARITY, WEIGHT_RATING, WEIGHT_PRICE = 0.60, 0.25, 0.15
PRIOR_WEIGHT = 50
DEFAULT_PRIOR_MEAN = 4.0 # used only when no product in the pool has a rating
MAX_ITEMS = 6
MIN_MATCHES = 3
SIMILAR_PRICE_BAND = 0.30 # +/-30% of the product's best price
BETTER_PRICE_BAND = 0.20 # +/-20%
MIN_REVIEWS_BETTER = 5
def prior_mean(candidates: List[Dict[str, Any]]) -> float:
"""Average published rating across the pool."""
rated = [c["rating"] for c in candidates if c.get("rating") is not None]
return sum(rated) / len(rated) if rated else DEFAULT_PRIOR_MEAN
def bayesian_rating(rating: Optional[float], count: Optional[int], mean: float) -> float:
if rating is None:
return mean
n = max(int(count or 0), 1) # a rating with no stated count weighs as one
return (PRIOR_WEIGHT * mean + n * float(rating)) / (PRIOR_WEIGHT + n)
def price_closeness(price: Optional[float], target_price: Optional[float]) -> float:
if price is None or not target_price:
return 0.5 # unknown: neither helps nor hurts
return max(0.0, 1.0 - abs(float(price) - float(target_price)) / float(target_price))
def score(candidate: Dict[str, Any], target_price: Optional[float], mean: float) -> float:
quality = bayesian_rating(candidate.get("rating"), candidate.get("rating_count"), mean) / 5
return (WEIGHT_SIMILARITY * float(candidate.get("similarity") or 0.0)
+ WEIGHT_RATING * quality
+ WEIGHT_PRICE * price_closeness(candidate.get("best_price"), target_price))
def reason(candidate: Dict[str, Any], target_rating: Optional[float], basis: str) -> str:
label = "Similar specs" if basis == "similar" else "Top rated in this category"
rating = candidate.get("rating")
if rating is None:
return label
if target_rating is not None and rating > target_rating:
return f"{label} · {rating:.1f}★ vs {target_rating:.1f}★"
count = candidate.get("rating_count")
return f"{label} · {rating:.1f}★" + (f" ({count:,} rating{'' if count == 1 else 's'})" if count else "")
def recommend(
target: Dict[str, Any],
similar: List[Dict[str, Any]],
top_rated: List[Dict[str, Any]],
limit: int = MAX_ITEMS,
) -> List[Dict[str, Any]]:
"""Order `similar` by score; when fewer than MIN_MATCHES are found, fill up
to `limit` from `top_rated` (best Bayesian rating first).
Each candidate is a dict with product_id, best_price, rating, rating_count
and (for `similar`) similarity. `target` has best_price and rating.
Returns the chosen candidates with `reason` and `basis` added."""
mean = prior_mean(similar + top_rated)
target_price, target_rating = target.get("best_price"), target.get("rating")
ranked = sorted(similar, key=lambda c: score(c, target_price, mean), reverse=True)[:limit]
picked = [{**c, "basis": "similar"} for c in ranked]
if len(picked) < MIN_MATCHES:
seen = {c["product_id"] for c in picked}
fill = sorted((c for c in top_rated if c["product_id"] not in seen and c.get("rating") is not None),
key=lambda c: bayesian_rating(c["rating"], c.get("rating_count"), mean), reverse=True)
picked += [{**c, "basis": "top_rated"} for c in fill[: limit - len(picked)]]
for c in picked:
c["reason"] = reason(c, target_rating, c["basis"])
return picked
def sentiment_balance(sentiment: Optional[Dict[str, int]]) -> float:
"""Share of positive minus share of negative stored reviews (-1..1); 0 with none."""
if not sentiment:
return 0.0
total = sum(sentiment.values())
if not total:
return 0.0
return (sentiment.get("positive", 0) - sentiment.get("negative", 0)) / total
def better_rated(
target: Dict[str, Any],
rated: List[Dict[str, Any]],
limit: int = MAX_ITEMS,
) -> List[Dict[str, Any]]:
"""The products in `rated` rated higher than `target` (any rated product
when the target has no rating), each rated by at least MIN_REVIEWS_BETTER
people. Each candidate has rating, rating_count and optionally `sentiment`
({"positive": n, "neutral": n, "negative": n})."""
target_rating = target.get("rating")
mean = prior_mean(rated + [target])
keep = [c for c in rated
if c.get("rating") is not None and (c.get("rating_count") or 0) >= MIN_REVIEWS_BETTER
and (target_rating is None or c["rating"] > target_rating)]
keep.sort(key=lambda c: (round(bayesian_rating(c["rating"], c["rating_count"], mean), 1),
sentiment_balance(c.get("sentiment"))), reverse=True)
picked = []
for c in keep[:limit]:
vs = f" vs {target_rating:.1f}★" if target_rating is not None else ""
picked.append({**c, "basis": "better_rated",
"reason": f"{c['rating']:.1f}★{vs} · {c['rating_count']:,} ratings"})
return picked

View File

@@ -27,13 +27,15 @@ mcp = FastMCP(
"Every product is confirmed by real listings on at least two retail platforms; "
"prices, ratings and reviews come with the page they were read from. Prices are "
"rupee strings. Use search_products to find products, then get_product for "
"per-platform offers, specs, images, rating and reviews."
"per-platform offers, specs, images, rating and reviews, and recommend_products "
"for similar alternatives."
),
)
_SEARCH_FIELDS = (
"product_id", "brand", "category", "display_name", "ram_gb", "storage_gb",
"best_price", "best_price_site", "platform_count", "sold_by_tn_retailer", "image_url",
"rating", "rating_count",
)
@@ -71,7 +73,8 @@ async def search_products(
limit: Maximum products to return (1-100).
Returns the total match count and, per product: id, name, variant, best price (rupee
string) and the platform offering it, number of platforms, and an image URL (or null).
string) and the platform offering it, number of platforms, an image URL (or null), and the
overall rating and rating count (null when no platform publishes a rating).
"""
limit = max(1, min(int(limit), 100))
result = await _run(
@@ -115,6 +118,36 @@ async def get_product(product_id: int) -> Dict[str, Any]:
}
@mcp.tool
async def recommend_products(product_id: int, kind: str = "similar", limit: int = 6) -> Dict[str, Any]:
"""Alternatives to suggest for one product (by its product_id).
Args:
product_id: The product to find alternatives for.
kind: "similar" - closest specs, ranked by spec similarity, rating (weighted by how
many people rated it) and price closeness; when few exist, the best-rated in the
category fill the list. "better_rated" - products rated higher than this one by
at least 5 people.
limit: Maximum products to return (1-12).
Always same category, in stock, within a similar price (+/-30% for similar, +/-20% for
better_rated), with other variants of the same model left out. Each item has a short
reason (e.g. "Similar specs · 4.5★ vs 4.1★"). The same model's other RAM/storage
variants are listed separately under other_variants.
"""
if kind not in ("similar", "better_rated"):
raise ToolError('kind must be "similar" or "better_rated"')
d = await _run(elec.recommendations, int(product_id), kind, max(1, min(int(limit), 12)), False)
return {
"items": [
{k: i.get(k) for k in ("product_id", "brand", "display_name", "ram_gb", "storage_gb",
"best_price", "best_price_site", "rating", "rating_count", "reason")}
for i in d["items"]
],
"other_variants": d["other_variants"],
}
@mcp.tool
async def price_history(product_id: int) -> List[Dict[str, Any]]:
"""Every price observed for a product, per platform, oldest first (rupee strings, ISO times)."""