backend updates on recommendation system
This commit is contained in:
@@ -7,12 +7,20 @@ Money is returned as a decimal string, never a float.
|
||||
from __future__ import annotations
|
||||
|
||||
from decimal import Decimal
|
||||
from typing import Any, Dict, List, Optional
|
||||
from typing import Any, Dict, List, Literal, Optional
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Query
|
||||
|
||||
from app.electronics.db.connection import connect
|
||||
from app.electronics.db.repository import product_rating_and_reviews
|
||||
from app.electronics import recommend as rec
|
||||
from app.electronics.db.repository import (
|
||||
other_variants,
|
||||
product_rating_and_reviews,
|
||||
rated_products,
|
||||
rating_sources_for,
|
||||
review_sentiment_counts,
|
||||
similar_products,
|
||||
)
|
||||
from app.electronics.reviews import select_reviews
|
||||
|
||||
router = APIRouter(prefix="/elec", tags=["electronics"])
|
||||
@@ -106,7 +114,13 @@ def products(
|
||||
f"LIMIT %(limit)s OFFSET %(offset)s",
|
||||
{**params, "limit": limit, "offset": offset},
|
||||
).fetchall()
|
||||
return {"total": total, "products": [_clean(r) for r in rows]}
|
||||
sources = rating_sources_for(conn, [r["product_id"] for r in rows])
|
||||
out = []
|
||||
for r in rows:
|
||||
rating = _overall_rating(sources[r["product_id"]])
|
||||
out.append({**_clean(r), "rating": rating["value"] if rating else None,
|
||||
"rating_count": rating["count"] if rating else None})
|
||||
return {"total": total, "products": out}
|
||||
|
||||
|
||||
@router.get("/products/{product_id}")
|
||||
@@ -185,6 +199,63 @@ def _breakdown(sources: List[dict]) -> Optional[List[dict]]:
|
||||
for k in ("5", "4", "3", "2", "1")]
|
||||
|
||||
|
||||
@router.get("/products/{product_id}/recommendations")
|
||||
def recommendations(
|
||||
product_id: int,
|
||||
kind: Literal["similar", "better_rated"] = Query("similar", alias="type"),
|
||||
limit: int = Query(rec.MAX_ITEMS, ge=1, le=12),
|
||||
tn_only: bool = False,
|
||||
) -> dict:
|
||||
"""Products to suggest under this one's ratings and reviews (see
|
||||
app/electronics/recommend.py), plus the same model's other variants.
|
||||
type=similar: closest specs; type=better_rated: rated higher, similar price."""
|
||||
with connect() as conn:
|
||||
target = conn.execute(
|
||||
"SELECT product_id, best_price FROM elec.v_brand_catalog WHERE product_id = %s", (product_id,)
|
||||
).fetchone()
|
||||
if not target:
|
||||
raise HTTPException(status_code=404, detail="Product not found or not verified")
|
||||
if kind == "similar":
|
||||
similar = similar_products(conn, product_id, band=rec.SIMILAR_PRICE_BAND, tn_only=tn_only)
|
||||
rated = (rated_products(conn, product_id, band=rec.SIMILAR_PRICE_BAND, tn_only=tn_only)
|
||||
if len(similar) < rec.MIN_MATCHES else [])
|
||||
else:
|
||||
similar = []
|
||||
rated = rated_products(conn, product_id, band=rec.BETTER_PRICE_BAND, tn_only=tn_only)
|
||||
ids = {product_id} | {c["product_id"] for c in similar + rated}
|
||||
ratings = {pid: _overall_rating(src) for pid, src in rating_sources_for(conn, list(ids)).items()}
|
||||
|
||||
def with_rating(c: dict) -> dict:
|
||||
r = ratings.get(c["product_id"])
|
||||
return {**c, "best_price": _float(c.get("best_price")),
|
||||
"rating": r["value"] if r else None, "rating_count": r["count"] if r else None}
|
||||
|
||||
if kind == "similar":
|
||||
picked = rec.recommend(with_rating(dict(target)), [with_rating(c) for c in similar],
|
||||
[with_rating(c) for c in rated], limit)
|
||||
else:
|
||||
sentiment = review_sentiment_counts(conn, [c["product_id"] for c in rated])
|
||||
picked = rec.better_rated(
|
||||
with_rating(dict(target)),
|
||||
[{**with_rating(c), "sentiment": sentiment.get(c["product_id"])} for c in rated], limit)
|
||||
cards = {r["product_id"]: r for r in conn.execute(
|
||||
"SELECT product_id, brand, display_name, ram_gb, storage_gb, image_url, best_price, best_price_site "
|
||||
"FROM elec.v_brand_catalog WHERE product_id = ANY(%s)", ([c["product_id"] for c in picked],)
|
||||
)}
|
||||
variants = other_variants(conn, product_id)
|
||||
items = [
|
||||
{**_clean(cards[c["product_id"]]), "rating": c["rating"], "rating_count": c["rating_count"],
|
||||
"basis": c["basis"], "reason": c["reason"]}
|
||||
for c in picked if c["product_id"] in cards
|
||||
]
|
||||
return {"product_id": product_id, "type": kind, "items": items,
|
||||
"other_variants": [_clean(v) for v in variants]}
|
||||
|
||||
|
||||
def _float(value: Any) -> Optional[float]:
|
||||
return None if value is None else float(value)
|
||||
|
||||
|
||||
@router.get("/products/{product_id}/price-history")
|
||||
def price_history(product_id: int) -> List[dict]:
|
||||
with connect() as conn:
|
||||
|
||||
@@ -305,6 +305,96 @@ def product_rating_and_reviews(conn, product_id: int) -> Dict[str, Any]:
|
||||
return {"sources": [dict(s) for s in sources], "reviews": [dict(r) for r in reviews]}
|
||||
|
||||
|
||||
def rating_sources_for(conn, product_ids: List[int]) -> Dict[int, List[dict]]:
|
||||
"""product_rating_and_reviews' per-platform ratings for many products at once."""
|
||||
out: Dict[int, List[dict]] = {pid: [] for pid in product_ids}
|
||||
rows = conn.execute(
|
||||
"SELECT a.product_id, a.site, a.source_url, l.rating, l.review_count, l.rating_breakdown "
|
||||
"FROM elec.v_product_availability a JOIN elec.source_listing l ON l.id = a.listing_id "
|
||||
"WHERE a.product_id = ANY(%s) AND l.rating > 0 ORDER BY l.review_count DESC NULLS LAST, a.site",
|
||||
(list(product_ids),),
|
||||
).fetchall()
|
||||
for r in rows:
|
||||
out[r["product_id"]].append(dict(r))
|
||||
return out
|
||||
|
||||
|
||||
# p is a variant of t when the two differ only in RAM/storage: same brand and
|
||||
# model and the same processor. One laptop model line ("HP 15") spans many
|
||||
# CPUs, so laptops count as variants only when both state the same processor;
|
||||
# phones state none, so for them the model alone decides. Written so it is
|
||||
# never NULL: NOT NULL would drop the product from recommendations too.
|
||||
_SAME_MODEL = """(p.brand_id = t.brand_id AND p.model_norm = t.model_norm
|
||||
AND p.processor IS NOT DISTINCT FROM t.processor
|
||||
AND (p.processor IS NOT NULL
|
||||
OR t.category_id IS DISTINCT FROM (SELECT id FROM elec.category WHERE slug = 'laptops')))"""
|
||||
|
||||
# Products that may be recommended for target t: verified, same category, not
|
||||
# t itself or another RAM/storage variant of the same model, with an in-stock
|
||||
# (or unknown-stock) best price within +/- %(band)s of t's own (no limit when t
|
||||
# has no price).
|
||||
_RECOMMENDABLE = """
|
||||
FROM elec.product t
|
||||
JOIN elec.product p ON p.category_id = t.category_id AND p.id <> t.id
|
||||
AND p.verification_status = 'verified'
|
||||
AND NOT """ + _SAME_MODEL + """
|
||||
JOIN elec.v_best_price bp ON bp.product_id = p.id AND bp.in_stock IS DISTINCT FROM FALSE
|
||||
LEFT JOIN elec.v_best_price tp ON tp.product_id = t.id
|
||||
WHERE t.id = %(pid)s
|
||||
AND (tp.price IS NULL OR bp.price BETWEEN tp.price * (1 - %(band)s) AND tp.price * (1 + %(band)s))
|
||||
"""
|
||||
_TN_ONLY = (" AND EXISTS (SELECT 1 FROM elec.v_product_availability a"
|
||||
" WHERE a.product_id = p.id AND a.site_region = 'TN')")
|
||||
|
||||
|
||||
def similar_products(conn, product_id: int, *, band: float, limit: int = 30, tn_only: bool = False) -> List[dict]:
|
||||
"""Recommendable products closest to this one by embedding (cosine), with
|
||||
their best price. Empty when the product has no embedding yet.
|
||||
|
||||
An exact scan, not the HNSW index: the category/stock filters would make an
|
||||
approximate index search drop matches, and a category is small enough."""
|
||||
sql = ("SELECT p.id AS product_id, 1 - (p.embedding <=> t.embedding) AS similarity, bp.price AS best_price"
|
||||
+ _RECOMMENDABLE + " AND t.embedding IS NOT NULL AND p.embedding IS NOT NULL"
|
||||
+ (_TN_ONLY if tn_only else "")
|
||||
+ " ORDER BY p.embedding <=> t.embedding LIMIT %(limit)s")
|
||||
return [dict(r) for r in conn.execute(sql, {"pid": product_id, "band": band, "limit": limit})]
|
||||
|
||||
|
||||
def rated_products(conn, product_id: int, *, band: float, tn_only: bool = False) -> List[dict]:
|
||||
"""Recommendable products that at least one platform has rated."""
|
||||
sql = ("SELECT p.id AS product_id, bp.price AS best_price" + _RECOMMENDABLE
|
||||
+ " AND EXISTS (SELECT 1 FROM elec.v_product_availability a JOIN elec.source_listing l"
|
||||
" ON l.id = a.listing_id WHERE a.product_id = p.id AND l.rating > 0)"
|
||||
+ (_TN_ONLY if tn_only else ""))
|
||||
return [dict(r) for r in conn.execute(sql, {"pid": product_id, "band": band})]
|
||||
|
||||
|
||||
def review_sentiment_counts(conn, product_ids: List[int]) -> Dict[int, Dict[str, int]]:
|
||||
"""How many stored reviews of each product are positive / neutral / negative."""
|
||||
out: Dict[int, Dict[str, int]] = {pid: {} for pid in product_ids}
|
||||
rows = conn.execute(
|
||||
"SELECT a.product_id, r.sentiment, count(*)::int AS n "
|
||||
"FROM elec.v_product_availability a JOIN elec.listing_review r ON r.listing_id = a.listing_id "
|
||||
"WHERE a.product_id = ANY(%s) AND r.sentiment IS NOT NULL GROUP BY 1, 2",
|
||||
(list(product_ids),),
|
||||
).fetchall()
|
||||
for r in rows:
|
||||
out[r["product_id"]][r["sentiment"]] = r["n"]
|
||||
return out
|
||||
|
||||
|
||||
def other_variants(conn, product_id: int) -> List[dict]:
|
||||
"""The same model's other verified RAM/storage variants (see _SAME_MODEL)."""
|
||||
return [dict(r) for r in conn.execute(
|
||||
"SELECT v.product_id, v.display_name, v.ram_gb, v.storage_gb, v.best_price "
|
||||
"FROM elec.product t JOIN elec.product p ON p.category_id = t.category_id AND p.id <> t.id "
|
||||
"AND " + _SAME_MODEL + " "
|
||||
"JOIN elec.v_brand_catalog v ON v.product_id = p.id "
|
||||
"WHERE t.id = %s ORDER BY v.ram_gb NULLS LAST, v.storage_gb NULLS LAST, v.display_name",
|
||||
(product_id,),
|
||||
)]
|
||||
|
||||
|
||||
def listings_for_review_backfill(category: Optional[str] = None) -> List[dict]:
|
||||
"""Page-read listings of verified products, for re-reading ratings/reviews."""
|
||||
sql = (
|
||||
|
||||
144
backend/app/electronics/recommend.py
Normal file
144
backend/app/electronics/recommend.py
Normal file
@@ -0,0 +1,144 @@
|
||||
"""Which other products to suggest under a product's ratings and reviews
|
||||
(docs/RECOMMENDATIONS.md): Phase 1 "Similar products" and Phase 2 "Better
|
||||
rated alternatives".
|
||||
|
||||
This module only scores and orders candidates; the database queries that find
|
||||
them (repository.similar_products / rated_products) and the endpoint live
|
||||
elsewhere. Nothing here invents a number: a product with no published rating
|
||||
is scored at the pool's average, and one with no price is never a candidate.
|
||||
|
||||
Hard price limit: every candidate's best price is within SIMILAR_PRICE_BAND
|
||||
(similar) or BETTER_PRICE_BAND (better rated) of the product's own, applied in
|
||||
the database query. A product with no price of its own gets no limit.
|
||||
|
||||
Similar products:
|
||||
|
||||
score = 0.60 x similarity + 0.25 x rating quality + 0.15 x price closeness
|
||||
|
||||
similarity cosine similarity of the two product embeddings (0..1)
|
||||
rating quality Bayesian average / 5, so 5.0 from 3 ratings does not beat
|
||||
4.4 from 2,000: each product's rating is pulled towards the
|
||||
pool average as if PRIOR_WEIGHT extra ratings at that average
|
||||
had been given
|
||||
price closeness 1 at the same price, falling to 0 at twice (or zero) the price
|
||||
|
||||
When fewer than MIN_MATCHES similar products qualify, the list is filled with
|
||||
the best-rated products of the same category (in the same price band).
|
||||
|
||||
Better rated alternatives: products rated higher than this one by at least
|
||||
MIN_REVIEWS_BETTER people, best Bayesian rating first. When two round to the
|
||||
same Bayesian rating, the one whose stored reviews are more positive (share
|
||||
of positive minus share of negative) goes first.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
WEIGHT_SIMILARITY, WEIGHT_RATING, WEIGHT_PRICE = 0.60, 0.25, 0.15
|
||||
PRIOR_WEIGHT = 50
|
||||
DEFAULT_PRIOR_MEAN = 4.0 # used only when no product in the pool has a rating
|
||||
MAX_ITEMS = 6
|
||||
MIN_MATCHES = 3
|
||||
SIMILAR_PRICE_BAND = 0.30 # +/-30% of the product's best price
|
||||
BETTER_PRICE_BAND = 0.20 # +/-20%
|
||||
MIN_REVIEWS_BETTER = 5
|
||||
|
||||
|
||||
def prior_mean(candidates: List[Dict[str, Any]]) -> float:
|
||||
"""Average published rating across the pool."""
|
||||
rated = [c["rating"] for c in candidates if c.get("rating") is not None]
|
||||
return sum(rated) / len(rated) if rated else DEFAULT_PRIOR_MEAN
|
||||
|
||||
|
||||
def bayesian_rating(rating: Optional[float], count: Optional[int], mean: float) -> float:
|
||||
if rating is None:
|
||||
return mean
|
||||
n = max(int(count or 0), 1) # a rating with no stated count weighs as one
|
||||
return (PRIOR_WEIGHT * mean + n * float(rating)) / (PRIOR_WEIGHT + n)
|
||||
|
||||
|
||||
def price_closeness(price: Optional[float], target_price: Optional[float]) -> float:
|
||||
if price is None or not target_price:
|
||||
return 0.5 # unknown: neither helps nor hurts
|
||||
return max(0.0, 1.0 - abs(float(price) - float(target_price)) / float(target_price))
|
||||
|
||||
|
||||
def score(candidate: Dict[str, Any], target_price: Optional[float], mean: float) -> float:
|
||||
quality = bayesian_rating(candidate.get("rating"), candidate.get("rating_count"), mean) / 5
|
||||
return (WEIGHT_SIMILARITY * float(candidate.get("similarity") or 0.0)
|
||||
+ WEIGHT_RATING * quality
|
||||
+ WEIGHT_PRICE * price_closeness(candidate.get("best_price"), target_price))
|
||||
|
||||
|
||||
def reason(candidate: Dict[str, Any], target_rating: Optional[float], basis: str) -> str:
|
||||
label = "Similar specs" if basis == "similar" else "Top rated in this category"
|
||||
rating = candidate.get("rating")
|
||||
if rating is None:
|
||||
return label
|
||||
if target_rating is not None and rating > target_rating:
|
||||
return f"{label} · {rating:.1f}★ vs {target_rating:.1f}★"
|
||||
count = candidate.get("rating_count")
|
||||
return f"{label} · {rating:.1f}★" + (f" ({count:,} rating{'' if count == 1 else 's'})" if count else "")
|
||||
|
||||
|
||||
def recommend(
|
||||
target: Dict[str, Any],
|
||||
similar: List[Dict[str, Any]],
|
||||
top_rated: List[Dict[str, Any]],
|
||||
limit: int = MAX_ITEMS,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Order `similar` by score; when fewer than MIN_MATCHES are found, fill up
|
||||
to `limit` from `top_rated` (best Bayesian rating first).
|
||||
|
||||
Each candidate is a dict with product_id, best_price, rating, rating_count
|
||||
and (for `similar`) similarity. `target` has best_price and rating.
|
||||
Returns the chosen candidates with `reason` and `basis` added."""
|
||||
mean = prior_mean(similar + top_rated)
|
||||
target_price, target_rating = target.get("best_price"), target.get("rating")
|
||||
|
||||
ranked = sorted(similar, key=lambda c: score(c, target_price, mean), reverse=True)[:limit]
|
||||
picked = [{**c, "basis": "similar"} for c in ranked]
|
||||
|
||||
if len(picked) < MIN_MATCHES:
|
||||
seen = {c["product_id"] for c in picked}
|
||||
fill = sorted((c for c in top_rated if c["product_id"] not in seen and c.get("rating") is not None),
|
||||
key=lambda c: bayesian_rating(c["rating"], c.get("rating_count"), mean), reverse=True)
|
||||
picked += [{**c, "basis": "top_rated"} for c in fill[: limit - len(picked)]]
|
||||
|
||||
for c in picked:
|
||||
c["reason"] = reason(c, target_rating, c["basis"])
|
||||
return picked
|
||||
|
||||
|
||||
def sentiment_balance(sentiment: Optional[Dict[str, int]]) -> float:
|
||||
"""Share of positive minus share of negative stored reviews (-1..1); 0 with none."""
|
||||
if not sentiment:
|
||||
return 0.0
|
||||
total = sum(sentiment.values())
|
||||
if not total:
|
||||
return 0.0
|
||||
return (sentiment.get("positive", 0) - sentiment.get("negative", 0)) / total
|
||||
|
||||
|
||||
def better_rated(
|
||||
target: Dict[str, Any],
|
||||
rated: List[Dict[str, Any]],
|
||||
limit: int = MAX_ITEMS,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""The products in `rated` rated higher than `target` (any rated product
|
||||
when the target has no rating), each rated by at least MIN_REVIEWS_BETTER
|
||||
people. Each candidate has rating, rating_count and optionally `sentiment`
|
||||
({"positive": n, "neutral": n, "negative": n})."""
|
||||
target_rating = target.get("rating")
|
||||
mean = prior_mean(rated + [target])
|
||||
keep = [c for c in rated
|
||||
if c.get("rating") is not None and (c.get("rating_count") or 0) >= MIN_REVIEWS_BETTER
|
||||
and (target_rating is None or c["rating"] > target_rating)]
|
||||
keep.sort(key=lambda c: (round(bayesian_rating(c["rating"], c["rating_count"], mean), 1),
|
||||
sentiment_balance(c.get("sentiment"))), reverse=True)
|
||||
picked = []
|
||||
for c in keep[:limit]:
|
||||
vs = f" vs {target_rating:.1f}★" if target_rating is not None else ""
|
||||
picked.append({**c, "basis": "better_rated",
|
||||
"reason": f"{c['rating']:.1f}★{vs} · {c['rating_count']:,} ratings"})
|
||||
return picked
|
||||
@@ -27,13 +27,15 @@ mcp = FastMCP(
|
||||
"Every product is confirmed by real listings on at least two retail platforms; "
|
||||
"prices, ratings and reviews come with the page they were read from. Prices are "
|
||||
"rupee strings. Use search_products to find products, then get_product for "
|
||||
"per-platform offers, specs, images, rating and reviews."
|
||||
"per-platform offers, specs, images, rating and reviews, and recommend_products "
|
||||
"for similar alternatives."
|
||||
),
|
||||
)
|
||||
|
||||
_SEARCH_FIELDS = (
|
||||
"product_id", "brand", "category", "display_name", "ram_gb", "storage_gb",
|
||||
"best_price", "best_price_site", "platform_count", "sold_by_tn_retailer", "image_url",
|
||||
"rating", "rating_count",
|
||||
)
|
||||
|
||||
|
||||
@@ -71,7 +73,8 @@ async def search_products(
|
||||
limit: Maximum products to return (1-100).
|
||||
|
||||
Returns the total match count and, per product: id, name, variant, best price (rupee
|
||||
string) and the platform offering it, number of platforms, and an image URL (or null).
|
||||
string) and the platform offering it, number of platforms, an image URL (or null), and the
|
||||
overall rating and rating count (null when no platform publishes a rating).
|
||||
"""
|
||||
limit = max(1, min(int(limit), 100))
|
||||
result = await _run(
|
||||
@@ -115,6 +118,36 @@ async def get_product(product_id: int) -> Dict[str, Any]:
|
||||
}
|
||||
|
||||
|
||||
@mcp.tool
|
||||
async def recommend_products(product_id: int, kind: str = "similar", limit: int = 6) -> Dict[str, Any]:
|
||||
"""Alternatives to suggest for one product (by its product_id).
|
||||
|
||||
Args:
|
||||
product_id: The product to find alternatives for.
|
||||
kind: "similar" - closest specs, ranked by spec similarity, rating (weighted by how
|
||||
many people rated it) and price closeness; when few exist, the best-rated in the
|
||||
category fill the list. "better_rated" - products rated higher than this one by
|
||||
at least 5 people.
|
||||
limit: Maximum products to return (1-12).
|
||||
|
||||
Always same category, in stock, within a similar price (+/-30% for similar, +/-20% for
|
||||
better_rated), with other variants of the same model left out. Each item has a short
|
||||
reason (e.g. "Similar specs · 4.5★ vs 4.1★"). The same model's other RAM/storage
|
||||
variants are listed separately under other_variants.
|
||||
"""
|
||||
if kind not in ("similar", "better_rated"):
|
||||
raise ToolError('kind must be "similar" or "better_rated"')
|
||||
d = await _run(elec.recommendations, int(product_id), kind, max(1, min(int(limit), 12)), False)
|
||||
return {
|
||||
"items": [
|
||||
{k: i.get(k) for k in ("product_id", "brand", "display_name", "ram_gb", "storage_gb",
|
||||
"best_price", "best_price_site", "rating", "rating_count", "reason")}
|
||||
for i in d["items"]
|
||||
],
|
||||
"other_variants": d["other_variants"],
|
||||
}
|
||||
|
||||
|
||||
@mcp.tool
|
||||
async def price_history(product_id: int) -> List[Dict[str, Any]]:
|
||||
"""Every price observed for a product, per platform, oldest first (rupee strings, ISO times)."""
|
||||
|
||||
Reference in New Issue
Block a user