backend updates on recommendation system
This commit is contained in:
@@ -7,12 +7,20 @@ Money is returned as a decimal string, never a float.
|
||||
from __future__ import annotations
|
||||
|
||||
from decimal import Decimal
|
||||
from typing import Any, Dict, List, Optional
|
||||
from typing import Any, Dict, List, Literal, Optional
|
||||
|
||||
from fastapi import APIRouter, HTTPException, Query
|
||||
|
||||
from app.electronics.db.connection import connect
|
||||
from app.electronics.db.repository import product_rating_and_reviews
|
||||
from app.electronics import recommend as rec
|
||||
from app.electronics.db.repository import (
|
||||
other_variants,
|
||||
product_rating_and_reviews,
|
||||
rated_products,
|
||||
rating_sources_for,
|
||||
review_sentiment_counts,
|
||||
similar_products,
|
||||
)
|
||||
from app.electronics.reviews import select_reviews
|
||||
|
||||
router = APIRouter(prefix="/elec", tags=["electronics"])
|
||||
@@ -106,7 +114,13 @@ def products(
|
||||
f"LIMIT %(limit)s OFFSET %(offset)s",
|
||||
{**params, "limit": limit, "offset": offset},
|
||||
).fetchall()
|
||||
return {"total": total, "products": [_clean(r) for r in rows]}
|
||||
sources = rating_sources_for(conn, [r["product_id"] for r in rows])
|
||||
out = []
|
||||
for r in rows:
|
||||
rating = _overall_rating(sources[r["product_id"]])
|
||||
out.append({**_clean(r), "rating": rating["value"] if rating else None,
|
||||
"rating_count": rating["count"] if rating else None})
|
||||
return {"total": total, "products": out}
|
||||
|
||||
|
||||
@router.get("/products/{product_id}")
|
||||
@@ -185,6 +199,63 @@ def _breakdown(sources: List[dict]) -> Optional[List[dict]]:
|
||||
for k in ("5", "4", "3", "2", "1")]
|
||||
|
||||
|
||||
@router.get("/products/{product_id}/recommendations")
|
||||
def recommendations(
|
||||
product_id: int,
|
||||
kind: Literal["similar", "better_rated"] = Query("similar", alias="type"),
|
||||
limit: int = Query(rec.MAX_ITEMS, ge=1, le=12),
|
||||
tn_only: bool = False,
|
||||
) -> dict:
|
||||
"""Products to suggest under this one's ratings and reviews (see
|
||||
app/electronics/recommend.py), plus the same model's other variants.
|
||||
type=similar: closest specs; type=better_rated: rated higher, similar price."""
|
||||
with connect() as conn:
|
||||
target = conn.execute(
|
||||
"SELECT product_id, best_price FROM elec.v_brand_catalog WHERE product_id = %s", (product_id,)
|
||||
).fetchone()
|
||||
if not target:
|
||||
raise HTTPException(status_code=404, detail="Product not found or not verified")
|
||||
if kind == "similar":
|
||||
similar = similar_products(conn, product_id, band=rec.SIMILAR_PRICE_BAND, tn_only=tn_only)
|
||||
rated = (rated_products(conn, product_id, band=rec.SIMILAR_PRICE_BAND, tn_only=tn_only)
|
||||
if len(similar) < rec.MIN_MATCHES else [])
|
||||
else:
|
||||
similar = []
|
||||
rated = rated_products(conn, product_id, band=rec.BETTER_PRICE_BAND, tn_only=tn_only)
|
||||
ids = {product_id} | {c["product_id"] for c in similar + rated}
|
||||
ratings = {pid: _overall_rating(src) for pid, src in rating_sources_for(conn, list(ids)).items()}
|
||||
|
||||
def with_rating(c: dict) -> dict:
|
||||
r = ratings.get(c["product_id"])
|
||||
return {**c, "best_price": _float(c.get("best_price")),
|
||||
"rating": r["value"] if r else None, "rating_count": r["count"] if r else None}
|
||||
|
||||
if kind == "similar":
|
||||
picked = rec.recommend(with_rating(dict(target)), [with_rating(c) for c in similar],
|
||||
[with_rating(c) for c in rated], limit)
|
||||
else:
|
||||
sentiment = review_sentiment_counts(conn, [c["product_id"] for c in rated])
|
||||
picked = rec.better_rated(
|
||||
with_rating(dict(target)),
|
||||
[{**with_rating(c), "sentiment": sentiment.get(c["product_id"])} for c in rated], limit)
|
||||
cards = {r["product_id"]: r for r in conn.execute(
|
||||
"SELECT product_id, brand, display_name, ram_gb, storage_gb, image_url, best_price, best_price_site "
|
||||
"FROM elec.v_brand_catalog WHERE product_id = ANY(%s)", ([c["product_id"] for c in picked],)
|
||||
)}
|
||||
variants = other_variants(conn, product_id)
|
||||
items = [
|
||||
{**_clean(cards[c["product_id"]]), "rating": c["rating"], "rating_count": c["rating_count"],
|
||||
"basis": c["basis"], "reason": c["reason"]}
|
||||
for c in picked if c["product_id"] in cards
|
||||
]
|
||||
return {"product_id": product_id, "type": kind, "items": items,
|
||||
"other_variants": [_clean(v) for v in variants]}
|
||||
|
||||
|
||||
def _float(value: Any) -> Optional[float]:
|
||||
return None if value is None else float(value)
|
||||
|
||||
|
||||
@router.get("/products/{product_id}/price-history")
|
||||
def price_history(product_id: int) -> List[dict]:
|
||||
with connect() as conn:
|
||||
|
||||
@@ -305,6 +305,96 @@ def product_rating_and_reviews(conn, product_id: int) -> Dict[str, Any]:
|
||||
return {"sources": [dict(s) for s in sources], "reviews": [dict(r) for r in reviews]}
|
||||
|
||||
|
||||
def rating_sources_for(conn, product_ids: List[int]) -> Dict[int, List[dict]]:
|
||||
"""product_rating_and_reviews' per-platform ratings for many products at once."""
|
||||
out: Dict[int, List[dict]] = {pid: [] for pid in product_ids}
|
||||
rows = conn.execute(
|
||||
"SELECT a.product_id, a.site, a.source_url, l.rating, l.review_count, l.rating_breakdown "
|
||||
"FROM elec.v_product_availability a JOIN elec.source_listing l ON l.id = a.listing_id "
|
||||
"WHERE a.product_id = ANY(%s) AND l.rating > 0 ORDER BY l.review_count DESC NULLS LAST, a.site",
|
||||
(list(product_ids),),
|
||||
).fetchall()
|
||||
for r in rows:
|
||||
out[r["product_id"]].append(dict(r))
|
||||
return out
|
||||
|
||||
|
||||
# p is a variant of t when the two differ only in RAM/storage: same brand and
|
||||
# model and the same processor. One laptop model line ("HP 15") spans many
|
||||
# CPUs, so laptops count as variants only when both state the same processor;
|
||||
# phones state none, so for them the model alone decides. Written so it is
|
||||
# never NULL: NOT NULL would drop the product from recommendations too.
|
||||
_SAME_MODEL = """(p.brand_id = t.brand_id AND p.model_norm = t.model_norm
|
||||
AND p.processor IS NOT DISTINCT FROM t.processor
|
||||
AND (p.processor IS NOT NULL
|
||||
OR t.category_id IS DISTINCT FROM (SELECT id FROM elec.category WHERE slug = 'laptops')))"""
|
||||
|
||||
# Products that may be recommended for target t: verified, same category, not
|
||||
# t itself or another RAM/storage variant of the same model, with an in-stock
|
||||
# (or unknown-stock) best price within +/- %(band)s of t's own (no limit when t
|
||||
# has no price).
|
||||
_RECOMMENDABLE = """
|
||||
FROM elec.product t
|
||||
JOIN elec.product p ON p.category_id = t.category_id AND p.id <> t.id
|
||||
AND p.verification_status = 'verified'
|
||||
AND NOT """ + _SAME_MODEL + """
|
||||
JOIN elec.v_best_price bp ON bp.product_id = p.id AND bp.in_stock IS DISTINCT FROM FALSE
|
||||
LEFT JOIN elec.v_best_price tp ON tp.product_id = t.id
|
||||
WHERE t.id = %(pid)s
|
||||
AND (tp.price IS NULL OR bp.price BETWEEN tp.price * (1 - %(band)s) AND tp.price * (1 + %(band)s))
|
||||
"""
|
||||
_TN_ONLY = (" AND EXISTS (SELECT 1 FROM elec.v_product_availability a"
|
||||
" WHERE a.product_id = p.id AND a.site_region = 'TN')")
|
||||
|
||||
|
||||
def similar_products(conn, product_id: int, *, band: float, limit: int = 30, tn_only: bool = False) -> List[dict]:
|
||||
"""Recommendable products closest to this one by embedding (cosine), with
|
||||
their best price. Empty when the product has no embedding yet.
|
||||
|
||||
An exact scan, not the HNSW index: the category/stock filters would make an
|
||||
approximate index search drop matches, and a category is small enough."""
|
||||
sql = ("SELECT p.id AS product_id, 1 - (p.embedding <=> t.embedding) AS similarity, bp.price AS best_price"
|
||||
+ _RECOMMENDABLE + " AND t.embedding IS NOT NULL AND p.embedding IS NOT NULL"
|
||||
+ (_TN_ONLY if tn_only else "")
|
||||
+ " ORDER BY p.embedding <=> t.embedding LIMIT %(limit)s")
|
||||
return [dict(r) for r in conn.execute(sql, {"pid": product_id, "band": band, "limit": limit})]
|
||||
|
||||
|
||||
def rated_products(conn, product_id: int, *, band: float, tn_only: bool = False) -> List[dict]:
|
||||
"""Recommendable products that at least one platform has rated."""
|
||||
sql = ("SELECT p.id AS product_id, bp.price AS best_price" + _RECOMMENDABLE
|
||||
+ " AND EXISTS (SELECT 1 FROM elec.v_product_availability a JOIN elec.source_listing l"
|
||||
" ON l.id = a.listing_id WHERE a.product_id = p.id AND l.rating > 0)"
|
||||
+ (_TN_ONLY if tn_only else ""))
|
||||
return [dict(r) for r in conn.execute(sql, {"pid": product_id, "band": band})]
|
||||
|
||||
|
||||
def review_sentiment_counts(conn, product_ids: List[int]) -> Dict[int, Dict[str, int]]:
|
||||
"""How many stored reviews of each product are positive / neutral / negative."""
|
||||
out: Dict[int, Dict[str, int]] = {pid: {} for pid in product_ids}
|
||||
rows = conn.execute(
|
||||
"SELECT a.product_id, r.sentiment, count(*)::int AS n "
|
||||
"FROM elec.v_product_availability a JOIN elec.listing_review r ON r.listing_id = a.listing_id "
|
||||
"WHERE a.product_id = ANY(%s) AND r.sentiment IS NOT NULL GROUP BY 1, 2",
|
||||
(list(product_ids),),
|
||||
).fetchall()
|
||||
for r in rows:
|
||||
out[r["product_id"]][r["sentiment"]] = r["n"]
|
||||
return out
|
||||
|
||||
|
||||
def other_variants(conn, product_id: int) -> List[dict]:
|
||||
"""The same model's other verified RAM/storage variants (see _SAME_MODEL)."""
|
||||
return [dict(r) for r in conn.execute(
|
||||
"SELECT v.product_id, v.display_name, v.ram_gb, v.storage_gb, v.best_price "
|
||||
"FROM elec.product t JOIN elec.product p ON p.category_id = t.category_id AND p.id <> t.id "
|
||||
"AND " + _SAME_MODEL + " "
|
||||
"JOIN elec.v_brand_catalog v ON v.product_id = p.id "
|
||||
"WHERE t.id = %s ORDER BY v.ram_gb NULLS LAST, v.storage_gb NULLS LAST, v.display_name",
|
||||
(product_id,),
|
||||
)]
|
||||
|
||||
|
||||
def listings_for_review_backfill(category: Optional[str] = None) -> List[dict]:
|
||||
"""Page-read listings of verified products, for re-reading ratings/reviews."""
|
||||
sql = (
|
||||
|
||||
144
backend/app/electronics/recommend.py
Normal file
144
backend/app/electronics/recommend.py
Normal file
@@ -0,0 +1,144 @@
|
||||
"""Which other products to suggest under a product's ratings and reviews
|
||||
(docs/RECOMMENDATIONS.md): Phase 1 "Similar products" and Phase 2 "Better
|
||||
rated alternatives".
|
||||
|
||||
This module only scores and orders candidates; the database queries that find
|
||||
them (repository.similar_products / rated_products) and the endpoint live
|
||||
elsewhere. Nothing here invents a number: a product with no published rating
|
||||
is scored at the pool's average, and one with no price is never a candidate.
|
||||
|
||||
Hard price limit: every candidate's best price is within SIMILAR_PRICE_BAND
|
||||
(similar) or BETTER_PRICE_BAND (better rated) of the product's own, applied in
|
||||
the database query. A product with no price of its own gets no limit.
|
||||
|
||||
Similar products:
|
||||
|
||||
score = 0.60 x similarity + 0.25 x rating quality + 0.15 x price closeness
|
||||
|
||||
similarity cosine similarity of the two product embeddings (0..1)
|
||||
rating quality Bayesian average / 5, so 5.0 from 3 ratings does not beat
|
||||
4.4 from 2,000: each product's rating is pulled towards the
|
||||
pool average as if PRIOR_WEIGHT extra ratings at that average
|
||||
had been given
|
||||
price closeness 1 at the same price, falling to 0 at twice (or zero) the price
|
||||
|
||||
When fewer than MIN_MATCHES similar products qualify, the list is filled with
|
||||
the best-rated products of the same category (in the same price band).
|
||||
|
||||
Better rated alternatives: products rated higher than this one by at least
|
||||
MIN_REVIEWS_BETTER people, best Bayesian rating first. When two round to the
|
||||
same Bayesian rating, the one whose stored reviews are more positive (share
|
||||
of positive minus share of negative) goes first.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
WEIGHT_SIMILARITY, WEIGHT_RATING, WEIGHT_PRICE = 0.60, 0.25, 0.15
|
||||
PRIOR_WEIGHT = 50
|
||||
DEFAULT_PRIOR_MEAN = 4.0 # used only when no product in the pool has a rating
|
||||
MAX_ITEMS = 6
|
||||
MIN_MATCHES = 3
|
||||
SIMILAR_PRICE_BAND = 0.30 # +/-30% of the product's best price
|
||||
BETTER_PRICE_BAND = 0.20 # +/-20%
|
||||
MIN_REVIEWS_BETTER = 5
|
||||
|
||||
|
||||
def prior_mean(candidates: List[Dict[str, Any]]) -> float:
|
||||
"""Average published rating across the pool."""
|
||||
rated = [c["rating"] for c in candidates if c.get("rating") is not None]
|
||||
return sum(rated) / len(rated) if rated else DEFAULT_PRIOR_MEAN
|
||||
|
||||
|
||||
def bayesian_rating(rating: Optional[float], count: Optional[int], mean: float) -> float:
|
||||
if rating is None:
|
||||
return mean
|
||||
n = max(int(count or 0), 1) # a rating with no stated count weighs as one
|
||||
return (PRIOR_WEIGHT * mean + n * float(rating)) / (PRIOR_WEIGHT + n)
|
||||
|
||||
|
||||
def price_closeness(price: Optional[float], target_price: Optional[float]) -> float:
|
||||
if price is None or not target_price:
|
||||
return 0.5 # unknown: neither helps nor hurts
|
||||
return max(0.0, 1.0 - abs(float(price) - float(target_price)) / float(target_price))
|
||||
|
||||
|
||||
def score(candidate: Dict[str, Any], target_price: Optional[float], mean: float) -> float:
|
||||
quality = bayesian_rating(candidate.get("rating"), candidate.get("rating_count"), mean) / 5
|
||||
return (WEIGHT_SIMILARITY * float(candidate.get("similarity") or 0.0)
|
||||
+ WEIGHT_RATING * quality
|
||||
+ WEIGHT_PRICE * price_closeness(candidate.get("best_price"), target_price))
|
||||
|
||||
|
||||
def reason(candidate: Dict[str, Any], target_rating: Optional[float], basis: str) -> str:
|
||||
label = "Similar specs" if basis == "similar" else "Top rated in this category"
|
||||
rating = candidate.get("rating")
|
||||
if rating is None:
|
||||
return label
|
||||
if target_rating is not None and rating > target_rating:
|
||||
return f"{label} · {rating:.1f}★ vs {target_rating:.1f}★"
|
||||
count = candidate.get("rating_count")
|
||||
return f"{label} · {rating:.1f}★" + (f" ({count:,} rating{'' if count == 1 else 's'})" if count else "")
|
||||
|
||||
|
||||
def recommend(
|
||||
target: Dict[str, Any],
|
||||
similar: List[Dict[str, Any]],
|
||||
top_rated: List[Dict[str, Any]],
|
||||
limit: int = MAX_ITEMS,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Order `similar` by score; when fewer than MIN_MATCHES are found, fill up
|
||||
to `limit` from `top_rated` (best Bayesian rating first).
|
||||
|
||||
Each candidate is a dict with product_id, best_price, rating, rating_count
|
||||
and (for `similar`) similarity. `target` has best_price and rating.
|
||||
Returns the chosen candidates with `reason` and `basis` added."""
|
||||
mean = prior_mean(similar + top_rated)
|
||||
target_price, target_rating = target.get("best_price"), target.get("rating")
|
||||
|
||||
ranked = sorted(similar, key=lambda c: score(c, target_price, mean), reverse=True)[:limit]
|
||||
picked = [{**c, "basis": "similar"} for c in ranked]
|
||||
|
||||
if len(picked) < MIN_MATCHES:
|
||||
seen = {c["product_id"] for c in picked}
|
||||
fill = sorted((c for c in top_rated if c["product_id"] not in seen and c.get("rating") is not None),
|
||||
key=lambda c: bayesian_rating(c["rating"], c.get("rating_count"), mean), reverse=True)
|
||||
picked += [{**c, "basis": "top_rated"} for c in fill[: limit - len(picked)]]
|
||||
|
||||
for c in picked:
|
||||
c["reason"] = reason(c, target_rating, c["basis"])
|
||||
return picked
|
||||
|
||||
|
||||
def sentiment_balance(sentiment: Optional[Dict[str, int]]) -> float:
|
||||
"""Share of positive minus share of negative stored reviews (-1..1); 0 with none."""
|
||||
if not sentiment:
|
||||
return 0.0
|
||||
total = sum(sentiment.values())
|
||||
if not total:
|
||||
return 0.0
|
||||
return (sentiment.get("positive", 0) - sentiment.get("negative", 0)) / total
|
||||
|
||||
|
||||
def better_rated(
|
||||
target: Dict[str, Any],
|
||||
rated: List[Dict[str, Any]],
|
||||
limit: int = MAX_ITEMS,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""The products in `rated` rated higher than `target` (any rated product
|
||||
when the target has no rating), each rated by at least MIN_REVIEWS_BETTER
|
||||
people. Each candidate has rating, rating_count and optionally `sentiment`
|
||||
({"positive": n, "neutral": n, "negative": n})."""
|
||||
target_rating = target.get("rating")
|
||||
mean = prior_mean(rated + [target])
|
||||
keep = [c for c in rated
|
||||
if c.get("rating") is not None and (c.get("rating_count") or 0) >= MIN_REVIEWS_BETTER
|
||||
and (target_rating is None or c["rating"] > target_rating)]
|
||||
keep.sort(key=lambda c: (round(bayesian_rating(c["rating"], c["rating_count"], mean), 1),
|
||||
sentiment_balance(c.get("sentiment"))), reverse=True)
|
||||
picked = []
|
||||
for c in keep[:limit]:
|
||||
vs = f" vs {target_rating:.1f}★" if target_rating is not None else ""
|
||||
picked.append({**c, "basis": "better_rated",
|
||||
"reason": f"{c['rating']:.1f}★{vs} · {c['rating_count']:,} ratings"})
|
||||
return picked
|
||||
@@ -27,13 +27,15 @@ mcp = FastMCP(
|
||||
"Every product is confirmed by real listings on at least two retail platforms; "
|
||||
"prices, ratings and reviews come with the page they were read from. Prices are "
|
||||
"rupee strings. Use search_products to find products, then get_product for "
|
||||
"per-platform offers, specs, images, rating and reviews."
|
||||
"per-platform offers, specs, images, rating and reviews, and recommend_products "
|
||||
"for similar alternatives."
|
||||
),
|
||||
)
|
||||
|
||||
_SEARCH_FIELDS = (
|
||||
"product_id", "brand", "category", "display_name", "ram_gb", "storage_gb",
|
||||
"best_price", "best_price_site", "platform_count", "sold_by_tn_retailer", "image_url",
|
||||
"rating", "rating_count",
|
||||
)
|
||||
|
||||
|
||||
@@ -71,7 +73,8 @@ async def search_products(
|
||||
limit: Maximum products to return (1-100).
|
||||
|
||||
Returns the total match count and, per product: id, name, variant, best price (rupee
|
||||
string) and the platform offering it, number of platforms, and an image URL (or null).
|
||||
string) and the platform offering it, number of platforms, an image URL (or null), and the
|
||||
overall rating and rating count (null when no platform publishes a rating).
|
||||
"""
|
||||
limit = max(1, min(int(limit), 100))
|
||||
result = await _run(
|
||||
@@ -115,6 +118,36 @@ async def get_product(product_id: int) -> Dict[str, Any]:
|
||||
}
|
||||
|
||||
|
||||
@mcp.tool
|
||||
async def recommend_products(product_id: int, kind: str = "similar", limit: int = 6) -> Dict[str, Any]:
|
||||
"""Alternatives to suggest for one product (by its product_id).
|
||||
|
||||
Args:
|
||||
product_id: The product to find alternatives for.
|
||||
kind: "similar" - closest specs, ranked by spec similarity, rating (weighted by how
|
||||
many people rated it) and price closeness; when few exist, the best-rated in the
|
||||
category fill the list. "better_rated" - products rated higher than this one by
|
||||
at least 5 people.
|
||||
limit: Maximum products to return (1-12).
|
||||
|
||||
Always same category, in stock, within a similar price (+/-30% for similar, +/-20% for
|
||||
better_rated), with other variants of the same model left out. Each item has a short
|
||||
reason (e.g. "Similar specs · 4.5★ vs 4.1★"). The same model's other RAM/storage
|
||||
variants are listed separately under other_variants.
|
||||
"""
|
||||
if kind not in ("similar", "better_rated"):
|
||||
raise ToolError('kind must be "similar" or "better_rated"')
|
||||
d = await _run(elec.recommendations, int(product_id), kind, max(1, min(int(limit), 12)), False)
|
||||
return {
|
||||
"items": [
|
||||
{k: i.get(k) for k in ("product_id", "brand", "display_name", "ram_gb", "storage_gb",
|
||||
"best_price", "best_price_site", "rating", "rating_count", "reason")}
|
||||
for i in d["items"]
|
||||
],
|
||||
"other_variants": d["other_variants"],
|
||||
}
|
||||
|
||||
|
||||
@mcp.tool
|
||||
async def price_history(product_id: int) -> List[Dict[str, Any]]:
|
||||
"""Every price observed for a product, per platform, oldest first (rupee strings, ISO times)."""
|
||||
|
||||
246
backend/tests/test_elec_recommend.py
Normal file
246
backend/tests/test_elec_recommend.py
Normal file
@@ -0,0 +1,246 @@
|
||||
"""Recommendations under a product's ratings and reviews (docs/RECOMMENDATIONS.md,
|
||||
Phase 1). Offline scoring tests first; the database tests are skipped when the
|
||||
local Postgres container is not running."""
|
||||
from __future__ import annotations
|
||||
|
||||
from decimal import Decimal
|
||||
|
||||
import pytest
|
||||
|
||||
from app.electronics import recommend as rec
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Scoring
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_bayesian_rating_trusts_many_ratings_over_few():
|
||||
mean = 4.0
|
||||
few_perfect = rec.bayesian_rating(5.0, 3, mean)
|
||||
many_good = rec.bayesian_rating(4.4, 2000, mean)
|
||||
assert many_good > few_perfect
|
||||
assert rec.bayesian_rating(None, None, mean) == mean # unrated: the pool average
|
||||
assert rec.bayesian_rating(5.0, None, mean) == rec.bayesian_rating(5.0, 1, mean)
|
||||
|
||||
|
||||
def test_price_closeness():
|
||||
assert rec.price_closeness(20000, 20000) == 1.0
|
||||
assert rec.price_closeness(15000, 20000) == pytest.approx(0.75)
|
||||
assert rec.price_closeness(45000, 20000) == 0.0 # never negative
|
||||
assert rec.price_closeness(20000, None) == 0.5 # unknown target price
|
||||
|
||||
|
||||
def test_similarity_dominates_but_rating_and_price_count():
|
||||
target = {"best_price": 20000.0, "rating": 4.1}
|
||||
close = {"product_id": 1, "similarity": 0.95, "best_price": 21000.0, "rating": 4.0, "rating_count": 500}
|
||||
far = {"product_id": 2, "similarity": 0.60, "best_price": 20000.0, "rating": 4.8, "rating_count": 5000}
|
||||
tie_better_rated = {"product_id": 3, "similarity": 0.95, "best_price": 21000.0, "rating": 4.6, "rating_count": 3000}
|
||||
out = rec.recommend(target, [far, close, tie_better_rated], [])
|
||||
assert [c["product_id"] for c in out] == [3, 1, 2]
|
||||
assert out[0]["reason"] == "Similar specs · 4.6★ vs 4.1★"
|
||||
assert out[1]["reason"] == "Similar specs · 4.0★ (500 ratings)"
|
||||
assert all(c["basis"] == "similar" for c in out)
|
||||
assert rec.reason({"rating": 4.0, "rating_count": 1}, None, "similar") == "Similar specs · 4.0★ (1 rating)"
|
||||
|
||||
|
||||
def test_limit_and_top_rated_fill_when_few_similar():
|
||||
target = {"best_price": 20000.0, "rating": None}
|
||||
similar = [{"product_id": 1, "similarity": 0.9, "best_price": 20000.0, "rating": None, "rating_count": None}]
|
||||
rated = [
|
||||
{"product_id": 1, "best_price": 20000.0, "rating": 4.9, "rating_count": 9000}, # already picked
|
||||
{"product_id": 2, "best_price": 30000.0, "rating": 4.2, "rating_count": 900},
|
||||
{"product_id": 3, "best_price": 25000.0, "rating": 4.7, "rating_count": 1200},
|
||||
{"product_id": 4, "best_price": 25000.0, "rating": None, "rating_count": None}, # unrated: never a "top rated"
|
||||
]
|
||||
out = rec.recommend(target, similar, rated, limit=3)
|
||||
assert [c["product_id"] for c in out] == [1, 3, 2]
|
||||
assert [c["basis"] for c in out] == ["similar", "top_rated", "top_rated"]
|
||||
assert out[0]["reason"] == "Similar specs"
|
||||
assert out[1]["reason"] == "Top rated in this category · 4.7★ (1,200 ratings)"
|
||||
|
||||
|
||||
def test_enough_similar_means_no_fill():
|
||||
similar = [{"product_id": i, "similarity": 0.5, "best_price": 1.0, "rating": None, "rating_count": None}
|
||||
for i in range(rec.MIN_MATCHES)]
|
||||
rated = [{"product_id": 99, "best_price": 1.0, "rating": 5.0, "rating_count": 10}]
|
||||
out = rec.recommend({"best_price": 1.0, "rating": None}, similar, rated)
|
||||
assert 99 not in {c["product_id"] for c in out}
|
||||
|
||||
|
||||
def test_better_rated_rules():
|
||||
target = {"rating": 4.1}
|
||||
rated = [
|
||||
{"product_id": 1, "rating": 4.5, "rating_count": 4000},
|
||||
{"product_id": 2, "rating": 4.9, "rating_count": 4}, # too few ratings
|
||||
{"product_id": 3, "rating": 4.1, "rating_count": 9000}, # not higher
|
||||
{"product_id": 4, "rating": 4.6, "rating_count": 3000},
|
||||
{"product_id": 5, "rating": None, "rating_count": None},
|
||||
]
|
||||
out = rec.better_rated(target, rated)
|
||||
assert [c["product_id"] for c in out] == [4, 1]
|
||||
assert out[0]["reason"] == "4.6★ vs 4.1★ · 3,000 ratings"
|
||||
# Unrated product: any well-rated product counts as better rated.
|
||||
out = rec.better_rated({"rating": None}, rated)
|
||||
assert [c["product_id"] for c in out] == [4, 1, 3]
|
||||
assert out[0]["reason"] == "4.6★ · 3,000 ratings"
|
||||
|
||||
|
||||
def test_better_rated_tie_goes_to_more_positive_reviews():
|
||||
rated = [
|
||||
{"product_id": 1, "rating": 4.5, "rating_count": 1000, "sentiment": {"positive": 2, "negative": 8}},
|
||||
{"product_id": 2, "rating": 4.5, "rating_count": 1000, "sentiment": {"positive": 8, "negative": 2}},
|
||||
{"product_id": 3, "rating": 4.5, "rating_count": 1000}, # no stored reviews: neutral
|
||||
]
|
||||
assert [c["product_id"] for c in rec.better_rated({"rating": 4.0}, rated)] == [2, 3, 1]
|
||||
assert rec.sentiment_balance({"positive": 3, "neutral": 1}) == 0.75
|
||||
assert rec.sentiment_balance({}) == 0.0
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# API (database)
|
||||
# ---------------------------------------------------------------------------
|
||||
def _vector(*head: float) -> list:
|
||||
v = list(head) + [0.0] * (384 - len(head))
|
||||
norm = sum(x * x for x in v) ** 0.5
|
||||
return [x / norm for x in v]
|
||||
|
||||
|
||||
def _seed():
|
||||
"""Samsung phones on two sites each, plus embeddings:
|
||||
S24 8/256 (target, 4.1★, ₹74,999), S24 8/128 (its variant), S23 (closest,
|
||||
4.5★ from 4,000), S22 (further, 4.3★ from only 4), A55 (close in specs but
|
||||
₹39,999 - outside both price bands), and an out-of-stock Z Flip6."""
|
||||
from app.electronics.collector import Collector, RunOptions, RunStats
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.db.connection import connect
|
||||
from app.electronics.models import Listing
|
||||
from app.electronics.normalise.title_parser import parse_title, variant_key
|
||||
|
||||
c = Collector.__new__(Collector)
|
||||
c.opt = RunOptions(category="mobiles", brands=["samsung"])
|
||||
c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats()
|
||||
|
||||
def store(title, sku, price, *, rating=None, count=None):
|
||||
p = parse_title(title, "mobiles")
|
||||
for site in ("amazon.in", "croma.com"):
|
||||
l = Listing(site_domain=site, source_sku=f"{site}-{sku}", source_url=f"https://www.{site}/p/{sku}",
|
||||
source_type="search_snippet", brand_slug="samsung", category="mobiles", title=title,
|
||||
evidence_text=f"{title} ₹{price}", confidence=0.5, parser="test",
|
||||
model=p.model, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=Decimal(price))
|
||||
l.model_norm, l.variant_key = p.model_norm, variant_key(p, "mobiles")
|
||||
l.rating, l.review_count = rating, count
|
||||
c.store(l)
|
||||
|
||||
store("Samsung Galaxy S24 5G (8GB RAM, 256GB)", "s24-256", 74999, rating=Decimal("4.1"), count=300)
|
||||
store("Samsung Galaxy S24 5G (8GB RAM, 128GB)", "s24-128", 69999)
|
||||
store("Samsung Galaxy S23 5G (8GB RAM, 256GB)", "s23", 64999, rating=Decimal("4.5"), count=2000)
|
||||
store("Samsung Galaxy S22 5G (8GB RAM, 256GB)", "s22", 79999, rating=Decimal("4.3"), count=2)
|
||||
store("Samsung Galaxy A55 5G (8GB RAM, 128GB)", "a55", 39999, rating=Decimal("4.2"), count=800)
|
||||
store("Samsung Galaxy Z Flip6 5G (12GB RAM, 256GB)", "flip6", 59999)
|
||||
repo.refresh_verification()
|
||||
|
||||
with connect() as conn: # every phone has its own price
|
||||
by_price = {int(r["best_price"]): r["product_id"] for r in conn.execute(
|
||||
"SELECT product_id, best_price FROM elec.v_brand_catalog")}
|
||||
pids = {"s24": by_price[74999], "s24_128": by_price[69999], "s23": by_price[64999],
|
||||
"s22": by_price[79999], "a55": by_price[39999], "flip6": by_price[59999]}
|
||||
with connect(autocommit=True) as conn: # the Flip6 sells out after it was verified
|
||||
conn.execute("UPDATE elec.source_listing SET in_stock = FALSE WHERE source_sku LIKE '%%-flip6'")
|
||||
vectors = {"s24": _vector(1, 0), "s24_128": _vector(1, 0), "s23": _vector(1, 0.2),
|
||||
"s22": _vector(1, 1), "a55": _vector(1, 0.1), "flip6": _vector(1, 0.1)}
|
||||
for key, pid in pids.items():
|
||||
repo.set_embedding(pid, vectors[key])
|
||||
return pids
|
||||
|
||||
|
||||
def test_api_recommends_similar_in_stock_products_without_variants(db, client):
|
||||
pids = _seed()
|
||||
body = client.get(f"/api/elec/products/{pids['s24']}/recommendations").json()
|
||||
got = [i["product_id"] for i in body["items"]]
|
||||
# Closest first; the variant, the out-of-stock Flip6 and the A55 (outside
|
||||
# the price limit, though close in specs) are left out.
|
||||
assert got == [pids["s23"], pids["s22"]]
|
||||
s23 = body["items"][0]
|
||||
assert s23["rating"] == 4.5 and s23["rating_count"] == 4000 # 2,000 on each of two sites
|
||||
assert s23["reason"] == "Similar specs · 4.5★ vs 4.1★"
|
||||
assert s23["best_price"] == "64999.00"
|
||||
assert [v["product_id"] for v in body["other_variants"]] == [pids["s24_128"]]
|
||||
|
||||
|
||||
def test_api_recommendations_unknown_product_and_bad_type(db, client):
|
||||
assert client.get("/api/elec/products/999999/recommendations").status_code == 404
|
||||
pids = _seed()
|
||||
assert client.get(f"/api/elec/products/{pids['s24']}/recommendations",
|
||||
params={"type": "cheapest"}).status_code == 422
|
||||
|
||||
|
||||
def test_api_falls_back_to_top_rated_without_embeddings(db, client):
|
||||
from app.electronics.db.connection import connect
|
||||
|
||||
pids = _seed()
|
||||
with connect(autocommit=True) as conn:
|
||||
conn.execute("UPDATE elec.product SET embedding = NULL")
|
||||
items = client.get(f"/api/elec/products/{pids['s24']}/recommendations").json()["items"]
|
||||
assert [i["product_id"] for i in items] == [pids["s23"], pids["s22"]]
|
||||
assert all(i["basis"] == "top_rated" for i in items)
|
||||
|
||||
|
||||
def test_api_better_rated_needs_higher_rating_enough_reviews_and_close_price(db, client):
|
||||
pids = _seed()
|
||||
body = client.get(f"/api/elec/products/{pids['s24']}/recommendations", params={"type": "better_rated"}).json()
|
||||
# S22 is rated higher but by only 4 people; A55 is outside +/-20% of the price.
|
||||
assert body["type"] == "better_rated"
|
||||
assert [i["product_id"] for i in body["items"]] == [pids["s23"]]
|
||||
assert body["items"][0]["reason"] == "4.5★ vs 4.1★ · 4,000 ratings"
|
||||
assert body["items"][0]["basis"] == "better_rated"
|
||||
|
||||
|
||||
def test_product_list_carries_the_overall_rating_for_card_badges(db, client):
|
||||
pids = _seed()
|
||||
products = {p["product_id"]: p for p in
|
||||
client.get("/api/elec/products", params={"category": "mobiles"}).json()["products"]}
|
||||
assert (products[pids["s23"]]["rating"], products[pids["s23"]]["rating_count"]) == (4.5, 4000)
|
||||
assert (products[pids["s24_128"]]["rating"], products[pids["s24_128"]]["rating_count"]) == (None, None)
|
||||
|
||||
|
||||
def test_laptop_variants_need_the_same_processor(db, client):
|
||||
"""One laptop line ("HP 15") spans many CPUs: only another RAM/storage of the
|
||||
same processor is a variant. Other CPUs, and part-number-only listings that
|
||||
state no processor, are separate products that can be recommended."""
|
||||
from app.electronics.collector import Collector, RunOptions, RunStats
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.db.connection import connect
|
||||
from app.electronics.models import Listing
|
||||
from app.electronics.normalise.title_parser import parse_title, variant_key
|
||||
|
||||
c = Collector.__new__(Collector)
|
||||
c.opt = RunOptions(category="laptops", brands=["hp"])
|
||||
c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats()
|
||||
for title, price in (("HP 15 Laptop AMD Ryzen 3 7320U (8GB RAM, 512GB SSD)", 40000),
|
||||
("HP 15 Laptop AMD Ryzen 3 7320U (16GB RAM, 512GB SSD)", 45000),
|
||||
("HP 15 Laptop AMD Ryzen 5 7520U (8GB RAM, 512GB SSD)", 47000),
|
||||
("HP 15 Laptop 15-FC0805AU (8GB RAM, 512GB SSD)", 41000),
|
||||
("HP 15 Laptop 15-FD0682TU (16GB RAM, 512GB SSD)", 42000)):
|
||||
p = parse_title(title, "laptops")
|
||||
for site in ("amazon.in", "croma.com"):
|
||||
l = Listing(site_domain=site, source_sku=f"{site}-{price}", source_url=f"https://www.{site}/p/{price}",
|
||||
source_type="search_snippet", brand_slug="hp", category="laptops", title=title,
|
||||
evidence_text=f"{title} ₹{price}", confidence=0.5, parser="test", model=p.model,
|
||||
model_number=p.mpn, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=Decimal(price))
|
||||
l.model_norm, l.variant_key, l.processor = p.model_norm, variant_key(p, "laptops"), p.processor
|
||||
c.store(l)
|
||||
repo.refresh_verification()
|
||||
with connect() as conn:
|
||||
by_price = {int(r["best_price"]): r["product_id"] for r in conn.execute(
|
||||
"SELECT product_id, best_price FROM elec.v_brand_catalog")}
|
||||
r3_8, r3_16, r5, fc, fd = (by_price[n] for n in (40000, 45000, 47000, 41000, 42000))
|
||||
for i, pid in enumerate((r3_8, r3_16, r5, fc, fd)):
|
||||
repo.set_embedding(pid, _vector(1, 0.1 * i))
|
||||
|
||||
body = client.get(f"/api/elec/products/{r3_8}/recommendations").json()
|
||||
assert [v["product_id"] for v in body["other_variants"]] == [r3_16]
|
||||
assert {r5, fc, fd} <= {i["product_id"] for i in body["items"]}
|
||||
assert r3_16 not in {i["product_id"] for i in body["items"]}
|
||||
|
||||
body = client.get(f"/api/elec/products/{fc}/recommendations").json()
|
||||
assert body["other_variants"] == []
|
||||
assert fd in {i["product_id"] for i in body["items"]}
|
||||
@@ -23,7 +23,8 @@ def test_only_read_only_catalogue_tools_are_exposed():
|
||||
async with Client(mcp) as c:
|
||||
return {t.name: set(t.input_schema.get("properties", {})) for t in await c.list_tools()}
|
||||
tools = anyio.run(go)
|
||||
assert set(tools) == {"list_categories", "search_products", "get_product", "price_history"}
|
||||
assert set(tools) == {"list_categories", "search_products", "get_product", "price_history",
|
||||
"recommend_products"}
|
||||
assert tools["search_products"] == {"query", "category", "brand", "max_price", "min_price", "limit"}
|
||||
# Nothing that can start a run, log in, or change data.
|
||||
assert not any(w in name for name in tools for w in ("admin", "run", "login", "probe", "review"))
|
||||
@@ -91,3 +92,11 @@ def test_unknown_product_is_a_tool_error(db):
|
||||
|
||||
with pytest.raises(ToolError, match="not found"):
|
||||
_call("get_product", {"product_id": 999999})
|
||||
|
||||
|
||||
def test_recommend_products_rejects_an_unknown_kind():
|
||||
import pytest
|
||||
from fastmcp.exceptions import ToolError
|
||||
|
||||
with pytest.raises(ToolError, match="better_rated"):
|
||||
_call("recommend_products", {"product_id": 1, "kind": "cheapest"})
|
||||
|
||||
Reference in New Issue
Block a user