diff --git a/backend/app/api/routers/elec.py b/backend/app/api/routers/elec.py index cb14ca4..e008256 100644 --- a/backend/app/api/routers/elec.py +++ b/backend/app/api/routers/elec.py @@ -7,12 +7,20 @@ Money is returned as a decimal string, never a float. from __future__ import annotations from decimal import Decimal -from typing import Any, Dict, List, Optional +from typing import Any, Dict, List, Literal, Optional from fastapi import APIRouter, HTTPException, Query from app.electronics.db.connection import connect -from app.electronics.db.repository import product_rating_and_reviews +from app.electronics import recommend as rec +from app.electronics.db.repository import ( + other_variants, + product_rating_and_reviews, + rated_products, + rating_sources_for, + review_sentiment_counts, + similar_products, +) from app.electronics.reviews import select_reviews router = APIRouter(prefix="/elec", tags=["electronics"]) @@ -106,7 +114,13 @@ def products( f"LIMIT %(limit)s OFFSET %(offset)s", {**params, "limit": limit, "offset": offset}, ).fetchall() - return {"total": total, "products": [_clean(r) for r in rows]} + sources = rating_sources_for(conn, [r["product_id"] for r in rows]) + out = [] + for r in rows: + rating = _overall_rating(sources[r["product_id"]]) + out.append({**_clean(r), "rating": rating["value"] if rating else None, + "rating_count": rating["count"] if rating else None}) + return {"total": total, "products": out} @router.get("/products/{product_id}") @@ -185,6 +199,63 @@ def _breakdown(sources: List[dict]) -> Optional[List[dict]]: for k in ("5", "4", "3", "2", "1")] +@router.get("/products/{product_id}/recommendations") +def recommendations( + product_id: int, + kind: Literal["similar", "better_rated"] = Query("similar", alias="type"), + limit: int = Query(rec.MAX_ITEMS, ge=1, le=12), + tn_only: bool = False, +) -> dict: + """Products to suggest under this one's ratings and reviews (see + app/electronics/recommend.py), plus the same model's other variants. + type=similar: closest specs; type=better_rated: rated higher, similar price.""" + with connect() as conn: + target = conn.execute( + "SELECT product_id, best_price FROM elec.v_brand_catalog WHERE product_id = %s", (product_id,) + ).fetchone() + if not target: + raise HTTPException(status_code=404, detail="Product not found or not verified") + if kind == "similar": + similar = similar_products(conn, product_id, band=rec.SIMILAR_PRICE_BAND, tn_only=tn_only) + rated = (rated_products(conn, product_id, band=rec.SIMILAR_PRICE_BAND, tn_only=tn_only) + if len(similar) < rec.MIN_MATCHES else []) + else: + similar = [] + rated = rated_products(conn, product_id, band=rec.BETTER_PRICE_BAND, tn_only=tn_only) + ids = {product_id} | {c["product_id"] for c in similar + rated} + ratings = {pid: _overall_rating(src) for pid, src in rating_sources_for(conn, list(ids)).items()} + + def with_rating(c: dict) -> dict: + r = ratings.get(c["product_id"]) + return {**c, "best_price": _float(c.get("best_price")), + "rating": r["value"] if r else None, "rating_count": r["count"] if r else None} + + if kind == "similar": + picked = rec.recommend(with_rating(dict(target)), [with_rating(c) for c in similar], + [with_rating(c) for c in rated], limit) + else: + sentiment = review_sentiment_counts(conn, [c["product_id"] for c in rated]) + picked = rec.better_rated( + with_rating(dict(target)), + [{**with_rating(c), "sentiment": sentiment.get(c["product_id"])} for c in rated], limit) + cards = {r["product_id"]: r for r in conn.execute( + "SELECT product_id, brand, display_name, ram_gb, storage_gb, image_url, best_price, best_price_site " + "FROM elec.v_brand_catalog WHERE product_id = ANY(%s)", ([c["product_id"] for c in picked],) + )} + variants = other_variants(conn, product_id) + items = [ + {**_clean(cards[c["product_id"]]), "rating": c["rating"], "rating_count": c["rating_count"], + "basis": c["basis"], "reason": c["reason"]} + for c in picked if c["product_id"] in cards + ] + return {"product_id": product_id, "type": kind, "items": items, + "other_variants": [_clean(v) for v in variants]} + + +def _float(value: Any) -> Optional[float]: + return None if value is None else float(value) + + @router.get("/products/{product_id}/price-history") def price_history(product_id: int) -> List[dict]: with connect() as conn: diff --git a/backend/app/electronics/db/repository.py b/backend/app/electronics/db/repository.py index 85dca58..b1db1b4 100644 --- a/backend/app/electronics/db/repository.py +++ b/backend/app/electronics/db/repository.py @@ -305,6 +305,96 @@ def product_rating_and_reviews(conn, product_id: int) -> Dict[str, Any]: return {"sources": [dict(s) for s in sources], "reviews": [dict(r) for r in reviews]} +def rating_sources_for(conn, product_ids: List[int]) -> Dict[int, List[dict]]: + """product_rating_and_reviews' per-platform ratings for many products at once.""" + out: Dict[int, List[dict]] = {pid: [] for pid in product_ids} + rows = conn.execute( + "SELECT a.product_id, a.site, a.source_url, l.rating, l.review_count, l.rating_breakdown " + "FROM elec.v_product_availability a JOIN elec.source_listing l ON l.id = a.listing_id " + "WHERE a.product_id = ANY(%s) AND l.rating > 0 ORDER BY l.review_count DESC NULLS LAST, a.site", + (list(product_ids),), + ).fetchall() + for r in rows: + out[r["product_id"]].append(dict(r)) + return out + + +# p is a variant of t when the two differ only in RAM/storage: same brand and +# model and the same processor. One laptop model line ("HP 15") spans many +# CPUs, so laptops count as variants only when both state the same processor; +# phones state none, so for them the model alone decides. Written so it is +# never NULL: NOT NULL would drop the product from recommendations too. +_SAME_MODEL = """(p.brand_id = t.brand_id AND p.model_norm = t.model_norm + AND p.processor IS NOT DISTINCT FROM t.processor + AND (p.processor IS NOT NULL + OR t.category_id IS DISTINCT FROM (SELECT id FROM elec.category WHERE slug = 'laptops')))""" + +# Products that may be recommended for target t: verified, same category, not +# t itself or another RAM/storage variant of the same model, with an in-stock +# (or unknown-stock) best price within +/- %(band)s of t's own (no limit when t +# has no price). +_RECOMMENDABLE = """ + FROM elec.product t + JOIN elec.product p ON p.category_id = t.category_id AND p.id <> t.id + AND p.verification_status = 'verified' + AND NOT """ + _SAME_MODEL + """ + JOIN elec.v_best_price bp ON bp.product_id = p.id AND bp.in_stock IS DISTINCT FROM FALSE + LEFT JOIN elec.v_best_price tp ON tp.product_id = t.id + WHERE t.id = %(pid)s + AND (tp.price IS NULL OR bp.price BETWEEN tp.price * (1 - %(band)s) AND tp.price * (1 + %(band)s)) +""" +_TN_ONLY = (" AND EXISTS (SELECT 1 FROM elec.v_product_availability a" + " WHERE a.product_id = p.id AND a.site_region = 'TN')") + + +def similar_products(conn, product_id: int, *, band: float, limit: int = 30, tn_only: bool = False) -> List[dict]: + """Recommendable products closest to this one by embedding (cosine), with + their best price. Empty when the product has no embedding yet. + + An exact scan, not the HNSW index: the category/stock filters would make an + approximate index search drop matches, and a category is small enough.""" + sql = ("SELECT p.id AS product_id, 1 - (p.embedding <=> t.embedding) AS similarity, bp.price AS best_price" + + _RECOMMENDABLE + " AND t.embedding IS NOT NULL AND p.embedding IS NOT NULL" + + (_TN_ONLY if tn_only else "") + + " ORDER BY p.embedding <=> t.embedding LIMIT %(limit)s") + return [dict(r) for r in conn.execute(sql, {"pid": product_id, "band": band, "limit": limit})] + + +def rated_products(conn, product_id: int, *, band: float, tn_only: bool = False) -> List[dict]: + """Recommendable products that at least one platform has rated.""" + sql = ("SELECT p.id AS product_id, bp.price AS best_price" + _RECOMMENDABLE + + " AND EXISTS (SELECT 1 FROM elec.v_product_availability a JOIN elec.source_listing l" + " ON l.id = a.listing_id WHERE a.product_id = p.id AND l.rating > 0)" + + (_TN_ONLY if tn_only else "")) + return [dict(r) for r in conn.execute(sql, {"pid": product_id, "band": band})] + + +def review_sentiment_counts(conn, product_ids: List[int]) -> Dict[int, Dict[str, int]]: + """How many stored reviews of each product are positive / neutral / negative.""" + out: Dict[int, Dict[str, int]] = {pid: {} for pid in product_ids} + rows = conn.execute( + "SELECT a.product_id, r.sentiment, count(*)::int AS n " + "FROM elec.v_product_availability a JOIN elec.listing_review r ON r.listing_id = a.listing_id " + "WHERE a.product_id = ANY(%s) AND r.sentiment IS NOT NULL GROUP BY 1, 2", + (list(product_ids),), + ).fetchall() + for r in rows: + out[r["product_id"]][r["sentiment"]] = r["n"] + return out + + +def other_variants(conn, product_id: int) -> List[dict]: + """The same model's other verified RAM/storage variants (see _SAME_MODEL).""" + return [dict(r) for r in conn.execute( + "SELECT v.product_id, v.display_name, v.ram_gb, v.storage_gb, v.best_price " + "FROM elec.product t JOIN elec.product p ON p.category_id = t.category_id AND p.id <> t.id " + "AND " + _SAME_MODEL + " " + "JOIN elec.v_brand_catalog v ON v.product_id = p.id " + "WHERE t.id = %s ORDER BY v.ram_gb NULLS LAST, v.storage_gb NULLS LAST, v.display_name", + (product_id,), + )] + + def listings_for_review_backfill(category: Optional[str] = None) -> List[dict]: """Page-read listings of verified products, for re-reading ratings/reviews.""" sql = ( diff --git a/backend/app/electronics/recommend.py b/backend/app/electronics/recommend.py new file mode 100644 index 0000000..ba9a7e4 --- /dev/null +++ b/backend/app/electronics/recommend.py @@ -0,0 +1,144 @@ +"""Which other products to suggest under a product's ratings and reviews +(docs/RECOMMENDATIONS.md): Phase 1 "Similar products" and Phase 2 "Better +rated alternatives". + +This module only scores and orders candidates; the database queries that find +them (repository.similar_products / rated_products) and the endpoint live +elsewhere. Nothing here invents a number: a product with no published rating +is scored at the pool's average, and one with no price is never a candidate. + +Hard price limit: every candidate's best price is within SIMILAR_PRICE_BAND +(similar) or BETTER_PRICE_BAND (better rated) of the product's own, applied in +the database query. A product with no price of its own gets no limit. + +Similar products: + + score = 0.60 x similarity + 0.25 x rating quality + 0.15 x price closeness + +similarity cosine similarity of the two product embeddings (0..1) +rating quality Bayesian average / 5, so 5.0 from 3 ratings does not beat + 4.4 from 2,000: each product's rating is pulled towards the + pool average as if PRIOR_WEIGHT extra ratings at that average + had been given +price closeness 1 at the same price, falling to 0 at twice (or zero) the price + +When fewer than MIN_MATCHES similar products qualify, the list is filled with +the best-rated products of the same category (in the same price band). + +Better rated alternatives: products rated higher than this one by at least +MIN_REVIEWS_BETTER people, best Bayesian rating first. When two round to the +same Bayesian rating, the one whose stored reviews are more positive (share +of positive minus share of negative) goes first. +""" +from __future__ import annotations + +from typing import Any, Dict, List, Optional + +WEIGHT_SIMILARITY, WEIGHT_RATING, WEIGHT_PRICE = 0.60, 0.25, 0.15 +PRIOR_WEIGHT = 50 +DEFAULT_PRIOR_MEAN = 4.0 # used only when no product in the pool has a rating +MAX_ITEMS = 6 +MIN_MATCHES = 3 +SIMILAR_PRICE_BAND = 0.30 # +/-30% of the product's best price +BETTER_PRICE_BAND = 0.20 # +/-20% +MIN_REVIEWS_BETTER = 5 + + +def prior_mean(candidates: List[Dict[str, Any]]) -> float: + """Average published rating across the pool.""" + rated = [c["rating"] for c in candidates if c.get("rating") is not None] + return sum(rated) / len(rated) if rated else DEFAULT_PRIOR_MEAN + + +def bayesian_rating(rating: Optional[float], count: Optional[int], mean: float) -> float: + if rating is None: + return mean + n = max(int(count or 0), 1) # a rating with no stated count weighs as one + return (PRIOR_WEIGHT * mean + n * float(rating)) / (PRIOR_WEIGHT + n) + + +def price_closeness(price: Optional[float], target_price: Optional[float]) -> float: + if price is None or not target_price: + return 0.5 # unknown: neither helps nor hurts + return max(0.0, 1.0 - abs(float(price) - float(target_price)) / float(target_price)) + + +def score(candidate: Dict[str, Any], target_price: Optional[float], mean: float) -> float: + quality = bayesian_rating(candidate.get("rating"), candidate.get("rating_count"), mean) / 5 + return (WEIGHT_SIMILARITY * float(candidate.get("similarity") or 0.0) + + WEIGHT_RATING * quality + + WEIGHT_PRICE * price_closeness(candidate.get("best_price"), target_price)) + + +def reason(candidate: Dict[str, Any], target_rating: Optional[float], basis: str) -> str: + label = "Similar specs" if basis == "similar" else "Top rated in this category" + rating = candidate.get("rating") + if rating is None: + return label + if target_rating is not None and rating > target_rating: + return f"{label} · {rating:.1f}★ vs {target_rating:.1f}★" + count = candidate.get("rating_count") + return f"{label} · {rating:.1f}★" + (f" ({count:,} rating{'' if count == 1 else 's'})" if count else "") + + +def recommend( + target: Dict[str, Any], + similar: List[Dict[str, Any]], + top_rated: List[Dict[str, Any]], + limit: int = MAX_ITEMS, +) -> List[Dict[str, Any]]: + """Order `similar` by score; when fewer than MIN_MATCHES are found, fill up + to `limit` from `top_rated` (best Bayesian rating first). + + Each candidate is a dict with product_id, best_price, rating, rating_count + and (for `similar`) similarity. `target` has best_price and rating. + Returns the chosen candidates with `reason` and `basis` added.""" + mean = prior_mean(similar + top_rated) + target_price, target_rating = target.get("best_price"), target.get("rating") + + ranked = sorted(similar, key=lambda c: score(c, target_price, mean), reverse=True)[:limit] + picked = [{**c, "basis": "similar"} for c in ranked] + + if len(picked) < MIN_MATCHES: + seen = {c["product_id"] for c in picked} + fill = sorted((c for c in top_rated if c["product_id"] not in seen and c.get("rating") is not None), + key=lambda c: bayesian_rating(c["rating"], c.get("rating_count"), mean), reverse=True) + picked += [{**c, "basis": "top_rated"} for c in fill[: limit - len(picked)]] + + for c in picked: + c["reason"] = reason(c, target_rating, c["basis"]) + return picked + + +def sentiment_balance(sentiment: Optional[Dict[str, int]]) -> float: + """Share of positive minus share of negative stored reviews (-1..1); 0 with none.""" + if not sentiment: + return 0.0 + total = sum(sentiment.values()) + if not total: + return 0.0 + return (sentiment.get("positive", 0) - sentiment.get("negative", 0)) / total + + +def better_rated( + target: Dict[str, Any], + rated: List[Dict[str, Any]], + limit: int = MAX_ITEMS, +) -> List[Dict[str, Any]]: + """The products in `rated` rated higher than `target` (any rated product + when the target has no rating), each rated by at least MIN_REVIEWS_BETTER + people. Each candidate has rating, rating_count and optionally `sentiment` + ({"positive": n, "neutral": n, "negative": n}).""" + target_rating = target.get("rating") + mean = prior_mean(rated + [target]) + keep = [c for c in rated + if c.get("rating") is not None and (c.get("rating_count") or 0) >= MIN_REVIEWS_BETTER + and (target_rating is None or c["rating"] > target_rating)] + keep.sort(key=lambda c: (round(bayesian_rating(c["rating"], c["rating_count"], mean), 1), + sentiment_balance(c.get("sentiment"))), reverse=True) + picked = [] + for c in keep[:limit]: + vs = f" vs {target_rating:.1f}★" if target_rating is not None else "" + picked.append({**c, "basis": "better_rated", + "reason": f"{c['rating']:.1f}★{vs} · {c['rating_count']:,} ratings"}) + return picked diff --git a/backend/app/mcp_server.py b/backend/app/mcp_server.py index 0204e1f..55ce9cf 100644 --- a/backend/app/mcp_server.py +++ b/backend/app/mcp_server.py @@ -27,13 +27,15 @@ mcp = FastMCP( "Every product is confirmed by real listings on at least two retail platforms; " "prices, ratings and reviews come with the page they were read from. Prices are " "rupee strings. Use search_products to find products, then get_product for " - "per-platform offers, specs, images, rating and reviews." + "per-platform offers, specs, images, rating and reviews, and recommend_products " + "for similar alternatives." ), ) _SEARCH_FIELDS = ( "product_id", "brand", "category", "display_name", "ram_gb", "storage_gb", "best_price", "best_price_site", "platform_count", "sold_by_tn_retailer", "image_url", + "rating", "rating_count", ) @@ -71,7 +73,8 @@ async def search_products( limit: Maximum products to return (1-100). Returns the total match count and, per product: id, name, variant, best price (rupee - string) and the platform offering it, number of platforms, and an image URL (or null). + string) and the platform offering it, number of platforms, an image URL (or null), and the + overall rating and rating count (null when no platform publishes a rating). """ limit = max(1, min(int(limit), 100)) result = await _run( @@ -115,6 +118,36 @@ async def get_product(product_id: int) -> Dict[str, Any]: } +@mcp.tool +async def recommend_products(product_id: int, kind: str = "similar", limit: int = 6) -> Dict[str, Any]: + """Alternatives to suggest for one product (by its product_id). + + Args: + product_id: The product to find alternatives for. + kind: "similar" - closest specs, ranked by spec similarity, rating (weighted by how + many people rated it) and price closeness; when few exist, the best-rated in the + category fill the list. "better_rated" - products rated higher than this one by + at least 5 people. + limit: Maximum products to return (1-12). + + Always same category, in stock, within a similar price (+/-30% for similar, +/-20% for + better_rated), with other variants of the same model left out. Each item has a short + reason (e.g. "Similar specs · 4.5★ vs 4.1★"). The same model's other RAM/storage + variants are listed separately under other_variants. + """ + if kind not in ("similar", "better_rated"): + raise ToolError('kind must be "similar" or "better_rated"') + d = await _run(elec.recommendations, int(product_id), kind, max(1, min(int(limit), 12)), False) + return { + "items": [ + {k: i.get(k) for k in ("product_id", "brand", "display_name", "ram_gb", "storage_gb", + "best_price", "best_price_site", "rating", "rating_count", "reason")} + for i in d["items"] + ], + "other_variants": d["other_variants"], + } + + @mcp.tool async def price_history(product_id: int) -> List[Dict[str, Any]]: """Every price observed for a product, per platform, oldest first (rupee strings, ISO times).""" diff --git a/backend/tests/test_elec_recommend.py b/backend/tests/test_elec_recommend.py new file mode 100644 index 0000000..8411b8e --- /dev/null +++ b/backend/tests/test_elec_recommend.py @@ -0,0 +1,246 @@ +"""Recommendations under a product's ratings and reviews (docs/RECOMMENDATIONS.md, +Phase 1). Offline scoring tests first; the database tests are skipped when the +local Postgres container is not running.""" +from __future__ import annotations + +from decimal import Decimal + +import pytest + +from app.electronics import recommend as rec + + +# --------------------------------------------------------------------------- +# Scoring +# --------------------------------------------------------------------------- +def test_bayesian_rating_trusts_many_ratings_over_few(): + mean = 4.0 + few_perfect = rec.bayesian_rating(5.0, 3, mean) + many_good = rec.bayesian_rating(4.4, 2000, mean) + assert many_good > few_perfect + assert rec.bayesian_rating(None, None, mean) == mean # unrated: the pool average + assert rec.bayesian_rating(5.0, None, mean) == rec.bayesian_rating(5.0, 1, mean) + + +def test_price_closeness(): + assert rec.price_closeness(20000, 20000) == 1.0 + assert rec.price_closeness(15000, 20000) == pytest.approx(0.75) + assert rec.price_closeness(45000, 20000) == 0.0 # never negative + assert rec.price_closeness(20000, None) == 0.5 # unknown target price + + +def test_similarity_dominates_but_rating_and_price_count(): + target = {"best_price": 20000.0, "rating": 4.1} + close = {"product_id": 1, "similarity": 0.95, "best_price": 21000.0, "rating": 4.0, "rating_count": 500} + far = {"product_id": 2, "similarity": 0.60, "best_price": 20000.0, "rating": 4.8, "rating_count": 5000} + tie_better_rated = {"product_id": 3, "similarity": 0.95, "best_price": 21000.0, "rating": 4.6, "rating_count": 3000} + out = rec.recommend(target, [far, close, tie_better_rated], []) + assert [c["product_id"] for c in out] == [3, 1, 2] + assert out[0]["reason"] == "Similar specs · 4.6★ vs 4.1★" + assert out[1]["reason"] == "Similar specs · 4.0★ (500 ratings)" + assert all(c["basis"] == "similar" for c in out) + assert rec.reason({"rating": 4.0, "rating_count": 1}, None, "similar") == "Similar specs · 4.0★ (1 rating)" + + +def test_limit_and_top_rated_fill_when_few_similar(): + target = {"best_price": 20000.0, "rating": None} + similar = [{"product_id": 1, "similarity": 0.9, "best_price": 20000.0, "rating": None, "rating_count": None}] + rated = [ + {"product_id": 1, "best_price": 20000.0, "rating": 4.9, "rating_count": 9000}, # already picked + {"product_id": 2, "best_price": 30000.0, "rating": 4.2, "rating_count": 900}, + {"product_id": 3, "best_price": 25000.0, "rating": 4.7, "rating_count": 1200}, + {"product_id": 4, "best_price": 25000.0, "rating": None, "rating_count": None}, # unrated: never a "top rated" + ] + out = rec.recommend(target, similar, rated, limit=3) + assert [c["product_id"] for c in out] == [1, 3, 2] + assert [c["basis"] for c in out] == ["similar", "top_rated", "top_rated"] + assert out[0]["reason"] == "Similar specs" + assert out[1]["reason"] == "Top rated in this category · 4.7★ (1,200 ratings)" + + +def test_enough_similar_means_no_fill(): + similar = [{"product_id": i, "similarity": 0.5, "best_price": 1.0, "rating": None, "rating_count": None} + for i in range(rec.MIN_MATCHES)] + rated = [{"product_id": 99, "best_price": 1.0, "rating": 5.0, "rating_count": 10}] + out = rec.recommend({"best_price": 1.0, "rating": None}, similar, rated) + assert 99 not in {c["product_id"] for c in out} + + +def test_better_rated_rules(): + target = {"rating": 4.1} + rated = [ + {"product_id": 1, "rating": 4.5, "rating_count": 4000}, + {"product_id": 2, "rating": 4.9, "rating_count": 4}, # too few ratings + {"product_id": 3, "rating": 4.1, "rating_count": 9000}, # not higher + {"product_id": 4, "rating": 4.6, "rating_count": 3000}, + {"product_id": 5, "rating": None, "rating_count": None}, + ] + out = rec.better_rated(target, rated) + assert [c["product_id"] for c in out] == [4, 1] + assert out[0]["reason"] == "4.6★ vs 4.1★ · 3,000 ratings" + # Unrated product: any well-rated product counts as better rated. + out = rec.better_rated({"rating": None}, rated) + assert [c["product_id"] for c in out] == [4, 1, 3] + assert out[0]["reason"] == "4.6★ · 3,000 ratings" + + +def test_better_rated_tie_goes_to_more_positive_reviews(): + rated = [ + {"product_id": 1, "rating": 4.5, "rating_count": 1000, "sentiment": {"positive": 2, "negative": 8}}, + {"product_id": 2, "rating": 4.5, "rating_count": 1000, "sentiment": {"positive": 8, "negative": 2}}, + {"product_id": 3, "rating": 4.5, "rating_count": 1000}, # no stored reviews: neutral + ] + assert [c["product_id"] for c in rec.better_rated({"rating": 4.0}, rated)] == [2, 3, 1] + assert rec.sentiment_balance({"positive": 3, "neutral": 1}) == 0.75 + assert rec.sentiment_balance({}) == 0.0 + + +# --------------------------------------------------------------------------- +# API (database) +# --------------------------------------------------------------------------- +def _vector(*head: float) -> list: + v = list(head) + [0.0] * (384 - len(head)) + norm = sum(x * x for x in v) ** 0.5 + return [x / norm for x in v] + + +def _seed(): + """Samsung phones on two sites each, plus embeddings: + S24 8/256 (target, 4.1★, ₹74,999), S24 8/128 (its variant), S23 (closest, + 4.5★ from 4,000), S22 (further, 4.3★ from only 4), A55 (close in specs but + ₹39,999 - outside both price bands), and an out-of-stock Z Flip6.""" + from app.electronics.collector import Collector, RunOptions, RunStats + from app.electronics.db import repository as repo + from app.electronics.db.connection import connect + from app.electronics.models import Listing + from app.electronics.normalise.title_parser import parse_title, variant_key + + c = Collector.__new__(Collector) + c.opt = RunOptions(category="mobiles", brands=["samsung"]) + c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats() + + def store(title, sku, price, *, rating=None, count=None): + p = parse_title(title, "mobiles") + for site in ("amazon.in", "croma.com"): + l = Listing(site_domain=site, source_sku=f"{site}-{sku}", source_url=f"https://www.{site}/p/{sku}", + source_type="search_snippet", brand_slug="samsung", category="mobiles", title=title, + evidence_text=f"{title} ₹{price}", confidence=0.5, parser="test", + model=p.model, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=Decimal(price)) + l.model_norm, l.variant_key = p.model_norm, variant_key(p, "mobiles") + l.rating, l.review_count = rating, count + c.store(l) + + store("Samsung Galaxy S24 5G (8GB RAM, 256GB)", "s24-256", 74999, rating=Decimal("4.1"), count=300) + store("Samsung Galaxy S24 5G (8GB RAM, 128GB)", "s24-128", 69999) + store("Samsung Galaxy S23 5G (8GB RAM, 256GB)", "s23", 64999, rating=Decimal("4.5"), count=2000) + store("Samsung Galaxy S22 5G (8GB RAM, 256GB)", "s22", 79999, rating=Decimal("4.3"), count=2) + store("Samsung Galaxy A55 5G (8GB RAM, 128GB)", "a55", 39999, rating=Decimal("4.2"), count=800) + store("Samsung Galaxy Z Flip6 5G (12GB RAM, 256GB)", "flip6", 59999) + repo.refresh_verification() + + with connect() as conn: # every phone has its own price + by_price = {int(r["best_price"]): r["product_id"] for r in conn.execute( + "SELECT product_id, best_price FROM elec.v_brand_catalog")} + pids = {"s24": by_price[74999], "s24_128": by_price[69999], "s23": by_price[64999], + "s22": by_price[79999], "a55": by_price[39999], "flip6": by_price[59999]} + with connect(autocommit=True) as conn: # the Flip6 sells out after it was verified + conn.execute("UPDATE elec.source_listing SET in_stock = FALSE WHERE source_sku LIKE '%%-flip6'") + vectors = {"s24": _vector(1, 0), "s24_128": _vector(1, 0), "s23": _vector(1, 0.2), + "s22": _vector(1, 1), "a55": _vector(1, 0.1), "flip6": _vector(1, 0.1)} + for key, pid in pids.items(): + repo.set_embedding(pid, vectors[key]) + return pids + + +def test_api_recommends_similar_in_stock_products_without_variants(db, client): + pids = _seed() + body = client.get(f"/api/elec/products/{pids['s24']}/recommendations").json() + got = [i["product_id"] for i in body["items"]] + # Closest first; the variant, the out-of-stock Flip6 and the A55 (outside + # the price limit, though close in specs) are left out. + assert got == [pids["s23"], pids["s22"]] + s23 = body["items"][0] + assert s23["rating"] == 4.5 and s23["rating_count"] == 4000 # 2,000 on each of two sites + assert s23["reason"] == "Similar specs · 4.5★ vs 4.1★" + assert s23["best_price"] == "64999.00" + assert [v["product_id"] for v in body["other_variants"]] == [pids["s24_128"]] + + +def test_api_recommendations_unknown_product_and_bad_type(db, client): + assert client.get("/api/elec/products/999999/recommendations").status_code == 404 + pids = _seed() + assert client.get(f"/api/elec/products/{pids['s24']}/recommendations", + params={"type": "cheapest"}).status_code == 422 + + +def test_api_falls_back_to_top_rated_without_embeddings(db, client): + from app.electronics.db.connection import connect + + pids = _seed() + with connect(autocommit=True) as conn: + conn.execute("UPDATE elec.product SET embedding = NULL") + items = client.get(f"/api/elec/products/{pids['s24']}/recommendations").json()["items"] + assert [i["product_id"] for i in items] == [pids["s23"], pids["s22"]] + assert all(i["basis"] == "top_rated" for i in items) + + +def test_api_better_rated_needs_higher_rating_enough_reviews_and_close_price(db, client): + pids = _seed() + body = client.get(f"/api/elec/products/{pids['s24']}/recommendations", params={"type": "better_rated"}).json() + # S22 is rated higher but by only 4 people; A55 is outside +/-20% of the price. + assert body["type"] == "better_rated" + assert [i["product_id"] for i in body["items"]] == [pids["s23"]] + assert body["items"][0]["reason"] == "4.5★ vs 4.1★ · 4,000 ratings" + assert body["items"][0]["basis"] == "better_rated" + + +def test_product_list_carries_the_overall_rating_for_card_badges(db, client): + pids = _seed() + products = {p["product_id"]: p for p in + client.get("/api/elec/products", params={"category": "mobiles"}).json()["products"]} + assert (products[pids["s23"]]["rating"], products[pids["s23"]]["rating_count"]) == (4.5, 4000) + assert (products[pids["s24_128"]]["rating"], products[pids["s24_128"]]["rating_count"]) == (None, None) + + +def test_laptop_variants_need_the_same_processor(db, client): + """One laptop line ("HP 15") spans many CPUs: only another RAM/storage of the + same processor is a variant. Other CPUs, and part-number-only listings that + state no processor, are separate products that can be recommended.""" + from app.electronics.collector import Collector, RunOptions, RunStats + from app.electronics.db import repository as repo + from app.electronics.db.connection import connect + from app.electronics.models import Listing + from app.electronics.normalise.title_parser import parse_title, variant_key + + c = Collector.__new__(Collector) + c.opt = RunOptions(category="laptops", brands=["hp"]) + c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats() + for title, price in (("HP 15 Laptop AMD Ryzen 3 7320U (8GB RAM, 512GB SSD)", 40000), + ("HP 15 Laptop AMD Ryzen 3 7320U (16GB RAM, 512GB SSD)", 45000), + ("HP 15 Laptop AMD Ryzen 5 7520U (8GB RAM, 512GB SSD)", 47000), + ("HP 15 Laptop 15-FC0805AU (8GB RAM, 512GB SSD)", 41000), + ("HP 15 Laptop 15-FD0682TU (16GB RAM, 512GB SSD)", 42000)): + p = parse_title(title, "laptops") + for site in ("amazon.in", "croma.com"): + l = Listing(site_domain=site, source_sku=f"{site}-{price}", source_url=f"https://www.{site}/p/{price}", + source_type="search_snippet", brand_slug="hp", category="laptops", title=title, + evidence_text=f"{title} ₹{price}", confidence=0.5, parser="test", model=p.model, + model_number=p.mpn, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=Decimal(price)) + l.model_norm, l.variant_key, l.processor = p.model_norm, variant_key(p, "laptops"), p.processor + c.store(l) + repo.refresh_verification() + with connect() as conn: + by_price = {int(r["best_price"]): r["product_id"] for r in conn.execute( + "SELECT product_id, best_price FROM elec.v_brand_catalog")} + r3_8, r3_16, r5, fc, fd = (by_price[n] for n in (40000, 45000, 47000, 41000, 42000)) + for i, pid in enumerate((r3_8, r3_16, r5, fc, fd)): + repo.set_embedding(pid, _vector(1, 0.1 * i)) + + body = client.get(f"/api/elec/products/{r3_8}/recommendations").json() + assert [v["product_id"] for v in body["other_variants"]] == [r3_16] + assert {r5, fc, fd} <= {i["product_id"] for i in body["items"]} + assert r3_16 not in {i["product_id"] for i in body["items"]} + + body = client.get(f"/api/elec/products/{fc}/recommendations").json() + assert body["other_variants"] == [] + assert fd in {i["product_id"] for i in body["items"]} diff --git a/backend/tests/test_mcp.py b/backend/tests/test_mcp.py index f281ca6..4a5654a 100644 --- a/backend/tests/test_mcp.py +++ b/backend/tests/test_mcp.py @@ -23,7 +23,8 @@ def test_only_read_only_catalogue_tools_are_exposed(): async with Client(mcp) as c: return {t.name: set(t.input_schema.get("properties", {})) for t in await c.list_tools()} tools = anyio.run(go) - assert set(tools) == {"list_categories", "search_products", "get_product", "price_history"} + assert set(tools) == {"list_categories", "search_products", "get_product", "price_history", + "recommend_products"} assert tools["search_products"] == {"query", "category", "brand", "max_price", "min_price", "limit"} # Nothing that can start a run, log in, or change data. assert not any(w in name for name in tools for w in ("admin", "run", "login", "probe", "review")) @@ -91,3 +92,11 @@ def test_unknown_product_is_a_tool_error(db): with pytest.raises(ToolError, match="not found"): _call("get_product", {"product_id": 999999}) + + +def test_recommend_products_rejects_an_unknown_kind(): + import pytest + from fastmcp.exceptions import ToolError + + with pytest.raises(ToolError, match="better_rated"): + _call("recommend_products", {"product_id": 1, "kind": "cheapest"})