backend updates on recommendation system

This commit is contained in:
sriram
2026-10-07 11:03:42 +05:30
parent 82f5db1250
commit 71dbb2a6e9
6 changed files with 599 additions and 6 deletions

View File

@@ -7,12 +7,20 @@ Money is returned as a decimal string, never a float.
from __future__ import annotations
from decimal import Decimal
from typing import Any, Dict, List, Optional
from typing import Any, Dict, List, Literal, Optional
from fastapi import APIRouter, HTTPException, Query
from app.electronics.db.connection import connect
from app.electronics.db.repository import product_rating_and_reviews
from app.electronics import recommend as rec
from app.electronics.db.repository import (
other_variants,
product_rating_and_reviews,
rated_products,
rating_sources_for,
review_sentiment_counts,
similar_products,
)
from app.electronics.reviews import select_reviews
router = APIRouter(prefix="/elec", tags=["electronics"])
@@ -106,7 +114,13 @@ def products(
f"LIMIT %(limit)s OFFSET %(offset)s",
{**params, "limit": limit, "offset": offset},
).fetchall()
return {"total": total, "products": [_clean(r) for r in rows]}
sources = rating_sources_for(conn, [r["product_id"] for r in rows])
out = []
for r in rows:
rating = _overall_rating(sources[r["product_id"]])
out.append({**_clean(r), "rating": rating["value"] if rating else None,
"rating_count": rating["count"] if rating else None})
return {"total": total, "products": out}
@router.get("/products/{product_id}")
@@ -185,6 +199,63 @@ def _breakdown(sources: List[dict]) -> Optional[List[dict]]:
for k in ("5", "4", "3", "2", "1")]
@router.get("/products/{product_id}/recommendations")
def recommendations(
product_id: int,
kind: Literal["similar", "better_rated"] = Query("similar", alias="type"),
limit: int = Query(rec.MAX_ITEMS, ge=1, le=12),
tn_only: bool = False,
) -> dict:
"""Products to suggest under this one's ratings and reviews (see
app/electronics/recommend.py), plus the same model's other variants.
type=similar: closest specs; type=better_rated: rated higher, similar price."""
with connect() as conn:
target = conn.execute(
"SELECT product_id, best_price FROM elec.v_brand_catalog WHERE product_id = %s", (product_id,)
).fetchone()
if not target:
raise HTTPException(status_code=404, detail="Product not found or not verified")
if kind == "similar":
similar = similar_products(conn, product_id, band=rec.SIMILAR_PRICE_BAND, tn_only=tn_only)
rated = (rated_products(conn, product_id, band=rec.SIMILAR_PRICE_BAND, tn_only=tn_only)
if len(similar) < rec.MIN_MATCHES else [])
else:
similar = []
rated = rated_products(conn, product_id, band=rec.BETTER_PRICE_BAND, tn_only=tn_only)
ids = {product_id} | {c["product_id"] for c in similar + rated}
ratings = {pid: _overall_rating(src) for pid, src in rating_sources_for(conn, list(ids)).items()}
def with_rating(c: dict) -> dict:
r = ratings.get(c["product_id"])
return {**c, "best_price": _float(c.get("best_price")),
"rating": r["value"] if r else None, "rating_count": r["count"] if r else None}
if kind == "similar":
picked = rec.recommend(with_rating(dict(target)), [with_rating(c) for c in similar],
[with_rating(c) for c in rated], limit)
else:
sentiment = review_sentiment_counts(conn, [c["product_id"] for c in rated])
picked = rec.better_rated(
with_rating(dict(target)),
[{**with_rating(c), "sentiment": sentiment.get(c["product_id"])} for c in rated], limit)
cards = {r["product_id"]: r for r in conn.execute(
"SELECT product_id, brand, display_name, ram_gb, storage_gb, image_url, best_price, best_price_site "
"FROM elec.v_brand_catalog WHERE product_id = ANY(%s)", ([c["product_id"] for c in picked],)
)}
variants = other_variants(conn, product_id)
items = [
{**_clean(cards[c["product_id"]]), "rating": c["rating"], "rating_count": c["rating_count"],
"basis": c["basis"], "reason": c["reason"]}
for c in picked if c["product_id"] in cards
]
return {"product_id": product_id, "type": kind, "items": items,
"other_variants": [_clean(v) for v in variants]}
def _float(value: Any) -> Optional[float]:
return None if value is None else float(value)
@router.get("/products/{product_id}/price-history")
def price_history(product_id: int) -> List[dict]:
with connect() as conn:

View File

@@ -305,6 +305,96 @@ def product_rating_and_reviews(conn, product_id: int) -> Dict[str, Any]:
return {"sources": [dict(s) for s in sources], "reviews": [dict(r) for r in reviews]}
def rating_sources_for(conn, product_ids: List[int]) -> Dict[int, List[dict]]:
"""product_rating_and_reviews' per-platform ratings for many products at once."""
out: Dict[int, List[dict]] = {pid: [] for pid in product_ids}
rows = conn.execute(
"SELECT a.product_id, a.site, a.source_url, l.rating, l.review_count, l.rating_breakdown "
"FROM elec.v_product_availability a JOIN elec.source_listing l ON l.id = a.listing_id "
"WHERE a.product_id = ANY(%s) AND l.rating > 0 ORDER BY l.review_count DESC NULLS LAST, a.site",
(list(product_ids),),
).fetchall()
for r in rows:
out[r["product_id"]].append(dict(r))
return out
# p is a variant of t when the two differ only in RAM/storage: same brand and
# model and the same processor. One laptop model line ("HP 15") spans many
# CPUs, so laptops count as variants only when both state the same processor;
# phones state none, so for them the model alone decides. Written so it is
# never NULL: NOT NULL would drop the product from recommendations too.
_SAME_MODEL = """(p.brand_id = t.brand_id AND p.model_norm = t.model_norm
AND p.processor IS NOT DISTINCT FROM t.processor
AND (p.processor IS NOT NULL
OR t.category_id IS DISTINCT FROM (SELECT id FROM elec.category WHERE slug = 'laptops')))"""
# Products that may be recommended for target t: verified, same category, not
# t itself or another RAM/storage variant of the same model, with an in-stock
# (or unknown-stock) best price within +/- %(band)s of t's own (no limit when t
# has no price).
_RECOMMENDABLE = """
FROM elec.product t
JOIN elec.product p ON p.category_id = t.category_id AND p.id <> t.id
AND p.verification_status = 'verified'
AND NOT """ + _SAME_MODEL + """
JOIN elec.v_best_price bp ON bp.product_id = p.id AND bp.in_stock IS DISTINCT FROM FALSE
LEFT JOIN elec.v_best_price tp ON tp.product_id = t.id
WHERE t.id = %(pid)s
AND (tp.price IS NULL OR bp.price BETWEEN tp.price * (1 - %(band)s) AND tp.price * (1 + %(band)s))
"""
_TN_ONLY = (" AND EXISTS (SELECT 1 FROM elec.v_product_availability a"
" WHERE a.product_id = p.id AND a.site_region = 'TN')")
def similar_products(conn, product_id: int, *, band: float, limit: int = 30, tn_only: bool = False) -> List[dict]:
"""Recommendable products closest to this one by embedding (cosine), with
their best price. Empty when the product has no embedding yet.
An exact scan, not the HNSW index: the category/stock filters would make an
approximate index search drop matches, and a category is small enough."""
sql = ("SELECT p.id AS product_id, 1 - (p.embedding <=> t.embedding) AS similarity, bp.price AS best_price"
+ _RECOMMENDABLE + " AND t.embedding IS NOT NULL AND p.embedding IS NOT NULL"
+ (_TN_ONLY if tn_only else "")
+ " ORDER BY p.embedding <=> t.embedding LIMIT %(limit)s")
return [dict(r) for r in conn.execute(sql, {"pid": product_id, "band": band, "limit": limit})]
def rated_products(conn, product_id: int, *, band: float, tn_only: bool = False) -> List[dict]:
"""Recommendable products that at least one platform has rated."""
sql = ("SELECT p.id AS product_id, bp.price AS best_price" + _RECOMMENDABLE
+ " AND EXISTS (SELECT 1 FROM elec.v_product_availability a JOIN elec.source_listing l"
" ON l.id = a.listing_id WHERE a.product_id = p.id AND l.rating > 0)"
+ (_TN_ONLY if tn_only else ""))
return [dict(r) for r in conn.execute(sql, {"pid": product_id, "band": band})]
def review_sentiment_counts(conn, product_ids: List[int]) -> Dict[int, Dict[str, int]]:
"""How many stored reviews of each product are positive / neutral / negative."""
out: Dict[int, Dict[str, int]] = {pid: {} for pid in product_ids}
rows = conn.execute(
"SELECT a.product_id, r.sentiment, count(*)::int AS n "
"FROM elec.v_product_availability a JOIN elec.listing_review r ON r.listing_id = a.listing_id "
"WHERE a.product_id = ANY(%s) AND r.sentiment IS NOT NULL GROUP BY 1, 2",
(list(product_ids),),
).fetchall()
for r in rows:
out[r["product_id"]][r["sentiment"]] = r["n"]
return out
def other_variants(conn, product_id: int) -> List[dict]:
"""The same model's other verified RAM/storage variants (see _SAME_MODEL)."""
return [dict(r) for r in conn.execute(
"SELECT v.product_id, v.display_name, v.ram_gb, v.storage_gb, v.best_price "
"FROM elec.product t JOIN elec.product p ON p.category_id = t.category_id AND p.id <> t.id "
"AND " + _SAME_MODEL + " "
"JOIN elec.v_brand_catalog v ON v.product_id = p.id "
"WHERE t.id = %s ORDER BY v.ram_gb NULLS LAST, v.storage_gb NULLS LAST, v.display_name",
(product_id,),
)]
def listings_for_review_backfill(category: Optional[str] = None) -> List[dict]:
"""Page-read listings of verified products, for re-reading ratings/reviews."""
sql = (

View File

@@ -0,0 +1,144 @@
"""Which other products to suggest under a product's ratings and reviews
(docs/RECOMMENDATIONS.md): Phase 1 "Similar products" and Phase 2 "Better
rated alternatives".
This module only scores and orders candidates; the database queries that find
them (repository.similar_products / rated_products) and the endpoint live
elsewhere. Nothing here invents a number: a product with no published rating
is scored at the pool's average, and one with no price is never a candidate.
Hard price limit: every candidate's best price is within SIMILAR_PRICE_BAND
(similar) or BETTER_PRICE_BAND (better rated) of the product's own, applied in
the database query. A product with no price of its own gets no limit.
Similar products:
score = 0.60 x similarity + 0.25 x rating quality + 0.15 x price closeness
similarity cosine similarity of the two product embeddings (0..1)
rating quality Bayesian average / 5, so 5.0 from 3 ratings does not beat
4.4 from 2,000: each product's rating is pulled towards the
pool average as if PRIOR_WEIGHT extra ratings at that average
had been given
price closeness 1 at the same price, falling to 0 at twice (or zero) the price
When fewer than MIN_MATCHES similar products qualify, the list is filled with
the best-rated products of the same category (in the same price band).
Better rated alternatives: products rated higher than this one by at least
MIN_REVIEWS_BETTER people, best Bayesian rating first. When two round to the
same Bayesian rating, the one whose stored reviews are more positive (share
of positive minus share of negative) goes first.
"""
from __future__ import annotations
from typing import Any, Dict, List, Optional
WEIGHT_SIMILARITY, WEIGHT_RATING, WEIGHT_PRICE = 0.60, 0.25, 0.15
PRIOR_WEIGHT = 50
DEFAULT_PRIOR_MEAN = 4.0 # used only when no product in the pool has a rating
MAX_ITEMS = 6
MIN_MATCHES = 3
SIMILAR_PRICE_BAND = 0.30 # +/-30% of the product's best price
BETTER_PRICE_BAND = 0.20 # +/-20%
MIN_REVIEWS_BETTER = 5
def prior_mean(candidates: List[Dict[str, Any]]) -> float:
"""Average published rating across the pool."""
rated = [c["rating"] for c in candidates if c.get("rating") is not None]
return sum(rated) / len(rated) if rated else DEFAULT_PRIOR_MEAN
def bayesian_rating(rating: Optional[float], count: Optional[int], mean: float) -> float:
if rating is None:
return mean
n = max(int(count or 0), 1) # a rating with no stated count weighs as one
return (PRIOR_WEIGHT * mean + n * float(rating)) / (PRIOR_WEIGHT + n)
def price_closeness(price: Optional[float], target_price: Optional[float]) -> float:
if price is None or not target_price:
return 0.5 # unknown: neither helps nor hurts
return max(0.0, 1.0 - abs(float(price) - float(target_price)) / float(target_price))
def score(candidate: Dict[str, Any], target_price: Optional[float], mean: float) -> float:
quality = bayesian_rating(candidate.get("rating"), candidate.get("rating_count"), mean) / 5
return (WEIGHT_SIMILARITY * float(candidate.get("similarity") or 0.0)
+ WEIGHT_RATING * quality
+ WEIGHT_PRICE * price_closeness(candidate.get("best_price"), target_price))
def reason(candidate: Dict[str, Any], target_rating: Optional[float], basis: str) -> str:
label = "Similar specs" if basis == "similar" else "Top rated in this category"
rating = candidate.get("rating")
if rating is None:
return label
if target_rating is not None and rating > target_rating:
return f"{label} · {rating:.1f}★ vs {target_rating:.1f}★"
count = candidate.get("rating_count")
return f"{label} · {rating:.1f}★" + (f" ({count:,} rating{'' if count == 1 else 's'})" if count else "")
def recommend(
target: Dict[str, Any],
similar: List[Dict[str, Any]],
top_rated: List[Dict[str, Any]],
limit: int = MAX_ITEMS,
) -> List[Dict[str, Any]]:
"""Order `similar` by score; when fewer than MIN_MATCHES are found, fill up
to `limit` from `top_rated` (best Bayesian rating first).
Each candidate is a dict with product_id, best_price, rating, rating_count
and (for `similar`) similarity. `target` has best_price and rating.
Returns the chosen candidates with `reason` and `basis` added."""
mean = prior_mean(similar + top_rated)
target_price, target_rating = target.get("best_price"), target.get("rating")
ranked = sorted(similar, key=lambda c: score(c, target_price, mean), reverse=True)[:limit]
picked = [{**c, "basis": "similar"} for c in ranked]
if len(picked) < MIN_MATCHES:
seen = {c["product_id"] for c in picked}
fill = sorted((c for c in top_rated if c["product_id"] not in seen and c.get("rating") is not None),
key=lambda c: bayesian_rating(c["rating"], c.get("rating_count"), mean), reverse=True)
picked += [{**c, "basis": "top_rated"} for c in fill[: limit - len(picked)]]
for c in picked:
c["reason"] = reason(c, target_rating, c["basis"])
return picked
def sentiment_balance(sentiment: Optional[Dict[str, int]]) -> float:
"""Share of positive minus share of negative stored reviews (-1..1); 0 with none."""
if not sentiment:
return 0.0
total = sum(sentiment.values())
if not total:
return 0.0
return (sentiment.get("positive", 0) - sentiment.get("negative", 0)) / total
def better_rated(
target: Dict[str, Any],
rated: List[Dict[str, Any]],
limit: int = MAX_ITEMS,
) -> List[Dict[str, Any]]:
"""The products in `rated` rated higher than `target` (any rated product
when the target has no rating), each rated by at least MIN_REVIEWS_BETTER
people. Each candidate has rating, rating_count and optionally `sentiment`
({"positive": n, "neutral": n, "negative": n})."""
target_rating = target.get("rating")
mean = prior_mean(rated + [target])
keep = [c for c in rated
if c.get("rating") is not None and (c.get("rating_count") or 0) >= MIN_REVIEWS_BETTER
and (target_rating is None or c["rating"] > target_rating)]
keep.sort(key=lambda c: (round(bayesian_rating(c["rating"], c["rating_count"], mean), 1),
sentiment_balance(c.get("sentiment"))), reverse=True)
picked = []
for c in keep[:limit]:
vs = f" vs {target_rating:.1f}★" if target_rating is not None else ""
picked.append({**c, "basis": "better_rated",
"reason": f"{c['rating']:.1f}★{vs} · {c['rating_count']:,} ratings"})
return picked

View File

@@ -27,13 +27,15 @@ mcp = FastMCP(
"Every product is confirmed by real listings on at least two retail platforms; "
"prices, ratings and reviews come with the page they were read from. Prices are "
"rupee strings. Use search_products to find products, then get_product for "
"per-platform offers, specs, images, rating and reviews."
"per-platform offers, specs, images, rating and reviews, and recommend_products "
"for similar alternatives."
),
)
_SEARCH_FIELDS = (
"product_id", "brand", "category", "display_name", "ram_gb", "storage_gb",
"best_price", "best_price_site", "platform_count", "sold_by_tn_retailer", "image_url",
"rating", "rating_count",
)
@@ -71,7 +73,8 @@ async def search_products(
limit: Maximum products to return (1-100).
Returns the total match count and, per product: id, name, variant, best price (rupee
string) and the platform offering it, number of platforms, and an image URL (or null).
string) and the platform offering it, number of platforms, an image URL (or null), and the
overall rating and rating count (null when no platform publishes a rating).
"""
limit = max(1, min(int(limit), 100))
result = await _run(
@@ -115,6 +118,36 @@ async def get_product(product_id: int) -> Dict[str, Any]:
}
@mcp.tool
async def recommend_products(product_id: int, kind: str = "similar", limit: int = 6) -> Dict[str, Any]:
"""Alternatives to suggest for one product (by its product_id).
Args:
product_id: The product to find alternatives for.
kind: "similar" - closest specs, ranked by spec similarity, rating (weighted by how
many people rated it) and price closeness; when few exist, the best-rated in the
category fill the list. "better_rated" - products rated higher than this one by
at least 5 people.
limit: Maximum products to return (1-12).
Always same category, in stock, within a similar price (+/-30% for similar, +/-20% for
better_rated), with other variants of the same model left out. Each item has a short
reason (e.g. "Similar specs · 4.5★ vs 4.1★"). The same model's other RAM/storage
variants are listed separately under other_variants.
"""
if kind not in ("similar", "better_rated"):
raise ToolError('kind must be "similar" or "better_rated"')
d = await _run(elec.recommendations, int(product_id), kind, max(1, min(int(limit), 12)), False)
return {
"items": [
{k: i.get(k) for k in ("product_id", "brand", "display_name", "ram_gb", "storage_gb",
"best_price", "best_price_site", "rating", "rating_count", "reason")}
for i in d["items"]
],
"other_variants": d["other_variants"],
}
@mcp.tool
async def price_history(product_id: int) -> List[Dict[str, Any]]:
"""Every price observed for a product, per platform, oldest first (rupee strings, ISO times)."""

View File

@@ -0,0 +1,246 @@
"""Recommendations under a product's ratings and reviews (docs/RECOMMENDATIONS.md,
Phase 1). Offline scoring tests first; the database tests are skipped when the
local Postgres container is not running."""
from __future__ import annotations
from decimal import Decimal
import pytest
from app.electronics import recommend as rec
# ---------------------------------------------------------------------------
# Scoring
# ---------------------------------------------------------------------------
def test_bayesian_rating_trusts_many_ratings_over_few():
mean = 4.0
few_perfect = rec.bayesian_rating(5.0, 3, mean)
many_good = rec.bayesian_rating(4.4, 2000, mean)
assert many_good > few_perfect
assert rec.bayesian_rating(None, None, mean) == mean # unrated: the pool average
assert rec.bayesian_rating(5.0, None, mean) == rec.bayesian_rating(5.0, 1, mean)
def test_price_closeness():
assert rec.price_closeness(20000, 20000) == 1.0
assert rec.price_closeness(15000, 20000) == pytest.approx(0.75)
assert rec.price_closeness(45000, 20000) == 0.0 # never negative
assert rec.price_closeness(20000, None) == 0.5 # unknown target price
def test_similarity_dominates_but_rating_and_price_count():
target = {"best_price": 20000.0, "rating": 4.1}
close = {"product_id": 1, "similarity": 0.95, "best_price": 21000.0, "rating": 4.0, "rating_count": 500}
far = {"product_id": 2, "similarity": 0.60, "best_price": 20000.0, "rating": 4.8, "rating_count": 5000}
tie_better_rated = {"product_id": 3, "similarity": 0.95, "best_price": 21000.0, "rating": 4.6, "rating_count": 3000}
out = rec.recommend(target, [far, close, tie_better_rated], [])
assert [c["product_id"] for c in out] == [3, 1, 2]
assert out[0]["reason"] == "Similar specs · 4.6★ vs 4.1★"
assert out[1]["reason"] == "Similar specs · 4.0★ (500 ratings)"
assert all(c["basis"] == "similar" for c in out)
assert rec.reason({"rating": 4.0, "rating_count": 1}, None, "similar") == "Similar specs · 4.0★ (1 rating)"
def test_limit_and_top_rated_fill_when_few_similar():
target = {"best_price": 20000.0, "rating": None}
similar = [{"product_id": 1, "similarity": 0.9, "best_price": 20000.0, "rating": None, "rating_count": None}]
rated = [
{"product_id": 1, "best_price": 20000.0, "rating": 4.9, "rating_count": 9000}, # already picked
{"product_id": 2, "best_price": 30000.0, "rating": 4.2, "rating_count": 900},
{"product_id": 3, "best_price": 25000.0, "rating": 4.7, "rating_count": 1200},
{"product_id": 4, "best_price": 25000.0, "rating": None, "rating_count": None}, # unrated: never a "top rated"
]
out = rec.recommend(target, similar, rated, limit=3)
assert [c["product_id"] for c in out] == [1, 3, 2]
assert [c["basis"] for c in out] == ["similar", "top_rated", "top_rated"]
assert out[0]["reason"] == "Similar specs"
assert out[1]["reason"] == "Top rated in this category · 4.7★ (1,200 ratings)"
def test_enough_similar_means_no_fill():
similar = [{"product_id": i, "similarity": 0.5, "best_price": 1.0, "rating": None, "rating_count": None}
for i in range(rec.MIN_MATCHES)]
rated = [{"product_id": 99, "best_price": 1.0, "rating": 5.0, "rating_count": 10}]
out = rec.recommend({"best_price": 1.0, "rating": None}, similar, rated)
assert 99 not in {c["product_id"] for c in out}
def test_better_rated_rules():
target = {"rating": 4.1}
rated = [
{"product_id": 1, "rating": 4.5, "rating_count": 4000},
{"product_id": 2, "rating": 4.9, "rating_count": 4}, # too few ratings
{"product_id": 3, "rating": 4.1, "rating_count": 9000}, # not higher
{"product_id": 4, "rating": 4.6, "rating_count": 3000},
{"product_id": 5, "rating": None, "rating_count": None},
]
out = rec.better_rated(target, rated)
assert [c["product_id"] for c in out] == [4, 1]
assert out[0]["reason"] == "4.6★ vs 4.1★ · 3,000 ratings"
# Unrated product: any well-rated product counts as better rated.
out = rec.better_rated({"rating": None}, rated)
assert [c["product_id"] for c in out] == [4, 1, 3]
assert out[0]["reason"] == "4.6★ · 3,000 ratings"
def test_better_rated_tie_goes_to_more_positive_reviews():
rated = [
{"product_id": 1, "rating": 4.5, "rating_count": 1000, "sentiment": {"positive": 2, "negative": 8}},
{"product_id": 2, "rating": 4.5, "rating_count": 1000, "sentiment": {"positive": 8, "negative": 2}},
{"product_id": 3, "rating": 4.5, "rating_count": 1000}, # no stored reviews: neutral
]
assert [c["product_id"] for c in rec.better_rated({"rating": 4.0}, rated)] == [2, 3, 1]
assert rec.sentiment_balance({"positive": 3, "neutral": 1}) == 0.75
assert rec.sentiment_balance({}) == 0.0
# ---------------------------------------------------------------------------
# API (database)
# ---------------------------------------------------------------------------
def _vector(*head: float) -> list:
v = list(head) + [0.0] * (384 - len(head))
norm = sum(x * x for x in v) ** 0.5
return [x / norm for x in v]
def _seed():
"""Samsung phones on two sites each, plus embeddings:
S24 8/256 (target, 4.1★, ₹74,999), S24 8/128 (its variant), S23 (closest,
4.5★ from 4,000), S22 (further, 4.3★ from only 4), A55 (close in specs but
₹39,999 - outside both price bands), and an out-of-stock Z Flip6."""
from app.electronics.collector import Collector, RunOptions, RunStats
from app.electronics.db import repository as repo
from app.electronics.db.connection import connect
from app.electronics.models import Listing
from app.electronics.normalise.title_parser import parse_title, variant_key
c = Collector.__new__(Collector)
c.opt = RunOptions(category="mobiles", brands=["samsung"])
c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats()
def store(title, sku, price, *, rating=None, count=None):
p = parse_title(title, "mobiles")
for site in ("amazon.in", "croma.com"):
l = Listing(site_domain=site, source_sku=f"{site}-{sku}", source_url=f"https://www.{site}/p/{sku}",
source_type="search_snippet", brand_slug="samsung", category="mobiles", title=title,
evidence_text=f"{title} ₹{price}", confidence=0.5, parser="test",
model=p.model, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=Decimal(price))
l.model_norm, l.variant_key = p.model_norm, variant_key(p, "mobiles")
l.rating, l.review_count = rating, count
c.store(l)
store("Samsung Galaxy S24 5G (8GB RAM, 256GB)", "s24-256", 74999, rating=Decimal("4.1"), count=300)
store("Samsung Galaxy S24 5G (8GB RAM, 128GB)", "s24-128", 69999)
store("Samsung Galaxy S23 5G (8GB RAM, 256GB)", "s23", 64999, rating=Decimal("4.5"), count=2000)
store("Samsung Galaxy S22 5G (8GB RAM, 256GB)", "s22", 79999, rating=Decimal("4.3"), count=2)
store("Samsung Galaxy A55 5G (8GB RAM, 128GB)", "a55", 39999, rating=Decimal("4.2"), count=800)
store("Samsung Galaxy Z Flip6 5G (12GB RAM, 256GB)", "flip6", 59999)
repo.refresh_verification()
with connect() as conn: # every phone has its own price
by_price = {int(r["best_price"]): r["product_id"] for r in conn.execute(
"SELECT product_id, best_price FROM elec.v_brand_catalog")}
pids = {"s24": by_price[74999], "s24_128": by_price[69999], "s23": by_price[64999],
"s22": by_price[79999], "a55": by_price[39999], "flip6": by_price[59999]}
with connect(autocommit=True) as conn: # the Flip6 sells out after it was verified
conn.execute("UPDATE elec.source_listing SET in_stock = FALSE WHERE source_sku LIKE '%%-flip6'")
vectors = {"s24": _vector(1, 0), "s24_128": _vector(1, 0), "s23": _vector(1, 0.2),
"s22": _vector(1, 1), "a55": _vector(1, 0.1), "flip6": _vector(1, 0.1)}
for key, pid in pids.items():
repo.set_embedding(pid, vectors[key])
return pids
def test_api_recommends_similar_in_stock_products_without_variants(db, client):
pids = _seed()
body = client.get(f"/api/elec/products/{pids['s24']}/recommendations").json()
got = [i["product_id"] for i in body["items"]]
# Closest first; the variant, the out-of-stock Flip6 and the A55 (outside
# the price limit, though close in specs) are left out.
assert got == [pids["s23"], pids["s22"]]
s23 = body["items"][0]
assert s23["rating"] == 4.5 and s23["rating_count"] == 4000 # 2,000 on each of two sites
assert s23["reason"] == "Similar specs · 4.5★ vs 4.1★"
assert s23["best_price"] == "64999.00"
assert [v["product_id"] for v in body["other_variants"]] == [pids["s24_128"]]
def test_api_recommendations_unknown_product_and_bad_type(db, client):
assert client.get("/api/elec/products/999999/recommendations").status_code == 404
pids = _seed()
assert client.get(f"/api/elec/products/{pids['s24']}/recommendations",
params={"type": "cheapest"}).status_code == 422
def test_api_falls_back_to_top_rated_without_embeddings(db, client):
from app.electronics.db.connection import connect
pids = _seed()
with connect(autocommit=True) as conn:
conn.execute("UPDATE elec.product SET embedding = NULL")
items = client.get(f"/api/elec/products/{pids['s24']}/recommendations").json()["items"]
assert [i["product_id"] for i in items] == [pids["s23"], pids["s22"]]
assert all(i["basis"] == "top_rated" for i in items)
def test_api_better_rated_needs_higher_rating_enough_reviews_and_close_price(db, client):
pids = _seed()
body = client.get(f"/api/elec/products/{pids['s24']}/recommendations", params={"type": "better_rated"}).json()
# S22 is rated higher but by only 4 people; A55 is outside +/-20% of the price.
assert body["type"] == "better_rated"
assert [i["product_id"] for i in body["items"]] == [pids["s23"]]
assert body["items"][0]["reason"] == "4.5★ vs 4.1★ · 4,000 ratings"
assert body["items"][0]["basis"] == "better_rated"
def test_product_list_carries_the_overall_rating_for_card_badges(db, client):
pids = _seed()
products = {p["product_id"]: p for p in
client.get("/api/elec/products", params={"category": "mobiles"}).json()["products"]}
assert (products[pids["s23"]]["rating"], products[pids["s23"]]["rating_count"]) == (4.5, 4000)
assert (products[pids["s24_128"]]["rating"], products[pids["s24_128"]]["rating_count"]) == (None, None)
def test_laptop_variants_need_the_same_processor(db, client):
"""One laptop line ("HP 15") spans many CPUs: only another RAM/storage of the
same processor is a variant. Other CPUs, and part-number-only listings that
state no processor, are separate products that can be recommended."""
from app.electronics.collector import Collector, RunOptions, RunStats
from app.electronics.db import repository as repo
from app.electronics.db.connection import connect
from app.electronics.models import Listing
from app.electronics.normalise.title_parser import parse_title, variant_key
c = Collector.__new__(Collector)
c.opt = RunOptions(category="laptops", brands=["hp"])
c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats()
for title, price in (("HP 15 Laptop AMD Ryzen 3 7320U (8GB RAM, 512GB SSD)", 40000),
("HP 15 Laptop AMD Ryzen 3 7320U (16GB RAM, 512GB SSD)", 45000),
("HP 15 Laptop AMD Ryzen 5 7520U (8GB RAM, 512GB SSD)", 47000),
("HP 15 Laptop 15-FC0805AU (8GB RAM, 512GB SSD)", 41000),
("HP 15 Laptop 15-FD0682TU (16GB RAM, 512GB SSD)", 42000)):
p = parse_title(title, "laptops")
for site in ("amazon.in", "croma.com"):
l = Listing(site_domain=site, source_sku=f"{site}-{price}", source_url=f"https://www.{site}/p/{price}",
source_type="search_snippet", brand_slug="hp", category="laptops", title=title,
evidence_text=f"{title} ₹{price}", confidence=0.5, parser="test", model=p.model,
model_number=p.mpn, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=Decimal(price))
l.model_norm, l.variant_key, l.processor = p.model_norm, variant_key(p, "laptops"), p.processor
c.store(l)
repo.refresh_verification()
with connect() as conn:
by_price = {int(r["best_price"]): r["product_id"] for r in conn.execute(
"SELECT product_id, best_price FROM elec.v_brand_catalog")}
r3_8, r3_16, r5, fc, fd = (by_price[n] for n in (40000, 45000, 47000, 41000, 42000))
for i, pid in enumerate((r3_8, r3_16, r5, fc, fd)):
repo.set_embedding(pid, _vector(1, 0.1 * i))
body = client.get(f"/api/elec/products/{r3_8}/recommendations").json()
assert [v["product_id"] for v in body["other_variants"]] == [r3_16]
assert {r5, fc, fd} <= {i["product_id"] for i in body["items"]}
assert r3_16 not in {i["product_id"] for i in body["items"]}
body = client.get(f"/api/elec/products/{fc}/recommendations").json()
assert body["other_variants"] == []
assert fd in {i["product_id"] for i in body["items"]}

View File

@@ -23,7 +23,8 @@ def test_only_read_only_catalogue_tools_are_exposed():
async with Client(mcp) as c:
return {t.name: set(t.input_schema.get("properties", {})) for t in await c.list_tools()}
tools = anyio.run(go)
assert set(tools) == {"list_categories", "search_products", "get_product", "price_history"}
assert set(tools) == {"list_categories", "search_products", "get_product", "price_history",
"recommend_products"}
assert tools["search_products"] == {"query", "category", "brand", "max_price", "min_price", "limit"}
# Nothing that can start a run, log in, or change data.
assert not any(w in name for name in tools for w in ("admin", "run", "login", "probe", "review"))
@@ -91,3 +92,11 @@ def test_unknown_product_is_a_tool_error(db):
with pytest.raises(ToolError, match="not found"):
_call("get_product", {"product_id": 999999})
def test_recommend_products_rejects_an_unknown_kind():
import pytest
from fastmcp.exceptions import ToolError
with pytest.raises(ToolError, match="better_rated"):
_call("recommend_products", {"product_id": 1, "kind": "cheapest"})