Files
loyaly-catalogue/backend/tests/test_elec_reviews.py
sriram c7e4d59188 Electronics Catalog: API, MCP server, frontend and deployment
Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
2026-10-01 12:17:42 +05:30

145 lines
7.4 KiB
Python

"""Ratings and reviews: read only what a page or search result states, and pick
the review mix by the product's rating. Offline tests first; the database
tests are skipped when the local Postgres container is not running."""
from __future__ import annotations
import json
from decimal import Decimal
from app.electronics.extract.jsonld import extract_products
from app.electronics.extract.serp_parser import read_rating
from app.electronics.reviews import select_reviews, sentiment_for
from app.electronics.search.providers import SearchHit, pagemap_rating
# ---------------------------------------------------------------------------
# Search-result ratings
# ---------------------------------------------------------------------------
def test_read_rating_accepts_explicit_statements():
r = read_rating("Samsung Galaxy S24 5G ... 4.3 out of 5 stars 1,234 ratings. ₹74,999")
assert r.rating == Decimal("4.3") and r.review_count == 1234
assert read_rating("Rating: 4.1/5 based on reviews").rating == Decimal("4.1")
r = read_rating("4.4★ (12,345 ratings)")
assert r.rating == Decimal("4.4") and r.review_count == 12345
def test_read_rating_refuses_guesses():
assert read_rating("Galaxy S24 8GB 256GB ₹74,999").rating is None
assert read_rating("1/5 inch sensor, 50MP").rating is None # a fraction, not a rating
assert read_rating("5/5G phone").rating is None
assert read_rating("4.2 out of 5 ... 3.9 out of 5").rating is None # two products: ambiguous
assert read_rating("7 out of 5").rating is None
def test_pagemap_rating_and_cached_hits_without_rating():
got = pagemap_rating({"aggregaterating": [{"ratingvalue": "4.5", "reviewcount": "2,310", "bestrating": "5"}]})
assert got["rating"] == 4.5 and got["review_count"] == 2310
assert pagemap_rating({"aggregaterating": [{"ratingvalue": "9", "bestrating": "10"}]}) is None
# Search results cached before the rating field existed still load.
hit = SearchHit.from_dict({"url": "https://a.in/p", "title": "t", "snippet": "s", "provider": "ddg", "rank": 0})
assert hit.rating is None
# ---------------------------------------------------------------------------
# Page reviews (schema.org JSON-LD)
# ---------------------------------------------------------------------------
def test_jsonld_reviews_are_read_verbatim():
ld = {
"@context": "https://schema.org", "@type": "Product", "name": "Samsung Galaxy S24",
"aggregateRating": {"ratingValue": "4.4", "reviewCount": "120"},
"review": [
{"@type": "Review", "author": {"@type": "Person", "name": "Arun"}, "name": "Great phone",
"reviewBody": "Battery lasts all day.", "datePublished": "2026-05-01",
"reviewRating": {"ratingValue": "5", "bestRating": "5"}},
{"@type": "Review", "author": "Priya", "reviewBody": "Heats up while gaming.",
"reviewRating": {"ratingValue": "4", "bestRating": "10"}},
{"@type": "Review", "author": "No words", "reviewRating": {"ratingValue": "1"}},
],
}
html = f'<script type="application/ld+json">{json.dumps(ld)}</script>'
p = extract_products(html)[0]
assert p["rating"] == Decimal("4.4") and p["review_count"] == 120
assert [r["body"] for r in p["reviews"]] == ["Battery lasts all day.", "Heats up while gaming."]
assert p["reviews"][0]["author"] == "Arun" and p["reviews"][0]["title"] == "Great phone"
assert p["reviews"][1]["rating"] == Decimal("2.0") # 4 out of 10, rescaled
# ---------------------------------------------------------------------------
# Review mix
# ---------------------------------------------------------------------------
def _pool(pos: int, neu: int, neg: int) -> list:
out = []
for label, n, stars in (("p", pos, 5), ("u", neu, 3), ("n", neg, 1)):
out += [{"body": f"{label}{i}", "rating": stars} for i in range(n)]
return out
def _counts(picked: list) -> tuple:
return tuple(sum(1 for r in picked if r["sentiment"] == s) for s in ("positive", "neutral", "negative"))
def test_sentiment_is_the_reviewers_own_stars():
assert [sentiment_for(x) for x in (5, 4, 3.5, 3, 2.9, 1, None)] == [
"positive", "positive", "neutral", "neutral", "negative", "negative", None]
def test_high_rating_shows_mostly_positive():
assert _counts(select_reviews(4.7, _pool(20, 20, 20))) == (6, 3, 1)
def test_middling_rating_shows_mostly_neutral():
assert _counts(select_reviews(3.6, _pool(20, 20, 20))) == (3, 5, 2)
def test_low_rating_shows_mostly_negative():
assert _counts(select_reviews(2.5, _pool(20, 20, 20))) == (2, 2, 6)
def test_short_groups_hand_slots_on_and_nothing_is_padded():
picked = select_reviews(4.8, _pool(3, 20, 0))
assert len(picked) == 10 and _counts(picked) == (3, 7, 0)
assert len(select_reviews(4.8, _pool(1, 1, 1))) == 3
assert select_reviews(4.8, [{"body": "no stars", "rating": None}]) == []
# ---------------------------------------------------------------------------
# Database + API
# ---------------------------------------------------------------------------
def test_api_serves_ratings_reviews_and_out_of_stock_price(db, client):
from app.electronics.collector import Collector, RunOptions, RunStats
from app.electronics.db import repository as repo
from app.electronics.models import Listing
from app.electronics.normalise.title_parser import parse_title, variant_key
def _listing(site, sku, title, *, price, source_type="search_snippet", evidence=None):
p = parse_title(title, "mobiles")
l = Listing(site_domain=site, source_sku=sku, source_url=f"https://www.{site}/p/{sku}",
source_type=source_type, brand_slug=p.brand.brand_slug, category="mobiles", title=title,
evidence_text=evidence or f"{title} ₹{price}", confidence=0.5, parser="test",
model=p.model, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=price)
l.model_norm, l.variant_key = p.model_norm, variant_key(p, "mobiles")
return l
c = Collector.__new__(Collector)
c.opt = RunOptions(category="mobiles", brands=["samsung"])
c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats()
title = "Samsung Galaxy S24 5G (8GB RAM, 256GB)"
a = _listing("amazon.in", "B0CS5XW6TN", title, price=Decimal(74999))
a.in_stock, a.rating, a.review_count = False, Decimal("4.6"), 300
b = _listing("croma.com", "303838", title, price=Decimal(73999), source_type="scraped_page",
evidence='{"price": "73999"}')
b.in_stock, b.rating, b.review_count = False, Decimal("4.0"), 100
b.reviews = [{"author": "Arun", "rating": Decimal(5), "title": "Great", "body": "Battery lasts all day."},
{"author": "Priya", "rating": Decimal(2), "body": "Heats up."}]
c.store(a)
c.store(b)
repo.refresh_verification()
product = client.get("/api/elec/products", params={"category": "mobiles"}).json()["products"][0]
assert product["best_price"] == "73999.00" # every listing out of stock, price still shown
detail = client.get(f"/api/elec/products/{product['product_id']}").json()
assert detail["rating"]["value"] == 4.5 and detail["rating"]["count"] == 400
assert {s["site"] for s in detail["rating"]["sources"]} == {"Amazon.in", "Croma"}
assert [r["body"] for r in detail["reviews"]] == ["Battery lasts all day.", "Heats up."]
assert all(r["source_url"].startswith("https://") for r in detail["reviews"])