"""Ratings and reviews: read only what a page or search result states, and pick the review mix by the product's rating. Offline tests first; the database tests are skipped when the local Postgres container is not running.""" from __future__ import annotations import json from decimal import Decimal from app.electronics.extract.jsonld import extract_products from app.electronics.extract.serp_parser import read_rating from app.electronics.reviews import select_reviews, sentiment_for from app.electronics.search.providers import SearchHit, pagemap_rating # --------------------------------------------------------------------------- # Search-result ratings # --------------------------------------------------------------------------- def test_read_rating_accepts_explicit_statements(): r = read_rating("Samsung Galaxy S24 5G ... 4.3 out of 5 stars 1,234 ratings. ₹74,999") assert r.rating == Decimal("4.3") and r.review_count == 1234 assert read_rating("Rating: 4.1/5 based on reviews").rating == Decimal("4.1") r = read_rating("4.4★ (12,345 ratings)") assert r.rating == Decimal("4.4") and r.review_count == 12345 def test_read_rating_refuses_guesses(): assert read_rating("Galaxy S24 8GB 256GB ₹74,999").rating is None assert read_rating("1/5 inch sensor, 50MP").rating is None # a fraction, not a rating assert read_rating("5/5G phone").rating is None assert read_rating("4.2 out of 5 ... 3.9 out of 5").rating is None # two products: ambiguous assert read_rating("7 out of 5").rating is None def test_pagemap_rating_and_cached_hits_without_rating(): got = pagemap_rating({"aggregaterating": [{"ratingvalue": "4.5", "reviewcount": "2,310", "bestrating": "5"}]}) assert got["rating"] == 4.5 and got["review_count"] == 2310 assert pagemap_rating({"aggregaterating": [{"ratingvalue": "9", "bestrating": "10"}]}) is None # Search results cached before the rating field existed still load. hit = SearchHit.from_dict({"url": "https://a.in/p", "title": "t", "snippet": "s", "provider": "ddg", "rank": 0}) assert hit.rating is None # --------------------------------------------------------------------------- # Page reviews (schema.org JSON-LD) # --------------------------------------------------------------------------- def test_jsonld_reviews_are_read_verbatim(): ld = { "@context": "https://schema.org", "@type": "Product", "name": "Samsung Galaxy S24", "aggregateRating": {"ratingValue": "4.4", "reviewCount": "120"}, "review": [ {"@type": "Review", "author": {"@type": "Person", "name": "Arun"}, "name": "Great phone", "reviewBody": "Battery lasts all day.", "datePublished": "2026-05-01", "reviewRating": {"ratingValue": "5", "bestRating": "5"}}, {"@type": "Review", "author": "Priya", "reviewBody": "Heats up while gaming.", "reviewRating": {"ratingValue": "4", "bestRating": "10"}}, {"@type": "Review", "author": "No words", "reviewRating": {"ratingValue": "1"}}, ], } html = f'' p = extract_products(html)[0] assert p["rating"] == Decimal("4.4") and p["review_count"] == 120 assert [r["body"] for r in p["reviews"]] == ["Battery lasts all day.", "Heats up while gaming."] assert p["reviews"][0]["author"] == "Arun" and p["reviews"][0]["title"] == "Great phone" assert p["reviews"][1]["rating"] == Decimal("2.0") # 4 out of 10, rescaled # --------------------------------------------------------------------------- # Review mix # --------------------------------------------------------------------------- def _pool(pos: int, neu: int, neg: int) -> list: out = [] for label, n, stars in (("p", pos, 5), ("u", neu, 3), ("n", neg, 1)): out += [{"body": f"{label}{i}", "rating": stars} for i in range(n)] return out def _counts(picked: list) -> tuple: return tuple(sum(1 for r in picked if r["sentiment"] == s) for s in ("positive", "neutral", "negative")) def test_sentiment_is_the_reviewers_own_stars(): assert [sentiment_for(x) for x in (5, 4, 3.5, 3, 2.9, 1, None)] == [ "positive", "positive", "neutral", "neutral", "negative", "negative", None] def test_high_rating_shows_mostly_positive(): assert _counts(select_reviews(4.7, _pool(20, 20, 20))) == (6, 3, 1) def test_middling_rating_shows_mostly_neutral(): assert _counts(select_reviews(3.6, _pool(20, 20, 20))) == (3, 5, 2) def test_low_rating_shows_mostly_negative(): assert _counts(select_reviews(2.5, _pool(20, 20, 20))) == (2, 2, 6) def test_short_groups_hand_slots_on_and_nothing_is_padded(): picked = select_reviews(4.8, _pool(3, 20, 0)) assert len(picked) == 10 and _counts(picked) == (3, 7, 0) assert len(select_reviews(4.8, _pool(1, 1, 1))) == 3 assert select_reviews(4.8, [{"body": "no stars", "rating": None}]) == [] # --------------------------------------------------------------------------- # Database + API # --------------------------------------------------------------------------- def test_api_serves_ratings_reviews_and_out_of_stock_price(db, client): from app.electronics.collector import Collector, RunOptions, RunStats from app.electronics.db import repository as repo from app.electronics.models import Listing from app.electronics.normalise.title_parser import parse_title, variant_key def _listing(site, sku, title, *, price, source_type="search_snippet", evidence=None): p = parse_title(title, "mobiles") l = Listing(site_domain=site, source_sku=sku, source_url=f"https://www.{site}/p/{sku}", source_type=source_type, brand_slug=p.brand.brand_slug, category="mobiles", title=title, evidence_text=evidence or f"{title} ₹{price}", confidence=0.5, parser="test", model=p.model, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=price) l.model_norm, l.variant_key = p.model_norm, variant_key(p, "mobiles") return l c = Collector.__new__(Collector) c.opt = RunOptions(category="mobiles", brands=["samsung"]) c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats() title = "Samsung Galaxy S24 5G (8GB RAM, 256GB)" a = _listing("amazon.in", "B0CS5XW6TN", title, price=Decimal(74999)) a.in_stock, a.rating, a.review_count = False, Decimal("4.6"), 300 b = _listing("croma.com", "303838", title, price=Decimal(73999), source_type="scraped_page", evidence='{"price": "73999"}') b.in_stock, b.rating, b.review_count = False, Decimal("4.0"), 100 b.reviews = [{"author": "Arun", "rating": Decimal(5), "title": "Great", "body": "Battery lasts all day."}, {"author": "Priya", "rating": Decimal(2), "body": "Heats up."}] c.store(a) c.store(b) repo.refresh_verification() product = client.get("/api/elec/products", params={"category": "mobiles"}).json()["products"][0] assert product["best_price"] == "73999.00" # every listing out of stock, price still shown detail = client.get(f"/api/elec/products/{product['product_id']}").json() assert detail["rating"]["value"] == 4.5 and detail["rating"]["count"] == 400 assert {s["site"] for s in detail["rating"]["sources"]} == {"Amazon.in", "Croma"} assert [r["body"] for r in detail["reviews"]] == ["Battery lasts all day.", "Heats up."] assert all(r["source_url"].startswith("https://") for r in detail["reviews"])