"""Ratings and reviews: read only what a page or search result states, and pick the review mix by the product's rating. Offline tests first; the database tests are skipped when the local Postgres container is not running.""" from __future__ import annotations import json from decimal import Decimal from app.electronics.extract.jsonld import extract_products from app.electronics.extract.serp_parser import read_rating from app.electronics.reviews import select_reviews, sentiment_for from app.electronics.search.providers import SearchHit, pagemap_rating # --------------------------------------------------------------------------- # Search-result ratings # --------------------------------------------------------------------------- def test_read_rating_accepts_explicit_statements(): r = read_rating("Samsung Galaxy S24 5G ... 4.3 out of 5 stars 1,234 ratings. ₹74,999") assert r.rating == Decimal("4.3") and r.review_count == 1234 assert read_rating("Rating: 4.1/5 based on reviews").rating == Decimal("4.1") r = read_rating("4.4★ (12,345 ratings)") assert r.rating == Decimal("4.4") and r.review_count == 12345 def test_read_rating_refuses_guesses(): assert read_rating("Galaxy S24 8GB 256GB ₹74,999").rating is None assert read_rating("1/5 inch sensor, 50MP").rating is None # a fraction, not a rating assert read_rating("5/5G phone").rating is None assert read_rating("4.2 out of 5 ... 3.9 out of 5").rating is None # two products: ambiguous assert read_rating("7 out of 5").rating is None def test_pagemap_rating_and_cached_hits_without_rating(): got = pagemap_rating({"aggregaterating": [{"ratingvalue": "4.5", "reviewcount": "2,310", "bestrating": "5"}]}) assert got["rating"] == 4.5 and got["review_count"] == 2310 assert pagemap_rating({"aggregaterating": [{"ratingvalue": "9", "bestrating": "10"}]}) is None # Search results cached before the rating field existed still load. hit = SearchHit.from_dict({"url": "https://a.in/p", "title": "t", "snippet": "s", "provider": "ddg", "rank": 0}) assert hit.rating is None # --------------------------------------------------------------------------- # Page reviews (schema.org JSON-LD) # --------------------------------------------------------------------------- def test_jsonld_reviews_are_read_verbatim(): ld = { "@context": "https://schema.org", "@type": "Product", "name": "Samsung Galaxy S24", "aggregateRating": {"ratingValue": "4.4", "reviewCount": "120"}, "review": [ {"@type": "Review", "author": {"@type": "Person", "name": "Arun"}, "name": "Great phone", "reviewBody": "Battery lasts all day.", "datePublished": "2026-05-01", "reviewRating": {"ratingValue": "5", "bestRating": "5"}}, {"@type": "Review", "author": "Priya", "reviewBody": "Heats up while gaming.", "reviewRating": {"ratingValue": "4", "bestRating": "10"}}, {"@type": "Review", "author": "No words", "reviewRating": {"ratingValue": "1"}}, ], } html = f'' p = extract_products(html)[0] assert p["rating"] == Decimal("4.4") and p["review_count"] == 120 assert [r["body"] for r in p["reviews"]] == ["Battery lasts all day.", "Heats up while gaming."] assert p["reviews"][0]["author"] == "Arun" and p["reviews"][0]["title"] == "Great phone" assert p["reviews"][1]["rating"] == Decimal("2.0") # 4 out of 10, rescaled # --------------------------------------------------------------------------- # Review mix # --------------------------------------------------------------------------- def _pool(pos: int, neu: int, neg: int) -> list: out = [] for label, n, stars in (("p", pos, 5), ("u", neu, 3), ("n", neg, 1)): out += [{"body": f"{label}{i}", "rating": stars} for i in range(n)] return out def _counts(picked: list) -> tuple: return tuple(sum(1 for r in picked if r["sentiment"] == s) for s in ("positive", "neutral", "negative")) def test_sentiment_is_the_reviewers_own_stars(): assert [sentiment_for(x) for x in (5, 4, 3.5, 3, 2.9, 1, None)] == [ "positive", "positive", "neutral", "neutral", "negative", "negative", None] def test_high_rating_shows_mostly_positive(): assert _counts(select_reviews(4.7, _pool(20, 20, 20))) == (6, 3, 1) def test_middling_rating_shows_mostly_neutral(): assert _counts(select_reviews(3.6, _pool(20, 20, 20))) == (3, 5, 2) def test_low_rating_shows_mostly_negative(): assert _counts(select_reviews(2.5, _pool(20, 20, 20))) == (2, 2, 6) def test_short_groups_hand_slots_on_and_nothing_is_padded(): picked = select_reviews(4.8, _pool(3, 20, 0)) assert len(picked) == 10 and _counts(picked) == (3, 7, 0) assert len(select_reviews(4.8, _pool(1, 1, 1))) == 3 assert select_reviews(4.8, [{"body": "no stars", "rating": None}]) == [] # --------------------------------------------------------------------------- # Database + API # --------------------------------------------------------------------------- def test_api_serves_ratings_reviews_and_out_of_stock_price(db, client): from app.electronics.collector import Collector, RunOptions, RunStats from app.electronics.db import repository as repo from app.electronics.models import Listing from app.electronics.normalise.title_parser import parse_title, variant_key def _listing(site, sku, title, *, price, source_type="search_snippet", evidence=None): p = parse_title(title, "mobiles") l = Listing(site_domain=site, source_sku=sku, source_url=f"https://www.{site}/p/{sku}", source_type=source_type, brand_slug=p.brand.brand_slug, category="mobiles", title=title, evidence_text=evidence or f"{title} ₹{price}", confidence=0.5, parser="test", model=p.model, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=price) l.model_norm, l.variant_key = p.model_norm, variant_key(p, "mobiles") return l c = Collector.__new__(Collector) c.opt = RunOptions(category="mobiles", brands=["samsung"]) c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats() title = "Samsung Galaxy S24 5G (8GB RAM, 256GB)" a = _listing("amazon.in", "B0CS5XW6TN", title, price=Decimal(74999)) a.in_stock, a.rating, a.review_count = False, Decimal("4.6"), 300 b = _listing("croma.com", "303838", title, price=Decimal(73999), source_type="scraped_page", evidence='{"price": "73999"}') b.in_stock, b.rating, b.review_count = False, Decimal("4.0"), 100 b.reviews = [{"author": "Arun", "rating": Decimal(5), "title": "Great", "body": "Battery lasts all day."}, {"author": "Priya", "rating": Decimal(2), "body": "Heats up."}] c.store(a) c.store(b) repo.refresh_verification() product = client.get("/api/elec/products", params={"category": "mobiles"}).json()["products"][0] assert product["best_price"] == "73999.00" # every listing out of stock, price still shown detail = client.get(f"/api/elec/products/{product['product_id']}").json() assert detail["rating"]["value"] == 4.5 and detail["rating"]["count"] == 400 assert {s["site"] for s in detail["rating"]["sources"]} == {"Amazon.in", "Croma"} assert [r["body"] for r in detail["reviews"]] == ["Battery lasts all day.", "Heats up."] assert all(r["source_url"].startswith("https://") for r in detail["reviews"]) assert detail["rating"]["breakdown"] is None # no platform published one # --------------------------------------------------------------------------- # Free sources: brand-store JSON-LD, retailer-embedded ratings, Vijay Sales # --------------------------------------------------------------------------- def _ld(*nodes) -> str: return "".join(f'' for n in nodes) def _review(author, stars, body, date="2026-09-30T17:06:05.000+00:00"): return {"@type": "Review", "author": {"@type": "Person", "name": author}, "reviewBody": body, "datePublished": date, "reviewRating": {"@type": "Rating", "bestRating": 5, "ratingValue": stars}} def test_nameless_review_node_is_merged_by_id(): # The samsung.com/in shape: Bazaarvoice publishes ratings in a second, # nameless Product node that shares the real product's @id. pid = "https://www.samsung.com/in/smartphones/galaxy-a/galaxy-a56-5g-awesome-olive-256gb-sm-a566ezggins/" html = _ld( {"@context": "https://schema.org", "@type": "Product", "@id": pid, "name": "Galaxy A56 5G (8 GB Memory)", "sku": "SM-A566EZGG", "offers": {"@type": "Offer", "price": "48999", "priceCurrency": "INR"}}, {"@context": "https://schema.org", "@type": "Product", "@id": pid, "aggregateRating": {"@type": "AggregateRating", "ratingValue": "4.5", "ratingCount": 1224}, "review": [_review("Vedant", 5, "Great and useful AI features"), _review("Ravi", 1, "Heats up")]}, ) [p] = extract_products(html) assert p["name"] == "Galaxy A56 5G (8 GB Memory)" and p["price"] == Decimal("48999") assert p["rating"] == Decimal("4.5") and p["review_count"] == 1224 assert [(r["author"], r["rating"], r["body"]) for r in p["reviews"]] == [ ("Vedant", Decimal("5.0"), "Great and useful AI features"), ("Ravi", Decimal("1.0"), "Heats up")] def test_nameless_review_node_without_id_is_not_guessed_between_products(): lone = {"@type": "Product", "aggregateRating": {"ratingValue": "4.1", "ratingCount": 9}} [only] = extract_products(_ld({"@type": "Product", "name": "Galaxy A56"}, lone)) assert only["rating"] == Decimal("4.1") # one product on the page: unambiguous two = extract_products(_ld({"@type": "Product", "name": "Galaxy A56"}, {"@type": "Product", "name": "Galaxy A36"}, lone)) assert all(p["rating"] is None for p in two) # two products: never guess which def test_reliance_embedded_rating_is_read_from_the_pages_own_product_only(): from app.electronics.extract.embedded_ratings import embedded_rating def page(app, recommended=None): state = {"productDetailsPage": {"product": {"name": "Dell 15", "_custom_json": {"_app": app}}}, "recommendations": [{"_custom_json": {"_app": recommended}}] if recommended else []} return f"" url = "https://www.reliancedigital.in/product/dell-15-9991568" got = embedded_rating("reliancedigital.in", page( {"averageRating": 4, "ratingsCount": 3, "ratingsCountDetails": {"5": 1, "4": 1, "3": 1}}), url) assert got == {"rating": Decimal("4.00"), "review_count": 3, "breakdown": {"5": 1, "4": 1, "3": 1}} # "No ratings yet" is published as zeros; a recommended product's rating is not ours. zero = page({"averageRating": 0, "ratingsCount": 0}, recommended={"averageRating": 4.8, "ratingsCount": 90}) assert embedded_rating("reliancedigital.in", zero, url) is None def test_poorvika_rating_needs_the_pages_own_slug(): from app.electronics.extract.embedded_ratings import embedded_rating def page(slug, rating, count): data = {"props": {"pageProps": {"additionalData": {"slug": slug, "rating": rating, "ratingCount": count}}}} return f'' slug = "xiaomi-14-civi-5g-shadow-black-256gb-8gb-ram" url = f"https://www.poorvika.com/{slug}/p" assert embedded_rating("poorvika.com", page(slug, 5, 1), url)["rating"] == Decimal("5.00") assert embedded_rating("poorvika.com", page("some-other-phone", 5, 1), url) is None assert embedded_rating("poorvika.com", page(slug, None, 0), url) is None def test_vasanth_template_rating_is_never_read(): from app.electronics.extract.embedded_ratings import embedded_rating template = ('

3.3

(3 Reviews)' + _ld({"@type": "Product", "name": "HP 15", "aggregateRating": {"ratingValue": "3.3", "reviewCount": 3}})) assert embedded_rating("vasanthandco.in", template, "https://vasanthandco.in/product/174200102203/") is None def test_vijaysales_graphql_review_feed(): from app.electronics.extract import vijaysales_reviews as vs url = "https://www.vijaysales.com/p/239904/acer-aspire-lite-al15-41-laptop-amd-ryzen-7-16gb" assert vs.url_key(url) == "acer-aspire-lite-al15-41-laptop-amd-ryzen-7-16gb" assert "url_key" in vs.graphql_url(url) assert vs.url_key('https://x.in/p/a"b') is None # cannot break out of the query product_name = "Acer Aspire Lite AL15-41 Laptop (AMD Ryzen 7/ 16GB DDR4 RAM/ 512GB SSD)" body = json.dumps({"data": {"products": {"items": [ {"sku": "P1", "rating_summary": 0, "review_count": 0, "reviews": {"items": []}}, {"sku": "239904", "rating_summary": 90, "review_count": 2, "reviews": {"items": [ {"nickname": "Jay", "summary": product_name, "text": "All over performance is good", "average_rating": 100, "created_at": "2026-05-02 10:00:00"}, {"nickname": "guest user", "summary": "Okay", "text": "", "average_rating": 60}]}}]}}}) got = vs.parse(body) assert got["rating"] == Decimal("4.50") and got["review_count"] == 2 # Product-name "title" dropped; a review with no text is skipped. assert got["reviews"] == [{"author": "Jay", "rating": Decimal("5.00"), "title": None, "body": "All over performance is good", "review_date": "2026-05-02 10:00:00", "sentiment": "positive"}] assert vs.parse('{"data":{"products":{"items":[]}}}') is None def test_api_breakdown_only_when_a_platform_publishes_one(db, client): from app.electronics.collector import Collector, RunOptions, RunStats from app.electronics.db import repository as repo from app.electronics.models import Listing from app.electronics.normalise.title_parser import parse_title, variant_key def listing(site, sku, price, source_type="search_snippet"): title = "Samsung Galaxy S24 5G (8GB RAM, 256GB)" p = parse_title(title, "mobiles") l = Listing(site_domain=site, source_sku=sku, source_url=f"https://www.{site}/p/{sku}", source_type=source_type, brand_slug="samsung", category="mobiles", title=title, evidence_text=f"{title} ₹{price}", confidence=0.5, parser="test", model=p.model, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=Decimal(price)) l.model_norm, l.variant_key = p.model_norm, variant_key(p, "mobiles") return l c = Collector.__new__(Collector) c.opt = RunOptions(category="mobiles", brands=["samsung"]) c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats() b = listing("reliancedigital.in", "rd-1", 73999, "scraped_page") b.rating, b.review_count, b.rating_breakdown = Decimal("4.0"), 4, {"5": 2, "4": 1, "1": 1} c.store(listing("amazon.in", "B0CS5XW6TN", 74999)) c.store(b) repo.refresh_verification() pid = client.get("/api/elec/products", params={"category": "mobiles"}).json()["products"][0]["product_id"] rating = client.get(f"/api/elec/products/{pid}").json()["rating"] assert rating["breakdown"] == [ {"stars": 5, "count": 2, "percent": 50}, {"stars": 4, "count": 1, "percent": 25}, {"stars": 3, "count": 0, "percent": 0}, {"stars": 2, "count": 0, "percent": 0}, {"stars": 1, "count": 1, "percent": 25}] assert rating["sources"][0]["breakdown"] == {"5": 2, "4": 1, "1": 1} def test_brand_store_page_must_be_the_exact_model_and_variant(): from app.electronics.normalise.title_parser import parse_title from app.electronics.review_refresh import _same_variant a56 = {"model_norm": "galaxy a56", "ram_gb": Decimal("8"), "storage_gb": Decimal("256")} page = lambda t: parse_title(t, "mobiles", expected_brand="samsung") # noqa: E731 assert _same_variant(page("Samsung Galaxy A56 5G 8GB/256GB (Olive)"), a56) # A neighbouring model scores ~90 on fuzzy matching; it must still be refused. assert not _same_variant(page("Samsung Galaxy A57 5G 8GB/256GB (Navy)"), a56) assert not _same_variant(page("Samsung Galaxy A56 5G 12GB/256GB (Olive)"), a56) assert not _same_variant(page("Samsung Galaxy A56 5G 8GB/128GB (Blue)"), a56) s25 = {"model_norm": "galaxy s25", "ram_gb": Decimal("12"), "storage_gb": Decimal("256")} assert _same_variant(page("Samsung Galaxy S25 12GB/256GB (Icyblue)"), s25) assert not _same_variant(page("Samsung Galaxy S25 Ultra 12GB/256GB (Titanium Black)"), s25) assert not _same_variant(page("Samsung Galaxy S25+ 12GB/256GB (Navy)"), s25)