Files
loyaly-catalogue/backend/tests/test_elec_reviews.py
2026-10-05 11:21:39 +05:30

297 lines
16 KiB
Python

"""Ratings and reviews: read only what a page or search result states, and pick
the review mix by the product's rating. Offline tests first; the database
tests are skipped when the local Postgres container is not running."""
from __future__ import annotations
import json
from decimal import Decimal
from app.electronics.extract.jsonld import extract_products
from app.electronics.extract.serp_parser import read_rating
from app.electronics.reviews import select_reviews, sentiment_for
from app.electronics.search.providers import SearchHit, pagemap_rating
# ---------------------------------------------------------------------------
# Search-result ratings
# ---------------------------------------------------------------------------
def test_read_rating_accepts_explicit_statements():
r = read_rating("Samsung Galaxy S24 5G ... 4.3 out of 5 stars 1,234 ratings. ₹74,999")
assert r.rating == Decimal("4.3") and r.review_count == 1234
assert read_rating("Rating: 4.1/5 based on reviews").rating == Decimal("4.1")
r = read_rating("4.4★ (12,345 ratings)")
assert r.rating == Decimal("4.4") and r.review_count == 12345
def test_read_rating_refuses_guesses():
assert read_rating("Galaxy S24 8GB 256GB ₹74,999").rating is None
assert read_rating("1/5 inch sensor, 50MP").rating is None # a fraction, not a rating
assert read_rating("5/5G phone").rating is None
assert read_rating("4.2 out of 5 ... 3.9 out of 5").rating is None # two products: ambiguous
assert read_rating("7 out of 5").rating is None
def test_pagemap_rating_and_cached_hits_without_rating():
got = pagemap_rating({"aggregaterating": [{"ratingvalue": "4.5", "reviewcount": "2,310", "bestrating": "5"}]})
assert got["rating"] == 4.5 and got["review_count"] == 2310
assert pagemap_rating({"aggregaterating": [{"ratingvalue": "9", "bestrating": "10"}]}) is None
# Search results cached before the rating field existed still load.
hit = SearchHit.from_dict({"url": "https://a.in/p", "title": "t", "snippet": "s", "provider": "ddg", "rank": 0})
assert hit.rating is None
# ---------------------------------------------------------------------------
# Page reviews (schema.org JSON-LD)
# ---------------------------------------------------------------------------
def test_jsonld_reviews_are_read_verbatim():
ld = {
"@context": "https://schema.org", "@type": "Product", "name": "Samsung Galaxy S24",
"aggregateRating": {"ratingValue": "4.4", "reviewCount": "120"},
"review": [
{"@type": "Review", "author": {"@type": "Person", "name": "Arun"}, "name": "Great phone",
"reviewBody": "Battery lasts all day.", "datePublished": "2026-05-01",
"reviewRating": {"ratingValue": "5", "bestRating": "5"}},
{"@type": "Review", "author": "Priya", "reviewBody": "Heats up while gaming.",
"reviewRating": {"ratingValue": "4", "bestRating": "10"}},
{"@type": "Review", "author": "No words", "reviewRating": {"ratingValue": "1"}},
],
}
html = f'<script type="application/ld+json">{json.dumps(ld)}</script>'
p = extract_products(html)[0]
assert p["rating"] == Decimal("4.4") and p["review_count"] == 120
assert [r["body"] for r in p["reviews"]] == ["Battery lasts all day.", "Heats up while gaming."]
assert p["reviews"][0]["author"] == "Arun" and p["reviews"][0]["title"] == "Great phone"
assert p["reviews"][1]["rating"] == Decimal("2.0") # 4 out of 10, rescaled
# ---------------------------------------------------------------------------
# Review mix
# ---------------------------------------------------------------------------
def _pool(pos: int, neu: int, neg: int) -> list:
out = []
for label, n, stars in (("p", pos, 5), ("u", neu, 3), ("n", neg, 1)):
out += [{"body": f"{label}{i}", "rating": stars} for i in range(n)]
return out
def _counts(picked: list) -> tuple:
return tuple(sum(1 for r in picked if r["sentiment"] == s) for s in ("positive", "neutral", "negative"))
def test_sentiment_is_the_reviewers_own_stars():
assert [sentiment_for(x) for x in (5, 4, 3.5, 3, 2.9, 1, None)] == [
"positive", "positive", "neutral", "neutral", "negative", "negative", None]
def test_high_rating_shows_mostly_positive():
assert _counts(select_reviews(4.7, _pool(20, 20, 20))) == (6, 3, 1)
def test_middling_rating_shows_mostly_neutral():
assert _counts(select_reviews(3.6, _pool(20, 20, 20))) == (3, 5, 2)
def test_low_rating_shows_mostly_negative():
assert _counts(select_reviews(2.5, _pool(20, 20, 20))) == (2, 2, 6)
def test_short_groups_hand_slots_on_and_nothing_is_padded():
picked = select_reviews(4.8, _pool(3, 20, 0))
assert len(picked) == 10 and _counts(picked) == (3, 7, 0)
assert len(select_reviews(4.8, _pool(1, 1, 1))) == 3
assert select_reviews(4.8, [{"body": "no stars", "rating": None}]) == []
# ---------------------------------------------------------------------------
# Database + API
# ---------------------------------------------------------------------------
def test_api_serves_ratings_reviews_and_out_of_stock_price(db, client):
from app.electronics.collector import Collector, RunOptions, RunStats
from app.electronics.db import repository as repo
from app.electronics.models import Listing
from app.electronics.normalise.title_parser import parse_title, variant_key
def _listing(site, sku, title, *, price, source_type="search_snippet", evidence=None):
p = parse_title(title, "mobiles")
l = Listing(site_domain=site, source_sku=sku, source_url=f"https://www.{site}/p/{sku}",
source_type=source_type, brand_slug=p.brand.brand_slug, category="mobiles", title=title,
evidence_text=evidence or f"{title} ₹{price}", confidence=0.5, parser="test",
model=p.model, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=price)
l.model_norm, l.variant_key = p.model_norm, variant_key(p, "mobiles")
return l
c = Collector.__new__(Collector)
c.opt = RunOptions(category="mobiles", brands=["samsung"])
c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats()
title = "Samsung Galaxy S24 5G (8GB RAM, 256GB)"
a = _listing("amazon.in", "B0CS5XW6TN", title, price=Decimal(74999))
a.in_stock, a.rating, a.review_count = False, Decimal("4.6"), 300
b = _listing("croma.com", "303838", title, price=Decimal(73999), source_type="scraped_page",
evidence='{"price": "73999"}')
b.in_stock, b.rating, b.review_count = False, Decimal("4.0"), 100
b.reviews = [{"author": "Arun", "rating": Decimal(5), "title": "Great", "body": "Battery lasts all day."},
{"author": "Priya", "rating": Decimal(2), "body": "Heats up."}]
c.store(a)
c.store(b)
repo.refresh_verification()
product = client.get("/api/elec/products", params={"category": "mobiles"}).json()["products"][0]
assert product["best_price"] == "73999.00" # every listing out of stock, price still shown
detail = client.get(f"/api/elec/products/{product['product_id']}").json()
assert detail["rating"]["value"] == 4.5 and detail["rating"]["count"] == 400
assert {s["site"] for s in detail["rating"]["sources"]} == {"Amazon.in", "Croma"}
assert [r["body"] for r in detail["reviews"]] == ["Battery lasts all day.", "Heats up."]
assert all(r["source_url"].startswith("https://") for r in detail["reviews"])
assert detail["rating"]["breakdown"] is None # no platform published one
# ---------------------------------------------------------------------------
# Free sources: brand-store JSON-LD, retailer-embedded ratings, Vijay Sales
# ---------------------------------------------------------------------------
def _ld(*nodes) -> str:
return "".join(f'<script type="application/ld+json">{json.dumps(n)}</script>' for n in nodes)
def _review(author, stars, body, date="2026-09-30T17:06:05.000+00:00"):
return {"@type": "Review", "author": {"@type": "Person", "name": author}, "reviewBody": body,
"datePublished": date, "reviewRating": {"@type": "Rating", "bestRating": 5, "ratingValue": stars}}
def test_nameless_review_node_is_merged_by_id():
# The samsung.com/in shape: Bazaarvoice publishes ratings in a second,
# nameless Product node that shares the real product's @id.
pid = "https://www.samsung.com/in/smartphones/galaxy-a/galaxy-a56-5g-awesome-olive-256gb-sm-a566ezggins/"
html = _ld(
{"@context": "https://schema.org", "@type": "Product", "@id": pid, "name": "Galaxy A56 5G (8 GB Memory)",
"sku": "SM-A566EZGG", "offers": {"@type": "Offer", "price": "48999", "priceCurrency": "INR"}},
{"@context": "https://schema.org", "@type": "Product", "@id": pid,
"aggregateRating": {"@type": "AggregateRating", "ratingValue": "4.5", "ratingCount": 1224},
"review": [_review("Vedant", 5, "Great and useful AI features"), _review("Ravi", 1, "Heats up")]},
)
[p] = extract_products(html)
assert p["name"] == "Galaxy A56 5G (8 GB Memory)" and p["price"] == Decimal("48999")
assert p["rating"] == Decimal("4.5") and p["review_count"] == 1224
assert [(r["author"], r["rating"], r["body"]) for r in p["reviews"]] == [
("Vedant", Decimal("5.0"), "Great and useful AI features"), ("Ravi", Decimal("1.0"), "Heats up")]
def test_nameless_review_node_without_id_is_not_guessed_between_products():
lone = {"@type": "Product", "aggregateRating": {"ratingValue": "4.1", "ratingCount": 9}}
[only] = extract_products(_ld({"@type": "Product", "name": "Galaxy A56"}, lone))
assert only["rating"] == Decimal("4.1") # one product on the page: unambiguous
two = extract_products(_ld({"@type": "Product", "name": "Galaxy A56"},
{"@type": "Product", "name": "Galaxy A36"}, lone))
assert all(p["rating"] is None for p in two) # two products: never guess which
def test_reliance_embedded_rating_is_read_from_the_pages_own_product_only():
from app.electronics.extract.embedded_ratings import embedded_rating
def page(app, recommended=None):
state = {"productDetailsPage": {"product": {"name": "Dell 15", "_custom_json": {"_app": app}}},
"recommendations": [{"_custom_json": {"_app": recommended}}] if recommended else []}
return f"<script>window.__INITIAL_STATE__={json.dumps(state)};window.x=1</script>"
url = "https://www.reliancedigital.in/product/dell-15-9991568"
got = embedded_rating("reliancedigital.in", page(
{"averageRating": 4, "ratingsCount": 3, "ratingsCountDetails": {"5": 1, "4": 1, "3": 1}}), url)
assert got == {"rating": Decimal("4.00"), "review_count": 3, "breakdown": {"5": 1, "4": 1, "3": 1}}
# "No ratings yet" is published as zeros; a recommended product's rating is not ours.
zero = page({"averageRating": 0, "ratingsCount": 0}, recommended={"averageRating": 4.8, "ratingsCount": 90})
assert embedded_rating("reliancedigital.in", zero, url) is None
def test_poorvika_rating_needs_the_pages_own_slug():
from app.electronics.extract.embedded_ratings import embedded_rating
def page(slug, rating, count):
data = {"props": {"pageProps": {"additionalData": {"slug": slug, "rating": rating, "ratingCount": count}}}}
return f'<script id="__NEXT_DATA__" type="application/json">{json.dumps(data)}</script>'
slug = "xiaomi-14-civi-5g-shadow-black-256gb-8gb-ram"
url = f"https://www.poorvika.com/{slug}/p"
assert embedded_rating("poorvika.com", page(slug, 5, 1), url)["rating"] == Decimal("5.00")
assert embedded_rating("poorvika.com", page("some-other-phone", 5, 1), url) is None
assert embedded_rating("poorvika.com", page(slug, None, 0), url) is None
def test_vasanth_template_rating_is_never_read():
from app.electronics.extract.embedded_ratings import embedded_rating
template = ('<h4 class="avg-mark">3.3</h4><a class="rating-reviews">(3 Reviews)</a>'
+ _ld({"@type": "Product", "name": "HP 15", "aggregateRating": {"ratingValue": "3.3", "reviewCount": 3}}))
assert embedded_rating("vasanthandco.in", template, "https://vasanthandco.in/product/174200102203/") is None
def test_vijaysales_graphql_review_feed():
from app.electronics.extract import vijaysales_reviews as vs
url = "https://www.vijaysales.com/p/239904/acer-aspire-lite-al15-41-laptop-amd-ryzen-7-16gb"
assert vs.url_key(url) == "acer-aspire-lite-al15-41-laptop-amd-ryzen-7-16gb"
assert "url_key" in vs.graphql_url(url)
assert vs.url_key('https://x.in/p/a"b') is None # cannot break out of the query
product_name = "Acer Aspire Lite AL15-41 Laptop (AMD Ryzen 7/ 16GB DDR4 RAM/ 512GB SSD)"
body = json.dumps({"data": {"products": {"items": [
{"sku": "P1", "rating_summary": 0, "review_count": 0, "reviews": {"items": []}},
{"sku": "239904", "rating_summary": 90, "review_count": 2, "reviews": {"items": [
{"nickname": "Jay", "summary": product_name, "text": "All over performance is good",
"average_rating": 100, "created_at": "2026-05-02 10:00:00"},
{"nickname": "guest user", "summary": "Okay", "text": "", "average_rating": 60}]}}]}}})
got = vs.parse(body)
assert got["rating"] == Decimal("4.50") and got["review_count"] == 2
# Product-name "title" dropped; a review with no text is skipped.
assert got["reviews"] == [{"author": "Jay", "rating": Decimal("5.00"), "title": None,
"body": "All over performance is good", "review_date": "2026-05-02 10:00:00",
"sentiment": "positive"}]
assert vs.parse('{"data":{"products":{"items":[]}}}') is None
def test_api_breakdown_only_when_a_platform_publishes_one(db, client):
from app.electronics.collector import Collector, RunOptions, RunStats
from app.electronics.db import repository as repo
from app.electronics.models import Listing
from app.electronics.normalise.title_parser import parse_title, variant_key
def listing(site, sku, price, source_type="search_snippet"):
title = "Samsung Galaxy S24 5G (8GB RAM, 256GB)"
p = parse_title(title, "mobiles")
l = Listing(site_domain=site, source_sku=sku, source_url=f"https://www.{site}/p/{sku}",
source_type=source_type, brand_slug="samsung", category="mobiles", title=title,
evidence_text=f"{title} ₹{price}", confidence=0.5, parser="test",
model=p.model, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=Decimal(price))
l.model_norm, l.variant_key = p.model_norm, variant_key(p, "mobiles")
return l
c = Collector.__new__(Collector)
c.opt = RunOptions(category="mobiles", brands=["samsung"])
c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats()
b = listing("reliancedigital.in", "rd-1", 73999, "scraped_page")
b.rating, b.review_count, b.rating_breakdown = Decimal("4.0"), 4, {"5": 2, "4": 1, "1": 1}
c.store(listing("amazon.in", "B0CS5XW6TN", 74999))
c.store(b)
repo.refresh_verification()
pid = client.get("/api/elec/products", params={"category": "mobiles"}).json()["products"][0]["product_id"]
rating = client.get(f"/api/elec/products/{pid}").json()["rating"]
assert rating["breakdown"] == [
{"stars": 5, "count": 2, "percent": 50}, {"stars": 4, "count": 1, "percent": 25},
{"stars": 3, "count": 0, "percent": 0}, {"stars": 2, "count": 0, "percent": 0},
{"stars": 1, "count": 1, "percent": 25}]
assert rating["sources"][0]["breakdown"] == {"5": 2, "4": 1, "1": 1}
def test_brand_store_page_must_be_the_exact_model_and_variant():
from app.electronics.normalise.title_parser import parse_title
from app.electronics.review_refresh import _same_variant
a56 = {"model_norm": "galaxy a56", "ram_gb": Decimal("8"), "storage_gb": Decimal("256")}
page = lambda t: parse_title(t, "mobiles", expected_brand="samsung") # noqa: E731
assert _same_variant(page("Samsung Galaxy A56 5G 8GB/256GB (Olive)"), a56)
# A neighbouring model scores ~90 on fuzzy matching; it must still be refused.
assert not _same_variant(page("Samsung Galaxy A57 5G 8GB/256GB (Navy)"), a56)
assert not _same_variant(page("Samsung Galaxy A56 5G 12GB/256GB (Olive)"), a56)
assert not _same_variant(page("Samsung Galaxy A56 5G 8GB/128GB (Blue)"), a56)
s25 = {"model_norm": "galaxy s25", "ram_gb": Decimal("12"), "storage_gb": Decimal("256")}
assert _same_variant(page("Samsung Galaxy S25 12GB/256GB (Icyblue)"), s25)
assert not _same_variant(page("Samsung Galaxy S25 Ultra 12GB/256GB (Titanium Black)"), s25)
assert not _same_variant(page("Samsung Galaxy S25+ 12GB/256GB (Navy)"), s25)