297 lines
16 KiB
Python
297 lines
16 KiB
Python
"""Ratings and reviews: read only what a page or search result states, and pick
|
|
the review mix by the product's rating. Offline tests first; the database
|
|
tests are skipped when the local Postgres container is not running."""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from decimal import Decimal
|
|
|
|
from app.electronics.extract.jsonld import extract_products
|
|
from app.electronics.extract.serp_parser import read_rating
|
|
from app.electronics.reviews import select_reviews, sentiment_for
|
|
from app.electronics.search.providers import SearchHit, pagemap_rating
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Search-result ratings
|
|
# ---------------------------------------------------------------------------
|
|
def test_read_rating_accepts_explicit_statements():
|
|
r = read_rating("Samsung Galaxy S24 5G ... 4.3 out of 5 stars 1,234 ratings. ₹74,999")
|
|
assert r.rating == Decimal("4.3") and r.review_count == 1234
|
|
assert read_rating("Rating: 4.1/5 based on reviews").rating == Decimal("4.1")
|
|
r = read_rating("4.4★ (12,345 ratings)")
|
|
assert r.rating == Decimal("4.4") and r.review_count == 12345
|
|
|
|
|
|
def test_read_rating_refuses_guesses():
|
|
assert read_rating("Galaxy S24 8GB 256GB ₹74,999").rating is None
|
|
assert read_rating("1/5 inch sensor, 50MP").rating is None # a fraction, not a rating
|
|
assert read_rating("5/5G phone").rating is None
|
|
assert read_rating("4.2 out of 5 ... 3.9 out of 5").rating is None # two products: ambiguous
|
|
assert read_rating("7 out of 5").rating is None
|
|
|
|
|
|
def test_pagemap_rating_and_cached_hits_without_rating():
|
|
got = pagemap_rating({"aggregaterating": [{"ratingvalue": "4.5", "reviewcount": "2,310", "bestrating": "5"}]})
|
|
assert got["rating"] == 4.5 and got["review_count"] == 2310
|
|
assert pagemap_rating({"aggregaterating": [{"ratingvalue": "9", "bestrating": "10"}]}) is None
|
|
# Search results cached before the rating field existed still load.
|
|
hit = SearchHit.from_dict({"url": "https://a.in/p", "title": "t", "snippet": "s", "provider": "ddg", "rank": 0})
|
|
assert hit.rating is None
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Page reviews (schema.org JSON-LD)
|
|
# ---------------------------------------------------------------------------
|
|
def test_jsonld_reviews_are_read_verbatim():
|
|
ld = {
|
|
"@context": "https://schema.org", "@type": "Product", "name": "Samsung Galaxy S24",
|
|
"aggregateRating": {"ratingValue": "4.4", "reviewCount": "120"},
|
|
"review": [
|
|
{"@type": "Review", "author": {"@type": "Person", "name": "Arun"}, "name": "Great phone",
|
|
"reviewBody": "Battery lasts all day.", "datePublished": "2026-05-01",
|
|
"reviewRating": {"ratingValue": "5", "bestRating": "5"}},
|
|
{"@type": "Review", "author": "Priya", "reviewBody": "Heats up while gaming.",
|
|
"reviewRating": {"ratingValue": "4", "bestRating": "10"}},
|
|
{"@type": "Review", "author": "No words", "reviewRating": {"ratingValue": "1"}},
|
|
],
|
|
}
|
|
html = f'<script type="application/ld+json">{json.dumps(ld)}</script>'
|
|
p = extract_products(html)[0]
|
|
assert p["rating"] == Decimal("4.4") and p["review_count"] == 120
|
|
assert [r["body"] for r in p["reviews"]] == ["Battery lasts all day.", "Heats up while gaming."]
|
|
assert p["reviews"][0]["author"] == "Arun" and p["reviews"][0]["title"] == "Great phone"
|
|
assert p["reviews"][1]["rating"] == Decimal("2.0") # 4 out of 10, rescaled
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Review mix
|
|
# ---------------------------------------------------------------------------
|
|
def _pool(pos: int, neu: int, neg: int) -> list:
|
|
out = []
|
|
for label, n, stars in (("p", pos, 5), ("u", neu, 3), ("n", neg, 1)):
|
|
out += [{"body": f"{label}{i}", "rating": stars} for i in range(n)]
|
|
return out
|
|
|
|
|
|
def _counts(picked: list) -> tuple:
|
|
return tuple(sum(1 for r in picked if r["sentiment"] == s) for s in ("positive", "neutral", "negative"))
|
|
|
|
|
|
def test_sentiment_is_the_reviewers_own_stars():
|
|
assert [sentiment_for(x) for x in (5, 4, 3.5, 3, 2.9, 1, None)] == [
|
|
"positive", "positive", "neutral", "neutral", "negative", "negative", None]
|
|
|
|
|
|
def test_high_rating_shows_mostly_positive():
|
|
assert _counts(select_reviews(4.7, _pool(20, 20, 20))) == (6, 3, 1)
|
|
|
|
|
|
def test_middling_rating_shows_mostly_neutral():
|
|
assert _counts(select_reviews(3.6, _pool(20, 20, 20))) == (3, 5, 2)
|
|
|
|
|
|
def test_low_rating_shows_mostly_negative():
|
|
assert _counts(select_reviews(2.5, _pool(20, 20, 20))) == (2, 2, 6)
|
|
|
|
|
|
def test_short_groups_hand_slots_on_and_nothing_is_padded():
|
|
picked = select_reviews(4.8, _pool(3, 20, 0))
|
|
assert len(picked) == 10 and _counts(picked) == (3, 7, 0)
|
|
assert len(select_reviews(4.8, _pool(1, 1, 1))) == 3
|
|
assert select_reviews(4.8, [{"body": "no stars", "rating": None}]) == []
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Database + API
|
|
# ---------------------------------------------------------------------------
|
|
def test_api_serves_ratings_reviews_and_out_of_stock_price(db, client):
|
|
from app.electronics.collector import Collector, RunOptions, RunStats
|
|
from app.electronics.db import repository as repo
|
|
from app.electronics.models import Listing
|
|
from app.electronics.normalise.title_parser import parse_title, variant_key
|
|
|
|
def _listing(site, sku, title, *, price, source_type="search_snippet", evidence=None):
|
|
p = parse_title(title, "mobiles")
|
|
l = Listing(site_domain=site, source_sku=sku, source_url=f"https://www.{site}/p/{sku}",
|
|
source_type=source_type, brand_slug=p.brand.brand_slug, category="mobiles", title=title,
|
|
evidence_text=evidence or f"{title} ₹{price}", confidence=0.5, parser="test",
|
|
model=p.model, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=price)
|
|
l.model_norm, l.variant_key = p.model_norm, variant_key(p, "mobiles")
|
|
return l
|
|
|
|
c = Collector.__new__(Collector)
|
|
c.opt = RunOptions(category="mobiles", brands=["samsung"])
|
|
c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats()
|
|
title = "Samsung Galaxy S24 5G (8GB RAM, 256GB)"
|
|
a = _listing("amazon.in", "B0CS5XW6TN", title, price=Decimal(74999))
|
|
a.in_stock, a.rating, a.review_count = False, Decimal("4.6"), 300
|
|
b = _listing("croma.com", "303838", title, price=Decimal(73999), source_type="scraped_page",
|
|
evidence='{"price": "73999"}')
|
|
b.in_stock, b.rating, b.review_count = False, Decimal("4.0"), 100
|
|
b.reviews = [{"author": "Arun", "rating": Decimal(5), "title": "Great", "body": "Battery lasts all day."},
|
|
{"author": "Priya", "rating": Decimal(2), "body": "Heats up."}]
|
|
c.store(a)
|
|
c.store(b)
|
|
repo.refresh_verification()
|
|
|
|
product = client.get("/api/elec/products", params={"category": "mobiles"}).json()["products"][0]
|
|
assert product["best_price"] == "73999.00" # every listing out of stock, price still shown
|
|
detail = client.get(f"/api/elec/products/{product['product_id']}").json()
|
|
assert detail["rating"]["value"] == 4.5 and detail["rating"]["count"] == 400
|
|
assert {s["site"] for s in detail["rating"]["sources"]} == {"Amazon.in", "Croma"}
|
|
assert [r["body"] for r in detail["reviews"]] == ["Battery lasts all day.", "Heats up."]
|
|
assert all(r["source_url"].startswith("https://") for r in detail["reviews"])
|
|
assert detail["rating"]["breakdown"] is None # no platform published one
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Free sources: brand-store JSON-LD, retailer-embedded ratings, Vijay Sales
|
|
# ---------------------------------------------------------------------------
|
|
def _ld(*nodes) -> str:
|
|
return "".join(f'<script type="application/ld+json">{json.dumps(n)}</script>' for n in nodes)
|
|
|
|
|
|
def _review(author, stars, body, date="2026-09-30T17:06:05.000+00:00"):
|
|
return {"@type": "Review", "author": {"@type": "Person", "name": author}, "reviewBody": body,
|
|
"datePublished": date, "reviewRating": {"@type": "Rating", "bestRating": 5, "ratingValue": stars}}
|
|
|
|
|
|
def test_nameless_review_node_is_merged_by_id():
|
|
# The samsung.com/in shape: Bazaarvoice publishes ratings in a second,
|
|
# nameless Product node that shares the real product's @id.
|
|
pid = "https://www.samsung.com/in/smartphones/galaxy-a/galaxy-a56-5g-awesome-olive-256gb-sm-a566ezggins/"
|
|
html = _ld(
|
|
{"@context": "https://schema.org", "@type": "Product", "@id": pid, "name": "Galaxy A56 5G (8 GB Memory)",
|
|
"sku": "SM-A566EZGG", "offers": {"@type": "Offer", "price": "48999", "priceCurrency": "INR"}},
|
|
{"@context": "https://schema.org", "@type": "Product", "@id": pid,
|
|
"aggregateRating": {"@type": "AggregateRating", "ratingValue": "4.5", "ratingCount": 1224},
|
|
"review": [_review("Vedant", 5, "Great and useful AI features"), _review("Ravi", 1, "Heats up")]},
|
|
)
|
|
[p] = extract_products(html)
|
|
assert p["name"] == "Galaxy A56 5G (8 GB Memory)" and p["price"] == Decimal("48999")
|
|
assert p["rating"] == Decimal("4.5") and p["review_count"] == 1224
|
|
assert [(r["author"], r["rating"], r["body"]) for r in p["reviews"]] == [
|
|
("Vedant", Decimal("5.0"), "Great and useful AI features"), ("Ravi", Decimal("1.0"), "Heats up")]
|
|
|
|
|
|
def test_nameless_review_node_without_id_is_not_guessed_between_products():
|
|
lone = {"@type": "Product", "aggregateRating": {"ratingValue": "4.1", "ratingCount": 9}}
|
|
[only] = extract_products(_ld({"@type": "Product", "name": "Galaxy A56"}, lone))
|
|
assert only["rating"] == Decimal("4.1") # one product on the page: unambiguous
|
|
two = extract_products(_ld({"@type": "Product", "name": "Galaxy A56"},
|
|
{"@type": "Product", "name": "Galaxy A36"}, lone))
|
|
assert all(p["rating"] is None for p in two) # two products: never guess which
|
|
|
|
|
|
def test_reliance_embedded_rating_is_read_from_the_pages_own_product_only():
|
|
from app.electronics.extract.embedded_ratings import embedded_rating
|
|
|
|
def page(app, recommended=None):
|
|
state = {"productDetailsPage": {"product": {"name": "Dell 15", "_custom_json": {"_app": app}}},
|
|
"recommendations": [{"_custom_json": {"_app": recommended}}] if recommended else []}
|
|
return f"<script>window.__INITIAL_STATE__={json.dumps(state)};window.x=1</script>"
|
|
|
|
url = "https://www.reliancedigital.in/product/dell-15-9991568"
|
|
got = embedded_rating("reliancedigital.in", page(
|
|
{"averageRating": 4, "ratingsCount": 3, "ratingsCountDetails": {"5": 1, "4": 1, "3": 1}}), url)
|
|
assert got == {"rating": Decimal("4.00"), "review_count": 3, "breakdown": {"5": 1, "4": 1, "3": 1}}
|
|
# "No ratings yet" is published as zeros; a recommended product's rating is not ours.
|
|
zero = page({"averageRating": 0, "ratingsCount": 0}, recommended={"averageRating": 4.8, "ratingsCount": 90})
|
|
assert embedded_rating("reliancedigital.in", zero, url) is None
|
|
|
|
|
|
def test_poorvika_rating_needs_the_pages_own_slug():
|
|
from app.electronics.extract.embedded_ratings import embedded_rating
|
|
|
|
def page(slug, rating, count):
|
|
data = {"props": {"pageProps": {"additionalData": {"slug": slug, "rating": rating, "ratingCount": count}}}}
|
|
return f'<script id="__NEXT_DATA__" type="application/json">{json.dumps(data)}</script>'
|
|
|
|
slug = "xiaomi-14-civi-5g-shadow-black-256gb-8gb-ram"
|
|
url = f"https://www.poorvika.com/{slug}/p"
|
|
assert embedded_rating("poorvika.com", page(slug, 5, 1), url)["rating"] == Decimal("5.00")
|
|
assert embedded_rating("poorvika.com", page("some-other-phone", 5, 1), url) is None
|
|
assert embedded_rating("poorvika.com", page(slug, None, 0), url) is None
|
|
|
|
|
|
def test_vasanth_template_rating_is_never_read():
|
|
from app.electronics.extract.embedded_ratings import embedded_rating
|
|
|
|
template = ('<h4 class="avg-mark">3.3</h4><a class="rating-reviews">(3 Reviews)</a>'
|
|
+ _ld({"@type": "Product", "name": "HP 15", "aggregateRating": {"ratingValue": "3.3", "reviewCount": 3}}))
|
|
assert embedded_rating("vasanthandco.in", template, "https://vasanthandco.in/product/174200102203/") is None
|
|
|
|
|
|
def test_vijaysales_graphql_review_feed():
|
|
from app.electronics.extract import vijaysales_reviews as vs
|
|
|
|
url = "https://www.vijaysales.com/p/239904/acer-aspire-lite-al15-41-laptop-amd-ryzen-7-16gb"
|
|
assert vs.url_key(url) == "acer-aspire-lite-al15-41-laptop-amd-ryzen-7-16gb"
|
|
assert "url_key" in vs.graphql_url(url)
|
|
assert vs.url_key('https://x.in/p/a"b') is None # cannot break out of the query
|
|
product_name = "Acer Aspire Lite AL15-41 Laptop (AMD Ryzen 7/ 16GB DDR4 RAM/ 512GB SSD)"
|
|
body = json.dumps({"data": {"products": {"items": [
|
|
{"sku": "P1", "rating_summary": 0, "review_count": 0, "reviews": {"items": []}},
|
|
{"sku": "239904", "rating_summary": 90, "review_count": 2, "reviews": {"items": [
|
|
{"nickname": "Jay", "summary": product_name, "text": "All over performance is good",
|
|
"average_rating": 100, "created_at": "2026-05-02 10:00:00"},
|
|
{"nickname": "guest user", "summary": "Okay", "text": "", "average_rating": 60}]}}]}}})
|
|
got = vs.parse(body)
|
|
assert got["rating"] == Decimal("4.50") and got["review_count"] == 2
|
|
# Product-name "title" dropped; a review with no text is skipped.
|
|
assert got["reviews"] == [{"author": "Jay", "rating": Decimal("5.00"), "title": None,
|
|
"body": "All over performance is good", "review_date": "2026-05-02 10:00:00",
|
|
"sentiment": "positive"}]
|
|
assert vs.parse('{"data":{"products":{"items":[]}}}') is None
|
|
|
|
|
|
def test_api_breakdown_only_when_a_platform_publishes_one(db, client):
|
|
from app.electronics.collector import Collector, RunOptions, RunStats
|
|
from app.electronics.db import repository as repo
|
|
from app.electronics.models import Listing
|
|
from app.electronics.normalise.title_parser import parse_title, variant_key
|
|
|
|
def listing(site, sku, price, source_type="search_snippet"):
|
|
title = "Samsung Galaxy S24 5G (8GB RAM, 256GB)"
|
|
p = parse_title(title, "mobiles")
|
|
l = Listing(site_domain=site, source_sku=sku, source_url=f"https://www.{site}/p/{sku}",
|
|
source_type=source_type, brand_slug="samsung", category="mobiles", title=title,
|
|
evidence_text=f"{title} ₹{price}", confidence=0.5, parser="test",
|
|
model=p.model, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=Decimal(price))
|
|
l.model_norm, l.variant_key = p.model_norm, variant_key(p, "mobiles")
|
|
return l
|
|
|
|
c = Collector.__new__(Collector)
|
|
c.opt = RunOptions(category="mobiles", brands=["samsung"])
|
|
c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats()
|
|
b = listing("reliancedigital.in", "rd-1", 73999, "scraped_page")
|
|
b.rating, b.review_count, b.rating_breakdown = Decimal("4.0"), 4, {"5": 2, "4": 1, "1": 1}
|
|
c.store(listing("amazon.in", "B0CS5XW6TN", 74999))
|
|
c.store(b)
|
|
repo.refresh_verification()
|
|
pid = client.get("/api/elec/products", params={"category": "mobiles"}).json()["products"][0]["product_id"]
|
|
rating = client.get(f"/api/elec/products/{pid}").json()["rating"]
|
|
assert rating["breakdown"] == [
|
|
{"stars": 5, "count": 2, "percent": 50}, {"stars": 4, "count": 1, "percent": 25},
|
|
{"stars": 3, "count": 0, "percent": 0}, {"stars": 2, "count": 0, "percent": 0},
|
|
{"stars": 1, "count": 1, "percent": 25}]
|
|
assert rating["sources"][0]["breakdown"] == {"5": 2, "4": 1, "1": 1}
|
|
|
|
|
|
def test_brand_store_page_must_be_the_exact_model_and_variant():
|
|
from app.electronics.normalise.title_parser import parse_title
|
|
from app.electronics.review_refresh import _same_variant
|
|
|
|
a56 = {"model_norm": "galaxy a56", "ram_gb": Decimal("8"), "storage_gb": Decimal("256")}
|
|
page = lambda t: parse_title(t, "mobiles", expected_brand="samsung") # noqa: E731
|
|
assert _same_variant(page("Samsung Galaxy A56 5G 8GB/256GB (Olive)"), a56)
|
|
# A neighbouring model scores ~90 on fuzzy matching; it must still be refused.
|
|
assert not _same_variant(page("Samsung Galaxy A57 5G 8GB/256GB (Navy)"), a56)
|
|
assert not _same_variant(page("Samsung Galaxy A56 5G 12GB/256GB (Olive)"), a56)
|
|
assert not _same_variant(page("Samsung Galaxy A56 5G 8GB/128GB (Blue)"), a56)
|
|
s25 = {"model_norm": "galaxy s25", "ram_gb": Decimal("12"), "storage_gb": Decimal("256")}
|
|
assert _same_variant(page("Samsung Galaxy S25 12GB/256GB (Icyblue)"), s25)
|
|
assert not _same_variant(page("Samsung Galaxy S25 Ultra 12GB/256GB (Titanium Black)"), s25)
|
|
assert not _same_variant(page("Samsung Galaxy S25+ 12GB/256GB (Navy)"), s25)
|