Customer Rating and review changes

This commit is contained in:
sriram
2026-10-05 11:21:39 +05:30
parent c7e4d59188
commit abe7ad8450
15 changed files with 697 additions and 95 deletions

View File

@@ -0,0 +1,95 @@
"""Ratings a retailer embeds in its own product-page HTML (not in JSON-LD).
Only the page we already fetched is read - no extra requests - and only the
record of the page's OWN product (never recommendations on the same page).
Zero values mean "no ratings yet" and are dropped. Review text is not available
here: those retailers load it from feeds their robots.txt does not allow.
Deliberately absent: vasanthandco.in. Its product pages carry an identical
"3.3 average, 3 reviews" template block on every product - placeholder
content, not ratings - so nothing is ever read from that domain.
"""
from __future__ import annotations
import json
import re
from decimal import Decimal, InvalidOperation
from typing import Any, Callable, Dict, Optional
from urllib.parse import urlparse
from bs4 import BeautifulSoup
# Domains whose rating-looking markup is known to be fake/placeholder.
IGNORED_DOMAINS = frozenset({"vasanthandco.in"})
def _dec(value: Any) -> Optional[Decimal]:
try:
d = Decimal(str(value))
except (InvalidOperation, TypeError, ValueError):
return None
return d if Decimal(0) < d <= Decimal(5) else None
def _int(value: Any) -> Optional[int]:
try:
n = int(value)
except (TypeError, ValueError):
return None
return n if n > 0 else None
def _reading(rating: Any, count: Any, breakdown: Optional[Dict[str, int]] = None) -> Optional[Dict[str, Any]]:
r = _dec(rating)
if r is None:
return None
clean = None
if breakdown:
clean = {str(k): int(v) for k, v in breakdown.items()
if str(k) in {"1", "2", "3", "4", "5"} and _int(v)}
return {"rating": r.quantize(Decimal("0.01")), "review_count": _int(count), "breakdown": clean or None}
def _reliance(html: str, url: str) -> Optional[Dict[str, Any]]:
"""window.__INITIAL_STATE__.productDetailsPage.product._custom_json._app"""
marker = "window.__INITIAL_STATE__="
i = html.find(marker)
if i < 0:
return None
try:
state, _ = json.JSONDecoder().raw_decode(html, i + len(marker))
app = state["productDetailsPage"]["product"]["_custom_json"]["_app"]
except (ValueError, KeyError, TypeError):
return None
if not isinstance(app, dict):
return None
return _reading(app.get("averageRating"), app.get("ratingsCount"), app.get("ratingsCountDetails"))
def _poorvika(html: str, url: str) -> Optional[Dict[str, Any]]:
"""__NEXT_DATA__ props.pageProps.additionalData, only when its slug is this URL's."""
tag = BeautifulSoup(html, "lxml").find("script", id="__NEXT_DATA__")
if tag is None:
return None
try:
data = json.loads(tag.string or tag.get_text())["props"]["pageProps"]["additionalData"]
except (ValueError, KeyError, TypeError):
return None
slug = str(data.get("slug") or "")
if not slug or slug not in urlparse(url).path:
return None
return _reading(data.get("rating"), data.get("ratingCount"))
_READERS: Dict[str, Callable[[str, str], Optional[Dict[str, Any]]]] = {
"reliancedigital.in": _reliance,
"poorvika.com": _poorvika,
}
def embedded_rating(domain: str, html: str, url: str) -> Optional[Dict[str, Any]]:
"""{"rating": Decimal, "review_count": int|None, "breakdown": {"5": n, ...}|None} or None."""
if domain in IGNORED_DOMAINS:
return None
reader = _READERS.get(domain)
return reader(html, url) if reader else None

View File

@@ -161,42 +161,67 @@ def _reviews(node: dict) -> List[Dict[str, Any]]:
return out
def _rating_only_nodes(nodes: List[dict]) -> Dict[Optional[str], dict]:
"""Nameless Product nodes that only carry ratings/reviews, keyed by @id.
Review widgets (Bazaarvoice on samsung.com/in, for one) publish the
product's aggregateRating and reviews as a SEPARATE Product node with no
name, linked to the real product by the same @id. Without merging, those
real reviews would be skipped for lack of a name.
"""
out: Dict[Optional[str], dict] = {}
for node in nodes:
if _text(node.get("name")):
continue
if not (isinstance(node.get("aggregateRating"), dict) or node.get("review") or node.get("reviews")):
continue
out.setdefault(_text(node.get("@id")), node)
return out
def extract_products(html: str) -> List[Dict[str, Any]]:
"""All schema.org Product nodes on the page, flattened to plain fields."""
products: List[Dict[str, Any]] = []
for block in json_ld_blocks(html):
for node in _walk(block):
if not (_types(node) & _PRODUCT_TYPES):
continue
name = _text(node.get("name"))
if not name:
continue
offer = _offer(node.get("offers")) if node.get("offers") else {}
if not offer and isinstance(node.get("hasVariant"), list):
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
props = {}
for p in node.get("additionalProperty") or []:
if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""):
props[str(p["name"])] = str(p["value"])
rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {}
products.append({
"name": name,
"brand": _text(node.get("brand")),
"sku": _text(node.get("sku")) or _text(node.get("productID")),
"mpn": _text(node.get("mpn")),
"gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None),
"color": _text(node.get("color")),
"images": _images(node.get("image")),
"description": _text(node.get("description")),
"price": offer.get("price"),
"currency": offer.get("currency"),
"availability": offer.get("availability"),
"in_stock": offer.get("in_stock"),
# Sites publish 0 for "no ratings yet"; that is not a rating.
"rating": (_dec(rating.get("ratingValue")) or None),
"review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None,
"reviews": _reviews(node),
"properties": props,
"evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500],
})
nodes = [n for block in json_ld_blocks(html) for n in _walk(block) if _types(n) & _PRODUCT_TYPES]
rating_nodes = _rating_only_nodes(nodes)
named = [n for n in nodes if _text(n.get("name"))]
for node in named:
# Attach a ratings-only node: same @id, or - when it has no @id - the
# page's only named product (there is then no doubt which it is about).
extra = rating_nodes.get(_text(node.get("@id"))) if node.get("@id") else None
if extra is None and len(named) == 1:
extra = rating_nodes.get(None)
if extra is not None:
node = {**node,
"aggregateRating": node.get("aggregateRating") or extra.get("aggregateRating"),
"review": node.get("review") or extra.get("review") or extra.get("reviews")}
name = _text(node.get("name"))
offer = _offer(node.get("offers")) if node.get("offers") else {}
if not offer and isinstance(node.get("hasVariant"), list):
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
props = {}
for p in node.get("additionalProperty") or []:
if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""):
props[str(p["name"])] = str(p["value"])
rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {}
products.append({
"name": name,
"brand": _text(node.get("brand")),
"sku": _text(node.get("sku")) or _text(node.get("productID")),
"mpn": _text(node.get("mpn")),
"gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None),
"color": _text(node.get("color")),
"images": _images(node.get("image")),
"description": _text(node.get("description")),
"price": offer.get("price"),
"currency": offer.get("currency"),
"availability": offer.get("availability"),
"in_stock": offer.get("in_stock"),
# Sites publish 0 for "no ratings yet"; that is not a rating.
"rating": (_dec(rating.get("ratingValue")) or None),
"review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None,
"reviews": _reviews(node),
"properties": props,
"evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500],
})
return products

View File

@@ -0,0 +1,80 @@
"""Vijay Sales ratings and reviews from its public GraphQL endpoint.
vijaysales.com serves product data to its own pages from GET /api/graphql;
robots.txt does not disallow it, and no login or token is involved. One GET
per listing, made through PoliteClient (robots, pacing, circuit breaker).
The product is looked up by url_key - the last path segment of the listing URL
we already hold - so the answer is about exactly that listing.
"""
from __future__ import annotations
import json
from decimal import Decimal, InvalidOperation
from typing import Any, Dict, List, Optional
from urllib.parse import urlencode, urlparse
from app.electronics.reviews import sentiment_for
ENDPOINT = "https://www.vijaysales.com/api/graphql"
_QUERY = (
'{products(filter:{url_key:{eq:"%s"}}){items{sku rating_summary review_count '
"reviews(pageSize:20){items{nickname summary text average_rating created_at}}}}}"
)
def url_key(listing_url: str) -> Optional[str]:
path = urlparse(listing_url).path.rstrip("/")
key = path.rsplit("/", 1)[-1] if path else ""
# url_keys are slugs; anything else would break out of the query string.
return key if key and all(c.isalnum() or c in "-_" for c in key) else None
def graphql_url(listing_url: str) -> Optional[str]:
key = url_key(listing_url)
return f"{ENDPOINT}?{urlencode({'query': _QUERY % key})}" if key else None
def _five(percent: Any) -> Optional[Decimal]:
"""Magento ratings are percentages (80 = 4 stars)."""
try:
p = Decimal(str(percent))
except (InvalidOperation, TypeError, ValueError):
return None
return (p / 20).quantize(Decimal("0.01")) if Decimal(0) < p <= Decimal(100) else None
def parse(body: str) -> Optional[Dict[str, Any]]:
"""{"rating", "review_count", "reviews": [...]} for the best-reviewed matching item, or None."""
try:
items = json.loads(body)["data"]["products"]["items"] or []
except (ValueError, KeyError, TypeError):
return None
items = [i for i in items if isinstance(i, dict)]
if not items:
return None
item = max(items, key=lambda i: i.get("review_count") or 0)
rating = _five(item.get("rating_summary"))
if rating is None:
return None
reviews: List[Dict[str, Any]] = []
for r in ((item.get("reviews") or {}).get("items") or []):
body_text = (r.get("text") or "").strip()
if not body_text:
continue
stars = _five(r.get("average_rating"))
title = (r.get("summary") or "").strip()
# When a customer leaves the title blank the site fills in the full
# product name; that is not something the reviewer wrote.
if len(title) > 60 or "/" in title:
title = ""
reviews.append({
"author": (r.get("nickname") or "").strip() or None,
"rating": stars,
"title": title or None,
"body": body_text[:4000],
"review_date": r.get("created_at"),
"sentiment": sentiment_for(stars),
})
return {"rating": rating, "review_count": int(item.get("review_count") or 0) or None, "reviews": reviews}