Customer Rating and review changes
This commit is contained in:
95
backend/app/electronics/extract/embedded_ratings.py
Normal file
95
backend/app/electronics/extract/embedded_ratings.py
Normal file
@@ -0,0 +1,95 @@
|
||||
"""Ratings a retailer embeds in its own product-page HTML (not in JSON-LD).
|
||||
|
||||
Only the page we already fetched is read - no extra requests - and only the
|
||||
record of the page's OWN product (never recommendations on the same page).
|
||||
Zero values mean "no ratings yet" and are dropped. Review text is not available
|
||||
here: those retailers load it from feeds their robots.txt does not allow.
|
||||
|
||||
Deliberately absent: vasanthandco.in. Its product pages carry an identical
|
||||
"3.3 average, 3 reviews" template block on every product - placeholder
|
||||
content, not ratings - so nothing is ever read from that domain.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from typing import Any, Callable, Dict, Optional
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
# Domains whose rating-looking markup is known to be fake/placeholder.
|
||||
IGNORED_DOMAINS = frozenset({"vasanthandco.in"})
|
||||
|
||||
|
||||
def _dec(value: Any) -> Optional[Decimal]:
|
||||
try:
|
||||
d = Decimal(str(value))
|
||||
except (InvalidOperation, TypeError, ValueError):
|
||||
return None
|
||||
return d if Decimal(0) < d <= Decimal(5) else None
|
||||
|
||||
|
||||
def _int(value: Any) -> Optional[int]:
|
||||
try:
|
||||
n = int(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
return n if n > 0 else None
|
||||
|
||||
|
||||
def _reading(rating: Any, count: Any, breakdown: Optional[Dict[str, int]] = None) -> Optional[Dict[str, Any]]:
|
||||
r = _dec(rating)
|
||||
if r is None:
|
||||
return None
|
||||
clean = None
|
||||
if breakdown:
|
||||
clean = {str(k): int(v) for k, v in breakdown.items()
|
||||
if str(k) in {"1", "2", "3", "4", "5"} and _int(v)}
|
||||
return {"rating": r.quantize(Decimal("0.01")), "review_count": _int(count), "breakdown": clean or None}
|
||||
|
||||
|
||||
def _reliance(html: str, url: str) -> Optional[Dict[str, Any]]:
|
||||
"""window.__INITIAL_STATE__.productDetailsPage.product._custom_json._app"""
|
||||
marker = "window.__INITIAL_STATE__="
|
||||
i = html.find(marker)
|
||||
if i < 0:
|
||||
return None
|
||||
try:
|
||||
state, _ = json.JSONDecoder().raw_decode(html, i + len(marker))
|
||||
app = state["productDetailsPage"]["product"]["_custom_json"]["_app"]
|
||||
except (ValueError, KeyError, TypeError):
|
||||
return None
|
||||
if not isinstance(app, dict):
|
||||
return None
|
||||
return _reading(app.get("averageRating"), app.get("ratingsCount"), app.get("ratingsCountDetails"))
|
||||
|
||||
|
||||
def _poorvika(html: str, url: str) -> Optional[Dict[str, Any]]:
|
||||
"""__NEXT_DATA__ props.pageProps.additionalData, only when its slug is this URL's."""
|
||||
tag = BeautifulSoup(html, "lxml").find("script", id="__NEXT_DATA__")
|
||||
if tag is None:
|
||||
return None
|
||||
try:
|
||||
data = json.loads(tag.string or tag.get_text())["props"]["pageProps"]["additionalData"]
|
||||
except (ValueError, KeyError, TypeError):
|
||||
return None
|
||||
slug = str(data.get("slug") or "")
|
||||
if not slug or slug not in urlparse(url).path:
|
||||
return None
|
||||
return _reading(data.get("rating"), data.get("ratingCount"))
|
||||
|
||||
|
||||
_READERS: Dict[str, Callable[[str, str], Optional[Dict[str, Any]]]] = {
|
||||
"reliancedigital.in": _reliance,
|
||||
"poorvika.com": _poorvika,
|
||||
}
|
||||
|
||||
|
||||
def embedded_rating(domain: str, html: str, url: str) -> Optional[Dict[str, Any]]:
|
||||
"""{"rating": Decimal, "review_count": int|None, "breakdown": {"5": n, ...}|None} or None."""
|
||||
if domain in IGNORED_DOMAINS:
|
||||
return None
|
||||
reader = _READERS.get(domain)
|
||||
return reader(html, url) if reader else None
|
||||
@@ -161,42 +161,67 @@ def _reviews(node: dict) -> List[Dict[str, Any]]:
|
||||
return out
|
||||
|
||||
|
||||
def _rating_only_nodes(nodes: List[dict]) -> Dict[Optional[str], dict]:
|
||||
"""Nameless Product nodes that only carry ratings/reviews, keyed by @id.
|
||||
|
||||
Review widgets (Bazaarvoice on samsung.com/in, for one) publish the
|
||||
product's aggregateRating and reviews as a SEPARATE Product node with no
|
||||
name, linked to the real product by the same @id. Without merging, those
|
||||
real reviews would be skipped for lack of a name.
|
||||
"""
|
||||
out: Dict[Optional[str], dict] = {}
|
||||
for node in nodes:
|
||||
if _text(node.get("name")):
|
||||
continue
|
||||
if not (isinstance(node.get("aggregateRating"), dict) or node.get("review") or node.get("reviews")):
|
||||
continue
|
||||
out.setdefault(_text(node.get("@id")), node)
|
||||
return out
|
||||
|
||||
|
||||
def extract_products(html: str) -> List[Dict[str, Any]]:
|
||||
"""All schema.org Product nodes on the page, flattened to plain fields."""
|
||||
products: List[Dict[str, Any]] = []
|
||||
for block in json_ld_blocks(html):
|
||||
for node in _walk(block):
|
||||
if not (_types(node) & _PRODUCT_TYPES):
|
||||
continue
|
||||
name = _text(node.get("name"))
|
||||
if not name:
|
||||
continue
|
||||
offer = _offer(node.get("offers")) if node.get("offers") else {}
|
||||
if not offer and isinstance(node.get("hasVariant"), list):
|
||||
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
|
||||
props = {}
|
||||
for p in node.get("additionalProperty") or []:
|
||||
if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""):
|
||||
props[str(p["name"])] = str(p["value"])
|
||||
rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {}
|
||||
products.append({
|
||||
"name": name,
|
||||
"brand": _text(node.get("brand")),
|
||||
"sku": _text(node.get("sku")) or _text(node.get("productID")),
|
||||
"mpn": _text(node.get("mpn")),
|
||||
"gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None),
|
||||
"color": _text(node.get("color")),
|
||||
"images": _images(node.get("image")),
|
||||
"description": _text(node.get("description")),
|
||||
"price": offer.get("price"),
|
||||
"currency": offer.get("currency"),
|
||||
"availability": offer.get("availability"),
|
||||
"in_stock": offer.get("in_stock"),
|
||||
# Sites publish 0 for "no ratings yet"; that is not a rating.
|
||||
"rating": (_dec(rating.get("ratingValue")) or None),
|
||||
"review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None,
|
||||
"reviews": _reviews(node),
|
||||
"properties": props,
|
||||
"evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500],
|
||||
})
|
||||
nodes = [n for block in json_ld_blocks(html) for n in _walk(block) if _types(n) & _PRODUCT_TYPES]
|
||||
rating_nodes = _rating_only_nodes(nodes)
|
||||
named = [n for n in nodes if _text(n.get("name"))]
|
||||
for node in named:
|
||||
# Attach a ratings-only node: same @id, or - when it has no @id - the
|
||||
# page's only named product (there is then no doubt which it is about).
|
||||
extra = rating_nodes.get(_text(node.get("@id"))) if node.get("@id") else None
|
||||
if extra is None and len(named) == 1:
|
||||
extra = rating_nodes.get(None)
|
||||
if extra is not None:
|
||||
node = {**node,
|
||||
"aggregateRating": node.get("aggregateRating") or extra.get("aggregateRating"),
|
||||
"review": node.get("review") or extra.get("review") or extra.get("reviews")}
|
||||
name = _text(node.get("name"))
|
||||
offer = _offer(node.get("offers")) if node.get("offers") else {}
|
||||
if not offer and isinstance(node.get("hasVariant"), list):
|
||||
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
|
||||
props = {}
|
||||
for p in node.get("additionalProperty") or []:
|
||||
if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""):
|
||||
props[str(p["name"])] = str(p["value"])
|
||||
rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {}
|
||||
products.append({
|
||||
"name": name,
|
||||
"brand": _text(node.get("brand")),
|
||||
"sku": _text(node.get("sku")) or _text(node.get("productID")),
|
||||
"mpn": _text(node.get("mpn")),
|
||||
"gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None),
|
||||
"color": _text(node.get("color")),
|
||||
"images": _images(node.get("image")),
|
||||
"description": _text(node.get("description")),
|
||||
"price": offer.get("price"),
|
||||
"currency": offer.get("currency"),
|
||||
"availability": offer.get("availability"),
|
||||
"in_stock": offer.get("in_stock"),
|
||||
# Sites publish 0 for "no ratings yet"; that is not a rating.
|
||||
"rating": (_dec(rating.get("ratingValue")) or None),
|
||||
"review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None,
|
||||
"reviews": _reviews(node),
|
||||
"properties": props,
|
||||
"evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500],
|
||||
})
|
||||
return products
|
||||
|
||||
80
backend/app/electronics/extract/vijaysales_reviews.py
Normal file
80
backend/app/electronics/extract/vijaysales_reviews.py
Normal file
@@ -0,0 +1,80 @@
|
||||
"""Vijay Sales ratings and reviews from its public GraphQL endpoint.
|
||||
|
||||
vijaysales.com serves product data to its own pages from GET /api/graphql;
|
||||
robots.txt does not disallow it, and no login or token is involved. One GET
|
||||
per listing, made through PoliteClient (robots, pacing, circuit breaker).
|
||||
|
||||
The product is looked up by url_key - the last path segment of the listing URL
|
||||
we already hold - so the answer is about exactly that listing.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from typing import Any, Dict, List, Optional
|
||||
from urllib.parse import urlencode, urlparse
|
||||
|
||||
from app.electronics.reviews import sentiment_for
|
||||
|
||||
ENDPOINT = "https://www.vijaysales.com/api/graphql"
|
||||
|
||||
_QUERY = (
|
||||
'{products(filter:{url_key:{eq:"%s"}}){items{sku rating_summary review_count '
|
||||
"reviews(pageSize:20){items{nickname summary text average_rating created_at}}}}}"
|
||||
)
|
||||
|
||||
|
||||
def url_key(listing_url: str) -> Optional[str]:
|
||||
path = urlparse(listing_url).path.rstrip("/")
|
||||
key = path.rsplit("/", 1)[-1] if path else ""
|
||||
# url_keys are slugs; anything else would break out of the query string.
|
||||
return key if key and all(c.isalnum() or c in "-_" for c in key) else None
|
||||
|
||||
|
||||
def graphql_url(listing_url: str) -> Optional[str]:
|
||||
key = url_key(listing_url)
|
||||
return f"{ENDPOINT}?{urlencode({'query': _QUERY % key})}" if key else None
|
||||
|
||||
|
||||
def _five(percent: Any) -> Optional[Decimal]:
|
||||
"""Magento ratings are percentages (80 = 4 stars)."""
|
||||
try:
|
||||
p = Decimal(str(percent))
|
||||
except (InvalidOperation, TypeError, ValueError):
|
||||
return None
|
||||
return (p / 20).quantize(Decimal("0.01")) if Decimal(0) < p <= Decimal(100) else None
|
||||
|
||||
|
||||
def parse(body: str) -> Optional[Dict[str, Any]]:
|
||||
"""{"rating", "review_count", "reviews": [...]} for the best-reviewed matching item, or None."""
|
||||
try:
|
||||
items = json.loads(body)["data"]["products"]["items"] or []
|
||||
except (ValueError, KeyError, TypeError):
|
||||
return None
|
||||
items = [i for i in items if isinstance(i, dict)]
|
||||
if not items:
|
||||
return None
|
||||
item = max(items, key=lambda i: i.get("review_count") or 0)
|
||||
rating = _five(item.get("rating_summary"))
|
||||
if rating is None:
|
||||
return None
|
||||
reviews: List[Dict[str, Any]] = []
|
||||
for r in ((item.get("reviews") or {}).get("items") or []):
|
||||
body_text = (r.get("text") or "").strip()
|
||||
if not body_text:
|
||||
continue
|
||||
stars = _five(r.get("average_rating"))
|
||||
title = (r.get("summary") or "").strip()
|
||||
# When a customer leaves the title blank the site fills in the full
|
||||
# product name; that is not something the reviewer wrote.
|
||||
if len(title) > 60 or "/" in title:
|
||||
title = ""
|
||||
reviews.append({
|
||||
"author": (r.get("nickname") or "").strip() or None,
|
||||
"rating": stars,
|
||||
"title": title or None,
|
||||
"body": body_text[:4000],
|
||||
"review_date": r.get("created_at"),
|
||||
"sentiment": sentiment_for(stars),
|
||||
})
|
||||
return {"rating": rating, "review_count": int(item.get("review_count") or 0) or None, "reviews": reviews}
|
||||
Reference in New Issue
Block a user