Customer Rating and review changes

This commit is contained in:
sriram
2026-10-05 11:21:39 +05:30
parent c7e4d59188
commit abe7ad8450
15 changed files with 697 additions and 95 deletions

View File

@@ -141,23 +141,50 @@ def product(product_id: int) -> dict:
def _overall_rating(sources: List[dict]) -> Optional[dict]:
"""The product's rating across the platforms that state one: the mean
weighted by each platform's rating count (a platform that states no count
weighs as 1). None when no platform states a rating - never a guess."""
weighs as 1). None when no platform states a rating - never a guess.
One reading per platform: a site with a page per colour repeats the same
model rating on each, and adding those up would multiply its count."""
if not sources:
return None
weight = lambda s: max(int(s["review_count"] or 0), 1) # noqa: E731
per_site: Dict[str, dict] = {}
for s in sources:
if s["site"] not in per_site or weight(s) > weight(per_site[s["site"]]):
per_site[s["site"]] = s
sources = list(per_site.values())
total = sum(weight(s) for s in sources)
value = sum(Decimal(s["rating"]) * weight(s) for s in sources) / total
counts = [s["review_count"] for s in sources if s["review_count"]]
return {
"value": round(float(value), 1),
"count": sum(counts) if counts else None,
"breakdown": _breakdown(sources),
"sources": [
{"site": s["site"], "rating": float(s["rating"]), "review_count": s["review_count"], "source_url": s["source_url"]}
{"site": s["site"], "rating": float(s["rating"]), "review_count": s["review_count"],
"breakdown": s.get("rating_breakdown"), "source_url": s["source_url"]}
for s in sources
],
}
def _breakdown(sources: List[dict]) -> Optional[List[dict]]:
"""5→1 star counts summed over the platforms that PUBLISH a breakdown.
None when none does - never estimated from the few review texts held."""
totals = {str(n): 0 for n in range(1, 6)}
stated = False
for s in sources:
for star, n in (s.get("rating_breakdown") or {}).items():
if star in totals and isinstance(n, int) and n > 0:
totals[star] += n
stated = True
if not stated:
return None
total = sum(totals.values())
return [{"stars": int(k), "count": totals[k], "percent": round(100 * totals[k] / total)}
for k in ("5", "4", "3", "2", "1")]
@router.get("/products/{product_id}/price-history")
def price_history(product_id: int) -> List[dict]:
with connect() as conn:

View File

@@ -143,58 +143,22 @@ def rematch(category: str = typer.Option(..., help="mobiles | laptops"),
@app.command()
def reviews(category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)"),
limit: int = typer.Option(200, help="Max product pages to re-read"),
limit: int = typer.Option(500, help="Max product pages to re-read"),
budget: int = typer.Option(60, help="Max web searches for finding brand-store pages"),
no_discover: bool = typer.Option(False, "--no-discover", help="Skip finding brand-store pages"),
verbose: bool = False) -> None:
"""Re-read ratings and customer reviews from the product pages already on file.
"""Refresh real customer ratings and reviews from free, robots-allowed sources.
Only pages the collector itself reads (scraped / brand official listings)
are fetched, politely (robots.txt, per-site pacing, circuit breaker). A
rating or review is stored only when the page's own schema.org data states
it; nothing is generated.
Finds each verified product's page on brand stores that publish reviews
(brands.yaml `reviews_site`), then re-reads every readable product page for
its rating, star breakdown and reviews, plus Vijay Sales' public review
feed. A rating or review is stored only when its source states it.
"""
_setup_logging(verbose)
from rapidfuzz import fuzz
from app.electronics.review_refresh import refresh_reviews
from app.electronics.db import repository as repo
from app.electronics.extract.jsonld import extract_products
from app.electronics.net.polite_client import PoliteClient
rows = repo.listings_for_review_backfill(category)[:limit]
run_id = repo.start_run("reviews", {"category": category, "pages": len(rows)})
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
r.outcome, r.robots_allowed))
stats = {"pages": 0, "pages_ok": 0, "rated": 0, "reviews_stored": 0, "no_matching_product": 0}
status, error = "done", None
try:
for row in rows:
stats["pages"] += 1
res = client.get(row["source_url"])
if not res.ok:
continue
stats["pages_ok"] += 1
products = extract_products(res.text)
# The same product the listing was stored from: its SKU, else its name.
match = next((p for p in products if p.get("sku") and p["sku"] == row["source_sku"]), None)
if match is None:
title = (row["title"] or "").lower()
scored = [(fuzz.token_set_ratio(p["name"].lower(), title), p) for p in products]
scored = [sp for sp in scored if sp[0] >= 85]
match = max(scored, key=lambda sp: sp[0])[1] if scored else None
if match is None:
stats["no_matching_product"] += 1
continue
if match.get("rating") is not None and Decimal(0) < match["rating"] <= Decimal(5):
repo.update_listing_rating(row["listing_id"], match["rating"], match.get("review_count"))
stats["rated"] += 1
if match.get("reviews"):
stats["reviews_stored"] += repo.save_reviews(row["listing_id"], match["reviews"])
typer.echo(f" {row['domain']:22} rating={match.get('rating')} reviews={len(match.get('reviews') or [])}")
except Exception as exc: # noqa: BLE001
status, error = "failed", repr(exc)
raise
finally:
client.close()
repo.finish_run(run_id, status, stats, error)
stats = refresh_reviews(category=category, limit=limit, budget=budget, discover=not no_discover,
progress=typer.echo)
typer.echo(json.dumps(stats, indent=2))

View File

@@ -36,6 +36,7 @@ from urllib.parse import urlparse
from rapidfuzz import fuzz
from app.electronics.db import repository as repo
from app.electronics.extract.embedded_ratings import IGNORED_DOMAINS, embedded_rating
from app.electronics.extract.html_fallback import extract_page, spec_tables, visible_text
from app.electronics.extract.jsonld import extract_products
from app.electronics.extract.serp_parser import clean_result_title, read_price, read_rating, read_stock
@@ -371,6 +372,13 @@ class Collector:
html: str, final_url: str) -> Optional[Listing]:
products = extract_products(html)
page = extract_page(html)
if site.kind == "brand_official":
# A brand's own store names products without the brand ("Galaxy A56 5G
# (8 GB Memory)"); on its own site the brand is not in doubt.
brand_name = self.ref.brands[parsed_hit.brand.brand_slug].name
for p in products:
if not p["name"].lower().startswith(brand_name.lower()):
p["name"] = f"{brand_name} {p['name']}"
name = None
product = None
for p in products:
@@ -428,6 +436,17 @@ class Collector:
listing.review_count = product.get("review_count")
listing.reviews = list(product.get("reviews") or [])
listing.colour = listing.colour or product.get("color")
# Retailers that embed their rating outside JSON-LD (page's own product
# only): the rating when JSON-LD has none, and the star breakdown.
embedded = embedded_rating(site.domain, html, final_url or hit.url)
if embedded:
if listing.rating is None:
listing.rating, listing.review_count = embedded["rating"], embedded["review_count"]
listing.rating_breakdown = embedded["breakdown"]
if site.domain in IGNORED_DOMAINS:
# Known placeholder rating markup: nothing rating-shaped is kept.
listing.rating = listing.review_count = listing.rating_breakdown = None
listing.reviews = []
raw_specs = dict((product or {}).get("properties") or {})
raw_specs.update({k: v for k, v in spec_tables(BeautifulSoup(html, "lxml")).items() if k not in raw_specs})
listing.specs_raw = dict(list(raw_specs.items())[:150])

View File

@@ -0,0 +1,4 @@
-- Star breakdown of a listing's ratings, exactly as the source states it:
-- {"5": 612, "4": 300, "3": 80, "2": 20, "1": 12}. NULL when the source does
-- not publish one - never derived from the few review texts we hold.
ALTER TABLE elec.source_listing ADD COLUMN rating_breakdown JSONB;

View File

@@ -199,7 +199,8 @@ def upsert_listing(listing: Listing, ids: Dict[str, Dict[str, int]], run_id: Opt
model_number=listing.model_number, ram_gb=listing.ram_gb, storage_gb=listing.storage_gb,
colour=listing.colour, price=listing.price, mrp=listing.mrp, availability=listing.availability,
in_stock=listing.in_stock, pincode=listing.pincode, pincode_applied=listing.pincode_applied,
rating=listing.rating, review_count=listing.review_count, gtin=listing.gtin,
rating=listing.rating, review_count=listing.review_count,
rating_breakdown=_json(listing.rating_breakdown) if listing.rating_breakdown else None, gtin=listing.gtin,
image_urls=listing.image_urls[:12], specs_raw=_json(listing.specs_raw), specs=_json(listing.specs),
evidence_text=listing.evidence_text[:4000], search_query=listing.search_query,
confidence=round(listing.confidence, 2), parser=listing.parser, content_hash=listing.content_hash,
@@ -210,13 +211,13 @@ def upsert_listing(listing: Listing, ids: Dict[str, Dict[str, int]], run_id: Opt
INSERT INTO elec.source_listing (
site_id, source_sku, source_url, source_type, brand_id, category_id, family, title, model,
model_number, ram_gb, storage_gb, colour, price, mrp, availability, in_stock, pincode,
pincode_applied, rating, review_count, gtin, image_urls, specs_raw, specs, evidence_text,
pincode_applied, rating, review_count, rating_breakdown, gtin, image_urls, specs_raw, specs, evidence_text,
search_query, confidence, parser, content_hash, crawl_run_id)
VALUES (
%(site_id)s, %(source_sku)s, %(source_url)s, %(source_type)s, %(brand_id)s, %(category_id)s,
%(family)s, %(title)s, %(model)s, %(model_number)s, %(ram_gb)s, %(storage_gb)s, %(colour)s,
%(price)s, %(mrp)s, %(availability)s, %(in_stock)s, %(pincode)s, %(pincode_applied)s,
%(rating)s, %(review_count)s, %(gtin)s, %(image_urls)s, %(specs_raw)s, %(specs)s,
%(rating)s, %(review_count)s, %(rating_breakdown)s, %(gtin)s, %(image_urls)s, %(specs_raw)s, %(specs)s,
%(evidence_text)s, %(search_query)s, %(confidence)s, %(parser)s, %(content_hash)s,
%(crawl_run_id)s)
ON CONFLICT (site_id, source_sku) DO UPDATE SET
@@ -227,7 +228,7 @@ def upsert_listing(listing: Listing, ids: Dict[str, Dict[str, int]], run_id: Opt
price = EXCLUDED.price, mrp = EXCLUDED.mrp, availability = EXCLUDED.availability,
in_stock = EXCLUDED.in_stock, pincode = EXCLUDED.pincode,
pincode_applied = EXCLUDED.pincode_applied, rating = EXCLUDED.rating,
review_count = EXCLUDED.review_count, gtin = EXCLUDED.gtin, image_urls = EXCLUDED.image_urls,
review_count = EXCLUDED.review_count, rating_breakdown = EXCLUDED.rating_breakdown, gtin = EXCLUDED.gtin, image_urls = EXCLUDED.image_urls,
specs_raw = EXCLUDED.specs_raw, specs = EXCLUDED.specs, evidence_text = EXCLUDED.evidence_text,
search_query = EXCLUDED.search_query, confidence = EXCLUDED.confidence, parser = EXCLUDED.parser,
content_hash = EXCLUDED.content_hash, crawl_run_id = EXCLUDED.crawl_run_id, last_seen_at = now()
@@ -276,12 +277,13 @@ def save_reviews(listing_id: int, reviews: List[Dict[str, Any]]) -> int:
return added
def update_listing_rating(listing_id: int, rating: Optional[Decimal], review_count: Optional[int]) -> None:
def update_listing_rating(listing_id: int, rating: Optional[Decimal], review_count: Optional[int],
breakdown: Optional[Dict[str, int]] = None) -> None:
"""Refresh only the rating fields of a listing (used by the review backfill)."""
with transaction() as conn:
conn.execute(
"UPDATE elec.source_listing SET rating = %s, review_count = %s WHERE id = %s",
(rating, review_count, listing_id),
"UPDATE elec.source_listing SET rating = %s, review_count = %s, rating_breakdown = %s WHERE id = %s",
(rating, review_count, _json(breakdown) if breakdown else None, listing_id),
)
@@ -289,7 +291,7 @@ def product_rating_and_reviews(conn, product_id: int) -> Dict[str, Any]:
"""Per-platform ratings and all stored reviews for a verified product's
approved listings, each with the page it was read from."""
sources = conn.execute(
"SELECT a.site, a.source_url, l.rating, l.review_count FROM elec.v_product_availability a "
"SELECT a.site, a.source_url, l.rating, l.review_count, l.rating_breakdown FROM elec.v_product_availability a "
"JOIN elec.source_listing l ON l.id = a.listing_id "
"WHERE a.product_id = %s AND l.rating > 0 ORDER BY l.review_count DESC NULLS LAST, a.site",
(product_id,),

View File

@@ -0,0 +1,95 @@
"""Ratings a retailer embeds in its own product-page HTML (not in JSON-LD).
Only the page we already fetched is read - no extra requests - and only the
record of the page's OWN product (never recommendations on the same page).
Zero values mean "no ratings yet" and are dropped. Review text is not available
here: those retailers load it from feeds their robots.txt does not allow.
Deliberately absent: vasanthandco.in. Its product pages carry an identical
"3.3 average, 3 reviews" template block on every product - placeholder
content, not ratings - so nothing is ever read from that domain.
"""
from __future__ import annotations
import json
import re
from decimal import Decimal, InvalidOperation
from typing import Any, Callable, Dict, Optional
from urllib.parse import urlparse
from bs4 import BeautifulSoup
# Domains whose rating-looking markup is known to be fake/placeholder.
IGNORED_DOMAINS = frozenset({"vasanthandco.in"})
def _dec(value: Any) -> Optional[Decimal]:
try:
d = Decimal(str(value))
except (InvalidOperation, TypeError, ValueError):
return None
return d if Decimal(0) < d <= Decimal(5) else None
def _int(value: Any) -> Optional[int]:
try:
n = int(value)
except (TypeError, ValueError):
return None
return n if n > 0 else None
def _reading(rating: Any, count: Any, breakdown: Optional[Dict[str, int]] = None) -> Optional[Dict[str, Any]]:
r = _dec(rating)
if r is None:
return None
clean = None
if breakdown:
clean = {str(k): int(v) for k, v in breakdown.items()
if str(k) in {"1", "2", "3", "4", "5"} and _int(v)}
return {"rating": r.quantize(Decimal("0.01")), "review_count": _int(count), "breakdown": clean or None}
def _reliance(html: str, url: str) -> Optional[Dict[str, Any]]:
"""window.__INITIAL_STATE__.productDetailsPage.product._custom_json._app"""
marker = "window.__INITIAL_STATE__="
i = html.find(marker)
if i < 0:
return None
try:
state, _ = json.JSONDecoder().raw_decode(html, i + len(marker))
app = state["productDetailsPage"]["product"]["_custom_json"]["_app"]
except (ValueError, KeyError, TypeError):
return None
if not isinstance(app, dict):
return None
return _reading(app.get("averageRating"), app.get("ratingsCount"), app.get("ratingsCountDetails"))
def _poorvika(html: str, url: str) -> Optional[Dict[str, Any]]:
"""__NEXT_DATA__ props.pageProps.additionalData, only when its slug is this URL's."""
tag = BeautifulSoup(html, "lxml").find("script", id="__NEXT_DATA__")
if tag is None:
return None
try:
data = json.loads(tag.string or tag.get_text())["props"]["pageProps"]["additionalData"]
except (ValueError, KeyError, TypeError):
return None
slug = str(data.get("slug") or "")
if not slug or slug not in urlparse(url).path:
return None
return _reading(data.get("rating"), data.get("ratingCount"))
_READERS: Dict[str, Callable[[str, str], Optional[Dict[str, Any]]]] = {
"reliancedigital.in": _reliance,
"poorvika.com": _poorvika,
}
def embedded_rating(domain: str, html: str, url: str) -> Optional[Dict[str, Any]]:
"""{"rating": Decimal, "review_count": int|None, "breakdown": {"5": n, ...}|None} or None."""
if domain in IGNORED_DOMAINS:
return None
reader = _READERS.get(domain)
return reader(html, url) if reader else None

View File

@@ -161,42 +161,67 @@ def _reviews(node: dict) -> List[Dict[str, Any]]:
return out
def _rating_only_nodes(nodes: List[dict]) -> Dict[Optional[str], dict]:
"""Nameless Product nodes that only carry ratings/reviews, keyed by @id.
Review widgets (Bazaarvoice on samsung.com/in, for one) publish the
product's aggregateRating and reviews as a SEPARATE Product node with no
name, linked to the real product by the same @id. Without merging, those
real reviews would be skipped for lack of a name.
"""
out: Dict[Optional[str], dict] = {}
for node in nodes:
if _text(node.get("name")):
continue
if not (isinstance(node.get("aggregateRating"), dict) or node.get("review") or node.get("reviews")):
continue
out.setdefault(_text(node.get("@id")), node)
return out
def extract_products(html: str) -> List[Dict[str, Any]]:
"""All schema.org Product nodes on the page, flattened to plain fields."""
products: List[Dict[str, Any]] = []
for block in json_ld_blocks(html):
for node in _walk(block):
if not (_types(node) & _PRODUCT_TYPES):
continue
name = _text(node.get("name"))
if not name:
continue
offer = _offer(node.get("offers")) if node.get("offers") else {}
if not offer and isinstance(node.get("hasVariant"), list):
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
props = {}
for p in node.get("additionalProperty") or []:
if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""):
props[str(p["name"])] = str(p["value"])
rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {}
products.append({
"name": name,
"brand": _text(node.get("brand")),
"sku": _text(node.get("sku")) or _text(node.get("productID")),
"mpn": _text(node.get("mpn")),
"gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None),
"color": _text(node.get("color")),
"images": _images(node.get("image")),
"description": _text(node.get("description")),
"price": offer.get("price"),
"currency": offer.get("currency"),
"availability": offer.get("availability"),
"in_stock": offer.get("in_stock"),
# Sites publish 0 for "no ratings yet"; that is not a rating.
"rating": (_dec(rating.get("ratingValue")) or None),
"review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None,
"reviews": _reviews(node),
"properties": props,
"evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500],
})
nodes = [n for block in json_ld_blocks(html) for n in _walk(block) if _types(n) & _PRODUCT_TYPES]
rating_nodes = _rating_only_nodes(nodes)
named = [n for n in nodes if _text(n.get("name"))]
for node in named:
# Attach a ratings-only node: same @id, or - when it has no @id - the
# page's only named product (there is then no doubt which it is about).
extra = rating_nodes.get(_text(node.get("@id"))) if node.get("@id") else None
if extra is None and len(named) == 1:
extra = rating_nodes.get(None)
if extra is not None:
node = {**node,
"aggregateRating": node.get("aggregateRating") or extra.get("aggregateRating"),
"review": node.get("review") or extra.get("review") or extra.get("reviews")}
name = _text(node.get("name"))
offer = _offer(node.get("offers")) if node.get("offers") else {}
if not offer and isinstance(node.get("hasVariant"), list):
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
props = {}
for p in node.get("additionalProperty") or []:
if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""):
props[str(p["name"])] = str(p["value"])
rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {}
products.append({
"name": name,
"brand": _text(node.get("brand")),
"sku": _text(node.get("sku")) or _text(node.get("productID")),
"mpn": _text(node.get("mpn")),
"gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None),
"color": _text(node.get("color")),
"images": _images(node.get("image")),
"description": _text(node.get("description")),
"price": offer.get("price"),
"currency": offer.get("currency"),
"availability": offer.get("availability"),
"in_stock": offer.get("in_stock"),
# Sites publish 0 for "no ratings yet"; that is not a rating.
"rating": (_dec(rating.get("ratingValue")) or None),
"review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None,
"reviews": _reviews(node),
"properties": props,
"evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500],
})
return products

View File

@@ -0,0 +1,80 @@
"""Vijay Sales ratings and reviews from its public GraphQL endpoint.
vijaysales.com serves product data to its own pages from GET /api/graphql;
robots.txt does not disallow it, and no login or token is involved. One GET
per listing, made through PoliteClient (robots, pacing, circuit breaker).
The product is looked up by url_key - the last path segment of the listing URL
we already hold - so the answer is about exactly that listing.
"""
from __future__ import annotations
import json
from decimal import Decimal, InvalidOperation
from typing import Any, Dict, List, Optional
from urllib.parse import urlencode, urlparse
from app.electronics.reviews import sentiment_for
ENDPOINT = "https://www.vijaysales.com/api/graphql"
_QUERY = (
'{products(filter:{url_key:{eq:"%s"}}){items{sku rating_summary review_count '
"reviews(pageSize:20){items{nickname summary text average_rating created_at}}}}}"
)
def url_key(listing_url: str) -> Optional[str]:
path = urlparse(listing_url).path.rstrip("/")
key = path.rsplit("/", 1)[-1] if path else ""
# url_keys are slugs; anything else would break out of the query string.
return key if key and all(c.isalnum() or c in "-_" for c in key) else None
def graphql_url(listing_url: str) -> Optional[str]:
key = url_key(listing_url)
return f"{ENDPOINT}?{urlencode({'query': _QUERY % key})}" if key else None
def _five(percent: Any) -> Optional[Decimal]:
"""Magento ratings are percentages (80 = 4 stars)."""
try:
p = Decimal(str(percent))
except (InvalidOperation, TypeError, ValueError):
return None
return (p / 20).quantize(Decimal("0.01")) if Decimal(0) < p <= Decimal(100) else None
def parse(body: str) -> Optional[Dict[str, Any]]:
"""{"rating", "review_count", "reviews": [...]} for the best-reviewed matching item, or None."""
try:
items = json.loads(body)["data"]["products"]["items"] or []
except (ValueError, KeyError, TypeError):
return None
items = [i for i in items if isinstance(i, dict)]
if not items:
return None
item = max(items, key=lambda i: i.get("review_count") or 0)
rating = _five(item.get("rating_summary"))
if rating is None:
return None
reviews: List[Dict[str, Any]] = []
for r in ((item.get("reviews") or {}).get("items") or []):
body_text = (r.get("text") or "").strip()
if not body_text:
continue
stars = _five(r.get("average_rating"))
title = (r.get("summary") or "").strip()
# When a customer leaves the title blank the site fills in the full
# product name; that is not something the reviewer wrote.
if len(title) > 60 or "/" in title:
title = ""
reviews.append({
"author": (r.get("nickname") or "").strip() or None,
"rating": stars,
"title": title or None,
"body": body_text[:4000],
"review_date": r.get("created_at"),
"sentiment": sentiment_for(stars),
})
return {"rating": rating, "review_count": int(item.get("review_count") or 0) or None, "reviews": reviews}

View File

@@ -34,6 +34,8 @@ class Listing:
pincode_applied: bool = False
rating: Optional[Decimal] = None
review_count: Optional[int] = None
# {"5": n, ..., "1": n} when the source publishes a star breakdown.
rating_breakdown: Optional[Dict[str, int]] = None
# Customer reviews the page itself publishes (schema.org Review); stored
# in elec.listing_review, not on the listing row.
reviews: List[Dict[str, Any]] = field(default_factory=list)

View File

@@ -24,6 +24,11 @@ class BrandRef:
aliases: tuple
sub_brands: tuple
official: tuple
# Brand-store path whose product pages publish real customer reviews as
# schema.org data (e.g. "samsung.com/in"); None when no free source exists.
reviews_site: Optional[str] = None
# Regex a reviews_site URL must match to be a single-variant product page.
reviews_product_url: Optional[str] = None
@dataclass(frozen=True)
@@ -78,6 +83,8 @@ def load_reference() -> Reference:
aliases=tuple(a.lower() for a in b.get("aliases", [])),
sub_brands=tuple(s.lower() for s in b.get("sub_brands", [])),
official=tuple(b.get("official", [])),
reviews_site=b.get("reviews_site"),
reviews_product_url=b.get("reviews_product_url"),
)
categories = {
c["slug"]: CategoryRef(c["slug"], c["name"], tuple(c.get("search_terms", [])),

View File

@@ -7,11 +7,19 @@
# parent, and are kept as the product family.
# official the brand's own Indian web domains. A product page on one of
# these is the strongest evidence that a product exists.
# reviews_site optional brand-store path whose product pages publish real
# customer ratings + reviews as schema.org JSON-LD, robots-allowed
# (checked 2026-10-01). `cli reviews` finds each verified product's
# page there. Add a brand only after checking its pages.
# reviews_product_url regex for that store's single-variant product pages
# (family/marketing pages carry no product data or reviews).
brands:
- name: Samsung
categories: [mobiles, laptops]
aliases: [samsung]
official: [samsung.com]
reviews_site: samsung.com/in
reviews_product_url: '-sm-[a-z0-9]+/?$' # .../galaxy-a56-5g-awesome-olive-256gb-sm-a566ezggins/
- name: Apple
categories: [mobiles, laptops]
aliases: [apple]

View File

@@ -0,0 +1,215 @@
"""Refresh customer ratings and reviews for the verified catalogue, from free
sources that publish them and allow reading them:
1. Brand stores with real reviews in schema.org JSON-LD (brands.yaml
`reviews_site`, e.g. samsung.com/in): find each verified product's own page
by web search, accept it only for the same model AND variant, read it with
the normal collector page path, and store it as a brand_official listing.
2. Every page-read listing of a verified product is re-read for its rating,
star breakdown and reviews (JSON-LD, then the retailer's embedded data).
3. Vijay Sales listings additionally get their reviews from the site's own
public GraphQL endpoint.
Everything goes through PoliteClient (robots.txt, per-site pacing, circuit
breaker). Nothing is generated; a source that states nothing leaves nothing.
"""
from __future__ import annotations
import dataclasses
import logging
import re
from decimal import Decimal
from typing import Callable, Dict, List, Optional
from rapidfuzz import fuzz
from app.electronics.collector import Collector, RunOptions
from app.electronics.db import repository as repo
from app.electronics.db.connection import connect
from app.electronics.extract import vijaysales_reviews
from app.electronics.extract.embedded_ratings import IGNORED_DOMAINS, embedded_rating
from app.electronics.extract.jsonld import extract_products
from app.electronics.reference import load_reference
logger = logging.getLogger(__name__)
def _verified_products(category: Optional[str]) -> List[dict]:
sql = ("SELECT v.product_id, v.brand_slug, v.category, p.model, p.model_norm, p.ram_gb, p.storage_gb "
"FROM elec.v_brand_catalog v JOIN elec.product p ON p.id = v.product_id")
params: tuple = ()
if category:
sql += " WHERE v.category = %s"
params = (category,)
with connect() as conn:
return list(conn.execute(sql + " ORDER BY v.product_id", params))
def _has_page_on(product_id: int, domain: str) -> bool:
with connect() as conn:
return conn.execute(
"SELECT 1 FROM elec.v_product_availability WHERE product_id = %s AND domain = %s "
"AND source_type IN ('scraped_page','brand_official') LIMIT 1", (product_id, domain),
).fetchone() is not None
def _gb(value) -> str:
return f"{format(Decimal(value).normalize(), 'f')}GB" if value is not None else ""
# Words that make a different model, not a different colour of the same one.
_MODEL_QUALIFIERS = frozenset({"ultra", "plus", "pro", "max", "fe", "lite", "edge", "mini", "neo",
"prime", "flip", "fold", "slim", "+"})
def _same_model(page_model: str, product_model: str) -> bool:
"""Same model line: every product word present, identical model-number
words, and nothing extra but non-model words (e.g. a colour the parser
left in: "galaxy a56 olive"). Fuzzy scores are no use here - they rate
"galaxy a57" vs "galaxy a56" at ~90."""
page, mine = set(page_model.split()), set(product_model.split())
numbered = lambda words: {w for w in words if any(c.isdigit() for c in w)} # noqa: E731
extra = page - mine
return mine <= page and numbered(page) == numbered(mine) and not (extra & _MODEL_QUALIFIERS)
def _same_variant(parsed, product: dict) -> bool:
if not parsed.model_norm or not _same_model(parsed.model_norm, product["model_norm"] or ""):
return False
for attr in ("ram_gb", "storage_gb"):
mine, theirs = product[attr], getattr(parsed, attr)
if mine is not None and theirs is not None and Decimal(mine) != Decimal(theirs):
return False
return True
def discover_brand_pages(category: Optional[str], budget: int, stats: Dict[str, int],
progress: Callable[[str], None]) -> None:
ref = load_reference()
by_category: Dict[str, List[dict]] = {}
for p in _verified_products(category):
brand = ref.brands.get(p["brand_slug"])
if brand and brand.reviews_site:
by_category.setdefault(p["category"], []).append(p)
for cat, products in by_category.items():
brands = sorted({p["brand_slug"] for p in products})
col = Collector(RunOptions(category=cat, brands=brands, search_budget=budget, use_llm=False),
progress=progress)
try:
for p in products:
brand = ref.brands[p["brand_slug"]]
domain = brand.reviews_site.split("/")[0]
if _has_page_on(p["product_id"], domain):
stats["brand_page_already_known"] = stats.get("brand_page_already_known", 0) + 1
continue
model = p["model"] or p["model_norm"]
queries = [f"site:{brand.reviews_site} {brand.name} {model} {_gb(p['ram_gb'])} {_gb(p['storage_gb'])}",
f"site:{brand.reviews_site} {model} 5G {_gb(p['storage_gb'])} buy"]
hits, query = [], ""
for q in (re.sub(r"\s+", " ", q).strip() for q in queries):
hits += [(h, q) for h in (col.engine.text(q, max_results=10) or [])]
seen = set()
for hit, query in hits:
if hit.url in seen:
continue
seen.add(hit.url)
# Brand-store titles often omit the brand ("Galaxy A56 5G 8GB/256GB ...",
# sometimes behind a "Business |" prefix). On the brand's own store
# the brand is not in doubt, so state it for the title parser.
title = re.sub(r"^\s*Business\s*\|\s*", "", hit.title or "")
if not title.lower().startswith(brand.name.lower()):
title = f"{brand.name} {title}"
hit = dataclasses.replace(hit, title=title)
if f"{brand.reviews_site}/" not in hit.url:
continue
if brand.reviews_product_url and not re.search(brand.reviews_product_url, hit.url.split("?")[0]):
continue # a family/marketing page, not one variant's product page
accepted = col._accept_hit(hit, brand)
if not accepted:
continue
site, parsed = accepted
if not _same_variant(parsed, p):
continue
res = col.client.get(hit.url)
if not res.ok:
continue
listing = col.listing_from_page(hit, site, parsed, query, res.text, res.final_url)
# Only a page with schema.org product data for this exact variant.
if listing is None or not listing.parser.startswith("jsonld") or not _same_variant(listing, p):
continue
col.store(listing)
if _has_page_on(p["product_id"], domain): # actually matched to this product
stats["brand_pages_added"] = stats.get("brand_pages_added", 0) + 1
progress(f" {brand.name}: {listing.title} -> rating {listing.rating}, "
f"{len(listing.reviews)} reviews")
break
finally:
col.client.close()
repo.refresh_verification()
def reread_pages(category: Optional[str], limit: int, stats: Dict[str, int],
progress: Callable[[str], None]) -> None:
from app.electronics.net.polite_client import PoliteClient
rows = repo.listings_for_review_backfill(category)[:limit]
with PoliteClient() as client:
for row in rows:
domain = row["domain"]
if domain in IGNORED_DOMAINS:
continue
stats["pages"] = stats.get("pages", 0) + 1
rating = count = breakdown = None
reviews: List[dict] = []
res = client.get(row["source_url"])
if res.ok:
products = extract_products(res.text)
match = next((x for x in products if x.get("sku") and x["sku"] == row["source_sku"]), None)
if match is None:
title = (row["title"] or "").lower()
scored = [(fuzz.token_set_ratio(x["name"].lower(), title), x) for x in products]
scored = [s for s in scored if s[0] >= 85]
match = max(scored, key=lambda s: s[0])[1] if scored else None
if match is not None:
rating, count, reviews = match.get("rating"), match.get("review_count"), match.get("reviews") or []
# The retailer's own embedded data: the rating when JSON-LD has none,
# and the star breakdown, which only it publishes.
emb = embedded_rating(domain, res.text, res.final_url or row["source_url"])
if emb:
if rating is None:
rating, count = emb["rating"], emb["review_count"]
breakdown = emb["breakdown"]
if domain == "vijaysales.com":
gql = vijaysales_reviews.graphql_url(row["source_url"])
vres = client.get(gql, accept_non_html=True) if gql else None
vs = vijaysales_reviews.parse(vres.text) if vres is not None and vres.ok else None
if vs:
rating, count = rating or vs["rating"], count or vs["review_count"]
reviews = reviews or vs["reviews"]
if rating is not None and Decimal(0) < Decimal(rating) <= Decimal(5):
repo.update_listing_rating(row["listing_id"], rating, count, breakdown)
stats["rated"] = stats.get("rated", 0) + 1
if reviews:
stats["reviews_stored"] = stats.get("reviews_stored", 0) + repo.save_reviews(row["listing_id"], reviews)
progress(f" {domain:22} rating={rating} count={count} breakdown={'yes' if breakdown else 'no'} "
f"reviews={len(reviews)}")
def refresh_reviews(category: Optional[str] = None, limit: int = 500, budget: int = 60,
discover: bool = True, progress: Optional[Callable[[str], None]] = None) -> Dict[str, int]:
progress = progress or (lambda m: logger.info(m))
stats: Dict[str, int] = {}
run_id = repo.start_run("reviews", {"category": category, "limit": limit, "discover": discover})
status, error = "done", None
try:
if discover:
progress("Finding brand-store pages with reviews ...")
discover_brand_pages(category, budget, stats, progress)
progress("Re-reading ratings and reviews from product pages ...")
reread_pages(category, limit, stats, progress)
except Exception as exc:
status, error = "failed", repr(exc)
raise
finally:
repo.finish_run(run_id, status, stats, error)
return stats

View File

@@ -199,7 +199,8 @@ USER_AGENT = os.getenv(
REQUEST_TIMEOUT_SECONDS = int(os.getenv("REQUEST_TIMEOUT_SECONDS", "20"))
ELEC_SITE_MIN_INTERVAL_SECONDS = float(os.getenv("ELEC_SITE_MIN_INTERVAL_SECONDS", "3"))
ELEC_BREAKER_COOLDOWN_HOURS = float(os.getenv("ELEC_BREAKER_COOLDOWN_HOURS", "24"))
ELEC_MAX_PAGE_BYTES = int(os.getenv("ELEC_MAX_PAGE_BYTES", str(3 * 1024 * 1024)))
# Brand stores embed their reviews in the page: samsung.com/in pages run ~3.1 MB.
ELEC_MAX_PAGE_BYTES = int(os.getenv("ELEC_MAX_PAGE_BYTES", str(6 * 1024 * 1024)))
ELEC_PROBE_TTL_DAYS = int(os.getenv("ELEC_PROBE_TTL_DAYS", "7"))
MIN_IMAGE_BYTES = int(os.getenv("MIN_IMAGE_BYTES", "3000"))

View File

@@ -92,8 +92,9 @@ async def get_product(product_id: int) -> Dict[str, Any]:
"""Full details of one product by its product_id (from search_products).
Returns per-platform offers (price, MRP, source URL, when seen), normalised specs,
image URLs, the overall rating with per-platform sources (null if none is published),
and up to 10 real customer reviews (often empty).
image URLs, the overall rating with per-platform sources and, when a platform publishes
one, a 5-to-1 star breakdown (rating is null if no platform publishes a rating), and up
to 10 real customer reviews with the site each came from (often empty).
"""
d = await _run(elec.product, int(product_id))
return {

View File

@@ -142,3 +142,155 @@ def test_api_serves_ratings_reviews_and_out_of_stock_price(db, client):
assert {s["site"] for s in detail["rating"]["sources"]} == {"Amazon.in", "Croma"}
assert [r["body"] for r in detail["reviews"]] == ["Battery lasts all day.", "Heats up."]
assert all(r["source_url"].startswith("https://") for r in detail["reviews"])
assert detail["rating"]["breakdown"] is None # no platform published one
# ---------------------------------------------------------------------------
# Free sources: brand-store JSON-LD, retailer-embedded ratings, Vijay Sales
# ---------------------------------------------------------------------------
def _ld(*nodes) -> str:
return "".join(f'<script type="application/ld+json">{json.dumps(n)}</script>' for n in nodes)
def _review(author, stars, body, date="2026-09-30T17:06:05.000+00:00"):
return {"@type": "Review", "author": {"@type": "Person", "name": author}, "reviewBody": body,
"datePublished": date, "reviewRating": {"@type": "Rating", "bestRating": 5, "ratingValue": stars}}
def test_nameless_review_node_is_merged_by_id():
# The samsung.com/in shape: Bazaarvoice publishes ratings in a second,
# nameless Product node that shares the real product's @id.
pid = "https://www.samsung.com/in/smartphones/galaxy-a/galaxy-a56-5g-awesome-olive-256gb-sm-a566ezggins/"
html = _ld(
{"@context": "https://schema.org", "@type": "Product", "@id": pid, "name": "Galaxy A56 5G (8 GB Memory)",
"sku": "SM-A566EZGG", "offers": {"@type": "Offer", "price": "48999", "priceCurrency": "INR"}},
{"@context": "https://schema.org", "@type": "Product", "@id": pid,
"aggregateRating": {"@type": "AggregateRating", "ratingValue": "4.5", "ratingCount": 1224},
"review": [_review("Vedant", 5, "Great and useful AI features"), _review("Ravi", 1, "Heats up")]},
)
[p] = extract_products(html)
assert p["name"] == "Galaxy A56 5G (8 GB Memory)" and p["price"] == Decimal("48999")
assert p["rating"] == Decimal("4.5") and p["review_count"] == 1224
assert [(r["author"], r["rating"], r["body"]) for r in p["reviews"]] == [
("Vedant", Decimal("5.0"), "Great and useful AI features"), ("Ravi", Decimal("1.0"), "Heats up")]
def test_nameless_review_node_without_id_is_not_guessed_between_products():
lone = {"@type": "Product", "aggregateRating": {"ratingValue": "4.1", "ratingCount": 9}}
[only] = extract_products(_ld({"@type": "Product", "name": "Galaxy A56"}, lone))
assert only["rating"] == Decimal("4.1") # one product on the page: unambiguous
two = extract_products(_ld({"@type": "Product", "name": "Galaxy A56"},
{"@type": "Product", "name": "Galaxy A36"}, lone))
assert all(p["rating"] is None for p in two) # two products: never guess which
def test_reliance_embedded_rating_is_read_from_the_pages_own_product_only():
from app.electronics.extract.embedded_ratings import embedded_rating
def page(app, recommended=None):
state = {"productDetailsPage": {"product": {"name": "Dell 15", "_custom_json": {"_app": app}}},
"recommendations": [{"_custom_json": {"_app": recommended}}] if recommended else []}
return f"<script>window.__INITIAL_STATE__={json.dumps(state)};window.x=1</script>"
url = "https://www.reliancedigital.in/product/dell-15-9991568"
got = embedded_rating("reliancedigital.in", page(
{"averageRating": 4, "ratingsCount": 3, "ratingsCountDetails": {"5": 1, "4": 1, "3": 1}}), url)
assert got == {"rating": Decimal("4.00"), "review_count": 3, "breakdown": {"5": 1, "4": 1, "3": 1}}
# "No ratings yet" is published as zeros; a recommended product's rating is not ours.
zero = page({"averageRating": 0, "ratingsCount": 0}, recommended={"averageRating": 4.8, "ratingsCount": 90})
assert embedded_rating("reliancedigital.in", zero, url) is None
def test_poorvika_rating_needs_the_pages_own_slug():
from app.electronics.extract.embedded_ratings import embedded_rating
def page(slug, rating, count):
data = {"props": {"pageProps": {"additionalData": {"slug": slug, "rating": rating, "ratingCount": count}}}}
return f'<script id="__NEXT_DATA__" type="application/json">{json.dumps(data)}</script>'
slug = "xiaomi-14-civi-5g-shadow-black-256gb-8gb-ram"
url = f"https://www.poorvika.com/{slug}/p"
assert embedded_rating("poorvika.com", page(slug, 5, 1), url)["rating"] == Decimal("5.00")
assert embedded_rating("poorvika.com", page("some-other-phone", 5, 1), url) is None
assert embedded_rating("poorvika.com", page(slug, None, 0), url) is None
def test_vasanth_template_rating_is_never_read():
from app.electronics.extract.embedded_ratings import embedded_rating
template = ('<h4 class="avg-mark">3.3</h4><a class="rating-reviews">(3 Reviews)</a>'
+ _ld({"@type": "Product", "name": "HP 15", "aggregateRating": {"ratingValue": "3.3", "reviewCount": 3}}))
assert embedded_rating("vasanthandco.in", template, "https://vasanthandco.in/product/174200102203/") is None
def test_vijaysales_graphql_review_feed():
from app.electronics.extract import vijaysales_reviews as vs
url = "https://www.vijaysales.com/p/239904/acer-aspire-lite-al15-41-laptop-amd-ryzen-7-16gb"
assert vs.url_key(url) == "acer-aspire-lite-al15-41-laptop-amd-ryzen-7-16gb"
assert "url_key" in vs.graphql_url(url)
assert vs.url_key('https://x.in/p/a"b') is None # cannot break out of the query
product_name = "Acer Aspire Lite AL15-41 Laptop (AMD Ryzen 7/ 16GB DDR4 RAM/ 512GB SSD)"
body = json.dumps({"data": {"products": {"items": [
{"sku": "P1", "rating_summary": 0, "review_count": 0, "reviews": {"items": []}},
{"sku": "239904", "rating_summary": 90, "review_count": 2, "reviews": {"items": [
{"nickname": "Jay", "summary": product_name, "text": "All over performance is good",
"average_rating": 100, "created_at": "2026-05-02 10:00:00"},
{"nickname": "guest user", "summary": "Okay", "text": "", "average_rating": 60}]}}]}}})
got = vs.parse(body)
assert got["rating"] == Decimal("4.50") and got["review_count"] == 2
# Product-name "title" dropped; a review with no text is skipped.
assert got["reviews"] == [{"author": "Jay", "rating": Decimal("5.00"), "title": None,
"body": "All over performance is good", "review_date": "2026-05-02 10:00:00",
"sentiment": "positive"}]
assert vs.parse('{"data":{"products":{"items":[]}}}') is None
def test_api_breakdown_only_when_a_platform_publishes_one(db, client):
from app.electronics.collector import Collector, RunOptions, RunStats
from app.electronics.db import repository as repo
from app.electronics.models import Listing
from app.electronics.normalise.title_parser import parse_title, variant_key
def listing(site, sku, price, source_type="search_snippet"):
title = "Samsung Galaxy S24 5G (8GB RAM, 256GB)"
p = parse_title(title, "mobiles")
l = Listing(site_domain=site, source_sku=sku, source_url=f"https://www.{site}/p/{sku}",
source_type=source_type, brand_slug="samsung", category="mobiles", title=title,
evidence_text=f"{title} ₹{price}", confidence=0.5, parser="test",
model=p.model, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=Decimal(price))
l.model_norm, l.variant_key = p.model_norm, variant_key(p, "mobiles")
return l
c = Collector.__new__(Collector)
c.opt = RunOptions(category="mobiles", brands=["samsung"])
c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats()
b = listing("reliancedigital.in", "rd-1", 73999, "scraped_page")
b.rating, b.review_count, b.rating_breakdown = Decimal("4.0"), 4, {"5": 2, "4": 1, "1": 1}
c.store(listing("amazon.in", "B0CS5XW6TN", 74999))
c.store(b)
repo.refresh_verification()
pid = client.get("/api/elec/products", params={"category": "mobiles"}).json()["products"][0]["product_id"]
rating = client.get(f"/api/elec/products/{pid}").json()["rating"]
assert rating["breakdown"] == [
{"stars": 5, "count": 2, "percent": 50}, {"stars": 4, "count": 1, "percent": 25},
{"stars": 3, "count": 0, "percent": 0}, {"stars": 2, "count": 0, "percent": 0},
{"stars": 1, "count": 1, "percent": 25}]
assert rating["sources"][0]["breakdown"] == {"5": 2, "4": 1, "1": 1}
def test_brand_store_page_must_be_the_exact_model_and_variant():
from app.electronics.normalise.title_parser import parse_title
from app.electronics.review_refresh import _same_variant
a56 = {"model_norm": "galaxy a56", "ram_gb": Decimal("8"), "storage_gb": Decimal("256")}
page = lambda t: parse_title(t, "mobiles", expected_brand="samsung") # noqa: E731
assert _same_variant(page("Samsung Galaxy A56 5G 8GB/256GB (Olive)"), a56)
# A neighbouring model scores ~90 on fuzzy matching; it must still be refused.
assert not _same_variant(page("Samsung Galaxy A57 5G 8GB/256GB (Navy)"), a56)
assert not _same_variant(page("Samsung Galaxy A56 5G 12GB/256GB (Olive)"), a56)
assert not _same_variant(page("Samsung Galaxy A56 5G 8GB/128GB (Blue)"), a56)
s25 = {"model_norm": "galaxy s25", "ram_gb": Decimal("12"), "storage_gb": Decimal("256")}
assert _same_variant(page("Samsung Galaxy S25 12GB/256GB (Icyblue)"), s25)
assert not _same_variant(page("Samsung Galaxy S25 Ultra 12GB/256GB (Titanium Black)"), s25)
assert not _same_variant(page("Samsung Galaxy S25+ 12GB/256GB (Navy)"), s25)