Customer Rating and review changes
This commit is contained in:
@@ -141,23 +141,50 @@ def product(product_id: int) -> dict:
|
||||
def _overall_rating(sources: List[dict]) -> Optional[dict]:
|
||||
"""The product's rating across the platforms that state one: the mean
|
||||
weighted by each platform's rating count (a platform that states no count
|
||||
weighs as 1). None when no platform states a rating - never a guess."""
|
||||
weighs as 1). None when no platform states a rating - never a guess.
|
||||
|
||||
One reading per platform: a site with a page per colour repeats the same
|
||||
model rating on each, and adding those up would multiply its count."""
|
||||
if not sources:
|
||||
return None
|
||||
weight = lambda s: max(int(s["review_count"] or 0), 1) # noqa: E731
|
||||
per_site: Dict[str, dict] = {}
|
||||
for s in sources:
|
||||
if s["site"] not in per_site or weight(s) > weight(per_site[s["site"]]):
|
||||
per_site[s["site"]] = s
|
||||
sources = list(per_site.values())
|
||||
total = sum(weight(s) for s in sources)
|
||||
value = sum(Decimal(s["rating"]) * weight(s) for s in sources) / total
|
||||
counts = [s["review_count"] for s in sources if s["review_count"]]
|
||||
return {
|
||||
"value": round(float(value), 1),
|
||||
"count": sum(counts) if counts else None,
|
||||
"breakdown": _breakdown(sources),
|
||||
"sources": [
|
||||
{"site": s["site"], "rating": float(s["rating"]), "review_count": s["review_count"], "source_url": s["source_url"]}
|
||||
{"site": s["site"], "rating": float(s["rating"]), "review_count": s["review_count"],
|
||||
"breakdown": s.get("rating_breakdown"), "source_url": s["source_url"]}
|
||||
for s in sources
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def _breakdown(sources: List[dict]) -> Optional[List[dict]]:
|
||||
"""5→1 star counts summed over the platforms that PUBLISH a breakdown.
|
||||
None when none does - never estimated from the few review texts held."""
|
||||
totals = {str(n): 0 for n in range(1, 6)}
|
||||
stated = False
|
||||
for s in sources:
|
||||
for star, n in (s.get("rating_breakdown") or {}).items():
|
||||
if star in totals and isinstance(n, int) and n > 0:
|
||||
totals[star] += n
|
||||
stated = True
|
||||
if not stated:
|
||||
return None
|
||||
total = sum(totals.values())
|
||||
return [{"stars": int(k), "count": totals[k], "percent": round(100 * totals[k] / total)}
|
||||
for k in ("5", "4", "3", "2", "1")]
|
||||
|
||||
|
||||
@router.get("/products/{product_id}/price-history")
|
||||
def price_history(product_id: int) -> List[dict]:
|
||||
with connect() as conn:
|
||||
|
||||
@@ -143,58 +143,22 @@ def rematch(category: str = typer.Option(..., help="mobiles | laptops"),
|
||||
|
||||
@app.command()
|
||||
def reviews(category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)"),
|
||||
limit: int = typer.Option(200, help="Max product pages to re-read"),
|
||||
limit: int = typer.Option(500, help="Max product pages to re-read"),
|
||||
budget: int = typer.Option(60, help="Max web searches for finding brand-store pages"),
|
||||
no_discover: bool = typer.Option(False, "--no-discover", help="Skip finding brand-store pages"),
|
||||
verbose: bool = False) -> None:
|
||||
"""Re-read ratings and customer reviews from the product pages already on file.
|
||||
"""Refresh real customer ratings and reviews from free, robots-allowed sources.
|
||||
|
||||
Only pages the collector itself reads (scraped / brand official listings)
|
||||
are fetched, politely (robots.txt, per-site pacing, circuit breaker). A
|
||||
rating or review is stored only when the page's own schema.org data states
|
||||
it; nothing is generated.
|
||||
Finds each verified product's page on brand stores that publish reviews
|
||||
(brands.yaml `reviews_site`), then re-reads every readable product page for
|
||||
its rating, star breakdown and reviews, plus Vijay Sales' public review
|
||||
feed. A rating or review is stored only when its source states it.
|
||||
"""
|
||||
_setup_logging(verbose)
|
||||
from rapidfuzz import fuzz
|
||||
from app.electronics.review_refresh import refresh_reviews
|
||||
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.extract.jsonld import extract_products
|
||||
from app.electronics.net.polite_client import PoliteClient
|
||||
|
||||
rows = repo.listings_for_review_backfill(category)[:limit]
|
||||
run_id = repo.start_run("reviews", {"category": category, "pages": len(rows)})
|
||||
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
|
||||
r.outcome, r.robots_allowed))
|
||||
stats = {"pages": 0, "pages_ok": 0, "rated": 0, "reviews_stored": 0, "no_matching_product": 0}
|
||||
status, error = "done", None
|
||||
try:
|
||||
for row in rows:
|
||||
stats["pages"] += 1
|
||||
res = client.get(row["source_url"])
|
||||
if not res.ok:
|
||||
continue
|
||||
stats["pages_ok"] += 1
|
||||
products = extract_products(res.text)
|
||||
# The same product the listing was stored from: its SKU, else its name.
|
||||
match = next((p for p in products if p.get("sku") and p["sku"] == row["source_sku"]), None)
|
||||
if match is None:
|
||||
title = (row["title"] or "").lower()
|
||||
scored = [(fuzz.token_set_ratio(p["name"].lower(), title), p) for p in products]
|
||||
scored = [sp for sp in scored if sp[0] >= 85]
|
||||
match = max(scored, key=lambda sp: sp[0])[1] if scored else None
|
||||
if match is None:
|
||||
stats["no_matching_product"] += 1
|
||||
continue
|
||||
if match.get("rating") is not None and Decimal(0) < match["rating"] <= Decimal(5):
|
||||
repo.update_listing_rating(row["listing_id"], match["rating"], match.get("review_count"))
|
||||
stats["rated"] += 1
|
||||
if match.get("reviews"):
|
||||
stats["reviews_stored"] += repo.save_reviews(row["listing_id"], match["reviews"])
|
||||
typer.echo(f" {row['domain']:22} rating={match.get('rating')} reviews={len(match.get('reviews') or [])}")
|
||||
except Exception as exc: # noqa: BLE001
|
||||
status, error = "failed", repr(exc)
|
||||
raise
|
||||
finally:
|
||||
client.close()
|
||||
repo.finish_run(run_id, status, stats, error)
|
||||
stats = refresh_reviews(category=category, limit=limit, budget=budget, discover=not no_discover,
|
||||
progress=typer.echo)
|
||||
typer.echo(json.dumps(stats, indent=2))
|
||||
|
||||
|
||||
|
||||
@@ -36,6 +36,7 @@ from urllib.parse import urlparse
|
||||
from rapidfuzz import fuzz
|
||||
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.extract.embedded_ratings import IGNORED_DOMAINS, embedded_rating
|
||||
from app.electronics.extract.html_fallback import extract_page, spec_tables, visible_text
|
||||
from app.electronics.extract.jsonld import extract_products
|
||||
from app.electronics.extract.serp_parser import clean_result_title, read_price, read_rating, read_stock
|
||||
@@ -371,6 +372,13 @@ class Collector:
|
||||
html: str, final_url: str) -> Optional[Listing]:
|
||||
products = extract_products(html)
|
||||
page = extract_page(html)
|
||||
if site.kind == "brand_official":
|
||||
# A brand's own store names products without the brand ("Galaxy A56 5G
|
||||
# (8 GB Memory)"); on its own site the brand is not in doubt.
|
||||
brand_name = self.ref.brands[parsed_hit.brand.brand_slug].name
|
||||
for p in products:
|
||||
if not p["name"].lower().startswith(brand_name.lower()):
|
||||
p["name"] = f"{brand_name} {p['name']}"
|
||||
name = None
|
||||
product = None
|
||||
for p in products:
|
||||
@@ -428,6 +436,17 @@ class Collector:
|
||||
listing.review_count = product.get("review_count")
|
||||
listing.reviews = list(product.get("reviews") or [])
|
||||
listing.colour = listing.colour or product.get("color")
|
||||
# Retailers that embed their rating outside JSON-LD (page's own product
|
||||
# only): the rating when JSON-LD has none, and the star breakdown.
|
||||
embedded = embedded_rating(site.domain, html, final_url or hit.url)
|
||||
if embedded:
|
||||
if listing.rating is None:
|
||||
listing.rating, listing.review_count = embedded["rating"], embedded["review_count"]
|
||||
listing.rating_breakdown = embedded["breakdown"]
|
||||
if site.domain in IGNORED_DOMAINS:
|
||||
# Known placeholder rating markup: nothing rating-shaped is kept.
|
||||
listing.rating = listing.review_count = listing.rating_breakdown = None
|
||||
listing.reviews = []
|
||||
raw_specs = dict((product or {}).get("properties") or {})
|
||||
raw_specs.update({k: v for k, v in spec_tables(BeautifulSoup(html, "lxml")).items() if k not in raw_specs})
|
||||
listing.specs_raw = dict(list(raw_specs.items())[:150])
|
||||
|
||||
@@ -0,0 +1,4 @@
|
||||
-- Star breakdown of a listing's ratings, exactly as the source states it:
|
||||
-- {"5": 612, "4": 300, "3": 80, "2": 20, "1": 12}. NULL when the source does
|
||||
-- not publish one - never derived from the few review texts we hold.
|
||||
ALTER TABLE elec.source_listing ADD COLUMN rating_breakdown JSONB;
|
||||
@@ -199,7 +199,8 @@ def upsert_listing(listing: Listing, ids: Dict[str, Dict[str, int]], run_id: Opt
|
||||
model_number=listing.model_number, ram_gb=listing.ram_gb, storage_gb=listing.storage_gb,
|
||||
colour=listing.colour, price=listing.price, mrp=listing.mrp, availability=listing.availability,
|
||||
in_stock=listing.in_stock, pincode=listing.pincode, pincode_applied=listing.pincode_applied,
|
||||
rating=listing.rating, review_count=listing.review_count, gtin=listing.gtin,
|
||||
rating=listing.rating, review_count=listing.review_count,
|
||||
rating_breakdown=_json(listing.rating_breakdown) if listing.rating_breakdown else None, gtin=listing.gtin,
|
||||
image_urls=listing.image_urls[:12], specs_raw=_json(listing.specs_raw), specs=_json(listing.specs),
|
||||
evidence_text=listing.evidence_text[:4000], search_query=listing.search_query,
|
||||
confidence=round(listing.confidence, 2), parser=listing.parser, content_hash=listing.content_hash,
|
||||
@@ -210,13 +211,13 @@ def upsert_listing(listing: Listing, ids: Dict[str, Dict[str, int]], run_id: Opt
|
||||
INSERT INTO elec.source_listing (
|
||||
site_id, source_sku, source_url, source_type, brand_id, category_id, family, title, model,
|
||||
model_number, ram_gb, storage_gb, colour, price, mrp, availability, in_stock, pincode,
|
||||
pincode_applied, rating, review_count, gtin, image_urls, specs_raw, specs, evidence_text,
|
||||
pincode_applied, rating, review_count, rating_breakdown, gtin, image_urls, specs_raw, specs, evidence_text,
|
||||
search_query, confidence, parser, content_hash, crawl_run_id)
|
||||
VALUES (
|
||||
%(site_id)s, %(source_sku)s, %(source_url)s, %(source_type)s, %(brand_id)s, %(category_id)s,
|
||||
%(family)s, %(title)s, %(model)s, %(model_number)s, %(ram_gb)s, %(storage_gb)s, %(colour)s,
|
||||
%(price)s, %(mrp)s, %(availability)s, %(in_stock)s, %(pincode)s, %(pincode_applied)s,
|
||||
%(rating)s, %(review_count)s, %(gtin)s, %(image_urls)s, %(specs_raw)s, %(specs)s,
|
||||
%(rating)s, %(review_count)s, %(rating_breakdown)s, %(gtin)s, %(image_urls)s, %(specs_raw)s, %(specs)s,
|
||||
%(evidence_text)s, %(search_query)s, %(confidence)s, %(parser)s, %(content_hash)s,
|
||||
%(crawl_run_id)s)
|
||||
ON CONFLICT (site_id, source_sku) DO UPDATE SET
|
||||
@@ -227,7 +228,7 @@ def upsert_listing(listing: Listing, ids: Dict[str, Dict[str, int]], run_id: Opt
|
||||
price = EXCLUDED.price, mrp = EXCLUDED.mrp, availability = EXCLUDED.availability,
|
||||
in_stock = EXCLUDED.in_stock, pincode = EXCLUDED.pincode,
|
||||
pincode_applied = EXCLUDED.pincode_applied, rating = EXCLUDED.rating,
|
||||
review_count = EXCLUDED.review_count, gtin = EXCLUDED.gtin, image_urls = EXCLUDED.image_urls,
|
||||
review_count = EXCLUDED.review_count, rating_breakdown = EXCLUDED.rating_breakdown, gtin = EXCLUDED.gtin, image_urls = EXCLUDED.image_urls,
|
||||
specs_raw = EXCLUDED.specs_raw, specs = EXCLUDED.specs, evidence_text = EXCLUDED.evidence_text,
|
||||
search_query = EXCLUDED.search_query, confidence = EXCLUDED.confidence, parser = EXCLUDED.parser,
|
||||
content_hash = EXCLUDED.content_hash, crawl_run_id = EXCLUDED.crawl_run_id, last_seen_at = now()
|
||||
@@ -276,12 +277,13 @@ def save_reviews(listing_id: int, reviews: List[Dict[str, Any]]) -> int:
|
||||
return added
|
||||
|
||||
|
||||
def update_listing_rating(listing_id: int, rating: Optional[Decimal], review_count: Optional[int]) -> None:
|
||||
def update_listing_rating(listing_id: int, rating: Optional[Decimal], review_count: Optional[int],
|
||||
breakdown: Optional[Dict[str, int]] = None) -> None:
|
||||
"""Refresh only the rating fields of a listing (used by the review backfill)."""
|
||||
with transaction() as conn:
|
||||
conn.execute(
|
||||
"UPDATE elec.source_listing SET rating = %s, review_count = %s WHERE id = %s",
|
||||
(rating, review_count, listing_id),
|
||||
"UPDATE elec.source_listing SET rating = %s, review_count = %s, rating_breakdown = %s WHERE id = %s",
|
||||
(rating, review_count, _json(breakdown) if breakdown else None, listing_id),
|
||||
)
|
||||
|
||||
|
||||
@@ -289,7 +291,7 @@ def product_rating_and_reviews(conn, product_id: int) -> Dict[str, Any]:
|
||||
"""Per-platform ratings and all stored reviews for a verified product's
|
||||
approved listings, each with the page it was read from."""
|
||||
sources = conn.execute(
|
||||
"SELECT a.site, a.source_url, l.rating, l.review_count FROM elec.v_product_availability a "
|
||||
"SELECT a.site, a.source_url, l.rating, l.review_count, l.rating_breakdown FROM elec.v_product_availability a "
|
||||
"JOIN elec.source_listing l ON l.id = a.listing_id "
|
||||
"WHERE a.product_id = %s AND l.rating > 0 ORDER BY l.review_count DESC NULLS LAST, a.site",
|
||||
(product_id,),
|
||||
|
||||
95
backend/app/electronics/extract/embedded_ratings.py
Normal file
95
backend/app/electronics/extract/embedded_ratings.py
Normal file
@@ -0,0 +1,95 @@
|
||||
"""Ratings a retailer embeds in its own product-page HTML (not in JSON-LD).
|
||||
|
||||
Only the page we already fetched is read - no extra requests - and only the
|
||||
record of the page's OWN product (never recommendations on the same page).
|
||||
Zero values mean "no ratings yet" and are dropped. Review text is not available
|
||||
here: those retailers load it from feeds their robots.txt does not allow.
|
||||
|
||||
Deliberately absent: vasanthandco.in. Its product pages carry an identical
|
||||
"3.3 average, 3 reviews" template block on every product - placeholder
|
||||
content, not ratings - so nothing is ever read from that domain.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from typing import Any, Callable, Dict, Optional
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
# Domains whose rating-looking markup is known to be fake/placeholder.
|
||||
IGNORED_DOMAINS = frozenset({"vasanthandco.in"})
|
||||
|
||||
|
||||
def _dec(value: Any) -> Optional[Decimal]:
|
||||
try:
|
||||
d = Decimal(str(value))
|
||||
except (InvalidOperation, TypeError, ValueError):
|
||||
return None
|
||||
return d if Decimal(0) < d <= Decimal(5) else None
|
||||
|
||||
|
||||
def _int(value: Any) -> Optional[int]:
|
||||
try:
|
||||
n = int(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
return n if n > 0 else None
|
||||
|
||||
|
||||
def _reading(rating: Any, count: Any, breakdown: Optional[Dict[str, int]] = None) -> Optional[Dict[str, Any]]:
|
||||
r = _dec(rating)
|
||||
if r is None:
|
||||
return None
|
||||
clean = None
|
||||
if breakdown:
|
||||
clean = {str(k): int(v) for k, v in breakdown.items()
|
||||
if str(k) in {"1", "2", "3", "4", "5"} and _int(v)}
|
||||
return {"rating": r.quantize(Decimal("0.01")), "review_count": _int(count), "breakdown": clean or None}
|
||||
|
||||
|
||||
def _reliance(html: str, url: str) -> Optional[Dict[str, Any]]:
|
||||
"""window.__INITIAL_STATE__.productDetailsPage.product._custom_json._app"""
|
||||
marker = "window.__INITIAL_STATE__="
|
||||
i = html.find(marker)
|
||||
if i < 0:
|
||||
return None
|
||||
try:
|
||||
state, _ = json.JSONDecoder().raw_decode(html, i + len(marker))
|
||||
app = state["productDetailsPage"]["product"]["_custom_json"]["_app"]
|
||||
except (ValueError, KeyError, TypeError):
|
||||
return None
|
||||
if not isinstance(app, dict):
|
||||
return None
|
||||
return _reading(app.get("averageRating"), app.get("ratingsCount"), app.get("ratingsCountDetails"))
|
||||
|
||||
|
||||
def _poorvika(html: str, url: str) -> Optional[Dict[str, Any]]:
|
||||
"""__NEXT_DATA__ props.pageProps.additionalData, only when its slug is this URL's."""
|
||||
tag = BeautifulSoup(html, "lxml").find("script", id="__NEXT_DATA__")
|
||||
if tag is None:
|
||||
return None
|
||||
try:
|
||||
data = json.loads(tag.string or tag.get_text())["props"]["pageProps"]["additionalData"]
|
||||
except (ValueError, KeyError, TypeError):
|
||||
return None
|
||||
slug = str(data.get("slug") or "")
|
||||
if not slug or slug not in urlparse(url).path:
|
||||
return None
|
||||
return _reading(data.get("rating"), data.get("ratingCount"))
|
||||
|
||||
|
||||
_READERS: Dict[str, Callable[[str, str], Optional[Dict[str, Any]]]] = {
|
||||
"reliancedigital.in": _reliance,
|
||||
"poorvika.com": _poorvika,
|
||||
}
|
||||
|
||||
|
||||
def embedded_rating(domain: str, html: str, url: str) -> Optional[Dict[str, Any]]:
|
||||
"""{"rating": Decimal, "review_count": int|None, "breakdown": {"5": n, ...}|None} or None."""
|
||||
if domain in IGNORED_DOMAINS:
|
||||
return None
|
||||
reader = _READERS.get(domain)
|
||||
return reader(html, url) if reader else None
|
||||
@@ -161,42 +161,67 @@ def _reviews(node: dict) -> List[Dict[str, Any]]:
|
||||
return out
|
||||
|
||||
|
||||
def _rating_only_nodes(nodes: List[dict]) -> Dict[Optional[str], dict]:
|
||||
"""Nameless Product nodes that only carry ratings/reviews, keyed by @id.
|
||||
|
||||
Review widgets (Bazaarvoice on samsung.com/in, for one) publish the
|
||||
product's aggregateRating and reviews as a SEPARATE Product node with no
|
||||
name, linked to the real product by the same @id. Without merging, those
|
||||
real reviews would be skipped for lack of a name.
|
||||
"""
|
||||
out: Dict[Optional[str], dict] = {}
|
||||
for node in nodes:
|
||||
if _text(node.get("name")):
|
||||
continue
|
||||
if not (isinstance(node.get("aggregateRating"), dict) or node.get("review") or node.get("reviews")):
|
||||
continue
|
||||
out.setdefault(_text(node.get("@id")), node)
|
||||
return out
|
||||
|
||||
|
||||
def extract_products(html: str) -> List[Dict[str, Any]]:
|
||||
"""All schema.org Product nodes on the page, flattened to plain fields."""
|
||||
products: List[Dict[str, Any]] = []
|
||||
for block in json_ld_blocks(html):
|
||||
for node in _walk(block):
|
||||
if not (_types(node) & _PRODUCT_TYPES):
|
||||
continue
|
||||
name = _text(node.get("name"))
|
||||
if not name:
|
||||
continue
|
||||
offer = _offer(node.get("offers")) if node.get("offers") else {}
|
||||
if not offer and isinstance(node.get("hasVariant"), list):
|
||||
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
|
||||
props = {}
|
||||
for p in node.get("additionalProperty") or []:
|
||||
if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""):
|
||||
props[str(p["name"])] = str(p["value"])
|
||||
rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {}
|
||||
products.append({
|
||||
"name": name,
|
||||
"brand": _text(node.get("brand")),
|
||||
"sku": _text(node.get("sku")) or _text(node.get("productID")),
|
||||
"mpn": _text(node.get("mpn")),
|
||||
"gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None),
|
||||
"color": _text(node.get("color")),
|
||||
"images": _images(node.get("image")),
|
||||
"description": _text(node.get("description")),
|
||||
"price": offer.get("price"),
|
||||
"currency": offer.get("currency"),
|
||||
"availability": offer.get("availability"),
|
||||
"in_stock": offer.get("in_stock"),
|
||||
# Sites publish 0 for "no ratings yet"; that is not a rating.
|
||||
"rating": (_dec(rating.get("ratingValue")) or None),
|
||||
"review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None,
|
||||
"reviews": _reviews(node),
|
||||
"properties": props,
|
||||
"evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500],
|
||||
})
|
||||
nodes = [n for block in json_ld_blocks(html) for n in _walk(block) if _types(n) & _PRODUCT_TYPES]
|
||||
rating_nodes = _rating_only_nodes(nodes)
|
||||
named = [n for n in nodes if _text(n.get("name"))]
|
||||
for node in named:
|
||||
# Attach a ratings-only node: same @id, or - when it has no @id - the
|
||||
# page's only named product (there is then no doubt which it is about).
|
||||
extra = rating_nodes.get(_text(node.get("@id"))) if node.get("@id") else None
|
||||
if extra is None and len(named) == 1:
|
||||
extra = rating_nodes.get(None)
|
||||
if extra is not None:
|
||||
node = {**node,
|
||||
"aggregateRating": node.get("aggregateRating") or extra.get("aggregateRating"),
|
||||
"review": node.get("review") or extra.get("review") or extra.get("reviews")}
|
||||
name = _text(node.get("name"))
|
||||
offer = _offer(node.get("offers")) if node.get("offers") else {}
|
||||
if not offer and isinstance(node.get("hasVariant"), list):
|
||||
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
|
||||
props = {}
|
||||
for p in node.get("additionalProperty") or []:
|
||||
if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""):
|
||||
props[str(p["name"])] = str(p["value"])
|
||||
rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {}
|
||||
products.append({
|
||||
"name": name,
|
||||
"brand": _text(node.get("brand")),
|
||||
"sku": _text(node.get("sku")) or _text(node.get("productID")),
|
||||
"mpn": _text(node.get("mpn")),
|
||||
"gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None),
|
||||
"color": _text(node.get("color")),
|
||||
"images": _images(node.get("image")),
|
||||
"description": _text(node.get("description")),
|
||||
"price": offer.get("price"),
|
||||
"currency": offer.get("currency"),
|
||||
"availability": offer.get("availability"),
|
||||
"in_stock": offer.get("in_stock"),
|
||||
# Sites publish 0 for "no ratings yet"; that is not a rating.
|
||||
"rating": (_dec(rating.get("ratingValue")) or None),
|
||||
"review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None,
|
||||
"reviews": _reviews(node),
|
||||
"properties": props,
|
||||
"evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500],
|
||||
})
|
||||
return products
|
||||
|
||||
80
backend/app/electronics/extract/vijaysales_reviews.py
Normal file
80
backend/app/electronics/extract/vijaysales_reviews.py
Normal file
@@ -0,0 +1,80 @@
|
||||
"""Vijay Sales ratings and reviews from its public GraphQL endpoint.
|
||||
|
||||
vijaysales.com serves product data to its own pages from GET /api/graphql;
|
||||
robots.txt does not disallow it, and no login or token is involved. One GET
|
||||
per listing, made through PoliteClient (robots, pacing, circuit breaker).
|
||||
|
||||
The product is looked up by url_key - the last path segment of the listing URL
|
||||
we already hold - so the answer is about exactly that listing.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from typing import Any, Dict, List, Optional
|
||||
from urllib.parse import urlencode, urlparse
|
||||
|
||||
from app.electronics.reviews import sentiment_for
|
||||
|
||||
ENDPOINT = "https://www.vijaysales.com/api/graphql"
|
||||
|
||||
_QUERY = (
|
||||
'{products(filter:{url_key:{eq:"%s"}}){items{sku rating_summary review_count '
|
||||
"reviews(pageSize:20){items{nickname summary text average_rating created_at}}}}}"
|
||||
)
|
||||
|
||||
|
||||
def url_key(listing_url: str) -> Optional[str]:
|
||||
path = urlparse(listing_url).path.rstrip("/")
|
||||
key = path.rsplit("/", 1)[-1] if path else ""
|
||||
# url_keys are slugs; anything else would break out of the query string.
|
||||
return key if key and all(c.isalnum() or c in "-_" for c in key) else None
|
||||
|
||||
|
||||
def graphql_url(listing_url: str) -> Optional[str]:
|
||||
key = url_key(listing_url)
|
||||
return f"{ENDPOINT}?{urlencode({'query': _QUERY % key})}" if key else None
|
||||
|
||||
|
||||
def _five(percent: Any) -> Optional[Decimal]:
|
||||
"""Magento ratings are percentages (80 = 4 stars)."""
|
||||
try:
|
||||
p = Decimal(str(percent))
|
||||
except (InvalidOperation, TypeError, ValueError):
|
||||
return None
|
||||
return (p / 20).quantize(Decimal("0.01")) if Decimal(0) < p <= Decimal(100) else None
|
||||
|
||||
|
||||
def parse(body: str) -> Optional[Dict[str, Any]]:
|
||||
"""{"rating", "review_count", "reviews": [...]} for the best-reviewed matching item, or None."""
|
||||
try:
|
||||
items = json.loads(body)["data"]["products"]["items"] or []
|
||||
except (ValueError, KeyError, TypeError):
|
||||
return None
|
||||
items = [i for i in items if isinstance(i, dict)]
|
||||
if not items:
|
||||
return None
|
||||
item = max(items, key=lambda i: i.get("review_count") or 0)
|
||||
rating = _five(item.get("rating_summary"))
|
||||
if rating is None:
|
||||
return None
|
||||
reviews: List[Dict[str, Any]] = []
|
||||
for r in ((item.get("reviews") or {}).get("items") or []):
|
||||
body_text = (r.get("text") or "").strip()
|
||||
if not body_text:
|
||||
continue
|
||||
stars = _five(r.get("average_rating"))
|
||||
title = (r.get("summary") or "").strip()
|
||||
# When a customer leaves the title blank the site fills in the full
|
||||
# product name; that is not something the reviewer wrote.
|
||||
if len(title) > 60 or "/" in title:
|
||||
title = ""
|
||||
reviews.append({
|
||||
"author": (r.get("nickname") or "").strip() or None,
|
||||
"rating": stars,
|
||||
"title": title or None,
|
||||
"body": body_text[:4000],
|
||||
"review_date": r.get("created_at"),
|
||||
"sentiment": sentiment_for(stars),
|
||||
})
|
||||
return {"rating": rating, "review_count": int(item.get("review_count") or 0) or None, "reviews": reviews}
|
||||
@@ -34,6 +34,8 @@ class Listing:
|
||||
pincode_applied: bool = False
|
||||
rating: Optional[Decimal] = None
|
||||
review_count: Optional[int] = None
|
||||
# {"5": n, ..., "1": n} when the source publishes a star breakdown.
|
||||
rating_breakdown: Optional[Dict[str, int]] = None
|
||||
# Customer reviews the page itself publishes (schema.org Review); stored
|
||||
# in elec.listing_review, not on the listing row.
|
||||
reviews: List[Dict[str, Any]] = field(default_factory=list)
|
||||
|
||||
@@ -24,6 +24,11 @@ class BrandRef:
|
||||
aliases: tuple
|
||||
sub_brands: tuple
|
||||
official: tuple
|
||||
# Brand-store path whose product pages publish real customer reviews as
|
||||
# schema.org data (e.g. "samsung.com/in"); None when no free source exists.
|
||||
reviews_site: Optional[str] = None
|
||||
# Regex a reviews_site URL must match to be a single-variant product page.
|
||||
reviews_product_url: Optional[str] = None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
@@ -78,6 +83,8 @@ def load_reference() -> Reference:
|
||||
aliases=tuple(a.lower() for a in b.get("aliases", [])),
|
||||
sub_brands=tuple(s.lower() for s in b.get("sub_brands", [])),
|
||||
official=tuple(b.get("official", [])),
|
||||
reviews_site=b.get("reviews_site"),
|
||||
reviews_product_url=b.get("reviews_product_url"),
|
||||
)
|
||||
categories = {
|
||||
c["slug"]: CategoryRef(c["slug"], c["name"], tuple(c.get("search_terms", [])),
|
||||
|
||||
@@ -7,11 +7,19 @@
|
||||
# parent, and are kept as the product family.
|
||||
# official the brand's own Indian web domains. A product page on one of
|
||||
# these is the strongest evidence that a product exists.
|
||||
# reviews_site optional brand-store path whose product pages publish real
|
||||
# customer ratings + reviews as schema.org JSON-LD, robots-allowed
|
||||
# (checked 2026-10-01). `cli reviews` finds each verified product's
|
||||
# page there. Add a brand only after checking its pages.
|
||||
# reviews_product_url regex for that store's single-variant product pages
|
||||
# (family/marketing pages carry no product data or reviews).
|
||||
brands:
|
||||
- name: Samsung
|
||||
categories: [mobiles, laptops]
|
||||
aliases: [samsung]
|
||||
official: [samsung.com]
|
||||
reviews_site: samsung.com/in
|
||||
reviews_product_url: '-sm-[a-z0-9]+/?$' # .../galaxy-a56-5g-awesome-olive-256gb-sm-a566ezggins/
|
||||
- name: Apple
|
||||
categories: [mobiles, laptops]
|
||||
aliases: [apple]
|
||||
|
||||
215
backend/app/electronics/review_refresh.py
Normal file
215
backend/app/electronics/review_refresh.py
Normal file
@@ -0,0 +1,215 @@
|
||||
"""Refresh customer ratings and reviews for the verified catalogue, from free
|
||||
sources that publish them and allow reading them:
|
||||
|
||||
1. Brand stores with real reviews in schema.org JSON-LD (brands.yaml
|
||||
`reviews_site`, e.g. samsung.com/in): find each verified product's own page
|
||||
by web search, accept it only for the same model AND variant, read it with
|
||||
the normal collector page path, and store it as a brand_official listing.
|
||||
2. Every page-read listing of a verified product is re-read for its rating,
|
||||
star breakdown and reviews (JSON-LD, then the retailer's embedded data).
|
||||
3. Vijay Sales listings additionally get their reviews from the site's own
|
||||
public GraphQL endpoint.
|
||||
|
||||
Everything goes through PoliteClient (robots.txt, per-site pacing, circuit
|
||||
breaker). Nothing is generated; a source that states nothing leaves nothing.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import dataclasses
|
||||
import logging
|
||||
import re
|
||||
from decimal import Decimal
|
||||
from typing import Callable, Dict, List, Optional
|
||||
|
||||
from rapidfuzz import fuzz
|
||||
|
||||
from app.electronics.collector import Collector, RunOptions
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.db.connection import connect
|
||||
from app.electronics.extract import vijaysales_reviews
|
||||
from app.electronics.extract.embedded_ratings import IGNORED_DOMAINS, embedded_rating
|
||||
from app.electronics.extract.jsonld import extract_products
|
||||
from app.electronics.reference import load_reference
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _verified_products(category: Optional[str]) -> List[dict]:
|
||||
sql = ("SELECT v.product_id, v.brand_slug, v.category, p.model, p.model_norm, p.ram_gb, p.storage_gb "
|
||||
"FROM elec.v_brand_catalog v JOIN elec.product p ON p.id = v.product_id")
|
||||
params: tuple = ()
|
||||
if category:
|
||||
sql += " WHERE v.category = %s"
|
||||
params = (category,)
|
||||
with connect() as conn:
|
||||
return list(conn.execute(sql + " ORDER BY v.product_id", params))
|
||||
|
||||
|
||||
def _has_page_on(product_id: int, domain: str) -> bool:
|
||||
with connect() as conn:
|
||||
return conn.execute(
|
||||
"SELECT 1 FROM elec.v_product_availability WHERE product_id = %s AND domain = %s "
|
||||
"AND source_type IN ('scraped_page','brand_official') LIMIT 1", (product_id, domain),
|
||||
).fetchone() is not None
|
||||
|
||||
|
||||
def _gb(value) -> str:
|
||||
return f"{format(Decimal(value).normalize(), 'f')}GB" if value is not None else ""
|
||||
|
||||
|
||||
# Words that make a different model, not a different colour of the same one.
|
||||
_MODEL_QUALIFIERS = frozenset({"ultra", "plus", "pro", "max", "fe", "lite", "edge", "mini", "neo",
|
||||
"prime", "flip", "fold", "slim", "+"})
|
||||
|
||||
|
||||
def _same_model(page_model: str, product_model: str) -> bool:
|
||||
"""Same model line: every product word present, identical model-number
|
||||
words, and nothing extra but non-model words (e.g. a colour the parser
|
||||
left in: "galaxy a56 olive"). Fuzzy scores are no use here - they rate
|
||||
"galaxy a57" vs "galaxy a56" at ~90."""
|
||||
page, mine = set(page_model.split()), set(product_model.split())
|
||||
numbered = lambda words: {w for w in words if any(c.isdigit() for c in w)} # noqa: E731
|
||||
extra = page - mine
|
||||
return mine <= page and numbered(page) == numbered(mine) and not (extra & _MODEL_QUALIFIERS)
|
||||
|
||||
|
||||
def _same_variant(parsed, product: dict) -> bool:
|
||||
if not parsed.model_norm or not _same_model(parsed.model_norm, product["model_norm"] or ""):
|
||||
return False
|
||||
for attr in ("ram_gb", "storage_gb"):
|
||||
mine, theirs = product[attr], getattr(parsed, attr)
|
||||
if mine is not None and theirs is not None and Decimal(mine) != Decimal(theirs):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def discover_brand_pages(category: Optional[str], budget: int, stats: Dict[str, int],
|
||||
progress: Callable[[str], None]) -> None:
|
||||
ref = load_reference()
|
||||
by_category: Dict[str, List[dict]] = {}
|
||||
for p in _verified_products(category):
|
||||
brand = ref.brands.get(p["brand_slug"])
|
||||
if brand and brand.reviews_site:
|
||||
by_category.setdefault(p["category"], []).append(p)
|
||||
for cat, products in by_category.items():
|
||||
brands = sorted({p["brand_slug"] for p in products})
|
||||
col = Collector(RunOptions(category=cat, brands=brands, search_budget=budget, use_llm=False),
|
||||
progress=progress)
|
||||
try:
|
||||
for p in products:
|
||||
brand = ref.brands[p["brand_slug"]]
|
||||
domain = brand.reviews_site.split("/")[0]
|
||||
if _has_page_on(p["product_id"], domain):
|
||||
stats["brand_page_already_known"] = stats.get("brand_page_already_known", 0) + 1
|
||||
continue
|
||||
model = p["model"] or p["model_norm"]
|
||||
queries = [f"site:{brand.reviews_site} {brand.name} {model} {_gb(p['ram_gb'])} {_gb(p['storage_gb'])}",
|
||||
f"site:{brand.reviews_site} {model} 5G {_gb(p['storage_gb'])} buy"]
|
||||
hits, query = [], ""
|
||||
for q in (re.sub(r"\s+", " ", q).strip() for q in queries):
|
||||
hits += [(h, q) for h in (col.engine.text(q, max_results=10) or [])]
|
||||
seen = set()
|
||||
for hit, query in hits:
|
||||
if hit.url in seen:
|
||||
continue
|
||||
seen.add(hit.url)
|
||||
# Brand-store titles often omit the brand ("Galaxy A56 5G 8GB/256GB ...",
|
||||
# sometimes behind a "Business |" prefix). On the brand's own store
|
||||
# the brand is not in doubt, so state it for the title parser.
|
||||
title = re.sub(r"^\s*Business\s*\|\s*", "", hit.title or "")
|
||||
if not title.lower().startswith(brand.name.lower()):
|
||||
title = f"{brand.name} {title}"
|
||||
hit = dataclasses.replace(hit, title=title)
|
||||
if f"{brand.reviews_site}/" not in hit.url:
|
||||
continue
|
||||
if brand.reviews_product_url and not re.search(brand.reviews_product_url, hit.url.split("?")[0]):
|
||||
continue # a family/marketing page, not one variant's product page
|
||||
accepted = col._accept_hit(hit, brand)
|
||||
if not accepted:
|
||||
continue
|
||||
site, parsed = accepted
|
||||
if not _same_variant(parsed, p):
|
||||
continue
|
||||
res = col.client.get(hit.url)
|
||||
if not res.ok:
|
||||
continue
|
||||
listing = col.listing_from_page(hit, site, parsed, query, res.text, res.final_url)
|
||||
# Only a page with schema.org product data for this exact variant.
|
||||
if listing is None or not listing.parser.startswith("jsonld") or not _same_variant(listing, p):
|
||||
continue
|
||||
col.store(listing)
|
||||
if _has_page_on(p["product_id"], domain): # actually matched to this product
|
||||
stats["brand_pages_added"] = stats.get("brand_pages_added", 0) + 1
|
||||
progress(f" {brand.name}: {listing.title} -> rating {listing.rating}, "
|
||||
f"{len(listing.reviews)} reviews")
|
||||
break
|
||||
finally:
|
||||
col.client.close()
|
||||
repo.refresh_verification()
|
||||
|
||||
|
||||
def reread_pages(category: Optional[str], limit: int, stats: Dict[str, int],
|
||||
progress: Callable[[str], None]) -> None:
|
||||
from app.electronics.net.polite_client import PoliteClient
|
||||
|
||||
rows = repo.listings_for_review_backfill(category)[:limit]
|
||||
with PoliteClient() as client:
|
||||
for row in rows:
|
||||
domain = row["domain"]
|
||||
if domain in IGNORED_DOMAINS:
|
||||
continue
|
||||
stats["pages"] = stats.get("pages", 0) + 1
|
||||
rating = count = breakdown = None
|
||||
reviews: List[dict] = []
|
||||
res = client.get(row["source_url"])
|
||||
if res.ok:
|
||||
products = extract_products(res.text)
|
||||
match = next((x for x in products if x.get("sku") and x["sku"] == row["source_sku"]), None)
|
||||
if match is None:
|
||||
title = (row["title"] or "").lower()
|
||||
scored = [(fuzz.token_set_ratio(x["name"].lower(), title), x) for x in products]
|
||||
scored = [s for s in scored if s[0] >= 85]
|
||||
match = max(scored, key=lambda s: s[0])[1] if scored else None
|
||||
if match is not None:
|
||||
rating, count, reviews = match.get("rating"), match.get("review_count"), match.get("reviews") or []
|
||||
# The retailer's own embedded data: the rating when JSON-LD has none,
|
||||
# and the star breakdown, which only it publishes.
|
||||
emb = embedded_rating(domain, res.text, res.final_url or row["source_url"])
|
||||
if emb:
|
||||
if rating is None:
|
||||
rating, count = emb["rating"], emb["review_count"]
|
||||
breakdown = emb["breakdown"]
|
||||
if domain == "vijaysales.com":
|
||||
gql = vijaysales_reviews.graphql_url(row["source_url"])
|
||||
vres = client.get(gql, accept_non_html=True) if gql else None
|
||||
vs = vijaysales_reviews.parse(vres.text) if vres is not None and vres.ok else None
|
||||
if vs:
|
||||
rating, count = rating or vs["rating"], count or vs["review_count"]
|
||||
reviews = reviews or vs["reviews"]
|
||||
if rating is not None and Decimal(0) < Decimal(rating) <= Decimal(5):
|
||||
repo.update_listing_rating(row["listing_id"], rating, count, breakdown)
|
||||
stats["rated"] = stats.get("rated", 0) + 1
|
||||
if reviews:
|
||||
stats["reviews_stored"] = stats.get("reviews_stored", 0) + repo.save_reviews(row["listing_id"], reviews)
|
||||
progress(f" {domain:22} rating={rating} count={count} breakdown={'yes' if breakdown else 'no'} "
|
||||
f"reviews={len(reviews)}")
|
||||
|
||||
|
||||
def refresh_reviews(category: Optional[str] = None, limit: int = 500, budget: int = 60,
|
||||
discover: bool = True, progress: Optional[Callable[[str], None]] = None) -> Dict[str, int]:
|
||||
progress = progress or (lambda m: logger.info(m))
|
||||
stats: Dict[str, int] = {}
|
||||
run_id = repo.start_run("reviews", {"category": category, "limit": limit, "discover": discover})
|
||||
status, error = "done", None
|
||||
try:
|
||||
if discover:
|
||||
progress("Finding brand-store pages with reviews ...")
|
||||
discover_brand_pages(category, budget, stats, progress)
|
||||
progress("Re-reading ratings and reviews from product pages ...")
|
||||
reread_pages(category, limit, stats, progress)
|
||||
except Exception as exc:
|
||||
status, error = "failed", repr(exc)
|
||||
raise
|
||||
finally:
|
||||
repo.finish_run(run_id, status, stats, error)
|
||||
return stats
|
||||
@@ -199,7 +199,8 @@ USER_AGENT = os.getenv(
|
||||
REQUEST_TIMEOUT_SECONDS = int(os.getenv("REQUEST_TIMEOUT_SECONDS", "20"))
|
||||
ELEC_SITE_MIN_INTERVAL_SECONDS = float(os.getenv("ELEC_SITE_MIN_INTERVAL_SECONDS", "3"))
|
||||
ELEC_BREAKER_COOLDOWN_HOURS = float(os.getenv("ELEC_BREAKER_COOLDOWN_HOURS", "24"))
|
||||
ELEC_MAX_PAGE_BYTES = int(os.getenv("ELEC_MAX_PAGE_BYTES", str(3 * 1024 * 1024)))
|
||||
# Brand stores embed their reviews in the page: samsung.com/in pages run ~3.1 MB.
|
||||
ELEC_MAX_PAGE_BYTES = int(os.getenv("ELEC_MAX_PAGE_BYTES", str(6 * 1024 * 1024)))
|
||||
ELEC_PROBE_TTL_DAYS = int(os.getenv("ELEC_PROBE_TTL_DAYS", "7"))
|
||||
MIN_IMAGE_BYTES = int(os.getenv("MIN_IMAGE_BYTES", "3000"))
|
||||
|
||||
|
||||
@@ -92,8 +92,9 @@ async def get_product(product_id: int) -> Dict[str, Any]:
|
||||
"""Full details of one product by its product_id (from search_products).
|
||||
|
||||
Returns per-platform offers (price, MRP, source URL, when seen), normalised specs,
|
||||
image URLs, the overall rating with per-platform sources (null if none is published),
|
||||
and up to 10 real customer reviews (often empty).
|
||||
image URLs, the overall rating with per-platform sources and, when a platform publishes
|
||||
one, a 5-to-1 star breakdown (rating is null if no platform publishes a rating), and up
|
||||
to 10 real customer reviews with the site each came from (often empty).
|
||||
"""
|
||||
d = await _run(elec.product, int(product_id))
|
||||
return {
|
||||
|
||||
@@ -142,3 +142,155 @@ def test_api_serves_ratings_reviews_and_out_of_stock_price(db, client):
|
||||
assert {s["site"] for s in detail["rating"]["sources"]} == {"Amazon.in", "Croma"}
|
||||
assert [r["body"] for r in detail["reviews"]] == ["Battery lasts all day.", "Heats up."]
|
||||
assert all(r["source_url"].startswith("https://") for r in detail["reviews"])
|
||||
assert detail["rating"]["breakdown"] is None # no platform published one
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Free sources: brand-store JSON-LD, retailer-embedded ratings, Vijay Sales
|
||||
# ---------------------------------------------------------------------------
|
||||
def _ld(*nodes) -> str:
|
||||
return "".join(f'<script type="application/ld+json">{json.dumps(n)}</script>' for n in nodes)
|
||||
|
||||
|
||||
def _review(author, stars, body, date="2026-09-30T17:06:05.000+00:00"):
|
||||
return {"@type": "Review", "author": {"@type": "Person", "name": author}, "reviewBody": body,
|
||||
"datePublished": date, "reviewRating": {"@type": "Rating", "bestRating": 5, "ratingValue": stars}}
|
||||
|
||||
|
||||
def test_nameless_review_node_is_merged_by_id():
|
||||
# The samsung.com/in shape: Bazaarvoice publishes ratings in a second,
|
||||
# nameless Product node that shares the real product's @id.
|
||||
pid = "https://www.samsung.com/in/smartphones/galaxy-a/galaxy-a56-5g-awesome-olive-256gb-sm-a566ezggins/"
|
||||
html = _ld(
|
||||
{"@context": "https://schema.org", "@type": "Product", "@id": pid, "name": "Galaxy A56 5G (8 GB Memory)",
|
||||
"sku": "SM-A566EZGG", "offers": {"@type": "Offer", "price": "48999", "priceCurrency": "INR"}},
|
||||
{"@context": "https://schema.org", "@type": "Product", "@id": pid,
|
||||
"aggregateRating": {"@type": "AggregateRating", "ratingValue": "4.5", "ratingCount": 1224},
|
||||
"review": [_review("Vedant", 5, "Great and useful AI features"), _review("Ravi", 1, "Heats up")]},
|
||||
)
|
||||
[p] = extract_products(html)
|
||||
assert p["name"] == "Galaxy A56 5G (8 GB Memory)" and p["price"] == Decimal("48999")
|
||||
assert p["rating"] == Decimal("4.5") and p["review_count"] == 1224
|
||||
assert [(r["author"], r["rating"], r["body"]) for r in p["reviews"]] == [
|
||||
("Vedant", Decimal("5.0"), "Great and useful AI features"), ("Ravi", Decimal("1.0"), "Heats up")]
|
||||
|
||||
|
||||
def test_nameless_review_node_without_id_is_not_guessed_between_products():
|
||||
lone = {"@type": "Product", "aggregateRating": {"ratingValue": "4.1", "ratingCount": 9}}
|
||||
[only] = extract_products(_ld({"@type": "Product", "name": "Galaxy A56"}, lone))
|
||||
assert only["rating"] == Decimal("4.1") # one product on the page: unambiguous
|
||||
two = extract_products(_ld({"@type": "Product", "name": "Galaxy A56"},
|
||||
{"@type": "Product", "name": "Galaxy A36"}, lone))
|
||||
assert all(p["rating"] is None for p in two) # two products: never guess which
|
||||
|
||||
|
||||
def test_reliance_embedded_rating_is_read_from_the_pages_own_product_only():
|
||||
from app.electronics.extract.embedded_ratings import embedded_rating
|
||||
|
||||
def page(app, recommended=None):
|
||||
state = {"productDetailsPage": {"product": {"name": "Dell 15", "_custom_json": {"_app": app}}},
|
||||
"recommendations": [{"_custom_json": {"_app": recommended}}] if recommended else []}
|
||||
return f"<script>window.__INITIAL_STATE__={json.dumps(state)};window.x=1</script>"
|
||||
|
||||
url = "https://www.reliancedigital.in/product/dell-15-9991568"
|
||||
got = embedded_rating("reliancedigital.in", page(
|
||||
{"averageRating": 4, "ratingsCount": 3, "ratingsCountDetails": {"5": 1, "4": 1, "3": 1}}), url)
|
||||
assert got == {"rating": Decimal("4.00"), "review_count": 3, "breakdown": {"5": 1, "4": 1, "3": 1}}
|
||||
# "No ratings yet" is published as zeros; a recommended product's rating is not ours.
|
||||
zero = page({"averageRating": 0, "ratingsCount": 0}, recommended={"averageRating": 4.8, "ratingsCount": 90})
|
||||
assert embedded_rating("reliancedigital.in", zero, url) is None
|
||||
|
||||
|
||||
def test_poorvika_rating_needs_the_pages_own_slug():
|
||||
from app.electronics.extract.embedded_ratings import embedded_rating
|
||||
|
||||
def page(slug, rating, count):
|
||||
data = {"props": {"pageProps": {"additionalData": {"slug": slug, "rating": rating, "ratingCount": count}}}}
|
||||
return f'<script id="__NEXT_DATA__" type="application/json">{json.dumps(data)}</script>'
|
||||
|
||||
slug = "xiaomi-14-civi-5g-shadow-black-256gb-8gb-ram"
|
||||
url = f"https://www.poorvika.com/{slug}/p"
|
||||
assert embedded_rating("poorvika.com", page(slug, 5, 1), url)["rating"] == Decimal("5.00")
|
||||
assert embedded_rating("poorvika.com", page("some-other-phone", 5, 1), url) is None
|
||||
assert embedded_rating("poorvika.com", page(slug, None, 0), url) is None
|
||||
|
||||
|
||||
def test_vasanth_template_rating_is_never_read():
|
||||
from app.electronics.extract.embedded_ratings import embedded_rating
|
||||
|
||||
template = ('<h4 class="avg-mark">3.3</h4><a class="rating-reviews">(3 Reviews)</a>'
|
||||
+ _ld({"@type": "Product", "name": "HP 15", "aggregateRating": {"ratingValue": "3.3", "reviewCount": 3}}))
|
||||
assert embedded_rating("vasanthandco.in", template, "https://vasanthandco.in/product/174200102203/") is None
|
||||
|
||||
|
||||
def test_vijaysales_graphql_review_feed():
|
||||
from app.electronics.extract import vijaysales_reviews as vs
|
||||
|
||||
url = "https://www.vijaysales.com/p/239904/acer-aspire-lite-al15-41-laptop-amd-ryzen-7-16gb"
|
||||
assert vs.url_key(url) == "acer-aspire-lite-al15-41-laptop-amd-ryzen-7-16gb"
|
||||
assert "url_key" in vs.graphql_url(url)
|
||||
assert vs.url_key('https://x.in/p/a"b') is None # cannot break out of the query
|
||||
product_name = "Acer Aspire Lite AL15-41 Laptop (AMD Ryzen 7/ 16GB DDR4 RAM/ 512GB SSD)"
|
||||
body = json.dumps({"data": {"products": {"items": [
|
||||
{"sku": "P1", "rating_summary": 0, "review_count": 0, "reviews": {"items": []}},
|
||||
{"sku": "239904", "rating_summary": 90, "review_count": 2, "reviews": {"items": [
|
||||
{"nickname": "Jay", "summary": product_name, "text": "All over performance is good",
|
||||
"average_rating": 100, "created_at": "2026-05-02 10:00:00"},
|
||||
{"nickname": "guest user", "summary": "Okay", "text": "", "average_rating": 60}]}}]}}})
|
||||
got = vs.parse(body)
|
||||
assert got["rating"] == Decimal("4.50") and got["review_count"] == 2
|
||||
# Product-name "title" dropped; a review with no text is skipped.
|
||||
assert got["reviews"] == [{"author": "Jay", "rating": Decimal("5.00"), "title": None,
|
||||
"body": "All over performance is good", "review_date": "2026-05-02 10:00:00",
|
||||
"sentiment": "positive"}]
|
||||
assert vs.parse('{"data":{"products":{"items":[]}}}') is None
|
||||
|
||||
|
||||
def test_api_breakdown_only_when_a_platform_publishes_one(db, client):
|
||||
from app.electronics.collector import Collector, RunOptions, RunStats
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.models import Listing
|
||||
from app.electronics.normalise.title_parser import parse_title, variant_key
|
||||
|
||||
def listing(site, sku, price, source_type="search_snippet"):
|
||||
title = "Samsung Galaxy S24 5G (8GB RAM, 256GB)"
|
||||
p = parse_title(title, "mobiles")
|
||||
l = Listing(site_domain=site, source_sku=sku, source_url=f"https://www.{site}/p/{sku}",
|
||||
source_type=source_type, brand_slug="samsung", category="mobiles", title=title,
|
||||
evidence_text=f"{title} ₹{price}", confidence=0.5, parser="test",
|
||||
model=p.model, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=Decimal(price))
|
||||
l.model_norm, l.variant_key = p.model_norm, variant_key(p, "mobiles")
|
||||
return l
|
||||
|
||||
c = Collector.__new__(Collector)
|
||||
c.opt = RunOptions(category="mobiles", brands=["samsung"])
|
||||
c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats()
|
||||
b = listing("reliancedigital.in", "rd-1", 73999, "scraped_page")
|
||||
b.rating, b.review_count, b.rating_breakdown = Decimal("4.0"), 4, {"5": 2, "4": 1, "1": 1}
|
||||
c.store(listing("amazon.in", "B0CS5XW6TN", 74999))
|
||||
c.store(b)
|
||||
repo.refresh_verification()
|
||||
pid = client.get("/api/elec/products", params={"category": "mobiles"}).json()["products"][0]["product_id"]
|
||||
rating = client.get(f"/api/elec/products/{pid}").json()["rating"]
|
||||
assert rating["breakdown"] == [
|
||||
{"stars": 5, "count": 2, "percent": 50}, {"stars": 4, "count": 1, "percent": 25},
|
||||
{"stars": 3, "count": 0, "percent": 0}, {"stars": 2, "count": 0, "percent": 0},
|
||||
{"stars": 1, "count": 1, "percent": 25}]
|
||||
assert rating["sources"][0]["breakdown"] == {"5": 2, "4": 1, "1": 1}
|
||||
|
||||
|
||||
def test_brand_store_page_must_be_the_exact_model_and_variant():
|
||||
from app.electronics.normalise.title_parser import parse_title
|
||||
from app.electronics.review_refresh import _same_variant
|
||||
|
||||
a56 = {"model_norm": "galaxy a56", "ram_gb": Decimal("8"), "storage_gb": Decimal("256")}
|
||||
page = lambda t: parse_title(t, "mobiles", expected_brand="samsung") # noqa: E731
|
||||
assert _same_variant(page("Samsung Galaxy A56 5G 8GB/256GB (Olive)"), a56)
|
||||
# A neighbouring model scores ~90 on fuzzy matching; it must still be refused.
|
||||
assert not _same_variant(page("Samsung Galaxy A57 5G 8GB/256GB (Navy)"), a56)
|
||||
assert not _same_variant(page("Samsung Galaxy A56 5G 12GB/256GB (Olive)"), a56)
|
||||
assert not _same_variant(page("Samsung Galaxy A56 5G 8GB/128GB (Blue)"), a56)
|
||||
s25 = {"model_norm": "galaxy s25", "ram_gb": Decimal("12"), "storage_gb": Decimal("256")}
|
||||
assert _same_variant(page("Samsung Galaxy S25 12GB/256GB (Icyblue)"), s25)
|
||||
assert not _same_variant(page("Samsung Galaxy S25 Ultra 12GB/256GB (Titanium Black)"), s25)
|
||||
assert not _same_variant(page("Samsung Galaxy S25+ 12GB/256GB (Navy)"), s25)
|
||||
|
||||
Reference in New Issue
Block a user