Customer Rating and review changes
This commit is contained in:
@@ -141,23 +141,50 @@ def product(product_id: int) -> dict:
|
|||||||
def _overall_rating(sources: List[dict]) -> Optional[dict]:
|
def _overall_rating(sources: List[dict]) -> Optional[dict]:
|
||||||
"""The product's rating across the platforms that state one: the mean
|
"""The product's rating across the platforms that state one: the mean
|
||||||
weighted by each platform's rating count (a platform that states no count
|
weighted by each platform's rating count (a platform that states no count
|
||||||
weighs as 1). None when no platform states a rating - never a guess."""
|
weighs as 1). None when no platform states a rating - never a guess.
|
||||||
|
|
||||||
|
One reading per platform: a site with a page per colour repeats the same
|
||||||
|
model rating on each, and adding those up would multiply its count."""
|
||||||
if not sources:
|
if not sources:
|
||||||
return None
|
return None
|
||||||
weight = lambda s: max(int(s["review_count"] or 0), 1) # noqa: E731
|
weight = lambda s: max(int(s["review_count"] or 0), 1) # noqa: E731
|
||||||
|
per_site: Dict[str, dict] = {}
|
||||||
|
for s in sources:
|
||||||
|
if s["site"] not in per_site or weight(s) > weight(per_site[s["site"]]):
|
||||||
|
per_site[s["site"]] = s
|
||||||
|
sources = list(per_site.values())
|
||||||
total = sum(weight(s) for s in sources)
|
total = sum(weight(s) for s in sources)
|
||||||
value = sum(Decimal(s["rating"]) * weight(s) for s in sources) / total
|
value = sum(Decimal(s["rating"]) * weight(s) for s in sources) / total
|
||||||
counts = [s["review_count"] for s in sources if s["review_count"]]
|
counts = [s["review_count"] for s in sources if s["review_count"]]
|
||||||
return {
|
return {
|
||||||
"value": round(float(value), 1),
|
"value": round(float(value), 1),
|
||||||
"count": sum(counts) if counts else None,
|
"count": sum(counts) if counts else None,
|
||||||
|
"breakdown": _breakdown(sources),
|
||||||
"sources": [
|
"sources": [
|
||||||
{"site": s["site"], "rating": float(s["rating"]), "review_count": s["review_count"], "source_url": s["source_url"]}
|
{"site": s["site"], "rating": float(s["rating"]), "review_count": s["review_count"],
|
||||||
|
"breakdown": s.get("rating_breakdown"), "source_url": s["source_url"]}
|
||||||
for s in sources
|
for s in sources
|
||||||
],
|
],
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _breakdown(sources: List[dict]) -> Optional[List[dict]]:
|
||||||
|
"""5→1 star counts summed over the platforms that PUBLISH a breakdown.
|
||||||
|
None when none does - never estimated from the few review texts held."""
|
||||||
|
totals = {str(n): 0 for n in range(1, 6)}
|
||||||
|
stated = False
|
||||||
|
for s in sources:
|
||||||
|
for star, n in (s.get("rating_breakdown") or {}).items():
|
||||||
|
if star in totals and isinstance(n, int) and n > 0:
|
||||||
|
totals[star] += n
|
||||||
|
stated = True
|
||||||
|
if not stated:
|
||||||
|
return None
|
||||||
|
total = sum(totals.values())
|
||||||
|
return [{"stars": int(k), "count": totals[k], "percent": round(100 * totals[k] / total)}
|
||||||
|
for k in ("5", "4", "3", "2", "1")]
|
||||||
|
|
||||||
|
|
||||||
@router.get("/products/{product_id}/price-history")
|
@router.get("/products/{product_id}/price-history")
|
||||||
def price_history(product_id: int) -> List[dict]:
|
def price_history(product_id: int) -> List[dict]:
|
||||||
with connect() as conn:
|
with connect() as conn:
|
||||||
|
|||||||
@@ -143,58 +143,22 @@ def rematch(category: str = typer.Option(..., help="mobiles | laptops"),
|
|||||||
|
|
||||||
@app.command()
|
@app.command()
|
||||||
def reviews(category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)"),
|
def reviews(category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)"),
|
||||||
limit: int = typer.Option(200, help="Max product pages to re-read"),
|
limit: int = typer.Option(500, help="Max product pages to re-read"),
|
||||||
|
budget: int = typer.Option(60, help="Max web searches for finding brand-store pages"),
|
||||||
|
no_discover: bool = typer.Option(False, "--no-discover", help="Skip finding brand-store pages"),
|
||||||
verbose: bool = False) -> None:
|
verbose: bool = False) -> None:
|
||||||
"""Re-read ratings and customer reviews from the product pages already on file.
|
"""Refresh real customer ratings and reviews from free, robots-allowed sources.
|
||||||
|
|
||||||
Only pages the collector itself reads (scraped / brand official listings)
|
Finds each verified product's page on brand stores that publish reviews
|
||||||
are fetched, politely (robots.txt, per-site pacing, circuit breaker). A
|
(brands.yaml `reviews_site`), then re-reads every readable product page for
|
||||||
rating or review is stored only when the page's own schema.org data states
|
its rating, star breakdown and reviews, plus Vijay Sales' public review
|
||||||
it; nothing is generated.
|
feed. A rating or review is stored only when its source states it.
|
||||||
"""
|
"""
|
||||||
_setup_logging(verbose)
|
_setup_logging(verbose)
|
||||||
from rapidfuzz import fuzz
|
from app.electronics.review_refresh import refresh_reviews
|
||||||
|
|
||||||
from app.electronics.db import repository as repo
|
stats = refresh_reviews(category=category, limit=limit, budget=budget, discover=not no_discover,
|
||||||
from app.electronics.extract.jsonld import extract_products
|
progress=typer.echo)
|
||||||
from app.electronics.net.polite_client import PoliteClient
|
|
||||||
|
|
||||||
rows = repo.listings_for_review_backfill(category)[:limit]
|
|
||||||
run_id = repo.start_run("reviews", {"category": category, "pages": len(rows)})
|
|
||||||
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
|
|
||||||
r.outcome, r.robots_allowed))
|
|
||||||
stats = {"pages": 0, "pages_ok": 0, "rated": 0, "reviews_stored": 0, "no_matching_product": 0}
|
|
||||||
status, error = "done", None
|
|
||||||
try:
|
|
||||||
for row in rows:
|
|
||||||
stats["pages"] += 1
|
|
||||||
res = client.get(row["source_url"])
|
|
||||||
if not res.ok:
|
|
||||||
continue
|
|
||||||
stats["pages_ok"] += 1
|
|
||||||
products = extract_products(res.text)
|
|
||||||
# The same product the listing was stored from: its SKU, else its name.
|
|
||||||
match = next((p for p in products if p.get("sku") and p["sku"] == row["source_sku"]), None)
|
|
||||||
if match is None:
|
|
||||||
title = (row["title"] or "").lower()
|
|
||||||
scored = [(fuzz.token_set_ratio(p["name"].lower(), title), p) for p in products]
|
|
||||||
scored = [sp for sp in scored if sp[0] >= 85]
|
|
||||||
match = max(scored, key=lambda sp: sp[0])[1] if scored else None
|
|
||||||
if match is None:
|
|
||||||
stats["no_matching_product"] += 1
|
|
||||||
continue
|
|
||||||
if match.get("rating") is not None and Decimal(0) < match["rating"] <= Decimal(5):
|
|
||||||
repo.update_listing_rating(row["listing_id"], match["rating"], match.get("review_count"))
|
|
||||||
stats["rated"] += 1
|
|
||||||
if match.get("reviews"):
|
|
||||||
stats["reviews_stored"] += repo.save_reviews(row["listing_id"], match["reviews"])
|
|
||||||
typer.echo(f" {row['domain']:22} rating={match.get('rating')} reviews={len(match.get('reviews') or [])}")
|
|
||||||
except Exception as exc: # noqa: BLE001
|
|
||||||
status, error = "failed", repr(exc)
|
|
||||||
raise
|
|
||||||
finally:
|
|
||||||
client.close()
|
|
||||||
repo.finish_run(run_id, status, stats, error)
|
|
||||||
typer.echo(json.dumps(stats, indent=2))
|
typer.echo(json.dumps(stats, indent=2))
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -36,6 +36,7 @@ from urllib.parse import urlparse
|
|||||||
from rapidfuzz import fuzz
|
from rapidfuzz import fuzz
|
||||||
|
|
||||||
from app.electronics.db import repository as repo
|
from app.electronics.db import repository as repo
|
||||||
|
from app.electronics.extract.embedded_ratings import IGNORED_DOMAINS, embedded_rating
|
||||||
from app.electronics.extract.html_fallback import extract_page, spec_tables, visible_text
|
from app.electronics.extract.html_fallback import extract_page, spec_tables, visible_text
|
||||||
from app.electronics.extract.jsonld import extract_products
|
from app.electronics.extract.jsonld import extract_products
|
||||||
from app.electronics.extract.serp_parser import clean_result_title, read_price, read_rating, read_stock
|
from app.electronics.extract.serp_parser import clean_result_title, read_price, read_rating, read_stock
|
||||||
@@ -371,6 +372,13 @@ class Collector:
|
|||||||
html: str, final_url: str) -> Optional[Listing]:
|
html: str, final_url: str) -> Optional[Listing]:
|
||||||
products = extract_products(html)
|
products = extract_products(html)
|
||||||
page = extract_page(html)
|
page = extract_page(html)
|
||||||
|
if site.kind == "brand_official":
|
||||||
|
# A brand's own store names products without the brand ("Galaxy A56 5G
|
||||||
|
# (8 GB Memory)"); on its own site the brand is not in doubt.
|
||||||
|
brand_name = self.ref.brands[parsed_hit.brand.brand_slug].name
|
||||||
|
for p in products:
|
||||||
|
if not p["name"].lower().startswith(brand_name.lower()):
|
||||||
|
p["name"] = f"{brand_name} {p['name']}"
|
||||||
name = None
|
name = None
|
||||||
product = None
|
product = None
|
||||||
for p in products:
|
for p in products:
|
||||||
@@ -428,6 +436,17 @@ class Collector:
|
|||||||
listing.review_count = product.get("review_count")
|
listing.review_count = product.get("review_count")
|
||||||
listing.reviews = list(product.get("reviews") or [])
|
listing.reviews = list(product.get("reviews") or [])
|
||||||
listing.colour = listing.colour or product.get("color")
|
listing.colour = listing.colour or product.get("color")
|
||||||
|
# Retailers that embed their rating outside JSON-LD (page's own product
|
||||||
|
# only): the rating when JSON-LD has none, and the star breakdown.
|
||||||
|
embedded = embedded_rating(site.domain, html, final_url or hit.url)
|
||||||
|
if embedded:
|
||||||
|
if listing.rating is None:
|
||||||
|
listing.rating, listing.review_count = embedded["rating"], embedded["review_count"]
|
||||||
|
listing.rating_breakdown = embedded["breakdown"]
|
||||||
|
if site.domain in IGNORED_DOMAINS:
|
||||||
|
# Known placeholder rating markup: nothing rating-shaped is kept.
|
||||||
|
listing.rating = listing.review_count = listing.rating_breakdown = None
|
||||||
|
listing.reviews = []
|
||||||
raw_specs = dict((product or {}).get("properties") or {})
|
raw_specs = dict((product or {}).get("properties") or {})
|
||||||
raw_specs.update({k: v for k, v in spec_tables(BeautifulSoup(html, "lxml")).items() if k not in raw_specs})
|
raw_specs.update({k: v for k, v in spec_tables(BeautifulSoup(html, "lxml")).items() if k not in raw_specs})
|
||||||
listing.specs_raw = dict(list(raw_specs.items())[:150])
|
listing.specs_raw = dict(list(raw_specs.items())[:150])
|
||||||
|
|||||||
@@ -0,0 +1,4 @@
|
|||||||
|
-- Star breakdown of a listing's ratings, exactly as the source states it:
|
||||||
|
-- {"5": 612, "4": 300, "3": 80, "2": 20, "1": 12}. NULL when the source does
|
||||||
|
-- not publish one - never derived from the few review texts we hold.
|
||||||
|
ALTER TABLE elec.source_listing ADD COLUMN rating_breakdown JSONB;
|
||||||
@@ -199,7 +199,8 @@ def upsert_listing(listing: Listing, ids: Dict[str, Dict[str, int]], run_id: Opt
|
|||||||
model_number=listing.model_number, ram_gb=listing.ram_gb, storage_gb=listing.storage_gb,
|
model_number=listing.model_number, ram_gb=listing.ram_gb, storage_gb=listing.storage_gb,
|
||||||
colour=listing.colour, price=listing.price, mrp=listing.mrp, availability=listing.availability,
|
colour=listing.colour, price=listing.price, mrp=listing.mrp, availability=listing.availability,
|
||||||
in_stock=listing.in_stock, pincode=listing.pincode, pincode_applied=listing.pincode_applied,
|
in_stock=listing.in_stock, pincode=listing.pincode, pincode_applied=listing.pincode_applied,
|
||||||
rating=listing.rating, review_count=listing.review_count, gtin=listing.gtin,
|
rating=listing.rating, review_count=listing.review_count,
|
||||||
|
rating_breakdown=_json(listing.rating_breakdown) if listing.rating_breakdown else None, gtin=listing.gtin,
|
||||||
image_urls=listing.image_urls[:12], specs_raw=_json(listing.specs_raw), specs=_json(listing.specs),
|
image_urls=listing.image_urls[:12], specs_raw=_json(listing.specs_raw), specs=_json(listing.specs),
|
||||||
evidence_text=listing.evidence_text[:4000], search_query=listing.search_query,
|
evidence_text=listing.evidence_text[:4000], search_query=listing.search_query,
|
||||||
confidence=round(listing.confidence, 2), parser=listing.parser, content_hash=listing.content_hash,
|
confidence=round(listing.confidence, 2), parser=listing.parser, content_hash=listing.content_hash,
|
||||||
@@ -210,13 +211,13 @@ def upsert_listing(listing: Listing, ids: Dict[str, Dict[str, int]], run_id: Opt
|
|||||||
INSERT INTO elec.source_listing (
|
INSERT INTO elec.source_listing (
|
||||||
site_id, source_sku, source_url, source_type, brand_id, category_id, family, title, model,
|
site_id, source_sku, source_url, source_type, brand_id, category_id, family, title, model,
|
||||||
model_number, ram_gb, storage_gb, colour, price, mrp, availability, in_stock, pincode,
|
model_number, ram_gb, storage_gb, colour, price, mrp, availability, in_stock, pincode,
|
||||||
pincode_applied, rating, review_count, gtin, image_urls, specs_raw, specs, evidence_text,
|
pincode_applied, rating, review_count, rating_breakdown, gtin, image_urls, specs_raw, specs, evidence_text,
|
||||||
search_query, confidence, parser, content_hash, crawl_run_id)
|
search_query, confidence, parser, content_hash, crawl_run_id)
|
||||||
VALUES (
|
VALUES (
|
||||||
%(site_id)s, %(source_sku)s, %(source_url)s, %(source_type)s, %(brand_id)s, %(category_id)s,
|
%(site_id)s, %(source_sku)s, %(source_url)s, %(source_type)s, %(brand_id)s, %(category_id)s,
|
||||||
%(family)s, %(title)s, %(model)s, %(model_number)s, %(ram_gb)s, %(storage_gb)s, %(colour)s,
|
%(family)s, %(title)s, %(model)s, %(model_number)s, %(ram_gb)s, %(storage_gb)s, %(colour)s,
|
||||||
%(price)s, %(mrp)s, %(availability)s, %(in_stock)s, %(pincode)s, %(pincode_applied)s,
|
%(price)s, %(mrp)s, %(availability)s, %(in_stock)s, %(pincode)s, %(pincode_applied)s,
|
||||||
%(rating)s, %(review_count)s, %(gtin)s, %(image_urls)s, %(specs_raw)s, %(specs)s,
|
%(rating)s, %(review_count)s, %(rating_breakdown)s, %(gtin)s, %(image_urls)s, %(specs_raw)s, %(specs)s,
|
||||||
%(evidence_text)s, %(search_query)s, %(confidence)s, %(parser)s, %(content_hash)s,
|
%(evidence_text)s, %(search_query)s, %(confidence)s, %(parser)s, %(content_hash)s,
|
||||||
%(crawl_run_id)s)
|
%(crawl_run_id)s)
|
||||||
ON CONFLICT (site_id, source_sku) DO UPDATE SET
|
ON CONFLICT (site_id, source_sku) DO UPDATE SET
|
||||||
@@ -227,7 +228,7 @@ def upsert_listing(listing: Listing, ids: Dict[str, Dict[str, int]], run_id: Opt
|
|||||||
price = EXCLUDED.price, mrp = EXCLUDED.mrp, availability = EXCLUDED.availability,
|
price = EXCLUDED.price, mrp = EXCLUDED.mrp, availability = EXCLUDED.availability,
|
||||||
in_stock = EXCLUDED.in_stock, pincode = EXCLUDED.pincode,
|
in_stock = EXCLUDED.in_stock, pincode = EXCLUDED.pincode,
|
||||||
pincode_applied = EXCLUDED.pincode_applied, rating = EXCLUDED.rating,
|
pincode_applied = EXCLUDED.pincode_applied, rating = EXCLUDED.rating,
|
||||||
review_count = EXCLUDED.review_count, gtin = EXCLUDED.gtin, image_urls = EXCLUDED.image_urls,
|
review_count = EXCLUDED.review_count, rating_breakdown = EXCLUDED.rating_breakdown, gtin = EXCLUDED.gtin, image_urls = EXCLUDED.image_urls,
|
||||||
specs_raw = EXCLUDED.specs_raw, specs = EXCLUDED.specs, evidence_text = EXCLUDED.evidence_text,
|
specs_raw = EXCLUDED.specs_raw, specs = EXCLUDED.specs, evidence_text = EXCLUDED.evidence_text,
|
||||||
search_query = EXCLUDED.search_query, confidence = EXCLUDED.confidence, parser = EXCLUDED.parser,
|
search_query = EXCLUDED.search_query, confidence = EXCLUDED.confidence, parser = EXCLUDED.parser,
|
||||||
content_hash = EXCLUDED.content_hash, crawl_run_id = EXCLUDED.crawl_run_id, last_seen_at = now()
|
content_hash = EXCLUDED.content_hash, crawl_run_id = EXCLUDED.crawl_run_id, last_seen_at = now()
|
||||||
@@ -276,12 +277,13 @@ def save_reviews(listing_id: int, reviews: List[Dict[str, Any]]) -> int:
|
|||||||
return added
|
return added
|
||||||
|
|
||||||
|
|
||||||
def update_listing_rating(listing_id: int, rating: Optional[Decimal], review_count: Optional[int]) -> None:
|
def update_listing_rating(listing_id: int, rating: Optional[Decimal], review_count: Optional[int],
|
||||||
|
breakdown: Optional[Dict[str, int]] = None) -> None:
|
||||||
"""Refresh only the rating fields of a listing (used by the review backfill)."""
|
"""Refresh only the rating fields of a listing (used by the review backfill)."""
|
||||||
with transaction() as conn:
|
with transaction() as conn:
|
||||||
conn.execute(
|
conn.execute(
|
||||||
"UPDATE elec.source_listing SET rating = %s, review_count = %s WHERE id = %s",
|
"UPDATE elec.source_listing SET rating = %s, review_count = %s, rating_breakdown = %s WHERE id = %s",
|
||||||
(rating, review_count, listing_id),
|
(rating, review_count, _json(breakdown) if breakdown else None, listing_id),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -289,7 +291,7 @@ def product_rating_and_reviews(conn, product_id: int) -> Dict[str, Any]:
|
|||||||
"""Per-platform ratings and all stored reviews for a verified product's
|
"""Per-platform ratings and all stored reviews for a verified product's
|
||||||
approved listings, each with the page it was read from."""
|
approved listings, each with the page it was read from."""
|
||||||
sources = conn.execute(
|
sources = conn.execute(
|
||||||
"SELECT a.site, a.source_url, l.rating, l.review_count FROM elec.v_product_availability a "
|
"SELECT a.site, a.source_url, l.rating, l.review_count, l.rating_breakdown FROM elec.v_product_availability a "
|
||||||
"JOIN elec.source_listing l ON l.id = a.listing_id "
|
"JOIN elec.source_listing l ON l.id = a.listing_id "
|
||||||
"WHERE a.product_id = %s AND l.rating > 0 ORDER BY l.review_count DESC NULLS LAST, a.site",
|
"WHERE a.product_id = %s AND l.rating > 0 ORDER BY l.review_count DESC NULLS LAST, a.site",
|
||||||
(product_id,),
|
(product_id,),
|
||||||
|
|||||||
95
backend/app/electronics/extract/embedded_ratings.py
Normal file
95
backend/app/electronics/extract/embedded_ratings.py
Normal file
@@ -0,0 +1,95 @@
|
|||||||
|
"""Ratings a retailer embeds in its own product-page HTML (not in JSON-LD).
|
||||||
|
|
||||||
|
Only the page we already fetched is read - no extra requests - and only the
|
||||||
|
record of the page's OWN product (never recommendations on the same page).
|
||||||
|
Zero values mean "no ratings yet" and are dropped. Review text is not available
|
||||||
|
here: those retailers load it from feeds their robots.txt does not allow.
|
||||||
|
|
||||||
|
Deliberately absent: vasanthandco.in. Its product pages carry an identical
|
||||||
|
"3.3 average, 3 reviews" template block on every product - placeholder
|
||||||
|
content, not ratings - so nothing is ever read from that domain.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
from decimal import Decimal, InvalidOperation
|
||||||
|
from typing import Any, Callable, Dict, Optional
|
||||||
|
from urllib.parse import urlparse
|
||||||
|
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
|
||||||
|
# Domains whose rating-looking markup is known to be fake/placeholder.
|
||||||
|
IGNORED_DOMAINS = frozenset({"vasanthandco.in"})
|
||||||
|
|
||||||
|
|
||||||
|
def _dec(value: Any) -> Optional[Decimal]:
|
||||||
|
try:
|
||||||
|
d = Decimal(str(value))
|
||||||
|
except (InvalidOperation, TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
return d if Decimal(0) < d <= Decimal(5) else None
|
||||||
|
|
||||||
|
|
||||||
|
def _int(value: Any) -> Optional[int]:
|
||||||
|
try:
|
||||||
|
n = int(value)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
return n if n > 0 else None
|
||||||
|
|
||||||
|
|
||||||
|
def _reading(rating: Any, count: Any, breakdown: Optional[Dict[str, int]] = None) -> Optional[Dict[str, Any]]:
|
||||||
|
r = _dec(rating)
|
||||||
|
if r is None:
|
||||||
|
return None
|
||||||
|
clean = None
|
||||||
|
if breakdown:
|
||||||
|
clean = {str(k): int(v) for k, v in breakdown.items()
|
||||||
|
if str(k) in {"1", "2", "3", "4", "5"} and _int(v)}
|
||||||
|
return {"rating": r.quantize(Decimal("0.01")), "review_count": _int(count), "breakdown": clean or None}
|
||||||
|
|
||||||
|
|
||||||
|
def _reliance(html: str, url: str) -> Optional[Dict[str, Any]]:
|
||||||
|
"""window.__INITIAL_STATE__.productDetailsPage.product._custom_json._app"""
|
||||||
|
marker = "window.__INITIAL_STATE__="
|
||||||
|
i = html.find(marker)
|
||||||
|
if i < 0:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
state, _ = json.JSONDecoder().raw_decode(html, i + len(marker))
|
||||||
|
app = state["productDetailsPage"]["product"]["_custom_json"]["_app"]
|
||||||
|
except (ValueError, KeyError, TypeError):
|
||||||
|
return None
|
||||||
|
if not isinstance(app, dict):
|
||||||
|
return None
|
||||||
|
return _reading(app.get("averageRating"), app.get("ratingsCount"), app.get("ratingsCountDetails"))
|
||||||
|
|
||||||
|
|
||||||
|
def _poorvika(html: str, url: str) -> Optional[Dict[str, Any]]:
|
||||||
|
"""__NEXT_DATA__ props.pageProps.additionalData, only when its slug is this URL's."""
|
||||||
|
tag = BeautifulSoup(html, "lxml").find("script", id="__NEXT_DATA__")
|
||||||
|
if tag is None:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
data = json.loads(tag.string or tag.get_text())["props"]["pageProps"]["additionalData"]
|
||||||
|
except (ValueError, KeyError, TypeError):
|
||||||
|
return None
|
||||||
|
slug = str(data.get("slug") or "")
|
||||||
|
if not slug or slug not in urlparse(url).path:
|
||||||
|
return None
|
||||||
|
return _reading(data.get("rating"), data.get("ratingCount"))
|
||||||
|
|
||||||
|
|
||||||
|
_READERS: Dict[str, Callable[[str, str], Optional[Dict[str, Any]]]] = {
|
||||||
|
"reliancedigital.in": _reliance,
|
||||||
|
"poorvika.com": _poorvika,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def embedded_rating(domain: str, html: str, url: str) -> Optional[Dict[str, Any]]:
|
||||||
|
"""{"rating": Decimal, "review_count": int|None, "breakdown": {"5": n, ...}|None} or None."""
|
||||||
|
if domain in IGNORED_DOMAINS:
|
||||||
|
return None
|
||||||
|
reader = _READERS.get(domain)
|
||||||
|
return reader(html, url) if reader else None
|
||||||
@@ -161,16 +161,41 @@ def _reviews(node: dict) -> List[Dict[str, Any]]:
|
|||||||
return out
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _rating_only_nodes(nodes: List[dict]) -> Dict[Optional[str], dict]:
|
||||||
|
"""Nameless Product nodes that only carry ratings/reviews, keyed by @id.
|
||||||
|
|
||||||
|
Review widgets (Bazaarvoice on samsung.com/in, for one) publish the
|
||||||
|
product's aggregateRating and reviews as a SEPARATE Product node with no
|
||||||
|
name, linked to the real product by the same @id. Without merging, those
|
||||||
|
real reviews would be skipped for lack of a name.
|
||||||
|
"""
|
||||||
|
out: Dict[Optional[str], dict] = {}
|
||||||
|
for node in nodes:
|
||||||
|
if _text(node.get("name")):
|
||||||
|
continue
|
||||||
|
if not (isinstance(node.get("aggregateRating"), dict) or node.get("review") or node.get("reviews")):
|
||||||
|
continue
|
||||||
|
out.setdefault(_text(node.get("@id")), node)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
def extract_products(html: str) -> List[Dict[str, Any]]:
|
def extract_products(html: str) -> List[Dict[str, Any]]:
|
||||||
"""All schema.org Product nodes on the page, flattened to plain fields."""
|
"""All schema.org Product nodes on the page, flattened to plain fields."""
|
||||||
products: List[Dict[str, Any]] = []
|
products: List[Dict[str, Any]] = []
|
||||||
for block in json_ld_blocks(html):
|
nodes = [n for block in json_ld_blocks(html) for n in _walk(block) if _types(n) & _PRODUCT_TYPES]
|
||||||
for node in _walk(block):
|
rating_nodes = _rating_only_nodes(nodes)
|
||||||
if not (_types(node) & _PRODUCT_TYPES):
|
named = [n for n in nodes if _text(n.get("name"))]
|
||||||
continue
|
for node in named:
|
||||||
|
# Attach a ratings-only node: same @id, or - when it has no @id - the
|
||||||
|
# page's only named product (there is then no doubt which it is about).
|
||||||
|
extra = rating_nodes.get(_text(node.get("@id"))) if node.get("@id") else None
|
||||||
|
if extra is None and len(named) == 1:
|
||||||
|
extra = rating_nodes.get(None)
|
||||||
|
if extra is not None:
|
||||||
|
node = {**node,
|
||||||
|
"aggregateRating": node.get("aggregateRating") or extra.get("aggregateRating"),
|
||||||
|
"review": node.get("review") or extra.get("review") or extra.get("reviews")}
|
||||||
name = _text(node.get("name"))
|
name = _text(node.get("name"))
|
||||||
if not name:
|
|
||||||
continue
|
|
||||||
offer = _offer(node.get("offers")) if node.get("offers") else {}
|
offer = _offer(node.get("offers")) if node.get("offers") else {}
|
||||||
if not offer and isinstance(node.get("hasVariant"), list):
|
if not offer and isinstance(node.get("hasVariant"), list):
|
||||||
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
|
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
|
||||||
|
|||||||
80
backend/app/electronics/extract/vijaysales_reviews.py
Normal file
80
backend/app/electronics/extract/vijaysales_reviews.py
Normal file
@@ -0,0 +1,80 @@
|
|||||||
|
"""Vijay Sales ratings and reviews from its public GraphQL endpoint.
|
||||||
|
|
||||||
|
vijaysales.com serves product data to its own pages from GET /api/graphql;
|
||||||
|
robots.txt does not disallow it, and no login or token is involved. One GET
|
||||||
|
per listing, made through PoliteClient (robots, pacing, circuit breaker).
|
||||||
|
|
||||||
|
The product is looked up by url_key - the last path segment of the listing URL
|
||||||
|
we already hold - so the answer is about exactly that listing.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
from decimal import Decimal, InvalidOperation
|
||||||
|
from typing import Any, Dict, List, Optional
|
||||||
|
from urllib.parse import urlencode, urlparse
|
||||||
|
|
||||||
|
from app.electronics.reviews import sentiment_for
|
||||||
|
|
||||||
|
ENDPOINT = "https://www.vijaysales.com/api/graphql"
|
||||||
|
|
||||||
|
_QUERY = (
|
||||||
|
'{products(filter:{url_key:{eq:"%s"}}){items{sku rating_summary review_count '
|
||||||
|
"reviews(pageSize:20){items{nickname summary text average_rating created_at}}}}}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def url_key(listing_url: str) -> Optional[str]:
|
||||||
|
path = urlparse(listing_url).path.rstrip("/")
|
||||||
|
key = path.rsplit("/", 1)[-1] if path else ""
|
||||||
|
# url_keys are slugs; anything else would break out of the query string.
|
||||||
|
return key if key and all(c.isalnum() or c in "-_" for c in key) else None
|
||||||
|
|
||||||
|
|
||||||
|
def graphql_url(listing_url: str) -> Optional[str]:
|
||||||
|
key = url_key(listing_url)
|
||||||
|
return f"{ENDPOINT}?{urlencode({'query': _QUERY % key})}" if key else None
|
||||||
|
|
||||||
|
|
||||||
|
def _five(percent: Any) -> Optional[Decimal]:
|
||||||
|
"""Magento ratings are percentages (80 = 4 stars)."""
|
||||||
|
try:
|
||||||
|
p = Decimal(str(percent))
|
||||||
|
except (InvalidOperation, TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
return (p / 20).quantize(Decimal("0.01")) if Decimal(0) < p <= Decimal(100) else None
|
||||||
|
|
||||||
|
|
||||||
|
def parse(body: str) -> Optional[Dict[str, Any]]:
|
||||||
|
"""{"rating", "review_count", "reviews": [...]} for the best-reviewed matching item, or None."""
|
||||||
|
try:
|
||||||
|
items = json.loads(body)["data"]["products"]["items"] or []
|
||||||
|
except (ValueError, KeyError, TypeError):
|
||||||
|
return None
|
||||||
|
items = [i for i in items if isinstance(i, dict)]
|
||||||
|
if not items:
|
||||||
|
return None
|
||||||
|
item = max(items, key=lambda i: i.get("review_count") or 0)
|
||||||
|
rating = _five(item.get("rating_summary"))
|
||||||
|
if rating is None:
|
||||||
|
return None
|
||||||
|
reviews: List[Dict[str, Any]] = []
|
||||||
|
for r in ((item.get("reviews") or {}).get("items") or []):
|
||||||
|
body_text = (r.get("text") or "").strip()
|
||||||
|
if not body_text:
|
||||||
|
continue
|
||||||
|
stars = _five(r.get("average_rating"))
|
||||||
|
title = (r.get("summary") or "").strip()
|
||||||
|
# When a customer leaves the title blank the site fills in the full
|
||||||
|
# product name; that is not something the reviewer wrote.
|
||||||
|
if len(title) > 60 or "/" in title:
|
||||||
|
title = ""
|
||||||
|
reviews.append({
|
||||||
|
"author": (r.get("nickname") or "").strip() or None,
|
||||||
|
"rating": stars,
|
||||||
|
"title": title or None,
|
||||||
|
"body": body_text[:4000],
|
||||||
|
"review_date": r.get("created_at"),
|
||||||
|
"sentiment": sentiment_for(stars),
|
||||||
|
})
|
||||||
|
return {"rating": rating, "review_count": int(item.get("review_count") or 0) or None, "reviews": reviews}
|
||||||
@@ -34,6 +34,8 @@ class Listing:
|
|||||||
pincode_applied: bool = False
|
pincode_applied: bool = False
|
||||||
rating: Optional[Decimal] = None
|
rating: Optional[Decimal] = None
|
||||||
review_count: Optional[int] = None
|
review_count: Optional[int] = None
|
||||||
|
# {"5": n, ..., "1": n} when the source publishes a star breakdown.
|
||||||
|
rating_breakdown: Optional[Dict[str, int]] = None
|
||||||
# Customer reviews the page itself publishes (schema.org Review); stored
|
# Customer reviews the page itself publishes (schema.org Review); stored
|
||||||
# in elec.listing_review, not on the listing row.
|
# in elec.listing_review, not on the listing row.
|
||||||
reviews: List[Dict[str, Any]] = field(default_factory=list)
|
reviews: List[Dict[str, Any]] = field(default_factory=list)
|
||||||
|
|||||||
@@ -24,6 +24,11 @@ class BrandRef:
|
|||||||
aliases: tuple
|
aliases: tuple
|
||||||
sub_brands: tuple
|
sub_brands: tuple
|
||||||
official: tuple
|
official: tuple
|
||||||
|
# Brand-store path whose product pages publish real customer reviews as
|
||||||
|
# schema.org data (e.g. "samsung.com/in"); None when no free source exists.
|
||||||
|
reviews_site: Optional[str] = None
|
||||||
|
# Regex a reviews_site URL must match to be a single-variant product page.
|
||||||
|
reviews_product_url: Optional[str] = None
|
||||||
|
|
||||||
|
|
||||||
@dataclass(frozen=True)
|
@dataclass(frozen=True)
|
||||||
@@ -78,6 +83,8 @@ def load_reference() -> Reference:
|
|||||||
aliases=tuple(a.lower() for a in b.get("aliases", [])),
|
aliases=tuple(a.lower() for a in b.get("aliases", [])),
|
||||||
sub_brands=tuple(s.lower() for s in b.get("sub_brands", [])),
|
sub_brands=tuple(s.lower() for s in b.get("sub_brands", [])),
|
||||||
official=tuple(b.get("official", [])),
|
official=tuple(b.get("official", [])),
|
||||||
|
reviews_site=b.get("reviews_site"),
|
||||||
|
reviews_product_url=b.get("reviews_product_url"),
|
||||||
)
|
)
|
||||||
categories = {
|
categories = {
|
||||||
c["slug"]: CategoryRef(c["slug"], c["name"], tuple(c.get("search_terms", [])),
|
c["slug"]: CategoryRef(c["slug"], c["name"], tuple(c.get("search_terms", [])),
|
||||||
|
|||||||
@@ -7,11 +7,19 @@
|
|||||||
# parent, and are kept as the product family.
|
# parent, and are kept as the product family.
|
||||||
# official the brand's own Indian web domains. A product page on one of
|
# official the brand's own Indian web domains. A product page on one of
|
||||||
# these is the strongest evidence that a product exists.
|
# these is the strongest evidence that a product exists.
|
||||||
|
# reviews_site optional brand-store path whose product pages publish real
|
||||||
|
# customer ratings + reviews as schema.org JSON-LD, robots-allowed
|
||||||
|
# (checked 2026-10-01). `cli reviews` finds each verified product's
|
||||||
|
# page there. Add a brand only after checking its pages.
|
||||||
|
# reviews_product_url regex for that store's single-variant product pages
|
||||||
|
# (family/marketing pages carry no product data or reviews).
|
||||||
brands:
|
brands:
|
||||||
- name: Samsung
|
- name: Samsung
|
||||||
categories: [mobiles, laptops]
|
categories: [mobiles, laptops]
|
||||||
aliases: [samsung]
|
aliases: [samsung]
|
||||||
official: [samsung.com]
|
official: [samsung.com]
|
||||||
|
reviews_site: samsung.com/in
|
||||||
|
reviews_product_url: '-sm-[a-z0-9]+/?$' # .../galaxy-a56-5g-awesome-olive-256gb-sm-a566ezggins/
|
||||||
- name: Apple
|
- name: Apple
|
||||||
categories: [mobiles, laptops]
|
categories: [mobiles, laptops]
|
||||||
aliases: [apple]
|
aliases: [apple]
|
||||||
|
|||||||
215
backend/app/electronics/review_refresh.py
Normal file
215
backend/app/electronics/review_refresh.py
Normal file
@@ -0,0 +1,215 @@
|
|||||||
|
"""Refresh customer ratings and reviews for the verified catalogue, from free
|
||||||
|
sources that publish them and allow reading them:
|
||||||
|
|
||||||
|
1. Brand stores with real reviews in schema.org JSON-LD (brands.yaml
|
||||||
|
`reviews_site`, e.g. samsung.com/in): find each verified product's own page
|
||||||
|
by web search, accept it only for the same model AND variant, read it with
|
||||||
|
the normal collector page path, and store it as a brand_official listing.
|
||||||
|
2. Every page-read listing of a verified product is re-read for its rating,
|
||||||
|
star breakdown and reviews (JSON-LD, then the retailer's embedded data).
|
||||||
|
3. Vijay Sales listings additionally get their reviews from the site's own
|
||||||
|
public GraphQL endpoint.
|
||||||
|
|
||||||
|
Everything goes through PoliteClient (robots.txt, per-site pacing, circuit
|
||||||
|
breaker). Nothing is generated; a source that states nothing leaves nothing.
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import dataclasses
|
||||||
|
import logging
|
||||||
|
import re
|
||||||
|
from decimal import Decimal
|
||||||
|
from typing import Callable, Dict, List, Optional
|
||||||
|
|
||||||
|
from rapidfuzz import fuzz
|
||||||
|
|
||||||
|
from app.electronics.collector import Collector, RunOptions
|
||||||
|
from app.electronics.db import repository as repo
|
||||||
|
from app.electronics.db.connection import connect
|
||||||
|
from app.electronics.extract import vijaysales_reviews
|
||||||
|
from app.electronics.extract.embedded_ratings import IGNORED_DOMAINS, embedded_rating
|
||||||
|
from app.electronics.extract.jsonld import extract_products
|
||||||
|
from app.electronics.reference import load_reference
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def _verified_products(category: Optional[str]) -> List[dict]:
|
||||||
|
sql = ("SELECT v.product_id, v.brand_slug, v.category, p.model, p.model_norm, p.ram_gb, p.storage_gb "
|
||||||
|
"FROM elec.v_brand_catalog v JOIN elec.product p ON p.id = v.product_id")
|
||||||
|
params: tuple = ()
|
||||||
|
if category:
|
||||||
|
sql += " WHERE v.category = %s"
|
||||||
|
params = (category,)
|
||||||
|
with connect() as conn:
|
||||||
|
return list(conn.execute(sql + " ORDER BY v.product_id", params))
|
||||||
|
|
||||||
|
|
||||||
|
def _has_page_on(product_id: int, domain: str) -> bool:
|
||||||
|
with connect() as conn:
|
||||||
|
return conn.execute(
|
||||||
|
"SELECT 1 FROM elec.v_product_availability WHERE product_id = %s AND domain = %s "
|
||||||
|
"AND source_type IN ('scraped_page','brand_official') LIMIT 1", (product_id, domain),
|
||||||
|
).fetchone() is not None
|
||||||
|
|
||||||
|
|
||||||
|
def _gb(value) -> str:
|
||||||
|
return f"{format(Decimal(value).normalize(), 'f')}GB" if value is not None else ""
|
||||||
|
|
||||||
|
|
||||||
|
# Words that make a different model, not a different colour of the same one.
|
||||||
|
_MODEL_QUALIFIERS = frozenset({"ultra", "plus", "pro", "max", "fe", "lite", "edge", "mini", "neo",
|
||||||
|
"prime", "flip", "fold", "slim", "+"})
|
||||||
|
|
||||||
|
|
||||||
|
def _same_model(page_model: str, product_model: str) -> bool:
|
||||||
|
"""Same model line: every product word present, identical model-number
|
||||||
|
words, and nothing extra but non-model words (e.g. a colour the parser
|
||||||
|
left in: "galaxy a56 olive"). Fuzzy scores are no use here - they rate
|
||||||
|
"galaxy a57" vs "galaxy a56" at ~90."""
|
||||||
|
page, mine = set(page_model.split()), set(product_model.split())
|
||||||
|
numbered = lambda words: {w for w in words if any(c.isdigit() for c in w)} # noqa: E731
|
||||||
|
extra = page - mine
|
||||||
|
return mine <= page and numbered(page) == numbered(mine) and not (extra & _MODEL_QUALIFIERS)
|
||||||
|
|
||||||
|
|
||||||
|
def _same_variant(parsed, product: dict) -> bool:
|
||||||
|
if not parsed.model_norm or not _same_model(parsed.model_norm, product["model_norm"] or ""):
|
||||||
|
return False
|
||||||
|
for attr in ("ram_gb", "storage_gb"):
|
||||||
|
mine, theirs = product[attr], getattr(parsed, attr)
|
||||||
|
if mine is not None and theirs is not None and Decimal(mine) != Decimal(theirs):
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def discover_brand_pages(category: Optional[str], budget: int, stats: Dict[str, int],
|
||||||
|
progress: Callable[[str], None]) -> None:
|
||||||
|
ref = load_reference()
|
||||||
|
by_category: Dict[str, List[dict]] = {}
|
||||||
|
for p in _verified_products(category):
|
||||||
|
brand = ref.brands.get(p["brand_slug"])
|
||||||
|
if brand and brand.reviews_site:
|
||||||
|
by_category.setdefault(p["category"], []).append(p)
|
||||||
|
for cat, products in by_category.items():
|
||||||
|
brands = sorted({p["brand_slug"] for p in products})
|
||||||
|
col = Collector(RunOptions(category=cat, brands=brands, search_budget=budget, use_llm=False),
|
||||||
|
progress=progress)
|
||||||
|
try:
|
||||||
|
for p in products:
|
||||||
|
brand = ref.brands[p["brand_slug"]]
|
||||||
|
domain = brand.reviews_site.split("/")[0]
|
||||||
|
if _has_page_on(p["product_id"], domain):
|
||||||
|
stats["brand_page_already_known"] = stats.get("brand_page_already_known", 0) + 1
|
||||||
|
continue
|
||||||
|
model = p["model"] or p["model_norm"]
|
||||||
|
queries = [f"site:{brand.reviews_site} {brand.name} {model} {_gb(p['ram_gb'])} {_gb(p['storage_gb'])}",
|
||||||
|
f"site:{brand.reviews_site} {model} 5G {_gb(p['storage_gb'])} buy"]
|
||||||
|
hits, query = [], ""
|
||||||
|
for q in (re.sub(r"\s+", " ", q).strip() for q in queries):
|
||||||
|
hits += [(h, q) for h in (col.engine.text(q, max_results=10) or [])]
|
||||||
|
seen = set()
|
||||||
|
for hit, query in hits:
|
||||||
|
if hit.url in seen:
|
||||||
|
continue
|
||||||
|
seen.add(hit.url)
|
||||||
|
# Brand-store titles often omit the brand ("Galaxy A56 5G 8GB/256GB ...",
|
||||||
|
# sometimes behind a "Business |" prefix). On the brand's own store
|
||||||
|
# the brand is not in doubt, so state it for the title parser.
|
||||||
|
title = re.sub(r"^\s*Business\s*\|\s*", "", hit.title or "")
|
||||||
|
if not title.lower().startswith(brand.name.lower()):
|
||||||
|
title = f"{brand.name} {title}"
|
||||||
|
hit = dataclasses.replace(hit, title=title)
|
||||||
|
if f"{brand.reviews_site}/" not in hit.url:
|
||||||
|
continue
|
||||||
|
if brand.reviews_product_url and not re.search(brand.reviews_product_url, hit.url.split("?")[0]):
|
||||||
|
continue # a family/marketing page, not one variant's product page
|
||||||
|
accepted = col._accept_hit(hit, brand)
|
||||||
|
if not accepted:
|
||||||
|
continue
|
||||||
|
site, parsed = accepted
|
||||||
|
if not _same_variant(parsed, p):
|
||||||
|
continue
|
||||||
|
res = col.client.get(hit.url)
|
||||||
|
if not res.ok:
|
||||||
|
continue
|
||||||
|
listing = col.listing_from_page(hit, site, parsed, query, res.text, res.final_url)
|
||||||
|
# Only a page with schema.org product data for this exact variant.
|
||||||
|
if listing is None or not listing.parser.startswith("jsonld") or not _same_variant(listing, p):
|
||||||
|
continue
|
||||||
|
col.store(listing)
|
||||||
|
if _has_page_on(p["product_id"], domain): # actually matched to this product
|
||||||
|
stats["brand_pages_added"] = stats.get("brand_pages_added", 0) + 1
|
||||||
|
progress(f" {brand.name}: {listing.title} -> rating {listing.rating}, "
|
||||||
|
f"{len(listing.reviews)} reviews")
|
||||||
|
break
|
||||||
|
finally:
|
||||||
|
col.client.close()
|
||||||
|
repo.refresh_verification()
|
||||||
|
|
||||||
|
|
||||||
|
def reread_pages(category: Optional[str], limit: int, stats: Dict[str, int],
|
||||||
|
progress: Callable[[str], None]) -> None:
|
||||||
|
from app.electronics.net.polite_client import PoliteClient
|
||||||
|
|
||||||
|
rows = repo.listings_for_review_backfill(category)[:limit]
|
||||||
|
with PoliteClient() as client:
|
||||||
|
for row in rows:
|
||||||
|
domain = row["domain"]
|
||||||
|
if domain in IGNORED_DOMAINS:
|
||||||
|
continue
|
||||||
|
stats["pages"] = stats.get("pages", 0) + 1
|
||||||
|
rating = count = breakdown = None
|
||||||
|
reviews: List[dict] = []
|
||||||
|
res = client.get(row["source_url"])
|
||||||
|
if res.ok:
|
||||||
|
products = extract_products(res.text)
|
||||||
|
match = next((x for x in products if x.get("sku") and x["sku"] == row["source_sku"]), None)
|
||||||
|
if match is None:
|
||||||
|
title = (row["title"] or "").lower()
|
||||||
|
scored = [(fuzz.token_set_ratio(x["name"].lower(), title), x) for x in products]
|
||||||
|
scored = [s for s in scored if s[0] >= 85]
|
||||||
|
match = max(scored, key=lambda s: s[0])[1] if scored else None
|
||||||
|
if match is not None:
|
||||||
|
rating, count, reviews = match.get("rating"), match.get("review_count"), match.get("reviews") or []
|
||||||
|
# The retailer's own embedded data: the rating when JSON-LD has none,
|
||||||
|
# and the star breakdown, which only it publishes.
|
||||||
|
emb = embedded_rating(domain, res.text, res.final_url or row["source_url"])
|
||||||
|
if emb:
|
||||||
|
if rating is None:
|
||||||
|
rating, count = emb["rating"], emb["review_count"]
|
||||||
|
breakdown = emb["breakdown"]
|
||||||
|
if domain == "vijaysales.com":
|
||||||
|
gql = vijaysales_reviews.graphql_url(row["source_url"])
|
||||||
|
vres = client.get(gql, accept_non_html=True) if gql else None
|
||||||
|
vs = vijaysales_reviews.parse(vres.text) if vres is not None and vres.ok else None
|
||||||
|
if vs:
|
||||||
|
rating, count = rating or vs["rating"], count or vs["review_count"]
|
||||||
|
reviews = reviews or vs["reviews"]
|
||||||
|
if rating is not None and Decimal(0) < Decimal(rating) <= Decimal(5):
|
||||||
|
repo.update_listing_rating(row["listing_id"], rating, count, breakdown)
|
||||||
|
stats["rated"] = stats.get("rated", 0) + 1
|
||||||
|
if reviews:
|
||||||
|
stats["reviews_stored"] = stats.get("reviews_stored", 0) + repo.save_reviews(row["listing_id"], reviews)
|
||||||
|
progress(f" {domain:22} rating={rating} count={count} breakdown={'yes' if breakdown else 'no'} "
|
||||||
|
f"reviews={len(reviews)}")
|
||||||
|
|
||||||
|
|
||||||
|
def refresh_reviews(category: Optional[str] = None, limit: int = 500, budget: int = 60,
|
||||||
|
discover: bool = True, progress: Optional[Callable[[str], None]] = None) -> Dict[str, int]:
|
||||||
|
progress = progress or (lambda m: logger.info(m))
|
||||||
|
stats: Dict[str, int] = {}
|
||||||
|
run_id = repo.start_run("reviews", {"category": category, "limit": limit, "discover": discover})
|
||||||
|
status, error = "done", None
|
||||||
|
try:
|
||||||
|
if discover:
|
||||||
|
progress("Finding brand-store pages with reviews ...")
|
||||||
|
discover_brand_pages(category, budget, stats, progress)
|
||||||
|
progress("Re-reading ratings and reviews from product pages ...")
|
||||||
|
reread_pages(category, limit, stats, progress)
|
||||||
|
except Exception as exc:
|
||||||
|
status, error = "failed", repr(exc)
|
||||||
|
raise
|
||||||
|
finally:
|
||||||
|
repo.finish_run(run_id, status, stats, error)
|
||||||
|
return stats
|
||||||
@@ -199,7 +199,8 @@ USER_AGENT = os.getenv(
|
|||||||
REQUEST_TIMEOUT_SECONDS = int(os.getenv("REQUEST_TIMEOUT_SECONDS", "20"))
|
REQUEST_TIMEOUT_SECONDS = int(os.getenv("REQUEST_TIMEOUT_SECONDS", "20"))
|
||||||
ELEC_SITE_MIN_INTERVAL_SECONDS = float(os.getenv("ELEC_SITE_MIN_INTERVAL_SECONDS", "3"))
|
ELEC_SITE_MIN_INTERVAL_SECONDS = float(os.getenv("ELEC_SITE_MIN_INTERVAL_SECONDS", "3"))
|
||||||
ELEC_BREAKER_COOLDOWN_HOURS = float(os.getenv("ELEC_BREAKER_COOLDOWN_HOURS", "24"))
|
ELEC_BREAKER_COOLDOWN_HOURS = float(os.getenv("ELEC_BREAKER_COOLDOWN_HOURS", "24"))
|
||||||
ELEC_MAX_PAGE_BYTES = int(os.getenv("ELEC_MAX_PAGE_BYTES", str(3 * 1024 * 1024)))
|
# Brand stores embed their reviews in the page: samsung.com/in pages run ~3.1 MB.
|
||||||
|
ELEC_MAX_PAGE_BYTES = int(os.getenv("ELEC_MAX_PAGE_BYTES", str(6 * 1024 * 1024)))
|
||||||
ELEC_PROBE_TTL_DAYS = int(os.getenv("ELEC_PROBE_TTL_DAYS", "7"))
|
ELEC_PROBE_TTL_DAYS = int(os.getenv("ELEC_PROBE_TTL_DAYS", "7"))
|
||||||
MIN_IMAGE_BYTES = int(os.getenv("MIN_IMAGE_BYTES", "3000"))
|
MIN_IMAGE_BYTES = int(os.getenv("MIN_IMAGE_BYTES", "3000"))
|
||||||
|
|
||||||
|
|||||||
@@ -92,8 +92,9 @@ async def get_product(product_id: int) -> Dict[str, Any]:
|
|||||||
"""Full details of one product by its product_id (from search_products).
|
"""Full details of one product by its product_id (from search_products).
|
||||||
|
|
||||||
Returns per-platform offers (price, MRP, source URL, when seen), normalised specs,
|
Returns per-platform offers (price, MRP, source URL, when seen), normalised specs,
|
||||||
image URLs, the overall rating with per-platform sources (null if none is published),
|
image URLs, the overall rating with per-platform sources and, when a platform publishes
|
||||||
and up to 10 real customer reviews (often empty).
|
one, a 5-to-1 star breakdown (rating is null if no platform publishes a rating), and up
|
||||||
|
to 10 real customer reviews with the site each came from (often empty).
|
||||||
"""
|
"""
|
||||||
d = await _run(elec.product, int(product_id))
|
d = await _run(elec.product, int(product_id))
|
||||||
return {
|
return {
|
||||||
|
|||||||
@@ -142,3 +142,155 @@ def test_api_serves_ratings_reviews_and_out_of_stock_price(db, client):
|
|||||||
assert {s["site"] for s in detail["rating"]["sources"]} == {"Amazon.in", "Croma"}
|
assert {s["site"] for s in detail["rating"]["sources"]} == {"Amazon.in", "Croma"}
|
||||||
assert [r["body"] for r in detail["reviews"]] == ["Battery lasts all day.", "Heats up."]
|
assert [r["body"] for r in detail["reviews"]] == ["Battery lasts all day.", "Heats up."]
|
||||||
assert all(r["source_url"].startswith("https://") for r in detail["reviews"])
|
assert all(r["source_url"].startswith("https://") for r in detail["reviews"])
|
||||||
|
assert detail["rating"]["breakdown"] is None # no platform published one
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Free sources: brand-store JSON-LD, retailer-embedded ratings, Vijay Sales
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
def _ld(*nodes) -> str:
|
||||||
|
return "".join(f'<script type="application/ld+json">{json.dumps(n)}</script>' for n in nodes)
|
||||||
|
|
||||||
|
|
||||||
|
def _review(author, stars, body, date="2026-09-30T17:06:05.000+00:00"):
|
||||||
|
return {"@type": "Review", "author": {"@type": "Person", "name": author}, "reviewBody": body,
|
||||||
|
"datePublished": date, "reviewRating": {"@type": "Rating", "bestRating": 5, "ratingValue": stars}}
|
||||||
|
|
||||||
|
|
||||||
|
def test_nameless_review_node_is_merged_by_id():
|
||||||
|
# The samsung.com/in shape: Bazaarvoice publishes ratings in a second,
|
||||||
|
# nameless Product node that shares the real product's @id.
|
||||||
|
pid = "https://www.samsung.com/in/smartphones/galaxy-a/galaxy-a56-5g-awesome-olive-256gb-sm-a566ezggins/"
|
||||||
|
html = _ld(
|
||||||
|
{"@context": "https://schema.org", "@type": "Product", "@id": pid, "name": "Galaxy A56 5G (8 GB Memory)",
|
||||||
|
"sku": "SM-A566EZGG", "offers": {"@type": "Offer", "price": "48999", "priceCurrency": "INR"}},
|
||||||
|
{"@context": "https://schema.org", "@type": "Product", "@id": pid,
|
||||||
|
"aggregateRating": {"@type": "AggregateRating", "ratingValue": "4.5", "ratingCount": 1224},
|
||||||
|
"review": [_review("Vedant", 5, "Great and useful AI features"), _review("Ravi", 1, "Heats up")]},
|
||||||
|
)
|
||||||
|
[p] = extract_products(html)
|
||||||
|
assert p["name"] == "Galaxy A56 5G (8 GB Memory)" and p["price"] == Decimal("48999")
|
||||||
|
assert p["rating"] == Decimal("4.5") and p["review_count"] == 1224
|
||||||
|
assert [(r["author"], r["rating"], r["body"]) for r in p["reviews"]] == [
|
||||||
|
("Vedant", Decimal("5.0"), "Great and useful AI features"), ("Ravi", Decimal("1.0"), "Heats up")]
|
||||||
|
|
||||||
|
|
||||||
|
def test_nameless_review_node_without_id_is_not_guessed_between_products():
|
||||||
|
lone = {"@type": "Product", "aggregateRating": {"ratingValue": "4.1", "ratingCount": 9}}
|
||||||
|
[only] = extract_products(_ld({"@type": "Product", "name": "Galaxy A56"}, lone))
|
||||||
|
assert only["rating"] == Decimal("4.1") # one product on the page: unambiguous
|
||||||
|
two = extract_products(_ld({"@type": "Product", "name": "Galaxy A56"},
|
||||||
|
{"@type": "Product", "name": "Galaxy A36"}, lone))
|
||||||
|
assert all(p["rating"] is None for p in two) # two products: never guess which
|
||||||
|
|
||||||
|
|
||||||
|
def test_reliance_embedded_rating_is_read_from_the_pages_own_product_only():
|
||||||
|
from app.electronics.extract.embedded_ratings import embedded_rating
|
||||||
|
|
||||||
|
def page(app, recommended=None):
|
||||||
|
state = {"productDetailsPage": {"product": {"name": "Dell 15", "_custom_json": {"_app": app}}},
|
||||||
|
"recommendations": [{"_custom_json": {"_app": recommended}}] if recommended else []}
|
||||||
|
return f"<script>window.__INITIAL_STATE__={json.dumps(state)};window.x=1</script>"
|
||||||
|
|
||||||
|
url = "https://www.reliancedigital.in/product/dell-15-9991568"
|
||||||
|
got = embedded_rating("reliancedigital.in", page(
|
||||||
|
{"averageRating": 4, "ratingsCount": 3, "ratingsCountDetails": {"5": 1, "4": 1, "3": 1}}), url)
|
||||||
|
assert got == {"rating": Decimal("4.00"), "review_count": 3, "breakdown": {"5": 1, "4": 1, "3": 1}}
|
||||||
|
# "No ratings yet" is published as zeros; a recommended product's rating is not ours.
|
||||||
|
zero = page({"averageRating": 0, "ratingsCount": 0}, recommended={"averageRating": 4.8, "ratingsCount": 90})
|
||||||
|
assert embedded_rating("reliancedigital.in", zero, url) is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_poorvika_rating_needs_the_pages_own_slug():
|
||||||
|
from app.electronics.extract.embedded_ratings import embedded_rating
|
||||||
|
|
||||||
|
def page(slug, rating, count):
|
||||||
|
data = {"props": {"pageProps": {"additionalData": {"slug": slug, "rating": rating, "ratingCount": count}}}}
|
||||||
|
return f'<script id="__NEXT_DATA__" type="application/json">{json.dumps(data)}</script>'
|
||||||
|
|
||||||
|
slug = "xiaomi-14-civi-5g-shadow-black-256gb-8gb-ram"
|
||||||
|
url = f"https://www.poorvika.com/{slug}/p"
|
||||||
|
assert embedded_rating("poorvika.com", page(slug, 5, 1), url)["rating"] == Decimal("5.00")
|
||||||
|
assert embedded_rating("poorvika.com", page("some-other-phone", 5, 1), url) is None
|
||||||
|
assert embedded_rating("poorvika.com", page(slug, None, 0), url) is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_vasanth_template_rating_is_never_read():
|
||||||
|
from app.electronics.extract.embedded_ratings import embedded_rating
|
||||||
|
|
||||||
|
template = ('<h4 class="avg-mark">3.3</h4><a class="rating-reviews">(3 Reviews)</a>'
|
||||||
|
+ _ld({"@type": "Product", "name": "HP 15", "aggregateRating": {"ratingValue": "3.3", "reviewCount": 3}}))
|
||||||
|
assert embedded_rating("vasanthandco.in", template, "https://vasanthandco.in/product/174200102203/") is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_vijaysales_graphql_review_feed():
|
||||||
|
from app.electronics.extract import vijaysales_reviews as vs
|
||||||
|
|
||||||
|
url = "https://www.vijaysales.com/p/239904/acer-aspire-lite-al15-41-laptop-amd-ryzen-7-16gb"
|
||||||
|
assert vs.url_key(url) == "acer-aspire-lite-al15-41-laptop-amd-ryzen-7-16gb"
|
||||||
|
assert "url_key" in vs.graphql_url(url)
|
||||||
|
assert vs.url_key('https://x.in/p/a"b') is None # cannot break out of the query
|
||||||
|
product_name = "Acer Aspire Lite AL15-41 Laptop (AMD Ryzen 7/ 16GB DDR4 RAM/ 512GB SSD)"
|
||||||
|
body = json.dumps({"data": {"products": {"items": [
|
||||||
|
{"sku": "P1", "rating_summary": 0, "review_count": 0, "reviews": {"items": []}},
|
||||||
|
{"sku": "239904", "rating_summary": 90, "review_count": 2, "reviews": {"items": [
|
||||||
|
{"nickname": "Jay", "summary": product_name, "text": "All over performance is good",
|
||||||
|
"average_rating": 100, "created_at": "2026-05-02 10:00:00"},
|
||||||
|
{"nickname": "guest user", "summary": "Okay", "text": "", "average_rating": 60}]}}]}}})
|
||||||
|
got = vs.parse(body)
|
||||||
|
assert got["rating"] == Decimal("4.50") and got["review_count"] == 2
|
||||||
|
# Product-name "title" dropped; a review with no text is skipped.
|
||||||
|
assert got["reviews"] == [{"author": "Jay", "rating": Decimal("5.00"), "title": None,
|
||||||
|
"body": "All over performance is good", "review_date": "2026-05-02 10:00:00",
|
||||||
|
"sentiment": "positive"}]
|
||||||
|
assert vs.parse('{"data":{"products":{"items":[]}}}') is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_api_breakdown_only_when_a_platform_publishes_one(db, client):
|
||||||
|
from app.electronics.collector import Collector, RunOptions, RunStats
|
||||||
|
from app.electronics.db import repository as repo
|
||||||
|
from app.electronics.models import Listing
|
||||||
|
from app.electronics.normalise.title_parser import parse_title, variant_key
|
||||||
|
|
||||||
|
def listing(site, sku, price, source_type="search_snippet"):
|
||||||
|
title = "Samsung Galaxy S24 5G (8GB RAM, 256GB)"
|
||||||
|
p = parse_title(title, "mobiles")
|
||||||
|
l = Listing(site_domain=site, source_sku=sku, source_url=f"https://www.{site}/p/{sku}",
|
||||||
|
source_type=source_type, brand_slug="samsung", category="mobiles", title=title,
|
||||||
|
evidence_text=f"{title} ₹{price}", confidence=0.5, parser="test",
|
||||||
|
model=p.model, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=Decimal(price))
|
||||||
|
l.model_norm, l.variant_key = p.model_norm, variant_key(p, "mobiles")
|
||||||
|
return l
|
||||||
|
|
||||||
|
c = Collector.__new__(Collector)
|
||||||
|
c.opt = RunOptions(category="mobiles", brands=["samsung"])
|
||||||
|
c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats()
|
||||||
|
b = listing("reliancedigital.in", "rd-1", 73999, "scraped_page")
|
||||||
|
b.rating, b.review_count, b.rating_breakdown = Decimal("4.0"), 4, {"5": 2, "4": 1, "1": 1}
|
||||||
|
c.store(listing("amazon.in", "B0CS5XW6TN", 74999))
|
||||||
|
c.store(b)
|
||||||
|
repo.refresh_verification()
|
||||||
|
pid = client.get("/api/elec/products", params={"category": "mobiles"}).json()["products"][0]["product_id"]
|
||||||
|
rating = client.get(f"/api/elec/products/{pid}").json()["rating"]
|
||||||
|
assert rating["breakdown"] == [
|
||||||
|
{"stars": 5, "count": 2, "percent": 50}, {"stars": 4, "count": 1, "percent": 25},
|
||||||
|
{"stars": 3, "count": 0, "percent": 0}, {"stars": 2, "count": 0, "percent": 0},
|
||||||
|
{"stars": 1, "count": 1, "percent": 25}]
|
||||||
|
assert rating["sources"][0]["breakdown"] == {"5": 2, "4": 1, "1": 1}
|
||||||
|
|
||||||
|
|
||||||
|
def test_brand_store_page_must_be_the_exact_model_and_variant():
|
||||||
|
from app.electronics.normalise.title_parser import parse_title
|
||||||
|
from app.electronics.review_refresh import _same_variant
|
||||||
|
|
||||||
|
a56 = {"model_norm": "galaxy a56", "ram_gb": Decimal("8"), "storage_gb": Decimal("256")}
|
||||||
|
page = lambda t: parse_title(t, "mobiles", expected_brand="samsung") # noqa: E731
|
||||||
|
assert _same_variant(page("Samsung Galaxy A56 5G 8GB/256GB (Olive)"), a56)
|
||||||
|
# A neighbouring model scores ~90 on fuzzy matching; it must still be refused.
|
||||||
|
assert not _same_variant(page("Samsung Galaxy A57 5G 8GB/256GB (Navy)"), a56)
|
||||||
|
assert not _same_variant(page("Samsung Galaxy A56 5G 12GB/256GB (Olive)"), a56)
|
||||||
|
assert not _same_variant(page("Samsung Galaxy A56 5G 8GB/128GB (Blue)"), a56)
|
||||||
|
s25 = {"model_norm": "galaxy s25", "ram_gb": Decimal("12"), "storage_gb": Decimal("256")}
|
||||||
|
assert _same_variant(page("Samsung Galaxy S25 12GB/256GB (Icyblue)"), s25)
|
||||||
|
assert not _same_variant(page("Samsung Galaxy S25 Ultra 12GB/256GB (Titanium Black)"), s25)
|
||||||
|
assert not _same_variant(page("Samsung Galaxy S25+ 12GB/256GB (Navy)"), s25)
|
||||||
|
|||||||
Reference in New Issue
Block a user