"""Refresh customer ratings and reviews for the verified catalogue, from free sources that publish them and allow reading them: 1. Brand stores with real reviews in schema.org JSON-LD (brands.yaml `reviews_site`, e.g. samsung.com/in): find each verified product's own page by web search, accept it only for the same model AND variant, read it with the normal collector page path, and store it as a brand_official listing. 2. Every page-read listing of a verified product is re-read for its rating, star breakdown and reviews (JSON-LD, then the retailer's embedded data). 3. Vijay Sales listings additionally get their reviews from the site's own public GraphQL endpoint. Everything goes through PoliteClient (robots.txt, per-site pacing, circuit breaker). Nothing is generated; a source that states nothing leaves nothing. """ from __future__ import annotations import dataclasses import logging import re from decimal import Decimal from typing import Callable, Dict, List, Optional from rapidfuzz import fuzz from app.electronics.collector import Collector, RunOptions from app.electronics.db import repository as repo from app.electronics.db.connection import connect from app.electronics.extract import vijaysales_reviews from app.electronics.extract.embedded_ratings import IGNORED_DOMAINS, embedded_rating from app.electronics.extract.jsonld import extract_products from app.electronics.reference import load_reference logger = logging.getLogger(__name__) def _verified_products(category: Optional[str]) -> List[dict]: sql = ("SELECT v.product_id, v.brand_slug, v.category, p.model, p.model_norm, p.ram_gb, p.storage_gb " "FROM elec.v_brand_catalog v JOIN elec.product p ON p.id = v.product_id") params: tuple = () if category: sql += " WHERE v.category = %s" params = (category,) with connect() as conn: return list(conn.execute(sql + " ORDER BY v.product_id", params)) def _has_page_on(product_id: int, domain: str) -> bool: with connect() as conn: return conn.execute( "SELECT 1 FROM elec.v_product_availability WHERE product_id = %s AND domain = %s " "AND source_type IN ('scraped_page','brand_official') LIMIT 1", (product_id, domain), ).fetchone() is not None def _gb(value) -> str: return f"{format(Decimal(value).normalize(), 'f')}GB" if value is not None else "" # Words that make a different model, not a different colour of the same one. _MODEL_QUALIFIERS = frozenset({"ultra", "plus", "pro", "max", "fe", "lite", "edge", "mini", "neo", "prime", "flip", "fold", "slim", "+"}) def _same_model(page_model: str, product_model: str) -> bool: """Same model line: every product word present, identical model-number words, and nothing extra but non-model words (e.g. a colour the parser left in: "galaxy a56 olive"). Fuzzy scores are no use here - they rate "galaxy a57" vs "galaxy a56" at ~90.""" page, mine = set(page_model.split()), set(product_model.split()) numbered = lambda words: {w for w in words if any(c.isdigit() for c in w)} # noqa: E731 extra = page - mine return mine <= page and numbered(page) == numbered(mine) and not (extra & _MODEL_QUALIFIERS) def _same_variant(parsed, product: dict) -> bool: if not parsed.model_norm or not _same_model(parsed.model_norm, product["model_norm"] or ""): return False for attr in ("ram_gb", "storage_gb"): mine, theirs = product[attr], getattr(parsed, attr) if mine is not None and theirs is not None and Decimal(mine) != Decimal(theirs): return False return True def discover_brand_pages(category: Optional[str], budget: int, stats: Dict[str, int], progress: Callable[[str], None]) -> None: ref = load_reference() by_category: Dict[str, List[dict]] = {} for p in _verified_products(category): brand = ref.brands.get(p["brand_slug"]) if brand and brand.reviews_site: by_category.setdefault(p["category"], []).append(p) for cat, products in by_category.items(): brands = sorted({p["brand_slug"] for p in products}) col = Collector(RunOptions(category=cat, brands=brands, search_budget=budget, use_llm=False), progress=progress) try: for p in products: brand = ref.brands[p["brand_slug"]] domain = brand.reviews_site.split("/")[0] if _has_page_on(p["product_id"], domain): stats["brand_page_already_known"] = stats.get("brand_page_already_known", 0) + 1 continue model = p["model"] or p["model_norm"] queries = [f"site:{brand.reviews_site} {brand.name} {model} {_gb(p['ram_gb'])} {_gb(p['storage_gb'])}", f"site:{brand.reviews_site} {model} 5G {_gb(p['storage_gb'])} buy"] hits, query = [], "" for q in (re.sub(r"\s+", " ", q).strip() for q in queries): hits += [(h, q) for h in (col.engine.text(q, max_results=10) or [])] seen = set() for hit, query in hits: if hit.url in seen: continue seen.add(hit.url) # Brand-store titles often omit the brand ("Galaxy A56 5G 8GB/256GB ...", # sometimes behind a "Business |" prefix). On the brand's own store # the brand is not in doubt, so state it for the title parser. title = re.sub(r"^\s*Business\s*\|\s*", "", hit.title or "") if not title.lower().startswith(brand.name.lower()): title = f"{brand.name} {title}" hit = dataclasses.replace(hit, title=title) if f"{brand.reviews_site}/" not in hit.url: continue if brand.reviews_product_url and not re.search(brand.reviews_product_url, hit.url.split("?")[0]): continue # a family/marketing page, not one variant's product page accepted = col._accept_hit(hit, brand) if not accepted: continue site, parsed = accepted if not _same_variant(parsed, p): continue res = col.client.get(hit.url) if not res.ok: continue listing = col.listing_from_page(hit, site, parsed, query, res.text, res.final_url) # Only a page with schema.org product data for this exact variant. if listing is None or not listing.parser.startswith("jsonld") or not _same_variant(listing, p): continue col.store(listing) if _has_page_on(p["product_id"], domain): # actually matched to this product stats["brand_pages_added"] = stats.get("brand_pages_added", 0) + 1 progress(f" {brand.name}: {listing.title} -> rating {listing.rating}, " f"{len(listing.reviews)} reviews") break finally: col.client.close() repo.refresh_verification() def reread_pages(category: Optional[str], limit: int, stats: Dict[str, int], progress: Callable[[str], None]) -> None: from app.electronics.net.polite_client import PoliteClient rows = repo.listings_for_review_backfill(category)[:limit] with PoliteClient() as client: for row in rows: domain = row["domain"] if domain in IGNORED_DOMAINS: continue stats["pages"] = stats.get("pages", 0) + 1 rating = count = breakdown = None reviews: List[dict] = [] res = client.get(row["source_url"]) if res.ok: products = extract_products(res.text) match = next((x for x in products if x.get("sku") and x["sku"] == row["source_sku"]), None) if match is None: title = (row["title"] or "").lower() scored = [(fuzz.token_set_ratio(x["name"].lower(), title), x) for x in products] scored = [s for s in scored if s[0] >= 85] match = max(scored, key=lambda s: s[0])[1] if scored else None if match is not None: rating, count, reviews = match.get("rating"), match.get("review_count"), match.get("reviews") or [] # The retailer's own embedded data: the rating when JSON-LD has none, # and the star breakdown, which only it publishes. emb = embedded_rating(domain, res.text, res.final_url or row["source_url"]) if emb: if rating is None: rating, count = emb["rating"], emb["review_count"] breakdown = emb["breakdown"] if domain == "vijaysales.com": gql = vijaysales_reviews.graphql_url(row["source_url"]) vres = client.get(gql, accept_non_html=True) if gql else None vs = vijaysales_reviews.parse(vres.text) if vres is not None and vres.ok else None if vs: rating, count = rating or vs["rating"], count or vs["review_count"] reviews = reviews or vs["reviews"] if rating is not None and Decimal(0) < Decimal(rating) <= Decimal(5): repo.update_listing_rating(row["listing_id"], rating, count, breakdown) stats["rated"] = stats.get("rated", 0) + 1 if reviews: stats["reviews_stored"] = stats.get("reviews_stored", 0) + repo.save_reviews(row["listing_id"], reviews) progress(f" {domain:22} rating={rating} count={count} breakdown={'yes' if breakdown else 'no'} " f"reviews={len(reviews)}") def refresh_reviews(category: Optional[str] = None, limit: int = 500, budget: int = 60, discover: bool = True, progress: Optional[Callable[[str], None]] = None) -> Dict[str, int]: progress = progress or (lambda m: logger.info(m)) stats: Dict[str, int] = {} run_id = repo.start_run("reviews", {"category": category, "limit": limit, "discover": discover}) status, error = "done", None try: if discover: progress("Finding brand-store pages with reviews ...") discover_brand_pages(category, budget, stats, progress) progress("Re-reading ratings and reviews from product pages ...") reread_pages(category, limit, stats, progress) except Exception as exc: status, error = "failed", repr(exc) raise finally: repo.finish_run(run_id, status, stats, error) return stats