Files
loyaly-catalogue/backend/app/electronics/review_refresh.py
2026-10-05 11:21:39 +05:30

216 lines
11 KiB
Python

"""Refresh customer ratings and reviews for the verified catalogue, from free
sources that publish them and allow reading them:
1. Brand stores with real reviews in schema.org JSON-LD (brands.yaml
`reviews_site`, e.g. samsung.com/in): find each verified product's own page
by web search, accept it only for the same model AND variant, read it with
the normal collector page path, and store it as a brand_official listing.
2. Every page-read listing of a verified product is re-read for its rating,
star breakdown and reviews (JSON-LD, then the retailer's embedded data).
3. Vijay Sales listings additionally get their reviews from the site's own
public GraphQL endpoint.
Everything goes through PoliteClient (robots.txt, per-site pacing, circuit
breaker). Nothing is generated; a source that states nothing leaves nothing.
"""
from __future__ import annotations
import dataclasses
import logging
import re
from decimal import Decimal
from typing import Callable, Dict, List, Optional
from rapidfuzz import fuzz
from app.electronics.collector import Collector, RunOptions
from app.electronics.db import repository as repo
from app.electronics.db.connection import connect
from app.electronics.extract import vijaysales_reviews
from app.electronics.extract.embedded_ratings import IGNORED_DOMAINS, embedded_rating
from app.electronics.extract.jsonld import extract_products
from app.electronics.reference import load_reference
logger = logging.getLogger(__name__)
def _verified_products(category: Optional[str]) -> List[dict]:
sql = ("SELECT v.product_id, v.brand_slug, v.category, p.model, p.model_norm, p.ram_gb, p.storage_gb "
"FROM elec.v_brand_catalog v JOIN elec.product p ON p.id = v.product_id")
params: tuple = ()
if category:
sql += " WHERE v.category = %s"
params = (category,)
with connect() as conn:
return list(conn.execute(sql + " ORDER BY v.product_id", params))
def _has_page_on(product_id: int, domain: str) -> bool:
with connect() as conn:
return conn.execute(
"SELECT 1 FROM elec.v_product_availability WHERE product_id = %s AND domain = %s "
"AND source_type IN ('scraped_page','brand_official') LIMIT 1", (product_id, domain),
).fetchone() is not None
def _gb(value) -> str:
return f"{format(Decimal(value).normalize(), 'f')}GB" if value is not None else ""
# Words that make a different model, not a different colour of the same one.
_MODEL_QUALIFIERS = frozenset({"ultra", "plus", "pro", "max", "fe", "lite", "edge", "mini", "neo",
"prime", "flip", "fold", "slim", "+"})
def _same_model(page_model: str, product_model: str) -> bool:
"""Same model line: every product word present, identical model-number
words, and nothing extra but non-model words (e.g. a colour the parser
left in: "galaxy a56 olive"). Fuzzy scores are no use here - they rate
"galaxy a57" vs "galaxy a56" at ~90."""
page, mine = set(page_model.split()), set(product_model.split())
numbered = lambda words: {w for w in words if any(c.isdigit() for c in w)} # noqa: E731
extra = page - mine
return mine <= page and numbered(page) == numbered(mine) and not (extra & _MODEL_QUALIFIERS)
def _same_variant(parsed, product: dict) -> bool:
if not parsed.model_norm or not _same_model(parsed.model_norm, product["model_norm"] or ""):
return False
for attr in ("ram_gb", "storage_gb"):
mine, theirs = product[attr], getattr(parsed, attr)
if mine is not None and theirs is not None and Decimal(mine) != Decimal(theirs):
return False
return True
def discover_brand_pages(category: Optional[str], budget: int, stats: Dict[str, int],
progress: Callable[[str], None]) -> None:
ref = load_reference()
by_category: Dict[str, List[dict]] = {}
for p in _verified_products(category):
brand = ref.brands.get(p["brand_slug"])
if brand and brand.reviews_site:
by_category.setdefault(p["category"], []).append(p)
for cat, products in by_category.items():
brands = sorted({p["brand_slug"] for p in products})
col = Collector(RunOptions(category=cat, brands=brands, search_budget=budget, use_llm=False),
progress=progress)
try:
for p in products:
brand = ref.brands[p["brand_slug"]]
domain = brand.reviews_site.split("/")[0]
if _has_page_on(p["product_id"], domain):
stats["brand_page_already_known"] = stats.get("brand_page_already_known", 0) + 1
continue
model = p["model"] or p["model_norm"]
queries = [f"site:{brand.reviews_site} {brand.name} {model} {_gb(p['ram_gb'])} {_gb(p['storage_gb'])}",
f"site:{brand.reviews_site} {model} 5G {_gb(p['storage_gb'])} buy"]
hits, query = [], ""
for q in (re.sub(r"\s+", " ", q).strip() for q in queries):
hits += [(h, q) for h in (col.engine.text(q, max_results=10) or [])]
seen = set()
for hit, query in hits:
if hit.url in seen:
continue
seen.add(hit.url)
# Brand-store titles often omit the brand ("Galaxy A56 5G 8GB/256GB ...",
# sometimes behind a "Business |" prefix). On the brand's own store
# the brand is not in doubt, so state it for the title parser.
title = re.sub(r"^\s*Business\s*\|\s*", "", hit.title or "")
if not title.lower().startswith(brand.name.lower()):
title = f"{brand.name} {title}"
hit = dataclasses.replace(hit, title=title)
if f"{brand.reviews_site}/" not in hit.url:
continue
if brand.reviews_product_url and not re.search(brand.reviews_product_url, hit.url.split("?")[0]):
continue # a family/marketing page, not one variant's product page
accepted = col._accept_hit(hit, brand)
if not accepted:
continue
site, parsed = accepted
if not _same_variant(parsed, p):
continue
res = col.client.get(hit.url)
if not res.ok:
continue
listing = col.listing_from_page(hit, site, parsed, query, res.text, res.final_url)
# Only a page with schema.org product data for this exact variant.
if listing is None or not listing.parser.startswith("jsonld") or not _same_variant(listing, p):
continue
col.store(listing)
if _has_page_on(p["product_id"], domain): # actually matched to this product
stats["brand_pages_added"] = stats.get("brand_pages_added", 0) + 1
progress(f" {brand.name}: {listing.title} -> rating {listing.rating}, "
f"{len(listing.reviews)} reviews")
break
finally:
col.client.close()
repo.refresh_verification()
def reread_pages(category: Optional[str], limit: int, stats: Dict[str, int],
progress: Callable[[str], None]) -> None:
from app.electronics.net.polite_client import PoliteClient
rows = repo.listings_for_review_backfill(category)[:limit]
with PoliteClient() as client:
for row in rows:
domain = row["domain"]
if domain in IGNORED_DOMAINS:
continue
stats["pages"] = stats.get("pages", 0) + 1
rating = count = breakdown = None
reviews: List[dict] = []
res = client.get(row["source_url"])
if res.ok:
products = extract_products(res.text)
match = next((x for x in products if x.get("sku") and x["sku"] == row["source_sku"]), None)
if match is None:
title = (row["title"] or "").lower()
scored = [(fuzz.token_set_ratio(x["name"].lower(), title), x) for x in products]
scored = [s for s in scored if s[0] >= 85]
match = max(scored, key=lambda s: s[0])[1] if scored else None
if match is not None:
rating, count, reviews = match.get("rating"), match.get("review_count"), match.get("reviews") or []
# The retailer's own embedded data: the rating when JSON-LD has none,
# and the star breakdown, which only it publishes.
emb = embedded_rating(domain, res.text, res.final_url or row["source_url"])
if emb:
if rating is None:
rating, count = emb["rating"], emb["review_count"]
breakdown = emb["breakdown"]
if domain == "vijaysales.com":
gql = vijaysales_reviews.graphql_url(row["source_url"])
vres = client.get(gql, accept_non_html=True) if gql else None
vs = vijaysales_reviews.parse(vres.text) if vres is not None and vres.ok else None
if vs:
rating, count = rating or vs["rating"], count or vs["review_count"]
reviews = reviews or vs["reviews"]
if rating is not None and Decimal(0) < Decimal(rating) <= Decimal(5):
repo.update_listing_rating(row["listing_id"], rating, count, breakdown)
stats["rated"] = stats.get("rated", 0) + 1
if reviews:
stats["reviews_stored"] = stats.get("reviews_stored", 0) + repo.save_reviews(row["listing_id"], reviews)
progress(f" {domain:22} rating={rating} count={count} breakdown={'yes' if breakdown else 'no'} "
f"reviews={len(reviews)}")
def refresh_reviews(category: Optional[str] = None, limit: int = 500, budget: int = 60,
discover: bool = True, progress: Optional[Callable[[str], None]] = None) -> Dict[str, int]:
progress = progress or (lambda m: logger.info(m))
stats: Dict[str, int] = {}
run_id = repo.start_run("reviews", {"category": category, "limit": limit, "discover": discover})
status, error = "done", None
try:
if discover:
progress("Finding brand-store pages with reviews ...")
discover_brand_pages(category, budget, stats, progress)
progress("Re-reading ratings and reviews from product pages ...")
reread_pages(category, limit, stats, progress)
except Exception as exc:
status, error = "failed", repr(exc)
raise
finally:
repo.finish_run(run_id, status, stats, error)
return stats