216 lines
11 KiB
Python
216 lines
11 KiB
Python
"""Refresh customer ratings and reviews for the verified catalogue, from free
|
|
sources that publish them and allow reading them:
|
|
|
|
1. Brand stores with real reviews in schema.org JSON-LD (brands.yaml
|
|
`reviews_site`, e.g. samsung.com/in): find each verified product's own page
|
|
by web search, accept it only for the same model AND variant, read it with
|
|
the normal collector page path, and store it as a brand_official listing.
|
|
2. Every page-read listing of a verified product is re-read for its rating,
|
|
star breakdown and reviews (JSON-LD, then the retailer's embedded data).
|
|
3. Vijay Sales listings additionally get their reviews from the site's own
|
|
public GraphQL endpoint.
|
|
|
|
Everything goes through PoliteClient (robots.txt, per-site pacing, circuit
|
|
breaker). Nothing is generated; a source that states nothing leaves nothing.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import dataclasses
|
|
import logging
|
|
import re
|
|
from decimal import Decimal
|
|
from typing import Callable, Dict, List, Optional
|
|
|
|
from rapidfuzz import fuzz
|
|
|
|
from app.electronics.collector import Collector, RunOptions
|
|
from app.electronics.db import repository as repo
|
|
from app.electronics.db.connection import connect
|
|
from app.electronics.extract import vijaysales_reviews
|
|
from app.electronics.extract.embedded_ratings import IGNORED_DOMAINS, embedded_rating
|
|
from app.electronics.extract.jsonld import extract_products
|
|
from app.electronics.reference import load_reference
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
def _verified_products(category: Optional[str]) -> List[dict]:
|
|
sql = ("SELECT v.product_id, v.brand_slug, v.category, p.model, p.model_norm, p.ram_gb, p.storage_gb "
|
|
"FROM elec.v_brand_catalog v JOIN elec.product p ON p.id = v.product_id")
|
|
params: tuple = ()
|
|
if category:
|
|
sql += " WHERE v.category = %s"
|
|
params = (category,)
|
|
with connect() as conn:
|
|
return list(conn.execute(sql + " ORDER BY v.product_id", params))
|
|
|
|
|
|
def _has_page_on(product_id: int, domain: str) -> bool:
|
|
with connect() as conn:
|
|
return conn.execute(
|
|
"SELECT 1 FROM elec.v_product_availability WHERE product_id = %s AND domain = %s "
|
|
"AND source_type IN ('scraped_page','brand_official') LIMIT 1", (product_id, domain),
|
|
).fetchone() is not None
|
|
|
|
|
|
def _gb(value) -> str:
|
|
return f"{format(Decimal(value).normalize(), 'f')}GB" if value is not None else ""
|
|
|
|
|
|
# Words that make a different model, not a different colour of the same one.
|
|
_MODEL_QUALIFIERS = frozenset({"ultra", "plus", "pro", "max", "fe", "lite", "edge", "mini", "neo",
|
|
"prime", "flip", "fold", "slim", "+"})
|
|
|
|
|
|
def _same_model(page_model: str, product_model: str) -> bool:
|
|
"""Same model line: every product word present, identical model-number
|
|
words, and nothing extra but non-model words (e.g. a colour the parser
|
|
left in: "galaxy a56 olive"). Fuzzy scores are no use here - they rate
|
|
"galaxy a57" vs "galaxy a56" at ~90."""
|
|
page, mine = set(page_model.split()), set(product_model.split())
|
|
numbered = lambda words: {w for w in words if any(c.isdigit() for c in w)} # noqa: E731
|
|
extra = page - mine
|
|
return mine <= page and numbered(page) == numbered(mine) and not (extra & _MODEL_QUALIFIERS)
|
|
|
|
|
|
def _same_variant(parsed, product: dict) -> bool:
|
|
if not parsed.model_norm or not _same_model(parsed.model_norm, product["model_norm"] or ""):
|
|
return False
|
|
for attr in ("ram_gb", "storage_gb"):
|
|
mine, theirs = product[attr], getattr(parsed, attr)
|
|
if mine is not None and theirs is not None and Decimal(mine) != Decimal(theirs):
|
|
return False
|
|
return True
|
|
|
|
|
|
def discover_brand_pages(category: Optional[str], budget: int, stats: Dict[str, int],
|
|
progress: Callable[[str], None]) -> None:
|
|
ref = load_reference()
|
|
by_category: Dict[str, List[dict]] = {}
|
|
for p in _verified_products(category):
|
|
brand = ref.brands.get(p["brand_slug"])
|
|
if brand and brand.reviews_site:
|
|
by_category.setdefault(p["category"], []).append(p)
|
|
for cat, products in by_category.items():
|
|
brands = sorted({p["brand_slug"] for p in products})
|
|
col = Collector(RunOptions(category=cat, brands=brands, search_budget=budget, use_llm=False),
|
|
progress=progress)
|
|
try:
|
|
for p in products:
|
|
brand = ref.brands[p["brand_slug"]]
|
|
domain = brand.reviews_site.split("/")[0]
|
|
if _has_page_on(p["product_id"], domain):
|
|
stats["brand_page_already_known"] = stats.get("brand_page_already_known", 0) + 1
|
|
continue
|
|
model = p["model"] or p["model_norm"]
|
|
queries = [f"site:{brand.reviews_site} {brand.name} {model} {_gb(p['ram_gb'])} {_gb(p['storage_gb'])}",
|
|
f"site:{brand.reviews_site} {model} 5G {_gb(p['storage_gb'])} buy"]
|
|
hits, query = [], ""
|
|
for q in (re.sub(r"\s+", " ", q).strip() for q in queries):
|
|
hits += [(h, q) for h in (col.engine.text(q, max_results=10) or [])]
|
|
seen = set()
|
|
for hit, query in hits:
|
|
if hit.url in seen:
|
|
continue
|
|
seen.add(hit.url)
|
|
# Brand-store titles often omit the brand ("Galaxy A56 5G 8GB/256GB ...",
|
|
# sometimes behind a "Business |" prefix). On the brand's own store
|
|
# the brand is not in doubt, so state it for the title parser.
|
|
title = re.sub(r"^\s*Business\s*\|\s*", "", hit.title or "")
|
|
if not title.lower().startswith(brand.name.lower()):
|
|
title = f"{brand.name} {title}"
|
|
hit = dataclasses.replace(hit, title=title)
|
|
if f"{brand.reviews_site}/" not in hit.url:
|
|
continue
|
|
if brand.reviews_product_url and not re.search(brand.reviews_product_url, hit.url.split("?")[0]):
|
|
continue # a family/marketing page, not one variant's product page
|
|
accepted = col._accept_hit(hit, brand)
|
|
if not accepted:
|
|
continue
|
|
site, parsed = accepted
|
|
if not _same_variant(parsed, p):
|
|
continue
|
|
res = col.client.get(hit.url)
|
|
if not res.ok:
|
|
continue
|
|
listing = col.listing_from_page(hit, site, parsed, query, res.text, res.final_url)
|
|
# Only a page with schema.org product data for this exact variant.
|
|
if listing is None or not listing.parser.startswith("jsonld") or not _same_variant(listing, p):
|
|
continue
|
|
col.store(listing)
|
|
if _has_page_on(p["product_id"], domain): # actually matched to this product
|
|
stats["brand_pages_added"] = stats.get("brand_pages_added", 0) + 1
|
|
progress(f" {brand.name}: {listing.title} -> rating {listing.rating}, "
|
|
f"{len(listing.reviews)} reviews")
|
|
break
|
|
finally:
|
|
col.client.close()
|
|
repo.refresh_verification()
|
|
|
|
|
|
def reread_pages(category: Optional[str], limit: int, stats: Dict[str, int],
|
|
progress: Callable[[str], None]) -> None:
|
|
from app.electronics.net.polite_client import PoliteClient
|
|
|
|
rows = repo.listings_for_review_backfill(category)[:limit]
|
|
with PoliteClient() as client:
|
|
for row in rows:
|
|
domain = row["domain"]
|
|
if domain in IGNORED_DOMAINS:
|
|
continue
|
|
stats["pages"] = stats.get("pages", 0) + 1
|
|
rating = count = breakdown = None
|
|
reviews: List[dict] = []
|
|
res = client.get(row["source_url"])
|
|
if res.ok:
|
|
products = extract_products(res.text)
|
|
match = next((x for x in products if x.get("sku") and x["sku"] == row["source_sku"]), None)
|
|
if match is None:
|
|
title = (row["title"] or "").lower()
|
|
scored = [(fuzz.token_set_ratio(x["name"].lower(), title), x) for x in products]
|
|
scored = [s for s in scored if s[0] >= 85]
|
|
match = max(scored, key=lambda s: s[0])[1] if scored else None
|
|
if match is not None:
|
|
rating, count, reviews = match.get("rating"), match.get("review_count"), match.get("reviews") or []
|
|
# The retailer's own embedded data: the rating when JSON-LD has none,
|
|
# and the star breakdown, which only it publishes.
|
|
emb = embedded_rating(domain, res.text, res.final_url or row["source_url"])
|
|
if emb:
|
|
if rating is None:
|
|
rating, count = emb["rating"], emb["review_count"]
|
|
breakdown = emb["breakdown"]
|
|
if domain == "vijaysales.com":
|
|
gql = vijaysales_reviews.graphql_url(row["source_url"])
|
|
vres = client.get(gql, accept_non_html=True) if gql else None
|
|
vs = vijaysales_reviews.parse(vres.text) if vres is not None and vres.ok else None
|
|
if vs:
|
|
rating, count = rating or vs["rating"], count or vs["review_count"]
|
|
reviews = reviews or vs["reviews"]
|
|
if rating is not None and Decimal(0) < Decimal(rating) <= Decimal(5):
|
|
repo.update_listing_rating(row["listing_id"], rating, count, breakdown)
|
|
stats["rated"] = stats.get("rated", 0) + 1
|
|
if reviews:
|
|
stats["reviews_stored"] = stats.get("reviews_stored", 0) + repo.save_reviews(row["listing_id"], reviews)
|
|
progress(f" {domain:22} rating={rating} count={count} breakdown={'yes' if breakdown else 'no'} "
|
|
f"reviews={len(reviews)}")
|
|
|
|
|
|
def refresh_reviews(category: Optional[str] = None, limit: int = 500, budget: int = 60,
|
|
discover: bool = True, progress: Optional[Callable[[str], None]] = None) -> Dict[str, int]:
|
|
progress = progress or (lambda m: logger.info(m))
|
|
stats: Dict[str, int] = {}
|
|
run_id = repo.start_run("reviews", {"category": category, "limit": limit, "discover": discover})
|
|
status, error = "done", None
|
|
try:
|
|
if discover:
|
|
progress("Finding brand-store pages with reviews ...")
|
|
discover_brand_pages(category, budget, stats, progress)
|
|
progress("Re-reading ratings and reviews from product pages ...")
|
|
reread_pages(category, limit, stats, progress)
|
|
except Exception as exc:
|
|
status, error = "failed", repr(exc)
|
|
raise
|
|
finally:
|
|
repo.finish_run(run_id, status, stats, error)
|
|
return stats
|