Customer Rating and review changes

This commit is contained in:
sriram
2026-10-05 11:21:39 +05:30
parent c7e4d59188
commit abe7ad8450
15 changed files with 697 additions and 95 deletions

View File

@@ -143,58 +143,22 @@ def rematch(category: str = typer.Option(..., help="mobiles | laptops"),
@app.command()
def reviews(category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)"),
limit: int = typer.Option(200, help="Max product pages to re-read"),
limit: int = typer.Option(500, help="Max product pages to re-read"),
budget: int = typer.Option(60, help="Max web searches for finding brand-store pages"),
no_discover: bool = typer.Option(False, "--no-discover", help="Skip finding brand-store pages"),
verbose: bool = False) -> None:
"""Re-read ratings and customer reviews from the product pages already on file.
"""Refresh real customer ratings and reviews from free, robots-allowed sources.
Only pages the collector itself reads (scraped / brand official listings)
are fetched, politely (robots.txt, per-site pacing, circuit breaker). A
rating or review is stored only when the page's own schema.org data states
it; nothing is generated.
Finds each verified product's page on brand stores that publish reviews
(brands.yaml `reviews_site`), then re-reads every readable product page for
its rating, star breakdown and reviews, plus Vijay Sales' public review
feed. A rating or review is stored only when its source states it.
"""
_setup_logging(verbose)
from rapidfuzz import fuzz
from app.electronics.review_refresh import refresh_reviews
from app.electronics.db import repository as repo
from app.electronics.extract.jsonld import extract_products
from app.electronics.net.polite_client import PoliteClient
rows = repo.listings_for_review_backfill(category)[:limit]
run_id = repo.start_run("reviews", {"category": category, "pages": len(rows)})
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
r.outcome, r.robots_allowed))
stats = {"pages": 0, "pages_ok": 0, "rated": 0, "reviews_stored": 0, "no_matching_product": 0}
status, error = "done", None
try:
for row in rows:
stats["pages"] += 1
res = client.get(row["source_url"])
if not res.ok:
continue
stats["pages_ok"] += 1
products = extract_products(res.text)
# The same product the listing was stored from: its SKU, else its name.
match = next((p for p in products if p.get("sku") and p["sku"] == row["source_sku"]), None)
if match is None:
title = (row["title"] or "").lower()
scored = [(fuzz.token_set_ratio(p["name"].lower(), title), p) for p in products]
scored = [sp for sp in scored if sp[0] >= 85]
match = max(scored, key=lambda sp: sp[0])[1] if scored else None
if match is None:
stats["no_matching_product"] += 1
continue
if match.get("rating") is not None and Decimal(0) < match["rating"] <= Decimal(5):
repo.update_listing_rating(row["listing_id"], match["rating"], match.get("review_count"))
stats["rated"] += 1
if match.get("reviews"):
stats["reviews_stored"] += repo.save_reviews(row["listing_id"], match["reviews"])
typer.echo(f" {row['domain']:22} rating={match.get('rating')} reviews={len(match.get('reviews') or [])}")
except Exception as exc: # noqa: BLE001
status, error = "failed", repr(exc)
raise
finally:
client.close()
repo.finish_run(run_id, status, stats, error)
stats = refresh_reviews(category=category, limit=limit, budget=budget, discover=not no_discover,
progress=typer.echo)
typer.echo(json.dumps(stats, indent=2))