Customer Rating and review changes
This commit is contained in:
@@ -143,58 +143,22 @@ def rematch(category: str = typer.Option(..., help="mobiles | laptops"),
|
||||
|
||||
@app.command()
|
||||
def reviews(category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)"),
|
||||
limit: int = typer.Option(200, help="Max product pages to re-read"),
|
||||
limit: int = typer.Option(500, help="Max product pages to re-read"),
|
||||
budget: int = typer.Option(60, help="Max web searches for finding brand-store pages"),
|
||||
no_discover: bool = typer.Option(False, "--no-discover", help="Skip finding brand-store pages"),
|
||||
verbose: bool = False) -> None:
|
||||
"""Re-read ratings and customer reviews from the product pages already on file.
|
||||
"""Refresh real customer ratings and reviews from free, robots-allowed sources.
|
||||
|
||||
Only pages the collector itself reads (scraped / brand official listings)
|
||||
are fetched, politely (robots.txt, per-site pacing, circuit breaker). A
|
||||
rating or review is stored only when the page's own schema.org data states
|
||||
it; nothing is generated.
|
||||
Finds each verified product's page on brand stores that publish reviews
|
||||
(brands.yaml `reviews_site`), then re-reads every readable product page for
|
||||
its rating, star breakdown and reviews, plus Vijay Sales' public review
|
||||
feed. A rating or review is stored only when its source states it.
|
||||
"""
|
||||
_setup_logging(verbose)
|
||||
from rapidfuzz import fuzz
|
||||
from app.electronics.review_refresh import refresh_reviews
|
||||
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.extract.jsonld import extract_products
|
||||
from app.electronics.net.polite_client import PoliteClient
|
||||
|
||||
rows = repo.listings_for_review_backfill(category)[:limit]
|
||||
run_id = repo.start_run("reviews", {"category": category, "pages": len(rows)})
|
||||
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
|
||||
r.outcome, r.robots_allowed))
|
||||
stats = {"pages": 0, "pages_ok": 0, "rated": 0, "reviews_stored": 0, "no_matching_product": 0}
|
||||
status, error = "done", None
|
||||
try:
|
||||
for row in rows:
|
||||
stats["pages"] += 1
|
||||
res = client.get(row["source_url"])
|
||||
if not res.ok:
|
||||
continue
|
||||
stats["pages_ok"] += 1
|
||||
products = extract_products(res.text)
|
||||
# The same product the listing was stored from: its SKU, else its name.
|
||||
match = next((p for p in products if p.get("sku") and p["sku"] == row["source_sku"]), None)
|
||||
if match is None:
|
||||
title = (row["title"] or "").lower()
|
||||
scored = [(fuzz.token_set_ratio(p["name"].lower(), title), p) for p in products]
|
||||
scored = [sp for sp in scored if sp[0] >= 85]
|
||||
match = max(scored, key=lambda sp: sp[0])[1] if scored else None
|
||||
if match is None:
|
||||
stats["no_matching_product"] += 1
|
||||
continue
|
||||
if match.get("rating") is not None and Decimal(0) < match["rating"] <= Decimal(5):
|
||||
repo.update_listing_rating(row["listing_id"], match["rating"], match.get("review_count"))
|
||||
stats["rated"] += 1
|
||||
if match.get("reviews"):
|
||||
stats["reviews_stored"] += repo.save_reviews(row["listing_id"], match["reviews"])
|
||||
typer.echo(f" {row['domain']:22} rating={match.get('rating')} reviews={len(match.get('reviews') or [])}")
|
||||
except Exception as exc: # noqa: BLE001
|
||||
status, error = "failed", repr(exc)
|
||||
raise
|
||||
finally:
|
||||
client.close()
|
||||
repo.finish_run(run_id, status, stats, error)
|
||||
stats = refresh_reviews(category=category, limit=limit, budget=budget, discover=not no_discover,
|
||||
progress=typer.echo)
|
||||
typer.echo(json.dumps(stats, indent=2))
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user