Customer Rating and review changes

This commit is contained in:
sriram
2026-10-05 11:21:39 +05:30
parent c7e4d59188
commit abe7ad8450
15 changed files with 697 additions and 95 deletions

View File

@@ -36,6 +36,7 @@ from urllib.parse import urlparse
from rapidfuzz import fuzz
from app.electronics.db import repository as repo
from app.electronics.extract.embedded_ratings import IGNORED_DOMAINS, embedded_rating
from app.electronics.extract.html_fallback import extract_page, spec_tables, visible_text
from app.electronics.extract.jsonld import extract_products
from app.electronics.extract.serp_parser import clean_result_title, read_price, read_rating, read_stock
@@ -371,6 +372,13 @@ class Collector:
html: str, final_url: str) -> Optional[Listing]:
products = extract_products(html)
page = extract_page(html)
if site.kind == "brand_official":
# A brand's own store names products without the brand ("Galaxy A56 5G
# (8 GB Memory)"); on its own site the brand is not in doubt.
brand_name = self.ref.brands[parsed_hit.brand.brand_slug].name
for p in products:
if not p["name"].lower().startswith(brand_name.lower()):
p["name"] = f"{brand_name} {p['name']}"
name = None
product = None
for p in products:
@@ -428,6 +436,17 @@ class Collector:
listing.review_count = product.get("review_count")
listing.reviews = list(product.get("reviews") or [])
listing.colour = listing.colour or product.get("color")
# Retailers that embed their rating outside JSON-LD (page's own product
# only): the rating when JSON-LD has none, and the star breakdown.
embedded = embedded_rating(site.domain, html, final_url or hit.url)
if embedded:
if listing.rating is None:
listing.rating, listing.review_count = embedded["rating"], embedded["review_count"]
listing.rating_breakdown = embedded["breakdown"]
if site.domain in IGNORED_DOMAINS:
# Known placeholder rating markup: nothing rating-shaped is kept.
listing.rating = listing.review_count = listing.rating_breakdown = None
listing.reviews = []
raw_specs = dict((product or {}).get("properties") or {})
raw_specs.update({k: v for k, v in spec_tables(BeautifulSoup(html, "lxml")).items() if k not in raw_specs})
listing.specs_raw = dict(list(raw_specs.items())[:150])