diff --git a/backend/app/api/routers/elec.py b/backend/app/api/routers/elec.py index 5cc229d..cb14ca4 100644 --- a/backend/app/api/routers/elec.py +++ b/backend/app/api/routers/elec.py @@ -141,23 +141,50 @@ def product(product_id: int) -> dict: def _overall_rating(sources: List[dict]) -> Optional[dict]: """The product's rating across the platforms that state one: the mean weighted by each platform's rating count (a platform that states no count - weighs as 1). None when no platform states a rating - never a guess.""" + weighs as 1). None when no platform states a rating - never a guess. + + One reading per platform: a site with a page per colour repeats the same + model rating on each, and adding those up would multiply its count.""" if not sources: return None weight = lambda s: max(int(s["review_count"] or 0), 1) # noqa: E731 + per_site: Dict[str, dict] = {} + for s in sources: + if s["site"] not in per_site or weight(s) > weight(per_site[s["site"]]): + per_site[s["site"]] = s + sources = list(per_site.values()) total = sum(weight(s) for s in sources) value = sum(Decimal(s["rating"]) * weight(s) for s in sources) / total counts = [s["review_count"] for s in sources if s["review_count"]] return { "value": round(float(value), 1), "count": sum(counts) if counts else None, + "breakdown": _breakdown(sources), "sources": [ - {"site": s["site"], "rating": float(s["rating"]), "review_count": s["review_count"], "source_url": s["source_url"]} + {"site": s["site"], "rating": float(s["rating"]), "review_count": s["review_count"], + "breakdown": s.get("rating_breakdown"), "source_url": s["source_url"]} for s in sources ], } +def _breakdown(sources: List[dict]) -> Optional[List[dict]]: + """5→1 star counts summed over the platforms that PUBLISH a breakdown. + None when none does - never estimated from the few review texts held.""" + totals = {str(n): 0 for n in range(1, 6)} + stated = False + for s in sources: + for star, n in (s.get("rating_breakdown") or {}).items(): + if star in totals and isinstance(n, int) and n > 0: + totals[star] += n + stated = True + if not stated: + return None + total = sum(totals.values()) + return [{"stars": int(k), "count": totals[k], "percent": round(100 * totals[k] / total)} + for k in ("5", "4", "3", "2", "1")] + + @router.get("/products/{product_id}/price-history") def price_history(product_id: int) -> List[dict]: with connect() as conn: diff --git a/backend/app/electronics/cli.py b/backend/app/electronics/cli.py index 63a1090..c978a8e 100644 --- a/backend/app/electronics/cli.py +++ b/backend/app/electronics/cli.py @@ -143,58 +143,22 @@ def rematch(category: str = typer.Option(..., help="mobiles | laptops"), @app.command() def reviews(category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)"), - limit: int = typer.Option(200, help="Max product pages to re-read"), + limit: int = typer.Option(500, help="Max product pages to re-read"), + budget: int = typer.Option(60, help="Max web searches for finding brand-store pages"), + no_discover: bool = typer.Option(False, "--no-discover", help="Skip finding brand-store pages"), verbose: bool = False) -> None: - """Re-read ratings and customer reviews from the product pages already on file. + """Refresh real customer ratings and reviews from free, robots-allowed sources. - Only pages the collector itself reads (scraped / brand official listings) - are fetched, politely (robots.txt, per-site pacing, circuit breaker). A - rating or review is stored only when the page's own schema.org data states - it; nothing is generated. + Finds each verified product's page on brand stores that publish reviews + (brands.yaml `reviews_site`), then re-reads every readable product page for + its rating, star breakdown and reviews, plus Vijay Sales' public review + feed. A rating or review is stored only when its source states it. """ _setup_logging(verbose) - from rapidfuzz import fuzz + from app.electronics.review_refresh import refresh_reviews - from app.electronics.db import repository as repo - from app.electronics.extract.jsonld import extract_products - from app.electronics.net.polite_client import PoliteClient - - rows = repo.listings_for_review_backfill(category)[:limit] - run_id = repo.start_run("reviews", {"category": category, "pages": len(rows)}) - client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes, - r.outcome, r.robots_allowed)) - stats = {"pages": 0, "pages_ok": 0, "rated": 0, "reviews_stored": 0, "no_matching_product": 0} - status, error = "done", None - try: - for row in rows: - stats["pages"] += 1 - res = client.get(row["source_url"]) - if not res.ok: - continue - stats["pages_ok"] += 1 - products = extract_products(res.text) - # The same product the listing was stored from: its SKU, else its name. - match = next((p for p in products if p.get("sku") and p["sku"] == row["source_sku"]), None) - if match is None: - title = (row["title"] or "").lower() - scored = [(fuzz.token_set_ratio(p["name"].lower(), title), p) for p in products] - scored = [sp for sp in scored if sp[0] >= 85] - match = max(scored, key=lambda sp: sp[0])[1] if scored else None - if match is None: - stats["no_matching_product"] += 1 - continue - if match.get("rating") is not None and Decimal(0) < match["rating"] <= Decimal(5): - repo.update_listing_rating(row["listing_id"], match["rating"], match.get("review_count")) - stats["rated"] += 1 - if match.get("reviews"): - stats["reviews_stored"] += repo.save_reviews(row["listing_id"], match["reviews"]) - typer.echo(f" {row['domain']:22} rating={match.get('rating')} reviews={len(match.get('reviews') or [])}") - except Exception as exc: # noqa: BLE001 - status, error = "failed", repr(exc) - raise - finally: - client.close() - repo.finish_run(run_id, status, stats, error) + stats = refresh_reviews(category=category, limit=limit, budget=budget, discover=not no_discover, + progress=typer.echo) typer.echo(json.dumps(stats, indent=2)) diff --git a/backend/app/electronics/collector.py b/backend/app/electronics/collector.py index 209d860..2f38655 100644 --- a/backend/app/electronics/collector.py +++ b/backend/app/electronics/collector.py @@ -36,6 +36,7 @@ from urllib.parse import urlparse from rapidfuzz import fuzz from app.electronics.db import repository as repo +from app.electronics.extract.embedded_ratings import IGNORED_DOMAINS, embedded_rating from app.electronics.extract.html_fallback import extract_page, spec_tables, visible_text from app.electronics.extract.jsonld import extract_products from app.electronics.extract.serp_parser import clean_result_title, read_price, read_rating, read_stock @@ -371,6 +372,13 @@ class Collector: html: str, final_url: str) -> Optional[Listing]: products = extract_products(html) page = extract_page(html) + if site.kind == "brand_official": + # A brand's own store names products without the brand ("Galaxy A56 5G + # (8 GB Memory)"); on its own site the brand is not in doubt. + brand_name = self.ref.brands[parsed_hit.brand.brand_slug].name + for p in products: + if not p["name"].lower().startswith(brand_name.lower()): + p["name"] = f"{brand_name} {p['name']}" name = None product = None for p in products: @@ -428,6 +436,17 @@ class Collector: listing.review_count = product.get("review_count") listing.reviews = list(product.get("reviews") or []) listing.colour = listing.colour or product.get("color") + # Retailers that embed their rating outside JSON-LD (page's own product + # only): the rating when JSON-LD has none, and the star breakdown. + embedded = embedded_rating(site.domain, html, final_url or hit.url) + if embedded: + if listing.rating is None: + listing.rating, listing.review_count = embedded["rating"], embedded["review_count"] + listing.rating_breakdown = embedded["breakdown"] + if site.domain in IGNORED_DOMAINS: + # Known placeholder rating markup: nothing rating-shaped is kept. + listing.rating = listing.review_count = listing.rating_breakdown = None + listing.reviews = [] raw_specs = dict((product or {}).get("properties") or {}) raw_specs.update({k: v for k, v in spec_tables(BeautifulSoup(html, "lxml")).items() if k not in raw_specs}) listing.specs_raw = dict(list(raw_specs.items())[:150]) diff --git a/backend/app/electronics/db/migrations/0007_rating_breakdown.sql b/backend/app/electronics/db/migrations/0007_rating_breakdown.sql new file mode 100644 index 0000000..4fd5ba8 --- /dev/null +++ b/backend/app/electronics/db/migrations/0007_rating_breakdown.sql @@ -0,0 +1,4 @@ +-- Star breakdown of a listing's ratings, exactly as the source states it: +-- {"5": 612, "4": 300, "3": 80, "2": 20, "1": 12}. NULL when the source does +-- not publish one - never derived from the few review texts we hold. +ALTER TABLE elec.source_listing ADD COLUMN rating_breakdown JSONB; diff --git a/backend/app/electronics/db/repository.py b/backend/app/electronics/db/repository.py index 9c09c55..85dca58 100644 --- a/backend/app/electronics/db/repository.py +++ b/backend/app/electronics/db/repository.py @@ -199,7 +199,8 @@ def upsert_listing(listing: Listing, ids: Dict[str, Dict[str, int]], run_id: Opt model_number=listing.model_number, ram_gb=listing.ram_gb, storage_gb=listing.storage_gb, colour=listing.colour, price=listing.price, mrp=listing.mrp, availability=listing.availability, in_stock=listing.in_stock, pincode=listing.pincode, pincode_applied=listing.pincode_applied, - rating=listing.rating, review_count=listing.review_count, gtin=listing.gtin, + rating=listing.rating, review_count=listing.review_count, + rating_breakdown=_json(listing.rating_breakdown) if listing.rating_breakdown else None, gtin=listing.gtin, image_urls=listing.image_urls[:12], specs_raw=_json(listing.specs_raw), specs=_json(listing.specs), evidence_text=listing.evidence_text[:4000], search_query=listing.search_query, confidence=round(listing.confidence, 2), parser=listing.parser, content_hash=listing.content_hash, @@ -210,13 +211,13 @@ def upsert_listing(listing: Listing, ids: Dict[str, Dict[str, int]], run_id: Opt INSERT INTO elec.source_listing ( site_id, source_sku, source_url, source_type, brand_id, category_id, family, title, model, model_number, ram_gb, storage_gb, colour, price, mrp, availability, in_stock, pincode, - pincode_applied, rating, review_count, gtin, image_urls, specs_raw, specs, evidence_text, + pincode_applied, rating, review_count, rating_breakdown, gtin, image_urls, specs_raw, specs, evidence_text, search_query, confidence, parser, content_hash, crawl_run_id) VALUES ( %(site_id)s, %(source_sku)s, %(source_url)s, %(source_type)s, %(brand_id)s, %(category_id)s, %(family)s, %(title)s, %(model)s, %(model_number)s, %(ram_gb)s, %(storage_gb)s, %(colour)s, %(price)s, %(mrp)s, %(availability)s, %(in_stock)s, %(pincode)s, %(pincode_applied)s, - %(rating)s, %(review_count)s, %(gtin)s, %(image_urls)s, %(specs_raw)s, %(specs)s, + %(rating)s, %(review_count)s, %(rating_breakdown)s, %(gtin)s, %(image_urls)s, %(specs_raw)s, %(specs)s, %(evidence_text)s, %(search_query)s, %(confidence)s, %(parser)s, %(content_hash)s, %(crawl_run_id)s) ON CONFLICT (site_id, source_sku) DO UPDATE SET @@ -227,7 +228,7 @@ def upsert_listing(listing: Listing, ids: Dict[str, Dict[str, int]], run_id: Opt price = EXCLUDED.price, mrp = EXCLUDED.mrp, availability = EXCLUDED.availability, in_stock = EXCLUDED.in_stock, pincode = EXCLUDED.pincode, pincode_applied = EXCLUDED.pincode_applied, rating = EXCLUDED.rating, - review_count = EXCLUDED.review_count, gtin = EXCLUDED.gtin, image_urls = EXCLUDED.image_urls, + review_count = EXCLUDED.review_count, rating_breakdown = EXCLUDED.rating_breakdown, gtin = EXCLUDED.gtin, image_urls = EXCLUDED.image_urls, specs_raw = EXCLUDED.specs_raw, specs = EXCLUDED.specs, evidence_text = EXCLUDED.evidence_text, search_query = EXCLUDED.search_query, confidence = EXCLUDED.confidence, parser = EXCLUDED.parser, content_hash = EXCLUDED.content_hash, crawl_run_id = EXCLUDED.crawl_run_id, last_seen_at = now() @@ -276,12 +277,13 @@ def save_reviews(listing_id: int, reviews: List[Dict[str, Any]]) -> int: return added -def update_listing_rating(listing_id: int, rating: Optional[Decimal], review_count: Optional[int]) -> None: +def update_listing_rating(listing_id: int, rating: Optional[Decimal], review_count: Optional[int], + breakdown: Optional[Dict[str, int]] = None) -> None: """Refresh only the rating fields of a listing (used by the review backfill).""" with transaction() as conn: conn.execute( - "UPDATE elec.source_listing SET rating = %s, review_count = %s WHERE id = %s", - (rating, review_count, listing_id), + "UPDATE elec.source_listing SET rating = %s, review_count = %s, rating_breakdown = %s WHERE id = %s", + (rating, review_count, _json(breakdown) if breakdown else None, listing_id), ) @@ -289,7 +291,7 @@ def product_rating_and_reviews(conn, product_id: int) -> Dict[str, Any]: """Per-platform ratings and all stored reviews for a verified product's approved listings, each with the page it was read from.""" sources = conn.execute( - "SELECT a.site, a.source_url, l.rating, l.review_count FROM elec.v_product_availability a " + "SELECT a.site, a.source_url, l.rating, l.review_count, l.rating_breakdown FROM elec.v_product_availability a " "JOIN elec.source_listing l ON l.id = a.listing_id " "WHERE a.product_id = %s AND l.rating > 0 ORDER BY l.review_count DESC NULLS LAST, a.site", (product_id,), diff --git a/backend/app/electronics/extract/embedded_ratings.py b/backend/app/electronics/extract/embedded_ratings.py new file mode 100644 index 0000000..782eea7 --- /dev/null +++ b/backend/app/electronics/extract/embedded_ratings.py @@ -0,0 +1,95 @@ +"""Ratings a retailer embeds in its own product-page HTML (not in JSON-LD). + +Only the page we already fetched is read - no extra requests - and only the +record of the page's OWN product (never recommendations on the same page). +Zero values mean "no ratings yet" and are dropped. Review text is not available +here: those retailers load it from feeds their robots.txt does not allow. + +Deliberately absent: vasanthandco.in. Its product pages carry an identical +"3.3 average, 3 reviews" template block on every product - placeholder +content, not ratings - so nothing is ever read from that domain. +""" +from __future__ import annotations + +import json +import re +from decimal import Decimal, InvalidOperation +from typing import Any, Callable, Dict, Optional +from urllib.parse import urlparse + +from bs4 import BeautifulSoup + +# Domains whose rating-looking markup is known to be fake/placeholder. +IGNORED_DOMAINS = frozenset({"vasanthandco.in"}) + + +def _dec(value: Any) -> Optional[Decimal]: + try: + d = Decimal(str(value)) + except (InvalidOperation, TypeError, ValueError): + return None + return d if Decimal(0) < d <= Decimal(5) else None + + +def _int(value: Any) -> Optional[int]: + try: + n = int(value) + except (TypeError, ValueError): + return None + return n if n > 0 else None + + +def _reading(rating: Any, count: Any, breakdown: Optional[Dict[str, int]] = None) -> Optional[Dict[str, Any]]: + r = _dec(rating) + if r is None: + return None + clean = None + if breakdown: + clean = {str(k): int(v) for k, v in breakdown.items() + if str(k) in {"1", "2", "3", "4", "5"} and _int(v)} + return {"rating": r.quantize(Decimal("0.01")), "review_count": _int(count), "breakdown": clean or None} + + +def _reliance(html: str, url: str) -> Optional[Dict[str, Any]]: + """window.__INITIAL_STATE__.productDetailsPage.product._custom_json._app""" + marker = "window.__INITIAL_STATE__=" + i = html.find(marker) + if i < 0: + return None + try: + state, _ = json.JSONDecoder().raw_decode(html, i + len(marker)) + app = state["productDetailsPage"]["product"]["_custom_json"]["_app"] + except (ValueError, KeyError, TypeError): + return None + if not isinstance(app, dict): + return None + return _reading(app.get("averageRating"), app.get("ratingsCount"), app.get("ratingsCountDetails")) + + +def _poorvika(html: str, url: str) -> Optional[Dict[str, Any]]: + """__NEXT_DATA__ props.pageProps.additionalData, only when its slug is this URL's.""" + tag = BeautifulSoup(html, "lxml").find("script", id="__NEXT_DATA__") + if tag is None: + return None + try: + data = json.loads(tag.string or tag.get_text())["props"]["pageProps"]["additionalData"] + except (ValueError, KeyError, TypeError): + return None + slug = str(data.get("slug") or "") + if not slug or slug not in urlparse(url).path: + return None + return _reading(data.get("rating"), data.get("ratingCount")) + + +_READERS: Dict[str, Callable[[str, str], Optional[Dict[str, Any]]]] = { + "reliancedigital.in": _reliance, + "poorvika.com": _poorvika, +} + + +def embedded_rating(domain: str, html: str, url: str) -> Optional[Dict[str, Any]]: + """{"rating": Decimal, "review_count": int|None, "breakdown": {"5": n, ...}|None} or None.""" + if domain in IGNORED_DOMAINS: + return None + reader = _READERS.get(domain) + return reader(html, url) if reader else None diff --git a/backend/app/electronics/extract/jsonld.py b/backend/app/electronics/extract/jsonld.py index 5c7e01f..31b7913 100644 --- a/backend/app/electronics/extract/jsonld.py +++ b/backend/app/electronics/extract/jsonld.py @@ -161,42 +161,67 @@ def _reviews(node: dict) -> List[Dict[str, Any]]: return out +def _rating_only_nodes(nodes: List[dict]) -> Dict[Optional[str], dict]: + """Nameless Product nodes that only carry ratings/reviews, keyed by @id. + + Review widgets (Bazaarvoice on samsung.com/in, for one) publish the + product's aggregateRating and reviews as a SEPARATE Product node with no + name, linked to the real product by the same @id. Without merging, those + real reviews would be skipped for lack of a name. + """ + out: Dict[Optional[str], dict] = {} + for node in nodes: + if _text(node.get("name")): + continue + if not (isinstance(node.get("aggregateRating"), dict) or node.get("review") or node.get("reviews")): + continue + out.setdefault(_text(node.get("@id")), node) + return out + + def extract_products(html: str) -> List[Dict[str, Any]]: """All schema.org Product nodes on the page, flattened to plain fields.""" products: List[Dict[str, Any]] = [] - for block in json_ld_blocks(html): - for node in _walk(block): - if not (_types(node) & _PRODUCT_TYPES): - continue - name = _text(node.get("name")) - if not name: - continue - offer = _offer(node.get("offers")) if node.get("offers") else {} - if not offer and isinstance(node.get("hasVariant"), list): - offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")]) - props = {} - for p in node.get("additionalProperty") or []: - if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""): - props[str(p["name"])] = str(p["value"]) - rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {} - products.append({ - "name": name, - "brand": _text(node.get("brand")), - "sku": _text(node.get("sku")) or _text(node.get("productID")), - "mpn": _text(node.get("mpn")), - "gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None), - "color": _text(node.get("color")), - "images": _images(node.get("image")), - "description": _text(node.get("description")), - "price": offer.get("price"), - "currency": offer.get("currency"), - "availability": offer.get("availability"), - "in_stock": offer.get("in_stock"), - # Sites publish 0 for "no ratings yet"; that is not a rating. - "rating": (_dec(rating.get("ratingValue")) or None), - "review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None, - "reviews": _reviews(node), - "properties": props, - "evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500], - }) + nodes = [n for block in json_ld_blocks(html) for n in _walk(block) if _types(n) & _PRODUCT_TYPES] + rating_nodes = _rating_only_nodes(nodes) + named = [n for n in nodes if _text(n.get("name"))] + for node in named: + # Attach a ratings-only node: same @id, or - when it has no @id - the + # page's only named product (there is then no doubt which it is about). + extra = rating_nodes.get(_text(node.get("@id"))) if node.get("@id") else None + if extra is None and len(named) == 1: + extra = rating_nodes.get(None) + if extra is not None: + node = {**node, + "aggregateRating": node.get("aggregateRating") or extra.get("aggregateRating"), + "review": node.get("review") or extra.get("review") or extra.get("reviews")} + name = _text(node.get("name")) + offer = _offer(node.get("offers")) if node.get("offers") else {} + if not offer and isinstance(node.get("hasVariant"), list): + offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")]) + props = {} + for p in node.get("additionalProperty") or []: + if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""): + props[str(p["name"])] = str(p["value"]) + rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {} + products.append({ + "name": name, + "brand": _text(node.get("brand")), + "sku": _text(node.get("sku")) or _text(node.get("productID")), + "mpn": _text(node.get("mpn")), + "gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None), + "color": _text(node.get("color")), + "images": _images(node.get("image")), + "description": _text(node.get("description")), + "price": offer.get("price"), + "currency": offer.get("currency"), + "availability": offer.get("availability"), + "in_stock": offer.get("in_stock"), + # Sites publish 0 for "no ratings yet"; that is not a rating. + "rating": (_dec(rating.get("ratingValue")) or None), + "review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None, + "reviews": _reviews(node), + "properties": props, + "evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500], + }) return products diff --git a/backend/app/electronics/extract/vijaysales_reviews.py b/backend/app/electronics/extract/vijaysales_reviews.py new file mode 100644 index 0000000..42b1e15 --- /dev/null +++ b/backend/app/electronics/extract/vijaysales_reviews.py @@ -0,0 +1,80 @@ +"""Vijay Sales ratings and reviews from its public GraphQL endpoint. + +vijaysales.com serves product data to its own pages from GET /api/graphql; +robots.txt does not disallow it, and no login or token is involved. One GET +per listing, made through PoliteClient (robots, pacing, circuit breaker). + +The product is looked up by url_key - the last path segment of the listing URL +we already hold - so the answer is about exactly that listing. +""" +from __future__ import annotations + +import json +from decimal import Decimal, InvalidOperation +from typing import Any, Dict, List, Optional +from urllib.parse import urlencode, urlparse + +from app.electronics.reviews import sentiment_for + +ENDPOINT = "https://www.vijaysales.com/api/graphql" + +_QUERY = ( + '{products(filter:{url_key:{eq:"%s"}}){items{sku rating_summary review_count ' + "reviews(pageSize:20){items{nickname summary text average_rating created_at}}}}}" +) + + +def url_key(listing_url: str) -> Optional[str]: + path = urlparse(listing_url).path.rstrip("/") + key = path.rsplit("/", 1)[-1] if path else "" + # url_keys are slugs; anything else would break out of the query string. + return key if key and all(c.isalnum() or c in "-_" for c in key) else None + + +def graphql_url(listing_url: str) -> Optional[str]: + key = url_key(listing_url) + return f"{ENDPOINT}?{urlencode({'query': _QUERY % key})}" if key else None + + +def _five(percent: Any) -> Optional[Decimal]: + """Magento ratings are percentages (80 = 4 stars).""" + try: + p = Decimal(str(percent)) + except (InvalidOperation, TypeError, ValueError): + return None + return (p / 20).quantize(Decimal("0.01")) if Decimal(0) < p <= Decimal(100) else None + + +def parse(body: str) -> Optional[Dict[str, Any]]: + """{"rating", "review_count", "reviews": [...]} for the best-reviewed matching item, or None.""" + try: + items = json.loads(body)["data"]["products"]["items"] or [] + except (ValueError, KeyError, TypeError): + return None + items = [i for i in items if isinstance(i, dict)] + if not items: + return None + item = max(items, key=lambda i: i.get("review_count") or 0) + rating = _five(item.get("rating_summary")) + if rating is None: + return None + reviews: List[Dict[str, Any]] = [] + for r in ((item.get("reviews") or {}).get("items") or []): + body_text = (r.get("text") or "").strip() + if not body_text: + continue + stars = _five(r.get("average_rating")) + title = (r.get("summary") or "").strip() + # When a customer leaves the title blank the site fills in the full + # product name; that is not something the reviewer wrote. + if len(title) > 60 or "/" in title: + title = "" + reviews.append({ + "author": (r.get("nickname") or "").strip() or None, + "rating": stars, + "title": title or None, + "body": body_text[:4000], + "review_date": r.get("created_at"), + "sentiment": sentiment_for(stars), + }) + return {"rating": rating, "review_count": int(item.get("review_count") or 0) or None, "reviews": reviews} diff --git a/backend/app/electronics/models.py b/backend/app/electronics/models.py index 527aa6d..dddc0f4 100644 --- a/backend/app/electronics/models.py +++ b/backend/app/electronics/models.py @@ -34,6 +34,8 @@ class Listing: pincode_applied: bool = False rating: Optional[Decimal] = None review_count: Optional[int] = None + # {"5": n, ..., "1": n} when the source publishes a star breakdown. + rating_breakdown: Optional[Dict[str, int]] = None # Customer reviews the page itself publishes (schema.org Review); stored # in elec.listing_review, not on the listing row. reviews: List[Dict[str, Any]] = field(default_factory=list) diff --git a/backend/app/electronics/reference/__init__.py b/backend/app/electronics/reference/__init__.py index 8d98d8f..94d59f6 100644 --- a/backend/app/electronics/reference/__init__.py +++ b/backend/app/electronics/reference/__init__.py @@ -24,6 +24,11 @@ class BrandRef: aliases: tuple sub_brands: tuple official: tuple + # Brand-store path whose product pages publish real customer reviews as + # schema.org data (e.g. "samsung.com/in"); None when no free source exists. + reviews_site: Optional[str] = None + # Regex a reviews_site URL must match to be a single-variant product page. + reviews_product_url: Optional[str] = None @dataclass(frozen=True) @@ -78,6 +83,8 @@ def load_reference() -> Reference: aliases=tuple(a.lower() for a in b.get("aliases", [])), sub_brands=tuple(s.lower() for s in b.get("sub_brands", [])), official=tuple(b.get("official", [])), + reviews_site=b.get("reviews_site"), + reviews_product_url=b.get("reviews_product_url"), ) categories = { c["slug"]: CategoryRef(c["slug"], c["name"], tuple(c.get("search_terms", [])), diff --git a/backend/app/electronics/reference/brands.yaml b/backend/app/electronics/reference/brands.yaml index a8a97f7..d47fc5c 100644 --- a/backend/app/electronics/reference/brands.yaml +++ b/backend/app/electronics/reference/brands.yaml @@ -7,11 +7,19 @@ # parent, and are kept as the product family. # official the brand's own Indian web domains. A product page on one of # these is the strongest evidence that a product exists. +# reviews_site optional brand-store path whose product pages publish real +# customer ratings + reviews as schema.org JSON-LD, robots-allowed +# (checked 2026-10-01). `cli reviews` finds each verified product's +# page there. Add a brand only after checking its pages. +# reviews_product_url regex for that store's single-variant product pages +# (family/marketing pages carry no product data or reviews). brands: - name: Samsung categories: [mobiles, laptops] aliases: [samsung] official: [samsung.com] + reviews_site: samsung.com/in + reviews_product_url: '-sm-[a-z0-9]+/?$' # .../galaxy-a56-5g-awesome-olive-256gb-sm-a566ezggins/ - name: Apple categories: [mobiles, laptops] aliases: [apple] diff --git a/backend/app/electronics/review_refresh.py b/backend/app/electronics/review_refresh.py new file mode 100644 index 0000000..35d4503 --- /dev/null +++ b/backend/app/electronics/review_refresh.py @@ -0,0 +1,215 @@ +"""Refresh customer ratings and reviews for the verified catalogue, from free +sources that publish them and allow reading them: + +1. Brand stores with real reviews in schema.org JSON-LD (brands.yaml + `reviews_site`, e.g. samsung.com/in): find each verified product's own page + by web search, accept it only for the same model AND variant, read it with + the normal collector page path, and store it as a brand_official listing. +2. Every page-read listing of a verified product is re-read for its rating, + star breakdown and reviews (JSON-LD, then the retailer's embedded data). +3. Vijay Sales listings additionally get their reviews from the site's own + public GraphQL endpoint. + +Everything goes through PoliteClient (robots.txt, per-site pacing, circuit +breaker). Nothing is generated; a source that states nothing leaves nothing. +""" +from __future__ import annotations + +import dataclasses +import logging +import re +from decimal import Decimal +from typing import Callable, Dict, List, Optional + +from rapidfuzz import fuzz + +from app.electronics.collector import Collector, RunOptions +from app.electronics.db import repository as repo +from app.electronics.db.connection import connect +from app.electronics.extract import vijaysales_reviews +from app.electronics.extract.embedded_ratings import IGNORED_DOMAINS, embedded_rating +from app.electronics.extract.jsonld import extract_products +from app.electronics.reference import load_reference + +logger = logging.getLogger(__name__) + + +def _verified_products(category: Optional[str]) -> List[dict]: + sql = ("SELECT v.product_id, v.brand_slug, v.category, p.model, p.model_norm, p.ram_gb, p.storage_gb " + "FROM elec.v_brand_catalog v JOIN elec.product p ON p.id = v.product_id") + params: tuple = () + if category: + sql += " WHERE v.category = %s" + params = (category,) + with connect() as conn: + return list(conn.execute(sql + " ORDER BY v.product_id", params)) + + +def _has_page_on(product_id: int, domain: str) -> bool: + with connect() as conn: + return conn.execute( + "SELECT 1 FROM elec.v_product_availability WHERE product_id = %s AND domain = %s " + "AND source_type IN ('scraped_page','brand_official') LIMIT 1", (product_id, domain), + ).fetchone() is not None + + +def _gb(value) -> str: + return f"{format(Decimal(value).normalize(), 'f')}GB" if value is not None else "" + + +# Words that make a different model, not a different colour of the same one. +_MODEL_QUALIFIERS = frozenset({"ultra", "plus", "pro", "max", "fe", "lite", "edge", "mini", "neo", + "prime", "flip", "fold", "slim", "+"}) + + +def _same_model(page_model: str, product_model: str) -> bool: + """Same model line: every product word present, identical model-number + words, and nothing extra but non-model words (e.g. a colour the parser + left in: "galaxy a56 olive"). Fuzzy scores are no use here - they rate + "galaxy a57" vs "galaxy a56" at ~90.""" + page, mine = set(page_model.split()), set(product_model.split()) + numbered = lambda words: {w for w in words if any(c.isdigit() for c in w)} # noqa: E731 + extra = page - mine + return mine <= page and numbered(page) == numbered(mine) and not (extra & _MODEL_QUALIFIERS) + + +def _same_variant(parsed, product: dict) -> bool: + if not parsed.model_norm or not _same_model(parsed.model_norm, product["model_norm"] or ""): + return False + for attr in ("ram_gb", "storage_gb"): + mine, theirs = product[attr], getattr(parsed, attr) + if mine is not None and theirs is not None and Decimal(mine) != Decimal(theirs): + return False + return True + + +def discover_brand_pages(category: Optional[str], budget: int, stats: Dict[str, int], + progress: Callable[[str], None]) -> None: + ref = load_reference() + by_category: Dict[str, List[dict]] = {} + for p in _verified_products(category): + brand = ref.brands.get(p["brand_slug"]) + if brand and brand.reviews_site: + by_category.setdefault(p["category"], []).append(p) + for cat, products in by_category.items(): + brands = sorted({p["brand_slug"] for p in products}) + col = Collector(RunOptions(category=cat, brands=brands, search_budget=budget, use_llm=False), + progress=progress) + try: + for p in products: + brand = ref.brands[p["brand_slug"]] + domain = brand.reviews_site.split("/")[0] + if _has_page_on(p["product_id"], domain): + stats["brand_page_already_known"] = stats.get("brand_page_already_known", 0) + 1 + continue + model = p["model"] or p["model_norm"] + queries = [f"site:{brand.reviews_site} {brand.name} {model} {_gb(p['ram_gb'])} {_gb(p['storage_gb'])}", + f"site:{brand.reviews_site} {model} 5G {_gb(p['storage_gb'])} buy"] + hits, query = [], "" + for q in (re.sub(r"\s+", " ", q).strip() for q in queries): + hits += [(h, q) for h in (col.engine.text(q, max_results=10) or [])] + seen = set() + for hit, query in hits: + if hit.url in seen: + continue + seen.add(hit.url) + # Brand-store titles often omit the brand ("Galaxy A56 5G 8GB/256GB ...", + # sometimes behind a "Business |" prefix). On the brand's own store + # the brand is not in doubt, so state it for the title parser. + title = re.sub(r"^\s*Business\s*\|\s*", "", hit.title or "") + if not title.lower().startswith(brand.name.lower()): + title = f"{brand.name} {title}" + hit = dataclasses.replace(hit, title=title) + if f"{brand.reviews_site}/" not in hit.url: + continue + if brand.reviews_product_url and not re.search(brand.reviews_product_url, hit.url.split("?")[0]): + continue # a family/marketing page, not one variant's product page + accepted = col._accept_hit(hit, brand) + if not accepted: + continue + site, parsed = accepted + if not _same_variant(parsed, p): + continue + res = col.client.get(hit.url) + if not res.ok: + continue + listing = col.listing_from_page(hit, site, parsed, query, res.text, res.final_url) + # Only a page with schema.org product data for this exact variant. + if listing is None or not listing.parser.startswith("jsonld") or not _same_variant(listing, p): + continue + col.store(listing) + if _has_page_on(p["product_id"], domain): # actually matched to this product + stats["brand_pages_added"] = stats.get("brand_pages_added", 0) + 1 + progress(f" {brand.name}: {listing.title} -> rating {listing.rating}, " + f"{len(listing.reviews)} reviews") + break + finally: + col.client.close() + repo.refresh_verification() + + +def reread_pages(category: Optional[str], limit: int, stats: Dict[str, int], + progress: Callable[[str], None]) -> None: + from app.electronics.net.polite_client import PoliteClient + + rows = repo.listings_for_review_backfill(category)[:limit] + with PoliteClient() as client: + for row in rows: + domain = row["domain"] + if domain in IGNORED_DOMAINS: + continue + stats["pages"] = stats.get("pages", 0) + 1 + rating = count = breakdown = None + reviews: List[dict] = [] + res = client.get(row["source_url"]) + if res.ok: + products = extract_products(res.text) + match = next((x for x in products if x.get("sku") and x["sku"] == row["source_sku"]), None) + if match is None: + title = (row["title"] or "").lower() + scored = [(fuzz.token_set_ratio(x["name"].lower(), title), x) for x in products] + scored = [s for s in scored if s[0] >= 85] + match = max(scored, key=lambda s: s[0])[1] if scored else None + if match is not None: + rating, count, reviews = match.get("rating"), match.get("review_count"), match.get("reviews") or [] + # The retailer's own embedded data: the rating when JSON-LD has none, + # and the star breakdown, which only it publishes. + emb = embedded_rating(domain, res.text, res.final_url or row["source_url"]) + if emb: + if rating is None: + rating, count = emb["rating"], emb["review_count"] + breakdown = emb["breakdown"] + if domain == "vijaysales.com": + gql = vijaysales_reviews.graphql_url(row["source_url"]) + vres = client.get(gql, accept_non_html=True) if gql else None + vs = vijaysales_reviews.parse(vres.text) if vres is not None and vres.ok else None + if vs: + rating, count = rating or vs["rating"], count or vs["review_count"] + reviews = reviews or vs["reviews"] + if rating is not None and Decimal(0) < Decimal(rating) <= Decimal(5): + repo.update_listing_rating(row["listing_id"], rating, count, breakdown) + stats["rated"] = stats.get("rated", 0) + 1 + if reviews: + stats["reviews_stored"] = stats.get("reviews_stored", 0) + repo.save_reviews(row["listing_id"], reviews) + progress(f" {domain:22} rating={rating} count={count} breakdown={'yes' if breakdown else 'no'} " + f"reviews={len(reviews)}") + + +def refresh_reviews(category: Optional[str] = None, limit: int = 500, budget: int = 60, + discover: bool = True, progress: Optional[Callable[[str], None]] = None) -> Dict[str, int]: + progress = progress or (lambda m: logger.info(m)) + stats: Dict[str, int] = {} + run_id = repo.start_run("reviews", {"category": category, "limit": limit, "discover": discover}) + status, error = "done", None + try: + if discover: + progress("Finding brand-store pages with reviews ...") + discover_brand_pages(category, budget, stats, progress) + progress("Re-reading ratings and reviews from product pages ...") + reread_pages(category, limit, stats, progress) + except Exception as exc: + status, error = "failed", repr(exc) + raise + finally: + repo.finish_run(run_id, status, stats, error) + return stats diff --git a/backend/app/infrastructure/settings.py b/backend/app/infrastructure/settings.py index 70f3d15..e453c65 100644 --- a/backend/app/infrastructure/settings.py +++ b/backend/app/infrastructure/settings.py @@ -199,7 +199,8 @@ USER_AGENT = os.getenv( REQUEST_TIMEOUT_SECONDS = int(os.getenv("REQUEST_TIMEOUT_SECONDS", "20")) ELEC_SITE_MIN_INTERVAL_SECONDS = float(os.getenv("ELEC_SITE_MIN_INTERVAL_SECONDS", "3")) ELEC_BREAKER_COOLDOWN_HOURS = float(os.getenv("ELEC_BREAKER_COOLDOWN_HOURS", "24")) -ELEC_MAX_PAGE_BYTES = int(os.getenv("ELEC_MAX_PAGE_BYTES", str(3 * 1024 * 1024))) +# Brand stores embed their reviews in the page: samsung.com/in pages run ~3.1 MB. +ELEC_MAX_PAGE_BYTES = int(os.getenv("ELEC_MAX_PAGE_BYTES", str(6 * 1024 * 1024))) ELEC_PROBE_TTL_DAYS = int(os.getenv("ELEC_PROBE_TTL_DAYS", "7")) MIN_IMAGE_BYTES = int(os.getenv("MIN_IMAGE_BYTES", "3000")) diff --git a/backend/app/mcp_server.py b/backend/app/mcp_server.py index a9a99d7..0204e1f 100644 --- a/backend/app/mcp_server.py +++ b/backend/app/mcp_server.py @@ -92,8 +92,9 @@ async def get_product(product_id: int) -> Dict[str, Any]: """Full details of one product by its product_id (from search_products). Returns per-platform offers (price, MRP, source URL, when seen), normalised specs, - image URLs, the overall rating with per-platform sources (null if none is published), - and up to 10 real customer reviews (often empty). + image URLs, the overall rating with per-platform sources and, when a platform publishes + one, a 5-to-1 star breakdown (rating is null if no platform publishes a rating), and up + to 10 real customer reviews with the site each came from (often empty). """ d = await _run(elec.product, int(product_id)) return { diff --git a/backend/tests/test_elec_reviews.py b/backend/tests/test_elec_reviews.py index 953623c..3292ad4 100644 --- a/backend/tests/test_elec_reviews.py +++ b/backend/tests/test_elec_reviews.py @@ -142,3 +142,155 @@ def test_api_serves_ratings_reviews_and_out_of_stock_price(db, client): assert {s["site"] for s in detail["rating"]["sources"]} == {"Amazon.in", "Croma"} assert [r["body"] for r in detail["reviews"]] == ["Battery lasts all day.", "Heats up."] assert all(r["source_url"].startswith("https://") for r in detail["reviews"]) + assert detail["rating"]["breakdown"] is None # no platform published one + + +# --------------------------------------------------------------------------- +# Free sources: brand-store JSON-LD, retailer-embedded ratings, Vijay Sales +# --------------------------------------------------------------------------- +def _ld(*nodes) -> str: + return "".join(f'' for n in nodes) + + +def _review(author, stars, body, date="2026-09-30T17:06:05.000+00:00"): + return {"@type": "Review", "author": {"@type": "Person", "name": author}, "reviewBody": body, + "datePublished": date, "reviewRating": {"@type": "Rating", "bestRating": 5, "ratingValue": stars}} + + +def test_nameless_review_node_is_merged_by_id(): + # The samsung.com/in shape: Bazaarvoice publishes ratings in a second, + # nameless Product node that shares the real product's @id. + pid = "https://www.samsung.com/in/smartphones/galaxy-a/galaxy-a56-5g-awesome-olive-256gb-sm-a566ezggins/" + html = _ld( + {"@context": "https://schema.org", "@type": "Product", "@id": pid, "name": "Galaxy A56 5G (8 GB Memory)", + "sku": "SM-A566EZGG", "offers": {"@type": "Offer", "price": "48999", "priceCurrency": "INR"}}, + {"@context": "https://schema.org", "@type": "Product", "@id": pid, + "aggregateRating": {"@type": "AggregateRating", "ratingValue": "4.5", "ratingCount": 1224}, + "review": [_review("Vedant", 5, "Great and useful AI features"), _review("Ravi", 1, "Heats up")]}, + ) + [p] = extract_products(html) + assert p["name"] == "Galaxy A56 5G (8 GB Memory)" and p["price"] == Decimal("48999") + assert p["rating"] == Decimal("4.5") and p["review_count"] == 1224 + assert [(r["author"], r["rating"], r["body"]) for r in p["reviews"]] == [ + ("Vedant", Decimal("5.0"), "Great and useful AI features"), ("Ravi", Decimal("1.0"), "Heats up")] + + +def test_nameless_review_node_without_id_is_not_guessed_between_products(): + lone = {"@type": "Product", "aggregateRating": {"ratingValue": "4.1", "ratingCount": 9}} + [only] = extract_products(_ld({"@type": "Product", "name": "Galaxy A56"}, lone)) + assert only["rating"] == Decimal("4.1") # one product on the page: unambiguous + two = extract_products(_ld({"@type": "Product", "name": "Galaxy A56"}, + {"@type": "Product", "name": "Galaxy A36"}, lone)) + assert all(p["rating"] is None for p in two) # two products: never guess which + + +def test_reliance_embedded_rating_is_read_from_the_pages_own_product_only(): + from app.electronics.extract.embedded_ratings import embedded_rating + + def page(app, recommended=None): + state = {"productDetailsPage": {"product": {"name": "Dell 15", "_custom_json": {"_app": app}}}, + "recommendations": [{"_custom_json": {"_app": recommended}}] if recommended else []} + return f"" + + url = "https://www.reliancedigital.in/product/dell-15-9991568" + got = embedded_rating("reliancedigital.in", page( + {"averageRating": 4, "ratingsCount": 3, "ratingsCountDetails": {"5": 1, "4": 1, "3": 1}}), url) + assert got == {"rating": Decimal("4.00"), "review_count": 3, "breakdown": {"5": 1, "4": 1, "3": 1}} + # "No ratings yet" is published as zeros; a recommended product's rating is not ours. + zero = page({"averageRating": 0, "ratingsCount": 0}, recommended={"averageRating": 4.8, "ratingsCount": 90}) + assert embedded_rating("reliancedigital.in", zero, url) is None + + +def test_poorvika_rating_needs_the_pages_own_slug(): + from app.electronics.extract.embedded_ratings import embedded_rating + + def page(slug, rating, count): + data = {"props": {"pageProps": {"additionalData": {"slug": slug, "rating": rating, "ratingCount": count}}}} + return f'' + + slug = "xiaomi-14-civi-5g-shadow-black-256gb-8gb-ram" + url = f"https://www.poorvika.com/{slug}/p" + assert embedded_rating("poorvika.com", page(slug, 5, 1), url)["rating"] == Decimal("5.00") + assert embedded_rating("poorvika.com", page("some-other-phone", 5, 1), url) is None + assert embedded_rating("poorvika.com", page(slug, None, 0), url) is None + + +def test_vasanth_template_rating_is_never_read(): + from app.electronics.extract.embedded_ratings import embedded_rating + + template = ('

3.3

(3 Reviews)' + + _ld({"@type": "Product", "name": "HP 15", "aggregateRating": {"ratingValue": "3.3", "reviewCount": 3}})) + assert embedded_rating("vasanthandco.in", template, "https://vasanthandco.in/product/174200102203/") is None + + +def test_vijaysales_graphql_review_feed(): + from app.electronics.extract import vijaysales_reviews as vs + + url = "https://www.vijaysales.com/p/239904/acer-aspire-lite-al15-41-laptop-amd-ryzen-7-16gb" + assert vs.url_key(url) == "acer-aspire-lite-al15-41-laptop-amd-ryzen-7-16gb" + assert "url_key" in vs.graphql_url(url) + assert vs.url_key('https://x.in/p/a"b') is None # cannot break out of the query + product_name = "Acer Aspire Lite AL15-41 Laptop (AMD Ryzen 7/ 16GB DDR4 RAM/ 512GB SSD)" + body = json.dumps({"data": {"products": {"items": [ + {"sku": "P1", "rating_summary": 0, "review_count": 0, "reviews": {"items": []}}, + {"sku": "239904", "rating_summary": 90, "review_count": 2, "reviews": {"items": [ + {"nickname": "Jay", "summary": product_name, "text": "All over performance is good", + "average_rating": 100, "created_at": "2026-05-02 10:00:00"}, + {"nickname": "guest user", "summary": "Okay", "text": "", "average_rating": 60}]}}]}}}) + got = vs.parse(body) + assert got["rating"] == Decimal("4.50") and got["review_count"] == 2 + # Product-name "title" dropped; a review with no text is skipped. + assert got["reviews"] == [{"author": "Jay", "rating": Decimal("5.00"), "title": None, + "body": "All over performance is good", "review_date": "2026-05-02 10:00:00", + "sentiment": "positive"}] + assert vs.parse('{"data":{"products":{"items":[]}}}') is None + + +def test_api_breakdown_only_when_a_platform_publishes_one(db, client): + from app.electronics.collector import Collector, RunOptions, RunStats + from app.electronics.db import repository as repo + from app.electronics.models import Listing + from app.electronics.normalise.title_parser import parse_title, variant_key + + def listing(site, sku, price, source_type="search_snippet"): + title = "Samsung Galaxy S24 5G (8GB RAM, 256GB)" + p = parse_title(title, "mobiles") + l = Listing(site_domain=site, source_sku=sku, source_url=f"https://www.{site}/p/{sku}", + source_type=source_type, brand_slug="samsung", category="mobiles", title=title, + evidence_text=f"{title} ₹{price}", confidence=0.5, parser="test", + model=p.model, ram_gb=p.ram_gb, storage_gb=p.storage_gb, price=Decimal(price)) + l.model_norm, l.variant_key = p.model_norm, variant_key(p, "mobiles") + return l + + c = Collector.__new__(Collector) + c.opt = RunOptions(category="mobiles", brands=["samsung"]) + c.ids, c.run_id, c._touched_products, c.stats = repo.id_maps(), None, {}, RunStats() + b = listing("reliancedigital.in", "rd-1", 73999, "scraped_page") + b.rating, b.review_count, b.rating_breakdown = Decimal("4.0"), 4, {"5": 2, "4": 1, "1": 1} + c.store(listing("amazon.in", "B0CS5XW6TN", 74999)) + c.store(b) + repo.refresh_verification() + pid = client.get("/api/elec/products", params={"category": "mobiles"}).json()["products"][0]["product_id"] + rating = client.get(f"/api/elec/products/{pid}").json()["rating"] + assert rating["breakdown"] == [ + {"stars": 5, "count": 2, "percent": 50}, {"stars": 4, "count": 1, "percent": 25}, + {"stars": 3, "count": 0, "percent": 0}, {"stars": 2, "count": 0, "percent": 0}, + {"stars": 1, "count": 1, "percent": 25}] + assert rating["sources"][0]["breakdown"] == {"5": 2, "4": 1, "1": 1} + + +def test_brand_store_page_must_be_the_exact_model_and_variant(): + from app.electronics.normalise.title_parser import parse_title + from app.electronics.review_refresh import _same_variant + + a56 = {"model_norm": "galaxy a56", "ram_gb": Decimal("8"), "storage_gb": Decimal("256")} + page = lambda t: parse_title(t, "mobiles", expected_brand="samsung") # noqa: E731 + assert _same_variant(page("Samsung Galaxy A56 5G 8GB/256GB (Olive)"), a56) + # A neighbouring model scores ~90 on fuzzy matching; it must still be refused. + assert not _same_variant(page("Samsung Galaxy A57 5G 8GB/256GB (Navy)"), a56) + assert not _same_variant(page("Samsung Galaxy A56 5G 12GB/256GB (Olive)"), a56) + assert not _same_variant(page("Samsung Galaxy A56 5G 8GB/128GB (Blue)"), a56) + s25 = {"model_norm": "galaxy s25", "ram_gb": Decimal("12"), "storage_gb": Decimal("256")} + assert _same_variant(page("Samsung Galaxy S25 12GB/256GB (Icyblue)"), s25) + assert not _same_variant(page("Samsung Galaxy S25 Ultra 12GB/256GB (Titanium Black)"), s25) + assert not _same_variant(page("Samsung Galaxy S25+ 12GB/256GB (Navy)"), s25)