"""schema.org Product data embedded in a page as JSON-LD. This is the preferred source on any page: it is what the site publishes for search engines, so it is stable and states price, currency and availability explicitly. """ from __future__ import annotations import json import re from decimal import Decimal, InvalidOperation from typing import Any, Dict, Iterable, List, Optional from bs4 import BeautifulSoup _PRODUCT_TYPES = {"product", "productgroup", "productmodel", "individualproduct"} def _types(node: dict) -> set: t = node.get("@type") if isinstance(t, list): return {str(x).lower() for x in t} return {str(t).lower()} if t else set() def _walk(node: Any) -> Iterable[dict]: if isinstance(node, dict): yield node for v in node.values(): yield from _walk(v) elif isinstance(node, list): for item in node: yield from _walk(item) def json_ld_blocks(html: str) -> List[Any]: soup = BeautifulSoup(html, "lxml") blocks = [] for tag in soup.find_all("script", attrs={"type": re.compile(r"ld\+json", re.I)}): raw = (tag.string or tag.get_text() or "").strip() if not raw: continue try: blocks.append(json.loads(raw)) except json.JSONDecodeError: # Some sites put several objects or trailing commas in one tag. try: blocks.append(json.loads(re.sub(r",\s*([}\]])", r"\1", raw))) except json.JSONDecodeError: continue return blocks def _dec(value: Any) -> Optional[Decimal]: if value is None or value == "": return None try: return Decimal(str(value).replace(",", "").strip()) except InvalidOperation: return None def _text(value: Any) -> Optional[str]: if isinstance(value, dict): value = value.get("name") or value.get("@value") if isinstance(value, list): value = value[0] if value else None return str(value).strip() if value not in (None, "") else None def _images(value: Any) -> List[str]: out: List[str] = [] for v in value if isinstance(value, list) else [value]: if isinstance(v, dict): v = v.get("url") or v.get("contentUrl") if isinstance(v, str) and v.startswith(("http://", "https://")): out.append(v) return out def _availability(value: Any) -> tuple: text = (_text(value) or "").lower() if not text: return None, None if "instock" in text or "limitedavailability" in text or "onlineonly" in text: return "InStock", True if any(k in text for k in ("outofstock", "soldout", "discontinued", "preorder", "presale")): return text.rsplit("/", 1)[-1], False return text.rsplit("/", 1)[-1], None def _offer(offers: Any) -> Dict[str, Any]: """The price/availability of the product's (lowest) offer.""" candidates = offers if isinstance(offers, list) else [offers] best: Dict[str, Any] = {} for o in candidates: if not isinstance(o, dict): continue price = _dec(o.get("price")) if price is None: price = _dec(o.get("lowPrice")) if price is None and isinstance(o.get("priceSpecification"), dict): price = _dec(o["priceSpecification"].get("price")) currency = _text(o.get("priceCurrency")) or ( _text(o["priceSpecification"].get("priceCurrency")) if isinstance(o.get("priceSpecification"), dict) else None ) availability, in_stock = _availability(o.get("availability")) entry = {"price": price, "currency": currency, "availability": availability, "in_stock": in_stock, "raw": {k: o.get(k) for k in ("price", "lowPrice", "priceCurrency", "availability") if k in o}} if price is not None and (not best or best.get("price") is None or price < best["price"]): best = entry elif not best: best = entry return best MAX_REVIEWS_PER_PAGE = 30 def _review_rating(value: Any) -> Optional[Decimal]: """A reviewer's star rating, rescaled to 0-5 when the page uses another scale.""" if not isinstance(value, dict): return None rating = _dec(value.get("ratingValue")) if rating is None: return None best = _dec(value.get("bestRating")) or Decimal(5) if best <= 0: return None if best != 5: rating = rating * Decimal(5) / best if not (Decimal(0) <= rating <= Decimal(5)): return None return rating.quantize(Decimal("0.1")) def _reviews(node: dict) -> List[Dict[str, Any]]: """Customer reviews published on the Product node (schema.org Review). Only reviews with text are kept - a bare star with no words is not something a reader can weigh. Nothing is paraphrased or summarised: body, title and author are the page's own strings. """ raw = node.get("review") or node.get("reviews") or [] out: List[Dict[str, Any]] = [] for r in raw if isinstance(raw, list) else [raw]: if not isinstance(r, dict): continue body = _text(r.get("reviewBody")) or _text(r.get("description")) if not body: continue out.append({ "author": _text(r.get("author")), "rating": _review_rating(r.get("reviewRating")), "title": _text(r.get("name")) or _text(r.get("headline")), "body": body[:4000], "review_date": _text(r.get("datePublished")) or _text(r.get("dateCreated")), }) if len(out) >= MAX_REVIEWS_PER_PAGE: break return out def _rating_only_nodes(nodes: List[dict]) -> Dict[Optional[str], dict]: """Nameless Product nodes that only carry ratings/reviews, keyed by @id. Review widgets (Bazaarvoice on samsung.com/in, for one) publish the product's aggregateRating and reviews as a SEPARATE Product node with no name, linked to the real product by the same @id. Without merging, those real reviews would be skipped for lack of a name. """ out: Dict[Optional[str], dict] = {} for node in nodes: if _text(node.get("name")): continue if not (isinstance(node.get("aggregateRating"), dict) or node.get("review") or node.get("reviews")): continue out.setdefault(_text(node.get("@id")), node) return out def extract_products(html: str) -> List[Dict[str, Any]]: """All schema.org Product nodes on the page, flattened to plain fields.""" products: List[Dict[str, Any]] = [] nodes = [n for block in json_ld_blocks(html) for n in _walk(block) if _types(n) & _PRODUCT_TYPES] rating_nodes = _rating_only_nodes(nodes) named = [n for n in nodes if _text(n.get("name"))] for node in named: # Attach a ratings-only node: same @id, or - when it has no @id - the # page's only named product (there is then no doubt which it is about). extra = rating_nodes.get(_text(node.get("@id"))) if node.get("@id") else None if extra is None and len(named) == 1: extra = rating_nodes.get(None) if extra is not None: node = {**node, "aggregateRating": node.get("aggregateRating") or extra.get("aggregateRating"), "review": node.get("review") or extra.get("review") or extra.get("reviews")} name = _text(node.get("name")) offer = _offer(node.get("offers")) if node.get("offers") else {} if not offer and isinstance(node.get("hasVariant"), list): offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")]) props = {} for p in node.get("additionalProperty") or []: if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""): props[str(p["name"])] = str(p["value"]) rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {} products.append({ "name": name, "brand": _text(node.get("brand")), "sku": _text(node.get("sku")) or _text(node.get("productID")), "mpn": _text(node.get("mpn")), "gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None), "color": _text(node.get("color")), "images": _images(node.get("image")), "description": _text(node.get("description")), "price": offer.get("price"), "currency": offer.get("currency"), "availability": offer.get("availability"), "in_stock": offer.get("in_stock"), # Sites publish 0 for "no ratings yet"; that is not a rating. "rating": (_dec(rating.get("ratingValue")) or None), "review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None, "reviews": _reviews(node), "properties": props, "evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500], }) return products