Verified catalogue of mobiles and laptops sold in India, collected from real retail listings (FastAPI backend, React frontend, Postgres/pgvector). - REST API under /api/elec (read-only catalogue; admin endpoints need login) - MCP server (FastMCP) at /mcp/ with list_categories, search_products, get_product and price_history tools - Real ratings and reviews read from product pages and search results - Production Dockerfile (requirements-api.txt, no PyTorch) and .env.production.example; remote database only via an explicit ELEC_ALLOW_REMOTE_DB host/name allowlist - docs/API.md: endpoint and MCP reference with live examples Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
203 lines
7.7 KiB
Python
203 lines
7.7 KiB
Python
"""schema.org Product data embedded in a page as JSON-LD.
|
|
|
|
This is the preferred source on any page: it is what the site publishes for
|
|
search engines, so it is stable and states price, currency and availability
|
|
explicitly.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import re
|
|
from decimal import Decimal, InvalidOperation
|
|
from typing import Any, Dict, Iterable, List, Optional
|
|
|
|
from bs4 import BeautifulSoup
|
|
|
|
_PRODUCT_TYPES = {"product", "productgroup", "productmodel", "individualproduct"}
|
|
|
|
|
|
def _types(node: dict) -> set:
|
|
t = node.get("@type")
|
|
if isinstance(t, list):
|
|
return {str(x).lower() for x in t}
|
|
return {str(t).lower()} if t else set()
|
|
|
|
|
|
def _walk(node: Any) -> Iterable[dict]:
|
|
if isinstance(node, dict):
|
|
yield node
|
|
for v in node.values():
|
|
yield from _walk(v)
|
|
elif isinstance(node, list):
|
|
for item in node:
|
|
yield from _walk(item)
|
|
|
|
|
|
def json_ld_blocks(html: str) -> List[Any]:
|
|
soup = BeautifulSoup(html, "lxml")
|
|
blocks = []
|
|
for tag in soup.find_all("script", attrs={"type": re.compile(r"ld\+json", re.I)}):
|
|
raw = (tag.string or tag.get_text() or "").strip()
|
|
if not raw:
|
|
continue
|
|
try:
|
|
blocks.append(json.loads(raw))
|
|
except json.JSONDecodeError:
|
|
# Some sites put several objects or trailing commas in one tag.
|
|
try:
|
|
blocks.append(json.loads(re.sub(r",\s*([}\]])", r"\1", raw)))
|
|
except json.JSONDecodeError:
|
|
continue
|
|
return blocks
|
|
|
|
|
|
def _dec(value: Any) -> Optional[Decimal]:
|
|
if value is None or value == "":
|
|
return None
|
|
try:
|
|
return Decimal(str(value).replace(",", "").strip())
|
|
except InvalidOperation:
|
|
return None
|
|
|
|
|
|
def _text(value: Any) -> Optional[str]:
|
|
if isinstance(value, dict):
|
|
value = value.get("name") or value.get("@value")
|
|
if isinstance(value, list):
|
|
value = value[0] if value else None
|
|
return str(value).strip() if value not in (None, "") else None
|
|
|
|
|
|
def _images(value: Any) -> List[str]:
|
|
out: List[str] = []
|
|
for v in value if isinstance(value, list) else [value]:
|
|
if isinstance(v, dict):
|
|
v = v.get("url") or v.get("contentUrl")
|
|
if isinstance(v, str) and v.startswith(("http://", "https://")):
|
|
out.append(v)
|
|
return out
|
|
|
|
|
|
def _availability(value: Any) -> tuple:
|
|
text = (_text(value) or "").lower()
|
|
if not text:
|
|
return None, None
|
|
if "instock" in text or "limitedavailability" in text or "onlineonly" in text:
|
|
return "InStock", True
|
|
if any(k in text for k in ("outofstock", "soldout", "discontinued", "preorder", "presale")):
|
|
return text.rsplit("/", 1)[-1], False
|
|
return text.rsplit("/", 1)[-1], None
|
|
|
|
|
|
def _offer(offers: Any) -> Dict[str, Any]:
|
|
"""The price/availability of the product's (lowest) offer."""
|
|
candidates = offers if isinstance(offers, list) else [offers]
|
|
best: Dict[str, Any] = {}
|
|
for o in candidates:
|
|
if not isinstance(o, dict):
|
|
continue
|
|
price = _dec(o.get("price"))
|
|
if price is None:
|
|
price = _dec(o.get("lowPrice"))
|
|
if price is None and isinstance(o.get("priceSpecification"), dict):
|
|
price = _dec(o["priceSpecification"].get("price"))
|
|
currency = _text(o.get("priceCurrency")) or (
|
|
_text(o["priceSpecification"].get("priceCurrency")) if isinstance(o.get("priceSpecification"), dict) else None
|
|
)
|
|
availability, in_stock = _availability(o.get("availability"))
|
|
entry = {"price": price, "currency": currency, "availability": availability, "in_stock": in_stock,
|
|
"raw": {k: o.get(k) for k in ("price", "lowPrice", "priceCurrency", "availability") if k in o}}
|
|
if price is not None and (not best or best.get("price") is None or price < best["price"]):
|
|
best = entry
|
|
elif not best:
|
|
best = entry
|
|
return best
|
|
|
|
|
|
MAX_REVIEWS_PER_PAGE = 30
|
|
|
|
|
|
def _review_rating(value: Any) -> Optional[Decimal]:
|
|
"""A reviewer's star rating, rescaled to 0-5 when the page uses another scale."""
|
|
if not isinstance(value, dict):
|
|
return None
|
|
rating = _dec(value.get("ratingValue"))
|
|
if rating is None:
|
|
return None
|
|
best = _dec(value.get("bestRating")) or Decimal(5)
|
|
if best <= 0:
|
|
return None
|
|
if best != 5:
|
|
rating = rating * Decimal(5) / best
|
|
if not (Decimal(0) <= rating <= Decimal(5)):
|
|
return None
|
|
return rating.quantize(Decimal("0.1"))
|
|
|
|
|
|
def _reviews(node: dict) -> List[Dict[str, Any]]:
|
|
"""Customer reviews published on the Product node (schema.org Review).
|
|
|
|
Only reviews with text are kept - a bare star with no words is not
|
|
something a reader can weigh. Nothing is paraphrased or summarised: body,
|
|
title and author are the page's own strings.
|
|
"""
|
|
raw = node.get("review") or node.get("reviews") or []
|
|
out: List[Dict[str, Any]] = []
|
|
for r in raw if isinstance(raw, list) else [raw]:
|
|
if not isinstance(r, dict):
|
|
continue
|
|
body = _text(r.get("reviewBody")) or _text(r.get("description"))
|
|
if not body:
|
|
continue
|
|
out.append({
|
|
"author": _text(r.get("author")),
|
|
"rating": _review_rating(r.get("reviewRating")),
|
|
"title": _text(r.get("name")) or _text(r.get("headline")),
|
|
"body": body[:4000],
|
|
"review_date": _text(r.get("datePublished")) or _text(r.get("dateCreated")),
|
|
})
|
|
if len(out) >= MAX_REVIEWS_PER_PAGE:
|
|
break
|
|
return out
|
|
|
|
|
|
def extract_products(html: str) -> List[Dict[str, Any]]:
|
|
"""All schema.org Product nodes on the page, flattened to plain fields."""
|
|
products: List[Dict[str, Any]] = []
|
|
for block in json_ld_blocks(html):
|
|
for node in _walk(block):
|
|
if not (_types(node) & _PRODUCT_TYPES):
|
|
continue
|
|
name = _text(node.get("name"))
|
|
if not name:
|
|
continue
|
|
offer = _offer(node.get("offers")) if node.get("offers") else {}
|
|
if not offer and isinstance(node.get("hasVariant"), list):
|
|
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
|
|
props = {}
|
|
for p in node.get("additionalProperty") or []:
|
|
if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""):
|
|
props[str(p["name"])] = str(p["value"])
|
|
rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {}
|
|
products.append({
|
|
"name": name,
|
|
"brand": _text(node.get("brand")),
|
|
"sku": _text(node.get("sku")) or _text(node.get("productID")),
|
|
"mpn": _text(node.get("mpn")),
|
|
"gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None),
|
|
"color": _text(node.get("color")),
|
|
"images": _images(node.get("image")),
|
|
"description": _text(node.get("description")),
|
|
"price": offer.get("price"),
|
|
"currency": offer.get("currency"),
|
|
"availability": offer.get("availability"),
|
|
"in_stock": offer.get("in_stock"),
|
|
# Sites publish 0 for "no ratings yet"; that is not a rating.
|
|
"rating": (_dec(rating.get("ratingValue")) or None),
|
|
"review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None,
|
|
"reviews": _reviews(node),
|
|
"properties": props,
|
|
"evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500],
|
|
})
|
|
return products
|