Files
sriram c7e4d59188 Electronics Catalog: API, MCP server, frontend and deployment
Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
2026-10-01 12:17:42 +05:30

203 lines
7.7 KiB
Python

"""schema.org Product data embedded in a page as JSON-LD.
This is the preferred source on any page: it is what the site publishes for
search engines, so it is stable and states price, currency and availability
explicitly.
"""
from __future__ import annotations
import json
import re
from decimal import Decimal, InvalidOperation
from typing import Any, Dict, Iterable, List, Optional
from bs4 import BeautifulSoup
_PRODUCT_TYPES = {"product", "productgroup", "productmodel", "individualproduct"}
def _types(node: dict) -> set:
t = node.get("@type")
if isinstance(t, list):
return {str(x).lower() for x in t}
return {str(t).lower()} if t else set()
def _walk(node: Any) -> Iterable[dict]:
if isinstance(node, dict):
yield node
for v in node.values():
yield from _walk(v)
elif isinstance(node, list):
for item in node:
yield from _walk(item)
def json_ld_blocks(html: str) -> List[Any]:
soup = BeautifulSoup(html, "lxml")
blocks = []
for tag in soup.find_all("script", attrs={"type": re.compile(r"ld\+json", re.I)}):
raw = (tag.string or tag.get_text() or "").strip()
if not raw:
continue
try:
blocks.append(json.loads(raw))
except json.JSONDecodeError:
# Some sites put several objects or trailing commas in one tag.
try:
blocks.append(json.loads(re.sub(r",\s*([}\]])", r"\1", raw)))
except json.JSONDecodeError:
continue
return blocks
def _dec(value: Any) -> Optional[Decimal]:
if value is None or value == "":
return None
try:
return Decimal(str(value).replace(",", "").strip())
except InvalidOperation:
return None
def _text(value: Any) -> Optional[str]:
if isinstance(value, dict):
value = value.get("name") or value.get("@value")
if isinstance(value, list):
value = value[0] if value else None
return str(value).strip() if value not in (None, "") else None
def _images(value: Any) -> List[str]:
out: List[str] = []
for v in value if isinstance(value, list) else [value]:
if isinstance(v, dict):
v = v.get("url") or v.get("contentUrl")
if isinstance(v, str) and v.startswith(("http://", "https://")):
out.append(v)
return out
def _availability(value: Any) -> tuple:
text = (_text(value) or "").lower()
if not text:
return None, None
if "instock" in text or "limitedavailability" in text or "onlineonly" in text:
return "InStock", True
if any(k in text for k in ("outofstock", "soldout", "discontinued", "preorder", "presale")):
return text.rsplit("/", 1)[-1], False
return text.rsplit("/", 1)[-1], None
def _offer(offers: Any) -> Dict[str, Any]:
"""The price/availability of the product's (lowest) offer."""
candidates = offers if isinstance(offers, list) else [offers]
best: Dict[str, Any] = {}
for o in candidates:
if not isinstance(o, dict):
continue
price = _dec(o.get("price"))
if price is None:
price = _dec(o.get("lowPrice"))
if price is None and isinstance(o.get("priceSpecification"), dict):
price = _dec(o["priceSpecification"].get("price"))
currency = _text(o.get("priceCurrency")) or (
_text(o["priceSpecification"].get("priceCurrency")) if isinstance(o.get("priceSpecification"), dict) else None
)
availability, in_stock = _availability(o.get("availability"))
entry = {"price": price, "currency": currency, "availability": availability, "in_stock": in_stock,
"raw": {k: o.get(k) for k in ("price", "lowPrice", "priceCurrency", "availability") if k in o}}
if price is not None and (not best or best.get("price") is None or price < best["price"]):
best = entry
elif not best:
best = entry
return best
MAX_REVIEWS_PER_PAGE = 30
def _review_rating(value: Any) -> Optional[Decimal]:
"""A reviewer's star rating, rescaled to 0-5 when the page uses another scale."""
if not isinstance(value, dict):
return None
rating = _dec(value.get("ratingValue"))
if rating is None:
return None
best = _dec(value.get("bestRating")) or Decimal(5)
if best <= 0:
return None
if best != 5:
rating = rating * Decimal(5) / best
if not (Decimal(0) <= rating <= Decimal(5)):
return None
return rating.quantize(Decimal("0.1"))
def _reviews(node: dict) -> List[Dict[str, Any]]:
"""Customer reviews published on the Product node (schema.org Review).
Only reviews with text are kept - a bare star with no words is not
something a reader can weigh. Nothing is paraphrased or summarised: body,
title and author are the page's own strings.
"""
raw = node.get("review") or node.get("reviews") or []
out: List[Dict[str, Any]] = []
for r in raw if isinstance(raw, list) else [raw]:
if not isinstance(r, dict):
continue
body = _text(r.get("reviewBody")) or _text(r.get("description"))
if not body:
continue
out.append({
"author": _text(r.get("author")),
"rating": _review_rating(r.get("reviewRating")),
"title": _text(r.get("name")) or _text(r.get("headline")),
"body": body[:4000],
"review_date": _text(r.get("datePublished")) or _text(r.get("dateCreated")),
})
if len(out) >= MAX_REVIEWS_PER_PAGE:
break
return out
def extract_products(html: str) -> List[Dict[str, Any]]:
"""All schema.org Product nodes on the page, flattened to plain fields."""
products: List[Dict[str, Any]] = []
for block in json_ld_blocks(html):
for node in _walk(block):
if not (_types(node) & _PRODUCT_TYPES):
continue
name = _text(node.get("name"))
if not name:
continue
offer = _offer(node.get("offers")) if node.get("offers") else {}
if not offer and isinstance(node.get("hasVariant"), list):
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
props = {}
for p in node.get("additionalProperty") or []:
if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""):
props[str(p["name"])] = str(p["value"])
rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {}
products.append({
"name": name,
"brand": _text(node.get("brand")),
"sku": _text(node.get("sku")) or _text(node.get("productID")),
"mpn": _text(node.get("mpn")),
"gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None),
"color": _text(node.get("color")),
"images": _images(node.get("image")),
"description": _text(node.get("description")),
"price": offer.get("price"),
"currency": offer.get("currency"),
"availability": offer.get("availability"),
"in_stock": offer.get("in_stock"),
# Sites publish 0 for "no ratings yet"; that is not a rating.
"rating": (_dec(rating.get("ratingValue")) or None),
"review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None,
"reviews": _reviews(node),
"properties": props,
"evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500],
})
return products