Electronics Catalog: API, MCP server, frontend and deployment

Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
sriram
2026-10-01 12:17:42 +05:30
commit c7e4d59188
115 changed files with 14329 additions and 0 deletions

View File

@@ -0,0 +1,123 @@
"""Product facts from page markup when there is no usable JSON-LD.
Only machine-readable markup is trusted for the price: OpenGraph/product meta
tags and schema.org microdata (itemprop="price"). Free text on the page is not
scanned for rupee amounts - a product page shows EMIs, offers and other
products' prices, and picking the wrong one is worse than picking none.
Spec tables (<table>, <dl>) supply specifications.
"""
from __future__ import annotations
import json
import re
from decimal import Decimal, InvalidOperation
from typing import Any, Dict, List, Optional
from bs4 import BeautifulSoup
def _dec(value: Optional[str]) -> Optional[Decimal]:
if not value:
return None
try:
return Decimal(re.sub(r"[^\d.]", "", value))
except InvalidOperation:
return None
def _meta(soup: BeautifulSoup, *names: str) -> Optional[str]:
for name in names:
tag = soup.find("meta", attrs={"property": name}) or soup.find("meta", attrs={"name": name})
if tag and tag.get("content"):
return tag["content"].strip()
return None
def spec_tables(soup: BeautifulSoup, limit: int = 200) -> Dict[str, str]:
specs: Dict[str, str] = {}
for row in soup.select("table tr"):
cells = row.find_all(["th", "td"])
if len(cells) == 2:
k, v = (c.get_text(" ", strip=True) for c in cells)
if k and v and len(k) <= 60 and len(v) <= 200:
specs.setdefault(k, v)
if len(specs) >= limit:
return specs
for dl in soup.find_all("dl"):
for dt in dl.find_all("dt"):
dd = dt.find_next_sibling("dd")
if dd:
k, v = dt.get_text(" ", strip=True), dd.get_text(" ", strip=True)
if k and v and len(k) <= 60 and len(v) <= 200:
specs.setdefault(k, v)
return specs
def extract_page(html: str) -> Dict[str, Any]:
soup = BeautifulSoup(html, "lxml")
title = _meta(soup, "og:title", "twitter:title")
if not title:
h1 = soup.find("h1")
title = h1.get_text(" ", strip=True) if h1 else None
images: List[str] = []
for name in ("og:image", "og:image:secure_url", "twitter:image"):
v = _meta(soup, name)
if v and v.startswith("http") and v not in images:
images.append(v)
price = _dec(_meta(soup, "product:price:amount", "og:price:amount"))
currency = _meta(soup, "product:price:currency", "og:price:currency")
evidence = ""
if price is not None:
evidence = f"meta product:price:amount={price} currency={currency}"
else:
tag = soup.find(attrs={"itemprop": "price"})
if tag is not None:
raw = tag.get("content") or tag.get_text(" ", strip=True)
price = _dec(raw)
cur_tag = soup.find(attrs={"itemprop": "priceCurrency"})
currency = (cur_tag.get("content") if cur_tag else None) or currency
if price is not None:
evidence = f'itemprop="price" {raw} currency={currency}'
availability = _meta(soup, "product:availability", "og:availability")
in_stock = None
if availability:
low = availability.lower().replace(" ", "")
in_stock = True if "instock" in low else False if ("outofstock" in low or "oos" == low) else None
return {
"name": title,
"images": images,
"price": price,
"currency": currency,
"availability": availability,
"in_stock": in_stock,
"properties": spec_tables(soup),
"evidence": evidence,
}
_STATE_RE = re.compile(
r"<script[^>]*id=\"__NEXT_DATA__\"[^>]*>(.*?)</script>"
r"|window\.__(?:INITIAL|PRELOADED)_STATE__\s*=\s*(\{.*?\})\s*;?\s*</script>",
re.DOTALL,
)
def embedded_state(html: str) -> Optional[Any]:
"""The page's server-rendered application state, when it embeds one."""
for m in _STATE_RE.finditer(html):
raw = m.group(1) or m.group(2)
try:
return json.loads(raw)
except (json.JSONDecodeError, TypeError):
continue
return None
def visible_text(html: str, limit: int = 6000) -> str:
soup = BeautifulSoup(html, "lxml")
for tag in soup(["script", "style", "noscript", "svg", "header", "footer", "nav"]):
tag.decompose()
return re.sub(r"\s+", " ", soup.get_text(" ", strip=True))[:limit]

View File

@@ -0,0 +1,202 @@
"""schema.org Product data embedded in a page as JSON-LD.
This is the preferred source on any page: it is what the site publishes for
search engines, so it is stable and states price, currency and availability
explicitly.
"""
from __future__ import annotations
import json
import re
from decimal import Decimal, InvalidOperation
from typing import Any, Dict, Iterable, List, Optional
from bs4 import BeautifulSoup
_PRODUCT_TYPES = {"product", "productgroup", "productmodel", "individualproduct"}
def _types(node: dict) -> set:
t = node.get("@type")
if isinstance(t, list):
return {str(x).lower() for x in t}
return {str(t).lower()} if t else set()
def _walk(node: Any) -> Iterable[dict]:
if isinstance(node, dict):
yield node
for v in node.values():
yield from _walk(v)
elif isinstance(node, list):
for item in node:
yield from _walk(item)
def json_ld_blocks(html: str) -> List[Any]:
soup = BeautifulSoup(html, "lxml")
blocks = []
for tag in soup.find_all("script", attrs={"type": re.compile(r"ld\+json", re.I)}):
raw = (tag.string or tag.get_text() or "").strip()
if not raw:
continue
try:
blocks.append(json.loads(raw))
except json.JSONDecodeError:
# Some sites put several objects or trailing commas in one tag.
try:
blocks.append(json.loads(re.sub(r",\s*([}\]])", r"\1", raw)))
except json.JSONDecodeError:
continue
return blocks
def _dec(value: Any) -> Optional[Decimal]:
if value is None or value == "":
return None
try:
return Decimal(str(value).replace(",", "").strip())
except InvalidOperation:
return None
def _text(value: Any) -> Optional[str]:
if isinstance(value, dict):
value = value.get("name") or value.get("@value")
if isinstance(value, list):
value = value[0] if value else None
return str(value).strip() if value not in (None, "") else None
def _images(value: Any) -> List[str]:
out: List[str] = []
for v in value if isinstance(value, list) else [value]:
if isinstance(v, dict):
v = v.get("url") or v.get("contentUrl")
if isinstance(v, str) and v.startswith(("http://", "https://")):
out.append(v)
return out
def _availability(value: Any) -> tuple:
text = (_text(value) or "").lower()
if not text:
return None, None
if "instock" in text or "limitedavailability" in text or "onlineonly" in text:
return "InStock", True
if any(k in text for k in ("outofstock", "soldout", "discontinued", "preorder", "presale")):
return text.rsplit("/", 1)[-1], False
return text.rsplit("/", 1)[-1], None
def _offer(offers: Any) -> Dict[str, Any]:
"""The price/availability of the product's (lowest) offer."""
candidates = offers if isinstance(offers, list) else [offers]
best: Dict[str, Any] = {}
for o in candidates:
if not isinstance(o, dict):
continue
price = _dec(o.get("price"))
if price is None:
price = _dec(o.get("lowPrice"))
if price is None and isinstance(o.get("priceSpecification"), dict):
price = _dec(o["priceSpecification"].get("price"))
currency = _text(o.get("priceCurrency")) or (
_text(o["priceSpecification"].get("priceCurrency")) if isinstance(o.get("priceSpecification"), dict) else None
)
availability, in_stock = _availability(o.get("availability"))
entry = {"price": price, "currency": currency, "availability": availability, "in_stock": in_stock,
"raw": {k: o.get(k) for k in ("price", "lowPrice", "priceCurrency", "availability") if k in o}}
if price is not None and (not best or best.get("price") is None or price < best["price"]):
best = entry
elif not best:
best = entry
return best
MAX_REVIEWS_PER_PAGE = 30
def _review_rating(value: Any) -> Optional[Decimal]:
"""A reviewer's star rating, rescaled to 0-5 when the page uses another scale."""
if not isinstance(value, dict):
return None
rating = _dec(value.get("ratingValue"))
if rating is None:
return None
best = _dec(value.get("bestRating")) or Decimal(5)
if best <= 0:
return None
if best != 5:
rating = rating * Decimal(5) / best
if not (Decimal(0) <= rating <= Decimal(5)):
return None
return rating.quantize(Decimal("0.1"))
def _reviews(node: dict) -> List[Dict[str, Any]]:
"""Customer reviews published on the Product node (schema.org Review).
Only reviews with text are kept - a bare star with no words is not
something a reader can weigh. Nothing is paraphrased or summarised: body,
title and author are the page's own strings.
"""
raw = node.get("review") or node.get("reviews") or []
out: List[Dict[str, Any]] = []
for r in raw if isinstance(raw, list) else [raw]:
if not isinstance(r, dict):
continue
body = _text(r.get("reviewBody")) or _text(r.get("description"))
if not body:
continue
out.append({
"author": _text(r.get("author")),
"rating": _review_rating(r.get("reviewRating")),
"title": _text(r.get("name")) or _text(r.get("headline")),
"body": body[:4000],
"review_date": _text(r.get("datePublished")) or _text(r.get("dateCreated")),
})
if len(out) >= MAX_REVIEWS_PER_PAGE:
break
return out
def extract_products(html: str) -> List[Dict[str, Any]]:
"""All schema.org Product nodes on the page, flattened to plain fields."""
products: List[Dict[str, Any]] = []
for block in json_ld_blocks(html):
for node in _walk(block):
if not (_types(node) & _PRODUCT_TYPES):
continue
name = _text(node.get("name"))
if not name:
continue
offer = _offer(node.get("offers")) if node.get("offers") else {}
if not offer and isinstance(node.get("hasVariant"), list):
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
props = {}
for p in node.get("additionalProperty") or []:
if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""):
props[str(p["name"])] = str(p["value"])
rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {}
products.append({
"name": name,
"brand": _text(node.get("brand")),
"sku": _text(node.get("sku")) or _text(node.get("productID")),
"mpn": _text(node.get("mpn")),
"gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None),
"color": _text(node.get("color")),
"images": _images(node.get("image")),
"description": _text(node.get("description")),
"price": offer.get("price"),
"currency": offer.get("currency"),
"availability": offer.get("availability"),
"in_stock": offer.get("in_stock"),
# Sites publish 0 for "no ratings yet"; that is not a rating.
"rating": (_dec(rating.get("ratingValue")) or None),
"review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None,
"reviews": _reviews(node),
"properties": props,
"evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500],
})
return products

View File

@@ -0,0 +1,207 @@
"""Read prices and stock state out of text we did not render ourselves:
search-result titles/snippets, and visible page text.
The rules lean hard towards NOT returning a price. A snippet usually carries
several rupee amounts - the selling price, the MRP, an EMI, a bank discount, an
exchange value, "₹X off" - and taking the wrong one is worse than taking none.
An amount is only a price when nothing around it says it is something else.
"""
from __future__ import annotations
import re
from dataclasses import dataclass
from decimal import Decimal, InvalidOperation
from typing import List, Optional
PRICE_MIN = Decimal("500")
PRICE_MAX = Decimal("1000000")
# ₹ / Rs / Rs. / INR followed by an amount with Indian (1,29,999) or western
# (129,999) grouping, or none.
_AMOUNT = r"(\d{1,3}(?:,\d{2,3})+(?:\.\d{1,2})?|\d+(?:\.\d{1,2})?)"
_MONEY_RE = re.compile(r"(?:₹|\bRs\.?|\bINR)\s?" + _AMOUNT, re.IGNORECASE)
# Words that make an amount something other than the selling price.
_REJECT_BEFORE = re.compile(
r"(?:emi|save|saving|savings|cashback|cash\s*back|exchange|bank|discount|coupon|"
r"extra|instant|up\s*to|upto|flat|off\s+upto|worth|delivery|shipping|fee|charges?|"
r"starting|starts|from|onwards|min(?:imum)?|as\s+low\s+as|down\s*payment|per\s+month)\W*$",
re.IGNORECASE,
)
_REJECT_AFTER = re.compile(
r"^\W{0,3}(?:off\b|/\s*m(?:o|onth)?\b|per\s+month|p\.?m\.?\b|a\s+month|emi\b|/-?\s*emi|"
r"cashback|discount|savings?|onwards|\+\s*shipping|delivery)",
re.IGNORECASE,
)
_MRP_BEFORE = re.compile(r"(?:m\.?\s?r\.?\s?p\.?|list\s+price|was|original\s+price)[\s:]*$", re.IGNORECASE)
_RANGE_BETWEEN = re.compile(r"^\s*(?:-|–|—|to)\s*$", re.IGNORECASE)
_OUT_OF_STOCK = re.compile(
r"\b(?:out\s+of\s+stock|currently\s+unavailable|sold\s+out|coming\s+soon|notify\s+me|"
r"temporarily\s+unavailable|not\s+available)\b",
re.IGNORECASE,
)
_IN_STOCK = re.compile(r"\b(?:in\s+stock|available\s+now|buy\s+now|add\s+to\s+cart)\b", re.IGNORECASE)
@dataclass(frozen=True)
class Amount:
value: Decimal
kind: str # price | mrp | rejected
reason: str
start: int
end: int
raw: str
def parse_amount(raw: str) -> Optional[Decimal]:
try:
value = Decimal(raw.replace(",", ""))
except InvalidOperation:
return None
return value
def find_amounts(text: str) -> List[Amount]:
"""Every rupee amount in `text`, each classified as price, mrp or rejected."""
out: List[Amount] = []
if not text:
return out
matches = list(_MONEY_RE.finditer(text))
for i, m in enumerate(matches):
value = parse_amount(m.group(1))
if value is None:
continue
before = text[max(0, m.start() - 28): m.start()]
after = text[m.end(): m.end() + 22]
kind, reason = "price", ""
if _MRP_BEFORE.search(before):
kind, reason = "mrp", "labelled MRP"
elif _REJECT_BEFORE.search(before):
kind, reason = "rejected", f"preceded by {_REJECT_BEFORE.search(before).group(0).strip()!r}"
elif _REJECT_AFTER.search(after):
kind, reason = "rejected", f"followed by {_REJECT_AFTER.search(after).group(0).strip()!r}"
# A range ("₹10,999 - ₹12,999") names no single price.
if kind == "price":
if i + 1 < len(matches) and _RANGE_BETWEEN.match(text[m.end(): matches[i + 1].start()]):
kind, reason = "rejected", "start of a price range"
elif i > 0 and _RANGE_BETWEEN.match(text[matches[i - 1].end(): m.start()]):
kind, reason = "rejected", "end of a price range"
if kind != "rejected" and not (PRICE_MIN <= value <= PRICE_MAX):
kind, reason = "rejected", "outside plausible range"
out.append(Amount(value, kind, reason, m.start(), m.end(), m.group(0)))
return out
@dataclass(frozen=True)
class PriceReading:
price: Optional[Decimal]
mrp: Optional[Decimal]
evidence: str # the exact substring the price was read from ("" if none)
def read_price(text: str) -> PriceReading:
"""The single selling price stated in `text`, or None.
If the text states two different unlabelled prices, it is ambiguous (a
listing page snippet often shows several variants) and None is returned.
"""
amounts = find_amounts(text)
prices = [a for a in amounts if a.kind == "price"]
mrps = [a for a in amounts if a.kind == "mrp"]
distinct = {a.value for a in prices}
price: Optional[Decimal] = None
evidence = ""
if len(distinct) == 1:
price = prices[0].value
evidence = prices[0].raw
mrp = mrps[0].value if mrps else None
if price is not None and mrp is not None and mrp < price:
mrp = None # an "MRP" below the selling price was misread; drop it
return PriceReading(price, mrp, evidence)
def read_stock(text: str) -> Optional[bool]:
"""True/False only when the text says so; None when it does not."""
if not text:
return None
if _OUT_OF_STOCK.search(text):
return False
if _IN_STOCK.search(text):
return True
return None
@dataclass(frozen=True)
class RatingReading:
rating: Optional[Decimal]
review_count: Optional[int]
evidence: str # the exact substring the rating was read from ("" if none)
# Only ratings the text states explicitly on a 5-point scale:
# "4.3 out of 5 stars", "Rating: 4.3/5", "Rated 4.3 / 5", "4.3★", "4.3 ★ (1,234 ratings)"
_RATING_PATTERNS = (
re.compile(r"\b([0-5](?:\.\d{1,2})?)\s*out\s+of\s*5(?:\.0)?\b(?:\s*stars?)?", re.IGNORECASE),
# "x/5" only with a rating word before it or "stars" after it - a bare
# "1/5" is as likely a sensor size or a fraction.
re.compile(r"\brat(?:ing|ed)\s*[:\-]?\s*([0-5](?:\.\d{1,2})?)\s*/\s*5(?:\.0)?\b", re.IGNORECASE),
re.compile(r"\b([0-5](?:\.\d{1,2})?)\s*/\s*5(?:\.0)?\s*stars?\b", re.IGNORECASE),
re.compile(r"\b([0-5](?:\.\d{1,2})?)\s*(?:★|☆|⭐)"),
re.compile(r"\brat(?:ing|ed)\s*[:\-]?\s*([0-5](?:\.\d{1,2})?)\s*(?:stars?|★)", re.IGNORECASE),
)
_RATING_COUNT = re.compile(
r"^[\s()\-|·,.:]*(?:stars?)?[\s()\-|·,.:]*(\d{1,3}(?:,\d{2,3})+|\d+)\s*(?:customer\s+)?(?:ratings?|reviews?|votes?)\b",
re.IGNORECASE,
)
def read_rating(text: str) -> RatingReading:
"""The product rating a search title/snippet states, or None.
Only an explicit "x out of 5" / "x/5" / "x★" statement counts; bare
numbers never do. If the text states two different ratings it is
ambiguous (several products on one results page) and None is returned.
"""
if not text:
return RatingReading(None, None, "")
found = []
for pattern in _RATING_PATTERNS:
for m in pattern.finditer(text):
try:
value = Decimal(m.group(1))
except InvalidOperation:
continue
if Decimal(0) < value <= Decimal(5):
found.append((value, m))
if not found or len({v for v, _ in found}) != 1:
return RatingReading(None, None, "")
value, m = min(found, key=lambda f: f[1].start())
count = None
tail = _RATING_COUNT.match(text[m.end(): m.end() + 40])
if tail:
count = int(tail.group(1).replace(",", ""))
evidence = text[m.start(): m.end() + (tail.end() if tail else 0)].strip()
return RatingReading(value, count, evidence)
# Titles returned by search engines carry the site name; it is not part of the
# product title.
_TITLE_SUFFIX = re.compile(
r"\s*(?:[|\-–:]\s*)?(?:buy\s+online.*|online\s+at\s+best\s+price.*|"
r"at\s+best\s+price.*|price\s+in\s+india.*|"
r"amazon\.in.*|flipkart(?:\.com)?.*|croma.*|reliance\s+digital.*|vijay\s+sales.*|"
r"tata\s+cliq.*|poorvika.*|sangeetha.*|vasanth.*|viveks.*)$",
re.IGNORECASE,
)
_TITLE_PREFIX = re.compile(r"^(?:buy\s+|amazon\.in\s*:\s*)", re.IGNORECASE)
def clean_result_title(title: str) -> str:
t = (title or "").strip()
# Engines truncate with "..." and sometimes run several results' titles
# together after it; everything past the first ellipsis is not this page.
t = re.split(r"\s*(?:\.\.\.|…)", t, maxsplit=1)[0]
t = _TITLE_PREFIX.sub("", t)
t = _TITLE_SUFFIX.sub("", t)
return t.strip(" -|:–")