Electronics Catalog: API, MCP server, frontend and deployment

Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
sriram
2026-10-01 12:17:42 +05:30
commit c7e4d59188
115 changed files with 14329 additions and 0 deletions

View File

@@ -0,0 +1,65 @@
"""Cached, budgeted access to the search providers.
Results are cached in elec.search_cache so a re-run does not query again
within SEARCH_CACHE_TTL_HOURS, and each run has a query budget so a large
brand list cannot hammer the providers.
"""
from __future__ import annotations
import logging
from typing import Dict, List, Optional
from app.electronics.db import repository as repo
from app.electronics.search.providers import DuckDuckGoProvider, GoogleCseProvider, SearchHit
from app.infrastructure.settings import GOOGLE_CSE_DAILY_QUOTA, SEARCH_CACHE_TTL_HOURS
logger = logging.getLogger(__name__)
class SearchEngine:
def __init__(self, *, budget: int = 200, use_cache: bool = True) -> None:
self.budget = budget
self.use_cache = use_cache
self.used = 0
self.stats: Dict[str, int] = {"cache_hits": 0, "queries": 0, "unavailable": 0}
self.ddg = DuckDuckGoProvider()
self.google = GoogleCseProvider(
quota_left=lambda: GOOGLE_CSE_DAILY_QUOTA - repo.google_queries_today()
)
def _ask(self, provider, kind: str, query: str, max_results: int) -> Optional[List[SearchHit]]:
if not provider.enabled:
return None
cached = repo.search_cache_get(provider.name, kind, query, SEARCH_CACHE_TTL_HOURS) if self.use_cache else None
if cached is not None:
self.stats["cache_hits"] += 1
return [SearchHit.from_dict(d) for d in cached]
if self.used >= self.budget:
logger.info("Search budget (%d) spent; skipping %r", self.budget, query)
return None
self.used += 1
self.stats["queries"] += 1
self.stats[f"queries_{provider.name}"] = self.stats.get(f"queries_{provider.name}", 0) + 1
hits = provider.text(query, max_results) if kind == "text" else provider.images(query, max_results)
if hits is None:
self.stats["unavailable"] += 1
return None
repo.search_cache_put(provider.name, kind, query, [h.to_dict() for h in hits])
return hits
def _run(self, kind: str, query: str, max_results: int, providers: str) -> Optional[List[SearchHit]]:
"""providers: "default" = DuckDuckGo, with Google only when DuckDuckGo
gives no answer (keeps the 100/day Google quota for price lookups);
"google" = Google only."""
if providers == "google":
return self._ask(self.google, kind, query, max_results)
hits = self._ask(self.ddg, kind, query, max_results)
if hits is None:
hits = self._ask(self.google, kind, query, max_results)
return hits
def text(self, query: str, max_results: int = 20, *, providers: str = "default") -> Optional[List[SearchHit]]:
return self._run("text", query, max_results, providers)
def images(self, query: str, max_results: int = 15) -> Optional[List[SearchHit]]:
return self._run("images", query, max_results, "default")

View File

@@ -0,0 +1,234 @@
"""Web search: DuckDuckGo (ddgs, no key) and, when configured, Google
Programmable Search.
Three outcomes, never collapsed: a list of hits, an empty list ("we asked and
nothing matched"), or None ("we could not ask" - throttled, offline, no
quota). A throttle is not evidence that a product is not sold anywhere.
"""
from __future__ import annotations
import logging
import threading
import time
from dataclasses import asdict, dataclass
from typing import Callable, List, Optional
import requests
from app.infrastructure.settings import (
GOOGLE_API_KEY,
GOOGLE_CSE_ID,
SEARCH_MIN_INTERVAL_SECONDS,
SEARCH_REGION,
USE_DDG_SEARCH,
USE_GOOGLE_CSE,
)
logger = logging.getLogger(__name__)
@dataclass
class SearchHit:
url: str
title: str
snippet: str
provider: str
rank: int
image_url: Optional[str] = None # image searches: the image itself (url = page it is on)
# Structured offer data the search engine itself extracted from the page
# (Google CSE "pagemap"): {"price", "currency", "availability", "raw"}.
offer: Optional[dict] = None
# Aggregate rating the search engine extracted from the page's own
# structured data (Google CSE "pagemap"): {"rating", "review_count", "raw"}.
rating: Optional[dict] = None
def to_dict(self) -> dict:
return asdict(self)
@classmethod
def from_dict(cls, d: dict) -> "SearchHit":
return cls(**{k: d.get(k) for k in ("url", "title", "snippet", "provider", "rank", "image_url", "offer", "rating")})
class _Pacer:
def __init__(self, interval: float, sleep: Callable[[float], None] = time.sleep) -> None:
self.interval = interval
self._sleep = sleep
self._last = 0.0
self._lock = threading.Lock()
def wait(self) -> None:
with self._lock:
gap = self.interval - (time.monotonic() - self._last)
if gap > 0:
self._sleep(gap)
self._last = time.monotonic()
class DuckDuckGoProvider:
name = "ddg"
def __init__(self, interval: float = SEARCH_MIN_INTERVAL_SECONDS) -> None:
self._pacer = _Pacer(interval)
self.enabled = USE_DDG_SEARCH
# "auto" rotates ddgs's engines; yahoo is a second opinion when it is throttled.
BACKENDS = ("auto", "yahoo")
def text(self, query: str, max_results: int = 20) -> Optional[List[SearchHit]]:
"""Hits, or None when no backend answered. ddgs reports a throttle and
a genuinely empty result the same way ("No results found"), so an
empty answer is treated as unknown rather than as "not listed"."""
if not self.enabled:
return None
try:
from ddgs import DDGS
except ImportError:
logger.warning("ddgs is not installed; DuckDuckGo search unavailable")
return None
for backend in self.BACKENDS:
self._pacer.wait()
try:
with DDGS(timeout=20) as ddgs:
rows = list(ddgs.text(query, region=SEARCH_REGION, safesearch="moderate",
max_results=max_results, backend=backend) or [])
except Exception as exc: # noqa: BLE001 - ddgs raises many types on throttling
logger.info("DuckDuckGo(%s) text search gave no answer (%s): %s", backend, query, exc)
continue
hits = [
SearchHit(r.get("href") or r.get("url") or "", r.get("title") or "", r.get("body") or "",
f"{self.name}", i)
for i, r in enumerate(rows)
if (r.get("href") or r.get("url") or "").startswith("http")
]
if hits:
return hits
return None
def images(self, query: str, max_results: int = 15) -> Optional[List[SearchHit]]:
if not self.enabled:
return None
try:
from ddgs import DDGS
except ImportError:
return None
self._pacer.wait()
try:
with DDGS(timeout=20) as ddgs:
rows = list(ddgs.images(query, region=SEARCH_REGION, safesearch="moderate",
max_results=max_results) or [])
except Exception as exc: # noqa: BLE001
if "no results" in str(exc).lower():
return []
logger.info("DuckDuckGo image search failed (%s): %s", query, exc)
return None
return [
SearchHit(r.get("url") or "", r.get("title") or "", "", self.name, i, image_url=r.get("image"))
for i, r in enumerate(rows)
if str(r.get("image") or "").startswith("http") and str(r.get("url") or "").startswith("http")
]
class GoogleCseProvider:
name = "google"
ENDPOINT = "https://www.googleapis.com/customsearch/v1"
def __init__(self, quota_left: Callable[[], int] = lambda: 100) -> None:
self.enabled = USE_GOOGLE_CSE
self.error: Optional[str] = None
self._quota_left = quota_left
self._pacer = _Pacer(1.0)
def _call(self, query: str, extra: dict) -> Optional[List[dict]]:
if not self.enabled:
return None
if self._quota_left() <= 0:
self.error = "daily query quota used up"
return None
self._pacer.wait()
try:
resp = requests.get(
self.ENDPOINT,
params={"key": GOOGLE_API_KEY, "cx": GOOGLE_CSE_ID, "q": query, "gl": "in", "num": 10, **extra},
timeout=20,
)
except requests.RequestException as exc:
logger.info("Google CSE failed: %s", exc)
return None
if resp.status_code in (400, 401, 403):
# A key/project problem will not fix itself mid-run: stop asking.
try:
message = resp.json().get("error", {}).get("message", "")
except ValueError:
message = resp.text[:200]
self.enabled = False
self.error = f"HTTP {resp.status_code}: {message}"
logger.warning("Google Programmable Search disabled for this run - %s", self.error)
return None
if resp.status_code != 200:
logger.info("Google CSE HTTP %s: %s", resp.status_code, resp.text[:200])
return None
return resp.json().get("items", []) or []
def text(self, query: str, max_results: int = 10) -> Optional[List[SearchHit]]:
items = self._call(query, {})
if items is None:
return None
return [SearchHit(i.get("link", ""), i.get("title", ""), i.get("snippet", ""), self.name, n,
offer=pagemap_offer(i.get("pagemap") or {}),
rating=pagemap_rating(i.get("pagemap") or {}))
for n, i in enumerate(items[:max_results]) if i.get("link")]
def images(self, query: str, max_results: int = 10) -> Optional[List[SearchHit]]:
items = self._call(query, {"searchType": "image"})
if items is None:
return None
return [SearchHit((i.get("image") or {}).get("contextLink", ""), i.get("title", ""), "", self.name, n,
image_url=i.get("link"))
for n, i in enumerate(items[:max_results]) if i.get("link")]
def pagemap_offer(pagemap: dict) -> Optional[dict]:
"""The offer Google extracted from the page's own structured data
(schema.org Offer, or product:price meta tags), if any. INR only."""
candidates = []
for offer in pagemap.get("offer") or []:
candidates.append((offer.get("price"), offer.get("pricecurrency"), offer.get("availability"), offer))
for meta in pagemap.get("metatags") or []:
price = meta.get("product:price:amount") or meta.get("og:price:amount")
if price:
candidates.append((price, meta.get("product:price:currency") or meta.get("og:price:currency"),
meta.get("product:availability") or meta.get("og:availability"),
{k: v for k, v in meta.items() if "price" in k or "availability" in k}))
for price, currency, availability, raw in candidates:
if price and (currency or "").upper() == "INR":
return {"price": str(price), "currency": "INR", "availability": availability, "raw": raw}
return None
def pagemap_rating(pagemap: dict) -> Optional[dict]:
"""The aggregate rating Google extracted from the page's own structured
data (schema.org AggregateRating), if any. Only a value on a 5-point
scale is accepted."""
for node in pagemap.get("aggregaterating") or []:
try:
value = float(str(node.get("ratingvalue", "")).replace(",", "."))
except ValueError:
continue
best = node.get("bestrating")
try:
if best not in (None, "") and float(best) != 5:
continue
except ValueError:
continue
if not 0 < value <= 5:
continue
count = None
for key in ("reviewcount", "ratingcount"):
digits = "".join(ch for ch in str(node.get(key) or "") if ch.isdigit())
if digits:
count = int(digits)
break
return {"rating": round(value, 2), "review_count": count,
"raw": {k: v for k, v in node.items() if k in ("ratingvalue", "reviewcount", "ratingcount", "bestrating")}}
return None