Files
loyaly-catalogue/backend/app/electronics/search/providers.py
sriram c7e4d59188 Electronics Catalog: API, MCP server, frontend and deployment
Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
2026-10-01 12:17:42 +05:30

235 lines
9.3 KiB
Python

"""Web search: DuckDuckGo (ddgs, no key) and, when configured, Google
Programmable Search.
Three outcomes, never collapsed: a list of hits, an empty list ("we asked and
nothing matched"), or None ("we could not ask" - throttled, offline, no
quota). A throttle is not evidence that a product is not sold anywhere.
"""
from __future__ import annotations
import logging
import threading
import time
from dataclasses import asdict, dataclass
from typing import Callable, List, Optional
import requests
from app.infrastructure.settings import (
GOOGLE_API_KEY,
GOOGLE_CSE_ID,
SEARCH_MIN_INTERVAL_SECONDS,
SEARCH_REGION,
USE_DDG_SEARCH,
USE_GOOGLE_CSE,
)
logger = logging.getLogger(__name__)
@dataclass
class SearchHit:
url: str
title: str
snippet: str
provider: str
rank: int
image_url: Optional[str] = None # image searches: the image itself (url = page it is on)
# Structured offer data the search engine itself extracted from the page
# (Google CSE "pagemap"): {"price", "currency", "availability", "raw"}.
offer: Optional[dict] = None
# Aggregate rating the search engine extracted from the page's own
# structured data (Google CSE "pagemap"): {"rating", "review_count", "raw"}.
rating: Optional[dict] = None
def to_dict(self) -> dict:
return asdict(self)
@classmethod
def from_dict(cls, d: dict) -> "SearchHit":
return cls(**{k: d.get(k) for k in ("url", "title", "snippet", "provider", "rank", "image_url", "offer", "rating")})
class _Pacer:
def __init__(self, interval: float, sleep: Callable[[float], None] = time.sleep) -> None:
self.interval = interval
self._sleep = sleep
self._last = 0.0
self._lock = threading.Lock()
def wait(self) -> None:
with self._lock:
gap = self.interval - (time.monotonic() - self._last)
if gap > 0:
self._sleep(gap)
self._last = time.monotonic()
class DuckDuckGoProvider:
name = "ddg"
def __init__(self, interval: float = SEARCH_MIN_INTERVAL_SECONDS) -> None:
self._pacer = _Pacer(interval)
self.enabled = USE_DDG_SEARCH
# "auto" rotates ddgs's engines; yahoo is a second opinion when it is throttled.
BACKENDS = ("auto", "yahoo")
def text(self, query: str, max_results: int = 20) -> Optional[List[SearchHit]]:
"""Hits, or None when no backend answered. ddgs reports a throttle and
a genuinely empty result the same way ("No results found"), so an
empty answer is treated as unknown rather than as "not listed"."""
if not self.enabled:
return None
try:
from ddgs import DDGS
except ImportError:
logger.warning("ddgs is not installed; DuckDuckGo search unavailable")
return None
for backend in self.BACKENDS:
self._pacer.wait()
try:
with DDGS(timeout=20) as ddgs:
rows = list(ddgs.text(query, region=SEARCH_REGION, safesearch="moderate",
max_results=max_results, backend=backend) or [])
except Exception as exc: # noqa: BLE001 - ddgs raises many types on throttling
logger.info("DuckDuckGo(%s) text search gave no answer (%s): %s", backend, query, exc)
continue
hits = [
SearchHit(r.get("href") or r.get("url") or "", r.get("title") or "", r.get("body") or "",
f"{self.name}", i)
for i, r in enumerate(rows)
if (r.get("href") or r.get("url") or "").startswith("http")
]
if hits:
return hits
return None
def images(self, query: str, max_results: int = 15) -> Optional[List[SearchHit]]:
if not self.enabled:
return None
try:
from ddgs import DDGS
except ImportError:
return None
self._pacer.wait()
try:
with DDGS(timeout=20) as ddgs:
rows = list(ddgs.images(query, region=SEARCH_REGION, safesearch="moderate",
max_results=max_results) or [])
except Exception as exc: # noqa: BLE001
if "no results" in str(exc).lower():
return []
logger.info("DuckDuckGo image search failed (%s): %s", query, exc)
return None
return [
SearchHit(r.get("url") or "", r.get("title") or "", "", self.name, i, image_url=r.get("image"))
for i, r in enumerate(rows)
if str(r.get("image") or "").startswith("http") and str(r.get("url") or "").startswith("http")
]
class GoogleCseProvider:
name = "google"
ENDPOINT = "https://www.googleapis.com/customsearch/v1"
def __init__(self, quota_left: Callable[[], int] = lambda: 100) -> None:
self.enabled = USE_GOOGLE_CSE
self.error: Optional[str] = None
self._quota_left = quota_left
self._pacer = _Pacer(1.0)
def _call(self, query: str, extra: dict) -> Optional[List[dict]]:
if not self.enabled:
return None
if self._quota_left() <= 0:
self.error = "daily query quota used up"
return None
self._pacer.wait()
try:
resp = requests.get(
self.ENDPOINT,
params={"key": GOOGLE_API_KEY, "cx": GOOGLE_CSE_ID, "q": query, "gl": "in", "num": 10, **extra},
timeout=20,
)
except requests.RequestException as exc:
logger.info("Google CSE failed: %s", exc)
return None
if resp.status_code in (400, 401, 403):
# A key/project problem will not fix itself mid-run: stop asking.
try:
message = resp.json().get("error", {}).get("message", "")
except ValueError:
message = resp.text[:200]
self.enabled = False
self.error = f"HTTP {resp.status_code}: {message}"
logger.warning("Google Programmable Search disabled for this run - %s", self.error)
return None
if resp.status_code != 200:
logger.info("Google CSE HTTP %s: %s", resp.status_code, resp.text[:200])
return None
return resp.json().get("items", []) or []
def text(self, query: str, max_results: int = 10) -> Optional[List[SearchHit]]:
items = self._call(query, {})
if items is None:
return None
return [SearchHit(i.get("link", ""), i.get("title", ""), i.get("snippet", ""), self.name, n,
offer=pagemap_offer(i.get("pagemap") or {}),
rating=pagemap_rating(i.get("pagemap") or {}))
for n, i in enumerate(items[:max_results]) if i.get("link")]
def images(self, query: str, max_results: int = 10) -> Optional[List[SearchHit]]:
items = self._call(query, {"searchType": "image"})
if items is None:
return None
return [SearchHit((i.get("image") or {}).get("contextLink", ""), i.get("title", ""), "", self.name, n,
image_url=i.get("link"))
for n, i in enumerate(items[:max_results]) if i.get("link")]
def pagemap_offer(pagemap: dict) -> Optional[dict]:
"""The offer Google extracted from the page's own structured data
(schema.org Offer, or product:price meta tags), if any. INR only."""
candidates = []
for offer in pagemap.get("offer") or []:
candidates.append((offer.get("price"), offer.get("pricecurrency"), offer.get("availability"), offer))
for meta in pagemap.get("metatags") or []:
price = meta.get("product:price:amount") or meta.get("og:price:amount")
if price:
candidates.append((price, meta.get("product:price:currency") or meta.get("og:price:currency"),
meta.get("product:availability") or meta.get("og:availability"),
{k: v for k, v in meta.items() if "price" in k or "availability" in k}))
for price, currency, availability, raw in candidates:
if price and (currency or "").upper() == "INR":
return {"price": str(price), "currency": "INR", "availability": availability, "raw": raw}
return None
def pagemap_rating(pagemap: dict) -> Optional[dict]:
"""The aggregate rating Google extracted from the page's own structured
data (schema.org AggregateRating), if any. Only a value on a 5-point
scale is accepted."""
for node in pagemap.get("aggregaterating") or []:
try:
value = float(str(node.get("ratingvalue", "")).replace(",", "."))
except ValueError:
continue
best = node.get("bestrating")
try:
if best not in (None, "") and float(best) != 5:
continue
except ValueError:
continue
if not 0 < value <= 5:
continue
count = None
for key in ("reviewcount", "ratingcount"):
digits = "".join(ch for ch in str(node.get(key) or "") if ch.isdigit())
if digits:
count = int(digits)
break
return {"rating": round(value, 2), "review_count": count,
"raw": {k: v for k, v in node.items() if k in ("ratingvalue", "reviewcount", "ratingcount", "bestrating")}}
return None