Verified catalogue of mobiles and laptops sold in India, collected from real retail listings (FastAPI backend, React frontend, Postgres/pgvector). - REST API under /api/elec (read-only catalogue; admin endpoints need login) - MCP server (FastMCP) at /mcp/ with list_categories, search_products, get_product and price_history tools - Real ratings and reviews read from product pages and search results - Production Dockerfile (requirements-api.txt, no PyTorch) and .env.production.example; remote database only via an explicit ELEC_ALLOW_REMOTE_DB host/name allowlist - docs/API.md: endpoint and MCP reference with live examples Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
235 lines
9.3 KiB
Python
235 lines
9.3 KiB
Python
"""Web search: DuckDuckGo (ddgs, no key) and, when configured, Google
|
|
Programmable Search.
|
|
|
|
Three outcomes, never collapsed: a list of hits, an empty list ("we asked and
|
|
nothing matched"), or None ("we could not ask" - throttled, offline, no
|
|
quota). A throttle is not evidence that a product is not sold anywhere.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import threading
|
|
import time
|
|
from dataclasses import asdict, dataclass
|
|
from typing import Callable, List, Optional
|
|
|
|
import requests
|
|
|
|
from app.infrastructure.settings import (
|
|
GOOGLE_API_KEY,
|
|
GOOGLE_CSE_ID,
|
|
SEARCH_MIN_INTERVAL_SECONDS,
|
|
SEARCH_REGION,
|
|
USE_DDG_SEARCH,
|
|
USE_GOOGLE_CSE,
|
|
)
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
@dataclass
|
|
class SearchHit:
|
|
url: str
|
|
title: str
|
|
snippet: str
|
|
provider: str
|
|
rank: int
|
|
image_url: Optional[str] = None # image searches: the image itself (url = page it is on)
|
|
# Structured offer data the search engine itself extracted from the page
|
|
# (Google CSE "pagemap"): {"price", "currency", "availability", "raw"}.
|
|
offer: Optional[dict] = None
|
|
# Aggregate rating the search engine extracted from the page's own
|
|
# structured data (Google CSE "pagemap"): {"rating", "review_count", "raw"}.
|
|
rating: Optional[dict] = None
|
|
|
|
def to_dict(self) -> dict:
|
|
return asdict(self)
|
|
|
|
@classmethod
|
|
def from_dict(cls, d: dict) -> "SearchHit":
|
|
return cls(**{k: d.get(k) for k in ("url", "title", "snippet", "provider", "rank", "image_url", "offer", "rating")})
|
|
|
|
|
|
class _Pacer:
|
|
def __init__(self, interval: float, sleep: Callable[[float], None] = time.sleep) -> None:
|
|
self.interval = interval
|
|
self._sleep = sleep
|
|
self._last = 0.0
|
|
self._lock = threading.Lock()
|
|
|
|
def wait(self) -> None:
|
|
with self._lock:
|
|
gap = self.interval - (time.monotonic() - self._last)
|
|
if gap > 0:
|
|
self._sleep(gap)
|
|
self._last = time.monotonic()
|
|
|
|
|
|
class DuckDuckGoProvider:
|
|
name = "ddg"
|
|
|
|
def __init__(self, interval: float = SEARCH_MIN_INTERVAL_SECONDS) -> None:
|
|
self._pacer = _Pacer(interval)
|
|
self.enabled = USE_DDG_SEARCH
|
|
|
|
# "auto" rotates ddgs's engines; yahoo is a second opinion when it is throttled.
|
|
BACKENDS = ("auto", "yahoo")
|
|
|
|
def text(self, query: str, max_results: int = 20) -> Optional[List[SearchHit]]:
|
|
"""Hits, or None when no backend answered. ddgs reports a throttle and
|
|
a genuinely empty result the same way ("No results found"), so an
|
|
empty answer is treated as unknown rather than as "not listed"."""
|
|
if not self.enabled:
|
|
return None
|
|
try:
|
|
from ddgs import DDGS
|
|
except ImportError:
|
|
logger.warning("ddgs is not installed; DuckDuckGo search unavailable")
|
|
return None
|
|
for backend in self.BACKENDS:
|
|
self._pacer.wait()
|
|
try:
|
|
with DDGS(timeout=20) as ddgs:
|
|
rows = list(ddgs.text(query, region=SEARCH_REGION, safesearch="moderate",
|
|
max_results=max_results, backend=backend) or [])
|
|
except Exception as exc: # noqa: BLE001 - ddgs raises many types on throttling
|
|
logger.info("DuckDuckGo(%s) text search gave no answer (%s): %s", backend, query, exc)
|
|
continue
|
|
hits = [
|
|
SearchHit(r.get("href") or r.get("url") or "", r.get("title") or "", r.get("body") or "",
|
|
f"{self.name}", i)
|
|
for i, r in enumerate(rows)
|
|
if (r.get("href") or r.get("url") or "").startswith("http")
|
|
]
|
|
if hits:
|
|
return hits
|
|
return None
|
|
|
|
def images(self, query: str, max_results: int = 15) -> Optional[List[SearchHit]]:
|
|
if not self.enabled:
|
|
return None
|
|
try:
|
|
from ddgs import DDGS
|
|
except ImportError:
|
|
return None
|
|
self._pacer.wait()
|
|
try:
|
|
with DDGS(timeout=20) as ddgs:
|
|
rows = list(ddgs.images(query, region=SEARCH_REGION, safesearch="moderate",
|
|
max_results=max_results) or [])
|
|
except Exception as exc: # noqa: BLE001
|
|
if "no results" in str(exc).lower():
|
|
return []
|
|
logger.info("DuckDuckGo image search failed (%s): %s", query, exc)
|
|
return None
|
|
return [
|
|
SearchHit(r.get("url") or "", r.get("title") or "", "", self.name, i, image_url=r.get("image"))
|
|
for i, r in enumerate(rows)
|
|
if str(r.get("image") or "").startswith("http") and str(r.get("url") or "").startswith("http")
|
|
]
|
|
|
|
|
|
class GoogleCseProvider:
|
|
name = "google"
|
|
ENDPOINT = "https://www.googleapis.com/customsearch/v1"
|
|
|
|
def __init__(self, quota_left: Callable[[], int] = lambda: 100) -> None:
|
|
self.enabled = USE_GOOGLE_CSE
|
|
self.error: Optional[str] = None
|
|
self._quota_left = quota_left
|
|
self._pacer = _Pacer(1.0)
|
|
|
|
def _call(self, query: str, extra: dict) -> Optional[List[dict]]:
|
|
if not self.enabled:
|
|
return None
|
|
if self._quota_left() <= 0:
|
|
self.error = "daily query quota used up"
|
|
return None
|
|
self._pacer.wait()
|
|
try:
|
|
resp = requests.get(
|
|
self.ENDPOINT,
|
|
params={"key": GOOGLE_API_KEY, "cx": GOOGLE_CSE_ID, "q": query, "gl": "in", "num": 10, **extra},
|
|
timeout=20,
|
|
)
|
|
except requests.RequestException as exc:
|
|
logger.info("Google CSE failed: %s", exc)
|
|
return None
|
|
if resp.status_code in (400, 401, 403):
|
|
# A key/project problem will not fix itself mid-run: stop asking.
|
|
try:
|
|
message = resp.json().get("error", {}).get("message", "")
|
|
except ValueError:
|
|
message = resp.text[:200]
|
|
self.enabled = False
|
|
self.error = f"HTTP {resp.status_code}: {message}"
|
|
logger.warning("Google Programmable Search disabled for this run - %s", self.error)
|
|
return None
|
|
if resp.status_code != 200:
|
|
logger.info("Google CSE HTTP %s: %s", resp.status_code, resp.text[:200])
|
|
return None
|
|
return resp.json().get("items", []) or []
|
|
|
|
def text(self, query: str, max_results: int = 10) -> Optional[List[SearchHit]]:
|
|
items = self._call(query, {})
|
|
if items is None:
|
|
return None
|
|
return [SearchHit(i.get("link", ""), i.get("title", ""), i.get("snippet", ""), self.name, n,
|
|
offer=pagemap_offer(i.get("pagemap") or {}),
|
|
rating=pagemap_rating(i.get("pagemap") or {}))
|
|
for n, i in enumerate(items[:max_results]) if i.get("link")]
|
|
|
|
def images(self, query: str, max_results: int = 10) -> Optional[List[SearchHit]]:
|
|
items = self._call(query, {"searchType": "image"})
|
|
if items is None:
|
|
return None
|
|
return [SearchHit((i.get("image") or {}).get("contextLink", ""), i.get("title", ""), "", self.name, n,
|
|
image_url=i.get("link"))
|
|
for n, i in enumerate(items[:max_results]) if i.get("link")]
|
|
|
|
|
|
def pagemap_offer(pagemap: dict) -> Optional[dict]:
|
|
"""The offer Google extracted from the page's own structured data
|
|
(schema.org Offer, or product:price meta tags), if any. INR only."""
|
|
candidates = []
|
|
for offer in pagemap.get("offer") or []:
|
|
candidates.append((offer.get("price"), offer.get("pricecurrency"), offer.get("availability"), offer))
|
|
for meta in pagemap.get("metatags") or []:
|
|
price = meta.get("product:price:amount") or meta.get("og:price:amount")
|
|
if price:
|
|
candidates.append((price, meta.get("product:price:currency") or meta.get("og:price:currency"),
|
|
meta.get("product:availability") or meta.get("og:availability"),
|
|
{k: v for k, v in meta.items() if "price" in k or "availability" in k}))
|
|
for price, currency, availability, raw in candidates:
|
|
if price and (currency or "").upper() == "INR":
|
|
return {"price": str(price), "currency": "INR", "availability": availability, "raw": raw}
|
|
return None
|
|
|
|
|
|
def pagemap_rating(pagemap: dict) -> Optional[dict]:
|
|
"""The aggregate rating Google extracted from the page's own structured
|
|
data (schema.org AggregateRating), if any. Only a value on a 5-point
|
|
scale is accepted."""
|
|
for node in pagemap.get("aggregaterating") or []:
|
|
try:
|
|
value = float(str(node.get("ratingvalue", "")).replace(",", "."))
|
|
except ValueError:
|
|
continue
|
|
best = node.get("bestrating")
|
|
try:
|
|
if best not in (None, "") and float(best) != 5:
|
|
continue
|
|
except ValueError:
|
|
continue
|
|
if not 0 < value <= 5:
|
|
continue
|
|
count = None
|
|
for key in ("reviewcount", "ratingcount"):
|
|
digits = "".join(ch for ch in str(node.get(key) or "") if ch.isdigit())
|
|
if digits:
|
|
count = int(digits)
|
|
break
|
|
return {"rating": round(value, 2), "review_count": count,
|
|
"raw": {k: v for k, v in node.items() if k in ("ratingvalue", "reviewcount", "ratingcount", "bestrating")}}
|
|
return None
|