Verified catalogue of mobiles and laptops sold in India, collected from real retail listings (FastAPI backend, React frontend, Postgres/pgvector). - REST API under /api/elec (read-only catalogue; admin endpoints need login) - MCP server (FastMCP) at /mcp/ with list_categories, search_products, get_product and price_history tools - Real ratings and reviews read from product pages and search results - Production Dockerfile (requirements-api.txt, no PyTorch) and .env.production.example; remote database only via an explicit ELEC_ALLOW_REMOTE_DB host/name allowlist - docs/API.md: endpoint and MCP reference with live examples Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
66 lines
2.9 KiB
Python
66 lines
2.9 KiB
Python
"""Cached, budgeted access to the search providers.
|
|
|
|
Results are cached in elec.search_cache so a re-run does not query again
|
|
within SEARCH_CACHE_TTL_HOURS, and each run has a query budget so a large
|
|
brand list cannot hammer the providers.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
from typing import Dict, List, Optional
|
|
|
|
from app.electronics.db import repository as repo
|
|
from app.electronics.search.providers import DuckDuckGoProvider, GoogleCseProvider, SearchHit
|
|
from app.infrastructure.settings import GOOGLE_CSE_DAILY_QUOTA, SEARCH_CACHE_TTL_HOURS
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
class SearchEngine:
|
|
def __init__(self, *, budget: int = 200, use_cache: bool = True) -> None:
|
|
self.budget = budget
|
|
self.use_cache = use_cache
|
|
self.used = 0
|
|
self.stats: Dict[str, int] = {"cache_hits": 0, "queries": 0, "unavailable": 0}
|
|
self.ddg = DuckDuckGoProvider()
|
|
self.google = GoogleCseProvider(
|
|
quota_left=lambda: GOOGLE_CSE_DAILY_QUOTA - repo.google_queries_today()
|
|
)
|
|
|
|
def _ask(self, provider, kind: str, query: str, max_results: int) -> Optional[List[SearchHit]]:
|
|
if not provider.enabled:
|
|
return None
|
|
cached = repo.search_cache_get(provider.name, kind, query, SEARCH_CACHE_TTL_HOURS) if self.use_cache else None
|
|
if cached is not None:
|
|
self.stats["cache_hits"] += 1
|
|
return [SearchHit.from_dict(d) for d in cached]
|
|
if self.used >= self.budget:
|
|
logger.info("Search budget (%d) spent; skipping %r", self.budget, query)
|
|
return None
|
|
self.used += 1
|
|
self.stats["queries"] += 1
|
|
self.stats[f"queries_{provider.name}"] = self.stats.get(f"queries_{provider.name}", 0) + 1
|
|
hits = provider.text(query, max_results) if kind == "text" else provider.images(query, max_results)
|
|
if hits is None:
|
|
self.stats["unavailable"] += 1
|
|
return None
|
|
repo.search_cache_put(provider.name, kind, query, [h.to_dict() for h in hits])
|
|
return hits
|
|
|
|
def _run(self, kind: str, query: str, max_results: int, providers: str) -> Optional[List[SearchHit]]:
|
|
"""providers: "default" = DuckDuckGo, with Google only when DuckDuckGo
|
|
gives no answer (keeps the 100/day Google quota for price lookups);
|
|
"google" = Google only."""
|
|
if providers == "google":
|
|
return self._ask(self.google, kind, query, max_results)
|
|
hits = self._ask(self.ddg, kind, query, max_results)
|
|
if hits is None:
|
|
hits = self._ask(self.google, kind, query, max_results)
|
|
return hits
|
|
|
|
def text(self, query: str, max_results: int = 20, *, providers: str = "default") -> Optional[List[SearchHit]]:
|
|
return self._run("text", query, max_results, providers)
|
|
|
|
def images(self, query: str, max_results: int = 15) -> Optional[List[SearchHit]]:
|
|
return self._run("images", query, max_results, "default")
|