Electronics Catalog: API, MCP server, frontend and deployment

Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
sriram
2026-10-01 12:17:42 +05:30
commit c7e4d59188
115 changed files with 14329 additions and 0 deletions

View File

View File

@@ -0,0 +1,61 @@
"""Per-host circuit breaker.
One 403, 429, 503 or CAPTCHA page opens the breaker for that host for
ELEC_BREAKER_COOLDOWN_HOURS. While it is open the host is not requested at all
and its products are collected from web search results instead. There is no
retry-with-a-different-identity: a block is an answer.
"""
from __future__ import annotations
import threading
import time
from typing import Callable, Dict, Optional, Tuple
from app.infrastructure.settings import ELEC_BREAKER_COOLDOWN_HOURS
class CircuitBreaker:
def __init__(
self,
cooldown_seconds: float = ELEC_BREAKER_COOLDOWN_HOURS * 3600,
on_trip: Optional[Callable[[str, str, float], None]] = None,
clock: Callable[[], float] = time.time,
) -> None:
self.cooldown = cooldown_seconds
self.on_trip = on_trip
self._clock = clock
self._open: Dict[str, Tuple[float, str]] = {}
self._lock = threading.Lock()
@staticmethod
def _key(host: str) -> str:
host = host.lower()
return host[4:] if host.startswith("www.") else host
def preload(self, host: str, until_epoch: float, reason: str) -> None:
"""Restore a breaker that was opened in an earlier run (elec.site)."""
if until_epoch > self._clock():
with self._lock:
self._open[self._key(host)] = (until_epoch, reason)
def trip(self, host: str, reason: str) -> None:
until = self._clock() + self.cooldown
with self._lock:
self._open[self._key(host)] = (until, reason)
if self.on_trip:
self.on_trip(self._key(host), reason, until)
def is_open(self, host: str) -> bool:
key = self._key(host)
with self._lock:
entry = self._open.get(key)
if not entry:
return False
if entry[0] <= self._clock():
del self._open[key]
return False
return True
def reason(self, host: str) -> Optional[str]:
entry = self._open.get(self._key(host))
return entry[1] if entry else None

View File

@@ -0,0 +1,223 @@
"""The only way this project fetches a retail or brand web page.
What it guarantees, for every request:
* robots.txt is consulted first (protego). If robots.txt cannot be read
because the server errors or blocks it, the site is treated as disallowed.
* at least ELEC_SITE_MIN_INTERVAL_SECONDS between requests to one host.
* an honest User-Agent naming the project and a contact address.
* no JavaScript, no cookies kept between requests, no proxies, no retries on
403/429 - a block is respected, not worked around.
* a size cap on the response body.
* a circuit breaker: a 403/429/CAPTCHA response opens it for the host, and
every later request to that host is refused until the cooldown passes.
* every request is reported to `on_fetch` (the fetch_log table).
"""
from __future__ import annotations
import logging
import re
import threading
import time
from dataclasses import dataclass
from typing import Callable, Dict, Optional, Tuple
from urllib.parse import urlparse
import httpx
from protego import Protego
from app.electronics.net.breaker import CircuitBreaker
from app.infrastructure.settings import (
ELEC_MAX_PAGE_BYTES,
ELEC_SITE_MIN_INTERVAL_SECONDS,
REQUEST_TIMEOUT_SECONDS,
USER_AGENT,
)
logger = logging.getLogger(__name__)
ROBOTS_TTL_SECONDS = 24 * 3600
# Pages that are a bot check rather than content. Matched on the first 20 KB.
_CAPTCHA_MARKERS = re.compile(
r"captcha|robot\s*check|are\s+you\s+a\s+robot|verify\s+you\s+are\s+human|"
r"/errors/validatecaptcha|px-captcha|cf-challenge|challenge-platform|access\s+denied|"
r"unusual\s+traffic|request\s+blocked|bot\s+detection|akamai.*reference",
re.IGNORECASE,
)
@dataclass
class FetchResult:
url: str
final_url: str
status: Optional[int]
text: str
outcome: str # ok | robots_disallowed | blocked | captcha | breaker_open | http_error | network_error | too_large | not_html
robots_allowed: Optional[bool]
bytes: int = 0
@property
def ok(self) -> bool:
return self.outcome == "ok"
class PoliteClient:
def __init__(
self,
*,
min_interval: float = ELEC_SITE_MIN_INTERVAL_SECONDS,
breaker: Optional[CircuitBreaker] = None,
on_fetch: Optional[Callable[[FetchResult, str], None]] = None,
transport: Optional[httpx.BaseTransport] = None,
sleep: Callable[[float], None] = time.sleep,
clock: Callable[[], float] = time.monotonic,
) -> None:
self.min_interval = min_interval
self.breaker = breaker or CircuitBreaker()
self.on_fetch = on_fetch
self._sleep = sleep
self._clock = clock
self._last: Dict[str, float] = {}
self._locks: Dict[str, threading.Lock] = {}
self._robots: Dict[str, Tuple[float, Optional[Protego], bool]] = {}
self._guard = threading.Lock()
self._client = httpx.Client(
headers={
"User-Agent": USER_AGENT,
"Accept": "text/html,application/xhtml+xml,application/json;q=0.9,*/*;q=0.5",
"Accept-Language": "en-IN,en;q=0.9",
},
follow_redirects=True,
timeout=REQUEST_TIMEOUT_SECONDS,
transport=transport,
)
def close(self) -> None:
self._client.close()
def __enter__(self) -> "PoliteClient":
return self
def __exit__(self, *exc) -> None:
self.close()
# -- pacing --------------------------------------------------------------
def _host_lock(self, host: str) -> threading.Lock:
with self._guard:
return self._locks.setdefault(host, threading.Lock())
def _wait_turn(self, host: str) -> None:
last = self._last.get(host)
if last is not None:
gap = self.min_interval - (self._clock() - last)
if gap > 0:
self._sleep(gap)
self._last[host] = self._clock()
# -- robots.txt ----------------------------------------------------------
def _robots_for(self, scheme: str, host: str) -> Tuple[Optional[Protego], bool]:
"""(parser, reachable). parser None + reachable True = no robots.txt
(everything allowed); reachable False = could not read it (deny)."""
cached = self._robots.get(host)
if cached and self._clock() - cached[0] < ROBOTS_TTL_SECONDS:
return cached[1], cached[2]
url = f"{scheme}://{host}/robots.txt"
parser: Optional[Protego] = None
reachable = False
self._wait_turn(host)
try:
resp = self._client.get(url)
if resp.status_code == 200:
parser, reachable = Protego.parse(resp.text), True
elif resp.status_code in (404, 410):
parser, reachable = None, True
else:
reachable = False
if resp.status_code in (403, 429):
self.breaker.trip(host, f"robots.txt returned HTTP {resp.status_code}")
except httpx.HTTPError as exc:
logger.info("robots.txt unreachable for %s: %s", host, exc)
self._robots[host] = (self._clock(), parser, reachable)
return parser, reachable
def robots_allowed(self, url: str) -> bool:
p = urlparse(url)
parser, reachable = self._robots_for(p.scheme or "https", p.netloc.lower())
if not reachable:
return False
return True if parser is None else bool(parser.can_fetch(url, USER_AGENT))
# -- fetch ---------------------------------------------------------------
def _report(self, result: FetchResult) -> FetchResult:
if self.on_fetch:
try:
self.on_fetch(result, urlparse(result.url).netloc.lower())
except Exception as exc: # noqa: BLE001 - logging must never break a crawl
logger.debug("fetch log failed: %s", exc)
return result
def get(self, url: str, *, check_robots: bool = True, accept_non_html: bool = False) -> FetchResult:
host = urlparse(url).netloc.lower()
if self.breaker.is_open(host):
return FetchResult(url, url, None, "", "breaker_open", None)
with self._host_lock(host):
allowed: Optional[bool] = None
if check_robots:
allowed = self.robots_allowed(url)
if not allowed:
return self._report(FetchResult(url, url, None, "", "robots_disallowed", False))
self._wait_turn(host)
try:
with self._client.stream("GET", url) as resp:
status = resp.status_code
final = str(resp.url)
ctype = resp.headers.get("content-type", "").lower()
body = bytearray()
too_large = False
for chunk in resp.iter_bytes():
body.extend(chunk)
if len(body) > ELEC_MAX_PAGE_BYTES:
too_large = True
break
encoding = resp.encoding or "utf-8"
except httpx.HTTPError as exc:
logger.info("fetch failed %s: %s", url, exc)
return self._report(FetchResult(url, url, None, "", "network_error", allowed))
text = bytes(body).decode(encoding, errors="replace") if body else ""
n = len(body)
if status in (403, 429, 503) or (status == 200 and _CAPTCHA_MARKERS.search(text[:20000]) and len(text) < 60000):
outcome = "captcha" if status == 200 or _CAPTCHA_MARKERS.search(text[:20000]) else "blocked"
self.breaker.trip(host, f"HTTP {status} ({outcome})")
return self._report(FetchResult(url, final, status, "", outcome, allowed, n))
if status != 200:
return self._report(FetchResult(url, final, status, "", "http_error", allowed, n))
if too_large:
return self._report(FetchResult(url, final, status, "", "too_large", allowed, n))
if not accept_non_html and "html" not in ctype and "json" not in ctype:
return self._report(FetchResult(url, final, status, "", "not_html", allowed, n))
return self._report(FetchResult(url, final, status, text, "ok", allowed, n))
def check_image(self, url: str, min_bytes: int) -> bool:
"""One ranged GET to confirm a URL serves a real image. Paced per host
like any request; robots.txt is not consulted because this fetches a
single file the product page itself references, as a browser would."""
host = urlparse(url).netloc.lower()
if not url.startswith(("http://", "https://")) or self.breaker.is_open(host):
return False
with self._host_lock(host):
self._wait_turn(host)
try:
with self._client.stream("GET", url, headers={"Accept": "image/*", "Range": f"bytes=0-{min_bytes * 4}"}) as resp:
if resp.status_code not in (200, 206):
return False
if not resp.headers.get("content-type", "").lower().startswith("image/"):
return False
got = 0
for chunk in resp.iter_bytes():
got += len(chunk)
if got >= min_bytes:
return True
return got >= min_bytes
except httpx.HTTPError:
return False