Electronics Catalog: API, MCP server, frontend and deployment
Verified catalogue of mobiles and laptops sold in India, collected from real retail listings (FastAPI backend, React frontend, Postgres/pgvector). - REST API under /api/elec (read-only catalogue; admin endpoints need login) - MCP server (FastMCP) at /mcp/ with list_categories, search_products, get_product and price_history tools - Real ratings and reviews read from product pages and search results - Production Dockerfile (requirements-api.txt, no PyTorch) and .env.production.example; remote database only via an explicit ELEC_ALLOW_REMOTE_DB host/name allowlist - docs/API.md: endpoint and MCP reference with live examples Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
0
backend/app/electronics/net/__init__.py
Normal file
0
backend/app/electronics/net/__init__.py
Normal file
61
backend/app/electronics/net/breaker.py
Normal file
61
backend/app/electronics/net/breaker.py
Normal file
@@ -0,0 +1,61 @@
|
||||
"""Per-host circuit breaker.
|
||||
|
||||
One 403, 429, 503 or CAPTCHA page opens the breaker for that host for
|
||||
ELEC_BREAKER_COOLDOWN_HOURS. While it is open the host is not requested at all
|
||||
and its products are collected from web search results instead. There is no
|
||||
retry-with-a-different-identity: a block is an answer.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
import time
|
||||
from typing import Callable, Dict, Optional, Tuple
|
||||
|
||||
from app.infrastructure.settings import ELEC_BREAKER_COOLDOWN_HOURS
|
||||
|
||||
|
||||
class CircuitBreaker:
|
||||
def __init__(
|
||||
self,
|
||||
cooldown_seconds: float = ELEC_BREAKER_COOLDOWN_HOURS * 3600,
|
||||
on_trip: Optional[Callable[[str, str, float], None]] = None,
|
||||
clock: Callable[[], float] = time.time,
|
||||
) -> None:
|
||||
self.cooldown = cooldown_seconds
|
||||
self.on_trip = on_trip
|
||||
self._clock = clock
|
||||
self._open: Dict[str, Tuple[float, str]] = {}
|
||||
self._lock = threading.Lock()
|
||||
|
||||
@staticmethod
|
||||
def _key(host: str) -> str:
|
||||
host = host.lower()
|
||||
return host[4:] if host.startswith("www.") else host
|
||||
|
||||
def preload(self, host: str, until_epoch: float, reason: str) -> None:
|
||||
"""Restore a breaker that was opened in an earlier run (elec.site)."""
|
||||
if until_epoch > self._clock():
|
||||
with self._lock:
|
||||
self._open[self._key(host)] = (until_epoch, reason)
|
||||
|
||||
def trip(self, host: str, reason: str) -> None:
|
||||
until = self._clock() + self.cooldown
|
||||
with self._lock:
|
||||
self._open[self._key(host)] = (until, reason)
|
||||
if self.on_trip:
|
||||
self.on_trip(self._key(host), reason, until)
|
||||
|
||||
def is_open(self, host: str) -> bool:
|
||||
key = self._key(host)
|
||||
with self._lock:
|
||||
entry = self._open.get(key)
|
||||
if not entry:
|
||||
return False
|
||||
if entry[0] <= self._clock():
|
||||
del self._open[key]
|
||||
return False
|
||||
return True
|
||||
|
||||
def reason(self, host: str) -> Optional[str]:
|
||||
entry = self._open.get(self._key(host))
|
||||
return entry[1] if entry else None
|
||||
223
backend/app/electronics/net/polite_client.py
Normal file
223
backend/app/electronics/net/polite_client.py
Normal file
@@ -0,0 +1,223 @@
|
||||
"""The only way this project fetches a retail or brand web page.
|
||||
|
||||
What it guarantees, for every request:
|
||||
* robots.txt is consulted first (protego). If robots.txt cannot be read
|
||||
because the server errors or blocks it, the site is treated as disallowed.
|
||||
* at least ELEC_SITE_MIN_INTERVAL_SECONDS between requests to one host.
|
||||
* an honest User-Agent naming the project and a contact address.
|
||||
* no JavaScript, no cookies kept between requests, no proxies, no retries on
|
||||
403/429 - a block is respected, not worked around.
|
||||
* a size cap on the response body.
|
||||
* a circuit breaker: a 403/429/CAPTCHA response opens it for the host, and
|
||||
every later request to that host is refused until the cooldown passes.
|
||||
* every request is reported to `on_fetch` (the fetch_log table).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
import threading
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import Callable, Dict, Optional, Tuple
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import httpx
|
||||
from protego import Protego
|
||||
|
||||
from app.electronics.net.breaker import CircuitBreaker
|
||||
from app.infrastructure.settings import (
|
||||
ELEC_MAX_PAGE_BYTES,
|
||||
ELEC_SITE_MIN_INTERVAL_SECONDS,
|
||||
REQUEST_TIMEOUT_SECONDS,
|
||||
USER_AGENT,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
ROBOTS_TTL_SECONDS = 24 * 3600
|
||||
|
||||
# Pages that are a bot check rather than content. Matched on the first 20 KB.
|
||||
_CAPTCHA_MARKERS = re.compile(
|
||||
r"captcha|robot\s*check|are\s+you\s+a\s+robot|verify\s+you\s+are\s+human|"
|
||||
r"/errors/validatecaptcha|px-captcha|cf-challenge|challenge-platform|access\s+denied|"
|
||||
r"unusual\s+traffic|request\s+blocked|bot\s+detection|akamai.*reference",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class FetchResult:
|
||||
url: str
|
||||
final_url: str
|
||||
status: Optional[int]
|
||||
text: str
|
||||
outcome: str # ok | robots_disallowed | blocked | captcha | breaker_open | http_error | network_error | too_large | not_html
|
||||
robots_allowed: Optional[bool]
|
||||
bytes: int = 0
|
||||
|
||||
@property
|
||||
def ok(self) -> bool:
|
||||
return self.outcome == "ok"
|
||||
|
||||
|
||||
class PoliteClient:
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
min_interval: float = ELEC_SITE_MIN_INTERVAL_SECONDS,
|
||||
breaker: Optional[CircuitBreaker] = None,
|
||||
on_fetch: Optional[Callable[[FetchResult, str], None]] = None,
|
||||
transport: Optional[httpx.BaseTransport] = None,
|
||||
sleep: Callable[[float], None] = time.sleep,
|
||||
clock: Callable[[], float] = time.monotonic,
|
||||
) -> None:
|
||||
self.min_interval = min_interval
|
||||
self.breaker = breaker or CircuitBreaker()
|
||||
self.on_fetch = on_fetch
|
||||
self._sleep = sleep
|
||||
self._clock = clock
|
||||
self._last: Dict[str, float] = {}
|
||||
self._locks: Dict[str, threading.Lock] = {}
|
||||
self._robots: Dict[str, Tuple[float, Optional[Protego], bool]] = {}
|
||||
self._guard = threading.Lock()
|
||||
self._client = httpx.Client(
|
||||
headers={
|
||||
"User-Agent": USER_AGENT,
|
||||
"Accept": "text/html,application/xhtml+xml,application/json;q=0.9,*/*;q=0.5",
|
||||
"Accept-Language": "en-IN,en;q=0.9",
|
||||
},
|
||||
follow_redirects=True,
|
||||
timeout=REQUEST_TIMEOUT_SECONDS,
|
||||
transport=transport,
|
||||
)
|
||||
|
||||
def close(self) -> None:
|
||||
self._client.close()
|
||||
|
||||
def __enter__(self) -> "PoliteClient":
|
||||
return self
|
||||
|
||||
def __exit__(self, *exc) -> None:
|
||||
self.close()
|
||||
|
||||
# -- pacing --------------------------------------------------------------
|
||||
def _host_lock(self, host: str) -> threading.Lock:
|
||||
with self._guard:
|
||||
return self._locks.setdefault(host, threading.Lock())
|
||||
|
||||
def _wait_turn(self, host: str) -> None:
|
||||
last = self._last.get(host)
|
||||
if last is not None:
|
||||
gap = self.min_interval - (self._clock() - last)
|
||||
if gap > 0:
|
||||
self._sleep(gap)
|
||||
self._last[host] = self._clock()
|
||||
|
||||
# -- robots.txt ----------------------------------------------------------
|
||||
def _robots_for(self, scheme: str, host: str) -> Tuple[Optional[Protego], bool]:
|
||||
"""(parser, reachable). parser None + reachable True = no robots.txt
|
||||
(everything allowed); reachable False = could not read it (deny)."""
|
||||
cached = self._robots.get(host)
|
||||
if cached and self._clock() - cached[0] < ROBOTS_TTL_SECONDS:
|
||||
return cached[1], cached[2]
|
||||
url = f"{scheme}://{host}/robots.txt"
|
||||
parser: Optional[Protego] = None
|
||||
reachable = False
|
||||
self._wait_turn(host)
|
||||
try:
|
||||
resp = self._client.get(url)
|
||||
if resp.status_code == 200:
|
||||
parser, reachable = Protego.parse(resp.text), True
|
||||
elif resp.status_code in (404, 410):
|
||||
parser, reachable = None, True
|
||||
else:
|
||||
reachable = False
|
||||
if resp.status_code in (403, 429):
|
||||
self.breaker.trip(host, f"robots.txt returned HTTP {resp.status_code}")
|
||||
except httpx.HTTPError as exc:
|
||||
logger.info("robots.txt unreachable for %s: %s", host, exc)
|
||||
self._robots[host] = (self._clock(), parser, reachable)
|
||||
return parser, reachable
|
||||
|
||||
def robots_allowed(self, url: str) -> bool:
|
||||
p = urlparse(url)
|
||||
parser, reachable = self._robots_for(p.scheme or "https", p.netloc.lower())
|
||||
if not reachable:
|
||||
return False
|
||||
return True if parser is None else bool(parser.can_fetch(url, USER_AGENT))
|
||||
|
||||
# -- fetch ---------------------------------------------------------------
|
||||
def _report(self, result: FetchResult) -> FetchResult:
|
||||
if self.on_fetch:
|
||||
try:
|
||||
self.on_fetch(result, urlparse(result.url).netloc.lower())
|
||||
except Exception as exc: # noqa: BLE001 - logging must never break a crawl
|
||||
logger.debug("fetch log failed: %s", exc)
|
||||
return result
|
||||
|
||||
def get(self, url: str, *, check_robots: bool = True, accept_non_html: bool = False) -> FetchResult:
|
||||
host = urlparse(url).netloc.lower()
|
||||
if self.breaker.is_open(host):
|
||||
return FetchResult(url, url, None, "", "breaker_open", None)
|
||||
with self._host_lock(host):
|
||||
allowed: Optional[bool] = None
|
||||
if check_robots:
|
||||
allowed = self.robots_allowed(url)
|
||||
if not allowed:
|
||||
return self._report(FetchResult(url, url, None, "", "robots_disallowed", False))
|
||||
self._wait_turn(host)
|
||||
try:
|
||||
with self._client.stream("GET", url) as resp:
|
||||
status = resp.status_code
|
||||
final = str(resp.url)
|
||||
ctype = resp.headers.get("content-type", "").lower()
|
||||
body = bytearray()
|
||||
too_large = False
|
||||
for chunk in resp.iter_bytes():
|
||||
body.extend(chunk)
|
||||
if len(body) > ELEC_MAX_PAGE_BYTES:
|
||||
too_large = True
|
||||
break
|
||||
encoding = resp.encoding or "utf-8"
|
||||
except httpx.HTTPError as exc:
|
||||
logger.info("fetch failed %s: %s", url, exc)
|
||||
return self._report(FetchResult(url, url, None, "", "network_error", allowed))
|
||||
|
||||
text = bytes(body).decode(encoding, errors="replace") if body else ""
|
||||
n = len(body)
|
||||
if status in (403, 429, 503) or (status == 200 and _CAPTCHA_MARKERS.search(text[:20000]) and len(text) < 60000):
|
||||
outcome = "captcha" if status == 200 or _CAPTCHA_MARKERS.search(text[:20000]) else "blocked"
|
||||
self.breaker.trip(host, f"HTTP {status} ({outcome})")
|
||||
return self._report(FetchResult(url, final, status, "", outcome, allowed, n))
|
||||
if status != 200:
|
||||
return self._report(FetchResult(url, final, status, "", "http_error", allowed, n))
|
||||
if too_large:
|
||||
return self._report(FetchResult(url, final, status, "", "too_large", allowed, n))
|
||||
if not accept_non_html and "html" not in ctype and "json" not in ctype:
|
||||
return self._report(FetchResult(url, final, status, "", "not_html", allowed, n))
|
||||
return self._report(FetchResult(url, final, status, text, "ok", allowed, n))
|
||||
|
||||
def check_image(self, url: str, min_bytes: int) -> bool:
|
||||
"""One ranged GET to confirm a URL serves a real image. Paced per host
|
||||
like any request; robots.txt is not consulted because this fetches a
|
||||
single file the product page itself references, as a browser would."""
|
||||
host = urlparse(url).netloc.lower()
|
||||
if not url.startswith(("http://", "https://")) or self.breaker.is_open(host):
|
||||
return False
|
||||
with self._host_lock(host):
|
||||
self._wait_turn(host)
|
||||
try:
|
||||
with self._client.stream("GET", url, headers={"Accept": "image/*", "Range": f"bytes=0-{min_bytes * 4}"}) as resp:
|
||||
if resp.status_code not in (200, 206):
|
||||
return False
|
||||
if not resp.headers.get("content-type", "").lower().startswith("image/"):
|
||||
return False
|
||||
got = 0
|
||||
for chunk in resp.iter_bytes():
|
||||
got += len(chunk)
|
||||
if got >= min_bytes:
|
||||
return True
|
||||
return got >= min_bytes
|
||||
except httpx.HTTPError:
|
||||
return False
|
||||
Reference in New Issue
Block a user