Electronics Catalog: API, MCP server, frontend and deployment

Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
sriram
2026-10-01 12:17:42 +05:30
commit c7e4d59188
115 changed files with 14329 additions and 0 deletions

View File

View File

@@ -0,0 +1,52 @@
from __future__ import annotations
from typing import List, Optional, TYPE_CHECKING
from app.infrastructure.settings import EMBEDDINGS_MODEL
if TYPE_CHECKING: # pragma: no cover - typing only, no runtime cost
from sentence_transformers import SentenceTransformer
_model_singleton: Optional["SentenceTransformer"] = None
def get_device() -> str:
"""Prefer CUDA if available, otherwise CPU.
Imports torch lazily: on an 8GB RAM / CPU-only laptop there is no
benefit to importing torch (and paying its startup/memory cost) until
an embedding is actually requested, so the FastAPI process can boot
and answer /api/health almost instantly.
"""
import torch # local import - see docstring
return "cuda" if torch.cuda.is_available() else "cpu"
def get_embedding_model() -> "SentenceTransformer":
global _model_singleton
if _model_singleton is None:
from sentence_transformers import SentenceTransformer # local import - see get_device()
device = get_device()
_model_singleton = SentenceTransformer(EMBEDDINGS_MODEL, device=device)
return _model_singleton
def embed_texts(texts: List[str]) -> List[List[float]]:
"""Embed a batch of texts into normalized 384-dim vectors (MiniLM-L6-v2).
Normalized so that pgvector's cosine-distance operator (`<=>`) behaves
consistently for the RAG retrieval step in `app.services.vector_store`.
"""
if not texts:
return []
model = get_embedding_model()
embeddings = model.encode(
texts,
batch_size=32,
normalize_embeddings=True,
convert_to_numpy=True,
show_progress_bar=False,
)
return embeddings.tolist()

View File

@@ -0,0 +1,171 @@
"""
Thin client for the local Ollama server.
In this project the LLM is an EXTRACTOR, never a source: it is only ever
handed text that was fetched from a real page or search result, and every
value it returns is checked against that text by
app.electronics.normalise.grounding before it is kept. There are deliberately
no functions here that ask the model to list products, prices or images.
"""
from __future__ import annotations
import json
import re
import time
import requests
from app.infrastructure.settings import OLLAMA_BASE_URL, OLLAMA_MODEL_NAME, USE_OLLAMA, OLLAMA_TIMEOUT_SECONDS
# Reachability is asked once per this many seconds, not once per caller.
#
# WHY THIS CACHE EXISTS. The probe below costs up to 5 seconds when nothing is
# listening, and `_ensure_client` is called per ROW by stage 2 of the ingestion
# pipeline (store_catalog_pipeline.stage_2_row_intake -> fetch_product_details).
# Uncached, a 2000-row sheet ingested with use_llm on, against a configured but
# unreachable Ollama, spends up to ~2.8 hours doing nothing but timing out - and
# presents as a batch that has hung rather than one that has failed. That is not
# hypothetical: USE_OLLAMA=true pointing at localhost:11434 is the default
# developer configuration, and `ollama serve` is not always running beside it.
#
# /api/health calls this too (app/api/routers/system.py), so the same cache
# stops a down Ollama adding 5s to every health request.
#
# A TTL rather than a permanent memo, deliberately: this is a liveness fact, not
# configuration. Cached forever, an Ollama started after the API would never be
# noticed and /api/health would report it down until a redeploy.
_PROBE_TTL_SECONDS = 30.0
_probe_cache: tuple[float, bool] | None = None
def reset_reachability_cache() -> None:
"""Forget the cached probe. For tests, and for anything that knows the
answer just changed."""
global _probe_cache
_probe_cache = None
def _ensure_client():
"""None when Ollama is switched off, True/False for reachable or not.
Three return values, not two - `system.py` relies on telling "disabled" from
"configured but down", so do not collapse this to a bool.
"""
if not USE_OLLAMA:
# No network call on this path, so nothing worth caching.
return None
global _probe_cache
now = time.monotonic()
if _probe_cache is not None and now - _probe_cache[0] < _PROBE_TTL_SECONDS:
return _probe_cache[1]
# Verify Ollama is reachable
try:
resp = requests.get(f"{OLLAMA_BASE_URL}/api/tags", timeout=5)
reachable = resp.status_code == 200
except Exception:
reachable = False
_probe_cache = (now, reachable)
return reachable
def _generate(
system: str,
user_prompt: str,
max_retries: int = 2,
*,
temperature: float = 0.0,
json_mode: bool = False,
) -> str:
"""Call Ollama's chat endpoint and return text safely.
Retries up to `max_retries` times when the response is empty, since
small local models (e.g. qwen2.5:1.5b) sometimes return empty content
for complex JSON prompts on the first attempt.
"""
if not _ensure_client():
return ""
for attempt in range(max_retries + 1):
try:
resp = requests.post(
f"{OLLAMA_BASE_URL}/api/chat",
json={
"model": OLLAMA_MODEL_NAME,
"messages": [
{"role": "system", "content": system},
{"role": "user", "content": user_prompt},
],
"stream": False,
"options": {"temperature": temperature},
**({"format": "json"} if json_mode else {}),
},
timeout=OLLAMA_TIMEOUT_SECONDS,
)
resp.raise_for_status()
data = resp.json()
content = (data.get("message", {}).get("content", "") or "").strip()
if content:
return content
if attempt < max_retries:
import time
time.sleep(1.0)
except Exception:
if attempt >= max_retries:
return ""
import time
time.sleep(1.0)
return ""
def _extract_json(text: str) -> dict | None:
"""Extract JSON from model response, trying multiple strategies.
Handles both JSON objects {...} and JSON arrays [...] since small
local models frequently return bare arrays instead of an object with
a ``products`` key.
"""
if not text:
return None
# Try fenced code block (object or array)
match = re.search(r"```(?:json)?\s*(\{[\s\S]*?\}|\[[\s\S]*?\])\s*```", text)
if match:
try:
return json.loads(match.group(1))
except json.JSONDecodeError:
pass
# Try first JSON value in text (greedy - object)
brace = re.search(r"\{[\s\S]*\}", text)
if brace:
try:
return json.loads(brace.group(0))
except json.JSONDecodeError:
pass
# Try first JSON value in text (greedy - array)
bracket = re.search(r"\[[\s\S]*\]", text)
if bracket:
try:
return json.loads(bracket.group(0))
except json.JSONDecodeError:
pass
# Try parsing entire text
try:
return json.loads(text.strip())
except json.JSONDecodeError:
pass
# Fallback: try to fix common issues
cleaned = text.strip()
cleaned = re.sub(r"(?<=[:,\[])\s*'", '"', cleaned)
cleaned = re.sub(r"'\s*(?=[,:\}\]])", '"', cleaned)
try:
return json.loads(cleaned)
except json.JSONDecodeError:
return None
def generate_json(system_prompt: str, user_prompt: str) -> dict | list | None:
"""Deterministic JSON-mode completion. None when Ollama is unavailable or
the reply is not JSON - callers must treat that as "no extra data"."""
return _extract_json(_generate(system_prompt, user_prompt, max_retries=1, json_mode=True))