Electronics Catalog: API, MCP server, frontend and deployment
Verified catalogue of mobiles and laptops sold in India, collected from real retail listings (FastAPI backend, React frontend, Postgres/pgvector). - REST API under /api/elec (read-only catalogue; admin endpoints need login) - MCP server (FastMCP) at /mcp/ with list_categories, search_products, get_product and price_history tools - Real ratings and reviews read from product pages and search results - Production Dockerfile (requirements-api.txt, no PyTorch) and .env.production.example; remote database only via an explicit ELEC_ALLOW_REMOTE_DB host/name allowlist - docs/API.md: endpoint and MCP reference with live examples Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
171
backend/app/services/ollama_service.py
Normal file
171
backend/app/services/ollama_service.py
Normal file
@@ -0,0 +1,171 @@
|
||||
"""
|
||||
Thin client for the local Ollama server.
|
||||
|
||||
In this project the LLM is an EXTRACTOR, never a source: it is only ever
|
||||
handed text that was fetched from a real page or search result, and every
|
||||
value it returns is checked against that text by
|
||||
app.electronics.normalise.grounding before it is kept. There are deliberately
|
||||
no functions here that ask the model to list products, prices or images.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
import time
|
||||
|
||||
import requests
|
||||
|
||||
from app.infrastructure.settings import OLLAMA_BASE_URL, OLLAMA_MODEL_NAME, USE_OLLAMA, OLLAMA_TIMEOUT_SECONDS
|
||||
|
||||
|
||||
# Reachability is asked once per this many seconds, not once per caller.
|
||||
#
|
||||
# WHY THIS CACHE EXISTS. The probe below costs up to 5 seconds when nothing is
|
||||
# listening, and `_ensure_client` is called per ROW by stage 2 of the ingestion
|
||||
# pipeline (store_catalog_pipeline.stage_2_row_intake -> fetch_product_details).
|
||||
# Uncached, a 2000-row sheet ingested with use_llm on, against a configured but
|
||||
# unreachable Ollama, spends up to ~2.8 hours doing nothing but timing out - and
|
||||
# presents as a batch that has hung rather than one that has failed. That is not
|
||||
# hypothetical: USE_OLLAMA=true pointing at localhost:11434 is the default
|
||||
# developer configuration, and `ollama serve` is not always running beside it.
|
||||
#
|
||||
# /api/health calls this too (app/api/routers/system.py), so the same cache
|
||||
# stops a down Ollama adding 5s to every health request.
|
||||
#
|
||||
# A TTL rather than a permanent memo, deliberately: this is a liveness fact, not
|
||||
# configuration. Cached forever, an Ollama started after the API would never be
|
||||
# noticed and /api/health would report it down until a redeploy.
|
||||
_PROBE_TTL_SECONDS = 30.0
|
||||
_probe_cache: tuple[float, bool] | None = None
|
||||
|
||||
|
||||
def reset_reachability_cache() -> None:
|
||||
"""Forget the cached probe. For tests, and for anything that knows the
|
||||
answer just changed."""
|
||||
global _probe_cache
|
||||
_probe_cache = None
|
||||
|
||||
|
||||
def _ensure_client():
|
||||
"""None when Ollama is switched off, True/False for reachable or not.
|
||||
|
||||
Three return values, not two - `system.py` relies on telling "disabled" from
|
||||
"configured but down", so do not collapse this to a bool.
|
||||
"""
|
||||
if not USE_OLLAMA:
|
||||
# No network call on this path, so nothing worth caching.
|
||||
return None
|
||||
|
||||
global _probe_cache
|
||||
now = time.monotonic()
|
||||
if _probe_cache is not None and now - _probe_cache[0] < _PROBE_TTL_SECONDS:
|
||||
return _probe_cache[1]
|
||||
|
||||
# Verify Ollama is reachable
|
||||
try:
|
||||
resp = requests.get(f"{OLLAMA_BASE_URL}/api/tags", timeout=5)
|
||||
reachable = resp.status_code == 200
|
||||
except Exception:
|
||||
reachable = False
|
||||
|
||||
_probe_cache = (now, reachable)
|
||||
return reachable
|
||||
|
||||
|
||||
def _generate(
|
||||
system: str,
|
||||
user_prompt: str,
|
||||
max_retries: int = 2,
|
||||
*,
|
||||
temperature: float = 0.0,
|
||||
json_mode: bool = False,
|
||||
) -> str:
|
||||
"""Call Ollama's chat endpoint and return text safely.
|
||||
|
||||
Retries up to `max_retries` times when the response is empty, since
|
||||
small local models (e.g. qwen2.5:1.5b) sometimes return empty content
|
||||
for complex JSON prompts on the first attempt.
|
||||
"""
|
||||
if not _ensure_client():
|
||||
return ""
|
||||
for attempt in range(max_retries + 1):
|
||||
try:
|
||||
resp = requests.post(
|
||||
f"{OLLAMA_BASE_URL}/api/chat",
|
||||
json={
|
||||
"model": OLLAMA_MODEL_NAME,
|
||||
"messages": [
|
||||
{"role": "system", "content": system},
|
||||
{"role": "user", "content": user_prompt},
|
||||
],
|
||||
"stream": False,
|
||||
"options": {"temperature": temperature},
|
||||
**({"format": "json"} if json_mode else {}),
|
||||
},
|
||||
timeout=OLLAMA_TIMEOUT_SECONDS,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
content = (data.get("message", {}).get("content", "") or "").strip()
|
||||
if content:
|
||||
return content
|
||||
if attempt < max_retries:
|
||||
import time
|
||||
time.sleep(1.0)
|
||||
except Exception:
|
||||
if attempt >= max_retries:
|
||||
return ""
|
||||
import time
|
||||
time.sleep(1.0)
|
||||
return ""
|
||||
|
||||
|
||||
def _extract_json(text: str) -> dict | None:
|
||||
"""Extract JSON from model response, trying multiple strategies.
|
||||
|
||||
Handles both JSON objects {...} and JSON arrays [...] since small
|
||||
local models frequently return bare arrays instead of an object with
|
||||
a ``products`` key.
|
||||
"""
|
||||
if not text:
|
||||
return None
|
||||
# Try fenced code block (object or array)
|
||||
match = re.search(r"```(?:json)?\s*(\{[\s\S]*?\}|\[[\s\S]*?\])\s*```", text)
|
||||
if match:
|
||||
try:
|
||||
return json.loads(match.group(1))
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
# Try first JSON value in text (greedy - object)
|
||||
brace = re.search(r"\{[\s\S]*\}", text)
|
||||
if brace:
|
||||
try:
|
||||
return json.loads(brace.group(0))
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
# Try first JSON value in text (greedy - array)
|
||||
bracket = re.search(r"\[[\s\S]*\]", text)
|
||||
if bracket:
|
||||
try:
|
||||
return json.loads(bracket.group(0))
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
# Try parsing entire text
|
||||
try:
|
||||
return json.loads(text.strip())
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
# Fallback: try to fix common issues
|
||||
cleaned = text.strip()
|
||||
cleaned = re.sub(r"(?<=[:,\[])\s*'", '"', cleaned)
|
||||
cleaned = re.sub(r"'\s*(?=[,:\}\]])", '"', cleaned)
|
||||
try:
|
||||
return json.loads(cleaned)
|
||||
except json.JSONDecodeError:
|
||||
return None
|
||||
|
||||
|
||||
def generate_json(system_prompt: str, user_prompt: str) -> dict | list | None:
|
||||
"""Deterministic JSON-mode completion. None when Ollama is unavailable or
|
||||
the reply is not JSON - callers must treat that as "no extra data"."""
|
||||
return _extract_json(_generate(system_prompt, user_prompt, max_retries=1, json_mode=True))
|
||||
Reference in New Issue
Block a user