Verified catalogue of mobiles and laptops sold in India, collected from real retail listings (FastAPI backend, React frontend, Postgres/pgvector). - REST API under /api/elec (read-only catalogue; admin endpoints need login) - MCP server (FastMCP) at /mcp/ with list_categories, search_products, get_product and price_history tools - Real ratings and reviews read from product pages and search results - Production Dockerfile (requirements-api.txt, no PyTorch) and .env.production.example; remote database only via an explicit ELEC_ALLOW_REMOTE_DB host/name allowlist - docs/API.md: endpoint and MCP reference with live examples Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
172 lines
5.9 KiB
Python
172 lines
5.9 KiB
Python
"""
|
|
Thin client for the local Ollama server.
|
|
|
|
In this project the LLM is an EXTRACTOR, never a source: it is only ever
|
|
handed text that was fetched from a real page or search result, and every
|
|
value it returns is checked against that text by
|
|
app.electronics.normalise.grounding before it is kept. There are deliberately
|
|
no functions here that ask the model to list products, prices or images.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import re
|
|
import time
|
|
|
|
import requests
|
|
|
|
from app.infrastructure.settings import OLLAMA_BASE_URL, OLLAMA_MODEL_NAME, USE_OLLAMA, OLLAMA_TIMEOUT_SECONDS
|
|
|
|
|
|
# Reachability is asked once per this many seconds, not once per caller.
|
|
#
|
|
# WHY THIS CACHE EXISTS. The probe below costs up to 5 seconds when nothing is
|
|
# listening, and `_ensure_client` is called per ROW by stage 2 of the ingestion
|
|
# pipeline (store_catalog_pipeline.stage_2_row_intake -> fetch_product_details).
|
|
# Uncached, a 2000-row sheet ingested with use_llm on, against a configured but
|
|
# unreachable Ollama, spends up to ~2.8 hours doing nothing but timing out - and
|
|
# presents as a batch that has hung rather than one that has failed. That is not
|
|
# hypothetical: USE_OLLAMA=true pointing at localhost:11434 is the default
|
|
# developer configuration, and `ollama serve` is not always running beside it.
|
|
#
|
|
# /api/health calls this too (app/api/routers/system.py), so the same cache
|
|
# stops a down Ollama adding 5s to every health request.
|
|
#
|
|
# A TTL rather than a permanent memo, deliberately: this is a liveness fact, not
|
|
# configuration. Cached forever, an Ollama started after the API would never be
|
|
# noticed and /api/health would report it down until a redeploy.
|
|
_PROBE_TTL_SECONDS = 30.0
|
|
_probe_cache: tuple[float, bool] | None = None
|
|
|
|
|
|
def reset_reachability_cache() -> None:
|
|
"""Forget the cached probe. For tests, and for anything that knows the
|
|
answer just changed."""
|
|
global _probe_cache
|
|
_probe_cache = None
|
|
|
|
|
|
def _ensure_client():
|
|
"""None when Ollama is switched off, True/False for reachable or not.
|
|
|
|
Three return values, not two - `system.py` relies on telling "disabled" from
|
|
"configured but down", so do not collapse this to a bool.
|
|
"""
|
|
if not USE_OLLAMA:
|
|
# No network call on this path, so nothing worth caching.
|
|
return None
|
|
|
|
global _probe_cache
|
|
now = time.monotonic()
|
|
if _probe_cache is not None and now - _probe_cache[0] < _PROBE_TTL_SECONDS:
|
|
return _probe_cache[1]
|
|
|
|
# Verify Ollama is reachable
|
|
try:
|
|
resp = requests.get(f"{OLLAMA_BASE_URL}/api/tags", timeout=5)
|
|
reachable = resp.status_code == 200
|
|
except Exception:
|
|
reachable = False
|
|
|
|
_probe_cache = (now, reachable)
|
|
return reachable
|
|
|
|
|
|
def _generate(
|
|
system: str,
|
|
user_prompt: str,
|
|
max_retries: int = 2,
|
|
*,
|
|
temperature: float = 0.0,
|
|
json_mode: bool = False,
|
|
) -> str:
|
|
"""Call Ollama's chat endpoint and return text safely.
|
|
|
|
Retries up to `max_retries` times when the response is empty, since
|
|
small local models (e.g. qwen2.5:1.5b) sometimes return empty content
|
|
for complex JSON prompts on the first attempt.
|
|
"""
|
|
if not _ensure_client():
|
|
return ""
|
|
for attempt in range(max_retries + 1):
|
|
try:
|
|
resp = requests.post(
|
|
f"{OLLAMA_BASE_URL}/api/chat",
|
|
json={
|
|
"model": OLLAMA_MODEL_NAME,
|
|
"messages": [
|
|
{"role": "system", "content": system},
|
|
{"role": "user", "content": user_prompt},
|
|
],
|
|
"stream": False,
|
|
"options": {"temperature": temperature},
|
|
**({"format": "json"} if json_mode else {}),
|
|
},
|
|
timeout=OLLAMA_TIMEOUT_SECONDS,
|
|
)
|
|
resp.raise_for_status()
|
|
data = resp.json()
|
|
content = (data.get("message", {}).get("content", "") or "").strip()
|
|
if content:
|
|
return content
|
|
if attempt < max_retries:
|
|
import time
|
|
time.sleep(1.0)
|
|
except Exception:
|
|
if attempt >= max_retries:
|
|
return ""
|
|
import time
|
|
time.sleep(1.0)
|
|
return ""
|
|
|
|
|
|
def _extract_json(text: str) -> dict | None:
|
|
"""Extract JSON from model response, trying multiple strategies.
|
|
|
|
Handles both JSON objects {...} and JSON arrays [...] since small
|
|
local models frequently return bare arrays instead of an object with
|
|
a ``products`` key.
|
|
"""
|
|
if not text:
|
|
return None
|
|
# Try fenced code block (object or array)
|
|
match = re.search(r"```(?:json)?\s*(\{[\s\S]*?\}|\[[\s\S]*?\])\s*```", text)
|
|
if match:
|
|
try:
|
|
return json.loads(match.group(1))
|
|
except json.JSONDecodeError:
|
|
pass
|
|
# Try first JSON value in text (greedy - object)
|
|
brace = re.search(r"\{[\s\S]*\}", text)
|
|
if brace:
|
|
try:
|
|
return json.loads(brace.group(0))
|
|
except json.JSONDecodeError:
|
|
pass
|
|
# Try first JSON value in text (greedy - array)
|
|
bracket = re.search(r"\[[\s\S]*\]", text)
|
|
if bracket:
|
|
try:
|
|
return json.loads(bracket.group(0))
|
|
except json.JSONDecodeError:
|
|
pass
|
|
# Try parsing entire text
|
|
try:
|
|
return json.loads(text.strip())
|
|
except json.JSONDecodeError:
|
|
pass
|
|
# Fallback: try to fix common issues
|
|
cleaned = text.strip()
|
|
cleaned = re.sub(r"(?<=[:,\[])\s*'", '"', cleaned)
|
|
cleaned = re.sub(r"'\s*(?=[,:\}\]])", '"', cleaned)
|
|
try:
|
|
return json.loads(cleaned)
|
|
except json.JSONDecodeError:
|
|
return None
|
|
|
|
|
|
def generate_json(system_prompt: str, user_prompt: str) -> dict | list | None:
|
|
"""Deterministic JSON-mode completion. None when Ollama is unavailable or
|
|
the reply is not JSON - callers must treat that as "no extra data"."""
|
|
return _extract_json(_generate(system_prompt, user_prompt, max_retries=1, json_mode=True))
|