""" Thin client for the local Ollama server. In this project the LLM is an EXTRACTOR, never a source: it is only ever handed text that was fetched from a real page or search result, and every value it returns is checked against that text by app.electronics.normalise.grounding before it is kept. There are deliberately no functions here that ask the model to list products, prices or images. """ from __future__ import annotations import json import re import time import requests from app.infrastructure.settings import OLLAMA_BASE_URL, OLLAMA_MODEL_NAME, USE_OLLAMA, OLLAMA_TIMEOUT_SECONDS # Reachability is asked once per this many seconds, not once per caller. # # WHY THIS CACHE EXISTS. The probe below costs up to 5 seconds when nothing is # listening, and `_ensure_client` is called per ROW by stage 2 of the ingestion # pipeline (store_catalog_pipeline.stage_2_row_intake -> fetch_product_details). # Uncached, a 2000-row sheet ingested with use_llm on, against a configured but # unreachable Ollama, spends up to ~2.8 hours doing nothing but timing out - and # presents as a batch that has hung rather than one that has failed. That is not # hypothetical: USE_OLLAMA=true pointing at localhost:11434 is the default # developer configuration, and `ollama serve` is not always running beside it. # # /api/health calls this too (app/api/routers/system.py), so the same cache # stops a down Ollama adding 5s to every health request. # # A TTL rather than a permanent memo, deliberately: this is a liveness fact, not # configuration. Cached forever, an Ollama started after the API would never be # noticed and /api/health would report it down until a redeploy. _PROBE_TTL_SECONDS = 30.0 _probe_cache: tuple[float, bool] | None = None def reset_reachability_cache() -> None: """Forget the cached probe. For tests, and for anything that knows the answer just changed.""" global _probe_cache _probe_cache = None def _ensure_client(): """None when Ollama is switched off, True/False for reachable or not. Three return values, not two - `system.py` relies on telling "disabled" from "configured but down", so do not collapse this to a bool. """ if not USE_OLLAMA: # No network call on this path, so nothing worth caching. return None global _probe_cache now = time.monotonic() if _probe_cache is not None and now - _probe_cache[0] < _PROBE_TTL_SECONDS: return _probe_cache[1] # Verify Ollama is reachable try: resp = requests.get(f"{OLLAMA_BASE_URL}/api/tags", timeout=5) reachable = resp.status_code == 200 except Exception: reachable = False _probe_cache = (now, reachable) return reachable def _generate( system: str, user_prompt: str, max_retries: int = 2, *, temperature: float = 0.0, json_mode: bool = False, ) -> str: """Call Ollama's chat endpoint and return text safely. Retries up to `max_retries` times when the response is empty, since small local models (e.g. qwen2.5:1.5b) sometimes return empty content for complex JSON prompts on the first attempt. """ if not _ensure_client(): return "" for attempt in range(max_retries + 1): try: resp = requests.post( f"{OLLAMA_BASE_URL}/api/chat", json={ "model": OLLAMA_MODEL_NAME, "messages": [ {"role": "system", "content": system}, {"role": "user", "content": user_prompt}, ], "stream": False, "options": {"temperature": temperature}, **({"format": "json"} if json_mode else {}), }, timeout=OLLAMA_TIMEOUT_SECONDS, ) resp.raise_for_status() data = resp.json() content = (data.get("message", {}).get("content", "") or "").strip() if content: return content if attempt < max_retries: import time time.sleep(1.0) except Exception: if attempt >= max_retries: return "" import time time.sleep(1.0) return "" def _extract_json(text: str) -> dict | None: """Extract JSON from model response, trying multiple strategies. Handles both JSON objects {...} and JSON arrays [...] since small local models frequently return bare arrays instead of an object with a ``products`` key. """ if not text: return None # Try fenced code block (object or array) match = re.search(r"```(?:json)?\s*(\{[\s\S]*?\}|\[[\s\S]*?\])\s*```", text) if match: try: return json.loads(match.group(1)) except json.JSONDecodeError: pass # Try first JSON value in text (greedy - object) brace = re.search(r"\{[\s\S]*\}", text) if brace: try: return json.loads(brace.group(0)) except json.JSONDecodeError: pass # Try first JSON value in text (greedy - array) bracket = re.search(r"\[[\s\S]*\]", text) if bracket: try: return json.loads(bracket.group(0)) except json.JSONDecodeError: pass # Try parsing entire text try: return json.loads(text.strip()) except json.JSONDecodeError: pass # Fallback: try to fix common issues cleaned = text.strip() cleaned = re.sub(r"(?<=[:,\[])\s*'", '"', cleaned) cleaned = re.sub(r"'\s*(?=[,:\}\]])", '"', cleaned) try: return json.loads(cleaned) except json.JSONDecodeError: return None def generate_json(system_prompt: str, user_prompt: str) -> dict | list | None: """Deterministic JSON-mode completion. None when Ollama is unavailable or the reply is not JSON - callers must treat that as "no extra data".""" return _extract_json(_generate(system_prompt, user_prompt, max_retries=1, json_mode=True))