Automate checklist-backend updates
This commit is contained in:
@@ -190,11 +190,21 @@ UPLOAD_AUTORUN = _bool("UPLOAD_AUTORUN", "true")
|
||||
# exists to avoid - stage 6 is the slowest stage and reaches the network, but
|
||||
# only one batch runs at a time so nothing else is competing with it.
|
||||
#
|
||||
# LLM OFF, because `use_llm` gates only description generation in
|
||||
# stage_2_row_intake, and production runs USE_OLLAMA=false: turning it on there
|
||||
# buys nothing and costs a connection timeout per row.
|
||||
# LLM ON. `use_llm` gates only description generation in stage_2_row_intake, and
|
||||
# in production it is currently a no-op: USE_OLLAMA is false there, so
|
||||
# ollama_service._ensure_client() returns on its first line without a request
|
||||
# and the row simply keeps its blank description. It is set true so the pipeline
|
||||
# is already configured correctly for the day an Ollama server exists.
|
||||
#
|
||||
# This default USED to be false, on the grounds that turning it on "costs a
|
||||
# connection timeout per row". That was true, and it was about the OTHER branch
|
||||
# of _ensure_client - USE_OLLAMA=true with nothing listening, which is any
|
||||
# developer machine that has not run `ollama serve`. It is answered now by the
|
||||
# TTL cache on that probe rather than by leaving the feature off: one probe per
|
||||
# batch instead of one per row. Do not remove that cache and this default
|
||||
# together without re-reading why both exist.
|
||||
UPLOAD_AUTORUN_FETCH_IMAGES = _bool("UPLOAD_AUTORUN_FETCH_IMAGES", "true")
|
||||
UPLOAD_AUTORUN_USE_LLM = _bool("UPLOAD_AUTORUN_USE_LLM", "false")
|
||||
UPLOAD_AUTORUN_USE_LLM = _bool("UPLOAD_AUTORUN_USE_LLM", "true")
|
||||
|
||||
# --- Review inbox ----------------------------------------------------------
|
||||
# The bound that applies only when UPLOAD_AUTORUN is false. Files then wait in
|
||||
|
||||
@@ -3,6 +3,8 @@ from __future__ import annotations
|
||||
from typing import List, Dict, Any, Optional
|
||||
import json
|
||||
import re
|
||||
import time
|
||||
|
||||
import requests
|
||||
|
||||
from app.infrastructure.settings import OLLAMA_BASE_URL, OLLAMA_MODEL_NAME, USE_OLLAMA, OLLAMA_TIMEOUT_SECONDS
|
||||
@@ -20,15 +22,58 @@ SYSTEM_PROMPT = (
|
||||
)
|
||||
|
||||
|
||||
# Reachability is asked once per this many seconds, not once per caller.
|
||||
#
|
||||
# WHY THIS CACHE EXISTS. The probe below costs up to 5 seconds when nothing is
|
||||
# listening, and `_ensure_client` is called per ROW by stage 2 of the ingestion
|
||||
# pipeline (store_catalog_pipeline.stage_2_row_intake -> fetch_product_details).
|
||||
# Uncached, a 2000-row sheet ingested with use_llm on, against a configured but
|
||||
# unreachable Ollama, spends up to ~2.8 hours doing nothing but timing out - and
|
||||
# presents as a batch that has hung rather than one that has failed. That is not
|
||||
# hypothetical: USE_OLLAMA=true pointing at localhost:11434 is the default
|
||||
# developer configuration, and `ollama serve` is not always running beside it.
|
||||
#
|
||||
# /api/health calls this too (app/api/routers/system.py), so the same cache
|
||||
# stops a down Ollama adding 5s to every health request.
|
||||
#
|
||||
# A TTL rather than a permanent memo, deliberately: this is a liveness fact, not
|
||||
# configuration. Cached forever, an Ollama started after the API would never be
|
||||
# noticed and /api/health would report it down until a redeploy.
|
||||
_PROBE_TTL_SECONDS = 30.0
|
||||
_probe_cache: tuple[float, bool] | None = None
|
||||
|
||||
|
||||
def reset_reachability_cache() -> None:
|
||||
"""Forget the cached probe. For tests, and for anything that knows the
|
||||
answer just changed."""
|
||||
global _probe_cache
|
||||
_probe_cache = None
|
||||
|
||||
|
||||
def _ensure_client():
|
||||
"""None when Ollama is switched off, True/False for reachable or not.
|
||||
|
||||
Three return values, not two - `system.py` relies on telling "disabled" from
|
||||
"configured but down", so do not collapse this to a bool.
|
||||
"""
|
||||
if not USE_OLLAMA:
|
||||
# No network call on this path, so nothing worth caching.
|
||||
return None
|
||||
|
||||
global _probe_cache
|
||||
now = time.monotonic()
|
||||
if _probe_cache is not None and now - _probe_cache[0] < _PROBE_TTL_SECONDS:
|
||||
return _probe_cache[1]
|
||||
|
||||
# Verify Ollama is reachable
|
||||
try:
|
||||
resp = requests.get(f"{OLLAMA_BASE_URL}/api/tags", timeout=5)
|
||||
return resp.status_code == 200
|
||||
reachable = resp.status_code == 200
|
||||
except Exception:
|
||||
return False
|
||||
reachable = False
|
||||
|
||||
_probe_cache = (now, reachable)
|
||||
return reachable
|
||||
|
||||
|
||||
def _generate(system: str, user_prompt: str, max_retries: int = 2) -> str:
|
||||
|
||||
Reference in New Issue
Block a user