Automate checklist-backend updates

This commit is contained in:
sriram
2026-08-31 15:32:30 +05:30
parent 3df2dc5991
commit 164ef90b31
6 changed files with 221 additions and 21 deletions

View File

@@ -190,11 +190,21 @@ UPLOAD_AUTORUN = _bool("UPLOAD_AUTORUN", "true")
# exists to avoid - stage 6 is the slowest stage and reaches the network, but
# only one batch runs at a time so nothing else is competing with it.
#
# LLM OFF, because `use_llm` gates only description generation in
# stage_2_row_intake, and production runs USE_OLLAMA=false: turning it on there
# buys nothing and costs a connection timeout per row.
# LLM ON. `use_llm` gates only description generation in stage_2_row_intake, and
# in production it is currently a no-op: USE_OLLAMA is false there, so
# ollama_service._ensure_client() returns on its first line without a request
# and the row simply keeps its blank description. It is set true so the pipeline
# is already configured correctly for the day an Ollama server exists.
#
# This default USED to be false, on the grounds that turning it on "costs a
# connection timeout per row". That was true, and it was about the OTHER branch
# of _ensure_client - USE_OLLAMA=true with nothing listening, which is any
# developer machine that has not run `ollama serve`. It is answered now by the
# TTL cache on that probe rather than by leaving the feature off: one probe per
# batch instead of one per row. Do not remove that cache and this default
# together without re-reading why both exist.
UPLOAD_AUTORUN_FETCH_IMAGES = _bool("UPLOAD_AUTORUN_FETCH_IMAGES", "true")
UPLOAD_AUTORUN_USE_LLM = _bool("UPLOAD_AUTORUN_USE_LLM", "false")
UPLOAD_AUTORUN_USE_LLM = _bool("UPLOAD_AUTORUN_USE_LLM", "true")
# --- Review inbox ----------------------------------------------------------
# The bound that applies only when UPLOAD_AUTORUN is false. Files then wait in

View File

@@ -3,6 +3,8 @@ from __future__ import annotations
from typing import List, Dict, Any, Optional
import json
import re
import time
import requests
from app.infrastructure.settings import OLLAMA_BASE_URL, OLLAMA_MODEL_NAME, USE_OLLAMA, OLLAMA_TIMEOUT_SECONDS
@@ -20,15 +22,58 @@ SYSTEM_PROMPT = (
)
# Reachability is asked once per this many seconds, not once per caller.
#
# WHY THIS CACHE EXISTS. The probe below costs up to 5 seconds when nothing is
# listening, and `_ensure_client` is called per ROW by stage 2 of the ingestion
# pipeline (store_catalog_pipeline.stage_2_row_intake -> fetch_product_details).
# Uncached, a 2000-row sheet ingested with use_llm on, against a configured but
# unreachable Ollama, spends up to ~2.8 hours doing nothing but timing out - and
# presents as a batch that has hung rather than one that has failed. That is not
# hypothetical: USE_OLLAMA=true pointing at localhost:11434 is the default
# developer configuration, and `ollama serve` is not always running beside it.
#
# /api/health calls this too (app/api/routers/system.py), so the same cache
# stops a down Ollama adding 5s to every health request.
#
# A TTL rather than a permanent memo, deliberately: this is a liveness fact, not
# configuration. Cached forever, an Ollama started after the API would never be
# noticed and /api/health would report it down until a redeploy.
_PROBE_TTL_SECONDS = 30.0
_probe_cache: tuple[float, bool] | None = None
def reset_reachability_cache() -> None:
"""Forget the cached probe. For tests, and for anything that knows the
answer just changed."""
global _probe_cache
_probe_cache = None
def _ensure_client():
"""None when Ollama is switched off, True/False for reachable or not.
Three return values, not two - `system.py` relies on telling "disabled" from
"configured but down", so do not collapse this to a bool.
"""
if not USE_OLLAMA:
# No network call on this path, so nothing worth caching.
return None
global _probe_cache
now = time.monotonic()
if _probe_cache is not None and now - _probe_cache[0] < _PROBE_TTL_SECONDS:
return _probe_cache[1]
# Verify Ollama is reachable
try:
resp = requests.get(f"{OLLAMA_BASE_URL}/api/tags", timeout=5)
return resp.status_code == 200
reachable = resp.status_code == 200
except Exception:
return False
reachable = False
_probe_cache = (now, reachable)
return reachable
def _generate(system: str, user_prompt: str, max_retries: int = 2) -> str: