Files
AI_engine/core/llm.py
Suriyakumarvijayanayagam 4283c602f6 fix(dispatch): liveness-aware coverage check, escalation rate limit, real alert sink
Prod findings (2026-09-22): a backend retry sweep failed 1,001 bookings in
60s; the agent called each a "coverage gap" because GEORADIUS found a miler
in the geo index (last seen in June), then sent 1,001 ops_alert tasks to
CUSTOMER_AGENT, which has no such handler.

- _find_zone filters GEORADIUS candidates by the backend's miler_status:<id>
  key; only status=Available counts. Facts now carry
  nearest_available_miler_within_km plus milers_in_geo_index_within_30km so
  the decision can separate "no riders here" from "riders exist, none on duty".
- Rate limit per zone per day: after the first alert, further failures only
  bump the counter (no LLM call); a summary re-alert goes out every
  DISPATCH_REALERT_EVERY (default 100).
- _ops_alert / _escalate_dispatch send EXCEPTION_DETECTED to JARVIS (the path
  that is actually handled); customer delay notice uses CUSTOMER_AGENT's real
  send_notification contract.
- JARVIS: escalation inbox (_escalations, pending_escalations()) and
  human_review/ops_alert task types are recorded instead of dropped.
- ExceptionAgent pull loops: also catch asyncio.TimeoutError (distinct from
  nats.errors.TimeoutError on 3.11) and log the exception type — the blank
  "pull loop error:" lines.
- Prompt + eval cases updated for the renamed facts; new case for the
  observed index-full/nobody-on-duty pattern. Tests for liveness filtering,
  burst suppression, fallback heuristic, sinks, and the JARVIS inbox.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_012AJLYcbTHCe45fyFnMfEin
2026-09-22 16:26:27 +05:30

225 lines
9.8 KiB
Python

"""
LLM-backed decision-making for agents.
Currently backed by Claude (Anthropic). It is deliberately kept behind a thin
function boundary (`decide_stall_response`) and a `dataclass` result so the
model — or the whole provider — can be swapped later (e.g. a self-hosted model)
without touching any agent logic.
Configuration (env):
ANTHROPIC_API_KEY required — the Anthropic API key
LLM_MODEL default "claude-opus-4-8"
LLM_EFFORT default "high" (low | medium | high | xhigh | max)
"""
import json
import os
from dataclasses import dataclass
from datetime import datetime, timezone
from typing import Optional
from core.logger import logger
LLM_MODEL = os.getenv("LLM_MODEL", "claude-opus-4-8")
LLM_EFFORT = os.getenv("LLM_EFFORT", "high")
# Thinking tokens count against max_tokens; adaptive thinking at high effort can
# use a lot, so the budget must be generous or the JSON output gets truncated.
LLM_MAX_TOKENS = int(os.getenv("LLM_MAX_TOKENS", "8192"))
LLM_TIMEOUT_S = float(os.getenv("LLM_TIMEOUT_S", "30"))
_client = None
def _get_client():
"""Lazily construct the *async* Anthropic client so a missing dependency or
key never breaks agent startup — only the actual decision call fails, and the
caller falls back to deterministic behaviour. Async so the decision never
blocks the event loop (a sync client would freeze every other coroutine for
the full duration of the call)."""
global _client
if _client is None:
import anthropic # lazy import
_client = anthropic.AsyncAnthropic() # reads ANTHROPIC_API_KEY from env
return _client
@dataclass
class StallDecision:
action: str # "reassign" | "notify_only" | "wait" | "escalate"
reasoning: str
confidence: float # 0.0 .. 1.0
DEFAULT_STALL_THRESHOLD_MIN = 10
def build_stall_context(minutes_stalled, *, stall_threshold_min=DEFAULT_STALL_THRESHOLD_MIN,
now=None, facts=None) -> str:
"""Assemble the prompt context for a stall decision.
Shared by the ExceptionAgent and the eval harness so the eval tests exactly
what runs in production. `facts` is an optional dict of extra situational
detail (location, notes, prior-stall count, ...) — today the agent passes
none, but context tools will populate it later.
"""
now = now or datetime.now(timezone.utc)
lines = [
"A miler has stalled during an active delivery.",
f"- minutes_stalled: {round(float(minutes_stalled), 1)}",
f"- stall_threshold_minutes: {stall_threshold_min}",
f"- current_time_utc: {now.isoformat()}",
f"- hour_of_day_utc: {now.hour}",
]
for key, value in (facts or {}).items():
lines.append(f"- {key}: {value}")
lines.append("Decide the best response.")
return "\n".join(lines)
_VALID_ACTIONS = {"reassign", "notify_only", "wait", "escalate"}
_STALL_SYSTEM = """You are the exception controller for a last-mile delivery network.
A "miler" (delivery rider) assigned to an active booking has stopped moving for a while.
Decide the single best response. Choose exactly one action:
- "wait": the stall is plausibly benign (traffic, a short pickup wait, a customer handoff). Take no action yet.
- "notify_only": the miler can likely recover, but the customer deserves reassurance. Notify the customer; keep the miler.
- "reassign": the miler is genuinely stuck and unlikely to recover soon. Hand the booking to another miler. This is customer-visible and hard to reverse — choose it only when the evidence clearly supports it.
- "escalate": the situation is ambiguous or high-stakes; a human dispatcher should decide.
Weigh how long it has been stalled relative to the threshold, the time of day, and how far into the job it is.
Prefer the least disruptive action that fits the evidence. Report an honest confidence in [0, 1] —
a low confidence is a signal to escalate rather than act."""
_STALL_SCHEMA = {
"type": "object",
"properties": {
"action": {"type": "string", "enum": sorted(_VALID_ACTIONS)},
"reasoning": {"type": "string"},
"confidence": {"type": "number"},
},
"required": ["action", "reasoning", "confidence"],
"additionalProperties": False,
}
async def _decide(system: str, schema: dict, valid_actions, context: str):
"""Run one structured-output decision against Claude.
Returns (action, reasoning, confidence), or None on any failure (network,
timeout, refusal, truncation, bad parse) so callers can fall back to
deterministic behaviour. Retries once on a transient error before giving up.
"""
resp = None
last_err = None
for attempt in range(2): # initial try + one retry on transient failure
try:
client = _get_client()
resp = await client.messages.create(
model=LLM_MODEL,
max_tokens=LLM_MAX_TOKENS,
system=system,
thinking={"type": "adaptive"},
output_config={
"effort": LLM_EFFORT,
"format": {"type": "json_schema", "schema": schema},
},
messages=[{"role": "user", "content": context}],
timeout=LLM_TIMEOUT_S,
)
break
except Exception as e:
last_err = e
logger.warning(f"LLM decision request failed (attempt {attempt + 1}/2): {e}")
if resp is None:
logger.error(f"LLM decision failed after retries: {last_err}")
return None
stop = getattr(resp, "stop_reason", None)
if stop == "refusal":
logger.warning("LLM refused the decision request")
return None
if stop == "max_tokens":
logger.error("LLM decision truncated (max_tokens); raise LLM_MAX_TOKENS. Treating as no decision.")
return None
try:
text = next(b.text for b in resp.content if b.type == "text")
data = json.loads(text)
action = data["action"]
if action not in valid_actions:
raise ValueError(f"unexpected action {action!r}")
return action, str(data.get("reasoning", "")), float(data.get("confidence", 0.0))
except (StopIteration, KeyError, ValueError, json.JSONDecodeError) as e:
logger.error(f"LLM decision parse failed: {e}")
return None
async def decide_stall_response(context: str) -> Optional[StallDecision]:
"""Ask Claude how to handle a stalled miler. None on failure (caller falls
back to deterministic behaviour)."""
r = await _decide(_STALL_SYSTEM, _STALL_SCHEMA, _VALID_ACTIONS, context)
return StallDecision(*r) if r else None
# ── Assignment-failure decision (DispatchAgent) ──────────────────────────────
@dataclass
class AssignmentDecision:
action: str # "monitor" | "notify_customer" | "ops_alert" | "escalate"
reasoning: str
confidence: float
_ASSIGNMENT_ACTIONS = {"monitor", "notify_customer", "ops_alert", "escalate"}
_ASSIGNMENT_SYSTEM = """You are the dispatch controller for a last-mile delivery network.
The backend tried to assign a booking to a miler (rider) — both its AI assignment and the
fallback failed, so no rider was assigned. Decide how to react. Choose exactly one action:
- "monitor": likely a transient gap (a momentary lack of free riders); the backend will retry. Take no action.
- "notify_customer": a real but recoverable delay; tell the customer we're finding a rider so they aren't left guessing.
- "ops_alert": a genuine coverage gap in this zone (repeated failures, no nearby riders). Alert operations to onboard or redirect riders here.
- "escalate": ambiguous or contradictory — e.g. riders ARE nearby yet assignment keeps failing, which suggests a systemic issue rather than a coverage gap. A human dispatcher should look.
Weigh how many times assignment has failed in this zone today, how far the nearest *available* rider is,
the time of day, and whether coordinates were even available. "nearest_available_miler_within_km" counts
only riders whose live status is Available; "milers_in_geo_index_within_30km" is the raw count of riders
ever seen nearby (it includes off-duty and stale entries), so a large index count with no available rider
means riders exist here but nobody is on duty — a staffing gap, not a systemic fault. A single failure with
an available rider nearby is usually transient; repeated failures with no available rider is a coverage
gap; repeated failures *despite* available riders nearby is not a coverage gap and warrants a human.
Report an honest confidence in [0, 1]."""
_ASSIGNMENT_SCHEMA = {
"type": "object",
"properties": {
"action": {"type": "string", "enum": sorted(_ASSIGNMENT_ACTIONS)},
"reasoning": {"type": "string"},
"confidence": {"type": "number"},
},
"required": ["action", "reasoning", "confidence"],
"additionalProperties": False,
}
def build_assignment_failure_context(facts, *, now=None) -> str:
"""Assemble the prompt context for an assignment-failure decision. Shared by
DispatchAgent and the eval harness so the eval tests exactly what runs."""
now = now or datetime.now(timezone.utc)
lines = [
"A booking could not be assigned to a miler — the backend's AI and fallback assignment both failed.",
f"- current_time_utc: {now.isoformat()}",
f"- hour_of_day_utc: {now.hour}",
]
for key, value in (facts or {}).items():
lines.append(f"- {key}: {value}")
lines.append("Decide the best response.")
return "\n".join(lines)
async def decide_assignment_failure(context: str) -> Optional[AssignmentDecision]:
"""Ask Claude how to react to a failed assignment. None on failure (caller
falls back to deterministic behaviour)."""
r = await _decide(_ASSIGNMENT_SYSTEM, _ASSIGNMENT_SCHEMA, _ASSIGNMENT_ACTIONS, context)
return AssignmentDecision(*r) if r else None