Files
AI_engine/core/llm.py
Suriyakumarvijayanayagam 5159c8d1a7 fix(dispatch): presence is three-state — absent status key is unknown, not off-duty
The previous commit treated a missing miler_status:<id> key as "not available".
In production only 2 of 34 milers have that key at all, so the agent would have
reported "no available rider" for nearly every failure — a confident wrong
answer in the opposite direction from the bug it fixed.

- _miler_presence returns available / unavailable / unknown. No key, an
  unparseable value, or an unrecognised status reads as unknown.
- _find_zone prefers a confirmed-available miler, otherwise reports the nearest
  unknown-presence one (it may well be assignable), and returns None only when
  every nearby candidate is confirmed off duty.
- Facts carry nearest_miler_within_km + nearest_miler_presence; the prompt
  states plainly that unknown presence is not evidence of a coverage gap and
  should lean to monitor/escalate rather than ops_alert.
- New eval case for riders-nearby-but-no-presence-data; tests for all three
  presence states.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_012AJLYcbTHCe45fyFnMfEin
2026-09-22 16:39:33 +05:30

232 lines
10 KiB
Python

"""
LLM-backed decision-making for agents.
Currently backed by Claude (Anthropic). It is deliberately kept behind a thin
function boundary (`decide_stall_response`) and a `dataclass` result so the
model — or the whole provider — can be swapped later (e.g. a self-hosted model)
without touching any agent logic.
Configuration (env):
ANTHROPIC_API_KEY required — the Anthropic API key
LLM_MODEL default "claude-opus-4-8"
LLM_EFFORT default "high" (low | medium | high | xhigh | max)
"""
import json
import os
from dataclasses import dataclass
from datetime import datetime, timezone
from typing import Optional
from core.logger import logger
LLM_MODEL = os.getenv("LLM_MODEL", "claude-opus-4-8")
LLM_EFFORT = os.getenv("LLM_EFFORT", "high")
# Thinking tokens count against max_tokens; adaptive thinking at high effort can
# use a lot, so the budget must be generous or the JSON output gets truncated.
LLM_MAX_TOKENS = int(os.getenv("LLM_MAX_TOKENS", "8192"))
LLM_TIMEOUT_S = float(os.getenv("LLM_TIMEOUT_S", "30"))
_client = None
def _get_client():
"""Lazily construct the *async* Anthropic client so a missing dependency or
key never breaks agent startup — only the actual decision call fails, and the
caller falls back to deterministic behaviour. Async so the decision never
blocks the event loop (a sync client would freeze every other coroutine for
the full duration of the call)."""
global _client
if _client is None:
import anthropic # lazy import
_client = anthropic.AsyncAnthropic() # reads ANTHROPIC_API_KEY from env
return _client
@dataclass
class StallDecision:
action: str # "reassign" | "notify_only" | "wait" | "escalate"
reasoning: str
confidence: float # 0.0 .. 1.0
DEFAULT_STALL_THRESHOLD_MIN = 10
def build_stall_context(minutes_stalled, *, stall_threshold_min=DEFAULT_STALL_THRESHOLD_MIN,
now=None, facts=None) -> str:
"""Assemble the prompt context for a stall decision.
Shared by the ExceptionAgent and the eval harness so the eval tests exactly
what runs in production. `facts` is an optional dict of extra situational
detail (location, notes, prior-stall count, ...) — today the agent passes
none, but context tools will populate it later.
"""
now = now or datetime.now(timezone.utc)
lines = [
"A miler has stalled during an active delivery.",
f"- minutes_stalled: {round(float(minutes_stalled), 1)}",
f"- stall_threshold_minutes: {stall_threshold_min}",
f"- current_time_utc: {now.isoformat()}",
f"- hour_of_day_utc: {now.hour}",
]
for key, value in (facts or {}).items():
lines.append(f"- {key}: {value}")
lines.append("Decide the best response.")
return "\n".join(lines)
_VALID_ACTIONS = {"reassign", "notify_only", "wait", "escalate"}
_STALL_SYSTEM = """You are the exception controller for a last-mile delivery network.
A "miler" (delivery rider) assigned to an active booking has stopped moving for a while.
Decide the single best response. Choose exactly one action:
- "wait": the stall is plausibly benign (traffic, a short pickup wait, a customer handoff). Take no action yet.
- "notify_only": the miler can likely recover, but the customer deserves reassurance. Notify the customer; keep the miler.
- "reassign": the miler is genuinely stuck and unlikely to recover soon. Hand the booking to another miler. This is customer-visible and hard to reverse — choose it only when the evidence clearly supports it.
- "escalate": the situation is ambiguous or high-stakes; a human dispatcher should decide.
Weigh how long it has been stalled relative to the threshold, the time of day, and how far into the job it is.
Prefer the least disruptive action that fits the evidence. Report an honest confidence in [0, 1] —
a low confidence is a signal to escalate rather than act."""
_STALL_SCHEMA = {
"type": "object",
"properties": {
"action": {"type": "string", "enum": sorted(_VALID_ACTIONS)},
"reasoning": {"type": "string"},
"confidence": {"type": "number"},
},
"required": ["action", "reasoning", "confidence"],
"additionalProperties": False,
}
async def _decide(system: str, schema: dict, valid_actions, context: str):
"""Run one structured-output decision against Claude.
Returns (action, reasoning, confidence), or None on any failure (network,
timeout, refusal, truncation, bad parse) so callers can fall back to
deterministic behaviour. Retries once on a transient error before giving up.
"""
resp = None
last_err = None
for attempt in range(2): # initial try + one retry on transient failure
try:
client = _get_client()
resp = await client.messages.create(
model=LLM_MODEL,
max_tokens=LLM_MAX_TOKENS,
system=system,
thinking={"type": "adaptive"},
output_config={
"effort": LLM_EFFORT,
"format": {"type": "json_schema", "schema": schema},
},
messages=[{"role": "user", "content": context}],
timeout=LLM_TIMEOUT_S,
)
break
except Exception as e:
last_err = e
logger.warning(f"LLM decision request failed (attempt {attempt + 1}/2): {e}")
if resp is None:
logger.error(f"LLM decision failed after retries: {last_err}")
return None
stop = getattr(resp, "stop_reason", None)
if stop == "refusal":
logger.warning("LLM refused the decision request")
return None
if stop == "max_tokens":
logger.error("LLM decision truncated (max_tokens); raise LLM_MAX_TOKENS. Treating as no decision.")
return None
try:
text = next(b.text for b in resp.content if b.type == "text")
data = json.loads(text)
action = data["action"]
if action not in valid_actions:
raise ValueError(f"unexpected action {action!r}")
return action, str(data.get("reasoning", "")), float(data.get("confidence", 0.0))
except (StopIteration, KeyError, ValueError, json.JSONDecodeError) as e:
logger.error(f"LLM decision parse failed: {e}")
return None
async def decide_stall_response(context: str) -> Optional[StallDecision]:
"""Ask Claude how to handle a stalled miler. None on failure (caller falls
back to deterministic behaviour)."""
r = await _decide(_STALL_SYSTEM, _STALL_SCHEMA, _VALID_ACTIONS, context)
return StallDecision(*r) if r else None
# ── Assignment-failure decision (DispatchAgent) ──────────────────────────────
@dataclass
class AssignmentDecision:
action: str # "monitor" | "notify_customer" | "ops_alert" | "escalate"
reasoning: str
confidence: float
_ASSIGNMENT_ACTIONS = {"monitor", "notify_customer", "ops_alert", "escalate"}
_ASSIGNMENT_SYSTEM = """You are the dispatch controller for a last-mile delivery network.
The backend tried to assign a booking to a miler (rider) — both its AI assignment and the
fallback failed, so no rider was assigned. Decide how to react. Choose exactly one action:
- "monitor": likely a transient gap (a momentary lack of free riders); the backend will retry. Take no action.
- "notify_customer": a real but recoverable delay; tell the customer we're finding a rider so they aren't left guessing.
- "ops_alert": a genuine coverage gap in this zone (repeated failures, no nearby riders). Alert operations to onboard or redirect riders here.
- "escalate": ambiguous or contradictory — e.g. riders ARE nearby yet assignment keeps failing, which suggests a systemic issue rather than a coverage gap. A human dispatcher should look.
Weigh how many times assignment has failed in this zone today, how far the nearest rider is and whether
that rider is actually on duty, the time of day, and whether coordinates were even available.
Read the rider facts carefully:
- "nearest_miler_within_km": distance to the nearest rider in the location index.
- "nearest_miler_presence": "available" = the backend confirms that rider is on duty;
"unavailable" = confirmed off duty / on break; "unknown" = there is NO presence data for nearby
riders. Unknown is not evidence of a coverage gap — do not treat it as "nobody is working".
- "milers_in_geo_index_within_30km": raw count of riders ever seen nearby, including stale entries.
A single failure with a rider nearby is usually transient. Repeated failures with no rider at all within
30 km is a genuine coverage gap. Repeated failures *despite* a confirmed-available rider nearby is not a
coverage gap but a systemic issue — escalate to a human. When presence is unknown, lean toward monitoring
or escalating for a human to check rather than declaring a coverage gap.
Report an honest confidence in [0, 1]."""
_ASSIGNMENT_SCHEMA = {
"type": "object",
"properties": {
"action": {"type": "string", "enum": sorted(_ASSIGNMENT_ACTIONS)},
"reasoning": {"type": "string"},
"confidence": {"type": "number"},
},
"required": ["action", "reasoning", "confidence"],
"additionalProperties": False,
}
def build_assignment_failure_context(facts, *, now=None) -> str:
"""Assemble the prompt context for an assignment-failure decision. Shared by
DispatchAgent and the eval harness so the eval tests exactly what runs."""
now = now or datetime.now(timezone.utc)
lines = [
"A booking could not be assigned to a miler — the backend's AI and fallback assignment both failed.",
f"- current_time_utc: {now.isoformat()}",
f"- hour_of_day_utc: {now.hour}",
]
for key, value in (facts or {}).items():
lines.append(f"- {key}: {value}")
lines.append("Decide the best response.")
return "\n".join(lines)
async def decide_assignment_failure(context: str) -> Optional[AssignmentDecision]:
"""Ask Claude how to react to a failed assignment. None on failure (caller
falls back to deterministic behaviour)."""
r = await _decide(_ASSIGNMENT_SYSTEM, _ASSIGNMENT_SCHEMA, _ASSIGNMENT_ACTIONS, context)
return AssignmentDecision(*r) if r else None