Files
AI_engine/core/llm.py

221 lines
9.4 KiB
Python

"""
LLM-backed decision-making for agents.
Currently backed by Claude (Anthropic). It is deliberately kept behind a thin
function boundary (`decide_stall_response`) and a `dataclass` result so the
model — or the whole provider — can be swapped later (e.g. a self-hosted model)
without touching any agent logic.
Configuration (env):
ANTHROPIC_API_KEY required — the Anthropic API key
LLM_MODEL default "claude-opus-4-8"
LLM_EFFORT default "high" (low | medium | high | xhigh | max)
"""
import json
import os
from dataclasses import dataclass
from datetime import datetime, timezone
from typing import Optional
from core.logger import logger
LLM_MODEL = os.getenv("LLM_MODEL", "claude-opus-4-8")
LLM_EFFORT = os.getenv("LLM_EFFORT", "high")
# Thinking tokens count against max_tokens; adaptive thinking at high effort can
# use a lot, so the budget must be generous or the JSON output gets truncated.
LLM_MAX_TOKENS = int(os.getenv("LLM_MAX_TOKENS", "8192"))
LLM_TIMEOUT_S = float(os.getenv("LLM_TIMEOUT_S", "30"))
_client = None
def _get_client():
"""Lazily construct the *async* Anthropic client so a missing dependency or
key never breaks agent startup — only the actual decision call fails, and the
caller falls back to deterministic behaviour. Async so the decision never
blocks the event loop (a sync client would freeze every other coroutine for
the full duration of the call)."""
global _client
if _client is None:
import anthropic # lazy import
_client = anthropic.AsyncAnthropic() # reads ANTHROPIC_API_KEY from env
return _client
@dataclass
class StallDecision:
action: str # "reassign" | "notify_only" | "wait" | "escalate"
reasoning: str
confidence: float # 0.0 .. 1.0
DEFAULT_STALL_THRESHOLD_MIN = 10
def build_stall_context(minutes_stalled, *, stall_threshold_min=DEFAULT_STALL_THRESHOLD_MIN,
now=None, facts=None) -> str:
"""Assemble the prompt context for a stall decision.
Shared by the ExceptionAgent and the eval harness so the eval tests exactly
what runs in production. `facts` is an optional dict of extra situational
detail (location, notes, prior-stall count, ...) — today the agent passes
none, but context tools will populate it later.
"""
now = now or datetime.now(timezone.utc)
lines = [
"A miler has stalled during an active delivery.",
f"- minutes_stalled: {round(float(minutes_stalled), 1)}",
f"- stall_threshold_minutes: {stall_threshold_min}",
f"- current_time_utc: {now.isoformat()}",
f"- hour_of_day_utc: {now.hour}",
]
for key, value in (facts or {}).items():
lines.append(f"- {key}: {value}")
lines.append("Decide the best response.")
return "\n".join(lines)
_VALID_ACTIONS = {"reassign", "notify_only", "wait", "escalate"}
_STALL_SYSTEM = """You are the exception controller for a last-mile delivery network.
A "miler" (delivery rider) assigned to an active booking has stopped moving for a while.
Decide the single best response. Choose exactly one action:
- "wait": the stall is plausibly benign (traffic, a short pickup wait, a customer handoff). Take no action yet.
- "notify_only": the miler can likely recover, but the customer deserves reassurance. Notify the customer; keep the miler.
- "reassign": the miler is genuinely stuck and unlikely to recover soon. Hand the booking to another miler. This is customer-visible and hard to reverse — choose it only when the evidence clearly supports it.
- "escalate": the situation is ambiguous or high-stakes; a human dispatcher should decide.
Weigh how long it has been stalled relative to the threshold, the time of day, and how far into the job it is.
Prefer the least disruptive action that fits the evidence. Report an honest confidence in [0, 1] —
a low confidence is a signal to escalate rather than act."""
_STALL_SCHEMA = {
"type": "object",
"properties": {
"action": {"type": "string", "enum": sorted(_VALID_ACTIONS)},
"reasoning": {"type": "string"},
"confidence": {"type": "number"},
},
"required": ["action", "reasoning", "confidence"],
"additionalProperties": False,
}
async def _decide(system: str, schema: dict, valid_actions, context: str):
"""Run one structured-output decision against Claude.
Returns (action, reasoning, confidence), or None on any failure (network,
timeout, refusal, truncation, bad parse) so callers can fall back to
deterministic behaviour. Retries once on a transient error before giving up.
"""
resp = None
last_err = None
for attempt in range(2): # initial try + one retry on transient failure
try:
client = _get_client()
resp = await client.messages.create(
model=LLM_MODEL,
max_tokens=LLM_MAX_TOKENS,
system=system,
thinking={"type": "adaptive"},
output_config={
"effort": LLM_EFFORT,
"format": {"type": "json_schema", "schema": schema},
},
messages=[{"role": "user", "content": context}],
timeout=LLM_TIMEOUT_S,
)
break
except Exception as e:
last_err = e
logger.warning(f"LLM decision request failed (attempt {attempt + 1}/2): {e}")
if resp is None:
logger.error(f"LLM decision failed after retries: {last_err}")
return None
stop = getattr(resp, "stop_reason", None)
if stop == "refusal":
logger.warning("LLM refused the decision request")
return None
if stop == "max_tokens":
logger.error("LLM decision truncated (max_tokens); raise LLM_MAX_TOKENS. Treating as no decision.")
return None
try:
text = next(b.text for b in resp.content if b.type == "text")
data = json.loads(text)
action = data["action"]
if action not in valid_actions:
raise ValueError(f"unexpected action {action!r}")
return action, str(data.get("reasoning", "")), float(data.get("confidence", 0.0))
except (StopIteration, KeyError, ValueError, json.JSONDecodeError) as e:
logger.error(f"LLM decision parse failed: {e}")
return None
async def decide_stall_response(context: str) -> Optional[StallDecision]:
"""Ask Claude how to handle a stalled miler. None on failure (caller falls
back to deterministic behaviour)."""
r = await _decide(_STALL_SYSTEM, _STALL_SCHEMA, _VALID_ACTIONS, context)
return StallDecision(*r) if r else None
# ── Assignment-failure decision (DispatchAgent) ──────────────────────────────
@dataclass
class AssignmentDecision:
action: str # "monitor" | "notify_customer" | "ops_alert" | "escalate"
reasoning: str
confidence: float
_ASSIGNMENT_ACTIONS = {"monitor", "notify_customer", "ops_alert", "escalate"}
_ASSIGNMENT_SYSTEM = """You are the dispatch controller for a last-mile delivery network.
The backend tried to assign a booking to a miler (rider) — both its AI assignment and the
fallback failed, so no rider was assigned. Decide how to react. Choose exactly one action:
- "monitor": likely a transient gap (a momentary lack of free riders); the backend will retry. Take no action.
- "notify_customer": a real but recoverable delay; tell the customer we're finding a rider so they aren't left guessing.
- "ops_alert": a genuine coverage gap in this zone (repeated failures, no nearby riders). Alert operations to onboard or redirect riders here.
- "escalate": ambiguous or contradictory — e.g. riders ARE nearby yet assignment keeps failing, which suggests a systemic issue rather than a coverage gap. A human dispatcher should look.
Weigh how many times assignment has failed in this zone today, how far the nearest available rider is,
the time of day, and whether coordinates were even available. A single failure with a rider nearby is
usually transient; repeated failures with no nearby rider is a coverage gap; repeated failures *despite*
nearby riders is not a coverage gap and warrants a human. Report an honest confidence in [0, 1]."""
_ASSIGNMENT_SCHEMA = {
"type": "object",
"properties": {
"action": {"type": "string", "enum": sorted(_ASSIGNMENT_ACTIONS)},
"reasoning": {"type": "string"},
"confidence": {"type": "number"},
},
"required": ["action", "reasoning", "confidence"],
"additionalProperties": False,
}
def build_assignment_failure_context(facts, *, now=None) -> str:
"""Assemble the prompt context for an assignment-failure decision. Shared by
DispatchAgent and the eval harness so the eval tests exactly what runs."""
now = now or datetime.now(timezone.utc)
lines = [
"A booking could not be assigned to a miler — the backend's AI and fallback assignment both failed.",
f"- current_time_utc: {now.isoformat()}",
f"- hour_of_day_utc: {now.hour}",
]
for key, value in (facts or {}).items():
lines.append(f"- {key}: {value}")
lines.append("Decide the best response.")
return "\n".join(lines)
async def decide_assignment_failure(context: str) -> Optional[AssignmentDecision]:
"""Ask Claude how to react to a failed assignment. None on failure (caller
falls back to deterministic behaviour)."""
r = await _decide(_ASSIGNMENT_SYSTEM, _ASSIGNMENT_SCHEMA, _ASSIGNMENT_ACTIONS, context)
return AssignmentDecision(*r) if r else None