247 lines
11 KiB
Python
247 lines
11 KiB
Python
"""
|
|
LLM-backed decision-making for agents.
|
|
|
|
Currently backed by Claude (Anthropic). It is deliberately kept behind a thin
|
|
function boundary (`decide_stall_response`) and a `dataclass` result so the
|
|
model — or the whole provider — can be swapped later (e.g. a self-hosted model)
|
|
without touching any agent logic.
|
|
|
|
Configuration (env):
|
|
ANTHROPIC_API_KEY required — the Anthropic API key
|
|
LLM_MODEL default "claude-opus-4-8"
|
|
LLM_EFFORT default "high" (low | medium | high | xhigh | max)
|
|
"""
|
|
import json
|
|
import os
|
|
from dataclasses import dataclass
|
|
from datetime import datetime, timezone
|
|
from typing import Optional
|
|
|
|
from core.logger import logger
|
|
|
|
LLM_MODEL = os.getenv("LLM_MODEL", "claude-opus-4-8")
|
|
LLM_EFFORT = os.getenv("LLM_EFFORT", "high")
|
|
# Thinking tokens count against max_tokens; adaptive thinking at high effort can
|
|
# use a lot, so the budget must be generous or the JSON output gets truncated.
|
|
LLM_MAX_TOKENS = int(os.getenv("LLM_MAX_TOKENS", "8192"))
|
|
LLM_TIMEOUT_S = float(os.getenv("LLM_TIMEOUT_S", "30"))
|
|
|
|
_client = None
|
|
|
|
|
|
def _get_client():
|
|
"""Lazily construct the *async* Anthropic client so a missing dependency or
|
|
key never breaks agent startup — only the actual decision call fails, and the
|
|
caller falls back to deterministic behaviour. Async so the decision never
|
|
blocks the event loop (a sync client would freeze every other coroutine for
|
|
the full duration of the call)."""
|
|
global _client
|
|
if _client is None:
|
|
import anthropic # lazy import
|
|
_client = anthropic.AsyncAnthropic() # reads ANTHROPIC_API_KEY from env
|
|
return _client
|
|
|
|
|
|
@dataclass
|
|
class StallDecision:
|
|
action: str # "reassign" | "notify_only" | "wait" | "escalate"
|
|
reasoning: str
|
|
confidence: float # 0.0 .. 1.0
|
|
|
|
|
|
DEFAULT_STALL_THRESHOLD_MIN = 10
|
|
|
|
|
|
def build_stall_context(minutes_stalled, *, stall_threshold_min=DEFAULT_STALL_THRESHOLD_MIN,
|
|
now=None, facts=None) -> str:
|
|
"""Assemble the prompt context for a stall decision.
|
|
|
|
Shared by the ExceptionAgent and the eval harness so the eval tests exactly
|
|
what runs in production. `facts` is an optional dict of extra situational
|
|
detail (location, notes, prior-stall count, ...) — today the agent passes
|
|
none, but context tools will populate it later.
|
|
"""
|
|
now = now or datetime.now(timezone.utc)
|
|
lines = [
|
|
"A miler has stalled during an active delivery.",
|
|
f"- minutes_stalled: {round(float(minutes_stalled), 1)}",
|
|
f"- stall_threshold_minutes: {stall_threshold_min}",
|
|
f"- current_time_utc: {now.isoformat()}",
|
|
f"- hour_of_day_utc: {now.hour}",
|
|
]
|
|
for key, value in (facts or {}).items():
|
|
lines.append(f"- {key}: {value}")
|
|
lines.append("Decide the best response.")
|
|
return "\n".join(lines)
|
|
|
|
|
|
_VALID_ACTIONS = {"reassign", "notify_only", "wait", "escalate"}
|
|
|
|
_STALL_SYSTEM = """You are the exception controller for a last-mile delivery network.
|
|
A "miler" (delivery rider) assigned to an active booking has stopped moving for a while.
|
|
Decide the single best response. Choose exactly one action:
|
|
|
|
- "wait": the stall is plausibly benign (traffic, a short pickup wait, a customer handoff). Take no action yet.
|
|
- "notify_only": the miler can likely recover, but the customer deserves reassurance. Notify the customer; keep the miler.
|
|
- "reassign": the miler is genuinely stuck and unlikely to recover soon. Hand the booking to another miler. This is customer-visible and hard to reverse — choose it only when the evidence clearly supports it.
|
|
- "escalate": the situation is ambiguous or high-stakes; a human dispatcher should decide.
|
|
|
|
Weigh how long it has been stalled relative to the threshold, the time of day, and how far into the job it is.
|
|
Prefer the least disruptive action that fits the evidence. Report an honest confidence in [0, 1] —
|
|
a low confidence is a signal to escalate rather than act."""
|
|
|
|
_STALL_SCHEMA = {
|
|
"type": "object",
|
|
"properties": {
|
|
"action": {"type": "string", "enum": sorted(_VALID_ACTIONS)},
|
|
"reasoning": {"type": "string"},
|
|
"confidence": {"type": "number"},
|
|
},
|
|
"required": ["action", "reasoning", "confidence"],
|
|
"additionalProperties": False,
|
|
}
|
|
|
|
|
|
def request_params(model: Optional[str] = None) -> dict:
|
|
"""The model-dependent part of a decision request.
|
|
|
|
`model` is a per-agent override from the agent registry; None means the
|
|
engine default (LLM_MODEL). Claude Haiku 4.5 rejects adaptive thinking and
|
|
`effort`, so both are left out for it — otherwise a Haiku pin set from the
|
|
console would make every decision fail and fall back to the heuristic.
|
|
"""
|
|
m = model or LLM_MODEL
|
|
output_config = {}
|
|
params = {"model": m}
|
|
if "haiku" not in m:
|
|
params["thinking"] = {"type": "adaptive"}
|
|
output_config["effort"] = LLM_EFFORT
|
|
params["output_config"] = output_config
|
|
return params
|
|
|
|
|
|
async def _decide(system: str, schema: dict, valid_actions, context: str, model: Optional[str] = None):
|
|
"""Run one structured-output decision against Claude.
|
|
|
|
Returns (action, reasoning, confidence), or None on any failure (network,
|
|
timeout, refusal, truncation, bad parse) so callers can fall back to
|
|
deterministic behaviour. Retries once on a transient error before giving up.
|
|
"""
|
|
resp = None
|
|
last_err = None
|
|
params = request_params(model)
|
|
params["output_config"]["format"] = {"type": "json_schema", "schema": schema}
|
|
for attempt in range(2): # initial try + one retry on transient failure
|
|
try:
|
|
client = _get_client()
|
|
resp = await client.messages.create(
|
|
max_tokens=LLM_MAX_TOKENS,
|
|
system=system,
|
|
messages=[{"role": "user", "content": context}],
|
|
timeout=LLM_TIMEOUT_S,
|
|
**params,
|
|
)
|
|
break
|
|
except Exception as e:
|
|
last_err = e
|
|
logger.warning(f"LLM decision request failed (attempt {attempt + 1}/2): {e}")
|
|
if resp is None:
|
|
logger.error(f"LLM decision failed after retries: {last_err}")
|
|
return None
|
|
|
|
stop = getattr(resp, "stop_reason", None)
|
|
if stop == "refusal":
|
|
logger.warning("LLM refused the decision request")
|
|
return None
|
|
if stop == "max_tokens":
|
|
logger.error("LLM decision truncated (max_tokens); raise LLM_MAX_TOKENS. Treating as no decision.")
|
|
return None
|
|
|
|
try:
|
|
text = next(b.text for b in resp.content if b.type == "text")
|
|
data = json.loads(text)
|
|
action = data["action"]
|
|
if action not in valid_actions:
|
|
raise ValueError(f"unexpected action {action!r}")
|
|
return action, str(data.get("reasoning", "")), float(data.get("confidence", 0.0))
|
|
except (StopIteration, KeyError, ValueError, json.JSONDecodeError) as e:
|
|
logger.error(f"LLM decision parse failed: {e}")
|
|
return None
|
|
|
|
|
|
async def decide_stall_response(context: str, model: Optional[str] = None) -> Optional[StallDecision]:
|
|
"""Ask Claude how to handle a stalled miler. None on failure (caller falls
|
|
back to deterministic behaviour). `model` overrides LLM_MODEL."""
|
|
r = await _decide(_STALL_SYSTEM, _STALL_SCHEMA, _VALID_ACTIONS, context, model)
|
|
return StallDecision(*r) if r else None
|
|
|
|
|
|
# ── Assignment-failure decision (DispatchAgent) ──────────────────────────────
|
|
|
|
@dataclass
|
|
class AssignmentDecision:
|
|
action: str # "monitor" | "notify_customer" | "ops_alert" | "escalate"
|
|
reasoning: str
|
|
confidence: float
|
|
|
|
|
|
_ASSIGNMENT_ACTIONS = {"monitor", "notify_customer", "ops_alert", "escalate"}
|
|
|
|
_ASSIGNMENT_SYSTEM = """You are the dispatch controller for a last-mile delivery network.
|
|
The backend tried to assign a booking to a miler (rider) — both its AI assignment and the
|
|
fallback failed, so no rider was assigned. Decide how to react. Choose exactly one action:
|
|
|
|
- "monitor": likely a transient gap (a momentary lack of free riders); the backend will retry. Take no action.
|
|
- "notify_customer": a real but recoverable delay; tell the customer we're finding a rider so they aren't left guessing.
|
|
- "ops_alert": a genuine coverage gap in this zone (repeated failures, no nearby riders). Alert operations to onboard or redirect riders here.
|
|
- "escalate": ambiguous or contradictory — e.g. riders ARE nearby yet assignment keeps failing, which suggests a systemic issue rather than a coverage gap. A human dispatcher should look.
|
|
|
|
Weigh how many times assignment has failed in this zone today, how far the nearest rider is and whether
|
|
that rider is actually on duty, the time of day, and whether coordinates were even available.
|
|
|
|
Read the rider facts carefully:
|
|
- "nearest_miler_within_km": distance to the nearest rider in the location index.
|
|
- "nearest_miler_presence": "available" = the backend confirms that rider is on duty;
|
|
"unavailable" = confirmed off duty / on break; "unknown" = there is NO presence data for nearby
|
|
riders. Unknown is not evidence of a coverage gap — do not treat it as "nobody is working".
|
|
- "milers_in_geo_index_within_30km": raw count of riders ever seen nearby, including stale entries.
|
|
|
|
A single failure with a rider nearby is usually transient. Repeated failures with no rider at all within
|
|
30 km is a genuine coverage gap. Repeated failures *despite* a confirmed-available rider nearby is not a
|
|
coverage gap but a systemic issue — escalate to a human. When presence is unknown, lean toward monitoring
|
|
or escalating for a human to check rather than declaring a coverage gap.
|
|
Report an honest confidence in [0, 1]."""
|
|
|
|
_ASSIGNMENT_SCHEMA = {
|
|
"type": "object",
|
|
"properties": {
|
|
"action": {"type": "string", "enum": sorted(_ASSIGNMENT_ACTIONS)},
|
|
"reasoning": {"type": "string"},
|
|
"confidence": {"type": "number"},
|
|
},
|
|
"required": ["action", "reasoning", "confidence"],
|
|
"additionalProperties": False,
|
|
}
|
|
|
|
|
|
def build_assignment_failure_context(facts, *, now=None) -> str:
|
|
"""Assemble the prompt context for an assignment-failure decision. Shared by
|
|
DispatchAgent and the eval harness so the eval tests exactly what runs."""
|
|
now = now or datetime.now(timezone.utc)
|
|
lines = [
|
|
"A booking could not be assigned to a miler — the backend's AI and fallback assignment both failed.",
|
|
f"- current_time_utc: {now.isoformat()}",
|
|
f"- hour_of_day_utc: {now.hour}",
|
|
]
|
|
for key, value in (facts or {}).items():
|
|
lines.append(f"- {key}: {value}")
|
|
lines.append("Decide the best response.")
|
|
return "\n".join(lines)
|
|
|
|
|
|
async def decide_assignment_failure(context: str, model: Optional[str] = None) -> Optional[AssignmentDecision]:
|
|
"""Ask Claude how to react to a failed assignment. None on failure (caller
|
|
falls back to deterministic behaviour). `model` overrides LLM_MODEL."""
|
|
r = await _decide(_ASSIGNMENT_SYSTEM, _ASSIGNMENT_SCHEMA, _ASSIGNMENT_ACTIONS, context, model)
|
|
return AssignmentDecision(*r) if r else None
|