""" LLM-backed decision-making for agents. Currently backed by Claude (Anthropic). It is deliberately kept behind a thin function boundary (`decide_stall_response`) and a `dataclass` result so the model — or the whole provider — can be swapped later (e.g. a self-hosted model) without touching any agent logic. Configuration (env): ANTHROPIC_API_KEY required — the Anthropic API key LLM_MODEL default "claude-opus-4-8" LLM_EFFORT default "high" (low | medium | high | xhigh | max) """ import json import os from dataclasses import dataclass from datetime import datetime, timezone from typing import Optional from core.logger import logger LLM_MODEL = os.getenv("LLM_MODEL", "claude-opus-4-8") LLM_EFFORT = os.getenv("LLM_EFFORT", "high") # Thinking tokens count against max_tokens; adaptive thinking at high effort can # use a lot, so the budget must be generous or the JSON output gets truncated. LLM_MAX_TOKENS = int(os.getenv("LLM_MAX_TOKENS", "8192")) LLM_TIMEOUT_S = float(os.getenv("LLM_TIMEOUT_S", "30")) _client = None def _get_client(): """Lazily construct the *async* Anthropic client so a missing dependency or key never breaks agent startup — only the actual decision call fails, and the caller falls back to deterministic behaviour. Async so the decision never blocks the event loop (a sync client would freeze every other coroutine for the full duration of the call).""" global _client if _client is None: import anthropic # lazy import _client = anthropic.AsyncAnthropic() # reads ANTHROPIC_API_KEY from env return _client @dataclass class StallDecision: action: str # "reassign" | "notify_only" | "wait" | "escalate" reasoning: str confidence: float # 0.0 .. 1.0 DEFAULT_STALL_THRESHOLD_MIN = 10 def build_stall_context(minutes_stalled, *, stall_threshold_min=DEFAULT_STALL_THRESHOLD_MIN, now=None, facts=None) -> str: """Assemble the prompt context for a stall decision. Shared by the ExceptionAgent and the eval harness so the eval tests exactly what runs in production. `facts` is an optional dict of extra situational detail (location, notes, prior-stall count, ...) — today the agent passes none, but context tools will populate it later. """ now = now or datetime.now(timezone.utc) lines = [ "A miler has stalled during an active delivery.", f"- minutes_stalled: {round(float(minutes_stalled), 1)}", f"- stall_threshold_minutes: {stall_threshold_min}", f"- current_time_utc: {now.isoformat()}", f"- hour_of_day_utc: {now.hour}", ] for key, value in (facts or {}).items(): lines.append(f"- {key}: {value}") lines.append("Decide the best response.") return "\n".join(lines) _VALID_ACTIONS = {"reassign", "notify_only", "wait", "escalate"} _STALL_SYSTEM = """You are the exception controller for a last-mile delivery network. A "miler" (delivery rider) assigned to an active booking has stopped moving for a while. Decide the single best response. Choose exactly one action: - "wait": the stall is plausibly benign (traffic, a short pickup wait, a customer handoff). Take no action yet. - "notify_only": the miler can likely recover, but the customer deserves reassurance. Notify the customer; keep the miler. - "reassign": the miler is genuinely stuck and unlikely to recover soon. Hand the booking to another miler. This is customer-visible and hard to reverse — choose it only when the evidence clearly supports it. - "escalate": the situation is ambiguous or high-stakes; a human dispatcher should decide. Weigh how long it has been stalled relative to the threshold, the time of day, and how far into the job it is. Prefer the least disruptive action that fits the evidence. Report an honest confidence in [0, 1] — a low confidence is a signal to escalate rather than act.""" _STALL_SCHEMA = { "type": "object", "properties": { "action": {"type": "string", "enum": sorted(_VALID_ACTIONS)}, "reasoning": {"type": "string"}, "confidence": {"type": "number"}, }, "required": ["action", "reasoning", "confidence"], "additionalProperties": False, } async def _decide(system: str, schema: dict, valid_actions, context: str): """Run one structured-output decision against Claude. Returns (action, reasoning, confidence), or None on any failure (network, timeout, refusal, truncation, bad parse) so callers can fall back to deterministic behaviour. Retries once on a transient error before giving up. """ resp = None last_err = None for attempt in range(2): # initial try + one retry on transient failure try: client = _get_client() resp = await client.messages.create( model=LLM_MODEL, max_tokens=LLM_MAX_TOKENS, system=system, thinking={"type": "adaptive"}, output_config={ "effort": LLM_EFFORT, "format": {"type": "json_schema", "schema": schema}, }, messages=[{"role": "user", "content": context}], timeout=LLM_TIMEOUT_S, ) break except Exception as e: last_err = e logger.warning(f"LLM decision request failed (attempt {attempt + 1}/2): {e}") if resp is None: logger.error(f"LLM decision failed after retries: {last_err}") return None stop = getattr(resp, "stop_reason", None) if stop == "refusal": logger.warning("LLM refused the decision request") return None if stop == "max_tokens": logger.error("LLM decision truncated (max_tokens); raise LLM_MAX_TOKENS. Treating as no decision.") return None try: text = next(b.text for b in resp.content if b.type == "text") data = json.loads(text) action = data["action"] if action not in valid_actions: raise ValueError(f"unexpected action {action!r}") return action, str(data.get("reasoning", "")), float(data.get("confidence", 0.0)) except (StopIteration, KeyError, ValueError, json.JSONDecodeError) as e: logger.error(f"LLM decision parse failed: {e}") return None async def decide_stall_response(context: str) -> Optional[StallDecision]: """Ask Claude how to handle a stalled miler. None on failure (caller falls back to deterministic behaviour).""" r = await _decide(_STALL_SYSTEM, _STALL_SCHEMA, _VALID_ACTIONS, context) return StallDecision(*r) if r else None # ── Assignment-failure decision (DispatchAgent) ────────────────────────────── @dataclass class AssignmentDecision: action: str # "monitor" | "notify_customer" | "ops_alert" | "escalate" reasoning: str confidence: float _ASSIGNMENT_ACTIONS = {"monitor", "notify_customer", "ops_alert", "escalate"} _ASSIGNMENT_SYSTEM = """You are the dispatch controller for a last-mile delivery network. The backend tried to assign a booking to a miler (rider) — both its AI assignment and the fallback failed, so no rider was assigned. Decide how to react. Choose exactly one action: - "monitor": likely a transient gap (a momentary lack of free riders); the backend will retry. Take no action. - "notify_customer": a real but recoverable delay; tell the customer we're finding a rider so they aren't left guessing. - "ops_alert": a genuine coverage gap in this zone (repeated failures, no nearby riders). Alert operations to onboard or redirect riders here. - "escalate": ambiguous or contradictory — e.g. riders ARE nearby yet assignment keeps failing, which suggests a systemic issue rather than a coverage gap. A human dispatcher should look. Weigh how many times assignment has failed in this zone today, how far the nearest rider is and whether that rider is actually on duty, the time of day, and whether coordinates were even available. Read the rider facts carefully: - "nearest_miler_within_km": distance to the nearest rider in the location index. - "nearest_miler_presence": "available" = the backend confirms that rider is on duty; "unavailable" = confirmed off duty / on break; "unknown" = there is NO presence data for nearby riders. Unknown is not evidence of a coverage gap — do not treat it as "nobody is working". - "milers_in_geo_index_within_30km": raw count of riders ever seen nearby, including stale entries. A single failure with a rider nearby is usually transient. Repeated failures with no rider at all within 30 km is a genuine coverage gap. Repeated failures *despite* a confirmed-available rider nearby is not a coverage gap but a systemic issue — escalate to a human. When presence is unknown, lean toward monitoring or escalating for a human to check rather than declaring a coverage gap. Report an honest confidence in [0, 1].""" _ASSIGNMENT_SCHEMA = { "type": "object", "properties": { "action": {"type": "string", "enum": sorted(_ASSIGNMENT_ACTIONS)}, "reasoning": {"type": "string"}, "confidence": {"type": "number"}, }, "required": ["action", "reasoning", "confidence"], "additionalProperties": False, } def build_assignment_failure_context(facts, *, now=None) -> str: """Assemble the prompt context for an assignment-failure decision. Shared by DispatchAgent and the eval harness so the eval tests exactly what runs.""" now = now or datetime.now(timezone.utc) lines = [ "A booking could not be assigned to a miler — the backend's AI and fallback assignment both failed.", f"- current_time_utc: {now.isoformat()}", f"- hour_of_day_utc: {now.hour}", ] for key, value in (facts or {}).items(): lines.append(f"- {key}: {value}") lines.append("Decide the best response.") return "\n".join(lines) async def decide_assignment_failure(context: str) -> Optional[AssignmentDecision]: """Ask Claude how to react to a failed assignment. None on failure (caller falls back to deterministic behaviour).""" r = await _decide(_ASSIGNMENT_SYSTEM, _ASSIGNMENT_SCHEMA, _ASSIGNMENT_ACTIONS, context) return AssignmentDecision(*r) if r else None