route optimizer agent with dailygrubs ai assign
This commit is contained in:
@@ -1,5 +1,6 @@
|
||||
"""Base Agent class - All agents inherit from this."""
|
||||
import asyncio
|
||||
import time
|
||||
import uuid
|
||||
from datetime import datetime
|
||||
from typing import Dict, List, Optional, Any
|
||||
@@ -28,6 +29,7 @@ class Agent(ABC):
|
||||
self._task_queue: asyncio.Queue = asyncio.Queue()
|
||||
self._running = False
|
||||
self._task_handlers: Dict[str, callable] = {}
|
||||
self._last_state_emit = 0.0
|
||||
|
||||
# Register with message bus
|
||||
message_bus.register_agent(self)
|
||||
@@ -49,6 +51,9 @@ class Agent(ABC):
|
||||
except asyncio.TimeoutError:
|
||||
await self._heartbeat()
|
||||
|
||||
if time.monotonic() - self._last_state_emit >= 5.0:
|
||||
await self._emit_state()
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error in agent {self.agent_id}: {e}")
|
||||
self.state.status = "error"
|
||||
@@ -64,6 +69,8 @@ class Agent(ABC):
|
||||
"""Process a task from the queue."""
|
||||
self.state.status = "working"
|
||||
self.state.current_task = task.task_id
|
||||
started = time.monotonic()
|
||||
await self._emit_state()
|
||||
|
||||
try:
|
||||
if task.task_type in self._task_handlers:
|
||||
@@ -85,9 +92,19 @@ class Agent(ABC):
|
||||
self.state.tasks_failed += 1
|
||||
|
||||
finally:
|
||||
duration_ms = int((time.monotonic() - started) * 1000)
|
||||
await message_bus.publish_telemetry("task", {
|
||||
"agent_id": self.agent_id,
|
||||
"task_id": task.task_id,
|
||||
"task_type": task.task_type,
|
||||
"status": task.status,
|
||||
"error": task.error,
|
||||
"duration_ms": duration_ms,
|
||||
})
|
||||
self.state.current_task = None
|
||||
self.state.last_active = datetime.now()
|
||||
self.state.status = "idle"
|
||||
await self._emit_state()
|
||||
|
||||
@abstractmethod
|
||||
async def handle_task(self, task: AgentTask) -> Dict[str, Any]:
|
||||
@@ -98,6 +115,18 @@ class Agent(ABC):
|
||||
"""Called periodically when idle. Override for custom behavior."""
|
||||
pass
|
||||
|
||||
async def _emit_state(self):
|
||||
"""Publish this agent's live state as telemetry (for the command center)."""
|
||||
self._last_state_emit = time.monotonic()
|
||||
await message_bus.publish_telemetry("agent", {
|
||||
"agent_id": self.agent_id,
|
||||
"agent_type": self.agent_type,
|
||||
"status": self.state.status,
|
||||
"current_task": self.state.current_task,
|
||||
"tasks_completed": self.state.tasks_completed,
|
||||
"tasks_failed": self.state.tasks_failed,
|
||||
})
|
||||
|
||||
def register_task_handler(self, task_type: str, handler: callable):
|
||||
self._task_handlers[task_type] = handler
|
||||
|
||||
@@ -124,6 +153,39 @@ class Agent(ABC):
|
||||
async def receive_messages(self) -> List[AgentMessage]:
|
||||
return await message_bus.get_messages(self.agent_id)
|
||||
|
||||
async def deliver(self, message: AgentMessage):
|
||||
"""Consume an inbound directed message.
|
||||
|
||||
This is what makes agent-to-agent messaging actually work: the message
|
||||
bus calls it for every directed message addressed to this agent. Messages
|
||||
that carry a ``task_type`` in their payload are enqueued as tasks so
|
||||
``handle_task`` processes them exactly like a submitted task; everything
|
||||
else is handed to ``handle_message`` so the agent can react. Without this,
|
||||
directed messages sat unread in the bus queue forever.
|
||||
"""
|
||||
payload = message.payload if isinstance(message.payload, dict) else {}
|
||||
task_type = payload.get("task_type")
|
||||
if task_type:
|
||||
await self._task_queue.put(AgentTask(
|
||||
task_id=f"{message.correlation_id or message.message_id}:{task_type}",
|
||||
agent_type=self.agent_type,
|
||||
task_type=task_type,
|
||||
data=payload,
|
||||
))
|
||||
else:
|
||||
try:
|
||||
await self.handle_message(message)
|
||||
except Exception as e:
|
||||
logger.error(f"{self.agent_id} handle_message error: {e}")
|
||||
|
||||
async def handle_message(self, message: AgentMessage):
|
||||
"""React to a non-task directed message (a notification). Default just
|
||||
logs it; agents override this to act on events like EXCEPTION_DETECTED."""
|
||||
logger.debug(
|
||||
f"{self.agent_id} received {message.message_type.value} "
|
||||
f"from {message.sender} (no task_type; not handled)"
|
||||
)
|
||||
|
||||
def subscribe_to(self, message_type: MessageType, callback: callable):
|
||||
message_bus.subscribe(message_type, callback)
|
||||
|
||||
@@ -151,6 +213,24 @@ class MasterAgent(Agent):
|
||||
self._sub_agents[agent.agent_id] = agent
|
||||
logger.info(f"JARVIS registered sub-agent: {agent.agent_id}")
|
||||
|
||||
async def handle_message(self, message: AgentMessage):
|
||||
"""Surface notifications from sub-agents. Escalations (proposals a human
|
||||
must decide) are logged at WARNING and recorded so they are visible
|
||||
rather than silently dropped — the endpoint of the human-review path."""
|
||||
if message.message_type == MessageType.EXCEPTION_DETECTED:
|
||||
logger.warning(f"JARVIS: exception escalation from {message.sender}: {message.payload}")
|
||||
else:
|
||||
logger.info(f"JARVIS: {message.message_type.value} from {message.sender}")
|
||||
self._decision_log.append({
|
||||
"timestamp": datetime.now(),
|
||||
"action": "received_notification",
|
||||
"from": message.sender,
|
||||
"type": message.message_type.value,
|
||||
"payload": message.payload,
|
||||
})
|
||||
if len(self._decision_log) > 500:
|
||||
self._decision_log = self._decision_log[-500:]
|
||||
|
||||
async def handle_task(self, task: AgentTask) -> Dict[str, Any]:
|
||||
if task.task_type == "orchestrate_order":
|
||||
return await self._orchestrate_order(task)
|
||||
|
||||
@@ -43,8 +43,22 @@ async def api_patch(url: str, **kwargs) -> Optional[Dict[str, Any]]:
|
||||
return await _request("patch", url, **kwargs)
|
||||
|
||||
|
||||
async def _safe_text(resp) -> str:
|
||||
try:
|
||||
return (await resp.text())[:300]
|
||||
except Exception:
|
||||
return "<unreadable body>"
|
||||
|
||||
|
||||
async def _request(method: str, url: str, **kwargs) -> Optional[Dict[str, Any]]:
|
||||
"""Execute an HTTP request with up to 3 attempts and exponential backoff."""
|
||||
"""Execute an HTTP request with up to 3 attempts and exponential backoff.
|
||||
|
||||
Returns the parsed body ONLY on a 2xx/3xx response. A 4xx is a client error
|
||||
and returns None immediately (no retry — it won't fix itself). A 5xx is
|
||||
retried like a network error, then returns None. This is the difference
|
||||
between "the action succeeded" and "the server rejected it": callers gate on
|
||||
a non-None result, so an error response must not look like success.
|
||||
"""
|
||||
max_retries = 3
|
||||
backoff = 1.0
|
||||
session = await get_session()
|
||||
@@ -52,6 +66,25 @@ async def _request(method: str, url: str, **kwargs) -> Optional[Dict[str, Any]]:
|
||||
for attempt in range(max_retries):
|
||||
try:
|
||||
async with getattr(session, method)(url, **kwargs) as resp:
|
||||
if resp.status >= 500:
|
||||
body = await _safe_text(resp)
|
||||
if attempt < max_retries - 1:
|
||||
logger.warning(
|
||||
f"{method.upper()} {url} -> {resp.status} (server error), "
|
||||
f"retrying in {backoff:.0f}s: {body}"
|
||||
)
|
||||
await asyncio.sleep(backoff)
|
||||
backoff *= 2
|
||||
continue
|
||||
logger.error(
|
||||
f"{method.upper()} {url} -> {resp.status} after {max_retries} attempts: {body}"
|
||||
)
|
||||
return None
|
||||
if resp.status >= 400:
|
||||
logger.error(
|
||||
f"{method.upper()} {url} -> {resp.status} (client error): {await _safe_text(resp)}"
|
||||
)
|
||||
return None
|
||||
if resp.content_type == "application/json":
|
||||
return await resp.json()
|
||||
return {"status_code": resp.status}
|
||||
|
||||
220
core/llm.py
Normal file
220
core/llm.py
Normal file
@@ -0,0 +1,220 @@
|
||||
"""
|
||||
LLM-backed decision-making for agents.
|
||||
|
||||
Currently backed by Claude (Anthropic). It is deliberately kept behind a thin
|
||||
function boundary (`decide_stall_response`) and a `dataclass` result so the
|
||||
model — or the whole provider — can be swapped later (e.g. a self-hosted model)
|
||||
without touching any agent logic.
|
||||
|
||||
Configuration (env):
|
||||
ANTHROPIC_API_KEY required — the Anthropic API key
|
||||
LLM_MODEL default "claude-opus-4-8"
|
||||
LLM_EFFORT default "high" (low | medium | high | xhigh | max)
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timezone
|
||||
from typing import Optional
|
||||
|
||||
from core.logger import logger
|
||||
|
||||
LLM_MODEL = os.getenv("LLM_MODEL", "claude-opus-4-8")
|
||||
LLM_EFFORT = os.getenv("LLM_EFFORT", "high")
|
||||
# Thinking tokens count against max_tokens; adaptive thinking at high effort can
|
||||
# use a lot, so the budget must be generous or the JSON output gets truncated.
|
||||
LLM_MAX_TOKENS = int(os.getenv("LLM_MAX_TOKENS", "8192"))
|
||||
LLM_TIMEOUT_S = float(os.getenv("LLM_TIMEOUT_S", "30"))
|
||||
|
||||
_client = None
|
||||
|
||||
|
||||
def _get_client():
|
||||
"""Lazily construct the *async* Anthropic client so a missing dependency or
|
||||
key never breaks agent startup — only the actual decision call fails, and the
|
||||
caller falls back to deterministic behaviour. Async so the decision never
|
||||
blocks the event loop (a sync client would freeze every other coroutine for
|
||||
the full duration of the call)."""
|
||||
global _client
|
||||
if _client is None:
|
||||
import anthropic # lazy import
|
||||
_client = anthropic.AsyncAnthropic() # reads ANTHROPIC_API_KEY from env
|
||||
return _client
|
||||
|
||||
|
||||
@dataclass
|
||||
class StallDecision:
|
||||
action: str # "reassign" | "notify_only" | "wait" | "escalate"
|
||||
reasoning: str
|
||||
confidence: float # 0.0 .. 1.0
|
||||
|
||||
|
||||
DEFAULT_STALL_THRESHOLD_MIN = 10
|
||||
|
||||
|
||||
def build_stall_context(minutes_stalled, *, stall_threshold_min=DEFAULT_STALL_THRESHOLD_MIN,
|
||||
now=None, facts=None) -> str:
|
||||
"""Assemble the prompt context for a stall decision.
|
||||
|
||||
Shared by the ExceptionAgent and the eval harness so the eval tests exactly
|
||||
what runs in production. `facts` is an optional dict of extra situational
|
||||
detail (location, notes, prior-stall count, ...) — today the agent passes
|
||||
none, but context tools will populate it later.
|
||||
"""
|
||||
now = now or datetime.now(timezone.utc)
|
||||
lines = [
|
||||
"A miler has stalled during an active delivery.",
|
||||
f"- minutes_stalled: {round(float(minutes_stalled), 1)}",
|
||||
f"- stall_threshold_minutes: {stall_threshold_min}",
|
||||
f"- current_time_utc: {now.isoformat()}",
|
||||
f"- hour_of_day_utc: {now.hour}",
|
||||
]
|
||||
for key, value in (facts or {}).items():
|
||||
lines.append(f"- {key}: {value}")
|
||||
lines.append("Decide the best response.")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
_VALID_ACTIONS = {"reassign", "notify_only", "wait", "escalate"}
|
||||
|
||||
_STALL_SYSTEM = """You are the exception controller for a last-mile delivery network.
|
||||
A "miler" (delivery rider) assigned to an active booking has stopped moving for a while.
|
||||
Decide the single best response. Choose exactly one action:
|
||||
|
||||
- "wait": the stall is plausibly benign (traffic, a short pickup wait, a customer handoff). Take no action yet.
|
||||
- "notify_only": the miler can likely recover, but the customer deserves reassurance. Notify the customer; keep the miler.
|
||||
- "reassign": the miler is genuinely stuck and unlikely to recover soon. Hand the booking to another miler. This is customer-visible and hard to reverse — choose it only when the evidence clearly supports it.
|
||||
- "escalate": the situation is ambiguous or high-stakes; a human dispatcher should decide.
|
||||
|
||||
Weigh how long it has been stalled relative to the threshold, the time of day, and how far into the job it is.
|
||||
Prefer the least disruptive action that fits the evidence. Report an honest confidence in [0, 1] —
|
||||
a low confidence is a signal to escalate rather than act."""
|
||||
|
||||
_STALL_SCHEMA = {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"action": {"type": "string", "enum": sorted(_VALID_ACTIONS)},
|
||||
"reasoning": {"type": "string"},
|
||||
"confidence": {"type": "number"},
|
||||
},
|
||||
"required": ["action", "reasoning", "confidence"],
|
||||
"additionalProperties": False,
|
||||
}
|
||||
|
||||
|
||||
async def _decide(system: str, schema: dict, valid_actions, context: str):
|
||||
"""Run one structured-output decision against Claude.
|
||||
|
||||
Returns (action, reasoning, confidence), or None on any failure (network,
|
||||
timeout, refusal, truncation, bad parse) so callers can fall back to
|
||||
deterministic behaviour. Retries once on a transient error before giving up.
|
||||
"""
|
||||
resp = None
|
||||
last_err = None
|
||||
for attempt in range(2): # initial try + one retry on transient failure
|
||||
try:
|
||||
client = _get_client()
|
||||
resp = await client.messages.create(
|
||||
model=LLM_MODEL,
|
||||
max_tokens=LLM_MAX_TOKENS,
|
||||
system=system,
|
||||
thinking={"type": "adaptive"},
|
||||
output_config={
|
||||
"effort": LLM_EFFORT,
|
||||
"format": {"type": "json_schema", "schema": schema},
|
||||
},
|
||||
messages=[{"role": "user", "content": context}],
|
||||
timeout=LLM_TIMEOUT_S,
|
||||
)
|
||||
break
|
||||
except Exception as e:
|
||||
last_err = e
|
||||
logger.warning(f"LLM decision request failed (attempt {attempt + 1}/2): {e}")
|
||||
if resp is None:
|
||||
logger.error(f"LLM decision failed after retries: {last_err}")
|
||||
return None
|
||||
|
||||
stop = getattr(resp, "stop_reason", None)
|
||||
if stop == "refusal":
|
||||
logger.warning("LLM refused the decision request")
|
||||
return None
|
||||
if stop == "max_tokens":
|
||||
logger.error("LLM decision truncated (max_tokens); raise LLM_MAX_TOKENS. Treating as no decision.")
|
||||
return None
|
||||
|
||||
try:
|
||||
text = next(b.text for b in resp.content if b.type == "text")
|
||||
data = json.loads(text)
|
||||
action = data["action"]
|
||||
if action not in valid_actions:
|
||||
raise ValueError(f"unexpected action {action!r}")
|
||||
return action, str(data.get("reasoning", "")), float(data.get("confidence", 0.0))
|
||||
except (StopIteration, KeyError, ValueError, json.JSONDecodeError) as e:
|
||||
logger.error(f"LLM decision parse failed: {e}")
|
||||
return None
|
||||
|
||||
|
||||
async def decide_stall_response(context: str) -> Optional[StallDecision]:
|
||||
"""Ask Claude how to handle a stalled miler. None on failure (caller falls
|
||||
back to deterministic behaviour)."""
|
||||
r = await _decide(_STALL_SYSTEM, _STALL_SCHEMA, _VALID_ACTIONS, context)
|
||||
return StallDecision(*r) if r else None
|
||||
|
||||
|
||||
# ── Assignment-failure decision (DispatchAgent) ──────────────────────────────
|
||||
|
||||
@dataclass
|
||||
class AssignmentDecision:
|
||||
action: str # "monitor" | "notify_customer" | "ops_alert" | "escalate"
|
||||
reasoning: str
|
||||
confidence: float
|
||||
|
||||
|
||||
_ASSIGNMENT_ACTIONS = {"monitor", "notify_customer", "ops_alert", "escalate"}
|
||||
|
||||
_ASSIGNMENT_SYSTEM = """You are the dispatch controller for a last-mile delivery network.
|
||||
The backend tried to assign a booking to a miler (rider) — both its AI assignment and the
|
||||
fallback failed, so no rider was assigned. Decide how to react. Choose exactly one action:
|
||||
|
||||
- "monitor": likely a transient gap (a momentary lack of free riders); the backend will retry. Take no action.
|
||||
- "notify_customer": a real but recoverable delay; tell the customer we're finding a rider so they aren't left guessing.
|
||||
- "ops_alert": a genuine coverage gap in this zone (repeated failures, no nearby riders). Alert operations to onboard or redirect riders here.
|
||||
- "escalate": ambiguous or contradictory — e.g. riders ARE nearby yet assignment keeps failing, which suggests a systemic issue rather than a coverage gap. A human dispatcher should look.
|
||||
|
||||
Weigh how many times assignment has failed in this zone today, how far the nearest available rider is,
|
||||
the time of day, and whether coordinates were even available. A single failure with a rider nearby is
|
||||
usually transient; repeated failures with no nearby rider is a coverage gap; repeated failures *despite*
|
||||
nearby riders is not a coverage gap and warrants a human. Report an honest confidence in [0, 1]."""
|
||||
|
||||
_ASSIGNMENT_SCHEMA = {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"action": {"type": "string", "enum": sorted(_ASSIGNMENT_ACTIONS)},
|
||||
"reasoning": {"type": "string"},
|
||||
"confidence": {"type": "number"},
|
||||
},
|
||||
"required": ["action", "reasoning", "confidence"],
|
||||
"additionalProperties": False,
|
||||
}
|
||||
|
||||
|
||||
def build_assignment_failure_context(facts, *, now=None) -> str:
|
||||
"""Assemble the prompt context for an assignment-failure decision. Shared by
|
||||
DispatchAgent and the eval harness so the eval tests exactly what runs."""
|
||||
now = now or datetime.now(timezone.utc)
|
||||
lines = [
|
||||
"A booking could not be assigned to a miler — the backend's AI and fallback assignment both failed.",
|
||||
f"- current_time_utc: {now.isoformat()}",
|
||||
f"- hour_of_day_utc: {now.hour}",
|
||||
]
|
||||
for key, value in (facts or {}).items():
|
||||
lines.append(f"- {key}: {value}")
|
||||
lines.append("Decide the best response.")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
async def decide_assignment_failure(context: str) -> Optional[AssignmentDecision]:
|
||||
"""Ask Claude how to react to a failed assignment. None on failure (caller
|
||||
falls back to deterministic behaviour)."""
|
||||
r = await _decide(_ASSIGNMENT_SYSTEM, _ASSIGNMENT_SCHEMA, _ASSIGNMENT_ACTIONS, context)
|
||||
return AssignmentDecision(*r) if r else None
|
||||
@@ -171,6 +171,22 @@ class MessageBus:
|
||||
) -> str:
|
||||
return await self.send_to_agent(sender, "ALL", message_type, payload, correlation_id)
|
||||
|
||||
async def _deliver_to_agent(self, agent_id: str, message: AgentMessage):
|
||||
"""Hand a directed message to a locally-registered agent so it is
|
||||
actually consumed (via its ``deliver``). Only if no such agent exists in
|
||||
this process does it fall back to the pull queue — so a message is never
|
||||
silently lost, but is delivered live whenever the recipient is present."""
|
||||
agent = self._agents.get(agent_id)
|
||||
if agent is not None and hasattr(agent, "deliver"):
|
||||
try:
|
||||
await agent.deliver(message)
|
||||
return
|
||||
except Exception as e:
|
||||
logger.error(f"Delivery to {agent_id} failed: {e}")
|
||||
return
|
||||
async with self._lock:
|
||||
self._queues[agent_id].append(QueuedMessage(message))
|
||||
|
||||
async def get_messages(self, agent_id: str) -> List[AgentMessage]:
|
||||
async with self._lock:
|
||||
queued = self._queues.pop(agent_id, [])
|
||||
@@ -179,6 +195,26 @@ class MessageBus:
|
||||
async def peek_messages(self, agent_id: str) -> List[AgentMessage]:
|
||||
return [q.message for q in self._queues.get(agent_id, [])]
|
||||
|
||||
# ------------------------------------------------------------------ #
|
||||
# Telemetry #
|
||||
# ------------------------------------------------------------------ #
|
||||
|
||||
async def publish_telemetry(self, kind: str, payload: Dict[str, Any]):
|
||||
"""
|
||||
Fire-and-forget observability event on plain NATS (subject `telemetry.<kind>`).
|
||||
|
||||
Deliberately NOT JetStream: telemetry is ephemeral fan-out for dashboards
|
||||
and must never accumulate in the persistent `logistics` stream. No-op
|
||||
when NATS is not connected — telemetry must never affect agent behavior.
|
||||
"""
|
||||
if self._nc is None or not self._nc.is_connected:
|
||||
return
|
||||
try:
|
||||
body = {"kind": kind, "ts": datetime.now().isoformat(), **payload}
|
||||
await self._nc.publish(f"telemetry.{kind}", json.dumps(body).encode())
|
||||
except Exception as e:
|
||||
logger.debug(f"Telemetry publish failed [{kind}]: {e}")
|
||||
|
||||
# ------------------------------------------------------------------ #
|
||||
# Hooks #
|
||||
# ------------------------------------------------------------------ #
|
||||
@@ -266,8 +302,7 @@ class MessageBus:
|
||||
async def handler(msg):
|
||||
try:
|
||||
agent_msg = AgentMessage.from_json(msg.data.decode())
|
||||
async with self._lock:
|
||||
self._queues[agent_id].append(QueuedMessage(agent_msg))
|
||||
await self._deliver_to_agent(agent_id, agent_msg)
|
||||
except Exception as e:
|
||||
logger.error(f"NATS agent-sub decode error [{agent_id}]: {e}")
|
||||
finally:
|
||||
@@ -282,8 +317,7 @@ class MessageBus:
|
||||
async def _local_dispatch(self, message: AgentMessage):
|
||||
"""In-process dispatch used when NATS is not connected."""
|
||||
if message.recipient != "ALL":
|
||||
async with self._lock:
|
||||
self._queues[message.recipient].append(QueuedMessage(message))
|
||||
await self._deliver_to_agent(message.recipient, message)
|
||||
else:
|
||||
for callback in list(self._subscribers.get(message.message_type, [])):
|
||||
try:
|
||||
|
||||
Reference in New Issue
Block a user