route optimizer agent with dailygrubs ai assign

This commit is contained in:
2026-09-01 13:48:58 +05:30
parent aa4d4c6549
commit 90cdb57a38
28 changed files with 3281 additions and 135 deletions

View File

@@ -1,5 +1,6 @@
"""Base Agent class - All agents inherit from this."""
import asyncio
import time
import uuid
from datetime import datetime
from typing import Dict, List, Optional, Any
@@ -28,6 +29,7 @@ class Agent(ABC):
self._task_queue: asyncio.Queue = asyncio.Queue()
self._running = False
self._task_handlers: Dict[str, callable] = {}
self._last_state_emit = 0.0
# Register with message bus
message_bus.register_agent(self)
@@ -49,6 +51,9 @@ class Agent(ABC):
except asyncio.TimeoutError:
await self._heartbeat()
if time.monotonic() - self._last_state_emit >= 5.0:
await self._emit_state()
except Exception as e:
logger.error(f"Error in agent {self.agent_id}: {e}")
self.state.status = "error"
@@ -64,6 +69,8 @@ class Agent(ABC):
"""Process a task from the queue."""
self.state.status = "working"
self.state.current_task = task.task_id
started = time.monotonic()
await self._emit_state()
try:
if task.task_type in self._task_handlers:
@@ -85,9 +92,19 @@ class Agent(ABC):
self.state.tasks_failed += 1
finally:
duration_ms = int((time.monotonic() - started) * 1000)
await message_bus.publish_telemetry("task", {
"agent_id": self.agent_id,
"task_id": task.task_id,
"task_type": task.task_type,
"status": task.status,
"error": task.error,
"duration_ms": duration_ms,
})
self.state.current_task = None
self.state.last_active = datetime.now()
self.state.status = "idle"
await self._emit_state()
@abstractmethod
async def handle_task(self, task: AgentTask) -> Dict[str, Any]:
@@ -98,6 +115,18 @@ class Agent(ABC):
"""Called periodically when idle. Override for custom behavior."""
pass
async def _emit_state(self):
"""Publish this agent's live state as telemetry (for the command center)."""
self._last_state_emit = time.monotonic()
await message_bus.publish_telemetry("agent", {
"agent_id": self.agent_id,
"agent_type": self.agent_type,
"status": self.state.status,
"current_task": self.state.current_task,
"tasks_completed": self.state.tasks_completed,
"tasks_failed": self.state.tasks_failed,
})
def register_task_handler(self, task_type: str, handler: callable):
self._task_handlers[task_type] = handler
@@ -124,6 +153,39 @@ class Agent(ABC):
async def receive_messages(self) -> List[AgentMessage]:
return await message_bus.get_messages(self.agent_id)
async def deliver(self, message: AgentMessage):
"""Consume an inbound directed message.
This is what makes agent-to-agent messaging actually work: the message
bus calls it for every directed message addressed to this agent. Messages
that carry a ``task_type`` in their payload are enqueued as tasks so
``handle_task`` processes them exactly like a submitted task; everything
else is handed to ``handle_message`` so the agent can react. Without this,
directed messages sat unread in the bus queue forever.
"""
payload = message.payload if isinstance(message.payload, dict) else {}
task_type = payload.get("task_type")
if task_type:
await self._task_queue.put(AgentTask(
task_id=f"{message.correlation_id or message.message_id}:{task_type}",
agent_type=self.agent_type,
task_type=task_type,
data=payload,
))
else:
try:
await self.handle_message(message)
except Exception as e:
logger.error(f"{self.agent_id} handle_message error: {e}")
async def handle_message(self, message: AgentMessage):
"""React to a non-task directed message (a notification). Default just
logs it; agents override this to act on events like EXCEPTION_DETECTED."""
logger.debug(
f"{self.agent_id} received {message.message_type.value} "
f"from {message.sender} (no task_type; not handled)"
)
def subscribe_to(self, message_type: MessageType, callback: callable):
message_bus.subscribe(message_type, callback)
@@ -151,6 +213,24 @@ class MasterAgent(Agent):
self._sub_agents[agent.agent_id] = agent
logger.info(f"JARVIS registered sub-agent: {agent.agent_id}")
async def handle_message(self, message: AgentMessage):
"""Surface notifications from sub-agents. Escalations (proposals a human
must decide) are logged at WARNING and recorded so they are visible
rather than silently dropped — the endpoint of the human-review path."""
if message.message_type == MessageType.EXCEPTION_DETECTED:
logger.warning(f"JARVIS: exception escalation from {message.sender}: {message.payload}")
else:
logger.info(f"JARVIS: {message.message_type.value} from {message.sender}")
self._decision_log.append({
"timestamp": datetime.now(),
"action": "received_notification",
"from": message.sender,
"type": message.message_type.value,
"payload": message.payload,
})
if len(self._decision_log) > 500:
self._decision_log = self._decision_log[-500:]
async def handle_task(self, task: AgentTask) -> Dict[str, Any]:
if task.task_type == "orchestrate_order":
return await self._orchestrate_order(task)

View File

@@ -43,8 +43,22 @@ async def api_patch(url: str, **kwargs) -> Optional[Dict[str, Any]]:
return await _request("patch", url, **kwargs)
async def _safe_text(resp) -> str:
try:
return (await resp.text())[:300]
except Exception:
return "<unreadable body>"
async def _request(method: str, url: str, **kwargs) -> Optional[Dict[str, Any]]:
"""Execute an HTTP request with up to 3 attempts and exponential backoff."""
"""Execute an HTTP request with up to 3 attempts and exponential backoff.
Returns the parsed body ONLY on a 2xx/3xx response. A 4xx is a client error
and returns None immediately (no retry — it won't fix itself). A 5xx is
retried like a network error, then returns None. This is the difference
between "the action succeeded" and "the server rejected it": callers gate on
a non-None result, so an error response must not look like success.
"""
max_retries = 3
backoff = 1.0
session = await get_session()
@@ -52,6 +66,25 @@ async def _request(method: str, url: str, **kwargs) -> Optional[Dict[str, Any]]:
for attempt in range(max_retries):
try:
async with getattr(session, method)(url, **kwargs) as resp:
if resp.status >= 500:
body = await _safe_text(resp)
if attempt < max_retries - 1:
logger.warning(
f"{method.upper()} {url} -> {resp.status} (server error), "
f"retrying in {backoff:.0f}s: {body}"
)
await asyncio.sleep(backoff)
backoff *= 2
continue
logger.error(
f"{method.upper()} {url} -> {resp.status} after {max_retries} attempts: {body}"
)
return None
if resp.status >= 400:
logger.error(
f"{method.upper()} {url} -> {resp.status} (client error): {await _safe_text(resp)}"
)
return None
if resp.content_type == "application/json":
return await resp.json()
return {"status_code": resp.status}

220
core/llm.py Normal file
View File

@@ -0,0 +1,220 @@
"""
LLM-backed decision-making for agents.
Currently backed by Claude (Anthropic). It is deliberately kept behind a thin
function boundary (`decide_stall_response`) and a `dataclass` result so the
model — or the whole provider — can be swapped later (e.g. a self-hosted model)
without touching any agent logic.
Configuration (env):
ANTHROPIC_API_KEY required — the Anthropic API key
LLM_MODEL default "claude-opus-4-8"
LLM_EFFORT default "high" (low | medium | high | xhigh | max)
"""
import json
import os
from dataclasses import dataclass
from datetime import datetime, timezone
from typing import Optional
from core.logger import logger
LLM_MODEL = os.getenv("LLM_MODEL", "claude-opus-4-8")
LLM_EFFORT = os.getenv("LLM_EFFORT", "high")
# Thinking tokens count against max_tokens; adaptive thinking at high effort can
# use a lot, so the budget must be generous or the JSON output gets truncated.
LLM_MAX_TOKENS = int(os.getenv("LLM_MAX_TOKENS", "8192"))
LLM_TIMEOUT_S = float(os.getenv("LLM_TIMEOUT_S", "30"))
_client = None
def _get_client():
"""Lazily construct the *async* Anthropic client so a missing dependency or
key never breaks agent startup — only the actual decision call fails, and the
caller falls back to deterministic behaviour. Async so the decision never
blocks the event loop (a sync client would freeze every other coroutine for
the full duration of the call)."""
global _client
if _client is None:
import anthropic # lazy import
_client = anthropic.AsyncAnthropic() # reads ANTHROPIC_API_KEY from env
return _client
@dataclass
class StallDecision:
action: str # "reassign" | "notify_only" | "wait" | "escalate"
reasoning: str
confidence: float # 0.0 .. 1.0
DEFAULT_STALL_THRESHOLD_MIN = 10
def build_stall_context(minutes_stalled, *, stall_threshold_min=DEFAULT_STALL_THRESHOLD_MIN,
now=None, facts=None) -> str:
"""Assemble the prompt context for a stall decision.
Shared by the ExceptionAgent and the eval harness so the eval tests exactly
what runs in production. `facts` is an optional dict of extra situational
detail (location, notes, prior-stall count, ...) — today the agent passes
none, but context tools will populate it later.
"""
now = now or datetime.now(timezone.utc)
lines = [
"A miler has stalled during an active delivery.",
f"- minutes_stalled: {round(float(minutes_stalled), 1)}",
f"- stall_threshold_minutes: {stall_threshold_min}",
f"- current_time_utc: {now.isoformat()}",
f"- hour_of_day_utc: {now.hour}",
]
for key, value in (facts or {}).items():
lines.append(f"- {key}: {value}")
lines.append("Decide the best response.")
return "\n".join(lines)
_VALID_ACTIONS = {"reassign", "notify_only", "wait", "escalate"}
_STALL_SYSTEM = """You are the exception controller for a last-mile delivery network.
A "miler" (delivery rider) assigned to an active booking has stopped moving for a while.
Decide the single best response. Choose exactly one action:
- "wait": the stall is plausibly benign (traffic, a short pickup wait, a customer handoff). Take no action yet.
- "notify_only": the miler can likely recover, but the customer deserves reassurance. Notify the customer; keep the miler.
- "reassign": the miler is genuinely stuck and unlikely to recover soon. Hand the booking to another miler. This is customer-visible and hard to reverse — choose it only when the evidence clearly supports it.
- "escalate": the situation is ambiguous or high-stakes; a human dispatcher should decide.
Weigh how long it has been stalled relative to the threshold, the time of day, and how far into the job it is.
Prefer the least disruptive action that fits the evidence. Report an honest confidence in [0, 1] —
a low confidence is a signal to escalate rather than act."""
_STALL_SCHEMA = {
"type": "object",
"properties": {
"action": {"type": "string", "enum": sorted(_VALID_ACTIONS)},
"reasoning": {"type": "string"},
"confidence": {"type": "number"},
},
"required": ["action", "reasoning", "confidence"],
"additionalProperties": False,
}
async def _decide(system: str, schema: dict, valid_actions, context: str):
"""Run one structured-output decision against Claude.
Returns (action, reasoning, confidence), or None on any failure (network,
timeout, refusal, truncation, bad parse) so callers can fall back to
deterministic behaviour. Retries once on a transient error before giving up.
"""
resp = None
last_err = None
for attempt in range(2): # initial try + one retry on transient failure
try:
client = _get_client()
resp = await client.messages.create(
model=LLM_MODEL,
max_tokens=LLM_MAX_TOKENS,
system=system,
thinking={"type": "adaptive"},
output_config={
"effort": LLM_EFFORT,
"format": {"type": "json_schema", "schema": schema},
},
messages=[{"role": "user", "content": context}],
timeout=LLM_TIMEOUT_S,
)
break
except Exception as e:
last_err = e
logger.warning(f"LLM decision request failed (attempt {attempt + 1}/2): {e}")
if resp is None:
logger.error(f"LLM decision failed after retries: {last_err}")
return None
stop = getattr(resp, "stop_reason", None)
if stop == "refusal":
logger.warning("LLM refused the decision request")
return None
if stop == "max_tokens":
logger.error("LLM decision truncated (max_tokens); raise LLM_MAX_TOKENS. Treating as no decision.")
return None
try:
text = next(b.text for b in resp.content if b.type == "text")
data = json.loads(text)
action = data["action"]
if action not in valid_actions:
raise ValueError(f"unexpected action {action!r}")
return action, str(data.get("reasoning", "")), float(data.get("confidence", 0.0))
except (StopIteration, KeyError, ValueError, json.JSONDecodeError) as e:
logger.error(f"LLM decision parse failed: {e}")
return None
async def decide_stall_response(context: str) -> Optional[StallDecision]:
"""Ask Claude how to handle a stalled miler. None on failure (caller falls
back to deterministic behaviour)."""
r = await _decide(_STALL_SYSTEM, _STALL_SCHEMA, _VALID_ACTIONS, context)
return StallDecision(*r) if r else None
# ── Assignment-failure decision (DispatchAgent) ──────────────────────────────
@dataclass
class AssignmentDecision:
action: str # "monitor" | "notify_customer" | "ops_alert" | "escalate"
reasoning: str
confidence: float
_ASSIGNMENT_ACTIONS = {"monitor", "notify_customer", "ops_alert", "escalate"}
_ASSIGNMENT_SYSTEM = """You are the dispatch controller for a last-mile delivery network.
The backend tried to assign a booking to a miler (rider) — both its AI assignment and the
fallback failed, so no rider was assigned. Decide how to react. Choose exactly one action:
- "monitor": likely a transient gap (a momentary lack of free riders); the backend will retry. Take no action.
- "notify_customer": a real but recoverable delay; tell the customer we're finding a rider so they aren't left guessing.
- "ops_alert": a genuine coverage gap in this zone (repeated failures, no nearby riders). Alert operations to onboard or redirect riders here.
- "escalate": ambiguous or contradictory — e.g. riders ARE nearby yet assignment keeps failing, which suggests a systemic issue rather than a coverage gap. A human dispatcher should look.
Weigh how many times assignment has failed in this zone today, how far the nearest available rider is,
the time of day, and whether coordinates were even available. A single failure with a rider nearby is
usually transient; repeated failures with no nearby rider is a coverage gap; repeated failures *despite*
nearby riders is not a coverage gap and warrants a human. Report an honest confidence in [0, 1]."""
_ASSIGNMENT_SCHEMA = {
"type": "object",
"properties": {
"action": {"type": "string", "enum": sorted(_ASSIGNMENT_ACTIONS)},
"reasoning": {"type": "string"},
"confidence": {"type": "number"},
},
"required": ["action", "reasoning", "confidence"],
"additionalProperties": False,
}
def build_assignment_failure_context(facts, *, now=None) -> str:
"""Assemble the prompt context for an assignment-failure decision. Shared by
DispatchAgent and the eval harness so the eval tests exactly what runs."""
now = now or datetime.now(timezone.utc)
lines = [
"A booking could not be assigned to a miler — the backend's AI and fallback assignment both failed.",
f"- current_time_utc: {now.isoformat()}",
f"- hour_of_day_utc: {now.hour}",
]
for key, value in (facts or {}).items():
lines.append(f"- {key}: {value}")
lines.append("Decide the best response.")
return "\n".join(lines)
async def decide_assignment_failure(context: str) -> Optional[AssignmentDecision]:
"""Ask Claude how to react to a failed assignment. None on failure (caller
falls back to deterministic behaviour)."""
r = await _decide(_ASSIGNMENT_SYSTEM, _ASSIGNMENT_SCHEMA, _ASSIGNMENT_ACTIONS, context)
return AssignmentDecision(*r) if r else None

View File

@@ -171,6 +171,22 @@ class MessageBus:
) -> str:
return await self.send_to_agent(sender, "ALL", message_type, payload, correlation_id)
async def _deliver_to_agent(self, agent_id: str, message: AgentMessage):
"""Hand a directed message to a locally-registered agent so it is
actually consumed (via its ``deliver``). Only if no such agent exists in
this process does it fall back to the pull queue — so a message is never
silently lost, but is delivered live whenever the recipient is present."""
agent = self._agents.get(agent_id)
if agent is not None and hasattr(agent, "deliver"):
try:
await agent.deliver(message)
return
except Exception as e:
logger.error(f"Delivery to {agent_id} failed: {e}")
return
async with self._lock:
self._queues[agent_id].append(QueuedMessage(message))
async def get_messages(self, agent_id: str) -> List[AgentMessage]:
async with self._lock:
queued = self._queues.pop(agent_id, [])
@@ -179,6 +195,26 @@ class MessageBus:
async def peek_messages(self, agent_id: str) -> List[AgentMessage]:
return [q.message for q in self._queues.get(agent_id, [])]
# ------------------------------------------------------------------ #
# Telemetry #
# ------------------------------------------------------------------ #
async def publish_telemetry(self, kind: str, payload: Dict[str, Any]):
"""
Fire-and-forget observability event on plain NATS (subject `telemetry.<kind>`).
Deliberately NOT JetStream: telemetry is ephemeral fan-out for dashboards
and must never accumulate in the persistent `logistics` stream. No-op
when NATS is not connected — telemetry must never affect agent behavior.
"""
if self._nc is None or not self._nc.is_connected:
return
try:
body = {"kind": kind, "ts": datetime.now().isoformat(), **payload}
await self._nc.publish(f"telemetry.{kind}", json.dumps(body).encode())
except Exception as e:
logger.debug(f"Telemetry publish failed [{kind}]: {e}")
# ------------------------------------------------------------------ #
# Hooks #
# ------------------------------------------------------------------ #
@@ -266,8 +302,7 @@ class MessageBus:
async def handler(msg):
try:
agent_msg = AgentMessage.from_json(msg.data.decode())
async with self._lock:
self._queues[agent_id].append(QueuedMessage(agent_msg))
await self._deliver_to_agent(agent_id, agent_msg)
except Exception as e:
logger.error(f"NATS agent-sub decode error [{agent_id}]: {e}")
finally:
@@ -282,8 +317,7 @@ class MessageBus:
async def _local_dispatch(self, message: AgentMessage):
"""In-process dispatch used when NATS is not connected."""
if message.recipient != "ALL":
async with self._lock:
self._queues[message.recipient].append(QueuedMessage(message))
await self._deliver_to_agent(message.recipient, message)
else:
for callback in list(self._subscribers.get(message.message_type, [])):
try: