route optimizer agent with dailygrubs ai assign
This commit is contained in:
@@ -1,7 +1,8 @@
|
||||
"""Dispatch Agent — watches NATS assignment events and escalates coverage gaps."""
|
||||
import asyncio
|
||||
import json
|
||||
from datetime import datetime
|
||||
import os
|
||||
from datetime import datetime, timezone
|
||||
from typing import Dict, List, Any, Optional
|
||||
|
||||
import nats
|
||||
@@ -11,18 +12,28 @@ import redis.asyncio as aioredis
|
||||
from core.agent import SpecializedAgent
|
||||
from core.types import AgentTask, MessageType
|
||||
from core.logger import logger
|
||||
from core.llm import decide_assignment_failure, build_assignment_failure_context
|
||||
from config.system_config import (
|
||||
NATS_HOST, NATS_PORT, NATS_USER, NATS_PASSWORD,
|
||||
REDIS_HOST, REDIS_PORT, REDIS_PASSWORD,
|
||||
)
|
||||
|
||||
# Notifying a customer is outward-facing, so it is gated: without autonomy the
|
||||
# agent proposes it as an internal ops alert instead of messaging the customer.
|
||||
DISPATCH_AGENT_AUTONOMOUS = os.getenv("DISPATCH_AGENT_AUTONOMOUS", "false").lower() == "true"
|
||||
|
||||
# The Go backend owns the JetStream stream carrying booking.* events; this agent
|
||||
# is a pure consumer and must not create it. Leave empty to auto-discover the
|
||||
# stream by subject, or pin it to a stream name if discovery isn't desired.
|
||||
DISPATCH_STREAM = os.getenv("DISPATCH_STREAM", "")
|
||||
|
||||
|
||||
class DispatchAgent(SpecializedAgent):
|
||||
"""
|
||||
Watches NATS ASSIGNMENTS stream for booking.assigned and
|
||||
booking.assignment_failed events published by the Go backend
|
||||
after routemate AI or fallback assignment. Does not trigger
|
||||
assignment itself.
|
||||
Consumes booking.assigned and booking.assignment_failed events published by
|
||||
the Go backend (after routemate AI or fallback assignment) by binding to the
|
||||
backend-owned JetStream stream that carries them. It is a pure consumer — it
|
||||
does not create the stream, and does not trigger assignment itself.
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
@@ -71,34 +82,46 @@ class DispatchAgent(SpecializedAgent):
|
||||
)
|
||||
self._nats_js = self._nats_nc.jetstream()
|
||||
|
||||
try:
|
||||
await self._nats_js.add_stream(
|
||||
name="ASSIGNMENTS",
|
||||
subjects=["booking.>"],
|
||||
)
|
||||
logger.info("NATS stream 'ASSIGNMENTS' created")
|
||||
except nats.js.errors.BadRequestError:
|
||||
pass # stream already exists
|
||||
|
||||
sub_assigned = await self._nats_js.subscribe(
|
||||
"booking.assigned",
|
||||
durable="dispatch-booking-assigned",
|
||||
cb=self._on_nats_booking_assigned,
|
||||
# Bind consumers to the *existing* backend-owned stream. Do NOT create
|
||||
# a stream here: booking.* is published by the Go backend, so creating
|
||||
# a second stream over the same subjects raises an overlap error that
|
||||
# (previously swallowed) left the consumer unbound. Mirrors the
|
||||
# explicit stream= binding ExceptionAgent uses for the TRACKING stream.
|
||||
sub_assigned = await self._bind_consumer(
|
||||
"booking.assigned", "dispatch-booking-assigned",
|
||||
self._on_nats_booking_assigned,
|
||||
)
|
||||
sub_failed = await self._nats_js.subscribe(
|
||||
"booking.assignment_failed",
|
||||
durable="dispatch-booking-assignment-failed",
|
||||
cb=self._on_nats_booking_assignment_failed,
|
||||
)
|
||||
self._nats_subs.extend([sub_assigned, sub_failed])
|
||||
logger.info(
|
||||
"DispatchAgent subscribed to booking.assigned "
|
||||
"and booking.assignment_failed"
|
||||
sub_failed = await self._bind_consumer(
|
||||
"booking.assignment_failed", "dispatch-booking-assignment-failed",
|
||||
self._on_nats_booking_assignment_failed,
|
||||
)
|
||||
self._nats_subs.extend(s for s in (sub_assigned, sub_failed) if s is not None)
|
||||
if self._nats_subs:
|
||||
logger.info(f"DispatchAgent bound {len(self._nats_subs)} booking-event consumer(s)")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"DispatchAgent NATS connect error: {e}")
|
||||
|
||||
async def _bind_consumer(self, subject, durable, cb):
|
||||
"""Bind a durable push consumer to whichever existing stream carries
|
||||
`subject`. Returns the subscription, or None (with a clear, actionable
|
||||
log) if no stream carries it — never a swallowed or opaque error."""
|
||||
try:
|
||||
stream = DISPATCH_STREAM or await self._nats_js.find_stream_name_by_subject(subject)
|
||||
sub = await self._nats_js.subscribe(subject, durable=durable, stream=stream, cb=cb)
|
||||
logger.info(f"DispatchAgent bound '{subject}' on stream '{stream}' (durable={durable})")
|
||||
return sub
|
||||
except nats.js.errors.NotFoundError:
|
||||
logger.error(
|
||||
f"DispatchAgent: no JetStream stream carries '{subject}'. The publisher "
|
||||
f"(Go backend) must create a stream covering it, or set DISPATCH_STREAM to "
|
||||
f"the stream name. Not subscribing to '{subject}'."
|
||||
)
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.error(f"DispatchAgent failed to bind '{subject}': {e}")
|
||||
return None
|
||||
|
||||
# ------------------------------------------------------------------ #
|
||||
# Redis GEO lookup #
|
||||
# ------------------------------------------------------------------ #
|
||||
@@ -177,9 +200,10 @@ class DispatchAgent(SpecializedAgent):
|
||||
booking.assignment_failed — Go publishes this when decide-assignment
|
||||
returns escalate=true or both AI and fallback find nobody.
|
||||
|
||||
Widens the GEORADIUS search to gauge coverage depth, tracks daily
|
||||
failures per zone in Redis, and escalates to CUSTOMER_AGENT as an
|
||||
ops_alert when a zone hits 3 failures in a day.
|
||||
Gathers coverage context (nearest-rider sweep + daily failure counter),
|
||||
then asks Claude to choose: monitor / notify_customer / ops_alert /
|
||||
escalate. Customer notification is gated behind autonomy. Falls back to
|
||||
the previous count>=3 heuristic if the LLM is unavailable.
|
||||
"""
|
||||
try:
|
||||
data = json.loads(msg.data.decode())
|
||||
@@ -188,64 +212,135 @@ class DispatchAgent(SpecializedAgent):
|
||||
lat = data.get("lat")
|
||||
lon = data.get("lon")
|
||||
|
||||
logger.warning(
|
||||
f"[DISPATCH] booking.assignment_failed — "
|
||||
f"booking={booking_id} zone={zone_id}"
|
||||
logger.warning(f"[DISPATCH] booking.assignment_failed — booking={booking_id} zone={zone_id}")
|
||||
|
||||
facts, count = await self._gather_assignment_facts(zone_id, lat, lon)
|
||||
logger.info(f"[DISPATCH] booking={booking_id} context facts: {facts}")
|
||||
|
||||
decision = await decide_assignment_failure(build_assignment_failure_context(facts))
|
||||
|
||||
if decision is None:
|
||||
logger.warning(f"No LLM decision for booking {booking_id}; falling back to count>=3 heuristic")
|
||||
if count >= 3:
|
||||
await self._ops_alert(
|
||||
zone_id, booking_id,
|
||||
f"Zone {zone_id} has had {count} failed assignments today — recommend onboarding milers here.",
|
||||
)
|
||||
else:
|
||||
logger.info(f"[DISPATCH] zone {zone_id} failure count {count}/3 — transient, no escalation")
|
||||
return
|
||||
|
||||
logger.info(
|
||||
f"[DISPATCH] booking={booking_id} zone={zone_id} decision={decision.action} "
|
||||
f"confidence={decision.confidence:.2f} reasoning={decision.reasoning!r}"
|
||||
)
|
||||
|
||||
# Wider sweeps to gauge how far the nearest miler actually is.
|
||||
has_coords = lat is not None and lon is not None
|
||||
found_20km = await self._find_zone(lat, lon, radius_km=20) if has_coords else None
|
||||
found_30km = await self._find_zone(lat, lon, radius_km=30) if has_coords else None
|
||||
if decision.action == "monitor":
|
||||
pass # transient — the backend will retry
|
||||
|
||||
if found_20km:
|
||||
logger.info(
|
||||
f"[DISPATCH] Nearest miler within 20 km: {found_20km['miler_id']}"
|
||||
)
|
||||
elif found_30km:
|
||||
logger.info(
|
||||
f"[DISPATCH] Nearest miler within 30 km: {found_30km['miler_id']}"
|
||||
)
|
||||
else:
|
||||
logger.warning(
|
||||
f"[DISPATCH] No miler found within 30 km of zone {zone_id}"
|
||||
)
|
||||
elif decision.action == "notify_customer":
|
||||
if DISPATCH_AGENT_AUTONOMOUS:
|
||||
await self._notify_customer_delay(booking_id)
|
||||
else:
|
||||
# Outward-facing action not authorised — raise an internal proposal instead.
|
||||
await self._ops_alert(
|
||||
zone_id, booking_id,
|
||||
f"[proposed: notify_customer, conf={decision.confidence:.2f}] {decision.reasoning}",
|
||||
)
|
||||
|
||||
# Increment daily failure counter (48 h TTL covers day rollover).
|
||||
today = datetime.utcnow().strftime("%Y-%m-%d")
|
||||
counter_key = f"failed_assignment:{zone_id}:{today}"
|
||||
count = await self._redis.incr(counter_key)
|
||||
await self._redis.expire(counter_key, 172800)
|
||||
elif decision.action == "ops_alert":
|
||||
await self._ops_alert(zone_id, booking_id, decision.reasoning)
|
||||
|
||||
if count >= 3:
|
||||
logger.warning(
|
||||
f"[DISPATCH] Coverage gap detected: zone {zone_id} has "
|
||||
f"{count} failed assignments today — escalating"
|
||||
)
|
||||
await self.send_message(
|
||||
recipient="CUSTOMER_AGENT",
|
||||
message_type=MessageType.AGENT_TASK,
|
||||
payload={
|
||||
"task_type": "ops_alert",
|
||||
"zone_id": zone_id,
|
||||
"reasoning": (
|
||||
f"Zone {zone_id} has had {count} failed assignments today "
|
||||
f"— recommend onboarding milers here."
|
||||
),
|
||||
},
|
||||
correlation_id=str(booking_id),
|
||||
)
|
||||
else:
|
||||
logger.info(
|
||||
f"[DISPATCH] zone {zone_id} failure count {count}/3 — "
|
||||
f"transient, no escalation"
|
||||
)
|
||||
else: # "escalate"
|
||||
await self._escalate_dispatch(zone_id, booking_id, decision)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"DispatchAgent booking.assignment_failed handler error: {e}")
|
||||
finally:
|
||||
await msg.ack()
|
||||
|
||||
# ------------------------------------------------------------------ #
|
||||
# Context gathering + actions #
|
||||
# ------------------------------------------------------------------ #
|
||||
|
||||
async def _gather_assignment_facts(self, zone_id, lat, lon):
|
||||
"""Read-only coverage context plus the daily failure counter for this
|
||||
zone. Returns (facts, failures_today). Best-effort — any failure just
|
||||
yields fewer facts, never raises."""
|
||||
facts: Dict[str, Any] = {"zone_id": zone_id}
|
||||
has_coords = lat is not None and lon is not None
|
||||
facts["has_coordinates"] = has_coords
|
||||
|
||||
nearest_km = None
|
||||
if has_coords:
|
||||
for radius in (10, 20, 30):
|
||||
try:
|
||||
if await self._find_zone(lat, lon, radius_km=radius):
|
||||
nearest_km = radius
|
||||
break
|
||||
except Exception as e:
|
||||
logger.warning(f"assignment facts: GEORADIUS {radius}km failed: {e}")
|
||||
break
|
||||
|
||||
if not has_coords:
|
||||
facts["nearest_miler_within_km"] = "unknown (no coordinates)"
|
||||
elif nearest_km is None:
|
||||
facts["nearest_miler_within_km"] = "none within 30km"
|
||||
else:
|
||||
facts["nearest_miler_within_km"] = nearest_km
|
||||
|
||||
count = 0
|
||||
try:
|
||||
today = datetime.now(timezone.utc).strftime("%Y-%m-%d")
|
||||
key = f"failed_assignment:{zone_id}:{today}"
|
||||
count = await self._redis.incr(key)
|
||||
await self._redis.expire(key, 172800) # 48h covers day rollover
|
||||
except Exception as e:
|
||||
logger.warning(f"assignment facts: failure counter failed for zone {zone_id}: {e}")
|
||||
facts["failures_today"] = count
|
||||
|
||||
return facts, count
|
||||
|
||||
async def _ops_alert(self, zone_id, booking_id, reasoning):
|
||||
await self.send_message(
|
||||
recipient="CUSTOMER_AGENT",
|
||||
message_type=MessageType.AGENT_TASK,
|
||||
payload={"task_type": "ops_alert", "zone_id": zone_id, "reasoning": reasoning},
|
||||
correlation_id=str(booking_id),
|
||||
)
|
||||
|
||||
async def _notify_customer_delay(self, booking_id):
|
||||
await self.send_message(
|
||||
recipient="CUSTOMER_AGENT",
|
||||
message_type=MessageType.AGENT_TASK,
|
||||
payload={
|
||||
"task_type": "customer_notify",
|
||||
"booking_id": booking_id,
|
||||
"message": "We're finding the right rider for your delivery — thanks for your patience.",
|
||||
},
|
||||
correlation_id=str(booking_id),
|
||||
)
|
||||
|
||||
async def _escalate_dispatch(self, zone_id, booking_id, decision):
|
||||
logger.warning(
|
||||
f"[DISPATCH] Escalating booking {booking_id} (zone {zone_id}) to a human "
|
||||
f"(proposed={decision.action}, confidence={decision.confidence:.2f}): {decision.reasoning}"
|
||||
)
|
||||
await self.send_message(
|
||||
recipient="JARVIS",
|
||||
message_type=MessageType.AGENT_TASK,
|
||||
payload={
|
||||
"task_type": "human_review",
|
||||
"reason": "assignment_failed",
|
||||
"zone_id": zone_id,
|
||||
"booking_id": booking_id,
|
||||
"proposed_action": decision.action,
|
||||
"confidence": decision.confidence,
|
||||
"reasoning": decision.reasoning,
|
||||
},
|
||||
correlation_id=str(booking_id),
|
||||
)
|
||||
|
||||
# ------------------------------------------------------------------ #
|
||||
# Task handler (no inbound task types remain) #
|
||||
# ------------------------------------------------------------------ #
|
||||
|
||||
@@ -10,8 +10,9 @@ StallDetector sweep (every 60 s):
|
||||
"""
|
||||
import asyncio
|
||||
import json
|
||||
import os
|
||||
import uuid
|
||||
from datetime import datetime, timedelta
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Dict, List, Any, Optional, Set
|
||||
from dataclasses import dataclass, field
|
||||
from enum import Enum
|
||||
@@ -25,6 +26,7 @@ from core.agent import SpecializedAgent
|
||||
from core.types import AgentTask, MessageType
|
||||
from core.logger import logger
|
||||
from core.http_client import api_post
|
||||
from core.llm import decide_stall_response, build_stall_context, StallDecision
|
||||
from config.system_config import (
|
||||
GO_API_BASE_URL, INTERNAL_API_KEY,
|
||||
DB_HOST, DB_PORT, DB_NAME, DB_USER, DB_PASSWORD,
|
||||
@@ -36,6 +38,35 @@ STALL_MINUTES = 10
|
||||
ACTIVE_STATUSES = ["Miler_Assigned", "Pickup_Scheduled"]
|
||||
TRACKING_STREAM = "TRACKING"
|
||||
|
||||
# Reassignment is customer-visible and hard to reverse, so it is gated: the LLM
|
||||
# only *proposes* a reassign unless the agent is explicitly put in autonomous
|
||||
# mode AND the model is confident enough. Otherwise the proposal is escalated to
|
||||
# a human via JARVIS.
|
||||
AUTONOMOUS_REASSIGN = os.getenv("EXCEPTION_AGENT_AUTONOMOUS", "false").lower() == "true"
|
||||
REASSIGN_MIN_CONFIDENCE = float(os.getenv("EXCEPTION_AGENT_REASSIGN_CONFIDENCE", "0.7"))
|
||||
# How long a booking stays "already alerted" so we don't re-alert the same stall.
|
||||
# Self-expiring so dedup state is bounded and survives restarts (unlike a set).
|
||||
STALL_NOTIFIED_TTL = int(os.getenv("STALL_NOTIFIED_TTL_SEC", "21600")) # 6h
|
||||
|
||||
|
||||
def _utcnow() -> datetime:
|
||||
"""Timezone-aware current time in UTC. All stall math uses this so it stays
|
||||
correct regardless of the container's local timezone, and never mixes naive
|
||||
with tz-aware timestamps."""
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def _parse_ts(value: Optional[str]) -> Optional[datetime]:
|
||||
"""Parse an ISO timestamp as tz-aware UTC. Naive values are assumed UTC so
|
||||
pre-existing (naive) Redis entries remain comparable during rollout."""
|
||||
if not value:
|
||||
return None
|
||||
try:
|
||||
dt = datetime.fromisoformat(value)
|
||||
except (ValueError, TypeError):
|
||||
return None
|
||||
return dt if dt.tzinfo else dt.replace(tzinfo=timezone.utc)
|
||||
|
||||
|
||||
class ExceptionType(str, Enum):
|
||||
DELAY = "delay"
|
||||
@@ -95,7 +126,6 @@ class ExceptionAgent(SpecializedAgent):
|
||||
|
||||
self._location_sub = None
|
||||
self._stalled_sub = None
|
||||
self._notified_stalls: Set[str] = set()
|
||||
self._exceptions: Dict[str, ExceptionRecord] = {}
|
||||
self._strategies = self._init_strategies()
|
||||
self._escalation_rules = self._init_escalation_rules()
|
||||
@@ -211,7 +241,7 @@ class ExceptionAgent(SpecializedAgent):
|
||||
miler_id = str(data.get("miler_id", ""))
|
||||
lat = str(round(float(data.get("lat", 0)), 6))
|
||||
lon = str(round(float(data.get("lon", 0)), 6))
|
||||
now_ts = datetime.now().isoformat()
|
||||
now_ts = _utcnow().isoformat()
|
||||
redis_key = f"miler:{miler_id}:movement"
|
||||
|
||||
prev = await self._redis.hgetall(redis_key)
|
||||
@@ -227,20 +257,13 @@ class ExceptionAgent(SpecializedAgent):
|
||||
else:
|
||||
await self._redis.hset(redis_key, mapping={"lat": lat, "lon": lon, "updated_at": now_ts})
|
||||
|
||||
unchanged_since_str = prev.get("position_unchanged_since", now_ts)
|
||||
try:
|
||||
unchanged_since = datetime.fromisoformat(unchanged_since_str)
|
||||
except ValueError:
|
||||
unchanged_since = datetime.now()
|
||||
|
||||
minutes_stalled = (datetime.now() - unchanged_since).total_seconds() / 60
|
||||
unchanged_since = _parse_ts(prev.get("position_unchanged_since")) or _utcnow()
|
||||
minutes_stalled = (_utcnow() - unchanged_since).total_seconds() / 60
|
||||
|
||||
if minutes_stalled >= STALL_MINUTES:
|
||||
booking = await self._get_active_booking(miler_id)
|
||||
if booking:
|
||||
booking_id = booking["booking_id"]
|
||||
if booking_id not in self._notified_stalls:
|
||||
await self._publish_stall(miler_id, booking_id, minutes_stalled)
|
||||
await self._publish_stall(miler_id, booking["booking_id"], minutes_stalled)
|
||||
|
||||
# ── Stall detection: background sweep ────────────────────────────────────
|
||||
|
||||
@@ -274,23 +297,19 @@ class ExceptionAgent(SpecializedAgent):
|
||||
logger.error(f"StallDetector Postgres error: {e}")
|
||||
return
|
||||
|
||||
now = datetime.now()
|
||||
now = _utcnow()
|
||||
for row in rows:
|
||||
miler_id = row["miler_id"]
|
||||
booking_id = row["booking_id"]
|
||||
|
||||
if booking_id in self._notified_stalls:
|
||||
continue
|
||||
|
||||
movement = await self._redis.hgetall(f"miler:{miler_id}:movement")
|
||||
if not movement or "updated_at" not in movement:
|
||||
continue
|
||||
|
||||
try:
|
||||
updated_at = datetime.fromisoformat(movement["updated_at"])
|
||||
minutes_stale = (now - updated_at).total_seconds() / 60
|
||||
except ValueError:
|
||||
updated_at = _parse_ts(movement.get("updated_at"))
|
||||
if updated_at is None:
|
||||
continue
|
||||
minutes_stale = (now - updated_at).total_seconds() / 60
|
||||
|
||||
if minutes_stale >= STALL_MINUTES:
|
||||
logger.warning(f"StallDetector: miler {miler_id} stale {minutes_stale:.1f} min (booking {booking_id})")
|
||||
@@ -298,8 +317,57 @@ class ExceptionAgent(SpecializedAgent):
|
||||
|
||||
# ── Stall: publish + handle ───────────────────────────────────────────────
|
||||
|
||||
def _stall_counter_key(self, miler_id) -> str:
|
||||
return f"miler:{miler_id}:stalls:{_utcnow().strftime('%Y-%m-%d')}"
|
||||
|
||||
def _stall_notified_key(self, booking_id) -> str:
|
||||
return f"stall_notified:{booking_id}"
|
||||
|
||||
def _stall_handled_key(self, booking_id) -> str:
|
||||
return f"stall_handled:{booking_id}"
|
||||
|
||||
async def _claim_stall(self, booking_id) -> bool:
|
||||
"""Atomically claim the first stall ALERT for a booking (Redis SET NX with
|
||||
TTL). True = publish this alert; False = already alerted within the window.
|
||||
Replaces the old unbounded in-memory set: self-expiring (bounded memory),
|
||||
durable across restarts, and race-free between the ping and sweep paths.
|
||||
Fail-open on Redis error so a genuine stall is never silently muted."""
|
||||
try:
|
||||
got = await self._redis.set(
|
||||
self._stall_notified_key(booking_id), "1", ex=STALL_NOTIFIED_TTL, nx=True,
|
||||
)
|
||||
return bool(got)
|
||||
except Exception as e:
|
||||
logger.warning(f"stall claim failed for {booking_id}: {e}")
|
||||
return True
|
||||
|
||||
async def _claim_stall_handled(self, booking_id) -> bool:
|
||||
"""Atomically claim HANDLING of a stall event (consumer side). Guards
|
||||
against JetStream at-least-once redelivery re-running the decision and
|
||||
repeating an irreversible reassign. Fail-open on Redis error."""
|
||||
try:
|
||||
got = await self._redis.set(
|
||||
self._stall_handled_key(booking_id), "1", ex=STALL_NOTIFIED_TTL, nx=True,
|
||||
)
|
||||
return bool(got)
|
||||
except Exception as e:
|
||||
logger.warning(f"stall handled-claim failed for {booking_id}: {e}")
|
||||
return True
|
||||
|
||||
async def _incr_stall_counter(self, miler_id):
|
||||
"""Bump this miler's per-day stall count (Redis, best-effort). 48h TTL
|
||||
covers the day rollover; failures never block stall handling."""
|
||||
try:
|
||||
key = self._stall_counter_key(miler_id)
|
||||
await self._redis.incr(key)
|
||||
await self._redis.expire(key, 172800)
|
||||
except Exception as e:
|
||||
logger.warning(f"stall counter incr failed for miler {miler_id}: {e}")
|
||||
|
||||
async def _publish_stall(self, miler_id: str, booking_id: str, minutes_stalled: float):
|
||||
self._notified_stalls.add(booking_id)
|
||||
if not await self._claim_stall(booking_id):
|
||||
return # already alerted for this booking within the dedup window
|
||||
await self._incr_stall_counter(miler_id)
|
||||
payload = json.dumps({
|
||||
"miler_id": miler_id,
|
||||
"booking_id": booking_id,
|
||||
@@ -312,7 +380,15 @@ class ExceptionAgent(SpecializedAgent):
|
||||
logger.error(f"Failed to publish miler.stalled: {e}")
|
||||
|
||||
async def _on_miler_stalled(self, msg):
|
||||
"""Reassign booking and notify customer via Go API."""
|
||||
"""Decide (via Claude) how to handle a stalled miler, then act.
|
||||
|
||||
Claude chooses one of wait / notify_only / reassign / escalate. The only
|
||||
irreversible, customer-visible action (reassign) is gated behind an
|
||||
autonomy flag and a confidence threshold; otherwise it is escalated to a
|
||||
human via JARVIS. If the LLM is unavailable, we fall back to the previous
|
||||
deterministic behaviour (reassign + notify) so detection never silently
|
||||
stops acting.
|
||||
"""
|
||||
try:
|
||||
data = json.loads(msg.data.decode())
|
||||
except Exception:
|
||||
@@ -324,36 +400,175 @@ class ExceptionAgent(SpecializedAgent):
|
||||
|
||||
logger.warning(f"EXCEPTION_AGENT stall: miler={miler_id} booking={booking_id} ({minutes_stalled} min)")
|
||||
|
||||
headers = {"X-Internal-Key": INTERNAL_API_KEY}
|
||||
if not await self._claim_stall_handled(booking_id):
|
||||
logger.info(f"Stall for booking {booking_id} already handled; skipping redelivery")
|
||||
return
|
||||
|
||||
reassign_result = await api_post(
|
||||
facts = await self._gather_stall_facts(miler_id, booking_id)
|
||||
if facts:
|
||||
logger.info(f"[EXCEPTION] booking={booking_id} context facts: {facts}")
|
||||
|
||||
context = build_stall_context(
|
||||
minutes_stalled, stall_threshold_min=STALL_MINUTES, now=_utcnow(), facts=facts,
|
||||
)
|
||||
decision = await decide_stall_response(context)
|
||||
|
||||
if decision is None:
|
||||
logger.warning(f"No LLM decision for booking {booking_id}; falling back to reassign + notify")
|
||||
await self._reassign(booking_id, "miler_stalled")
|
||||
await self._notify_customer(booking_id, "We detected a delay, finding you a new miler")
|
||||
self._record_stall_exception(
|
||||
miler_id, booking_id, minutes_stalled,
|
||||
resolution="Fallback (LLM unavailable): reassignment triggered + customer notified",
|
||||
actions=["reassign", "notify_customer"],
|
||||
)
|
||||
return
|
||||
|
||||
logger.info(
|
||||
f"[EXCEPTION] booking={booking_id} decision={decision.action} "
|
||||
f"confidence={decision.confidence:.2f} reasoning={decision.reasoning!r}"
|
||||
)
|
||||
|
||||
actions: List[str] = []
|
||||
|
||||
if decision.action == "wait":
|
||||
actions.append("wait")
|
||||
|
||||
elif decision.action == "notify_only":
|
||||
await self._notify_customer(
|
||||
booking_id, "We're keeping an eye on your delivery — thanks for your patience."
|
||||
)
|
||||
actions.append("notify_customer")
|
||||
|
||||
elif decision.action == "reassign":
|
||||
if AUTONOMOUS_REASSIGN and decision.confidence >= REASSIGN_MIN_CONFIDENCE:
|
||||
await self._reassign(booking_id, "miler_stalled")
|
||||
await self._notify_customer(booking_id, "We detected a delay, finding you a new miler")
|
||||
actions += ["reassign", "notify_customer"]
|
||||
else:
|
||||
# Not authorised to auto-reassign (mode off or low confidence):
|
||||
# hand the proposal to a human rather than taking the action.
|
||||
await self._escalate_to_human(miler_id, booking_id, minutes_stalled, decision)
|
||||
actions.append("escalate_reassign_for_review")
|
||||
|
||||
else: # "escalate"
|
||||
await self._escalate_to_human(miler_id, booking_id, minutes_stalled, decision)
|
||||
actions.append("escalate")
|
||||
|
||||
self._record_stall_exception(
|
||||
miler_id, booking_id, minutes_stalled,
|
||||
resolution=f"[{decision.action}] {decision.reasoning}",
|
||||
actions=actions,
|
||||
)
|
||||
|
||||
# ── Stall: action helpers ─────────────────────────────────────────────────
|
||||
|
||||
async def _reassign(self, booking_id: str, reason: str) -> bool:
|
||||
result = await api_post(
|
||||
f"{GO_API_BASE_URL}/api/v1/internal/bookings/{booking_id}/reassign",
|
||||
json={"reason": "miler_stalled"},
|
||||
headers=headers,
|
||||
json={"reason": reason},
|
||||
headers={"X-Internal-Key": INTERNAL_API_KEY},
|
||||
)
|
||||
logger.info(f"Reassign {'OK' if reassign_result is not None else 'FAILED'} for booking {booking_id}")
|
||||
ok = result is not None
|
||||
logger.info(f"Reassign {'OK' if ok else 'FAILED'} for booking {booking_id}")
|
||||
return ok
|
||||
|
||||
notify_result = await api_post(
|
||||
async def _notify_customer(self, booking_id: str, message: str) -> bool:
|
||||
result = await api_post(
|
||||
f"{GO_API_BASE_URL}/api/v1/internal/notify",
|
||||
json={
|
||||
"booking_id": booking_id,
|
||||
"message": "We detected a delay, finding you a new miler",
|
||||
"target": "customer",
|
||||
},
|
||||
headers=headers,
|
||||
json={"booking_id": booking_id, "message": message, "target": "customer"},
|
||||
headers={"X-Internal-Key": INTERNAL_API_KEY},
|
||||
)
|
||||
logger.info(f"Notify {'OK' if notify_result is not None else 'FAILED'} for booking {booking_id}")
|
||||
ok = result is not None
|
||||
logger.info(f"Notify {'OK' if ok else 'FAILED'} for booking {booking_id}")
|
||||
return ok
|
||||
|
||||
exc_id = f"EXC-{datetime.now().strftime('%Y%m%d')}-{uuid.uuid4().hex[:6].upper()}"
|
||||
async def _escalate_to_human(self, miler_id, booking_id, minutes_stalled, decision: StallDecision):
|
||||
logger.warning(
|
||||
f"[EXCEPTION] Escalating booking {booking_id} to human dispatcher "
|
||||
f"(proposed={decision.action}, confidence={decision.confidence:.2f}): {decision.reasoning}"
|
||||
)
|
||||
await self.send_message(
|
||||
recipient="JARVIS",
|
||||
message_type=MessageType.EXCEPTION_DETECTED,
|
||||
payload={
|
||||
"order_id": booking_id,
|
||||
"miler_id": miler_id,
|
||||
"exception_type": ExceptionType.MILER_STALLED.value,
|
||||
"severity": ExceptionSeverity.HIGH.value,
|
||||
"proposed_action": decision.action,
|
||||
"confidence": decision.confidence,
|
||||
"reasoning": decision.reasoning,
|
||||
"minutes_stalled": minutes_stalled,
|
||||
"urgency": "high",
|
||||
},
|
||||
)
|
||||
|
||||
async def _gather_stall_facts(self, miler_id, booking_id) -> Dict[str, Any]:
|
||||
"""Read-only context gathering: pull real signals from Postgres + Redis
|
||||
to inform the decision. Best-effort — any failure is logged and skipped,
|
||||
never raised, so a thin-context decision still runs (which skews safe)."""
|
||||
facts: Dict[str, Any] = {}
|
||||
now = _utcnow()
|
||||
|
||||
# Booking phase + age (Postgres, read-only).
|
||||
if self._pg:
|
||||
try:
|
||||
async with self._pg.acquire() as conn:
|
||||
row = await conn.fetchrow(
|
||||
"SELECT status, createdat FROM pickupbookings WHERE bookingid::text = $1",
|
||||
str(booking_id),
|
||||
)
|
||||
if row:
|
||||
facts["booking_status"] = row["status"]
|
||||
created = row["createdat"]
|
||||
if created is not None:
|
||||
if created.tzinfo is None:
|
||||
created = created.replace(tzinfo=timezone.utc)
|
||||
facts["minutes_since_booking_created"] = round((now - created).total_seconds() / 60, 1)
|
||||
except Exception as e:
|
||||
logger.warning(f"stall facts: booking query failed for {booking_id}: {e}")
|
||||
|
||||
# Movement freshness (Redis, read-only).
|
||||
try:
|
||||
movement = await self._redis.hgetall(f"miler:{miler_id}:movement")
|
||||
except Exception as e:
|
||||
logger.warning(f"stall facts: redis read failed for miler {miler_id}: {e}")
|
||||
movement = {}
|
||||
|
||||
if movement:
|
||||
last_ping = _parse_ts(movement.get("updated_at"))
|
||||
if last_ping:
|
||||
facts["last_gps_ping_minutes_ago"] = round((now - last_ping).total_seconds() / 60, 1)
|
||||
unchanged = _parse_ts(movement.get("position_unchanged_since"))
|
||||
if unchanged:
|
||||
facts["position_unchanged_minutes"] = round((now - unchanged).total_seconds() / 60, 1)
|
||||
|
||||
# Per-miler stall count today (Redis, read-only). Already includes the
|
||||
# current stall, since _publish_stall increments before the consumer runs.
|
||||
try:
|
||||
raw = await self._redis.get(self._stall_counter_key(miler_id))
|
||||
if raw is not None:
|
||||
facts["stalls_today"] = int(raw)
|
||||
except Exception as e:
|
||||
logger.warning(f"stall facts: stall counter read failed for miler {miler_id}: {e}")
|
||||
|
||||
return facts
|
||||
|
||||
def _record_stall_exception(self, miler_id, booking_id, minutes_stalled, resolution, actions):
|
||||
escalated = any(a.startswith("escalate") for a in actions)
|
||||
exc_id = f"EXC-{_utcnow().strftime('%Y%m%d')}-{uuid.uuid4().hex[:6].upper()}"
|
||||
self._exceptions[exc_id] = ExceptionRecord(
|
||||
exception_id=exc_id, order_id=booking_id,
|
||||
exception_type=ExceptionType.MILER_STALLED,
|
||||
severity=ExceptionSeverity.HIGH,
|
||||
description=f"Miler {miler_id} stalled {minutes_stalled} min",
|
||||
detected_at=datetime.now(), resolved_at=datetime.now(),
|
||||
resolution="Reassignment triggered + customer notified",
|
||||
assigned_to=None, status="resolved",
|
||||
actions_taken=["reassign", "notify_customer"],
|
||||
detected_at=_utcnow(),
|
||||
resolved_at=None if escalated else _utcnow(),
|
||||
resolution=resolution,
|
||||
assigned_to=None,
|
||||
status="escalated" if escalated else "resolved",
|
||||
actions_taken=list(actions),
|
||||
)
|
||||
|
||||
# ── Postgres helper ───────────────────────────────────────────────────────
|
||||
|
||||
@@ -117,9 +117,12 @@ class FleetAgent(SpecializedAgent):
|
||||
|
||||
vehicle_id = suitable_vehicles[0]
|
||||
vehicle = self._vehicles[vehicle_id]
|
||||
was_available = vehicle["status"] == "available"
|
||||
vehicle["status"] = "assigned"
|
||||
|
||||
if vehicle["hub"] in self._hub_capacity:
|
||||
# Only move the counter when the availability boundary is actually
|
||||
# crossed, so it can't drift out of [0, total].
|
||||
if was_available and vehicle["hub"] in self._hub_capacity:
|
||||
self._hub_capacity[vehicle["hub"]]["available"] -= 1
|
||||
|
||||
assignment_id = f"ASN-{uuid.uuid4().hex[:8].upper()}"
|
||||
@@ -160,11 +163,20 @@ class FleetAgent(SpecializedAgent):
|
||||
return {"status": "error", "message": f"Vehicle {vehicle_id} not found"}
|
||||
|
||||
vehicle = self._vehicles[vehicle_id]
|
||||
was_available = vehicle["status"] == "available"
|
||||
vehicle["status"] = "available"
|
||||
|
||||
if vehicle["hub"] in self._hub_capacity:
|
||||
# Only increment when the vehicle was actually unavailable, so a
|
||||
# double-release (or releasing an available vehicle) can't exceed total.
|
||||
if not was_available and vehicle["hub"] in self._hub_capacity:
|
||||
self._hub_capacity[vehicle["hub"]]["available"] += 1
|
||||
|
||||
# Close any active assignment for this vehicle so it isn't reported as
|
||||
# still on the road (and stale 'active' rows don't accumulate).
|
||||
for assignment in self._assignments.values():
|
||||
if assignment.vehicle_id == vehicle_id and assignment.status == "active":
|
||||
assignment.status = "returned"
|
||||
|
||||
logger.info(f"Vehicle {vehicle_id} released")
|
||||
|
||||
return {"status": "released", "vehicle_id": vehicle_id, "hub": vehicle["hub"]}
|
||||
@@ -239,7 +251,12 @@ class FleetAgent(SpecializedAgent):
|
||||
else:
|
||||
self._maintenance_schedule[vehicle_id] = datetime.now() + timedelta(days=3)
|
||||
|
||||
self._vehicles[vehicle_id]["status"] = "maintenance"
|
||||
# Taking a vehicle out of service reduces availability (only if it was
|
||||
# counted as available), keeping hub capacity accurate.
|
||||
vehicle = self._vehicles[vehicle_id]
|
||||
if vehicle["status"] == "available" and vehicle["hub"] in self._hub_capacity:
|
||||
self._hub_capacity[vehicle["hub"]]["available"] -= 1
|
||||
vehicle["status"] = "maintenance"
|
||||
|
||||
return {
|
||||
"status": "scheduled",
|
||||
|
||||
@@ -254,11 +254,11 @@ class OrderAgent(SpecializedAgent):
|
||||
if email and not self._is_valid_email(email):
|
||||
warnings.append("Invalid email format")
|
||||
|
||||
pickup_pincode = order_data.get("pickup_address", {}).get("pincode", "")
|
||||
pickup_pincode = self._addr_pincode(order_data, "pickup_address")
|
||||
if not self._is_valid_pincode(pickup_pincode):
|
||||
errors.append("Invalid pickup pincode")
|
||||
|
||||
delivery_pincode = order_data.get("delivery_address", {}).get("pincode", "")
|
||||
delivery_pincode = self._addr_pincode(order_data, "delivery_address")
|
||||
if not self._is_valid_pincode(delivery_pincode):
|
||||
errors.append("Invalid delivery pincode")
|
||||
|
||||
@@ -280,8 +280,8 @@ class OrderAgent(SpecializedAgent):
|
||||
}
|
||||
|
||||
def _categorize_order_data(self, order_data: Dict) -> Dict[str, Any]:
|
||||
pickup_pincode = order_data.get("pickup_address", {}).get("pincode", "")[:3]
|
||||
delivery_pincode = order_data.get("delivery_address", {}).get("pincode", "")[:3]
|
||||
pickup_pincode = self._addr_pincode(order_data, "pickup_address")[:3]
|
||||
delivery_pincode = self._addr_pincode(order_data, "delivery_address")[:3]
|
||||
|
||||
if pickup_pincode == delivery_pincode:
|
||||
zone_type = ZoneType.LAST_MILE.value
|
||||
@@ -330,7 +330,18 @@ class OrderAgent(SpecializedAgent):
|
||||
import re
|
||||
return bool(re.match(self._validation_rules["email_pattern"], email))
|
||||
|
||||
def _is_valid_pincode(self, pincode: str) -> bool:
|
||||
@staticmethod
|
||||
def _addr_pincode(order_data: Dict, key: str) -> str:
|
||||
"""Pincode as a string, tolerant of a missing/None address block or a
|
||||
numeric pincode (valid JSON may send either). Never raises."""
|
||||
addr = order_data.get(key) or {}
|
||||
if not isinstance(addr, dict):
|
||||
return ""
|
||||
pincode = addr.get("pincode")
|
||||
return str(pincode) if pincode is not None else ""
|
||||
|
||||
def _is_valid_pincode(self, pincode) -> bool:
|
||||
pincode = str(pincode) if pincode is not None else ""
|
||||
return len(pincode) == self._validation_rules["pincode_length"] and pincode.isdigit()
|
||||
|
||||
async def think(self, context: str, options: List[str] = None) -> str:
|
||||
|
||||
Reference in New Issue
Block a user