route optimizer agent with dailygrubs ai assign

This commit is contained in:
2026-09-01 13:48:58 +05:30
parent aa4d4c6549
commit 90cdb57a38
28 changed files with 3281 additions and 135 deletions

View File

@@ -10,8 +10,9 @@ StallDetector sweep (every 60 s):
"""
import asyncio
import json
import os
import uuid
from datetime import datetime, timedelta
from datetime import datetime, timedelta, timezone
from typing import Dict, List, Any, Optional, Set
from dataclasses import dataclass, field
from enum import Enum
@@ -25,6 +26,7 @@ from core.agent import SpecializedAgent
from core.types import AgentTask, MessageType
from core.logger import logger
from core.http_client import api_post
from core.llm import decide_stall_response, build_stall_context, StallDecision
from config.system_config import (
GO_API_BASE_URL, INTERNAL_API_KEY,
DB_HOST, DB_PORT, DB_NAME, DB_USER, DB_PASSWORD,
@@ -36,6 +38,35 @@ STALL_MINUTES = 10
ACTIVE_STATUSES = ["Miler_Assigned", "Pickup_Scheduled"]
TRACKING_STREAM = "TRACKING"
# Reassignment is customer-visible and hard to reverse, so it is gated: the LLM
# only *proposes* a reassign unless the agent is explicitly put in autonomous
# mode AND the model is confident enough. Otherwise the proposal is escalated to
# a human via JARVIS.
AUTONOMOUS_REASSIGN = os.getenv("EXCEPTION_AGENT_AUTONOMOUS", "false").lower() == "true"
REASSIGN_MIN_CONFIDENCE = float(os.getenv("EXCEPTION_AGENT_REASSIGN_CONFIDENCE", "0.7"))
# How long a booking stays "already alerted" so we don't re-alert the same stall.
# Self-expiring so dedup state is bounded and survives restarts (unlike a set).
STALL_NOTIFIED_TTL = int(os.getenv("STALL_NOTIFIED_TTL_SEC", "21600")) # 6h
def _utcnow() -> datetime:
"""Timezone-aware current time in UTC. All stall math uses this so it stays
correct regardless of the container's local timezone, and never mixes naive
with tz-aware timestamps."""
return datetime.now(timezone.utc)
def _parse_ts(value: Optional[str]) -> Optional[datetime]:
"""Parse an ISO timestamp as tz-aware UTC. Naive values are assumed UTC so
pre-existing (naive) Redis entries remain comparable during rollout."""
if not value:
return None
try:
dt = datetime.fromisoformat(value)
except (ValueError, TypeError):
return None
return dt if dt.tzinfo else dt.replace(tzinfo=timezone.utc)
class ExceptionType(str, Enum):
DELAY = "delay"
@@ -95,7 +126,6 @@ class ExceptionAgent(SpecializedAgent):
self._location_sub = None
self._stalled_sub = None
self._notified_stalls: Set[str] = set()
self._exceptions: Dict[str, ExceptionRecord] = {}
self._strategies = self._init_strategies()
self._escalation_rules = self._init_escalation_rules()
@@ -211,7 +241,7 @@ class ExceptionAgent(SpecializedAgent):
miler_id = str(data.get("miler_id", ""))
lat = str(round(float(data.get("lat", 0)), 6))
lon = str(round(float(data.get("lon", 0)), 6))
now_ts = datetime.now().isoformat()
now_ts = _utcnow().isoformat()
redis_key = f"miler:{miler_id}:movement"
prev = await self._redis.hgetall(redis_key)
@@ -227,20 +257,13 @@ class ExceptionAgent(SpecializedAgent):
else:
await self._redis.hset(redis_key, mapping={"lat": lat, "lon": lon, "updated_at": now_ts})
unchanged_since_str = prev.get("position_unchanged_since", now_ts)
try:
unchanged_since = datetime.fromisoformat(unchanged_since_str)
except ValueError:
unchanged_since = datetime.now()
minutes_stalled = (datetime.now() - unchanged_since).total_seconds() / 60
unchanged_since = _parse_ts(prev.get("position_unchanged_since")) or _utcnow()
minutes_stalled = (_utcnow() - unchanged_since).total_seconds() / 60
if minutes_stalled >= STALL_MINUTES:
booking = await self._get_active_booking(miler_id)
if booking:
booking_id = booking["booking_id"]
if booking_id not in self._notified_stalls:
await self._publish_stall(miler_id, booking_id, minutes_stalled)
await self._publish_stall(miler_id, booking["booking_id"], minutes_stalled)
# ── Stall detection: background sweep ────────────────────────────────────
@@ -274,23 +297,19 @@ class ExceptionAgent(SpecializedAgent):
logger.error(f"StallDetector Postgres error: {e}")
return
now = datetime.now()
now = _utcnow()
for row in rows:
miler_id = row["miler_id"]
booking_id = row["booking_id"]
if booking_id in self._notified_stalls:
continue
movement = await self._redis.hgetall(f"miler:{miler_id}:movement")
if not movement or "updated_at" not in movement:
continue
try:
updated_at = datetime.fromisoformat(movement["updated_at"])
minutes_stale = (now - updated_at).total_seconds() / 60
except ValueError:
updated_at = _parse_ts(movement.get("updated_at"))
if updated_at is None:
continue
minutes_stale = (now - updated_at).total_seconds() / 60
if minutes_stale >= STALL_MINUTES:
logger.warning(f"StallDetector: miler {miler_id} stale {minutes_stale:.1f} min (booking {booking_id})")
@@ -298,8 +317,57 @@ class ExceptionAgent(SpecializedAgent):
# ── Stall: publish + handle ───────────────────────────────────────────────
def _stall_counter_key(self, miler_id) -> str:
return f"miler:{miler_id}:stalls:{_utcnow().strftime('%Y-%m-%d')}"
def _stall_notified_key(self, booking_id) -> str:
return f"stall_notified:{booking_id}"
def _stall_handled_key(self, booking_id) -> str:
return f"stall_handled:{booking_id}"
async def _claim_stall(self, booking_id) -> bool:
"""Atomically claim the first stall ALERT for a booking (Redis SET NX with
TTL). True = publish this alert; False = already alerted within the window.
Replaces the old unbounded in-memory set: self-expiring (bounded memory),
durable across restarts, and race-free between the ping and sweep paths.
Fail-open on Redis error so a genuine stall is never silently muted."""
try:
got = await self._redis.set(
self._stall_notified_key(booking_id), "1", ex=STALL_NOTIFIED_TTL, nx=True,
)
return bool(got)
except Exception as e:
logger.warning(f"stall claim failed for {booking_id}: {e}")
return True
async def _claim_stall_handled(self, booking_id) -> bool:
"""Atomically claim HANDLING of a stall event (consumer side). Guards
against JetStream at-least-once redelivery re-running the decision and
repeating an irreversible reassign. Fail-open on Redis error."""
try:
got = await self._redis.set(
self._stall_handled_key(booking_id), "1", ex=STALL_NOTIFIED_TTL, nx=True,
)
return bool(got)
except Exception as e:
logger.warning(f"stall handled-claim failed for {booking_id}: {e}")
return True
async def _incr_stall_counter(self, miler_id):
"""Bump this miler's per-day stall count (Redis, best-effort). 48h TTL
covers the day rollover; failures never block stall handling."""
try:
key = self._stall_counter_key(miler_id)
await self._redis.incr(key)
await self._redis.expire(key, 172800)
except Exception as e:
logger.warning(f"stall counter incr failed for miler {miler_id}: {e}")
async def _publish_stall(self, miler_id: str, booking_id: str, minutes_stalled: float):
self._notified_stalls.add(booking_id)
if not await self._claim_stall(booking_id):
return # already alerted for this booking within the dedup window
await self._incr_stall_counter(miler_id)
payload = json.dumps({
"miler_id": miler_id,
"booking_id": booking_id,
@@ -312,7 +380,15 @@ class ExceptionAgent(SpecializedAgent):
logger.error(f"Failed to publish miler.stalled: {e}")
async def _on_miler_stalled(self, msg):
"""Reassign booking and notify customer via Go API."""
"""Decide (via Claude) how to handle a stalled miler, then act.
Claude chooses one of wait / notify_only / reassign / escalate. The only
irreversible, customer-visible action (reassign) is gated behind an
autonomy flag and a confidence threshold; otherwise it is escalated to a
human via JARVIS. If the LLM is unavailable, we fall back to the previous
deterministic behaviour (reassign + notify) so detection never silently
stops acting.
"""
try:
data = json.loads(msg.data.decode())
except Exception:
@@ -324,36 +400,175 @@ class ExceptionAgent(SpecializedAgent):
logger.warning(f"EXCEPTION_AGENT stall: miler={miler_id} booking={booking_id} ({minutes_stalled} min)")
headers = {"X-Internal-Key": INTERNAL_API_KEY}
if not await self._claim_stall_handled(booking_id):
logger.info(f"Stall for booking {booking_id} already handled; skipping redelivery")
return
reassign_result = await api_post(
facts = await self._gather_stall_facts(miler_id, booking_id)
if facts:
logger.info(f"[EXCEPTION] booking={booking_id} context facts: {facts}")
context = build_stall_context(
minutes_stalled, stall_threshold_min=STALL_MINUTES, now=_utcnow(), facts=facts,
)
decision = await decide_stall_response(context)
if decision is None:
logger.warning(f"No LLM decision for booking {booking_id}; falling back to reassign + notify")
await self._reassign(booking_id, "miler_stalled")
await self._notify_customer(booking_id, "We detected a delay, finding you a new miler")
self._record_stall_exception(
miler_id, booking_id, minutes_stalled,
resolution="Fallback (LLM unavailable): reassignment triggered + customer notified",
actions=["reassign", "notify_customer"],
)
return
logger.info(
f"[EXCEPTION] booking={booking_id} decision={decision.action} "
f"confidence={decision.confidence:.2f} reasoning={decision.reasoning!r}"
)
actions: List[str] = []
if decision.action == "wait":
actions.append("wait")
elif decision.action == "notify_only":
await self._notify_customer(
booking_id, "We're keeping an eye on your delivery — thanks for your patience."
)
actions.append("notify_customer")
elif decision.action == "reassign":
if AUTONOMOUS_REASSIGN and decision.confidence >= REASSIGN_MIN_CONFIDENCE:
await self._reassign(booking_id, "miler_stalled")
await self._notify_customer(booking_id, "We detected a delay, finding you a new miler")
actions += ["reassign", "notify_customer"]
else:
# Not authorised to auto-reassign (mode off or low confidence):
# hand the proposal to a human rather than taking the action.
await self._escalate_to_human(miler_id, booking_id, minutes_stalled, decision)
actions.append("escalate_reassign_for_review")
else: # "escalate"
await self._escalate_to_human(miler_id, booking_id, minutes_stalled, decision)
actions.append("escalate")
self._record_stall_exception(
miler_id, booking_id, minutes_stalled,
resolution=f"[{decision.action}] {decision.reasoning}",
actions=actions,
)
# ── Stall: action helpers ─────────────────────────────────────────────────
async def _reassign(self, booking_id: str, reason: str) -> bool:
result = await api_post(
f"{GO_API_BASE_URL}/api/v1/internal/bookings/{booking_id}/reassign",
json={"reason": "miler_stalled"},
headers=headers,
json={"reason": reason},
headers={"X-Internal-Key": INTERNAL_API_KEY},
)
logger.info(f"Reassign {'OK' if reassign_result is not None else 'FAILED'} for booking {booking_id}")
ok = result is not None
logger.info(f"Reassign {'OK' if ok else 'FAILED'} for booking {booking_id}")
return ok
notify_result = await api_post(
async def _notify_customer(self, booking_id: str, message: str) -> bool:
result = await api_post(
f"{GO_API_BASE_URL}/api/v1/internal/notify",
json={
"booking_id": booking_id,
"message": "We detected a delay, finding you a new miler",
"target": "customer",
},
headers=headers,
json={"booking_id": booking_id, "message": message, "target": "customer"},
headers={"X-Internal-Key": INTERNAL_API_KEY},
)
logger.info(f"Notify {'OK' if notify_result is not None else 'FAILED'} for booking {booking_id}")
ok = result is not None
logger.info(f"Notify {'OK' if ok else 'FAILED'} for booking {booking_id}")
return ok
exc_id = f"EXC-{datetime.now().strftime('%Y%m%d')}-{uuid.uuid4().hex[:6].upper()}"
async def _escalate_to_human(self, miler_id, booking_id, minutes_stalled, decision: StallDecision):
logger.warning(
f"[EXCEPTION] Escalating booking {booking_id} to human dispatcher "
f"(proposed={decision.action}, confidence={decision.confidence:.2f}): {decision.reasoning}"
)
await self.send_message(
recipient="JARVIS",
message_type=MessageType.EXCEPTION_DETECTED,
payload={
"order_id": booking_id,
"miler_id": miler_id,
"exception_type": ExceptionType.MILER_STALLED.value,
"severity": ExceptionSeverity.HIGH.value,
"proposed_action": decision.action,
"confidence": decision.confidence,
"reasoning": decision.reasoning,
"minutes_stalled": minutes_stalled,
"urgency": "high",
},
)
async def _gather_stall_facts(self, miler_id, booking_id) -> Dict[str, Any]:
"""Read-only context gathering: pull real signals from Postgres + Redis
to inform the decision. Best-effort — any failure is logged and skipped,
never raised, so a thin-context decision still runs (which skews safe)."""
facts: Dict[str, Any] = {}
now = _utcnow()
# Booking phase + age (Postgres, read-only).
if self._pg:
try:
async with self._pg.acquire() as conn:
row = await conn.fetchrow(
"SELECT status, createdat FROM pickupbookings WHERE bookingid::text = $1",
str(booking_id),
)
if row:
facts["booking_status"] = row["status"]
created = row["createdat"]
if created is not None:
if created.tzinfo is None:
created = created.replace(tzinfo=timezone.utc)
facts["minutes_since_booking_created"] = round((now - created).total_seconds() / 60, 1)
except Exception as e:
logger.warning(f"stall facts: booking query failed for {booking_id}: {e}")
# Movement freshness (Redis, read-only).
try:
movement = await self._redis.hgetall(f"miler:{miler_id}:movement")
except Exception as e:
logger.warning(f"stall facts: redis read failed for miler {miler_id}: {e}")
movement = {}
if movement:
last_ping = _parse_ts(movement.get("updated_at"))
if last_ping:
facts["last_gps_ping_minutes_ago"] = round((now - last_ping).total_seconds() / 60, 1)
unchanged = _parse_ts(movement.get("position_unchanged_since"))
if unchanged:
facts["position_unchanged_minutes"] = round((now - unchanged).total_seconds() / 60, 1)
# Per-miler stall count today (Redis, read-only). Already includes the
# current stall, since _publish_stall increments before the consumer runs.
try:
raw = await self._redis.get(self._stall_counter_key(miler_id))
if raw is not None:
facts["stalls_today"] = int(raw)
except Exception as e:
logger.warning(f"stall facts: stall counter read failed for miler {miler_id}: {e}")
return facts
def _record_stall_exception(self, miler_id, booking_id, minutes_stalled, resolution, actions):
escalated = any(a.startswith("escalate") for a in actions)
exc_id = f"EXC-{_utcnow().strftime('%Y%m%d')}-{uuid.uuid4().hex[:6].upper()}"
self._exceptions[exc_id] = ExceptionRecord(
exception_id=exc_id, order_id=booking_id,
exception_type=ExceptionType.MILER_STALLED,
severity=ExceptionSeverity.HIGH,
description=f"Miler {miler_id} stalled {minutes_stalled} min",
detected_at=datetime.now(), resolved_at=datetime.now(),
resolution="Reassignment triggered + customer notified",
assigned_to=None, status="resolved",
actions_taken=["reassign", "notify_customer"],
detected_at=_utcnow(),
resolved_at=None if escalated else _utcnow(),
resolution=resolution,
assigned_to=None,
status="escalated" if escalated else "resolved",
actions_taken=list(actions),
)
# ── Postgres helper ───────────────────────────────────────────────────────