fix(dispatch): liveness-aware coverage check, escalation rate limit, real alert sink
Prod findings (2026-09-22): a backend retry sweep failed 1,001 bookings in 60s; the agent called each a "coverage gap" because GEORADIUS found a miler in the geo index (last seen in June), then sent 1,001 ops_alert tasks to CUSTOMER_AGENT, which has no such handler. - _find_zone filters GEORADIUS candidates by the backend's miler_status:<id> key; only status=Available counts. Facts now carry nearest_available_miler_within_km plus milers_in_geo_index_within_30km so the decision can separate "no riders here" from "riders exist, none on duty". - Rate limit per zone per day: after the first alert, further failures only bump the counter (no LLM call); a summary re-alert goes out every DISPATCH_REALERT_EVERY (default 100). - _ops_alert / _escalate_dispatch send EXCEPTION_DETECTED to JARVIS (the path that is actually handled); customer delay notice uses CUSTOMER_AGENT's real send_notification contract. - JARVIS: escalation inbox (_escalations, pending_escalations()) and human_review/ops_alert task types are recorded instead of dropped. - ExceptionAgent pull loops: also catch asyncio.TimeoutError (distinct from nats.errors.TimeoutError on 3.11) and log the exception type — the blank "pull loop error:" lines. - Prompt + eval cases updated for the renamed facts; new case for the observed index-full/nobody-on-duty pattern. Tests for liveness filtering, burst suppression, fallback heuristic, sinks, and the JARVIS inbox. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_012AJLYcbTHCe45fyFnMfEin
This commit is contained in:
@@ -27,6 +27,18 @@ DISPATCH_AGENT_AUTONOMOUS = os.getenv("DISPATCH_AGENT_AUTONOMOUS", "false").lowe
|
||||
# stream by subject, or pin it to a stream name if discovery isn't desired.
|
||||
DISPATCH_STREAM = os.getenv("DISPATCH_STREAM", "")
|
||||
|
||||
# Escalation rate limit. A backend retry sweep can fail hundreds of bookings in
|
||||
# one zone within a minute (observed: 1,001 in 60 s). Once a zone has been
|
||||
# alerted today, further failures only bump the counter; a summary re-alert goes
|
||||
# out every DISPATCH_REALERT_EVERY failures so ops can see it's still growing.
|
||||
# The LLM is only consulted before the first alert, so cost is bounded per zone.
|
||||
DISPATCH_REALERT_EVERY = int(os.getenv("DISPATCH_REALERT_EVERY", "100"))
|
||||
|
||||
# A miler counts as available only if the Go backend's presence key says so.
|
||||
# The milers:locations geo index keeps last-known positions indefinitely, so
|
||||
# membership alone says nothing about whether anyone is actually on duty.
|
||||
MILER_AVAILABLE_STATUSES = {"available"}
|
||||
|
||||
|
||||
class DispatchAgent(SpecializedAgent):
|
||||
"""
|
||||
@@ -126,34 +138,60 @@ class DispatchAgent(SpecializedAgent):
|
||||
# Redis GEO lookup #
|
||||
# ------------------------------------------------------------------ #
|
||||
|
||||
async def _miler_is_available(self, miler_id) -> bool:
|
||||
"""Liveness check against the backend-owned miler_status:<id> key
|
||||
({"userid": .., "status": "Available" | "Break" | ...}). Missing or
|
||||
unparseable → not available: geo-index presence alone is not evidence."""
|
||||
try:
|
||||
raw = await self._redis.get(f"miler_status:{miler_id}")
|
||||
if not raw:
|
||||
return False
|
||||
status = json.loads(raw).get("status", "")
|
||||
return str(status).lower() in MILER_AVAILABLE_STATUSES
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
async def _find_zone(
|
||||
self, lat: float, lon: float, radius_km: int = 10
|
||||
) -> Optional[Dict[str, Any]]:
|
||||
"""GEORADIUS sweep on milers:locations; returns nearest miler or None."""
|
||||
"""GEORADIUS sweep on milers:locations filtered to milers whose status
|
||||
key says they are available. Returns the nearest available miler (with
|
||||
how many index entries were skipped as stale/off-duty), or None."""
|
||||
try:
|
||||
results = await self._redis.georadius(
|
||||
"milers:locations",
|
||||
lon, lat,
|
||||
radius_km, "km",
|
||||
sort="ASC",
|
||||
count=5,
|
||||
count=10,
|
||||
)
|
||||
if not results:
|
||||
return None
|
||||
|
||||
nearest_miler = results[0]
|
||||
miler_info = await self._redis.hgetall(f"miler:{nearest_miler}")
|
||||
|
||||
return {
|
||||
"miler_id": nearest_miler,
|
||||
"hub_id": miler_info.get("hub_id", "UNKNOWN"),
|
||||
"zone_id": miler_info.get("zone_id", "unknown"),
|
||||
"avg_delivery_time": int(miler_info.get("avg_delivery_time", 60)),
|
||||
}
|
||||
for candidate in results:
|
||||
if await self._miler_is_available(candidate):
|
||||
miler_info = await self._redis.hgetall(f"miler:{candidate}")
|
||||
return {
|
||||
"miler_id": candidate,
|
||||
"hub_id": miler_info.get("hub_id", "UNKNOWN"),
|
||||
"zone_id": miler_info.get("zone_id", "unknown"),
|
||||
"avg_delivery_time": int(miler_info.get("avg_delivery_time", 60)),
|
||||
"stale_candidates_skipped": len(results) - 1,
|
||||
}
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.warning(f"Redis GEORADIUS error: {e}")
|
||||
return None
|
||||
|
||||
async def _count_in_geo_index(self, lat: float, lon: float, radius_km: int = 30) -> Optional[int]:
|
||||
"""Raw geo-index membership (any status) — reported separately so the
|
||||
decision can tell 'nobody registered here' from 'riders exist but none on duty'."""
|
||||
try:
|
||||
results = await self._redis.georadius("milers:locations", lon, lat, radius_km, "km", count=50)
|
||||
return len(results or [])
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
# ------------------------------------------------------------------ #
|
||||
# NATS event handlers #
|
||||
# ------------------------------------------------------------------ #
|
||||
@@ -217,6 +255,23 @@ class DispatchAgent(SpecializedAgent):
|
||||
facts, count = await self._gather_assignment_facts(zone_id, lat, lon)
|
||||
logger.info(f"[DISPATCH] booking={booking_id} context facts: {facts}")
|
||||
|
||||
# Rate limit: once this zone has been alerted today, don't re-decide
|
||||
# (and don't pay for an LLM call) on every further failure.
|
||||
if facts.get("alert_already_sent_today"):
|
||||
if DISPATCH_REALERT_EVERY and count % DISPATCH_REALERT_EVERY == 0:
|
||||
await self._ops_alert(
|
||||
zone_id, booking_id,
|
||||
f"Zone {zone_id}: {count} failed assignments today and still climbing "
|
||||
f"(re-alert every {DISPATCH_REALERT_EVERY}; earlier alert already raised).",
|
||||
severity="high",
|
||||
)
|
||||
else:
|
||||
logger.info(
|
||||
f"[DISPATCH] zone {zone_id} already alerted today — "
|
||||
f"failure #{count} counted, no new alert"
|
||||
)
|
||||
return
|
||||
|
||||
decision = await decide_assignment_failure(build_assignment_failure_context(facts))
|
||||
|
||||
if decision is None:
|
||||
@@ -283,11 +338,17 @@ class DispatchAgent(SpecializedAgent):
|
||||
break
|
||||
|
||||
if not has_coords:
|
||||
facts["nearest_miler_within_km"] = "unknown (no coordinates)"
|
||||
facts["nearest_available_miler_within_km"] = "unknown (no coordinates)"
|
||||
elif nearest_km is None:
|
||||
facts["nearest_miler_within_km"] = "none within 30km"
|
||||
facts["nearest_available_miler_within_km"] = "none within 30km"
|
||||
else:
|
||||
facts["nearest_miler_within_km"] = nearest_km
|
||||
facts["nearest_available_miler_within_km"] = nearest_km
|
||||
|
||||
# Distinguish "no riders registered here" from "riders exist, none on duty".
|
||||
if has_coords:
|
||||
indexed = await self._count_in_geo_index(lat, lon, radius_km=30)
|
||||
if indexed is not None:
|
||||
facts["milers_in_geo_index_within_30km"] = indexed
|
||||
|
||||
count = 0
|
||||
try:
|
||||
@@ -298,25 +359,57 @@ class DispatchAgent(SpecializedAgent):
|
||||
except Exception as e:
|
||||
logger.warning(f"assignment facts: failure counter failed for zone {zone_id}: {e}")
|
||||
facts["failures_today"] = count
|
||||
facts["alert_already_sent_today"] = await self._alert_already_sent_today(zone_id)
|
||||
|
||||
return facts, count
|
||||
|
||||
async def _ops_alert(self, zone_id, booking_id, reasoning):
|
||||
def _alert_key(self, zone_id) -> str:
|
||||
today = datetime.now(timezone.utc).strftime("%Y-%m-%d")
|
||||
return f"assignment_alert_sent:{zone_id}:{today}"
|
||||
|
||||
async def _alert_already_sent_today(self, zone_id) -> bool:
|
||||
try:
|
||||
return bool(await self._redis.exists(self._alert_key(zone_id)))
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
async def _mark_alert_sent(self, zone_id):
|
||||
try:
|
||||
await self._redis.set(self._alert_key(zone_id), "1", ex=172800, nx=True)
|
||||
except Exception as e:
|
||||
logger.warning(f"could not mark alert sent for zone {zone_id}: {e}")
|
||||
|
||||
async def _ops_alert(self, zone_id, booking_id, reasoning, severity="medium"):
|
||||
"""Internal ops alert. Goes to JARVIS as EXCEPTION_DETECTED — the same
|
||||
path the ExceptionAgent uses — which is logged at WARNING and kept in
|
||||
JARVIS's escalation inbox. (Previously sent as an 'ops_alert' task to
|
||||
CUSTOMER_AGENT, which had no handler for it: 1,001 alerts went nowhere.)"""
|
||||
logger.warning(f"[DISPATCH] OPS ALERT zone={zone_id} booking={booking_id}: {reasoning}")
|
||||
await self.send_message(
|
||||
recipient="CUSTOMER_AGENT",
|
||||
message_type=MessageType.AGENT_TASK,
|
||||
payload={"task_type": "ops_alert", "zone_id": zone_id, "reasoning": reasoning},
|
||||
recipient="JARVIS",
|
||||
message_type=MessageType.EXCEPTION_DETECTED,
|
||||
payload={
|
||||
"exception_type": "coverage_gap",
|
||||
"severity": severity,
|
||||
"zone_id": zone_id,
|
||||
"order_id": booking_id,
|
||||
"reasoning": reasoning,
|
||||
"source": self.agent_id,
|
||||
},
|
||||
correlation_id=str(booking_id),
|
||||
)
|
||||
await self._mark_alert_sent(zone_id)
|
||||
|
||||
async def _notify_customer_delay(self, booking_id):
|
||||
await self.send_message(
|
||||
recipient="CUSTOMER_AGENT",
|
||||
message_type=MessageType.AGENT_TASK,
|
||||
payload={
|
||||
"task_type": "customer_notify",
|
||||
"booking_id": booking_id,
|
||||
"message": "We're finding the right rider for your delivery — thanks for your patience.",
|
||||
# CUSTOMER_AGENT's real task contract (send_notification + DELAYED template)
|
||||
"task_type": "send_notification",
|
||||
"order_id": booking_id,
|
||||
"notification_type": "delayed",
|
||||
"template_vars": {"order_id": booking_id, "reason": "finding the right rider"},
|
||||
},
|
||||
correlation_id=str(booking_id),
|
||||
)
|
||||
@@ -328,18 +421,20 @@ class DispatchAgent(SpecializedAgent):
|
||||
)
|
||||
await self.send_message(
|
||||
recipient="JARVIS",
|
||||
message_type=MessageType.AGENT_TASK,
|
||||
message_type=MessageType.EXCEPTION_DETECTED,
|
||||
payload={
|
||||
"task_type": "human_review",
|
||||
"reason": "assignment_failed",
|
||||
"exception_type": "assignment_failed",
|
||||
"severity": "high",
|
||||
"zone_id": zone_id,
|
||||
"booking_id": booking_id,
|
||||
"order_id": booking_id,
|
||||
"proposed_action": decision.action,
|
||||
"confidence": decision.confidence,
|
||||
"reasoning": decision.reasoning,
|
||||
"source": self.agent_id,
|
||||
},
|
||||
correlation_id=str(booking_id),
|
||||
)
|
||||
await self._mark_alert_sent(zone_id)
|
||||
|
||||
# ------------------------------------------------------------------ #
|
||||
# Task handler (no inbound task types remain) #
|
||||
|
||||
Reference in New Issue
Block a user