fix(dispatch): liveness-aware coverage check, escalation rate limit, real alert sink

Prod findings (2026-09-22): a backend retry sweep failed 1,001 bookings in
60s; the agent called each a "coverage gap" because GEORADIUS found a miler
in the geo index (last seen in June), then sent 1,001 ops_alert tasks to
CUSTOMER_AGENT, which has no such handler.

- _find_zone filters GEORADIUS candidates by the backend's miler_status:<id>
  key; only status=Available counts. Facts now carry
  nearest_available_miler_within_km plus milers_in_geo_index_within_30km so
  the decision can separate "no riders here" from "riders exist, none on duty".
- Rate limit per zone per day: after the first alert, further failures only
  bump the counter (no LLM call); a summary re-alert goes out every
  DISPATCH_REALERT_EVERY (default 100).
- _ops_alert / _escalate_dispatch send EXCEPTION_DETECTED to JARVIS (the path
  that is actually handled); customer delay notice uses CUSTOMER_AGENT's real
  send_notification contract.
- JARVIS: escalation inbox (_escalations, pending_escalations()) and
  human_review/ops_alert task types are recorded instead of dropped.
- ExceptionAgent pull loops: also catch asyncio.TimeoutError (distinct from
  nats.errors.TimeoutError on 3.11) and log the exception type — the blank
  "pull loop error:" lines.
- Prompt + eval cases updated for the renamed facts; new case for the
  observed index-full/nobody-on-duty pattern. Tests for liveness filtering,
  burst suppression, fallback heuristic, sinks, and the JARVIS inbox.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_012AJLYcbTHCe45fyFnMfEin
This commit is contained in:
2026-09-22 16:26:27 +05:30
parent 58bfa07385
commit 4283c602f6
8 changed files with 376 additions and 54 deletions

View File

@@ -27,6 +27,18 @@ DISPATCH_AGENT_AUTONOMOUS = os.getenv("DISPATCH_AGENT_AUTONOMOUS", "false").lowe
# stream by subject, or pin it to a stream name if discovery isn't desired.
DISPATCH_STREAM = os.getenv("DISPATCH_STREAM", "")
# Escalation rate limit. A backend retry sweep can fail hundreds of bookings in
# one zone within a minute (observed: 1,001 in 60 s). Once a zone has been
# alerted today, further failures only bump the counter; a summary re-alert goes
# out every DISPATCH_REALERT_EVERY failures so ops can see it's still growing.
# The LLM is only consulted before the first alert, so cost is bounded per zone.
DISPATCH_REALERT_EVERY = int(os.getenv("DISPATCH_REALERT_EVERY", "100"))
# A miler counts as available only if the Go backend's presence key says so.
# The milers:locations geo index keeps last-known positions indefinitely, so
# membership alone says nothing about whether anyone is actually on duty.
MILER_AVAILABLE_STATUSES = {"available"}
class DispatchAgent(SpecializedAgent):
"""
@@ -126,34 +138,60 @@ class DispatchAgent(SpecializedAgent):
# Redis GEO lookup #
# ------------------------------------------------------------------ #
async def _miler_is_available(self, miler_id) -> bool:
"""Liveness check against the backend-owned miler_status:<id> key
({"userid": .., "status": "Available" | "Break" | ...}). Missing or
unparseable → not available: geo-index presence alone is not evidence."""
try:
raw = await self._redis.get(f"miler_status:{miler_id}")
if not raw:
return False
status = json.loads(raw).get("status", "")
return str(status).lower() in MILER_AVAILABLE_STATUSES
except Exception:
return False
async def _find_zone(
self, lat: float, lon: float, radius_km: int = 10
) -> Optional[Dict[str, Any]]:
"""GEORADIUS sweep on milers:locations; returns nearest miler or None."""
"""GEORADIUS sweep on milers:locations filtered to milers whose status
key says they are available. Returns the nearest available miler (with
how many index entries were skipped as stale/off-duty), or None."""
try:
results = await self._redis.georadius(
"milers:locations",
lon, lat,
radius_km, "km",
sort="ASC",
count=5,
count=10,
)
if not results:
return None
nearest_miler = results[0]
miler_info = await self._redis.hgetall(f"miler:{nearest_miler}")
return {
"miler_id": nearest_miler,
"hub_id": miler_info.get("hub_id", "UNKNOWN"),
"zone_id": miler_info.get("zone_id", "unknown"),
"avg_delivery_time": int(miler_info.get("avg_delivery_time", 60)),
}
for candidate in results:
if await self._miler_is_available(candidate):
miler_info = await self._redis.hgetall(f"miler:{candidate}")
return {
"miler_id": candidate,
"hub_id": miler_info.get("hub_id", "UNKNOWN"),
"zone_id": miler_info.get("zone_id", "unknown"),
"avg_delivery_time": int(miler_info.get("avg_delivery_time", 60)),
"stale_candidates_skipped": len(results) - 1,
}
return None
except Exception as e:
logger.warning(f"Redis GEORADIUS error: {e}")
return None
async def _count_in_geo_index(self, lat: float, lon: float, radius_km: int = 30) -> Optional[int]:
"""Raw geo-index membership (any status) — reported separately so the
decision can tell 'nobody registered here' from 'riders exist but none on duty'."""
try:
results = await self._redis.georadius("milers:locations", lon, lat, radius_km, "km", count=50)
return len(results or [])
except Exception:
return None
# ------------------------------------------------------------------ #
# NATS event handlers #
# ------------------------------------------------------------------ #
@@ -217,6 +255,23 @@ class DispatchAgent(SpecializedAgent):
facts, count = await self._gather_assignment_facts(zone_id, lat, lon)
logger.info(f"[DISPATCH] booking={booking_id} context facts: {facts}")
# Rate limit: once this zone has been alerted today, don't re-decide
# (and don't pay for an LLM call) on every further failure.
if facts.get("alert_already_sent_today"):
if DISPATCH_REALERT_EVERY and count % DISPATCH_REALERT_EVERY == 0:
await self._ops_alert(
zone_id, booking_id,
f"Zone {zone_id}: {count} failed assignments today and still climbing "
f"(re-alert every {DISPATCH_REALERT_EVERY}; earlier alert already raised).",
severity="high",
)
else:
logger.info(
f"[DISPATCH] zone {zone_id} already alerted today — "
f"failure #{count} counted, no new alert"
)
return
decision = await decide_assignment_failure(build_assignment_failure_context(facts))
if decision is None:
@@ -283,11 +338,17 @@ class DispatchAgent(SpecializedAgent):
break
if not has_coords:
facts["nearest_miler_within_km"] = "unknown (no coordinates)"
facts["nearest_available_miler_within_km"] = "unknown (no coordinates)"
elif nearest_km is None:
facts["nearest_miler_within_km"] = "none within 30km"
facts["nearest_available_miler_within_km"] = "none within 30km"
else:
facts["nearest_miler_within_km"] = nearest_km
facts["nearest_available_miler_within_km"] = nearest_km
# Distinguish "no riders registered here" from "riders exist, none on duty".
if has_coords:
indexed = await self._count_in_geo_index(lat, lon, radius_km=30)
if indexed is not None:
facts["milers_in_geo_index_within_30km"] = indexed
count = 0
try:
@@ -298,25 +359,57 @@ class DispatchAgent(SpecializedAgent):
except Exception as e:
logger.warning(f"assignment facts: failure counter failed for zone {zone_id}: {e}")
facts["failures_today"] = count
facts["alert_already_sent_today"] = await self._alert_already_sent_today(zone_id)
return facts, count
async def _ops_alert(self, zone_id, booking_id, reasoning):
def _alert_key(self, zone_id) -> str:
today = datetime.now(timezone.utc).strftime("%Y-%m-%d")
return f"assignment_alert_sent:{zone_id}:{today}"
async def _alert_already_sent_today(self, zone_id) -> bool:
try:
return bool(await self._redis.exists(self._alert_key(zone_id)))
except Exception:
return False
async def _mark_alert_sent(self, zone_id):
try:
await self._redis.set(self._alert_key(zone_id), "1", ex=172800, nx=True)
except Exception as e:
logger.warning(f"could not mark alert sent for zone {zone_id}: {e}")
async def _ops_alert(self, zone_id, booking_id, reasoning, severity="medium"):
"""Internal ops alert. Goes to JARVIS as EXCEPTION_DETECTED — the same
path the ExceptionAgent uses — which is logged at WARNING and kept in
JARVIS's escalation inbox. (Previously sent as an 'ops_alert' task to
CUSTOMER_AGENT, which had no handler for it: 1,001 alerts went nowhere.)"""
logger.warning(f"[DISPATCH] OPS ALERT zone={zone_id} booking={booking_id}: {reasoning}")
await self.send_message(
recipient="CUSTOMER_AGENT",
message_type=MessageType.AGENT_TASK,
payload={"task_type": "ops_alert", "zone_id": zone_id, "reasoning": reasoning},
recipient="JARVIS",
message_type=MessageType.EXCEPTION_DETECTED,
payload={
"exception_type": "coverage_gap",
"severity": severity,
"zone_id": zone_id,
"order_id": booking_id,
"reasoning": reasoning,
"source": self.agent_id,
},
correlation_id=str(booking_id),
)
await self._mark_alert_sent(zone_id)
async def _notify_customer_delay(self, booking_id):
await self.send_message(
recipient="CUSTOMER_AGENT",
message_type=MessageType.AGENT_TASK,
payload={
"task_type": "customer_notify",
"booking_id": booking_id,
"message": "We're finding the right rider for your delivery — thanks for your patience.",
# CUSTOMER_AGENT's real task contract (send_notification + DELAYED template)
"task_type": "send_notification",
"order_id": booking_id,
"notification_type": "delayed",
"template_vars": {"order_id": booking_id, "reason": "finding the right rider"},
},
correlation_id=str(booking_id),
)
@@ -328,18 +421,20 @@ class DispatchAgent(SpecializedAgent):
)
await self.send_message(
recipient="JARVIS",
message_type=MessageType.AGENT_TASK,
message_type=MessageType.EXCEPTION_DETECTED,
payload={
"task_type": "human_review",
"reason": "assignment_failed",
"exception_type": "assignment_failed",
"severity": "high",
"zone_id": zone_id,
"booking_id": booking_id,
"order_id": booking_id,
"proposed_action": decision.action,
"confidence": decision.confidence,
"reasoning": decision.reasoning,
"source": self.agent_id,
},
correlation_id=str(booking_id),
)
await self._mark_alert_sent(zone_id)
# ------------------------------------------------------------------ #
# Task handler (no inbound task types remain) #

View File

@@ -204,10 +204,10 @@ class ExceptionAgent(SpecializedAgent):
logger.error(f"Location handler error: {e}")
finally:
await msg.ack()
except nats.errors.TimeoutError:
pass
except (nats.errors.TimeoutError, asyncio.TimeoutError):
pass # empty fetch — normal
except Exception as e:
logger.error(f"Location pull loop error: {e}")
logger.error(f"Location pull loop error: {type(e).__name__}: {e}")
await asyncio.sleep(2)
async def _pull_stalled_loop(self):
@@ -224,10 +224,10 @@ class ExceptionAgent(SpecializedAgent):
logger.error(f"Stalled handler error: {e}")
finally:
await msg.ack()
except nats.errors.TimeoutError:
pass
except (nats.errors.TimeoutError, asyncio.TimeoutError):
pass # empty fetch — normal
except Exception as e:
logger.error(f"Stalled pull loop error: {e}")
logger.error(f"Stalled pull loop error: {type(e).__name__}: {e}")
await asyncio.sleep(2)
# ── Stall detection: per location ping ───────────────────────────────────