implemenation on the ai agents and registry

This commit is contained in:
2026-09-30 14:56:08 +05:30
parent aa3bb35733
commit 5a7d32cc04
15 changed files with 803 additions and 72 deletions

View File

@@ -26,7 +26,9 @@ from core.agent import SpecializedAgent
from core.types import AgentTask, MessageType
from core.logger import logger
from core.http_client import api_post
from core.llm import decide_stall_response, build_stall_context, StallDecision
from core.llm import decide_stall_response, build_stall_context, StallDecision, LLM_MODEL
from core.registry import registry
from core.decisions import record_decision
from config.system_config import (
GO_API_BASE_URL, INTERNAL_API_KEY,
DB_HOST, DB_PORT, DB_NAME, DB_USER, DB_PASSWORD,
@@ -34,7 +36,12 @@ from config.system_config import (
REDIS_HOST, REDIS_PORT, REDIS_PASSWORD,
)
STALL_MINUTES = 10
STALL_MINUTES = 10 # default; the agent registry's stall_response.stallMinutes overrides it
def stall_minutes() -> float:
"""Minutes without movement before a rider counts as stalled (registry, else 10)."""
return registry.threshold("stall_response", "stallMinutes", STALL_MINUTES)
ACTIVE_STATUSES = ["Miler_Assigned", "Pickup_Scheduled"]
TRACKING_STREAM = "TRACKING"
@@ -260,7 +267,7 @@ class ExceptionAgent(SpecializedAgent):
unchanged_since = _parse_ts(prev.get("position_unchanged_since")) or _utcnow()
minutes_stalled = (_utcnow() - unchanged_since).total_seconds() / 60
if minutes_stalled >= STALL_MINUTES:
if minutes_stalled >= stall_minutes():
booking = await self._get_active_booking(miler_id)
if booking:
await self._publish_stall(miler_id, booking["booking_id"], minutes_stalled)
@@ -311,7 +318,7 @@ class ExceptionAgent(SpecializedAgent):
continue
minutes_stale = (now - updated_at).total_seconds() / 60
if minutes_stale >= STALL_MINUTES:
if minutes_stale >= stall_minutes():
logger.warning(f"StallDetector: miler {miler_id} stale {minutes_stale:.1f} min (booking {booking_id})")
await self._publish_stall(miler_id, booking_id, minutes_stale)
@@ -385,15 +392,24 @@ class ExceptionAgent(SpecializedAgent):
Claude chooses one of wait / notify_only / reassign / escalate. The only
irreversible, customer-visible action (reassign) is gated behind an
autonomy flag and a confidence threshold; otherwise it is escalated to a
human via JARVIS. If the LLM is unavailable, we fall back to the previous
deterministic behaviour (reassign + notify) so detection never silently
stops acting.
human via JARVIS. If the LLM is unavailable: an autonomous agent falls
back to reassign + notify so detection never silently stops acting; a
non-autonomous one escalates to a human instead (it used to reassign
regardless, which made "autonomy off" untrue during an LLM outage).
Settings come from the agent registry (core/registry.py), falling back
to env: skill `stall_response` on/off, its `stallMinutes` and
`reassignConfidence`, and this agent's autonomy and model.
"""
try:
data = json.loads(msg.data.decode())
except Exception:
return
if not registry.skill_enabled("stall_response"):
logger.info("Stall received but skill stall_response is disabled in the agent registry; not acting")
return
miler_id = data.get("miler_id", "")
booking_id = data.get("booking_id", "")
minutes_stalled = data.get("minutes_stalled", 0)
@@ -409,21 +425,37 @@ class ExceptionAgent(SpecializedAgent):
logger.info(f"[EXCEPTION] booking={booking_id} context facts: {facts}")
context = build_stall_context(
minutes_stalled, stall_threshold_min=STALL_MINUTES, now=_utcnow(), facts=facts,
minutes_stalled, stall_threshold_min=stall_minutes(), now=_utcnow(), facts=facts,
)
decision = await decide_stall_response(context)
model = registry.model("EXCEPTION_AGENT")
decision = await decide_stall_response(context, model)
autonomous = registry.autonomous("EXCEPTION_AGENT", AUTONOMOUS_REASSIGN)
if decision is None:
logger.warning(f"No LLM decision for booking {booking_id}; falling back to reassign + notify")
await self._reassign(booking_id, "miler_stalled")
await self._notify_customer(booking_id, "We detected a delay, finding you a new miler")
self._record_stall_exception(
miler_id, booking_id, minutes_stalled,
resolution="Fallback (LLM unavailable): reassignment triggered + customer notified",
actions=["reassign", "notify_customer"],
)
if autonomous:
logger.warning(f"No LLM decision for booking {booking_id}; autonomous — falling back to reassign + notify")
await self._reassign(booking_id, "miler_stalled")
await self._notify_customer(booking_id, "We detected a delay, finding you a new miler")
self._record_stall_exception(
miler_id, booking_id, minutes_stalled,
resolution="Fallback (LLM unavailable): reassignment triggered + customer notified",
actions=["reassign", "notify_customer"],
)
else:
logger.warning(f"No LLM decision for booking {booking_id}; not autonomous — escalating to a human")
await self._escalate_to_human(miler_id, booking_id, minutes_stalled, StallDecision(
action="escalate", reasoning="LLM unavailable; autonomy is off, so a human decides.", confidence=0.0,
))
self._record_stall_exception(
miler_id, booking_id, minutes_stalled,
resolution="Fallback (LLM unavailable, autonomy off): escalated to a human",
actions=["escalate"],
)
return
record_decision("stall_response", booking_id, facts, decision.action, decision.confidence,
decision.reasoning, model or LLM_MODEL)
logger.info(
f"[EXCEPTION] booking={booking_id} decision={decision.action} "
f"confidence={decision.confidence:.2f} reasoning={decision.reasoning!r}"
@@ -441,7 +473,8 @@ class ExceptionAgent(SpecializedAgent):
actions.append("notify_customer")
elif decision.action == "reassign":
if AUTONOMOUS_REASSIGN and decision.confidence >= REASSIGN_MIN_CONFIDENCE:
min_confidence = registry.threshold("stall_response", "reassignConfidence", REASSIGN_MIN_CONFIDENCE)
if autonomous and decision.confidence >= min_confidence:
await self._reassign(booking_id, "miler_stalled")
await self._notify_customer(booking_id, "We detected a delay, finding you a new miler")
actions += ["reassign", "notify_customer"]