implemenation on the ai agents and registry

This commit is contained in:
2026-09-30 14:56:08 +05:30
parent aa3bb35733
commit 5a7d32cc04
15 changed files with 803 additions and 72 deletions

View File

@@ -11,6 +11,7 @@ from core.agent import SpecializedAgent
from core.types import AgentTask, MessageType, OrderStatus
from core.logger import logger
from core.http_client import api_post, api_get
from core.registry import registry
from config.system_config import GO_API_BASE_URL, INTERNAL_API_KEY
@@ -64,6 +65,32 @@ class CustomerAgent(SpecializedAgent):
self._notifications: Dict[str, List[Notification]] = {}
self._templates = self._init_templates()
self._preferences: Dict[str, Dict] = {}
self._received_events: List[Dict[str, Any]] = []
# Event notifications other agents send without a task_type. They used to
# fall through to the base handle_message and be dropped at DEBUG level.
# They come from ExceptionAgent's generic exception tasks, which run over
# in-memory, simulated records (their order ids are not real bookings), so
# they are RECORDED here and logged — deliberately not turned into a real
# customer message through the backend's /internal/notify.
_RECORDED_EVENTS = {MessageType.ORDER_CANCELLED, MessageType.NOTIFICATION_SENT}
async def handle_message(self, message) -> None:
if message.message_type not in self._RECORDED_EVENTS:
await super().handle_message(message)
return
payload = message.payload if isinstance(message.payload, dict) else {}
event = {
"type": message.message_type.value,
"from": message.sender,
"order_id": payload.get("order_id"),
"received_at": datetime.now().isoformat(),
}
self._received_events = (self._received_events + [event])[-200:]
logger.info(
f"Customer Agent: recorded {event['type']} from {event['from']} for order {event['order_id']} "
"(simulated exception flow; no customer message sent)"
)
def _init_templates(self) -> Dict[str, Dict]:
return {
@@ -186,6 +213,9 @@ class CustomerAgent(SpecializedAgent):
return {"status": "sent", "order_id": order_id, "notifications": notifications_sent}
async def _send_notification(self, task: AgentTask) -> Dict[str, Any]:
if not registry.skill_enabled("customer_notifications"):
logger.info("Customer Agent: skill customer_notifications is disabled in the agent registry; not sending")
return {"status": "skipped", "reason": "customer_notifications disabled in the agent registry"}
order_id = task.data.get("order_id")
notification_type = task.data.get("notification_type")
template_vars = task.data.get("template_vars", {})

View File

@@ -12,7 +12,9 @@ import redis.asyncio as aioredis
from core.agent import SpecializedAgent
from core.types import AgentTask, MessageType
from core.logger import logger
from core.llm import decide_assignment_failure, build_assignment_failure_context
from core.llm import decide_assignment_failure, build_assignment_failure_context, LLM_MODEL
from core.registry import registry
from core.decisions import record_decision
from config.system_config import (
NATS_HOST, NATS_PORT, NATS_USER, NATS_PASSWORD,
REDIS_HOST, REDIS_PORT, REDIS_PASSWORD,
@@ -264,18 +266,12 @@ class DispatchAgent(SpecializedAgent):
f"miler={miler_id} hub={hub_id} confidence={confidence} "
f"reasoning={reasoning!r}"
)
await self.send_message(
recipient="HUB_AGENT",
message_type=MessageType.AGENT_TASK,
payload={
"task_type": "prepare_receiving",
"booking_id": booking_id,
"miler_id": miler_id,
"hub_id": hub_id,
},
correlation_id=str(booking_id),
)
# No longer forwarded to HUB_AGENT as `prepare_receiving`. HUB_AGENT
# is a simulation over eight fictional hubs keyed "XX-HUB-01"; handed a
# real backend hub id and booking id it answered "Hub not found" for
# every assignment, so the hand-off did nothing but log noise. Restore
# it only once HUB_AGENT reads real hubs (agent registry status:
# simulation).
except Exception as e:
logger.error(f"DispatchAgent booking.assigned handler error: {e}")
@@ -305,17 +301,25 @@ class DispatchAgent(SpecializedAgent):
f"reason={data.get('reason')!r}"
)
if not registry.skill_enabled("assignment_failure_triage"):
logger.info(
f"[DISPATCH] booking={booking_id}: skill assignment_failure_triage is disabled "
"in the agent registry; not triaging"
)
return
facts, count = await self._gather_assignment_facts(zone_id, lat, lon)
realert_every = int(registry.threshold("assignment_failure_triage", "realertEvery", DISPATCH_REALERT_EVERY))
logger.info(f"[DISPATCH] booking={booking_id} context facts: {facts}")
# Rate limit: once this zone has been alerted today, don't re-decide
# (and don't pay for an LLM call) on every further failure.
if facts.get("alert_already_sent_today"):
if DISPATCH_REALERT_EVERY and count % DISPATCH_REALERT_EVERY == 0:
if realert_every and count % realert_every == 0:
await self._ops_alert(
zone_id, booking_id,
f"Zone {zone_id}: {count} failed assignments today and still climbing "
f"(re-alert every {DISPATCH_REALERT_EVERY}; earlier alert already raised).",
f"(re-alert every {realert_every}; earlier alert already raised).",
severity="high",
)
else:
@@ -325,7 +329,11 @@ class DispatchAgent(SpecializedAgent):
)
return
decision = await decide_assignment_failure(build_assignment_failure_context(facts))
model = registry.model("DISPATCH_AGENT")
decision = await decide_assignment_failure(build_assignment_failure_context(facts), model)
if decision is not None:
record_decision("assignment_failure", booking_id, facts, decision.action, decision.confidence,
decision.reasoning, model or LLM_MODEL)
if decision is None:
logger.warning(f"No LLM decision for booking {booking_id}; falling back to count>=3 heuristic")
@@ -347,7 +355,7 @@ class DispatchAgent(SpecializedAgent):
pass # transient — the backend will retry
elif decision.action == "notify_customer":
if DISPATCH_AGENT_AUTONOMOUS:
if registry.autonomous("DISPATCH_AGENT", DISPATCH_AGENT_AUTONOMOUS):
await self._notify_customer_delay(booking_id)
else:
# Outward-facing action not authorised — raise an internal proposal instead.

View File

@@ -26,7 +26,9 @@ from core.agent import SpecializedAgent
from core.types import AgentTask, MessageType
from core.logger import logger
from core.http_client import api_post
from core.llm import decide_stall_response, build_stall_context, StallDecision
from core.llm import decide_stall_response, build_stall_context, StallDecision, LLM_MODEL
from core.registry import registry
from core.decisions import record_decision
from config.system_config import (
GO_API_BASE_URL, INTERNAL_API_KEY,
DB_HOST, DB_PORT, DB_NAME, DB_USER, DB_PASSWORD,
@@ -34,7 +36,12 @@ from config.system_config import (
REDIS_HOST, REDIS_PORT, REDIS_PASSWORD,
)
STALL_MINUTES = 10
STALL_MINUTES = 10 # default; the agent registry's stall_response.stallMinutes overrides it
def stall_minutes() -> float:
"""Minutes without movement before a rider counts as stalled (registry, else 10)."""
return registry.threshold("stall_response", "stallMinutes", STALL_MINUTES)
ACTIVE_STATUSES = ["Miler_Assigned", "Pickup_Scheduled"]
TRACKING_STREAM = "TRACKING"
@@ -260,7 +267,7 @@ class ExceptionAgent(SpecializedAgent):
unchanged_since = _parse_ts(prev.get("position_unchanged_since")) or _utcnow()
minutes_stalled = (_utcnow() - unchanged_since).total_seconds() / 60
if minutes_stalled >= STALL_MINUTES:
if minutes_stalled >= stall_minutes():
booking = await self._get_active_booking(miler_id)
if booking:
await self._publish_stall(miler_id, booking["booking_id"], minutes_stalled)
@@ -311,7 +318,7 @@ class ExceptionAgent(SpecializedAgent):
continue
minutes_stale = (now - updated_at).total_seconds() / 60
if minutes_stale >= STALL_MINUTES:
if minutes_stale >= stall_minutes():
logger.warning(f"StallDetector: miler {miler_id} stale {minutes_stale:.1f} min (booking {booking_id})")
await self._publish_stall(miler_id, booking_id, minutes_stale)
@@ -385,15 +392,24 @@ class ExceptionAgent(SpecializedAgent):
Claude chooses one of wait / notify_only / reassign / escalate. The only
irreversible, customer-visible action (reassign) is gated behind an
autonomy flag and a confidence threshold; otherwise it is escalated to a
human via JARVIS. If the LLM is unavailable, we fall back to the previous
deterministic behaviour (reassign + notify) so detection never silently
stops acting.
human via JARVIS. If the LLM is unavailable: an autonomous agent falls
back to reassign + notify so detection never silently stops acting; a
non-autonomous one escalates to a human instead (it used to reassign
regardless, which made "autonomy off" untrue during an LLM outage).
Settings come from the agent registry (core/registry.py), falling back
to env: skill `stall_response` on/off, its `stallMinutes` and
`reassignConfidence`, and this agent's autonomy and model.
"""
try:
data = json.loads(msg.data.decode())
except Exception:
return
if not registry.skill_enabled("stall_response"):
logger.info("Stall received but skill stall_response is disabled in the agent registry; not acting")
return
miler_id = data.get("miler_id", "")
booking_id = data.get("booking_id", "")
minutes_stalled = data.get("minutes_stalled", 0)
@@ -409,21 +425,37 @@ class ExceptionAgent(SpecializedAgent):
logger.info(f"[EXCEPTION] booking={booking_id} context facts: {facts}")
context = build_stall_context(
minutes_stalled, stall_threshold_min=STALL_MINUTES, now=_utcnow(), facts=facts,
minutes_stalled, stall_threshold_min=stall_minutes(), now=_utcnow(), facts=facts,
)
decision = await decide_stall_response(context)
model = registry.model("EXCEPTION_AGENT")
decision = await decide_stall_response(context, model)
autonomous = registry.autonomous("EXCEPTION_AGENT", AUTONOMOUS_REASSIGN)
if decision is None:
logger.warning(f"No LLM decision for booking {booking_id}; falling back to reassign + notify")
await self._reassign(booking_id, "miler_stalled")
await self._notify_customer(booking_id, "We detected a delay, finding you a new miler")
self._record_stall_exception(
miler_id, booking_id, minutes_stalled,
resolution="Fallback (LLM unavailable): reassignment triggered + customer notified",
actions=["reassign", "notify_customer"],
)
if autonomous:
logger.warning(f"No LLM decision for booking {booking_id}; autonomous — falling back to reassign + notify")
await self._reassign(booking_id, "miler_stalled")
await self._notify_customer(booking_id, "We detected a delay, finding you a new miler")
self._record_stall_exception(
miler_id, booking_id, minutes_stalled,
resolution="Fallback (LLM unavailable): reassignment triggered + customer notified",
actions=["reassign", "notify_customer"],
)
else:
logger.warning(f"No LLM decision for booking {booking_id}; not autonomous — escalating to a human")
await self._escalate_to_human(miler_id, booking_id, minutes_stalled, StallDecision(
action="escalate", reasoning="LLM unavailable; autonomy is off, so a human decides.", confidence=0.0,
))
self._record_stall_exception(
miler_id, booking_id, minutes_stalled,
resolution="Fallback (LLM unavailable, autonomy off): escalated to a human",
actions=["escalate"],
)
return
record_decision("stall_response", booking_id, facts, decision.action, decision.confidence,
decision.reasoning, model or LLM_MODEL)
logger.info(
f"[EXCEPTION] booking={booking_id} decision={decision.action} "
f"confidence={decision.confidence:.2f} reasoning={decision.reasoning!r}"
@@ -441,7 +473,8 @@ class ExceptionAgent(SpecializedAgent):
actions.append("notify_customer")
elif decision.action == "reassign":
if AUTONOMOUS_REASSIGN and decision.confidence >= REASSIGN_MIN_CONFIDENCE:
min_confidence = registry.threshold("stall_response", "reassignConfidence", REASSIGN_MIN_CONFIDENCE)
if autonomous and decision.confidence >= min_confidence:
await self._reassign(booking_id, "miler_stalled")
await self._notify_customer(booking_id, "We detected a delay, finding you a new miler")
actions += ["reassign", "notify_customer"]

View File

@@ -36,6 +36,7 @@ from core.agent import SpecializedAgent
from core.types import AgentTask
from core.logger import logger
from core.http_client import api_get, api_post
from core.registry import registry
from config.system_config import (
NATS_HOST, NATS_PORT, NATS_USER, NATS_PASSWORD,
GO_API_BASE_URL, INTERNAL_API_KEY, ROUTE_OPTIMIZER_URL,
@@ -162,7 +163,9 @@ class ExpressDispatchAgent(SpecializedAgent):
logger.info(
f"[EXPRESS] dispatch received — tenant={tenant_id} bookings={len(booking_ids)}"
)
if tenant_id and booking_ids:
if not registry.skill_enabled("express_batch_dispatch"):
logger.info("[EXPRESS] skill express_batch_dispatch is disabled in the agent registry; batch left for the console")
elif tenant_id and booking_ids:
await self._handle_batch(tenant_id, booking_ids)
except Exception as e:
logger.error(f"ExpressDispatchAgent batch handler error: {e}")
@@ -212,10 +215,10 @@ class ExpressDispatchAgent(SpecializedAgent):
return
# 3. Write back (or, in observation mode, only log the plan).
if not EXPRESS_AGENT_AUTONOMOUS:
if not registry.autonomous("EXPRESS_DISPATCH_AGENT", EXPRESS_AGENT_AUTONOMOUS):
logger.info(
f"[EXPRESS] tenant={tenant_id} OBSERVE-ONLY plan "
f"(EXPRESS_AGENT_AUTONOMOUS=false): {json.dumps(assignments)}"
f"(autonomy off): {json.dumps(assignments)}"
)
return
@@ -247,6 +250,10 @@ class ExpressDispatchAgent(SpecializedAgent):
part (ordering); this only decides who."""
load: Dict[int, int] = {r["miler_user_id"]: 0 for r in riders}
rider_stops: Dict[int, List[Dict]] = {}
# Tuned in the agent registry (skill express_batch_dispatch), else env.
max_per_rider = int(registry.threshold("express_batch_dispatch", "maxPerRider", MAX_PER_RIDER))
max_radius_km = registry.threshold("express_batch_dispatch", "maxRadiusKm", MAX_RADIUS_KM)
load_penalty_km = registry.threshold("express_batch_dispatch", "loadPenaltyKm", LOAD_PENALTY_KM)
# Assign larger-pickup-cluster bookings first is unnecessary; simple
# stable order keeps it predictable and testable.
@@ -255,13 +262,13 @@ class ExpressDispatchAgent(SpecializedAgent):
best_rider = None
best_score = None
for r in riders:
if load[r["miler_user_id"]] >= MAX_PER_RIDER:
if load[r["miler_user_id"]] >= max_per_rider:
continue
rlat, rlon = _rider_location(r, plat, plon)
dist = _haversine_km(rlat, rlon, plat, plon)
if dist > MAX_RADIUS_KM:
if dist > max_radius_km:
continue
score = dist + LOAD_PENALTY_KM * load[r["miler_user_id"]]
score = dist + load_penalty_km * load[r["miler_user_id"]]
if best_score is None or score < best_score:
best_score = score
best_rider = r

View File

@@ -89,6 +89,10 @@ class FleetAgent(SpecializedAgent):
handlers = {
"assign_vehicle": self._assign_vehicle,
"release_vehicle": self._release_vehicle,
# Sent by ExceptionAgent._cancel_order, which knows the order, not
# the vehicle. Had no handler, so every cancellation hit
# _unknown_task and the vehicle stayed marked in use.
"release_vehicle_for_cancel": self._release_vehicle_for_order,
"get_availability": self._get_availability,
"track_vehicle": self._track_vehicle,
"update_location": self._update_location,
@@ -181,6 +185,20 @@ class FleetAgent(SpecializedAgent):
return {"status": "released", "vehicle_id": vehicle_id, "hub": vehicle["hub"]}
async def _release_vehicle_for_order(self, task: AgentTask) -> Dict[str, Any]:
"""Release the vehicle carrying a cancelled order, found by order id."""
order_id = task.data.get("order_id")
for assignment in self._assignments.values():
if assignment.status == "active" and order_id in (assignment.order_ids or []):
return await self._release_vehicle(AgentTask(
task_id=f"{task.task_id}:release",
agent_type=task.agent_type,
task_type="release_vehicle",
data={"vehicle_id": assignment.vehicle_id},
))
return {"status": "no_assignment", "order_id": order_id,
"message": f"No active vehicle assignment carries order {order_id}"}
async def _get_availability(self, task: AgentTask) -> Dict[str, Any]:
hub_id = task.data.get("hub_id")
vehicle_type = task.data.get("vehicle_type")

View File

@@ -8,16 +8,18 @@ from core.types import (
AgentTask, MessageType, Priority, OrderStatus, ZoneType
)
from core.logger import logger
from core.http_client import api_get, api_post, api_patch
from config.system_config import GO_API_BASE_URL
class OrderAgent(SpecializedAgent):
"""
Order Agent - Manages the entire order lifecycle from intake to validation.
Writes go to the Doormile Go backend (POST /api/v1/admin/crmbooking).
Reads and status updates also go through the Go API.
NOT CONNECTED to the Doormile backend (agent registry status: broken).
It was written against /api/v1/admin/crmbooking, which was renamed to
/admin/expressbooking, and every /admin/* route needs a console login's
JWT — this engine holds only the internal key, which /admin/* refuses. So
the helpers below refuse up front instead of sending requests that can only
fail. Wiring it needs an /internal/* booking route on the backend first.
"""
def __init__(self):
@@ -70,23 +72,22 @@ class OrderAgent(SpecializedAgent):
# Go API helpers (use shared session + retry from http_client) #
# ------------------------------------------------------------------ #
# Returns None (what every caller already treats as "no backend result").
NO_BACKEND_ROUTE = ("OrderAgent has no working backend route: /admin/* needs a console JWT "
"and /admin/crmbooking no longer exists")
def _refuse(self, method: str, path: str) -> None:
logger.warning(f"OrderAgent: not calling {method} {path} — {self.NO_BACKEND_ROUTE}")
return None
async def _api_post(self, path: str, payload: Dict) -> Optional[Dict]:
result = await api_post(f"{GO_API_BASE_URL}{path}", json=payload)
if result is None:
logger.error(f"Go API POST {path} returned no response")
return result
return self._refuse("POST", path)
async def _api_get(self, path: str, params: Dict = None) -> Optional[Dict]:
result = await api_get(f"{GO_API_BASE_URL}{path}", params=params)
if result is None:
logger.error(f"Go API GET {path} returned no response")
return result
return self._refuse("GET", path)
async def _api_patch(self, path: str, payload: Dict) -> Optional[Dict]:
result = await api_patch(f"{GO_API_BASE_URL}{path}", json=payload)
if result is None:
logger.error(f"Go API PATCH {path} returned no response")
return result
return self._refuse("PATCH", path)
# ------------------------------------------------------------------ #
# Task handlers #