implemenation on the ai agents and registry

This commit is contained in:
2026-09-30 14:56:08 +05:30
parent aa3bb35733
commit 5a7d32cc04
15 changed files with 803 additions and 72 deletions

View File

@@ -280,7 +280,9 @@ class MasterAgent(Agent):
task_id=f"{order_id}_validate",
agent_type="order",
task_type="validate_order",
data={"order": order_data}
# OrderAgent._validate_order reads `order_id`; sending only
# {"order": ...} made every orchestrated order "not found".
data={"order_id": order_id, "order": order_data}
))
customer_agent = self._sub_agents.get("CUSTOMER_AGENT")

66
core/decisions.py Normal file
View File

@@ -0,0 +1,66 @@
"""
Records the engine's model decisions in the backend's decision log.
Until Phase 5 of the plan (krow_talent_app/docs/agent-platform-plan.md) the
stall and assignment-failure decisions existed only as log lines, so the
console's Insights tab could not show them. Each decision is now sent to
``POST /api/v1/internal/agent-decisions`` — the same log routemate's
assignment decisions go to.
Fire-and-forget: the post runs as a background task, so a slow or down
backend never delays an agent's reaction to a stall. A failure is logged by
the HTTP client and otherwise ignored — the decision was still made and acted
on; only its record is missing.
"""
import asyncio
from typing import Any, Dict, Optional, Set
from config.system_config import GO_API_BASE_URL, INTERNAL_API_KEY
from core.http_client import api_post
_pending: Set[asyncio.Task] = set() # keep references so tasks are not garbage-collected
def _as_booking_id(value: Any) -> Optional[int]:
"""The backend's booking_id is an unsigned integer; anything else is omitted."""
try:
n = int(value)
except (TypeError, ValueError):
return None
return n if n > 0 else None
def build_payload(decision_type: str, booking_id: Any, facts: Optional[Dict[str, Any]],
action: str, confidence: float, reasoning: str, model: str) -> Dict[str, Any]:
"""The request body. Pure, so its shape is tested without a network."""
return {
"decision_type": decision_type,
"booking_id": _as_booking_id(booking_id),
"context": {"facts": facts or {}, "model": model},
"decision": {"action": action, "confidence": round(float(confidence), 3)},
"reasoning": reasoning or "",
}
async def _post(payload: Dict[str, Any]) -> None:
await api_post(
f"{GO_API_BASE_URL}/api/v1/internal/agent-decisions",
json=payload,
headers={"X-Internal-Key": INTERNAL_API_KEY},
)
def record_decision(decision_type: str, booking_id: Any, facts: Optional[Dict[str, Any]],
action: str, confidence: float, reasoning: str, model: str) -> Optional[asyncio.Task]:
"""Schedule the record and return at once. Returns the task (for tests), or
None when there is no running loop or no key to authenticate with."""
if not INTERNAL_API_KEY:
return None
try:
loop = asyncio.get_running_loop()
except RuntimeError:
return None
task = loop.create_task(_post(build_payload(decision_type, booking_id, facts, action, confidence, reasoning, model)))
_pending.add(task)
task.add_done_callback(_pending.discard)
return task

View File

@@ -102,7 +102,25 @@ _STALL_SCHEMA = {
}
async def _decide(system: str, schema: dict, valid_actions, context: str):
def request_params(model: Optional[str] = None) -> dict:
"""The model-dependent part of a decision request.
`model` is a per-agent override from the agent registry; None means the
engine default (LLM_MODEL). Claude Haiku 4.5 rejects adaptive thinking and
`effort`, so both are left out for it — otherwise a Haiku pin set from the
console would make every decision fail and fall back to the heuristic.
"""
m = model or LLM_MODEL
output_config = {}
params = {"model": m}
if "haiku" not in m:
params["thinking"] = {"type": "adaptive"}
output_config["effort"] = LLM_EFFORT
params["output_config"] = output_config
return params
async def _decide(system: str, schema: dict, valid_actions, context: str, model: Optional[str] = None):
"""Run one structured-output decision against Claude.
Returns (action, reasoning, confidence), or None on any failure (network,
@@ -111,20 +129,17 @@ async def _decide(system: str, schema: dict, valid_actions, context: str):
"""
resp = None
last_err = None
params = request_params(model)
params["output_config"]["format"] = {"type": "json_schema", "schema": schema}
for attempt in range(2): # initial try + one retry on transient failure
try:
client = _get_client()
resp = await client.messages.create(
model=LLM_MODEL,
max_tokens=LLM_MAX_TOKENS,
system=system,
thinking={"type": "adaptive"},
output_config={
"effort": LLM_EFFORT,
"format": {"type": "json_schema", "schema": schema},
},
messages=[{"role": "user", "content": context}],
timeout=LLM_TIMEOUT_S,
**params,
)
break
except Exception as e:
@@ -154,10 +169,10 @@ async def _decide(system: str, schema: dict, valid_actions, context: str):
return None
async def decide_stall_response(context: str) -> Optional[StallDecision]:
async def decide_stall_response(context: str, model: Optional[str] = None) -> Optional[StallDecision]:
"""Ask Claude how to handle a stalled miler. None on failure (caller falls
back to deterministic behaviour)."""
r = await _decide(_STALL_SYSTEM, _STALL_SCHEMA, _VALID_ACTIONS, context)
back to deterministic behaviour). `model` overrides LLM_MODEL."""
r = await _decide(_STALL_SYSTEM, _STALL_SCHEMA, _VALID_ACTIONS, context, model)
return StallDecision(*r) if r else None
@@ -224,8 +239,8 @@ def build_assignment_failure_context(facts, *, now=None) -> str:
return "\n".join(lines)
async def decide_assignment_failure(context: str) -> Optional[AssignmentDecision]:
async def decide_assignment_failure(context: str, model: Optional[str] = None) -> Optional[AssignmentDecision]:
"""Ask Claude how to react to a failed assignment. None on failure (caller
falls back to deterministic behaviour)."""
r = await _decide(_ASSIGNMENT_SYSTEM, _ASSIGNMENT_SCHEMA, _ASSIGNMENT_ACTIONS, context)
falls back to deterministic behaviour). `model` overrides LLM_MODEL."""
r = await _decide(_ASSIGNMENT_SYSTEM, _ASSIGNMENT_SCHEMA, _ASSIGNMENT_ACTIONS, context, model)
return AssignmentDecision(*r) if r else None

122
core/registry.py Normal file
View File

@@ -0,0 +1,122 @@
"""
The agent registry, as this engine sees it.
Settings an operator changes in the console (Settings → Skills & Tools) are
stored in the Doormile backend's agent registry. This module polls
``GET /api/v1/internal/ai/registry`` and answers, at the moment an agent needs
it, three questions:
registry.autonomous(agent_id, env_default) may this agent act on its own?
registry.model(agent_id) which Claude model, if pinned?
registry.skill_enabled(skill_id) is this behaviour switched on?
registry.threshold(skill_id, key, default) the tuned value of a knob
Precedence — deliberate, and the same for every setting:
registry loaded → the registry's value (it is the operator's decision)
not yet / never → the env default this engine always had
So a backend that is down, unreachable, or not yet deployed leaves the engine
exactly as it behaved before the registry existed. Once the registry has been
read, the last good copy is kept through later failures: a flapping backend
does not flip autonomy back to an env default mid-shift.
Polling uses the ETag the backend sends; an unchanged registry costs a 304
with no body. Before Phase 5 of the plan (krow_talent_app/docs/
agent-platform-plan.md) every one of these settings was a module constant read
from the environment at import time, so a change needed a redeploy.
"""
import asyncio
import os
from typing import Any, Dict, Optional
import aiohttp
from core.logger import logger
POLL_SECONDS = float(os.getenv("REGISTRY_POLL_SECONDS", "30"))
class RegistryClient:
def __init__(self):
self._agents: Dict[str, Dict[str, Any]] = {}
self._skills: Dict[str, Dict[str, Any]] = {}
self._etag: Optional[str] = None
self.loaded = False
# ── Reading (sync, cheap, safe to call on every event) ──────────────────
def autonomous(self, agent_id: str, env_default: bool) -> bool:
if not self.loaded or agent_id not in self._agents:
return env_default
return bool(self._agents[agent_id].get("autonomous", False))
def model(self, agent_id: str) -> Optional[str]:
"""A pinned model id, or None for the engine's own LLM_MODEL."""
if not self.loaded:
return None
return self._agents.get(agent_id, {}).get("model") or None
def skill_enabled(self, skill_id: str, default: bool = True) -> bool:
if not self.loaded or skill_id not in self._skills:
return default
return bool(self._skills[skill_id].get("enabled", default))
def threshold(self, skill_id: str, key: str, default: float) -> float:
if not self.loaded:
return default
value = (self._skills.get(skill_id, {}).get("thresholds") or {}).get(key)
return value if isinstance(value, (int, float)) and not isinstance(value, bool) else default
# ── Loading ──────────────────────────────────────────────────────────────
def apply(self, snapshot: Dict[str, Any]) -> None:
"""Adopt a registry snapshot (the `data` of /internal/ai/registry)."""
agents = {a["agentid"]: a for a in snapshot.get("agents") or [] if a.get("agentid")}
skills = {s["skillid"]: s for s in snapshot.get("skills") or [] if s.get("skillid")}
self._agents, self._skills = agents, skills
if not self.loaded:
logger.info(f"Agent registry loaded: {len(agents)} agents, {len(skills)} skills")
self.loaded = True
async def fetch_once(self, session: aiohttp.ClientSession, base_url: str, api_key: str) -> str:
"""One poll. Returns 'updated', 'unchanged', 'unavailable' or 'refused'."""
headers = {"X-Internal-Key": api_key}
if self._etag:
headers["If-None-Match"] = self._etag
try:
async with session.get(f"{base_url}/api/v1/internal/ai/registry", headers=headers,
timeout=aiohttp.ClientTimeout(total=10)) as resp:
if resp.status == 304:
return "unchanged"
if resp.status in (401, 403):
return "refused"
if resp.status != 200:
return "unavailable"
body = await resp.json()
self.apply(body.get("data") or {})
self._etag = resp.headers.get("ETag")
return "updated"
except Exception as exc: # network, timeout, bad JSON — keep the last good copy
logger.debug(f"Agent registry poll failed: {exc}")
return "unavailable"
async def run(self, base_url: str, api_key: str, poll_seconds: float = POLL_SECONDS) -> None:
"""Poll forever. Never raises; the engine runs on env defaults meanwhile."""
if not api_key:
logger.warning("INTERNAL_API_KEY unset — agent registry not read; running on env defaults")
return
last = None
async with aiohttp.ClientSession() as session:
while True:
outcome = await self.fetch_once(session, base_url, api_key)
if outcome != last and outcome in ("refused", "unavailable"):
logger.warning(
f"Agent registry {outcome} at {base_url}; "
+ ("keeping the last loaded copy" if self.loaded else "running on env defaults")
)
last = outcome
await asyncio.sleep(poll_seconds)
registry = RegistryClient()

View File

@@ -16,6 +16,9 @@ class MessageType(str, Enum):
ORDER_DELIVERED = "ORDER_DELIVERED"
ORDER_CANCELLED = "ORDER_CANCELLED"
ORDER_DELAYED = "ORDER_DELAYED"
# OrderAgent._update_status publishes this; it was missing from the enum, so
# every update_status task raised AttributeError.
ORDER_STATUS_UPDATE = "ORDER_STATUS_UPDATE"
HUB_STATUS_UPDATE = "HUB_STATUS_UPDATE"
VEHICLE_ASSIGNED = "VEHICLE_ASSIGNED"
ROUTE_OPTIMIZED = "ROUTE_OPTIMIZED"