implemenation on the ai agents and registry
This commit is contained in:
@@ -280,7 +280,9 @@ class MasterAgent(Agent):
|
||||
task_id=f"{order_id}_validate",
|
||||
agent_type="order",
|
||||
task_type="validate_order",
|
||||
data={"order": order_data}
|
||||
# OrderAgent._validate_order reads `order_id`; sending only
|
||||
# {"order": ...} made every orchestrated order "not found".
|
||||
data={"order_id": order_id, "order": order_data}
|
||||
))
|
||||
|
||||
customer_agent = self._sub_agents.get("CUSTOMER_AGENT")
|
||||
|
||||
66
core/decisions.py
Normal file
66
core/decisions.py
Normal file
@@ -0,0 +1,66 @@
|
||||
"""
|
||||
Records the engine's model decisions in the backend's decision log.
|
||||
|
||||
Until Phase 5 of the plan (krow_talent_app/docs/agent-platform-plan.md) the
|
||||
stall and assignment-failure decisions existed only as log lines, so the
|
||||
console's Insights tab could not show them. Each decision is now sent to
|
||||
``POST /api/v1/internal/agent-decisions`` — the same log routemate's
|
||||
assignment decisions go to.
|
||||
|
||||
Fire-and-forget: the post runs as a background task, so a slow or down
|
||||
backend never delays an agent's reaction to a stall. A failure is logged by
|
||||
the HTTP client and otherwise ignored — the decision was still made and acted
|
||||
on; only its record is missing.
|
||||
"""
|
||||
import asyncio
|
||||
from typing import Any, Dict, Optional, Set
|
||||
|
||||
from config.system_config import GO_API_BASE_URL, INTERNAL_API_KEY
|
||||
from core.http_client import api_post
|
||||
|
||||
_pending: Set[asyncio.Task] = set() # keep references so tasks are not garbage-collected
|
||||
|
||||
|
||||
def _as_booking_id(value: Any) -> Optional[int]:
|
||||
"""The backend's booking_id is an unsigned integer; anything else is omitted."""
|
||||
try:
|
||||
n = int(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
return n if n > 0 else None
|
||||
|
||||
|
||||
def build_payload(decision_type: str, booking_id: Any, facts: Optional[Dict[str, Any]],
|
||||
action: str, confidence: float, reasoning: str, model: str) -> Dict[str, Any]:
|
||||
"""The request body. Pure, so its shape is tested without a network."""
|
||||
return {
|
||||
"decision_type": decision_type,
|
||||
"booking_id": _as_booking_id(booking_id),
|
||||
"context": {"facts": facts or {}, "model": model},
|
||||
"decision": {"action": action, "confidence": round(float(confidence), 3)},
|
||||
"reasoning": reasoning or "",
|
||||
}
|
||||
|
||||
|
||||
async def _post(payload: Dict[str, Any]) -> None:
|
||||
await api_post(
|
||||
f"{GO_API_BASE_URL}/api/v1/internal/agent-decisions",
|
||||
json=payload,
|
||||
headers={"X-Internal-Key": INTERNAL_API_KEY},
|
||||
)
|
||||
|
||||
|
||||
def record_decision(decision_type: str, booking_id: Any, facts: Optional[Dict[str, Any]],
|
||||
action: str, confidence: float, reasoning: str, model: str) -> Optional[asyncio.Task]:
|
||||
"""Schedule the record and return at once. Returns the task (for tests), or
|
||||
None when there is no running loop or no key to authenticate with."""
|
||||
if not INTERNAL_API_KEY:
|
||||
return None
|
||||
try:
|
||||
loop = asyncio.get_running_loop()
|
||||
except RuntimeError:
|
||||
return None
|
||||
task = loop.create_task(_post(build_payload(decision_type, booking_id, facts, action, confidence, reasoning, model)))
|
||||
_pending.add(task)
|
||||
task.add_done_callback(_pending.discard)
|
||||
return task
|
||||
41
core/llm.py
41
core/llm.py
@@ -102,7 +102,25 @@ _STALL_SCHEMA = {
|
||||
}
|
||||
|
||||
|
||||
async def _decide(system: str, schema: dict, valid_actions, context: str):
|
||||
def request_params(model: Optional[str] = None) -> dict:
|
||||
"""The model-dependent part of a decision request.
|
||||
|
||||
`model` is a per-agent override from the agent registry; None means the
|
||||
engine default (LLM_MODEL). Claude Haiku 4.5 rejects adaptive thinking and
|
||||
`effort`, so both are left out for it — otherwise a Haiku pin set from the
|
||||
console would make every decision fail and fall back to the heuristic.
|
||||
"""
|
||||
m = model or LLM_MODEL
|
||||
output_config = {}
|
||||
params = {"model": m}
|
||||
if "haiku" not in m:
|
||||
params["thinking"] = {"type": "adaptive"}
|
||||
output_config["effort"] = LLM_EFFORT
|
||||
params["output_config"] = output_config
|
||||
return params
|
||||
|
||||
|
||||
async def _decide(system: str, schema: dict, valid_actions, context: str, model: Optional[str] = None):
|
||||
"""Run one structured-output decision against Claude.
|
||||
|
||||
Returns (action, reasoning, confidence), or None on any failure (network,
|
||||
@@ -111,20 +129,17 @@ async def _decide(system: str, schema: dict, valid_actions, context: str):
|
||||
"""
|
||||
resp = None
|
||||
last_err = None
|
||||
params = request_params(model)
|
||||
params["output_config"]["format"] = {"type": "json_schema", "schema": schema}
|
||||
for attempt in range(2): # initial try + one retry on transient failure
|
||||
try:
|
||||
client = _get_client()
|
||||
resp = await client.messages.create(
|
||||
model=LLM_MODEL,
|
||||
max_tokens=LLM_MAX_TOKENS,
|
||||
system=system,
|
||||
thinking={"type": "adaptive"},
|
||||
output_config={
|
||||
"effort": LLM_EFFORT,
|
||||
"format": {"type": "json_schema", "schema": schema},
|
||||
},
|
||||
messages=[{"role": "user", "content": context}],
|
||||
timeout=LLM_TIMEOUT_S,
|
||||
**params,
|
||||
)
|
||||
break
|
||||
except Exception as e:
|
||||
@@ -154,10 +169,10 @@ async def _decide(system: str, schema: dict, valid_actions, context: str):
|
||||
return None
|
||||
|
||||
|
||||
async def decide_stall_response(context: str) -> Optional[StallDecision]:
|
||||
async def decide_stall_response(context: str, model: Optional[str] = None) -> Optional[StallDecision]:
|
||||
"""Ask Claude how to handle a stalled miler. None on failure (caller falls
|
||||
back to deterministic behaviour)."""
|
||||
r = await _decide(_STALL_SYSTEM, _STALL_SCHEMA, _VALID_ACTIONS, context)
|
||||
back to deterministic behaviour). `model` overrides LLM_MODEL."""
|
||||
r = await _decide(_STALL_SYSTEM, _STALL_SCHEMA, _VALID_ACTIONS, context, model)
|
||||
return StallDecision(*r) if r else None
|
||||
|
||||
|
||||
@@ -224,8 +239,8 @@ def build_assignment_failure_context(facts, *, now=None) -> str:
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
async def decide_assignment_failure(context: str) -> Optional[AssignmentDecision]:
|
||||
async def decide_assignment_failure(context: str, model: Optional[str] = None) -> Optional[AssignmentDecision]:
|
||||
"""Ask Claude how to react to a failed assignment. None on failure (caller
|
||||
falls back to deterministic behaviour)."""
|
||||
r = await _decide(_ASSIGNMENT_SYSTEM, _ASSIGNMENT_SCHEMA, _ASSIGNMENT_ACTIONS, context)
|
||||
falls back to deterministic behaviour). `model` overrides LLM_MODEL."""
|
||||
r = await _decide(_ASSIGNMENT_SYSTEM, _ASSIGNMENT_SCHEMA, _ASSIGNMENT_ACTIONS, context, model)
|
||||
return AssignmentDecision(*r) if r else None
|
||||
|
||||
122
core/registry.py
Normal file
122
core/registry.py
Normal file
@@ -0,0 +1,122 @@
|
||||
"""
|
||||
The agent registry, as this engine sees it.
|
||||
|
||||
Settings an operator changes in the console (Settings → Skills & Tools) are
|
||||
stored in the Doormile backend's agent registry. This module polls
|
||||
``GET /api/v1/internal/ai/registry`` and answers, at the moment an agent needs
|
||||
it, three questions:
|
||||
|
||||
registry.autonomous(agent_id, env_default) may this agent act on its own?
|
||||
registry.model(agent_id) which Claude model, if pinned?
|
||||
registry.skill_enabled(skill_id) is this behaviour switched on?
|
||||
registry.threshold(skill_id, key, default) the tuned value of a knob
|
||||
|
||||
Precedence — deliberate, and the same for every setting:
|
||||
|
||||
registry loaded → the registry's value (it is the operator's decision)
|
||||
not yet / never → the env default this engine always had
|
||||
|
||||
So a backend that is down, unreachable, or not yet deployed leaves the engine
|
||||
exactly as it behaved before the registry existed. Once the registry has been
|
||||
read, the last good copy is kept through later failures: a flapping backend
|
||||
does not flip autonomy back to an env default mid-shift.
|
||||
|
||||
Polling uses the ETag the backend sends; an unchanged registry costs a 304
|
||||
with no body. Before Phase 5 of the plan (krow_talent_app/docs/
|
||||
agent-platform-plan.md) every one of these settings was a module constant read
|
||||
from the environment at import time, so a change needed a redeploy.
|
||||
"""
|
||||
import asyncio
|
||||
import os
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
import aiohttp
|
||||
|
||||
from core.logger import logger
|
||||
|
||||
POLL_SECONDS = float(os.getenv("REGISTRY_POLL_SECONDS", "30"))
|
||||
|
||||
|
||||
class RegistryClient:
|
||||
def __init__(self):
|
||||
self._agents: Dict[str, Dict[str, Any]] = {}
|
||||
self._skills: Dict[str, Dict[str, Any]] = {}
|
||||
self._etag: Optional[str] = None
|
||||
self.loaded = False
|
||||
|
||||
# ── Reading (sync, cheap, safe to call on every event) ──────────────────
|
||||
|
||||
def autonomous(self, agent_id: str, env_default: bool) -> bool:
|
||||
if not self.loaded or agent_id not in self._agents:
|
||||
return env_default
|
||||
return bool(self._agents[agent_id].get("autonomous", False))
|
||||
|
||||
def model(self, agent_id: str) -> Optional[str]:
|
||||
"""A pinned model id, or None for the engine's own LLM_MODEL."""
|
||||
if not self.loaded:
|
||||
return None
|
||||
return self._agents.get(agent_id, {}).get("model") or None
|
||||
|
||||
def skill_enabled(self, skill_id: str, default: bool = True) -> bool:
|
||||
if not self.loaded or skill_id not in self._skills:
|
||||
return default
|
||||
return bool(self._skills[skill_id].get("enabled", default))
|
||||
|
||||
def threshold(self, skill_id: str, key: str, default: float) -> float:
|
||||
if not self.loaded:
|
||||
return default
|
||||
value = (self._skills.get(skill_id, {}).get("thresholds") or {}).get(key)
|
||||
return value if isinstance(value, (int, float)) and not isinstance(value, bool) else default
|
||||
|
||||
# ── Loading ──────────────────────────────────────────────────────────────
|
||||
|
||||
def apply(self, snapshot: Dict[str, Any]) -> None:
|
||||
"""Adopt a registry snapshot (the `data` of /internal/ai/registry)."""
|
||||
agents = {a["agentid"]: a for a in snapshot.get("agents") or [] if a.get("agentid")}
|
||||
skills = {s["skillid"]: s for s in snapshot.get("skills") or [] if s.get("skillid")}
|
||||
self._agents, self._skills = agents, skills
|
||||
if not self.loaded:
|
||||
logger.info(f"Agent registry loaded: {len(agents)} agents, {len(skills)} skills")
|
||||
self.loaded = True
|
||||
|
||||
async def fetch_once(self, session: aiohttp.ClientSession, base_url: str, api_key: str) -> str:
|
||||
"""One poll. Returns 'updated', 'unchanged', 'unavailable' or 'refused'."""
|
||||
headers = {"X-Internal-Key": api_key}
|
||||
if self._etag:
|
||||
headers["If-None-Match"] = self._etag
|
||||
try:
|
||||
async with session.get(f"{base_url}/api/v1/internal/ai/registry", headers=headers,
|
||||
timeout=aiohttp.ClientTimeout(total=10)) as resp:
|
||||
if resp.status == 304:
|
||||
return "unchanged"
|
||||
if resp.status in (401, 403):
|
||||
return "refused"
|
||||
if resp.status != 200:
|
||||
return "unavailable"
|
||||
body = await resp.json()
|
||||
self.apply(body.get("data") or {})
|
||||
self._etag = resp.headers.get("ETag")
|
||||
return "updated"
|
||||
except Exception as exc: # network, timeout, bad JSON — keep the last good copy
|
||||
logger.debug(f"Agent registry poll failed: {exc}")
|
||||
return "unavailable"
|
||||
|
||||
async def run(self, base_url: str, api_key: str, poll_seconds: float = POLL_SECONDS) -> None:
|
||||
"""Poll forever. Never raises; the engine runs on env defaults meanwhile."""
|
||||
if not api_key:
|
||||
logger.warning("INTERNAL_API_KEY unset — agent registry not read; running on env defaults")
|
||||
return
|
||||
last = None
|
||||
async with aiohttp.ClientSession() as session:
|
||||
while True:
|
||||
outcome = await self.fetch_once(session, base_url, api_key)
|
||||
if outcome != last and outcome in ("refused", "unavailable"):
|
||||
logger.warning(
|
||||
f"Agent registry {outcome} at {base_url}; "
|
||||
+ ("keeping the last loaded copy" if self.loaded else "running on env defaults")
|
||||
)
|
||||
last = outcome
|
||||
await asyncio.sleep(poll_seconds)
|
||||
|
||||
|
||||
registry = RegistryClient()
|
||||
@@ -16,6 +16,9 @@ class MessageType(str, Enum):
|
||||
ORDER_DELIVERED = "ORDER_DELIVERED"
|
||||
ORDER_CANCELLED = "ORDER_CANCELLED"
|
||||
ORDER_DELAYED = "ORDER_DELAYED"
|
||||
# OrderAgent._update_status publishes this; it was missing from the enum, so
|
||||
# every update_status task raised AttributeError.
|
||||
ORDER_STATUS_UPDATE = "ORDER_STATUS_UPDATE"
|
||||
HUB_STATUS_UPDATE = "HUB_STATUS_UPDATE"
|
||||
VEHICLE_ASSIGNED = "VEHICLE_ASSIGNED"
|
||||
ROUTE_OPTIMIZED = "ROUTE_OPTIMIZED"
|
||||
|
||||
Reference in New Issue
Block a user