Files
doormile_backend/internal/ai/registry/registry_test.go

481 lines
18 KiB
Go

package registry
import (
"encoding/json"
"errors"
"strings"
"testing"
"doormile/models"
)
// No database here: these pin the seed's integrity and every rule a write must
// pass. Seed/Load/Update against Postgres are not exercised by this file.
func boolp(b bool) *bool { return &b }
func strp(s string) *string { return &s }
func isValidation(err error) bool { var v *ValidationError; return errors.As(err, &v) }
// ── Seed integrity ──────────────────────────────────────────────────────────
func TestSeedIDsAreUnique(t *testing.T) {
seen := map[string]bool{}
for _, a := range SeedAgents {
if seen["agent:"+a.Agentid] {
t.Errorf("agent %s seeded twice", a.Agentid)
}
seen["agent:"+a.Agentid] = true
}
for _, tl := range SeedTools {
if seen["tool:"+tl.Toolname] {
t.Errorf("tool %s seeded twice", tl.Toolname)
}
seen["tool:"+tl.Toolname] = true
}
for _, s := range SeedSkills {
if seen["skill:"+s.Skill.Skillid] {
t.Errorf("skill %s seeded twice", s.Skill.Skillid)
}
seen["skill:"+s.Skill.Skillid] = true
}
}
func TestSeedAgentsUseKnownValues(t *testing.T) {
for _, a := range SeedAgents {
if !validStatuses[a.Status] {
t.Errorf("%s: unknown status %q", a.Agentid, a.Status)
}
if !validRuntimes[a.Runtime] {
t.Errorf("%s: unknown runtime %q", a.Agentid, a.Runtime)
}
if a.Autonomous {
t.Errorf("%s is seeded autonomous; every agent must start with autonomy off", a.Agentid)
}
if a.Name == "" || a.Purpose == "" || a.Classref == "" {
t.Errorf("%s: name, purpose and classref are all required", a.Agentid)
}
}
}
// Autonomy can only be switched on the three agents whose writes AI_engine
// actually gates. A gate on any other agent would be a switch wired to nothing.
func TestOnlyTheThreeGatedAgentsHaveAutonomyGates(t *testing.T) {
want := map[string]bool{"DISPATCH_AGENT": true, "EXCEPTION_AGENT": true, "EXPRESS_DISPATCH_AGENT": true}
for _, a := range SeedAgents {
if a.Hasautonomygate != want[a.Agentid] {
t.Errorf("%s: hasautonomygate = %v, want %v", a.Agentid, a.Hasautonomygate, want[a.Agentid])
}
}
}
// The simulated agents must say so — the console renders this badge, and an
// agent that looks live and is not is the failure this registry exists to end.
func TestSimulatedAgentsAreSeededAsSimulation(t *testing.T) {
for _, id := range []string{"HUB_AGENT", "FLEET_AGENT", "ROUTE_OPTIMIZER"} {
for _, a := range SeedAgents {
if a.Agentid == id && a.Status != StatusSimulation {
t.Errorf("%s status = %q, want simulation", id, a.Status)
}
}
}
}
func TestSeedToolsAreWellFormed(t *testing.T) {
for _, tl := range SeedTools {
if !validKinds[tl.Kind] {
t.Errorf("%s: unknown kind %q", tl.Toolname, tl.Kind)
}
if (tl.Kind == KindWrite || tl.Kind == KindNotify) && !tl.Requiresconfirmation {
t.Errorf("%s is a %s tool but does not require confirmation", tl.Toolname, tl.Kind)
}
if tl.Kind == KindRead && tl.Requiresconfirmation {
t.Errorf("%s is read-only but requires confirmation", tl.Toolname)
}
var schema map[string]any
if err := json.Unmarshal([]byte(tl.Inputschema), &schema); err != nil || schema["type"] != "object" {
t.Errorf("%s: inputschema is not a JSON-schema object: %s", tl.Toolname, tl.Inputschema)
}
if tl.Description == "" || tl.Target == "" || tl.Implementedat == "" {
t.Errorf("%s: description, target and implementedat are all required", tl.Toolname)
}
}
}
func TestEverySkillPointsAtRealAgentsAndTools(t *testing.T) {
agents := map[string]bool{}
for _, a := range SeedAgents {
agents[a.Agentid] = true
}
tools := map[string]bool{}
for _, tl := range SeedTools {
tools[tl.Toolname] = true
}
for _, s := range SeedSkills {
if !agents[s.Skill.Agentid] {
t.Errorf("skill %s belongs to unknown agent %s", s.Skill.Skillid, s.Skill.Agentid)
}
if len(s.Tools) == 0 {
t.Errorf("skill %s has no tools", s.Skill.Skillid)
}
for _, tl := range s.Tools {
if !tools[tl] {
t.Errorf("skill %s uses unknown tool %s", s.Skill.Skillid, tl)
}
}
if s.Skill.Source != SourceEngine && s.Skill.Source != SourceConsole {
t.Errorf("seeded skill %s has source %q; custom is for operator-made skills only", s.Skill.Skillid, s.Skill.Source)
}
}
}
// Every seeded tool is used by some skill — the registry lists capabilities in
// use, not a catalogue of ideas.
func TestEveryToolIsUsedBySomeSkill(t *testing.T) {
used := map[string]bool{}
for _, s := range SeedSkills {
for _, tl := range s.Tools {
used[tl] = true
}
}
for _, tl := range SeedTools {
if !used[tl.Toolname] {
t.Errorf("tool %s is seeded but no skill uses it", tl.Toolname)
}
}
}
func TestSeedThresholdDefaultsAreValid(t *testing.T) {
for _, s := range SeedSkills {
keys := map[string]bool{}
for _, spec := range s.Schema {
if keys[spec.Key] {
t.Errorf("%s: threshold %s declared twice", s.Skill.Skillid, spec.Key)
}
keys[spec.Key] = true
if spec.Min >= spec.Max {
t.Errorf("%s.%s: min %v is not below max %v", s.Skill.Skillid, spec.Key, spec.Min, spec.Max)
}
if err := spec.check(spec.Default); err != nil {
t.Errorf("%s.%s: default is itself invalid: %v", s.Skill.Skillid, spec.Key, err)
}
}
}
}
// Rebalancing has no endpoint behind it. It must never ship switched on.
func TestRebalanceShipsDisabled(t *testing.T) {
for _, s := range SeedSkills {
if s.Skill.Skillid == "dispatch_rebalance" && s.Skill.Enabled {
t.Fatal("dispatch_rebalance is seeded enabled; nothing implements it")
}
}
}
// The ops-layer skills keep the branch's ids and threshold keys so the Phase 3
// port maps one to one. Pin them.
func TestConsoleOpsSkillsKeepBranchIDsAndKeys(t *testing.T) {
want := map[string][]string{
"skill_sla_guardian": {"slaRiskWindowMin", "unassignedAgingMin"},
"skill_doorstep_stall": {"arrivedStalledMin"},
"skill_fleet_balancer": {"riderActiveCap"},
"skill_high_value_cod": {"codRiskThresholdAmount"},
"skill_rider_battery_safety": {"criticalBatteryPercent"},
"skill_hub_congestion": {"hubDwellMinutes"},
"skill_late_dispatch": {"lateDispatchMinutes", "criticalDispatchMinutes"},
"skill_cash_exposure": {"maxCashPerRider", "warningCashPercent"},
}
for _, s := range SeedSkills {
keys, ok := want[s.Skill.Skillid]
if !ok {
continue
}
delete(want, s.Skill.Skillid)
var got []string
for _, spec := range s.Schema {
got = append(got, spec.Key)
}
if strings.Join(got, ",") != strings.Join(keys, ",") {
t.Errorf("%s threshold keys = %v, want %v", s.Skill.Skillid, got, keys)
}
}
for id := range want {
t.Errorf("branch skill %s is missing from the seed", id)
}
}
// ── Thresholds ──────────────────────────────────────────────────────────────
var testSchema = []ThresholdSpec{
{Key: "minutes", Default: 20, Min: 10, Max: 60, Step: 5},
{Key: "confidence", Default: 0.7, Min: 0.5, Max: 1, Step: 0.05},
}
func TestEffectiveThresholdsFillsDefaultsAndDropsStrays(t *testing.T) {
got := EffectiveThresholds(testSchema, map[string]float64{"minutes": 30, "removed": 9, "confidence": 5})
if got["minutes"] != 30 {
t.Errorf("a valid stored value was not kept: %v", got["minutes"])
}
if got["confidence"] != 0.7 {
t.Errorf("an out-of-range stored value was not replaced by the default: %v", got["confidence"])
}
if _, stray := got["removed"]; stray {
t.Error("a key the schema no longer has was kept")
}
}
func TestApplyThresholdPatchAcceptsValidValues(t *testing.T) {
got, err := ApplyThresholdPatch(testSchema, nil, map[string]any{"minutes": 45.0, "confidence": 0.85})
if err != nil {
t.Fatalf("valid patch refused: %v", err)
}
if got["minutes"] != 45 || got["confidence"] != 0.85 {
t.Errorf("patch not applied: %v", got)
}
}
func TestApplyThresholdPatchRefusesBadInput(t *testing.T) {
cases := map[string]map[string]any{
"unknown key": {"minuts": 30.0},
"not a number": {"minutes": "30"},
"a boolean": {"minutes": true},
"below min": {"minutes": 5.0},
"above max": {"minutes": 65.0},
"off the step": {"minutes": 33.0},
"off the fstep": {"confidence": 0.72},
}
for name, patch := range cases {
if _, err := ApplyThresholdPatch(testSchema, nil, patch); !isValidation(err) {
t.Errorf("%s: want a ValidationError, got %v", name, err)
}
}
}
// Every problem is reported at once, so an operator fixes the form in one pass.
func TestApplyThresholdPatchReportsEveryProblem(t *testing.T) {
_, err := ApplyThresholdPatch(testSchema, nil, map[string]any{"minutes": 5.0, "nope": 1.0})
if err == nil || !strings.Contains(err.Error(), "minutes") || !strings.Contains(err.Error(), "nope") {
t.Fatalf("want both problems named, got %v", err)
}
}
// A refused patch changes nothing — the whole set, not the valid half, is rejected.
func TestApplyThresholdPatchIsAllOrNothing(t *testing.T) {
got, err := ApplyThresholdPatch(testSchema, map[string]float64{"minutes": 20}, map[string]any{"minutes": 40.0, "confidence": 9.0})
if err == nil || got != nil {
t.Fatalf("partial patch was applied: %v, %v", got, err)
}
}
func TestParseThresholdsToleratesJunk(t *testing.T) {
for _, raw := range []string{"", "null", "not json", "[]"} {
if got := ParseThresholds(raw); got == nil || len(got) != 0 {
t.Errorf("ParseThresholds(%q) = %v, want an empty map", raw, got)
}
}
if got := ParseSchema("garbage"); got == nil || len(got) != 0 {
t.Errorf("ParseSchema(garbage) = %v, want empty", got)
}
}
// ── Agent patches ───────────────────────────────────────────────────────────
var gated = models.AIAgent{Agentid: "EXCEPTION_AGENT", Runtime: RuntimeEngine, Hasautonomygate: true}
func TestAutonomyOnNeedsTypedConfirmation(t *testing.T) {
if err := CheckAgentPatch(gated, AgentPatch{Autonomous: boolp(true)}); !isValidation(err) {
t.Errorf("autonomy switched on with no confirmation: %v", err)
}
if err := CheckAgentPatch(gated, AgentPatch{Autonomous: boolp(true), Confirm: "exception_agent"}); !isValidation(err) {
t.Errorf("a near-miss confirmation was accepted: %v", err)
}
if err := CheckAgentPatch(gated, AgentPatch{Autonomous: boolp(true), Confirm: "EXCEPTION_AGENT"}); err != nil {
t.Errorf("correct confirmation refused: %v", err)
}
}
// Switching autonomy OFF is always allowed without ceremony — the safe direction.
func TestAutonomyOffNeedsNoConfirmation(t *testing.T) {
on := gated
on.Autonomous = true
if err := CheckAgentPatch(on, AgentPatch{Autonomous: boolp(false)}); err != nil {
t.Errorf("switching autonomy off was refused: %v", err)
}
}
func TestAutonomyRefusedOnUngatedAgent(t *testing.T) {
hub := models.AIAgent{Agentid: "HUB_AGENT", Runtime: RuntimeEngine}
if err := CheckAgentPatch(hub, AgentPatch{Autonomous: boolp(true), Confirm: "HUB_AGENT"}); !isValidation(err) {
t.Errorf("autonomy set on an agent with no gate: %v", err)
}
}
func TestModelIDRules(t *testing.T) {
for _, ok := range []string{"", "claude-sonnet-5-5", "claude-haiku-4-5-20251001", "claude-opus-5-5"} {
if err := CheckAgentPatch(gated, AgentPatch{Model: strp(ok)}); err != nil {
t.Errorf("model %q refused: %v", ok, err)
}
}
for _, bad := range []string{"gpt-4o", "claude-", "Claude-Sonnet", "claude-x; drop table", strings.Repeat("claude-a", 20)} {
if err := CheckAgentPatch(gated, AgentPatch{Model: strp(bad)}); !isValidation(err) {
t.Errorf("model %q accepted", bad)
}
}
console := models.AIAgent{Agentid: "CONSOLE_ASSISTANT", Runtime: RuntimeConsole}
if err := CheckAgentPatch(console, AgentPatch{Model: strp("claude-sonnet-5-5")}); !isValidation(err) {
t.Error("a model was set on a console agent, which has no model setting")
}
}
func TestEmptyAgentPatchRefused(t *testing.T) {
if err := CheckAgentPatch(gated, AgentPatch{}); !isValidation(err) {
t.Error("an empty patch was accepted")
}
}
// ── New skills ──────────────────────────────────────────────────────────────
func TestCheckNewSkill(t *testing.T) {
good := NewSkill{Agentid: "CONSOLE_OPS_AGENT", Title: "Night shift watch", Tools: []string{"lookup_order"}}
if err := CheckNewSkill(good); err != nil {
t.Fatalf("valid skill refused: %v", err)
}
bad := map[string]NewSkill{
"no agent": {Title: "x", Tools: []string{"a"}},
"no title": {Agentid: "A", Title: " ", Tools: []string{"a"}},
"no tools": {Agentid: "A", Title: "x"},
"duplicate tool": {Agentid: "A", Title: "x", Tools: []string{"a", "a"}},
"title too long": {Agentid: "A", Title: strings.Repeat("x", 121), Tools: []string{"a"}},
}
for name, n := range bad {
if err := CheckNewSkill(n); !isValidation(err) {
t.Errorf("%s: accepted", name)
}
}
}
func TestCustomSkillID(t *testing.T) {
cases := map[string]string{
"Night Shift Watch": "custom_night_shift_watch",
" COD > ₹5,000 alerts!! ": "custom_cod_5_000_alerts",
"!!!": "custom_skill",
}
for in, want := range cases {
if got := customSkillID(in); got != want {
t.Errorf("customSkillID(%q) = %q, want %q", in, got, want)
}
}
if got := customSkillID(strings.Repeat("abc ", 40)); len(got) > 64 {
t.Errorf("id %q exceeds the 64-char column", got)
}
}
// ── Build & ETag ────────────────────────────────────────────────────────────
func seededRows() ([]models.AIAgent, []models.AITool, []models.AISkill, []models.AISkillTool) {
var skills []models.AISkill
var links []models.AISkillTool
for _, s := range SeedSkills {
row := s.Skill
row.Thresholdsschema = mustJSON(schemaOrEmpty(s.Schema))
row.Thresholds = mustJSON(DefaultThresholds(s.Schema))
skills = append(skills, row)
for _, tl := range s.Tools {
links = append(links, models.AISkillTool{Skillid: s.Skill.Skillid, Toolname: tl})
}
}
return SeedAgents, SeedTools, skills, links
}
func TestBuildCountsSkillsAndToolsPerAgent(t *testing.T) {
snap := Build(seededRows())
for _, a := range snap.Agents {
if a.Agentid == "EXPRESS_DISPATCH_AGENT" && (a.Skillcount != 1 || a.Toolcount != 4) {
t.Errorf("EXPRESS_DISPATCH_AGENT: %d skills, %d tools; want 1 and 4", a.Skillcount, a.Toolcount)
}
if a.Agentid == "HUB_AGENT" && (a.Skillcount != 0 || a.Toolcount != 0) {
t.Errorf("HUB_AGENT has no skills, got %d/%d", a.Skillcount, a.Toolcount)
}
}
}
// The API must emit thresholds and schemas as JSON values, not as strings of JSON.
func TestSnapshotSerialisesJSONColumnsAsJSON(t *testing.T) {
b, err := json.Marshal(Build(seededRows()))
if err != nil {
t.Fatal(err)
}
var out struct {
Skills []struct {
Skillid string `json:"skillid"`
Thresholds map[string]float64 `json:"thresholds"`
Thresholdsschema []ThresholdSpec `json:"thresholdsschema"`
Tools []string `json:"tools"`
} `json:"skills"`
Tools []struct {
Inputschema map[string]any `json:"inputschema"`
} `json:"tools"`
}
if err := json.Unmarshal(b, &out); err != nil {
t.Fatalf("snapshot JSON does not decode as structured values: %v", err)
}
for _, s := range out.Skills {
if s.Skillid == "skill_cash_exposure" {
if s.Thresholds["maxCashPerRider"] != 10000 || len(s.Thresholdsschema) != 2 || len(s.Tools) != 2 {
t.Errorf("skill_cash_exposure serialised wrongly: %+v", s)
}
}
}
if len(out.Tools) == 0 || out.Tools[0].Inputschema["type"] != "object" {
t.Error("tool inputschema is not emitted as a JSON object")
}
}
func TestETagIsStableAndChangesWithContent(t *testing.T) {
a := ETag(Build(seededRows()))
if a != ETag(Build(seededRows())) {
t.Fatal("ETag differs for identical content")
}
agents, tools, skills, links := seededRows()
skills[0].Enabled = !skills[0].Enabled
if a == ETag(Build(agents, tools, skills, links)) {
t.Fatal("ETag did not change when a skill was toggled")
}
if !strings.HasPrefix(a, `W/"`) {
t.Errorf("ETag %s is not a weak validator", a)
}
}
// These three read fields /admin/bookings rows do not carry (payment amounts,
// battery). Enabled, they would read undefined on every row and report an
// all-clear board. They must ship off, in step with the console's defaults.
func TestSkillsWithNoDataSourceShipDisabled(t *testing.T) {
off := map[string]bool{"skill_high_value_cod": true, "skill_cash_exposure": true, "skill_rider_battery_safety": true}
for _, s := range SeedSkills {
if off[s.Skill.Skillid] {
delete(off, s.Skill.Skillid)
if s.Skill.Enabled {
t.Errorf("%s is seeded enabled but its rows carry no data for its rule", s.Skill.Skillid)
}
if !strings.Contains(s.Skill.Description, "OFF:") {
t.Errorf("%s does not say why it is off", s.Skill.Skillid)
}
}
}
for id := range off {
t.Errorf("%s missing from the seed", id)
}
}
// Only notify_riders has an executor in the console. Every other console write
// must say REVIEW ONLY, or Agent Studio would advertise an action that cannot run.
func TestConsoleWritesWithoutExecutorSayReviewOnly(t *testing.T) {
for _, tl := range SeedTools {
if !strings.HasPrefix(tl.Implementedat, consoleSrc+"lib/assistant/skills/") && !strings.Contains(tl.Implementedat, "(no executor)") {
continue
}
if tl.Kind != KindRead && !strings.HasPrefix(tl.Description, "REVIEW ONLY") {
t.Errorf("%s has no executor but its description does not start with REVIEW ONLY", tl.Toolname)
}
}
}