krow-workforce-agent has declared five subagents since it was written and
answered every question by itself. Everything for §6 existed except the
delegation: the parser read `subagents:`, runtime.Agent carried them, the
loader populated them, agent_runs had a parent_run_id column with a
self-reference and a no-self-parent constraint, and budget.go's comments
already described sharing a budget with subagents. Nothing called any of it.
A subagent is offered to the parent's model as a tool, because §6 says that is
what delegation is from the parent's side. Three rules are enforced rather than
assumed, each with a test that fails if it stops holding:
I1 The subagent runs as the ORIGINAL caller. It cannot read anything the
person could not read directly.
§6 It SHARES the parent's budget. The test sets MaxSteps to 1, spends it in
the parent, and asserts the child terminates BudgetExceeded — an
assertion that only passes when the budget is shared, and that a fresh
budget would quietly turn green.
§3 Depth is capped at 2. At the cap no subagent is loaded or offered, so a
cycle reaching run time is bounded rather than unbounded.
I4 survives too: a write a SUBAGENT wants approved still stops the whole run
and asks a person, rather than being performed because it happened one level
down.
Two bugs found by running it rather than by reading it:
- delegate() read the error before the result. finish returns a non-nil
error for every termination that is not Completed, INCLUDING
ConfirmationPending — which is not a failure but a run that stopped to ask
a question. Reading the error first discarded the result and with it the
confirmation, so a subagent's write silently never happened and nobody was
asked.
- Delegated trajectories were never persisted at all. parent_run_id is a
foreign key and a subagent finishes BEFORE the run that delegated to it,
so every child insert named a parent row that did not exist yet. The
database refused it; finish deliberately does not fail a run over a sink
error; and the entry recording that the trajectory could not be saved was
itself in the trajectory that was not saved. Children are now buffered and
written by finish after the parent's own row, each arriving with its
descendants already ordered behind it, so one pass writes a whole tree
parent-first. The regression test asserts on save ORDER, because a
MemorySink has no foreign key and will pass either way.
Verified end to end against a live model: an agent with no tools of its own and
one subagent produced
delegation-probe run=run_16622d7de6 parent=(root)
talent-pool-agent run=run_64160684b1 parent=run_16622d7de6
with the subagent's answer reaching the parent's model. Full suite green, only
TestLive* skipped.
Not addressed: §3's publish-time cycle detection, which needs the whole agent
set in hand. The depth cap is what holds without it, and is the half that
matters at run time.
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01PJvibeSc1JYXjatankqM1g
363 lines
13 KiB
Go
363 lines
13 KiB
Go
package runtime
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"strings"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/krow/krow-backend/go-api/internal/authctx"
|
|
"github.com/krow/krow-backend/go-api/internal/gateway"
|
|
"github.com/krow/krow-backend/go-api/internal/tools"
|
|
)
|
|
|
|
// fakeSubagents resolves subagents from a map, and remembers who asked — which
|
|
// is how I1 is checked below.
|
|
type fakeSubagents struct {
|
|
agents map[string]*Agent
|
|
asked []string
|
|
ident authctx.Identity
|
|
}
|
|
|
|
func (f *fakeSubagents) LoadExecutableAgent(_ context.Context, ident authctx.Identity, id string) (*Agent, error) {
|
|
f.asked = append(f.asked, id)
|
|
f.ident = ident
|
|
if a, ok := f.agents[id]; ok {
|
|
return a, nil
|
|
}
|
|
return nil, fmt.Errorf("no agent %q", id)
|
|
}
|
|
|
|
func childAgent() *Agent {
|
|
return &Agent{
|
|
ID: "talent-pool-agent", Name: "Talent Pool Agent", Version: 1,
|
|
Description: "Who is available in the pool.",
|
|
Reasoning: "balanced",
|
|
Pages: []string{"talent-pool"},
|
|
}
|
|
}
|
|
|
|
// parentWith returns an agent declaring the given subagents, and a resolver
|
|
// that can supply the child.
|
|
func parentWith(subs ...string) (*Agent, *fakeSubagents) {
|
|
p := testAgent()
|
|
p.Subagents = subs
|
|
return p, &fakeSubagents{agents: map[string]*Agent{"talent-pool-agent": childAgent()}}
|
|
}
|
|
|
|
func delegationCall(id, tool, question string) gateway.ToolCall {
|
|
return gateway.ToolCall{
|
|
ID: id, Name: tool,
|
|
Input: json.RawMessage(fmt.Sprintf(`{"question":%q}`, question)),
|
|
}
|
|
}
|
|
|
|
// A subagent is offered to the model as a tool. Before this, `subagents:`
|
|
// parsed, loaded, and was dropped — the model was never told the agent existed.
|
|
func TestDelegationOffersSubagentsAsTools(t *testing.T) {
|
|
parent, res := parentWith("talent-pool-agent")
|
|
gw := &scriptedGateway{}
|
|
exec := NewModelExecutor(gw, &MemorySink{}, tools.NewRegistry()).WithSubagents(res)
|
|
|
|
if _, err := exec.ExecuteAgent(context.Background(), parent, testInput("who is free?")); err != nil {
|
|
t.Fatalf("unexpected error: %v", err)
|
|
}
|
|
if len(gw.seen) == 0 {
|
|
t.Fatal("the model was never called")
|
|
}
|
|
var names []string
|
|
for _, td := range gw.seen[0].Tools {
|
|
names = append(names, td.Name)
|
|
}
|
|
want := delegationToolName("talent-pool-agent")
|
|
if len(names) != 1 || names[0] != want {
|
|
t.Fatalf("offered tools = %v, want exactly [%s]", names, want)
|
|
}
|
|
if len(res.asked) != 1 {
|
|
t.Errorf("the resolver was asked %d times, want 1 — subagents resolve once per run", len(res.asked))
|
|
}
|
|
}
|
|
|
|
// Without a resolver, delegation is off and the agent runs alone. This is the
|
|
// behaviour every deployment had, and it must stay a quiet degrade rather than
|
|
// a failure.
|
|
func TestDelegationIsOffWithoutAResolver(t *testing.T) {
|
|
parent, _ := parentWith("talent-pool-agent")
|
|
gw := &scriptedGateway{}
|
|
exec := NewModelExecutor(gw, &MemorySink{}, tools.NewRegistry())
|
|
|
|
res, err := exec.ExecuteAgent(context.Background(), parent, testInput("who is free?"))
|
|
if err != nil {
|
|
t.Fatalf("unexpected error: %v", err)
|
|
}
|
|
if res.Termination != TerminationCompleted {
|
|
t.Errorf("Termination = %q, want Completed", res.Termination)
|
|
}
|
|
if len(gw.seen[0].Tools) != 0 {
|
|
t.Errorf("offered %d tools with no resolver, want 0", len(gw.seen[0].Tools))
|
|
}
|
|
}
|
|
|
|
// The subagent runs, its answer reaches the parent's model, and it writes its
|
|
// OWN trajectory linked by parent_run_id (§6).
|
|
func TestDelegationRunsTheSubagentAndLinksItsTrajectory(t *testing.T) {
|
|
parent, resolver := parentWith("talent-pool-agent")
|
|
tool := delegationToolName("talent-pool-agent")
|
|
|
|
gw := &scriptedGateway{steps: []*gateway.Response{
|
|
// parent asks
|
|
{ToolCalls: []gateway.ToolCall{delegationCall("c1", tool, "who is available?")},
|
|
StopReason: "tool_use", Model: "fake-model"},
|
|
// child answers
|
|
{Text: "Five workers are available.", StopReason: "end_turn", Model: "fake-model"},
|
|
// parent answers from it
|
|
{Text: "Five are free this week.", StopReason: "end_turn", Model: "fake-model"},
|
|
}}
|
|
sink := &MemorySink{}
|
|
exec := NewModelExecutor(gw, sink, tools.NewRegistry()).WithSubagents(resolver)
|
|
|
|
res, err := exec.ExecuteAgent(context.Background(), parent, testInput("who is free?"))
|
|
if err != nil {
|
|
t.Fatalf("unexpected error: %v", err)
|
|
}
|
|
if res.Termination != TerminationCompleted || res.Output != "Five are free this week." {
|
|
t.Fatalf("Termination=%q Output=%q", res.Termination, res.Output)
|
|
}
|
|
|
|
if len(sink.Runs) != 2 {
|
|
t.Fatalf("%d trajectories saved, want 2 — the subagent gets its own", len(sink.Runs))
|
|
}
|
|
var child, root *Trajectory
|
|
for _, tr := range sink.Runs {
|
|
if tr.AgentID == "talent-pool-agent" {
|
|
child = tr
|
|
} else {
|
|
root = tr
|
|
}
|
|
}
|
|
if child == nil || root == nil {
|
|
t.Fatal("expected one trajectory per agent")
|
|
}
|
|
// ORDER MATTERS, and a MemorySink will not tell you so on its own.
|
|
// agent_runs.parent_run_id is a foreign key, and a subagent finishes
|
|
// before the run that delegated to it — so saving in completion order
|
|
// makes every child insert name a parent row that does not exist yet. The
|
|
// database refuses it, finish does not fail a run over a sink error, and
|
|
// every delegated trajectory disappears without trace. Parent first.
|
|
if sink.Runs[0].AgentID != root.AgentID {
|
|
t.Errorf("saved %q first, want the parent %q — a child written before its "+
|
|
"parent violates the parent_run_id foreign key and is silently dropped",
|
|
sink.Runs[0].AgentID, root.AgentID)
|
|
}
|
|
if child.ParentRunID != root.RunID {
|
|
t.Errorf("child.ParentRunID = %q, want the parent's run id %q", child.ParentRunID, root.RunID)
|
|
}
|
|
if root.ParentRunID != "" {
|
|
t.Errorf("the root run has ParentRunID %q, want empty", root.ParentRunID)
|
|
}
|
|
|
|
// The parent's model must actually have received the child's answer.
|
|
last := gw.seen[len(gw.seen)-1]
|
|
var sawAnswer bool
|
|
for _, msg := range last.Messages {
|
|
for _, tr := range msg.ToolResults {
|
|
if strings.Contains(tr.Content, "Five workers are available.") {
|
|
sawAnswer = true
|
|
}
|
|
}
|
|
}
|
|
if !sawAnswer {
|
|
t.Error("the subagent's answer never reached the parent's model")
|
|
}
|
|
}
|
|
|
|
// §6, and the reason delegation cannot be a way to buy more budget.
|
|
//
|
|
// MaxSteps is 1. The parent spends it, then delegates. If the subagent got a
|
|
// FRESH budget it would have a step of its own and answer; sharing the
|
|
// parent's means it has nothing left and terminates BudgetExceeded. The
|
|
// assertion is on the child's termination, which differs between the two
|
|
// designs and cannot be produced by accident.
|
|
func TestDelegationSharesTheParentsBudget(t *testing.T) {
|
|
parent, resolver := parentWith("talent-pool-agent")
|
|
tool := delegationToolName("talent-pool-agent")
|
|
|
|
gw := &scriptedGateway{steps: []*gateway.Response{
|
|
{ToolCalls: []gateway.ToolCall{delegationCall("c1", tool, "who is available?")},
|
|
StopReason: "tool_use", Model: "fake-model"},
|
|
{Text: "the child should never get this far", StopReason: "end_turn", Model: "fake-model"},
|
|
}}
|
|
sink := &MemorySink{}
|
|
exec := NewModelExecutor(gw, sink, tools.NewRegistry()).WithSubagents(resolver)
|
|
|
|
// The parent exhausts the budget too and finish returns an error with its
|
|
// result; that is expected here and not what this test is about.
|
|
_, _ = exec.executeRun(context.Background(), parent, testInput("who is free?"),
|
|
Limits{MaxSteps: 1, MaxToolCalls: 5, MaxTokens: 100_000, Deadline: 30 * time.Second},
|
|
delegation{})
|
|
|
|
var child *Trajectory
|
|
for _, tr := range sink.Runs {
|
|
if tr.AgentID == "talent-pool-agent" {
|
|
child = tr
|
|
}
|
|
}
|
|
if child == nil {
|
|
t.Fatal("the subagent never ran")
|
|
}
|
|
if child.Termination != TerminationBudgetExceeded {
|
|
t.Errorf("child Termination = %q, want BudgetExceeded — it was given a fresh budget "+
|
|
"instead of sharing its parent's, which §6 forbids", child.Termination)
|
|
}
|
|
}
|
|
|
|
// I1: a subagent executes as the ORIGINAL caller and never a widened one.
|
|
func TestDelegationRunsAsTheOriginalCaller(t *testing.T) {
|
|
parent, resolver := parentWith("talent-pool-agent")
|
|
gw := &scriptedGateway{}
|
|
exec := NewModelExecutor(gw, &MemorySink{}, tools.NewRegistry()).WithSubagents(resolver)
|
|
|
|
in := testInput("who is free?")
|
|
if _, err := exec.ExecuteAgent(context.Background(), parent, in); err != nil {
|
|
t.Fatalf("unexpected error: %v", err)
|
|
}
|
|
if resolver.ident.UserID != in.Identity.UserID || resolver.ident.OrgID != in.Identity.OrgID {
|
|
t.Errorf("subagent resolved as %+v, want the caller %+v", resolver.ident, in.Identity)
|
|
}
|
|
}
|
|
|
|
// §3's depth cap. At the cap, no subagent is offered at all, so a cycle in the
|
|
// spec graph is bounded rather than unbounded.
|
|
func TestDelegationCapsDepth(t *testing.T) {
|
|
parent, resolver := parentWith("talent-pool-agent")
|
|
gw := &scriptedGateway{}
|
|
sink := &MemorySink{}
|
|
exec := NewModelExecutor(gw, sink, tools.NewRegistry()).WithSubagents(resolver)
|
|
|
|
_, err := exec.executeRun(context.Background(), parent, testInput("who is free?"),
|
|
LimitsForTier("balanced"), delegation{depth: MaxDelegationDepth})
|
|
if err != nil {
|
|
t.Fatalf("unexpected error: %v", err)
|
|
}
|
|
if len(gw.seen[0].Tools) != 0 {
|
|
t.Errorf("offered %d tools at depth %d, want 0", len(gw.seen[0].Tools), MaxDelegationDepth)
|
|
}
|
|
if len(resolver.asked) != 0 {
|
|
t.Errorf("resolved %d subagents at the cap, want 0 — the cap should short-circuit "+
|
|
"before loading anything", len(resolver.asked))
|
|
}
|
|
// And it must be visible, not silent.
|
|
if !hasError(sink.Last(), "runtime.delegation_depth") {
|
|
t.Error("hitting the depth cap was not recorded in the trajectory")
|
|
}
|
|
}
|
|
|
|
// An agent naming itself is refused before the model is offered the call: the
|
|
// database has a no-self-parent constraint, and a run that tried would fail on
|
|
// insert rather than on anything legible.
|
|
func TestDelegationRefusesSelfReference(t *testing.T) {
|
|
parent := testAgent()
|
|
parent.Subagents = []string{parent.ID}
|
|
resolver := &fakeSubagents{agents: map[string]*Agent{}}
|
|
|
|
gw := &scriptedGateway{}
|
|
sink := &MemorySink{}
|
|
exec := NewModelExecutor(gw, sink, tools.NewRegistry()).WithSubagents(resolver)
|
|
|
|
if _, err := exec.ExecuteAgent(context.Background(), parent, testInput("hello")); err != nil {
|
|
t.Fatalf("unexpected error: %v", err)
|
|
}
|
|
if len(gw.seen[0].Tools) != 0 {
|
|
t.Errorf("a self-referencing subagent was offered as a tool")
|
|
}
|
|
if !hasError(sink.Last(), "runtime.subagent_self") {
|
|
t.Error("the self-reference was not recorded")
|
|
}
|
|
}
|
|
|
|
// A subagent that cannot be loaded degrades rather than failing the run — the
|
|
// same rule as an unknown tool — but is recorded so somebody can fix the spec.
|
|
func TestDelegationRecordsAnUnloadableSubagent(t *testing.T) {
|
|
parent := testAgent()
|
|
parent.Subagents = []string{"no-such-agent"}
|
|
resolver := &fakeSubagents{agents: map[string]*Agent{}}
|
|
|
|
gw := &scriptedGateway{}
|
|
sink := &MemorySink{}
|
|
exec := NewModelExecutor(gw, sink, tools.NewRegistry()).WithSubagents(resolver)
|
|
|
|
res, err := exec.ExecuteAgent(context.Background(), parent, testInput("hello"))
|
|
if err != nil {
|
|
t.Fatalf("unexpected error: %v", err)
|
|
}
|
|
if res.Termination != TerminationCompleted {
|
|
t.Errorf("Termination = %q, want Completed — an unloadable subagent degrades", res.Termination)
|
|
}
|
|
if !hasError(sink.Last(), "runtime.unknown_subagent") {
|
|
t.Error("the unloadable subagent was not recorded")
|
|
}
|
|
}
|
|
|
|
// I4 survives delegation. A write a SUBAGENT wants approved still stops
|
|
// everything and asks a person — it does not get performed because it happened
|
|
// one level down.
|
|
func TestSubagentConfirmationStopsTheParentRun(t *testing.T) {
|
|
writeTool := tools.Tool{
|
|
Name: "assign_worker",
|
|
Description: "Assigns a worker to a shift, which is a real change to a real rota.",
|
|
InputSchema: map[string]any{"type": "object"},
|
|
Effect: tools.EffectWrite,
|
|
Confirm: func(context.Context, tools.Context, json.RawMessage) (*tools.Confirmation, *tools.Result) {
|
|
return &tools.Confirmation{Token: "tok_1", Tool: "assign_worker", Title: "Assign Maya to Bar"}, nil
|
|
},
|
|
Handler: func(context.Context, tools.Context, json.RawMessage) tools.Result {
|
|
t.Error("the write ran without approval")
|
|
return tools.OK(map[string]any{})
|
|
},
|
|
}
|
|
reg := tools.NewRegistry()
|
|
reg.MustRegister(writeTool)
|
|
|
|
child := childAgent()
|
|
child.Tools = []string{"assign_worker"}
|
|
parent := testAgent()
|
|
parent.Subagents = []string{child.ID}
|
|
resolver := &fakeSubagents{agents: map[string]*Agent{child.ID: child}}
|
|
|
|
tool := delegationToolName(child.ID)
|
|
gw := &scriptedGateway{steps: []*gateway.Response{
|
|
{ToolCalls: []gateway.ToolCall{delegationCall("c1", tool, "assign Maya")},
|
|
StopReason: "tool_use", Model: "fake-model"},
|
|
{ToolCalls: []gateway.ToolCall{{ID: "c2", Name: "assign_worker", Input: json.RawMessage(`{}`)}},
|
|
StopReason: "tool_use", Model: "fake-model"},
|
|
}}
|
|
exec := NewModelExecutor(gw, &MemorySink{}, reg).WithSubagents(resolver)
|
|
|
|
// The error beside the result is how the loop reports every termination
|
|
// that is not Completed; ConfirmationPending is not a failure and the
|
|
// existing confirmation tests ignore it the same way.
|
|
res, _ := exec.ExecuteAgent(context.Background(), parent, testInput("assign someone"))
|
|
if res.Termination != TerminationConfirmationPending {
|
|
t.Fatalf("Termination = %q, want ConfirmationPending — a subagent's write must "+
|
|
"still stop and ask", res.Termination)
|
|
}
|
|
if len(res.Confirmations) != 1 || res.Confirmations[0].Tool != "assign_worker" {
|
|
t.Fatalf("Confirmations = %+v, want the subagent's pending write", res.Confirmations)
|
|
}
|
|
}
|
|
|
|
// hasError reports whether a trajectory recorded an entry with the given code.
|
|
func hasError(tr *Trajectory, code string) bool {
|
|
if tr == nil {
|
|
return false
|
|
}
|
|
for _, e := range tr.Entries {
|
|
if strings.Contains(fmt.Sprint(e), code) {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|