package runtime import ( "context" "encoding/json" "fmt" "strings" "github.com/krow/krow-backend/go-api/internal/authctx" "github.com/krow/krow-backend/go-api/internal/gateway" "github.com/krow/krow-backend/go-api/internal/tools" ) /* ── Delegation ───────────────────────────────────────────────────────────── §6: "Delegation is a tool call from the parent's perspective. Subagent runs get their own trajectory, linked by parent_run_id." Everything for this existed except the delegation. The parser read `subagents:`, the runtime type carried them, the loader populated them, agent_runs had a parent_run_id column with a self-reference and a no-self-parent constraint, and budget.go's comments described sharing a budget with subagents. Nothing called any of it, so krow-workforce-agent declared five subagents and answered every question alone. Three rules from §3 and §6 are load-bearing here, and each is enforced below rather than assumed: I1 A subagent runs as the ORIGINAL caller. It never receives a widened principal, so it can read exactly what the person could read directly. §6 A subagent SHARES the parent's budget. It never gets a fresh one, or a run could buy itself unlimited steps by delegating in a loop. §3 Depth is capped. The cap is what makes a cycle in the spec graph a bounded waste rather than an unbounded one. ──────────────────────────────────────────────────────────────────────────── */ // MaxDelegationDepth is §3's cap. // // The root run is depth 0, so a run at depth 2 is offered no subagents at all // and delegation stops there. Publish-time cycle detection is the other half // of §3 and is not here — this is what holds when a cycle reaches run time // anyway, which is the case worth defending against. const MaxDelegationDepth = 2 // SubagentResolver loads a subagent ready to run, with its own skills and // dependencies already resolved. // // An interface, satisfied by *Loader, so the loop keeps depending on behaviour // rather than on the repository — the same reason Retriever is one. type SubagentResolver interface { LoadExecutableAgent(ctx context.Context, ident authctx.Identity, idOrDefID string) (*Agent, error) } // WithSubagents attaches the resolver that turns a spec's `subagents:` into // agents this loop can actually run. Without it, delegation is silently off — // which is the behaviour every deployment had until now. func (m *ModelExecutor) WithSubagents(r SubagentResolver) *ModelExecutor { m.subagents = r return m } // delegation is what a subagent run inherits from its parent. // // The zero value is a root run: its own budget, no parent, depth 0. type delegation struct { budget *Budget parentRunID string depth int } // delegationToolPrefix marks a tool call as a delegation rather than a tool. const delegationToolPrefix = "ask_" // delegationToolName is the name a subagent is offered to the model under. // // Hyphens become underscores because agent ids are kebab-case and tool names // across this registry are snake_case; a model offered both conventions at // once picks badly. func delegationToolName(agentID string) string { return delegationToolPrefix + strings.ReplaceAll(agentID, "-", "_") } // resolveSubagents loads the agents this one may delegate to, keyed by the // tool name each is offered under. // // Every reason a subagent cannot be offered is RECORDED rather than silently // dropped. A spec that names five subagents and gets three is a spec somebody // needs to fix, and the trajectory is where they will look. func (m *ModelExecutor) resolveSubagents( ctx context.Context, rec *Recorder, agent *Agent, ident authctx.Identity, depth int, ) map[string]*Agent { if m.subagents == nil || len(agent.Subagents) == 0 { return nil } if depth >= MaxDelegationDepth { rec.Error("runtime.delegation_depth", fmt.Sprintf( "at depth %d of %d; %d subagent(s) were not offered", depth, MaxDelegationDepth, len(agent.Subagents))) return nil } out := make(map[string]*Agent, len(agent.Subagents)) for _, id := range agent.Subagents { if id == agent.ID { // The database forbids a run being its own parent; refusing it // here means the model is never offered the call in the first // place. rec.Error("runtime.subagent_self", fmt.Sprintf("%q names itself as a subagent", id)) continue } name := delegationToolName(id) if m.tools != nil { if _, taken := m.tools.Get(name); taken { rec.Error("runtime.subagent_shadowed", fmt.Sprintf( "subagent %q would be offered as %q, which is a registered tool", id, name)) continue } } sub, err := m.subagents.LoadExecutableAgent(ctx, ident, id) if err != nil || sub == nil { // Same reasoning as an unknown tool: §3 says this fails at // publish, so reaching run time means the spec changed underneath // a live agent. Degrade and say so. rec.Error("runtime.unknown_subagent", fmt.Sprintf( "%q could not be loaded and was not offered", id)) continue } out[name] = sub } return out } // delegateTools describes each subagent to the model as a tool it can call. // // The description is the subagent's own, because that is what was written for // a model to read. A parent choosing between five subagents is doing the same // job the router does when choosing between five tools, and it needs the same // quality of description to do it. func delegateTools(subs map[string]*Agent) []gateway.ToolDef { if len(subs) == 0 { return nil } defs := make([]gateway.ToolDef, 0, len(subs)) for name, sub := range subs { desc := strings.TrimSpace(sub.Description) if t := strings.TrimSpace(sub.Trigger); t != "" { desc = strings.TrimSpace(desc + " " + t) } if desc == "" { desc = fmt.Sprintf("Ask the %s agent.", sub.Name) } defs = append(defs, gateway.ToolDef{ Name: name, Description: fmt.Sprintf( "Delegate to %s and return its answer. %s "+ "Ask one self-contained question: this agent cannot see your conversation.", sub.Name, desc), InputSchema: map[string]any{ "type": "object", "properties": map[string]any{ "question": map[string]any{ "type": "string", "description": "The question to ask, complete on its own. " + "Include any names, dates or ids it needs.", }, }, "required": []string{"question"}, "additionalProperties": false, }, }) } return defs } // delegationRequest is the one argument a delegation takes. type delegationRequest struct { Question string `json:"question"` } // delegationAnswer is what the parent's model receives back. // // Structured, not prose: §4 says handlers return data and formatting is the // model's job, and a delegation is a tool call from the parent's side. RunID // travels with it so a bad answer inside a delegated branch can be found. type delegationAnswer struct { Agent string `json:"agent"` RunID string `json:"runId"` Termination string `json:"termination"` Answer string `json:"answer,omitempty"` Error string `json:"error,omitempty"` } // delegate runs one subagent and returns its answer, plus anything it wants a // person to approve. // // The subagent gets the caller's identity and the PARENT'S budget object, so // its steps, tool calls, tokens and deadline all come out of the same // allowance. It does not get the parent's confirmation token: a token // authorises one specific write that one person was shown, and handing it down // would let a different tool spend it. func (m *ModelExecutor) delegate( ctx context.Context, rec *Recorder, budget *Budget, sub *Agent, input ExecutionInput, raw json.RawMessage, depth int, ) (answer delegationAnswer, pending []*tools.Confirmation) { var req delegationRequest if err := json.Unmarshal(raw, &req); err != nil || strings.TrimSpace(req.Question) == "" { return delegationAnswer{ Agent: sub.ID, Termination: string(TerminationToolFailure), Error: "a delegation needs a question", }, nil } // The subagent writes into a buffer rather than straight to the sink: its // row cannot be inserted until this run's row exists (parent_run_id is a // foreign key). The buffer is handed to the recorder and flushed by finish // once this run has been written. buffered := *m collected := &MemorySink{} buffered.sink = collected m = &buffered res, err := m.executeRun(ctx, sub, ExecutionInput{ Identity: input.Identity, // I1 — the caller, never widened Input: req.Question, }, LimitsForTier(sub.Reasoning), delegation{ budget: budget, // §6 — shared, never fresh parentRunID: rec.RunID(), depth: depth + 1, }) // The RESULT is what matters, not the error beside it. finish returns a // non-nil error for every termination that is not Completed — including // ConfirmationPending, which is not a failure at all but a run that // stopped to ask a person something. Reading the error first and // discarding the result loses that question, and the write it was // guarding silently never happens. // Whatever happened, keep what the subagent recorded. A run that failed is // the one somebody will want to read. rec.AddChildren(collected.Runs) if res == nil { msg := "the subagent returned nothing" if err != nil { msg = err.Error() } return delegationAnswer{ Agent: sub.ID, Termination: string(TerminationToolFailure), Error: msg, }, nil } out := delegationAnswer{ Agent: sub.ID, RunID: res.RunID, Termination: string(res.Termination), Answer: res.Output, } switch { case res.Error != nil: out.Error = res.Error.Error() case err != nil && res.Termination != TerminationCompleted && res.Termination != TerminationConfirmationPending: out.Error = err.Error() } return out, res.Confirmations }