package tools import ( "context" "encoding/json" "fmt" "sort" "sync" "time" ) // MaxToolsPerAgent is §8's cap. // // Not a technical limit. Past roughly twenty tools a model's choice degrades // faster than the extra capability helps, and an agent that needs more is an // agent that should be a parent with subagents. Enforced at resolution so a // spec cannot quietly exceed it. const MaxToolsPerAgent = 20 // Registry holds every tool this service can run. // // One registry, not one per agent: a spec selects from it by name, and a name // that resolves to nothing fails at publish rather than at run time. That is // the same rule the frontend registry already applies to skills. type Registry struct { mu sync.RWMutex tools map[string]Tool // confirmations is where a pending write waits for a person. Held by the // registry rather than passed to Dispatch so that no call site can supply // its own — a store is the thing that says a write was approved, and a // caller able to swap it is a caller able to approve on the user's behalf. confirmations Store } // NewRegistry builds an empty registry with an in-process confirmation store. // // Fine for one instance and for tests. A deployment with more than one replica // must pass a shared store — see NewRegistryWithStore and the note on // MemoryStore. func NewRegistry() *Registry { return NewRegistryWithStore(NewMemoryStore()) } // NewRegistryWithStore builds a registry over a specific confirmation store. func NewRegistryWithStore(s Store) *Registry { if s == nil { s = NewMemoryStore() } return &Registry{tools: make(map[string]Tool), confirmations: s} } // Register adds a tool, or returns why it cannot be added. // // Called at startup, so a malformed tool is a boot failure rather than a // surprise on the first run that reaches it. func (r *Registry) Register(t Tool) error { if t.Name == "" { return fmt.Errorf("tools: a tool needs a name") } if t.Handler == nil { return fmt.Errorf("tools: %s has no handler", t.Name) } if t.Description == "" { // The description is what the model reads instead of documentation. A // tool without one is a tool that will be called wrongly. return fmt.Errorf("tools: %s has no description", t.Name) } switch t.Effect { case EffectRead: case EffectWrite: // I4, and the reason RequiresConfirmation is not left to the author: // a write that forgot to set it would run unconfirmed forever, and // nothing downstream could tell it apart from a deliberate choice. t.RequiresConfirmation = true default: return fmt.Errorf("tools: %s declares effect %q, want read or write", t.Name, t.Effect) } // §8 step 3. A write with no renderer could only ever be confirmed by // showing someone its raw arguments, and nobody can meaningfully approve // a pair of uuids. Refused at registration, so it is a boot failure rather // than a bad dialog discovered in production. if t.RequiresConfirmation && t.Confirm == nil { return fmt.Errorf( "tools: %s needs confirmation but has no Confirm renderer; "+ "a write must be able to say in plain language what it will do", t.Name) } if t.Effect == EffectRead && t.Confirm != nil && !t.RequiresConfirmation { // A renderer that will never run is a renderer nobody maintains, and // the day the tool becomes a write it will be wrong. return fmt.Errorf("tools: %s has a Confirm renderer but never asks for confirmation", t.Name) } if t.MaxResultBytes <= 0 { t.MaxResultBytes = DefaultMaxResultBytes } r.mu.Lock() defer r.mu.Unlock() if _, exists := r.tools[t.Name]; exists { return fmt.Errorf("tools: %s is already registered", t.Name) } r.tools[t.Name] = t return nil } // MustRegister adds a tool or panics. For startup wiring, where the alternative // to a panic is a service that boots without a capability it claims to have. func (r *Registry) MustRegister(t Tool) { if err := r.Register(t); err != nil { panic(err) } } // Get returns a tool by name. func (r *Registry) Get(name string) (Tool, bool) { r.mu.RLock() defer r.mu.RUnlock() t, ok := r.tools[name] return t, ok } // Names lists every registered tool, sorted. // // Sorted because this list is rendered into the prompt, and the prompt is a // cache prefix: an unstable order would invalidate the cache on every request // for no reason at all. func (r *Registry) Names() []string { r.mu.RLock() defer r.mu.RUnlock() names := make([]string, 0, len(r.tools)) for name := range r.tools { names = append(names, name) } sort.Strings(names) return names } // Resolve turns a spec's tool names into tools. // // Unknown names are returned rather than dropped: §3 says a spec naming a tool // that does not exist fails validation at publish, and this is the function // that lets publish say which one. At run time the caller decides — dropping a // missing tool is better than refusing the run, but only if someone is told. func (r *Registry) Resolve(names []string) (resolved []Tool, unknown []string, err error) { if len(names) > MaxToolsPerAgent { return nil, nil, fmt.Errorf( "tools: %d tools requested, the cap is %d — split this agent into a parent with subagents", len(names), MaxToolsPerAgent) } r.mu.RLock() defer r.mu.RUnlock() seen := make(map[string]bool, len(names)) for _, name := range names { if seen[name] { continue } seen[name] = true t, ok := r.tools[name] if !ok { unknown = append(unknown, name) continue } resolved = append(resolved, t) } return resolved, unknown, nil } // Dispatch runs one tool call and returns its result. // // This is the one place a tool is invoked, and it holds the two gates a handler // must not be trusted to hold itself: // // 1. **The confirmation gate.** A write with no resolved confirmation never // reaches its handler. The model cannot argue its way past this because the // model is not consulted — the check is on the tool's declared effect and // the token in the context, both of which are set outside the conversation. // 2. **The truncation cap.** Applied to what the handler returned, with the // flag set. A handler that forgets to bound its own output cannot flood the // next turn. // // A panicking handler is contained here too. A tool is the least trusted code // in the runtime — it is where new integrations land — and one bad handler // must cost its own call, not the run. func (r *Registry) Dispatch(ctx context.Context, tc Context, name string, inputs json.RawMessage) (res Result) { t, ok := r.Get(name) if !ok { return Failf(CodeUnavailable, "there is no tool called %q", name) } // The gate. Everything past this line has either no effect or an approval. if t.RequiresConfirmation { if pending, blocked := r.gate(ctx, tc, t, inputs); blocked { return pending } } defer func() { if p := recover(); p != nil { res = Failf(CodeFailed, "%s failed unexpectedly", name) } }() res = t.Handler(ctx, tc, inputs) return truncate(res, t.MaxResultBytes) } // gate decides whether a confirmed tool may run now. // // Two outcomes, and the interesting thing is how few: // // - **A token bound to exactly this call.** Consumed, and the handler runs. // - **Anything else.** Render what the call would do, issue a token bound to // it, and return that as a pending Result. Nothing is written. The run ends // at ConfirmationPending and a person is asked. // // "Anything else" deliberately includes a token that does not match: expired, // already spent, issued to somebody else, or issued for a different worker on a // different day. None of those is an error to report — they all mean the same // thing, which is that nobody has approved THIS call, and the honest response // to that is to describe it and ask. Returning a failure instead would leave // the model holding an error it cannot act on, in a run whose token is fixed // for its whole duration. // // A non-matching token is NOT consumed. An earlier version spent it on any // attempt, reasoning that a mismatch was a replay or a guess. It is neither — // the token is 24 random bytes, so guessing is not on the table — and the cost // was real: a model that makes two write calls in one turn would destroy a // perfectly good approval with whichever call happened to be dispatched first. // // Returns (result, true) when the call must not proceed. func (r *Registry) gate(ctx context.Context, tc Context, t Tool, inputs json.RawMessage) (Result, bool) { b := bind(tc, t.Name, inputs) if tc.Confirmation != "" && r.confirmations != nil && r.confirmations.Resolve(ctx, tc.Confirmation, b) { return Result{}, false } // Rendered under the same authorization as the write. A renderer that // resolved a name the caller may not see would have leaked it in the act // of asking permission not to. confirmation, denied := r.render(ctx, tc, t, inputs) if denied != nil { return *denied, true } token, err := newToken() if err != nil { return Failf(CodeUnavailable, "%s could not be prepared for confirmation", t.Name), true } confirmation.Token = token confirmation.Tool = t.Name if confirmation.ExpiresAt.IsZero() { confirmation.ExpiresAt = time.Now().Add(ConfirmationTTL) } if r.confirmations == nil { return Failf(CodeUnavailable, "%s cannot run: there is nowhere to record a confirmation", t.Name), true } if err := r.confirmations.Issue(ctx, b, confirmation); err != nil { // Failing closed. A confirmation that was shown but not recorded is one // that can never be honoured, and asking a person a question whose // answer will be discarded is worse than saying the tool is unavailable. return Failf(CodeUnavailable, "%s could not be prepared for confirmation", t.Name), true } return Result{Confirmation: confirmation}, true } // render runs a tool's Confirm renderer, containing its failures. // // A renderer is author-written code that runs before any approval exists, so it // gets the same panic containment a handler does — and a renderer that returns // nothing at all is treated as a refusal rather than as an empty dialog. func (r *Registry) render(ctx context.Context, tc Context, t Tool, inputs json.RawMessage) (c *Confirmation, denied *Result) { defer func() { if p := recover(); p != nil { failed := Failf(CodeFailed, "%s could not describe what it would do", t.Name) c, denied = nil, &failed } }() if t.Confirm == nil { // Register refuses this, so reaching it means a Tool was built by hand // and bypassed registration. Fail closed rather than trust it. failed := Failf(CodeUnavailable, "%s cannot describe what it would do", t.Name) return nil, &failed } c, denied = t.Confirm(ctx, tc, inputs) if denied != nil { return nil, denied } if c == nil { failed := Failf(CodeUnavailable, "%s cannot describe what it would do", t.Name) return nil, &failed } return c, nil } // truncate bounds a result, marking it when it had to. // // The data is replaced wholesale rather than cut mid-encoding: a JSON document // sliced at a byte offset is not a JSON document, and a model handed one will // either fail to parse it or — worse — parse the fragment and reason about it // as if it were the whole. func truncate(res Result, maxBytes int) Result { if res.Data == nil || maxBytes <= 0 { return res } encoded, err := json.Marshal(res.Data) if err != nil { return Failf(CodeFailed, "the result could not be encoded") } if len(encoded) <= maxBytes { return res } return Result{ Data: map[string]any{ "note": fmt.Sprintf( "This result was %d bytes, over the %d-byte limit, and has been withheld rather than cut. "+ "Ask again with a narrower filter, a period, or a smaller limit.", len(encoded), maxBytes), }, Truncated: true, } } /* ── Redeeming an approval ──────────────────────────────────────────────── */ // Redeemed is what an approved write did. type Redeemed struct { // Tool is the tool that ran, so the caller can record and report it. Tool string // Inputs are the arguments a person approved, replayed verbatim. Inputs json.RawMessage Result Result } // DispatchApproved performs the call a token authorised. // // The other half of the confirmation flow, and the one that makes it reliable. // // The original design had only one path: the model, on a resumed turn, makes // the same tool call again, and the token is matched against it. That works // when it works and fails silently when it does not — a model is not // deterministic, and asked a second time it may reasonably seek clarification // instead of repeating itself. Observed: a person clicked Approve, the model // asked a follow-up question, the token was never presented, and nothing // happened. No error. No write. Nothing to tell the user why. // // So this path does not ask the model anything. It takes the token, gets back // the exact call that was described to the person, and performs THAT. What // somebody approved and what happens are the same thing by construction rather // than by the model's cooperation. // // The gate is not bypassed — this IS the gate. Redeem checks the caller, the // tenant and the expiry and consumes the token atomically, so an approval still // buys exactly one write and only for the person who was asked. // // Returns ok=false when the token authorises nothing: unknown, expired, spent, // or somebody else's. Indistinguishably, as everywhere else. func (r *Registry) DispatchApproved(ctx context.Context, tc Context, token string) (out Redeemed, ok bool) { if r.confirmations == nil || token == "" { return Redeemed{}, false } approved, ok := r.confirmations.Redeem(ctx, token, Principal{ UserID: tc.Principal.UserID, OrgID: tc.Principal.OrgID, }) if !ok { return Redeemed{}, false } t, exists := r.Get(approved.Tool) if !exists { // The tool was withdrawn between the asking and the answering. The // token is spent either way — it has been consumed by Redeem — which is // correct: re-offering it would let a person approve something this // service can no longer describe. return Redeemed{ Tool: approved.Tool, Inputs: approved.Inputs, Result: Failf(CodeUnavailable, "%s is no longer available", approved.Tool), }, true } res := func() (res Result) { defer func() { if p := recover(); p != nil { res = Failf(CodeFailed, "%s failed unexpectedly", t.Name) } }() return t.Handler(ctx, tc, approved.Inputs) }() return Redeemed{ Tool: approved.Tool, Inputs: approved.Inputs, Result: truncate(res, t.MaxResultBytes), }, true } // ToolInfo is what a tool looks like to somebody choosing one, rather than to // the model calling it. // // The InputSchema is deliberately absent: an author picks a capability, and the // schema is the model's business. Effect is present because it is the one thing // an author must understand — a write tool means their agent can propose // changes, which a person will then be asked to approve. type ToolInfo struct { Name string `json:"name"` Description string `json:"description"` Effect string `json:"effect"` RequiresConfirmation bool `json:"requiresConfirmation"` } // Catalogue lists every registered tool, sorted, as choosable metadata. // // This exists so an agent author can be shown the real tool set rather than a // hand-maintained copy of it in the frontend. A second list would drift, and // the failure would be silent: an author picks a tool that no longer exists and // gets an agent that quietly cannot do the thing they picked. func (r *Registry) Catalogue() []ToolInfo { r.mu.RLock() defer r.mu.RUnlock() out := make([]ToolInfo, 0, len(r.tools)) for _, t := range r.tools { out = append(out, ToolInfo{ Name: t.Name, Description: t.Description, Effect: string(t.Effect), RequiresConfirmation: t.RequiresConfirmation, }) } sort.Slice(out, func(i, j int) bool { return out[i].Name < out[j].Name }) return out } // Known reports whether every name is a registered tool, returning the ones // that are not. // // §3 requires an unknown tool name to fail validation at PUBLISH. Without this // the runtime records and drops the name, so a typo becomes an agent that is // silently missing a capability its author believes it has. func (r *Registry) Known(names []string) (unknown []string) { r.mu.RLock() defer r.mu.RUnlock() for _, n := range names { if _, ok := r.tools[n]; !ok { unknown = append(unknown, n) } } return unknown }