492 lines
18 KiB
Go
492 lines
18 KiB
Go
package tools
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"sort"
|
|
"sync"
|
|
"time"
|
|
)
|
|
|
|
// MaxToolsPerAgent is §8's cap.
|
|
//
|
|
// Not a technical limit. Past roughly twenty tools a model's choice degrades
|
|
// faster than the extra capability helps, and an agent that needs more is an
|
|
// agent that should be a parent with subagents. Enforced at resolution so a
|
|
// spec cannot quietly exceed it.
|
|
const MaxToolsPerAgent = 20
|
|
|
|
// Registry holds every tool this service can run.
|
|
//
|
|
// One registry, not one per agent: a spec selects from it by name, and a name
|
|
// that resolves to nothing fails at publish rather than at run time. That is
|
|
// the same rule the frontend registry already applies to skills.
|
|
type Registry struct {
|
|
mu sync.RWMutex
|
|
tools map[string]Tool
|
|
|
|
// confirmations is where a pending write waits for a person. Held by the
|
|
// registry rather than passed to Dispatch so that no call site can supply
|
|
// its own — a store is the thing that says a write was approved, and a
|
|
// caller able to swap it is a caller able to approve on the user's behalf.
|
|
confirmations Store
|
|
}
|
|
|
|
// NewRegistry builds an empty registry with an in-process confirmation store.
|
|
//
|
|
// Fine for one instance and for tests. A deployment with more than one replica
|
|
// must pass a shared store — see NewRegistryWithStore and the note on
|
|
// MemoryStore.
|
|
func NewRegistry() *Registry {
|
|
return NewRegistryWithStore(NewMemoryStore())
|
|
}
|
|
|
|
// NewRegistryWithStore builds a registry over a specific confirmation store.
|
|
func NewRegistryWithStore(s Store) *Registry {
|
|
if s == nil {
|
|
s = NewMemoryStore()
|
|
}
|
|
return &Registry{tools: make(map[string]Tool), confirmations: s}
|
|
}
|
|
|
|
// Register adds a tool, or returns why it cannot be added.
|
|
//
|
|
// Called at startup, so a malformed tool is a boot failure rather than a
|
|
// surprise on the first run that reaches it.
|
|
func (r *Registry) Register(t Tool) error {
|
|
if t.Name == "" {
|
|
return fmt.Errorf("tools: a tool needs a name")
|
|
}
|
|
if t.Handler == nil {
|
|
return fmt.Errorf("tools: %s has no handler", t.Name)
|
|
}
|
|
if t.Description == "" {
|
|
// The description is what the model reads instead of documentation. A
|
|
// tool without one is a tool that will be called wrongly.
|
|
return fmt.Errorf("tools: %s has no description", t.Name)
|
|
}
|
|
switch t.Effect {
|
|
case EffectRead:
|
|
case EffectWrite:
|
|
// I4, and the reason RequiresConfirmation is not left to the author:
|
|
// a write that forgot to set it would run unconfirmed forever, and
|
|
// nothing downstream could tell it apart from a deliberate choice.
|
|
t.RequiresConfirmation = true
|
|
default:
|
|
return fmt.Errorf("tools: %s declares effect %q, want read or write", t.Name, t.Effect)
|
|
}
|
|
// §8 step 3. A write with no renderer could only ever be confirmed by
|
|
// showing someone its raw arguments, and nobody can meaningfully approve
|
|
// a pair of uuids. Refused at registration, so it is a boot failure rather
|
|
// than a bad dialog discovered in production.
|
|
if t.RequiresConfirmation && t.Confirm == nil {
|
|
return fmt.Errorf(
|
|
"tools: %s needs confirmation but has no Confirm renderer; "+
|
|
"a write must be able to say in plain language what it will do", t.Name)
|
|
}
|
|
if t.Effect == EffectRead && t.Confirm != nil && !t.RequiresConfirmation {
|
|
// A renderer that will never run is a renderer nobody maintains, and
|
|
// the day the tool becomes a write it will be wrong.
|
|
return fmt.Errorf("tools: %s has a Confirm renderer but never asks for confirmation", t.Name)
|
|
}
|
|
if t.MaxResultBytes <= 0 {
|
|
t.MaxResultBytes = DefaultMaxResultBytes
|
|
}
|
|
|
|
r.mu.Lock()
|
|
defer r.mu.Unlock()
|
|
if _, exists := r.tools[t.Name]; exists {
|
|
return fmt.Errorf("tools: %s is already registered", t.Name)
|
|
}
|
|
r.tools[t.Name] = t
|
|
return nil
|
|
}
|
|
|
|
// MustRegister adds a tool or panics. For startup wiring, where the alternative
|
|
// to a panic is a service that boots without a capability it claims to have.
|
|
func (r *Registry) MustRegister(t Tool) {
|
|
if err := r.Register(t); err != nil {
|
|
panic(err)
|
|
}
|
|
}
|
|
|
|
// Get returns a tool by name.
|
|
func (r *Registry) Get(name string) (Tool, bool) {
|
|
r.mu.RLock()
|
|
defer r.mu.RUnlock()
|
|
t, ok := r.tools[name]
|
|
return t, ok
|
|
}
|
|
|
|
// Names lists every registered tool, sorted.
|
|
//
|
|
// Sorted because this list is rendered into the prompt, and the prompt is a
|
|
// cache prefix: an unstable order would invalidate the cache on every request
|
|
// for no reason at all.
|
|
func (r *Registry) Names() []string {
|
|
r.mu.RLock()
|
|
defer r.mu.RUnlock()
|
|
|
|
names := make([]string, 0, len(r.tools))
|
|
for name := range r.tools {
|
|
names = append(names, name)
|
|
}
|
|
sort.Strings(names)
|
|
return names
|
|
}
|
|
|
|
// Resolve turns a spec's tool names into tools.
|
|
//
|
|
// Unknown names are returned rather than dropped: §3 says a spec naming a tool
|
|
// that does not exist fails validation at publish, and this is the function
|
|
// that lets publish say which one. At run time the caller decides — dropping a
|
|
// missing tool is better than refusing the run, but only if someone is told.
|
|
func (r *Registry) Resolve(names []string) (resolved []Tool, unknown []string, err error) {
|
|
if len(names) > MaxToolsPerAgent {
|
|
return nil, nil, fmt.Errorf(
|
|
"tools: %d tools requested, the cap is %d — split this agent into a parent with subagents",
|
|
len(names), MaxToolsPerAgent)
|
|
}
|
|
|
|
r.mu.RLock()
|
|
defer r.mu.RUnlock()
|
|
|
|
seen := make(map[string]bool, len(names))
|
|
for _, name := range names {
|
|
if seen[name] {
|
|
continue
|
|
}
|
|
seen[name] = true
|
|
|
|
t, ok := r.tools[name]
|
|
if !ok {
|
|
unknown = append(unknown, name)
|
|
continue
|
|
}
|
|
resolved = append(resolved, t)
|
|
}
|
|
return resolved, unknown, nil
|
|
}
|
|
|
|
// Dispatch runs one tool call and returns its result.
|
|
//
|
|
// This is the one place a tool is invoked, and it holds the two gates a handler
|
|
// must not be trusted to hold itself:
|
|
//
|
|
// 1. **The confirmation gate.** A write with no resolved confirmation never
|
|
// reaches its handler. The model cannot argue its way past this because the
|
|
// model is not consulted — the check is on the tool's declared effect and
|
|
// the token in the context, both of which are set outside the conversation.
|
|
// 2. **The truncation cap.** Applied to what the handler returned, with the
|
|
// flag set. A handler that forgets to bound its own output cannot flood the
|
|
// next turn.
|
|
//
|
|
// A panicking handler is contained here too. A tool is the least trusted code
|
|
// in the runtime — it is where new integrations land — and one bad handler
|
|
// must cost its own call, not the run.
|
|
func (r *Registry) Dispatch(ctx context.Context, tc Context, name string, inputs json.RawMessage) (res Result) {
|
|
t, ok := r.Get(name)
|
|
if !ok {
|
|
return Failf(CodeUnavailable, "there is no tool called %q", name)
|
|
}
|
|
|
|
// The gate. Everything past this line has either no effect or an approval.
|
|
if t.RequiresConfirmation {
|
|
if pending, blocked := r.gate(ctx, tc, t, inputs); blocked {
|
|
return pending
|
|
}
|
|
}
|
|
|
|
defer func() {
|
|
if p := recover(); p != nil {
|
|
res = Failf(CodeFailed, "%s failed unexpectedly", name)
|
|
}
|
|
}()
|
|
|
|
res = t.Handler(ctx, tc, inputs)
|
|
return truncate(res, t.MaxResultBytes)
|
|
}
|
|
|
|
// gate decides whether a confirmed tool may run now.
|
|
//
|
|
// Two outcomes, and the interesting thing is how few:
|
|
//
|
|
// - **A token bound to exactly this call.** Consumed, and the handler runs.
|
|
// - **Anything else.** Render what the call would do, issue a token bound to
|
|
// it, and return that as a pending Result. Nothing is written. The run ends
|
|
// at ConfirmationPending and a person is asked.
|
|
//
|
|
// "Anything else" deliberately includes a token that does not match: expired,
|
|
// already spent, issued to somebody else, or issued for a different worker on a
|
|
// different day. None of those is an error to report — they all mean the same
|
|
// thing, which is that nobody has approved THIS call, and the honest response
|
|
// to that is to describe it and ask. Returning a failure instead would leave
|
|
// the model holding an error it cannot act on, in a run whose token is fixed
|
|
// for its whole duration.
|
|
//
|
|
// A non-matching token is NOT consumed. An earlier version spent it on any
|
|
// attempt, reasoning that a mismatch was a replay or a guess. It is neither —
|
|
// the token is 24 random bytes, so guessing is not on the table — and the cost
|
|
// was real: a model that makes two write calls in one turn would destroy a
|
|
// perfectly good approval with whichever call happened to be dispatched first.
|
|
//
|
|
// Returns (result, true) when the call must not proceed.
|
|
func (r *Registry) gate(ctx context.Context, tc Context, t Tool, inputs json.RawMessage) (Result, bool) {
|
|
b := bind(tc, t.Name, inputs)
|
|
|
|
if tc.Confirmation != "" && r.confirmations != nil &&
|
|
r.confirmations.Resolve(ctx, tc.Confirmation, b) {
|
|
return Result{}, false
|
|
}
|
|
|
|
// Rendered under the same authorization as the write. A renderer that
|
|
// resolved a name the caller may not see would have leaked it in the act
|
|
// of asking permission not to.
|
|
confirmation, denied := r.render(ctx, tc, t, inputs)
|
|
if denied != nil {
|
|
return *denied, true
|
|
}
|
|
|
|
token, err := newToken()
|
|
if err != nil {
|
|
return Failf(CodeUnavailable, "%s could not be prepared for confirmation", t.Name), true
|
|
}
|
|
confirmation.Token = token
|
|
confirmation.Tool = t.Name
|
|
if confirmation.ExpiresAt.IsZero() {
|
|
confirmation.ExpiresAt = time.Now().Add(ConfirmationTTL)
|
|
}
|
|
|
|
if r.confirmations == nil {
|
|
return Failf(CodeUnavailable,
|
|
"%s cannot run: there is nowhere to record a confirmation", t.Name), true
|
|
}
|
|
if err := r.confirmations.Issue(ctx, b, confirmation); err != nil {
|
|
// Failing closed. A confirmation that was shown but not recorded is one
|
|
// that can never be honoured, and asking a person a question whose
|
|
// answer will be discarded is worse than saying the tool is unavailable.
|
|
return Failf(CodeUnavailable, "%s could not be prepared for confirmation", t.Name), true
|
|
}
|
|
|
|
return Result{Confirmation: confirmation}, true
|
|
}
|
|
|
|
// render runs a tool's Confirm renderer, containing its failures.
|
|
//
|
|
// A renderer is author-written code that runs before any approval exists, so it
|
|
// gets the same panic containment a handler does — and a renderer that returns
|
|
// nothing at all is treated as a refusal rather than as an empty dialog.
|
|
func (r *Registry) render(ctx context.Context, tc Context, t Tool, inputs json.RawMessage) (c *Confirmation, denied *Result) {
|
|
defer func() {
|
|
if p := recover(); p != nil {
|
|
failed := Failf(CodeFailed, "%s could not describe what it would do", t.Name)
|
|
c, denied = nil, &failed
|
|
}
|
|
}()
|
|
|
|
if t.Confirm == nil {
|
|
// Register refuses this, so reaching it means a Tool was built by hand
|
|
// and bypassed registration. Fail closed rather than trust it.
|
|
failed := Failf(CodeUnavailable, "%s cannot describe what it would do", t.Name)
|
|
return nil, &failed
|
|
}
|
|
|
|
c, denied = t.Confirm(ctx, tc, inputs)
|
|
if denied != nil {
|
|
return nil, denied
|
|
}
|
|
if c == nil {
|
|
failed := Failf(CodeUnavailable, "%s cannot describe what it would do", t.Name)
|
|
return nil, &failed
|
|
}
|
|
return c, nil
|
|
}
|
|
|
|
// truncate bounds a result, marking it when it had to.
|
|
//
|
|
// The data is replaced wholesale rather than cut mid-encoding: a JSON document
|
|
// sliced at a byte offset is not a JSON document, and a model handed one will
|
|
// either fail to parse it or — worse — parse the fragment and reason about it
|
|
// as if it were the whole.
|
|
func truncate(res Result, maxBytes int) Result {
|
|
if res.Data == nil || maxBytes <= 0 {
|
|
return res
|
|
}
|
|
encoded, err := json.Marshal(res.Data)
|
|
if err != nil {
|
|
return Failf(CodeFailed, "the result could not be encoded")
|
|
}
|
|
if len(encoded) <= maxBytes {
|
|
return res
|
|
}
|
|
return Result{
|
|
Data: map[string]any{
|
|
"note": fmt.Sprintf(
|
|
"This result was %d bytes, over the %d-byte limit, and has been withheld rather than cut. "+
|
|
"Ask again with a narrower filter, a period, or a smaller limit.",
|
|
len(encoded), maxBytes),
|
|
},
|
|
Truncated: true,
|
|
}
|
|
}
|
|
|
|
/* ── Redeeming an approval ──────────────────────────────────────────────── */
|
|
|
|
// Redeemed is what an approved write did.
|
|
type Redeemed struct {
|
|
// Tool is the tool that ran, so the caller can record and report it.
|
|
Tool string
|
|
|
|
// Inputs are the arguments a person approved, replayed verbatim.
|
|
Inputs json.RawMessage
|
|
|
|
Result Result
|
|
}
|
|
|
|
// DispatchApproved performs the call a token authorised.
|
|
//
|
|
// The other half of the confirmation flow, and the one that makes it reliable.
|
|
//
|
|
// The original design had only one path: the model, on a resumed turn, makes
|
|
// the same tool call again, and the token is matched against it. That works
|
|
// when it works and fails silently when it does not — a model is not
|
|
// deterministic, and asked a second time it may reasonably seek clarification
|
|
// instead of repeating itself. Observed: a person clicked Approve, the model
|
|
// asked a follow-up question, the token was never presented, and nothing
|
|
// happened. No error. No write. Nothing to tell the user why.
|
|
//
|
|
// So this path does not ask the model anything. It takes the token, gets back
|
|
// the exact call that was described to the person, and performs THAT. What
|
|
// somebody approved and what happens are the same thing by construction rather
|
|
// than by the model's cooperation.
|
|
//
|
|
// The gate is not bypassed — this IS the gate. Redeem checks the caller, the
|
|
// tenant and the expiry and consumes the token atomically, so an approval still
|
|
// buys exactly one write and only for the person who was asked.
|
|
//
|
|
// Returns ok=false when the token authorises nothing: unknown, expired, spent,
|
|
// or somebody else's. Indistinguishably, as everywhere else.
|
|
func (r *Registry) DispatchApproved(ctx context.Context, tc Context, token string) (out Redeemed, ok bool) {
|
|
if r.confirmations == nil || token == "" {
|
|
return Redeemed{}, false
|
|
}
|
|
|
|
approved, ok := r.confirmations.Redeem(ctx, token, Principal{
|
|
UserID: tc.Principal.UserID,
|
|
OrgID: tc.Principal.OrgID,
|
|
})
|
|
if !ok {
|
|
return Redeemed{}, false
|
|
}
|
|
|
|
t, exists := r.Get(approved.Tool)
|
|
if !exists {
|
|
// The tool was withdrawn between the asking and the answering. The
|
|
// token is spent either way — it has been consumed by Redeem — which is
|
|
// correct: re-offering it would let a person approve something this
|
|
// service can no longer describe.
|
|
return Redeemed{
|
|
Tool: approved.Tool,
|
|
Inputs: approved.Inputs,
|
|
Result: Failf(CodeUnavailable, "%s is no longer available", approved.Tool),
|
|
}, true
|
|
}
|
|
|
|
res := func() (res Result) {
|
|
defer func() {
|
|
if p := recover(); p != nil {
|
|
res = Failf(CodeFailed, "%s failed unexpectedly", t.Name)
|
|
}
|
|
}()
|
|
return t.Handler(ctx, tc, approved.Inputs)
|
|
}()
|
|
|
|
return Redeemed{
|
|
Tool: approved.Tool,
|
|
Inputs: approved.Inputs,
|
|
Result: truncate(res, t.MaxResultBytes),
|
|
}, true
|
|
}
|
|
|
|
// ToolInfo is what a tool looks like to somebody choosing one, and — since the
|
|
// MCP surface — to a client that must publish the schema before calling it.
|
|
//
|
|
// Effect is present because it is the one thing an author must understand: a
|
|
// write tool means their agent can propose changes, which a person will then be
|
|
// asked to approve.
|
|
//
|
|
// InputSchema used to be deliberately absent, on the reasoning that an author
|
|
// picks a capability and the schema is the model's business. That is still true
|
|
// of the authoring UI, which simply ignores the field. It stopped being true of
|
|
// the catalogue as a whole once a second consumer appeared: MCP's tools/list
|
|
// must publish a JSON Schema per tool, and the only alternative to carrying
|
|
// this one is maintaining a copy. A copy is a second definition of the same
|
|
// thing, and the way it fails is silent — a field renamed on the handler side
|
|
// leaves a schema that still validates and no longer matches, so the call
|
|
// succeeds and the argument is quietly ignored.
|
|
//
|
|
// Both new fields carry `omitempty`, but be clear about what that does and does
|
|
// not buy: every currently registered tool declares a schema and a byte cap, so
|
|
// GET /api/v1/tools genuinely does get larger — roughly 700 bytes per tool. The
|
|
// change is additive rather than invisible. It is compatible because JSON
|
|
// consumers ignore keys they do not read, and the one consumer that exists (the
|
|
// agent editor's tool picker) reads name, description and effect; `omitempty`
|
|
// covers the remaining case of a tool registered with neither field.
|
|
type ToolInfo struct {
|
|
Name string `json:"name"`
|
|
Description string `json:"description"`
|
|
Effect string `json:"effect"`
|
|
RequiresConfirmation bool `json:"requiresConfirmation"`
|
|
|
|
// InputSchema is the tool's JSON Schema, exactly as registered.
|
|
InputSchema map[string]any `json:"inputSchema,omitempty"`
|
|
|
|
// MaxResultBytes is the cap Dispatch truncates at, so a caller can state
|
|
// the limit rather than discover it by hitting it.
|
|
MaxResultBytes int `json:"maxResultBytes,omitempty"`
|
|
}
|
|
|
|
// Catalogue lists every registered tool, sorted, as choosable metadata.
|
|
//
|
|
// This exists so an agent author can be shown the real tool set rather than a
|
|
// hand-maintained copy of it in the frontend. A second list would drift, and
|
|
// the failure would be silent: an author picks a tool that no longer exists and
|
|
// gets an agent that quietly cannot do the thing they picked.
|
|
func (r *Registry) Catalogue() []ToolInfo {
|
|
r.mu.RLock()
|
|
defer r.mu.RUnlock()
|
|
|
|
out := make([]ToolInfo, 0, len(r.tools))
|
|
for _, t := range r.tools {
|
|
out = append(out, ToolInfo{
|
|
Name: t.Name,
|
|
Description: t.Description,
|
|
Effect: string(t.Effect),
|
|
RequiresConfirmation: t.RequiresConfirmation,
|
|
InputSchema: t.InputSchema,
|
|
MaxResultBytes: t.MaxResultBytes,
|
|
})
|
|
}
|
|
sort.Slice(out, func(i, j int) bool { return out[i].Name < out[j].Name })
|
|
return out
|
|
}
|
|
|
|
// Known reports whether every name is a registered tool, returning the ones
|
|
// that are not.
|
|
//
|
|
// §3 requires an unknown tool name to fail validation at PUBLISH. Without this
|
|
// the runtime records and drops the name, so a typo becomes an agent that is
|
|
// silently missing a capability its author believes it has.
|
|
func (r *Registry) Known(names []string) (unknown []string) {
|
|
r.mu.RLock()
|
|
defer r.mu.RUnlock()
|
|
|
|
for _, n := range names {
|
|
if _, ok := r.tools[n]; !ok {
|
|
unknown = append(unknown, n)
|
|
}
|
|
}
|
|
return unknown
|
|
}
|