576 lines
19 KiB
Go
576 lines
19 KiB
Go
// Package tools is the registry every assistant call goes through.
|
|
//
|
|
// An agent does not reach the database. It names a tool, and this package
|
|
// decides whether that is allowed, whether the arguments make sense, who is
|
|
// asking, and what gets recorded — then runs a handler that was written by a
|
|
// person and tested.
|
|
//
|
|
// ── Why a registry rather than generated SQL ────────────────────────────────
|
|
//
|
|
// The usual reason is safety. Here there is a harder one: the fields on this
|
|
// backend do not mean what their names say, and it is measured and documented.
|
|
// `orders.deliverystatus` is an empty string on all 181 rows of tenant 1147.
|
|
// `orders.orderstatus` only ever carries pending, delivered or cancelled, so
|
|
// the six delivery stages in between never reach it. `deliveries.ridername`
|
|
// holds delivery statuses as often as names. `orders.deliverytype` is empty on
|
|
// every row in production, so filtering on it hides the entire list.
|
|
// `billedat` is local wall-clock labelled `Z`, which puts 19 of 20 bills in the
|
|
// future.
|
|
//
|
|
// A model writing SQL gets every one of those wrong, confidently, with no
|
|
// error — it reports a cancel rate from a column of empty strings and nobody
|
|
// can tell. A model calling a tool cannot, because the correction lives inside
|
|
// the handler with the measurement that justified it written beside it.
|
|
//
|
|
// So: no agent gets raw table access, and a tool that accepts a `where` string
|
|
// is a table with extra steps.
|
|
package tools
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"fmt"
|
|
"sort"
|
|
"strings"
|
|
"time"
|
|
)
|
|
|
|
// Scope separates reads from writes.
|
|
//
|
|
// Not decoration. A write tool goes through human approval before it runs
|
|
// (that is a later phase), and the registry is where the two are told apart, so
|
|
// the distinction has to be declared on the tool rather than inferred from its
|
|
// name.
|
|
type Scope string
|
|
|
|
const (
|
|
ScopeRead Scope = "read"
|
|
ScopeWrite Scope = "write"
|
|
)
|
|
|
|
// Kind is the type of one argument. Deliberately three.
|
|
//
|
|
// Enough for every tool that exists, and small enough that the validator is
|
|
// readable in one sitting. A tool wanting a nested object is a tool doing two
|
|
// things.
|
|
type Kind string
|
|
|
|
const (
|
|
KindInt Kind = "integer"
|
|
KindString Kind = "string"
|
|
KindBool Kind = "boolean"
|
|
)
|
|
|
|
// Field is one argument a tool accepts.
|
|
//
|
|
// `Max == 0` means unbounded, which is why no field here wants a negative
|
|
// range — none do, and a pointer per bound to express "unset" would cost every
|
|
// call site clarity to buy a case that has not come up.
|
|
type Field struct {
|
|
Name string
|
|
// What the model reads to decide what to put here. Written for the model,
|
|
// not for a developer: "minutes a job may sit unaccepted before it counts
|
|
// as stuck" beats "threshold".
|
|
Description string
|
|
Kind Kind
|
|
Required bool
|
|
Min, Max int
|
|
Default any
|
|
}
|
|
|
|
// Schema is a tool's argument contract.
|
|
type Schema struct{ Fields []Field }
|
|
|
|
var (
|
|
ErrUnknownTool = errors.New("no such tool")
|
|
ErrNotAllowed = errors.New("this agent may not use that tool")
|
|
ErrBadArgument = errors.New("argument is not valid")
|
|
ErrNoTenant = errors.New("caller names no tenant")
|
|
)
|
|
|
|
// Validate checks arguments and fills in defaults.
|
|
//
|
|
// Returns a NEW map rather than editing the caller's, and the returned map is
|
|
// what the handler sees. Anything not declared is dropped rather than passed
|
|
// through — a handler must never receive a key it did not ask for, or an
|
|
// argument the model invented becomes an argument the handler might one day
|
|
// start reading.
|
|
func (s Schema) Validate(args map[string]any) (map[string]any, error) {
|
|
clean := make(map[string]any, len(s.Fields))
|
|
|
|
for _, field := range s.Fields {
|
|
raw, sent := args[field.Name]
|
|
if !sent || raw == nil {
|
|
if field.Required {
|
|
return nil, fmt.Errorf("%w: %s is required", ErrBadArgument, field.Name)
|
|
}
|
|
if field.Default != nil {
|
|
clean[field.Name] = field.Default
|
|
}
|
|
continue
|
|
}
|
|
|
|
value, err := coerce(field, raw)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
clean[field.Name] = value
|
|
}
|
|
|
|
return clean, nil
|
|
}
|
|
|
|
// coerce turns what arrived into what the field declared.
|
|
//
|
|
// JSON numbers arrive as float64 whatever they looked like on the wire, so an
|
|
// integer field has to accept one and check it is whole. Reading it as an int
|
|
// directly would fail every call made over HTTP, which is all of them.
|
|
func coerce(field Field, raw any) (any, error) {
|
|
switch field.Kind {
|
|
case KindInt:
|
|
var n int
|
|
switch v := raw.(type) {
|
|
case int:
|
|
n = v
|
|
case int64:
|
|
n = int(v)
|
|
case float64:
|
|
if v != float64(int(v)) {
|
|
return nil, fmt.Errorf("%w: %s must be a whole number", ErrBadArgument, field.Name)
|
|
}
|
|
n = int(v)
|
|
default:
|
|
return nil, fmt.Errorf("%w: %s must be a number", ErrBadArgument, field.Name)
|
|
}
|
|
if n < field.Min {
|
|
return nil, fmt.Errorf("%w: %s must be at least %d", ErrBadArgument, field.Name, field.Min)
|
|
}
|
|
if field.Max > 0 && n > field.Max {
|
|
return nil, fmt.Errorf("%w: %s must be at most %d", ErrBadArgument, field.Name, field.Max)
|
|
}
|
|
return n, nil
|
|
|
|
case KindString:
|
|
text, ok := raw.(string)
|
|
if !ok {
|
|
return nil, fmt.Errorf("%w: %s must be text", ErrBadArgument, field.Name)
|
|
}
|
|
if field.Max > 0 && len(text) > field.Max {
|
|
return nil, fmt.Errorf("%w: %s is longer than %d characters", ErrBadArgument, field.Name, field.Max)
|
|
}
|
|
return text, nil
|
|
|
|
case KindBool:
|
|
flag, ok := raw.(bool)
|
|
if !ok {
|
|
return nil, fmt.Errorf("%w: %s must be true or false", ErrBadArgument, field.Name)
|
|
}
|
|
return flag, nil
|
|
}
|
|
|
|
return nil, fmt.Errorf("%w: %s has no type", ErrBadArgument, field.Name)
|
|
}
|
|
|
|
// JSONSchema renders the contract in the form a model and MCP both expect.
|
|
//
|
|
// Kept as a projection of `Schema` rather than the source of truth, so the
|
|
// validator and the description a model is given cannot drift: there is one
|
|
// declaration and this is a view of it.
|
|
func (s Schema) JSONSchema() map[string]any {
|
|
properties := map[string]any{}
|
|
required := []string{}
|
|
|
|
for _, field := range s.Fields {
|
|
property := map[string]any{
|
|
"type": string(field.Kind),
|
|
"description": field.Description,
|
|
}
|
|
if field.Kind == KindInt {
|
|
property["minimum"] = field.Min
|
|
if field.Max > 0 {
|
|
property["maximum"] = field.Max
|
|
}
|
|
}
|
|
if field.Default != nil {
|
|
property["default"] = field.Default
|
|
}
|
|
properties[field.Name] = property
|
|
if field.Required {
|
|
required = append(required, field.Name)
|
|
}
|
|
}
|
|
|
|
sort.Strings(required)
|
|
schema := map[string]any{
|
|
"type": "object",
|
|
"properties": properties,
|
|
// The model may not invent arguments. A tool that tolerated extras
|
|
// would make the schema a suggestion.
|
|
"additionalProperties": false,
|
|
}
|
|
if len(required) > 0 {
|
|
schema["required"] = required
|
|
}
|
|
return schema
|
|
}
|
|
|
|
// Requires is the scope a tool needs before it may run.
|
|
//
|
|
// Declared on the tool and enforced by the registry, not written out inside
|
|
// each handler. Seven tools carrying the same four lines is seven chances for
|
|
// the eighth to be written without them — and a tool that forgets does not
|
|
// fail, it reads whatever a zero tenant returns.
|
|
//
|
|
// The zero value is the strictest, deliberately. A tool that declares nothing
|
|
// is confined to one merchant, so forgetting is safe rather than silent.
|
|
type Requires int
|
|
|
|
const (
|
|
// RequiresTenant confines the tool to one merchant. The default.
|
|
RequiresTenant Requires = iota
|
|
// RequiresBranch additionally needs a branch in view — for reads that exist
|
|
// per outlet and have no all-branches form, like till presence.
|
|
RequiresBranch
|
|
// RequiresNothing is for tools that touch no shop data at all. There is one:
|
|
// the product help corpus, which describes how Nearle works and carries
|
|
// nothing about anybody.
|
|
RequiresNothing
|
|
)
|
|
|
|
// scopingArguments are names no tool may accept.
|
|
//
|
|
// The registry refuses to register a tool whose schema offers one, because an
|
|
// argument is something the MODEL fills in — and the model is the one part of
|
|
// this system that can be argued with. Whose data is read is decided by the
|
|
// session and never by the conversation.
|
|
var scopingArguments = map[string]bool{
|
|
"tenantid": true, "tenant_id": true, "tenant": true,
|
|
"locationid": true, "location_id": true, "store_id": true, "branch": true,
|
|
"partnerid": true, "customerid": true, "appuserid": true, "userid": true,
|
|
}
|
|
|
|
// Caller is the verified session a tool runs on behalf of.
|
|
//
|
|
// Built from `middleware.WebAuth`'s claims and never from anything the model
|
|
// said. That is the whole arrangement: the model chooses the tool and the
|
|
// arguments, and has no say at all in whose data it reads.
|
|
type Caller struct {
|
|
Userid int
|
|
Tenantid int
|
|
Locationid int
|
|
// Nearle staff, from `app_users.issuperadmin` — the only thing that reads
|
|
// across tenants. NOT a role: `app_roles` calls roleid 1 "Super admin" and
|
|
// tenant onboarding wrote 1 for every shop owner.
|
|
Superadmin bool
|
|
}
|
|
|
|
// Request is what a handler receives.
|
|
type Request struct {
|
|
Args map[string]any
|
|
Caller Caller
|
|
}
|
|
|
|
// Int reads a validated integer argument.
|
|
//
|
|
// No error return, on purpose: by the time a handler runs, the schema has
|
|
// already refused anything that is not an int in range, so a second check here
|
|
// would be unreachable code that still has to be read.
|
|
//
|
|
// ── Why it also accepts the JSON number types ───────────────────────────────
|
|
//
|
|
// `Validate` produces a real `int`, so in the ordinary path the first case is
|
|
// the only one that ever matches. The others exist because the cost of missing
|
|
// one is a SILENT ZERO, and a zero id is a plausible-looking argument rather
|
|
// than an obvious fault.
|
|
//
|
|
// That is not hypothetical. An approval card seals its arguments as JSON, and
|
|
// `41` comes back from that as the float64 `41`; the first version of the
|
|
// approval path handed those to a handler unvalidated, which looked up request
|
|
// 0 and answered "request 0 is not waiting for approval" — a sentence that
|
|
// reads like a stale card rather than a bug. The real fix was to validate the
|
|
// card's arguments, and that is done. This is the second line of defence, so
|
|
// the next path that forgets is merely redundant instead of quietly wrong.
|
|
func (r Request) Int(name string) int {
|
|
switch v := r.Args[name].(type) {
|
|
case int:
|
|
return v
|
|
case int64:
|
|
return int(v)
|
|
case float64:
|
|
// Whole numbers only. A fractional value here means something upstream
|
|
// skipped validation AND the caller sent a fraction, and truncating it
|
|
// silently would be inventing an answer.
|
|
if v == float64(int(v)) {
|
|
return int(v)
|
|
}
|
|
}
|
|
return 0
|
|
}
|
|
|
|
func (r Request) String(name string) string {
|
|
if v, ok := r.Args[name].(string); ok {
|
|
return v
|
|
}
|
|
return ""
|
|
}
|
|
|
|
func (r Request) Bool(name string) bool {
|
|
if v, ok := r.Args[name].(bool); ok {
|
|
return v
|
|
}
|
|
return false
|
|
}
|
|
|
|
// Result is what a tool answers with.
|
|
//
|
|
// Rows, never prose. The handler returns structured data and the model does the
|
|
// phrasing; a handler that wrote sentences would mean two layers formatting the
|
|
// same fact, and they drift.
|
|
type Result struct {
|
|
// Concrete typed slices, marshalled by the caller. `any` here so handlers
|
|
// stay typed rather than every one of them building maps.
|
|
Rows any
|
|
Count int
|
|
// True when there were more rows than were returned, with `Note` saying so
|
|
// in words. An empty answer and a capped answer look identical to a model,
|
|
// and it will describe both as "none".
|
|
Truncated bool
|
|
Note string
|
|
// The console route showing the same rows, so an answer can carry a link to
|
|
// what proves it. Buddy states a conclusion; this is where a person checks.
|
|
Source string
|
|
// What the answer covers — "all branches", or a named one. A tool that
|
|
// omits it lets an answer about one branch be read as an answer about all.
|
|
Scope string
|
|
}
|
|
|
|
// Agent is a caller's allow-list.
|
|
//
|
|
// The registry refuses a tool an agent has not named, rather than the model
|
|
// declining to call it. A prompt is a request; this is a rule.
|
|
type Agent struct {
|
|
Name string
|
|
Tools []string
|
|
}
|
|
|
|
func (a Agent) Allows(tool string) bool {
|
|
for _, name := range a.Tools {
|
|
if name == tool {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// Tool is one thing an agent can do.
|
|
type Tool struct {
|
|
Name string
|
|
// What the model reads to choose between tools. The most load-bearing
|
|
// string in the package: a vague one produces a model that calls the wrong
|
|
// tool and explains the wrong number confidently.
|
|
Description string
|
|
Scope Scope
|
|
// Needs is the scope required before the handler runs. Zero value is
|
|
// RequiresTenant, so a tool that says nothing is confined to one merchant.
|
|
Needs Requires
|
|
Schema Schema
|
|
Handler func(ctx context.Context, req Request) (Result, error)
|
|
|
|
// The two halves of a write, settable only through WriteTool. Unexported so
|
|
// there is no shape of Tool a caller can build that performs a write when
|
|
// the model asks for it.
|
|
propose ProposeFunc
|
|
execute ExecuteFunc
|
|
}
|
|
|
|
// Registry holds the tools and is the only way to reach one.
|
|
type Registry struct {
|
|
tools map[string]Tool
|
|
audit AuditSink
|
|
now func() time.Time
|
|
}
|
|
|
|
func New(audit AuditSink) *Registry {
|
|
if audit == nil {
|
|
audit = DiscardAudit{}
|
|
}
|
|
return &Registry{tools: map[string]Tool{}, audit: audit, now: time.Now}
|
|
}
|
|
|
|
// Register adds a tool. Duplicate names are refused rather than overwritten:
|
|
// silently replacing a tool is how a permission check disappears.
|
|
func (r *Registry) Register(t Tool) error {
|
|
if t.Name == "" {
|
|
return errors.New("a tool needs a name")
|
|
}
|
|
if t.Handler == nil {
|
|
return fmt.Errorf("tool %q has no handler", t.Name)
|
|
}
|
|
if t.Description == "" {
|
|
return fmt.Errorf("tool %q has no description; a model cannot choose it", t.Name)
|
|
}
|
|
if _, taken := r.tools[t.Name]; taken {
|
|
return fmt.Errorf("tool %q is already registered", t.Name)
|
|
}
|
|
if t.Scope == ScopeWrite && (t.propose == nil || t.execute == nil) {
|
|
return fmt.Errorf(
|
|
"tool %q is a write but was not built with WriteTool; a write must resolve to a card and execute separately",
|
|
t.Name)
|
|
}
|
|
for _, field := range t.Schema.Fields {
|
|
if scopingArguments[strings.ToLower(field.Name)] {
|
|
return fmt.Errorf(
|
|
"tool %q offers %q as an argument; whose data is read comes from the session, never from the model",
|
|
t.Name, field.Name)
|
|
}
|
|
}
|
|
r.tools[t.Name] = t
|
|
return nil
|
|
}
|
|
|
|
// Tool returns one registered tool.
|
|
//
|
|
// Exposed for the MCP door, which has to know a tool's SCOPE before offering
|
|
// it — a write is left out of the listing entirely rather than described and
|
|
// then refused. Read-only: the returned copy cannot change what is registered.
|
|
func (r *Registry) Tool(name string) (Tool, bool) {
|
|
tool, ok := r.tools[name]
|
|
return tool, ok
|
|
}
|
|
|
|
// Has reports whether a tool exists.
|
|
//
|
|
// Used to validate an agent definition at startup. A typo in a tool name is
|
|
// otherwise invisible: the agent never calls it, the model says it cannot look
|
|
// something up, and everything reports healthy.
|
|
func (r *Registry) Has(name string) bool {
|
|
_, ok := r.tools[name]
|
|
return ok
|
|
}
|
|
|
|
// Definitions describes the tools one agent may use, for a model or for MCP.
|
|
//
|
|
// Built from the agent's allow-list rather than from everything registered, so
|
|
// a model is never told about a tool it would then be refused — which reads to
|
|
// a model as a malfunction and to a person as the assistant being broken.
|
|
func (r *Registry) Definitions(agent Agent) []map[string]any {
|
|
names := make([]string, 0, len(agent.Tools))
|
|
for _, name := range agent.Tools {
|
|
if _, ok := r.tools[name]; ok {
|
|
names = append(names, name)
|
|
}
|
|
}
|
|
sort.Strings(names)
|
|
|
|
out := make([]map[string]any, 0, len(names))
|
|
for _, name := range names {
|
|
tool := r.tools[name]
|
|
out = append(out, map[string]any{
|
|
"name": tool.Name,
|
|
"description": tool.Description,
|
|
"input_schema": tool.Schema.JSONSchema(),
|
|
})
|
|
}
|
|
return out
|
|
}
|
|
|
|
// satisfies reports whether this caller may run this tool.
|
|
func satisfies(tool Tool, caller Caller) error {
|
|
switch tool.Needs {
|
|
case RequiresNothing:
|
|
return nil
|
|
case RequiresBranch:
|
|
if caller.Tenantid <= 0 {
|
|
return fmt.Errorf("%w: %s needs a shop; pick one first", ErrNoTenant, tool.Name)
|
|
}
|
|
if caller.Locationid <= 0 {
|
|
return fmt.Errorf("%w: %s covers one branch at a time; pick a branch first", ErrNoTenant, tool.Name)
|
|
}
|
|
return nil
|
|
default:
|
|
if caller.Tenantid <= 0 {
|
|
return fmt.Errorf("%w: %s needs a shop; staff must pick one first", ErrNoTenant, tool.Name)
|
|
}
|
|
return nil
|
|
}
|
|
}
|
|
|
|
// Call is the one entry point, and it does five things in this order:
|
|
// find the tool, check the agent may use it, validate the arguments, confirm
|
|
// the caller is scoped to something, and run the handler — recording exactly
|
|
// one audit row whatever happens, including every refusal.
|
|
//
|
|
// The order matters. Arguments are validated before the handler sees them so a
|
|
// handler never defends itself, and the caller is checked before the handler
|
|
// runs so a tool cannot forget to.
|
|
func (r *Registry) Call(ctx context.Context, agent Agent, name string, args map[string]any, caller Caller) (Result, error) {
|
|
started := r.now()
|
|
entry := AuditEntry{
|
|
At: started,
|
|
Agent: agent.Name,
|
|
Tool: name,
|
|
Userid: caller.Userid,
|
|
Tenantid: caller.Tenantid,
|
|
Args: args,
|
|
}
|
|
|
|
finish := func(result Result, outcome string, detail string, err error) (Result, error) {
|
|
entry.Outcome = outcome
|
|
entry.Detail = detail
|
|
entry.Rows = result.Count
|
|
entry.Took = r.now().Sub(started)
|
|
r.audit.Write(ctx, entry)
|
|
return result, err
|
|
}
|
|
|
|
tool, known := r.tools[name]
|
|
if !known {
|
|
return finish(Result{}, OutcomeRefused, "unknown tool", fmt.Errorf("%w: %s", ErrUnknownTool, name))
|
|
}
|
|
entry.Scope = string(tool.Scope)
|
|
|
|
if !agent.Allows(name) {
|
|
return finish(Result{}, OutcomeRefused, "not on the agent's allow-list", fmt.Errorf("%w: %s cannot use %s", ErrNotAllowed, agent.Name, name))
|
|
}
|
|
|
|
// A caller scoped to nothing must not be treated as a caller scoped to
|
|
// everything. Go's zero value is 0, so an unset tenant and a platform
|
|
// account look identical unless staff status is asked for separately.
|
|
if caller.Tenantid <= 0 && !caller.Superadmin {
|
|
return finish(Result{}, OutcomeRefused, "no tenant on the caller", ErrNoTenant)
|
|
}
|
|
|
|
// The scope the tool declared. Enforced here rather than inside the handler
|
|
// so a tool written next year cannot forget — and refused BEFORE the handler
|
|
// runs, so no query is built from a scope that was never established.
|
|
//
|
|
// Staff are not exempt. A platform account carries no tenant, and "every
|
|
// merchant at once" is not an answer to "what is stuck?" — so they are told
|
|
// to pick one, in the same words a branch user would get.
|
|
if err := satisfies(tool, caller); err != nil {
|
|
return finish(Result{}, OutcomeRefused, err.Error(), err)
|
|
}
|
|
|
|
clean, err := tool.Schema.Validate(args)
|
|
if err != nil {
|
|
return finish(Result{}, OutcomeRefused, err.Error(), err)
|
|
}
|
|
entry.Args = clean
|
|
|
|
// A write does not run here. It resolves into a card and stops — the only
|
|
// path to the write itself is Approve, with a person in between.
|
|
if tool.Scope == ScopeWrite {
|
|
result, err := r.proposeWrite(ctx, tool, Request{Args: clean, Caller: caller}, started)
|
|
if err != nil {
|
|
return finish(Result{}, OutcomeRefused, err.Error(), err)
|
|
}
|
|
return finish(result, "proposed", "awaiting approval", nil)
|
|
}
|
|
|
|
result, err := tool.Handler(ctx, Request{Args: clean, Caller: caller})
|
|
if err != nil {
|
|
return finish(Result{}, OutcomeFailed, err.Error(), err)
|
|
}
|
|
return finish(result, OutcomeOK, "", nil)
|
|
}
|