Phase 1 of Nearle Buddy: an agent names a tool, and the registry decides whether that is allowed, whether the arguments make sense, who is asking, and what gets recorded — then runs a handler a person wrote and tested. No agent gets raw table access. The usual argument for tools over generated SQL is safety; here there is a harder one. The fields on this backend do not mean what their names say, and it is measured: orders.deliverystatus is an empty string on all 181 rows of tenant 1147, orders.orderstatus never carries the six middle delivery stages, deliveries.ridername holds statuses as often as names, deliverytype is empty on every row in production. A model writing SQL gets each of those wrong with no error — it reports a cancel rate from a column of empty strings and nobody can tell. A model calling a tool cannot, because the correction lives in the handler beside the measurement that justified it. Call does five things in order: find the tool, check the agent's allow-list, validate arguments, confirm the caller is scoped to something, run the handler — writing exactly one audit row whatever happens, refusals included. A trail of successes answers "did anything try to read another tenant?" with silence, which reads the same as no. The model has no say in whose data is read. stuck_orders has no tenantid field on its schema — absent, not rejected — and the tenant comes from the session claims added in the previous commit. Arguments the tool did not declare are dropped rather than passed on, so a model sending a `where` clause gets it discarded. stuck_orders: deliveries a rider was given and has not accepted, ten minutes for a look, twenty-five for somebody now. Derived from assigntime and orderstatus, so it does not depend on anyone having been watching. Carries the wait in minutes, what to do, where to check it, and what it covered. A capped answer says so — an empty result and a truncated one look identical to a model and it will call both "none". The audit sink writes to the log for now; a database sink is phase 8. Nothing calls the registry yet: the loop and the model gateway are phase 2. 37 tests. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
328 lines
11 KiB
Go
328 lines
11 KiB
Go
package tools
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"testing"
|
|
"time"
|
|
)
|
|
|
|
// The registry is the only way to reach a tool, so these are the checks that
|
|
// stand between a model and the data. Each one is a thing the model could ask
|
|
// for and must not get.
|
|
|
|
func okTool(name string) Tool {
|
|
return Tool{
|
|
Name: name,
|
|
Description: "a tool, for testing",
|
|
Scope: ScopeRead,
|
|
Schema: Schema{Fields: []Field{{
|
|
Name: "limit", Description: "how many", Kind: KindInt, Min: 1, Max: 50, Default: 10,
|
|
}}},
|
|
Handler: func(_ context.Context, req Request) (Result, error) {
|
|
return Result{Rows: []int{1, 2}, Count: 2, Scope: "all branches"}, nil
|
|
},
|
|
}
|
|
}
|
|
|
|
func registryWith(t *testing.T, tools ...Tool) (*Registry, *CollectAudit) {
|
|
t.Helper()
|
|
audit := &CollectAudit{}
|
|
r := New(audit)
|
|
for _, tool := range tools {
|
|
if err := r.Register(tool); err != nil {
|
|
t.Fatalf("registering %s: %v", tool.Name, err)
|
|
}
|
|
}
|
|
return r, audit
|
|
}
|
|
|
|
var anyone = Caller{Userid: 904, Tenantid: 1147}
|
|
|
|
/* ── The five jobs ─────────────────────────────────────────────────────── */
|
|
|
|
func TestAToolRunsAndReturnsRows(t *testing.T) {
|
|
r, _ := registryWith(t, okTool("thing"))
|
|
agent := Agent{Name: "orders", Tools: []string{"thing"}}
|
|
|
|
result, err := r.Call(context.Background(), agent, "thing", nil, anyone)
|
|
if err != nil {
|
|
t.Fatalf("calling: %v", err)
|
|
}
|
|
if result.Count != 2 {
|
|
t.Fatalf("rows lost: %d", result.Count)
|
|
}
|
|
}
|
|
|
|
func TestAnUnknownToolIsRefused(t *testing.T) {
|
|
r, _ := registryWith(t, okTool("thing"))
|
|
agent := Agent{Name: "orders", Tools: []string{"thing"}}
|
|
|
|
_, err := r.Call(context.Background(), agent, "invented", nil, anyone)
|
|
if !errors.Is(err, ErrUnknownTool) {
|
|
t.Fatalf("a tool the model made up was not refused: %v", err)
|
|
}
|
|
}
|
|
|
|
func TestAToolOffTheAllowListIsRefusedByTheRegistryNotTheModel(t *testing.T) {
|
|
// The whole point of the allow-list: a prompt is a request, this is a rule.
|
|
r, _ := registryWith(t, okTool("thing"), okTool("other"))
|
|
agent := Agent{Name: "orders", Tools: []string{"thing"}}
|
|
|
|
_, err := r.Call(context.Background(), agent, "other", nil, anyone)
|
|
if !errors.Is(err, ErrNotAllowed) {
|
|
t.Fatalf("an agent reached a tool it does not name: %v", err)
|
|
}
|
|
}
|
|
|
|
func TestBadArgumentsNeverReachTheHandler(t *testing.T) {
|
|
reached := false
|
|
tool := okTool("thing")
|
|
tool.Handler = func(context.Context, Request) (Result, error) {
|
|
reached = true
|
|
return Result{}, nil
|
|
}
|
|
r, _ := registryWith(t, tool)
|
|
agent := Agent{Name: "orders", Tools: []string{"thing"}}
|
|
|
|
_, err := r.Call(context.Background(), agent, "thing", map[string]any{"limit": 999}, anyone)
|
|
if !errors.Is(err, ErrBadArgument) {
|
|
t.Fatalf("an out-of-range argument was accepted: %v", err)
|
|
}
|
|
if reached {
|
|
t.Fatal("the handler ran on arguments the schema refused")
|
|
}
|
|
}
|
|
|
|
func TestACallerScopedToNothingIsNotACallerScopedToEverything(t *testing.T) {
|
|
// Go's zero value is 0, so an unset tenant and a platform account look
|
|
// identical unless staff status is asked for separately.
|
|
r, _ := registryWith(t, okTool("thing"))
|
|
agent := Agent{Name: "orders", Tools: []string{"thing"}}
|
|
|
|
_, err := r.Call(context.Background(), agent, "thing", nil, Caller{Userid: 904})
|
|
if !errors.Is(err, ErrNoTenant) {
|
|
t.Fatalf("a caller with no tenant was let through: %v", err)
|
|
}
|
|
|
|
if _, err := r.Call(context.Background(), agent, "thing", nil, Caller{Userid: 12, Superadmin: true}); err != nil {
|
|
t.Fatalf("staff were refused: %v", err)
|
|
}
|
|
}
|
|
|
|
/* ── Arguments ─────────────────────────────────────────────────────────── */
|
|
|
|
func TestDefaultsAreFilledIn(t *testing.T) {
|
|
var seen Request
|
|
tool := okTool("thing")
|
|
tool.Handler = func(_ context.Context, req Request) (Result, error) {
|
|
seen = req
|
|
return Result{}, nil
|
|
}
|
|
r, _ := registryWith(t, tool)
|
|
|
|
if _, err := r.Call(context.Background(), Agent{Name: "a", Tools: []string{"thing"}}, "thing", nil, anyone); err != nil {
|
|
t.Fatalf("calling: %v", err)
|
|
}
|
|
if seen.Int("limit") != 10 {
|
|
t.Fatalf("the default did not arrive: %d", seen.Int("limit"))
|
|
}
|
|
}
|
|
|
|
func TestAJSONNumberIsAcceptedAsAnInteger(t *testing.T) {
|
|
// Every argument arrives over HTTP, so an integer field that only accepted
|
|
// Go ints would refuse every real call.
|
|
var seen Request
|
|
tool := okTool("thing")
|
|
tool.Handler = func(_ context.Context, req Request) (Result, error) {
|
|
seen = req
|
|
return Result{}, nil
|
|
}
|
|
r, _ := registryWith(t, tool)
|
|
|
|
if _, err := r.Call(context.Background(), Agent{Name: "a", Tools: []string{"thing"}}, "thing", map[string]any{"limit": float64(20)}, anyone); err != nil {
|
|
t.Fatalf("a JSON number was refused: %v", err)
|
|
}
|
|
if seen.Int("limit") != 20 {
|
|
t.Fatalf("the value did not survive: %d", seen.Int("limit"))
|
|
}
|
|
}
|
|
|
|
func TestAFractionIsNotAWholeNumber(t *testing.T) {
|
|
r, _ := registryWith(t, okTool("thing"))
|
|
_, err := r.Call(context.Background(), Agent{Name: "a", Tools: []string{"thing"}}, "thing", map[string]any{"limit": 2.5}, anyone)
|
|
if !errors.Is(err, ErrBadArgument) {
|
|
t.Fatalf("2.5 was accepted as a count: %v", err)
|
|
}
|
|
}
|
|
|
|
func TestAnInventedArgumentIsDroppedNotPassedOn(t *testing.T) {
|
|
// A handler must never receive a key it did not declare, or an argument the
|
|
// model made up becomes one a handler might later start reading.
|
|
var seen Request
|
|
tool := okTool("thing")
|
|
tool.Handler = func(_ context.Context, req Request) (Result, error) {
|
|
seen = req
|
|
return Result{}, nil
|
|
}
|
|
r, _ := registryWith(t, tool)
|
|
|
|
args := map[string]any{"limit": 5, "tenantid": 916, "where": "1=1"}
|
|
if _, err := r.Call(context.Background(), Agent{Name: "a", Tools: []string{"thing"}}, "thing", args, anyone); err != nil {
|
|
t.Fatalf("calling: %v", err)
|
|
}
|
|
if _, present := seen.Args["tenantid"]; present {
|
|
t.Fatal("the model got to name a tenant")
|
|
}
|
|
if _, present := seen.Args["where"]; present {
|
|
t.Fatal("the model got to pass a where clause")
|
|
}
|
|
}
|
|
|
|
/* ── Registration ──────────────────────────────────────────────────────── */
|
|
|
|
func TestAToolCannotBeSilentlyReplaced(t *testing.T) {
|
|
// Overwriting a registered tool is how a permission check disappears.
|
|
r, _ := registryWith(t, okTool("thing"))
|
|
if err := r.Register(okTool("thing")); err == nil {
|
|
t.Fatal("a second tool took the same name")
|
|
}
|
|
}
|
|
|
|
func TestAToolWithoutADescriptionIsRefused(t *testing.T) {
|
|
// A model chooses between tools by their descriptions. One without is a
|
|
// tool that gets called for the wrong question.
|
|
r := New(nil)
|
|
tool := okTool("thing")
|
|
tool.Description = ""
|
|
if err := r.Register(tool); err == nil {
|
|
t.Fatal("a tool with no description was registered")
|
|
}
|
|
}
|
|
|
|
/* ── What the model is told ────────────────────────────────────────────── */
|
|
|
|
func TestAnAgentIsOnlyToldAboutToolsItMayUse(t *testing.T) {
|
|
// Describing a tool the agent would then be refused reads to a model as a
|
|
// malfunction, and to a person as the assistant being broken.
|
|
r, _ := registryWith(t, okTool("thing"), okTool("other"))
|
|
defs := r.Definitions(Agent{Name: "orders", Tools: []string{"thing"}})
|
|
|
|
if len(defs) != 1 || defs[0]["name"] != "thing" {
|
|
t.Fatalf("the agent was told about the wrong tools: %v", defs)
|
|
}
|
|
}
|
|
|
|
func TestTheSchemaForbidsInventedArguments(t *testing.T) {
|
|
r, _ := registryWith(t, okTool("thing"))
|
|
defs := r.Definitions(Agent{Name: "orders", Tools: []string{"thing"}})
|
|
schema, _ := defs[0]["input_schema"].(map[string]any)
|
|
|
|
if schema["additionalProperties"] != false {
|
|
t.Fatal("the schema lets the model add its own arguments")
|
|
}
|
|
}
|
|
|
|
/* ── The audit trail ───────────────────────────────────────────────────── */
|
|
|
|
func TestEveryCallLeavesExactlyOneRow(t *testing.T) {
|
|
r, audit := registryWith(t, okTool("thing"))
|
|
agent := Agent{Name: "orders", Tools: []string{"thing"}}
|
|
|
|
_, _ = r.Call(context.Background(), agent, "thing", nil, anyone)
|
|
if len(audit.Entries) != 1 {
|
|
t.Fatalf("a successful call wrote %d rows", len(audit.Entries))
|
|
}
|
|
entry, _ := audit.Last()
|
|
if entry.Outcome != OutcomeOK || entry.Tool != "thing" || entry.Tenantid != 1147 {
|
|
t.Fatalf("the row does not describe the call: %+v", entry)
|
|
}
|
|
}
|
|
|
|
func TestARefusalIsAudited(t *testing.T) {
|
|
// The refusals are the interesting ones. A trail of successes answers
|
|
// "did anything try to read another tenant?" with silence, which reads the
|
|
// same as "no".
|
|
r, audit := registryWith(t, okTool("thing"), okTool("other"))
|
|
|
|
_, _ = r.Call(context.Background(), Agent{Name: "orders", Tools: []string{"thing"}}, "other", nil, anyone)
|
|
entry, ok := audit.Last()
|
|
if !ok || entry.Outcome != OutcomeRefused {
|
|
t.Fatalf("a refusal left no trace: %+v", entry)
|
|
}
|
|
if entry.Detail == "" {
|
|
t.Fatal("the refusal does not say why")
|
|
}
|
|
}
|
|
|
|
func TestABrokenToolIsFailedNotRefused(t *testing.T) {
|
|
// Collapsing the two hides a broken tool inside a count of things working
|
|
// as designed.
|
|
tool := okTool("thing")
|
|
tool.Handler = func(context.Context, Request) (Result, error) {
|
|
return Result{}, errors.New("the database is down")
|
|
}
|
|
r, audit := registryWith(t, tool)
|
|
|
|
_, err := r.Call(context.Background(), Agent{Name: "a", Tools: []string{"thing"}}, "thing", nil, anyone)
|
|
if err == nil {
|
|
t.Fatal("a broken handler reported success")
|
|
}
|
|
entry, _ := audit.Last()
|
|
if entry.Outcome != OutcomeFailed {
|
|
t.Fatalf("a handler error was recorded as %q", entry.Outcome)
|
|
}
|
|
}
|
|
|
|
func TestTheAuditKeepsWhatRanNotWhatWasSent(t *testing.T) {
|
|
// Defaults applied, invented keys dropped. What actually executed is the
|
|
// thing worth being able to read back.
|
|
r, audit := registryWith(t, okTool("thing"))
|
|
|
|
_, _ = r.Call(context.Background(), Agent{Name: "a", Tools: []string{"thing"}}, "thing",
|
|
map[string]any{"tenantid": 916}, anyone)
|
|
|
|
entry, _ := audit.Last()
|
|
if _, present := entry.Args["tenantid"]; present {
|
|
t.Fatal("the audit kept an argument the handler never saw")
|
|
}
|
|
if entry.Args["limit"] != 10 {
|
|
t.Fatalf("the applied default is missing from the trail: %+v", entry.Args)
|
|
}
|
|
}
|
|
|
|
func TestAnAuditLineIsStableBetweenIdenticalCalls(t *testing.T) {
|
|
// Go randomises map iteration, so without sorting the same call logs
|
|
// differently every time and a grep for one of them finds one of them.
|
|
entry := AuditEntry{
|
|
At: time.Now(), Agent: "orders", Tool: "thing", Userid: 904, Tenantid: 1147,
|
|
Args: map[string]any{"b": 2, "a": 1, "c": 3}, Outcome: OutcomeOK,
|
|
}
|
|
first := entry.Line()
|
|
for range 20 {
|
|
if entry.Line() != first {
|
|
t.Fatalf("two renderings of one entry differ:\n%s\n%s", first, entry.Line())
|
|
}
|
|
}
|
|
}
|
|
|
|
func TestAnAuditLineSaysWhenThereWasNoTenant(t *testing.T) {
|
|
// Explicitly, rather than by omission: no tenant on an assistant call is
|
|
// either staff or a bug, and both are worth being able to search for.
|
|
entry := AuditEntry{Agent: "orders", Tool: "thing", Userid: 12, Outcome: OutcomeOK}
|
|
if got := entry.Line(); !contains(got, "tenant=none") {
|
|
t.Fatalf("a tenantless call is invisible in the log: %s", got)
|
|
}
|
|
}
|
|
|
|
func contains(haystack, needle string) bool {
|
|
return len(haystack) >= len(needle) && (func() bool {
|
|
for i := 0; i+len(needle) <= len(haystack); i++ {
|
|
if haystack[i:i+len(needle)] == needle {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
})()
|
|
}
|