Phase 1 of Nearle Buddy: an agent names a tool, and the registry decides whether that is allowed, whether the arguments make sense, who is asking, and what gets recorded — then runs a handler a person wrote and tested. No agent gets raw table access. The usual argument for tools over generated SQL is safety; here there is a harder one. The fields on this backend do not mean what their names say, and it is measured: orders.deliverystatus is an empty string on all 181 rows of tenant 1147, orders.orderstatus never carries the six middle delivery stages, deliveries.ridername holds statuses as often as names, deliverytype is empty on every row in production. A model writing SQL gets each of those wrong with no error — it reports a cancel rate from a column of empty strings and nobody can tell. A model calling a tool cannot, because the correction lives in the handler beside the measurement that justified it. Call does five things in order: find the tool, check the agent's allow-list, validate arguments, confirm the caller is scoped to something, run the handler — writing exactly one audit row whatever happens, refusals included. A trail of successes answers "did anything try to read another tenant?" with silence, which reads the same as no. The model has no say in whose data is read. stuck_orders has no tenantid field on its schema — absent, not rejected — and the tenant comes from the session claims added in the previous commit. Arguments the tool did not declare are dropped rather than passed on, so a model sending a `where` clause gets it discarded. stuck_orders: deliveries a rider was given and has not accepted, ten minutes for a look, twenty-five for somebody now. Derived from assigntime and orderstatus, so it does not depend on anyone having been watching. Carries the wait in minutes, what to do, where to check it, and what it covered. A capped answer says so — an empty result and a truncated one look identical to a model and it will call both "none". The audit sink writes to the log for now; a database sink is phase 8. Nothing calls the registry yet: the loop and the model gateway are phase 2. 37 tests. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
306 lines
11 KiB
Go
306 lines
11 KiB
Go
package tools
|
|
|
|
import (
|
|
"context"
|
|
"testing"
|
|
"time"
|
|
|
|
"nearle/models"
|
|
)
|
|
|
|
var stuckNow = time.Date(2026, 9, 23, 14, 0, 0, 0, time.Local)
|
|
|
|
// fakeDeliveries stands in for the deliveries service, and records what it was
|
|
// asked — the query matters as much as the answer, because that is where the
|
|
// tenant scope either is or is not.
|
|
type fakeDeliveries struct {
|
|
rows []models.Deliveryinfo
|
|
last models.DeliveryQuery
|
|
}
|
|
|
|
func (f *fakeDeliveries) GetDeliveries(input models.DeliveryQuery) []models.Deliveryinfo {
|
|
f.last = input
|
|
return f.rows
|
|
}
|
|
|
|
// assignedMinutesAgo writes the stamp the way `stampNow` does: local
|
|
// wall-clock, no zone.
|
|
func assignedMinutesAgo(n int) string {
|
|
return stuckNow.Add(-time.Duration(n) * time.Minute).Format("2006-01-02 15:04:05")
|
|
}
|
|
|
|
func job(over models.Deliveryinfo) models.Deliveryinfo {
|
|
if over.Orderstatus == "" {
|
|
over.Orderstatus = "pending"
|
|
}
|
|
if over.Assigntime == "" {
|
|
over.Assigntime = assignedMinutesAgo(30)
|
|
}
|
|
return over
|
|
}
|
|
|
|
func runStuck(t *testing.T, rows []models.Deliveryinfo, args map[string]any, caller Caller) (Result, *fakeDeliveries) {
|
|
t.Helper()
|
|
deliveries := &fakeDeliveries{rows: rows}
|
|
r := New(nil)
|
|
if err := r.Register(StuckOrders(deliveries, func() time.Time { return stuckNow })); err != nil {
|
|
t.Fatalf("registering: %v", err)
|
|
}
|
|
result, err := r.Call(context.Background(), Agent{Name: "orders", Tools: []string{"stuck_orders"}}, "stuck_orders", args, caller)
|
|
if err != nil {
|
|
t.Fatalf("calling: %v", err)
|
|
}
|
|
return result, deliveries
|
|
}
|
|
|
|
func rowsOf(t *testing.T, result Result) []StuckOrder {
|
|
t.Helper()
|
|
rows, ok := result.Rows.([]StuckOrder)
|
|
if !ok {
|
|
t.Fatalf("rows are not stuck orders: %T", result.Rows)
|
|
}
|
|
return rows
|
|
}
|
|
|
|
/* ── The tenant is never the model's to choose ─────────────────────────── */
|
|
|
|
func TestTheTenantComesFromTheSessionNotTheArguments(t *testing.T) {
|
|
// The single most important property of the whole registry. The model picks
|
|
// the tool and the arguments; it has no say in whose data is read.
|
|
_, deliveries := runStuck(t, nil, map[string]any{"tenantid": 916}, Caller{Userid: 904, Tenantid: 1147})
|
|
|
|
if deliveries.last.Tenantid != 1147 {
|
|
t.Fatalf("the query ran against tenant %d", deliveries.last.Tenantid)
|
|
}
|
|
}
|
|
|
|
func TestTheToolDoesNotEvenAcceptATenantArgument(t *testing.T) {
|
|
// Belt and braces: the schema must not have the field at all, so there is
|
|
// nothing to argue the model into filling in.
|
|
tool := StuckOrders(&fakeDeliveries{}, nil)
|
|
for _, field := range tool.Schema.Fields {
|
|
if field.Name == "tenantid" || field.Name == "locationid" {
|
|
t.Fatalf("the schema offers %q for the model to set", field.Name)
|
|
}
|
|
}
|
|
}
|
|
|
|
func TestABranchUserIsScopedToTheirBranch(t *testing.T) {
|
|
_, deliveries := runStuck(t, nil, nil, Caller{Userid: 904, Tenantid: 1147, Locationid: 1172})
|
|
if deliveries.last.Locationid != 1172 {
|
|
t.Fatalf("a branch user read branch %d", deliveries.last.Locationid)
|
|
}
|
|
}
|
|
|
|
func TestStaffMustPickATenantFirst(t *testing.T) {
|
|
// A platform account passes the registry's caller check but cannot ask this
|
|
// question of "everyone" — the answer would span merchants.
|
|
deliveries := &fakeDeliveries{}
|
|
r := New(nil)
|
|
_ = r.Register(StuckOrders(deliveries, func() time.Time { return stuckNow }))
|
|
|
|
_, err := r.Call(context.Background(), Agent{Name: "orders", Tools: []string{"stuck_orders"}},
|
|
"stuck_orders", nil, Caller{Userid: 12, Superadmin: true})
|
|
if err == nil {
|
|
t.Fatal("staff read stuck orders across every tenant at once")
|
|
}
|
|
}
|
|
|
|
/* ── What counts as stuck ──────────────────────────────────────────────── */
|
|
|
|
func TestOnlyPendingJobsAreStuck(t *testing.T) {
|
|
// A job the rider accepted, picked up or delivered is not stuck however old
|
|
// it is. Rejected and skipped are a different problem with a different fix.
|
|
rows := []models.Deliveryinfo{
|
|
job(models.Deliveryinfo{Deliveryid: 1, Orderstatus: "pending"}),
|
|
job(models.Deliveryinfo{Deliveryid: 2, Orderstatus: "accepted"}),
|
|
job(models.Deliveryinfo{Deliveryid: 3, Orderstatus: "picked"}),
|
|
job(models.Deliveryinfo{Deliveryid: 4, Orderstatus: "delivered"}),
|
|
job(models.Deliveryinfo{Deliveryid: 5, Orderstatus: "rejected"}),
|
|
job(models.Deliveryinfo{Deliveryid: 6, Orderstatus: "skipped"}),
|
|
}
|
|
result, _ := runStuck(t, rows, nil, anyone)
|
|
|
|
if result.Count != 1 || rowsOf(t, result)[0].Deliveryid != 1 {
|
|
t.Fatalf("expected only the pending job: %+v", result.Rows)
|
|
}
|
|
}
|
|
|
|
func TestTheCasingTheRiderAppActuallyWrites(t *testing.T) {
|
|
// Fiesta stores status as free text and the casing varies between writers.
|
|
rows := []models.Deliveryinfo{job(models.Deliveryinfo{Deliveryid: 1, Orderstatus: "Pending"})}
|
|
result, _ := runStuck(t, rows, nil, anyone)
|
|
if result.Count != 1 {
|
|
t.Fatal("a capitalised status stopped counting")
|
|
}
|
|
}
|
|
|
|
func TestAFewMinutesOfSilenceIsOrdinary(t *testing.T) {
|
|
// Flagging this would train people to ignore the flag, and an ignored flag
|
|
// is worse than none.
|
|
rows := []models.Deliveryinfo{
|
|
job(models.Deliveryinfo{Deliveryid: 1, Assigntime: assignedMinutesAgo(3)}),
|
|
job(models.Deliveryinfo{Deliveryid: 2, Assigntime: assignedMinutesAgo(9)}),
|
|
}
|
|
result, _ := runStuck(t, rows, nil, anyone)
|
|
if result.Count != 0 {
|
|
t.Fatalf("a nine-minute wait was reported as stuck: %+v", result.Rows)
|
|
}
|
|
}
|
|
|
|
func TestTenMinutesIsALookAndTwentyFiveNeedsSomebody(t *testing.T) {
|
|
rows := []models.Deliveryinfo{
|
|
job(models.Deliveryinfo{Deliveryid: 1, Assigntime: assignedMinutesAgo(12)}),
|
|
job(models.Deliveryinfo{Deliveryid: 2, Assigntime: assignedMinutesAgo(40)}),
|
|
}
|
|
result, _ := runStuck(t, rows, nil, anyone)
|
|
got := rowsOf(t, result)
|
|
|
|
// Worst first: the forty-minute job leads.
|
|
if got[0].Deliveryid != 2 || got[0].Urgency != "now" {
|
|
t.Fatalf("the urgent job is not first: %+v", got)
|
|
}
|
|
if got[1].Urgency != "look" {
|
|
t.Fatalf("a twelve-minute wait was not a look: %+v", got[1])
|
|
}
|
|
if got[0].Action == got[1].Action {
|
|
t.Fatal("both buckets suggest the same action, so the bucket says nothing")
|
|
}
|
|
}
|
|
|
|
func TestTheThresholdCanBeRaised(t *testing.T) {
|
|
rows := []models.Deliveryinfo{
|
|
job(models.Deliveryinfo{Deliveryid: 1, Assigntime: assignedMinutesAgo(12)}),
|
|
job(models.Deliveryinfo{Deliveryid: 2, Assigntime: assignedMinutesAgo(40)}),
|
|
}
|
|
result, _ := runStuck(t, rows, map[string]any{"minutes_waiting": 30}, anyone)
|
|
if result.Count != 1 || rowsOf(t, result)[0].Deliveryid != 2 {
|
|
t.Fatalf("the threshold was ignored: %+v", result.Rows)
|
|
}
|
|
}
|
|
|
|
func TestTheWaitIsReportedInMinutes(t *testing.T) {
|
|
rows := []models.Deliveryinfo{job(models.Deliveryinfo{Deliveryid: 1, Assigntime: assignedMinutesAgo(40)})}
|
|
result, _ := runStuck(t, rows, nil, anyone)
|
|
if got := rowsOf(t, result)[0].WaitingMinutes; got != 40 {
|
|
t.Fatalf("waiting minutes reported as %d", got)
|
|
}
|
|
}
|
|
|
|
/* ── Stamps that cannot be read ────────────────────────────────────────── */
|
|
|
|
func TestAnUnreadableAssigntimeIsSkippedNotDatedTo1970(t *testing.T) {
|
|
// Reading it as the epoch would report the row as fifty years late, which
|
|
// is the kind of number that gets a whole screen ignored.
|
|
// Built without `job()`, which fills a default stamp in — the helper would
|
|
// hand the parser a valid date and the test would pass without testing.
|
|
rows := []models.Deliveryinfo{
|
|
{Deliveryid: 1, Orderstatus: "pending", Assigntime: ""},
|
|
{Deliveryid: 2, Orderstatus: "pending", Assigntime: "not a date"},
|
|
{Deliveryid: 3, Orderstatus: "pending", Assigntime: " "},
|
|
}
|
|
result, _ := runStuck(t, rows, nil, anyone)
|
|
if result.Count != 0 {
|
|
t.Fatalf("an unreadable stamp produced a row: %+v", result.Rows)
|
|
}
|
|
}
|
|
|
|
func TestALocalStampIsNotReadAsUTC(t *testing.T) {
|
|
// `assigntime` is local wall-clock with no zone. Read as UTC it would be
|
|
// five and a half hours out in India, turning a fresh assignment into a
|
|
// four-hour wait.
|
|
rows := []models.Deliveryinfo{job(models.Deliveryinfo{Deliveryid: 1, Assigntime: assignedMinutesAgo(30)})}
|
|
result, _ := runStuck(t, rows, nil, anyone)
|
|
if got := rowsOf(t, result)[0].WaitingMinutes; got != 30 {
|
|
t.Fatalf("a local stamp read as %d minutes instead of 30", got)
|
|
}
|
|
}
|
|
|
|
func TestAJobFromTheFutureIsAClockNotAWait(t *testing.T) {
|
|
rows := []models.Deliveryinfo{job(models.Deliveryinfo{Deliveryid: 1, Assigntime: assignedMinutesAgo(-20)})}
|
|
result, _ := runStuck(t, rows, nil, anyone)
|
|
if result.Count != 0 {
|
|
t.Fatalf("a stamp in the future was reported as a wait: %+v", result.Rows)
|
|
}
|
|
}
|
|
|
|
/* ── The rider's name ──────────────────────────────────────────────────── */
|
|
|
|
func TestAStatusInTheRiderColumnIsNotAName(t *testing.T) {
|
|
// Measured on tenant 916: rider 897 carries "Varun" 69 times and
|
|
// "delivered" 75, so neither "first non-empty" nor "most common" finds the
|
|
// name. A model handed "delivered" would tell somebody to call a rider
|
|
// called Delivered.
|
|
rows := []models.Deliveryinfo{
|
|
job(models.Deliveryinfo{Deliveryid: 1, Ridername: "delivered"}),
|
|
job(models.Deliveryinfo{Deliveryid: 2, Ridername: "Varun"}),
|
|
}
|
|
result, _ := runStuck(t, rows, nil, anyone)
|
|
got := rowsOf(t, result)
|
|
|
|
byID := map[int]string{}
|
|
for _, row := range got {
|
|
byID[row.Deliveryid] = row.Rider
|
|
}
|
|
if byID[1] != "" {
|
|
t.Fatalf("a status was reported as a rider: %q", byID[1])
|
|
}
|
|
if byID[2] != "Varun" {
|
|
t.Fatalf("a real name was dropped: %q", byID[2])
|
|
}
|
|
}
|
|
|
|
/* ── Saying when the answer is partial ─────────────────────────────────── */
|
|
|
|
func TestACappedAnswerSaysSo(t *testing.T) {
|
|
// An empty answer and a capped answer look identical to a model, and it
|
|
// will describe both as "none".
|
|
rows := make([]models.Deliveryinfo, 0, 60)
|
|
for i := range 60 {
|
|
rows = append(rows, job(models.Deliveryinfo{Deliveryid: i + 1, Assigntime: assignedMinutesAgo(30 + i)}))
|
|
}
|
|
result, _ := runStuck(t, rows, nil, anyone)
|
|
|
|
if !result.Truncated {
|
|
t.Fatal("sixty rows came back as a complete answer")
|
|
}
|
|
if result.Note == "" {
|
|
t.Fatal("the cap is not explained in words the model will repeat")
|
|
}
|
|
if len(rowsOf(t, result)) != stuckMaxRows {
|
|
t.Fatalf("the cap did not apply: %d rows", len(rowsOf(t, result)))
|
|
}
|
|
if result.Count != 60 {
|
|
t.Fatalf("the true total was lost: %d", result.Count)
|
|
}
|
|
}
|
|
|
|
func TestAnEmptyBoardIsAnAnswer(t *testing.T) {
|
|
result, _ := runStuck(t, nil, nil, anyone)
|
|
if result.Count != 0 || result.Truncated {
|
|
t.Fatalf("an empty board was not answered cleanly: %+v", result)
|
|
}
|
|
if result.Scope == "" {
|
|
t.Fatal("even an empty answer must say what it covered")
|
|
}
|
|
}
|
|
|
|
/* ── Evidence ──────────────────────────────────────────────────────────── */
|
|
|
|
func TestTheAnswerCarriesWhereToCheckIt(t *testing.T) {
|
|
// Buddy states a conclusion; this is where a person goes to see the rows.
|
|
result, _ := runStuck(t, []models.Deliveryinfo{job(models.Deliveryinfo{Deliveryid: 1})}, nil, anyone)
|
|
if result.Source == "" {
|
|
t.Fatal("the answer links to nothing")
|
|
}
|
|
}
|
|
|
|
func TestTheScopeIsStatedSoOneBranchIsNotReadAsAll(t *testing.T) {
|
|
all, _ := runStuck(t, nil, nil, Caller{Userid: 904, Tenantid: 1147})
|
|
one, _ := runStuck(t, nil, nil, Caller{Userid: 904, Tenantid: 1147, Locationid: 1172})
|
|
|
|
if all.Scope == one.Scope {
|
|
t.Fatalf("one branch and all branches report the same scope: %q", all.Scope)
|
|
}
|
|
}
|