agent
This commit is contained in:
262
services/tools/evals_test.go
Normal file
262
services/tools/evals_test.go
Normal file
@@ -0,0 +1,262 @@
|
||||
package tools
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"flag"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"nearle/models"
|
||||
)
|
||||
|
||||
// Evals: the answers this assistant is known to give.
|
||||
//
|
||||
// Every case below is a fixed shop, a fixed clock and a real tool, with its
|
||||
// whole answer written down in `testdata`. Change the arithmetic, the sorting,
|
||||
// a threshold, a field name or a sentence the model is told to repeat, and the
|
||||
// diff turns up here rather than in front of a shopkeeper.
|
||||
//
|
||||
// ── Why whole answers rather than assertions ────────────────────────────────
|
||||
//
|
||||
// The per-tool tests already assert the things somebody thought to check. These
|
||||
// catch what nobody thought to check — a field quietly renamed, a note that
|
||||
// stopped mentioning truncation, a rounding change in the fourth decimal. An
|
||||
// eval that only checked the numbers it was told to check would have been
|
||||
// written by the same person who wrote the bug.
|
||||
//
|
||||
// ── Reading a failure ───────────────────────────────────────────────────────
|
||||
//
|
||||
// A failing eval is not automatically a bug: an intended change shows up here
|
||||
// too. Run with `-update` to rewrite the golden files, then READ THE DIFF — it
|
||||
// is the summary of what this change does to every answer the assistant gives.
|
||||
// Committing an update without reading it is how a regression ships wearing the
|
||||
// clothes of an improvement.
|
||||
//
|
||||
// go test ./services/tools/ -run TestEval -update
|
||||
|
||||
var updateGolden = flag.Bool("update", false, "rewrite the golden answers in testdata")
|
||||
|
||||
// evalNow is the instant every eval is run at, so "40 minutes ago" is the same
|
||||
// forty minutes in a year's time.
|
||||
var evalNow = time.Date(2026, 9, 23, 14, 0, 0, 0, time.Local)
|
||||
|
||||
func evalStamp(minutesAgo int) string {
|
||||
return evalNow.Add(-time.Duration(minutesAgo) * time.Minute).Format("2006-01-02 15:04:05")
|
||||
}
|
||||
|
||||
// evalShop is one fixed merchant, used by every case that needs data.
|
||||
//
|
||||
// Deliberately awkward in the ways production is: a branch with no orders, a
|
||||
// rider column carrying a status instead of a name, a stamp that will not parse.
|
||||
// An eval over tidy data proves the tool works on data that does not exist.
|
||||
type evalShop struct {
|
||||
deliveries []models.Deliveryinfo
|
||||
branches []models.Ordersummarylocation
|
||||
requests []models.StockRequest
|
||||
stocks []models.Productstocks
|
||||
}
|
||||
|
||||
func theEvalShop() evalShop {
|
||||
return evalShop{
|
||||
deliveries: []models.Deliveryinfo{
|
||||
{Deliveryid: 4412, Orderid: "ORD-4412", Orderstatus: "pending", Assigntime: evalStamp(41),
|
||||
Ridername: "Varun", Locationname: "R Mart", Deliverycustomer: "S Kumar"},
|
||||
{Deliveryid: 4407, Orderid: "ORD-4407", Orderstatus: "pending", Assigntime: evalStamp(33),
|
||||
Ridername: "delivered", Locationname: "R Mart"}, // a status in the name column
|
||||
{Deliveryid: 4419, Orderid: "ORD-4419", Orderstatus: "pending", Assigntime: evalStamp(12),
|
||||
Ridername: "Murali", Locationname: "Anna Nagar"},
|
||||
{Deliveryid: 4420, Orderid: "ORD-4420", Orderstatus: "pending", Assigntime: evalStamp(4)},
|
||||
{Deliveryid: 4421, Orderid: "ORD-4421", Orderstatus: "picked", Assigntime: evalStamp(90)},
|
||||
{Deliveryid: 4422, Orderid: "ORD-4422", Orderstatus: "active", Assigntime: evalStamp(70)},
|
||||
{Deliveryid: 4423, Orderid: "ORD-4423", Orderstatus: "delivered", Assigntime: evalStamp(200)},
|
||||
{Deliveryid: 4424, Orderid: "ORD-4424", Orderstatus: "rejected", Assigntime: evalStamp(150)},
|
||||
{Deliveryid: 4425, Orderid: "ORD-4425", Orderstatus: "pending", Assigntime: "not a date"},
|
||||
},
|
||||
branches: []models.Ordersummarylocation{
|
||||
{Locationid: 1172, Locationname: "R Mart", Total: 240, Delivered: 198, Cancelled: 9},
|
||||
{Locationid: 1173, Locationname: "Anna Nagar", Total: 180, Delivered: 121, Cancelled: 41},
|
||||
{Locationid: 1174, Locationname: "New Shop", Total: 0},
|
||||
},
|
||||
requests: []models.StockRequest{
|
||||
{Requestid: 41, Qty: 12, Productname: "Basmati Rice 5kg", Locationname: "R Mart",
|
||||
Status: "Pending", Created: evalNow.AddDate(0, 0, -3)},
|
||||
{Requestid: 42, Qty: 40, Productname: "Sunflower Oil 1L", Locationname: "Anna Nagar",
|
||||
Status: "Pending", Created: evalNow.AddDate(0, 0, -9)},
|
||||
},
|
||||
stocks: []models.Productstocks{
|
||||
{Productid: 88, Productname: "Atta 10kg", Quantity: 0},
|
||||
{Productid: 91, Productname: "Sugar 1kg", Quantity: 2},
|
||||
{Productid: 92, Productname: "Salt 1kg", Quantity: 2},
|
||||
{Productid: 93, Productname: "Tea 250g", Quantity: 60},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// evalCase is one question, as a tool call.
|
||||
type evalCase struct {
|
||||
// The file in testdata, and the name in test output.
|
||||
name string
|
||||
// What a person would have asked to get here. Not executed — it is the
|
||||
// reason this case exists, and without it a golden file is a wall of JSON
|
||||
// nobody can review.
|
||||
question string
|
||||
tool func(evalShop) Tool
|
||||
args map[string]any
|
||||
caller Caller
|
||||
}
|
||||
|
||||
var merchantAll = Caller{Userid: 904, Tenantid: 1147}
|
||||
var merchantBranch = Caller{Userid: 904, Tenantid: 1147, Locationid: 1172}
|
||||
|
||||
func evalCases() []evalCase {
|
||||
return []evalCase{
|
||||
{
|
||||
name: "stuck_orders",
|
||||
question: "Which orders are stuck?",
|
||||
tool: func(s evalShop) Tool {
|
||||
return StuckOrders(&fakeDeliveries{rows: s.deliveries}, func() time.Time { return evalNow })
|
||||
},
|
||||
caller: merchantAll,
|
||||
},
|
||||
{
|
||||
name: "stuck_orders_half_an_hour",
|
||||
question: "Anything waiting more than half an hour?",
|
||||
tool: func(s evalShop) Tool {
|
||||
return StuckOrders(&fakeDeliveries{rows: s.deliveries}, func() time.Time { return evalNow })
|
||||
},
|
||||
args: map[string]any{"minutes_waiting": 30},
|
||||
caller: merchantAll,
|
||||
},
|
||||
{
|
||||
name: "delivery_progress",
|
||||
question: "What is out for delivery?",
|
||||
tool: func(s evalShop) Tool { return DeliveryProgress(&fakeDeliveries{rows: s.deliveries}) },
|
||||
caller: merchantAll,
|
||||
},
|
||||
{
|
||||
name: "branch_performance",
|
||||
question: "Which branch is underperforming?",
|
||||
tool: func(s evalShop) Tool { return BranchPerformance(&fakeBranches{rows: s.branches}) },
|
||||
caller: merchantAll,
|
||||
},
|
||||
{
|
||||
name: "pending_approvals",
|
||||
question: "What needs my approval?",
|
||||
tool: func(s evalShop) Tool {
|
||||
return PendingApprovals(&fakeApprovals{rows: s.requests}, func() time.Time { return evalNow })
|
||||
},
|
||||
caller: merchantAll,
|
||||
},
|
||||
{
|
||||
name: "low_stock",
|
||||
question: "Where is stock running out?",
|
||||
tool: func(s evalShop) Tool { return LowStock(&fakeStocks{rows: s.stocks}) },
|
||||
caller: merchantBranch,
|
||||
},
|
||||
{
|
||||
name: "help_cashier",
|
||||
question: "How do I add a cashier?",
|
||||
tool: func(evalShop) Tool {
|
||||
corpus, err := LoadHelp()
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
return Help(corpus)
|
||||
},
|
||||
args: map[string]any{"question": "how do I add a cashier"},
|
||||
caller: merchantAll,
|
||||
},
|
||||
{
|
||||
name: "help_unanswerable",
|
||||
question: "What is the capital of France?",
|
||||
tool: func(evalShop) Tool {
|
||||
corpus, err := LoadHelp()
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
return Help(corpus)
|
||||
},
|
||||
args: map[string]any{"question": "what is the capital of france"},
|
||||
caller: merchantAll,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// golden is the whole answer, in the order a reader wants it.
|
||||
//
|
||||
// `Question` rides along so a diff reads as "this is what changed about the
|
||||
// answer to THAT" rather than as an anonymous blob.
|
||||
type golden struct {
|
||||
Question string `json:"question"`
|
||||
Rows any `json:"rows"`
|
||||
Count int `json:"count"`
|
||||
Truncated bool `json:"truncated,omitempty"`
|
||||
Note string `json:"note,omitempty"`
|
||||
Source string `json:"source,omitempty"`
|
||||
Covers string `json:"covers,omitempty"`
|
||||
}
|
||||
|
||||
func TestEvalAnswersHaveNotChanged(t *testing.T) {
|
||||
for _, eval := range evalCases() {
|
||||
t.Run(eval.name, func(t *testing.T) {
|
||||
shop := theEvalShop()
|
||||
tool := eval.tool(shop)
|
||||
|
||||
r := New(nil)
|
||||
if err := r.Register(tool); err != nil {
|
||||
t.Fatalf("registering: %v", err)
|
||||
}
|
||||
result, err := r.Call(context.Background(),
|
||||
Agent{Name: "eval", Tools: []string{tool.Name}}, tool.Name, eval.args, eval.caller)
|
||||
if err != nil {
|
||||
t.Fatalf("%s: %v", eval.name, err)
|
||||
}
|
||||
|
||||
got, err := json.MarshalIndent(golden{
|
||||
Question: eval.question, Rows: result.Rows, Count: result.Count,
|
||||
Truncated: result.Truncated, Note: result.Note,
|
||||
Source: result.Source, Covers: result.Scope,
|
||||
}, "", " ")
|
||||
if err != nil {
|
||||
t.Fatalf("encoding: %v", err)
|
||||
}
|
||||
got = append(got, '\n')
|
||||
|
||||
path := filepath.Join("testdata", eval.name+".json")
|
||||
if *updateGolden {
|
||||
if err := os.MkdirAll("testdata", 0o755); err != nil {
|
||||
t.Fatalf("testdata: %v", err)
|
||||
}
|
||||
if err := os.WriteFile(path, got, 0o600); err != nil {
|
||||
t.Fatalf("writing %s: %v", path, err)
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
want, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Fatalf("no golden answer for %q — run with -update and read the diff: %v", eval.name, err)
|
||||
}
|
||||
if string(got) != string(want) {
|
||||
t.Fatalf(
|
||||
"the answer to %q changed.\n\nIf that was intended, run:\n go test ./services/tools/ -run TestEval -update\nand read the diff before committing it.\n\nwant:\n%s\ngot:\n%s",
|
||||
eval.question, want, got)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestEveryEvalCaseHasAQuestion keeps the golden files reviewable.
|
||||
//
|
||||
// A case with no question is a wall of JSON nobody can judge, and an
|
||||
// unreviewable golden file gets updated rather than read.
|
||||
func TestEveryEvalCaseHasAQuestion(t *testing.T) {
|
||||
for _, eval := range evalCases() {
|
||||
if eval.question == "" {
|
||||
t.Fatalf("eval %q does not say what was asked", eval.name)
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user