Files
backend_fiesta/services/tools/evals_test.go
2026-09-23 17:26:13 +05:30

263 lines
9.4 KiB
Go

package tools
import (
"context"
"encoding/json"
"flag"
"os"
"path/filepath"
"testing"
"time"
"nearle/models"
)
// Evals: the answers this assistant is known to give.
//
// Every case below is a fixed shop, a fixed clock and a real tool, with its
// whole answer written down in `testdata`. Change the arithmetic, the sorting,
// a threshold, a field name or a sentence the model is told to repeat, and the
// diff turns up here rather than in front of a shopkeeper.
//
// ── Why whole answers rather than assertions ────────────────────────────────
//
// The per-tool tests already assert the things somebody thought to check. These
// catch what nobody thought to check — a field quietly renamed, a note that
// stopped mentioning truncation, a rounding change in the fourth decimal. An
// eval that only checked the numbers it was told to check would have been
// written by the same person who wrote the bug.
//
// ── Reading a failure ───────────────────────────────────────────────────────
//
// A failing eval is not automatically a bug: an intended change shows up here
// too. Run with `-update` to rewrite the golden files, then READ THE DIFF — it
// is the summary of what this change does to every answer the assistant gives.
// Committing an update without reading it is how a regression ships wearing the
// clothes of an improvement.
//
// go test ./services/tools/ -run TestEval -update
var updateGolden = flag.Bool("update", false, "rewrite the golden answers in testdata")
// evalNow is the instant every eval is run at, so "40 minutes ago" is the same
// forty minutes in a year's time.
var evalNow = time.Date(2026, 9, 23, 14, 0, 0, 0, time.Local)
func evalStamp(minutesAgo int) string {
return evalNow.Add(-time.Duration(minutesAgo) * time.Minute).Format("2006-01-02 15:04:05")
}
// evalShop is one fixed merchant, used by every case that needs data.
//
// Deliberately awkward in the ways production is: a branch with no orders, a
// rider column carrying a status instead of a name, a stamp that will not parse.
// An eval over tidy data proves the tool works on data that does not exist.
type evalShop struct {
deliveries []models.Deliveryinfo
branches []models.Ordersummarylocation
requests []models.StockRequest
stocks []models.Productstocks
}
func theEvalShop() evalShop {
return evalShop{
deliveries: []models.Deliveryinfo{
{Deliveryid: 4412, Orderid: "ORD-4412", Orderstatus: "pending", Assigntime: evalStamp(41),
Ridername: "Varun", Locationname: "R Mart", Deliverycustomer: "S Kumar"},
{Deliveryid: 4407, Orderid: "ORD-4407", Orderstatus: "pending", Assigntime: evalStamp(33),
Ridername: "delivered", Locationname: "R Mart"}, // a status in the name column
{Deliveryid: 4419, Orderid: "ORD-4419", Orderstatus: "pending", Assigntime: evalStamp(12),
Ridername: "Murali", Locationname: "Anna Nagar"},
{Deliveryid: 4420, Orderid: "ORD-4420", Orderstatus: "pending", Assigntime: evalStamp(4)},
{Deliveryid: 4421, Orderid: "ORD-4421", Orderstatus: "picked", Assigntime: evalStamp(90)},
{Deliveryid: 4422, Orderid: "ORD-4422", Orderstatus: "active", Assigntime: evalStamp(70)},
{Deliveryid: 4423, Orderid: "ORD-4423", Orderstatus: "delivered", Assigntime: evalStamp(200)},
{Deliveryid: 4424, Orderid: "ORD-4424", Orderstatus: "rejected", Assigntime: evalStamp(150)},
{Deliveryid: 4425, Orderid: "ORD-4425", Orderstatus: "pending", Assigntime: "not a date"},
},
branches: []models.Ordersummarylocation{
{Locationid: 1172, Locationname: "R Mart", Total: 240, Delivered: 198, Cancelled: 9},
{Locationid: 1173, Locationname: "Anna Nagar", Total: 180, Delivered: 121, Cancelled: 41},
{Locationid: 1174, Locationname: "New Shop", Total: 0},
},
requests: []models.StockRequest{
{Requestid: 41, Qty: 12, Productname: "Basmati Rice 5kg", Locationname: "R Mart",
Status: "Pending", Created: evalNow.AddDate(0, 0, -3)},
{Requestid: 42, Qty: 40, Productname: "Sunflower Oil 1L", Locationname: "Anna Nagar",
Status: "Pending", Created: evalNow.AddDate(0, 0, -9)},
},
stocks: []models.Productstocks{
{Productid: 88, Productname: "Atta 10kg", Quantity: 0},
{Productid: 91, Productname: "Sugar 1kg", Quantity: 2},
{Productid: 92, Productname: "Salt 1kg", Quantity: 2},
{Productid: 93, Productname: "Tea 250g", Quantity: 60},
},
}
}
// evalCase is one question, as a tool call.
type evalCase struct {
// The file in testdata, and the name in test output.
name string
// What a person would have asked to get here. Not executed — it is the
// reason this case exists, and without it a golden file is a wall of JSON
// nobody can review.
question string
tool func(evalShop) Tool
args map[string]any
caller Caller
}
var merchantAll = Caller{Userid: 904, Tenantid: 1147}
var merchantBranch = Caller{Userid: 904, Tenantid: 1147, Locationid: 1172}
func evalCases() []evalCase {
return []evalCase{
{
name: "stuck_orders",
question: "Which orders are stuck?",
tool: func(s evalShop) Tool {
return StuckOrders(&fakeDeliveries{rows: s.deliveries}, func() time.Time { return evalNow })
},
caller: merchantAll,
},
{
name: "stuck_orders_half_an_hour",
question: "Anything waiting more than half an hour?",
tool: func(s evalShop) Tool {
return StuckOrders(&fakeDeliveries{rows: s.deliveries}, func() time.Time { return evalNow })
},
args: map[string]any{"minutes_waiting": 30},
caller: merchantAll,
},
{
name: "delivery_progress",
question: "What is out for delivery?",
tool: func(s evalShop) Tool { return DeliveryProgress(&fakeDeliveries{rows: s.deliveries}) },
caller: merchantAll,
},
{
name: "branch_performance",
question: "Which branch is underperforming?",
tool: func(s evalShop) Tool { return BranchPerformance(&fakeBranches{rows: s.branches}) },
caller: merchantAll,
},
{
name: "pending_approvals",
question: "What needs my approval?",
tool: func(s evalShop) Tool {
return PendingApprovals(&fakeApprovals{rows: s.requests}, func() time.Time { return evalNow })
},
caller: merchantAll,
},
{
name: "low_stock",
question: "Where is stock running out?",
tool: func(s evalShop) Tool { return LowStock(&fakeStocks{rows: s.stocks}) },
caller: merchantBranch,
},
{
name: "help_cashier",
question: "How do I add a cashier?",
tool: func(evalShop) Tool {
corpus, err := LoadHelp()
if err != nil {
panic(err)
}
return Help(corpus)
},
args: map[string]any{"question": "how do I add a cashier"},
caller: merchantAll,
},
{
name: "help_unanswerable",
question: "What is the capital of France?",
tool: func(evalShop) Tool {
corpus, err := LoadHelp()
if err != nil {
panic(err)
}
return Help(corpus)
},
args: map[string]any{"question": "what is the capital of france"},
caller: merchantAll,
},
}
}
// golden is the whole answer, in the order a reader wants it.
//
// `Question` rides along so a diff reads as "this is what changed about the
// answer to THAT" rather than as an anonymous blob.
type golden struct {
Question string `json:"question"`
Rows any `json:"rows"`
Count int `json:"count"`
Truncated bool `json:"truncated,omitempty"`
Note string `json:"note,omitempty"`
Source string `json:"source,omitempty"`
Covers string `json:"covers,omitempty"`
}
func TestEvalAnswersHaveNotChanged(t *testing.T) {
for _, eval := range evalCases() {
t.Run(eval.name, func(t *testing.T) {
shop := theEvalShop()
tool := eval.tool(shop)
r := New(nil)
if err := r.Register(tool); err != nil {
t.Fatalf("registering: %v", err)
}
result, err := r.Call(context.Background(),
Agent{Name: "eval", Tools: []string{tool.Name}}, tool.Name, eval.args, eval.caller)
if err != nil {
t.Fatalf("%s: %v", eval.name, err)
}
got, err := json.MarshalIndent(golden{
Question: eval.question, Rows: result.Rows, Count: result.Count,
Truncated: result.Truncated, Note: result.Note,
Source: result.Source, Covers: result.Scope,
}, "", " ")
if err != nil {
t.Fatalf("encoding: %v", err)
}
got = append(got, '\n')
path := filepath.Join("testdata", eval.name+".json")
if *updateGolden {
if err := os.MkdirAll("testdata", 0o755); err != nil {
t.Fatalf("testdata: %v", err)
}
if err := os.WriteFile(path, got, 0o600); err != nil {
t.Fatalf("writing %s: %v", path, err)
}
return
}
want, err := os.ReadFile(path)
if err != nil {
t.Fatalf("no golden answer for %q — run with -update and read the diff: %v", eval.name, err)
}
if string(got) != string(want) {
t.Fatalf(
"the answer to %q changed.\n\nIf that was intended, run:\n go test ./services/tools/ -run TestEval -update\nand read the diff before committing it.\n\nwant:\n%s\ngot:\n%s",
eval.question, want, got)
}
})
}
}
// TestEveryEvalCaseHasAQuestion keeps the golden files reviewable.
//
// A case with no question is a wall of JSON nobody can judge, and an
// unreviewable golden file gets updated rather than read.
func TestEveryEvalCaseHasAQuestion(t *testing.T) {
for _, eval := range evalCases() {
if eval.question == "" {
t.Fatalf("eval %q does not say what was asked", eval.name)
}
}
}