package tools import ( "context" "encoding/json" "flag" "os" "path/filepath" "testing" "time" "nearle/models" ) // Evals: the answers this assistant is known to give. // // Every case below is a fixed shop, a fixed clock and a real tool, with its // whole answer written down in `testdata`. Change the arithmetic, the sorting, // a threshold, a field name or a sentence the model is told to repeat, and the // diff turns up here rather than in front of a shopkeeper. // // ── Why whole answers rather than assertions ──────────────────────────────── // // The per-tool tests already assert the things somebody thought to check. These // catch what nobody thought to check — a field quietly renamed, a note that // stopped mentioning truncation, a rounding change in the fourth decimal. An // eval that only checked the numbers it was told to check would have been // written by the same person who wrote the bug. // // ── Reading a failure ─────────────────────────────────────────────────────── // // A failing eval is not automatically a bug: an intended change shows up here // too. Run with `-update` to rewrite the golden files, then READ THE DIFF — it // is the summary of what this change does to every answer the assistant gives. // Committing an update without reading it is how a regression ships wearing the // clothes of an improvement. // // go test ./services/tools/ -run TestEval -update var updateGolden = flag.Bool("update", false, "rewrite the golden answers in testdata") // evalNow is the instant every eval is run at, so "40 minutes ago" is the same // forty minutes in a year's time. var evalNow = time.Date(2026, 9, 23, 14, 0, 0, 0, time.Local) func evalStamp(minutesAgo int) string { return evalNow.Add(-time.Duration(minutesAgo) * time.Minute).Format("2006-01-02 15:04:05") } // evalShop is one fixed merchant, used by every case that needs data. // // Deliberately awkward in the ways production is: a branch with no orders, a // rider column carrying a status instead of a name, a stamp that will not parse. // An eval over tidy data proves the tool works on data that does not exist. type evalShop struct { deliveries []models.Deliveryinfo branches []models.Ordersummarylocation requests []models.StockRequest stocks []models.Productstocks } func theEvalShop() evalShop { return evalShop{ deliveries: []models.Deliveryinfo{ {Deliveryid: 4412, Orderid: "ORD-4412", Orderstatus: "pending", Assigntime: evalStamp(41), Ridername: "Varun", Locationname: "R Mart", Deliverycustomer: "S Kumar"}, {Deliveryid: 4407, Orderid: "ORD-4407", Orderstatus: "pending", Assigntime: evalStamp(33), Ridername: "delivered", Locationname: "R Mart"}, // a status in the name column {Deliveryid: 4419, Orderid: "ORD-4419", Orderstatus: "pending", Assigntime: evalStamp(12), Ridername: "Murali", Locationname: "Anna Nagar"}, {Deliveryid: 4420, Orderid: "ORD-4420", Orderstatus: "pending", Assigntime: evalStamp(4)}, {Deliveryid: 4421, Orderid: "ORD-4421", Orderstatus: "picked", Assigntime: evalStamp(90)}, {Deliveryid: 4422, Orderid: "ORD-4422", Orderstatus: "active", Assigntime: evalStamp(70)}, {Deliveryid: 4423, Orderid: "ORD-4423", Orderstatus: "delivered", Assigntime: evalStamp(200)}, {Deliveryid: 4424, Orderid: "ORD-4424", Orderstatus: "rejected", Assigntime: evalStamp(150)}, {Deliveryid: 4425, Orderid: "ORD-4425", Orderstatus: "pending", Assigntime: "not a date"}, }, branches: []models.Ordersummarylocation{ {Locationid: 1172, Locationname: "R Mart", Total: 240, Delivered: 198, Cancelled: 9}, {Locationid: 1173, Locationname: "Anna Nagar", Total: 180, Delivered: 121, Cancelled: 41}, {Locationid: 1174, Locationname: "New Shop", Total: 0}, }, requests: []models.StockRequest{ {Requestid: 41, Qty: 12, Productname: "Basmati Rice 5kg", Locationname: "R Mart", Status: "Pending", Created: evalNow.AddDate(0, 0, -3)}, {Requestid: 42, Qty: 40, Productname: "Sunflower Oil 1L", Locationname: "Anna Nagar", Status: "Pending", Created: evalNow.AddDate(0, 0, -9)}, }, stocks: []models.Productstocks{ {Productid: 88, Productname: "Atta 10kg", Quantity: 0}, {Productid: 91, Productname: "Sugar 1kg", Quantity: 2}, {Productid: 92, Productname: "Salt 1kg", Quantity: 2}, {Productid: 93, Productname: "Tea 250g", Quantity: 60}, }, } } // evalCase is one question, as a tool call. type evalCase struct { // The file in testdata, and the name in test output. name string // What a person would have asked to get here. Not executed — it is the // reason this case exists, and without it a golden file is a wall of JSON // nobody can review. question string tool func(evalShop) Tool args map[string]any caller Caller } var merchantAll = Caller{Userid: 904, Tenantid: 1147} var merchantBranch = Caller{Userid: 904, Tenantid: 1147, Locationid: 1172} func evalCases() []evalCase { return []evalCase{ { name: "stuck_orders", question: "Which orders are stuck?", tool: func(s evalShop) Tool { return StuckOrders(&fakeDeliveries{rows: s.deliveries}, func() time.Time { return evalNow }) }, caller: merchantAll, }, { name: "stuck_orders_half_an_hour", question: "Anything waiting more than half an hour?", tool: func(s evalShop) Tool { return StuckOrders(&fakeDeliveries{rows: s.deliveries}, func() time.Time { return evalNow }) }, args: map[string]any{"minutes_waiting": 30}, caller: merchantAll, }, { name: "delivery_progress", question: "What is out for delivery?", tool: func(s evalShop) Tool { return DeliveryProgress(&fakeDeliveries{rows: s.deliveries}) }, caller: merchantAll, }, { name: "branch_performance", question: "Which branch is underperforming?", tool: func(s evalShop) Tool { return BranchPerformance(&fakeBranches{rows: s.branches}) }, caller: merchantAll, }, { name: "pending_approvals", question: "What needs my approval?", tool: func(s evalShop) Tool { return PendingApprovals(&fakeApprovals{rows: s.requests}, func() time.Time { return evalNow }) }, caller: merchantAll, }, { name: "low_stock", question: "Where is stock running out?", tool: func(s evalShop) Tool { return LowStock(&fakeStocks{rows: s.stocks}) }, caller: merchantBranch, }, { name: "help_cashier", question: "How do I add a cashier?", tool: func(evalShop) Tool { corpus, err := LoadHelp() if err != nil { panic(err) } return Help(corpus) }, args: map[string]any{"question": "how do I add a cashier"}, caller: merchantAll, }, { name: "help_unanswerable", question: "What is the capital of France?", tool: func(evalShop) Tool { corpus, err := LoadHelp() if err != nil { panic(err) } return Help(corpus) }, args: map[string]any{"question": "what is the capital of france"}, caller: merchantAll, }, } } // golden is the whole answer, in the order a reader wants it. // // `Question` rides along so a diff reads as "this is what changed about the // answer to THAT" rather than as an anonymous blob. type golden struct { Question string `json:"question"` Rows any `json:"rows"` Count int `json:"count"` Truncated bool `json:"truncated,omitempty"` Note string `json:"note,omitempty"` Source string `json:"source,omitempty"` Covers string `json:"covers,omitempty"` } func TestEvalAnswersHaveNotChanged(t *testing.T) { for _, eval := range evalCases() { t.Run(eval.name, func(t *testing.T) { shop := theEvalShop() tool := eval.tool(shop) r := New(nil) if err := r.Register(tool); err != nil { t.Fatalf("registering: %v", err) } result, err := r.Call(context.Background(), Agent{Name: "eval", Tools: []string{tool.Name}}, tool.Name, eval.args, eval.caller) if err != nil { t.Fatalf("%s: %v", eval.name, err) } got, err := json.MarshalIndent(golden{ Question: eval.question, Rows: result.Rows, Count: result.Count, Truncated: result.Truncated, Note: result.Note, Source: result.Source, Covers: result.Scope, }, "", " ") if err != nil { t.Fatalf("encoding: %v", err) } got = append(got, '\n') path := filepath.Join("testdata", eval.name+".json") if *updateGolden { if err := os.MkdirAll("testdata", 0o755); err != nil { t.Fatalf("testdata: %v", err) } if err := os.WriteFile(path, got, 0o600); err != nil { t.Fatalf("writing %s: %v", path, err) } return } want, err := os.ReadFile(path) if err != nil { t.Fatalf("no golden answer for %q — run with -update and read the diff: %v", eval.name, err) } if string(got) != string(want) { t.Fatalf( "the answer to %q changed.\n\nIf that was intended, run:\n go test ./services/tools/ -run TestEval -update\nand read the diff before committing it.\n\nwant:\n%s\ngot:\n%s", eval.question, want, got) } }) } } // TestEveryEvalCaseHasAQuestion keeps the golden files reviewable. // // A case with no question is a wall of JSON nobody can judge, and an // unreviewable golden file gets updated rather than read. func TestEveryEvalCaseHasAQuestion(t *testing.T) { for _, eval := range evalCases() { if eval.question == "" { t.Fatalf("eval %q does not say what was asked", eval.name) } } }