263 lines
9.4 KiB
Go
263 lines
9.4 KiB
Go
package tools
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"flag"
|
|
"os"
|
|
"path/filepath"
|
|
"testing"
|
|
"time"
|
|
|
|
"nearle/models"
|
|
)
|
|
|
|
// Evals: the answers this assistant is known to give.
|
|
//
|
|
// Every case below is a fixed shop, a fixed clock and a real tool, with its
|
|
// whole answer written down in `testdata`. Change the arithmetic, the sorting,
|
|
// a threshold, a field name or a sentence the model is told to repeat, and the
|
|
// diff turns up here rather than in front of a shopkeeper.
|
|
//
|
|
// ── Why whole answers rather than assertions ────────────────────────────────
|
|
//
|
|
// The per-tool tests already assert the things somebody thought to check. These
|
|
// catch what nobody thought to check — a field quietly renamed, a note that
|
|
// stopped mentioning truncation, a rounding change in the fourth decimal. An
|
|
// eval that only checked the numbers it was told to check would have been
|
|
// written by the same person who wrote the bug.
|
|
//
|
|
// ── Reading a failure ───────────────────────────────────────────────────────
|
|
//
|
|
// A failing eval is not automatically a bug: an intended change shows up here
|
|
// too. Run with `-update` to rewrite the golden files, then READ THE DIFF — it
|
|
// is the summary of what this change does to every answer the assistant gives.
|
|
// Committing an update without reading it is how a regression ships wearing the
|
|
// clothes of an improvement.
|
|
//
|
|
// go test ./services/tools/ -run TestEval -update
|
|
|
|
var updateGolden = flag.Bool("update", false, "rewrite the golden answers in testdata")
|
|
|
|
// evalNow is the instant every eval is run at, so "40 minutes ago" is the same
|
|
// forty minutes in a year's time.
|
|
var evalNow = time.Date(2026, 9, 23, 14, 0, 0, 0, time.Local)
|
|
|
|
func evalStamp(minutesAgo int) string {
|
|
return evalNow.Add(-time.Duration(minutesAgo) * time.Minute).Format("2006-01-02 15:04:05")
|
|
}
|
|
|
|
// evalShop is one fixed merchant, used by every case that needs data.
|
|
//
|
|
// Deliberately awkward in the ways production is: a branch with no orders, a
|
|
// rider column carrying a status instead of a name, a stamp that will not parse.
|
|
// An eval over tidy data proves the tool works on data that does not exist.
|
|
type evalShop struct {
|
|
deliveries []models.Deliveryinfo
|
|
branches []models.Ordersummarylocation
|
|
requests []models.StockRequest
|
|
stocks []models.Productstocks
|
|
}
|
|
|
|
func theEvalShop() evalShop {
|
|
return evalShop{
|
|
deliveries: []models.Deliveryinfo{
|
|
{Deliveryid: 4412, Orderid: "ORD-4412", Orderstatus: "pending", Assigntime: evalStamp(41),
|
|
Ridername: "Varun", Locationname: "R Mart", Deliverycustomer: "S Kumar"},
|
|
{Deliveryid: 4407, Orderid: "ORD-4407", Orderstatus: "pending", Assigntime: evalStamp(33),
|
|
Ridername: "delivered", Locationname: "R Mart"}, // a status in the name column
|
|
{Deliveryid: 4419, Orderid: "ORD-4419", Orderstatus: "pending", Assigntime: evalStamp(12),
|
|
Ridername: "Murali", Locationname: "Anna Nagar"},
|
|
{Deliveryid: 4420, Orderid: "ORD-4420", Orderstatus: "pending", Assigntime: evalStamp(4)},
|
|
{Deliveryid: 4421, Orderid: "ORD-4421", Orderstatus: "picked", Assigntime: evalStamp(90)},
|
|
{Deliveryid: 4422, Orderid: "ORD-4422", Orderstatus: "active", Assigntime: evalStamp(70)},
|
|
{Deliveryid: 4423, Orderid: "ORD-4423", Orderstatus: "delivered", Assigntime: evalStamp(200)},
|
|
{Deliveryid: 4424, Orderid: "ORD-4424", Orderstatus: "rejected", Assigntime: evalStamp(150)},
|
|
{Deliveryid: 4425, Orderid: "ORD-4425", Orderstatus: "pending", Assigntime: "not a date"},
|
|
},
|
|
branches: []models.Ordersummarylocation{
|
|
{Locationid: 1172, Locationname: "R Mart", Total: 240, Delivered: 198, Cancelled: 9},
|
|
{Locationid: 1173, Locationname: "Anna Nagar", Total: 180, Delivered: 121, Cancelled: 41},
|
|
{Locationid: 1174, Locationname: "New Shop", Total: 0},
|
|
},
|
|
requests: []models.StockRequest{
|
|
{Requestid: 41, Qty: 12, Productname: "Basmati Rice 5kg", Locationname: "R Mart",
|
|
Status: "Pending", Created: evalNow.AddDate(0, 0, -3)},
|
|
{Requestid: 42, Qty: 40, Productname: "Sunflower Oil 1L", Locationname: "Anna Nagar",
|
|
Status: "Pending", Created: evalNow.AddDate(0, 0, -9)},
|
|
},
|
|
stocks: []models.Productstocks{
|
|
{Productid: 88, Productname: "Atta 10kg", Quantity: 0},
|
|
{Productid: 91, Productname: "Sugar 1kg", Quantity: 2},
|
|
{Productid: 92, Productname: "Salt 1kg", Quantity: 2},
|
|
{Productid: 93, Productname: "Tea 250g", Quantity: 60},
|
|
},
|
|
}
|
|
}
|
|
|
|
// evalCase is one question, as a tool call.
|
|
type evalCase struct {
|
|
// The file in testdata, and the name in test output.
|
|
name string
|
|
// What a person would have asked to get here. Not executed — it is the
|
|
// reason this case exists, and without it a golden file is a wall of JSON
|
|
// nobody can review.
|
|
question string
|
|
tool func(evalShop) Tool
|
|
args map[string]any
|
|
caller Caller
|
|
}
|
|
|
|
var merchantAll = Caller{Userid: 904, Tenantid: 1147}
|
|
var merchantBranch = Caller{Userid: 904, Tenantid: 1147, Locationid: 1172}
|
|
|
|
func evalCases() []evalCase {
|
|
return []evalCase{
|
|
{
|
|
name: "stuck_orders",
|
|
question: "Which orders are stuck?",
|
|
tool: func(s evalShop) Tool {
|
|
return StuckOrders(&fakeDeliveries{rows: s.deliveries}, func() time.Time { return evalNow })
|
|
},
|
|
caller: merchantAll,
|
|
},
|
|
{
|
|
name: "stuck_orders_half_an_hour",
|
|
question: "Anything waiting more than half an hour?",
|
|
tool: func(s evalShop) Tool {
|
|
return StuckOrders(&fakeDeliveries{rows: s.deliveries}, func() time.Time { return evalNow })
|
|
},
|
|
args: map[string]any{"minutes_waiting": 30},
|
|
caller: merchantAll,
|
|
},
|
|
{
|
|
name: "delivery_progress",
|
|
question: "What is out for delivery?",
|
|
tool: func(s evalShop) Tool { return DeliveryProgress(&fakeDeliveries{rows: s.deliveries}) },
|
|
caller: merchantAll,
|
|
},
|
|
{
|
|
name: "branch_performance",
|
|
question: "Which branch is underperforming?",
|
|
tool: func(s evalShop) Tool { return BranchPerformance(&fakeBranches{rows: s.branches}) },
|
|
caller: merchantAll,
|
|
},
|
|
{
|
|
name: "pending_approvals",
|
|
question: "What needs my approval?",
|
|
tool: func(s evalShop) Tool {
|
|
return PendingApprovals(&fakeApprovals{rows: s.requests}, func() time.Time { return evalNow })
|
|
},
|
|
caller: merchantAll,
|
|
},
|
|
{
|
|
name: "low_stock",
|
|
question: "Where is stock running out?",
|
|
tool: func(s evalShop) Tool { return LowStock(&fakeStocks{rows: s.stocks}) },
|
|
caller: merchantBranch,
|
|
},
|
|
{
|
|
name: "help_cashier",
|
|
question: "How do I add a cashier?",
|
|
tool: func(evalShop) Tool {
|
|
corpus, err := LoadHelp()
|
|
if err != nil {
|
|
panic(err)
|
|
}
|
|
return Help(corpus)
|
|
},
|
|
args: map[string]any{"question": "how do I add a cashier"},
|
|
caller: merchantAll,
|
|
},
|
|
{
|
|
name: "help_unanswerable",
|
|
question: "What is the capital of France?",
|
|
tool: func(evalShop) Tool {
|
|
corpus, err := LoadHelp()
|
|
if err != nil {
|
|
panic(err)
|
|
}
|
|
return Help(corpus)
|
|
},
|
|
args: map[string]any{"question": "what is the capital of france"},
|
|
caller: merchantAll,
|
|
},
|
|
}
|
|
}
|
|
|
|
// golden is the whole answer, in the order a reader wants it.
|
|
//
|
|
// `Question` rides along so a diff reads as "this is what changed about the
|
|
// answer to THAT" rather than as an anonymous blob.
|
|
type golden struct {
|
|
Question string `json:"question"`
|
|
Rows any `json:"rows"`
|
|
Count int `json:"count"`
|
|
Truncated bool `json:"truncated,omitempty"`
|
|
Note string `json:"note,omitempty"`
|
|
Source string `json:"source,omitempty"`
|
|
Covers string `json:"covers,omitempty"`
|
|
}
|
|
|
|
func TestEvalAnswersHaveNotChanged(t *testing.T) {
|
|
for _, eval := range evalCases() {
|
|
t.Run(eval.name, func(t *testing.T) {
|
|
shop := theEvalShop()
|
|
tool := eval.tool(shop)
|
|
|
|
r := New(nil)
|
|
if err := r.Register(tool); err != nil {
|
|
t.Fatalf("registering: %v", err)
|
|
}
|
|
result, err := r.Call(context.Background(),
|
|
Agent{Name: "eval", Tools: []string{tool.Name}}, tool.Name, eval.args, eval.caller)
|
|
if err != nil {
|
|
t.Fatalf("%s: %v", eval.name, err)
|
|
}
|
|
|
|
got, err := json.MarshalIndent(golden{
|
|
Question: eval.question, Rows: result.Rows, Count: result.Count,
|
|
Truncated: result.Truncated, Note: result.Note,
|
|
Source: result.Source, Covers: result.Scope,
|
|
}, "", " ")
|
|
if err != nil {
|
|
t.Fatalf("encoding: %v", err)
|
|
}
|
|
got = append(got, '\n')
|
|
|
|
path := filepath.Join("testdata", eval.name+".json")
|
|
if *updateGolden {
|
|
if err := os.MkdirAll("testdata", 0o755); err != nil {
|
|
t.Fatalf("testdata: %v", err)
|
|
}
|
|
if err := os.WriteFile(path, got, 0o600); err != nil {
|
|
t.Fatalf("writing %s: %v", path, err)
|
|
}
|
|
return
|
|
}
|
|
|
|
want, err := os.ReadFile(path)
|
|
if err != nil {
|
|
t.Fatalf("no golden answer for %q — run with -update and read the diff: %v", eval.name, err)
|
|
}
|
|
if string(got) != string(want) {
|
|
t.Fatalf(
|
|
"the answer to %q changed.\n\nIf that was intended, run:\n go test ./services/tools/ -run TestEval -update\nand read the diff before committing it.\n\nwant:\n%s\ngot:\n%s",
|
|
eval.question, want, got)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// TestEveryEvalCaseHasAQuestion keeps the golden files reviewable.
|
|
//
|
|
// A case with no question is a wall of JSON nobody can judge, and an
|
|
// unreviewable golden file gets updated rather than read.
|
|
func TestEveryEvalCaseHasAQuestion(t *testing.T) {
|
|
for _, eval := range evalCases() {
|
|
if eval.question == "" {
|
|
t.Fatalf("eval %q does not say what was asked", eval.name)
|
|
}
|
|
}
|
|
}
|