package services import ( "context" "strings" "testing" "time" "nearle/config" "nearle/models" "nearle/services/tools" "nearle/utils" ) // The one test that talks to a real model. // // Everything else in this package runs against a scripted chat, because the // loop's job is to be safe whatever a model does and a scripted one can be made // to misbehave on demand. This is the opposite question: does a REAL model, // handed our tool definitions, pick the right tool and use the answer? // // That cannot be settled by reasoning. Descriptions are the only thing a model // chooses by, and whether ours are good enough is a fact about a particular // model on a particular day. // // ── Skipped unless a key is present ───────────────────────────────────────── // // No provider, no run — so CI stays offline, free and deterministic by default. // Point it at anything OpenAI-compatible: // // ASSISTANT_PROVIDER=openai \ // ASSISTANT_BASE_URL=https://api.groq.com/openai/v1 \ // ASSISTANT_API_KEY=... \ // ASSISTANT_MODEL=openai/gpt-oss-120b \ // go test ./services/ -run TestLive -v // // The key comes from the environment and never from a file in this repository: // `.env.local` is tracked by git, so a secret written there is a secret pushed. func liveChat(t *testing.T) (utils.Chat, string) { t.Helper() // Read exactly as production reads it, defaults and all. // // This was a hand-built struct twice, and it was wrong both times: first it // read ASSISTANT_PROVIDER straight and skipped silently once that stopped // being required, then it kept demanding ASSISTANT_MODEL after that gained // a default. A test that builds its own configuration is a test of a // configuration nobody runs. cfg := config.AssistantFromEnv() if !cfg.Enabled() { t.Skipf("no model configured, skipping the live test: %s", cfg.Why()) } chat, err := utils.NewChat(cfg) if err != nil { t.Fatalf("building the gateway: %v", err) } if chat == nil { t.Skip("gateway not configured") } return chat, cfg.Balanced } // liveShop is a small, unambiguous shop. The point is whether the model reaches // for the right tool, not whether it can summarise a crowd. func liveAssistant(t *testing.T, chat utils.Chat) AssistantService { t.Helper() now := func() time.Time { return time.Date(2026, 9, 24, 14, 0, 0, 0, time.Local) } stamp := func(minutesAgo int) string { return now().Add(-time.Duration(minutesAgo) * time.Minute).Format("2006-01-02 15:04:05") } deliveries := &fakeLiveDeliveries{rows: []models.Deliveryinfo{ {Deliveryid: 4412, Orderid: "ORD-4412", Orderstatus: "pending", Assigntime: stamp(41), Ridername: "Varun", Locationname: "R Mart"}, {Deliveryid: 4419, Orderid: "ORD-4419", Orderstatus: "pending", Assigntime: stamp(12), Ridername: "Murali", Locationname: "Anna Nagar"}, {Deliveryid: 4421, Orderid: "ORD-4421", Orderstatus: "delivered", Assigntime: stamp(200)}, }} corpus, err := tools.LoadHelp() if err != nil { t.Fatalf("help corpus: %v", err) } registry := tools.New(nil) for _, tool := range []tools.Tool{ tools.StuckOrders(deliveries, now), tools.DeliveryProgress(deliveries), tools.Help(corpus), } { if err := registry.Register(tool); err != nil { t.Fatalf("registering %s: %v", tool.Name, err) } } agents, err := LoadAgents("", registry.Has) if err != nil { // The shipped agents name tools this cut-down registry does not hold, // so build one by hand rather than pretending to run the real config. agents = map[string]Agent{"orders": { Name: "orders", Tier: utils.TierBalanced, System: basePrompt, Tools: []string{"stuck_orders", "delivery_progress", "help"}, MaxSteps: 4, MaxToolCalls: 6, }} } return NewAssistantService(registry, chat, agents) } type fakeLiveDeliveries struct{ rows []models.Deliveryinfo } func (f *fakeLiveDeliveries) GetDeliveries(models.DeliveryQuery) []models.Deliveryinfo { return f.rows } var liveMerchant = tools.Caller{Userid: 904, Tenantid: 1147} func TestLiveModelPicksTheRightToolAndAnswersFromIt(t *testing.T) { chat, model := liveChat(t) assistant := liveAssistant(t, chat) ctx, cancel := context.WithTimeout(context.Background(), 90*time.Second) defer cancel() answer, err := assistant.Ask(ctx, "orders", "Which orders are stuck?", liveMerchant) if err != nil { t.Fatalf("asking %s: %v", model, err) } t.Logf("model: %s", answer.Model) t.Logf("used: %+v", answer.Used) t.Logf("reply: %s", answer.Reply) if len(answer.Used) == 0 { t.Fatal("the model answered without calling any tool — it invented the answer") } if answer.Used[0].Tool != "stuck_orders" { t.Fatalf("reached for %q instead of stuck_orders", answer.Used[0].Tool) } if answer.Reply == "" { t.Fatal("a tool ran but nothing came back in words") } // The two waiting jobs are 41 and 12 minutes. An answer that mentions // neither has run the tool and then ignored it, which is worse than not // running it at all. if !strings.Contains(answer.Reply, "41") && !strings.Contains(answer.Reply, "12") { t.Fatalf("the reply does not use the numbers the tool returned: %q", answer.Reply) } } func TestLiveModelUsesHelpForAHowDoIQuestion(t *testing.T) { // The two kinds of question have to route differently, or the help corpus // is decoration. chat, _ := liveChat(t) assistant := liveAssistant(t, chat) ctx, cancel := context.WithTimeout(context.Background(), 90*time.Second) defer cancel() answer, err := assistant.Ask(ctx, "orders", "How do I add a cashier?", liveMerchant) if err != nil { t.Fatalf("asking: %v", err) } t.Logf("used: %+v", answer.Used) t.Logf("reply: %s", answer.Reply) if len(answer.Used) == 0 || answer.Used[0].Tool != "help" { t.Fatalf("a how-do-I question did not reach the help corpus: %+v", answer.Used) } } func TestLiveModelSaysSoWhenNothingCanAnswer(t *testing.T) { // The behaviour the whole design exists to produce: no tool covers this, so // it must decline rather than answer from what it knows about retail. chat, _ := liveChat(t) assistant := liveAssistant(t, chat) ctx, cancel := context.WithTimeout(context.Background(), 90*time.Second) defer cancel() answer, err := assistant.Ask(ctx, "orders", "What was my total revenue last quarter, and what will it be next quarter?", liveMerchant) if err != nil { t.Fatalf("asking: %v", err) } t.Logf("used: %+v", answer.Used) t.Logf("reply: %s", answer.Reply) // Not asserting particular words — models phrase a refusal differently every // time. Asserting the thing that matters: it did not invent a figure. for _, invented := range []string{"₹", "lakh", "crore", "$"} { if strings.Contains(answer.Reply, invented) { t.Fatalf("the model produced a money figure no tool gave it: %q", answer.Reply) } } }