209 lines
7.0 KiB
Go
209 lines
7.0 KiB
Go
package services
|
|
|
|
import (
|
|
"context"
|
|
"os"
|
|
"strings"
|
|
"testing"
|
|
"time"
|
|
|
|
"nearle/config"
|
|
"nearle/models"
|
|
"nearle/services/tools"
|
|
"nearle/utils"
|
|
)
|
|
|
|
// The one test that talks to a real model.
|
|
//
|
|
// Everything else in this package runs against a scripted chat, because the
|
|
// loop's job is to be safe whatever a model does and a scripted one can be made
|
|
// to misbehave on demand. This is the opposite question: does a REAL model,
|
|
// handed our tool definitions, pick the right tool and use the answer?
|
|
//
|
|
// That cannot be settled by reasoning. Descriptions are the only thing a model
|
|
// chooses by, and whether ours are good enough is a fact about a particular
|
|
// model on a particular day.
|
|
//
|
|
// ── Skipped unless a key is present ─────────────────────────────────────────
|
|
//
|
|
// No provider, no run — so CI stays offline, free and deterministic by default.
|
|
// Point it at anything OpenAI-compatible:
|
|
//
|
|
// ASSISTANT_PROVIDER=openai \
|
|
// ASSISTANT_BASE_URL=https://api.groq.com/openai/v1 \
|
|
// ASSISTANT_API_KEY=... \
|
|
// ASSISTANT_MODEL=openai/gpt-oss-120b \
|
|
// go test ./services/ -run TestLive -v
|
|
//
|
|
// The key comes from the environment and never from a file in this repository:
|
|
// `.env.local` is tracked by git, so a secret written there is a secret pushed.
|
|
|
|
func liveChat(t *testing.T) (utils.Chat, string) {
|
|
t.Helper()
|
|
|
|
// Provider defaults exactly as production does, so this exercises the
|
|
// defaulting rather than working around it. The first version read the
|
|
// variable straight and skipped silently the moment `ASSISTANT_PROVIDER`
|
|
// was left unset — which is precisely the configuration this is meant to
|
|
// prove works.
|
|
provider := strings.ToLower(strings.TrimSpace(os.Getenv("ASSISTANT_PROVIDER")))
|
|
model := os.Getenv("ASSISTANT_MODEL")
|
|
if provider == "" && strings.TrimSpace(model) != "" {
|
|
provider = "openai"
|
|
}
|
|
|
|
cfg := config.AssistantConfig{
|
|
Provider: provider,
|
|
BaseURL: os.Getenv("ASSISTANT_BASE_URL"),
|
|
APIKey: os.Getenv("ASSISTANT_API_KEY"),
|
|
Balanced: model,
|
|
}
|
|
if !cfg.Enabled() {
|
|
t.Skipf("no model configured, skipping the live test: %s", cfg.Why())
|
|
}
|
|
|
|
chat, err := utils.NewChat(cfg)
|
|
if err != nil {
|
|
t.Fatalf("building the gateway: %v", err)
|
|
}
|
|
if chat == nil {
|
|
t.Skip("gateway not configured")
|
|
}
|
|
return chat, cfg.Balanced
|
|
}
|
|
|
|
// liveShop is a small, unambiguous shop. The point is whether the model reaches
|
|
// for the right tool, not whether it can summarise a crowd.
|
|
func liveAssistant(t *testing.T, chat utils.Chat) AssistantService {
|
|
t.Helper()
|
|
|
|
now := func() time.Time { return time.Date(2026, 9, 24, 14, 0, 0, 0, time.Local) }
|
|
stamp := func(minutesAgo int) string {
|
|
return now().Add(-time.Duration(minutesAgo) * time.Minute).Format("2006-01-02 15:04:05")
|
|
}
|
|
|
|
deliveries := &fakeLiveDeliveries{rows: []models.Deliveryinfo{
|
|
{Deliveryid: 4412, Orderid: "ORD-4412", Orderstatus: "pending", Assigntime: stamp(41),
|
|
Ridername: "Varun", Locationname: "R Mart"},
|
|
{Deliveryid: 4419, Orderid: "ORD-4419", Orderstatus: "pending", Assigntime: stamp(12),
|
|
Ridername: "Murali", Locationname: "Anna Nagar"},
|
|
{Deliveryid: 4421, Orderid: "ORD-4421", Orderstatus: "delivered", Assigntime: stamp(200)},
|
|
}}
|
|
|
|
corpus, err := tools.LoadHelp()
|
|
if err != nil {
|
|
t.Fatalf("help corpus: %v", err)
|
|
}
|
|
|
|
registry := tools.New(nil)
|
|
for _, tool := range []tools.Tool{
|
|
tools.StuckOrders(deliveries, now),
|
|
tools.DeliveryProgress(deliveries),
|
|
tools.Help(corpus),
|
|
} {
|
|
if err := registry.Register(tool); err != nil {
|
|
t.Fatalf("registering %s: %v", tool.Name, err)
|
|
}
|
|
}
|
|
|
|
agents, err := LoadAgents("", registry.Has)
|
|
if err != nil {
|
|
// The shipped agents name tools this cut-down registry does not hold,
|
|
// so build one by hand rather than pretending to run the real config.
|
|
agents = map[string]Agent{"orders": {
|
|
Name: "orders", Tier: utils.TierBalanced, System: basePrompt,
|
|
Tools: []string{"stuck_orders", "delivery_progress", "help"},
|
|
MaxSteps: 4, MaxToolCalls: 6,
|
|
}}
|
|
}
|
|
return NewAssistantService(registry, chat, agents)
|
|
}
|
|
|
|
type fakeLiveDeliveries struct{ rows []models.Deliveryinfo }
|
|
|
|
func (f *fakeLiveDeliveries) GetDeliveries(models.DeliveryQuery) []models.Deliveryinfo {
|
|
return f.rows
|
|
}
|
|
|
|
var liveMerchant = tools.Caller{Userid: 904, Tenantid: 1147}
|
|
|
|
func TestLiveModelPicksTheRightToolAndAnswersFromIt(t *testing.T) {
|
|
chat, model := liveChat(t)
|
|
assistant := liveAssistant(t, chat)
|
|
|
|
ctx, cancel := context.WithTimeout(context.Background(), 90*time.Second)
|
|
defer cancel()
|
|
|
|
answer, err := assistant.Ask(ctx, "orders", "Which orders are stuck?", liveMerchant)
|
|
if err != nil {
|
|
t.Fatalf("asking %s: %v", model, err)
|
|
}
|
|
|
|
t.Logf("model: %s", answer.Model)
|
|
t.Logf("used: %+v", answer.Used)
|
|
t.Logf("reply: %s", answer.Reply)
|
|
|
|
if len(answer.Used) == 0 {
|
|
t.Fatal("the model answered without calling any tool — it invented the answer")
|
|
}
|
|
if answer.Used[0].Tool != "stuck_orders" {
|
|
t.Fatalf("reached for %q instead of stuck_orders", answer.Used[0].Tool)
|
|
}
|
|
if answer.Reply == "" {
|
|
t.Fatal("a tool ran but nothing came back in words")
|
|
}
|
|
// The two waiting jobs are 41 and 12 minutes. An answer that mentions
|
|
// neither has run the tool and then ignored it, which is worse than not
|
|
// running it at all.
|
|
if !strings.Contains(answer.Reply, "41") && !strings.Contains(answer.Reply, "12") {
|
|
t.Fatalf("the reply does not use the numbers the tool returned: %q", answer.Reply)
|
|
}
|
|
}
|
|
|
|
func TestLiveModelUsesHelpForAHowDoIQuestion(t *testing.T) {
|
|
// The two kinds of question have to route differently, or the help corpus
|
|
// is decoration.
|
|
chat, _ := liveChat(t)
|
|
assistant := liveAssistant(t, chat)
|
|
|
|
ctx, cancel := context.WithTimeout(context.Background(), 90*time.Second)
|
|
defer cancel()
|
|
|
|
answer, err := assistant.Ask(ctx, "orders", "How do I add a cashier?", liveMerchant)
|
|
if err != nil {
|
|
t.Fatalf("asking: %v", err)
|
|
}
|
|
t.Logf("used: %+v", answer.Used)
|
|
t.Logf("reply: %s", answer.Reply)
|
|
|
|
if len(answer.Used) == 0 || answer.Used[0].Tool != "help" {
|
|
t.Fatalf("a how-do-I question did not reach the help corpus: %+v", answer.Used)
|
|
}
|
|
}
|
|
|
|
func TestLiveModelSaysSoWhenNothingCanAnswer(t *testing.T) {
|
|
// The behaviour the whole design exists to produce: no tool covers this, so
|
|
// it must decline rather than answer from what it knows about retail.
|
|
chat, _ := liveChat(t)
|
|
assistant := liveAssistant(t, chat)
|
|
|
|
ctx, cancel := context.WithTimeout(context.Background(), 90*time.Second)
|
|
defer cancel()
|
|
|
|
answer, err := assistant.Ask(ctx, "orders",
|
|
"What was my total revenue last quarter, and what will it be next quarter?", liveMerchant)
|
|
if err != nil {
|
|
t.Fatalf("asking: %v", err)
|
|
}
|
|
t.Logf("used: %+v", answer.Used)
|
|
t.Logf("reply: %s", answer.Reply)
|
|
|
|
// Not asserting particular words — models phrase a refusal differently every
|
|
// time. Asserting the thing that matters: it did not invent a figure.
|
|
for _, invented := range []string{"₹", "lakh", "crore", "$"} {
|
|
if strings.Contains(answer.Reply, invented) {
|
|
t.Fatalf("the model produced a money figure no tool gave it: %q", answer.Reply)
|
|
}
|
|
}
|
|
}
|