Files
backend_fiesta/services/assistantLive_test.go
2026-09-24 17:20:04 +05:30

199 lines
6.7 KiB
Go

package services
import (
"context"
"strings"
"testing"
"time"
"nearle/config"
"nearle/models"
"nearle/services/tools"
"nearle/utils"
)
// The one test that talks to a real model.
//
// Everything else in this package runs against a scripted chat, because the
// loop's job is to be safe whatever a model does and a scripted one can be made
// to misbehave on demand. This is the opposite question: does a REAL model,
// handed our tool definitions, pick the right tool and use the answer?
//
// That cannot be settled by reasoning. Descriptions are the only thing a model
// chooses by, and whether ours are good enough is a fact about a particular
// model on a particular day.
//
// ── Skipped unless a key is present ─────────────────────────────────────────
//
// No provider, no run — so CI stays offline, free and deterministic by default.
// Point it at anything OpenAI-compatible:
//
// ASSISTANT_PROVIDER=openai \
// ASSISTANT_BASE_URL=https://api.groq.com/openai/v1 \
// ASSISTANT_API_KEY=... \
// ASSISTANT_MODEL=openai/gpt-oss-120b \
// go test ./services/ -run TestLive -v
//
// The key comes from the environment and never from a file in this repository:
// `.env.local` is tracked by git, so a secret written there is a secret pushed.
func liveChat(t *testing.T) (utils.Chat, string) {
t.Helper()
// Read exactly as production reads it, defaults and all.
//
// This was a hand-built struct twice, and it was wrong both times: first it
// read ASSISTANT_PROVIDER straight and skipped silently once that stopped
// being required, then it kept demanding ASSISTANT_MODEL after that gained
// a default. A test that builds its own configuration is a test of a
// configuration nobody runs.
cfg := config.AssistantFromEnv()
if !cfg.Enabled() {
t.Skipf("no model configured, skipping the live test: %s", cfg.Why())
}
chat, err := utils.NewChat(cfg)
if err != nil {
t.Fatalf("building the gateway: %v", err)
}
if chat == nil {
t.Skip("gateway not configured")
}
return chat, cfg.Balanced
}
// liveShop is a small, unambiguous shop. The point is whether the model reaches
// for the right tool, not whether it can summarise a crowd.
func liveAssistant(t *testing.T, chat utils.Chat) AssistantService {
t.Helper()
now := func() time.Time { return time.Date(2026, 9, 24, 14, 0, 0, 0, time.Local) }
stamp := func(minutesAgo int) string {
return now().Add(-time.Duration(minutesAgo) * time.Minute).Format("2006-01-02 15:04:05")
}
deliveries := &fakeLiveDeliveries{rows: []models.Deliveryinfo{
{Deliveryid: 4412, Orderid: "ORD-4412", Orderstatus: "pending", Assigntime: stamp(41),
Ridername: "Varun", Locationname: "R Mart"},
{Deliveryid: 4419, Orderid: "ORD-4419", Orderstatus: "pending", Assigntime: stamp(12),
Ridername: "Murali", Locationname: "Anna Nagar"},
{Deliveryid: 4421, Orderid: "ORD-4421", Orderstatus: "delivered", Assigntime: stamp(200)},
}}
corpus, err := tools.LoadHelp()
if err != nil {
t.Fatalf("help corpus: %v", err)
}
registry := tools.New(nil)
for _, tool := range []tools.Tool{
tools.StuckOrders(deliveries, now),
tools.DeliveryProgress(deliveries),
tools.Help(corpus),
} {
if err := registry.Register(tool); err != nil {
t.Fatalf("registering %s: %v", tool.Name, err)
}
}
agents, err := LoadAgents("", registry.Has)
if err != nil {
// The shipped agents name tools this cut-down registry does not hold,
// so build one by hand rather than pretending to run the real config.
agents = map[string]Agent{"orders": {
Name: "orders", Tier: utils.TierBalanced, System: basePrompt,
Tools: []string{"stuck_orders", "delivery_progress", "help"},
MaxSteps: 4, MaxToolCalls: 6,
}}
}
return NewAssistantService(registry, chat, agents)
}
type fakeLiveDeliveries struct{ rows []models.Deliveryinfo }
func (f *fakeLiveDeliveries) GetDeliveries(models.DeliveryQuery) []models.Deliveryinfo {
return f.rows
}
var liveMerchant = tools.Caller{Userid: 904, Tenantid: 1147}
func TestLiveModelPicksTheRightToolAndAnswersFromIt(t *testing.T) {
chat, model := liveChat(t)
assistant := liveAssistant(t, chat)
ctx, cancel := context.WithTimeout(context.Background(), 90*time.Second)
defer cancel()
answer, err := assistant.Ask(ctx, "orders", "Which orders are stuck?", liveMerchant)
if err != nil {
t.Fatalf("asking %s: %v", model, err)
}
t.Logf("model: %s", answer.Model)
t.Logf("used: %+v", answer.Used)
t.Logf("reply: %s", answer.Reply)
if len(answer.Used) == 0 {
t.Fatal("the model answered without calling any tool — it invented the answer")
}
if answer.Used[0].Tool != "stuck_orders" {
t.Fatalf("reached for %q instead of stuck_orders", answer.Used[0].Tool)
}
if answer.Reply == "" {
t.Fatal("a tool ran but nothing came back in words")
}
// The two waiting jobs are 41 and 12 minutes. An answer that mentions
// neither has run the tool and then ignored it, which is worse than not
// running it at all.
if !strings.Contains(answer.Reply, "41") && !strings.Contains(answer.Reply, "12") {
t.Fatalf("the reply does not use the numbers the tool returned: %q", answer.Reply)
}
}
func TestLiveModelUsesHelpForAHowDoIQuestion(t *testing.T) {
// The two kinds of question have to route differently, or the help corpus
// is decoration.
chat, _ := liveChat(t)
assistant := liveAssistant(t, chat)
ctx, cancel := context.WithTimeout(context.Background(), 90*time.Second)
defer cancel()
answer, err := assistant.Ask(ctx, "orders", "How do I add a cashier?", liveMerchant)
if err != nil {
t.Fatalf("asking: %v", err)
}
t.Logf("used: %+v", answer.Used)
t.Logf("reply: %s", answer.Reply)
if len(answer.Used) == 0 || answer.Used[0].Tool != "help" {
t.Fatalf("a how-do-I question did not reach the help corpus: %+v", answer.Used)
}
}
func TestLiveModelSaysSoWhenNothingCanAnswer(t *testing.T) {
// The behaviour the whole design exists to produce: no tool covers this, so
// it must decline rather than answer from what it knows about retail.
chat, _ := liveChat(t)
assistant := liveAssistant(t, chat)
ctx, cancel := context.WithTimeout(context.Background(), 90*time.Second)
defer cancel()
answer, err := assistant.Ask(ctx, "orders",
"What was my total revenue last quarter, and what will it be next quarter?", liveMerchant)
if err != nil {
t.Fatalf("asking: %v", err)
}
t.Logf("used: %+v", answer.Used)
t.Logf("reply: %s", answer.Reply)
// Not asserting particular words — models phrase a refusal differently every
// time. Asserting the thing that matters: it did not invent a figure.
for _, invented := range []string{"₹", "lakh", "crore", "$"} {
if strings.Contains(answer.Reply, invented) {
t.Fatalf("the model produced a money figure no tool gave it: %q", answer.Reply)
}
}
}