package evals_test import ( "context" "encoding/json" "fmt" "net/http" "os" "sort" "strings" "testing" "time" "github.com/krow/krow-backend/go-api/internal/config" ) // TestConfiguredModelsAreServed asks the provider whether it still serves the // three ids this deployment is configured with. // // THIS TEST EXISTS BECAUSE THE DEFAULTS WERE WRONG THE DAY THEY SHIPPED. The // gateway was pointed at Groq with llama-3.1-8b-instant and // llama-3.3-70b-versatile, both chosen from memory and neither served by Groq // any more. Startup validation passed — it can reject a claude-* prefix, but // "an id this provider retired" is not a property of the string — so the // configuration booted clean and would have failed every single agent run with // a 400. // // That is the shape of the failure worth defending against, and it is not a // one-off: model ids are retired on the provider's schedule, not this repo's, so // a configuration that is correct today goes stale without anything here // changing. No amount of local validation can see it. Only asking can. // // Skipped without a credential, like the rest of the live suite, so // `go test ./...` stays green offline and the scripted suites remain the gate. func TestConfiguredModelsAreServed(t *testing.T) { key := strings.TrimSpace(os.Getenv("MODEL_API_KEY")) baseURL := strings.TrimSpace(os.Getenv("MODEL_BASE_URL")) if baseURL == "" { baseURL = "https://api.groq.com/openai/v1" } if key == "" { t.Skip("no MODEL_API_KEY; the live suite is skipped") } // The ids this deployment would actually use: an explicit override if the // environment carries one, otherwise the shipped default. Both are worth // checking — an override is just as capable of naming a retired model, and // is likelier to, having been written by hand. fast, balanced, deep := config.DefaultModels() effective := func(env, dflt string) string { if v := strings.TrimSpace(os.Getenv(env)); v != "" { return v } return dflt } served, err := servedModels(baseURL, key) if err != nil { t.Skipf("could not list models at %s: %v", baseURL, err) } if len(served) == 0 { t.Skipf("%s returned no models; nothing to check against", baseURL) } for _, m := range []struct{ key, id string }{ {"MODEL_FAST", effective("MODEL_FAST", fast)}, {"MODEL_BALANCED", effective("MODEL_BALANCED", balanced)}, {"MODEL_DEEP", effective("MODEL_DEEP", deep)}, } { if !served[m.id] { available := make([]string, 0, len(served)) for id := range served { available = append(available, id) } sort.Strings(available) t.Errorf("%s is %q, which %s does not serve.\n"+ "Every run on this tier would fail with a 400 that no local check can predict.\n"+ "Available: %s", m.key, m.id, baseURL, strings.Join(available, ", ")) } } } // servedModels lists the model ids the provider will accept. // // GET /models is part of the same openai-compatible surface the gateway already // speaks, so every provider this platform supports answers it. func servedModels(baseURL, key string) (map[string]bool, error) { ctx, cancel := context.WithTimeout(context.Background(), 20*time.Second) defer cancel() req, err := http.NewRequestWithContext(ctx, http.MethodGet, strings.TrimSuffix(baseURL, "/")+"/models", nil) if err != nil { return nil, err } req.Header.Set("Authorization", "Bearer "+key) resp, err := http.DefaultClient.Do(req) if err != nil { return nil, err } defer resp.Body.Close() if resp.StatusCode != http.StatusOK { return nil, fmt.Errorf("http %d", resp.StatusCode) } var body struct { Data []struct { ID string `json:"id"` } `json:"data"` } if err := json.NewDecoder(resp.Body).Decode(&body); err != nil { return nil, err } served := make(map[string]bool, len(body.Data)) for _, m := range body.Data { served[m.ID] = true } return served, nil }