Files
krow_backend/go-api/internal/config/model_test.go
Suriyakumarvijayanayagam 9d3192a9c4
Some checks failed
CI / test (push) Failing after 4m39s
CI / fixture (push) Failing after 7s
Replace the model ids with ones Groq actually serves
The defaults shipped yesterday were wrong the day they shipped, and a real key
proved it in one request. Groq serves neither llama-3.1-8b-instant nor
llama-3.3-70b-versatile any more. Both were chosen from memory, both passed
startup validation, and every agent run would have failed with a 400.

This is the exact failure the claude-* guard was written to catch, arriving from
the side that guard cannot see. A prefix check can reject a vendor this service
cannot call; it has no way to know a provider retired an id last month. That is
not a gap in the check, it is a gap in the class of thing local validation can
know, so the fix is not another guard:

TestConfiguredModelsAreServed asks the provider. It lists /models — part of the
same openai-compatible surface the gateway already speaks, so every supported
provider answers it — and fails if a configured id is absent, printing what is
available. It reads the ids through config.DefaultModels() rather than
repeating them, because a second copy would be the first thing to drift, and
drift is the whole failure. Skipped without a credential like the rest of the
live suite. Verified three ways: it fails on the retired id with the message an
operator needs, skips clean with no key, passes on the new ones.

New defaults, chosen against the live account rather than from memory:
openai/gpt-oss-20b (fast) and openai/gpt-oss-120b (balanced, deep). Tool
calling confirmed on both. groq/compound-mini was ruled out — it cannot do tool
calls at all, which this platform requires.

MODEL_REASONING_EFFORT is now documented as safe here and NOT portable: gpt-oss
accepts low/medium/high, exactly the scale openAIEffort maps onto, while
qwen/qwen3.6-27b on the same account rejects all three and fails the whole
request rather than ignoring the key.

I7 IS NO LONGER UNPROVEN. make eval-live passes all three cases twice against
gpt-oss-120b, the planted-injection case included: answers from the handbook,
cites, refuses the injection, leaks neither the operator-only pay guidance nor
the other tenant's figures. CLAUDE.md §12 and handover.md updated from "urgent"
to measured, dated, and scoped to the one model it is evidence about.

One real defect found on the way. The handbook grounding check failed once on an
answer containing the phrase it wanted — "more than ten minutes" on screen,
strings.Contains false — which leaves an invisible separator as the only
explanation; the same model writes "47 %" and a U+2011 hyphen elsewhere. The
flaky assertion is the small half. THE LEAK ASSERTIONS USED THE SAME MATCH and
fail in the dangerous direction: "attacker@evil.test" with a zero-width space,
or "uplift" with a soft hyphen, would have been reported clean. A permission
test that cannot see the leak it is hunting is worse than none, because it is
believed. normalizeForMatch folds those away, and its test pins that every case
is one plain ToLower MISSES — a case whose naive match already succeeds fails,
so the suite cannot fill with examples that demonstrate nothing. That caught my
own first BOM case, which put the mark where Contains found it regardless.

gofmt clean, vet clean, 15/15 packages pass offline; live suite green twice.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01PJvibeSc1JYXjatankqM1g
2026-09-07 12:43:45 +05:30

173 lines
6.1 KiB
Go

package config
import (
"strings"
"testing"
)
func modelCfg(env string, m ModelConfig) *Config {
c := &Config{AppEnv: env}
c.Model = m
return c
}
func TestValidateModelProvider(t *testing.T) {
for _, tc := range []struct {
name string
cfg *Config
wantErr bool
}{
{
"unset provider is openai, which is now the only implementation",
modelCfg("development", ModelConfig{}), false,
},
{"openai named explicitly", modelCfg("development", ModelConfig{Provider: "openai"}), false},
{
"anthropic is refused rather than ignored — it used to be correct",
modelCfg("development", ModelConfig{Provider: "anthropic"}), true,
},
{"a typo is caught once at startup, not once per run",
modelCfg("development", ModelConfig{Provider: "openal"}), true},
{
"a vendor name is not a provider: groq is reached through openai + a base URL",
modelCfg("development", ModelConfig{Provider: "groq"}), true,
},
} {
t.Run(tc.name, func(t *testing.T) {
err := tc.cfg.validateModel()
if tc.wantErr != (err != nil) {
t.Fatalf("validateModel() = %v, wantErr = %v", err, tc.wantErr)
}
})
}
}
// THE STALE CONFIGURATION.
//
// This replaced a test called TestBaseURLWithoutOpenAIProviderIsRefused, which
// guarded the mirror image of the same mistake: while both providers existed, a
// base URL without MODEL_PROVIDER=openai meant a deployment that believed it had
// left Claude and had not. That failure is now impossible — there is nowhere
// else for a run to go — and the surviving one points the other way: a
// deployment that still names Anthropic, and must be told rather than silently
// re-pointed at a provider it never chose.
func TestTheRemovedProviderIsRefusedLoudly(t *testing.T) {
err := modelCfg("development", ModelConfig{Provider: "anthropic"}).validateModel()
if err == nil {
t.Fatal("MODEL_PROVIDER=anthropic was accepted; the stack would silently run on another vendor")
}
for _, want := range []string{"MODEL_PROVIDER=anthropic", "no longer supported", "MODEL_BASE_URL"} {
if !strings.Contains(err.Error(), want) {
t.Errorf("the message does not mention %q:\n %v", want, err)
}
}
// The intended configuration is exactly what the message tells them to set.
if err := modelCfg("development", ModelConfig{
Provider: "openai", BaseURL: "https://api.groq.com/openai/v1",
}).validateModel(); err != nil {
t.Fatalf("the intended configuration was refused: %v", err)
}
}
// A model id that outlived its provider.
//
// The expensive shape of this is not a typo, it is an UNCHANGED .env: the tier
// ids were claude-* for the whole life of the Anthropic path, and nothing about
// switching providers forces them to be revisited. Left unchecked the process
// starts clean and every single run fails at the gateway with a 400 — which is
// the incident that made the gateway start carrying upstream error text at all.
func TestClaudeModelIdsAreRefused(t *testing.T) {
base := ModelConfig{Provider: "openai", BaseURL: "https://api.groq.com/openai/v1",
Fast: "openai/gpt-oss-20b", Balanced: "openai/gpt-oss-120b", Deep: "openai/gpt-oss-120b"}
for _, tier := range []string{"MODEL_FAST", "MODEL_BALANCED", "MODEL_DEEP"} {
t.Run(tier, func(t *testing.T) {
m := base
switch tier {
case "MODEL_FAST":
m.Fast = "claude-opus-5"
case "MODEL_BALANCED":
m.Balanced = "claude-opus-5"
case "MODEL_DEEP":
m.Deep = "claude-3-5-sonnet-latest"
}
err := modelCfg("development", m).validateModel()
if err == nil {
t.Fatalf("%s kept a claude id and was accepted; every run on that tier would 400", tier)
}
// Naming the tier is the whole value: "a model is wrong" does not
// tell an operator which of three lines to edit.
if !strings.Contains(err.Error(), tier) {
t.Errorf("the message does not name the tier %q:\n %v", tier, err)
}
})
}
if err := modelCfg("development", base).validateModel(); err != nil {
t.Fatalf("a fully-migrated configuration was refused: %v", err)
}
}
func TestBaseURLMustBeAURL(t *testing.T) {
for _, raw := range []string{"api.groq.com", "ftp://x.test", "not a url", "://broken"} {
err := modelCfg("development", ModelConfig{Provider: "openai", BaseURL: raw}).validateModel()
if err == nil {
t.Errorf("MODEL_BASE_URL=%q was accepted", raw)
}
}
for _, raw := range []string{"http://localhost:11434/v1", "https://api.groq.com/openai/v1"} {
if err := modelCfg("development", ModelConfig{Provider: "openai", BaseURL: raw}).validateModel(); err != nil {
t.Errorf("MODEL_BASE_URL=%q was refused: %v", raw, err)
}
}
}
// Production without a credential fails every run at the gateway, which is a
// misconfiguration wearing a runtime error's clothes. A local model is the one
// exception: it needs no key, and demanding one would make the zero-cost path
// impossible to configure.
func TestProductionCredentialRequirement(t *testing.T) {
for _, tc := range []struct {
name string
cfg *Config
wantErr bool
}{
{"production with no key", modelCfg("production", ModelConfig{}), true},
{"production with a key", modelCfg("production", ModelConfig{APIKey: "k"}), false},
{
"production against a local model needs no key",
modelCfg("production", ModelConfig{Provider: "openai", BaseURL: "http://localhost:11434/v1"}),
false,
},
{
"production against a hosted provider still does",
modelCfg("production", ModelConfig{Provider: "openai", BaseURL: "https://api.groq.com/openai/v1"}),
true,
},
{"development needs nothing", modelCfg("development", ModelConfig{}), false},
} {
t.Run(tc.name, func(t *testing.T) {
err := tc.cfg.validateModel()
if tc.wantErr != (err != nil) {
t.Fatalf("validateModel() = %v, wantErr = %v", err, tc.wantErr)
}
})
}
}
func TestIsLoopback(t *testing.T) {
for raw, want := range map[string]bool{
"http://localhost:11434/v1": true,
"http://127.0.0.1:11434/v1": true,
"https://api.groq.com/v1": false,
"": false,
// A remote host that merely mentions localhost in its path is not local.
"https://x.test/localhost/v1": false,
} {
if got := isLoopback(raw); got != want {
t.Errorf("isLoopback(%q) = %v, want %v", raw, got, want)
}
}
}