The defaults shipped yesterday were wrong the day they shipped, and a real key proved it in one request. Groq serves neither llama-3.1-8b-instant nor llama-3.3-70b-versatile any more. Both were chosen from memory, both passed startup validation, and every agent run would have failed with a 400. This is the exact failure the claude-* guard was written to catch, arriving from the side that guard cannot see. A prefix check can reject a vendor this service cannot call; it has no way to know a provider retired an id last month. That is not a gap in the check, it is a gap in the class of thing local validation can know, so the fix is not another guard: TestConfiguredModelsAreServed asks the provider. It lists /models — part of the same openai-compatible surface the gateway already speaks, so every supported provider answers it — and fails if a configured id is absent, printing what is available. It reads the ids through config.DefaultModels() rather than repeating them, because a second copy would be the first thing to drift, and drift is the whole failure. Skipped without a credential like the rest of the live suite. Verified three ways: it fails on the retired id with the message an operator needs, skips clean with no key, passes on the new ones. New defaults, chosen against the live account rather than from memory: openai/gpt-oss-20b (fast) and openai/gpt-oss-120b (balanced, deep). Tool calling confirmed on both. groq/compound-mini was ruled out — it cannot do tool calls at all, which this platform requires. MODEL_REASONING_EFFORT is now documented as safe here and NOT portable: gpt-oss accepts low/medium/high, exactly the scale openAIEffort maps onto, while qwen/qwen3.6-27b on the same account rejects all three and fails the whole request rather than ignoring the key. I7 IS NO LONGER UNPROVEN. make eval-live passes all three cases twice against gpt-oss-120b, the planted-injection case included: answers from the handbook, cites, refuses the injection, leaks neither the operator-only pay guidance nor the other tenant's figures. CLAUDE.md §12 and handover.md updated from "urgent" to measured, dated, and scoped to the one model it is evidence about. One real defect found on the way. The handbook grounding check failed once on an answer containing the phrase it wanted — "more than ten minutes" on screen, strings.Contains false — which leaves an invisible separator as the only explanation; the same model writes "47 %" and a U+2011 hyphen elsewhere. The flaky assertion is the small half. THE LEAK ASSERTIONS USED THE SAME MATCH and fail in the dangerous direction: "attacker@evil.test" with a zero-width space, or "uplift" with a soft hyphen, would have been reported clean. A permission test that cannot see the leak it is hunting is worse than none, because it is believed. normalizeForMatch folds those away, and its test pins that every case is one plain ToLower MISSES — a case whose naive match already succeeds fails, so the suite cannot fill with examples that demonstrate nothing. That caught my own first BOM case, which put the mark where Contains found it regardless. gofmt clean, vet clean, 15/15 packages pass offline; live suite green twice. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01PJvibeSc1JYXjatankqM1g
165 lines
5.2 KiB
Go
165 lines
5.2 KiB
Go
package runtime_test
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/krow/krow-backend/go-api/internal/runtime"
|
|
"github.com/krow/krow-backend/go-api/internal/testutil"
|
|
)
|
|
|
|
// trajectory builds a saveable run for the given harness org.
|
|
func trajectory(orgID, runID string) *runtime.Trajectory {
|
|
started := time.Now().Add(-2 * time.Second).UTC()
|
|
return &runtime.Trajectory{
|
|
RunID: runID,
|
|
OrgID: orgID,
|
|
AgentID: "activity-agent",
|
|
AgentVersion: 3,
|
|
Tier: "balanced",
|
|
Model: "openai/gpt-oss-120b",
|
|
StartedAt: started,
|
|
EndedAt: started.Add(1200 * time.Millisecond),
|
|
Termination: runtime.TerminationCompleted,
|
|
Entries: []runtime.Entry{
|
|
{Seq: 1, At: started, Kind: runtime.EntryMessage, Role: "user", Text: "what happened?"},
|
|
{Seq: 2, At: started, Kind: runtime.EntryBudget, Budget: &runtime.Snapshot{StepsLeft: 8, TokensLeft: 120000}},
|
|
{Seq: 3, At: started, Kind: runtime.EntryMessage, Role: "assistant", Text: "Twelve events."},
|
|
},
|
|
Usage: runtime.RunUsage{InputTokens: 900, OutputTokens: 120, TotalTokens: 1020, ModelCalls: 1},
|
|
}
|
|
}
|
|
|
|
func TestPostgresSinkSavesAndReadsBack(t *testing.T) {
|
|
h := testutil.New(t)
|
|
ctx := context.Background()
|
|
sink := runtime.NewPostgresSink(h.Pool)
|
|
|
|
traj := trajectory(h.OrgID, "run_store_basic")
|
|
if err := sink.Save(ctx, traj); err != nil {
|
|
t.Fatalf("save: %v", err)
|
|
}
|
|
|
|
var (
|
|
agentID, tier, model, termination string
|
|
version, modelCalls int
|
|
total int64
|
|
entries []byte
|
|
)
|
|
err := h.Pool.QueryRow(ctx, `
|
|
SELECT agent_id, agent_version, tier, model, termination, total_tokens, model_calls, entries
|
|
FROM agent_runs WHERE run_id = $1`, traj.RunID,
|
|
).Scan(&agentID, &version, &tier, &model, &termination, &total, &modelCalls, &entries)
|
|
if err != nil {
|
|
t.Fatalf("read back: %v", err)
|
|
}
|
|
|
|
if agentID != "activity-agent" || version != 3 {
|
|
t.Errorf("stored %s v%d, want activity-agent v3", agentID, version)
|
|
}
|
|
if termination != string(runtime.TerminationCompleted) {
|
|
t.Errorf("termination = %q, want Completed", termination)
|
|
}
|
|
// Both the tier asked for and the model that answered, so a trajectory read
|
|
// a year later does not require knowing that week's routing.
|
|
if tier != "balanced" || model != "openai/gpt-oss-120b" {
|
|
t.Errorf("tier/model = %q/%q, want balanced/openai/gpt-oss-120b", tier, model)
|
|
}
|
|
if total != 1020 || modelCalls != 1 {
|
|
t.Errorf("usage = %d tokens over %d calls, want 1020/1", total, modelCalls)
|
|
}
|
|
|
|
var round []runtime.Entry
|
|
if err := json.Unmarshal(entries, &round); err != nil {
|
|
t.Fatalf("entries did not round-trip: %v", err)
|
|
}
|
|
if len(round) != 3 || round[0].Role != "user" || round[2].Text != "Twelve events." {
|
|
t.Errorf("entries round-tripped as %+v", round)
|
|
}
|
|
}
|
|
|
|
func TestPostgresSinkIsIdempotentPerRun(t *testing.T) {
|
|
// A run id is generated once and written once. A second save is a retry of
|
|
// one that already landed — failing it would turn a harmless duplicate
|
|
// into a reported error on a run that succeeded.
|
|
h := testutil.New(t)
|
|
ctx := context.Background()
|
|
sink := runtime.NewPostgresSink(h.Pool)
|
|
|
|
traj := trajectory(h.OrgID, "run_store_twice")
|
|
if err := sink.Save(ctx, traj); err != nil {
|
|
t.Fatalf("first save: %v", err)
|
|
}
|
|
if err := sink.Save(ctx, traj); err != nil {
|
|
t.Fatalf("second save should be a no-op, got: %v", err)
|
|
}
|
|
|
|
var n int
|
|
if err := h.Pool.QueryRow(ctx,
|
|
`SELECT count(*) FROM agent_runs WHERE run_id = $1`, traj.RunID).Scan(&n); err != nil {
|
|
t.Fatalf("count: %v", err)
|
|
}
|
|
if n != 1 {
|
|
t.Errorf("%d rows for one run id, want 1", n)
|
|
}
|
|
}
|
|
|
|
func TestPostgresSinkRefusesRunsItCannotAttribute(t *testing.T) {
|
|
h := testutil.New(t)
|
|
ctx := context.Background()
|
|
sink := runtime.NewPostgresSink(h.Pool)
|
|
|
|
cases := map[string]*runtime.Trajectory{
|
|
"no run id": func() *runtime.Trajectory {
|
|
tr := trajectory(h.OrgID, "")
|
|
return tr
|
|
}(),
|
|
// I5: tenancy is not optional. Caught here so the failure names the
|
|
// cause rather than surfacing as a NOT NULL violation.
|
|
"no org": func() *runtime.Trajectory {
|
|
tr := trajectory("", "run_no_org")
|
|
return tr
|
|
}(),
|
|
// An invented termination must never reach the column that evals and
|
|
// dashboards group by.
|
|
"invented termination": func() *runtime.Trajectory {
|
|
tr := trajectory(h.OrgID, "run_bad_term")
|
|
tr.Termination = "Finished"
|
|
return tr
|
|
}(),
|
|
}
|
|
|
|
for name, tr := range cases {
|
|
if err := sink.Save(ctx, tr); err == nil {
|
|
t.Errorf("%s: save should have been refused", name)
|
|
}
|
|
}
|
|
}
|
|
|
|
func TestPostgresSinkStoresAnEmptyTrajectory(t *testing.T) {
|
|
// A run that recorded nothing is still a run worth keeping, and a nil
|
|
// slice marshals to "null", which the jsonb_typeof CHECK refuses.
|
|
h := testutil.New(t)
|
|
ctx := context.Background()
|
|
sink := runtime.NewPostgresSink(h.Pool)
|
|
|
|
traj := trajectory(h.OrgID, "run_store_empty")
|
|
traj.Entries = nil
|
|
traj.Termination = runtime.TerminationBudgetExceeded
|
|
|
|
if err := sink.Save(ctx, traj); err != nil {
|
|
t.Fatalf("save: %v", err)
|
|
}
|
|
|
|
var kind string
|
|
if err := h.Pool.QueryRow(ctx,
|
|
`SELECT jsonb_typeof(entries) FROM agent_runs WHERE run_id = $1`, traj.RunID).Scan(&kind); err != nil {
|
|
t.Fatalf("read back: %v", err)
|
|
}
|
|
if kind != "array" {
|
|
t.Errorf("entries stored as %q, want array", kind)
|
|
}
|
|
}
|