package evals_test import ( "context" "encoding/json" "fmt" "os" "path/filepath" "strings" "testing" "github.com/krow/krow-backend/go-api/internal/definition" "github.com/krow/krow-backend/go-api/internal/evals" "github.com/krow/krow-backend/go-api/internal/gateway" "github.com/krow/krow-backend/go-api/internal/knowledge" "github.com/krow/krow-backend/go-api/internal/runtime" "github.com/krow/krow-backend/go-api/internal/testutil" ) // Suites for the agents this product actually ships. // // §9 says no agent ships without evals. Eight of the nine shipped without any: // `activity-agent` had a suite, and the other two suites in evals/ — coverage // and handbook — are fixtures built for the harness rather than agents in the // registry. So the rule was being met by one agent in nine. // // Two things are done differently here from the activity suite, both because // the point is to test what ships: // // - the agent is loaded from its REAL spec in agents/*.md, not written out // again in Go. A hand-copied agent tests the copy: it keeps passing after // somebody edits the spec, which is the moment it most needed to fail. // - the tool set is whatever that spec declares. If a spec names a tool the // registry does not have, the suite says so rather than quietly running an // agent with one capability fewer. // // What these prove is the boundary, not the prose. The model is scripted // (`toolThenAnswer`) and answers with the tool's output verbatim, so a case // asserts that a tool ran, that what it returned carries what it should, and — // the part that matters — that it carries nothing belonging to anyone else. // seedWorkspace fills both tenants with the records these agents read. // // Both, always. A leak test against an empty second tenant is a test that // cannot fail: `mustNotLeak` looks for the other tenant's rows in the answer, // and if that tenant has no rows there is nothing to find. Every table an // agent's tools touch is populated on both sides, with values distinctive // enough to spot in a blob of JSON. func seedWorkspace(t *testing.T, h *testutil.Harness) (otherOrg string) { t.Helper() ctx := context.Background() if err := h.Pool.QueryRow(ctx, `INSERT INTO organizations (name, slug) VALUES ('Rival Staffing', 'rival-staffing') RETURNING id::text`).Scan(&otherOrg); err != nil { t.Fatalf("create rival org: %v", err) } type tenant struct { org, tag string } for _, tn := range []tenant{{h.OrgID, "Ours"}, {otherOrg, "RIVAL"}} { var postingID string if err := h.Pool.QueryRow(ctx, ` INSERT INTO job_postings (org_id, title, status, headcount, location, priority) VALUES ($1::uuid, $2, 'active', 3, $3, 'high') RETURNING id::text`, tn.org, tn.tag+" Bar Supervisor", tn.tag+" Shoreditch").Scan(&postingID); err != nil { t.Fatalf("seed posting (%s): %v", tn.tag, err) } for i, st := range []string{"applied", "ai_screened", "shortlisted", "interview", "hired"} { if _, err := h.Pool.Exec(ctx, ` INSERT INTO job_applications (org_id, job_posting_id, applicant_name, email, status, ai_score, job_title) VALUES ($1::uuid, $2::uuid, $3, $4, $5::application_status, $6, $7)`, tn.org, postingID, fmt.Sprintf("%s Applicant %d", tn.tag, i), fmt.Sprintf("%s-applicant-%d@example.test", strings.ToLower(tn.tag), i), st, 60+i*8, tn.tag+" Bar Supervisor"); err != nil { t.Fatalf("seed application (%s): %v", tn.tag, err) } } for i, name := range []string{"Worker One", "Worker Two"} { if _, err := h.Pool.Exec(ctx, ` INSERT INTO worker_profiles (org_id, full_name, email, krow_score, reliability_score, attendance_score, performance_score, client_rating, experience_years, shifts_completed, current_position) VALUES ($1::uuid, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11)`, tn.org, tn.tag+" "+name, fmt.Sprintf("%s-worker-%d@example.test", strings.ToLower(tn.tag), i), 80+i*7, 85+i*5, 90+i*3, 82+i*4, 4.5, 3+i, 20+i*10, tn.tag+" Bartender"); err != nil { t.Fatalf("seed worker (%s): %v", tn.tag, err) } } if _, err := h.Pool.Exec(ctx, ` INSERT INTO staff (org_id, name, email, role, status, ai_score, hire_date) VALUES ($1::uuid, $2, $3, $4, 'active', 91, current_date - 30)`, tn.org, tn.tag+" Hired Person", fmt.Sprintf("%s-hire@example.test", strings.ToLower(tn.tag)), tn.tag+" Bar Supervisor"); err != nil { t.Fatalf("seed staff (%s): %v", tn.tag, err) } for i, st := range []string{"present", "present", "late", "absent", "no_show"} { // A missed shift has no hours behind it — shift_records enforces // that, and seeding around the constraint would be seeding data the // product cannot hold. missed := st == "absent" || st == "no_show" worked, overtime, late := 8.0, float64(i), i*7 if missed { worked, overtime, late = 0, 0, 0 } if _, err := h.Pool.Exec(ctx, ` INSERT INTO shift_records (org_id, worker_name, worker_email, role, shift_date, scheduled_start, scheduled_end, created_date, status, scheduled_hours, actual_hours, overtime_hours, minutes_late) VALUES ($1::uuid, $2, $3, $4, current_date - $5::int, (current_date - $5::int) + time '18:00', (current_date - $5::int) + time '02:00' + interval '1 day', (current_date - $5::int) + time '18:00', $6::shift_status, 8, $7, $8, $9)`, tn.org, tn.tag+" Worker One", fmt.Sprintf("%s-worker-0@example.test", strings.ToLower(tn.tag)), tn.tag+" Bartender", i+1, st, worked, overtime, late); err != nil { t.Fatalf("seed shift (%s): %v", tn.tag, err) } } if _, err := h.Pool.Exec(ctx, ` INSERT INTO courses (org_id, title, category, status, xp) VALUES ($1::uuid, $2, 'Bar', 'active', 50)`, tn.org, tn.tag+" Cocktail Fundamentals"); err != nil { t.Fatalf("seed course (%s): %v", tn.tag, err) } for _, ev := range []string{"apply_job", "hire_candidate", "delete_position"} { if _, err := h.Pool.Exec(ctx, ` INSERT INTO user_activity (org_id, event_type, user_email, user_name) VALUES ($1::uuid, $2, $3, $4)`, tn.org, ev, fmt.Sprintf("%s-actor@example.test", strings.ToLower(tn.tag)), tn.tag+" Actor"); err != nil { t.Fatalf("seed activity (%s): %v", tn.tag, err) } } } return otherOrg } // callNamed exercises the tool a case names, rather than always the first one. // // `toolThenAnswer` calls req.Tools[0], which is right for an agent carrying one // or two tools and useless for one carrying eight: seven of them would never be // reached, and a boundary nothing calls is a boundary nothing tests. A case // says which capability it is about through `expect.toolsCalled`, and this // calls that one. The assertions are still the case's own — this decides what // runs, not whether it passed. type callNamed struct { want string done bool } func (m *callNamed) Complete(_ context.Context, req gateway.Request) (*gateway.Response, error) { last := req.Messages[len(req.Messages)-1] if len(last.ToolResults) > 0 { return &gateway.Response{ Text: "Here is everything I was given: " + last.ToolResults[0].Content, StopReason: "end_turn", Model: "scripted", }, nil } if len(req.Tools) == 0 { return &gateway.Response{ Text: "I have no way to look that up.", StopReason: "end_turn", Model: "scripted", }, nil } pick := req.Tools[0].Name for _, tool := range req.Tools { if tool.Name == m.want { pick = tool.Name break } } return &gateway.Response{ ToolCalls: []gateway.ToolCall{{ID: "call_1", Name: pick, Input: json.RawMessage(`{}`)}}, StopReason: "tool_use", Model: "scripted", }, nil } // loadShippedAgent reads an agent from the spec this product ships. func loadShippedAgent(t *testing.T, key string) *runtime.Agent { t.Helper() path := filepath.Join("..", "..", "..", "agents", key+".md") raw, err := os.ReadFile(path) if err != nil { t.Fatalf("read spec %s: %v", path, err) } parsed, err := definition.ParseAgent(string(raw), definition.Options{}) if err != nil { t.Fatalf("parse spec %s: %v", key, err) } return &runtime.Agent{ ID: parsed.ID, Name: parsed.Name, Version: parsed.Version, Description: parsed.Description, Reasoning: parsed.Reasoning, Pages: parsed.Pages, Instructions: parsed.Instructions, Skills: parsed.Skills, Tools: parsed.Tools, KnowledgeSources: parsed.Sources, } } // TestShippedAgentSuites runs every shipped agent against its own suite. // // One test over a table rather than eight near-identical functions: the agents // differ in their spec and their cases, not in how they are exercised, and // eight copies of this loop would drift apart one edit at a time. func TestShippedAgentSuites(t *testing.T) { for _, key := range []string{ "analytics-agent", "candidates-agent", "control-center-agent", "hired-history-agent", "krow-forge-agent", "krow-workforce-agent", "positions-agent", "talent-pool-agent", } { t.Run(key, func(t *testing.T) { h := testutil.New(t) ctx := context.Background() seedWorkspace(t, h) suite, err := evals.LoadSuite(resolveSuite(t, key+".json")) if err != nil { t.Fatalf("load suite: %v", err) } agent := loadShippedAgent(t, key) // The real registry, so a case exercises the tool that ships rather // than a stand-in written to pass. reg := runtime.DefaultTools(h.Pool, knowledge.NewRetriever(h.Pool, nil)) // A spec naming a tool the registry does not have is an agent with a // capability its author believes it has. Said here rather than left // for the runtime to drop in silence. if unknown := reg.Known(agent.Tools); len(unknown) > 0 { t.Fatalf("%s declares tools that are not registered: %s", key, strings.Join(unknown, ", ")) } users := seedPrincipals(t, h, map[string]string{ "$ADMIN_ID": "boss@example.test", "$TALENT_ID": "worker@example.test", }) var results []evals.Result for _, c := range suite.Cases { want := "" if len(c.Expect.ToolsCalled) > 0 { want = c.Expect.ToolsCalled[0] } runner := evals.NewRunner(func(sink runtime.Sink) runtime.AgentExecutor { return runtime.NewModelExecutor(&callNamed{want: want}, sink, reg) }, agent) results = append(results, runner.Run(ctx, substitute(c, h.OrgID, users))) } t.Log("\n" + evals.Report(suite.Agent, results)) for _, r := range results { if !r.Passed { t.Errorf("%s failed: %v", r.CaseID, r.Failures) } } if len(results) < 5 { t.Errorf("%s has %d cases; §9 requires at least 5", key, len(results)) } }) } }