Files
krow_backend/go-api/internal/evals/registry_agents_test.go
Aravind a222dcd3e4
Some checks failed
CI / test (push) Has been cancelled
CI / fixture (push) Has been cancelled
Add evals for every shipped agent, a policy corpus, and CI
§9 says no agent ships without evals. Eight of the nine had none: the two
other suites in evals/ are harness fixtures rather than agents in the
registry, so the rule was being met by one agent in nine.

Evals — 40 new cases, five per agent, every one carrying mustNotLeak:

  - the agent is loaded from its real spec in agents/*.md rather than
    written out again in Go. A hand-copied agent tests the copy: it keeps
    passing after somebody edits the spec, which is the moment it most
    needed to fail.
  - callNamed calls the tool a case names. toolThenAnswer always called
    tools[0], so seven of positions-agent's eight tools were unreachable,
    and a boundary nothing calls is a boundary nothing tests.
  - seedWorkspace fills BOTH tenants. A leak test against an empty second
    tenant cannot fail.

Verified by breaking workersByScore's org predicate: six cases across four
agents fail with LEAKED "RIVAL".

Knowledge — six policy documents, taking the corpus from 2 to 8 (34
chunks). Three restricted to admin and employer, five tenant-wide. They
cover what the tools cannot: a tool reports how many shifts went unworked,
a policy says what cover costs inside 24 hours.

corpus_test.go treats those documents as product rather than fixtures. The
first version was tautological — it read audience: from a file and checked
that file's audience was enforced, so opening a restricted document passed.
mustNotBeTenantWide now holds that judgement apart from the files, with the
reason recorded for each.

CI — the checks this repository already had, made unskippable. testutil
calls t.Skipf on an unreachable database, so a dead service container would
produce a green build over a suite that ran almost nothing. Simulated: go
test exits 0 with 74 tests skipped, including every tenant-isolation test.
The guard exits 1 and names them, while still allowing TestLive* to skip
without a model key.

This CI tests; it does not deploy. The README's claim that migrations are
run by CI against the target database remains aspirational.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0186JgqQUCDS8ZwGmyw3ymWu
2026-08-28 13:52:46 +05:30

278 lines
10 KiB
Go

package evals_test
import (
"context"
"encoding/json"
"fmt"
"os"
"path/filepath"
"strings"
"testing"
"github.com/krow/krow-backend/go-api/internal/definition"
"github.com/krow/krow-backend/go-api/internal/evals"
"github.com/krow/krow-backend/go-api/internal/gateway"
"github.com/krow/krow-backend/go-api/internal/knowledge"
"github.com/krow/krow-backend/go-api/internal/runtime"
"github.com/krow/krow-backend/go-api/internal/testutil"
)
// Suites for the agents this product actually ships.
//
// §9 says no agent ships without evals. Eight of the nine shipped without any:
// `activity-agent` had a suite, and the other two suites in evals/ — coverage
// and handbook — are fixtures built for the harness rather than agents in the
// registry. So the rule was being met by one agent in nine.
//
// Two things are done differently here from the activity suite, both because
// the point is to test what ships:
//
// - the agent is loaded from its REAL spec in agents/*.md, not written out
// again in Go. A hand-copied agent tests the copy: it keeps passing after
// somebody edits the spec, which is the moment it most needed to fail.
// - the tool set is whatever that spec declares. If a spec names a tool the
// registry does not have, the suite says so rather than quietly running an
// agent with one capability fewer.
//
// What these prove is the boundary, not the prose. The model is scripted
// (`toolThenAnswer`) and answers with the tool's output verbatim, so a case
// asserts that a tool ran, that what it returned carries what it should, and —
// the part that matters — that it carries nothing belonging to anyone else.
// seedWorkspace fills both tenants with the records these agents read.
//
// Both, always. A leak test against an empty second tenant is a test that
// cannot fail: `mustNotLeak` looks for the other tenant's rows in the answer,
// and if that tenant has no rows there is nothing to find. Every table an
// agent's tools touch is populated on both sides, with values distinctive
// enough to spot in a blob of JSON.
func seedWorkspace(t *testing.T, h *testutil.Harness) (otherOrg string) {
t.Helper()
ctx := context.Background()
if err := h.Pool.QueryRow(ctx,
`INSERT INTO organizations (name, slug) VALUES ('Rival Staffing', 'rival-staffing')
RETURNING id::text`).Scan(&otherOrg); err != nil {
t.Fatalf("create rival org: %v", err)
}
type tenant struct {
org, tag string
}
for _, tn := range []tenant{{h.OrgID, "Ours"}, {otherOrg, "RIVAL"}} {
var postingID string
if err := h.Pool.QueryRow(ctx, `
INSERT INTO job_postings (org_id, title, status, headcount, location, priority)
VALUES ($1::uuid, $2, 'active', 3, $3, 'high') RETURNING id::text`,
tn.org, tn.tag+" Bar Supervisor", tn.tag+" Shoreditch").Scan(&postingID); err != nil {
t.Fatalf("seed posting (%s): %v", tn.tag, err)
}
for i, st := range []string{"applied", "ai_screened", "shortlisted", "interview", "hired"} {
if _, err := h.Pool.Exec(ctx, `
INSERT INTO job_applications
(org_id, job_posting_id, applicant_name, email, status, ai_score, job_title)
VALUES ($1::uuid, $2::uuid, $3, $4, $5::application_status, $6, $7)`,
tn.org, postingID,
fmt.Sprintf("%s Applicant %d", tn.tag, i),
fmt.Sprintf("%s-applicant-%d@example.test", strings.ToLower(tn.tag), i),
st, 60+i*8, tn.tag+" Bar Supervisor"); err != nil {
t.Fatalf("seed application (%s): %v", tn.tag, err)
}
}
for i, name := range []string{"Worker One", "Worker Two"} {
if _, err := h.Pool.Exec(ctx, `
INSERT INTO worker_profiles
(org_id, full_name, email, krow_score, reliability_score,
attendance_score, performance_score, client_rating,
experience_years, shifts_completed, current_position)
VALUES ($1::uuid, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11)`,
tn.org, tn.tag+" "+name,
fmt.Sprintf("%s-worker-%d@example.test", strings.ToLower(tn.tag), i),
80+i*7, 85+i*5, 90+i*3, 82+i*4, 4.5, 3+i, 20+i*10,
tn.tag+" Bartender"); err != nil {
t.Fatalf("seed worker (%s): %v", tn.tag, err)
}
}
if _, err := h.Pool.Exec(ctx, `
INSERT INTO staff (org_id, name, email, role, status, ai_score, hire_date)
VALUES ($1::uuid, $2, $3, $4, 'active', 91, current_date - 30)`,
tn.org, tn.tag+" Hired Person",
fmt.Sprintf("%s-hire@example.test", strings.ToLower(tn.tag)),
tn.tag+" Bar Supervisor"); err != nil {
t.Fatalf("seed staff (%s): %v", tn.tag, err)
}
for i, st := range []string{"present", "present", "late", "absent", "no_show"} {
// A missed shift has no hours behind it — shift_records enforces
// that, and seeding around the constraint would be seeding data the
// product cannot hold.
missed := st == "absent" || st == "no_show"
worked, overtime, late := 8.0, float64(i), i*7
if missed {
worked, overtime, late = 0, 0, 0
}
if _, err := h.Pool.Exec(ctx, `
INSERT INTO shift_records
(org_id, worker_name, worker_email, role, shift_date,
scheduled_start, scheduled_end, created_date,
status, scheduled_hours, actual_hours, overtime_hours, minutes_late)
VALUES ($1::uuid, $2, $3, $4, current_date - $5::int,
(current_date - $5::int) + time '18:00',
(current_date - $5::int) + time '02:00' + interval '1 day',
(current_date - $5::int) + time '18:00',
$6::shift_status, 8, $7, $8, $9)`,
tn.org, tn.tag+" Worker One",
fmt.Sprintf("%s-worker-0@example.test", strings.ToLower(tn.tag)),
tn.tag+" Bartender", i+1, st, worked, overtime, late); err != nil {
t.Fatalf("seed shift (%s): %v", tn.tag, err)
}
}
if _, err := h.Pool.Exec(ctx, `
INSERT INTO courses (org_id, title, category, status, xp)
VALUES ($1::uuid, $2, 'Bar', 'active', 50)`,
tn.org, tn.tag+" Cocktail Fundamentals"); err != nil {
t.Fatalf("seed course (%s): %v", tn.tag, err)
}
for _, ev := range []string{"apply_job", "hire_candidate", "delete_position"} {
if _, err := h.Pool.Exec(ctx, `
INSERT INTO user_activity (org_id, event_type, user_email, user_name)
VALUES ($1::uuid, $2, $3, $4)`,
tn.org, ev,
fmt.Sprintf("%s-actor@example.test", strings.ToLower(tn.tag)),
tn.tag+" Actor"); err != nil {
t.Fatalf("seed activity (%s): %v", tn.tag, err)
}
}
}
return otherOrg
}
// callNamed exercises the tool a case names, rather than always the first one.
//
// `toolThenAnswer` calls req.Tools[0], which is right for an agent carrying one
// or two tools and useless for one carrying eight: seven of them would never be
// reached, and a boundary nothing calls is a boundary nothing tests. A case
// says which capability it is about through `expect.toolsCalled`, and this
// calls that one. The assertions are still the case's own — this decides what
// runs, not whether it passed.
type callNamed struct {
want string
done bool
}
func (m *callNamed) Complete(_ context.Context, req gateway.Request) (*gateway.Response, error) {
last := req.Messages[len(req.Messages)-1]
if len(last.ToolResults) > 0 {
return &gateway.Response{
Text: "Here is everything I was given: " + last.ToolResults[0].Content,
StopReason: "end_turn", Model: "scripted",
}, nil
}
if len(req.Tools) == 0 {
return &gateway.Response{
Text: "I have no way to look that up.", StopReason: "end_turn", Model: "scripted",
}, nil
}
pick := req.Tools[0].Name
for _, tool := range req.Tools {
if tool.Name == m.want {
pick = tool.Name
break
}
}
return &gateway.Response{
ToolCalls: []gateway.ToolCall{{ID: "call_1", Name: pick, Input: json.RawMessage(`{}`)}},
StopReason: "tool_use", Model: "scripted",
}, nil
}
// loadShippedAgent reads an agent from the spec this product ships.
func loadShippedAgent(t *testing.T, key string) *runtime.Agent {
t.Helper()
path := filepath.Join("..", "..", "..", "agents", key+".md")
raw, err := os.ReadFile(path)
if err != nil {
t.Fatalf("read spec %s: %v", path, err)
}
parsed, err := definition.ParseAgent(string(raw), definition.Options{})
if err != nil {
t.Fatalf("parse spec %s: %v", key, err)
}
return &runtime.Agent{
ID: parsed.ID, Name: parsed.Name, Version: parsed.Version,
Description: parsed.Description, Reasoning: parsed.Reasoning,
Pages: parsed.Pages, Instructions: parsed.Instructions,
Skills: parsed.Skills, Tools: parsed.Tools,
KnowledgeSources: parsed.Sources,
}
}
// TestShippedAgentSuites runs every shipped agent against its own suite.
//
// One test over a table rather than eight near-identical functions: the agents
// differ in their spec and their cases, not in how they are exercised, and
// eight copies of this loop would drift apart one edit at a time.
func TestShippedAgentSuites(t *testing.T) {
for _, key := range []string{
"analytics-agent", "candidates-agent", "control-center-agent",
"hired-history-agent", "krow-forge-agent", "krow-workforce-agent",
"positions-agent", "talent-pool-agent",
} {
t.Run(key, func(t *testing.T) {
h := testutil.New(t)
ctx := context.Background()
seedWorkspace(t, h)
suite, err := evals.LoadSuite(resolveSuite(t, key+".json"))
if err != nil {
t.Fatalf("load suite: %v", err)
}
agent := loadShippedAgent(t, key)
// The real registry, so a case exercises the tool that ships rather
// than a stand-in written to pass.
reg := runtime.DefaultTools(h.Pool, knowledge.NewRetriever(h.Pool, nil))
// A spec naming a tool the registry does not have is an agent with a
// capability its author believes it has. Said here rather than left
// for the runtime to drop in silence.
if unknown := reg.Known(agent.Tools); len(unknown) > 0 {
t.Fatalf("%s declares tools that are not registered: %s",
key, strings.Join(unknown, ", "))
}
users := seedPrincipals(t, h, map[string]string{
"$ADMIN_ID": "boss@example.test",
"$TALENT_ID": "worker@example.test",
})
var results []evals.Result
for _, c := range suite.Cases {
want := ""
if len(c.Expect.ToolsCalled) > 0 {
want = c.Expect.ToolsCalled[0]
}
runner := evals.NewRunner(func(sink runtime.Sink) runtime.AgentExecutor {
return runtime.NewModelExecutor(&callNamed{want: want}, sink, reg)
}, agent)
results = append(results, runner.Run(ctx, substitute(c, h.OrgID, users)))
}
t.Log("\n" + evals.Report(suite.Agent, results))
for _, r := range results {
if !r.Passed {
t.Errorf("%s failed: %v", r.CaseID, r.Failures)
}
}
if len(results) < 5 {
t.Errorf("%s has %d cases; §9 requires at least 5", key, len(results))
}
})
}
}