§9 says no agent ships without evals. Eight of the nine had none: the two
other suites in evals/ are harness fixtures rather than agents in the
registry, so the rule was being met by one agent in nine.
Evals — 40 new cases, five per agent, every one carrying mustNotLeak:
- the agent is loaded from its real spec in agents/*.md rather than
written out again in Go. A hand-copied agent tests the copy: it keeps
passing after somebody edits the spec, which is the moment it most
needed to fail.
- callNamed calls the tool a case names. toolThenAnswer always called
tools[0], so seven of positions-agent's eight tools were unreachable,
and a boundary nothing calls is a boundary nothing tests.
- seedWorkspace fills BOTH tenants. A leak test against an empty second
tenant cannot fail.
Verified by breaking workersByScore's org predicate: six cases across four
agents fail with LEAKED "RIVAL".
Knowledge — six policy documents, taking the corpus from 2 to 8 (34
chunks). Three restricted to admin and employer, five tenant-wide. They
cover what the tools cannot: a tool reports how many shifts went unworked,
a policy says what cover costs inside 24 hours.
corpus_test.go treats those documents as product rather than fixtures. The
first version was tautological — it read audience: from a file and checked
that file's audience was enforced, so opening a restricted document passed.
mustNotBeTenantWide now holds that judgement apart from the files, with the
reason recorded for each.
CI — the checks this repository already had, made unskippable. testutil
calls t.Skipf on an unreachable database, so a dead service container would
produce a green build over a suite that ran almost nothing. Simulated: go
test exits 0 with 74 tests skipped, including every tenant-isolation test.
The guard exits 1 and names them, while still allowing TestLive* to skip
without a model key.
This CI tests; it does not deploy. The README's claim that migrations are
run by CI against the target database remains aspirational.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0186JgqQUCDS8ZwGmyw3ymWu
278 lines
10 KiB
Go
278 lines
10 KiB
Go
package evals_test
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"testing"
|
|
|
|
"github.com/krow/krow-backend/go-api/internal/definition"
|
|
"github.com/krow/krow-backend/go-api/internal/evals"
|
|
"github.com/krow/krow-backend/go-api/internal/gateway"
|
|
"github.com/krow/krow-backend/go-api/internal/knowledge"
|
|
"github.com/krow/krow-backend/go-api/internal/runtime"
|
|
"github.com/krow/krow-backend/go-api/internal/testutil"
|
|
)
|
|
|
|
// Suites for the agents this product actually ships.
|
|
//
|
|
// §9 says no agent ships without evals. Eight of the nine shipped without any:
|
|
// `activity-agent` had a suite, and the other two suites in evals/ — coverage
|
|
// and handbook — are fixtures built for the harness rather than agents in the
|
|
// registry. So the rule was being met by one agent in nine.
|
|
//
|
|
// Two things are done differently here from the activity suite, both because
|
|
// the point is to test what ships:
|
|
//
|
|
// - the agent is loaded from its REAL spec in agents/*.md, not written out
|
|
// again in Go. A hand-copied agent tests the copy: it keeps passing after
|
|
// somebody edits the spec, which is the moment it most needed to fail.
|
|
// - the tool set is whatever that spec declares. If a spec names a tool the
|
|
// registry does not have, the suite says so rather than quietly running an
|
|
// agent with one capability fewer.
|
|
//
|
|
// What these prove is the boundary, not the prose. The model is scripted
|
|
// (`toolThenAnswer`) and answers with the tool's output verbatim, so a case
|
|
// asserts that a tool ran, that what it returned carries what it should, and —
|
|
// the part that matters — that it carries nothing belonging to anyone else.
|
|
|
|
// seedWorkspace fills both tenants with the records these agents read.
|
|
//
|
|
// Both, always. A leak test against an empty second tenant is a test that
|
|
// cannot fail: `mustNotLeak` looks for the other tenant's rows in the answer,
|
|
// and if that tenant has no rows there is nothing to find. Every table an
|
|
// agent's tools touch is populated on both sides, with values distinctive
|
|
// enough to spot in a blob of JSON.
|
|
func seedWorkspace(t *testing.T, h *testutil.Harness) (otherOrg string) {
|
|
t.Helper()
|
|
ctx := context.Background()
|
|
|
|
if err := h.Pool.QueryRow(ctx,
|
|
`INSERT INTO organizations (name, slug) VALUES ('Rival Staffing', 'rival-staffing')
|
|
RETURNING id::text`).Scan(&otherOrg); err != nil {
|
|
t.Fatalf("create rival org: %v", err)
|
|
}
|
|
|
|
type tenant struct {
|
|
org, tag string
|
|
}
|
|
for _, tn := range []tenant{{h.OrgID, "Ours"}, {otherOrg, "RIVAL"}} {
|
|
var postingID string
|
|
if err := h.Pool.QueryRow(ctx, `
|
|
INSERT INTO job_postings (org_id, title, status, headcount, location, priority)
|
|
VALUES ($1::uuid, $2, 'active', 3, $3, 'high') RETURNING id::text`,
|
|
tn.org, tn.tag+" Bar Supervisor", tn.tag+" Shoreditch").Scan(&postingID); err != nil {
|
|
t.Fatalf("seed posting (%s): %v", tn.tag, err)
|
|
}
|
|
|
|
for i, st := range []string{"applied", "ai_screened", "shortlisted", "interview", "hired"} {
|
|
if _, err := h.Pool.Exec(ctx, `
|
|
INSERT INTO job_applications
|
|
(org_id, job_posting_id, applicant_name, email, status, ai_score, job_title)
|
|
VALUES ($1::uuid, $2::uuid, $3, $4, $5::application_status, $6, $7)`,
|
|
tn.org, postingID,
|
|
fmt.Sprintf("%s Applicant %d", tn.tag, i),
|
|
fmt.Sprintf("%s-applicant-%d@example.test", strings.ToLower(tn.tag), i),
|
|
st, 60+i*8, tn.tag+" Bar Supervisor"); err != nil {
|
|
t.Fatalf("seed application (%s): %v", tn.tag, err)
|
|
}
|
|
}
|
|
|
|
for i, name := range []string{"Worker One", "Worker Two"} {
|
|
if _, err := h.Pool.Exec(ctx, `
|
|
INSERT INTO worker_profiles
|
|
(org_id, full_name, email, krow_score, reliability_score,
|
|
attendance_score, performance_score, client_rating,
|
|
experience_years, shifts_completed, current_position)
|
|
VALUES ($1::uuid, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11)`,
|
|
tn.org, tn.tag+" "+name,
|
|
fmt.Sprintf("%s-worker-%d@example.test", strings.ToLower(tn.tag), i),
|
|
80+i*7, 85+i*5, 90+i*3, 82+i*4, 4.5, 3+i, 20+i*10,
|
|
tn.tag+" Bartender"); err != nil {
|
|
t.Fatalf("seed worker (%s): %v", tn.tag, err)
|
|
}
|
|
}
|
|
|
|
if _, err := h.Pool.Exec(ctx, `
|
|
INSERT INTO staff (org_id, name, email, role, status, ai_score, hire_date)
|
|
VALUES ($1::uuid, $2, $3, $4, 'active', 91, current_date - 30)`,
|
|
tn.org, tn.tag+" Hired Person",
|
|
fmt.Sprintf("%s-hire@example.test", strings.ToLower(tn.tag)),
|
|
tn.tag+" Bar Supervisor"); err != nil {
|
|
t.Fatalf("seed staff (%s): %v", tn.tag, err)
|
|
}
|
|
|
|
for i, st := range []string{"present", "present", "late", "absent", "no_show"} {
|
|
// A missed shift has no hours behind it — shift_records enforces
|
|
// that, and seeding around the constraint would be seeding data the
|
|
// product cannot hold.
|
|
missed := st == "absent" || st == "no_show"
|
|
worked, overtime, late := 8.0, float64(i), i*7
|
|
if missed {
|
|
worked, overtime, late = 0, 0, 0
|
|
}
|
|
if _, err := h.Pool.Exec(ctx, `
|
|
INSERT INTO shift_records
|
|
(org_id, worker_name, worker_email, role, shift_date,
|
|
scheduled_start, scheduled_end, created_date,
|
|
status, scheduled_hours, actual_hours, overtime_hours, minutes_late)
|
|
VALUES ($1::uuid, $2, $3, $4, current_date - $5::int,
|
|
(current_date - $5::int) + time '18:00',
|
|
(current_date - $5::int) + time '02:00' + interval '1 day',
|
|
(current_date - $5::int) + time '18:00',
|
|
$6::shift_status, 8, $7, $8, $9)`,
|
|
tn.org, tn.tag+" Worker One",
|
|
fmt.Sprintf("%s-worker-0@example.test", strings.ToLower(tn.tag)),
|
|
tn.tag+" Bartender", i+1, st, worked, overtime, late); err != nil {
|
|
t.Fatalf("seed shift (%s): %v", tn.tag, err)
|
|
}
|
|
}
|
|
|
|
if _, err := h.Pool.Exec(ctx, `
|
|
INSERT INTO courses (org_id, title, category, status, xp)
|
|
VALUES ($1::uuid, $2, 'Bar', 'active', 50)`,
|
|
tn.org, tn.tag+" Cocktail Fundamentals"); err != nil {
|
|
t.Fatalf("seed course (%s): %v", tn.tag, err)
|
|
}
|
|
|
|
for _, ev := range []string{"apply_job", "hire_candidate", "delete_position"} {
|
|
if _, err := h.Pool.Exec(ctx, `
|
|
INSERT INTO user_activity (org_id, event_type, user_email, user_name)
|
|
VALUES ($1::uuid, $2, $3, $4)`,
|
|
tn.org, ev,
|
|
fmt.Sprintf("%s-actor@example.test", strings.ToLower(tn.tag)),
|
|
tn.tag+" Actor"); err != nil {
|
|
t.Fatalf("seed activity (%s): %v", tn.tag, err)
|
|
}
|
|
}
|
|
}
|
|
return otherOrg
|
|
}
|
|
|
|
// callNamed exercises the tool a case names, rather than always the first one.
|
|
//
|
|
// `toolThenAnswer` calls req.Tools[0], which is right for an agent carrying one
|
|
// or two tools and useless for one carrying eight: seven of them would never be
|
|
// reached, and a boundary nothing calls is a boundary nothing tests. A case
|
|
// says which capability it is about through `expect.toolsCalled`, and this
|
|
// calls that one. The assertions are still the case's own — this decides what
|
|
// runs, not whether it passed.
|
|
type callNamed struct {
|
|
want string
|
|
done bool
|
|
}
|
|
|
|
func (m *callNamed) Complete(_ context.Context, req gateway.Request) (*gateway.Response, error) {
|
|
last := req.Messages[len(req.Messages)-1]
|
|
if len(last.ToolResults) > 0 {
|
|
return &gateway.Response{
|
|
Text: "Here is everything I was given: " + last.ToolResults[0].Content,
|
|
StopReason: "end_turn", Model: "scripted",
|
|
}, nil
|
|
}
|
|
if len(req.Tools) == 0 {
|
|
return &gateway.Response{
|
|
Text: "I have no way to look that up.", StopReason: "end_turn", Model: "scripted",
|
|
}, nil
|
|
}
|
|
pick := req.Tools[0].Name
|
|
for _, tool := range req.Tools {
|
|
if tool.Name == m.want {
|
|
pick = tool.Name
|
|
break
|
|
}
|
|
}
|
|
return &gateway.Response{
|
|
ToolCalls: []gateway.ToolCall{{ID: "call_1", Name: pick, Input: json.RawMessage(`{}`)}},
|
|
StopReason: "tool_use", Model: "scripted",
|
|
}, nil
|
|
}
|
|
|
|
// loadShippedAgent reads an agent from the spec this product ships.
|
|
func loadShippedAgent(t *testing.T, key string) *runtime.Agent {
|
|
t.Helper()
|
|
path := filepath.Join("..", "..", "..", "agents", key+".md")
|
|
raw, err := os.ReadFile(path)
|
|
if err != nil {
|
|
t.Fatalf("read spec %s: %v", path, err)
|
|
}
|
|
parsed, err := definition.ParseAgent(string(raw), definition.Options{})
|
|
if err != nil {
|
|
t.Fatalf("parse spec %s: %v", key, err)
|
|
}
|
|
return &runtime.Agent{
|
|
ID: parsed.ID, Name: parsed.Name, Version: parsed.Version,
|
|
Description: parsed.Description, Reasoning: parsed.Reasoning,
|
|
Pages: parsed.Pages, Instructions: parsed.Instructions,
|
|
Skills: parsed.Skills, Tools: parsed.Tools,
|
|
KnowledgeSources: parsed.Sources,
|
|
}
|
|
}
|
|
|
|
// TestShippedAgentSuites runs every shipped agent against its own suite.
|
|
//
|
|
// One test over a table rather than eight near-identical functions: the agents
|
|
// differ in their spec and their cases, not in how they are exercised, and
|
|
// eight copies of this loop would drift apart one edit at a time.
|
|
func TestShippedAgentSuites(t *testing.T) {
|
|
for _, key := range []string{
|
|
"analytics-agent", "candidates-agent", "control-center-agent",
|
|
"hired-history-agent", "krow-forge-agent", "krow-workforce-agent",
|
|
"positions-agent", "talent-pool-agent",
|
|
} {
|
|
t.Run(key, func(t *testing.T) {
|
|
h := testutil.New(t)
|
|
ctx := context.Background()
|
|
seedWorkspace(t, h)
|
|
|
|
suite, err := evals.LoadSuite(resolveSuite(t, key+".json"))
|
|
if err != nil {
|
|
t.Fatalf("load suite: %v", err)
|
|
}
|
|
agent := loadShippedAgent(t, key)
|
|
|
|
// The real registry, so a case exercises the tool that ships rather
|
|
// than a stand-in written to pass.
|
|
reg := runtime.DefaultTools(h.Pool, knowledge.NewRetriever(h.Pool, nil))
|
|
|
|
// A spec naming a tool the registry does not have is an agent with a
|
|
// capability its author believes it has. Said here rather than left
|
|
// for the runtime to drop in silence.
|
|
if unknown := reg.Known(agent.Tools); len(unknown) > 0 {
|
|
t.Fatalf("%s declares tools that are not registered: %s",
|
|
key, strings.Join(unknown, ", "))
|
|
}
|
|
|
|
users := seedPrincipals(t, h, map[string]string{
|
|
"$ADMIN_ID": "boss@example.test",
|
|
"$TALENT_ID": "worker@example.test",
|
|
})
|
|
|
|
var results []evals.Result
|
|
for _, c := range suite.Cases {
|
|
want := ""
|
|
if len(c.Expect.ToolsCalled) > 0 {
|
|
want = c.Expect.ToolsCalled[0]
|
|
}
|
|
runner := evals.NewRunner(func(sink runtime.Sink) runtime.AgentExecutor {
|
|
return runtime.NewModelExecutor(&callNamed{want: want}, sink, reg)
|
|
}, agent)
|
|
results = append(results, runner.Run(ctx, substitute(c, h.OrgID, users)))
|
|
}
|
|
|
|
t.Log("\n" + evals.Report(suite.Agent, results))
|
|
for _, r := range results {
|
|
if !r.Passed {
|
|
t.Errorf("%s failed: %v", r.CaseID, r.Failures)
|
|
}
|
|
}
|
|
if len(results) < 5 {
|
|
t.Errorf("%s has %d cases; §9 requires at least 5", key, len(results))
|
|
}
|
|
})
|
|
}
|
|
}
|