// Package evals is the harness that makes an agent's behaviour assertable. // // §9: no agent ships without evals, and no change to the loop, retrieval or // prompt assembly merges without running the suite. That is only enforceable if // running a case is cheap and its assertions are precise, so this package does // two things and no more — it runs a case against a real runtime, and it checks // the trajectory against what the case declared. // // **`must_not_leak` is mandatory on every case.** Not a convention: LoadSuite // refuses a case without it. Every eval therefore doubles as a permission test, // which is the only reason I1 is testable at all — a leak is not something you // notice by reading an answer, it is something you notice by asserting that a // string which should be unreachable never appears. // // The check is deliberately blunt: the forbidden string must not appear // anywhere in the run — not in the answer, not in a tool result, not in an // error message. A leak that reaches the trajectory has already left the // boundary, whether or not the model chose to repeat it. package evals import ( "context" "encoding/json" "fmt" "os" "strings" "time" "github.com/krow/krow-backend/go-api/internal/authctx" "github.com/krow/krow-backend/go-api/internal/runtime" "github.com/krow/krow-backend/go-api/internal/tools" ) // Case is one eval. type Case struct { ID string `json:"id"` Input string `json:"input"` // Principal is who asks. A case that does not say runs as nobody, which // every tool refuses — so this is effectively required. Principal Principal `json:"principal"` Expect Expect `json:"expect"` } // Principal is the caller a case runs as. type Principal struct { UserID string `json:"userId"` OrgID string `json:"orgId"` Role string `json:"role"` Email string `json:"email"` } func (p Principal) identity() authctx.Identity { return authctx.Identity{UserID: p.UserID, OrgID: p.OrgID, Role: p.Role, Email: p.Email} } // Expect is what a case asserts. type Expect struct { Termination runtime.Termination `json:"termination"` ToolsCalled []string `json:"toolsCalled"` MustMention []string `json:"mustMention"` // MustNotLeak is mandatory. Strings that must appear nowhere in the run. MustNotLeak []string `json:"mustNotLeak"` // ConfirmationsRaised are tools that must have DESCRIBED a write without // performing it. The write path's version of an assertion: a case that // expects an agent to propose an assignment checks that it proposed one, // rather than that it talked about proposing one. ConfirmationsRaised []string `json:"confirmationsRaised,omitempty"` // MustNotWrite are tools that must not have executed. Distinct from // mustNotLeak, which is about what a run SAID: this is about what it DID. // A run can be word-perfect and still have assigned somebody to a shift. // // Left optional rather than mandatory, unlike mustNotLeak, because it is // checked structurally as well — see check(): ANY write that ran in a case // which did not expect one fails, whether or not the case named it. An // author cannot forget this the way they could forget a leak string. MustNotWrite []string `json:"mustNotWrite,omitempty"` // Writes are the tools this case expects to have actually executed, after // an approval. Naming one is what makes a write permissible in a case at // all. Writes []string `json:"writes,omitempty"` MaxSteps int `json:"maxSteps"` } // Suite is a set of cases for one agent. type Suite struct { Agent string `json:"agent"` Cases []Case `json:"cases"` } // LoadSuite reads a suite and refuses one that cannot assert what it must. func LoadSuite(path string) (*Suite, error) { raw, err := os.ReadFile(path) if err != nil { return nil, fmt.Errorf("evals: reading %s: %w", path, err) } var s Suite if err := json.Unmarshal(raw, &s); err != nil { return nil, fmt.Errorf("evals: parsing %s: %w", path, err) } if s.Agent == "" { return nil, fmt.Errorf("evals: %s names no agent", path) } // §9 puts the floor at five. Fewer than that is not a suite, it is an // example, and an example does not catch a regression. if len(s.Cases) < 5 { return nil, fmt.Errorf("evals: %s has %d cases; §9 requires at least 5", path, len(s.Cases)) } for i, c := range s.Cases { if c.ID == "" { return nil, fmt.Errorf("evals: %s case %d has no id", path, i) } if len(c.Expect.MustNotLeak) == 0 { return nil, fmt.Errorf( "evals: %s case %q declares no must_not_leak; it is mandatory on every case, "+ "because every eval doubles as a permission test", path, c.ID) } } return &s, nil } // Result is how one case went. type Result struct { CaseID string Passed bool Failures []string Run *runtime.Trajectory Elapsed time.Duration } // Runner executes cases against a real executor. // // Sink must be the SAME sink the executor was built with. The assertions read // the trajectory, not the answer — `toolsCalled` and `maxSteps` exist nowhere // else — so a runner holding its own sink would silently pass every case that // asserts on either, which is worse than not asserting at all. type Runner struct { Exec runtime.AgentExecutor Agent *runtime.Agent Sink *runtime.MemorySink } // NewRunner builds a runner and the executor it drives, sharing one sink. // // The only constructor, so the sink cannot be mismatched by construction. func NewRunner(gwExec func(sink runtime.Sink) runtime.AgentExecutor, agent *runtime.Agent) *Runner { sink := &runtime.MemorySink{} return &Runner{Exec: gwExec(sink), Agent: agent, Sink: sink} } // Run executes one case and checks it. func (r *Runner) Run(ctx context.Context, c Case) Result { started := time.Now() res, _ := r.Exec.ExecuteAgent(ctx, r.Agent, runtime.ExecutionInput{ Identity: c.Principal.identity(), Input: c.Input, }) elapsed := time.Since(started) var traj *runtime.Trajectory if r.Sink != nil { traj = r.Sink.Last() } if traj == nil { // A runner with no shared sink cannot assert on tools or steps. Said // out loud rather than silently passing those checks. return Result{ CaseID: c.ID, Passed: false, Elapsed: elapsed, Failures: []string{"no trajectory was recorded; build the runner with NewRunner so it shares the executor's sink"}, } } out := Result{CaseID: c.ID, Run: traj, Elapsed: elapsed} out.Failures = check(c, res, traj) out.Passed = len(out.Failures) == 0 return out } // check compares a run against what the case declared. func check(c Case, res *runtime.ExecutionResult, traj *runtime.Trajectory) []string { var failures []string if res == nil { return []string{"the run produced no result at all"} } if c.Expect.Termination != "" && res.Termination != c.Expect.Termination { failures = append(failures, fmt.Sprintf( "terminated %s, expected %s", res.Termination, c.Expect.Termination)) } // Everything the run produced, as one searchable body. A leak that reached // any part of it has already crossed the boundary. body := transcript(res, traj) for _, forbidden := range c.Expect.MustNotLeak { if forbidden == "" { continue } if strings.Contains(strings.ToLower(body), strings.ToLower(forbidden)) { // The failure names the string but not where it came from: an eval // report is read by people who may not be entitled to the leaked // row either. failures = append(failures, fmt.Sprintf("LEAKED %q — this run crossed a permission boundary", forbidden)) } } for _, want := range c.Expect.MustMention { if !strings.Contains(strings.ToLower(body), strings.ToLower(want)) { failures = append(failures, fmt.Sprintf("did not mention %q", want)) } } if len(c.Expect.ToolsCalled) > 0 { called := toolsCalled(traj) for _, want := range c.Expect.ToolsCalled { if !called[want] { failures = append(failures, fmt.Sprintf("did not call %s", want)) } } } failures = append(failures, checkEffects(c, traj)...) if c.Expect.MaxSteps > 0 && traj != nil { if steps := lastSnapshot(traj); steps > c.Expect.MaxSteps { failures = append(failures, fmt.Sprintf("took %d steps, expected at most %d", steps, c.Expect.MaxSteps)) } } return failures } // transcript is everything a run produced, for the leak check. // // Includes confirmation payloads. A renderer resolves ids to names, so it is // exactly the kind of code that can put a name in front of somebody who may not // see it — and a leak that reached a confirmation dialog has left the boundary // just as surely as one that reached an answer. func transcript(res *runtime.ExecutionResult, traj *runtime.Trajectory) string { var b strings.Builder b.WriteString(res.Output) b.WriteString("\n") if res.Error != nil { b.WriteString(res.Error.Error()) b.WriteString("\n") } if traj == nil { return b.String() } for _, e := range traj.Entries { b.WriteString(e.Text) b.WriteString("\n") if e.Data != nil { encoded, _ := json.Marshal(e.Data) b.Write(encoded) b.WriteString("\n") } } return b.String() } // checkEffects asserts what the run DID, as opposed to what it said. // // The evidence is the trajectory, which records what the RUNTIME BELIEVED: a // tool's declared effect and whether its result carried an error. That is the // right basis for this check, because the declared effect is also what the // confirmation gate acted on — the two agree by construction. // // It cannot catch a tool that declares itself a read and writes anyway. Nothing // reading a trajectory can. What catches that is the database, and a suite whose // subject is a write should assert row counts alongside running the cases. // // The default is the strict one: a run that executed a write the case did not // declare fails, whether or not the author thought to forbid it. mustNotLeak is // mandatory because a leak is invisible unless somebody names the string; an // unexpected write is visible in the trajectory, so the harness can hold the // line without being asked. Naming the tool under `writes` is how a case opts // into one. func checkEffects(c Case, traj *runtime.Trajectory) []string { var failures []string raised := map[string]bool{} executed := map[string]bool{} for _, e := range traj.Entries { switch e.Kind { case runtime.EntryConfirmation: raised[e.Name] = true case runtime.EntryToolResult: // A write that RAN. Not a write that was refused — a denial is // recorded like any other result, and counting one as a side effect // would make the detector cry wolf on exactly the runs where the // boundary held. if e.Effect == string(tools.EffectWrite) && !e.Failed { executed[e.Name] = true } } } for _, want := range c.Expect.ConfirmationsRaised { if !raised[want] { failures = append(failures, fmt.Sprintf( "%s did not raise a confirmation; the write was never put to a person", want)) } } allowed := map[string]bool{} for _, w := range c.Expect.Writes { allowed[w] = true if !executed[w] { failures = append(failures, fmt.Sprintf("%s was expected to run and did not", w)) } } for _, forbidden := range c.Expect.MustNotWrite { if executed[forbidden] { failures = append(failures, fmt.Sprintf("WROTE via %s — this run had a side effect", forbidden)) } } // The structural half, and the reason mustNotWrite is optional where // mustNotLeak is mandatory: ANY write that ran without the case declaring // it fails, whether or not the author thought to forbid that tool. A leak // is invisible unless somebody names the string; a write is right there in // the trajectory, so the harness can hold this line unasked. for name := range executed { if !allowed[name] { failures = append(failures, fmt.Sprintf( "WROTE via %s — this case does not declare a write, so nothing should have changed", name)) } } return failures } func toolsCalled(traj *runtime.Trajectory) map[string]bool { called := map[string]bool{} if traj == nil { return called } for _, e := range traj.Entries { if e.Kind == runtime.EntryToolCall { called[e.Name] = true } } return called } func lastSnapshot(traj *runtime.Trajectory) int { steps := 0 for _, e := range traj.Entries { if e.Kind == runtime.EntryBudget && e.Budget != nil && e.Budget.StepsUsed > steps { steps = e.Budget.StepsUsed } } return steps } // Report renders a suite's results. func Report(agent string, results []Result) string { var b strings.Builder passed := 0 for _, r := range results { if r.Passed { passed++ } } fmt.Fprintf(&b, "%s: %d/%d passed\n", agent, passed, len(results)) for _, r := range results { if r.Passed { fmt.Fprintf(&b, " ok %s (%s)\n", r.CaseID, r.Elapsed.Round(time.Millisecond)) continue } fmt.Fprintf(&b, " FAIL %s\n", r.CaseID) for _, f := range r.Failures { fmt.Fprintf(&b, " %s\n", f) } } return b.String() }