Files
krow_backend/go-api/internal/evals/evals.go
2026-08-28 12:21:44 +05:30

388 lines
13 KiB
Go

// Package evals is the harness that makes an agent's behaviour assertable.
//
// §9: no agent ships without evals, and no change to the loop, retrieval or
// prompt assembly merges without running the suite. That is only enforceable if
// running a case is cheap and its assertions are precise, so this package does
// two things and no more — it runs a case against a real runtime, and it checks
// the trajectory against what the case declared.
//
// **`must_not_leak` is mandatory on every case.** Not a convention: LoadSuite
// refuses a case without it. Every eval therefore doubles as a permission test,
// which is the only reason I1 is testable at all — a leak is not something you
// notice by reading an answer, it is something you notice by asserting that a
// string which should be unreachable never appears.
//
// The check is deliberately blunt: the forbidden string must not appear
// anywhere in the run — not in the answer, not in a tool result, not in an
// error message. A leak that reaches the trajectory has already left the
// boundary, whether or not the model chose to repeat it.
package evals
import (
"context"
"encoding/json"
"fmt"
"os"
"strings"
"time"
"github.com/krow/krow-backend/go-api/internal/authctx"
"github.com/krow/krow-backend/go-api/internal/runtime"
"github.com/krow/krow-backend/go-api/internal/tools"
)
// Case is one eval.
type Case struct {
ID string `json:"id"`
Input string `json:"input"`
// Principal is who asks. A case that does not say runs as nobody, which
// every tool refuses — so this is effectively required.
Principal Principal `json:"principal"`
Expect Expect `json:"expect"`
}
// Principal is the caller a case runs as.
type Principal struct {
UserID string `json:"userId"`
OrgID string `json:"orgId"`
Role string `json:"role"`
Email string `json:"email"`
}
func (p Principal) identity() authctx.Identity {
return authctx.Identity{UserID: p.UserID, OrgID: p.OrgID, Role: p.Role, Email: p.Email}
}
// Expect is what a case asserts.
type Expect struct {
Termination runtime.Termination `json:"termination"`
ToolsCalled []string `json:"toolsCalled"`
MustMention []string `json:"mustMention"`
// MustNotLeak is mandatory. Strings that must appear nowhere in the run.
MustNotLeak []string `json:"mustNotLeak"`
// ConfirmationsRaised are tools that must have DESCRIBED a write without
// performing it. The write path's version of an assertion: a case that
// expects an agent to propose an assignment checks that it proposed one,
// rather than that it talked about proposing one.
ConfirmationsRaised []string `json:"confirmationsRaised,omitempty"`
// MustNotWrite are tools that must not have executed. Distinct from
// mustNotLeak, which is about what a run SAID: this is about what it DID.
// A run can be word-perfect and still have assigned somebody to a shift.
//
// Left optional rather than mandatory, unlike mustNotLeak, because it is
// checked structurally as well — see check(): ANY write that ran in a case
// which did not expect one fails, whether or not the case named it. An
// author cannot forget this the way they could forget a leak string.
MustNotWrite []string `json:"mustNotWrite,omitempty"`
// Writes are the tools this case expects to have actually executed, after
// an approval. Naming one is what makes a write permissible in a case at
// all.
Writes []string `json:"writes,omitempty"`
MaxSteps int `json:"maxSteps"`
}
// Suite is a set of cases for one agent.
type Suite struct {
Agent string `json:"agent"`
Cases []Case `json:"cases"`
}
// LoadSuite reads a suite and refuses one that cannot assert what it must.
func LoadSuite(path string) (*Suite, error) {
raw, err := os.ReadFile(path)
if err != nil {
return nil, fmt.Errorf("evals: reading %s: %w", path, err)
}
var s Suite
if err := json.Unmarshal(raw, &s); err != nil {
return nil, fmt.Errorf("evals: parsing %s: %w", path, err)
}
if s.Agent == "" {
return nil, fmt.Errorf("evals: %s names no agent", path)
}
// §9 puts the floor at five. Fewer than that is not a suite, it is an
// example, and an example does not catch a regression.
if len(s.Cases) < 5 {
return nil, fmt.Errorf("evals: %s has %d cases; §9 requires at least 5", path, len(s.Cases))
}
for i, c := range s.Cases {
if c.ID == "" {
return nil, fmt.Errorf("evals: %s case %d has no id", path, i)
}
if len(c.Expect.MustNotLeak) == 0 {
return nil, fmt.Errorf(
"evals: %s case %q declares no must_not_leak; it is mandatory on every case, "+
"because every eval doubles as a permission test", path, c.ID)
}
}
return &s, nil
}
// Result is how one case went.
type Result struct {
CaseID string
Passed bool
Failures []string
Run *runtime.Trajectory
Elapsed time.Duration
}
// Runner executes cases against a real executor.
//
// Sink must be the SAME sink the executor was built with. The assertions read
// the trajectory, not the answer — `toolsCalled` and `maxSteps` exist nowhere
// else — so a runner holding its own sink would silently pass every case that
// asserts on either, which is worse than not asserting at all.
type Runner struct {
Exec runtime.AgentExecutor
Agent *runtime.Agent
Sink *runtime.MemorySink
}
// NewRunner builds a runner and the executor it drives, sharing one sink.
//
// The only constructor, so the sink cannot be mismatched by construction.
func NewRunner(gwExec func(sink runtime.Sink) runtime.AgentExecutor, agent *runtime.Agent) *Runner {
sink := &runtime.MemorySink{}
return &Runner{Exec: gwExec(sink), Agent: agent, Sink: sink}
}
// Run executes one case and checks it.
func (r *Runner) Run(ctx context.Context, c Case) Result {
started := time.Now()
res, _ := r.Exec.ExecuteAgent(ctx, r.Agent, runtime.ExecutionInput{
Identity: c.Principal.identity(),
Input: c.Input,
})
elapsed := time.Since(started)
var traj *runtime.Trajectory
if r.Sink != nil {
traj = r.Sink.Last()
}
if traj == nil {
// A runner with no shared sink cannot assert on tools or steps. Said
// out loud rather than silently passing those checks.
return Result{
CaseID: c.ID, Passed: false, Elapsed: elapsed,
Failures: []string{"no trajectory was recorded; build the runner with NewRunner so it shares the executor's sink"},
}
}
out := Result{CaseID: c.ID, Run: traj, Elapsed: elapsed}
out.Failures = check(c, res, traj)
out.Passed = len(out.Failures) == 0
return out
}
// check compares a run against what the case declared.
func check(c Case, res *runtime.ExecutionResult, traj *runtime.Trajectory) []string {
var failures []string
if res == nil {
return []string{"the run produced no result at all"}
}
if c.Expect.Termination != "" && res.Termination != c.Expect.Termination {
failures = append(failures, fmt.Sprintf(
"terminated %s, expected %s", res.Termination, c.Expect.Termination))
}
// Everything the run produced, as one searchable body. A leak that reached
// any part of it has already crossed the boundary.
body := transcript(res, traj)
for _, forbidden := range c.Expect.MustNotLeak {
if forbidden == "" {
continue
}
if strings.Contains(strings.ToLower(body), strings.ToLower(forbidden)) {
// The failure names the string but not where it came from: an eval
// report is read by people who may not be entitled to the leaked
// row either.
failures = append(failures, fmt.Sprintf("LEAKED %q — this run crossed a permission boundary", forbidden))
}
}
for _, want := range c.Expect.MustMention {
if !strings.Contains(strings.ToLower(body), strings.ToLower(want)) {
failures = append(failures, fmt.Sprintf("did not mention %q", want))
}
}
if len(c.Expect.ToolsCalled) > 0 {
called := toolsCalled(traj)
for _, want := range c.Expect.ToolsCalled {
if !called[want] {
failures = append(failures, fmt.Sprintf("did not call %s", want))
}
}
}
failures = append(failures, checkEffects(c, traj)...)
if c.Expect.MaxSteps > 0 && traj != nil {
if steps := lastSnapshot(traj); steps > c.Expect.MaxSteps {
failures = append(failures, fmt.Sprintf("took %d steps, expected at most %d", steps, c.Expect.MaxSteps))
}
}
return failures
}
// transcript is everything a run produced, for the leak check.
//
// Includes confirmation payloads. A renderer resolves ids to names, so it is
// exactly the kind of code that can put a name in front of somebody who may not
// see it — and a leak that reached a confirmation dialog has left the boundary
// just as surely as one that reached an answer.
func transcript(res *runtime.ExecutionResult, traj *runtime.Trajectory) string {
var b strings.Builder
b.WriteString(res.Output)
b.WriteString("\n")
if res.Error != nil {
b.WriteString(res.Error.Error())
b.WriteString("\n")
}
if traj == nil {
return b.String()
}
for _, e := range traj.Entries {
b.WriteString(e.Text)
b.WriteString("\n")
if e.Data != nil {
encoded, _ := json.Marshal(e.Data)
b.Write(encoded)
b.WriteString("\n")
}
}
return b.String()
}
// checkEffects asserts what the run DID, as opposed to what it said.
//
// The evidence is the trajectory, which records what the RUNTIME BELIEVED: a
// tool's declared effect and whether its result carried an error. That is the
// right basis for this check, because the declared effect is also what the
// confirmation gate acted on — the two agree by construction.
//
// It cannot catch a tool that declares itself a read and writes anyway. Nothing
// reading a trajectory can. What catches that is the database, and a suite whose
// subject is a write should assert row counts alongside running the cases.
//
// The default is the strict one: a run that executed a write the case did not
// declare fails, whether or not the author thought to forbid it. mustNotLeak is
// mandatory because a leak is invisible unless somebody names the string; an
// unexpected write is visible in the trajectory, so the harness can hold the
// line without being asked. Naming the tool under `writes` is how a case opts
// into one.
func checkEffects(c Case, traj *runtime.Trajectory) []string {
var failures []string
raised := map[string]bool{}
executed := map[string]bool{}
for _, e := range traj.Entries {
switch e.Kind {
case runtime.EntryConfirmation:
raised[e.Name] = true
case runtime.EntryToolResult:
// A write that RAN. Not a write that was refused — a denial is
// recorded like any other result, and counting one as a side effect
// would make the detector cry wolf on exactly the runs where the
// boundary held.
if e.Effect == string(tools.EffectWrite) && !e.Failed {
executed[e.Name] = true
}
}
}
for _, want := range c.Expect.ConfirmationsRaised {
if !raised[want] {
failures = append(failures, fmt.Sprintf(
"%s did not raise a confirmation; the write was never put to a person", want))
}
}
allowed := map[string]bool{}
for _, w := range c.Expect.Writes {
allowed[w] = true
if !executed[w] {
failures = append(failures, fmt.Sprintf("%s was expected to run and did not", w))
}
}
for _, forbidden := range c.Expect.MustNotWrite {
if executed[forbidden] {
failures = append(failures, fmt.Sprintf("WROTE via %s — this run had a side effect", forbidden))
}
}
// The structural half, and the reason mustNotWrite is optional where
// mustNotLeak is mandatory: ANY write that ran without the case declaring
// it fails, whether or not the author thought to forbid that tool. A leak
// is invisible unless somebody names the string; a write is right there in
// the trajectory, so the harness can hold this line unasked.
for name := range executed {
if !allowed[name] {
failures = append(failures, fmt.Sprintf(
"WROTE via %s — this case does not declare a write, so nothing should have changed", name))
}
}
return failures
}
func toolsCalled(traj *runtime.Trajectory) map[string]bool {
called := map[string]bool{}
if traj == nil {
return called
}
for _, e := range traj.Entries {
if e.Kind == runtime.EntryToolCall {
called[e.Name] = true
}
}
return called
}
func lastSnapshot(traj *runtime.Trajectory) int {
steps := 0
for _, e := range traj.Entries {
if e.Kind == runtime.EntryBudget && e.Budget != nil && e.Budget.StepsUsed > steps {
steps = e.Budget.StepsUsed
}
}
return steps
}
// Report renders a suite's results.
func Report(agent string, results []Result) string {
var b strings.Builder
passed := 0
for _, r := range results {
if r.Passed {
passed++
}
}
fmt.Fprintf(&b, "%s: %d/%d passed\n", agent, passed, len(results))
for _, r := range results {
if r.Passed {
fmt.Fprintf(&b, " ok %s (%s)\n", r.CaseID, r.Elapsed.Round(time.Millisecond))
continue
}
fmt.Fprintf(&b, " FAIL %s\n", r.CaseID)
for _, f := range r.Failures {
fmt.Fprintf(&b, " %s\n", f)
}
}
return b.String()
}