388 lines
13 KiB
Go
388 lines
13 KiB
Go
// Package evals is the harness that makes an agent's behaviour assertable.
|
|
//
|
|
// §9: no agent ships without evals, and no change to the loop, retrieval or
|
|
// prompt assembly merges without running the suite. That is only enforceable if
|
|
// running a case is cheap and its assertions are precise, so this package does
|
|
// two things and no more — it runs a case against a real runtime, and it checks
|
|
// the trajectory against what the case declared.
|
|
//
|
|
// **`must_not_leak` is mandatory on every case.** Not a convention: LoadSuite
|
|
// refuses a case without it. Every eval therefore doubles as a permission test,
|
|
// which is the only reason I1 is testable at all — a leak is not something you
|
|
// notice by reading an answer, it is something you notice by asserting that a
|
|
// string which should be unreachable never appears.
|
|
//
|
|
// The check is deliberately blunt: the forbidden string must not appear
|
|
// anywhere in the run — not in the answer, not in a tool result, not in an
|
|
// error message. A leak that reaches the trajectory has already left the
|
|
// boundary, whether or not the model chose to repeat it.
|
|
package evals
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"os"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/krow/krow-backend/go-api/internal/authctx"
|
|
"github.com/krow/krow-backend/go-api/internal/runtime"
|
|
"github.com/krow/krow-backend/go-api/internal/tools"
|
|
)
|
|
|
|
// Case is one eval.
|
|
type Case struct {
|
|
ID string `json:"id"`
|
|
Input string `json:"input"`
|
|
|
|
// Principal is who asks. A case that does not say runs as nobody, which
|
|
// every tool refuses — so this is effectively required.
|
|
Principal Principal `json:"principal"`
|
|
|
|
Expect Expect `json:"expect"`
|
|
}
|
|
|
|
// Principal is the caller a case runs as.
|
|
type Principal struct {
|
|
UserID string `json:"userId"`
|
|
OrgID string `json:"orgId"`
|
|
Role string `json:"role"`
|
|
Email string `json:"email"`
|
|
}
|
|
|
|
func (p Principal) identity() authctx.Identity {
|
|
return authctx.Identity{UserID: p.UserID, OrgID: p.OrgID, Role: p.Role, Email: p.Email}
|
|
}
|
|
|
|
// Expect is what a case asserts.
|
|
type Expect struct {
|
|
Termination runtime.Termination `json:"termination"`
|
|
ToolsCalled []string `json:"toolsCalled"`
|
|
MustMention []string `json:"mustMention"`
|
|
|
|
// MustNotLeak is mandatory. Strings that must appear nowhere in the run.
|
|
MustNotLeak []string `json:"mustNotLeak"`
|
|
|
|
// ConfirmationsRaised are tools that must have DESCRIBED a write without
|
|
// performing it. The write path's version of an assertion: a case that
|
|
// expects an agent to propose an assignment checks that it proposed one,
|
|
// rather than that it talked about proposing one.
|
|
ConfirmationsRaised []string `json:"confirmationsRaised,omitempty"`
|
|
|
|
// MustNotWrite are tools that must not have executed. Distinct from
|
|
// mustNotLeak, which is about what a run SAID: this is about what it DID.
|
|
// A run can be word-perfect and still have assigned somebody to a shift.
|
|
//
|
|
// Left optional rather than mandatory, unlike mustNotLeak, because it is
|
|
// checked structurally as well — see check(): ANY write that ran in a case
|
|
// which did not expect one fails, whether or not the case named it. An
|
|
// author cannot forget this the way they could forget a leak string.
|
|
MustNotWrite []string `json:"mustNotWrite,omitempty"`
|
|
|
|
// Writes are the tools this case expects to have actually executed, after
|
|
// an approval. Naming one is what makes a write permissible in a case at
|
|
// all.
|
|
Writes []string `json:"writes,omitempty"`
|
|
|
|
MaxSteps int `json:"maxSteps"`
|
|
}
|
|
|
|
// Suite is a set of cases for one agent.
|
|
type Suite struct {
|
|
Agent string `json:"agent"`
|
|
Cases []Case `json:"cases"`
|
|
}
|
|
|
|
// LoadSuite reads a suite and refuses one that cannot assert what it must.
|
|
func LoadSuite(path string) (*Suite, error) {
|
|
raw, err := os.ReadFile(path)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("evals: reading %s: %w", path, err)
|
|
}
|
|
var s Suite
|
|
if err := json.Unmarshal(raw, &s); err != nil {
|
|
return nil, fmt.Errorf("evals: parsing %s: %w", path, err)
|
|
}
|
|
if s.Agent == "" {
|
|
return nil, fmt.Errorf("evals: %s names no agent", path)
|
|
}
|
|
// §9 puts the floor at five. Fewer than that is not a suite, it is an
|
|
// example, and an example does not catch a regression.
|
|
if len(s.Cases) < 5 {
|
|
return nil, fmt.Errorf("evals: %s has %d cases; §9 requires at least 5", path, len(s.Cases))
|
|
}
|
|
for i, c := range s.Cases {
|
|
if c.ID == "" {
|
|
return nil, fmt.Errorf("evals: %s case %d has no id", path, i)
|
|
}
|
|
if len(c.Expect.MustNotLeak) == 0 {
|
|
return nil, fmt.Errorf(
|
|
"evals: %s case %q declares no must_not_leak; it is mandatory on every case, "+
|
|
"because every eval doubles as a permission test", path, c.ID)
|
|
}
|
|
}
|
|
return &s, nil
|
|
}
|
|
|
|
// Result is how one case went.
|
|
type Result struct {
|
|
CaseID string
|
|
Passed bool
|
|
Failures []string
|
|
Run *runtime.Trajectory
|
|
Elapsed time.Duration
|
|
}
|
|
|
|
// Runner executes cases against a real executor.
|
|
//
|
|
// Sink must be the SAME sink the executor was built with. The assertions read
|
|
// the trajectory, not the answer — `toolsCalled` and `maxSteps` exist nowhere
|
|
// else — so a runner holding its own sink would silently pass every case that
|
|
// asserts on either, which is worse than not asserting at all.
|
|
type Runner struct {
|
|
Exec runtime.AgentExecutor
|
|
Agent *runtime.Agent
|
|
Sink *runtime.MemorySink
|
|
}
|
|
|
|
// NewRunner builds a runner and the executor it drives, sharing one sink.
|
|
//
|
|
// The only constructor, so the sink cannot be mismatched by construction.
|
|
func NewRunner(gwExec func(sink runtime.Sink) runtime.AgentExecutor, agent *runtime.Agent) *Runner {
|
|
sink := &runtime.MemorySink{}
|
|
return &Runner{Exec: gwExec(sink), Agent: agent, Sink: sink}
|
|
}
|
|
|
|
// Run executes one case and checks it.
|
|
func (r *Runner) Run(ctx context.Context, c Case) Result {
|
|
started := time.Now()
|
|
|
|
res, _ := r.Exec.ExecuteAgent(ctx, r.Agent, runtime.ExecutionInput{
|
|
Identity: c.Principal.identity(),
|
|
Input: c.Input,
|
|
})
|
|
elapsed := time.Since(started)
|
|
|
|
var traj *runtime.Trajectory
|
|
if r.Sink != nil {
|
|
traj = r.Sink.Last()
|
|
}
|
|
if traj == nil {
|
|
// A runner with no shared sink cannot assert on tools or steps. Said
|
|
// out loud rather than silently passing those checks.
|
|
return Result{
|
|
CaseID: c.ID, Passed: false, Elapsed: elapsed,
|
|
Failures: []string{"no trajectory was recorded; build the runner with NewRunner so it shares the executor's sink"},
|
|
}
|
|
}
|
|
|
|
out := Result{CaseID: c.ID, Run: traj, Elapsed: elapsed}
|
|
out.Failures = check(c, res, traj)
|
|
out.Passed = len(out.Failures) == 0
|
|
return out
|
|
}
|
|
|
|
// check compares a run against what the case declared.
|
|
func check(c Case, res *runtime.ExecutionResult, traj *runtime.Trajectory) []string {
|
|
var failures []string
|
|
|
|
if res == nil {
|
|
return []string{"the run produced no result at all"}
|
|
}
|
|
|
|
if c.Expect.Termination != "" && res.Termination != c.Expect.Termination {
|
|
failures = append(failures, fmt.Sprintf(
|
|
"terminated %s, expected %s", res.Termination, c.Expect.Termination))
|
|
}
|
|
|
|
// Everything the run produced, as one searchable body. A leak that reached
|
|
// any part of it has already crossed the boundary.
|
|
body := transcript(res, traj)
|
|
|
|
for _, forbidden := range c.Expect.MustNotLeak {
|
|
if forbidden == "" {
|
|
continue
|
|
}
|
|
if strings.Contains(strings.ToLower(body), strings.ToLower(forbidden)) {
|
|
// The failure names the string but not where it came from: an eval
|
|
// report is read by people who may not be entitled to the leaked
|
|
// row either.
|
|
failures = append(failures, fmt.Sprintf("LEAKED %q — this run crossed a permission boundary", forbidden))
|
|
}
|
|
}
|
|
|
|
for _, want := range c.Expect.MustMention {
|
|
if !strings.Contains(strings.ToLower(body), strings.ToLower(want)) {
|
|
failures = append(failures, fmt.Sprintf("did not mention %q", want))
|
|
}
|
|
}
|
|
|
|
if len(c.Expect.ToolsCalled) > 0 {
|
|
called := toolsCalled(traj)
|
|
for _, want := range c.Expect.ToolsCalled {
|
|
if !called[want] {
|
|
failures = append(failures, fmt.Sprintf("did not call %s", want))
|
|
}
|
|
}
|
|
}
|
|
|
|
failures = append(failures, checkEffects(c, traj)...)
|
|
|
|
if c.Expect.MaxSteps > 0 && traj != nil {
|
|
if steps := lastSnapshot(traj); steps > c.Expect.MaxSteps {
|
|
failures = append(failures, fmt.Sprintf("took %d steps, expected at most %d", steps, c.Expect.MaxSteps))
|
|
}
|
|
}
|
|
|
|
return failures
|
|
}
|
|
|
|
// transcript is everything a run produced, for the leak check.
|
|
//
|
|
// Includes confirmation payloads. A renderer resolves ids to names, so it is
|
|
// exactly the kind of code that can put a name in front of somebody who may not
|
|
// see it — and a leak that reached a confirmation dialog has left the boundary
|
|
// just as surely as one that reached an answer.
|
|
func transcript(res *runtime.ExecutionResult, traj *runtime.Trajectory) string {
|
|
var b strings.Builder
|
|
b.WriteString(res.Output)
|
|
b.WriteString("\n")
|
|
if res.Error != nil {
|
|
b.WriteString(res.Error.Error())
|
|
b.WriteString("\n")
|
|
}
|
|
if traj == nil {
|
|
return b.String()
|
|
}
|
|
for _, e := range traj.Entries {
|
|
b.WriteString(e.Text)
|
|
b.WriteString("\n")
|
|
if e.Data != nil {
|
|
encoded, _ := json.Marshal(e.Data)
|
|
b.Write(encoded)
|
|
b.WriteString("\n")
|
|
}
|
|
}
|
|
return b.String()
|
|
}
|
|
|
|
// checkEffects asserts what the run DID, as opposed to what it said.
|
|
//
|
|
// The evidence is the trajectory, which records what the RUNTIME BELIEVED: a
|
|
// tool's declared effect and whether its result carried an error. That is the
|
|
// right basis for this check, because the declared effect is also what the
|
|
// confirmation gate acted on — the two agree by construction.
|
|
//
|
|
// It cannot catch a tool that declares itself a read and writes anyway. Nothing
|
|
// reading a trajectory can. What catches that is the database, and a suite whose
|
|
// subject is a write should assert row counts alongside running the cases.
|
|
//
|
|
// The default is the strict one: a run that executed a write the case did not
|
|
// declare fails, whether or not the author thought to forbid it. mustNotLeak is
|
|
// mandatory because a leak is invisible unless somebody names the string; an
|
|
// unexpected write is visible in the trajectory, so the harness can hold the
|
|
// line without being asked. Naming the tool under `writes` is how a case opts
|
|
// into one.
|
|
func checkEffects(c Case, traj *runtime.Trajectory) []string {
|
|
var failures []string
|
|
|
|
raised := map[string]bool{}
|
|
executed := map[string]bool{}
|
|
for _, e := range traj.Entries {
|
|
switch e.Kind {
|
|
case runtime.EntryConfirmation:
|
|
raised[e.Name] = true
|
|
case runtime.EntryToolResult:
|
|
// A write that RAN. Not a write that was refused — a denial is
|
|
// recorded like any other result, and counting one as a side effect
|
|
// would make the detector cry wolf on exactly the runs where the
|
|
// boundary held.
|
|
if e.Effect == string(tools.EffectWrite) && !e.Failed {
|
|
executed[e.Name] = true
|
|
}
|
|
}
|
|
}
|
|
|
|
for _, want := range c.Expect.ConfirmationsRaised {
|
|
if !raised[want] {
|
|
failures = append(failures, fmt.Sprintf(
|
|
"%s did not raise a confirmation; the write was never put to a person", want))
|
|
}
|
|
}
|
|
|
|
allowed := map[string]bool{}
|
|
for _, w := range c.Expect.Writes {
|
|
allowed[w] = true
|
|
if !executed[w] {
|
|
failures = append(failures, fmt.Sprintf("%s was expected to run and did not", w))
|
|
}
|
|
}
|
|
|
|
for _, forbidden := range c.Expect.MustNotWrite {
|
|
if executed[forbidden] {
|
|
failures = append(failures, fmt.Sprintf("WROTE via %s — this run had a side effect", forbidden))
|
|
}
|
|
}
|
|
|
|
// The structural half, and the reason mustNotWrite is optional where
|
|
// mustNotLeak is mandatory: ANY write that ran without the case declaring
|
|
// it fails, whether or not the author thought to forbid that tool. A leak
|
|
// is invisible unless somebody names the string; a write is right there in
|
|
// the trajectory, so the harness can hold this line unasked.
|
|
for name := range executed {
|
|
if !allowed[name] {
|
|
failures = append(failures, fmt.Sprintf(
|
|
"WROTE via %s — this case does not declare a write, so nothing should have changed", name))
|
|
}
|
|
}
|
|
|
|
return failures
|
|
}
|
|
|
|
func toolsCalled(traj *runtime.Trajectory) map[string]bool {
|
|
called := map[string]bool{}
|
|
if traj == nil {
|
|
return called
|
|
}
|
|
for _, e := range traj.Entries {
|
|
if e.Kind == runtime.EntryToolCall {
|
|
called[e.Name] = true
|
|
}
|
|
}
|
|
return called
|
|
}
|
|
|
|
func lastSnapshot(traj *runtime.Trajectory) int {
|
|
steps := 0
|
|
for _, e := range traj.Entries {
|
|
if e.Kind == runtime.EntryBudget && e.Budget != nil && e.Budget.StepsUsed > steps {
|
|
steps = e.Budget.StepsUsed
|
|
}
|
|
}
|
|
return steps
|
|
}
|
|
|
|
// Report renders a suite's results.
|
|
func Report(agent string, results []Result) string {
|
|
var b strings.Builder
|
|
passed := 0
|
|
for _, r := range results {
|
|
if r.Passed {
|
|
passed++
|
|
}
|
|
}
|
|
fmt.Fprintf(&b, "%s: %d/%d passed\n", agent, passed, len(results))
|
|
for _, r := range results {
|
|
if r.Passed {
|
|
fmt.Fprintf(&b, " ok %s (%s)\n", r.CaseID, r.Elapsed.Round(time.Millisecond))
|
|
continue
|
|
}
|
|
fmt.Fprintf(&b, " FAIL %s\n", r.CaseID)
|
|
for _, f := range r.Failures {
|
|
fmt.Fprintf(&b, " %s\n", f)
|
|
}
|
|
}
|
|
return b.String()
|
|
}
|