agent build
This commit is contained in:
387
go-api/internal/evals/evals.go
Normal file
387
go-api/internal/evals/evals.go
Normal file
@@ -0,0 +1,387 @@
|
||||
// Package evals is the harness that makes an agent's behaviour assertable.
|
||||
//
|
||||
// §9: no agent ships without evals, and no change to the loop, retrieval or
|
||||
// prompt assembly merges without running the suite. That is only enforceable if
|
||||
// running a case is cheap and its assertions are precise, so this package does
|
||||
// two things and no more — it runs a case against a real runtime, and it checks
|
||||
// the trajectory against what the case declared.
|
||||
//
|
||||
// **`must_not_leak` is mandatory on every case.** Not a convention: LoadSuite
|
||||
// refuses a case without it. Every eval therefore doubles as a permission test,
|
||||
// which is the only reason I1 is testable at all — a leak is not something you
|
||||
// notice by reading an answer, it is something you notice by asserting that a
|
||||
// string which should be unreachable never appears.
|
||||
//
|
||||
// The check is deliberately blunt: the forbidden string must not appear
|
||||
// anywhere in the run — not in the answer, not in a tool result, not in an
|
||||
// error message. A leak that reaches the trajectory has already left the
|
||||
// boundary, whether or not the model chose to repeat it.
|
||||
package evals
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/krow/krow-backend/go-api/internal/authctx"
|
||||
"github.com/krow/krow-backend/go-api/internal/runtime"
|
||||
"github.com/krow/krow-backend/go-api/internal/tools"
|
||||
)
|
||||
|
||||
// Case is one eval.
|
||||
type Case struct {
|
||||
ID string `json:"id"`
|
||||
Input string `json:"input"`
|
||||
|
||||
// Principal is who asks. A case that does not say runs as nobody, which
|
||||
// every tool refuses — so this is effectively required.
|
||||
Principal Principal `json:"principal"`
|
||||
|
||||
Expect Expect `json:"expect"`
|
||||
}
|
||||
|
||||
// Principal is the caller a case runs as.
|
||||
type Principal struct {
|
||||
UserID string `json:"userId"`
|
||||
OrgID string `json:"orgId"`
|
||||
Role string `json:"role"`
|
||||
Email string `json:"email"`
|
||||
}
|
||||
|
||||
func (p Principal) identity() authctx.Identity {
|
||||
return authctx.Identity{UserID: p.UserID, OrgID: p.OrgID, Role: p.Role, Email: p.Email}
|
||||
}
|
||||
|
||||
// Expect is what a case asserts.
|
||||
type Expect struct {
|
||||
Termination runtime.Termination `json:"termination"`
|
||||
ToolsCalled []string `json:"toolsCalled"`
|
||||
MustMention []string `json:"mustMention"`
|
||||
|
||||
// MustNotLeak is mandatory. Strings that must appear nowhere in the run.
|
||||
MustNotLeak []string `json:"mustNotLeak"`
|
||||
|
||||
// ConfirmationsRaised are tools that must have DESCRIBED a write without
|
||||
// performing it. The write path's version of an assertion: a case that
|
||||
// expects an agent to propose an assignment checks that it proposed one,
|
||||
// rather than that it talked about proposing one.
|
||||
ConfirmationsRaised []string `json:"confirmationsRaised,omitempty"`
|
||||
|
||||
// MustNotWrite are tools that must not have executed. Distinct from
|
||||
// mustNotLeak, which is about what a run SAID: this is about what it DID.
|
||||
// A run can be word-perfect and still have assigned somebody to a shift.
|
||||
//
|
||||
// Left optional rather than mandatory, unlike mustNotLeak, because it is
|
||||
// checked structurally as well — see check(): ANY write that ran in a case
|
||||
// which did not expect one fails, whether or not the case named it. An
|
||||
// author cannot forget this the way they could forget a leak string.
|
||||
MustNotWrite []string `json:"mustNotWrite,omitempty"`
|
||||
|
||||
// Writes are the tools this case expects to have actually executed, after
|
||||
// an approval. Naming one is what makes a write permissible in a case at
|
||||
// all.
|
||||
Writes []string `json:"writes,omitempty"`
|
||||
|
||||
MaxSteps int `json:"maxSteps"`
|
||||
}
|
||||
|
||||
// Suite is a set of cases for one agent.
|
||||
type Suite struct {
|
||||
Agent string `json:"agent"`
|
||||
Cases []Case `json:"cases"`
|
||||
}
|
||||
|
||||
// LoadSuite reads a suite and refuses one that cannot assert what it must.
|
||||
func LoadSuite(path string) (*Suite, error) {
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("evals: reading %s: %w", path, err)
|
||||
}
|
||||
var s Suite
|
||||
if err := json.Unmarshal(raw, &s); err != nil {
|
||||
return nil, fmt.Errorf("evals: parsing %s: %w", path, err)
|
||||
}
|
||||
if s.Agent == "" {
|
||||
return nil, fmt.Errorf("evals: %s names no agent", path)
|
||||
}
|
||||
// §9 puts the floor at five. Fewer than that is not a suite, it is an
|
||||
// example, and an example does not catch a regression.
|
||||
if len(s.Cases) < 5 {
|
||||
return nil, fmt.Errorf("evals: %s has %d cases; §9 requires at least 5", path, len(s.Cases))
|
||||
}
|
||||
for i, c := range s.Cases {
|
||||
if c.ID == "" {
|
||||
return nil, fmt.Errorf("evals: %s case %d has no id", path, i)
|
||||
}
|
||||
if len(c.Expect.MustNotLeak) == 0 {
|
||||
return nil, fmt.Errorf(
|
||||
"evals: %s case %q declares no must_not_leak; it is mandatory on every case, "+
|
||||
"because every eval doubles as a permission test", path, c.ID)
|
||||
}
|
||||
}
|
||||
return &s, nil
|
||||
}
|
||||
|
||||
// Result is how one case went.
|
||||
type Result struct {
|
||||
CaseID string
|
||||
Passed bool
|
||||
Failures []string
|
||||
Run *runtime.Trajectory
|
||||
Elapsed time.Duration
|
||||
}
|
||||
|
||||
// Runner executes cases against a real executor.
|
||||
//
|
||||
// Sink must be the SAME sink the executor was built with. The assertions read
|
||||
// the trajectory, not the answer — `toolsCalled` and `maxSteps` exist nowhere
|
||||
// else — so a runner holding its own sink would silently pass every case that
|
||||
// asserts on either, which is worse than not asserting at all.
|
||||
type Runner struct {
|
||||
Exec runtime.AgentExecutor
|
||||
Agent *runtime.Agent
|
||||
Sink *runtime.MemorySink
|
||||
}
|
||||
|
||||
// NewRunner builds a runner and the executor it drives, sharing one sink.
|
||||
//
|
||||
// The only constructor, so the sink cannot be mismatched by construction.
|
||||
func NewRunner(gwExec func(sink runtime.Sink) runtime.AgentExecutor, agent *runtime.Agent) *Runner {
|
||||
sink := &runtime.MemorySink{}
|
||||
return &Runner{Exec: gwExec(sink), Agent: agent, Sink: sink}
|
||||
}
|
||||
|
||||
// Run executes one case and checks it.
|
||||
func (r *Runner) Run(ctx context.Context, c Case) Result {
|
||||
started := time.Now()
|
||||
|
||||
res, _ := r.Exec.ExecuteAgent(ctx, r.Agent, runtime.ExecutionInput{
|
||||
Identity: c.Principal.identity(),
|
||||
Input: c.Input,
|
||||
})
|
||||
elapsed := time.Since(started)
|
||||
|
||||
var traj *runtime.Trajectory
|
||||
if r.Sink != nil {
|
||||
traj = r.Sink.Last()
|
||||
}
|
||||
if traj == nil {
|
||||
// A runner with no shared sink cannot assert on tools or steps. Said
|
||||
// out loud rather than silently passing those checks.
|
||||
return Result{
|
||||
CaseID: c.ID, Passed: false, Elapsed: elapsed,
|
||||
Failures: []string{"no trajectory was recorded; build the runner with NewRunner so it shares the executor's sink"},
|
||||
}
|
||||
}
|
||||
|
||||
out := Result{CaseID: c.ID, Run: traj, Elapsed: elapsed}
|
||||
out.Failures = check(c, res, traj)
|
||||
out.Passed = len(out.Failures) == 0
|
||||
return out
|
||||
}
|
||||
|
||||
// check compares a run against what the case declared.
|
||||
func check(c Case, res *runtime.ExecutionResult, traj *runtime.Trajectory) []string {
|
||||
var failures []string
|
||||
|
||||
if res == nil {
|
||||
return []string{"the run produced no result at all"}
|
||||
}
|
||||
|
||||
if c.Expect.Termination != "" && res.Termination != c.Expect.Termination {
|
||||
failures = append(failures, fmt.Sprintf(
|
||||
"terminated %s, expected %s", res.Termination, c.Expect.Termination))
|
||||
}
|
||||
|
||||
// Everything the run produced, as one searchable body. A leak that reached
|
||||
// any part of it has already crossed the boundary.
|
||||
body := transcript(res, traj)
|
||||
|
||||
for _, forbidden := range c.Expect.MustNotLeak {
|
||||
if forbidden == "" {
|
||||
continue
|
||||
}
|
||||
if strings.Contains(strings.ToLower(body), strings.ToLower(forbidden)) {
|
||||
// The failure names the string but not where it came from: an eval
|
||||
// report is read by people who may not be entitled to the leaked
|
||||
// row either.
|
||||
failures = append(failures, fmt.Sprintf("LEAKED %q — this run crossed a permission boundary", forbidden))
|
||||
}
|
||||
}
|
||||
|
||||
for _, want := range c.Expect.MustMention {
|
||||
if !strings.Contains(strings.ToLower(body), strings.ToLower(want)) {
|
||||
failures = append(failures, fmt.Sprintf("did not mention %q", want))
|
||||
}
|
||||
}
|
||||
|
||||
if len(c.Expect.ToolsCalled) > 0 {
|
||||
called := toolsCalled(traj)
|
||||
for _, want := range c.Expect.ToolsCalled {
|
||||
if !called[want] {
|
||||
failures = append(failures, fmt.Sprintf("did not call %s", want))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
failures = append(failures, checkEffects(c, traj)...)
|
||||
|
||||
if c.Expect.MaxSteps > 0 && traj != nil {
|
||||
if steps := lastSnapshot(traj); steps > c.Expect.MaxSteps {
|
||||
failures = append(failures, fmt.Sprintf("took %d steps, expected at most %d", steps, c.Expect.MaxSteps))
|
||||
}
|
||||
}
|
||||
|
||||
return failures
|
||||
}
|
||||
|
||||
// transcript is everything a run produced, for the leak check.
|
||||
//
|
||||
// Includes confirmation payloads. A renderer resolves ids to names, so it is
|
||||
// exactly the kind of code that can put a name in front of somebody who may not
|
||||
// see it — and a leak that reached a confirmation dialog has left the boundary
|
||||
// just as surely as one that reached an answer.
|
||||
func transcript(res *runtime.ExecutionResult, traj *runtime.Trajectory) string {
|
||||
var b strings.Builder
|
||||
b.WriteString(res.Output)
|
||||
b.WriteString("\n")
|
||||
if res.Error != nil {
|
||||
b.WriteString(res.Error.Error())
|
||||
b.WriteString("\n")
|
||||
}
|
||||
if traj == nil {
|
||||
return b.String()
|
||||
}
|
||||
for _, e := range traj.Entries {
|
||||
b.WriteString(e.Text)
|
||||
b.WriteString("\n")
|
||||
if e.Data != nil {
|
||||
encoded, _ := json.Marshal(e.Data)
|
||||
b.Write(encoded)
|
||||
b.WriteString("\n")
|
||||
}
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// checkEffects asserts what the run DID, as opposed to what it said.
|
||||
//
|
||||
// The evidence is the trajectory, which records what the RUNTIME BELIEVED: a
|
||||
// tool's declared effect and whether its result carried an error. That is the
|
||||
// right basis for this check, because the declared effect is also what the
|
||||
// confirmation gate acted on — the two agree by construction.
|
||||
//
|
||||
// It cannot catch a tool that declares itself a read and writes anyway. Nothing
|
||||
// reading a trajectory can. What catches that is the database, and a suite whose
|
||||
// subject is a write should assert row counts alongside running the cases.
|
||||
//
|
||||
// The default is the strict one: a run that executed a write the case did not
|
||||
// declare fails, whether or not the author thought to forbid it. mustNotLeak is
|
||||
// mandatory because a leak is invisible unless somebody names the string; an
|
||||
// unexpected write is visible in the trajectory, so the harness can hold the
|
||||
// line without being asked. Naming the tool under `writes` is how a case opts
|
||||
// into one.
|
||||
func checkEffects(c Case, traj *runtime.Trajectory) []string {
|
||||
var failures []string
|
||||
|
||||
raised := map[string]bool{}
|
||||
executed := map[string]bool{}
|
||||
for _, e := range traj.Entries {
|
||||
switch e.Kind {
|
||||
case runtime.EntryConfirmation:
|
||||
raised[e.Name] = true
|
||||
case runtime.EntryToolResult:
|
||||
// A write that RAN. Not a write that was refused — a denial is
|
||||
// recorded like any other result, and counting one as a side effect
|
||||
// would make the detector cry wolf on exactly the runs where the
|
||||
// boundary held.
|
||||
if e.Effect == string(tools.EffectWrite) && !e.Failed {
|
||||
executed[e.Name] = true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for _, want := range c.Expect.ConfirmationsRaised {
|
||||
if !raised[want] {
|
||||
failures = append(failures, fmt.Sprintf(
|
||||
"%s did not raise a confirmation; the write was never put to a person", want))
|
||||
}
|
||||
}
|
||||
|
||||
allowed := map[string]bool{}
|
||||
for _, w := range c.Expect.Writes {
|
||||
allowed[w] = true
|
||||
if !executed[w] {
|
||||
failures = append(failures, fmt.Sprintf("%s was expected to run and did not", w))
|
||||
}
|
||||
}
|
||||
|
||||
for _, forbidden := range c.Expect.MustNotWrite {
|
||||
if executed[forbidden] {
|
||||
failures = append(failures, fmt.Sprintf("WROTE via %s — this run had a side effect", forbidden))
|
||||
}
|
||||
}
|
||||
|
||||
// The structural half, and the reason mustNotWrite is optional where
|
||||
// mustNotLeak is mandatory: ANY write that ran without the case declaring
|
||||
// it fails, whether or not the author thought to forbid that tool. A leak
|
||||
// is invisible unless somebody names the string; a write is right there in
|
||||
// the trajectory, so the harness can hold this line unasked.
|
||||
for name := range executed {
|
||||
if !allowed[name] {
|
||||
failures = append(failures, fmt.Sprintf(
|
||||
"WROTE via %s — this case does not declare a write, so nothing should have changed", name))
|
||||
}
|
||||
}
|
||||
|
||||
return failures
|
||||
}
|
||||
|
||||
func toolsCalled(traj *runtime.Trajectory) map[string]bool {
|
||||
called := map[string]bool{}
|
||||
if traj == nil {
|
||||
return called
|
||||
}
|
||||
for _, e := range traj.Entries {
|
||||
if e.Kind == runtime.EntryToolCall {
|
||||
called[e.Name] = true
|
||||
}
|
||||
}
|
||||
return called
|
||||
}
|
||||
|
||||
func lastSnapshot(traj *runtime.Trajectory) int {
|
||||
steps := 0
|
||||
for _, e := range traj.Entries {
|
||||
if e.Kind == runtime.EntryBudget && e.Budget != nil && e.Budget.StepsUsed > steps {
|
||||
steps = e.Budget.StepsUsed
|
||||
}
|
||||
}
|
||||
return steps
|
||||
}
|
||||
|
||||
// Report renders a suite's results.
|
||||
func Report(agent string, results []Result) string {
|
||||
var b strings.Builder
|
||||
passed := 0
|
||||
for _, r := range results {
|
||||
if r.Passed {
|
||||
passed++
|
||||
}
|
||||
}
|
||||
fmt.Fprintf(&b, "%s: %d/%d passed\n", agent, passed, len(results))
|
||||
for _, r := range results {
|
||||
if r.Passed {
|
||||
fmt.Fprintf(&b, " ok %s (%s)\n", r.CaseID, r.Elapsed.Round(time.Millisecond))
|
||||
continue
|
||||
}
|
||||
fmt.Fprintf(&b, " FAIL %s\n", r.CaseID)
|
||||
for _, f := range r.Failures {
|
||||
fmt.Fprintf(&b, " %s\n", f)
|
||||
}
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
Reference in New Issue
Block a user