1078 lines
38 KiB
Go
1078 lines
38 KiB
Go
package runtime
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"errors"
|
|
"fmt"
|
|
"strings"
|
|
"sync"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/krow/krow-backend/go-api/internal/authctx"
|
|
"github.com/krow/krow-backend/go-api/internal/gateway"
|
|
"github.com/krow/krow-backend/go-api/internal/knowledge"
|
|
"github.com/krow/krow-backend/go-api/internal/tools"
|
|
)
|
|
|
|
// fakeGateway stands in for the model. The whole point of Gateway being a
|
|
// one-method interface is that this exists and the eval harness never needs a
|
|
// network.
|
|
type fakeGateway struct {
|
|
mu sync.Mutex
|
|
calls int
|
|
text string
|
|
err error
|
|
usage gateway.Usage
|
|
delay time.Duration
|
|
lastReq gateway.Request
|
|
}
|
|
|
|
func (f *fakeGateway) Complete(ctx context.Context, req gateway.Request) (*gateway.Response, error) {
|
|
f.mu.Lock()
|
|
f.calls++
|
|
f.lastReq = req
|
|
f.mu.Unlock()
|
|
|
|
if f.delay > 0 {
|
|
select {
|
|
case <-time.After(f.delay):
|
|
case <-ctx.Done():
|
|
return nil, &gateway.Error{Code: gateway.CodeTimeout, Message: "context ended", Cause: ctx.Err()}
|
|
}
|
|
}
|
|
if f.err != nil {
|
|
// A failed call is still billed — the loop must charge for it.
|
|
return &gateway.Response{Usage: f.usage, Model: "fake-model"}, f.err
|
|
}
|
|
return &gateway.Response{
|
|
Text: f.text, StopReason: "end_turn", Usage: f.usage, Model: "fake-model", Tier: req.Tier,
|
|
}, nil
|
|
}
|
|
|
|
func testAgent() *Agent {
|
|
return &Agent{
|
|
ID: "activity-agent", Name: "Activity Agent", Version: 3,
|
|
Description: "The audit trail.",
|
|
Reasoning: "balanced",
|
|
Pages: []string{"activity"},
|
|
Instructions: "Answer about what has happened in this workspace.",
|
|
}
|
|
}
|
|
|
|
func testInput(q string) ExecutionInput {
|
|
return ExecutionInput{
|
|
Identity: authctx.Identity{UserID: "11111111-1111-1111-1111-111111111111", OrgID: "22222222-2222-2222-2222-222222222222"},
|
|
Input: q,
|
|
}
|
|
}
|
|
|
|
func TestCompletedRunRecordsTrajectory(t *testing.T) {
|
|
gw := &fakeGateway{text: "Twelve events, mostly logins.", usage: gateway.Usage{InputTokens: 900, OutputTokens: 120}}
|
|
sink := &MemorySink{}
|
|
exec := NewModelExecutor(gw, sink, nil)
|
|
|
|
res, err := exec.ExecuteAgent(context.Background(), testAgent(), testInput("what happened this week?"))
|
|
if err != nil {
|
|
t.Fatalf("unexpected error: %v", err)
|
|
}
|
|
if !res.Success || res.Termination != TerminationCompleted {
|
|
t.Fatalf("Success=%v Termination=%q, want true/Completed", res.Success, res.Termination)
|
|
}
|
|
if res.Output != gw.text {
|
|
t.Errorf("Output = %q, want %q", res.Output, gw.text)
|
|
}
|
|
if res.RunID == "" {
|
|
t.Error("a run must return an id the caller can point at")
|
|
}
|
|
if res.Usage.TotalTokens != 1020 || res.Usage.ModelCalls != 1 {
|
|
t.Errorf("Usage = %+v, want 1020 tokens over 1 call", res.Usage)
|
|
}
|
|
|
|
traj := sink.Last()
|
|
if traj == nil {
|
|
t.Fatal("no trajectory was persisted")
|
|
}
|
|
if traj.AgentID != "activity-agent" || traj.AgentVersion != 3 {
|
|
t.Errorf("trajectory identifies %s v%d, want activity-agent v3", traj.AgentID, traj.AgentVersion)
|
|
}
|
|
if traj.Model != "fake-model" {
|
|
t.Errorf("Model = %q — the trajectory must record what actually answered", traj.Model)
|
|
}
|
|
|
|
// A budget snapshot must precede the dispatch, so an overrun is
|
|
// diagnosable from the line that permitted it.
|
|
var sawBudgetBeforeAssistant bool
|
|
for _, e := range traj.Entries {
|
|
if e.Kind == EntryBudget {
|
|
sawBudgetBeforeAssistant = true
|
|
}
|
|
if e.Kind == EntryMessage && e.Role == "assistant" {
|
|
break
|
|
}
|
|
}
|
|
if !sawBudgetBeforeAssistant {
|
|
t.Error("no budget snapshot was recorded before the model call")
|
|
}
|
|
}
|
|
|
|
func TestStepBudgetIsClaimedBeforeDispatch(t *testing.T) {
|
|
gw := &fakeGateway{text: "hi"}
|
|
exec := NewModelExecutor(gw, &MemorySink{}, nil)
|
|
|
|
agent := testAgent()
|
|
agent.Reasoning = "fast"
|
|
|
|
// Drain the budget by hand to prove the claim happens before the call
|
|
// rather than after it: with no steps left, the gateway must not be
|
|
// reached at all.
|
|
budget := NewBudget(Limits{MaxSteps: 0, MaxToolCalls: 0, MaxTokens: 1000, Deadline: time.Minute})
|
|
if got := budget.ClaimStep(); got != TerminationBudgetExceeded {
|
|
t.Fatalf("ClaimStep on an exhausted budget = %q, want BudgetExceeded", got)
|
|
}
|
|
|
|
res, err := exec.ExecuteAgent(context.Background(), agent, testInput("anything"))
|
|
if err != nil {
|
|
t.Fatalf("a normal run should still work: %v", err)
|
|
}
|
|
if gw.calls != 1 {
|
|
t.Errorf("gateway called %d times, want exactly 1 — there are no tools yet to justify a second step", gw.calls)
|
|
}
|
|
if res.Termination != TerminationCompleted {
|
|
t.Errorf("Termination = %q, want Completed", res.Termination)
|
|
}
|
|
}
|
|
|
|
func TestRefusalIsChargedAndNotRetryable(t *testing.T) {
|
|
// A refusal is billed. A ledger that forgives it is one a loop will
|
|
// happily repeat against.
|
|
gw := &fakeGateway{
|
|
err: &gateway.Error{Code: gateway.CodeRefused, Message: "declined", Category: "cyber"},
|
|
usage: gateway.Usage{InputTokens: 500, OutputTokens: 0},
|
|
}
|
|
sink := &MemorySink{}
|
|
exec := NewModelExecutor(gw, sink, nil)
|
|
|
|
res, err := exec.ExecuteAgent(context.Background(), testAgent(), testInput("something refused"))
|
|
if err == nil {
|
|
t.Fatal("a refused run should return an error alongside its result")
|
|
}
|
|
if res.Termination != TerminationRefused {
|
|
t.Errorf("Termination = %q, want Refused", res.Termination)
|
|
}
|
|
if res.Success {
|
|
t.Error("a refused run is not a success")
|
|
}
|
|
if res.Usage.TotalTokens != 500 {
|
|
t.Errorf("Usage.TotalTokens = %d, want 500 — a refusal is still billed", res.Usage.TotalTokens)
|
|
}
|
|
|
|
traj := sink.Last()
|
|
if traj == nil || traj.Termination != TerminationRefused {
|
|
t.Fatal("a refused run must still be persisted, with its reason")
|
|
}
|
|
}
|
|
|
|
func TestDeadlineTerminatesAsDeadlineNotBudget(t *testing.T) {
|
|
// "Too slow" and "too expensive" are different questions to an operator.
|
|
gw := &fakeGateway{delay: 200 * time.Millisecond, text: "too late"}
|
|
exec := NewModelExecutor(gw, &MemorySink{}, nil)
|
|
|
|
agent := testAgent()
|
|
// A deadline shorter than the fake's delay, reached through the run
|
|
// context rather than by draining a counter.
|
|
res, err := exec.executeWithLimits(context.Background(), agent, testInput("slow one"),
|
|
Limits{MaxSteps: 8, MaxToolCalls: 12, MaxTokens: 100000, Deadline: 20 * time.Millisecond})
|
|
if err == nil {
|
|
t.Fatal("a run past its deadline should report an error")
|
|
}
|
|
if res.Termination != TerminationDeadline {
|
|
t.Errorf("Termination = %q, want Deadline", res.Termination)
|
|
}
|
|
}
|
|
|
|
func TestEmptyInputFailsBeforeSpendingAnything(t *testing.T) {
|
|
gw := &fakeGateway{text: "should not be reached"}
|
|
exec := NewModelExecutor(gw, &MemorySink{}, nil)
|
|
|
|
res, err := exec.ExecuteAgent(context.Background(), testAgent(), testInput(" "))
|
|
if err == nil {
|
|
t.Fatal("an empty question should be refused")
|
|
}
|
|
if gw.calls != 0 {
|
|
t.Errorf("gateway called %d times — an empty question must cost nothing", gw.calls)
|
|
}
|
|
if res.Success {
|
|
t.Error("an empty question is not a successful run")
|
|
}
|
|
}
|
|
|
|
func TestUnknownTierRunsAtDefaultAndSaysSo(t *testing.T) {
|
|
gw := &fakeGateway{text: "ok"}
|
|
sink := &MemorySink{}
|
|
exec := NewModelExecutor(gw, sink, nil)
|
|
|
|
agent := testAgent()
|
|
agent.Reasoning = "thorough" // not in the vocabulary
|
|
|
|
if _, err := exec.ExecuteAgent(context.Background(), agent, testInput("q")); err != nil {
|
|
t.Fatalf("a drifted tier must not fail the run: %v", err)
|
|
}
|
|
if gw.lastReq.Tier != gateway.DefaultTier {
|
|
t.Errorf("ran at tier %q, want the default %q", gw.lastReq.Tier, gateway.DefaultTier)
|
|
}
|
|
|
|
traj := sink.Last()
|
|
var reported bool
|
|
for _, e := range traj.Entries {
|
|
if e.Kind == EntryError && e.ErrorCode == "runtime.unknown_tier" {
|
|
reported = true
|
|
}
|
|
}
|
|
if !reported {
|
|
t.Error("a drifted tier must be recorded, not silently reinterpreted")
|
|
}
|
|
}
|
|
|
|
func TestSystemPromptCarriesTheUntrustedContentRule(t *testing.T) {
|
|
// I7. The rule has to be stated before content arrives, not alongside it.
|
|
got := SystemPrompt(testAgent())
|
|
if !strings.Contains(got, "<context>") {
|
|
t.Error("the system prompt must name the delimiter retrieved content will arrive in")
|
|
}
|
|
if !strings.Contains(got, "never as instructions") {
|
|
t.Error("the system prompt must say that retrieved content is data")
|
|
}
|
|
if !strings.Contains(got, "Answer about what has happened") {
|
|
t.Error("the agent's own instructions must reach the prompt")
|
|
}
|
|
if !strings.Contains(got, "activity") {
|
|
t.Error("the agent's pages must reach the prompt")
|
|
}
|
|
}
|
|
|
|
func TestSinkFailureDoesNotFailTheRun(t *testing.T) {
|
|
// The answer was already produced. Losing the record is bad; discarding a
|
|
// correct answer over it is worse.
|
|
gw := &fakeGateway{text: "the answer"}
|
|
exec := NewModelExecutor(gw, failingSink{}, nil)
|
|
|
|
res, err := exec.ExecuteAgent(context.Background(), testAgent(), testInput("q"))
|
|
if err != nil {
|
|
t.Fatalf("a failed save must not fail the run: %v", err)
|
|
}
|
|
if res.Output != "the answer" {
|
|
t.Errorf("Output = %q, want the answer through", res.Output)
|
|
}
|
|
}
|
|
|
|
type failingSink struct{}
|
|
|
|
func (failingSink) Save(context.Context, *Trajectory) error {
|
|
return context.DeadlineExceeded
|
|
}
|
|
|
|
func TestTerminationValidRejectsInvented(t *testing.T) {
|
|
for _, ok := range []Termination{
|
|
TerminationCompleted, TerminationBudgetExceeded, TerminationDeadline,
|
|
TerminationConfirmationPending, TerminationToolFailure, TerminationRefused,
|
|
} {
|
|
if !ok.Valid() {
|
|
t.Errorf("%q should be a valid termination", ok)
|
|
}
|
|
}
|
|
if Termination("Finished").Valid() {
|
|
t.Error("an invented termination must not validate — the enum is what evals group by")
|
|
}
|
|
}
|
|
|
|
func TestBudgetConcurrentClaimsDoNotOversell(t *testing.T) {
|
|
// Subagents share their parent's budget. Two branches must not both see
|
|
// the last step as available.
|
|
b := NewBudget(Limits{MaxSteps: 10, MaxToolCalls: 10, MaxTokens: 1000, Deadline: time.Minute})
|
|
|
|
var wg sync.WaitGroup
|
|
var mu sync.Mutex
|
|
granted := 0
|
|
for i := 0; i < 50; i++ {
|
|
wg.Add(1)
|
|
go func() {
|
|
defer wg.Done()
|
|
if b.ClaimStep() == "" {
|
|
mu.Lock()
|
|
granted++
|
|
mu.Unlock()
|
|
}
|
|
}()
|
|
}
|
|
wg.Wait()
|
|
|
|
if granted != 10 {
|
|
t.Errorf("%d steps granted from a budget of 10", granted)
|
|
}
|
|
if s := b.Snapshot(); s.StepsLeft != 0 || s.StepsUsed != 10 {
|
|
t.Errorf("Snapshot = %+v, want 10 used / 0 left", s)
|
|
}
|
|
}
|
|
|
|
/* ── The tool loop ──────────────────────────────────────────────────────── */
|
|
|
|
// scriptedGateway returns a queued sequence of responses, so a test can drive
|
|
// the loop through a tool call and out the other side.
|
|
type scriptedGateway struct {
|
|
mu sync.Mutex
|
|
steps []*gateway.Response
|
|
seen []gateway.Request
|
|
}
|
|
|
|
func (s *scriptedGateway) Complete(_ context.Context, req gateway.Request) (*gateway.Response, error) {
|
|
s.mu.Lock()
|
|
defer s.mu.Unlock()
|
|
s.seen = append(s.seen, req)
|
|
if len(s.steps) == 0 {
|
|
return &gateway.Response{Text: "done", StopReason: "end_turn", Model: "fake-model"}, nil
|
|
}
|
|
next := s.steps[0]
|
|
s.steps = s.steps[1:]
|
|
return next, nil
|
|
}
|
|
|
|
func countingTool(name string, calls *int) tools.Tool {
|
|
return tools.Tool{
|
|
Name: name,
|
|
Description: "A tool that counts how often it was called.",
|
|
InputSchema: map[string]any{"type": "object"},
|
|
Effect: tools.EffectRead,
|
|
Handler: func(context.Context, tools.Context, json.RawMessage) tools.Result {
|
|
*calls++
|
|
return tools.OK(map[string]any{"totalEvents": 12})
|
|
},
|
|
}
|
|
}
|
|
|
|
func TestLoopDispatchesToolsAndContinues(t *testing.T) {
|
|
var toolCalls int
|
|
reg := tools.NewRegistry()
|
|
reg.MustRegister(countingTool("activity_breakdown", &toolCalls))
|
|
|
|
gw := &scriptedGateway{steps: []*gateway.Response{
|
|
{
|
|
ToolCalls: []gateway.ToolCall{{ID: "call_1", Name: "activity_breakdown", Input: json.RawMessage(`{}`)}},
|
|
StopReason: "tool_use", Model: "fake-model",
|
|
Usage: gateway.Usage{InputTokens: 400, OutputTokens: 30},
|
|
},
|
|
{
|
|
Text: "Twelve events.", StopReason: "end_turn", Model: "fake-model",
|
|
Usage: gateway.Usage{InputTokens: 600, OutputTokens: 40},
|
|
},
|
|
}}
|
|
|
|
sink := &MemorySink{}
|
|
exec := NewModelExecutor(gw, sink, reg)
|
|
|
|
agent := testAgent()
|
|
agent.Tools = []string{"activity_breakdown"}
|
|
|
|
res, err := exec.ExecuteAgent(context.Background(), agent, testInput("what happened?"))
|
|
if err != nil {
|
|
t.Fatalf("unexpected error: %v", err)
|
|
}
|
|
if res.Termination != TerminationCompleted || res.Output != "Twelve events." {
|
|
t.Fatalf("Termination=%q Output=%q", res.Termination, res.Output)
|
|
}
|
|
if toolCalls != 1 {
|
|
t.Errorf("the tool ran %d times, want 1", toolCalls)
|
|
}
|
|
if res.Usage.ModelCalls != 2 {
|
|
t.Errorf("ModelCalls = %d, want 2 — one to ask, one to answer", res.Usage.ModelCalls)
|
|
}
|
|
|
|
// The tool must actually have been offered on the first request.
|
|
if len(gw.seen) < 1 || len(gw.seen[0].Tools) != 1 {
|
|
t.Fatalf("the first request offered %d tools, want 1", len(gw.seen[0].Tools))
|
|
}
|
|
// And the second request must carry the assistant turn plus the results,
|
|
// in that order — a result with no preceding call is malformed.
|
|
second := gw.seen[1].Messages
|
|
if len(second) != 3 {
|
|
t.Fatalf("second request had %d messages, want 3 (question, assistant+calls, results)", len(second))
|
|
}
|
|
if len(second[1].ToolCalls) != 1 || len(second[2].ToolResults) != 1 {
|
|
t.Errorf("the tool call and its result did not round-trip: %+v", second)
|
|
}
|
|
|
|
traj := sink.Last()
|
|
var sawCall, sawResult bool
|
|
for _, e := range traj.Entries {
|
|
switch e.Kind {
|
|
case EntryToolCall:
|
|
sawCall = true
|
|
case EntryToolResult:
|
|
sawResult = true
|
|
}
|
|
}
|
|
if !sawCall || !sawResult {
|
|
t.Error("the trajectory must record the tool call and its result — this is what evals assert on")
|
|
}
|
|
}
|
|
|
|
func TestToolCallBudgetEndsTheRun(t *testing.T) {
|
|
// A model that has exhausted its tool calls cannot make progress. Letting
|
|
// it continue would spend the step budget on turns that can only apologise.
|
|
var toolCalls int
|
|
reg := tools.NewRegistry()
|
|
reg.MustRegister(countingTool("activity_breakdown", &toolCalls))
|
|
|
|
// Always asks for a tool, forever.
|
|
gw := &loopingGateway{}
|
|
exec := NewModelExecutor(gw, &MemorySink{}, reg)
|
|
|
|
agent := testAgent()
|
|
agent.Tools = []string{"activity_breakdown"}
|
|
|
|
res, _ := exec.executeWithLimits(context.Background(), agent, testInput("go"),
|
|
Limits{MaxSteps: 50, MaxToolCalls: 2, MaxTokens: 1_000_000, Deadline: 30 * time.Second})
|
|
|
|
if res.Termination != TerminationBudgetExceeded {
|
|
t.Errorf("Termination = %q, want BudgetExceeded", res.Termination)
|
|
}
|
|
if toolCalls != 2 {
|
|
t.Errorf("the tool ran %d times, want exactly the 2 the budget allowed", toolCalls)
|
|
}
|
|
}
|
|
|
|
type loopingGateway struct{ n int }
|
|
|
|
func (l *loopingGateway) Complete(context.Context, gateway.Request) (*gateway.Response, error) {
|
|
l.n++
|
|
return &gateway.Response{
|
|
ToolCalls: []gateway.ToolCall{{ID: fmt.Sprintf("c%d", l.n), Name: "activity_breakdown", Input: json.RawMessage(`{}`)}},
|
|
StopReason: "tool_use", Model: "fake-model",
|
|
}, nil
|
|
}
|
|
|
|
func TestFailingToolIsHandedToTheModelNotSwallowed(t *testing.T) {
|
|
// §13: swallowing a tool error and letting the model narrate around it is
|
|
// the anti-pattern. The failure must reach the model AS a failure.
|
|
reg := tools.NewRegistry()
|
|
reg.MustRegister(tools.Tool{
|
|
Name: "activity_breakdown", Description: "Fails on purpose.",
|
|
InputSchema: map[string]any{"type": "object"}, Effect: tools.EffectRead,
|
|
Handler: func(context.Context, tools.Context, json.RawMessage) tools.Result {
|
|
return tools.Denied()
|
|
},
|
|
})
|
|
|
|
gw := &scriptedGateway{steps: []*gateway.Response{
|
|
{
|
|
ToolCalls: []gateway.ToolCall{{ID: "call_1", Name: "activity_breakdown", Input: json.RawMessage(`{}`)}},
|
|
StopReason: "tool_use", Model: "fake-model",
|
|
},
|
|
{Text: "I could not read that.", StopReason: "end_turn", Model: "fake-model"},
|
|
}}
|
|
|
|
exec := NewModelExecutor(gw, &MemorySink{}, reg)
|
|
agent := testAgent()
|
|
agent.Tools = []string{"activity_breakdown"}
|
|
|
|
res, err := exec.ExecuteAgent(context.Background(), agent, testInput("q"))
|
|
if err != nil {
|
|
t.Fatalf("a denied tool must not fail the run: %v", err)
|
|
}
|
|
if res.Termination != TerminationCompleted {
|
|
t.Errorf("Termination = %q, want Completed", res.Termination)
|
|
}
|
|
|
|
results := gw.seen[1].Messages[2].ToolResults
|
|
if len(results) != 1 || !results[0].IsError {
|
|
t.Fatalf("the denial did not reach the model as an error: %+v", results)
|
|
}
|
|
if !strings.Contains(results[0].Content, "tool.denied") {
|
|
t.Errorf("the model was not told why: %s", results[0].Content)
|
|
}
|
|
}
|
|
|
|
func TestUnknownToolIsRecordedNotFatal(t *testing.T) {
|
|
gw := &scriptedGateway{}
|
|
sink := &MemorySink{}
|
|
exec := NewModelExecutor(gw, sink, tools.NewRegistry())
|
|
|
|
agent := testAgent()
|
|
agent.Tools = []string{"a_tool_that_was_withdrawn"}
|
|
|
|
if _, err := exec.ExecuteAgent(context.Background(), agent, testInput("q")); err != nil {
|
|
t.Fatalf("a withdrawn tool must not fail the run: %v", err)
|
|
}
|
|
|
|
var reported bool
|
|
for _, e := range sink.Last().Entries {
|
|
if e.ErrorCode == "runtime.unknown_tool" {
|
|
reported = true
|
|
}
|
|
}
|
|
if !reported {
|
|
t.Error("a tool that no longer resolves must be recorded, not silently dropped")
|
|
}
|
|
}
|
|
|
|
/* ── Confirmation ───────────────────────────────────────────────────────── */
|
|
|
|
// writeTool is a write whose executions are counted.
|
|
//
|
|
// Same shape as countingTool, and the difference is the whole subject of the
|
|
// tests below: this one has an effect, so the loop must not let it happen
|
|
// without a person.
|
|
func writeTool(name string, calls *int) tools.Tool {
|
|
return tools.Tool{
|
|
Name: name,
|
|
Description: "A tool that changes something in the world.",
|
|
InputSchema: map[string]any{"type": "object"},
|
|
Effect: tools.EffectWrite,
|
|
Confirm: func(context.Context, tools.Context, json.RawMessage) (*tools.Confirmation, *tools.Result) {
|
|
return &tools.Confirmation{
|
|
Title: "Assign Maya Chen to Bar Supervisor",
|
|
Summary: "Maya Chen will be scheduled to work Friday evening.",
|
|
}, nil
|
|
},
|
|
Handler: func(context.Context, tools.Context, json.RawMessage) tools.Result {
|
|
*calls++
|
|
return tools.OK(map[string]any{"assignmentId": "a1"})
|
|
},
|
|
}
|
|
}
|
|
|
|
func TestARunThatWantsToWriteStopsAndAsks(t *testing.T) {
|
|
// I4 at the loop level. The model asked for a write; the run ends waiting
|
|
// on a person rather than performing it, and ConfirmationPending is a
|
|
// termination rather than an error because nothing failed — the run is
|
|
// simply not finished, and only a human can finish it.
|
|
var writes int
|
|
reg := tools.NewRegistry()
|
|
reg.MustRegister(writeTool("assign_worker", &writes))
|
|
|
|
gw := &scriptedGateway{steps: []*gateway.Response{
|
|
{
|
|
ToolCalls: []gateway.ToolCall{{ID: "call_1", Name: "assign_worker", Input: json.RawMessage(`{"worker":"maya"}`)}},
|
|
StopReason: "tool_use", Model: "fake-model",
|
|
Usage: gateway.Usage{InputTokens: 400, OutputTokens: 30},
|
|
},
|
|
// Never reached. Queued so that a loop which wrongly continued would
|
|
// finish Completed and fail loudly rather than hang.
|
|
{Text: "Done, assigned.", StopReason: "end_turn", Model: "fake-model"},
|
|
}}
|
|
|
|
sink := &MemorySink{}
|
|
exec := NewModelExecutor(gw, sink, reg)
|
|
agent := testAgent()
|
|
agent.Tools = []string{"assign_worker"}
|
|
|
|
res, err := exec.ExecuteAgent(context.Background(), agent, testInput("cover Friday's bar shift"))
|
|
|
|
if res.Termination != TerminationConfirmationPending {
|
|
t.Fatalf("Termination = %q, want ConfirmationPending", res.Termination)
|
|
}
|
|
if writes != 0 {
|
|
t.Fatalf("the write ran %d times without an approval", writes)
|
|
}
|
|
if len(res.Confirmations) != 1 {
|
|
t.Fatalf("%d confirmations returned, want 1", len(res.Confirmations))
|
|
}
|
|
if res.Confirmations[0].Token == "" {
|
|
t.Error("a confirmation with no token can never be answered")
|
|
}
|
|
if res.Confirmations[0].Title == "" {
|
|
t.Error("a confirmation with nothing written on it cannot be approved")
|
|
}
|
|
// Structured, per §10 — the surface derives the wording from the code.
|
|
var rtErr *RuntimeError
|
|
if !errors.As(err, &rtErr) || rtErr.Code != "runtime.confirmationpending" {
|
|
t.Errorf("want a structured runtime error, got %v", err)
|
|
}
|
|
// Exactly one model call: the loop stopped rather than taking another turn
|
|
// to talk about what it was about to do.
|
|
if res.Usage.ModelCalls != 1 {
|
|
t.Errorf("ModelCalls = %d, want 1 — the run should stop, not deliberate", res.Usage.ModelCalls)
|
|
}
|
|
}
|
|
|
|
func TestTheModelNeverSeesAPendingConfirmation(t *testing.T) {
|
|
// A confirmation is a question for a person. Handing it back as a tool
|
|
// result would invite the model to reason about it — to argue for approval,
|
|
// or to look for a route that does not ask. Neither is its business, and
|
|
// the cheapest way to guarantee it is for the text never to reach the
|
|
// conversation at all.
|
|
var writes int
|
|
reg := tools.NewRegistry()
|
|
reg.MustRegister(writeTool("assign_worker", &writes))
|
|
|
|
gw := &scriptedGateway{steps: []*gateway.Response{
|
|
{
|
|
ToolCalls: []gateway.ToolCall{{ID: "call_1", Name: "assign_worker", Input: json.RawMessage(`{}`)}},
|
|
StopReason: "tool_use", Model: "fake-model",
|
|
},
|
|
}}
|
|
exec := NewModelExecutor(gw, &MemorySink{}, reg)
|
|
agent := testAgent()
|
|
agent.Tools = []string{"assign_worker"}
|
|
|
|
exec.ExecuteAgent(context.Background(), agent, testInput("assign somebody"))
|
|
|
|
if len(gw.seen) != 1 {
|
|
t.Fatalf("the model was called %d times; a pending confirmation must end the run", len(gw.seen))
|
|
}
|
|
for _, req := range gw.seen {
|
|
for _, m := range req.Messages {
|
|
for _, r := range m.ToolResults {
|
|
if strings.Contains(r.Content, "Maya Chen") || strings.Contains(r.Content, "cnf_") {
|
|
t.Errorf("a confirmation reached the model as a tool result: %s", r.Content)
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
func TestAPendingConfirmationIsRecordedInTheTrajectory(t *testing.T) {
|
|
// "What was this person asked to approve, and when" is the question an
|
|
// audit of an agent-initiated write actually asks. It is only answerable if
|
|
// the description is kept alongside everything else the run did.
|
|
var writes int
|
|
reg := tools.NewRegistry()
|
|
reg.MustRegister(writeTool("assign_worker", &writes))
|
|
|
|
gw := &scriptedGateway{steps: []*gateway.Response{
|
|
{
|
|
ToolCalls: []gateway.ToolCall{{ID: "call_1", Name: "assign_worker", Input: json.RawMessage(`{}`)}},
|
|
StopReason: "tool_use", Model: "fake-model",
|
|
},
|
|
}}
|
|
sink := &MemorySink{}
|
|
exec := NewModelExecutor(gw, sink, reg)
|
|
agent := testAgent()
|
|
agent.Tools = []string{"assign_worker"}
|
|
|
|
exec.ExecuteAgent(context.Background(), agent, testInput("assign somebody"))
|
|
|
|
traj := sink.Last()
|
|
if traj == nil {
|
|
t.Fatal("no trajectory was saved for a run that ended pending")
|
|
}
|
|
if traj.Termination != TerminationConfirmationPending {
|
|
t.Errorf("trajectory termination = %q", traj.Termination)
|
|
}
|
|
var found bool
|
|
for _, e := range traj.Entries {
|
|
if e.Kind == EntryConfirmation && e.Name == "assign_worker" {
|
|
found = true
|
|
}
|
|
if e.Kind == EntryToolResult && e.Name == "assign_worker" {
|
|
t.Error("a write that never ran was recorded as having produced a result")
|
|
}
|
|
}
|
|
if !found {
|
|
t.Error("the confirmation was not recorded in the trajectory")
|
|
}
|
|
}
|
|
|
|
func TestReadsInTheSameTurnStillRunAndAreRecorded(t *testing.T) {
|
|
// A turn that mixes reads with a write should not throw the reads away.
|
|
// They are safe, they were already dispatched, and their results are part
|
|
// of the evidence for the write a person is about to consider.
|
|
var reads, writes int
|
|
reg := tools.NewRegistry()
|
|
reg.MustRegister(countingTool("activity_breakdown", &reads))
|
|
reg.MustRegister(writeTool("assign_worker", &writes))
|
|
|
|
gw := &scriptedGateway{steps: []*gateway.Response{
|
|
{
|
|
ToolCalls: []gateway.ToolCall{
|
|
{ID: "c1", Name: "activity_breakdown", Input: json.RawMessage(`{}`)},
|
|
{ID: "c2", Name: "assign_worker", Input: json.RawMessage(`{}`)},
|
|
},
|
|
StopReason: "tool_use", Model: "fake-model",
|
|
},
|
|
}}
|
|
sink := &MemorySink{}
|
|
exec := NewModelExecutor(gw, sink, reg)
|
|
agent := testAgent()
|
|
agent.Tools = []string{"activity_breakdown", "assign_worker"}
|
|
|
|
res, _ := exec.ExecuteAgent(context.Background(), agent, testInput("what happened, and cover Friday"))
|
|
|
|
if reads != 1 {
|
|
t.Errorf("the read ran %d times, want 1", reads)
|
|
}
|
|
if writes != 0 {
|
|
t.Errorf("the write ran %d times, want 0", writes)
|
|
}
|
|
if res.Termination != TerminationConfirmationPending {
|
|
t.Errorf("Termination = %q, want ConfirmationPending", res.Termination)
|
|
}
|
|
}
|
|
|
|
func TestAnApprovedRunResumesAndWrites(t *testing.T) {
|
|
// The full round trip: the model asks, the run stops, a person approves,
|
|
// and the write happens.
|
|
//
|
|
// The write is performed from what the person was SHOWN, before the model
|
|
// gets a turn — so it does not depend on the model reproducing the same
|
|
// tool call. On the resumed turn this model reports rather than repeating,
|
|
// which is what the "already carried out, do not repeat" context asks for.
|
|
var writes int
|
|
reg := tools.NewRegistry()
|
|
reg.MustRegister(writeTool("assign_worker", &writes))
|
|
agent := testAgent()
|
|
agent.Tools = []string{"assign_worker"}
|
|
|
|
call := gateway.ToolCall{ID: "c1", Name: "assign_worker", Input: json.RawMessage(`{"worker":"maya"}`)}
|
|
|
|
asking := &scriptedGateway{steps: []*gateway.Response{{
|
|
ToolCalls: []gateway.ToolCall{call}, StopReason: "tool_use", Model: "fake-model",
|
|
}}}
|
|
res, _ := NewModelExecutor(asking, &MemorySink{}, reg).
|
|
ExecuteAgent(context.Background(), agent, testInput("cover Friday"))
|
|
if len(res.Confirmations) != 1 {
|
|
t.Fatalf("%d confirmations, want 1", len(res.Confirmations))
|
|
}
|
|
if writes != 0 {
|
|
t.Fatal("the write happened before approval")
|
|
}
|
|
|
|
// A person approves. The resumed model reports what happened.
|
|
resuming := &scriptedGateway{steps: []*gateway.Response{
|
|
{Text: "Maya is on Friday's bar shift.", StopReason: "end_turn", Model: "fake-model"},
|
|
}}
|
|
in := testInput("cover Friday")
|
|
in.Confirmation = res.Confirmations[0].Token
|
|
|
|
sink := &MemorySink{}
|
|
out, err := NewModelExecutor(resuming, sink, reg).ExecuteAgent(context.Background(), agent, in)
|
|
if err != nil {
|
|
t.Fatalf("the resumed run failed: %v", err)
|
|
}
|
|
if out.Termination != TerminationCompleted {
|
|
t.Fatalf("Termination = %q, want Completed", out.Termination)
|
|
}
|
|
if writes != 1 {
|
|
t.Fatalf("the approved write ran %d times, want 1", writes)
|
|
}
|
|
|
|
// The model was TOLD what happened, so it can report rather than invent.
|
|
if len(resuming.seen) == 0 {
|
|
t.Fatal("the model was never called")
|
|
}
|
|
first := resuming.seen[0].Messages[0].Text
|
|
if !strings.Contains(first, "assign_worker") || !strings.Contains(first, "already") {
|
|
t.Errorf("the model was not told the write had happened:\n%s", first)
|
|
}
|
|
|
|
// And it is in the trajectory as a write that ran, not as a proposal.
|
|
var recorded bool
|
|
for _, e := range sink.Last().Entries {
|
|
if e.Kind == EntryToolResult && e.Name == "assign_worker" && e.Effect == "write" {
|
|
recorded = true
|
|
}
|
|
}
|
|
if !recorded {
|
|
t.Error("the approved write is not recorded in the trajectory")
|
|
}
|
|
}
|
|
|
|
func TestAnApprovedWriteHappensEvenIfTheModelDoesNotRepeatItself(t *testing.T) {
|
|
// The failure this whole path exists to fix.
|
|
//
|
|
// Observed in practice against a real model: a person clicked Approve, the
|
|
// resumed model asked a clarifying question instead of repeating the tool
|
|
// call, the token was never presented, and nothing was written. No error,
|
|
// no write, and nothing to tell the user why.
|
|
//
|
|
// Here the model says something completely unrelated. The write must still
|
|
// happen, because it was already approved.
|
|
var writes int
|
|
reg := tools.NewRegistry()
|
|
reg.MustRegister(writeTool("assign_worker", &writes))
|
|
agent := testAgent()
|
|
agent.Tools = []string{"assign_worker"}
|
|
|
|
asking := &scriptedGateway{steps: []*gateway.Response{{
|
|
ToolCalls: []gateway.ToolCall{{ID: "c1", Name: "assign_worker", Input: json.RawMessage(`{"worker":"maya"}`)}},
|
|
StopReason: "tool_use", Model: "fake-model",
|
|
}}}
|
|
res, _ := NewModelExecutor(asking, &MemorySink{}, reg).
|
|
ExecuteAgent(context.Background(), agent, testInput("cover Friday"))
|
|
|
|
// A model that asks a question rather than repeating the call.
|
|
wandering := &scriptedGateway{steps: []*gateway.Response{
|
|
{Text: "Which of the two bar roles did you mean?", StopReason: "end_turn", Model: "fake-model"},
|
|
}}
|
|
in := testInput("cover Friday")
|
|
in.Confirmation = res.Confirmations[0].Token
|
|
|
|
out, _ := NewModelExecutor(wandering, &MemorySink{}, reg).ExecuteAgent(context.Background(), agent, in)
|
|
|
|
if writes != 1 {
|
|
t.Fatalf("the approved write ran %d times, want 1 — an approval must not depend "+
|
|
"on the model repeating itself", writes)
|
|
}
|
|
if out.Termination != TerminationCompleted {
|
|
t.Errorf("Termination = %q, want Completed", out.Termination)
|
|
}
|
|
}
|
|
|
|
func TestApprovingOneWriteDoesNotApproveASecondInTheSameRun(t *testing.T) {
|
|
// Resuming with a token is not a permissive mode. It performs the one call
|
|
// it was issued for; any OTHER write the model then attempts — including
|
|
// repeating the approved one — raises its own confirmation and stops the
|
|
// run again.
|
|
//
|
|
// This is the failure a boolean would wave straight through: the model
|
|
// slipping an extra call into the turn that carries the approval.
|
|
var writes int
|
|
reg := tools.NewRegistry()
|
|
reg.MustRegister(writeTool("assign_worker", &writes))
|
|
agent := testAgent()
|
|
agent.Tools = []string{"assign_worker"}
|
|
|
|
approved := gateway.ToolCall{ID: "c1", Name: "assign_worker", Input: json.RawMessage(`{"worker":"maya"}`)}
|
|
sneaked := gateway.ToolCall{ID: "c2", Name: "assign_worker", Input: json.RawMessage(`{"worker":"dan"}`)}
|
|
|
|
asking := &scriptedGateway{steps: []*gateway.Response{{
|
|
ToolCalls: []gateway.ToolCall{approved}, StopReason: "tool_use", Model: "fake-model",
|
|
}}}
|
|
res, _ := NewModelExecutor(asking, &MemorySink{}, reg).
|
|
ExecuteAgent(context.Background(), agent, testInput("cover Friday"))
|
|
|
|
resuming := &scriptedGateway{steps: []*gateway.Response{{
|
|
ToolCalls: []gateway.ToolCall{sneaked}, StopReason: "tool_use", Model: "fake-model",
|
|
}}}
|
|
in := testInput("cover Friday")
|
|
in.Confirmation = res.Confirmations[0].Token
|
|
|
|
out, _ := NewModelExecutor(resuming, &MemorySink{}, reg).ExecuteAgent(context.Background(), agent, in)
|
|
|
|
// One write: the approved one. Dan was never approved.
|
|
if writes != 1 {
|
|
t.Fatalf("%d writes, want 1 — an approval for one worker authorised another", writes)
|
|
}
|
|
if out.Termination != TerminationConfirmationPending {
|
|
t.Fatalf("Termination = %q; the unapproved write should have stopped the run again", out.Termination)
|
|
}
|
|
if len(out.Confirmations) != 1 {
|
|
t.Fatalf("%d confirmations raised for the second write, want 1", len(out.Confirmations))
|
|
}
|
|
}
|
|
|
|
func TestASpentApprovalDoesNotFailTheRun(t *testing.T) {
|
|
// The most common cause of an unredeemable token is a person clicking
|
|
// Approve twice. Failing the run would answer a double-click with an error;
|
|
// answering the question again is what somebody actually wants.
|
|
var writes int
|
|
reg := tools.NewRegistry()
|
|
reg.MustRegister(writeTool("assign_worker", &writes))
|
|
agent := testAgent()
|
|
agent.Tools = []string{"assign_worker"}
|
|
|
|
in := testInput("cover Friday")
|
|
in.Confirmation = "cnf_never-existed"
|
|
|
|
gw := &scriptedGateway{steps: []*gateway.Response{
|
|
{Text: "Nothing to report.", StopReason: "end_turn", Model: "fake-model"},
|
|
}}
|
|
sink := &MemorySink{}
|
|
out, err := NewModelExecutor(gw, sink, reg).ExecuteAgent(context.Background(), agent, in)
|
|
if err != nil {
|
|
t.Fatalf("a spent approval ended the run: %v", err)
|
|
}
|
|
if out.Termination != TerminationCompleted {
|
|
t.Errorf("Termination = %q, want Completed", out.Termination)
|
|
}
|
|
if writes != 0 {
|
|
t.Error("a token that authorises nothing produced a write")
|
|
}
|
|
// Recorded, so "why did my approval do nothing" is answerable.
|
|
var noted bool
|
|
for _, e := range sink.Last().Entries {
|
|
if e.Kind == EntryError && e.ErrorCode == "runtime.confirmation_not_redeemable" {
|
|
noted = true
|
|
}
|
|
}
|
|
if !noted {
|
|
t.Error("an unredeemable approval was not recorded")
|
|
}
|
|
}
|
|
|
|
/* ── Retrieval ──────────────────────────────────────────────────────────── */
|
|
|
|
// scriptedRetriever returns a fixed corpus, and records what it was asked.
|
|
//
|
|
// The assertions below are mostly about the ARGUMENTS it received, not the
|
|
// results it gave: whose principal reached it, and which sources. Those two are
|
|
// I1 as far as the loop is concerned, and a fake is the only way to see them.
|
|
type scriptedRetriever struct {
|
|
results *knowledge.Results
|
|
err error
|
|
lastQ knowledge.Query
|
|
calls int
|
|
}
|
|
|
|
func (s *scriptedRetriever) Retrieve(_ context.Context, q knowledge.Query) (*knowledge.Results, error) {
|
|
s.calls++
|
|
s.lastQ = q
|
|
return s.results, s.err
|
|
}
|
|
|
|
func knowledgeAgent() *Agent {
|
|
a := testAgent()
|
|
a.KnowledgeSources = []string{"policy_docs"}
|
|
return a
|
|
}
|
|
|
|
func onePassage(text string) *knowledge.Results {
|
|
return &knowledge.Results{Chunks: []knowledge.Result{{
|
|
ChunkID: "chunk-1", DocumentID: "doc-1", Source: "policy_docs",
|
|
Title: "Staff Handbook", Heading: "Attendance", Text: text, Score: 0.03,
|
|
}}}
|
|
}
|
|
|
|
func TestRetrievedTextGoesInAUserMessageAndNeverTheSystemPrompt(t *testing.T) {
|
|
// I7, and the reason it is a POSITION rule rather than a filtering one.
|
|
// There is no reliable way to detect "ignore your instructions and..." in a
|
|
// document, so the defence is that document text physically cannot reach the
|
|
// place where instructions live.
|
|
ret := &scriptedRetriever{results: onePassage(
|
|
"Staff arriving more than ten minutes after the shift start are recorded as late.")}
|
|
gw := &scriptedGateway{}
|
|
|
|
exec := NewModelExecutor(gw, &MemorySink{}, nil).WithRetriever(ret)
|
|
res, err := exec.ExecuteAgent(context.Background(), knowledgeAgent(), testInput("when am I late?"))
|
|
if err != nil {
|
|
t.Fatalf("unexpected error: %v", err)
|
|
}
|
|
if res.Termination != TerminationCompleted {
|
|
t.Fatalf("Termination = %q", res.Termination)
|
|
}
|
|
if ret.calls != 1 {
|
|
t.Fatalf("the retriever was called %d times, want 1", ret.calls)
|
|
}
|
|
|
|
req := gw.seen[0]
|
|
if strings.Contains(req.System, "ten minutes") {
|
|
t.Error("retrieved document text reached the SYSTEM prompt")
|
|
}
|
|
if len(req.Messages) == 0 || req.Messages[0].Role != gateway.RoleUser {
|
|
t.Fatalf("the first message is not a user turn: %+v", req.Messages)
|
|
}
|
|
if !strings.Contains(req.Messages[0].Text, "ten minutes") {
|
|
t.Error("the retrieved passage never reached the model at all")
|
|
}
|
|
if !strings.Contains(req.Messages[0].Text, "<"+knowledge.ContextTag+">") {
|
|
t.Error("the passage arrived undelimited")
|
|
}
|
|
// The question is last, so the model reads the evidence and then the thing
|
|
// it is being asked.
|
|
if !strings.HasSuffix(strings.TrimSpace(req.Messages[0].Text), "when am I late?") {
|
|
t.Error("the caller's question did not come after the context block")
|
|
}
|
|
if !strings.Contains(req.System, knowledge.ContextInstruction) {
|
|
t.Error("the system prompt does not say that content inside the fence is data")
|
|
}
|
|
}
|
|
|
|
func TestRetrievalUsesTheCallersPrincipalAndTheSpecsSources(t *testing.T) {
|
|
// I1: the identity that reaches the knowledge layer is the CALLER's, and
|
|
// the sources are the SPEC's. Neither is anything the model influences.
|
|
ret := &scriptedRetriever{results: onePassage("Text.")}
|
|
exec := NewModelExecutor(&scriptedGateway{}, &MemorySink{}, nil).WithRetriever(ret)
|
|
|
|
agent := knowledgeAgent()
|
|
agent.KnowledgeSources = []string{"policy_docs", "worker_notes"}
|
|
input := testInput("what does the handbook say?")
|
|
|
|
if _, err := exec.ExecuteAgent(context.Background(), agent, input); err != nil {
|
|
t.Fatalf("unexpected error: %v", err)
|
|
}
|
|
|
|
if ret.lastQ.Principal.UserID != input.Identity.UserID ||
|
|
ret.lastQ.Principal.OrgID != input.Identity.OrgID {
|
|
t.Errorf("retrieval ran as %+v, want the caller %+v", ret.lastQ.Principal, input.Identity)
|
|
}
|
|
if strings.Join(ret.lastQ.Sources, ",") != "policy_docs,worker_notes" {
|
|
t.Errorf("retrieval searched %v, want the spec's sources", ret.lastQ.Sources)
|
|
}
|
|
}
|
|
|
|
func TestAnAgentWithNoKnowledgeDoesNotRetrieve(t *testing.T) {
|
|
// An empty source list means no knowledge. Calling the retriever with one
|
|
// would be the moment "no knowledge" turned into "all of it" — retrieval
|
|
// refuses that, but the loop should not ask.
|
|
ret := &scriptedRetriever{results: onePassage("Text.")}
|
|
exec := NewModelExecutor(&scriptedGateway{}, &MemorySink{}, nil).WithRetriever(ret)
|
|
|
|
if _, err := exec.ExecuteAgent(context.Background(), testAgent(), testInput("hello")); err != nil {
|
|
t.Fatalf("unexpected error: %v", err)
|
|
}
|
|
if ret.calls != 0 {
|
|
t.Errorf("an agent with no declared knowledge retrieved anyway (%d calls)", ret.calls)
|
|
}
|
|
}
|
|
|
|
func TestAFailedRetrievalDegradesTheRunRatherThanEndingIt(t *testing.T) {
|
|
// A knowledge layer that is down should cost grounding, not the answer. But
|
|
// it is recorded, because an ungrounded answer that LOOKS grounded is the
|
|
// worse outcome, and "the agent answered from nothing" is only diagnosable
|
|
// afterwards if the failure was written down at the time.
|
|
ret := &scriptedRetriever{err: &knowledge.Error{
|
|
Code: knowledge.ErrRetrieveFailed, Message: "the index is unreachable"}}
|
|
sink := &MemorySink{}
|
|
exec := NewModelExecutor(&scriptedGateway{}, sink, nil).WithRetriever(ret)
|
|
|
|
res, err := exec.ExecuteAgent(context.Background(), knowledgeAgent(), testInput("what does it say?"))
|
|
if err != nil {
|
|
t.Fatalf("a retrieval failure ended the run: %v", err)
|
|
}
|
|
if res.Termination != TerminationCompleted {
|
|
t.Fatalf("Termination = %q, want Completed", res.Termination)
|
|
}
|
|
|
|
var recorded bool
|
|
for _, e := range sink.Last().Entries {
|
|
if e.Kind == EntryError && e.ErrorCode == knowledge.ErrRetrieveFailed {
|
|
recorded = true
|
|
}
|
|
}
|
|
if !recorded {
|
|
t.Error("a retrieval failure was swallowed; nothing says the answer was ungrounded")
|
|
}
|
|
}
|
|
|
|
func TestTheTrajectoryRecordsWhichChunksGroundedTheAnswer(t *testing.T) {
|
|
// Ids and ranks only — copying the text in would make every trajectory a
|
|
// partial copy of the corpus, with all of the corpus's access rules and
|
|
// none of its retention.
|
|
ret := &scriptedRetriever{results: onePassage("Ten minutes late is late.")}
|
|
sink := &MemorySink{}
|
|
exec := NewModelExecutor(&scriptedGateway{}, sink, nil).WithRetriever(ret)
|
|
|
|
if _, err := exec.ExecuteAgent(context.Background(), knowledgeAgent(), testInput("when?")); err != nil {
|
|
t.Fatalf("unexpected error: %v", err)
|
|
}
|
|
|
|
var found bool
|
|
for _, e := range sink.Last().Entries {
|
|
if e.Kind != EntryRetrieval {
|
|
continue
|
|
}
|
|
found = true
|
|
encoded, _ := json.Marshal(e.Data)
|
|
if !strings.Contains(string(encoded), "chunk-1") {
|
|
t.Errorf("the retrieval entry does not name the chunk it returned: %s", encoded)
|
|
}
|
|
if strings.Contains(string(encoded), "Ten minutes late") {
|
|
t.Error("the trajectory recorded the chunk TEXT; it should record ids only")
|
|
}
|
|
}
|
|
if !found {
|
|
t.Error("nothing in the trajectory says a retrieval happened")
|
|
}
|
|
}
|