The ids always travelled TO the model; nothing a reader could look at ever travelled back. So the panel stripped the citations the model wrote — there was nowhere to put them — and a grounded answer became indistinguishable from an invented one, which is the opposite of what citing is for. A run now carries its sources: the id the model was told to cite, the document title, the heading, and the opening of the passage. Both the JSON and the streamed paths return them, because both build the same response. Source is NOT knowledge.Result. That type carries ranks, scores and the whole chunk, which exist to debug a retrieval rather than to be shown: an RRF score is a rank and would be read as a percentage, and the full text would make the response larger than the answer. This is the subset a citation needs. Snippets are capped at 240 runes. A reader checking "the policy says X" needs to recognise the passage, not to receive the corpus one answer at a time. A run that retrieved nothing carries nothing rather than an empty list, so the panel has no heading to draw for absent evidence — which is most runs, since seven of nine agents answer from tools. Next, and only now possible: the panel can render these and stop stripping the citations that point at them. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
186 lines
6.8 KiB
Go
186 lines
6.8 KiB
Go
package runtime
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"strings"
|
|
"testing"
|
|
|
|
"github.com/krow/krow-backend/go-api/internal/knowledge"
|
|
|
|
"github.com/krow/krow-backend/go-api/internal/authctx"
|
|
"github.com/krow/krow-backend/go-api/internal/memory"
|
|
)
|
|
|
|
type fakeMemory struct {
|
|
records []memory.Record
|
|
degraded string
|
|
err error
|
|
asked string
|
|
}
|
|
|
|
func (f *fakeMemory) Recall(_ context.Context, _ authctx.Identity, question string, _ int) ([]memory.Record, string, error) {
|
|
f.asked = question
|
|
return f.records, f.degraded, f.err
|
|
}
|
|
|
|
func TestARecalledMemoryReachesThePrompt(t *testing.T) {
|
|
gw := &fakeGateway{text: "answered"}
|
|
mem := &fakeMemory{records: []memory.Record{
|
|
{SubjectType: memory.SubjectWorkspace, Author: memory.AuthorModel, Text: "Thursdays are short-staffed."},
|
|
}}
|
|
exec := NewModelExecutor(gw, &MemorySink{}, nil).WithMemory(mem)
|
|
|
|
if _, err := exec.ExecuteAgent(context.Background(), testAgent(), testInput("who is free?")); err != nil {
|
|
t.Fatalf("run failed: %v", err)
|
|
}
|
|
sent := gw.lastReq.Messages[0].Text
|
|
if !strings.Contains(sent, "Thursdays are short-staffed.") {
|
|
t.Errorf("the memory did not reach the model:\n%s", sent)
|
|
}
|
|
if !strings.Contains(sent, "<memory>") {
|
|
t.Error("the memory was not fenced")
|
|
}
|
|
}
|
|
|
|
// The question goes last. A model reads the last thing and answers it; put the
|
|
// evidence after it and the evidence becomes the prompt.
|
|
func TestTheQuestionStaysLastWhenMemoryIsCarried(t *testing.T) {
|
|
gw := &fakeGateway{text: "answered"}
|
|
mem := &fakeMemory{records: []memory.Record{
|
|
{SubjectType: memory.SubjectWorkspace, Author: memory.AuthorModel, Text: "A remembered thing."},
|
|
}}
|
|
exec := NewModelExecutor(gw, &MemorySink{}, nil).WithMemory(mem)
|
|
exec.ExecuteAgent(context.Background(), testAgent(), testInput("who is free?"))
|
|
|
|
sent := gw.lastReq.Messages[0].Text
|
|
if !strings.HasSuffix(strings.TrimSpace(sent), "who is free?") {
|
|
t.Errorf("the question is not last:\n%s", sent)
|
|
}
|
|
}
|
|
|
|
// A store that is down must not take the run with it: an answer without
|
|
// memory is worse, not wrong.
|
|
func TestAMemoryFailureDoesNotFailTheRun(t *testing.T) {
|
|
gw := &fakeGateway{text: "answered anyway"}
|
|
mem := &fakeMemory{err: errors.New("the memory store is unreachable")}
|
|
sink := &MemorySink{}
|
|
exec := NewModelExecutor(gw, sink, nil).WithMemory(mem)
|
|
|
|
res, err := exec.ExecuteAgent(context.Background(), testAgent(), testInput("who is free?"))
|
|
if err != nil {
|
|
t.Fatalf("a memory failure took the run with it: %v", err)
|
|
}
|
|
if res.Termination != TerminationCompleted {
|
|
t.Errorf("Termination = %q, want Completed", res.Termination)
|
|
}
|
|
|
|
// ...and it is recorded, so a thin answer is explainable afterwards.
|
|
var noted bool
|
|
for _, e := range sink.Last().Entries {
|
|
if strings.Contains(e.Name, "memory") || strings.Contains(e.Text, "memory store") {
|
|
noted = true
|
|
}
|
|
}
|
|
if !noted {
|
|
t.Error("the memory failure left no trace in the trajectory")
|
|
}
|
|
}
|
|
|
|
// Greetings skip memory for the same reason they skip retrieval: nobody needs
|
|
// remembering to say good morning, and paying for it is how "hi" came to cost
|
|
// six thousand tokens.
|
|
func TestSmalltalkCarriesNoMemory(t *testing.T) {
|
|
gw := &fakeGateway{text: "Good morning."}
|
|
mem := &fakeMemory{records: []memory.Record{
|
|
{SubjectType: memory.SubjectWorkspace, Author: memory.AuthorModel, Text: "A remembered thing."},
|
|
}}
|
|
exec := NewModelExecutor(gw, &MemorySink{}, nil).WithMemory(mem)
|
|
exec.ExecuteAgent(context.Background(), testAgent(), testInput("good morning"))
|
|
|
|
if strings.Contains(gw.lastReq.Messages[0].Text, "<memory>") {
|
|
t.Errorf("a greeting carried memory:\n%s", gw.lastReq.Messages[0].Text)
|
|
}
|
|
}
|
|
|
|
// With no store the loop is exactly what it was.
|
|
func TestWithoutAMemoryStoreNothingChanges(t *testing.T) {
|
|
gw := &fakeGateway{text: "answered"}
|
|
exec := NewModelExecutor(gw, &MemorySink{}, nil)
|
|
exec.ExecuteAgent(context.Background(), testAgent(), testInput("who is free?"))
|
|
|
|
if gw.lastReq.Messages[0].Text != "who is free?" {
|
|
t.Errorf("the question was altered with no memory configured:\n%q", gw.lastReq.Messages[0].Text)
|
|
}
|
|
}
|
|
|
|
/* ── Citable answers ─────────────────────────────────────────────────────── */
|
|
|
|
// Without this, a grounded answer and an invented one look identical to the
|
|
// reader: the ids reach the model and nothing reaches the panel, so the
|
|
// citations get stripped and the evidence disappears with them.
|
|
func TestSourcesComeBackWithTheAnswer(t *testing.T) {
|
|
gw := &fakeGateway{text: "The policy says shifts are offered for four hours."}
|
|
exec := NewModelExecutor(gw, &MemorySink{}, nil).WithRetriever(stubRetriever{})
|
|
|
|
agent := testAgent()
|
|
agent.KnowledgeSources = []string{"policy_docs"}
|
|
|
|
res, err := exec.ExecuteAgent(context.Background(), agent, testInput("how long is a shift offered?"))
|
|
if err != nil {
|
|
t.Fatalf("run failed: %v", err)
|
|
}
|
|
if len(res.Sources) == 0 {
|
|
t.Fatal("the answer carries no sources, so no claim in it can be checked")
|
|
}
|
|
s := res.Sources[0]
|
|
if s.ID == "" || s.Title == "" || s.Snippet == "" {
|
|
t.Errorf("a source is missing what a reader needs: %+v", s)
|
|
}
|
|
}
|
|
|
|
// The snippet is a recognisable opening, not the corpus delivered one answer
|
|
// at a time.
|
|
func TestASourceSnippetIsBounded(t *testing.T) {
|
|
gw := &fakeGateway{text: "answered"}
|
|
exec := NewModelExecutor(gw, &MemorySink{}, nil).WithRetriever(stubRetriever{long: true})
|
|
|
|
agent := testAgent()
|
|
agent.KnowledgeSources = []string{"policy_docs"}
|
|
res, _ := exec.ExecuteAgent(context.Background(), agent, testInput("anything?"))
|
|
|
|
if len(res.Sources) == 0 {
|
|
t.Fatal("no sources")
|
|
}
|
|
if n := len([]rune(res.Sources[0].Snippet)); n > 260 {
|
|
t.Errorf("snippet is %d runes; the response is becoming the corpus", n)
|
|
}
|
|
}
|
|
|
|
// A run that retrieved nothing says so by carrying nothing, rather than an
|
|
// empty shell the panel would draw a heading for.
|
|
func TestARunWithoutRetrievalCarriesNoSources(t *testing.T) {
|
|
gw := &fakeGateway{text: "answered from tools"}
|
|
exec := NewModelExecutor(gw, &MemorySink{}, nil)
|
|
res, _ := exec.ExecuteAgent(context.Background(), testAgent(), testInput("how many open roles?"))
|
|
if len(res.Sources) != 0 {
|
|
t.Errorf("got %d sources with no retriever", len(res.Sources))
|
|
}
|
|
}
|
|
|
|
// stubRetriever returns one chunk, so the citation path can be exercised
|
|
// without a corpus or an embedder.
|
|
type stubRetriever struct{ long bool }
|
|
|
|
func (s stubRetriever) Retrieve(_ context.Context, _ knowledge.Query) (*knowledge.Results, error) {
|
|
text := "Open shifts are offered to under-hour staff at the venue first, for four hours."
|
|
if s.long {
|
|
text = strings.Repeat("a long policy paragraph that goes on. ", 40)
|
|
}
|
|
return &knowledge.Results{Chunks: []knowledge.Result{{
|
|
ChunkID: "chunk-1", DocumentID: "doc-1",
|
|
Source: "policy_docs", Title: "Shift cover and cancellation",
|
|
Heading: "Offering an open shift", Text: text,
|
|
}}}, nil
|
|
}
|