diff --git a/go-api/internal/httpserver/runs.go b/go-api/internal/httpserver/runs.go index 74a91fb..5012678 100644 --- a/go-api/internal/httpserver/runs.go +++ b/go-api/internal/httpserver/runs.go @@ -113,6 +113,11 @@ type runResponse struct { // exactly when termination is ConfirmationPending. Confirmations []*tools.Confirmation `json:"confirmations,omitempty"` + // Sources are the passages the answer was given, so a claim can be + // checked. Absent when nothing was retrieved — which is most runs, since + // seven of nine agents answer from tools rather than from a corpus. + Sources []runtime.Source `json:"sources,omitempty"` + Usage runUsage `json:"usage"` } @@ -248,6 +253,7 @@ func buildRunResponse(res *runtime.ExecutionResult) runResponse { Termination: string(res.Termination), Output: res.Output, Confirmations: res.Confirmations, + Sources: res.Sources, Usage: runUsage{ InputTokens: res.Usage.InputTokens, OutputTokens: res.Usage.OutputTokens, diff --git a/go-api/internal/runtime/loop.go b/go-api/internal/runtime/loop.go index e8f553b..94f51d6 100644 --- a/go-api/internal/runtime/loop.go +++ b/go-api/internal/runtime/loop.go @@ -253,7 +253,7 @@ func (m *ModelExecutor) executeRun( question := strings.TrimSpace(input.Input) if question == "" { return m.finish(ctx, rec, budget, TerminationToolFailure, agent, skillIDs, - "", &RuntimeError{Code: "runtime.empty_input", Message: "a run needs a question"}) + "", nil, &RuntimeError{Code: "runtime.empty_input", Message: "a run needs a question"}) } rec.Message("user", question) @@ -317,8 +317,12 @@ func (m *ModelExecutor) executeRun( // from what the model says next. The model's job afterwards is to report // what happened, which is a job it cannot get wrong in a way that costs // anybody a shift. + // Declared before the approved-write path, which can finish the run before + // retrieval ever happens. Nil then, which is correct: nothing was read. + var sources []Source + if approved, done := m.performApproved(runCtx, rec, budget, agent, input); done != nil { - return m.finish(ctx, rec, budget, *done, agent, skillIDs, "", nil) + return m.finish(ctx, rec, budget, *done, agent, skillIDs, "", sources, nil) } else if approved != "" { // Prepended to the question so the model answers knowing the write // already happened. It is a tool result in everything but shape — @@ -359,6 +363,7 @@ func (m *ModelExecutor) executeRun( if block, retrieved := m.retrieve(runCtx, rec, agent, input, question); block != "" { blocks = append(blocks, block) rec.Retrieval(retrieved) + sources = sourcesFrom(retrieved) } } @@ -398,10 +403,10 @@ func (m *ModelExecutor) executeRun( // Claimed before dispatch, never after. A call that hangs until the // context dies has still spent the step it was given. if t := budget.ClaimStep(); t != "" { - return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, nil) + return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, sources, nil) } if t := budget.CheckTokens(); t != "" { - return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, nil) + return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, sources, nil) } rec.Budget(budget.Snapshot()) @@ -458,7 +463,7 @@ func (m *ModelExecutor) executeRun( rec.SetModel(resp.Model) } if err != nil { - return m.finish(ctx, rec, budget, terminationFor(err), agent, skillIDs, lastText, err) + return m.finish(ctx, rec, budget, terminationFor(err), agent, skillIDs, lastText, sources, err) } if resp.Text != "" { @@ -468,7 +473,7 @@ func (m *ModelExecutor) executeRun( // No tool calls means the model is done talking. if len(resp.ToolCalls) == 0 { - return m.finish(ctx, rec, budget, TerminationCompleted, agent, skillIDs, lastText, nil) + return m.finish(ctx, rec, budget, TerminationCompleted, agent, skillIDs, lastText, sources, nil) } // The assistant turn goes back verbatim, calls included, before any @@ -480,7 +485,7 @@ func (m *ModelExecutor) executeRun( results, pending, term := m.runTools(runCtx, rec, budget, agent, input, resp.ToolCalls, subs, del.depth) if term != "" { - return m.finish(ctx, rec, budget, term, agent, skillIDs, lastText, nil) + return m.finish(ctx, rec, budget, term, agent, skillIDs, lastText, sources, nil) } // I4. A run that wants to write stops here and asks. It does not @@ -489,7 +494,7 @@ func (m *ModelExecutor) executeRun( // person deciding, and the run resumes only if they say yes. if len(pending) > 0 { res, err := m.finish(ctx, rec, budget, - TerminationConfirmationPending, agent, skillIDs, lastText, nil) + TerminationConfirmationPending, agent, skillIDs, lastText, sources, nil) res.Confirmations = pending return res, err } @@ -779,6 +784,7 @@ func (m *ModelExecutor) finish( agent *Agent, skillIDs []string, output string, + sources []Source, cause error, ) (*ExecutionResult, error) { if cause != nil { @@ -838,6 +844,7 @@ func (m *ModelExecutor) finish( AgentVersion: agent.Version, ResolvedSkills: skillIDs, RunID: traj.RunID, + Sources: sources, Termination: term, Usage: traj.Usage, } diff --git a/go-api/internal/runtime/recall_test.go b/go-api/internal/runtime/recall_test.go index 8f2a655..b0c1595 100644 --- a/go-api/internal/runtime/recall_test.go +++ b/go-api/internal/runtime/recall_test.go @@ -6,6 +6,8 @@ import ( "strings" "testing" + "github.com/krow/krow-backend/go-api/internal/knowledge" + "github.com/krow/krow-backend/go-api/internal/authctx" "github.com/krow/krow-backend/go-api/internal/memory" ) @@ -111,3 +113,73 @@ func TestWithoutAMemoryStoreNothingChanges(t *testing.T) { t.Errorf("the question was altered with no memory configured:\n%q", gw.lastReq.Messages[0].Text) } } + +/* ── Citable answers ─────────────────────────────────────────────────────── */ + +// Without this, a grounded answer and an invented one look identical to the +// reader: the ids reach the model and nothing reaches the panel, so the +// citations get stripped and the evidence disappears with them. +func TestSourcesComeBackWithTheAnswer(t *testing.T) { + gw := &fakeGateway{text: "The policy says shifts are offered for four hours."} + exec := NewModelExecutor(gw, &MemorySink{}, nil).WithRetriever(stubRetriever{}) + + agent := testAgent() + agent.KnowledgeSources = []string{"policy_docs"} + + res, err := exec.ExecuteAgent(context.Background(), agent, testInput("how long is a shift offered?")) + if err != nil { + t.Fatalf("run failed: %v", err) + } + if len(res.Sources) == 0 { + t.Fatal("the answer carries no sources, so no claim in it can be checked") + } + s := res.Sources[0] + if s.ID == "" || s.Title == "" || s.Snippet == "" { + t.Errorf("a source is missing what a reader needs: %+v", s) + } +} + +// The snippet is a recognisable opening, not the corpus delivered one answer +// at a time. +func TestASourceSnippetIsBounded(t *testing.T) { + gw := &fakeGateway{text: "answered"} + exec := NewModelExecutor(gw, &MemorySink{}, nil).WithRetriever(stubRetriever{long: true}) + + agent := testAgent() + agent.KnowledgeSources = []string{"policy_docs"} + res, _ := exec.ExecuteAgent(context.Background(), agent, testInput("anything?")) + + if len(res.Sources) == 0 { + t.Fatal("no sources") + } + if n := len([]rune(res.Sources[0].Snippet)); n > 260 { + t.Errorf("snippet is %d runes; the response is becoming the corpus", n) + } +} + +// A run that retrieved nothing says so by carrying nothing, rather than an +// empty shell the panel would draw a heading for. +func TestARunWithoutRetrievalCarriesNoSources(t *testing.T) { + gw := &fakeGateway{text: "answered from tools"} + exec := NewModelExecutor(gw, &MemorySink{}, nil) + res, _ := exec.ExecuteAgent(context.Background(), testAgent(), testInput("how many open roles?")) + if len(res.Sources) != 0 { + t.Errorf("got %d sources with no retriever", len(res.Sources)) + } +} + +// stubRetriever returns one chunk, so the citation path can be exercised +// without a corpus or an embedder. +type stubRetriever struct{ long bool } + +func (s stubRetriever) Retrieve(_ context.Context, _ knowledge.Query) (*knowledge.Results, error) { + text := "Open shifts are offered to under-hour staff at the venue first, for four hours." + if s.long { + text = strings.Repeat("a long policy paragraph that goes on. ", 40) + } + return &knowledge.Results{Chunks: []knowledge.Result{{ + ChunkID: "chunk-1", DocumentID: "doc-1", + Source: "policy_docs", Title: "Shift cover and cancellation", + Heading: "Offering an open shift", Text: text, + }}}, nil +} diff --git a/go-api/internal/runtime/types.go b/go-api/internal/runtime/types.go index 7e90d66..4cac53f 100644 --- a/go-api/internal/runtime/types.go +++ b/go-api/internal/runtime/types.go @@ -3,8 +3,10 @@ package runtime import ( "errors" "fmt" + "strings" "github.com/krow/krow-backend/go-api/internal/authctx" + "github.com/krow/krow-backend/go-api/internal/knowledge" "github.com/krow/krow-backend/go-api/internal/tools" ) @@ -149,6 +151,26 @@ type ExecutionInput struct { Confirmation string `json:"confirmation,omitempty"` } +// Source is one retrieved passage, as a reader needs to see it. +// +// Deliberately not knowledge.Result. That type carries ranks, scores and the +// full chunk text, which exist to debug a retrieval and not to be shown: an +// RRF score is a rank and would be read as a percentage, and the whole chunk +// is more than the answer used. This is the subset a citation needs and +// nothing else. +type Source struct { + // ID is what the model was told to cite, so a [id] in the answer can be + // matched to the passage it came from. + ID string `json:"id"` + + Title string `json:"title"` + Heading string `json:"heading,omitempty"` + + // Snippet is the opening of the passage: enough to recognise it, short + // enough that the response does not become the corpus. + Snippet string `json:"snippet"` +} + // ExecutionResult captures the outcome of an execution attempt. type ExecutionResult struct { Success bool `json:"success"` @@ -162,6 +184,16 @@ type ExecutionResult struct { // a conversation about a bad answer has something to point at. RunID string `json:"runId,omitempty"` + // Sources are the passages this run was given, in the order it was given + // them. + // + // RETURNED SO A CLAIM CAN BE CHECKED. The ids already travelled to the + // model; what never travelled back was anything a reader could look at, so + // the panel stripped the citations the model wrote because there was + // nowhere to put them. That made every grounded answer indistinguishable + // from an ungrounded one — which is the opposite of what citing is for. + Sources []Source `json:"sources,omitempty"` + // Termination is why the run ended — exactly one of the six, always set by // the loop. Empty only on results built by Engine's pre-execution failure // paths, where no run was ever started. @@ -209,3 +241,31 @@ func (e *RuntimeError) Error() string { func (e *RuntimeError) Unwrap() error { return e.Cause } + +// sourcesFrom reduces a retrieval to what a reader needs to check a claim. +// +// The whole chunk is not returned. A reader checking "the policy says X" needs +// to recognise the passage, not to receive the corpus one answer at a time — +// and a response that carried every retrieved chunk in full would be larger +// than the answer, on a surface where size is latency. +func sourcesFrom(res *knowledge.Results) []Source { + if res == nil || len(res.Chunks) == 0 { + return nil + } + const snippetRunes = 240 + + out := make([]Source, 0, len(res.Chunks)) + for _, c := range res.Chunks { + text := strings.Join(strings.Fields(c.Text), " ") + if len([]rune(text)) > snippetRunes { + text = string([]rune(text)[:snippetRunes]) + "…" + } + out = append(out, Source{ + ID: c.ChunkID, + Title: c.Title, + Heading: c.Heading, + Snippet: text, + }) + } + return out +}