Return the passages an answer was given, so a claim can be checked
Some checks failed
CI / test (push) Failing after 4m38s
CI / fixture (push) Failing after 9s

The ids always travelled TO the model; nothing a reader could look at ever
travelled back. So the panel stripped the citations the model wrote — there was
nowhere to put them — and a grounded answer became indistinguishable from an
invented one, which is the opposite of what citing is for.

A run now carries its sources: the id the model was told to cite, the document
title, the heading, and the opening of the passage. Both the JSON and the
streamed paths return them, because both build the same response.

Source is NOT knowledge.Result. That type carries ranks, scores and the whole
chunk, which exist to debug a retrieval rather than to be shown: an RRF score
is a rank and would be read as a percentage, and the full text would make the
response larger than the answer. This is the subset a citation needs.

Snippets are capped at 240 runes. A reader checking "the policy says X" needs
to recognise the passage, not to receive the corpus one answer at a time.

A run that retrieved nothing carries nothing rather than an empty list, so the
panel has no heading to draw for absent evidence — which is most runs, since
seven of nine agents answer from tools.

Next, and only now possible: the panel can render these and stop stripping the
citations that point at them.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
2026-10-07 20:41:10 +05:30
parent 7d04b2f0b5
commit d5cc1f1800
4 changed files with 153 additions and 8 deletions

View File

@@ -113,6 +113,11 @@ type runResponse struct {
// exactly when termination is ConfirmationPending. // exactly when termination is ConfirmationPending.
Confirmations []*tools.Confirmation `json:"confirmations,omitempty"` Confirmations []*tools.Confirmation `json:"confirmations,omitempty"`
// Sources are the passages the answer was given, so a claim can be
// checked. Absent when nothing was retrieved — which is most runs, since
// seven of nine agents answer from tools rather than from a corpus.
Sources []runtime.Source `json:"sources,omitempty"`
Usage runUsage `json:"usage"` Usage runUsage `json:"usage"`
} }
@@ -248,6 +253,7 @@ func buildRunResponse(res *runtime.ExecutionResult) runResponse {
Termination: string(res.Termination), Termination: string(res.Termination),
Output: res.Output, Output: res.Output,
Confirmations: res.Confirmations, Confirmations: res.Confirmations,
Sources: res.Sources,
Usage: runUsage{ Usage: runUsage{
InputTokens: res.Usage.InputTokens, InputTokens: res.Usage.InputTokens,
OutputTokens: res.Usage.OutputTokens, OutputTokens: res.Usage.OutputTokens,

View File

@@ -253,7 +253,7 @@ func (m *ModelExecutor) executeRun(
question := strings.TrimSpace(input.Input) question := strings.TrimSpace(input.Input)
if question == "" { if question == "" {
return m.finish(ctx, rec, budget, TerminationToolFailure, agent, skillIDs, return m.finish(ctx, rec, budget, TerminationToolFailure, agent, skillIDs,
"", &RuntimeError{Code: "runtime.empty_input", Message: "a run needs a question"}) "", nil, &RuntimeError{Code: "runtime.empty_input", Message: "a run needs a question"})
} }
rec.Message("user", question) rec.Message("user", question)
@@ -317,8 +317,12 @@ func (m *ModelExecutor) executeRun(
// from what the model says next. The model's job afterwards is to report // from what the model says next. The model's job afterwards is to report
// what happened, which is a job it cannot get wrong in a way that costs // what happened, which is a job it cannot get wrong in a way that costs
// anybody a shift. // anybody a shift.
// Declared before the approved-write path, which can finish the run before
// retrieval ever happens. Nil then, which is correct: nothing was read.
var sources []Source
if approved, done := m.performApproved(runCtx, rec, budget, agent, input); done != nil { if approved, done := m.performApproved(runCtx, rec, budget, agent, input); done != nil {
return m.finish(ctx, rec, budget, *done, agent, skillIDs, "", nil) return m.finish(ctx, rec, budget, *done, agent, skillIDs, "", sources, nil)
} else if approved != "" { } else if approved != "" {
// Prepended to the question so the model answers knowing the write // Prepended to the question so the model answers knowing the write
// already happened. It is a tool result in everything but shape — // already happened. It is a tool result in everything but shape —
@@ -359,6 +363,7 @@ func (m *ModelExecutor) executeRun(
if block, retrieved := m.retrieve(runCtx, rec, agent, input, question); block != "" { if block, retrieved := m.retrieve(runCtx, rec, agent, input, question); block != "" {
blocks = append(blocks, block) blocks = append(blocks, block)
rec.Retrieval(retrieved) rec.Retrieval(retrieved)
sources = sourcesFrom(retrieved)
} }
} }
@@ -398,10 +403,10 @@ func (m *ModelExecutor) executeRun(
// Claimed before dispatch, never after. A call that hangs until the // Claimed before dispatch, never after. A call that hangs until the
// context dies has still spent the step it was given. // context dies has still spent the step it was given.
if t := budget.ClaimStep(); t != "" { if t := budget.ClaimStep(); t != "" {
return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, nil) return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, sources, nil)
} }
if t := budget.CheckTokens(); t != "" { if t := budget.CheckTokens(); t != "" {
return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, nil) return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, sources, nil)
} }
rec.Budget(budget.Snapshot()) rec.Budget(budget.Snapshot())
@@ -458,7 +463,7 @@ func (m *ModelExecutor) executeRun(
rec.SetModel(resp.Model) rec.SetModel(resp.Model)
} }
if err != nil { if err != nil {
return m.finish(ctx, rec, budget, terminationFor(err), agent, skillIDs, lastText, err) return m.finish(ctx, rec, budget, terminationFor(err), agent, skillIDs, lastText, sources, err)
} }
if resp.Text != "" { if resp.Text != "" {
@@ -468,7 +473,7 @@ func (m *ModelExecutor) executeRun(
// No tool calls means the model is done talking. // No tool calls means the model is done talking.
if len(resp.ToolCalls) == 0 { if len(resp.ToolCalls) == 0 {
return m.finish(ctx, rec, budget, TerminationCompleted, agent, skillIDs, lastText, nil) return m.finish(ctx, rec, budget, TerminationCompleted, agent, skillIDs, lastText, sources, nil)
} }
// The assistant turn goes back verbatim, calls included, before any // The assistant turn goes back verbatim, calls included, before any
@@ -480,7 +485,7 @@ func (m *ModelExecutor) executeRun(
results, pending, term := m.runTools(runCtx, rec, budget, agent, input, resp.ToolCalls, subs, del.depth) results, pending, term := m.runTools(runCtx, rec, budget, agent, input, resp.ToolCalls, subs, del.depth)
if term != "" { if term != "" {
return m.finish(ctx, rec, budget, term, agent, skillIDs, lastText, nil) return m.finish(ctx, rec, budget, term, agent, skillIDs, lastText, sources, nil)
} }
// I4. A run that wants to write stops here and asks. It does not // I4. A run that wants to write stops here and asks. It does not
@@ -489,7 +494,7 @@ func (m *ModelExecutor) executeRun(
// person deciding, and the run resumes only if they say yes. // person deciding, and the run resumes only if they say yes.
if len(pending) > 0 { if len(pending) > 0 {
res, err := m.finish(ctx, rec, budget, res, err := m.finish(ctx, rec, budget,
TerminationConfirmationPending, agent, skillIDs, lastText, nil) TerminationConfirmationPending, agent, skillIDs, lastText, sources, nil)
res.Confirmations = pending res.Confirmations = pending
return res, err return res, err
} }
@@ -779,6 +784,7 @@ func (m *ModelExecutor) finish(
agent *Agent, agent *Agent,
skillIDs []string, skillIDs []string,
output string, output string,
sources []Source,
cause error, cause error,
) (*ExecutionResult, error) { ) (*ExecutionResult, error) {
if cause != nil { if cause != nil {
@@ -838,6 +844,7 @@ func (m *ModelExecutor) finish(
AgentVersion: agent.Version, AgentVersion: agent.Version,
ResolvedSkills: skillIDs, ResolvedSkills: skillIDs,
RunID: traj.RunID, RunID: traj.RunID,
Sources: sources,
Termination: term, Termination: term,
Usage: traj.Usage, Usage: traj.Usage,
} }

View File

@@ -6,6 +6,8 @@ import (
"strings" "strings"
"testing" "testing"
"github.com/krow/krow-backend/go-api/internal/knowledge"
"github.com/krow/krow-backend/go-api/internal/authctx" "github.com/krow/krow-backend/go-api/internal/authctx"
"github.com/krow/krow-backend/go-api/internal/memory" "github.com/krow/krow-backend/go-api/internal/memory"
) )
@@ -111,3 +113,73 @@ func TestWithoutAMemoryStoreNothingChanges(t *testing.T) {
t.Errorf("the question was altered with no memory configured:\n%q", gw.lastReq.Messages[0].Text) t.Errorf("the question was altered with no memory configured:\n%q", gw.lastReq.Messages[0].Text)
} }
} }
/* ── Citable answers ─────────────────────────────────────────────────────── */
// Without this, a grounded answer and an invented one look identical to the
// reader: the ids reach the model and nothing reaches the panel, so the
// citations get stripped and the evidence disappears with them.
func TestSourcesComeBackWithTheAnswer(t *testing.T) {
gw := &fakeGateway{text: "The policy says shifts are offered for four hours."}
exec := NewModelExecutor(gw, &MemorySink{}, nil).WithRetriever(stubRetriever{})
agent := testAgent()
agent.KnowledgeSources = []string{"policy_docs"}
res, err := exec.ExecuteAgent(context.Background(), agent, testInput("how long is a shift offered?"))
if err != nil {
t.Fatalf("run failed: %v", err)
}
if len(res.Sources) == 0 {
t.Fatal("the answer carries no sources, so no claim in it can be checked")
}
s := res.Sources[0]
if s.ID == "" || s.Title == "" || s.Snippet == "" {
t.Errorf("a source is missing what a reader needs: %+v", s)
}
}
// The snippet is a recognisable opening, not the corpus delivered one answer
// at a time.
func TestASourceSnippetIsBounded(t *testing.T) {
gw := &fakeGateway{text: "answered"}
exec := NewModelExecutor(gw, &MemorySink{}, nil).WithRetriever(stubRetriever{long: true})
agent := testAgent()
agent.KnowledgeSources = []string{"policy_docs"}
res, _ := exec.ExecuteAgent(context.Background(), agent, testInput("anything?"))
if len(res.Sources) == 0 {
t.Fatal("no sources")
}
if n := len([]rune(res.Sources[0].Snippet)); n > 260 {
t.Errorf("snippet is %d runes; the response is becoming the corpus", n)
}
}
// A run that retrieved nothing says so by carrying nothing, rather than an
// empty shell the panel would draw a heading for.
func TestARunWithoutRetrievalCarriesNoSources(t *testing.T) {
gw := &fakeGateway{text: "answered from tools"}
exec := NewModelExecutor(gw, &MemorySink{}, nil)
res, _ := exec.ExecuteAgent(context.Background(), testAgent(), testInput("how many open roles?"))
if len(res.Sources) != 0 {
t.Errorf("got %d sources with no retriever", len(res.Sources))
}
}
// stubRetriever returns one chunk, so the citation path can be exercised
// without a corpus or an embedder.
type stubRetriever struct{ long bool }
func (s stubRetriever) Retrieve(_ context.Context, _ knowledge.Query) (*knowledge.Results, error) {
text := "Open shifts are offered to under-hour staff at the venue first, for four hours."
if s.long {
text = strings.Repeat("a long policy paragraph that goes on. ", 40)
}
return &knowledge.Results{Chunks: []knowledge.Result{{
ChunkID: "chunk-1", DocumentID: "doc-1",
Source: "policy_docs", Title: "Shift cover and cancellation",
Heading: "Offering an open shift", Text: text,
}}}, nil
}

View File

@@ -3,8 +3,10 @@ package runtime
import ( import (
"errors" "errors"
"fmt" "fmt"
"strings"
"github.com/krow/krow-backend/go-api/internal/authctx" "github.com/krow/krow-backend/go-api/internal/authctx"
"github.com/krow/krow-backend/go-api/internal/knowledge"
"github.com/krow/krow-backend/go-api/internal/tools" "github.com/krow/krow-backend/go-api/internal/tools"
) )
@@ -149,6 +151,26 @@ type ExecutionInput struct {
Confirmation string `json:"confirmation,omitempty"` Confirmation string `json:"confirmation,omitempty"`
} }
// Source is one retrieved passage, as a reader needs to see it.
//
// Deliberately not knowledge.Result. That type carries ranks, scores and the
// full chunk text, which exist to debug a retrieval and not to be shown: an
// RRF score is a rank and would be read as a percentage, and the whole chunk
// is more than the answer used. This is the subset a citation needs and
// nothing else.
type Source struct {
// ID is what the model was told to cite, so a [id] in the answer can be
// matched to the passage it came from.
ID string `json:"id"`
Title string `json:"title"`
Heading string `json:"heading,omitempty"`
// Snippet is the opening of the passage: enough to recognise it, short
// enough that the response does not become the corpus.
Snippet string `json:"snippet"`
}
// ExecutionResult captures the outcome of an execution attempt. // ExecutionResult captures the outcome of an execution attempt.
type ExecutionResult struct { type ExecutionResult struct {
Success bool `json:"success"` Success bool `json:"success"`
@@ -162,6 +184,16 @@ type ExecutionResult struct {
// a conversation about a bad answer has something to point at. // a conversation about a bad answer has something to point at.
RunID string `json:"runId,omitempty"` RunID string `json:"runId,omitempty"`
// Sources are the passages this run was given, in the order it was given
// them.
//
// RETURNED SO A CLAIM CAN BE CHECKED. The ids already travelled to the
// model; what never travelled back was anything a reader could look at, so
// the panel stripped the citations the model wrote because there was
// nowhere to put them. That made every grounded answer indistinguishable
// from an ungrounded one — which is the opposite of what citing is for.
Sources []Source `json:"sources,omitempty"`
// Termination is why the run ended — exactly one of the six, always set by // Termination is why the run ended — exactly one of the six, always set by
// the loop. Empty only on results built by Engine's pre-execution failure // the loop. Empty only on results built by Engine's pre-execution failure
// paths, where no run was ever started. // paths, where no run was ever started.
@@ -209,3 +241,31 @@ func (e *RuntimeError) Error() string {
func (e *RuntimeError) Unwrap() error { func (e *RuntimeError) Unwrap() error {
return e.Cause return e.Cause
} }
// sourcesFrom reduces a retrieval to what a reader needs to check a claim.
//
// The whole chunk is not returned. A reader checking "the policy says X" needs
// to recognise the passage, not to receive the corpus one answer at a time —
// and a response that carried every retrieved chunk in full would be larger
// than the answer, on a surface where size is latency.
func sourcesFrom(res *knowledge.Results) []Source {
if res == nil || len(res.Chunks) == 0 {
return nil
}
const snippetRunes = 240
out := make([]Source, 0, len(res.Chunks))
for _, c := range res.Chunks {
text := strings.Join(strings.Fields(c.Text), " ")
if len([]rune(text)) > snippetRunes {
text = string([]rune(text)[:snippetRunes]) + "…"
}
out = append(out, Source{
ID: c.ChunkID,
Title: c.Title,
Heading: c.Heading,
Snippet: text,
})
}
return out
}