Return the passages an answer was given, so a claim can be checked
Some checks failed
CI / test (push) Failing after 4m38s
CI / fixture (push) Failing after 9s

The ids always travelled TO the model; nothing a reader could look at ever
travelled back. So the panel stripped the citations the model wrote — there was
nowhere to put them — and a grounded answer became indistinguishable from an
invented one, which is the opposite of what citing is for.

A run now carries its sources: the id the model was told to cite, the document
title, the heading, and the opening of the passage. Both the JSON and the
streamed paths return them, because both build the same response.

Source is NOT knowledge.Result. That type carries ranks, scores and the whole
chunk, which exist to debug a retrieval rather than to be shown: an RRF score
is a rank and would be read as a percentage, and the full text would make the
response larger than the answer. This is the subset a citation needs.

Snippets are capped at 240 runes. A reader checking "the policy says X" needs
to recognise the passage, not to receive the corpus one answer at a time.

A run that retrieved nothing carries nothing rather than an empty list, so the
panel has no heading to draw for absent evidence — which is most runs, since
seven of nine agents answer from tools.

Next, and only now possible: the panel can render these and stop stripping the
citations that point at them.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
2026-10-07 20:41:10 +05:30
parent 7d04b2f0b5
commit d5cc1f1800
4 changed files with 153 additions and 8 deletions

View File

@@ -253,7 +253,7 @@ func (m *ModelExecutor) executeRun(
question := strings.TrimSpace(input.Input)
if question == "" {
return m.finish(ctx, rec, budget, TerminationToolFailure, agent, skillIDs,
"", &RuntimeError{Code: "runtime.empty_input", Message: "a run needs a question"})
"", nil, &RuntimeError{Code: "runtime.empty_input", Message: "a run needs a question"})
}
rec.Message("user", question)
@@ -317,8 +317,12 @@ func (m *ModelExecutor) executeRun(
// from what the model says next. The model's job afterwards is to report
// what happened, which is a job it cannot get wrong in a way that costs
// anybody a shift.
// Declared before the approved-write path, which can finish the run before
// retrieval ever happens. Nil then, which is correct: nothing was read.
var sources []Source
if approved, done := m.performApproved(runCtx, rec, budget, agent, input); done != nil {
return m.finish(ctx, rec, budget, *done, agent, skillIDs, "", nil)
return m.finish(ctx, rec, budget, *done, agent, skillIDs, "", sources, nil)
} else if approved != "" {
// Prepended to the question so the model answers knowing the write
// already happened. It is a tool result in everything but shape —
@@ -359,6 +363,7 @@ func (m *ModelExecutor) executeRun(
if block, retrieved := m.retrieve(runCtx, rec, agent, input, question); block != "" {
blocks = append(blocks, block)
rec.Retrieval(retrieved)
sources = sourcesFrom(retrieved)
}
}
@@ -398,10 +403,10 @@ func (m *ModelExecutor) executeRun(
// Claimed before dispatch, never after. A call that hangs until the
// context dies has still spent the step it was given.
if t := budget.ClaimStep(); t != "" {
return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, nil)
return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, sources, nil)
}
if t := budget.CheckTokens(); t != "" {
return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, nil)
return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, sources, nil)
}
rec.Budget(budget.Snapshot())
@@ -458,7 +463,7 @@ func (m *ModelExecutor) executeRun(
rec.SetModel(resp.Model)
}
if err != nil {
return m.finish(ctx, rec, budget, terminationFor(err), agent, skillIDs, lastText, err)
return m.finish(ctx, rec, budget, terminationFor(err), agent, skillIDs, lastText, sources, err)
}
if resp.Text != "" {
@@ -468,7 +473,7 @@ func (m *ModelExecutor) executeRun(
// No tool calls means the model is done talking.
if len(resp.ToolCalls) == 0 {
return m.finish(ctx, rec, budget, TerminationCompleted, agent, skillIDs, lastText, nil)
return m.finish(ctx, rec, budget, TerminationCompleted, agent, skillIDs, lastText, sources, nil)
}
// The assistant turn goes back verbatim, calls included, before any
@@ -480,7 +485,7 @@ func (m *ModelExecutor) executeRun(
results, pending, term := m.runTools(runCtx, rec, budget, agent, input, resp.ToolCalls, subs, del.depth)
if term != "" {
return m.finish(ctx, rec, budget, term, agent, skillIDs, lastText, nil)
return m.finish(ctx, rec, budget, term, agent, skillIDs, lastText, sources, nil)
}
// I4. A run that wants to write stops here and asks. It does not
@@ -489,7 +494,7 @@ func (m *ModelExecutor) executeRun(
// person deciding, and the run resumes only if they say yes.
if len(pending) > 0 {
res, err := m.finish(ctx, rec, budget,
TerminationConfirmationPending, agent, skillIDs, lastText, nil)
TerminationConfirmationPending, agent, skillIDs, lastText, sources, nil)
res.Confirmations = pending
return res, err
}
@@ -779,6 +784,7 @@ func (m *ModelExecutor) finish(
agent *Agent,
skillIDs []string,
output string,
sources []Source,
cause error,
) (*ExecutionResult, error) {
if cause != nil {
@@ -838,6 +844,7 @@ func (m *ModelExecutor) finish(
AgentVersion: agent.Version,
ResolvedSkills: skillIDs,
RunID: traj.RunID,
Sources: sources,
Termination: term,
Usage: traj.Usage,
}