Return the passages an answer was given, so a claim can be checked
The ids always travelled TO the model; nothing a reader could look at ever travelled back. So the panel stripped the citations the model wrote — there was nowhere to put them — and a grounded answer became indistinguishable from an invented one, which is the opposite of what citing is for. A run now carries its sources: the id the model was told to cite, the document title, the heading, and the opening of the passage. Both the JSON and the streamed paths return them, because both build the same response. Source is NOT knowledge.Result. That type carries ranks, scores and the whole chunk, which exist to debug a retrieval rather than to be shown: an RRF score is a rank and would be read as a percentage, and the full text would make the response larger than the answer. This is the subset a citation needs. Snippets are capped at 240 runes. A reader checking "the policy says X" needs to recognise the passage, not to receive the corpus one answer at a time. A run that retrieved nothing carries nothing rather than an empty list, so the panel has no heading to draw for absent evidence — which is most runs, since seven of nine agents answer from tools. Next, and only now possible: the panel can render these and stop stripping the citations that point at them. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -253,7 +253,7 @@ func (m *ModelExecutor) executeRun(
|
||||
question := strings.TrimSpace(input.Input)
|
||||
if question == "" {
|
||||
return m.finish(ctx, rec, budget, TerminationToolFailure, agent, skillIDs,
|
||||
"", &RuntimeError{Code: "runtime.empty_input", Message: "a run needs a question"})
|
||||
"", nil, &RuntimeError{Code: "runtime.empty_input", Message: "a run needs a question"})
|
||||
}
|
||||
rec.Message("user", question)
|
||||
|
||||
@@ -317,8 +317,12 @@ func (m *ModelExecutor) executeRun(
|
||||
// from what the model says next. The model's job afterwards is to report
|
||||
// what happened, which is a job it cannot get wrong in a way that costs
|
||||
// anybody a shift.
|
||||
// Declared before the approved-write path, which can finish the run before
|
||||
// retrieval ever happens. Nil then, which is correct: nothing was read.
|
||||
var sources []Source
|
||||
|
||||
if approved, done := m.performApproved(runCtx, rec, budget, agent, input); done != nil {
|
||||
return m.finish(ctx, rec, budget, *done, agent, skillIDs, "", nil)
|
||||
return m.finish(ctx, rec, budget, *done, agent, skillIDs, "", sources, nil)
|
||||
} else if approved != "" {
|
||||
// Prepended to the question so the model answers knowing the write
|
||||
// already happened. It is a tool result in everything but shape —
|
||||
@@ -359,6 +363,7 @@ func (m *ModelExecutor) executeRun(
|
||||
if block, retrieved := m.retrieve(runCtx, rec, agent, input, question); block != "" {
|
||||
blocks = append(blocks, block)
|
||||
rec.Retrieval(retrieved)
|
||||
sources = sourcesFrom(retrieved)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -398,10 +403,10 @@ func (m *ModelExecutor) executeRun(
|
||||
// Claimed before dispatch, never after. A call that hangs until the
|
||||
// context dies has still spent the step it was given.
|
||||
if t := budget.ClaimStep(); t != "" {
|
||||
return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, nil)
|
||||
return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, sources, nil)
|
||||
}
|
||||
if t := budget.CheckTokens(); t != "" {
|
||||
return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, nil)
|
||||
return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, sources, nil)
|
||||
}
|
||||
rec.Budget(budget.Snapshot())
|
||||
|
||||
@@ -458,7 +463,7 @@ func (m *ModelExecutor) executeRun(
|
||||
rec.SetModel(resp.Model)
|
||||
}
|
||||
if err != nil {
|
||||
return m.finish(ctx, rec, budget, terminationFor(err), agent, skillIDs, lastText, err)
|
||||
return m.finish(ctx, rec, budget, terminationFor(err), agent, skillIDs, lastText, sources, err)
|
||||
}
|
||||
|
||||
if resp.Text != "" {
|
||||
@@ -468,7 +473,7 @@ func (m *ModelExecutor) executeRun(
|
||||
|
||||
// No tool calls means the model is done talking.
|
||||
if len(resp.ToolCalls) == 0 {
|
||||
return m.finish(ctx, rec, budget, TerminationCompleted, agent, skillIDs, lastText, nil)
|
||||
return m.finish(ctx, rec, budget, TerminationCompleted, agent, skillIDs, lastText, sources, nil)
|
||||
}
|
||||
|
||||
// The assistant turn goes back verbatim, calls included, before any
|
||||
@@ -480,7 +485,7 @@ func (m *ModelExecutor) executeRun(
|
||||
|
||||
results, pending, term := m.runTools(runCtx, rec, budget, agent, input, resp.ToolCalls, subs, del.depth)
|
||||
if term != "" {
|
||||
return m.finish(ctx, rec, budget, term, agent, skillIDs, lastText, nil)
|
||||
return m.finish(ctx, rec, budget, term, agent, skillIDs, lastText, sources, nil)
|
||||
}
|
||||
|
||||
// I4. A run that wants to write stops here and asks. It does not
|
||||
@@ -489,7 +494,7 @@ func (m *ModelExecutor) executeRun(
|
||||
// person deciding, and the run resumes only if they say yes.
|
||||
if len(pending) > 0 {
|
||||
res, err := m.finish(ctx, rec, budget,
|
||||
TerminationConfirmationPending, agent, skillIDs, lastText, nil)
|
||||
TerminationConfirmationPending, agent, skillIDs, lastText, sources, nil)
|
||||
res.Confirmations = pending
|
||||
return res, err
|
||||
}
|
||||
@@ -779,6 +784,7 @@ func (m *ModelExecutor) finish(
|
||||
agent *Agent,
|
||||
skillIDs []string,
|
||||
output string,
|
||||
sources []Source,
|
||||
cause error,
|
||||
) (*ExecutionResult, error) {
|
||||
if cause != nil {
|
||||
@@ -838,6 +844,7 @@ func (m *ModelExecutor) finish(
|
||||
AgentVersion: agent.Version,
|
||||
ResolvedSkills: skillIDs,
|
||||
RunID: traj.RunID,
|
||||
Sources: sources,
|
||||
Termination: term,
|
||||
Usage: traj.Usage,
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user