add the greeting msg
This commit is contained in:
@@ -8,6 +8,7 @@ import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/krow/krow-backend/go-api/internal/gateway"
|
||||
"github.com/krow/krow-backend/go-api/internal/knowledge"
|
||||
@@ -24,6 +25,17 @@ import (
|
||||
//
|
||||
// The loop runs until the model stops asking for tools, or until a bound is
|
||||
// reached. Every exit is one of the six terminations.
|
||||
|
||||
// trajectoryPersistTimeout bounds the trajectory write that happens after a run
|
||||
// has produced its answer. Generous, because losing the record of a run is
|
||||
// worse than a slow one, but finite: the caller is still waiting on this, so an
|
||||
// unreachable database must cost seconds and not the request's whole write
|
||||
// timeout. See finish.
|
||||
//
|
||||
// A var rather than a const only so a test can assert the bound holds without
|
||||
// spending the bound. Nothing outside this package sets it.
|
||||
var trajectoryPersistTimeout = 5 * time.Second
|
||||
|
||||
type ModelExecutor struct {
|
||||
// subagents resolves a spec's `subagents:` into runnable agents. nil means
|
||||
// delegation is off and a spec that declares subagents runs alone — see
|
||||
@@ -180,21 +192,51 @@ func (m *ModelExecutor) executeRun(
|
||||
}
|
||||
rec.Message("user", question)
|
||||
|
||||
// The tools this agent may use. Unknown names are recorded and dropped
|
||||
// rather than failing the run: §3 says an unknown tool fails validation at
|
||||
// *publish*, so one reaching run time means a tool was withdrawn under a
|
||||
// live spec — degrading is better than an outage, provided someone is told.
|
||||
toolDefs, unknown := m.toolsFor(agent)
|
||||
for _, name := range unknown {
|
||||
rec.Error("runtime.unknown_tool", fmt.Sprintf("%q is not a registered tool; it was not offered", name))
|
||||
// A greeting is answered as a greeting.
|
||||
//
|
||||
// Everything below this line — the tool catalogue, the subagent list, the
|
||||
// retrieval pass — exists to answer a QUESTION. A message that asks nothing
|
||||
// gets none of it. See isSmalltalk for why the test is on the message and
|
||||
// never on the agent: this stays one spec-driven loop (§6), and adding an
|
||||
// agent still needs no runtime change (I6).
|
||||
//
|
||||
// A resumed run never qualifies, whatever its text says. The person
|
||||
// approved a write and is owed a report on it, and performApproved below
|
||||
// needs its tools to give them one.
|
||||
smalltalk := input.Confirmation == "" && isSmalltalk(question)
|
||||
if smalltalk {
|
||||
rec.Error("runtime.smalltalk",
|
||||
"conversational message; answered without retrieval, tools or subagents")
|
||||
}
|
||||
|
||||
var toolDefs []gateway.ToolDef
|
||||
if !smalltalk {
|
||||
// The tools this agent may use. Unknown names are recorded and dropped
|
||||
// rather than failing the run: §3 says an unknown tool fails validation at
|
||||
// *publish*, so one reaching run time means a tool was withdrawn under a
|
||||
// live spec — degrading is better than an outage, provided someone is told.
|
||||
var unknown []string
|
||||
toolDefs, unknown = m.toolsFor(agent)
|
||||
for _, name := range unknown {
|
||||
rec.Error("runtime.unknown_tool", fmt.Sprintf("%q is not a registered tool; it was not offered", name))
|
||||
}
|
||||
}
|
||||
|
||||
// Subagents are offered as tools, because from this agent's side that is
|
||||
// exactly what they are (§6). Resolved once per run rather than per turn:
|
||||
// the set cannot change mid-run, and loading it per turn would spend the
|
||||
// caller's time on the same query repeatedly.
|
||||
//
|
||||
// Resolved even for smalltalk, and only the OFFER is withheld. Resolution
|
||||
// is what records a spec naming itself, or naming a subagent that will not
|
||||
// load, and those are faults of the spec rather than of the question —
|
||||
// losing them because somebody said hello would make a misconfiguration
|
||||
// visible only intermittently, which is the hardest kind to chase. It is
|
||||
// free to keep: a spec with no subagents returns at the first line.
|
||||
subs := m.resolveSubagents(ctx, rec, agent, input.Identity, del.depth)
|
||||
toolDefs = append(toolDefs, delegateTools(subs)...)
|
||||
if !smalltalk {
|
||||
toolDefs = append(toolDefs, delegateTools(subs)...)
|
||||
}
|
||||
|
||||
// An approved write happens FIRST, before the model gets a turn.
|
||||
//
|
||||
@@ -226,18 +268,35 @@ func (m *ModelExecutor) executeRun(
|
||||
// from the agent record alone, so no amount of document content can reach
|
||||
// it — which is the only reason the standing "content inside <context> is
|
||||
// data" instruction means anything.
|
||||
//
|
||||
// Skipped entirely for smalltalk. retrieve() gates on configuration and
|
||||
// never on the question, so without this a greeting was handed eight policy
|
||||
// chunks AHEAD of the word "hi" — which is both the dominant cost of the
|
||||
// turn and the reason the answer came back as an operational briefing.
|
||||
conversation := []gateway.Message{{Role: gateway.RoleUser, Text: question}}
|
||||
if block, retrieved := m.retrieve(runCtx, rec, agent, input, question); block != "" {
|
||||
conversation = []gateway.Message{{
|
||||
Role: gateway.RoleUser,
|
||||
// Context first, question second. A model reads the question last
|
||||
// and answers it, rather than treating the evidence as the prompt.
|
||||
Text: block + "\n\n" + question,
|
||||
}}
|
||||
rec.Retrieval(retrieved)
|
||||
if !smalltalk {
|
||||
if block, retrieved := m.retrieve(runCtx, rec, agent, input, question); block != "" {
|
||||
conversation = []gateway.Message{{
|
||||
Role: gateway.RoleUser,
|
||||
// Context first, question second. A model reads the question last
|
||||
// and answers it, rather than treating the evidence as the prompt.
|
||||
Text: block + "\n\n" + question,
|
||||
}}
|
||||
rec.Retrieval(retrieved)
|
||||
}
|
||||
}
|
||||
|
||||
// Recorded once rather than on each remaining step, so a long run does not
|
||||
// fill its trajectory with the same note.
|
||||
var toolsWithheld bool
|
||||
|
||||
system := SystemPrompt(agent)
|
||||
if smalltalk {
|
||||
// Taking the evidence away removes the citations; it does not by itself
|
||||
// shorten the reply, because the agent's own instructions still
|
||||
// describe an operational analyst. See smalltalkDirective.
|
||||
system += smalltalkDirective
|
||||
}
|
||||
var lastText string
|
||||
|
||||
for {
|
||||
@@ -255,11 +314,43 @@ func (m *ModelExecutor) executeRun(
|
||||
// halves go through StreamComplete, so the loop has one call site and
|
||||
// no branch on transport — a run behaves identically whether its text
|
||||
// arrived in one piece or a hundred.
|
||||
// The catalogue has to be resent on every call — the wire protocol has
|
||||
// no way to refer back to one already sent — but a catalogue the model
|
||||
// is no longer ALLOWED to use is pure waste. Once the tool-call budget
|
||||
// is spent, every definition describes a call that would be refused,
|
||||
// and the step it is being sent on is the synthesis turn that just
|
||||
// needs to write the answer up.
|
||||
//
|
||||
// Measured on this deployment's control-center agent: seven tools,
|
||||
// ~1.2k tokens, resent on the final call of every tool-using run.
|
||||
stepTools := toolDefs
|
||||
if len(stepTools) > 0 && budget.Snapshot().ToolCallsLeft <= 0 {
|
||||
stepTools = nil
|
||||
if !toolsWithheld {
|
||||
toolsWithheld = true
|
||||
rec.Error("runtime.tools_withheld",
|
||||
"tool-call budget spent; catalogue not resent on the remaining steps")
|
||||
}
|
||||
}
|
||||
|
||||
// Smalltalk is capped far below the tier's ceiling. The directive in
|
||||
// the system prompt is what actually shortens the reply; this only
|
||||
// bounds the bill for a model that ignores it.
|
||||
maxOut := budget.Limits().MaxOutputTokens
|
||||
if smalltalk && (maxOut <= 0 || maxOut > smalltalkMaxOutputTokens) {
|
||||
maxOut = smalltalkMaxOutputTokens
|
||||
}
|
||||
|
||||
resp, err := gateway.StreamComplete(runCtx, m.gw, gateway.Request{
|
||||
Tier: tier,
|
||||
System: system,
|
||||
Messages: conversation,
|
||||
Tools: toolDefs,
|
||||
Tools: stepTools,
|
||||
// Per STEP, and taken from the budget rather than from the tier —
|
||||
// so a delegated run inherits the parent's ceiling along with the
|
||||
// parent's budget instead of reading its own tier and quietly
|
||||
// buying a longer answer than the parent was allowed.
|
||||
MaxOutputTokens: maxOut,
|
||||
}, input.OnDelta)
|
||||
|
||||
// Charged whatever happened. A refused or failed call was still billed,
|
||||
@@ -578,9 +669,13 @@ func terminationFor(err error) Termination {
|
||||
|
||||
// finish closes the trajectory, persists it, and builds the caller's result.
|
||||
//
|
||||
// Persistence uses the *caller's* context, not the run's: the run context is
|
||||
// Persistence deliberately outlives the run's context: that context is
|
||||
// cancelled at the deadline, and a run that ended by running out of time is
|
||||
// exactly the one whose record is most worth keeping.
|
||||
// exactly the one whose record is most worth keeping. It does NOT outlive the
|
||||
// caller's patience — trajectoryPersistTimeout bounds the whole write, because
|
||||
// a run that answered inside its deadline and then sat in the sink for a minute
|
||||
// is, to the person waiting, a slow run. I3 bounds the run; this bounds its
|
||||
// tail.
|
||||
func (m *ModelExecutor) finish(
|
||||
ctx context.Context,
|
||||
rec *Recorder,
|
||||
@@ -594,7 +689,16 @@ func (m *ModelExecutor) finish(
|
||||
if cause != nil {
|
||||
var gwErr *gateway.Error
|
||||
if errors.As(cause, &gwErr) {
|
||||
rec.Error(gwErr.Code, gwErr.Message)
|
||||
// The status rides in the message because `entries` has no column
|
||||
// for it and E5 forbids applying a migration from here. It matters:
|
||||
// `gateway.upstream` alone cannot tell a provider shedding load
|
||||
// (5xx, clears by itself) from an endpoint rejecting the request
|
||||
// (4xx, needs an administrator), and those are opposite actions.
|
||||
msg := gwErr.Message
|
||||
if gwErr.Status > 0 {
|
||||
msg = fmt.Sprintf("http %d: %s", gwErr.Status, msg)
|
||||
}
|
||||
rec.Error(gwErr.Code, msg)
|
||||
} else {
|
||||
rec.Error("runtime.failed", cause.Error())
|
||||
}
|
||||
@@ -602,11 +706,19 @@ func (m *ModelExecutor) finish(
|
||||
rec.Budget(budget.Snapshot())
|
||||
traj := rec.Finish(term)
|
||||
|
||||
// WithoutCancel so a deadline-terminated run still records itself; the
|
||||
// timeout so it cannot record itself forever. One budget covers the parent
|
||||
// and every child, since writing the tree is one logical act and a
|
||||
// per-trajectory timeout would multiply by the number of subagents.
|
||||
persistCtx, cancelPersist := context.WithTimeout(
|
||||
context.WithoutCancel(ctx), trajectoryPersistTimeout)
|
||||
defer cancelPersist()
|
||||
|
||||
// A sink that fails must not fail the run — the answer was already
|
||||
// produced. It is recorded in the trajectory we could not save, which is
|
||||
// the best available place for it.
|
||||
var unsaved []string
|
||||
if err := m.sink.Save(ctx, traj); err != nil {
|
||||
if err := m.sink.Save(persistCtx, traj); err != nil {
|
||||
rec.Error("runtime.trajectory_unsaved", err.Error())
|
||||
unsaved = append(unsaved, traj.RunID+": "+err.Error())
|
||||
}
|
||||
@@ -616,7 +728,7 @@ func (m *ModelExecutor) finish(
|
||||
// own descendants already ordered behind it, so one pass here writes a
|
||||
// whole tree parent-first.
|
||||
for _, child := range rec.Children() {
|
||||
if err := m.sink.Save(ctx, child); err != nil {
|
||||
if err := m.sink.Save(persistCtx, child); err != nil {
|
||||
rec.Error("runtime.subrun_unsaved",
|
||||
fmt.Sprintf("%s: %s", child.RunID, err.Error()))
|
||||
unsaved = append(unsaved, child.RunID+": "+err.Error())
|
||||
|
||||
Reference in New Issue
Block a user