agent build

This commit is contained in:
2026-08-28 12:21:44 +05:30
parent b6f8655909
commit f7df96c973
138 changed files with 24164 additions and 207 deletions

View File

@@ -0,0 +1,233 @@
package runtime
import (
"context"
"fmt"
"sync"
"time"
)
// Termination is how a run ended. Exactly one per run, always.
//
// An enum rather than a boolean and a message, because "what happened" is the
// first question asked of every trajectory — in a debugger, in an eval report,
// and in a support conversation — and a free-text reason cannot be grouped,
// counted or asserted on.
type Termination string
const (
TerminationCompleted Termination = "Completed"
TerminationBudgetExceeded Termination = "BudgetExceeded"
TerminationDeadline Termination = "Deadline"
TerminationConfirmationPending Termination = "ConfirmationPending"
TerminationToolFailure Termination = "ToolFailure"
TerminationRefused Termination = "Refused"
)
// Valid reports whether t is one of the six.
func (t Termination) Valid() bool {
switch t {
case TerminationCompleted, TerminationBudgetExceeded, TerminationDeadline,
TerminationConfirmationPending, TerminationToolFailure, TerminationRefused:
return true
}
return false
}
// Limits are the four bounds every run carries.
//
// I3: there is no "run until done" path. A run that reaches any of these ends
// with a structured result, never an exception into user-facing text.
type Limits struct {
MaxSteps int
MaxToolCalls int
MaxTokens int64
Deadline time.Duration
}
// LimitsForTier is what a run gets when its spec declares no limits of its own.
//
// Derived from the reasoning tier because that is the only thing a Krow agent
// definition says today about how much work it is worth. A `limits:` block in
// the frontmatter would override this per agent; adding one changes the spec
// contract and the authoring UI together, so it is a deliberate schema
// decision rather than something to infer here.
//
// The numbers are chosen so that the cheapest tier cannot quietly become the
// expensive one: a fast run gets a third of a deep run's steps and a sixth of
// its deadline, so a misrouted spec shows up as a truncated answer rather than
// as a bill.
func LimitsForTier(tier string) Limits {
switch tier {
case "fast":
return Limits{MaxSteps: 3, MaxToolCalls: 4, MaxTokens: 40_000, Deadline: 20 * time.Second}
case "deep":
return Limits{MaxSteps: 12, MaxToolCalls: 20, MaxTokens: 300_000, Deadline: 120 * time.Second}
default: // balanced, and anything unrecognised — ParseTier has already normalised it
return Limits{MaxSteps: 8, MaxToolCalls: 12, MaxTokens: 120_000, Deadline: 60 * time.Second}
}
}
// Snapshot is what a budget had left at one moment. Recorded into the
// trajectory before every dispatch, so "where did the budget go" is answerable
// after the fact instead of being reconstructed from timings.
type Snapshot struct {
StepsUsed int `json:"stepsUsed"`
StepsLeft int `json:"stepsLeft"`
ToolCallsUsed int `json:"toolCallsUsed"`
ToolCallsLeft int `json:"toolCallsLeft"`
TokensUsed int64 `json:"tokensUsed"`
TokensLeft int64 `json:"tokensLeft"`
MillisLeft int64 `json:"millisLeft"`
}
// Budget tracks one run against its limits.
//
// **Everything is claimed before dispatch, never after.** A step is spent the
// moment the loop decides to take it, not when it returns — otherwise a call
// that hangs until the context dies has consumed nothing on the ledger, and a
// loop that retries it can go round forever while the budget reads full.
//
// Tokens are the exception that proves the rule: their true cost is only known
// once a response comes back, so the budget is checked before dispatch and
// charged after. That leaves one turn of overshoot, bounded by MaxOutputTokens
// on the request, which is why the model gateway takes a hard per-call ceiling
// as well as this soft per-run one.
//
// Safe for concurrent use: subagents share their parent's budget, and two
// delegated branches must not both see the last step as available.
type Budget struct {
limits Limits
start time.Time
mu sync.Mutex
steps int
toolCalls int
tokens int64
}
// NewBudget starts a budget. The wall clock starts now: a run's deadline is
// measured from when it began, not from when it first reached a model.
func NewBudget(limits Limits) *Budget {
return &Budget{limits: limits, start: time.Now()}
}
// Limits returns the bounds this budget enforces.
func (b *Budget) Limits() Limits { return b.limits }
// ClaimStep takes one step up front, reporting the termination to end with if
// there was nothing left to take.
//
// The returned Termination is empty when the claim succeeded. Callers branch on
// that rather than on a boolean, so the reason a run stopped travels with the
// refusal instead of being re-derived at the call site.
func (b *Budget) ClaimStep() Termination {
if t := b.expired(); t != "" {
return t
}
b.mu.Lock()
defer b.mu.Unlock()
if b.steps >= b.limits.MaxSteps {
return TerminationBudgetExceeded
}
b.steps++
return ""
}
// ClaimToolCall takes one tool call up front.
func (b *Budget) ClaimToolCall() Termination {
if t := b.expired(); t != "" {
return t
}
b.mu.Lock()
defer b.mu.Unlock()
if b.toolCalls >= b.limits.MaxToolCalls {
return TerminationBudgetExceeded
}
b.toolCalls++
return ""
}
// CheckTokens reports whether there is token budget left to dispatch against.
//
// Checked before, charged after — see the type comment. A run that has already
// spent its allowance stops here rather than issuing one more call it cannot
// pay for.
func (b *Budget) CheckTokens() Termination {
if t := b.expired(); t != "" {
return t
}
b.mu.Lock()
defer b.mu.Unlock()
if b.tokens >= b.limits.MaxTokens {
return TerminationBudgetExceeded
}
return ""
}
// ChargeTokens records what a completed call actually cost.
//
// Called for refused and failed calls too. A turn that produced no text was
// still billed, and a ledger that forgives it is a ledger a loop will happily
// repeat against.
func (b *Budget) ChargeTokens(n int64) {
if n <= 0 {
return
}
b.mu.Lock()
defer b.mu.Unlock()
b.tokens += n
}
// expired reports the deadline having passed. Separate from the step and tool
// checks because it is a different termination reason: a run that ran out of
// time did not run out of budget, and conflating them hides which bound is
// actually being hit in production.
func (b *Budget) expired() Termination {
if time.Since(b.start) >= b.limits.Deadline {
return TerminationDeadline
}
return ""
}
// Context returns a context that is cancelled at the run's deadline.
//
// The same deadline the budget enforces, so an in-flight model call is torn
// down rather than being allowed to return into a run that has already ended.
func (b *Budget) Context(parent context.Context) (context.Context, context.CancelFunc) {
return context.WithDeadline(parent, b.start.Add(b.limits.Deadline))
}
// Snapshot reads the budget without changing it.
func (b *Budget) Snapshot() Snapshot {
b.mu.Lock()
defer b.mu.Unlock()
left := b.limits.Deadline - time.Since(b.start)
if left < 0 {
left = 0
}
return Snapshot{
StepsUsed: b.steps,
StepsLeft: max(0, b.limits.MaxSteps-b.steps),
ToolCallsUsed: b.toolCalls,
ToolCallsLeft: max(0, b.limits.MaxToolCalls-b.toolCalls),
TokensUsed: b.tokens,
TokensLeft: maxInt64(0, b.limits.MaxTokens-b.tokens),
MillisLeft: left.Milliseconds(),
}
}
func (s Snapshot) String() string {
return fmt.Sprintf("steps %d/%d, tools %d/%d, tokens %d, %dms left",
s.StepsUsed, s.StepsUsed+s.StepsLeft,
s.ToolCallsUsed, s.ToolCallsUsed+s.ToolCallsLeft,
s.TokensUsed, s.MillisLeft)
}
func maxInt64(a, b int64) int64 {
if a > b {
return a
}
return b
}

View File

@@ -0,0 +1,102 @@
package runtime_test
import (
"strings"
"testing"
"github.com/krow/krow-backend/go-api/internal/config"
"github.com/krow/krow-backend/go-api/internal/runtime"
)
// Which embedder a deployment actually gets.
//
// The reason this is worth testing at all: all three providers return vectors,
// and retrieval works with any of them. A deployment running the stand-in looks
// exactly like one running a real model — same shape of result, same citations,
// same confidence — until somebody phrases a question differently. There is no
// symptom to notice, so the choice has to be asserted rather than observed.
func embedderFor(k config.KnowledgeConfig, env string) string {
e := runtime.NewEmbedder(config.Config{AppEnv: env, Knowledge: k})
if e == nil {
return "none"
}
return e.Model()
}
func TestTheNamedProviderWins(t *testing.T) {
// Explicit beats inferred. A deployment that names ollama gets ollama even
// with a Voyage key sitting in the environment — otherwise a leftover
// credential silently decides where tenant text goes.
got := embedderFor(config.KnowledgeConfig{
EmbedProvider: "ollama",
EmbedAPIKey: "pa-a-real-looking-key",
}, "development")
if !strings.HasPrefix(got, "ollama/") {
t.Errorf("named ollama and got %q; a stray credential overrode an explicit choice", got)
}
got = embedderFor(config.KnowledgeConfig{
EmbedProvider: "voyage",
EmbedAPIKey: "pa-key",
EmbedBaseURL: "http://localhost:11434",
}, "development")
if strings.HasPrefix(got, "ollama/") {
t.Errorf("named voyage and got %q", got)
}
}
func TestWithNothingConfiguredThereIsNoEmbedder(t *testing.T) {
// Nil, not a hosted client with an empty key. Both end up keyword-only, but
// nil says so once at wiring time instead of failing one HTTP call per
// query to learn the same thing.
if got := embedderFor(config.KnowledgeConfig{}, "development"); got != "none" {
t.Errorf("an unconfigured deployment got %q, want no embedder", got)
}
}
func TestNamingVoyageWithoutAKeyIsNotAnEmbedder(t *testing.T) {
// Named but unusable. Degrading honestly beats failing a request per query
// on a credential nobody set.
if got := embedderFor(config.KnowledgeConfig{EmbedProvider: "voyage"}, "development"); got != "none" {
t.Errorf("voyage with no key produced %q", got)
}
}
func TestInferencePrefersTheLocalModel(t *testing.T) {
// With nothing named, a configured local model wins over a hosted one: it
// costs nothing and keeps tenant text on the host, and both are real
// semantic embedders.
got := embedderFor(config.KnowledgeConfig{
EmbedBaseURL: "http://localhost:11434",
EmbedAPIKey: "pa-key",
}, "development")
if !strings.HasPrefix(got, "ollama/") {
t.Errorf("inferred %q; a local model should win when both are available", got)
}
}
func TestTheStandInRefusesToRunInProduction(t *testing.T) {
// It is not semantic. A production corpus indexed with it retrieves on word
// overlap alone — which looks like working retrieval and is not, which is
// exactly why the refusal is in the code and not in a comment.
e := runtime.NewEmbedder(config.Config{
AppEnv: "production",
Knowledge: config.KnowledgeConfig{EmbedProvider: "lexical"},
})
if e == nil {
t.Fatal("expected the stand-in, refusing at use rather than at wiring")
}
if _, err := e.Embed(t.Context(), []string{"x"}, "document"); err == nil {
t.Error("the stand-in embedded text in production")
}
// And in development it works, because that is what it is for.
dev := runtime.NewEmbedder(config.Config{
AppEnv: "development",
Knowledge: config.KnowledgeConfig{EmbedProvider: "lexical"},
})
if _, err := dev.Embed(t.Context(), []string{"x"}, "document"); err != nil {
t.Errorf("the stand-in refused in development: %v", err)
}
}

View File

@@ -2,6 +2,7 @@ package runtime
import (
"context"
"fmt"
"github.com/krow/krow-backend/go-api/internal/authctx"
"github.com/krow/krow-backend/go-api/internal/repo"
@@ -88,13 +89,27 @@ func NewEngine(db repo.Querier, opts ...Option) *Engine {
// RunAgent loads an executable agent with dependencies and dispatches to the executor boundary.
func (e *Engine) RunAgent(ctx context.Context, ident authctx.Identity, idOrDefID string, input ExecutionInput) (*ExecutionResult, error) {
agent, err := e.Loader.LoadExecutableAgent(ctx, ident, idOrDefID)
// A pinned run loads the agent AS IT WAS. §3: running conversations pin the
// version they started with — which matters most on a resumed run, where a
// person approved a write while looking at one version and an edit may have
// landed since.
agent, unpinned, err := e.Loader.LoadAgentVersion(ctx, ident, idOrDefID, input.AgentVersion)
if err != nil {
return &ExecutionResult{
Success: false,
Error: err,
}, err
}
if unpinned {
// Asked for a version that has no snapshot — a definition published
// before versions were recorded. The run proceeds on the current
// definition rather than failing, and says so, because a silent
// substitution is the thing worth preventing.
input.Notes = append(input.Notes, fmt.Sprintf(
"version_unavailable: version %d of %s is not in the published history; "+
"this run used the current definition (v%d)",
input.AgentVersion, agent.ID, agent.Version))
}
res, err := e.AgentExec.ExecuteAgent(ctx, agent, input)
if res == nil {

View File

@@ -21,11 +21,18 @@ func isUUID(s string) bool {
// Loader loads and validates authored definitions into runtime representations with tenant isolation.
type Loader struct {
repo *repo.DefinitionsRepo
// versions resolves a pinned version back to the definition that answered.
// See LoadAgentVersion.
versions *repo.VersionsRepo
}
// NewLoader builds a runtime definition loader over a storage repository.
func NewLoader(db repo.Querier) *Loader {
return &Loader{repo: repo.NewDefinitionsRepo(db)}
return &Loader{
repo: repo.NewDefinitionsRepo(db),
versions: repo.NewVersionsRepo(db),
}
}
// LoadAgent loads an agent definition by id or definition_id, parsing it into a runtime representation.
@@ -77,8 +84,13 @@ func (l *Loader) LoadAgent(ctx context.Context, ident authctx.Identity, idOrDefI
WebSearch: parsed.WebSearch,
Instructions: parsed.Instructions,
Skills: parsed.Skills,
Subagents: parsed.Subagents,
RawMarkdown: rawMD,
Tools: parsed.Tools,
// The spec's corpora, and the ONLY place they come from. A model that
// asked to search a source its agent was not granted is asking for a
// list it has no way to set — see tools.Context.KnowledgeSources.
KnowledgeSources: parsed.Sources,
Subagents: parsed.Subagents,
RawMarkdown: rawMD,
}
if rec["owner_user_id"] != nil {
@@ -235,3 +247,59 @@ func (l *Loader) LoadExecutableSkill(ctx context.Context, ident authctx.Identity
return skill, nil
}
/* ── Pinned versions ────────────────────────────────────────────────────── */
// LoadAgentVersion loads an agent AS IT WAS at a published version.
//
// The current definition is not consulted at all — that is the point. An agent
// edited since a conversation began is a different agent, and a run that quietly
// switched to it would answer a question the reader never asked with tools they
// were never offered.
//
// Falls back to the current definition when the version is not in the history,
// and does so deliberately rather than failing. Every definition published
// before migration 000010 has no snapshot; refusing those would break every
// existing conversation to enforce a rule that could not have been followed
// when they started. The fallback is recorded by the caller, so a run that
// could not pin is visible rather than silent.
func (l *Loader) LoadAgentVersion(ctx context.Context, ident authctx.Identity,
idOrDefID string, version int) (*Agent, bool, error) {
current, err := l.LoadExecutableAgent(ctx, ident, idOrDefID)
if err != nil {
return nil, false, err
}
if version <= 0 || version == current.Version {
return current, false, nil
}
snapshot, err := l.versions.Load(ctx, ident, repo.KindAgent, current.ID, version)
if err != nil || snapshot == nil {
// No snapshot for that number. The run continues on the current
// definition — see the note above — and the caller records that it
// could not pin.
return current, true, nil
}
parsed, err := definition.ParseAgent(snapshot.Markdown, definition.Options{})
if err != nil {
return current, true, nil
}
// Rebuilt from the snapshot, keeping the identity fields that belong to the
// row rather than to the definition text.
pinned := *current
pinned.Name = parsed.Name
pinned.Description = parsed.Description
pinned.Version = snapshot.Version
pinned.Pages = parsed.Pages
pinned.Reasoning = parsed.Reasoning
pinned.Instructions = parsed.Instructions
pinned.Skills = parsed.Skills
pinned.Tools = parsed.Tools
pinned.KnowledgeSources = parsed.Sources
pinned.Subagents = parsed.Subagents
pinned.RawMarkdown = snapshot.Markdown
return &pinned, false, nil
}

View File

@@ -0,0 +1,630 @@
package runtime
import (
"context"
"crypto/rand"
"encoding/hex"
"encoding/json"
"errors"
"fmt"
"strings"
"github.com/krow/krow-backend/go-api/internal/gateway"
"github.com/krow/krow-backend/go-api/internal/knowledge"
"github.com/krow/krow-backend/go-api/internal/tools"
)
// ModelExecutor runs an agent against a model.
//
// This is the agent loop. It is spec-driven and there is exactly one of it: no
// branch anywhere below asks which agent it is running. An agent's identity
// reaches this code only as data — its instructions, its tier, its skills —
// which is what I6 means in practice and what makes adding an agent a data
// change rather than a deploy.
//
// The loop runs until the model stops asking for tools, or until a bound is
// reached. Every exit is one of the six terminations.
type ModelExecutor struct {
gw gateway.Gateway
sink Sink
tools *tools.Registry
// retriever is the knowledge layer, or nil for an agent platform with no
// documents in it. Nil is a supported state rather than a broken one: every
// agent built so far answers from the operational tables through tools, and
// none of them needs a corpus.
retriever Retriever
}
// Retriever is what the loop needs from the knowledge layer.
//
// An interface rather than the concrete type so the runtime does not import the
// knowledge package's whole surface, and so a test can drive the loop with a
// scripted corpus. Deliberately narrow: the loop retrieves, it does not ingest,
// and it has no way to ask for anything other than the caller's own rows —
// knowledge.Query requires a principal and this signature carries one.
type Retriever interface {
Retrieve(ctx context.Context, q knowledge.Query) (*knowledge.Results, error)
}
// WithRetriever attaches a knowledge layer to an executor.
func (m *ModelExecutor) WithRetriever(r Retriever) *ModelExecutor {
m.retriever = r
return m
}
var _ AgentExecutor = (*ModelExecutor)(nil)
// NewModelExecutor builds the loop over a model gateway.
//
// A nil sink is DiscardSink rather than a panic: a service wired without a
// trajectory store should still answer, and losing the record is a worse
// outcome than nothing but not one worth refusing a correct answer over.
// A nil registry is an empty one: an agent that names no tools does not need
// one, and a nil map dereference is a worse way to discover that than an agent
// that simply has nothing to call.
func NewModelExecutor(gw gateway.Gateway, sink Sink, reg *tools.Registry) *ModelExecutor {
if sink == nil {
sink = DiscardSink{}
}
if reg == nil {
reg = tools.NewRegistry()
}
return &ModelExecutor{gw: gw, sink: sink, tools: reg}
}
// newRunID returns an opaque run identifier.
//
// Random rather than sequential: a run id appears in logs and in support
// conversations, and a sequential one would leak how many runs a deployment
// has served.
func newRunID() string {
var b [16]byte
if _, err := rand.Read(b[:]); err != nil {
// crypto/rand does not fail in practice; if it ever does, a run
// without an id is still better than a run that refuses to start.
return "run-unknown"
}
return "run_" + hex.EncodeToString(b[:])
}
// ExecuteAgent runs one agent turn and returns a structured result.
//
// It never returns a bare error into user-facing text. Every exit is a
// termination reason plus a trajectory, because §6 requires exactly one
// termination per run and §10 requires user-facing text to be derived at the
// surface layer rather than raised from here.
func (m *ModelExecutor) ExecuteAgent(ctx context.Context, agent *Agent, input ExecutionInput) (*ExecutionResult, error) {
tier, _ := gateway.ParseTier(agent.Reasoning)
return m.executeWithLimits(ctx, agent, input, LimitsForTier(string(tier)))
}
// executeWithLimits is ExecuteAgent with the bounds supplied rather than
// derived.
//
// The seam exists for two reasons and will earn its keep for the second. Today
// it lets a test drive a real deadline instead of asserting on a counter. When
// the spec gains a `limits:` block, that block resolves here and ExecuteAgent
// stays the one-line default — so per-agent limits arrive without the loop
// itself changing shape.
func (m *ModelExecutor) executeWithLimits(
ctx context.Context, agent *Agent, input ExecutionInput, limits Limits,
) (*ExecutionResult, error) {
tier, known := gateway.ParseTier(agent.Reasoning)
budget := NewBudget(limits)
skillIDs := make([]string, len(agent.ResolvedSkills))
for i, s := range agent.ResolvedSkills {
skillIDs[i] = s.ID
}
rec := NewRecorder(&Trajectory{
RunID: newRunID(),
OrgID: input.Identity.OrgID,
UserID: input.Identity.UserID,
AgentID: agent.ID,
AgentVersion: agent.Version,
Tier: string(tier),
})
// A spec naming a tier the vocabulary does not have still runs, at the
// default — but it is recorded, so a definition that has drifted is
// visible in the trajectory rather than silently reinterpreted.
if !known {
rec.Error("runtime.unknown_tier",
fmt.Sprintf("%q is not a reasoning mode; running at %s", agent.Reasoning, tier))
}
// The deadline is the budget's, so an in-flight model call is torn down
// rather than returning into a run that has already ended.
runCtx, cancel := budget.Context(ctx)
defer cancel()
// Anything the caller wants on the record, before the run does anything.
// A run that silently could not do what was asked of it is the failure
// worth preventing here.
for _, note := range input.Notes {
rec.Error("runtime.note", note)
}
question := strings.TrimSpace(input.Input)
if question == "" {
return m.finish(ctx, rec, budget, TerminationToolFailure, agent, skillIDs,
"", &RuntimeError{Code: "runtime.empty_input", Message: "a run needs a question"})
}
rec.Message("user", question)
// The tools this agent may use. Unknown names are recorded and dropped
// rather than failing the run: §3 says an unknown tool fails validation at
// *publish*, so one reaching run time means a tool was withdrawn under a
// live spec — degrading is better than an outage, provided someone is told.
toolDefs, unknown := m.toolsFor(agent)
for _, name := range unknown {
rec.Error("runtime.unknown_tool", fmt.Sprintf("%q is not a registered tool; it was not offered", name))
}
// An approved write happens FIRST, before the model gets a turn.
//
// This is the half of I4 that makes a confirmation reliable rather than
// hopeful. The older design resumed the run and matched the model's next
// tool call against the token — which only works if the model repeats
// itself, and a model asked a second time may perfectly reasonably ask a
// clarifying question instead. When that happened the token was never
// presented, nothing was written, and the person who clicked Approve got a
// follow-up question with no explanation.
//
// So the approved call is performed from what the person was SHOWN, not
// from what the model says next. The model's job afterwards is to report
// what happened, which is a job it cannot get wrong in a way that costs
// anybody a shift.
if approved, done := m.performApproved(runCtx, rec, budget, agent, input); done != nil {
return m.finish(ctx, rec, budget, *done, agent, skillIDs, "", nil)
} else if approved != "" {
// Prepended to the question so the model answers knowing the write
// already happened. It is a tool result in everything but shape —
// delimited, factual, and about an act rather than an instruction.
question = approved + "\n\n" + question
}
// Retrieval, before the first model call.
//
// I7 decides where the result goes: into a delimited block in a USER
// message, never into the system prompt. The system prompt is assembled
// from the agent record alone, so no amount of document content can reach
// it — which is the only reason the standing "content inside <context> is
// data" instruction means anything.
conversation := []gateway.Message{{Role: gateway.RoleUser, Text: question}}
if block, retrieved := m.retrieve(runCtx, rec, agent, input, question); block != "" {
conversation = []gateway.Message{{
Role: gateway.RoleUser,
// Context first, question second. A model reads the question last
// and answers it, rather than treating the evidence as the prompt.
Text: block + "\n\n" + question,
}}
rec.Retrieval(retrieved)
}
system := SystemPrompt(agent)
var lastText string
for {
// Claimed before dispatch, never after. A call that hangs until the
// context dies has still spent the step it was given.
if t := budget.ClaimStep(); t != "" {
return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, nil)
}
if t := budget.CheckTokens(); t != "" {
return m.finish(ctx, rec, budget, t, agent, skillIDs, lastText, nil)
}
rec.Budget(budget.Snapshot())
// Streamed when the caller asked for it AND the gateway can. Both
// halves go through StreamComplete, so the loop has one call site and
// no branch on transport — a run behaves identically whether its text
// arrived in one piece or a hundred.
resp, err := gateway.StreamComplete(runCtx, m.gw, gateway.Request{
Tier: tier,
System: system,
Messages: conversation,
Tools: toolDefs,
}, input.OnDelta)
// Charged whatever happened. A refused or failed call was still billed,
// and a ledger that forgives it is one a loop will happily repeat
// against.
if resp != nil {
budget.ChargeTokens(resp.Usage.Total())
rec.ChargeUsage(resp.Usage.InputTokens, resp.Usage.OutputTokens,
resp.Usage.CacheReadTokens+resp.Usage.CacheCreationTokens)
rec.SetModel(resp.Model)
}
if err != nil {
return m.finish(ctx, rec, budget, terminationFor(err), agent, skillIDs, lastText, err)
}
if resp.Text != "" {
rec.Message("assistant", resp.Text)
lastText = resp.Text
}
// No tool calls means the model is done talking.
if len(resp.ToolCalls) == 0 {
return m.finish(ctx, rec, budget, TerminationCompleted, agent, skillIDs, lastText, nil)
}
// The assistant turn goes back verbatim, calls included, before any
// result is appended — a tool result with no preceding call is a
// malformed conversation the API will reject.
conversation = append(conversation, gateway.Message{
Role: gateway.RoleAssistant, Text: resp.Text, ToolCalls: resp.ToolCalls,
})
results, pending, term := m.runTools(runCtx, rec, budget, agent, input, resp.ToolCalls)
if term != "" {
return m.finish(ctx, rec, budget, term, agent, skillIDs, lastText, nil)
}
// I4. A run that wants to write stops here and asks. It does not
// continue with the reads it also made, does not summarise, and does
// not get another turn to reconsider — the next thing that happens is a
// person deciding, and the run resumes only if they say yes.
if len(pending) > 0 {
res, err := m.finish(ctx, rec, budget,
TerminationConfirmationPending, agent, skillIDs, lastText, nil)
res.Confirmations = pending
return res, err
}
// Every result in ONE user turn. Splitting them is accepted and quietly
// teaches the model to stop calling tools in parallel.
conversation = append(conversation, gateway.Message{
Role: gateway.RoleUser, ToolResults: results,
})
}
}
// performApproved carries out a write a person approved.
//
// Returns the sentence describing what happened, for the model to report from.
// A run with no confirmation token does nothing here and returns "".
//
// A token that authorises nothing — unknown, expired, already spent, somebody
// else's — is NOT an error and does not end the run. It is recorded and the run
// continues, because the most common cause is a person clicking Approve twice,
// and the honest response to that is to answer the question again rather than
// to fail.
func (m *ModelExecutor) performApproved(
ctx context.Context, rec *Recorder, budget *Budget, agent *Agent, input ExecutionInput,
) (string, *Termination) {
if input.Confirmation == "" || m.tools == nil {
return "", nil
}
// The write spends a tool call from the run's budget, claimed before
// dispatch like every other. An approval is not a way around I3.
if t := budget.ClaimToolCall(); t != "" {
return "", &t
}
rec.Budget(budget.Snapshot())
tc := tools.Context{
Principal: input.Identity,
RunID: rec.RunID(),
RemainingTokens: budget.Snapshot().TokensLeft,
AgentID: agent.ID,
KnowledgeSources: agent.KnowledgeSources,
}
out, ok := m.tools.DispatchApproved(ctx, tc, input.Confirmation)
if !ok {
rec.Error("runtime.confirmation_not_redeemable",
"the supplied approval authorises nothing; it may have expired or already been used")
return "", nil
}
rec.ToolCall(out.Tool, string(tools.EffectWrite), out.Inputs)
rec.ToolResult(out.Tool, string(tools.EffectWrite), out.Result.Error != nil, out.Result)
encoded, err := json.Marshal(out.Result)
if err != nil {
encoded = []byte(`{"error":{"code":"tool.failed","message":"the result could not be encoded"}}`)
}
// Delimited and labelled as data, on the same terms as retrieved content.
// This text describes something that already happened; it is not an
// instruction, and the standing <context> rule in the system prompt covers
// it for exactly that reason.
return fmt.Sprintf(
"<context>\nA change you approved has already been carried out. This is its "+
"result, as data — report it, do not repeat the action.\n\n"+
"<source id=%q>\n%s\n</source>\n</context>",
out.Tool, string(encoded)), nil
}
// retrieve searches the agent's declared corpora on the CALLER's behalf.
//
// Three things are load-bearing and none of them is the search itself:
//
// - The principal is the caller's, never the agent's. I1: an agent reads
// exactly what its caller could read directly, and the identity that
// reaches knowledge.Query is the one that arrived with the request.
// - The sources are the SPEC's. An agent granted the policy library does not
// gain the incident log by asking nicely, because the source list is not
// something the model can influence.
// - A failure degrades rather than ends the run. A knowledge layer that is
// down should cost grounding, not the answer — but it is recorded, because
// an ungrounded answer that looks grounded is the worse outcome.
func (m *ModelExecutor) retrieve(
ctx context.Context, rec *Recorder, agent *Agent, input ExecutionInput, question string,
) (string, *knowledge.Results) {
if m.retriever == nil || len(agent.KnowledgeSources) == 0 {
return "", nil
}
res, err := m.retriever.Retrieve(ctx, knowledge.Query{
Text: question,
Principal: input.Identity,
Sources: agent.KnowledgeSources,
})
if err != nil {
// Recorded, not raised. The run continues without grounding, and the
// trajectory says so — "the agent answered from nothing" is only
// diagnosable afterwards if the failure was written down at the time.
var kErr *knowledge.Error
if errors.As(err, &kErr) {
rec.Error(kErr.Code, kErr.Message)
} else {
rec.Error("knowledge.failed", err.Error())
}
return "", nil
}
if res == nil || len(res.Chunks) == 0 {
return "", res
}
return knowledge.RenderContext(res), res
}
// toolsFor resolves the tools an agent's spec names.
func (m *ModelExecutor) toolsFor(agent *Agent) (defs []gateway.ToolDef, unknown []string) {
if m.tools == nil || len(agent.Tools) == 0 {
return nil, nil
}
resolved, unknown, err := m.tools.Resolve(agent.Tools)
if err != nil {
// Over the per-agent cap. Offering none is the safe reading: an agent
// that silently got its first twenty tools would behave differently
// depending on the order someone happened to write them in.
return nil, agent.Tools
}
for _, t := range resolved {
defs = append(defs, gateway.ToolDef{
Name: t.Name, Description: t.Description, InputSchema: t.InputSchema,
})
}
return defs, unknown
}
// runTools dispatches one turn's calls and returns their results.
//
// A tool that fails returns its error TO THE MODEL rather than ending the run.
// §13 lists "swallowing a tool error and letting the model narrate around it"
// as an anti-pattern — the fix is not to hide the failure but to hand it over
// as a failure, so the model can say it could not look rather than inventing
// what it would have found.
//
// The tool-call budget is claimed per call, before dispatch. Running out ends
// the run: a model that has exhausted its calls cannot make progress, and
// letting it continue would spend the remaining step budget on turns that can
// only apologise.
//
// A write that needs approving comes back as a pending confirmation rather than
// a result. Those are collected across the whole turn rather than returned at
// the first one, so a person is asked about every write the model wanted in one
// go instead of being walked through them one dialog at a time — and so that
// the reads in the same turn, which are safe, still run and are still recorded.
func (m *ModelExecutor) runTools(
ctx context.Context, rec *Recorder, budget *Budget, agent *Agent,
input ExecutionInput, calls []gateway.ToolCall,
) (results []gateway.ToolResult, pending []*tools.Confirmation, term Termination) {
results = make([]gateway.ToolResult, 0, len(calls))
for _, call := range calls {
if t := budget.ClaimToolCall(); t != "" {
return nil, nil, t
}
rec.Budget(budget.Snapshot())
// The declared effect travels with the record. An eval asking "did this
// run change anything" reads it from here rather than keeping its own
// list of which tools write — a list that goes stale on the first tool
// anybody adds.
var effect string
if t, ok := m.tools.Get(call.Name); ok {
effect = string(t.Effect)
}
rec.ToolCall(call.Name, effect, json.RawMessage(call.Input))
res := m.tools.Dispatch(ctx, tools.Context{
Principal: input.Identity,
RunID: rec.RunID(),
RemainingTokens: budget.Snapshot().TokensLeft,
Confirmation: input.Confirmation,
// From the spec, never from the call. A model that asked to search
// a corpus its agent was not granted is asking for a source list it
// has no way to set.
KnowledgeSources: agent.KnowledgeSources,
}, call.Name, call.Input)
// A pending confirmation never reaches the model. It is a question for
// a person, and handing it back as a tool result would invite the model
// to reason about it — to explain why it should be approved, or to try
// a different tool that might not ask. Neither is its business.
if res.Confirmation != nil {
rec.Confirmation(call.Name, res.Confirmation)
pending = append(pending, res.Confirmation)
continue
}
rec.ToolResult(call.Name, effect, res.Error != nil, res)
encoded, err := json.Marshal(res)
if err != nil {
encoded = []byte(`{"error":{"code":"tool.failed","message":"the result could not be encoded"}}`)
}
results = append(results, gateway.ToolResult{
CallID: call.ID,
Content: string(encoded),
IsError: res.Error != nil,
})
}
return results, pending, ""
}
// terminationFor maps a failure to the reason a run ends with.
//
// The mapping matters more than it looks: Deadline and BudgetExceeded are
// different questions to an operator ("too slow" versus "too expensive"), and
// a Refused run is one that must not be retried. Flattening them into a single
// failure reason would make every one of those distinctions unanswerable from
// the trajectory.
func terminationFor(err error) Termination {
var gwErr *gateway.Error
if !errors.As(err, &gwErr) {
return TerminationToolFailure
}
switch gwErr.Code {
case gateway.CodeRefused:
return TerminationRefused
case gateway.CodeTimeout:
return TerminationDeadline
default:
return TerminationToolFailure
}
}
// finish closes the trajectory, persists it, and builds the caller's result.
//
// Persistence uses the *caller's* context, not the run's: the run context is
// cancelled at the deadline, and a run that ended by running out of time is
// exactly the one whose record is most worth keeping.
func (m *ModelExecutor) finish(
ctx context.Context,
rec *Recorder,
budget *Budget,
term Termination,
agent *Agent,
skillIDs []string,
output string,
cause error,
) (*ExecutionResult, error) {
if cause != nil {
var gwErr *gateway.Error
if errors.As(cause, &gwErr) {
rec.Error(gwErr.Code, gwErr.Message)
} else {
rec.Error("runtime.failed", cause.Error())
}
}
rec.Budget(budget.Snapshot())
traj := rec.Finish(term)
// A sink that fails must not fail the run — the answer was already
// produced. It is recorded in the trajectory we could not save, which is
// the best available place for it.
if err := m.sink.Save(ctx, traj); err != nil {
rec.Error("runtime.trajectory_unsaved", err.Error())
}
res := &ExecutionResult{
Success: term == TerminationCompleted,
Output: output,
AgentID: agent.ID,
AgentVersion: agent.Version,
ResolvedSkills: skillIDs,
RunID: traj.RunID,
Termination: term,
Usage: traj.Usage,
}
if term == TerminationCompleted {
return res, nil
}
// A bounded run is not an exception. The caller gets a result carrying the
// reason; the error exists so a Go caller that ignores the result still
// notices, and it is structured so the surface layer derives the wording.
rtErr := &RuntimeError{
Code: "runtime." + strings.ToLower(string(term)),
Message: terminationMessage(term),
Target: agent.ID,
Cause: cause,
}
res.Error = rtErr
return res, rtErr
}
// terminationMessage is the internal explanation for a termination. Not
// user-facing copy — §10 puts that at the surface layer, which is free to say
// something kinder using the code.
func terminationMessage(t Termination) string {
switch t {
case TerminationBudgetExceeded:
return "the run reached its budget before finishing"
case TerminationDeadline:
return "the run reached its deadline before finishing"
case TerminationRefused:
return "the model declined to answer"
case TerminationConfirmationPending:
return "the run is waiting on a confirmation"
case TerminationToolFailure:
return "the run failed"
default:
return string(t)
}
}
// SystemPrompt assembles an agent's system prompt from its spec.
//
// I7 is the whole design of this function. Retrieved document text, tool
// results and user messages are all untrusted, and none of them are reachable
// from here: it reads the agent record and nothing else. When Phase 2 adds
// retrieval, the retrieved chunks go into a delimited block in a *user*
// message — not into this string — and the standing instruction below is what
// makes that delimiter mean something.
func SystemPrompt(agent *Agent) string {
var b strings.Builder
b.WriteString("You are ")
b.WriteString(agent.Name)
if agent.Description != "" {
b.WriteString(", ")
b.WriteString(agent.Description)
}
b.WriteString(".\n\n")
if instructions := strings.TrimSpace(agent.Instructions); instructions != "" {
b.WriteString(instructions)
b.WriteString("\n\n")
}
if len(agent.Pages) > 0 {
b.WriteString("You answer on: ")
b.WriteString(strings.Join(agent.Pages, ", "))
b.WriteString(". Anywhere else, say plainly that you do not cover it.\n\n")
}
// Stated even when nothing was retrieved, because the boundary has to be
// established before content arrives rather than alongside it.
//
// The sentence comes from the knowledge package, beside the renderer that
// emits the fence. A prompt promising <context> while the renderer wrote
// <documents> would be a defence that had quietly stopped existing, and two
// copies of a string in two packages is exactly how that happens.
b.WriteString(knowledge.ContextInstruction)
b.WriteString("\n\n")
b.WriteString("State a figure only where the records you were given show it. " +
"When you cannot answer from them, say so rather than estimating.")
return b.String()
}

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,143 @@
package runtime
import (
"context"
"encoding/json"
"errors"
"strings"
"time"
"github.com/jackc/pgx/v5"
"github.com/krow/krow-backend/go-api/internal/authctx"
"github.com/krow/krow-backend/go-api/internal/domain"
"github.com/krow/krow-backend/go-api/internal/repo"
)
// Reading a recorded run back.
//
// The write side of trajectories is PostgresSink; this is the read side, and it
// exists because §6's requirement is only worth anything if somebody can look.
// "Why did the agent say that" should be answerable by pointing at a run id.
//
// Two rules shape what comes back:
//
// - **Tenant scope is in the query.** I5. A run id from another organization
// is absent rather than forbidden, so it answers 404 and cannot be used to
// discover that a given run exists somewhere else.
// - **A talent caller sees only their own runs.** An operator sees the
// organization's, which is what an operator console is. A trajectory
// contains the caller's question and the records retrieved for them, so
// "anyone in the tenant may read any run" would be a much larger grant than
// it looks.
// RunReader loads recorded trajectories.
type RunReader struct {
db repo.Querier
}
// NewRunReader builds a reader over a pool or transaction.
func NewRunReader(db repo.Querier) *RunReader { return &RunReader{db: db} }
// RunView is a trajectory as a caller sees it.
//
// Not the Trajectory struct. That one is the internal record and gains fields
// as the runtime does; this is a response shape, and the difference is what
// keeps a new internal field from silently becoming a new public one.
type RunView struct {
RunID string `json:"runId"`
ParentRunID string `json:"parentRunId,omitempty"`
AgentID string `json:"agentId"`
AgentVersion int `json:"agentVersion"`
Tier string `json:"tier"`
Model string `json:"model,omitempty"`
StartedAt time.Time `json:"startedAt"`
EndedAt time.Time `json:"endedAt"`
Termination string `json:"termination"`
Entries []Entry `json:"entries"`
Usage RunUsage `json:"usage"`
}
// Load returns one run, if this caller may read it.
func (r *RunReader) Load(ctx context.Context, ident authctx.Identity, runID string) (*RunView, error) {
if strings.TrimSpace(runID) == "" {
return nil, domain.NotFound("run", runID)
}
if strings.TrimSpace(ident.OrgID) == "" {
// I5. No tenant, no read — and answered as absent rather than
// forbidden, on the same terms as every other row in this service.
return nil, domain.NotFound("run", runID)
}
where, args := runScope(ident, runID)
var (
view RunView
parent *string
model *string
rawEntries []byte
termination string
)
err := r.db.QueryRow(ctx, `
SELECT run_id, parent_run_id, agent_id, agent_version, tier, model,
started_at, ended_at, termination, entries,
input_tokens, output_tokens, cached_tokens, total_tokens, model_calls
FROM agent_runs
WHERE `+where, args...,
).Scan(&view.RunID, &parent, &view.AgentID, &view.AgentVersion, &view.Tier, &model,
&view.StartedAt, &view.EndedAt, &termination, &rawEntries,
&view.Usage.InputTokens, &view.Usage.OutputTokens, &view.Usage.CachedTokens,
&view.Usage.TotalTokens, &view.Usage.ModelCalls)
if err != nil {
if errors.Is(err, pgx.ErrNoRows) {
return nil, domain.NotFound("run", runID)
}
return nil, domain.Internal(err)
}
view.Termination = termination
if parent != nil {
view.ParentRunID = *parent
}
if model != nil {
view.Model = *model
}
if len(rawEntries) > 0 {
if err := json.Unmarshal(rawEntries, &view.Entries); err != nil {
return nil, domain.Internal(err)
}
}
if view.Entries == nil {
view.Entries = []Entry{}
}
return &view, nil
}
// runScope builds the predicate a caller's runs are behind.
//
// Two conditions, and the second is the one that is easy to forget. Tenancy is
// obvious. The talent restriction is not: a trajectory holds the question that
// was asked and the records retrieved to answer it, so a tenant-wide read would
// let any worker read every colleague's conversation with an agent — including
// the ones about them.
//
// An unrecognised role gets `false`, so it matches nothing rather than
// everything. The safe direction, and loud enough to find.
func runScope(ident authctx.Identity, runID string) (string, []any) {
args := []any{ident.OrgID, runID}
where := "org_id = $1::uuid AND run_id = $2"
role, ok := domain.ParseRole(ident.Role)
if !ok {
return where + " AND false", args
}
if role == domain.RoleTalent {
if strings.TrimSpace(ident.UserID) == "" {
return where + " AND false", args
}
args = append(args, ident.UserID)
where += " AND user_id = $3::uuid"
}
return where, args
}

View File

@@ -4,6 +4,7 @@ import (
"context"
"errors"
"fmt"
"strings"
"testing"
"github.com/krow/krow-backend/go-api/internal/authctx"
@@ -888,3 +889,146 @@ pages:
t.Errorf("custom skill output mismatch: %+v", resCustom)
}
}
/* ── Pinned versions ────────────────────────────────────────────────────── */
// TestAPinnedRunUsesTheAgentAsItWas.
//
// §3: "Running conversations pin the version they started with."
//
// The reason this matters is not tidiness. A person approves a write while
// looking at version 1; somebody publishes version 2 with different
// instructions and a different tool list; the approval is then carried out. If
// the run silently moved to version 2, the thing performed would not be the
// thing that was shown — which is the failure the whole confirmation mechanism
// exists to prevent, arriving through the registry instead of through the gate.
func TestAPinnedRunUsesTheAgentAsItWas(t *testing.T) {
f := newFixture(t)
ctx := context.Background()
v1 := `---
id: pinned-agent
name: Pinned Agent
status: published
version: 1
pages:
- activity
tools:
- activity_breakdown
---
## Instructions
Version one instructions.
`
v2 := `---
id: pinned-agent
name: Pinned Agent
status: published
version: 2
pages:
- activity
tools:
- activity_breakdown
- activity_signals
---
## Instructions
Version two instructions.
`
if _, err := f.h.Pool.Exec(ctx, `
INSERT INTO agent_definitions (definition_id, org_id, visibility, created_by,
markdown, status, version, name, description, pages)
VALUES ('pinned-agent', $1::uuid, 'organization', $2::uuid, $3::text,
'published', 2, 'Pinned Agent', '', ARRAY['activity'])`,
f.org1, f.userA.UserID, v2); err != nil {
t.Fatalf("seed current definition: %v", err)
}
versions := repo.NewVersionsRepo(f.h.Pool)
for _, s := range []repo.SnapshotInput{
{Kind: repo.KindAgent, DefinitionID: "pinned-agent", Version: 1, Markdown: v1, Name: "Pinned Agent"},
{Kind: repo.KindAgent, DefinitionID: "pinned-agent", Version: 2, Markdown: v2, Name: "Pinned Agent"},
} {
if err := versions.Snapshot(ctx, f.userA, s); err != nil {
t.Fatalf("snapshot v%d: %v", s.Version, err)
}
}
// Unpinned: the current definition, which is version 2.
current, fellBack, err := f.loader.LoadAgentVersion(ctx, f.userA, "pinned-agent", 0)
if err != nil {
t.Fatalf("load current: %v", err)
}
if fellBack {
t.Error("an unpinned load reported a fallback")
}
if current.Version != 2 || !strings.Contains(current.Instructions, "Version two") {
t.Errorf("current is v%d: %q", current.Version, current.Instructions)
}
if len(current.Tools) != 2 {
t.Errorf("v2 should carry 2 tools, got %v", current.Tools)
}
// Pinned to 1: the agent as it was, INCLUDING its narrower tool list. That
// last part is the one that would let an approved write reach a tool the
// approver's version never offered.
pinned, fellBack, err := f.loader.LoadAgentVersion(ctx, f.userA, "pinned-agent", 1)
if err != nil {
t.Fatalf("load v1: %v", err)
}
if fellBack {
t.Error("a version that exists reported a fallback")
}
if pinned.Version != 1 {
t.Errorf("pinned version = %d, want 1", pinned.Version)
}
if !strings.Contains(pinned.Instructions, "Version one") {
t.Errorf("pinned instructions are v2's: %q", pinned.Instructions)
}
if len(pinned.Tools) != 1 || pinned.Tools[0] != "activity_breakdown" {
t.Errorf("pinned tools are %v; v1 offered only activity_breakdown", pinned.Tools)
}
}
func TestAVersionWithNoSnapshotFallsBackAndSaysSo(t *testing.T) {
// Every definition published before versions were recorded has no snapshot.
// Refusing those would break every conversation that predates the feature,
// to enforce a rule they could not have followed. The run continues on the
// current definition and the fallback is REPORTED, because a silent
// substitution is the thing worth preventing.
f := newFixture(t)
ctx := context.Background()
md := `---
id: unversioned-agent
name: Unversioned
status: published
version: 1
pages:
- activity
---
## Instructions
Only ever existed as one thing.
`
if _, err := f.h.Pool.Exec(ctx, `
INSERT INTO agent_definitions (definition_id, org_id, visibility, created_by,
markdown, status, version, name, description, pages)
VALUES ('unversioned-agent', $1::uuid, 'organization', $2::uuid, $3::text,
'published', 1, 'Unversioned', '', ARRAY['activity'])`,
f.org1, f.userA.UserID, md); err != nil {
t.Fatalf("seed: %v", err)
}
agent, fellBack, err := f.loader.LoadAgentVersion(ctx, f.userA, "unversioned-agent", 7)
if err != nil {
t.Fatalf("a missing snapshot failed the load: %v", err)
}
if !fellBack {
t.Error("a version with no snapshot did not report a fallback")
}
if agent == nil || agent.Version != 1 {
t.Errorf("the fallback did not return the current definition: %+v", agent)
}
}

View File

@@ -0,0 +1,111 @@
package runtime
import (
"context"
"encoding/json"
"fmt"
"github.com/krow/krow-backend/go-api/internal/repo"
)
// PostgresSink writes trajectories to agent_runs.
//
// Hand-written rather than built through repo.Repo's descriptor machinery, and
// deliberately so: that layer exists to serve the generic CRUD contract in
// docs/api-contract.md — filters, sorts, pagination, resource descriptors —
// and a trajectory has none of those. It is written once, whole, by the
// runtime, and read back by run id. One INSERT with explicit bind parameters
// is the honest shape for that, and inventing a resource descriptor to reach
// it would add a layer that only ever gets used one way.
type PostgresSink struct {
db repo.Querier
}
// NewPostgresSink builds a sink over a pool or transaction.
func NewPostgresSink(db repo.Querier) *PostgresSink {
return &PostgresSink{db: db}
}
var _ Sink = (*PostgresSink)(nil)
const insertRunSQL = `
INSERT INTO agent_runs (
run_id, parent_run_id, org_id, user_id,
agent_id, agent_version, tier, model,
started_at, ended_at, termination, entries,
input_tokens, output_tokens, cached_tokens, total_tokens, model_calls
) VALUES (
$1, $2, $3::uuid, $4::uuid,
$5, $6, $7, $8,
$9, $10, $11, $12::jsonb,
$13, $14, $15, $16, $17
)
ON CONFLICT (run_id) DO NOTHING`
// Save writes one finished trajectory.
//
// ON CONFLICT DO NOTHING because a run id is generated once and written once:
// a conflict means a retry of a save that already landed, and the first write
// is the authoritative one. Failing the second attempt would turn a harmless
// duplicate into a lost answer, since the caller treats a save error as
// something to report.
func (s *PostgresSink) Save(ctx context.Context, t *Trajectory) error {
if t == nil {
return fmt.Errorf("runtime: no trajectory to save")
}
if t.RunID == "" {
return fmt.Errorf("runtime: a trajectory needs a run id")
}
if !t.Termination.Valid() {
return fmt.Errorf("runtime: %q is not a termination reason", t.Termination)
}
// The table's tenancy column is NOT NULL, and a run with no organization is
// a bug upstream rather than a row to write. Caught here so the failure
// names the cause instead of surfacing as a constraint violation.
if t.OrgID == "" {
return fmt.Errorf("runtime: a trajectory needs an org id")
}
entries, err := json.Marshal(t.Entries)
if err != nil {
return fmt.Errorf("runtime: encoding trajectory entries: %w", err)
}
// A nil slice marshals to "null", which the jsonb_typeof CHECK refuses.
// A run that recorded nothing is still a run worth keeping.
if len(t.Entries) == 0 {
entries = []byte("[]")
}
_, err = s.db.Exec(ctx, insertRunSQL,
t.RunID,
nullIfEmpty(t.ParentRunID),
t.OrgID,
nullIfEmpty(t.UserID),
t.AgentID,
t.AgentVersion,
t.Tier,
t.Model,
t.StartedAt,
t.EndedAt,
string(t.Termination),
string(entries),
t.Usage.InputTokens,
t.Usage.OutputTokens,
t.Usage.CachedTokens,
t.Usage.TotalTokens,
t.Usage.ModelCalls,
)
if err != nil {
return fmt.Errorf("runtime: saving trajectory %s: %w", t.RunID, err)
}
return nil
}
// nullIfEmpty keeps an empty optional out of a uuid column, where "" is not a
// value the type accepts.
func nullIfEmpty(s string) any {
if s == "" {
return nil
}
return s
}

View File

@@ -0,0 +1,164 @@
package runtime_test
import (
"context"
"encoding/json"
"testing"
"time"
"github.com/krow/krow-backend/go-api/internal/runtime"
"github.com/krow/krow-backend/go-api/internal/testutil"
)
// trajectory builds a saveable run for the given harness org.
func trajectory(orgID, runID string) *runtime.Trajectory {
started := time.Now().Add(-2 * time.Second).UTC()
return &runtime.Trajectory{
RunID: runID,
OrgID: orgID,
AgentID: "activity-agent",
AgentVersion: 3,
Tier: "balanced",
Model: "claude-opus-5",
StartedAt: started,
EndedAt: started.Add(1200 * time.Millisecond),
Termination: runtime.TerminationCompleted,
Entries: []runtime.Entry{
{Seq: 1, At: started, Kind: runtime.EntryMessage, Role: "user", Text: "what happened?"},
{Seq: 2, At: started, Kind: runtime.EntryBudget, Budget: &runtime.Snapshot{StepsLeft: 8, TokensLeft: 120000}},
{Seq: 3, At: started, Kind: runtime.EntryMessage, Role: "assistant", Text: "Twelve events."},
},
Usage: runtime.RunUsage{InputTokens: 900, OutputTokens: 120, TotalTokens: 1020, ModelCalls: 1},
}
}
func TestPostgresSinkSavesAndReadsBack(t *testing.T) {
h := testutil.New(t)
ctx := context.Background()
sink := runtime.NewPostgresSink(h.Pool)
traj := trajectory(h.OrgID, "run_store_basic")
if err := sink.Save(ctx, traj); err != nil {
t.Fatalf("save: %v", err)
}
var (
agentID, tier, model, termination string
version, modelCalls int
total int64
entries []byte
)
err := h.Pool.QueryRow(ctx, `
SELECT agent_id, agent_version, tier, model, termination, total_tokens, model_calls, entries
FROM agent_runs WHERE run_id = $1`, traj.RunID,
).Scan(&agentID, &version, &tier, &model, &termination, &total, &modelCalls, &entries)
if err != nil {
t.Fatalf("read back: %v", err)
}
if agentID != "activity-agent" || version != 3 {
t.Errorf("stored %s v%d, want activity-agent v3", agentID, version)
}
if termination != string(runtime.TerminationCompleted) {
t.Errorf("termination = %q, want Completed", termination)
}
// Both the tier asked for and the model that answered, so a trajectory read
// a year later does not require knowing that week's routing.
if tier != "balanced" || model != "claude-opus-5" {
t.Errorf("tier/model = %q/%q, want balanced/claude-opus-5", tier, model)
}
if total != 1020 || modelCalls != 1 {
t.Errorf("usage = %d tokens over %d calls, want 1020/1", total, modelCalls)
}
var round []runtime.Entry
if err := json.Unmarshal(entries, &round); err != nil {
t.Fatalf("entries did not round-trip: %v", err)
}
if len(round) != 3 || round[0].Role != "user" || round[2].Text != "Twelve events." {
t.Errorf("entries round-tripped as %+v", round)
}
}
func TestPostgresSinkIsIdempotentPerRun(t *testing.T) {
// A run id is generated once and written once. A second save is a retry of
// one that already landed — failing it would turn a harmless duplicate
// into a reported error on a run that succeeded.
h := testutil.New(t)
ctx := context.Background()
sink := runtime.NewPostgresSink(h.Pool)
traj := trajectory(h.OrgID, "run_store_twice")
if err := sink.Save(ctx, traj); err != nil {
t.Fatalf("first save: %v", err)
}
if err := sink.Save(ctx, traj); err != nil {
t.Fatalf("second save should be a no-op, got: %v", err)
}
var n int
if err := h.Pool.QueryRow(ctx,
`SELECT count(*) FROM agent_runs WHERE run_id = $1`, traj.RunID).Scan(&n); err != nil {
t.Fatalf("count: %v", err)
}
if n != 1 {
t.Errorf("%d rows for one run id, want 1", n)
}
}
func TestPostgresSinkRefusesRunsItCannotAttribute(t *testing.T) {
h := testutil.New(t)
ctx := context.Background()
sink := runtime.NewPostgresSink(h.Pool)
cases := map[string]*runtime.Trajectory{
"no run id": func() *runtime.Trajectory {
tr := trajectory(h.OrgID, "")
return tr
}(),
// I5: tenancy is not optional. Caught here so the failure names the
// cause rather than surfacing as a NOT NULL violation.
"no org": func() *runtime.Trajectory {
tr := trajectory("", "run_no_org")
return tr
}(),
// An invented termination must never reach the column that evals and
// dashboards group by.
"invented termination": func() *runtime.Trajectory {
tr := trajectory(h.OrgID, "run_bad_term")
tr.Termination = "Finished"
return tr
}(),
}
for name, tr := range cases {
if err := sink.Save(ctx, tr); err == nil {
t.Errorf("%s: save should have been refused", name)
}
}
}
func TestPostgresSinkStoresAnEmptyTrajectory(t *testing.T) {
// A run that recorded nothing is still a run worth keeping, and a nil
// slice marshals to "null", which the jsonb_typeof CHECK refuses.
h := testutil.New(t)
ctx := context.Background()
sink := runtime.NewPostgresSink(h.Pool)
traj := trajectory(h.OrgID, "run_store_empty")
traj.Entries = nil
traj.Termination = runtime.TerminationBudgetExceeded
if err := sink.Save(ctx, traj); err != nil {
t.Fatalf("save: %v", err)
}
var kind string
if err := h.Pool.QueryRow(ctx,
`SELECT jsonb_typeof(entries) FROM agent_runs WHERE run_id = $1`, traj.RunID).Scan(&kind); err != nil {
t.Fatalf("read back: %v", err)
}
if kind != "array" {
t.Errorf("entries stored as %q, want array", kind)
}
}

View File

@@ -0,0 +1,271 @@
package runtime
import (
"context"
"sync"
"time"
"github.com/krow/krow-backend/go-api/internal/knowledge"
)
// EntryKind is what one line of a trajectory records.
type EntryKind string
const (
EntryMessage EntryKind = "message"
EntryToolCall EntryKind = "tool_call"
EntryToolResult EntryKind = "tool_result"
EntryBudget EntryKind = "budget"
EntryError EntryKind = "error"
// EntryConfirmation is a write that was described and not performed. It is
// recorded because "what was this person asked to approve, and when" is the
// question an audit of an agent-initiated write actually asks.
EntryConfirmation EntryKind = "confirmation"
// EntryRetrieval is what the knowledge layer returned for this run. The
// chunk ids are recorded, never the chunk text: a trajectory is already the
// most sensitive row in the database, and duplicating the corpus into it
// would mean a retention policy on runs quietly became a retention policy on
// every document too.
EntryRetrieval EntryKind = "retrieval"
)
// Entry is one recorded moment in a run.
//
// Deliberately flat and deliberately typed as data rather than prose: an eval
// asserts on `tools_called`, a debugger reads the budget line before the
// dispatch that overran, and neither can do that against a log string.
type Entry struct {
Seq int `json:"seq"`
At time.Time `json:"at"`
Kind EntryKind `json:"kind"`
Role string `json:"role,omitempty"`
Name string `json:"name,omitempty"`
Text string `json:"text,omitempty"`
Data any `json:"data,omitempty"`
Budget *Snapshot `json:"budget,omitempty"`
ErrorCode string `json:"errorCode,omitempty"`
// Effect is the tool's declared effect, on tool_call and tool_result
// entries. Recorded because "did this run change anything" is not answerable
// from a tool name — an eval reading the trajectory would otherwise have to
// keep its own list of which tools write, and that list would go stale on
// the first tool anyone added.
//
// It records what the RUNTIME BELIEVED, which is what the gate acted on. A
// tool that declares itself a read and writes anyway is invisible here, and
// is meant to be: the trajectory cannot be the check on a tool lying about
// itself. That check is the database.
Effect string `json:"effect,omitempty"`
// Failed says a tool result carried an error. A refusal is recorded like
// any other result, and an eval that could not tell the two apart would
// read every denial as a successful call.
Failed bool `json:"failed,omitempty"`
}
// Trajectory is the full record of one run.
//
// §6 is explicit that this is not optional telemetry: it is what makes
// debugging and evals possible at all. A run whose trajectory was dropped
// because the sink was busy is a run nobody can explain afterwards, which is
// why recording never blocks on persistence — see Recorder.
type Trajectory struct {
RunID string `json:"runId"`
ParentRunID string `json:"parentRunId,omitempty"`
OrgID string `json:"orgId"`
UserID string `json:"userId"`
AgentID string `json:"agentId"`
AgentVersion int `json:"agentVersion"`
Tier string `json:"tier"`
Model string `json:"model,omitempty"`
StartedAt time.Time `json:"startedAt"`
EndedAt time.Time `json:"endedAt"`
Termination Termination `json:"termination"`
Entries []Entry `json:"entries"`
Usage RunUsage `json:"usage"`
}
// RunUsage is what a whole run cost, across every call it made.
type RunUsage struct {
InputTokens int64 `json:"inputTokens"`
OutputTokens int64 `json:"outputTokens"`
CachedTokens int64 `json:"cachedTokens"`
TotalTokens int64 `json:"totalTokens"`
ModelCalls int `json:"modelCalls"`
}
// Sink persists a finished trajectory.
//
// An interface with one method so the eval harness can hold runs in memory and
// the service can write them to Postgres without either knowing about the
// other. A sink that fails must not fail the run: the answer was already
// produced, and losing the record is worse than losing nothing but is not
// worth discarding a correct answer over.
type Sink interface {
Save(ctx context.Context, t *Trajectory) error
}
// DiscardSink drops trajectories. The default, so a service wired without a
// store still runs — and so tests that do not care about persistence say so by
// choosing it rather than by leaving a nil that panics.
type DiscardSink struct{}
func (DiscardSink) Save(context.Context, *Trajectory) error { return nil }
// MemorySink keeps trajectories in memory. For tests and the eval harness.
type MemorySink struct {
mu sync.Mutex
Runs []*Trajectory
}
func (m *MemorySink) Save(_ context.Context, t *Trajectory) error {
m.mu.Lock()
defer m.mu.Unlock()
m.Runs = append(m.Runs, t)
return nil
}
// Last returns the most recent trajectory, or nil.
func (m *MemorySink) Last() *Trajectory {
m.mu.Lock()
defer m.mu.Unlock()
if len(m.Runs) == 0 {
return nil
}
return m.Runs[len(m.Runs)-1]
}
// Recorder accumulates a trajectory during a run.
//
// Entries are held in memory and written once at the end rather than streamed
// per line. A run is short and bounded by construction — I3 guarantees it —
// so the whole record fits, and one write means a trajectory is either wholly
// there or wholly absent, never a half-run that reads as a run that stopped.
//
// Safe for concurrent use: a subagent records into its own recorder, but tool
// calls within one run may be dispatched in parallel.
type Recorder struct {
mu sync.Mutex
t *Trajectory
}
// NewRecorder begins recording a run.
func NewRecorder(t *Trajectory) *Recorder {
if t.StartedAt.IsZero() {
t.StartedAt = time.Now()
}
return &Recorder{t: t}
}
func (r *Recorder) append(e Entry) {
r.mu.Lock()
defer r.mu.Unlock()
e.Seq = len(r.t.Entries) + 1
if e.At.IsZero() {
e.At = time.Now()
}
r.t.Entries = append(r.t.Entries, e)
}
// RunID is the run being recorded, so a tool handler can be told which
// trajectory its call belongs to.
func (r *Recorder) RunID() string {
r.mu.Lock()
defer r.mu.Unlock()
return r.t.RunID
}
// Message records one turn of the conversation.
func (r *Recorder) Message(role, text string) {
r.append(Entry{Kind: EntryMessage, Role: role, Text: text})
}
// Budget records what was left before a dispatch.
//
// Called *before* the call it precedes, so the last budget line in a trajectory
// is the state that permitted the dispatch that ended the run — which is the
// line anyone debugging an overrun actually wants.
func (r *Recorder) Budget(s Snapshot) {
r.append(Entry{Kind: EntryBudget, Budget: &s})
}
// ToolCall records a dispatch to a tool.
func (r *Recorder) ToolCall(name, effect string, input any) {
r.append(Entry{Kind: EntryToolCall, Name: name, Effect: effect, Data: input})
}
// ToolResult records what a tool returned, and whether it failed.
func (r *Recorder) ToolResult(name, effect string, failed bool, output any) {
r.append(Entry{Kind: EntryToolResult, Name: name, Effect: effect, Failed: failed, Data: output})
}
// Retrieval records what the knowledge layer returned.
//
// Ids and ranks, not text. "Which chunks grounded this answer" is the question
// an eval and a debugger both ask, and it is answerable from ids alone — while
// copying the text in would make every trajectory a partial copy of the corpus,
// with all of the corpus's access rules and none of its retention.
func (r *Recorder) Retrieval(res *knowledge.Results) {
if res == nil {
return
}
cited := make([]map[string]any, 0, len(res.Chunks))
for _, c := range res.Chunks {
cited = append(cited, map[string]any{
"chunkId": c.ChunkID, "documentId": c.DocumentID,
"source": c.Source, "score": c.Score,
"denseRank": c.DenseRank, "sparseRank": c.SparseRank,
})
}
data := map[string]any{"chunks": cited, "tokens": res.TotalTokens}
if res.DenseSkipped != "" {
data["degraded"] = res.DenseSkipped
}
r.append(Entry{Kind: EntryRetrieval, Name: "knowledge", Data: data})
}
// Confirmation records a write that was described and is awaiting approval.
func (r *Recorder) Confirmation(name string, c any) {
r.append(Entry{Kind: EntryConfirmation, Name: name, Data: c})
}
// Error records a failure with the code that classified it.
func (r *Recorder) Error(code, message string) {
r.append(Entry{Kind: EntryError, ErrorCode: code, Text: message})
}
// ChargeUsage adds one model call's cost to the run total.
func (r *Recorder) ChargeUsage(input, output, cached int64) {
r.mu.Lock()
defer r.mu.Unlock()
r.t.Usage.InputTokens += input
r.t.Usage.OutputTokens += output
r.t.Usage.CachedTokens += cached
r.t.Usage.TotalTokens += input + output + cached
r.t.Usage.ModelCalls++
}
// SetModel records which model actually answered, as opposed to the tier that
// was requested.
func (r *Recorder) SetModel(model string) {
r.mu.Lock()
defer r.mu.Unlock()
r.t.Model = model
}
// Finish closes the trajectory with its termination reason and returns it.
//
// A reason that is not one of the six is recorded as ToolFailure rather than
// stored as-is: an unrecognised termination is a bug in the loop, and writing
// it verbatim would let that bug propagate into every eval and dashboard that
// groups by this column.
func (r *Recorder) Finish(t Termination) *Trajectory {
r.mu.Lock()
defer r.mu.Unlock()
if !t.Valid() {
t = TerminationToolFailure
}
r.t.Termination = t
r.t.EndedAt = time.Now()
return r.t
}

View File

@@ -5,6 +5,7 @@ import (
"fmt"
"github.com/krow/krow-backend/go-api/internal/authctx"
"github.com/krow/krow-backend/go-api/internal/tools"
)
// Standard runtime errors.
@@ -24,24 +25,50 @@ var (
// Agent represents an authored agent prepared for runtime execution.
type Agent struct {
ID string `json:"id"`
DatabaseID string `json:"databaseId"`
Name string `json:"name"`
Description string `json:"description"`
Status string `json:"status"`
Version int `json:"version"`
Visibility string `json:"visibility"`
OwnerUserID *string `json:"ownerUserId,omitempty"`
Pages []string `json:"pages"`
Icon string `json:"icon,omitempty"`
Reasoning string `json:"reasoning,omitempty"`
Trigger string `json:"trigger,omitempty"`
WebSearch bool `json:"webSearch,omitempty"`
Instructions string `json:"instructions"`
Skills []string `json:"skills"`
ResolvedSkills []*Skill `json:"resolvedSkills,omitempty"`
Subagents []string `json:"subagents,omitempty"`
RawMarkdown string `json:"rawMarkdown"`
ID string `json:"id"`
DatabaseID string `json:"databaseId"`
Name string `json:"name"`
Description string `json:"description"`
Status string `json:"status"`
Version int `json:"version"`
Visibility string `json:"visibility"`
OwnerUserID *string `json:"ownerUserId,omitempty"`
Pages []string `json:"pages"`
Icon string `json:"icon,omitempty"`
Reasoning string `json:"reasoning,omitempty"`
Trigger string `json:"trigger,omitempty"`
WebSearch bool `json:"webSearch,omitempty"`
Instructions string `json:"instructions"`
Skills []string `json:"skills"`
// Tools this agent may call, by registry name. A name that resolves to
// nothing fails at publish (§3); one that reaches run time is recorded and
// dropped rather than taking the run with it.
Tools []string `json:"tools,omitempty"`
// KnowledgeSources are the corpora this agent may retrieve from, by source
// name. Empty means this agent has no knowledge — NOT that it may read
// everything. Retrieval refuses an empty source list for exactly that
// reason.
//
// NAMED `sources:` IN A SPEC, NOT `knowledge:`, AND THAT IS A DEVIATION.
// §3's spec contract calls this block `knowledge:`. The shipped product got
// there first and uses `knowledge:` for something else entirely — an
// author's notes to the agent, free text, no retrieval involved (see
// definition.Agent.Knowledge). Two different things under one key would be
// resolved wrongly by whichever parser ran second, silently, so the
// retrieval block is `sources:` until somebody decides which name wins.
// Flagged rather than settled: §12 says not to resolve a schema question
// unilaterally.
//
// A list of names rather than scope templates, for now. §3's
// `scope: "venue:{caller.venue_ids}"` resolves at run time against the
// caller, and the ACL tag on each chunk already carries the resolved
// version of that decision — see knowledge/acl.go. Templates become
// necessary when a source needs a narrower slice than its own tags express.
KnowledgeSources []string `json:"knowledgeSources,omitempty"`
ResolvedSkills []*Skill `json:"resolvedSkills,omitempty"`
Subagents []string `json:"subagents,omitempty"`
RawMarkdown string `json:"rawMarkdown"`
}
// Skill represents an authored skill prepared for runtime execution.
@@ -71,6 +98,45 @@ type ExecutionInput struct {
Input string `json:"input"`
Parameters map[string]any `json:"parameters,omitempty"`
Context map[string]any `json:"context,omitempty"`
// Notes are things the runtime should record about this run before it
// starts — a version that could not be pinned, a capability that was asked
// for and is not configured.
//
// A dedicated field rather than a key smuggled into Context. Context is
// opaque client state and nothing reads it, so a note put there is a note
// nobody sees — which is exactly what happened on the first attempt: the
// fallback was "recorded" into a map the trajectory never touches, and the
// only thing that noticed was a test looking for it.
Notes []string `json:"notes,omitempty"`
// AgentVersion pins the run to a published version of the agent.
//
// Zero means "whatever is current", which is what an ordinary question
// wants. A RESUMED run should pin, and that is the whole reason this
// exists: a person approved a write while looking at version 3, and
// carrying it out under version 4's tool list would perform something they
// were never shown. §3 puts it as "running conversations pin the version
// they started with".
AgentVersion int `json:"agentVersion,omitempty"`
// OnDelta receives assistant text as it arrives, if the caller wants it
// streamed. Nil for a caller that only wants the finished answer, which is
// every eval and every test — streaming is a delivery choice, not a
// different kind of run.
//
// Called from the run's own goroutine, in order. A slow handler here sits
// directly between the model and the reader.
OnDelta func(string) `json:"-"`
// Confirmation is a token a person approved, carried into a resumed run.
//
// It authorises ONE call — the exact tool, arguments, caller and run it was
// issued against — and nothing else. Supplying it does not put the run into
// a permissive mode: a second write in the same run raises its own
// confirmation, because a person approved one thing and only one thing.
// See tools/confirm.go.
Confirmation string `json:"confirmation,omitempty"`
}
// ExecutionResult captures the outcome of an execution attempt.
@@ -81,6 +147,26 @@ type ExecutionResult struct {
AgentVersion int `json:"agentVersion,omitempty"`
ResolvedSkills []string `json:"resolvedSkills,omitempty"`
Error error `json:"error,omitempty"`
// RunID addresses the trajectory this run wrote. Returned to the caller so
// a conversation about a bad answer has something to point at.
RunID string `json:"runId,omitempty"`
// Termination is why the run ended — exactly one of the six, always set by
// the loop. Empty only on results built by Engine's pre-execution failure
// paths, where no run was ever started.
Termination Termination `json:"termination,omitempty"`
// Usage is what the run cost across every model call it made.
Usage RunUsage `json:"usage"`
// Confirmations are writes the run described but did not perform. Present
// exactly when Termination is ConfirmationPending, and the reason that is a
// termination rather than an error: nothing failed and nothing happened —
// the run is waiting on a person. The surface renders these, and returns
// the token of whichever the person approves as ExecutionInput.Confirmation
// on the next call.
Confirmations []*tools.Confirmation `json:"confirmations,omitempty"`
}
// RuntimeError is a structured error containing context for execution failures.

View File

@@ -0,0 +1,157 @@
package runtime
import (
"github.com/krow/krow-backend/go-api/internal/config"
"github.com/krow/krow-backend/go-api/internal/gateway"
"github.com/krow/krow-backend/go-api/internal/knowledge"
"github.com/krow/krow-backend/go-api/internal/repo"
"github.com/krow/krow-backend/go-api/internal/tools"
)
// NewModelEngine builds the production runtime: the loader, the agent loop, a
// live model gateway and a Postgres trajectory sink.
//
// One call, because the alternative is four, and four assembled at a call site
// is how a deployment ends up running with a DiscardSink nobody chose. A test
// that wants a fake model still reaches for NewEngine with WithAgentExecutor —
// this function is the wiring, not a second way to configure the runtime.
//
// Skills keep the refusing stub. A skill has no executor of its own: the loop
// runs agents, and a skill reaches a model only as a capability an agent
// carries. Handing SkillExec a model would create a second, unbounded path to
// one — which is exactly the shape I3 exists to prevent.
func NewModelEngine(db repo.Querier, cfg config.Config) *Engine {
gw := gateway.NewAnthropic(gateway.FromConfig(cfg.Model))
retriever := knowledge.NewRetriever(db, NewEmbedder(cfg))
exec := NewModelExecutor(gw, NewPostgresSink(db), DefaultTools(db, retriever)).
WithRetriever(retriever)
return NewEngine(db, WithAgentExecutor(exec))
}
// NewEmbedder picks the embedding provider from configuration.
//
// Explicit first, then what is configured, then nothing. The order is the whole
// design: three providers all return vectors and retrieval works with any of
// them, so a deployment running the wrong one looks identical to one running
// the right one until somebody phrases a question differently. Naming the
// provider is how that stops being a silent condition.
//
// ollama A model on this machine. Real semantics, no credential, no
// per-token cost, no tenant text leaving the host. The default
// worth reaching for.
// voyage Hosted. Better on subtle retrieval over a large messy corpus,
// and the only one that needs a credential.
// lexical The deterministic stand-in. NOT semantic — it matches shared
// vocabulary and nothing else. Development only; config.validate
// refuses it in production.
//
// Returns nil when nothing is configured, and retrieval then runs keyword-only,
// saying so on every result. Nil rather than a hosted client with an empty key:
// both end up keyword-only, but nil says "no embedder is configured" once, at
// wiring time, instead of failing an HTTP call per query to learn the same
// thing.
func NewEmbedder(cfg config.Config) knowledge.Embedder {
k := cfg.Knowledge
provider := k.EmbedProvider
if provider == "" {
// Nothing named. Infer from what is actually present, preferring the
// one that costs nothing and keeps text local.
switch {
case k.UseLexicalEmbedder:
provider = "lexical"
case k.EmbedBaseURL != "":
provider = "ollama"
case k.EmbedAPIKey != "":
provider = "voyage"
default:
return nil
}
}
switch provider {
case "ollama":
return knowledge.NewOllama(k.EmbedBaseURL, k.EmbedModel, k.EmbedDims)
case "voyage":
if k.EmbedAPIKey == "" {
// Named but unusable. Nil, so retrieval degrades honestly rather
// than failing a request per query on a credential nobody set.
return nil
}
model, dims := k.EmbedModel, k.EmbedDims
if model == "" {
model = knowledge.DefaultVoyageModel
}
if dims == 0 {
dims = knowledge.DefaultVoyageDims
}
return knowledge.NewVoyage(k.EmbedAPIKey, model, dims)
case "lexical":
dims := k.EmbedDims
if dims == 0 {
dims = 256
}
e := knowledge.NewLexical(dims)
// Told what environment it is in, so its own refusal is the backstop
// behind config.validate's.
e.Production = cfg.AppEnv == "production"
return e
}
return nil
}
// DefaultTools is the tool registry this service ships with.
//
// One function, so "which tools exist" has a single answer that a test and the
// server reach the same way. Registration panics on a malformed tool: a
// service that booted without a capability its specs name would fail one run
// at a time instead of once, loudly, at startup.
func DefaultTools(db repo.Querier, retriever *knowledge.Retriever) *tools.Registry {
// The confirmation store is Postgres-backed, not in-process. A pending
// write is asked about in one request and approved in another, and nothing
// guarantees those two reach the same replica — an in-memory store would
// refuse a large share of perfectly good approvals, for a reason invisible
// to the person clicking. See tools.MemoryStore's own warning.
reg := tools.NewRegistryWithStore(tools.NewPostgresStore(db))
for _, t := range []tools.Tool{
// Activity
tools.ActivityBreakdown(db),
tools.ActivitySignals(db),
// Workforce
tools.WorkforceAttendance(db),
tools.WorkforceOvertime(db),
tools.WorkforceCoverage(db),
tools.WorkforceTraining(db),
// Hiring
tools.CandidatesQuality(db),
tools.HiresRecent(db),
tools.HiresPerformance(db),
tools.PositionsRisk(db),
tools.TalentPool(db),
// Cross-domain
tools.WorkspaceSummary(db),
tools.OperationsRisk(db),
// Assignments: the two lookups that yield ids, and the one write that
// consumes them. assign_worker is the only tool here with an effect,
// and it cannot run without an approval — see tools/confirm.go.
tools.OpenPositions(db),
tools.AvailableWorkers(db),
tools.AssignWorker(db),
// The hiring funnel: the lookup that yields application ids, and the
// write that moves somebody through it. Replaces the browser panel's
// interview matcher, which was the one capability the old templates had
// that the tool layer did not.
tools.CandidatesAwaiting(db),
tools.MoveApplication(db),
// Knowledge. Registered once; which corpora it may read comes from the
// running agent's spec by way of the tool Context, so this single
// registration serves every agent without any of them being able to
// name another's documents.
tools.KnowledgeSearch(retriever),
} {
reg.MustRegister(t)
}
return reg
}