Files
doormile_backend/internal/ai/playground/playground.go

342 lines
11 KiB
Go

// Package playground runs one operator prompt through Claude with a registry
// skill's tools, for Agent Studio's Test tab (Phase 6 of
// krow_talent_app/docs/agent-platform-plan.md).
//
// The rules that make it safe to point at production:
// - read tools the backend can serve run for real, and their results are
// redacted (names, phones, addresses, emails, free text) before Claude sees
// them — see redact.go;
// - write, notify and event tools NEVER run: the call is answered with a
// proposal and shown in the trace as "proposed";
// - a tool outside the selected skill is refused;
// - the loop is bounded (MaxTurns) and so is every tool call (ToolTimeout).
//
// Claude is reached through the Model interface, so the loop is tested with a
// fake and the server wires in a real client only when one is configured.
package playground
import (
"context"
"encoding/json"
"errors"
"fmt"
"strings"
"time"
"doormile/internal/ai/registry"
)
// DefaultModel is used when the agent has no model pinned in the registry.
const DefaultModel = "claude-opus-5-5"
const (
MaxTurns = 6
MaxTokens = 4096
MaxPromptChars = 2000
ToolTimeout = 5 * time.Second
maxResultBytes = 16 * 1024
)
// Tool kinds, as the registry stores them.
const (
kindRead = "read"
)
// Outcomes of a tool call, as the trace shows them.
const (
OutcomeExecuted = "executed"
OutcomeProposed = "proposed"
OutcomeUnavailable = "unavailable"
OutcomeError = "error"
OutcomeRejected = "rejected"
)
// ── The model boundary ──────────────────────────────────────────────────────
// ToolUse is a tool call Claude asked for.
type ToolUse struct {
ID string
Name string
Input json.RawMessage
}
// Block is one content block of a reply. Text and tool_use are read by the
// loop; anything else (thinking) is carried in Raw and sent back unchanged,
// which the API requires within a tool-use turn.
type Block struct {
Type string
Text string
ToolUse *ToolUse
Raw json.RawMessage
}
// Reply is one model response.
type Reply struct {
Blocks []Block
StopReason string
InputTokens int64
OutputTokens int64
}
// ToolResult answers one ToolUse.
type ToolResult struct {
ToolUseID string
Content string
IsError bool
}
// Turn is one message of the conversation: the user's prompt, an assistant
// reply, or the user turn carrying tool results.
type Turn struct {
Role string // "user" or "assistant"
Text string
Assistant []Block
Results []ToolResult
}
// ToolDef is a tool as offered to the model.
type ToolDef struct {
Name string
Description string
InputSchema json.RawMessage
}
// Request is one model call.
type Request struct {
Model string
System string
MaxTokens int64
Tools []ToolDef
Turns []Turn
}
// Model is the one call the loop needs from Claude.
type Model interface {
Next(ctx context.Context, req Request) (Reply, error)
}
// ── Plan: what a run may use ────────────────────────────────────────────────
// ErrNotFound is returned when the agent or skill does not exist.
var ErrNotFound = errors.New("not found")
// Plan is the resolved agent, skill and tools for one run.
type Plan struct {
AgentID string
AgentName string
SkillID string
Model string
System string
Tools []ToolDef
kinds map[string]string
}
// Prepare resolves a run from the registry. skillID may be empty: the run then
// gets every tool of the agent's enabled skills.
func Prepare(snap *registry.Snapshot, agentID, skillID string) (*Plan, error) {
var agent *registry.AgentView
for i := range snap.Agents {
if snap.Agents[i].Agentid == agentID {
agent = &snap.Agents[i]
}
}
if agent == nil {
return nil, fmt.Errorf("agent %q: %w", agentID, ErrNotFound)
}
toolNames := map[string]bool{}
var skillLines []string
found := skillID == ""
for _, s := range snap.Skills {
if s.Agentid != agentID {
continue
}
if skillID != "" && s.Skillid != skillID {
continue
}
if skillID == "" && !s.Enabled {
continue
}
found = true
skillLines = append(skillLines, fmt.Sprintf("- %s: %s", s.Title, s.Description))
for _, t := range s.Tools {
toolNames[t] = true
}
}
if !found {
return nil, fmt.Errorf("skill %q of agent %q: %w", skillID, agentID, ErrNotFound)
}
plan := &Plan{
AgentID: agent.Agentid, AgentName: agent.Name, SkillID: skillID,
Model: agent.Model, kinds: map[string]string{}, Tools: []ToolDef{},
}
if plan.Model == "" {
plan.Model = DefaultModel
}
for _, t := range snap.Tools {
if !toolNames[t.Toolname] {
continue
}
desc := t.Description
if t.Kind != kindRead {
desc += " [Playground: NOT executed — calling it records a proposal for a human.]"
}
plan.Tools = append(plan.Tools, ToolDef{Name: t.Toolname, Description: desc, InputSchema: t.Inputschema})
plan.kinds[t.Toolname] = t.Kind
}
plan.System = strings.Join([]string{
fmt.Sprintf("You are %s, an agent in Doormile's delivery operations, being tested by an operator in the Agent Studio playground.", agent.Name),
"Purpose: " + agent.Purpose,
"Skills in scope:\n" + strings.Join(skillLines, "\n"),
"Read tools return live data with personal details (names, phones, addresses, notes) removed; do not ask for them.",
"Write, notify and event tools are not executed here: calling one records a proposal for a human to review. Say plainly what you would do and why.",
"If a tool is unavailable, say so rather than guessing its result. Keep the final answer short and concrete.",
}, "\n\n")
return plan, nil
}
// ── The run ─────────────────────────────────────────────────────────────────
// Executor runs one read tool. Its result is redacted before the model sees it.
type Executor func(ctx context.Context, input json.RawMessage) (any, error)
// Step is one line of the trace the console shows.
type Step struct {
Kind string `json:"kind"` // "text" or "tool"
Text string `json:"text,omitempty"`
Tool string `json:"tool,omitempty"`
Toolkind string `json:"toolkind,omitempty"`
Input json.RawMessage `json:"input,omitempty"`
Outcome string `json:"outcome,omitempty"`
Result json.RawMessage `json:"result,omitempty"`
Ms int64 `json:"ms"`
}
// Trace is the whole run.
type Trace struct {
Agentid string `json:"agentid"`
Skillid string `json:"skillid"`
Model string `json:"model"`
Steps []Step `json:"steps"`
Final string `json:"final"`
Stopreason string `json:"stopreason"`
Turns int `json:"turns"`
Inputtokens int64 `json:"inputtokens"`
Outputtokens int64 `json:"outputtokens"`
Ms int64 `json:"ms"`
}
// Run executes the prompt. A model error ends the run with the error; the
// trace so far is still returned so the console can show how far it got.
func Run(ctx context.Context, m Model, plan *Plan, prompt string, execs map[string]Executor) (*Trace, error) {
started := time.Now()
tr := &Trace{Agentid: plan.AgentID, Skillid: plan.SkillID, Model: plan.Model, Steps: []Step{}}
turns := []Turn{{Role: "user", Text: prompt}}
defer func() { tr.Ms = time.Since(started).Milliseconds() }()
for tr.Turns < MaxTurns {
callStarted := time.Now()
reply, err := m.Next(ctx, Request{
Model: plan.Model, System: plan.System, MaxTokens: MaxTokens, Tools: plan.Tools, Turns: turns,
})
tr.Turns++
if err != nil {
tr.Stopreason = "error"
return tr, err
}
tr.Inputtokens += reply.InputTokens
tr.Outputtokens += reply.OutputTokens
tr.Stopreason = reply.StopReason
modelMs := time.Since(callStarted).Milliseconds()
turns = append(turns, Turn{Role: "assistant", Assistant: reply.Blocks})
var texts []string
var results []ToolResult
for _, b := range reply.Blocks {
switch {
case b.Type == "text" && strings.TrimSpace(b.Text) != "":
texts = append(texts, b.Text)
tr.Steps = append(tr.Steps, Step{Kind: "text", Text: b.Text, Ms: modelMs})
modelMs = 0
case b.Type == "tool_use" && b.ToolUse != nil:
step, res := callTool(ctx, plan, execs, *b.ToolUse)
tr.Steps = append(tr.Steps, step)
results = append(results, res)
}
}
if reply.StopReason != "tool_use" || len(results) == 0 {
tr.Final = strings.Join(texts, "\n\n")
return tr, nil
}
turns = append(turns, Turn{Role: "user", Results: results})
}
tr.Stopreason = "max_turns"
return tr, nil
}
func callTool(ctx context.Context, plan *Plan, execs map[string]Executor, use ToolUse) (Step, ToolResult) {
started := time.Now()
input := use.Input
if len(input) == 0 || !json.Valid(input) {
input = json.RawMessage(`{}`)
}
kind, inSkill := plan.kinds[use.Name]
step := Step{Kind: "tool", Tool: use.Name, Toolkind: kind, Input: input}
res := ToolResult{ToolUseID: use.ID}
finish := func(outcome string, payload any, isError bool) (Step, ToolResult) {
body := encode(payload)
step.Outcome, step.Result, step.Ms = outcome, body, time.Since(started).Milliseconds()
res.Content, res.IsError = string(body), isError
return step, res
}
switch {
case !inSkill:
return finish(OutcomeRejected, map[string]string{"error": "This tool is not part of the selected skill."}, true)
case kind != kindRead:
return finish(OutcomeProposed, map[string]any{
"executed": false,
"proposal": map[string]any{"tool": use.Name, "input": input},
"note": "Playground: recorded as a proposal for a human. Nothing was changed.",
}, false)
case execs[use.Name] == nil:
return finish(OutcomeUnavailable, map[string]string{
"error": "This read tool is not available in the playground (it runs inside AI_engine or calls an external service).",
}, true)
}
tctx, cancel := context.WithTimeout(ctx, ToolTimeout)
defer cancel()
out, err := execs[use.Name](tctx, input)
if err != nil {
return finish(OutcomeError, map[string]string{"error": err.Error()}, true)
}
return finish(OutcomeExecuted, Redact(out), false)
}
// encode marshals a tool result, capped so one large read cannot blow the
// context window or the console.
func encode(v any) json.RawMessage {
b, err := json.Marshal(v)
if err != nil {
b, _ = json.Marshal(map[string]string{"error": "result could not be encoded"})
}
if len(b) > maxResultBytes {
b, _ = json.Marshal(map[string]any{
"truncated": true,
"note": fmt.Sprintf("Result was %d bytes; showing the first %d.", len(b), maxResultBytes),
"partial": string(b[:maxResultBytes]),
})
}
return b
}