342 lines
11 KiB
Go
342 lines
11 KiB
Go
// Package playground runs one operator prompt through Claude with a registry
|
|
// skill's tools, for Agent Studio's Test tab (Phase 6 of
|
|
// krow_talent_app/docs/agent-platform-plan.md).
|
|
//
|
|
// The rules that make it safe to point at production:
|
|
// - read tools the backend can serve run for real, and their results are
|
|
// redacted (names, phones, addresses, emails, free text) before Claude sees
|
|
// them — see redact.go;
|
|
// - write, notify and event tools NEVER run: the call is answered with a
|
|
// proposal and shown in the trace as "proposed";
|
|
// - a tool outside the selected skill is refused;
|
|
// - the loop is bounded (MaxTurns) and so is every tool call (ToolTimeout).
|
|
//
|
|
// Claude is reached through the Model interface, so the loop is tested with a
|
|
// fake and the server wires in a real client only when one is configured.
|
|
package playground
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"errors"
|
|
"fmt"
|
|
"strings"
|
|
"time"
|
|
|
|
"doormile/internal/ai/registry"
|
|
)
|
|
|
|
// DefaultModel is used when the agent has no model pinned in the registry.
|
|
const DefaultModel = "claude-opus-5-5"
|
|
|
|
const (
|
|
MaxTurns = 6
|
|
MaxTokens = 4096
|
|
MaxPromptChars = 2000
|
|
ToolTimeout = 5 * time.Second
|
|
maxResultBytes = 16 * 1024
|
|
)
|
|
|
|
// Tool kinds, as the registry stores them.
|
|
const (
|
|
kindRead = "read"
|
|
)
|
|
|
|
// Outcomes of a tool call, as the trace shows them.
|
|
const (
|
|
OutcomeExecuted = "executed"
|
|
OutcomeProposed = "proposed"
|
|
OutcomeUnavailable = "unavailable"
|
|
OutcomeError = "error"
|
|
OutcomeRejected = "rejected"
|
|
)
|
|
|
|
// ── The model boundary ──────────────────────────────────────────────────────
|
|
|
|
// ToolUse is a tool call Claude asked for.
|
|
type ToolUse struct {
|
|
ID string
|
|
Name string
|
|
Input json.RawMessage
|
|
}
|
|
|
|
// Block is one content block of a reply. Text and tool_use are read by the
|
|
// loop; anything else (thinking) is carried in Raw and sent back unchanged,
|
|
// which the API requires within a tool-use turn.
|
|
type Block struct {
|
|
Type string
|
|
Text string
|
|
ToolUse *ToolUse
|
|
Raw json.RawMessage
|
|
}
|
|
|
|
// Reply is one model response.
|
|
type Reply struct {
|
|
Blocks []Block
|
|
StopReason string
|
|
InputTokens int64
|
|
OutputTokens int64
|
|
}
|
|
|
|
// ToolResult answers one ToolUse.
|
|
type ToolResult struct {
|
|
ToolUseID string
|
|
Content string
|
|
IsError bool
|
|
}
|
|
|
|
// Turn is one message of the conversation: the user's prompt, an assistant
|
|
// reply, or the user turn carrying tool results.
|
|
type Turn struct {
|
|
Role string // "user" or "assistant"
|
|
Text string
|
|
Assistant []Block
|
|
Results []ToolResult
|
|
}
|
|
|
|
// ToolDef is a tool as offered to the model.
|
|
type ToolDef struct {
|
|
Name string
|
|
Description string
|
|
InputSchema json.RawMessage
|
|
}
|
|
|
|
// Request is one model call.
|
|
type Request struct {
|
|
Model string
|
|
System string
|
|
MaxTokens int64
|
|
Tools []ToolDef
|
|
Turns []Turn
|
|
}
|
|
|
|
// Model is the one call the loop needs from Claude.
|
|
type Model interface {
|
|
Next(ctx context.Context, req Request) (Reply, error)
|
|
}
|
|
|
|
// ── Plan: what a run may use ────────────────────────────────────────────────
|
|
|
|
// ErrNotFound is returned when the agent or skill does not exist.
|
|
var ErrNotFound = errors.New("not found")
|
|
|
|
// Plan is the resolved agent, skill and tools for one run.
|
|
type Plan struct {
|
|
AgentID string
|
|
AgentName string
|
|
SkillID string
|
|
Model string
|
|
System string
|
|
Tools []ToolDef
|
|
kinds map[string]string
|
|
}
|
|
|
|
// Prepare resolves a run from the registry. skillID may be empty: the run then
|
|
// gets every tool of the agent's enabled skills.
|
|
func Prepare(snap *registry.Snapshot, agentID, skillID string) (*Plan, error) {
|
|
var agent *registry.AgentView
|
|
for i := range snap.Agents {
|
|
if snap.Agents[i].Agentid == agentID {
|
|
agent = &snap.Agents[i]
|
|
}
|
|
}
|
|
if agent == nil {
|
|
return nil, fmt.Errorf("agent %q: %w", agentID, ErrNotFound)
|
|
}
|
|
|
|
toolNames := map[string]bool{}
|
|
var skillLines []string
|
|
found := skillID == ""
|
|
for _, s := range snap.Skills {
|
|
if s.Agentid != agentID {
|
|
continue
|
|
}
|
|
if skillID != "" && s.Skillid != skillID {
|
|
continue
|
|
}
|
|
if skillID == "" && !s.Enabled {
|
|
continue
|
|
}
|
|
found = true
|
|
skillLines = append(skillLines, fmt.Sprintf("- %s: %s", s.Title, s.Description))
|
|
for _, t := range s.Tools {
|
|
toolNames[t] = true
|
|
}
|
|
}
|
|
if !found {
|
|
return nil, fmt.Errorf("skill %q of agent %q: %w", skillID, agentID, ErrNotFound)
|
|
}
|
|
|
|
plan := &Plan{
|
|
AgentID: agent.Agentid, AgentName: agent.Name, SkillID: skillID,
|
|
Model: agent.Model, kinds: map[string]string{}, Tools: []ToolDef{},
|
|
}
|
|
if plan.Model == "" {
|
|
plan.Model = DefaultModel
|
|
}
|
|
for _, t := range snap.Tools {
|
|
if !toolNames[t.Toolname] {
|
|
continue
|
|
}
|
|
desc := t.Description
|
|
if t.Kind != kindRead {
|
|
desc += " [Playground: NOT executed — calling it records a proposal for a human.]"
|
|
}
|
|
plan.Tools = append(plan.Tools, ToolDef{Name: t.Toolname, Description: desc, InputSchema: t.Inputschema})
|
|
plan.kinds[t.Toolname] = t.Kind
|
|
}
|
|
|
|
plan.System = strings.Join([]string{
|
|
fmt.Sprintf("You are %s, an agent in Doormile's delivery operations, being tested by an operator in the Agent Studio playground.", agent.Name),
|
|
"Purpose: " + agent.Purpose,
|
|
"Skills in scope:\n" + strings.Join(skillLines, "\n"),
|
|
"Read tools return live data with personal details (names, phones, addresses, notes) removed; do not ask for them.",
|
|
"Write, notify and event tools are not executed here: calling one records a proposal for a human to review. Say plainly what you would do and why.",
|
|
"If a tool is unavailable, say so rather than guessing its result. Keep the final answer short and concrete.",
|
|
}, "\n\n")
|
|
return plan, nil
|
|
}
|
|
|
|
// ── The run ─────────────────────────────────────────────────────────────────
|
|
|
|
// Executor runs one read tool. Its result is redacted before the model sees it.
|
|
type Executor func(ctx context.Context, input json.RawMessage) (any, error)
|
|
|
|
// Step is one line of the trace the console shows.
|
|
type Step struct {
|
|
Kind string `json:"kind"` // "text" or "tool"
|
|
Text string `json:"text,omitempty"`
|
|
Tool string `json:"tool,omitempty"`
|
|
Toolkind string `json:"toolkind,omitempty"`
|
|
Input json.RawMessage `json:"input,omitempty"`
|
|
Outcome string `json:"outcome,omitempty"`
|
|
Result json.RawMessage `json:"result,omitempty"`
|
|
Ms int64 `json:"ms"`
|
|
}
|
|
|
|
// Trace is the whole run.
|
|
type Trace struct {
|
|
Agentid string `json:"agentid"`
|
|
Skillid string `json:"skillid"`
|
|
Model string `json:"model"`
|
|
Steps []Step `json:"steps"`
|
|
Final string `json:"final"`
|
|
Stopreason string `json:"stopreason"`
|
|
Turns int `json:"turns"`
|
|
Inputtokens int64 `json:"inputtokens"`
|
|
Outputtokens int64 `json:"outputtokens"`
|
|
Ms int64 `json:"ms"`
|
|
}
|
|
|
|
// Run executes the prompt. A model error ends the run with the error; the
|
|
// trace so far is still returned so the console can show how far it got.
|
|
func Run(ctx context.Context, m Model, plan *Plan, prompt string, execs map[string]Executor) (*Trace, error) {
|
|
started := time.Now()
|
|
tr := &Trace{Agentid: plan.AgentID, Skillid: plan.SkillID, Model: plan.Model, Steps: []Step{}}
|
|
turns := []Turn{{Role: "user", Text: prompt}}
|
|
|
|
defer func() { tr.Ms = time.Since(started).Milliseconds() }()
|
|
|
|
for tr.Turns < MaxTurns {
|
|
callStarted := time.Now()
|
|
reply, err := m.Next(ctx, Request{
|
|
Model: plan.Model, System: plan.System, MaxTokens: MaxTokens, Tools: plan.Tools, Turns: turns,
|
|
})
|
|
tr.Turns++
|
|
if err != nil {
|
|
tr.Stopreason = "error"
|
|
return tr, err
|
|
}
|
|
tr.Inputtokens += reply.InputTokens
|
|
tr.Outputtokens += reply.OutputTokens
|
|
tr.Stopreason = reply.StopReason
|
|
modelMs := time.Since(callStarted).Milliseconds()
|
|
|
|
turns = append(turns, Turn{Role: "assistant", Assistant: reply.Blocks})
|
|
|
|
var texts []string
|
|
var results []ToolResult
|
|
for _, b := range reply.Blocks {
|
|
switch {
|
|
case b.Type == "text" && strings.TrimSpace(b.Text) != "":
|
|
texts = append(texts, b.Text)
|
|
tr.Steps = append(tr.Steps, Step{Kind: "text", Text: b.Text, Ms: modelMs})
|
|
modelMs = 0
|
|
case b.Type == "tool_use" && b.ToolUse != nil:
|
|
step, res := callTool(ctx, plan, execs, *b.ToolUse)
|
|
tr.Steps = append(tr.Steps, step)
|
|
results = append(results, res)
|
|
}
|
|
}
|
|
|
|
if reply.StopReason != "tool_use" || len(results) == 0 {
|
|
tr.Final = strings.Join(texts, "\n\n")
|
|
return tr, nil
|
|
}
|
|
turns = append(turns, Turn{Role: "user", Results: results})
|
|
}
|
|
|
|
tr.Stopreason = "max_turns"
|
|
return tr, nil
|
|
}
|
|
|
|
func callTool(ctx context.Context, plan *Plan, execs map[string]Executor, use ToolUse) (Step, ToolResult) {
|
|
started := time.Now()
|
|
input := use.Input
|
|
if len(input) == 0 || !json.Valid(input) {
|
|
input = json.RawMessage(`{}`)
|
|
}
|
|
kind, inSkill := plan.kinds[use.Name]
|
|
step := Step{Kind: "tool", Tool: use.Name, Toolkind: kind, Input: input}
|
|
res := ToolResult{ToolUseID: use.ID}
|
|
|
|
finish := func(outcome string, payload any, isError bool) (Step, ToolResult) {
|
|
body := encode(payload)
|
|
step.Outcome, step.Result, step.Ms = outcome, body, time.Since(started).Milliseconds()
|
|
res.Content, res.IsError = string(body), isError
|
|
return step, res
|
|
}
|
|
|
|
switch {
|
|
case !inSkill:
|
|
return finish(OutcomeRejected, map[string]string{"error": "This tool is not part of the selected skill."}, true)
|
|
|
|
case kind != kindRead:
|
|
return finish(OutcomeProposed, map[string]any{
|
|
"executed": false,
|
|
"proposal": map[string]any{"tool": use.Name, "input": input},
|
|
"note": "Playground: recorded as a proposal for a human. Nothing was changed.",
|
|
}, false)
|
|
|
|
case execs[use.Name] == nil:
|
|
return finish(OutcomeUnavailable, map[string]string{
|
|
"error": "This read tool is not available in the playground (it runs inside AI_engine or calls an external service).",
|
|
}, true)
|
|
}
|
|
|
|
tctx, cancel := context.WithTimeout(ctx, ToolTimeout)
|
|
defer cancel()
|
|
out, err := execs[use.Name](tctx, input)
|
|
if err != nil {
|
|
return finish(OutcomeError, map[string]string{"error": err.Error()}, true)
|
|
}
|
|
return finish(OutcomeExecuted, Redact(out), false)
|
|
}
|
|
|
|
// encode marshals a tool result, capped so one large read cannot blow the
|
|
// context window or the console.
|
|
func encode(v any) json.RawMessage {
|
|
b, err := json.Marshal(v)
|
|
if err != nil {
|
|
b, _ = json.Marshal(map[string]string{"error": "result could not be encoded"})
|
|
}
|
|
if len(b) > maxResultBytes {
|
|
b, _ = json.Marshal(map[string]any{
|
|
"truncated": true,
|
|
"note": fmt.Sprintf("Result was %d bytes; showing the first %d.", len(b), maxResultBytes),
|
|
"partial": string(b[:maxResultBytes]),
|
|
})
|
|
}
|
|
return b
|
|
}
|