// Package playground runs one operator prompt through Claude with a registry // skill's tools, for Agent Studio's Test tab (Phase 6 of // krow_talent_app/docs/agent-platform-plan.md). // // The rules that make it safe to point at production: // - read tools the backend can serve run for real, and their results are // redacted (names, phones, addresses, emails, free text) before Claude sees // them — see redact.go; // - write, notify and event tools NEVER run: the call is answered with a // proposal and shown in the trace as "proposed"; // - a tool outside the selected skill is refused; // - the loop is bounded (MaxTurns) and so is every tool call (ToolTimeout). // // Claude is reached through the Model interface, so the loop is tested with a // fake and the server wires in a real client only when one is configured. package playground import ( "context" "encoding/json" "errors" "fmt" "strings" "time" "doormile/internal/ai/registry" ) // DefaultModel is used when the agent has no model pinned in the registry. const DefaultModel = "claude-opus-5-5" const ( MaxTurns = 6 MaxTokens = 4096 MaxPromptChars = 2000 ToolTimeout = 5 * time.Second maxResultBytes = 16 * 1024 ) // Tool kinds, as the registry stores them. const ( kindRead = "read" ) // Outcomes of a tool call, as the trace shows them. const ( OutcomeExecuted = "executed" OutcomeProposed = "proposed" OutcomeUnavailable = "unavailable" OutcomeError = "error" OutcomeRejected = "rejected" ) // ── The model boundary ────────────────────────────────────────────────────── // ToolUse is a tool call Claude asked for. type ToolUse struct { ID string Name string Input json.RawMessage } // Block is one content block of a reply. Text and tool_use are read by the // loop; anything else (thinking) is carried in Raw and sent back unchanged, // which the API requires within a tool-use turn. type Block struct { Type string Text string ToolUse *ToolUse Raw json.RawMessage } // Reply is one model response. type Reply struct { Blocks []Block StopReason string InputTokens int64 OutputTokens int64 } // ToolResult answers one ToolUse. type ToolResult struct { ToolUseID string Content string IsError bool } // Turn is one message of the conversation: the user's prompt, an assistant // reply, or the user turn carrying tool results. type Turn struct { Role string // "user" or "assistant" Text string Assistant []Block Results []ToolResult } // ToolDef is a tool as offered to the model. type ToolDef struct { Name string Description string InputSchema json.RawMessage } // Request is one model call. type Request struct { Model string System string MaxTokens int64 Tools []ToolDef Turns []Turn } // Model is the one call the loop needs from Claude. type Model interface { Next(ctx context.Context, req Request) (Reply, error) } // ── Plan: what a run may use ──────────────────────────────────────────────── // ErrNotFound is returned when the agent or skill does not exist. var ErrNotFound = errors.New("not found") // Plan is the resolved agent, skill and tools for one run. type Plan struct { AgentID string AgentName string SkillID string Model string System string Tools []ToolDef kinds map[string]string } // Prepare resolves a run from the registry. skillID may be empty: the run then // gets every tool of the agent's enabled skills. func Prepare(snap *registry.Snapshot, agentID, skillID string) (*Plan, error) { var agent *registry.AgentView for i := range snap.Agents { if snap.Agents[i].Agentid == agentID { agent = &snap.Agents[i] } } if agent == nil { return nil, fmt.Errorf("agent %q: %w", agentID, ErrNotFound) } toolNames := map[string]bool{} var skillLines []string found := skillID == "" for _, s := range snap.Skills { if s.Agentid != agentID { continue } if skillID != "" && s.Skillid != skillID { continue } if skillID == "" && !s.Enabled { continue } found = true skillLines = append(skillLines, fmt.Sprintf("- %s: %s", s.Title, s.Description)) for _, t := range s.Tools { toolNames[t] = true } } if !found { return nil, fmt.Errorf("skill %q of agent %q: %w", skillID, agentID, ErrNotFound) } plan := &Plan{ AgentID: agent.Agentid, AgentName: agent.Name, SkillID: skillID, Model: agent.Model, kinds: map[string]string{}, Tools: []ToolDef{}, } if plan.Model == "" { plan.Model = DefaultModel } for _, t := range snap.Tools { if !toolNames[t.Toolname] { continue } desc := t.Description if t.Kind != kindRead { desc += " [Playground: NOT executed — calling it records a proposal for a human.]" } plan.Tools = append(plan.Tools, ToolDef{Name: t.Toolname, Description: desc, InputSchema: t.Inputschema}) plan.kinds[t.Toolname] = t.Kind } plan.System = strings.Join([]string{ fmt.Sprintf("You are %s, an agent in Doormile's delivery operations, being tested by an operator in the Agent Studio playground.", agent.Name), "Purpose: " + agent.Purpose, "Skills in scope:\n" + strings.Join(skillLines, "\n"), "Read tools return live data with personal details (names, phones, addresses, notes) removed; do not ask for them.", "Write, notify and event tools are not executed here: calling one records a proposal for a human to review. Say plainly what you would do and why.", "If a tool is unavailable, say so rather than guessing its result. Keep the final answer short and concrete.", }, "\n\n") return plan, nil } // ── The run ───────────────────────────────────────────────────────────────── // Executor runs one read tool. Its result is redacted before the model sees it. type Executor func(ctx context.Context, input json.RawMessage) (any, error) // Step is one line of the trace the console shows. type Step struct { Kind string `json:"kind"` // "text" or "tool" Text string `json:"text,omitempty"` Tool string `json:"tool,omitempty"` Toolkind string `json:"toolkind,omitempty"` Input json.RawMessage `json:"input,omitempty"` Outcome string `json:"outcome,omitempty"` Result json.RawMessage `json:"result,omitempty"` Ms int64 `json:"ms"` } // Trace is the whole run. type Trace struct { Agentid string `json:"agentid"` Skillid string `json:"skillid"` Model string `json:"model"` Steps []Step `json:"steps"` Final string `json:"final"` Stopreason string `json:"stopreason"` Turns int `json:"turns"` Inputtokens int64 `json:"inputtokens"` Outputtokens int64 `json:"outputtokens"` Ms int64 `json:"ms"` } // Run executes the prompt. A model error ends the run with the error; the // trace so far is still returned so the console can show how far it got. func Run(ctx context.Context, m Model, plan *Plan, prompt string, execs map[string]Executor) (*Trace, error) { started := time.Now() tr := &Trace{Agentid: plan.AgentID, Skillid: plan.SkillID, Model: plan.Model, Steps: []Step{}} turns := []Turn{{Role: "user", Text: prompt}} defer func() { tr.Ms = time.Since(started).Milliseconds() }() for tr.Turns < MaxTurns { callStarted := time.Now() reply, err := m.Next(ctx, Request{ Model: plan.Model, System: plan.System, MaxTokens: MaxTokens, Tools: plan.Tools, Turns: turns, }) tr.Turns++ if err != nil { tr.Stopreason = "error" return tr, err } tr.Inputtokens += reply.InputTokens tr.Outputtokens += reply.OutputTokens tr.Stopreason = reply.StopReason modelMs := time.Since(callStarted).Milliseconds() turns = append(turns, Turn{Role: "assistant", Assistant: reply.Blocks}) var texts []string var results []ToolResult for _, b := range reply.Blocks { switch { case b.Type == "text" && strings.TrimSpace(b.Text) != "": texts = append(texts, b.Text) tr.Steps = append(tr.Steps, Step{Kind: "text", Text: b.Text, Ms: modelMs}) modelMs = 0 case b.Type == "tool_use" && b.ToolUse != nil: step, res := callTool(ctx, plan, execs, *b.ToolUse) tr.Steps = append(tr.Steps, step) results = append(results, res) } } if reply.StopReason != "tool_use" || len(results) == 0 { tr.Final = strings.Join(texts, "\n\n") return tr, nil } turns = append(turns, Turn{Role: "user", Results: results}) } tr.Stopreason = "max_turns" return tr, nil } func callTool(ctx context.Context, plan *Plan, execs map[string]Executor, use ToolUse) (Step, ToolResult) { started := time.Now() input := use.Input if len(input) == 0 || !json.Valid(input) { input = json.RawMessage(`{}`) } kind, inSkill := plan.kinds[use.Name] step := Step{Kind: "tool", Tool: use.Name, Toolkind: kind, Input: input} res := ToolResult{ToolUseID: use.ID} finish := func(outcome string, payload any, isError bool) (Step, ToolResult) { body := encode(payload) step.Outcome, step.Result, step.Ms = outcome, body, time.Since(started).Milliseconds() res.Content, res.IsError = string(body), isError return step, res } switch { case !inSkill: return finish(OutcomeRejected, map[string]string{"error": "This tool is not part of the selected skill."}, true) case kind != kindRead: return finish(OutcomeProposed, map[string]any{ "executed": false, "proposal": map[string]any{"tool": use.Name, "input": input}, "note": "Playground: recorded as a proposal for a human. Nothing was changed.", }, false) case execs[use.Name] == nil: return finish(OutcomeUnavailable, map[string]string{ "error": "This read tool is not available in the playground (it runs inside AI_engine or calls an external service).", }, true) } tctx, cancel := context.WithTimeout(ctx, ToolTimeout) defer cancel() out, err := execs[use.Name](tctx, input) if err != nil { return finish(OutcomeError, map[string]string{"error": err.Error()}, true) } return finish(OutcomeExecuted, Redact(out), false) } // encode marshals a tool result, capped so one large read cannot blow the // context window or the console. func encode(v any) json.RawMessage { b, err := json.Marshal(v) if err != nil { b, _ = json.Marshal(map[string]string{"error": "result could not be encoded"}) } if len(b) > maxResultBytes { b, _ = json.Marshal(map[string]any{ "truncated": true, "note": fmt.Sprintf("Result was %d bytes; showing the first %d.", len(b), maxResultBytes), "partial": string(b[:maxResultBytes]), }) } return b }