agent build
This commit is contained in:
477
go-api/internal/gateway/anthropic.go
Normal file
477
go-api/internal/gateway/anthropic.go
Normal file
@@ -0,0 +1,477 @@
|
||||
package gateway
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/anthropics/anthropic-sdk-go"
|
||||
"github.com/anthropics/anthropic-sdk-go/option"
|
||||
)
|
||||
|
||||
// Routing is how a tier becomes a model and an effort level.
|
||||
//
|
||||
// The model per tier is a deployment knob — a tenant on a different contract,
|
||||
// or a deployment pinning a version through an incident, changes it without a
|
||||
// spec edit. The *effort* per tier is not: "fast" and "deep" mean something
|
||||
// specific about how much work an answer is worth, and letting a deployment
|
||||
// redefine that would make the same spec behave differently in two places
|
||||
// while claiming the same tier.
|
||||
type Routing struct {
|
||||
Model string
|
||||
Effort anthropic.OutputConfigEffort
|
||||
}
|
||||
|
||||
// Config is the gateway's whole configuration surface.
|
||||
//
|
||||
// Built once at startup from the environment and passed in frozen, per §10.
|
||||
// Nothing in this package reads the environment itself.
|
||||
type Config struct {
|
||||
APIKey string
|
||||
|
||||
Fast Routing
|
||||
Balanced Routing
|
||||
Deep Routing
|
||||
|
||||
// MaxOutputTokens applies when a request does not set its own.
|
||||
MaxOutputTokens int64
|
||||
}
|
||||
|
||||
// AnthropicGateway calls the Claude API.
|
||||
type AnthropicGateway struct {
|
||||
client anthropic.Client
|
||||
cfg Config
|
||||
}
|
||||
|
||||
// Compile-time proof that this satisfies the boundary.
|
||||
var _ Gateway = (*AnthropicGateway)(nil)
|
||||
|
||||
// NewAnthropic builds a gateway over the Claude API.
|
||||
//
|
||||
// A missing key is not an error here. The service has to boot without model
|
||||
// credentials — every endpoint that is not an agent run still works, and a
|
||||
// developer running migrations should not need a key to do it. The failure
|
||||
// surfaces at the first Complete, as a structured NotConfigured that the
|
||||
// runtime can end a run with, rather than as a panic at startup.
|
||||
func NewAnthropic(cfg Config) *AnthropicGateway {
|
||||
opts := []option.RequestOption{}
|
||||
if cfg.APIKey != "" {
|
||||
opts = append(opts, option.WithAPIKey(cfg.APIKey))
|
||||
}
|
||||
return &AnthropicGateway{client: anthropic.NewClient(opts...), cfg: cfg}
|
||||
}
|
||||
|
||||
// routing resolves a tier. An unknown tier has already been normalised by
|
||||
// ParseTier, so the default arm is reached only by a zero value.
|
||||
func (g *AnthropicGateway) routing(t Tier) Routing {
|
||||
switch t {
|
||||
case TierFast:
|
||||
return g.cfg.Fast
|
||||
case TierDeep:
|
||||
return g.cfg.Deep
|
||||
default:
|
||||
return g.cfg.Balanced
|
||||
}
|
||||
}
|
||||
|
||||
// Complete sends one request and reports one result.
|
||||
// MaxAttempts is how many times a transient failure is retried.
|
||||
//
|
||||
// Three total, not three retries. Past that the problem is not transient and a
|
||||
// fourth call is just spending money on the same answer.
|
||||
const MaxAttempts = 3
|
||||
|
||||
// retryBackoff is the pause before each retry.
|
||||
//
|
||||
// Short, and deliberately so: this sits inside a run that already has a
|
||||
// wall-clock deadline, and a backoff long enough to be polite to the API is
|
||||
// long enough to spend the caller's whole budget waiting. A run that cannot
|
||||
// afford the wait dies on its deadline instead, which is the correct failure.
|
||||
var retryBackoff = []time.Duration{400 * time.Millisecond, 1200 * time.Millisecond}
|
||||
|
||||
// Complete calls the model, retrying failures that are worth retrying.
|
||||
//
|
||||
// THE RETRY IS NOT DEFENSIVE POLISH. Error.Retryable() has existed since this
|
||||
// package was written and had ZERO callers — the classification was built and
|
||||
// never used, so a 529 "overloaded" killed a run that would have succeeded four
|
||||
// hundred milliseconds later. Found by a real overload during live testing,
|
||||
// where it presented as "the agent could not finish" with nothing to act on.
|
||||
//
|
||||
// Only genuinely transient failures qualify: rate limits, timeouts, and 5xx.
|
||||
// A 400 is a malformed request and will be malformed again; a 401 is a bad
|
||||
// credential and retrying it three times just gets refused three times.
|
||||
//
|
||||
// The run's context governs. A retry that would outlive the caller's deadline
|
||||
// does not happen — the deadline belongs to the run, not to this function, and
|
||||
// waiting past it would turn a bounded run into an unbounded one.
|
||||
func (g *AnthropicGateway) Complete(ctx context.Context, req Request) (*Response, error) {
|
||||
var last error
|
||||
for attempt := 0; attempt < MaxAttempts; attempt++ {
|
||||
if attempt > 0 {
|
||||
pause := retryBackoff[min(attempt-1, len(retryBackoff)-1)]
|
||||
select {
|
||||
case <-time.After(pause):
|
||||
case <-ctx.Done():
|
||||
// Out of time. The ORIGINAL failure is returned rather than the
|
||||
// context error: "the model was overloaded" is what an operator
|
||||
// needs to see, and "context deadline exceeded" would hide it.
|
||||
return nil, last
|
||||
}
|
||||
}
|
||||
|
||||
resp, err := g.complete(ctx, req)
|
||||
if err == nil {
|
||||
return resp, nil
|
||||
}
|
||||
last = err
|
||||
|
||||
var gwErr *Error
|
||||
if !errors.As(err, &gwErr) || !gwErr.Retryable() {
|
||||
return resp, err
|
||||
}
|
||||
}
|
||||
return nil, last
|
||||
}
|
||||
|
||||
// complete is one attempt.
|
||||
func (g *AnthropicGateway) complete(ctx context.Context, req Request) (*Response, error) {
|
||||
if g.cfg.APIKey == "" {
|
||||
return nil, &Error{
|
||||
Code: CodeNotConfigured,
|
||||
Message: "no model credentials are configured for this deployment",
|
||||
}
|
||||
}
|
||||
if err := req.Validate(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
params, err := g.params(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
msg, err := g.client.Messages.New(ctx, params)
|
||||
if err != nil {
|
||||
return nil, translate(err)
|
||||
}
|
||||
return g.decode(msg, req)
|
||||
}
|
||||
|
||||
// params builds the request both paths send.
|
||||
//
|
||||
// Extracted so the streaming and non-streaming calls cannot drift. They send
|
||||
// the same model, the same thinking config, the same cache breakpoint and the
|
||||
// same tools — an answer that differs depending on whether it was streamed
|
||||
// would be the worst kind of bug to chase, because the transport is the last
|
||||
// place anybody looks.
|
||||
func (g *AnthropicGateway) params(req Request) (anthropic.MessageNewParams, error) {
|
||||
route := g.routing(req.Tier)
|
||||
|
||||
maxTokens := req.MaxOutputTokens
|
||||
if maxTokens <= 0 {
|
||||
maxTokens = g.cfg.MaxOutputTokens
|
||||
}
|
||||
|
||||
messages, err := encodeMessages(req.Messages)
|
||||
if err != nil {
|
||||
return anthropic.MessageNewParams{}, err
|
||||
}
|
||||
|
||||
params := anthropic.MessageNewParams{
|
||||
Model: anthropic.Model(route.Model),
|
||||
MaxTokens: maxTokens,
|
||||
Messages: messages,
|
||||
// Adaptive thinking on every tier: the model decides how much to think,
|
||||
// and effort sets the ceiling on that. A fixed token budget for
|
||||
// reasoning is the deprecated shape and is rejected outright by the
|
||||
// current models.
|
||||
Thinking: anthropic.ThinkingConfigParamUnion{
|
||||
OfAdaptive: &anthropic.ThinkingConfigAdaptiveParam{},
|
||||
},
|
||||
OutputConfig: anthropic.OutputConfigParam{Effort: route.Effort},
|
||||
}
|
||||
|
||||
if len(req.Tools) > 0 {
|
||||
params.Tools = encodeTools(req.Tools)
|
||||
}
|
||||
|
||||
if s := strings.TrimSpace(req.System); s != "" {
|
||||
// One cached block. The system prompt is the stable prefix of every
|
||||
// turn in a run, and the render order is tools → system → messages, so
|
||||
// a breakpoint here is the one that survives the conversation growing.
|
||||
params.System = []anthropic.TextBlockParam{{
|
||||
Text: s,
|
||||
CacheControl: anthropic.NewCacheControlEphemeralParam(),
|
||||
}}
|
||||
}
|
||||
|
||||
return params, nil
|
||||
}
|
||||
|
||||
// decode turns a finished message into a Response.
|
||||
//
|
||||
// Shared by both paths for the same reason params() is: a streamed message and
|
||||
// a non-streamed one are the same object by the time they get here, and reading
|
||||
// them differently would make streaming a second implementation of the answer.
|
||||
func (g *AnthropicGateway) decode(msg *anthropic.Message, req Request) (*Response, error) {
|
||||
route := g.routing(req.Tier)
|
||||
|
||||
usage := Usage{
|
||||
InputTokens: msg.Usage.InputTokens,
|
||||
OutputTokens: msg.Usage.OutputTokens,
|
||||
CacheReadTokens: msg.Usage.CacheReadInputTokens,
|
||||
CacheCreationTokens: msg.Usage.CacheCreationInputTokens,
|
||||
}
|
||||
|
||||
// A refusal arrives as a successful HTTP response, so it has to be checked
|
||||
// before the content is read. It is still billed, and the usage is carried
|
||||
// on the error so the run's budget is charged for a turn that produced no
|
||||
// text — a refusal that cost nothing on the ledger is a refusal the loop
|
||||
// would happily repeat.
|
||||
if msg.StopReason == anthropic.StopReasonRefusal {
|
||||
return &Response{
|
||||
StopReason: string(msg.StopReason),
|
||||
Usage: usage,
|
||||
Model: route.Model,
|
||||
Tier: req.Tier,
|
||||
}, &Error{
|
||||
Code: CodeRefused,
|
||||
Message: "the model declined this request",
|
||||
Category: string(msg.StopDetails.Category),
|
||||
}
|
||||
}
|
||||
|
||||
var (
|
||||
text strings.Builder
|
||||
calls []ToolCall
|
||||
)
|
||||
for _, block := range msg.Content {
|
||||
switch b := block.AsAny().(type) {
|
||||
case anthropic.TextBlock:
|
||||
text.WriteString(b.Text)
|
||||
case anthropic.ToolUseBlock:
|
||||
// The raw JSON, not a parsed value: current models vary their
|
||||
// string escaping inside tool inputs, so this is handed to the
|
||||
// handler's own decoder rather than matched on as a string here.
|
||||
calls = append(calls, ToolCall{
|
||||
ID: b.ID,
|
||||
Name: b.Name,
|
||||
Input: json.RawMessage(b.JSON.Input.Raw()),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
return &Response{
|
||||
Text: text.String(),
|
||||
ToolCalls: calls,
|
||||
StopReason: string(msg.StopReason),
|
||||
Usage: usage,
|
||||
Model: route.Model,
|
||||
Tier: req.Tier,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// encodeTools renders the tool definitions for the wire.
|
||||
func encodeTools(defs []ToolDef) []anthropic.ToolUnionParam {
|
||||
out := make([]anthropic.ToolUnionParam, 0, len(defs))
|
||||
for _, d := range defs {
|
||||
schema := anthropic.ToolInputSchemaParam{}
|
||||
if props, ok := d.InputSchema["properties"].(map[string]any); ok {
|
||||
schema.Properties = props
|
||||
}
|
||||
if req, ok := d.InputSchema["required"].([]string); ok {
|
||||
schema.Required = req
|
||||
}
|
||||
tool := anthropic.ToolParam{
|
||||
Name: d.Name,
|
||||
Description: anthropic.String(d.Description),
|
||||
InputSchema: schema,
|
||||
}
|
||||
out = append(out, anthropic.ToolUnionParam{OfTool: &tool})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// encodeMessages renders a conversation for the wire.
|
||||
//
|
||||
// Tool results are variadic within ONE user message. Splitting them across
|
||||
// several messages is accepted by the API and quietly teaches the model to stop
|
||||
// making parallel calls — a performance regression with no error to trace it
|
||||
// to, so the grouping is done here rather than left to callers.
|
||||
func encodeMessages(msgs []Message) ([]anthropic.MessageParam, error) {
|
||||
out := make([]anthropic.MessageParam, 0, len(msgs))
|
||||
|
||||
for i, m := range msgs {
|
||||
var blocks []anthropic.ContentBlockParamUnion
|
||||
|
||||
if s := strings.TrimSpace(m.Text); s != "" {
|
||||
blocks = append(blocks, anthropic.NewTextBlock(m.Text))
|
||||
}
|
||||
for _, c := range m.ToolCalls {
|
||||
var input any
|
||||
if len(c.Input) > 0 {
|
||||
if err := json.Unmarshal(c.Input, &input); err != nil {
|
||||
return nil, &Error{
|
||||
Code: CodeInvalidRequest,
|
||||
Message: fmt.Sprintf("messages[%d]: tool call %s carries invalid JSON", i, c.Name),
|
||||
}
|
||||
}
|
||||
}
|
||||
blocks = append(blocks, anthropic.NewToolUseBlock(c.ID, input, c.Name))
|
||||
}
|
||||
for _, r := range m.ToolResults {
|
||||
blocks = append(blocks, anthropic.NewToolResultBlock(r.CallID, r.Content, r.IsError))
|
||||
}
|
||||
|
||||
if len(blocks) == 0 {
|
||||
continue
|
||||
}
|
||||
if m.Role == RoleAssistant {
|
||||
out = append(out, anthropic.NewAssistantMessage(blocks...))
|
||||
continue
|
||||
}
|
||||
out = append(out, anthropic.NewUserMessage(blocks...))
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// translate turns an SDK error into one the runtime can branch on.
|
||||
//
|
||||
// A single broad class would lose the distinction the loop actually needs:
|
||||
// whether sending the same request again could work. So the status is read and
|
||||
// mapped, and anything unrecognised stays CodeUpstream with its status intact
|
||||
// rather than being flattened into a generic failure.
|
||||
func translate(err error) error {
|
||||
if errors.Is(err, context.DeadlineExceeded) || errors.Is(err, context.Canceled) {
|
||||
return &Error{Code: CodeTimeout, Message: "the model call did not complete in time", Cause: err}
|
||||
}
|
||||
|
||||
var apierr *anthropic.Error
|
||||
if !errors.As(err, &apierr) {
|
||||
return &Error{Code: CodeUpstream, Message: "the model call failed", Cause: err}
|
||||
}
|
||||
|
||||
switch apierr.StatusCode {
|
||||
case 400:
|
||||
return &Error{Code: CodeInvalidRequest, Message: "the model rejected the request", Status: 400, Cause: err}
|
||||
case 401, 403:
|
||||
return &Error{Code: CodeUnauthorized, Message: "the model credentials were refused", Status: apierr.StatusCode, Cause: err}
|
||||
case 408:
|
||||
return &Error{Code: CodeTimeout, Message: "the model call timed out", Status: 408, Cause: err}
|
||||
case 429:
|
||||
return &Error{Code: CodeRateLimited, Message: "the model is rate limiting this deployment", Status: 429, Cause: err}
|
||||
case 529:
|
||||
// Anthropic's "overloaded" — the service is up and temporarily out of
|
||||
// capacity. Named separately from the 500s because it is the one that
|
||||
// actually happens, and because a run dying on it is a run that would
|
||||
// have succeeded a second later.
|
||||
return &Error{Code: CodeUpstream, Message: "the model is temporarily overloaded",
|
||||
Status: 529, Cause: err}
|
||||
default:
|
||||
// The status is IN the message, not only in the field. It cost an hour
|
||||
// of debugging to learn that "the model call failed" was a 529 rather
|
||||
// than a malformed tool schema, and the trajectory only records the
|
||||
// message.
|
||||
return &Error{
|
||||
Code: CodeUpstream,
|
||||
Message: fmt.Sprintf("the model call failed (http %d)", apierr.StatusCode),
|
||||
Status: apierr.StatusCode, Cause: err,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* ── Streaming ──────────────────────────────────────────────────────────── */
|
||||
|
||||
// Stream is Complete, with the assistant's text delivered as it arrives.
|
||||
//
|
||||
// §6: "Stream partial assistant text as it arrives; buffer tool calls until
|
||||
// complete." Both halves of that matter and they pull in opposite directions.
|
||||
//
|
||||
// TEXT IS STREAMED because a fifteen-second wait with nothing on screen reads
|
||||
// as broken. The reader wants the first sentence while the rest is still being
|
||||
// written, and that is the whole difference between a product and a spinner.
|
||||
//
|
||||
// TOOL CALLS ARE NOT. A tool call arrives as JSON assembled character by
|
||||
// character across many events, and a half-built argument object is not a
|
||||
// smaller version of the finished one — it is a different object, usually an
|
||||
// invalid one. Dispatching on a partial call would run a tool with arguments
|
||||
// the model had not finished choosing. So the accumulated message is decoded
|
||||
// only once the stream closes, by exactly the same code the non-streaming path
|
||||
// uses.
|
||||
//
|
||||
// onDelta is called from this goroutine, in order, and must not block for long
|
||||
// — it is on the path between the model and the reader.
|
||||
func (g *AnthropicGateway) Stream(ctx context.Context, req Request, onDelta func(string)) (*Response, error) {
|
||||
if g.cfg.APIKey == "" {
|
||||
return nil, &Error{
|
||||
Code: CodeNotConfigured,
|
||||
Message: "no model credentials are configured for this deployment",
|
||||
}
|
||||
}
|
||||
if err := req.Validate(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
params, err := g.params(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
stream := g.client.Messages.NewStreaming(ctx, params)
|
||||
defer stream.Close()
|
||||
|
||||
var msg anthropic.Message
|
||||
for stream.Next() {
|
||||
event := stream.Current()
|
||||
if err := msg.Accumulate(event); err != nil {
|
||||
return nil, &Error{
|
||||
Code: CodeUpstream,
|
||||
Message: "the streamed response could not be assembled",
|
||||
Cause: err,
|
||||
}
|
||||
}
|
||||
|
||||
// Text only. A thinking delta is the model's private reasoning and is
|
||||
// not the answer; a tool-input delta is a fragment of JSON. Neither is
|
||||
// something to put in front of a reader.
|
||||
if event.Type == "content_block_delta" && event.Delta.Type == "text_delta" {
|
||||
if d := event.Delta.Text; d != "" && onDelta != nil {
|
||||
onDelta(d)
|
||||
}
|
||||
}
|
||||
}
|
||||
if err := stream.Err(); err != nil {
|
||||
return nil, translate(err)
|
||||
}
|
||||
|
||||
return g.decode(&msg, req)
|
||||
}
|
||||
|
||||
// StreamComplete runs a request through whichever path the gateway supports.
|
||||
//
|
||||
// A gateway that cannot stream is not a broken gateway — every fake in the test
|
||||
// suite is one, and so is any future provider without a streaming API. Falling
|
||||
// back to Complete and delivering the finished text as a single delta keeps the
|
||||
// caller's code identical either way, which is what stops streaming from
|
||||
// becoming a second code path through the loop.
|
||||
func StreamComplete(ctx context.Context, gw Gateway, req Request, onDelta func(string)) (*Response, error) {
|
||||
// Normalised once, here, so no implementation has to guard it. A caller
|
||||
// that does not want deltas passes nil — every eval and every test does —
|
||||
// and an implementation that took that literally would panic on the first
|
||||
// fragment. Making each Streamer remember the check is how one of them
|
||||
// eventually forgets.
|
||||
if onDelta == nil {
|
||||
onDelta = func(string) {}
|
||||
}
|
||||
if s, ok := gw.(Streamer); ok {
|
||||
return s.Stream(ctx, req, onDelta)
|
||||
}
|
||||
resp, err := gw.Complete(ctx, req)
|
||||
if err == nil && resp != nil && resp.Text != "" && onDelta != nil {
|
||||
onDelta(resp.Text)
|
||||
}
|
||||
return resp, err
|
||||
}
|
||||
294
go-api/internal/gateway/gateway.go
Normal file
294
go-api/internal/gateway/gateway.go
Normal file
@@ -0,0 +1,294 @@
|
||||
// Package gateway is the model gateway: the one place in this service that
|
||||
// talks to a language model.
|
||||
//
|
||||
// Everything else — the runtime loop, the tool layer, retrieval — reaches a
|
||||
// model through this package and nowhere else. That is the whole point of it
|
||||
// being a layer rather than a helper:
|
||||
//
|
||||
// - **Routing lives here.** An agent spec declares a `reasoning` tier, not a
|
||||
// model id. Which model and how much thinking that tier buys is a
|
||||
// deployment decision, and it changes without touching a single spec.
|
||||
// - **Token accounting lives here.** Every call returns what it cost. A
|
||||
// budget the runtime cannot measure is a budget it cannot enforce, and
|
||||
// I3 requires it to enforce one.
|
||||
// - **Refusal is an outcome, not an exception.** A model that declines comes
|
||||
// back as a structured Refused, which is one of the six termination
|
||||
// reasons the runtime already knows how to end a run with.
|
||||
//
|
||||
// What this package deliberately does *not* do: assemble prompts, decide what
|
||||
// a caller may read, or loop. It sends one request and reports one result.
|
||||
// Composition is the runtime's job and authorization is the tool layer's, and
|
||||
// folding either of them in here would put policy behind a transport.
|
||||
package gateway
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// Tier is an agent spec's `reasoning` value.
|
||||
//
|
||||
// Three tiers, because an author choosing between "fast" and "deep" is making
|
||||
// a judgement about the work, not about a model. The mapping from a tier to a
|
||||
// model and an effort level is this package's business and is configured per
|
||||
// deployment — a spec that named a model directly would pin every tenant to
|
||||
// whatever was current the day it was written.
|
||||
type Tier string
|
||||
|
||||
const (
|
||||
TierFast Tier = "fast"
|
||||
TierBalanced Tier = "balanced"
|
||||
TierDeep Tier = "deep"
|
||||
)
|
||||
|
||||
// DefaultTier is what a spec that declares no reasoning mode gets. It matches
|
||||
// the frontend vocabulary's own default, so a definition means the same thing
|
||||
// on both sides of the wire.
|
||||
const DefaultTier = TierBalanced
|
||||
|
||||
// ParseTier resolves a spec's declared reasoning value.
|
||||
//
|
||||
// An unrecognised tier falls back rather than failing: the tier affects how
|
||||
// much a turn costs, never whether it is allowed, so refusing the run would
|
||||
// turn a typo in a definition into an outage. The caller is told, so a
|
||||
// definition that has drifted from the vocabulary is still visible.
|
||||
func ParseTier(raw string) (Tier, bool) {
|
||||
switch Tier(strings.ToLower(strings.TrimSpace(raw))) {
|
||||
case TierFast:
|
||||
return TierFast, true
|
||||
case TierBalanced:
|
||||
return TierBalanced, true
|
||||
case TierDeep:
|
||||
return TierDeep, true
|
||||
case "":
|
||||
return DefaultTier, true
|
||||
default:
|
||||
return DefaultTier, false
|
||||
}
|
||||
}
|
||||
|
||||
// Role is who said something.
|
||||
type Role string
|
||||
|
||||
const (
|
||||
RoleUser Role = "user"
|
||||
RoleAssistant Role = "assistant"
|
||||
)
|
||||
|
||||
// ToolCall is the model asking for a tool to be run.
|
||||
type ToolCall struct {
|
||||
// ID correlates the call with its result. Echoed back verbatim: it is the
|
||||
// model's own handle, and a result carrying a different one is a result
|
||||
// attached to the wrong question.
|
||||
ID string
|
||||
Name string
|
||||
Input json.RawMessage
|
||||
}
|
||||
|
||||
// ToolResult is what came back, on its way to the model.
|
||||
//
|
||||
// Content is a string because that is what crosses the wire, but it carries
|
||||
// encoded structured data — §4 keeps formatting the model's job, so a handler
|
||||
// never writes prose and this never carries any.
|
||||
type ToolResult struct {
|
||||
CallID string
|
||||
Content string
|
||||
IsError bool
|
||||
}
|
||||
|
||||
// Message is one turn of a conversation.
|
||||
//
|
||||
// A turn is text, or tool calls, or tool results — an assistant turn may carry
|
||||
// text and calls together, which is why these are fields rather than a union.
|
||||
type Message struct {
|
||||
Role Role
|
||||
Text string
|
||||
ToolCalls []ToolCall
|
||||
ToolResults []ToolResult
|
||||
}
|
||||
|
||||
// ToolDef is a tool as the model sees it.
|
||||
//
|
||||
// Deliberately not the tool layer's own type. The gateway must not import the
|
||||
// tool package: a model provider knowing what an `effect` or a confirmation
|
||||
// token is would put policy behind a transport, and the confirmation gate has
|
||||
// to sit where the model cannot reach it.
|
||||
type ToolDef struct {
|
||||
Name string
|
||||
Description string
|
||||
InputSchema map[string]any
|
||||
}
|
||||
|
||||
// Request is one model call.
|
||||
type Request struct {
|
||||
// Tier selects the model and effort. From the agent spec.
|
||||
Tier Tier
|
||||
|
||||
// System is the assembled system prompt.
|
||||
//
|
||||
// I7: retrieved document text must never reach this field. Retrieved
|
||||
// content belongs in a delimited context block inside a user message,
|
||||
// where the system prompt has already said that its contents are data.
|
||||
// Nothing here can enforce that — it is a property of what the runtime
|
||||
// passes — so it is stated where the field is declared.
|
||||
System string
|
||||
|
||||
// Messages is the conversation so far, oldest first.
|
||||
Messages []Message
|
||||
|
||||
// Tools the model may call this turn. Order matters: it is part of the
|
||||
// cached prefix, so the caller sorts it once and keeps it stable.
|
||||
Tools []ToolDef
|
||||
|
||||
// MaxOutputTokens caps this response. Zero takes the configured default.
|
||||
//
|
||||
// This is a hard ceiling the model is not aware of, so it truncates rather
|
||||
// than winding down. It is not the run's token budget — that is the
|
||||
// runtime's, and it spans every call in a run.
|
||||
MaxOutputTokens int64
|
||||
}
|
||||
|
||||
// Usage is what a call cost.
|
||||
type Usage struct {
|
||||
InputTokens int64
|
||||
OutputTokens int64
|
||||
CacheReadTokens int64
|
||||
CacheCreationTokens int64
|
||||
}
|
||||
|
||||
// Total is every token this call is billed for.
|
||||
//
|
||||
// Cache reads are counted: they are cheaper than fresh input, not free, and a
|
||||
// budget that ignored them would drift further from the truth the longer a
|
||||
// conversation ran — which is exactly when it matters most.
|
||||
func (u Usage) Total() int64 {
|
||||
return u.InputTokens + u.OutputTokens + u.CacheReadTokens + u.CacheCreationTokens
|
||||
}
|
||||
|
||||
// Response is one model reply.
|
||||
type Response struct {
|
||||
Text string
|
||||
|
||||
// ToolCalls the model wants run before it can continue. Non-empty exactly
|
||||
// when StopReason is "tool_use".
|
||||
ToolCalls []ToolCall
|
||||
|
||||
StopReason string
|
||||
Usage Usage
|
||||
|
||||
// Model is the id actually used, not the tier that was asked for. Logged
|
||||
// with every run so a change of routing is visible in the trajectory
|
||||
// rather than inferred from a deploy date.
|
||||
Model string
|
||||
Tier Tier
|
||||
}
|
||||
|
||||
// Error codes. Structured rather than bare strings, per §10 — user-facing text
|
||||
// is derived at the surface layer, never raised from here.
|
||||
const (
|
||||
CodeNotConfigured = "gateway.not_configured"
|
||||
CodeInvalidRequest = "gateway.invalid_request"
|
||||
CodeUnauthorized = "gateway.unauthorized"
|
||||
CodeRateLimited = "gateway.rate_limited"
|
||||
CodeTimeout = "gateway.timeout"
|
||||
CodeRefused = "gateway.refused"
|
||||
CodeUpstream = "gateway.upstream"
|
||||
)
|
||||
|
||||
// Error is a gateway failure with a code the runtime can branch on.
|
||||
type Error struct {
|
||||
Code string
|
||||
Message string
|
||||
|
||||
// Status is the upstream HTTP status, when there was one.
|
||||
Status int
|
||||
|
||||
// Category carries a refusal's reason when Code is CodeRefused. An open
|
||||
// set upstream, so it is a string and is never switched on exhaustively.
|
||||
Category string
|
||||
|
||||
Cause error
|
||||
}
|
||||
|
||||
func (e *Error) Error() string {
|
||||
if e.Status != 0 {
|
||||
return fmt.Sprintf("%s: %s (http %d)", e.Code, e.Message, e.Status)
|
||||
}
|
||||
return fmt.Sprintf("%s: %s", e.Code, e.Message)
|
||||
}
|
||||
|
||||
func (e *Error) Unwrap() error { return e.Cause }
|
||||
|
||||
// Retryable reports whether the same request could succeed if sent again.
|
||||
//
|
||||
// The runtime needs this to decide between a retry and a terminal
|
||||
// ToolFailure. A refusal is emphatically not retryable — re-sending a request
|
||||
// the model declined is how a loop burns a whole budget on one turn.
|
||||
func (e *Error) Retryable() bool {
|
||||
switch e.Code {
|
||||
case CodeRateLimited, CodeTimeout:
|
||||
return true
|
||||
case CodeUpstream:
|
||||
return e.Status >= 500
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// Streamer is a Gateway that can deliver text as it arrives.
|
||||
//
|
||||
// A SEPARATE interface, not a method on Gateway, and that is deliberate. Adding
|
||||
// Stream to Gateway would break every fake in the test suite and force each one
|
||||
// to implement a transport it does not care about — and those fakes exist to
|
||||
// test the LOOP, not the wire. StreamComplete bridges the two, so a caller
|
||||
// writes one line and gets streaming wherever it is available.
|
||||
type Streamer interface {
|
||||
// Stream calls the model, invoking onDelta with each fragment of assistant
|
||||
// text. Tool calls are NOT streamed: a partially-built argument object is a
|
||||
// different object from the finished one, and usually an invalid one.
|
||||
Stream(ctx context.Context, req Request, onDelta func(string)) (*Response, error)
|
||||
}
|
||||
|
||||
// Gateway is the model boundary.
|
||||
//
|
||||
// One method. A second implementation — a fake for tests, a recorded one for
|
||||
// evals — has one thing to satisfy, which is what keeps the eval harness from
|
||||
// needing a network.
|
||||
type Gateway interface {
|
||||
Complete(ctx context.Context, req Request) (*Response, error)
|
||||
}
|
||||
|
||||
// Validate checks a request before it costs anything.
|
||||
func (r Request) Validate() error {
|
||||
if len(r.Messages) == 0 {
|
||||
return &Error{Code: CodeInvalidRequest, Message: "a request needs at least one message"}
|
||||
}
|
||||
for i, m := range r.Messages {
|
||||
if m.Role != RoleUser && m.Role != RoleAssistant {
|
||||
return &Error{
|
||||
Code: CodeInvalidRequest,
|
||||
Message: fmt.Sprintf("messages[%d]: %q is not a role", i, m.Role),
|
||||
}
|
||||
}
|
||||
// A turn must say something, but "something" is text, tool calls or
|
||||
// tool results. A tool-result turn legitimately carries no text at all.
|
||||
if strings.TrimSpace(m.Text) == "" && len(m.ToolCalls) == 0 && len(m.ToolResults) == 0 {
|
||||
return &Error{
|
||||
Code: CodeInvalidRequest,
|
||||
Message: fmt.Sprintf("messages[%d]: a message cannot be empty", i),
|
||||
}
|
||||
}
|
||||
}
|
||||
for i, t := range r.Tools {
|
||||
if strings.TrimSpace(t.Name) == "" {
|
||||
return &Error{Code: CodeInvalidRequest, Message: fmt.Sprintf("tools[%d]: a tool needs a name", i)}
|
||||
}
|
||||
if strings.TrimSpace(t.Description) == "" {
|
||||
// The description is what the model reads instead of documentation.
|
||||
return &Error{Code: CodeInvalidRequest, Message: fmt.Sprintf("tools[%d]: %s has no description", i, t.Name)}
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
197
go-api/internal/gateway/gateway_test.go
Normal file
197
go-api/internal/gateway/gateway_test.go
Normal file
@@ -0,0 +1,197 @@
|
||||
package gateway
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/anthropics/anthropic-sdk-go"
|
||||
|
||||
"github.com/krow/krow-backend/go-api/internal/config"
|
||||
)
|
||||
|
||||
func TestParseTier(t *testing.T) {
|
||||
cases := []struct {
|
||||
in string
|
||||
want Tier
|
||||
known bool
|
||||
}{
|
||||
{"fast", TierFast, true},
|
||||
{"balanced", TierBalanced, true},
|
||||
{"deep", TierDeep, true},
|
||||
{" DEEP ", TierDeep, true},
|
||||
// Unset means the default, and is not a drift signal: most specs
|
||||
// simply do not declare a tier.
|
||||
{"", DefaultTier, true},
|
||||
// A tier that is not in the vocabulary still runs, at the default, but
|
||||
// reports itself so a drifted definition stays visible.
|
||||
{"thorough", DefaultTier, false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, known := ParseTier(c.in)
|
||||
if got != c.want || known != c.known {
|
||||
t.Errorf("ParseTier(%q) = (%q, %v), want (%q, %v)", c.in, got, known, c.want, c.known)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestUsageTotalCountsCacheReads(t *testing.T) {
|
||||
// A cache read is cheaper than fresh input, not free. Excluding it would
|
||||
// make the budget drift further from the truth the longer a run went on.
|
||||
u := Usage{InputTokens: 100, OutputTokens: 50, CacheReadTokens: 900, CacheCreationTokens: 10}
|
||||
if got := u.Total(); got != 1060 {
|
||||
t.Errorf("Total() = %d, want 1060", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRequestValidate(t *testing.T) {
|
||||
if err := (Request{}).Validate(); err == nil {
|
||||
t.Error("a request with no messages should be refused")
|
||||
}
|
||||
|
||||
blank := Request{Messages: []Message{{Role: RoleUser, Text: " "}}}
|
||||
if err := blank.Validate(); err == nil {
|
||||
t.Error("a whitespace-only message should be refused")
|
||||
}
|
||||
|
||||
bad := Request{Messages: []Message{{Role: "system", Text: "hi"}}}
|
||||
err := bad.Validate()
|
||||
var gwErr *Error
|
||||
if !errors.As(err, &gwErr) || gwErr.Code != CodeInvalidRequest {
|
||||
t.Errorf("a bad role should give CodeInvalidRequest, got %v", err)
|
||||
}
|
||||
|
||||
ok := Request{Messages: []Message{{Role: RoleUser, Text: "which shifts are uncovered?"}}}
|
||||
if err := ok.Validate(); err != nil {
|
||||
t.Errorf("a valid request was refused: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCompleteWithoutCredentialsIsStructured(t *testing.T) {
|
||||
// The service boots without a key on purpose. The failure has to arrive as
|
||||
// something a run can terminate with, not as a panic or a bare string.
|
||||
g := NewAnthropic(Config{})
|
||||
_, err := g.Complete(context.Background(), Request{
|
||||
Messages: []Message{{Role: RoleUser, Text: "anything"}},
|
||||
})
|
||||
|
||||
var gwErr *Error
|
||||
if !errors.As(err, &gwErr) {
|
||||
t.Fatalf("want a *gateway.Error, got %T: %v", err, err)
|
||||
}
|
||||
if gwErr.Code != CodeNotConfigured {
|
||||
t.Errorf("Code = %q, want %q", gwErr.Code, CodeNotConfigured)
|
||||
}
|
||||
if gwErr.Retryable() {
|
||||
t.Error("a missing key is not fixed by retrying")
|
||||
}
|
||||
}
|
||||
|
||||
func TestRetryable(t *testing.T) {
|
||||
cases := map[*Error]bool{
|
||||
{Code: CodeRateLimited}: true,
|
||||
{Code: CodeTimeout}: true,
|
||||
{Code: CodeUpstream, Status: 503}: true,
|
||||
{Code: CodeUpstream, Status: 400}: false,
|
||||
{Code: CodeUnauthorized, Status: 401}: false,
|
||||
{Code: CodeInvalidRequest}: false,
|
||||
// The one that matters: re-sending a request the model declined is how
|
||||
// a loop spends a whole budget on a single turn.
|
||||
{Code: CodeRefused, Category: "cyber"}: false,
|
||||
}
|
||||
for err, want := range cases {
|
||||
if got := err.Retryable(); got != want {
|
||||
t.Errorf("%s: Retryable() = %v, want %v", err.Code, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestFromConfigPinsEffortPerTier(t *testing.T) {
|
||||
cfg := FromConfig(config.ModelConfig{
|
||||
APIKey: "test", Fast: "m-fast", Balanced: "m-balanced", Deep: "m-deep",
|
||||
MaxOutputTokens: 8000,
|
||||
})
|
||||
|
||||
if cfg.Fast.Effort != anthropic.OutputConfigEffortLow {
|
||||
t.Errorf("fast effort = %q, want low", cfg.Fast.Effort)
|
||||
}
|
||||
if cfg.Balanced.Effort != anthropic.OutputConfigEffortHigh {
|
||||
t.Errorf("balanced effort = %q, want high", cfg.Balanced.Effort)
|
||||
}
|
||||
if cfg.Deep.Effort != anthropic.OutputConfigEffortXhigh {
|
||||
t.Errorf("deep effort = %q, want xhigh", cfg.Deep.Effort)
|
||||
}
|
||||
if cfg.MaxOutputTokens != 8000 {
|
||||
t.Errorf("MaxOutputTokens = %d, want 8000", cfg.MaxOutputTokens)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRoutingSelectsPerTier(t *testing.T) {
|
||||
g := NewAnthropic(Config{
|
||||
Fast: Routing{Model: "m-fast"},
|
||||
Balanced: Routing{Model: "m-balanced"},
|
||||
Deep: Routing{Model: "m-deep"},
|
||||
})
|
||||
|
||||
cases := map[Tier]string{
|
||||
TierFast: "m-fast",
|
||||
TierBalanced: "m-balanced",
|
||||
TierDeep: "m-deep",
|
||||
// A zero value routes to balanced rather than to an empty model id.
|
||||
Tier(""): "m-balanced",
|
||||
}
|
||||
for tier, want := range cases {
|
||||
if got := g.routing(tier).Model; got != want {
|
||||
t.Errorf("routing(%q) = %q, want %q", tier, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* ── Retrying what is worth retrying ────────────────────────────────────── */
|
||||
|
||||
func TestATransientOverloadIsWorthRetrying(t *testing.T) {
|
||||
// The classification this asserts existed from the start and had ZERO
|
||||
// callers, so a 529 killed runs that would have succeeded a moment later.
|
||||
// Found by a real overload during live testing.
|
||||
overloaded := &Error{Code: CodeUpstream, Message: "overloaded", Status: 529}
|
||||
if !overloaded.Retryable() {
|
||||
t.Error("a 529 overload should be retryable — it is the transient failure that actually happens")
|
||||
}
|
||||
|
||||
for _, e := range []*Error{
|
||||
{Code: CodeRateLimited, Status: 429},
|
||||
{Code: CodeTimeout},
|
||||
{Code: CodeUpstream, Status: 503},
|
||||
} {
|
||||
if !e.Retryable() {
|
||||
t.Errorf("%s (status %d) should be retryable", e.Code, e.Status)
|
||||
}
|
||||
}
|
||||
|
||||
// And the ones that will fail identically every time must not be.
|
||||
for _, e := range []*Error{
|
||||
{Code: CodeInvalidRequest, Status: 400},
|
||||
{Code: CodeUnauthorized, Status: 401},
|
||||
{Code: CodeNotConfigured},
|
||||
{Code: CodeRefused},
|
||||
} {
|
||||
if e.Retryable() {
|
||||
t.Errorf("%s should NOT be retryable — the same call will fail the same way", e.Code)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestAnUpstreamErrorNamesItsStatus(t *testing.T) {
|
||||
// "the model call failed" cost an hour of debugging, because the trajectory
|
||||
// records the message and the message did not say it was a 529. A failure
|
||||
// an operator cannot classify is a failure they cannot act on.
|
||||
e := &Error{
|
||||
Code: CodeUpstream,
|
||||
Message: "the model call failed (http 529)",
|
||||
Status: 529,
|
||||
}
|
||||
if !strings.Contains(e.Error(), "529") {
|
||||
t.Errorf("the rendered error hides its status: %s", e.Error())
|
||||
}
|
||||
}
|
||||
33
go-api/internal/gateway/routing.go
Normal file
33
go-api/internal/gateway/routing.go
Normal file
@@ -0,0 +1,33 @@
|
||||
package gateway
|
||||
|
||||
import (
|
||||
"github.com/anthropics/anthropic-sdk-go"
|
||||
|
||||
"github.com/krow/krow-backend/go-api/internal/config"
|
||||
)
|
||||
|
||||
// FromConfig builds the gateway's routing table from validated settings.
|
||||
//
|
||||
// The effort per tier is fixed here rather than configured, and that is the
|
||||
// point of the function existing at all: a deployment chooses *which model*
|
||||
// answers a tier, and the platform chooses *how hard it thinks*. If a
|
||||
// deployment could redefine effort, two installations running the same
|
||||
// definition would disagree about what "deep" means while both reporting the
|
||||
// tier as deep — and the tier is written into every trajectory.
|
||||
//
|
||||
// fast → low a lookup, a restatement, a short structured reading
|
||||
// balanced → high the default, and what most turns should cost
|
||||
// deep → xhigh a turn worth several tool calls and real deliberation
|
||||
//
|
||||
// `max` is deliberately not reachable from a spec. It is the setting for when
|
||||
// correctness matters more than cost, which is a judgement an operator makes
|
||||
// about a deployment, not one an agent author makes about a page.
|
||||
func FromConfig(c config.ModelConfig) Config {
|
||||
return Config{
|
||||
APIKey: c.APIKey,
|
||||
Fast: Routing{Model: c.Fast, Effort: anthropic.OutputConfigEffortLow},
|
||||
Balanced: Routing{Model: c.Balanced, Effort: anthropic.OutputConfigEffortHigh},
|
||||
Deep: Routing{Model: c.Deep, Effort: anthropic.OutputConfigEffortXhigh},
|
||||
MaxOutputTokens: int64(c.MaxOutputTokens),
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user