agent build

This commit is contained in:
2026-08-28 12:21:44 +05:30
parent b6f8655909
commit f7df96c973
138 changed files with 24164 additions and 207 deletions

View File

@@ -0,0 +1,477 @@
package gateway
import (
"context"
"encoding/json"
"errors"
"fmt"
"strings"
"time"
"github.com/anthropics/anthropic-sdk-go"
"github.com/anthropics/anthropic-sdk-go/option"
)
// Routing is how a tier becomes a model and an effort level.
//
// The model per tier is a deployment knob — a tenant on a different contract,
// or a deployment pinning a version through an incident, changes it without a
// spec edit. The *effort* per tier is not: "fast" and "deep" mean something
// specific about how much work an answer is worth, and letting a deployment
// redefine that would make the same spec behave differently in two places
// while claiming the same tier.
type Routing struct {
Model string
Effort anthropic.OutputConfigEffort
}
// Config is the gateway's whole configuration surface.
//
// Built once at startup from the environment and passed in frozen, per §10.
// Nothing in this package reads the environment itself.
type Config struct {
APIKey string
Fast Routing
Balanced Routing
Deep Routing
// MaxOutputTokens applies when a request does not set its own.
MaxOutputTokens int64
}
// AnthropicGateway calls the Claude API.
type AnthropicGateway struct {
client anthropic.Client
cfg Config
}
// Compile-time proof that this satisfies the boundary.
var _ Gateway = (*AnthropicGateway)(nil)
// NewAnthropic builds a gateway over the Claude API.
//
// A missing key is not an error here. The service has to boot without model
// credentials — every endpoint that is not an agent run still works, and a
// developer running migrations should not need a key to do it. The failure
// surfaces at the first Complete, as a structured NotConfigured that the
// runtime can end a run with, rather than as a panic at startup.
func NewAnthropic(cfg Config) *AnthropicGateway {
opts := []option.RequestOption{}
if cfg.APIKey != "" {
opts = append(opts, option.WithAPIKey(cfg.APIKey))
}
return &AnthropicGateway{client: anthropic.NewClient(opts...), cfg: cfg}
}
// routing resolves a tier. An unknown tier has already been normalised by
// ParseTier, so the default arm is reached only by a zero value.
func (g *AnthropicGateway) routing(t Tier) Routing {
switch t {
case TierFast:
return g.cfg.Fast
case TierDeep:
return g.cfg.Deep
default:
return g.cfg.Balanced
}
}
// Complete sends one request and reports one result.
// MaxAttempts is how many times a transient failure is retried.
//
// Three total, not three retries. Past that the problem is not transient and a
// fourth call is just spending money on the same answer.
const MaxAttempts = 3
// retryBackoff is the pause before each retry.
//
// Short, and deliberately so: this sits inside a run that already has a
// wall-clock deadline, and a backoff long enough to be polite to the API is
// long enough to spend the caller's whole budget waiting. A run that cannot
// afford the wait dies on its deadline instead, which is the correct failure.
var retryBackoff = []time.Duration{400 * time.Millisecond, 1200 * time.Millisecond}
// Complete calls the model, retrying failures that are worth retrying.
//
// THE RETRY IS NOT DEFENSIVE POLISH. Error.Retryable() has existed since this
// package was written and had ZERO callers — the classification was built and
// never used, so a 529 "overloaded" killed a run that would have succeeded four
// hundred milliseconds later. Found by a real overload during live testing,
// where it presented as "the agent could not finish" with nothing to act on.
//
// Only genuinely transient failures qualify: rate limits, timeouts, and 5xx.
// A 400 is a malformed request and will be malformed again; a 401 is a bad
// credential and retrying it three times just gets refused three times.
//
// The run's context governs. A retry that would outlive the caller's deadline
// does not happen — the deadline belongs to the run, not to this function, and
// waiting past it would turn a bounded run into an unbounded one.
func (g *AnthropicGateway) Complete(ctx context.Context, req Request) (*Response, error) {
var last error
for attempt := 0; attempt < MaxAttempts; attempt++ {
if attempt > 0 {
pause := retryBackoff[min(attempt-1, len(retryBackoff)-1)]
select {
case <-time.After(pause):
case <-ctx.Done():
// Out of time. The ORIGINAL failure is returned rather than the
// context error: "the model was overloaded" is what an operator
// needs to see, and "context deadline exceeded" would hide it.
return nil, last
}
}
resp, err := g.complete(ctx, req)
if err == nil {
return resp, nil
}
last = err
var gwErr *Error
if !errors.As(err, &gwErr) || !gwErr.Retryable() {
return resp, err
}
}
return nil, last
}
// complete is one attempt.
func (g *AnthropicGateway) complete(ctx context.Context, req Request) (*Response, error) {
if g.cfg.APIKey == "" {
return nil, &Error{
Code: CodeNotConfigured,
Message: "no model credentials are configured for this deployment",
}
}
if err := req.Validate(); err != nil {
return nil, err
}
params, err := g.params(req)
if err != nil {
return nil, err
}
msg, err := g.client.Messages.New(ctx, params)
if err != nil {
return nil, translate(err)
}
return g.decode(msg, req)
}
// params builds the request both paths send.
//
// Extracted so the streaming and non-streaming calls cannot drift. They send
// the same model, the same thinking config, the same cache breakpoint and the
// same tools — an answer that differs depending on whether it was streamed
// would be the worst kind of bug to chase, because the transport is the last
// place anybody looks.
func (g *AnthropicGateway) params(req Request) (anthropic.MessageNewParams, error) {
route := g.routing(req.Tier)
maxTokens := req.MaxOutputTokens
if maxTokens <= 0 {
maxTokens = g.cfg.MaxOutputTokens
}
messages, err := encodeMessages(req.Messages)
if err != nil {
return anthropic.MessageNewParams{}, err
}
params := anthropic.MessageNewParams{
Model: anthropic.Model(route.Model),
MaxTokens: maxTokens,
Messages: messages,
// Adaptive thinking on every tier: the model decides how much to think,
// and effort sets the ceiling on that. A fixed token budget for
// reasoning is the deprecated shape and is rejected outright by the
// current models.
Thinking: anthropic.ThinkingConfigParamUnion{
OfAdaptive: &anthropic.ThinkingConfigAdaptiveParam{},
},
OutputConfig: anthropic.OutputConfigParam{Effort: route.Effort},
}
if len(req.Tools) > 0 {
params.Tools = encodeTools(req.Tools)
}
if s := strings.TrimSpace(req.System); s != "" {
// One cached block. The system prompt is the stable prefix of every
// turn in a run, and the render order is tools → system → messages, so
// a breakpoint here is the one that survives the conversation growing.
params.System = []anthropic.TextBlockParam{{
Text: s,
CacheControl: anthropic.NewCacheControlEphemeralParam(),
}}
}
return params, nil
}
// decode turns a finished message into a Response.
//
// Shared by both paths for the same reason params() is: a streamed message and
// a non-streamed one are the same object by the time they get here, and reading
// them differently would make streaming a second implementation of the answer.
func (g *AnthropicGateway) decode(msg *anthropic.Message, req Request) (*Response, error) {
route := g.routing(req.Tier)
usage := Usage{
InputTokens: msg.Usage.InputTokens,
OutputTokens: msg.Usage.OutputTokens,
CacheReadTokens: msg.Usage.CacheReadInputTokens,
CacheCreationTokens: msg.Usage.CacheCreationInputTokens,
}
// A refusal arrives as a successful HTTP response, so it has to be checked
// before the content is read. It is still billed, and the usage is carried
// on the error so the run's budget is charged for a turn that produced no
// text — a refusal that cost nothing on the ledger is a refusal the loop
// would happily repeat.
if msg.StopReason == anthropic.StopReasonRefusal {
return &Response{
StopReason: string(msg.StopReason),
Usage: usage,
Model: route.Model,
Tier: req.Tier,
}, &Error{
Code: CodeRefused,
Message: "the model declined this request",
Category: string(msg.StopDetails.Category),
}
}
var (
text strings.Builder
calls []ToolCall
)
for _, block := range msg.Content {
switch b := block.AsAny().(type) {
case anthropic.TextBlock:
text.WriteString(b.Text)
case anthropic.ToolUseBlock:
// The raw JSON, not a parsed value: current models vary their
// string escaping inside tool inputs, so this is handed to the
// handler's own decoder rather than matched on as a string here.
calls = append(calls, ToolCall{
ID: b.ID,
Name: b.Name,
Input: json.RawMessage(b.JSON.Input.Raw()),
})
}
}
return &Response{
Text: text.String(),
ToolCalls: calls,
StopReason: string(msg.StopReason),
Usage: usage,
Model: route.Model,
Tier: req.Tier,
}, nil
}
// encodeTools renders the tool definitions for the wire.
func encodeTools(defs []ToolDef) []anthropic.ToolUnionParam {
out := make([]anthropic.ToolUnionParam, 0, len(defs))
for _, d := range defs {
schema := anthropic.ToolInputSchemaParam{}
if props, ok := d.InputSchema["properties"].(map[string]any); ok {
schema.Properties = props
}
if req, ok := d.InputSchema["required"].([]string); ok {
schema.Required = req
}
tool := anthropic.ToolParam{
Name: d.Name,
Description: anthropic.String(d.Description),
InputSchema: schema,
}
out = append(out, anthropic.ToolUnionParam{OfTool: &tool})
}
return out
}
// encodeMessages renders a conversation for the wire.
//
// Tool results are variadic within ONE user message. Splitting them across
// several messages is accepted by the API and quietly teaches the model to stop
// making parallel calls — a performance regression with no error to trace it
// to, so the grouping is done here rather than left to callers.
func encodeMessages(msgs []Message) ([]anthropic.MessageParam, error) {
out := make([]anthropic.MessageParam, 0, len(msgs))
for i, m := range msgs {
var blocks []anthropic.ContentBlockParamUnion
if s := strings.TrimSpace(m.Text); s != "" {
blocks = append(blocks, anthropic.NewTextBlock(m.Text))
}
for _, c := range m.ToolCalls {
var input any
if len(c.Input) > 0 {
if err := json.Unmarshal(c.Input, &input); err != nil {
return nil, &Error{
Code: CodeInvalidRequest,
Message: fmt.Sprintf("messages[%d]: tool call %s carries invalid JSON", i, c.Name),
}
}
}
blocks = append(blocks, anthropic.NewToolUseBlock(c.ID, input, c.Name))
}
for _, r := range m.ToolResults {
blocks = append(blocks, anthropic.NewToolResultBlock(r.CallID, r.Content, r.IsError))
}
if len(blocks) == 0 {
continue
}
if m.Role == RoleAssistant {
out = append(out, anthropic.NewAssistantMessage(blocks...))
continue
}
out = append(out, anthropic.NewUserMessage(blocks...))
}
return out, nil
}
// translate turns an SDK error into one the runtime can branch on.
//
// A single broad class would lose the distinction the loop actually needs:
// whether sending the same request again could work. So the status is read and
// mapped, and anything unrecognised stays CodeUpstream with its status intact
// rather than being flattened into a generic failure.
func translate(err error) error {
if errors.Is(err, context.DeadlineExceeded) || errors.Is(err, context.Canceled) {
return &Error{Code: CodeTimeout, Message: "the model call did not complete in time", Cause: err}
}
var apierr *anthropic.Error
if !errors.As(err, &apierr) {
return &Error{Code: CodeUpstream, Message: "the model call failed", Cause: err}
}
switch apierr.StatusCode {
case 400:
return &Error{Code: CodeInvalidRequest, Message: "the model rejected the request", Status: 400, Cause: err}
case 401, 403:
return &Error{Code: CodeUnauthorized, Message: "the model credentials were refused", Status: apierr.StatusCode, Cause: err}
case 408:
return &Error{Code: CodeTimeout, Message: "the model call timed out", Status: 408, Cause: err}
case 429:
return &Error{Code: CodeRateLimited, Message: "the model is rate limiting this deployment", Status: 429, Cause: err}
case 529:
// Anthropic's "overloaded" — the service is up and temporarily out of
// capacity. Named separately from the 500s because it is the one that
// actually happens, and because a run dying on it is a run that would
// have succeeded a second later.
return &Error{Code: CodeUpstream, Message: "the model is temporarily overloaded",
Status: 529, Cause: err}
default:
// The status is IN the message, not only in the field. It cost an hour
// of debugging to learn that "the model call failed" was a 529 rather
// than a malformed tool schema, and the trajectory only records the
// message.
return &Error{
Code: CodeUpstream,
Message: fmt.Sprintf("the model call failed (http %d)", apierr.StatusCode),
Status: apierr.StatusCode, Cause: err,
}
}
}
/* ── Streaming ──────────────────────────────────────────────────────────── */
// Stream is Complete, with the assistant's text delivered as it arrives.
//
// §6: "Stream partial assistant text as it arrives; buffer tool calls until
// complete." Both halves of that matter and they pull in opposite directions.
//
// TEXT IS STREAMED because a fifteen-second wait with nothing on screen reads
// as broken. The reader wants the first sentence while the rest is still being
// written, and that is the whole difference between a product and a spinner.
//
// TOOL CALLS ARE NOT. A tool call arrives as JSON assembled character by
// character across many events, and a half-built argument object is not a
// smaller version of the finished one — it is a different object, usually an
// invalid one. Dispatching on a partial call would run a tool with arguments
// the model had not finished choosing. So the accumulated message is decoded
// only once the stream closes, by exactly the same code the non-streaming path
// uses.
//
// onDelta is called from this goroutine, in order, and must not block for long
// — it is on the path between the model and the reader.
func (g *AnthropicGateway) Stream(ctx context.Context, req Request, onDelta func(string)) (*Response, error) {
if g.cfg.APIKey == "" {
return nil, &Error{
Code: CodeNotConfigured,
Message: "no model credentials are configured for this deployment",
}
}
if err := req.Validate(); err != nil {
return nil, err
}
params, err := g.params(req)
if err != nil {
return nil, err
}
stream := g.client.Messages.NewStreaming(ctx, params)
defer stream.Close()
var msg anthropic.Message
for stream.Next() {
event := stream.Current()
if err := msg.Accumulate(event); err != nil {
return nil, &Error{
Code: CodeUpstream,
Message: "the streamed response could not be assembled",
Cause: err,
}
}
// Text only. A thinking delta is the model's private reasoning and is
// not the answer; a tool-input delta is a fragment of JSON. Neither is
// something to put in front of a reader.
if event.Type == "content_block_delta" && event.Delta.Type == "text_delta" {
if d := event.Delta.Text; d != "" && onDelta != nil {
onDelta(d)
}
}
}
if err := stream.Err(); err != nil {
return nil, translate(err)
}
return g.decode(&msg, req)
}
// StreamComplete runs a request through whichever path the gateway supports.
//
// A gateway that cannot stream is not a broken gateway — every fake in the test
// suite is one, and so is any future provider without a streaming API. Falling
// back to Complete and delivering the finished text as a single delta keeps the
// caller's code identical either way, which is what stops streaming from
// becoming a second code path through the loop.
func StreamComplete(ctx context.Context, gw Gateway, req Request, onDelta func(string)) (*Response, error) {
// Normalised once, here, so no implementation has to guard it. A caller
// that does not want deltas passes nil — every eval and every test does —
// and an implementation that took that literally would panic on the first
// fragment. Making each Streamer remember the check is how one of them
// eventually forgets.
if onDelta == nil {
onDelta = func(string) {}
}
if s, ok := gw.(Streamer); ok {
return s.Stream(ctx, req, onDelta)
}
resp, err := gw.Complete(ctx, req)
if err == nil && resp != nil && resp.Text != "" && onDelta != nil {
onDelta(resp.Text)
}
return resp, err
}

View File

@@ -0,0 +1,294 @@
// Package gateway is the model gateway: the one place in this service that
// talks to a language model.
//
// Everything else — the runtime loop, the tool layer, retrieval — reaches a
// model through this package and nowhere else. That is the whole point of it
// being a layer rather than a helper:
//
// - **Routing lives here.** An agent spec declares a `reasoning` tier, not a
// model id. Which model and how much thinking that tier buys is a
// deployment decision, and it changes without touching a single spec.
// - **Token accounting lives here.** Every call returns what it cost. A
// budget the runtime cannot measure is a budget it cannot enforce, and
// I3 requires it to enforce one.
// - **Refusal is an outcome, not an exception.** A model that declines comes
// back as a structured Refused, which is one of the six termination
// reasons the runtime already knows how to end a run with.
//
// What this package deliberately does *not* do: assemble prompts, decide what
// a caller may read, or loop. It sends one request and reports one result.
// Composition is the runtime's job and authorization is the tool layer's, and
// folding either of them in here would put policy behind a transport.
package gateway
import (
"context"
"encoding/json"
"fmt"
"strings"
)
// Tier is an agent spec's `reasoning` value.
//
// Three tiers, because an author choosing between "fast" and "deep" is making
// a judgement about the work, not about a model. The mapping from a tier to a
// model and an effort level is this package's business and is configured per
// deployment — a spec that named a model directly would pin every tenant to
// whatever was current the day it was written.
type Tier string
const (
TierFast Tier = "fast"
TierBalanced Tier = "balanced"
TierDeep Tier = "deep"
)
// DefaultTier is what a spec that declares no reasoning mode gets. It matches
// the frontend vocabulary's own default, so a definition means the same thing
// on both sides of the wire.
const DefaultTier = TierBalanced
// ParseTier resolves a spec's declared reasoning value.
//
// An unrecognised tier falls back rather than failing: the tier affects how
// much a turn costs, never whether it is allowed, so refusing the run would
// turn a typo in a definition into an outage. The caller is told, so a
// definition that has drifted from the vocabulary is still visible.
func ParseTier(raw string) (Tier, bool) {
switch Tier(strings.ToLower(strings.TrimSpace(raw))) {
case TierFast:
return TierFast, true
case TierBalanced:
return TierBalanced, true
case TierDeep:
return TierDeep, true
case "":
return DefaultTier, true
default:
return DefaultTier, false
}
}
// Role is who said something.
type Role string
const (
RoleUser Role = "user"
RoleAssistant Role = "assistant"
)
// ToolCall is the model asking for a tool to be run.
type ToolCall struct {
// ID correlates the call with its result. Echoed back verbatim: it is the
// model's own handle, and a result carrying a different one is a result
// attached to the wrong question.
ID string
Name string
Input json.RawMessage
}
// ToolResult is what came back, on its way to the model.
//
// Content is a string because that is what crosses the wire, but it carries
// encoded structured data — §4 keeps formatting the model's job, so a handler
// never writes prose and this never carries any.
type ToolResult struct {
CallID string
Content string
IsError bool
}
// Message is one turn of a conversation.
//
// A turn is text, or tool calls, or tool results — an assistant turn may carry
// text and calls together, which is why these are fields rather than a union.
type Message struct {
Role Role
Text string
ToolCalls []ToolCall
ToolResults []ToolResult
}
// ToolDef is a tool as the model sees it.
//
// Deliberately not the tool layer's own type. The gateway must not import the
// tool package: a model provider knowing what an `effect` or a confirmation
// token is would put policy behind a transport, and the confirmation gate has
// to sit where the model cannot reach it.
type ToolDef struct {
Name string
Description string
InputSchema map[string]any
}
// Request is one model call.
type Request struct {
// Tier selects the model and effort. From the agent spec.
Tier Tier
// System is the assembled system prompt.
//
// I7: retrieved document text must never reach this field. Retrieved
// content belongs in a delimited context block inside a user message,
// where the system prompt has already said that its contents are data.
// Nothing here can enforce that — it is a property of what the runtime
// passes — so it is stated where the field is declared.
System string
// Messages is the conversation so far, oldest first.
Messages []Message
// Tools the model may call this turn. Order matters: it is part of the
// cached prefix, so the caller sorts it once and keeps it stable.
Tools []ToolDef
// MaxOutputTokens caps this response. Zero takes the configured default.
//
// This is a hard ceiling the model is not aware of, so it truncates rather
// than winding down. It is not the run's token budget — that is the
// runtime's, and it spans every call in a run.
MaxOutputTokens int64
}
// Usage is what a call cost.
type Usage struct {
InputTokens int64
OutputTokens int64
CacheReadTokens int64
CacheCreationTokens int64
}
// Total is every token this call is billed for.
//
// Cache reads are counted: they are cheaper than fresh input, not free, and a
// budget that ignored them would drift further from the truth the longer a
// conversation ran — which is exactly when it matters most.
func (u Usage) Total() int64 {
return u.InputTokens + u.OutputTokens + u.CacheReadTokens + u.CacheCreationTokens
}
// Response is one model reply.
type Response struct {
Text string
// ToolCalls the model wants run before it can continue. Non-empty exactly
// when StopReason is "tool_use".
ToolCalls []ToolCall
StopReason string
Usage Usage
// Model is the id actually used, not the tier that was asked for. Logged
// with every run so a change of routing is visible in the trajectory
// rather than inferred from a deploy date.
Model string
Tier Tier
}
// Error codes. Structured rather than bare strings, per §10 — user-facing text
// is derived at the surface layer, never raised from here.
const (
CodeNotConfigured = "gateway.not_configured"
CodeInvalidRequest = "gateway.invalid_request"
CodeUnauthorized = "gateway.unauthorized"
CodeRateLimited = "gateway.rate_limited"
CodeTimeout = "gateway.timeout"
CodeRefused = "gateway.refused"
CodeUpstream = "gateway.upstream"
)
// Error is a gateway failure with a code the runtime can branch on.
type Error struct {
Code string
Message string
// Status is the upstream HTTP status, when there was one.
Status int
// Category carries a refusal's reason when Code is CodeRefused. An open
// set upstream, so it is a string and is never switched on exhaustively.
Category string
Cause error
}
func (e *Error) Error() string {
if e.Status != 0 {
return fmt.Sprintf("%s: %s (http %d)", e.Code, e.Message, e.Status)
}
return fmt.Sprintf("%s: %s", e.Code, e.Message)
}
func (e *Error) Unwrap() error { return e.Cause }
// Retryable reports whether the same request could succeed if sent again.
//
// The runtime needs this to decide between a retry and a terminal
// ToolFailure. A refusal is emphatically not retryable — re-sending a request
// the model declined is how a loop burns a whole budget on one turn.
func (e *Error) Retryable() bool {
switch e.Code {
case CodeRateLimited, CodeTimeout:
return true
case CodeUpstream:
return e.Status >= 500
default:
return false
}
}
// Streamer is a Gateway that can deliver text as it arrives.
//
// A SEPARATE interface, not a method on Gateway, and that is deliberate. Adding
// Stream to Gateway would break every fake in the test suite and force each one
// to implement a transport it does not care about — and those fakes exist to
// test the LOOP, not the wire. StreamComplete bridges the two, so a caller
// writes one line and gets streaming wherever it is available.
type Streamer interface {
// Stream calls the model, invoking onDelta with each fragment of assistant
// text. Tool calls are NOT streamed: a partially-built argument object is a
// different object from the finished one, and usually an invalid one.
Stream(ctx context.Context, req Request, onDelta func(string)) (*Response, error)
}
// Gateway is the model boundary.
//
// One method. A second implementation — a fake for tests, a recorded one for
// evals — has one thing to satisfy, which is what keeps the eval harness from
// needing a network.
type Gateway interface {
Complete(ctx context.Context, req Request) (*Response, error)
}
// Validate checks a request before it costs anything.
func (r Request) Validate() error {
if len(r.Messages) == 0 {
return &Error{Code: CodeInvalidRequest, Message: "a request needs at least one message"}
}
for i, m := range r.Messages {
if m.Role != RoleUser && m.Role != RoleAssistant {
return &Error{
Code: CodeInvalidRequest,
Message: fmt.Sprintf("messages[%d]: %q is not a role", i, m.Role),
}
}
// A turn must say something, but "something" is text, tool calls or
// tool results. A tool-result turn legitimately carries no text at all.
if strings.TrimSpace(m.Text) == "" && len(m.ToolCalls) == 0 && len(m.ToolResults) == 0 {
return &Error{
Code: CodeInvalidRequest,
Message: fmt.Sprintf("messages[%d]: a message cannot be empty", i),
}
}
}
for i, t := range r.Tools {
if strings.TrimSpace(t.Name) == "" {
return &Error{Code: CodeInvalidRequest, Message: fmt.Sprintf("tools[%d]: a tool needs a name", i)}
}
if strings.TrimSpace(t.Description) == "" {
// The description is what the model reads instead of documentation.
return &Error{Code: CodeInvalidRequest, Message: fmt.Sprintf("tools[%d]: %s has no description", i, t.Name)}
}
}
return nil
}

View File

@@ -0,0 +1,197 @@
package gateway
import (
"context"
"errors"
"strings"
"testing"
"github.com/anthropics/anthropic-sdk-go"
"github.com/krow/krow-backend/go-api/internal/config"
)
func TestParseTier(t *testing.T) {
cases := []struct {
in string
want Tier
known bool
}{
{"fast", TierFast, true},
{"balanced", TierBalanced, true},
{"deep", TierDeep, true},
{" DEEP ", TierDeep, true},
// Unset means the default, and is not a drift signal: most specs
// simply do not declare a tier.
{"", DefaultTier, true},
// A tier that is not in the vocabulary still runs, at the default, but
// reports itself so a drifted definition stays visible.
{"thorough", DefaultTier, false},
}
for _, c := range cases {
got, known := ParseTier(c.in)
if got != c.want || known != c.known {
t.Errorf("ParseTier(%q) = (%q, %v), want (%q, %v)", c.in, got, known, c.want, c.known)
}
}
}
func TestUsageTotalCountsCacheReads(t *testing.T) {
// A cache read is cheaper than fresh input, not free. Excluding it would
// make the budget drift further from the truth the longer a run went on.
u := Usage{InputTokens: 100, OutputTokens: 50, CacheReadTokens: 900, CacheCreationTokens: 10}
if got := u.Total(); got != 1060 {
t.Errorf("Total() = %d, want 1060", got)
}
}
func TestRequestValidate(t *testing.T) {
if err := (Request{}).Validate(); err == nil {
t.Error("a request with no messages should be refused")
}
blank := Request{Messages: []Message{{Role: RoleUser, Text: " "}}}
if err := blank.Validate(); err == nil {
t.Error("a whitespace-only message should be refused")
}
bad := Request{Messages: []Message{{Role: "system", Text: "hi"}}}
err := bad.Validate()
var gwErr *Error
if !errors.As(err, &gwErr) || gwErr.Code != CodeInvalidRequest {
t.Errorf("a bad role should give CodeInvalidRequest, got %v", err)
}
ok := Request{Messages: []Message{{Role: RoleUser, Text: "which shifts are uncovered?"}}}
if err := ok.Validate(); err != nil {
t.Errorf("a valid request was refused: %v", err)
}
}
func TestCompleteWithoutCredentialsIsStructured(t *testing.T) {
// The service boots without a key on purpose. The failure has to arrive as
// something a run can terminate with, not as a panic or a bare string.
g := NewAnthropic(Config{})
_, err := g.Complete(context.Background(), Request{
Messages: []Message{{Role: RoleUser, Text: "anything"}},
})
var gwErr *Error
if !errors.As(err, &gwErr) {
t.Fatalf("want a *gateway.Error, got %T: %v", err, err)
}
if gwErr.Code != CodeNotConfigured {
t.Errorf("Code = %q, want %q", gwErr.Code, CodeNotConfigured)
}
if gwErr.Retryable() {
t.Error("a missing key is not fixed by retrying")
}
}
func TestRetryable(t *testing.T) {
cases := map[*Error]bool{
{Code: CodeRateLimited}: true,
{Code: CodeTimeout}: true,
{Code: CodeUpstream, Status: 503}: true,
{Code: CodeUpstream, Status: 400}: false,
{Code: CodeUnauthorized, Status: 401}: false,
{Code: CodeInvalidRequest}: false,
// The one that matters: re-sending a request the model declined is how
// a loop spends a whole budget on a single turn.
{Code: CodeRefused, Category: "cyber"}: false,
}
for err, want := range cases {
if got := err.Retryable(); got != want {
t.Errorf("%s: Retryable() = %v, want %v", err.Code, got, want)
}
}
}
func TestFromConfigPinsEffortPerTier(t *testing.T) {
cfg := FromConfig(config.ModelConfig{
APIKey: "test", Fast: "m-fast", Balanced: "m-balanced", Deep: "m-deep",
MaxOutputTokens: 8000,
})
if cfg.Fast.Effort != anthropic.OutputConfigEffortLow {
t.Errorf("fast effort = %q, want low", cfg.Fast.Effort)
}
if cfg.Balanced.Effort != anthropic.OutputConfigEffortHigh {
t.Errorf("balanced effort = %q, want high", cfg.Balanced.Effort)
}
if cfg.Deep.Effort != anthropic.OutputConfigEffortXhigh {
t.Errorf("deep effort = %q, want xhigh", cfg.Deep.Effort)
}
if cfg.MaxOutputTokens != 8000 {
t.Errorf("MaxOutputTokens = %d, want 8000", cfg.MaxOutputTokens)
}
}
func TestRoutingSelectsPerTier(t *testing.T) {
g := NewAnthropic(Config{
Fast: Routing{Model: "m-fast"},
Balanced: Routing{Model: "m-balanced"},
Deep: Routing{Model: "m-deep"},
})
cases := map[Tier]string{
TierFast: "m-fast",
TierBalanced: "m-balanced",
TierDeep: "m-deep",
// A zero value routes to balanced rather than to an empty model id.
Tier(""): "m-balanced",
}
for tier, want := range cases {
if got := g.routing(tier).Model; got != want {
t.Errorf("routing(%q) = %q, want %q", tier, got, want)
}
}
}
/* ── Retrying what is worth retrying ────────────────────────────────────── */
func TestATransientOverloadIsWorthRetrying(t *testing.T) {
// The classification this asserts existed from the start and had ZERO
// callers, so a 529 killed runs that would have succeeded a moment later.
// Found by a real overload during live testing.
overloaded := &Error{Code: CodeUpstream, Message: "overloaded", Status: 529}
if !overloaded.Retryable() {
t.Error("a 529 overload should be retryable — it is the transient failure that actually happens")
}
for _, e := range []*Error{
{Code: CodeRateLimited, Status: 429},
{Code: CodeTimeout},
{Code: CodeUpstream, Status: 503},
} {
if !e.Retryable() {
t.Errorf("%s (status %d) should be retryable", e.Code, e.Status)
}
}
// And the ones that will fail identically every time must not be.
for _, e := range []*Error{
{Code: CodeInvalidRequest, Status: 400},
{Code: CodeUnauthorized, Status: 401},
{Code: CodeNotConfigured},
{Code: CodeRefused},
} {
if e.Retryable() {
t.Errorf("%s should NOT be retryable — the same call will fail the same way", e.Code)
}
}
}
func TestAnUpstreamErrorNamesItsStatus(t *testing.T) {
// "the model call failed" cost an hour of debugging, because the trajectory
// records the message and the message did not say it was a 529. A failure
// an operator cannot classify is a failure they cannot act on.
e := &Error{
Code: CodeUpstream,
Message: "the model call failed (http 529)",
Status: 529,
}
if !strings.Contains(e.Error(), "529") {
t.Errorf("the rendered error hides its status: %s", e.Error())
}
}

View File

@@ -0,0 +1,33 @@
package gateway
import (
"github.com/anthropics/anthropic-sdk-go"
"github.com/krow/krow-backend/go-api/internal/config"
)
// FromConfig builds the gateway's routing table from validated settings.
//
// The effort per tier is fixed here rather than configured, and that is the
// point of the function existing at all: a deployment chooses *which model*
// answers a tier, and the platform chooses *how hard it thinks*. If a
// deployment could redefine effort, two installations running the same
// definition would disagree about what "deep" means while both reporting the
// tier as deep — and the tier is written into every trajectory.
//
// fast → low a lookup, a restatement, a short structured reading
// balanced → high the default, and what most turns should cost
// deep → xhigh a turn worth several tool calls and real deliberation
//
// `max` is deliberately not reachable from a spec. It is the setting for when
// correctness matters more than cost, which is a judgement an operator makes
// about a deployment, not one an agent author makes about a page.
func FromConfig(c config.ModelConfig) Config {
return Config{
APIKey: c.APIKey,
Fast: Routing{Model: c.Fast, Effort: anthropic.OutputConfigEffortLow},
Balanced: Routing{Model: c.Balanced, Effort: anthropic.OutputConfigEffortHigh},
Deep: Routing{Model: c.Deep, Effort: anthropic.OutputConfigEffortXhigh},
MaxOutputTokens: int64(c.MaxOutputTokens),
}
}