295 lines
10 KiB
Go
295 lines
10 KiB
Go
// Package gateway is the model gateway: the one place in this service that
|
|
// talks to a language model.
|
|
//
|
|
// Everything else — the runtime loop, the tool layer, retrieval — reaches a
|
|
// model through this package and nowhere else. That is the whole point of it
|
|
// being a layer rather than a helper:
|
|
//
|
|
// - **Routing lives here.** An agent spec declares a `reasoning` tier, not a
|
|
// model id. Which model and how much thinking that tier buys is a
|
|
// deployment decision, and it changes without touching a single spec.
|
|
// - **Token accounting lives here.** Every call returns what it cost. A
|
|
// budget the runtime cannot measure is a budget it cannot enforce, and
|
|
// I3 requires it to enforce one.
|
|
// - **Refusal is an outcome, not an exception.** A model that declines comes
|
|
// back as a structured Refused, which is one of the six termination
|
|
// reasons the runtime already knows how to end a run with.
|
|
//
|
|
// What this package deliberately does *not* do: assemble prompts, decide what
|
|
// a caller may read, or loop. It sends one request and reports one result.
|
|
// Composition is the runtime's job and authorization is the tool layer's, and
|
|
// folding either of them in here would put policy behind a transport.
|
|
package gateway
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"strings"
|
|
)
|
|
|
|
// Tier is an agent spec's `reasoning` value.
|
|
//
|
|
// Three tiers, because an author choosing between "fast" and "deep" is making
|
|
// a judgement about the work, not about a model. The mapping from a tier to a
|
|
// model and an effort level is this package's business and is configured per
|
|
// deployment — a spec that named a model directly would pin every tenant to
|
|
// whatever was current the day it was written.
|
|
type Tier string
|
|
|
|
const (
|
|
TierFast Tier = "fast"
|
|
TierBalanced Tier = "balanced"
|
|
TierDeep Tier = "deep"
|
|
)
|
|
|
|
// DefaultTier is what a spec that declares no reasoning mode gets. It matches
|
|
// the frontend vocabulary's own default, so a definition means the same thing
|
|
// on both sides of the wire.
|
|
const DefaultTier = TierBalanced
|
|
|
|
// ParseTier resolves a spec's declared reasoning value.
|
|
//
|
|
// An unrecognised tier falls back rather than failing: the tier affects how
|
|
// much a turn costs, never whether it is allowed, so refusing the run would
|
|
// turn a typo in a definition into an outage. The caller is told, so a
|
|
// definition that has drifted from the vocabulary is still visible.
|
|
func ParseTier(raw string) (Tier, bool) {
|
|
switch Tier(strings.ToLower(strings.TrimSpace(raw))) {
|
|
case TierFast:
|
|
return TierFast, true
|
|
case TierBalanced:
|
|
return TierBalanced, true
|
|
case TierDeep:
|
|
return TierDeep, true
|
|
case "":
|
|
return DefaultTier, true
|
|
default:
|
|
return DefaultTier, false
|
|
}
|
|
}
|
|
|
|
// Role is who said something.
|
|
type Role string
|
|
|
|
const (
|
|
RoleUser Role = "user"
|
|
RoleAssistant Role = "assistant"
|
|
)
|
|
|
|
// ToolCall is the model asking for a tool to be run.
|
|
type ToolCall struct {
|
|
// ID correlates the call with its result. Echoed back verbatim: it is the
|
|
// model's own handle, and a result carrying a different one is a result
|
|
// attached to the wrong question.
|
|
ID string
|
|
Name string
|
|
Input json.RawMessage
|
|
}
|
|
|
|
// ToolResult is what came back, on its way to the model.
|
|
//
|
|
// Content is a string because that is what crosses the wire, but it carries
|
|
// encoded structured data — §4 keeps formatting the model's job, so a handler
|
|
// never writes prose and this never carries any.
|
|
type ToolResult struct {
|
|
CallID string
|
|
Content string
|
|
IsError bool
|
|
}
|
|
|
|
// Message is one turn of a conversation.
|
|
//
|
|
// A turn is text, or tool calls, or tool results — an assistant turn may carry
|
|
// text and calls together, which is why these are fields rather than a union.
|
|
type Message struct {
|
|
Role Role
|
|
Text string
|
|
ToolCalls []ToolCall
|
|
ToolResults []ToolResult
|
|
}
|
|
|
|
// ToolDef is a tool as the model sees it.
|
|
//
|
|
// Deliberately not the tool layer's own type. The gateway must not import the
|
|
// tool package: a model provider knowing what an `effect` or a confirmation
|
|
// token is would put policy behind a transport, and the confirmation gate has
|
|
// to sit where the model cannot reach it.
|
|
type ToolDef struct {
|
|
Name string
|
|
Description string
|
|
InputSchema map[string]any
|
|
}
|
|
|
|
// Request is one model call.
|
|
type Request struct {
|
|
// Tier selects the model and effort. From the agent spec.
|
|
Tier Tier
|
|
|
|
// System is the assembled system prompt.
|
|
//
|
|
// I7: retrieved document text must never reach this field. Retrieved
|
|
// content belongs in a delimited context block inside a user message,
|
|
// where the system prompt has already said that its contents are data.
|
|
// Nothing here can enforce that — it is a property of what the runtime
|
|
// passes — so it is stated where the field is declared.
|
|
System string
|
|
|
|
// Messages is the conversation so far, oldest first.
|
|
Messages []Message
|
|
|
|
// Tools the model may call this turn. Order matters: it is part of the
|
|
// cached prefix, so the caller sorts it once and keeps it stable.
|
|
Tools []ToolDef
|
|
|
|
// MaxOutputTokens caps this response. Zero takes the configured default.
|
|
//
|
|
// This is a hard ceiling the model is not aware of, so it truncates rather
|
|
// than winding down. It is not the run's token budget — that is the
|
|
// runtime's, and it spans every call in a run.
|
|
MaxOutputTokens int64
|
|
}
|
|
|
|
// Usage is what a call cost.
|
|
type Usage struct {
|
|
InputTokens int64
|
|
OutputTokens int64
|
|
CacheReadTokens int64
|
|
CacheCreationTokens int64
|
|
}
|
|
|
|
// Total is every token this call is billed for.
|
|
//
|
|
// Cache reads are counted: they are cheaper than fresh input, not free, and a
|
|
// budget that ignored them would drift further from the truth the longer a
|
|
// conversation ran — which is exactly when it matters most.
|
|
func (u Usage) Total() int64 {
|
|
return u.InputTokens + u.OutputTokens + u.CacheReadTokens + u.CacheCreationTokens
|
|
}
|
|
|
|
// Response is one model reply.
|
|
type Response struct {
|
|
Text string
|
|
|
|
// ToolCalls the model wants run before it can continue. Non-empty exactly
|
|
// when StopReason is "tool_use".
|
|
ToolCalls []ToolCall
|
|
|
|
StopReason string
|
|
Usage Usage
|
|
|
|
// Model is the id actually used, not the tier that was asked for. Logged
|
|
// with every run so a change of routing is visible in the trajectory
|
|
// rather than inferred from a deploy date.
|
|
Model string
|
|
Tier Tier
|
|
}
|
|
|
|
// Error codes. Structured rather than bare strings, per §10 — user-facing text
|
|
// is derived at the surface layer, never raised from here.
|
|
const (
|
|
CodeNotConfigured = "gateway.not_configured"
|
|
CodeInvalidRequest = "gateway.invalid_request"
|
|
CodeUnauthorized = "gateway.unauthorized"
|
|
CodeRateLimited = "gateway.rate_limited"
|
|
CodeTimeout = "gateway.timeout"
|
|
CodeRefused = "gateway.refused"
|
|
CodeUpstream = "gateway.upstream"
|
|
)
|
|
|
|
// Error is a gateway failure with a code the runtime can branch on.
|
|
type Error struct {
|
|
Code string
|
|
Message string
|
|
|
|
// Status is the upstream HTTP status, when there was one.
|
|
Status int
|
|
|
|
// Category carries a refusal's reason when Code is CodeRefused. An open
|
|
// set upstream, so it is a string and is never switched on exhaustively.
|
|
Category string
|
|
|
|
Cause error
|
|
}
|
|
|
|
func (e *Error) Error() string {
|
|
if e.Status != 0 {
|
|
return fmt.Sprintf("%s: %s (http %d)", e.Code, e.Message, e.Status)
|
|
}
|
|
return fmt.Sprintf("%s: %s", e.Code, e.Message)
|
|
}
|
|
|
|
func (e *Error) Unwrap() error { return e.Cause }
|
|
|
|
// Retryable reports whether the same request could succeed if sent again.
|
|
//
|
|
// The runtime needs this to decide between a retry and a terminal
|
|
// ToolFailure. A refusal is emphatically not retryable — re-sending a request
|
|
// the model declined is how a loop burns a whole budget on one turn.
|
|
func (e *Error) Retryable() bool {
|
|
switch e.Code {
|
|
case CodeRateLimited, CodeTimeout:
|
|
return true
|
|
case CodeUpstream:
|
|
return e.Status >= 500
|
|
default:
|
|
return false
|
|
}
|
|
}
|
|
|
|
// Streamer is a Gateway that can deliver text as it arrives.
|
|
//
|
|
// A SEPARATE interface, not a method on Gateway, and that is deliberate. Adding
|
|
// Stream to Gateway would break every fake in the test suite and force each one
|
|
// to implement a transport it does not care about — and those fakes exist to
|
|
// test the LOOP, not the wire. StreamComplete bridges the two, so a caller
|
|
// writes one line and gets streaming wherever it is available.
|
|
type Streamer interface {
|
|
// Stream calls the model, invoking onDelta with each fragment of assistant
|
|
// text. Tool calls are NOT streamed: a partially-built argument object is a
|
|
// different object from the finished one, and usually an invalid one.
|
|
Stream(ctx context.Context, req Request, onDelta func(string)) (*Response, error)
|
|
}
|
|
|
|
// Gateway is the model boundary.
|
|
//
|
|
// One method. A second implementation — a fake for tests, a recorded one for
|
|
// evals — has one thing to satisfy, which is what keeps the eval harness from
|
|
// needing a network.
|
|
type Gateway interface {
|
|
Complete(ctx context.Context, req Request) (*Response, error)
|
|
}
|
|
|
|
// Validate checks a request before it costs anything.
|
|
func (r Request) Validate() error {
|
|
if len(r.Messages) == 0 {
|
|
return &Error{Code: CodeInvalidRequest, Message: "a request needs at least one message"}
|
|
}
|
|
for i, m := range r.Messages {
|
|
if m.Role != RoleUser && m.Role != RoleAssistant {
|
|
return &Error{
|
|
Code: CodeInvalidRequest,
|
|
Message: fmt.Sprintf("messages[%d]: %q is not a role", i, m.Role),
|
|
}
|
|
}
|
|
// A turn must say something, but "something" is text, tool calls or
|
|
// tool results. A tool-result turn legitimately carries no text at all.
|
|
if strings.TrimSpace(m.Text) == "" && len(m.ToolCalls) == 0 && len(m.ToolResults) == 0 {
|
|
return &Error{
|
|
Code: CodeInvalidRequest,
|
|
Message: fmt.Sprintf("messages[%d]: a message cannot be empty", i),
|
|
}
|
|
}
|
|
}
|
|
for i, t := range r.Tools {
|
|
if strings.TrimSpace(t.Name) == "" {
|
|
return &Error{Code: CodeInvalidRequest, Message: fmt.Sprintf("tools[%d]: a tool needs a name", i)}
|
|
}
|
|
if strings.TrimSpace(t.Description) == "" {
|
|
// The description is what the model reads instead of documentation.
|
|
return &Error{Code: CodeInvalidRequest, Message: fmt.Sprintf("tools[%d]: %s has no description", i, t.Name)}
|
|
}
|
|
}
|
|
return nil
|
|
}
|