// Package gateway is the model gateway: the one place in this service that // talks to a language model. // // Everything else — the runtime loop, the tool layer, retrieval — reaches a // model through this package and nowhere else. That is the whole point of it // being a layer rather than a helper: // // - **Routing lives here.** An agent spec declares a `reasoning` tier, not a // model id. Which model and how much thinking that tier buys is a // deployment decision, and it changes without touching a single spec. // - **Token accounting lives here.** Every call returns what it cost. A // budget the runtime cannot measure is a budget it cannot enforce, and // I3 requires it to enforce one. // - **Refusal is an outcome, not an exception.** A model that declines comes // back as a structured Refused, which is one of the six termination // reasons the runtime already knows how to end a run with. // // What this package deliberately does *not* do: assemble prompts, decide what // a caller may read, or loop. It sends one request and reports one result. // Composition is the runtime's job and authorization is the tool layer's, and // folding either of them in here would put policy behind a transport. package gateway import ( "context" "encoding/json" "fmt" "strings" ) // Tier is an agent spec's `reasoning` value. // // Three tiers, because an author choosing between "fast" and "deep" is making // a judgement about the work, not about a model. The mapping from a tier to a // model and an effort level is this package's business and is configured per // deployment — a spec that named a model directly would pin every tenant to // whatever was current the day it was written. type Tier string const ( TierFast Tier = "fast" TierBalanced Tier = "balanced" TierDeep Tier = "deep" ) // DefaultTier is what a spec that declares no reasoning mode gets. It matches // the frontend vocabulary's own default, so a definition means the same thing // on both sides of the wire. const DefaultTier = TierBalanced // ParseTier resolves a spec's declared reasoning value. // // An unrecognised tier falls back rather than failing: the tier affects how // much a turn costs, never whether it is allowed, so refusing the run would // turn a typo in a definition into an outage. The caller is told, so a // definition that has drifted from the vocabulary is still visible. func ParseTier(raw string) (Tier, bool) { switch Tier(strings.ToLower(strings.TrimSpace(raw))) { case TierFast: return TierFast, true case TierBalanced: return TierBalanced, true case TierDeep: return TierDeep, true case "": return DefaultTier, true default: return DefaultTier, false } } // Role is who said something. type Role string const ( RoleUser Role = "user" RoleAssistant Role = "assistant" ) // ToolCall is the model asking for a tool to be run. type ToolCall struct { // ID correlates the call with its result. Echoed back verbatim: it is the // model's own handle, and a result carrying a different one is a result // attached to the wrong question. ID string Name string Input json.RawMessage // Extra is provider metadata attached to the call, carried back to the // provider verbatim on the next turn and never read here. // // It exists because at least one provider requires it. Gemini 3 models // attach a "thought signature" to every function call and REJECT the // follow-up request — 400, "Function call is missing a thought_signature" // — if the assistant message that echoes the call does not carry it back. // A gateway that rebuilds the assistant turn from ID, Name and Input alone // drops it, and every tool-using run dies on its second model call while // the first one looked perfectly healthy. That is exactly what happened // on 2026-09-22 when production was pointed at Gemini. // // The gateway does not know what is in it and must not: the whole point // of speaking one wire shape is that a vendor's private fields pass // through untouched. It is the raw JSON of the call's extra_content // object, or nil when the provider sent none, in which case it is omitted // from the request again. Extra json.RawMessage } // ToolResult is what came back, on its way to the model. // // Content is a string because that is what crosses the wire, but it carries // encoded structured data — §4 keeps formatting the model's job, so a handler // never writes prose and this never carries any. type ToolResult struct { CallID string Content string IsError bool } // Message is one turn of a conversation. // // A turn is text, or tool calls, or tool results — an assistant turn may carry // text and calls together, which is why these are fields rather than a union. type Message struct { Role Role Text string ToolCalls []ToolCall ToolResults []ToolResult } // ToolDef is a tool as the model sees it. // // Deliberately not the tool layer's own type. The gateway must not import the // tool package: a model provider knowing what an `effect` or a confirmation // token is would put policy behind a transport, and the confirmation gate has // to sit where the model cannot reach it. type ToolDef struct { Name string Description string InputSchema map[string]any } // Request is one model call. type Request struct { // Tier selects the model and effort. From the agent spec. Tier Tier // System is the assembled system prompt. // // I7: retrieved document text must never reach this field. Retrieved // content belongs in a delimited context block inside a user message, // where the system prompt has already said that its contents are data. // Nothing here can enforce that — it is a property of what the runtime // passes — so it is stated where the field is declared. System string // Messages is the conversation so far, oldest first. Messages []Message // Tools the model may call this turn. Order matters: it is part of the // cached prefix, so the caller sorts it once and keeps it stable. Tools []ToolDef // MaxOutputTokens caps this response. Zero takes the configured default. // // This is a hard ceiling the model is not aware of, so it truncates rather // than winding down. It is not the run's token budget — that is the // runtime's, and it spans every call in a run. MaxOutputTokens int64 } // Usage is what a call cost. type Usage struct { InputTokens int64 OutputTokens int64 CacheReadTokens int64 CacheCreationTokens int64 } // Total is every token this call is billed for. // // Cache reads are counted: they are cheaper than fresh input, not free, and a // budget that ignored them would drift further from the truth the longer a // conversation ran — which is exactly when it matters most. func (u Usage) Total() int64 { return u.InputTokens + u.OutputTokens + u.CacheReadTokens + u.CacheCreationTokens } // Response is one model reply. type Response struct { Text string // ToolCalls the model wants run before it can continue. Non-empty exactly // when StopReason is "tool_use". ToolCalls []ToolCall StopReason string Usage Usage // Model is the id actually used, not the tier that was asked for. Logged // with every run so a change of routing is visible in the trajectory // rather than inferred from a deploy date. Model string Tier Tier } // Error codes. Structured rather than bare strings, per §10 — user-facing text // is derived at the surface layer, never raised from here. const ( CodeNotConfigured = "gateway.not_configured" CodeInvalidRequest = "gateway.invalid_request" CodeUnauthorized = "gateway.unauthorized" CodeRateLimited = "gateway.rate_limited" CodeTimeout = "gateway.timeout" CodeRefused = "gateway.refused" CodeUpstream = "gateway.upstream" ) // Error is a gateway failure with a code the runtime can branch on. type Error struct { Code string Message string // Status is the upstream HTTP status, when there was one. Status int // Category carries a refusal's reason when Code is CodeRefused. An open // set upstream, so it is a string and is never switched on exhaustively. Category string Cause error } func (e *Error) Error() string { if e.Status != 0 { return fmt.Sprintf("%s: %s (http %d)", e.Code, e.Message, e.Status) } return fmt.Sprintf("%s: %s", e.Code, e.Message) } func (e *Error) Unwrap() error { return e.Cause } // Retryable reports whether the same request could succeed if sent again. // // The runtime needs this to decide between a retry and a terminal // ToolFailure. A refusal is emphatically not retryable — re-sending a request // the model declined is how a loop burns a whole budget on one turn. func (e *Error) Retryable() bool { switch e.Code { case CodeRateLimited, CodeTimeout: return true case CodeUpstream: return e.Status >= 500 default: return false } } // Streamer is a Gateway that can deliver text as it arrives. // // A SEPARATE interface, not a method on Gateway, and that is deliberate. Adding // Stream to Gateway would break every fake in the test suite and force each one // to implement a transport it does not care about — and those fakes exist to // test the LOOP, not the wire. StreamComplete bridges the two, so a caller // writes one line and gets streaming wherever it is available. type Streamer interface { // Stream calls the model, invoking onDelta with each fragment of assistant // text. Tool calls are NOT streamed: a partially-built argument object is a // different object from the finished one, and usually an invalid one. Stream(ctx context.Context, req Request, onDelta func(string)) (*Response, error) } // Gateway is the model boundary. // // One method. A second implementation — a fake for tests, a recorded one for // evals — has one thing to satisfy, which is what keeps the eval harness from // needing a network. type Gateway interface { Complete(ctx context.Context, req Request) (*Response, error) } // Validate checks a request before it costs anything. func (r Request) Validate() error { if len(r.Messages) == 0 { return &Error{Code: CodeInvalidRequest, Message: "a request needs at least one message"} } for i, m := range r.Messages { if m.Role != RoleUser && m.Role != RoleAssistant { return &Error{ Code: CodeInvalidRequest, Message: fmt.Sprintf("messages[%d]: %q is not a role", i, m.Role), } } // A turn must say something, but "something" is text, tool calls or // tool results. A tool-result turn legitimately carries no text at all. if strings.TrimSpace(m.Text) == "" && len(m.ToolCalls) == 0 && len(m.ToolResults) == 0 { return &Error{ Code: CodeInvalidRequest, Message: fmt.Sprintf("messages[%d]: a message cannot be empty", i), } } } for i, t := range r.Tools { if strings.TrimSpace(t.Name) == "" { return &Error{Code: CodeInvalidRequest, Message: fmt.Sprintf("tools[%d]: a tool needs a name", i)} } if strings.TrimSpace(t.Description) == "" { // The description is what the model reads instead of documentation. return &Error{Code: CodeInvalidRequest, Message: fmt.Sprintf("tools[%d]: %s has no description", i, t.Name)} } } return nil } // StreamComplete runs a request through whichever path the gateway supports. // // A gateway that cannot stream is not a broken gateway — every fake in the test // suite is one, and so is any future provider without a streaming API. Falling // back to Complete and delivering the finished text as a single delta keeps the // caller's code identical either way, which is what stops streaming from // becoming a second code path through the loop. func StreamComplete(ctx context.Context, gw Gateway, req Request, onDelta func(string)) (*Response, error) { // Normalised once, here, so no implementation has to guard it. A caller // that does not want deltas passes nil — every eval and every test does — // and an implementation that took that literally would panic on the first // fragment. Making each Streamer remember the check is how one of them // eventually forgets. if onDelta == nil { onDelta = func(string) {} } if s, ok := gw.(Streamer); ok { return s.Stream(ctx, req, onDelta) } resp, err := gw.Complete(ctx, req) if err == nil && resp != nil && resp.Text != "" && onDelta != nil { onDelta(resp.Text) } return resp, err }