package gateway // Failover: a second and third provider, for when the first one says no. // // THE PROBLEM THIS SOLVES IS A CEILING, NOT A BUG. A free tier is a token // budget per minute, and one agent run can exceed a whole minute's worth by // itself — a three-call run measured 12,123 tokens against a ceiling of 8,000. // withRetry already fires three times, and on a rate limit all three are // refused, because waiting 1.6 seconds does not buy back a minute's budget. The // run then ends GatewayFailure and a person reads "the model did not answer". // // Retrying harder cannot fix that. Asking somebody else can: the ceilings are // per provider, so a second key is a second budget. Groq, Cerebras, Gemini, // Mistral and OpenRouter all serve the same chat-completions shape, which is // the whole reason this is a list of Configs and not a second implementation. // // WHAT IT DOES NOT DO, stated because the gap is where the next bug lives: // it does not make a run cheaper, it does not raise any one provider's ceiling, // and it does not help when every configured provider is exhausted at once. It // converts "one busy provider" from an outage into a slower answer. import ( "context" "errors" ) // failover tries each provider in order until one answers. type failover struct { providers []Gateway } // NewFailover builds a gateway that falls back through `rest` when `primary` // cannot answer. With no fallbacks it returns the primary unchanged, so a // single-provider deployment carries no wrapper and behaves exactly as before. func NewFailover(primary Gateway, rest ...Gateway) Gateway { if len(rest) == 0 { return primary } return &failover{providers: append([]Gateway{primary}, rest...)} } func (f *failover) Complete(ctx context.Context, req Request) (*Response, error) { var last error for i, p := range f.providers { if i > 0 && !canFailOver(req, last) { break } resp, err := p.Complete(ctx, req) if err == nil { return resp, nil } last = err // The caller's deadline governs. A deployment with four providers must // not spend four timeouts' worth of a person's patience discovering // that none of them is available. if ctx.Err() != nil { break } } return nil, last } // Stream falls over only before the first fragment has been delivered. // // After a delta reaches the client, the answer has begun in the reader's own // window. Starting a second provider would continue that sentence in a // different voice from a different model, or repeat its opening — so once text // is out, the error is the answer. func (f *failover) Stream(ctx context.Context, req Request, onDelta func(string)) (*Response, error) { var last error for i, p := range f.providers { if i > 0 && !canFailOver(req, last) { break } var delivered bool wrapped := func(s string) { delivered = true onDelta(s) } resp, err := StreamComplete(ctx, p, req, wrapped) if err == nil { return resp, nil } last = err if delivered || ctx.Err() != nil { break } } return nil, last } // canFailOver decides whether asking a DIFFERENT provider is sound. // // Two conditions, and both are necessary. // // 1. THE FAILURE MUST BE TRANSIENT. Error.Retryable() already draws that line // for retries and it is the same line here: a rate limit or a 5xx is the // provider being unable, and somebody else may be able. A 400 is a // malformed request and will be malformed for everyone; a 401 is this // deployment's own credential. Failing over on those turns one provider's // configuration error into every provider's, and buries the fault. // // 2. THE CONVERSATION MUST NOT BE BOUND TO ITS PROVIDER. ToolCall.Extra // carries provider metadata echoed back verbatim — Gemini 3's thought // signature is the known case, and it REJECTS a follow-up that does not // return it. That metadata is meaningless to a different provider and its // absence is fatal to the one that issued it, so a conversation that // already carries any is pinned to whoever produced it. In practice this // means failover is available on the first model call of a run, which is // where a rate limit usually lands anyway. func canFailOver(req Request, err error) bool { var gwErr *Error if !errors.As(err, &gwErr) || !gwErr.Retryable() { return false } for _, m := range req.Messages { for _, tc := range m.ToolCalls { if len(tc.Extra) > 0 { return false } } } return true }