secert updated
This commit is contained in:
@@ -8,6 +8,8 @@ import (
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"reflect"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
@@ -161,7 +163,41 @@ func (e *geminiEmbedder) Embed(ctx context.Context, text string) ([]float32, err
|
||||
// so a rate-limited assistant told a shopkeeper "embedding: HTTP 429" — a
|
||||
// sentence about a subsystem they have never heard of, describing something
|
||||
// that was not involved.
|
||||
// postJSON sends the request, and sends it a second time if the provider said
|
||||
// it was over its quota and named a wait we are willing to hold for.
|
||||
//
|
||||
// One retry, not a loop: past that, a queue forms behind a limit that is not
|
||||
// going to lift, and the person is better told to try again than left watching
|
||||
// a spinner. Both the chat gateway and the embedder go through here, so neither
|
||||
// can be the one that forgot.
|
||||
func postJSON(ctx context.Context, client *http.Client, what, url, auth string, body, out interface{}, headers ...string) error {
|
||||
err := postJSONOnce(ctx, client, what, url, auth, body, out, headers...)
|
||||
|
||||
var busy *tooManyRequests
|
||||
if errors.As(err, &busy) && waitBeforeRetry(ctx, busy.after) {
|
||||
// `out` must be emptied first. The first attempt decoded the provider's
|
||||
// error body into it, and `encoding/json` leaves fields the second
|
||||
// payload does not mention exactly as it found them — so a retry that
|
||||
// SUCCEEDED came back carrying the 429's `error` object, and every
|
||||
// caller here checks that field before the data. The result was a
|
||||
// successful call reported as the failure it had just recovered from.
|
||||
resetForRetry(out)
|
||||
return postJSONOnce(ctx, client, what, url, auth, body, out, headers...)
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
// resetForRetry empties a decode target so a second attempt cannot inherit the
|
||||
// first one's fields.
|
||||
func resetForRetry(out interface{}) {
|
||||
value := reflect.ValueOf(out)
|
||||
if value.Kind() != reflect.Ptr || value.IsNil() {
|
||||
return
|
||||
}
|
||||
value.Elem().Set(reflect.Zero(value.Elem().Type()))
|
||||
}
|
||||
|
||||
func postJSONOnce(ctx context.Context, client *http.Client, what, url, auth string, body, out interface{}, headers ...string) error {
|
||||
payload, err := json.Marshal(body)
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -194,6 +230,20 @@ func postJSON(ctx context.Context, client *http.Client, what, url, auth string,
|
||||
return fmt.Errorf("%s: HTTP %d, unreadable body: %w", what, resp.StatusCode, err)
|
||||
}
|
||||
if resp.StatusCode/100 != 2 {
|
||||
// Too many requests is the one status that is not about this request.
|
||||
// The provider's own sentence is unusable here — Groq's reads
|
||||
//
|
||||
// "Rate limit reached for model `openai/gpt-oss-120b` in organization
|
||||
// `org_01m38x8s72e759kn6g88ve2dhj` service tier `on_demand` on tokens
|
||||
// per minute (TPM): Limit 8000, Used 7320…"
|
||||
//
|
||||
// which is shown to a shopkeeper as Buddy's answer, names our billing
|
||||
// account, and tells them nothing they can act on. It is also usually
|
||||
// over within a second, so the honest handling is to wait and try again
|
||||
// rather than to report it at all.
|
||||
if resp.StatusCode == http.StatusTooManyRequests {
|
||||
return &tooManyRequests{what: what, after: retryAfter(resp, raw)}
|
||||
}
|
||||
// The decoded body carries the provider's message where there is one;
|
||||
// this is the fallback for a bare status.
|
||||
if msg := extractMessage(raw); msg != "" {
|
||||
@@ -230,3 +280,106 @@ func VectorLiteral(v []float32) string {
|
||||
b.WriteByte(']')
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// ── Being rate limited ──────────────────────────────────────────────────────
|
||||
//
|
||||
// A shared provider quota is not a fault in the request that happened to hit
|
||||
// it, and on Groq's free tier it is reached by ordinary use: 8,000 tokens a
|
||||
// minute is three or four Buddy questions. The waits are short — the provider
|
||||
// states them in milliseconds — so one retry turns almost all of them into a
|
||||
// slightly slower answer instead of an error.
|
||||
|
||||
// ErrBusy is what a caller sees when the wait did not help.
|
||||
//
|
||||
// Sentinel so the HTTP layer can answer 429 and the console can say "a moment"
|
||||
// rather than rendering a provider's billing details as an answer.
|
||||
var ErrBusy = errors.New("the assistant is busy right now — try again in a moment")
|
||||
|
||||
type tooManyRequests struct {
|
||||
what string
|
||||
after time.Duration
|
||||
}
|
||||
|
||||
func (e *tooManyRequests) Error() string { return e.what + ": " + ErrBusy.Error() }
|
||||
func (e *tooManyRequests) Unwrap() error { return ErrBusy }
|
||||
|
||||
// maxRetryWait bounds how long a request may be held. Beyond this the honest
|
||||
// answer is "busy" — a person watching a spinner has already decided something
|
||||
// is broken, and the provider's own suggestion can be a minute on a hard quota.
|
||||
const maxRetryWait = 3 * time.Second
|
||||
|
||||
// retryAfter reads how long the provider asked us to wait.
|
||||
//
|
||||
// `Retry-After` first, because it is the standard and a proxy may add it where
|
||||
// the body has nothing. Groq puts the number in prose instead — "Please try
|
||||
// again in 840ms" — so that is read next. Zero means "no idea", and the caller
|
||||
// uses its own floor rather than hammering immediately.
|
||||
func retryAfter(resp *http.Response, raw []byte) time.Duration {
|
||||
if header := strings.TrimSpace(resp.Header.Get("Retry-After")); header != "" {
|
||||
// Seconds, as an integer, is the only form worth reading: the HTTP-date
|
||||
// form is for caches and no model provider sends it.
|
||||
if secs, err := strconv.ParseFloat(header, 64); err == nil && secs > 0 {
|
||||
return time.Duration(secs * float64(time.Second))
|
||||
}
|
||||
}
|
||||
return waitFromMessage(extractMessage(raw))
|
||||
}
|
||||
|
||||
// waitFromMessage pulls "try again in 840ms" or "try again in 1.5s" out of prose.
|
||||
//
|
||||
// Its own function because it is the part worth testing: the wording comes from
|
||||
// somebody else's error strings and is the first thing that will change.
|
||||
func waitFromMessage(message string) time.Duration {
|
||||
lower := strings.ToLower(message)
|
||||
marker := "try again in "
|
||||
at := strings.Index(lower, marker)
|
||||
if at < 0 {
|
||||
return 0
|
||||
}
|
||||
|
||||
rest := lower[at+len(marker):]
|
||||
end := 0
|
||||
for end < len(rest) && (rest[end] == '.' || (rest[end] >= '0' && rest[end] <= '9')) {
|
||||
end++
|
||||
}
|
||||
if end == 0 {
|
||||
return 0
|
||||
}
|
||||
amount, err := strconv.ParseFloat(rest[:end], 64)
|
||||
if err != nil || amount <= 0 {
|
||||
return 0
|
||||
}
|
||||
|
||||
switch {
|
||||
case strings.HasPrefix(rest[end:], "ms"):
|
||||
return time.Duration(amount * float64(time.Millisecond))
|
||||
case strings.HasPrefix(rest[end:], "s"):
|
||||
return time.Duration(amount * float64(time.Second))
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// waitBeforeRetry sleeps for what the provider asked, bounded, and reports
|
||||
// whether waiting is worth it at all.
|
||||
//
|
||||
// Returns false when the ask is longer than we are prepared to hold a request
|
||||
// for, or when the caller's context is done — a retry after the browser has
|
||||
// given up is work nobody will see.
|
||||
func waitBeforeRetry(ctx context.Context, after time.Duration) bool {
|
||||
if after <= 0 {
|
||||
// No stated wait. A short one anyway: retrying instantly on a quota is
|
||||
// how a burst becomes two bursts.
|
||||
after = 250 * time.Millisecond
|
||||
}
|
||||
if after > maxRetryWait {
|
||||
return false
|
||||
}
|
||||
timer := time.NewTimer(after)
|
||||
defer timer.Stop()
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return false
|
||||
case <-timer.C:
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user