package services import ( "fmt" "sync" "time" ) // How often one person may ask. // // The assistant is the only endpoint in this backend that costs money per // request. Everything else is bounded by the database; this is bounded by // somebody's willingness to keep typing, and a held-down key or a bad retry // loop in a browser turns a shopkeeper's curiosity into a bill. // // ── Per user, not per tenant or per IP ────────────────────────────────────── // // Per tenant would let one impatient person in a shop lock out their // colleagues, which turns a cost control into an outage. Per IP is wrong twice // over: a shop behind one router shares an address, and the cost follows the // session rather than the network. // // ── What it is NOT ────────────────────────────────────────────────────────── // // Not a security control. Somebody with a valid session can already read their // own shop; this only decides how fast, and how expensively. The rules about // WHOSE data is read live in the registry and are not affected by any of this. // // ── One process ───────────────────────────────────────────────────────────── // // In memory, so the limit is per pod: two pods means twice the burst. That is // worth being clear about rather than hiding, and it is still the difference // between a bounded cost and an unbounded one. Redis is already in this // deployment if a shared limit is ever wanted — it is a repository swap, not a // redesign. const ( // askBurst is how many questions can be asked back to back. // // Six, because a person working through the prompt chips on a page will // fire four in a row and should not be stopped mid-thought. askBurst = 6 // askRefill is how long one question takes to come back. askRefill = 10 * time.Second // askIdle is when a quiet caller is forgotten, so the map does not grow // with every account that ever asked anything. askIdle = 30 * time.Minute ) // ErrTooFast is what a caller sees when they have run out of allowance. type ErrTooFast struct{ RetryIn time.Duration } func (e ErrTooFast) Error() string { return fmt.Sprintf("that is a lot of questions at once — try again in %d seconds", int(e.RetryIn.Seconds()+0.5)) } // askLimiter is a token bucket per user. type askLimiter struct { mu sync.Mutex buckets map[int]*bucket now func() time.Time } type bucket struct { tokens float64 seen time.Time } func newAskLimiter(now func() time.Time) *askLimiter { if now == nil { now = time.Now } return &askLimiter{buckets: map[int]*bucket{}, now: now} } // allow takes one token, or reports how long until the next is due. // // A caller with no user id gets through. Reachable only where the session did // not identify anybody, and every such request is already refused before this — // silently rate-limiting an unauthenticated caller would hide the real reason // behind a confusing one. func (l *askLimiter) allow(userid int) error { if userid <= 0 { return nil } l.mu.Lock() defer l.mu.Unlock() at := l.now() b, known := l.buckets[userid] if !known { l.buckets[userid] = &bucket{tokens: askBurst - 1, seen: at} l.sweep(at) return nil } // Refill by however long has passed, capped at the burst. Continuous rather // than a fixed window, so a person is never told to wait out a window that // started before they arrived. b.tokens += at.Sub(b.seen).Seconds() / askRefill.Seconds() if b.tokens > askBurst { b.tokens = askBurst } b.seen = at if b.tokens < 1 { return ErrTooFast{RetryIn: time.Duration((1 - b.tokens) * float64(askRefill))} } b.tokens-- return nil } // sweep drops callers nobody has heard from. // // Called on a new caller rather than on a timer: the map only grows when // somebody new arrives, so that is the moment it is worth tidying, and it // costs nothing on a quiet deployment. func (l *askLimiter) sweep(at time.Time) { if len(l.buckets) < 256 { return } for userid, b := range l.buckets { if at.Sub(b.seen) > askIdle { delete(l.buckets, userid) } } }