mcp connection
Some checks failed
CI / fixture (push) Has been cancelled
CI / test (push) Has been cancelled

This commit is contained in:
2026-09-22 10:58:02 +05:30
parent 4e1f746b22
commit f2aa3b3ad8
53 changed files with 12515 additions and 37 deletions

View File

@@ -0,0 +1,193 @@
package httpserver
import (
"context"
"log/slog"
"time"
"github.com/krow/krow-backend/go-api/internal/oauth"
"github.com/krow/krow-backend/go-api/internal/ratelimit"
)
// Scheduled maintenance for the OAuth and rate-limit tables.
//
// WHY THIS SHAPE AND NOT A NEW ONE
//
// The process already has a scheduled maintenance mechanism: sweepSessions in
// cmd/api/main.go, a ticker goroutine whose context is the server's, which runs
// once at startup and then on an interval, logs a failure and retries at the
// next tick. It is bounded, cancellable, non-blocking and failure-isolated, and
// it has been in production.
//
// So this is the same thing for two more tables rather than a second kind of
// thing. No new process, no cron dependency, no leader election, no library.
// The one addition is that both sweeps live behind a single type, so
// cmd/api/main.go gains one line rather than two more goroutines.
//
// MULTI-INSTANCE SAFETY COMES FROM THE STATEMENTS, NOT FROM COORDINATION
//
// Every instance runs this, on its own schedule, with no lock between them —
// deliberately. A lease or an advisory lock would be state to hold, to expire
// and to recover when the holder dies mid-sweep, in exchange for avoiding work
// that is already harmless: each sweep is a bounded DELETE whose predicate no
// longer matches once a row is gone. Two instances sweeping at the same moment
// delete disjoint sets and neither errors. A row deleted twice is not an error;
// it is a row that was already deleted.
//
// That is the same property Phase 5's concurrent-cleanup test asserts directly:
// four workers, six dead tokens, exactly six removed between them.
// maintenanceInterval is how often the sweep runs.
//
// Hourly. The grace period before anything is deleted is also an hour, so a
// row becomes eligible and is collected within roughly two — soon enough that
// nothing accumulates, and far enough apart that a DELETE never lands on a hot
// path. Shorter would buy nothing: nothing here is a correctness deadline.
//
// Deliberately NOT sweepInterval's fifteen minutes. Sessions churn with every
// sign-in; authorization codes live sixty seconds and tokens fifteen minutes,
// so an hour still collects them promptly while running a quarter as often.
const maintenanceInterval = time.Hour
// maintenanceTimeout bounds one pass.
//
// Generous for three bounded deletes and short enough that a wedged statement
// cannot hold this goroutine past shutdown. Matches sweepSessions' own bound in
// spirit; longer only because there are more statements.
const maintenanceTimeout = 60 * time.Second
// Maintenance sweeps the OAuth and rate-limit tables.
//
// Nil when the deployment does not serve MCP, which is why Server.Maintenance
// returns a pointer and the caller checks it — the same way routeOAuth simply
// registers nothing.
type Maintenance struct {
store *oauth.Store
limiter *ratelimit.Limiter
log *slog.Logger
}
// Maintenance exposes the sweeper, or nil when there is nothing to sweep.
//
// Mirrors Server.Sessions(), which exists for exactly this reason: the process
// owns the schedule, the server owns the things being swept.
func (s *Server) Maintenance() *Maintenance {
if !s.cfg.OAuth.Enabled() {
return nil
}
return &Maintenance{
store: oauth.NewStore(s.db.Pool),
limiter: s.limiter,
log: s.log,
}
}
// MaintenanceResult is what one pass removed.
type MaintenanceResult struct {
Grants int64
AccessTokens int64
RefreshTokens int64
RateLimits int64
}
// Total is the row count removed, for the log line.
func (r MaintenanceResult) Total() int64 {
return r.Grants + r.AccessTokens + r.RefreshTokens + r.RateLimits
}
// Sweep runs one maintenance pass.
//
// The two halves are independent on purpose: a failure sweeping OAuth rows must
// not prevent the rate-limit sweep, because the second is the one that would
// otherwise grow without bound. The first error is returned, after both have
// been attempted.
func (m *Maintenance) Sweep(ctx context.Context) (MaintenanceResult, error) {
var out MaintenanceResult
var firstErr error
// OAuth: codes, access tokens, and refresh tokens past their retention.
// The grace period and the reuse-detection retention are enforced inside
// Store.Cleanup — this schedules it, it does not reimplement it.
cleaned, err := m.store.Cleanup(ctx)
if err != nil {
firstErr = err
} else {
out.Grants = cleaned.Grants
out.AccessTokens = cleaned.AccessTokens
out.RefreshTokens = cleaned.RefreshTokens
}
if m.limiter != nil {
swept, err := m.limiter.Sweep(ctx, 0) // 0 = the package's own batch size
if err != nil && firstErr == nil {
firstErr = err
}
out.RateLimits = swept
}
return out, firstErr
}
// SweepMaintenance runs the sweep until the context is cancelled.
//
// Deliberately identical in shape to sweepSessions: one pass immediately so a
// process that has been down does not carry a backlog for a further hour, then
// on the ticker. A failed pass is logged and retried at the next tick — the
// tables being briefly larger than they should be is not worth stopping the API
// for, and it is certainly not worth a panic in a goroutine nobody is watching.
//
// Exported because cmd/api owns the process's goroutines and this package owns
// what they do.
func SweepMaintenance(ctx context.Context, m *Maintenance, log *slog.Logger) {
if m == nil {
// No OAuth surface, nothing to sweep. Returning rather than ticking
// uselessly for the life of the process.
return
}
ticker := time.NewTicker(maintenanceInterval)
defer ticker.Stop()
pass := func() {
// A deadline of its own, so a slow DELETE cannot leave this goroutine
// blocked past shutdown.
sweepCtx, cancel := context.WithTimeout(ctx, maintenanceTimeout)
defer cancel()
// A panic in a background goroutine takes the process with it, and
// this one runs unattended for the life of the deployment. Recovering
// turns a bug here into a logged failure and a retry at the next tick.
defer func() {
if p := recover(); p != nil {
log.Error("maintenance sweep panicked", "panic", p)
}
}()
result, err := m.Sweep(sweepCtx)
switch {
case err != nil && ctx.Err() != nil:
// Shutting down; the cancellation is expected, not a failure.
case err != nil:
log.Warn("maintenance sweep failed", "error", err,
"grants", result.Grants, "access_tokens", result.AccessTokens,
"refresh_tokens", result.RefreshTokens, "rate_limits", result.RateLimits)
case result.Total() > 0:
log.Info("maintenance sweep",
"grants", result.Grants, "access_tokens", result.AccessTokens,
"refresh_tokens", result.RefreshTokens, "rate_limits", result.RateLimits)
default:
log.Debug("maintenance sweep found nothing to delete")
}
}
pass()
for {
select {
case <-ctx.Done():
log.Debug("maintenance sweeper stopped")
return
case <-ticker.C:
pass()
}
}
}