updates on the ai agents and time series prediction and updates on the api to
This commit is contained in:
130
internal/prediction/sweeper.go
Normal file
130
internal/prediction/sweeper.go
Normal file
@@ -0,0 +1,130 @@
|
||||
package prediction
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"doormile/db"
|
||||
"doormile/utils"
|
||||
)
|
||||
|
||||
// The calibration sweeper.
|
||||
//
|
||||
// Deliberately the same shape as internal/assignment/sweeper.go: a ticker, a
|
||||
// recover() per tick so one bad run cannot take the process down, and a Redis
|
||||
// lock that expires before the next tick so only one replica of three does the
|
||||
// work. Without Redis every replica refreshes, which is wasteful but correct —
|
||||
// Refresh replaces the table in one transaction, so concurrent runs converge
|
||||
// rather than interleave.
|
||||
//
|
||||
// This is a nightly job by default. The calibration is a p80 over weeks of
|
||||
// deliveries; recomputing it more often costs a scan and changes nothing.
|
||||
|
||||
const (
|
||||
defaultCalibrationSweepSeconds = 6 * 60 * 60 // 6h
|
||||
calibrationLockKey = "prediction:calibration:lock"
|
||||
|
||||
// Below this a "sweep" is a scan loop against the whole delivery history.
|
||||
minCalibrationSweepSeconds = 600
|
||||
)
|
||||
|
||||
// calibrationInterval: PREDICTION_CALIBRATION_SECONDS, default 6h; 0 turns the
|
||||
// sweeper off and leaves whatever is in the store. Read once at start, matching
|
||||
// how assignment's sweepInterval behaves.
|
||||
func calibrationInterval() time.Duration {
|
||||
if v := strings.TrimSpace(os.Getenv("PREDICTION_CALIBRATION_SECONDS")); v != "" {
|
||||
if n, err := strconv.Atoi(v); err == nil && n >= 0 {
|
||||
if n > 0 && n < minCalibrationSweepSeconds {
|
||||
n = minCalibrationSweepSeconds
|
||||
}
|
||||
return time.Duration(n) * time.Second
|
||||
}
|
||||
utils.Warn("PREDICTION_CALIBRATION_SECONDS is not a non-negative integer, using the default",
|
||||
"value", v, "default_seconds", defaultCalibrationSweepSeconds)
|
||||
}
|
||||
return defaultCalibrationSweepSeconds * time.Second
|
||||
}
|
||||
|
||||
// StartCalibrationSweeper loads whatever calibration is stored, then refreshes
|
||||
// it on a timer. Call once at boot, in a goroutine.
|
||||
//
|
||||
// The load happens even when the sweeper is disabled: a stored calibration from
|
||||
// a previous deploy is still the best available answer, and ETAMinutes will
|
||||
// stop trusting it on its own once it passes staleAfter.
|
||||
func StartCalibrationSweeper() {
|
||||
LoadFromDB()
|
||||
|
||||
interval := calibrationInterval()
|
||||
if interval == 0 {
|
||||
utils.Info("CalibrationSweeper: disabled (PREDICTION_CALIBRATION_SECONDS=0)")
|
||||
return
|
||||
}
|
||||
utils.Info("CalibrationSweeper: started", "interval", interval.String())
|
||||
|
||||
// One refresh shortly after boot rather than waiting a full interval, so a
|
||||
// newly deployed replica is not serving a day-old calibration for six
|
||||
// hours. Offset so three replicas do not all wake together.
|
||||
time.Sleep(90 * time.Second)
|
||||
refreshOnce(interval)
|
||||
|
||||
ticker := time.NewTicker(interval)
|
||||
defer ticker.Stop()
|
||||
for range ticker.C {
|
||||
refreshOnce(interval)
|
||||
}
|
||||
}
|
||||
|
||||
// refreshOnce makes one refresh attempt. Only one replica at a time (a Redis
|
||||
// lock that expires before the next tick).
|
||||
func refreshOnce(interval time.Duration) {
|
||||
defer func() {
|
||||
if r := recover(); r != nil {
|
||||
utils.Error("CalibrationSweeper: panic recovered", "error", r)
|
||||
}
|
||||
}()
|
||||
if db.DB == nil {
|
||||
return
|
||||
}
|
||||
if db.Rdb != nil {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second)
|
||||
ttl := interval - 60*time.Second
|
||||
if ttl <= 0 {
|
||||
ttl = interval / 2
|
||||
}
|
||||
got, err := db.Rdb.SetNX(ctx, calibrationLockKey, "1", ttl).Result()
|
||||
cancel()
|
||||
if err == nil && !got {
|
||||
return // another replica has this refresh
|
||||
}
|
||||
}
|
||||
|
||||
started := time.Now()
|
||||
if err := Refresh(db.DB); err != nil {
|
||||
// A failed refresh leaves the previous calibration serving. That is the
|
||||
// intended behaviour: the last good factors beat falling back to flat
|
||||
// constants, and staleAfter puts a bound on how long that can last.
|
||||
utils.Error("CalibrationSweeper: refresh failed", "error", err,
|
||||
"took_ms", time.Since(started).Milliseconds())
|
||||
return
|
||||
}
|
||||
ok, builtAt, cells := Loaded()
|
||||
utils.Info("CalibrationSweeper: refresh complete",
|
||||
"loaded", ok, "cells", cells, "built_at", builtAt,
|
||||
"took_ms", time.Since(started).Milliseconds())
|
||||
|
||||
// Apply the fresh calibration to parcels still in flight. Writes
|
||||
// estimateddeliveryat only — never sladueat, which is the commitment made
|
||||
// at booking (see internal/prediction/refine.go). A no-op while nothing is
|
||||
// calibrated.
|
||||
refined, err := RefineInFlight(db.DB, time.Now())
|
||||
if err != nil {
|
||||
utils.Error("CalibrationSweeper: ETA refine failed", "error", err)
|
||||
return
|
||||
}
|
||||
if len(refined) > 0 {
|
||||
utils.Info("CalibrationSweeper: refined in-flight ETAs", "count", len(refined))
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user