updates on the ai agents and time series prediction and updates on the api to
This commit is contained in:
272
internal/prediction/calibration.go
Normal file
272
internal/prediction/calibration.go
Normal file
@@ -0,0 +1,272 @@
|
||||
package prediction
|
||||
|
||||
import (
|
||||
"time"
|
||||
|
||||
"doormile/constants"
|
||||
"doormile/db"
|
||||
"doormile/utils"
|
||||
|
||||
"gorm.io/gorm"
|
||||
)
|
||||
|
||||
// The calibration refresh.
|
||||
//
|
||||
// Every delivered consignment whose rider was sequenced carries two numbers:
|
||||
// what the Route Optimization API predicted (bookingassignments.etaminutes,
|
||||
// written by internal/routing) and what actually happened (the Delivered row in
|
||||
// consignmenthistory). The ratio between them, grouped and taken at p80, is the
|
||||
// factor ETAMinutes multiplies by.
|
||||
//
|
||||
// Why p80 and not the mean: docs/prediction-plan.md §8 decision 3. An ETA is a
|
||||
// promise, and a mean is late half the time.
|
||||
//
|
||||
// Why a view and not a join here: consignments has no bookingid column (hazard
|
||||
// H4 in the plan). The consignment_booking view resolves the two paths —
|
||||
// bookingdestinations for multi-destination pickups, the legacy
|
||||
// pickupbookings.consignmentid for console and pre-fan-out rows — once, in
|
||||
// migrate.go. Joining through consignments directly attaches features to the
|
||||
// wrong parcel for every multi-destination booking, silently.
|
||||
|
||||
// ETACalibration is one calibration cell as stored. Written only by Refresh,
|
||||
// read at boot and after each refresh.
|
||||
type ETACalibration struct {
|
||||
Calibrationid int `gorm:"primaryKey;column:calibrationid;autoIncrement"`
|
||||
Scope string `gorm:"column:scope;size:24;not null;index:idx_etacalibration_lookup,priority:1"`
|
||||
Zone string `gorm:"column:zone;size:8;index:idx_etacalibration_lookup,priority:2"`
|
||||
Hourbucket *int `gorm:"column:hourbucket;index:idx_etacalibration_lookup,priority:3"`
|
||||
Weekday *int `gorm:"column:weekday;index:idx_etacalibration_lookup,priority:4"`
|
||||
Factor float64 `gorm:"column:factor;not null"`
|
||||
Handling float64 `gorm:"column:handling;not null;default:0"`
|
||||
Samples int `gorm:"column:samples;not null"`
|
||||
Refreshedat time.Time `gorm:"column:refreshedat;not null"`
|
||||
}
|
||||
|
||||
func (ETACalibration) TableName() string { return "etacalibration" }
|
||||
|
||||
// row is one group from the refresh query.
|
||||
type row struct {
|
||||
Scope string
|
||||
Zone string
|
||||
Hourbucket *int
|
||||
Weekday *int
|
||||
Factor float64
|
||||
Samples int
|
||||
}
|
||||
|
||||
// Observed durations below this are almost certainly data errors — a Delivered
|
||||
// event written in the same second as the consignment, or a backfill. Including
|
||||
// them drags every factor toward zero.
|
||||
const minObservedMinutes = 5
|
||||
|
||||
// And above this, the parcel sat for days for a reason that has nothing to do
|
||||
// with road duration (held at hub, re-attempts, disputes). Calibrating a
|
||||
// routed-duration multiplier on those teaches it the wrong thing.
|
||||
const maxObservedMinutes = 48 * 60
|
||||
|
||||
// refreshSQL computes the p80 ratio of actual to routed duration, at three
|
||||
// grains plus a global row, in one pass.
|
||||
//
|
||||
// The `actual` term uses consignmenthistory.createdat - consignments.createdat.
|
||||
// Both come from CURRENT_TIMESTAMP defaults, so they share a tagging and their
|
||||
// difference is correct regardless of hazard H1 — which is why this does not
|
||||
// go near estimateddeliveryat, whose writes are not uniform
|
||||
// (adminController.go:2738 uses time.Now(), elsewhere it is CURRENT_TIMESTAMP).
|
||||
//
|
||||
// EXTRACT(ISODOW) is 1..7 Monday-first, matching weekdayOf in eta.go. The hour
|
||||
// bucket divides by 3, matching hourBucketSize. If either changes, both change.
|
||||
// A booking can hold SEVERAL bookingassignments rows — Assigned, Rejected,
|
||||
// Reassigned, Cancelled are all statuses it passes through. Joining them all
|
||||
// would multiply one delivered parcel by its whole assignment history and pull
|
||||
// the p80 toward whatever got reassigned most, so `assigned` picks exactly one
|
||||
// row per booking: the most recently sequenced one, which is the routed ETA
|
||||
// that was actually in force when the parcel was delivered.
|
||||
//
|
||||
// NULL::int / NULL::text are explicit because a UNION ALL resolves column types
|
||||
// across its branches; an untyped NULL can be inferred as text and fail against
|
||||
// the integer from the first branch — at runtime, against the real database,
|
||||
// which is exactly where it is most expensive to discover.
|
||||
const refreshSQL = `
|
||||
WITH assigned AS (
|
||||
SELECT DISTINCT ON (ba.bookingid)
|
||||
ba.bookingid,
|
||||
ba.etaminutes
|
||||
FROM bookingassignments ba
|
||||
WHERE ba.etaminutes > 0
|
||||
ORDER BY ba.bookingid, ba.sequencedat DESC NULLS LAST, ba.bookingassignmentid DESC
|
||||
),
|
||||
observed AS (
|
||||
SELECT left(c.deliverypincode, 3) AS zone,
|
||||
(EXTRACT(HOUR FROM c.createdat)::int / ?) AS hourbucket,
|
||||
EXTRACT(ISODOW FROM c.createdat)::int AS weekday,
|
||||
a.etaminutes::double precision AS routed,
|
||||
EXTRACT(EPOCH FROM (h.createdat - c.createdat)) / 60.0 AS actual
|
||||
FROM consignments c
|
||||
JOIN consignmenthistory h
|
||||
ON h.consignmentid = c.consignmentid
|
||||
AND h.eventstatus = ?
|
||||
JOIN consignment_booking cb
|
||||
ON cb.consignmentid = c.consignmentid
|
||||
JOIN assigned a
|
||||
ON a.bookingid = cb.bookingid
|
||||
WHERE c.deliverypincode IS NOT NULL
|
||||
AND length(c.deliverypincode) >= 3
|
||||
),
|
||||
clean AS (
|
||||
SELECT * FROM observed
|
||||
WHERE actual BETWEEN ? AND ?
|
||||
AND routed > 0
|
||||
)
|
||||
SELECT 'zone_hour_weekday'::text AS scope, zone, hourbucket, weekday,
|
||||
percentile_cont(0.8) WITHIN GROUP (ORDER BY actual / routed) AS factor,
|
||||
count(*)::int AS samples
|
||||
FROM clean GROUP BY zone, hourbucket, weekday
|
||||
UNION ALL
|
||||
SELECT 'zone_weekday'::text, zone, NULL::int, weekday,
|
||||
percentile_cont(0.8) WITHIN GROUP (ORDER BY actual / routed), count(*)::int
|
||||
FROM clean GROUP BY zone, weekday
|
||||
UNION ALL
|
||||
SELECT 'zone'::text, zone, NULL::int, NULL::int,
|
||||
percentile_cont(0.8) WITHIN GROUP (ORDER BY actual / routed), count(*)::int
|
||||
FROM clean GROUP BY zone
|
||||
UNION ALL
|
||||
SELECT 'global'::text, ''::text, NULL::int, NULL::int,
|
||||
percentile_cont(0.8) WITHIN GROUP (ORDER BY actual / routed), count(*)::int
|
||||
FROM clean
|
||||
`
|
||||
|
||||
// Refresh recomputes the calibration from history, replaces the stored table,
|
||||
// and swaps the in-memory snapshot.
|
||||
//
|
||||
// A failure leaves the previous calibration in place — a refresh that cannot
|
||||
// run is not a reason to stop answering with the last good factors. If they go
|
||||
// stale past staleAfter, ETAMinutes stops trusting them on its own.
|
||||
func Refresh(gdb *gorm.DB) error {
|
||||
if gdb == nil {
|
||||
return nil
|
||||
}
|
||||
|
||||
var rows []row
|
||||
if err := gdb.Raw(refreshSQL,
|
||||
hourBucketSize, constants.ConsignmentDelivered, minObservedMinutes, maxObservedMinutes,
|
||||
).Scan(&rows).Error; err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
kept := make([]ETACalibration, 0, len(rows))
|
||||
now := utils.DBNow()
|
||||
for _, r := range rows {
|
||||
if r.Samples < minSamples {
|
||||
continue // a p80 over fewer than minSamples is noise
|
||||
}
|
||||
if r.Factor < minFactor || r.Factor > maxFactor {
|
||||
// Out of bounds is a signal about the data, not a usable factor.
|
||||
// Log it once here rather than discovering it per request.
|
||||
utils.Warn("prediction: discarding out-of-bounds calibration factor",
|
||||
"scope", r.Scope, "zone", r.Zone, "factor", r.Factor, "samples", r.Samples)
|
||||
continue
|
||||
}
|
||||
kept = append(kept, ETACalibration{
|
||||
Scope: r.Scope,
|
||||
Zone: r.Zone,
|
||||
Hourbucket: r.Hourbucket,
|
||||
Weekday: r.Weekday,
|
||||
Factor: r.Factor,
|
||||
Samples: r.Samples,
|
||||
Refreshedat: now,
|
||||
})
|
||||
}
|
||||
|
||||
if len(kept) == 0 {
|
||||
// No cell cleared the floor. Expected until routing is on and history
|
||||
// accumulates; ETAMinutes keeps returning false and the promise tables
|
||||
// keep answering.
|
||||
utils.Info("prediction: no calibration cells met the sample floor",
|
||||
"groups_considered", len(rows), "min_samples", minSamples)
|
||||
return nil
|
||||
}
|
||||
|
||||
// Replace wholesale in one transaction: a partial table would serve a mix
|
||||
// of old and new factors for the same zone.
|
||||
if err := gdb.Transaction(func(tx *gorm.DB) error {
|
||||
if err := tx.Exec(`DELETE FROM etacalibration`).Error; err != nil {
|
||||
return err
|
||||
}
|
||||
return tx.CreateInBatches(&kept, 200).Error
|
||||
}); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
Load(kept)
|
||||
utils.Info("prediction: calibration refreshed", "cells", len(kept))
|
||||
return nil
|
||||
}
|
||||
|
||||
// Load swaps the in-memory snapshot. Exported so boot can populate from the
|
||||
// stored table without recomputing, and so tests can install a known
|
||||
// calibration without a database.
|
||||
//
|
||||
// builtAt is the NEWEST Refreshedat among the rows, not time.Now(): a restart
|
||||
// that loads month-old rows from the store must still look month-old to the
|
||||
// staleness guard in ETAMinutes. Taking the load time here would silently
|
||||
// re-arm a stale calibration on every deploy. A row with no Refreshedat (a test
|
||||
// fixture) is treated as fresh.
|
||||
// Refreshedat is stored through utils.DBNow, which writes IST wall-clock digits
|
||||
// labelled UTC. Reading it back as an instant is off by 5h30m, so it goes
|
||||
// through utils.IST first — the same correction utils/epoch.go exists for.
|
||||
// Without it a fresh calibration reads as 5.5 hours old, which is the very
|
||||
// hazard docs/prediction-plan.md calls H1.
|
||||
func Load(rows []ETACalibration) {
|
||||
t := &table{cells: make(map[string]cell, len(rows))}
|
||||
for _, r := range rows {
|
||||
if r.Refreshedat.IsZero() {
|
||||
continue
|
||||
}
|
||||
if at := utils.IST(r.Refreshedat); at.After(t.builtAt) {
|
||||
t.builtAt = at
|
||||
}
|
||||
}
|
||||
if t.builtAt.IsZero() {
|
||||
t.builtAt = time.Now() // a fixture with no Refreshedat counts as fresh
|
||||
}
|
||||
for _, r := range rows {
|
||||
c := cell{factor: r.Factor, handling: r.Handling, samples: r.Samples}
|
||||
switch r.Scope {
|
||||
case "global":
|
||||
t.global, t.hasGlobal = c, true
|
||||
case "zone":
|
||||
t.cells[keyZone(r.Zone)] = c
|
||||
case "zone_weekday":
|
||||
if r.Weekday != nil {
|
||||
t.cells[keyZoneWeekday(r.Zone, *r.Weekday)] = c
|
||||
}
|
||||
case "zone_hour_weekday":
|
||||
if r.Weekday != nil && r.Hourbucket != nil {
|
||||
t.cells[keyZoneHourWeekday(r.Zone, *r.Hourbucket, *r.Weekday)] = c
|
||||
}
|
||||
}
|
||||
}
|
||||
current.store(t)
|
||||
}
|
||||
|
||||
// Reset clears the in-memory calibration. Tests only — it makes ETAMinutes
|
||||
// return false, which is the state a fresh process is in before boot loads.
|
||||
func Reset() { current.store(nil) }
|
||||
|
||||
// LoadFromDB populates the snapshot from the stored table at boot, so a restart
|
||||
// does not wait for the next refresh to start answering.
|
||||
func LoadFromDB() {
|
||||
if db.DB == nil {
|
||||
return
|
||||
}
|
||||
var rows []ETACalibration
|
||||
if err := db.DB.Find(&rows).Error; err != nil {
|
||||
utils.Warn("prediction: could not load stored calibration", "error", err)
|
||||
return
|
||||
}
|
||||
if len(rows) == 0 {
|
||||
return
|
||||
}
|
||||
Load(rows)
|
||||
utils.Info("prediction: calibration loaded from store", "cells", len(rows))
|
||||
}
|
||||
Reference in New Issue
Block a user