227 lines
7.1 KiB
Go
227 lines
7.1 KiB
Go
package outcomes
|
|
|
|
import (
|
|
"context"
|
|
"os"
|
|
"strconv"
|
|
"strings"
|
|
"time"
|
|
|
|
"doormile/constants"
|
|
"doormile/db"
|
|
"doormile/utils"
|
|
|
|
"gorm.io/gorm"
|
|
)
|
|
|
|
// The outcome sweeper.
|
|
//
|
|
// Same shape as internal/assignment/sweeper.go and internal/prediction/sweeper.go:
|
|
// a ticker, a recover() per tick, and a Redis lock that expires before the next
|
|
// tick so one replica of three does the work. Without Redis every replica
|
|
// sweeps, which is wasteful but correct — each update is idempotent and scoped
|
|
// to rows that are still pending.
|
|
|
|
const (
|
|
defaultOutcomeSweepSeconds = 900 // 15m
|
|
minOutcomeSweepSeconds = 120
|
|
outcomeLockKey = "ai:outcome-sweep:lock"
|
|
)
|
|
|
|
func sweepInterval() time.Duration {
|
|
if v := strings.TrimSpace(os.Getenv("AGENT_OUTCOME_SWEEP_SECONDS")); v != "" {
|
|
if n, err := strconv.Atoi(v); err == nil && n >= 0 {
|
|
if n > 0 && n < minOutcomeSweepSeconds {
|
|
n = minOutcomeSweepSeconds
|
|
}
|
|
return time.Duration(n) * time.Second
|
|
}
|
|
utils.Warn("AGENT_OUTCOME_SWEEP_SECONDS is not a non-negative integer, using the default",
|
|
"value", v, "default_seconds", defaultOutcomeSweepSeconds)
|
|
}
|
|
return defaultOutcomeSweepSeconds * time.Second
|
|
}
|
|
|
|
// PruneFindings is set at boot by main.go to controllers.PruneAIFindings.
|
|
// A function variable rather than a direct call because controllers imports
|
|
// most of the codebase, and this package is imported BY controllers' siblings —
|
|
// calling it directly would be an import cycle.
|
|
var PruneFindings func(retentionDays int) (int, error)
|
|
|
|
// StartOutcomeSweeper judges pending decisions on a timer, and prunes old ones.
|
|
// Call once at boot, in a goroutine.
|
|
func StartOutcomeSweeper() {
|
|
interval := sweepInterval()
|
|
if interval == 0 {
|
|
utils.Info("OutcomeSweeper: disabled (AGENT_OUTCOME_SWEEP_SECONDS=0)")
|
|
return
|
|
}
|
|
utils.Info("OutcomeSweeper: started",
|
|
"interval", interval.String(), "window", OutcomeWindow().String(),
|
|
"retention_days", RetentionDays())
|
|
|
|
ticker := time.NewTicker(interval)
|
|
defer ticker.Stop()
|
|
for range ticker.C {
|
|
sweepOnce(interval)
|
|
}
|
|
}
|
|
|
|
func sweepOnce(interval time.Duration) {
|
|
defer func() {
|
|
if r := recover(); r != nil {
|
|
utils.Error("OutcomeSweeper: panic recovered", "error", r)
|
|
}
|
|
}()
|
|
if db.DB == nil {
|
|
return
|
|
}
|
|
if db.Rdb != nil {
|
|
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second)
|
|
ttl := interval - 30*time.Second
|
|
if ttl <= 0 {
|
|
ttl = interval / 2
|
|
}
|
|
got, err := db.Rdb.SetNX(ctx, outcomeLockKey, "1", ttl).Result()
|
|
cancel()
|
|
if err == nil && !got {
|
|
return
|
|
}
|
|
}
|
|
|
|
judged, err := JudgePending(db.DB, time.Now())
|
|
if err != nil {
|
|
utils.Error("OutcomeSweeper: judging failed", "error", err)
|
|
}
|
|
pruned, err := Prune(db.DB)
|
|
if err != nil {
|
|
utils.Error("OutcomeSweeper: prune failed", "error", err)
|
|
}
|
|
// Skill findings share this tick rather than carrying their own timer:
|
|
// one more table to keep tidy, not one more goroutine. Injected so this
|
|
// package does not import controllers (which imports everything).
|
|
findingsPruned := 0
|
|
if PruneFindings != nil {
|
|
if n, err := PruneFindings(RetentionDays()); err != nil {
|
|
utils.Error("OutcomeSweeper: finding prune failed", "error", err)
|
|
} else {
|
|
findingsPruned = n
|
|
}
|
|
}
|
|
if judged > 0 || pruned > 0 || findingsPruned > 0 {
|
|
utils.Info("OutcomeSweeper: swept",
|
|
"judged", judged, "decisions_pruned", pruned, "findings_pruned", findingsPruned)
|
|
}
|
|
}
|
|
|
|
// pendingRow is one unjudged decision joined to the booking it concerned.
|
|
//
|
|
// The join runs through pickupbookings on the decision's own booking_id, which
|
|
// the engine supplies — not through consignments, which has no bookingid column
|
|
// (the hazard CLAUDE.md §8.5 and docs/prediction-plan.md H4 describe). The
|
|
// delivered timestamp therefore comes from consignmenthistory via the
|
|
// consignment_booking view, the one place that resolution is written down.
|
|
type pendingRow struct {
|
|
ID uint64
|
|
DecisionType string
|
|
Bookingid *int
|
|
Status string
|
|
Assignedat *time.Time
|
|
Assigned bool
|
|
Createdat time.Time
|
|
Deliveredat *time.Time
|
|
Sladueat *time.Time
|
|
}
|
|
|
|
const pendingSQL = `
|
|
SELECT d.id AS id,
|
|
d.decision_type AS decision_type,
|
|
d.booking_id AS bookingid,
|
|
COALESCE(pb.status, '') AS status,
|
|
ba.assignedat AS assignedat,
|
|
(pb.assignedmileruserid IS NOT NULL) AS assigned,
|
|
d.created_at AS createdat,
|
|
del.createdat AS deliveredat,
|
|
con.sladueat AS sladueat
|
|
FROM agent_decisions d
|
|
LEFT JOIN pickupbookings pb ON pb.bookingid = d.booking_id
|
|
LEFT JOIN (
|
|
SELECT DISTINCT ON (bookingid) bookingid, assignedat
|
|
FROM bookingassignments
|
|
ORDER BY bookingid, assignedat DESC NULLS LAST, bookingassignmentid DESC
|
|
) ba ON ba.bookingid = d.booking_id
|
|
LEFT JOIN consignment_booking cb ON cb.bookingid = d.booking_id
|
|
LEFT JOIN consignments con ON con.consignmentid = cb.consignmentid
|
|
LEFT JOIN (
|
|
SELECT DISTINCT ON (consignmentid) consignmentid, createdat
|
|
FROM consignmenthistory
|
|
WHERE eventstatus = ?
|
|
ORDER BY consignmentid, createdat ASC
|
|
) del ON del.consignmentid = cb.consignmentid
|
|
WHERE d.outcome IS NULL
|
|
ORDER BY d.created_at ASC
|
|
LIMIT ?
|
|
`
|
|
|
|
// JudgePending labels every pending decision it can and returns how many it
|
|
// wrote. A decision it cannot judge yet is left pending for the next sweep.
|
|
func JudgePending(gdb *gorm.DB, now time.Time) (int, error) {
|
|
if gdb == nil {
|
|
return 0, nil
|
|
}
|
|
|
|
var rows []pendingRow
|
|
if err := gdb.Raw(pendingSQL, constants.ConsignmentDelivered, sweepBatch).Scan(&rows).Error; err != nil {
|
|
return 0, err
|
|
}
|
|
|
|
window := OutcomeWindow()
|
|
judged := 0
|
|
for _, r := range rows {
|
|
facts := bookingFacts{
|
|
Status: r.Status,
|
|
Assigned: r.Assigned,
|
|
DecidedAt: r.Createdat,
|
|
DeliveredAt: r.Deliveredat,
|
|
SLADueAt: r.Sladueat,
|
|
AssignedAt: r.Assignedat,
|
|
Cancelled: r.Status == constants.BookingCancelled,
|
|
}
|
|
if r.Bookingid != nil {
|
|
facts.Bookingid = *r.Bookingid
|
|
}
|
|
|
|
outcome, ok := judge(r.DecisionType, facts, now, window)
|
|
if !ok {
|
|
continue
|
|
}
|
|
// Guarded on outcome IS NULL so two replicas sweeping concurrently
|
|
// cannot overwrite each other, and a decision is judged exactly once.
|
|
res := gdb.Exec(
|
|
`UPDATE agent_decisions SET outcome = ?, outcome_recorded_at = ? WHERE id = ? AND outcome IS NULL`,
|
|
outcome, utils.DBNow(), r.ID,
|
|
)
|
|
if res.Error != nil {
|
|
utils.Warn("OutcomeSweeper: could not record outcome", "id", r.ID, "error", res.Error)
|
|
continue
|
|
}
|
|
judged += int(res.RowsAffected)
|
|
}
|
|
return judged, nil
|
|
}
|
|
|
|
// Prune drops decisions past the retention window. agent_decisions had no
|
|
// retention at all while aiagentruns purged at 30 days — and this is the table
|
|
// the similarity query scans, so unbounded growth degrades every recall.
|
|
func Prune(gdb *gorm.DB) (int, error) {
|
|
if gdb == nil {
|
|
return 0, nil
|
|
}
|
|
cutoff := utils.DBNow().AddDate(0, 0, -RetentionDays())
|
|
res := gdb.Exec(`DELETE FROM agent_decisions WHERE created_at < ?`, cutoff)
|
|
if res.Error != nil {
|
|
return 0, res.Error
|
|
}
|
|
return int(res.RowsAffected), nil
|
|
}
|