updates on the ai agents and time series prediction and updates on the api to
This commit is contained in:
240
internal/ai/outcomes/backfill.go
Normal file
240
internal/ai/outcomes/backfill.go
Normal file
@@ -0,0 +1,240 @@
|
||||
package outcomes
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"time"
|
||||
|
||||
"doormile/constants"
|
||||
"doormile/models"
|
||||
"doormile/utils"
|
||||
|
||||
"gorm.io/gorm"
|
||||
)
|
||||
|
||||
// Seeding the decision memory from history, so recall is not useless for weeks.
|
||||
//
|
||||
// ─── The cold-start problem this solves ────────────────────────────────────
|
||||
//
|
||||
// `/internal/agent-decisions/similar` filters on `outcome IS NOT NULL`. A row
|
||||
// becomes eligible only after the outcome sweeper has judged it, and a decision
|
||||
// is only judged once its window has closed. So the sequence for a freshly
|
||||
// switched-on memory is:
|
||||
//
|
||||
// switch embeddings on -> wait for stalls to happen -> wait 48h per decision
|
||||
// -> the sweeper labels them -> only now does recall return anything
|
||||
//
|
||||
// With autonomy gates off and two decision types, that is weeks of recall
|
||||
// returning [] — which reads as "retrieval does not help here" rather than
|
||||
// "retrieval has nothing to retrieve yet". The first conclusion is wrong and
|
||||
// expensive to un-learn.
|
||||
//
|
||||
// This backfills decisions from bookings whose outcome is ALREADY known.
|
||||
//
|
||||
// ─── What it does and does not claim ───────────────────────────────────────
|
||||
//
|
||||
// A backfilled row is not a decision the engine made. It is a record of a
|
||||
// situation that occurred and how it ended, shaped so the retrieval path can
|
||||
// use it as precedent. That distinction is recorded honestly:
|
||||
//
|
||||
// decision.action = "none" — nothing was decided; nobody intervened
|
||||
// decision.source = "backfill" — so these are distinguishable forever
|
||||
// reasoning = states plainly that this is historical, not a decision
|
||||
//
|
||||
// Why that matters: as_prompt_block renders `action` to the model. Writing a
|
||||
// plausible-looking action here would teach the model that an action it never
|
||||
// took produced the outcome that followed — which is worse than no memory.
|
||||
// "none" is honest: this is what happened when nothing was done.
|
||||
//
|
||||
// ─── It writes no embeddings ───────────────────────────────────────────────
|
||||
//
|
||||
// Embedding is the engine's job (AI_engine/core/embeddings.py) and needs the
|
||||
// provider key, which this process does not have. Backfilled rows land with
|
||||
// NULL context_embedding and are invisible to similarity search until
|
||||
// something embeds them. That is deliberate: a backfill that silently created
|
||||
// un-embedded rows AND claimed to have seeded the memory would be the worse
|
||||
// failure. BackfillStats reports the count so the caller knows what is owed.
|
||||
|
||||
// BackfillStats is what one run produced.
|
||||
type BackfillStats struct {
|
||||
Scanned int `json:"scanned"`
|
||||
Inserted int `json:"inserted"`
|
||||
Skipped int `json:"skipped"`
|
||||
// NeedsEmbedding is Inserted — every backfilled row still needs a vector
|
||||
// before it can be retrieved. Surfaced separately so it cannot be missed.
|
||||
NeedsEmbedding int `json:"needsembedding"`
|
||||
}
|
||||
|
||||
// historicalRow is one past booking whose ending is known.
|
||||
//
|
||||
// Joins through consignment_booking, never consignments.bookingid — that column
|
||||
// does not exist, and the legacy pickupbookings.consignmentid link names only
|
||||
// the first order of a multi-destination pickup (hazard H4).
|
||||
type historicalRow struct {
|
||||
Bookingid int
|
||||
Tenantid *uint64
|
||||
Deliverypincode string
|
||||
Status string
|
||||
Createdat time.Time
|
||||
Deliveredat *time.Time
|
||||
Sladueat *time.Time
|
||||
Attemptcount int
|
||||
Chargeableweight float64
|
||||
}
|
||||
|
||||
const historicalSQL = `
|
||||
SELECT pb.bookingid,
|
||||
pb.tenantid AS tenantid,
|
||||
COALESCE(c.deliverypincode, '') AS deliverypincode,
|
||||
COALESCE(c.status, pb.status) AS status,
|
||||
c.createdat AS createdat,
|
||||
del.createdat AS deliveredat,
|
||||
c.sladueat AS sladueat,
|
||||
COALESCE(c.attemptcount, 0) AS attemptcount,
|
||||
COALESCE(c.chargeableweight, 0) AS chargeableweight
|
||||
FROM consignments c
|
||||
JOIN consignment_booking cb ON cb.consignmentid = c.consignmentid
|
||||
JOIN pickupbookings pb ON pb.bookingid = cb.bookingid
|
||||
LEFT JOIN (
|
||||
SELECT DISTINCT ON (consignmentid) consignmentid, createdat
|
||||
FROM consignmenthistory
|
||||
WHERE eventstatus = ?
|
||||
ORDER BY consignmentid, createdat ASC
|
||||
) del ON del.consignmentid = c.consignmentid
|
||||
WHERE c.deletedat IS NULL
|
||||
-- Only parcels whose story has ended. An in-flight parcel has no outcome to
|
||||
-- learn from, and guessing one is exactly what this must not do.
|
||||
AND (del.createdat IS NOT NULL OR c.status IN ?)
|
||||
-- Not already backfilled or decided for. The uniqueness is on
|
||||
-- (decision_type, booking_id), enforced here rather than by a constraint
|
||||
-- because real decisions legitimately repeat for one booking.
|
||||
AND NOT EXISTS (
|
||||
SELECT 1 FROM agent_decisions d
|
||||
WHERE d.booking_id = pb.bookingid AND d.decision_type = ?
|
||||
)
|
||||
ORDER BY c.createdat DESC
|
||||
LIMIT ?
|
||||
`
|
||||
|
||||
// BackfillDecisionType is its own type, kept separate from the engine's
|
||||
// `stall_response` and `assignment_failure`. Recall is per-type, so backfilled
|
||||
// precedent is retrievable on purpose and never silently mixed into a type the
|
||||
// engine thinks it authored.
|
||||
const BackfillDecisionType = "historical_delivery"
|
||||
|
||||
// Backfill writes historical precedent rows. Idempotent: a booking that already
|
||||
// has a decision of this type is skipped, so re-running adds only what is new.
|
||||
//
|
||||
// limit bounds one run — this scans delivery history, which is the largest
|
||||
// table pair in the database.
|
||||
func Backfill(gdb *gorm.DB, limit int) (BackfillStats, error) {
|
||||
if gdb == nil {
|
||||
return BackfillStats{}, nil
|
||||
}
|
||||
if limit <= 0 {
|
||||
limit = 2000
|
||||
}
|
||||
|
||||
terminal := []string{
|
||||
constants.ConsignmentDelivered,
|
||||
"Returned_to_Sender",
|
||||
"Missing",
|
||||
"Damaged",
|
||||
}
|
||||
|
||||
var rows []historicalRow
|
||||
if err := gdb.Raw(historicalSQL,
|
||||
constants.ConsignmentDelivered, terminal, BackfillDecisionType, limit,
|
||||
).Scan(&rows).Error; err != nil {
|
||||
return BackfillStats{}, err
|
||||
}
|
||||
|
||||
stats := BackfillStats{Scanned: len(rows)}
|
||||
batch := make([]models.AgentDecision, 0, len(rows))
|
||||
|
||||
for _, r := range rows {
|
||||
outcome := OutcomeFailure
|
||||
switch {
|
||||
case r.Deliveredat == nil:
|
||||
// Terminal but never delivered: returned, lost or damaged.
|
||||
outcome = OutcomeFailure
|
||||
case r.Sladueat != nil && r.Deliveredat.After(*r.Sladueat):
|
||||
// Delivered late. Counting this a success would teach that any
|
||||
// eventual delivery is a good outcome — the same trap the
|
||||
// stall_response rule avoids.
|
||||
outcome = OutcomeFailure
|
||||
default:
|
||||
outcome = OutcomeSuccess
|
||||
}
|
||||
|
||||
// The facts shape mirrors what AI_engine puts in context.facts, so an
|
||||
// embedding of a backfilled row sits in the same space as a real one.
|
||||
// If these diverge, retrieval returns neighbours that are near in
|
||||
// vector space for the wrong reasons.
|
||||
facts := map[string]any{
|
||||
"booking_id": r.Bookingid,
|
||||
"delivery_pincode": r.Deliverypincode,
|
||||
"attempt_count": r.Attemptcount,
|
||||
"chargeable_weight": r.Chargeableweight,
|
||||
"final_status": r.Status,
|
||||
}
|
||||
if r.Deliveredat != nil {
|
||||
facts["hours_to_deliver"] = int(r.Deliveredat.Sub(r.Createdat).Hours())
|
||||
}
|
||||
|
||||
contextJSON, err := json.Marshal(map[string]any{
|
||||
"facts": facts,
|
||||
"model": "none",
|
||||
"source": "backfill",
|
||||
})
|
||||
if err != nil {
|
||||
stats.Skipped++
|
||||
continue
|
||||
}
|
||||
decisionJSON, err := json.Marshal(map[string]any{
|
||||
// Honest: nobody decided anything. See the package comment — a
|
||||
// plausible-looking action here would be a fabricated lesson.
|
||||
"action": "none",
|
||||
"confidence": 0.0,
|
||||
"source": "backfill",
|
||||
})
|
||||
if err != nil {
|
||||
stats.Skipped++
|
||||
continue
|
||||
}
|
||||
|
||||
recordedAt := utils.DBNow()
|
||||
batch = append(batch, models.AgentDecision{
|
||||
DecisionType: BackfillDecisionType,
|
||||
BookingID: u64(r.Bookingid),
|
||||
TenantID: r.Tenantid,
|
||||
Context: string(contextJSON),
|
||||
Decision: string(decisionJSON),
|
||||
Reasoning: "Historical outcome backfilled from delivery records. No agent decision was made for this booking; this row records what happened when nothing intervened.",
|
||||
Outcome: &outcome,
|
||||
OutcomeRecordedAt: &recordedAt,
|
||||
CreatedAt: r.Createdat,
|
||||
})
|
||||
}
|
||||
|
||||
if len(batch) == 0 {
|
||||
return stats, nil
|
||||
}
|
||||
if err := gdb.CreateInBatches(&batch, 200).Error; err != nil {
|
||||
return stats, err
|
||||
}
|
||||
stats.Inserted = len(batch)
|
||||
stats.NeedsEmbedding = len(batch)
|
||||
|
||||
utils.Info("outcomes: backfilled historical precedent",
|
||||
"scanned", stats.Scanned, "inserted", stats.Inserted, "skipped", stats.Skipped,
|
||||
"note", "rows have no embedding yet and are not retrievable until one is written")
|
||||
return stats, nil
|
||||
}
|
||||
|
||||
func u64(n int) *uint64 {
|
||||
if n <= 0 {
|
||||
return nil
|
||||
}
|
||||
v := uint64(n)
|
||||
return &v
|
||||
}
|
||||
158
internal/ai/outcomes/rules.go
Normal file
158
internal/ai/outcomes/rules.go
Normal file
@@ -0,0 +1,158 @@
|
||||
// Package outcomes decides, after the fact, whether an agent's decision worked.
|
||||
//
|
||||
// This is the half of the decision memory that was missing, and without it the
|
||||
// other half does nothing. `/internal/agent-decisions/similar` filters on
|
||||
// `outcome IS NOT NULL`, so until something judges a decision it is invisible
|
||||
// to retrieval. Embeddings could flow for months and every recall would still
|
||||
// come back empty.
|
||||
//
|
||||
// A decision is not precedent because it was made. It is precedent because we
|
||||
// know how it turned out.
|
||||
//
|
||||
// ─── What "worked" means ──────────────────────────────────────────────────
|
||||
//
|
||||
// Two decision types exist today, from exactly two call sites in the engine:
|
||||
//
|
||||
// stall_response AI_engine/agents/exception_agent.py:456
|
||||
// assignment_failure AI_engine/agents/dispatch_agent.py:335
|
||||
//
|
||||
// Each gets its own definition below. A third type appearing without a rule
|
||||
// here is left pending rather than guessed at — a wrong label is worse than no
|
||||
// label, because it teaches the model confidently.
|
||||
package outcomes
|
||||
|
||||
import (
|
||||
"os"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"doormile/utils"
|
||||
)
|
||||
|
||||
// The outcome vocabulary. Stored in agent_decisions.outcome (varchar 30) and
|
||||
// read back by the Insights tab, which groups by it.
|
||||
const (
|
||||
// OutcomeSuccess — the thing the decision was trying to achieve happened.
|
||||
OutcomeSuccess = "success"
|
||||
// OutcomeFailure — it did not.
|
||||
OutcomeFailure = "failure"
|
||||
// OutcomeUnknown — the window closed without enough evidence either way.
|
||||
// Deliberately recorded rather than left pending: a pending row is
|
||||
// retried every sweep forever, and an unjudgeable decision should stop
|
||||
// costing a scan. It is excluded from retrieval the same as pending,
|
||||
// because `outcome IS NOT NULL` is not the only filter that matters —
|
||||
// see judgeable().
|
||||
OutcomeUnknown = "unknown"
|
||||
)
|
||||
|
||||
const (
|
||||
defaultOutcomeWindowHours = 48
|
||||
// A decision younger than this is left alone: the booking it concerns is
|
||||
// probably still in flight, and judging it now would record a failure for
|
||||
// something that simply has not finished.
|
||||
minDecisionAge = 30 * time.Minute
|
||||
// Caps one sweep's work; the rest are next sweep's.
|
||||
sweepBatch = 500
|
||||
)
|
||||
|
||||
// OutcomeWindow is how long after a decision its result is judged.
|
||||
// AGENT_OUTCOME_WINDOW_HOURS, default 48.
|
||||
//
|
||||
// The window is a real tradeoff, not a tuning knob. Too short and a parcel
|
||||
// that was always going to take three days is recorded as a failure of the
|
||||
// decision rather than of the promise. Too long and the memory learns slowly.
|
||||
// 48h matches the Standard service SLA (36h) with headroom.
|
||||
func OutcomeWindow() time.Duration {
|
||||
if v := strings.TrimSpace(os.Getenv("AGENT_OUTCOME_WINDOW_HOURS")); v != "" {
|
||||
if n, err := strconv.Atoi(v); err == nil && n > 0 {
|
||||
return time.Duration(n) * time.Hour
|
||||
}
|
||||
utils.Warn("AGENT_OUTCOME_WINDOW_HOURS is not a positive integer, using the default",
|
||||
"value", v, "default_hours", defaultOutcomeWindowHours)
|
||||
}
|
||||
return defaultOutcomeWindowHours * time.Hour
|
||||
}
|
||||
|
||||
// RetentionDays bounds how long decisions are kept. Longer than the 30 days
|
||||
// aiagentruns keeps, because old precedent is the whole point of this table —
|
||||
// but not unbounded, which is what it was.
|
||||
func RetentionDays() int {
|
||||
if v := strings.TrimSpace(os.Getenv("AGENT_DECISION_RETENTION_DAYS")); v != "" {
|
||||
if n, err := strconv.Atoi(v); err == nil && n > 0 {
|
||||
return n
|
||||
}
|
||||
}
|
||||
return 180
|
||||
}
|
||||
|
||||
// bookingFacts is what the sweeper reads about the booking a decision concerned.
|
||||
// Timestamps come from columns written with CURRENT_TIMESTAMP defaults, so they
|
||||
// are consistent with each other and their differences are correct regardless
|
||||
// of the IST-digits-labelled-UTC convention (utils.DBNow) — the same reasoning
|
||||
// internal/prediction's calibration relies on.
|
||||
type bookingFacts struct {
|
||||
Bookingid int
|
||||
Status string
|
||||
Assigned bool
|
||||
DecidedAt time.Time
|
||||
DeliveredAt *time.Time
|
||||
SLADueAt *time.Time
|
||||
AssignedAt *time.Time
|
||||
Cancelled bool
|
||||
}
|
||||
|
||||
// judge applies the per-type rule. Returns the outcome and whether the decision
|
||||
// is judgeable at all; a false means leave it pending for now.
|
||||
func judge(decisionType string, f bookingFacts, now time.Time, window time.Duration) (string, bool) {
|
||||
age := now.Sub(f.DecidedAt)
|
||||
if age < minDecisionAge {
|
||||
return "", false // too soon; the booking is still in flight
|
||||
}
|
||||
|
||||
switch decisionType {
|
||||
case "stall_response":
|
||||
// The agent intervened because a rider had stopped making progress.
|
||||
// It worked if the parcel reached the customer, and reached them
|
||||
// within the promise that was in force.
|
||||
if f.Cancelled {
|
||||
return OutcomeFailure, true
|
||||
}
|
||||
if f.DeliveredAt != nil {
|
||||
if f.SLADueAt != nil && f.DeliveredAt.After(*f.SLADueAt) {
|
||||
// Delivered, but late. Counting this as success would teach
|
||||
// the model that any eventual delivery vindicates the action.
|
||||
return OutcomeFailure, true
|
||||
}
|
||||
return OutcomeSuccess, true
|
||||
}
|
||||
if age >= window {
|
||||
// Window closed, never delivered, not cancelled — stuck.
|
||||
return OutcomeFailure, true
|
||||
}
|
||||
return "", false
|
||||
|
||||
case "assignment_failure":
|
||||
// The agent reasoned about why no rider could be found. It worked if
|
||||
// the booking subsequently got one.
|
||||
if f.Cancelled {
|
||||
return OutcomeFailure, true
|
||||
}
|
||||
if f.Assigned && f.AssignedAt != nil && f.AssignedAt.After(f.DecidedAt) {
|
||||
return OutcomeSuccess, true
|
||||
}
|
||||
if age >= window {
|
||||
return OutcomeFailure, true
|
||||
}
|
||||
return "", false
|
||||
|
||||
default:
|
||||
// An unrecognised decision type. Judge it unknown once the window has
|
||||
// closed so it stops being rescanned, but never guess success or
|
||||
// failure — a wrong label is worse than no label.
|
||||
if age >= window {
|
||||
return OutcomeUnknown, true
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
}
|
||||
214
internal/ai/outcomes/rules_test.go
Normal file
214
internal/ai/outcomes/rules_test.go
Normal file
@@ -0,0 +1,214 @@
|
||||
package outcomes
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func at(h int) time.Time {
|
||||
return time.Date(2026, 10, 5, h, 0, 0, 0, time.UTC)
|
||||
}
|
||||
|
||||
func tp(t time.Time) *time.Time { return &t }
|
||||
|
||||
const window = 48 * time.Hour
|
||||
|
||||
// ─── Too soon to judge ─────────────────────────────────────────────────────
|
||||
|
||||
func TestYoungDecisionIsLeftPending(t *testing.T) {
|
||||
decided := at(10)
|
||||
for _, dt := range []string{"stall_response", "assignment_failure", "something_new"} {
|
||||
_, ok := judge(dt, bookingFacts{DecidedAt: decided}, decided.Add(5*time.Minute), window)
|
||||
if ok {
|
||||
t.Errorf("%s: judged a 5-minute-old decision; the booking is still in flight", dt)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestPendingWhileInsideTheWindow(t *testing.T) {
|
||||
decided := at(10)
|
||||
// An hour in: no delivery yet, but the window has not closed. Recording
|
||||
// failure here would blame the decision for a parcel still on its way.
|
||||
if _, ok := judge("stall_response", bookingFacts{DecidedAt: decided}, decided.Add(time.Hour), window); ok {
|
||||
t.Error("stall_response judged before its window closed")
|
||||
}
|
||||
if _, ok := judge("assignment_failure", bookingFacts{DecidedAt: decided}, decided.Add(time.Hour), window); ok {
|
||||
t.Error("assignment_failure judged before its window closed")
|
||||
}
|
||||
}
|
||||
|
||||
// ─── stall_response ────────────────────────────────────────────────────────
|
||||
|
||||
func TestStallDeliveredOnTimeIsSuccess(t *testing.T) {
|
||||
decided := at(10)
|
||||
f := bookingFacts{
|
||||
DecidedAt: decided,
|
||||
DeliveredAt: tp(decided.Add(3 * time.Hour)),
|
||||
SLADueAt: tp(decided.Add(12 * time.Hour)),
|
||||
}
|
||||
got, ok := judge("stall_response", f, decided.Add(4*time.Hour), window)
|
||||
if !ok || got != OutcomeSuccess {
|
||||
t.Errorf("got %q ok=%v, want success", got, ok)
|
||||
}
|
||||
}
|
||||
|
||||
// The case that matters most: counting any eventual delivery as success would
|
||||
// teach the model that the action is always vindicated, which is exactly the
|
||||
// wrong lesson.
|
||||
func TestStallDeliveredLateIsFailure(t *testing.T) {
|
||||
decided := at(10)
|
||||
f := bookingFacts{
|
||||
DecidedAt: decided,
|
||||
DeliveredAt: tp(decided.Add(20 * time.Hour)),
|
||||
SLADueAt: tp(decided.Add(12 * time.Hour)),
|
||||
}
|
||||
got, ok := judge("stall_response", f, decided.Add(21*time.Hour), window)
|
||||
if !ok || got != OutcomeFailure {
|
||||
t.Errorf("got %q ok=%v, want failure — delivered 8h past SLA", got, ok)
|
||||
}
|
||||
}
|
||||
|
||||
func TestStallDeliveredWithNoSLAIsSuccess(t *testing.T) {
|
||||
decided := at(10)
|
||||
// No sladueat recorded (an older row). Delivery is the best evidence
|
||||
// available and there is nothing to call it late against.
|
||||
f := bookingFacts{DecidedAt: decided, DeliveredAt: tp(decided.Add(5 * time.Hour))}
|
||||
got, ok := judge("stall_response", f, decided.Add(6*time.Hour), window)
|
||||
if !ok || got != OutcomeSuccess {
|
||||
t.Errorf("got %q ok=%v, want success", got, ok)
|
||||
}
|
||||
}
|
||||
|
||||
func TestStallNeverDeliveredAfterWindowIsFailure(t *testing.T) {
|
||||
decided := at(10)
|
||||
got, ok := judge("stall_response", bookingFacts{DecidedAt: decided}, decided.Add(window+time.Hour), window)
|
||||
if !ok || got != OutcomeFailure {
|
||||
t.Errorf("got %q ok=%v, want failure", got, ok)
|
||||
}
|
||||
}
|
||||
|
||||
func TestStallCancelledIsFailureImmediately(t *testing.T) {
|
||||
decided := at(10)
|
||||
f := bookingFacts{DecidedAt: decided, Cancelled: true}
|
||||
// Cancellation is conclusive — no need to wait out the window.
|
||||
got, ok := judge("stall_response", f, decided.Add(time.Hour), window)
|
||||
if !ok || got != OutcomeFailure {
|
||||
t.Errorf("got %q ok=%v, want failure on cancellation", got, ok)
|
||||
}
|
||||
}
|
||||
|
||||
// ─── assignment_failure ────────────────────────────────────────────────────
|
||||
|
||||
func TestAssignmentLaterAssignedIsSuccess(t *testing.T) {
|
||||
decided := at(10)
|
||||
f := bookingFacts{
|
||||
DecidedAt: decided,
|
||||
Assigned: true,
|
||||
AssignedAt: tp(decided.Add(2 * time.Hour)),
|
||||
}
|
||||
got, ok := judge("assignment_failure", f, decided.Add(3*time.Hour), window)
|
||||
if !ok || got != OutcomeSuccess {
|
||||
t.Errorf("got %q ok=%v, want success", got, ok)
|
||||
}
|
||||
}
|
||||
|
||||
// A rider assigned BEFORE the decision is not evidence the decision worked —
|
||||
// it is the assignment that was already there. Without the ordering check,
|
||||
// every reassignment decision would read as an instant success.
|
||||
func TestAssignmentPredatingTheDecisionIsNotSuccess(t *testing.T) {
|
||||
decided := at(10)
|
||||
f := bookingFacts{
|
||||
DecidedAt: decided,
|
||||
Assigned: true,
|
||||
AssignedAt: tp(decided.Add(-2 * time.Hour)),
|
||||
}
|
||||
got, ok := judge("assignment_failure", f, decided.Add(time.Hour), window)
|
||||
if ok && got == OutcomeSuccess {
|
||||
t.Error("counted a pre-existing assignment as the decision's success")
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssignmentNeverAssignedAfterWindowIsFailure(t *testing.T) {
|
||||
decided := at(10)
|
||||
got, ok := judge("assignment_failure", bookingFacts{DecidedAt: decided}, decided.Add(window+time.Hour), window)
|
||||
if !ok || got != OutcomeFailure {
|
||||
t.Errorf("got %q ok=%v, want failure", got, ok)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssignmentCancelledIsFailure(t *testing.T) {
|
||||
decided := at(10)
|
||||
f := bookingFacts{DecidedAt: decided, Cancelled: true}
|
||||
got, ok := judge("assignment_failure", f, decided.Add(time.Hour), window)
|
||||
if !ok || got != OutcomeFailure {
|
||||
t.Errorf("got %q ok=%v, want failure", got, ok)
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Unknown types are never guessed ───────────────────────────────────────
|
||||
|
||||
func TestUnknownTypeIsNeverSuccessOrFailure(t *testing.T) {
|
||||
decided := at(10)
|
||||
// Inside the window: pending.
|
||||
if _, ok := judge("a_new_decision_type", bookingFacts{DecidedAt: decided}, decided.Add(time.Hour), window); ok {
|
||||
t.Error("judged an unknown decision type inside its window")
|
||||
}
|
||||
// Past it: unknown, so it stops being rescanned — but never a label that
|
||||
// would teach the model something nobody defined.
|
||||
got, ok := judge("a_new_decision_type", bookingFacts{DecidedAt: decided}, decided.Add(window+time.Hour), window)
|
||||
if !ok || got != OutcomeUnknown {
|
||||
t.Errorf("got %q ok=%v, want unknown", got, ok)
|
||||
}
|
||||
}
|
||||
|
||||
// A delivered unknown type must still not be called a success: the rule for
|
||||
// what success means for that type does not exist yet.
|
||||
func TestUnknownTypeWithDeliveryIsStillUnknown(t *testing.T) {
|
||||
decided := at(10)
|
||||
f := bookingFacts{DecidedAt: decided, DeliveredAt: tp(decided.Add(time.Hour))}
|
||||
got, ok := judge("a_new_decision_type", f, decided.Add(window+time.Hour), window)
|
||||
if !ok || got != OutcomeUnknown {
|
||||
t.Errorf("got %q ok=%v, want unknown", got, ok)
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Configuration ─────────────────────────────────────────────────────────
|
||||
|
||||
func TestOutcomeWindowDefaultAndOverride(t *testing.T) {
|
||||
t.Setenv("AGENT_OUTCOME_WINDOW_HOURS", "")
|
||||
if got := OutcomeWindow(); got != defaultOutcomeWindowHours*time.Hour {
|
||||
t.Errorf("default window = %v, want %v", got, defaultOutcomeWindowHours*time.Hour)
|
||||
}
|
||||
t.Setenv("AGENT_OUTCOME_WINDOW_HOURS", "12")
|
||||
if got := OutcomeWindow(); got != 12*time.Hour {
|
||||
t.Errorf("window = %v, want 12h", got)
|
||||
}
|
||||
// Garbage falls back rather than producing a zero window, which would
|
||||
// judge every decision the instant it passed minDecisionAge.
|
||||
t.Setenv("AGENT_OUTCOME_WINDOW_HOURS", "not-a-number")
|
||||
if got := OutcomeWindow(); got != defaultOutcomeWindowHours*time.Hour {
|
||||
t.Errorf("window on garbage = %v, want the default", got)
|
||||
}
|
||||
t.Setenv("AGENT_OUTCOME_WINDOW_HOURS", "0")
|
||||
if got := OutcomeWindow(); got != defaultOutcomeWindowHours*time.Hour {
|
||||
t.Errorf("window on 0 = %v, want the default", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRetentionDaysDefaultAndOverride(t *testing.T) {
|
||||
t.Setenv("AGENT_DECISION_RETENTION_DAYS", "")
|
||||
if got := RetentionDays(); got != 180 {
|
||||
t.Errorf("default retention = %d, want 180", got)
|
||||
}
|
||||
t.Setenv("AGENT_DECISION_RETENTION_DAYS", "90")
|
||||
if got := RetentionDays(); got != 90 {
|
||||
t.Errorf("retention = %d, want 90", got)
|
||||
}
|
||||
// Retention must be longer than aiagentruns' 30 days, because old
|
||||
// precedent is the point of this table. Not enforced in code — asserted
|
||||
// here so a future change to the default has to confront it.
|
||||
t.Setenv("AGENT_DECISION_RETENTION_DAYS", "")
|
||||
if RetentionDays() <= 30 {
|
||||
t.Error("decision retention is no longer than telemetry retention; precedent will be purged before it is useful")
|
||||
}
|
||||
}
|
||||
226
internal/ai/outcomes/sweeper.go
Normal file
226
internal/ai/outcomes/sweeper.go
Normal file
@@ -0,0 +1,226 @@
|
||||
package outcomes
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"doormile/constants"
|
||||
"doormile/db"
|
||||
"doormile/utils"
|
||||
|
||||
"gorm.io/gorm"
|
||||
)
|
||||
|
||||
// The outcome sweeper.
|
||||
//
|
||||
// Same shape as internal/assignment/sweeper.go and internal/prediction/sweeper.go:
|
||||
// a ticker, a recover() per tick, and a Redis lock that expires before the next
|
||||
// tick so one replica of three does the work. Without Redis every replica
|
||||
// sweeps, which is wasteful but correct — each update is idempotent and scoped
|
||||
// to rows that are still pending.
|
||||
|
||||
const (
|
||||
defaultOutcomeSweepSeconds = 900 // 15m
|
||||
minOutcomeSweepSeconds = 120
|
||||
outcomeLockKey = "ai:outcome-sweep:lock"
|
||||
)
|
||||
|
||||
func sweepInterval() time.Duration {
|
||||
if v := strings.TrimSpace(os.Getenv("AGENT_OUTCOME_SWEEP_SECONDS")); v != "" {
|
||||
if n, err := strconv.Atoi(v); err == nil && n >= 0 {
|
||||
if n > 0 && n < minOutcomeSweepSeconds {
|
||||
n = minOutcomeSweepSeconds
|
||||
}
|
||||
return time.Duration(n) * time.Second
|
||||
}
|
||||
utils.Warn("AGENT_OUTCOME_SWEEP_SECONDS is not a non-negative integer, using the default",
|
||||
"value", v, "default_seconds", defaultOutcomeSweepSeconds)
|
||||
}
|
||||
return defaultOutcomeSweepSeconds * time.Second
|
||||
}
|
||||
|
||||
// PruneFindings is set at boot by main.go to controllers.PruneAIFindings.
|
||||
// A function variable rather than a direct call because controllers imports
|
||||
// most of the codebase, and this package is imported BY controllers' siblings —
|
||||
// calling it directly would be an import cycle.
|
||||
var PruneFindings func(retentionDays int) (int, error)
|
||||
|
||||
// StartOutcomeSweeper judges pending decisions on a timer, and prunes old ones.
|
||||
// Call once at boot, in a goroutine.
|
||||
func StartOutcomeSweeper() {
|
||||
interval := sweepInterval()
|
||||
if interval == 0 {
|
||||
utils.Info("OutcomeSweeper: disabled (AGENT_OUTCOME_SWEEP_SECONDS=0)")
|
||||
return
|
||||
}
|
||||
utils.Info("OutcomeSweeper: started",
|
||||
"interval", interval.String(), "window", OutcomeWindow().String(),
|
||||
"retention_days", RetentionDays())
|
||||
|
||||
ticker := time.NewTicker(interval)
|
||||
defer ticker.Stop()
|
||||
for range ticker.C {
|
||||
sweepOnce(interval)
|
||||
}
|
||||
}
|
||||
|
||||
func sweepOnce(interval time.Duration) {
|
||||
defer func() {
|
||||
if r := recover(); r != nil {
|
||||
utils.Error("OutcomeSweeper: panic recovered", "error", r)
|
||||
}
|
||||
}()
|
||||
if db.DB == nil {
|
||||
return
|
||||
}
|
||||
if db.Rdb != nil {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second)
|
||||
ttl := interval - 30*time.Second
|
||||
if ttl <= 0 {
|
||||
ttl = interval / 2
|
||||
}
|
||||
got, err := db.Rdb.SetNX(ctx, outcomeLockKey, "1", ttl).Result()
|
||||
cancel()
|
||||
if err == nil && !got {
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
judged, err := JudgePending(db.DB, time.Now())
|
||||
if err != nil {
|
||||
utils.Error("OutcomeSweeper: judging failed", "error", err)
|
||||
}
|
||||
pruned, err := Prune(db.DB)
|
||||
if err != nil {
|
||||
utils.Error("OutcomeSweeper: prune failed", "error", err)
|
||||
}
|
||||
// Skill findings share this tick rather than carrying their own timer:
|
||||
// one more table to keep tidy, not one more goroutine. Injected so this
|
||||
// package does not import controllers (which imports everything).
|
||||
findingsPruned := 0
|
||||
if PruneFindings != nil {
|
||||
if n, err := PruneFindings(RetentionDays()); err != nil {
|
||||
utils.Error("OutcomeSweeper: finding prune failed", "error", err)
|
||||
} else {
|
||||
findingsPruned = n
|
||||
}
|
||||
}
|
||||
if judged > 0 || pruned > 0 || findingsPruned > 0 {
|
||||
utils.Info("OutcomeSweeper: swept",
|
||||
"judged", judged, "decisions_pruned", pruned, "findings_pruned", findingsPruned)
|
||||
}
|
||||
}
|
||||
|
||||
// pendingRow is one unjudged decision joined to the booking it concerned.
|
||||
//
|
||||
// The join runs through pickupbookings on the decision's own booking_id, which
|
||||
// the engine supplies — not through consignments, which has no bookingid column
|
||||
// (the hazard CLAUDE.md §8.5 and docs/prediction-plan.md H4 describe). The
|
||||
// delivered timestamp therefore comes from consignmenthistory via the
|
||||
// consignment_booking view, the one place that resolution is written down.
|
||||
type pendingRow struct {
|
||||
ID uint64
|
||||
DecisionType string
|
||||
Bookingid *int
|
||||
Status string
|
||||
Assignedat *time.Time
|
||||
Assigned bool
|
||||
Createdat time.Time
|
||||
Deliveredat *time.Time
|
||||
Sladueat *time.Time
|
||||
}
|
||||
|
||||
const pendingSQL = `
|
||||
SELECT d.id AS id,
|
||||
d.decision_type AS decision_type,
|
||||
d.booking_id AS bookingid,
|
||||
COALESCE(pb.status, '') AS status,
|
||||
ba.assignedat AS assignedat,
|
||||
(pb.assignedmileruserid IS NOT NULL) AS assigned,
|
||||
d.created_at AS createdat,
|
||||
del.createdat AS deliveredat,
|
||||
con.sladueat AS sladueat
|
||||
FROM agent_decisions d
|
||||
LEFT JOIN pickupbookings pb ON pb.bookingid = d.booking_id
|
||||
LEFT JOIN (
|
||||
SELECT DISTINCT ON (bookingid) bookingid, assignedat
|
||||
FROM bookingassignments
|
||||
ORDER BY bookingid, assignedat DESC NULLS LAST, bookingassignmentid DESC
|
||||
) ba ON ba.bookingid = d.booking_id
|
||||
LEFT JOIN consignment_booking cb ON cb.bookingid = d.booking_id
|
||||
LEFT JOIN consignments con ON con.consignmentid = cb.consignmentid
|
||||
LEFT JOIN (
|
||||
SELECT DISTINCT ON (consignmentid) consignmentid, createdat
|
||||
FROM consignmenthistory
|
||||
WHERE eventstatus = ?
|
||||
ORDER BY consignmentid, createdat ASC
|
||||
) del ON del.consignmentid = cb.consignmentid
|
||||
WHERE d.outcome IS NULL
|
||||
ORDER BY d.created_at ASC
|
||||
LIMIT ?
|
||||
`
|
||||
|
||||
// JudgePending labels every pending decision it can and returns how many it
|
||||
// wrote. A decision it cannot judge yet is left pending for the next sweep.
|
||||
func JudgePending(gdb *gorm.DB, now time.Time) (int, error) {
|
||||
if gdb == nil {
|
||||
return 0, nil
|
||||
}
|
||||
|
||||
var rows []pendingRow
|
||||
if err := gdb.Raw(pendingSQL, constants.ConsignmentDelivered, sweepBatch).Scan(&rows).Error; err != nil {
|
||||
return 0, err
|
||||
}
|
||||
|
||||
window := OutcomeWindow()
|
||||
judged := 0
|
||||
for _, r := range rows {
|
||||
facts := bookingFacts{
|
||||
Status: r.Status,
|
||||
Assigned: r.Assigned,
|
||||
DecidedAt: r.Createdat,
|
||||
DeliveredAt: r.Deliveredat,
|
||||
SLADueAt: r.Sladueat,
|
||||
AssignedAt: r.Assignedat,
|
||||
Cancelled: r.Status == constants.BookingCancelled,
|
||||
}
|
||||
if r.Bookingid != nil {
|
||||
facts.Bookingid = *r.Bookingid
|
||||
}
|
||||
|
||||
outcome, ok := judge(r.DecisionType, facts, now, window)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
// Guarded on outcome IS NULL so two replicas sweeping concurrently
|
||||
// cannot overwrite each other, and a decision is judged exactly once.
|
||||
res := gdb.Exec(
|
||||
`UPDATE agent_decisions SET outcome = ?, outcome_recorded_at = ? WHERE id = ? AND outcome IS NULL`,
|
||||
outcome, utils.DBNow(), r.ID,
|
||||
)
|
||||
if res.Error != nil {
|
||||
utils.Warn("OutcomeSweeper: could not record outcome", "id", r.ID, "error", res.Error)
|
||||
continue
|
||||
}
|
||||
judged += int(res.RowsAffected)
|
||||
}
|
||||
return judged, nil
|
||||
}
|
||||
|
||||
// Prune drops decisions past the retention window. agent_decisions had no
|
||||
// retention at all while aiagentruns purged at 30 days — and this is the table
|
||||
// the similarity query scans, so unbounded growth degrades every recall.
|
||||
func Prune(gdb *gorm.DB) (int, error) {
|
||||
if gdb == nil {
|
||||
return 0, nil
|
||||
}
|
||||
cutoff := utils.DBNow().AddDate(0, 0, -RetentionDays())
|
||||
res := gdb.Exec(`DELETE FROM agent_decisions WHERE created_at < ?`, cutoff)
|
||||
if res.Error != nil {
|
||||
return 0, res.Error
|
||||
}
|
||||
return int(res.RowsAffected), nil
|
||||
}
|
||||
Reference in New Issue
Block a user