// Package outcomes decides, after the fact, whether an agent's decision worked. // // This is the half of the decision memory that was missing, and without it the // other half does nothing. `/internal/agent-decisions/similar` filters on // `outcome IS NOT NULL`, so until something judges a decision it is invisible // to retrieval. Embeddings could flow for months and every recall would still // come back empty. // // A decision is not precedent because it was made. It is precedent because we // know how it turned out. // // ─── What "worked" means ────────────────────────────────────────────────── // // Two decision types exist today, from exactly two call sites in the engine: // // stall_response AI_engine/agents/exception_agent.py:456 // assignment_failure AI_engine/agents/dispatch_agent.py:335 // // Each gets its own definition below. A third type appearing without a rule // here is left pending rather than guessed at — a wrong label is worse than no // label, because it teaches the model confidently. package outcomes import ( "os" "strconv" "strings" "time" "doormile/utils" ) // The outcome vocabulary. Stored in agent_decisions.outcome (varchar 30) and // read back by the Insights tab, which groups by it. const ( // OutcomeSuccess — the thing the decision was trying to achieve happened. OutcomeSuccess = "success" // OutcomeFailure — it did not. OutcomeFailure = "failure" // OutcomeUnknown — the window closed without enough evidence either way. // Deliberately recorded rather than left pending: a pending row is // retried every sweep forever, and an unjudgeable decision should stop // costing a scan. It is excluded from retrieval the same as pending, // because `outcome IS NOT NULL` is not the only filter that matters — // see judgeable(). OutcomeUnknown = "unknown" ) const ( defaultOutcomeWindowHours = 48 // A decision younger than this is left alone: the booking it concerns is // probably still in flight, and judging it now would record a failure for // something that simply has not finished. minDecisionAge = 30 * time.Minute // Caps one sweep's work; the rest are next sweep's. sweepBatch = 500 ) // OutcomeWindow is how long after a decision its result is judged. // AGENT_OUTCOME_WINDOW_HOURS, default 48. // // The window is a real tradeoff, not a tuning knob. Too short and a parcel // that was always going to take three days is recorded as a failure of the // decision rather than of the promise. Too long and the memory learns slowly. // 48h matches the Standard service SLA (36h) with headroom. func OutcomeWindow() time.Duration { if v := strings.TrimSpace(os.Getenv("AGENT_OUTCOME_WINDOW_HOURS")); v != "" { if n, err := strconv.Atoi(v); err == nil && n > 0 { return time.Duration(n) * time.Hour } utils.Warn("AGENT_OUTCOME_WINDOW_HOURS is not a positive integer, using the default", "value", v, "default_hours", defaultOutcomeWindowHours) } return defaultOutcomeWindowHours * time.Hour } // RetentionDays bounds how long decisions are kept. Longer than the 30 days // aiagentruns keeps, because old precedent is the whole point of this table — // but not unbounded, which is what it was. func RetentionDays() int { if v := strings.TrimSpace(os.Getenv("AGENT_DECISION_RETENTION_DAYS")); v != "" { if n, err := strconv.Atoi(v); err == nil && n > 0 { return n } } return 180 } // bookingFacts is what the sweeper reads about the booking a decision concerned. // Timestamps come from columns written with CURRENT_TIMESTAMP defaults, so they // are consistent with each other and their differences are correct regardless // of the IST-digits-labelled-UTC convention (utils.DBNow) — the same reasoning // internal/prediction's calibration relies on. type bookingFacts struct { Bookingid int Status string Assigned bool DecidedAt time.Time DeliveredAt *time.Time SLADueAt *time.Time AssignedAt *time.Time Cancelled bool } // judge applies the per-type rule. Returns the outcome and whether the decision // is judgeable at all; a false means leave it pending for now. func judge(decisionType string, f bookingFacts, now time.Time, window time.Duration) (string, bool) { age := now.Sub(f.DecidedAt) if age < minDecisionAge { return "", false // too soon; the booking is still in flight } switch decisionType { case "stall_response": // The agent intervened because a rider had stopped making progress. // It worked if the parcel reached the customer, and reached them // within the promise that was in force. if f.Cancelled { return OutcomeFailure, true } if f.DeliveredAt != nil { if f.SLADueAt != nil && f.DeliveredAt.After(*f.SLADueAt) { // Delivered, but late. Counting this as success would teach // the model that any eventual delivery vindicates the action. return OutcomeFailure, true } return OutcomeSuccess, true } if age >= window { // Window closed, never delivered, not cancelled — stuck. return OutcomeFailure, true } return "", false case "assignment_failure": // The agent reasoned about why no rider could be found. It worked if // the booking subsequently got one. if f.Cancelled { return OutcomeFailure, true } if f.Assigned && f.AssignedAt != nil && f.AssignedAt.After(f.DecidedAt) { return OutcomeSuccess, true } if age >= window { return OutcomeFailure, true } return "", false default: // An unrecognised decision type. Judge it unknown once the window has // closed so it stops being rescanned, but never guess success or // failure — a wrong label is worse than no label. if age >= window { return OutcomeUnknown, true } return "", false } }