route optimizer agent with dailygrubs ai assign

This commit is contained in:
2026-09-01 13:48:58 +05:30
parent aa4d4c6549
commit 90cdb57a38
28 changed files with 3281 additions and 135 deletions

34
evals/stall_eval.py Normal file
View File

@@ -0,0 +1,34 @@
"""
Offline eval for the ExceptionAgent stall decision (core.llm.decide_stall_response).
Scores whether the chosen action is *defensible* (in the case's ``acceptable``
set) and whether it matches the single ``ideal`` action. Stall handling is a
judgment task with more than one right answer, so acceptable-rate is the
headline metric. See evals/_harness.py for scoring details.
Usage:
export ANTHROPIC_API_KEY=...
python -m evals.stall_eval # built-in cases, 1 run each
python -m evals.stall_eval --runs 3 # sample each case 3x
python -m evals.stall_eval --model claude-haiku-4-5
python -m evals.stall_eval --dry-run # print contexts, no API calls
python -m evals.stall_eval --min-pass-rate 0.9 # CI gate
Add real production stall cases to stall_cases.jsonl as they occur — that is
what makes the number trustworthy before flipping EXCEPTION_AGENT_AUTONOMOUS.
"""
from pathlib import Path
import core.llm as llm
from core.llm import build_stall_context, decide_stall_response
from evals._harness import case_now, run_cli
CASES_PATH = Path(__file__).with_name("stall_cases.jsonl")
def to_context(case):
return build_stall_context(case["minutes_stalled"], now=case_now(case), facts=case.get("facts"))
if __name__ == "__main__":
run_cli(CASES_PATH, to_context, decide_stall_response, llm)