route optimizer agent with dailygrubs ai assign
This commit is contained in:
34
evals/stall_eval.py
Normal file
34
evals/stall_eval.py
Normal file
@@ -0,0 +1,34 @@
|
||||
"""
|
||||
Offline eval for the ExceptionAgent stall decision (core.llm.decide_stall_response).
|
||||
|
||||
Scores whether the chosen action is *defensible* (in the case's ``acceptable``
|
||||
set) and whether it matches the single ``ideal`` action. Stall handling is a
|
||||
judgment task with more than one right answer, so acceptable-rate is the
|
||||
headline metric. See evals/_harness.py for scoring details.
|
||||
|
||||
Usage:
|
||||
export ANTHROPIC_API_KEY=...
|
||||
python -m evals.stall_eval # built-in cases, 1 run each
|
||||
python -m evals.stall_eval --runs 3 # sample each case 3x
|
||||
python -m evals.stall_eval --model claude-haiku-4-5
|
||||
python -m evals.stall_eval --dry-run # print contexts, no API calls
|
||||
python -m evals.stall_eval --min-pass-rate 0.9 # CI gate
|
||||
|
||||
Add real production stall cases to stall_cases.jsonl as they occur — that is
|
||||
what makes the number trustworthy before flipping EXCEPTION_AGENT_AUTONOMOUS.
|
||||
"""
|
||||
from pathlib import Path
|
||||
|
||||
import core.llm as llm
|
||||
from core.llm import build_stall_context, decide_stall_response
|
||||
from evals._harness import case_now, run_cli
|
||||
|
||||
CASES_PATH = Path(__file__).with_name("stall_cases.jsonl")
|
||||
|
||||
|
||||
def to_context(case):
|
||||
return build_stall_context(case["minutes_stalled"], now=case_now(case), facts=case.get("facts"))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
run_cli(CASES_PATH, to_context, decide_stall_response, llm)
|
||||
Reference in New Issue
Block a user