35 lines
1.3 KiB
Python
35 lines
1.3 KiB
Python
"""
|
|
Offline eval for the ExceptionAgent stall decision (core.llm.decide_stall_response).
|
|
|
|
Scores whether the chosen action is *defensible* (in the case's ``acceptable``
|
|
set) and whether it matches the single ``ideal`` action. Stall handling is a
|
|
judgment task with more than one right answer, so acceptable-rate is the
|
|
headline metric. See evals/_harness.py for scoring details.
|
|
|
|
Usage:
|
|
export ANTHROPIC_API_KEY=...
|
|
python -m evals.stall_eval # built-in cases, 1 run each
|
|
python -m evals.stall_eval --runs 3 # sample each case 3x
|
|
python -m evals.stall_eval --model claude-haiku-4-5
|
|
python -m evals.stall_eval --dry-run # print contexts, no API calls
|
|
python -m evals.stall_eval --min-pass-rate 0.9 # CI gate
|
|
|
|
Add real production stall cases to stall_cases.jsonl as they occur — that is
|
|
what makes the number trustworthy before flipping EXCEPTION_AGENT_AUTONOMOUS.
|
|
"""
|
|
from pathlib import Path
|
|
|
|
import core.llm as llm
|
|
from core.llm import build_stall_context, decide_stall_response
|
|
from evals._harness import case_now, run_cli
|
|
|
|
CASES_PATH = Path(__file__).with_name("stall_cases.jsonl")
|
|
|
|
|
|
def to_context(case):
|
|
return build_stall_context(case["minutes_stalled"], now=case_now(case), facts=case.get("facts"))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
run_cli(CASES_PATH, to_context, decide_stall_response, llm)
|