""" Offline eval for the ExceptionAgent stall decision (core.llm.decide_stall_response). Scores whether the chosen action is *defensible* (in the case's ``acceptable`` set) and whether it matches the single ``ideal`` action. Stall handling is a judgment task with more than one right answer, so acceptable-rate is the headline metric. See evals/_harness.py for scoring details. Usage: export ANTHROPIC_API_KEY=... python -m evals.stall_eval # built-in cases, 1 run each python -m evals.stall_eval --runs 3 # sample each case 3x python -m evals.stall_eval --model claude-haiku-4-5 python -m evals.stall_eval --dry-run # print contexts, no API calls python -m evals.stall_eval --min-pass-rate 0.9 # CI gate Add real production stall cases to stall_cases.jsonl as they occur — that is what makes the number trustworthy before flipping EXCEPTION_AGENT_AUTONOMOUS. """ from pathlib import Path import core.llm as llm from core.llm import build_stall_context, decide_stall_response from evals._harness import case_now, run_cli CASES_PATH = Path(__file__).with_name("stall_cases.jsonl") def to_context(case): return build_stall_context(case["minutes_stalled"], now=case_now(case), facts=case.get("facts")) if __name__ == "__main__": run_cli(CASES_PATH, to_context, decide_stall_response, llm)