Files
AI_engine/evals/stall_eval.py

35 lines
1.3 KiB
Python

"""
Offline eval for the ExceptionAgent stall decision (core.llm.decide_stall_response).
Scores whether the chosen action is *defensible* (in the case's ``acceptable``
set) and whether it matches the single ``ideal`` action. Stall handling is a
judgment task with more than one right answer, so acceptable-rate is the
headline metric. See evals/_harness.py for scoring details.
Usage:
export ANTHROPIC_API_KEY=...
python -m evals.stall_eval # built-in cases, 1 run each
python -m evals.stall_eval --runs 3 # sample each case 3x
python -m evals.stall_eval --model claude-haiku-4-5
python -m evals.stall_eval --dry-run # print contexts, no API calls
python -m evals.stall_eval --min-pass-rate 0.9 # CI gate
Add real production stall cases to stall_cases.jsonl as they occur — that is
what makes the number trustworthy before flipping EXCEPTION_AGENT_AUTONOMOUS.
"""
from pathlib import Path
import core.llm as llm
from core.llm import build_stall_context, decide_stall_response
from evals._harness import case_now, run_cli
CASES_PATH = Path(__file__).with_name("stall_cases.jsonl")
def to_context(case):
return build_stall_context(case["minutes_stalled"], now=case_now(case), facts=case.get("facts"))
if __name__ == "__main__":
run_cli(CASES_PATH, to_context, decide_stall_response, llm)