Files
AI_engine/evals/assignment_eval.py

31 lines
1.0 KiB
Python

"""
Offline eval for the DispatchAgent assignment-failure decision
(core.llm.decide_assignment_failure).
Same scoring as the stall eval (see evals/_harness.py): acceptable-rate is the
headline, exact-rate secondary. The cases use the real gatherer keys
(zone_id, failures_today, nearest_miler_within_km, has_coordinates) so the eval
reflects what DispatchAgent actually sends.
Usage:
export ANTHROPIC_API_KEY=...
python -m evals.assignment_eval --runs 3
python -m evals.assignment_eval --dry-run
python -m evals.assignment_eval --model claude-haiku-4-5 --min-pass-rate 0.9
"""
from pathlib import Path
import core.llm as llm
from core.llm import build_assignment_failure_context, decide_assignment_failure
from evals._harness import case_now, run_cli
CASES_PATH = Path(__file__).with_name("assignment_cases.jsonl")
def to_context(case):
return build_assignment_failure_context(case.get("facts", {}), now=case_now(case))
if __name__ == "__main__":
run_cli(CASES_PATH, to_context, decide_assignment_failure, llm)