""" Offline eval for the DispatchAgent assignment-failure decision (core.llm.decide_assignment_failure). Same scoring as the stall eval (see evals/_harness.py): acceptable-rate is the headline, exact-rate secondary. The cases use the real gatherer keys (zone_id, failures_today, nearest_available_miler_within_km, milers_in_geo_index_within_30km, has_coordinates) so the eval reflects what DispatchAgent actually sends. Usage: export ANTHROPIC_API_KEY=... python -m evals.assignment_eval --runs 3 python -m evals.assignment_eval --dry-run python -m evals.assignment_eval --model claude-haiku-4-5 --min-pass-rate 0.9 """ from pathlib import Path import core.llm as llm from core.llm import build_assignment_failure_context, decide_assignment_failure from evals._harness import case_now, run_cli CASES_PATH = Path(__file__).with_name("assignment_cases.jsonl") def to_context(case): return build_assignment_failure_context(case.get("facts", {}), now=case_now(case)) if __name__ == "__main__": run_cli(CASES_PATH, to_context, decide_assignment_failure, llm)