31 lines
1.0 KiB
Python
31 lines
1.0 KiB
Python
"""
|
|
Offline eval for the DispatchAgent assignment-failure decision
|
|
(core.llm.decide_assignment_failure).
|
|
|
|
Same scoring as the stall eval (see evals/_harness.py): acceptable-rate is the
|
|
headline, exact-rate secondary. The cases use the real gatherer keys
|
|
(zone_id, failures_today, nearest_miler_within_km, has_coordinates) so the eval
|
|
reflects what DispatchAgent actually sends.
|
|
|
|
Usage:
|
|
export ANTHROPIC_API_KEY=...
|
|
python -m evals.assignment_eval --runs 3
|
|
python -m evals.assignment_eval --dry-run
|
|
python -m evals.assignment_eval --model claude-haiku-4-5 --min-pass-rate 0.9
|
|
"""
|
|
from pathlib import Path
|
|
|
|
import core.llm as llm
|
|
from core.llm import build_assignment_failure_context, decide_assignment_failure
|
|
from evals._harness import case_now, run_cli
|
|
|
|
CASES_PATH = Path(__file__).with_name("assignment_cases.jsonl")
|
|
|
|
|
|
def to_context(case):
|
|
return build_assignment_failure_context(case.get("facts", {}), now=case_now(case))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
run_cli(CASES_PATH, to_context, decide_assignment_failure, llm)
|