route optimizer agent with dailygrubs ai assign
This commit is contained in:
0
evals/__init__.py
Normal file
0
evals/__init__.py
Normal file
131
evals/_harness.py
Normal file
131
evals/_harness.py
Normal file
@@ -0,0 +1,131 @@
|
||||
"""
|
||||
Shared scoring / reporting for agent decision evals.
|
||||
|
||||
A concrete eval (stall, assignment, ...) supplies two callables:
|
||||
- to_context(case) -> str : turn a dataset case into the model prompt context
|
||||
- decide(context) -> obj : the decision function (returns an object with
|
||||
.action / .reasoning / .confidence, or None)
|
||||
|
||||
and calls run_cli(...). Scoring treats a decision as correct if the majority
|
||||
action across --runs samples is in the case's ``acceptable`` set (headline
|
||||
"acceptable-rate"); matching the single ``ideal`` is the secondary "exact-rate".
|
||||
"""
|
||||
import argparse
|
||||
import asyncio
|
||||
import json
|
||||
import sys
|
||||
from collections import Counter
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from statistics import mean
|
||||
|
||||
|
||||
def load_cases(path: Path):
|
||||
cases = []
|
||||
for lineno, raw in enumerate(path.read_text(encoding="utf-8").splitlines(), 1):
|
||||
line = raw.strip()
|
||||
if not line or line.startswith("#"):
|
||||
continue
|
||||
try:
|
||||
cases.append(json.loads(line))
|
||||
except json.JSONDecodeError as e:
|
||||
print(f"skipping malformed case on line {lineno}: {e}", file=sys.stderr)
|
||||
return cases
|
||||
|
||||
|
||||
def case_now(case):
|
||||
"""Fixed date so only hour-of-day varies — keeps contexts reproducible."""
|
||||
return datetime(2024, 1, 1, int(case.get("hour_of_day_utc", 12)), 0, tzinfo=timezone.utc)
|
||||
|
||||
|
||||
async def evaluate(cases, to_context, decide, runs: int):
|
||||
rows = []
|
||||
for case in cases:
|
||||
ctx = to_context(case)
|
||||
decisions = []
|
||||
for _ in range(runs):
|
||||
d = await decide(ctx)
|
||||
if d is not None:
|
||||
decisions.append(d)
|
||||
errors = runs - len(decisions)
|
||||
|
||||
if decisions:
|
||||
counts = Counter(d.action for d in decisions)
|
||||
majority, agree = counts.most_common(1)[0]
|
||||
avg_conf = mean(d.confidence for d in decisions)
|
||||
reasoning = next((d.reasoning for d in decisions if d.action == majority), "")
|
||||
else:
|
||||
majority, agree, avg_conf, reasoning = None, 0, 0.0, ""
|
||||
|
||||
rows.append({
|
||||
"id": case["id"], "ideal": case.get("ideal"), "acceptable_set": case["acceptable"],
|
||||
"got": majority, "agree": agree, "runs": runs, "errors": errors,
|
||||
"avg_conf": avg_conf,
|
||||
"acceptable": majority in case["acceptable"] if majority else False,
|
||||
"exact": majority == case.get("ideal") if majority else False,
|
||||
"reasoning": reasoning,
|
||||
})
|
||||
return rows
|
||||
|
||||
|
||||
def print_report(rows):
|
||||
print(f"\n{'case':<28} {'ideal':<14} {'got':<14} {'agree':<7} {'conf':<6} result")
|
||||
print("-" * 82)
|
||||
for r in rows:
|
||||
if r["got"] is None:
|
||||
result = "ERROR (no valid decision)"
|
||||
elif r["exact"]:
|
||||
result = "EXACT"
|
||||
elif r["acceptable"]:
|
||||
result = "ok (acceptable)"
|
||||
else:
|
||||
result = f"MISS (allowed: {', '.join(r['acceptable_set'])})"
|
||||
print(f"{r['id']:<28} {str(r['ideal']):<14} {str(r['got']):<14} "
|
||||
f"{r['agree']}/{r['runs']:<5} {r['avg_conf']:<6.2f} {result}")
|
||||
if r["got"] is not None and not r["acceptable"]:
|
||||
print(f"{'':<28} └─ why: {r['reasoning'][:110]}")
|
||||
|
||||
scored = [r for r in rows if r["got"] is not None]
|
||||
n_scored = len(scored)
|
||||
acc = sum(r["acceptable"] for r in scored)
|
||||
exact = sum(r["exact"] for r in scored)
|
||||
print("-" * 82)
|
||||
print(f"cases: {len(rows)} scored: {n_scored} model-errors: {sum(r['errors'] for r in rows)}")
|
||||
if n_scored:
|
||||
print(f"acceptable-rate: {acc}/{n_scored} = {acc / n_scored:.0%}")
|
||||
print(f"exact-rate: {exact}/{n_scored} = {exact / n_scored:.0%}")
|
||||
return (acc / n_scored) if n_scored else 0.0
|
||||
|
||||
|
||||
def run_cli(default_cases: Path, to_context, decide, llm_module):
|
||||
"""argparse + dry-run + evaluate + report, shared across eval scripts.
|
||||
``llm_module`` is core.llm so --model can override LLM_MODEL for the run."""
|
||||
ap = argparse.ArgumentParser(description="Eval an agent decision against labelled cases.")
|
||||
ap.add_argument("--cases", type=Path, default=default_cases, help="JSONL dataset")
|
||||
ap.add_argument("--runs", type=int, default=1, help="samples per case (majority vote)")
|
||||
ap.add_argument("--model", help="override LLM_MODEL for this run")
|
||||
ap.add_argument("--dry-run", action="store_true", help="print contexts and labels; no API calls")
|
||||
ap.add_argument("--min-pass-rate", type=float, default=0.0, help="exit non-zero if acceptable-rate below this")
|
||||
args = ap.parse_args()
|
||||
|
||||
if args.model:
|
||||
llm_module.LLM_MODEL = args.model
|
||||
|
||||
cases = load_cases(args.cases)
|
||||
if not cases:
|
||||
print(f"no cases found in {args.cases}", file=sys.stderr)
|
||||
sys.exit(2)
|
||||
|
||||
if args.dry_run:
|
||||
for case in cases:
|
||||
print(f"\n=== {case['id']} (ideal={case.get('ideal')}, allowed={case['acceptable']}) ===")
|
||||
print(to_context(case))
|
||||
print(f"\n[dry-run] {len(cases)} cases, no API calls made.")
|
||||
return
|
||||
|
||||
print(f"model: {llm_module.LLM_MODEL} cases: {len(cases)} runs/case: {args.runs}")
|
||||
rows = asyncio.run(evaluate(cases, to_context, decide, args.runs))
|
||||
pass_rate = print_report(rows)
|
||||
if pass_rate < args.min_pass_rate:
|
||||
print(f"\nFAIL: acceptable-rate {pass_rate:.0%} < required {args.min_pass_rate:.0%}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
10
evals/assignment_cases.jsonl
Normal file
10
evals/assignment_cases.jsonl
Normal file
@@ -0,0 +1,10 @@
|
||||
{"id": "first-failure-close-rider", "hour_of_day_utc": 14, "facts": {"zone_id": "hyderabad", "failures_today": 1, "nearest_miler_within_km": 10, "has_coordinates": true}, "acceptable": ["monitor", "notify_customer"], "ideal": "monitor", "rationale": "Single failure with a rider only 10km away — almost certainly transient; let the backend retry."}
|
||||
{"id": "repeated-zone-gap", "hour_of_day_utc": 12, "facts": {"zone_id": "pune", "failures_today": 4, "nearest_miler_within_km": "none within 30km", "has_coordinates": true}, "acceptable": ["ops_alert", "escalate"], "ideal": "ops_alert", "rationale": "Repeated failures and no rider within 30km — a genuine coverage gap; alert ops."}
|
||||
{"id": "second-failure-far-rider", "hour_of_day_utc": 18, "facts": {"zone_id": "bangalore", "failures_today": 2, "nearest_miler_within_km": 30, "has_coordinates": true}, "acceptable": ["notify_customer", "monitor"], "ideal": "notify_customer", "rationale": "A couple of failures with the nearest rider 30km out during rush — real delay; keep the customer informed."}
|
||||
{"id": "no-coordinates-first", "hour_of_day_utc": 10, "facts": {"zone_id": "unknown", "failures_today": 1, "nearest_miler_within_km": "unknown (no coordinates)", "has_coordinates": false}, "acceptable": ["monitor", "escalate"], "ideal": "monitor", "rationale": "Coverage can't be assessed without coordinates, but a single failure is most likely transient."}
|
||||
{"id": "many-failures-riders-near", "hour_of_day_utc": 15, "facts": {"zone_id": "mumbai_west", "failures_today": 5, "nearest_miler_within_km": 10, "has_coordinates": true}, "acceptable": ["escalate", "ops_alert"], "ideal": "escalate", "rationale": "Riders ARE nearby yet assignment keeps failing — not a coverage gap; a systemic issue for a human."}
|
||||
{"id": "night-coverage-gap", "hour_of_day_utc": 2, "facts": {"zone_id": "kolkata", "failures_today": 3, "nearest_miler_within_km": "none within 30km", "has_coordinates": true}, "acceptable": ["ops_alert", "escalate"], "ideal": "ops_alert", "rationale": "Repeated overnight failures with no nearby rider — a coverage gap worth flagging to ops."}
|
||||
{"id": "single-far-rider", "hour_of_day_utc": 9, "facts": {"zone_id": "hyderabad", "failures_today": 1, "nearest_miler_within_km": 30, "has_coordinates": true}, "acceptable": ["monitor", "notify_customer"], "ideal": "monitor", "rationale": "One failure, distant rider — likely resolves on retry."}
|
||||
{"id": "persistent-far-rush", "hour_of_day_utc": 18, "facts": {"zone_id": "mumbai_east", "failures_today": 3, "nearest_miler_within_km": 30, "has_coordinates": true}, "acceptable": ["notify_customer", "ops_alert"], "ideal": "notify_customer", "rationale": "Rush-hour, repeated, distant rider — the customer should hear about the delay."}
|
||||
{"id": "borderline-count", "hour_of_day_utc": 13, "facts": {"zone_id": "north_delhi", "failures_today": 3, "nearest_miler_within_km": 20, "has_coordinates": true}, "acceptable": ["notify_customer", "ops_alert", "monitor"], "ideal": "notify_customer", "rationale": "Genuinely borderline — a few failures with a moderately-close rider; any of these is defensible."}
|
||||
{"id": "severe-gap", "hour_of_day_utc": 11, "facts": {"zone_id": "pune", "failures_today": 8, "nearest_miler_within_km": "none within 30km", "has_coordinates": true}, "acceptable": ["ops_alert", "escalate"], "ideal": "ops_alert", "rationale": "Many failures and no rider anywhere near — a clear, severe coverage gap."}
|
||||
30
evals/assignment_eval.py
Normal file
30
evals/assignment_eval.py
Normal file
@@ -0,0 +1,30 @@
|
||||
"""
|
||||
Offline eval for the DispatchAgent assignment-failure decision
|
||||
(core.llm.decide_assignment_failure).
|
||||
|
||||
Same scoring as the stall eval (see evals/_harness.py): acceptable-rate is the
|
||||
headline, exact-rate secondary. The cases use the real gatherer keys
|
||||
(zone_id, failures_today, nearest_miler_within_km, has_coordinates) so the eval
|
||||
reflects what DispatchAgent actually sends.
|
||||
|
||||
Usage:
|
||||
export ANTHROPIC_API_KEY=...
|
||||
python -m evals.assignment_eval --runs 3
|
||||
python -m evals.assignment_eval --dry-run
|
||||
python -m evals.assignment_eval --model claude-haiku-4-5 --min-pass-rate 0.9
|
||||
"""
|
||||
from pathlib import Path
|
||||
|
||||
import core.llm as llm
|
||||
from core.llm import build_assignment_failure_context, decide_assignment_failure
|
||||
from evals._harness import case_now, run_cli
|
||||
|
||||
CASES_PATH = Path(__file__).with_name("assignment_cases.jsonl")
|
||||
|
||||
|
||||
def to_context(case):
|
||||
return build_assignment_failure_context(case.get("facts", {}), now=case_now(case))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
run_cli(CASES_PATH, to_context, decide_assignment_failure, llm)
|
||||
15
evals/stall_cases.jsonl
Normal file
15
evals/stall_cases.jsonl
Normal file
@@ -0,0 +1,15 @@
|
||||
{"id": "just-over-traffic", "minutes_stalled": 11, "hour_of_day_utc": 18, "facts": {"location": "major intersection", "note": "stop-go traffic reported in the area"}, "acceptable": ["wait", "notify_only"], "ideal": "wait", "rationale": "Just over threshold during evening rush; most likely ordinary traffic."}
|
||||
{"id": "moderate-at-delivery", "minutes_stalled": 18, "hour_of_day_utc": 13, "facts": {"location": "residential delivery address", "note": "arrived at the delivery point"}, "acceptable": ["notify_only", "wait"], "ideal": "notify_only", "rationale": "Stopped at the delivery point — likely completing handoff; reassure the customer, keep the miler."}
|
||||
{"id": "long-highway-no-stop", "minutes_stalled": 42, "hour_of_day_utc": 15, "facts": {"location": "on a highway stretch", "note": "no position change, not at any known stop"}, "acceptable": ["reassign", "escalate"], "ideal": "reassign", "rationale": "Long stall mid-highway, not at a stop — unlikely to recover on its own."}
|
||||
{"id": "late-night-thin-coverage", "minutes_stalled": 25, "hour_of_day_utc": 2, "facts": {"note": "very few active milers in this zone at this hour"}, "acceptable": ["escalate", "notify_only"], "ideal": "escalate", "rationale": "Reassignment is risky with thin overnight coverage; a human should decide."}
|
||||
{"id": "very-long-no-context", "minutes_stalled": 60, "hour_of_day_utc": 11, "facts": {}, "acceptable": ["reassign", "escalate"], "ideal": "escalate", "rationale": "Very long stall but zero situational detail — low information favors a human."}
|
||||
{"id": "repeat-staller", "minutes_stalled": 15, "hour_of_day_utc": 10, "facts": {"stalls_today": 3, "position_unchanged_minutes": 15}, "acceptable": ["reassign", "escalate"], "ideal": "reassign", "rationale": "Real gatherer signal: 3rd stall today for this miler — a pattern that says this miler is unreliable right now."}
|
||||
{"id": "near-completion", "minutes_stalled": 20, "hour_of_day_utc": 16, "facts": {"note": "route is ~90% complete, miler is at the final delivery cluster"}, "acceptable": ["wait", "notify_only"], "ideal": "notify_only", "rationale": "Almost done — reassigning would waste a near-complete trip."}
|
||||
{"id": "vehicle-breakdown", "minutes_stalled": 35, "hour_of_day_utc": 12, "facts": {"location": "roadside", "note": "miler reported a vehicle breakdown"}, "acceptable": ["reassign"], "ideal": "reassign", "rationale": "Explicit breakdown — the miler cannot recover; must reassign."}
|
||||
{"id": "ambiguous-mid", "minutes_stalled": 22, "hour_of_day_utc": 14, "facts": {"location": "commercial area"}, "acceptable": ["notify_only", "reassign", "escalate"], "ideal": "escalate", "rationale": "Genuinely ambiguous — moderate stall with thin info; hard to justify a strong action."}
|
||||
{"id": "morning-hub-queue", "minutes_stalled": 12, "hour_of_day_utc": 8, "facts": {"location": "at pickup hub", "note": "morning batch pickup, queues are common here"}, "acceptable": ["wait", "notify_only"], "ideal": "wait", "rationale": "Hub queue during the morning batch is expected; not a real stall."}
|
||||
{"id": "customer-unavailable", "minutes_stalled": 28, "hour_of_day_utc": 19, "facts": {"note": "at the delivery address, customer not answering calls"}, "acceptable": ["notify_only", "escalate"], "ideal": "notify_only", "rationale": "Customer-side delay, not the miler's fault; reassignment would not help."}
|
||||
{"id": "long-idle-mid-route", "minutes_stalled": 50, "hour_of_day_utc": 10, "facts": {"location": "mid-route, not at any stop", "note": "GPS static for 50 minutes"}, "acceptable": ["reassign", "escalate"], "ideal": "reassign", "rationale": "Long static period mid-route with no plausible benign cause."}
|
||||
{"id": "gathered-static-assigned", "minutes_stalled": 40, "hour_of_day_utc": 11, "facts": {"booking_status": "Miler_Assigned", "position_unchanged_minutes": 40, "last_gps_ping_minutes_ago": 1}, "acceptable": ["reassign", "escalate"], "ideal": "reassign", "rationale": "Real gatherer signals: device still pinging but frozen 40 min while only assigned (not yet at pickup) — genuine stall."}
|
||||
{"id": "gathered-pickup-phase", "minutes_stalled": 14, "hour_of_day_utc": 9, "facts": {"booking_status": "Pickup_Scheduled", "position_unchanged_minutes": 14, "minutes_since_booking_created": 20}, "acceptable": ["wait", "notify_only"], "ideal": "wait", "rationale": "Real gatherer signals: brief stationary period in the pickup phase — most likely waiting at the pickup point."}
|
||||
{"id": "gathered-stale-gps", "minutes_stalled": 25, "hour_of_day_utc": 13, "facts": {"booking_status": "Miler_Assigned", "last_gps_ping_minutes_ago": 25, "position_unchanged_minutes": 25}, "acceptable": ["escalate", "reassign"], "ideal": "escalate", "rationale": "Real gatherer signals: no GPS ping for 25 min — device may be offline (dead phone), which is ambiguous; prefer a human."}
|
||||
34
evals/stall_eval.py
Normal file
34
evals/stall_eval.py
Normal file
@@ -0,0 +1,34 @@
|
||||
"""
|
||||
Offline eval for the ExceptionAgent stall decision (core.llm.decide_stall_response).
|
||||
|
||||
Scores whether the chosen action is *defensible* (in the case's ``acceptable``
|
||||
set) and whether it matches the single ``ideal`` action. Stall handling is a
|
||||
judgment task with more than one right answer, so acceptable-rate is the
|
||||
headline metric. See evals/_harness.py for scoring details.
|
||||
|
||||
Usage:
|
||||
export ANTHROPIC_API_KEY=...
|
||||
python -m evals.stall_eval # built-in cases, 1 run each
|
||||
python -m evals.stall_eval --runs 3 # sample each case 3x
|
||||
python -m evals.stall_eval --model claude-haiku-4-5
|
||||
python -m evals.stall_eval --dry-run # print contexts, no API calls
|
||||
python -m evals.stall_eval --min-pass-rate 0.9 # CI gate
|
||||
|
||||
Add real production stall cases to stall_cases.jsonl as they occur — that is
|
||||
what makes the number trustworthy before flipping EXCEPTION_AGENT_AUTONOMOUS.
|
||||
"""
|
||||
from pathlib import Path
|
||||
|
||||
import core.llm as llm
|
||||
from core.llm import build_stall_context, decide_stall_response
|
||||
from evals._harness import case_now, run_cli
|
||||
|
||||
CASES_PATH = Path(__file__).with_name("stall_cases.jsonl")
|
||||
|
||||
|
||||
def to_context(case):
|
||||
return build_stall_context(case["minutes_stalled"], now=case_now(case), facts=case.get("facts"))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
run_cli(CASES_PATH, to_context, decide_stall_response, llm)
|
||||
Reference in New Issue
Block a user