import { INTENTS_COLLECTION, query } from './collections.js'; import fs from 'fs'; import path from 'path'; import { fileURLToPath } from 'url'; import { classify } from './confidence.js'; // Read rather than `import ... assert`: the import-assertion syntax changed // between Node 20 (`assert`) and Node 22 (`with`), and this has to run on both. const HERE = path.dirname(fileURLToPath(import.meta.url)); const heldOut = JSON.parse(fs.readFileSync(path.join(HERE, 'eval-set.json'), 'utf8')); // ==============================|| Routing evaluation ||============================== // // // The question this answers is NOT "does retrieval work" — it will always look // good against the phrasings it was seeded with. It is: // // does it route phrasings NOBODY tuned it against? // // eval-set.json is deliberately held out: none of these appear in // phrasings.json, and none were used to pick the confidence thresholds. // // Ship criteria (RAG_PLAN.md §7): // • accuracy and coverage beat the regex baseline // • ZERO false write routes — a wrong create is the worst failure here, // the same class as answering "0" when the page shows 19. const pct = (n, d) => (d ? `${((n / d) * 100).toFixed(1)}%` : '—'); const main = async () => { const cases = heldOut.cases || []; let correct = 0; let answered = 0; let falseWrite = 0; const wrong = []; const latencies = []; for (const c of cases) { const started = Date.now(); // eslint-disable-next-line no-await-in-loop const hits = await query(INTENTS_COLLECTION, c.text, 5); latencies.push(Date.now() - started); const { confidence, score, margin } = classify(hits); const top = hits[0]?.metadata || {}; const routed = confidence === 'low' ? null : top.intentId; if (routed) answered += 1; if (routed === c.expect) correct += 1; else if (routed) wrong.push({ text: c.text, expected: c.expect, got: routed, score: score.toFixed(2), margin: margin.toFixed(2) }); // A non-write phrasing that routes to a write intent is the failure that // matters most — it would open a create form the operator never asked for. if (top.isWrite && !c.isWrite && confidence === 'high') falseWrite += 1; } latencies.sort((a, b) => a - b); const p = (q) => latencies[Math.floor(latencies.length * q)] ?? 0; console.log('\n=== held-out routing evaluation ==='); console.log(`cases ${cases.length}`); console.log(`accuracy ${correct}/${cases.length} ${pct(correct, cases.length)}`); console.log(`coverage ${answered}/${cases.length} ${pct(answered, cases.length)} (answered at all)`); console.log(`false writes ${falseWrite} ${falseWrite === 0 ? '✓' : '✗ MUST BE ZERO'}`); console.log(`latency p50/p95 ${p(0.5)}ms / ${p(0.95)}ms`); if (wrong.length) { console.log('\n--- misroutes ---'); wrong.forEach((w) => console.log(` "${w.text}"\n expected ${w.expected}, got ${w.got} (score ${w.score}, margin ${w.margin})`)); } const ok = falseWrite === 0 && correct / Math.max(cases.length, 1) >= 0.8; console.log(`\n${ok ? 'PASS' : 'FAIL'} — threshold: >=80% accuracy and zero false writes\n`); process.exit(ok ? 0 : 1); }; main().catch((err) => { console.error('[eval] failed:', err.message); process.exit(1); });