80 lines
3.2 KiB
JavaScript
80 lines
3.2 KiB
JavaScript
import { INTENTS_COLLECTION, query } from './collections.js';
|
|
import fs from 'fs';
|
|
import path from 'path';
|
|
import { fileURLToPath } from 'url';
|
|
import { classify } from './confidence.js';
|
|
|
|
// Read rather than `import ... assert`: the import-assertion syntax changed
|
|
// between Node 20 (`assert`) and Node 22 (`with`), and this has to run on both.
|
|
const HERE = path.dirname(fileURLToPath(import.meta.url));
|
|
const heldOut = JSON.parse(fs.readFileSync(path.join(HERE, 'eval-set.json'), 'utf8'));
|
|
|
|
// ==============================|| Routing evaluation ||============================== //
|
|
//
|
|
// The question this answers is NOT "does retrieval work" — it will always look
|
|
// good against the phrasings it was seeded with. It is:
|
|
//
|
|
// does it route phrasings NOBODY tuned it against?
|
|
//
|
|
// eval-set.json is deliberately held out: none of these appear in
|
|
// phrasings.json, and none were used to pick the confidence thresholds.
|
|
//
|
|
// Ship criteria (RAG_PLAN.md §7):
|
|
// • accuracy and coverage beat the regex baseline
|
|
// • ZERO false write routes — a wrong create is the worst failure here,
|
|
// the same class as answering "0" when the page shows 19.
|
|
|
|
const pct = (n, d) => (d ? `${((n / d) * 100).toFixed(1)}%` : '—');
|
|
|
|
const main = async () => {
|
|
const cases = heldOut.cases || [];
|
|
let correct = 0;
|
|
let answered = 0;
|
|
let falseWrite = 0;
|
|
const wrong = [];
|
|
const latencies = [];
|
|
|
|
for (const c of cases) {
|
|
const started = Date.now();
|
|
// eslint-disable-next-line no-await-in-loop
|
|
const hits = await query(INTENTS_COLLECTION, c.text, 5);
|
|
latencies.push(Date.now() - started);
|
|
|
|
const { confidence, score, margin } = classify(hits);
|
|
const top = hits[0]?.metadata || {};
|
|
const routed = confidence === 'low' ? null : top.intentId;
|
|
|
|
if (routed) answered += 1;
|
|
if (routed === c.expect) correct += 1;
|
|
else if (routed) wrong.push({ text: c.text, expected: c.expect, got: routed, score: score.toFixed(2), margin: margin.toFixed(2) });
|
|
|
|
// A non-write phrasing that routes to a write intent is the failure that
|
|
// matters most — it would open a create form the operator never asked for.
|
|
if (top.isWrite && !c.isWrite && confidence === 'high') falseWrite += 1;
|
|
}
|
|
|
|
latencies.sort((a, b) => a - b);
|
|
const p = (q) => latencies[Math.floor(latencies.length * q)] ?? 0;
|
|
|
|
console.log('\n=== held-out routing evaluation ===');
|
|
console.log(`cases ${cases.length}`);
|
|
console.log(`accuracy ${correct}/${cases.length} ${pct(correct, cases.length)}`);
|
|
console.log(`coverage ${answered}/${cases.length} ${pct(answered, cases.length)} (answered at all)`);
|
|
console.log(`false writes ${falseWrite} ${falseWrite === 0 ? '✓' : '✗ MUST BE ZERO'}`);
|
|
console.log(`latency p50/p95 ${p(0.5)}ms / ${p(0.95)}ms`);
|
|
|
|
if (wrong.length) {
|
|
console.log('\n--- misroutes ---');
|
|
wrong.forEach((w) => console.log(` "${w.text}"\n expected ${w.expected}, got ${w.got} (score ${w.score}, margin ${w.margin})`));
|
|
}
|
|
|
|
const ok = falseWrite === 0 && correct / Math.max(cases.length, 1) >= 0.8;
|
|
console.log(`\n${ok ? 'PASS' : 'FAIL'} — threshold: >=80% accuracy and zero false writes\n`);
|
|
process.exit(ok ? 0 : 1);
|
|
};
|
|
|
|
main().catch((err) => {
|
|
console.error('[eval] failed:', err.message);
|
|
process.exit(1);
|
|
});
|