implemenation on the bot
This commit is contained in:
79
services/ai/eval.js
Normal file
79
services/ai/eval.js
Normal file
@@ -0,0 +1,79 @@
|
||||
import { INTENTS_COLLECTION, query } from './collections.js';
|
||||
import fs from 'fs';
|
||||
import path from 'path';
|
||||
import { fileURLToPath } from 'url';
|
||||
import { classify } from './confidence.js';
|
||||
|
||||
// Read rather than `import ... assert`: the import-assertion syntax changed
|
||||
// between Node 20 (`assert`) and Node 22 (`with`), and this has to run on both.
|
||||
const HERE = path.dirname(fileURLToPath(import.meta.url));
|
||||
const heldOut = JSON.parse(fs.readFileSync(path.join(HERE, 'eval-set.json'), 'utf8'));
|
||||
|
||||
// ==============================|| Routing evaluation ||============================== //
|
||||
//
|
||||
// The question this answers is NOT "does retrieval work" — it will always look
|
||||
// good against the phrasings it was seeded with. It is:
|
||||
//
|
||||
// does it route phrasings NOBODY tuned it against?
|
||||
//
|
||||
// eval-set.json is deliberately held out: none of these appear in
|
||||
// phrasings.json, and none were used to pick the confidence thresholds.
|
||||
//
|
||||
// Ship criteria (RAG_PLAN.md §7):
|
||||
// • accuracy and coverage beat the regex baseline
|
||||
// • ZERO false write routes — a wrong create is the worst failure here,
|
||||
// the same class as answering "0" when the page shows 19.
|
||||
|
||||
const pct = (n, d) => (d ? `${((n / d) * 100).toFixed(1)}%` : '—');
|
||||
|
||||
const main = async () => {
|
||||
const cases = heldOut.cases || [];
|
||||
let correct = 0;
|
||||
let answered = 0;
|
||||
let falseWrite = 0;
|
||||
const wrong = [];
|
||||
const latencies = [];
|
||||
|
||||
for (const c of cases) {
|
||||
const started = Date.now();
|
||||
// eslint-disable-next-line no-await-in-loop
|
||||
const hits = await query(INTENTS_COLLECTION, c.text, 5);
|
||||
latencies.push(Date.now() - started);
|
||||
|
||||
const { confidence, score, margin } = classify(hits);
|
||||
const top = hits[0]?.metadata || {};
|
||||
const routed = confidence === 'low' ? null : top.intentId;
|
||||
|
||||
if (routed) answered += 1;
|
||||
if (routed === c.expect) correct += 1;
|
||||
else if (routed) wrong.push({ text: c.text, expected: c.expect, got: routed, score: score.toFixed(2), margin: margin.toFixed(2) });
|
||||
|
||||
// A non-write phrasing that routes to a write intent is the failure that
|
||||
// matters most — it would open a create form the operator never asked for.
|
||||
if (top.isWrite && !c.isWrite && confidence === 'high') falseWrite += 1;
|
||||
}
|
||||
|
||||
latencies.sort((a, b) => a - b);
|
||||
const p = (q) => latencies[Math.floor(latencies.length * q)] ?? 0;
|
||||
|
||||
console.log('\n=== held-out routing evaluation ===');
|
||||
console.log(`cases ${cases.length}`);
|
||||
console.log(`accuracy ${correct}/${cases.length} ${pct(correct, cases.length)}`);
|
||||
console.log(`coverage ${answered}/${cases.length} ${pct(answered, cases.length)} (answered at all)`);
|
||||
console.log(`false writes ${falseWrite} ${falseWrite === 0 ? '✓' : '✗ MUST BE ZERO'}`);
|
||||
console.log(`latency p50/p95 ${p(0.5)}ms / ${p(0.95)}ms`);
|
||||
|
||||
if (wrong.length) {
|
||||
console.log('\n--- misroutes ---');
|
||||
wrong.forEach((w) => console.log(` "${w.text}"\n expected ${w.expected}, got ${w.got} (score ${w.score}, margin ${w.margin})`));
|
||||
}
|
||||
|
||||
const ok = falseWrite === 0 && correct / Math.max(cases.length, 1) >= 0.8;
|
||||
console.log(`\n${ok ? 'PASS' : 'FAIL'} — threshold: >=80% accuracy and zero false writes\n`);
|
||||
process.exit(ok ? 0 : 1);
|
||||
};
|
||||
|
||||
main().catch((err) => {
|
||||
console.error('[eval] failed:', err.message);
|
||||
process.exit(1);
|
||||
});
|
||||
Reference in New Issue
Block a user