// ==============================|| Confidence ||============================== // // // Its own module, not part of index.js, because index.js calls app.listen() at // module scope — importing it from eval.js would boot a second HTTP server as // a side effect of running the evaluation. // // ---- Why margin and not similarity ----------------------------------------- // // Cosine similarity is NOT a probability of correctness. A score of 0.87 does // not mean "87% likely right", and showing it as though it does would be the // same false precision this assistant avoids everywhere else. // // What carries information is the MARGIN between the best and second-best // match. A wide margin means the question is unambiguous. A narrow one means it // genuinely could be two things — and that is exactly when the bot should ask // instead of guessing. // // The answer's correctness never comes from this number. It comes from the // intent's deterministic run() hitting a real endpoint. // Starting values. Tune from eval.js output, not intuition. export const HIGH_SCORE = 0.75; export const HIGH_MARGIN = 0.1; export const MED_SCORE = 0.6; export const MED_MARGIN = 0.05; export const classify = (hits) => { if (!hits.length) return { confidence: 'low', score: 0, margin: 0 }; const score = hits[0].score; // With a single hit there is nothing to be ambiguous against, so the margin // is the score itself rather than a fabricated 0. const margin = hits.length > 1 ? score - hits[1].score : score; if (score >= HIGH_SCORE && margin >= HIGH_MARGIN) return { confidence: 'high', score, margin }; if (score >= MED_SCORE && margin >= MED_MARGIN) return { confidence: 'medium', score, margin }; return { confidence: 'low', score, margin }; };