Files
Doormilexpress_console/services/ai/confidence.js
2026-08-19 17:08:45 +05:30

37 lines
1.7 KiB
JavaScript

// ==============================|| Confidence ||============================== //
//
// Its own module, not part of index.js, because index.js calls app.listen() at
// module scope — importing it from eval.js would boot a second HTTP server as
// a side effect of running the evaluation.
//
// ---- Why margin and not similarity -----------------------------------------
//
// Cosine similarity is NOT a probability of correctness. A score of 0.87 does
// not mean "87% likely right", and showing it as though it does would be the
// same false precision this assistant avoids everywhere else.
//
// What carries information is the MARGIN between the best and second-best
// match. A wide margin means the question is unambiguous. A narrow one means it
// genuinely could be two things — and that is exactly when the bot should ask
// instead of guessing.
//
// The answer's correctness never comes from this number. It comes from the
// intent's deterministic run() hitting a real endpoint.
// Starting values. Tune from eval.js output, not intuition.
export const HIGH_SCORE = 0.75;
export const HIGH_MARGIN = 0.1;
export const MED_SCORE = 0.6;
export const MED_MARGIN = 0.05;
export const classify = (hits) => {
if (!hits.length) return { confidence: 'low', score: 0, margin: 0 };
const score = hits[0].score;
// With a single hit there is nothing to be ambiguous against, so the margin
// is the score itself rather than a fabricated 0.
const margin = hits.length > 1 ? score - hits[1].score : score;
if (score >= HIGH_SCORE && margin >= HIGH_MARGIN) return { confidence: 'high', score, margin };
if (score >= MED_SCORE && margin >= MED_MARGIN) return { confidence: 'medium', score, margin };
return { confidence: 'low', score, margin };
};