37 lines
1.7 KiB
JavaScript
37 lines
1.7 KiB
JavaScript
// ==============================|| Confidence ||============================== //
|
|
//
|
|
// Its own module, not part of index.js, because index.js calls app.listen() at
|
|
// module scope — importing it from eval.js would boot a second HTTP server as
|
|
// a side effect of running the evaluation.
|
|
//
|
|
// ---- Why margin and not similarity -----------------------------------------
|
|
//
|
|
// Cosine similarity is NOT a probability of correctness. A score of 0.87 does
|
|
// not mean "87% likely right", and showing it as though it does would be the
|
|
// same false precision this assistant avoids everywhere else.
|
|
//
|
|
// What carries information is the MARGIN between the best and second-best
|
|
// match. A wide margin means the question is unambiguous. A narrow one means it
|
|
// genuinely could be two things — and that is exactly when the bot should ask
|
|
// instead of guessing.
|
|
//
|
|
// The answer's correctness never comes from this number. It comes from the
|
|
// intent's deterministic run() hitting a real endpoint.
|
|
|
|
// Starting values. Tune from eval.js output, not intuition.
|
|
export const HIGH_SCORE = 0.75;
|
|
export const HIGH_MARGIN = 0.1;
|
|
export const MED_SCORE = 0.6;
|
|
export const MED_MARGIN = 0.05;
|
|
|
|
export const classify = (hits) => {
|
|
if (!hits.length) return { confidence: 'low', score: 0, margin: 0 };
|
|
const score = hits[0].score;
|
|
// With a single hit there is nothing to be ambiguous against, so the margin
|
|
// is the score itself rather than a fabricated 0.
|
|
const margin = hits.length > 1 ? score - hits[1].score : score;
|
|
if (score >= HIGH_SCORE && margin >= HIGH_MARGIN) return { confidence: 'high', score, margin };
|
|
if (score >= MED_SCORE && margin >= MED_MARGIN) return { confidence: 'medium', score, margin };
|
|
return { confidence: 'low', score, margin };
|
|
};
|