172 lines
6.6 KiB
JavaScript
172 lines
6.6 KiB
JavaScript
/**
|
|
* Ranking the panel's own suggestions against what is being typed.
|
|
*
|
|
* This is not a catalogue and deliberately does not own one. Owliver already
|
|
* decides what can be asked on a page — the agent's starters, every attached
|
|
* skill's declared suggestions, the page's own derived prompts — and each of
|
|
* those chips carries the metadata that makes it *executable*: a `capability`
|
|
* that names one of the page's answers, a `skillId`/`skillCapability` pair that
|
|
* names a skill's reading, a `positionId` that says which record, a `route` for
|
|
* the ones that navigate. All this module does is choose which of those
|
|
* already-built chips are worth showing for a given query, and hand them back
|
|
* unchanged so that clicking one runs exactly what it always ran.
|
|
*
|
|
* Returning the original object rather than a copy is the whole contract. A
|
|
* ranked chip is the same chip; `runPrompt` cannot tell it apart from one that
|
|
* arrived unfiltered, so nothing about how a suggestion executes depends on
|
|
* whether it was typed towards or offered outright.
|
|
*/
|
|
|
|
/**
|
|
* The shortest query worth ranking. Below it the panel shows nothing at all —
|
|
* one character matches most of the catalogue, which is the wall of chips this
|
|
* replaced.
|
|
*/
|
|
export const MIN_QUERY_CHARS = 2;
|
|
|
|
/** Never more than a row. The composer is what the reader came for. */
|
|
export const MAX_MATCHES = 3;
|
|
|
|
/** One shared empty array, so a keystroke that matches nothing is referentially
|
|
stable and does not rerender the chip row. */
|
|
const NONE = [];
|
|
|
|
const lower = (value) => String(value ?? '').toLowerCase();
|
|
|
|
/**
|
|
* Letters and digits only, Unicode aware — the same thing the reader would
|
|
* count. Punctuation alone never counts as having typed anything.
|
|
*/
|
|
const meaningful = (value) => (String(value ?? '').match(/[\p{L}\p{N}]/gu) || []).length;
|
|
|
|
/** A string as the words worth matching on. */
|
|
const words = (value) => lower(value).split(/[^\p{L}\p{N}]+/u).filter(Boolean);
|
|
|
|
/**
|
|
* Words too common to carry a subject.
|
|
*
|
|
* Only used to stop a query made *entirely* of them from matching the whole
|
|
* catalogue: "show me the" should offer nothing rather than everything. A
|
|
* stop word alongside a real term is still scored, because "show pipeline"
|
|
* should rank a chip saying both above one saying only the second.
|
|
*/
|
|
const STOP_WORDS = new Set([
|
|
'the', 'a', 'an', 'is', 'are', 'was', 'were', 'be', 'do', 'does', 'did',
|
|
'show', 'me', 'my', 'our', 'as', 'of', 'for', 'in', 'on', 'to', 'and', 'or',
|
|
'with', 'what', 'how', 'why', 'can', 'you', 'i', 'it', 'please', 'give',
|
|
'tell', 'about', 'this', 'that', 'any', 'all',
|
|
]);
|
|
|
|
/**
|
|
* How well one term matches one field.
|
|
*
|
|
* Prefix matching is what makes this feel like typing rather than searching:
|
|
* "platf" has to find "Platform health" four characters before the word is
|
|
* finished. Infix matching is allowed only from four characters, where a
|
|
* fragment is specific enough that finding it mid-word is a hit rather than an
|
|
* accident — "line" must not match "pipeline" while "peli" reasonably does.
|
|
*/
|
|
function fieldScore(term, fieldWords, exact, partial) {
|
|
let best = 0;
|
|
for (const word of fieldWords) {
|
|
if (word === term) return exact;
|
|
if (word.startsWith(term)) best = Math.max(best, partial);
|
|
else if (term.length >= 4 && word.includes(term)) best = Math.max(best, partial - 1);
|
|
}
|
|
return best;
|
|
}
|
|
|
|
/**
|
|
* What a chip resolves to, as words.
|
|
*
|
|
* A capability id is not decoration — `pipeline-health` is the name of the
|
|
* reading the chip runs, and it is frequently the only place the subject
|
|
* appears. The Control Center's bottleneck chip reads "Bottleneck at
|
|
* interviewed" and sends a sentence about candidates dropping between stages;
|
|
* nothing in either says "pipeline", and typing that word is exactly how a
|
|
* reader would look for it. So the chip's own routing metadata is matched too,
|
|
* at the lowest weight of the three fields — it is what the chip *is*, not what
|
|
* it says, and a chip that says the word should always rank above one that
|
|
* merely resolves to it.
|
|
*/
|
|
const intentWords = (chip) => words(
|
|
[chip?.capability, chip?.skillId, chip?.skillCapability].filter(Boolean).join(' ')
|
|
);
|
|
|
|
/**
|
|
* A chip's score for a query, or 0 for "do not offer this".
|
|
*
|
|
* The label is weighted above the prompt because the label is what the reader
|
|
* sees: a chip that reads "Platform health" is a better answer to "platform"
|
|
* than one that happens to mention the word in the sentence it sends, even
|
|
* though both would answer.
|
|
*/
|
|
function score(chip, terms, phrase) {
|
|
const label = lower(chip?.label);
|
|
const prompt = lower(chip?.prompt);
|
|
if (!label && !prompt) return 0;
|
|
|
|
let total = 0;
|
|
let matched = 0;
|
|
|
|
/* The whole query as one phrase, which is the strongest signal there is —
|
|
"pipeline health" typed in full should beat two chips that each carry one
|
|
of those words. */
|
|
if (label.includes(phrase)) total += 12;
|
|
else if (prompt.includes(phrase)) total += 7;
|
|
|
|
const labelWords = words(label);
|
|
const promptWords = words(prompt);
|
|
const routeWords = intentWords(chip);
|
|
|
|
for (const term of terms) {
|
|
const hit = Math.max(
|
|
fieldScore(term, labelWords, 6, 4),
|
|
fieldScore(term, promptWords, 4, 2),
|
|
fieldScore(term, routeWords, 3, 2)
|
|
);
|
|
if (hit) {
|
|
matched += 1;
|
|
total += hit;
|
|
}
|
|
}
|
|
|
|
/* A chip has to actually be about something that was typed. */
|
|
if (!matched) return 0;
|
|
|
|
/* Every term landing somewhere is worth more than most of them landing. */
|
|
if (matched === terms.length) total += 3;
|
|
|
|
/* A suggestion that would have to ask which record before it could answer
|
|
ranks below one that answers — the same order the resolver already puts
|
|
them in when they are offered unfiltered. */
|
|
if (chip?.deferred) total -= 3;
|
|
|
|
return total;
|
|
}
|
|
|
|
/**
|
|
* The best few of `prompts` for `query`, in the chips' own objects.
|
|
*
|
|
* Stable: chips scoring equally keep the order they arrived in, which is the
|
|
* order the panel already considers most useful — agent starters, then declared
|
|
* skill suggestions, then the page's derived prompts.
|
|
*/
|
|
export function rankPrompts(prompts = [], query = '', max = MAX_MATCHES) {
|
|
const text = String(query ?? '').trim();
|
|
if (meaningful(text) < MIN_QUERY_CHARS) return NONE;
|
|
|
|
const terms = words(text);
|
|
if (!terms.length || terms.every((term) => STOP_WORDS.has(term))) return NONE;
|
|
|
|
const phrase = lower(text);
|
|
const scored = [];
|
|
prompts.forEach((chip, index) => {
|
|
const value = score(chip, terms, phrase);
|
|
if (value > 0) scored.push({ chip, value, index });
|
|
});
|
|
|
|
scored.sort((a, b) => b.value - a.value || a.index - b.index);
|
|
return scored.length ? scored.slice(0, max).map((entry) => entry.chip) : NONE;
|
|
}
|