360 lines
13 KiB
Go
360 lines
13 KiB
Go
package owliver
|
|
|
|
import (
|
|
"sort"
|
|
"strings"
|
|
"unicode"
|
|
|
|
"github.com/krow/krow-backend/go-api/internal/domain"
|
|
)
|
|
|
|
// MaxSuggestions is the most a response may carry.
|
|
//
|
|
// Three, because the panel shows them under a composer the user is still typing
|
|
// into. A fourth line pushes the input off a phone screen, and a ranked list
|
|
// nobody reads to the bottom is a longer list, not a better one.
|
|
const MaxSuggestions = 3
|
|
|
|
// MinQueryChars is the shortest query that is worth ranking.
|
|
//
|
|
// Counted in letters and digits after normalization, so " a " and "?!" are
|
|
// both too short. One character matches a prefix of almost every term in the
|
|
// catalogue, which would make the first keystroke return three arbitrary
|
|
// readings and the second replace all three — noise that reads as a bug.
|
|
const MinQueryChars = 2
|
|
|
|
// MaxQueryChars bounds the work one request can ask for. The panel sends what
|
|
// is in the composer, and nothing about a suggestion improves past a couple of
|
|
// sentences. Beyond this the query is truncated, never rejected: a long paste
|
|
// should rank on its opening words, not fail.
|
|
const MaxQueryChars = 200
|
|
|
|
// Suggestion is one offered question.
|
|
//
|
|
// Text is what the user reads. Intent is the frontend capability id the panel
|
|
// dispatches on. Capability is the section type the answer should be drawn as,
|
|
// present only when the query asked for one — nothing internal is exposed here:
|
|
// no terms, no resource names, no policy detail, no scores.
|
|
type Suggestion struct {
|
|
Text string `json:"text"`
|
|
Intent string `json:"intent"`
|
|
Capability string `json:"capability,omitempty"`
|
|
}
|
|
|
|
/* ── Shapes ─────────────────────────────────────────────────────────────── */
|
|
|
|
// shape is one section type an answer can be drawn as.
|
|
//
|
|
// Transcribed from OWLIVER_CAPABILITIES in src/lib/skills/surfaces.js: the ids
|
|
// and the terms are that table's, and `phrase` is how the id reads inside a
|
|
// sentence. `summary` is first and has no component — it is prose, so it
|
|
// applies to any intent with a subject.
|
|
//
|
|
// Terms here describe the SHAPE and never a subject, which is what keeps a
|
|
// shape from dragging in another page's readings: "as a flow" belongs here,
|
|
// "hiring activity" belongs to an intent.
|
|
type shape struct {
|
|
id string
|
|
phrase string
|
|
terms []string
|
|
}
|
|
|
|
var shapes = []shape{
|
|
{id: "summary", phrase: "", terms: []string{
|
|
"summary", "summarise", "summarize", "summarised", "summarized",
|
|
"summarising", "summarizing", "sum up", "recap", "overview", "brief me",
|
|
"in short", "tell me about"}},
|
|
{id: "flow", phrase: "as a flow", terms: []string{
|
|
"flow", "as a flow", "chart", "graph", "diagram", "funnel", "visual",
|
|
"visualise", "visualize", "step by step"}},
|
|
{id: "stats", phrase: "as stats", terms: []string{
|
|
"stats", "statistics", "figures", "numbers", "counts"}},
|
|
{id: "list", phrase: "as a list", terms: []string{
|
|
"list", "which ones", "show me the records"}},
|
|
{id: "table", phrase: "as a table", terms: []string{
|
|
"table", "as a table", "rows", "grid", "spreadsheet"}},
|
|
{id: "timeline", phrase: "as a timeline", terms: []string{
|
|
"timeline", "chronology", "over time", "what happened"}},
|
|
{id: "progress", phrase: "as progress bars", terms: []string{
|
|
"progress", "bars", "completion", "how far"}},
|
|
{id: "weights", phrase: "as weights", terms: []string{
|
|
"weighting", "weightings", "set the weights", "adjust the weights",
|
|
"screening weight", "vetting weight"}},
|
|
{id: "insight", phrase: "as an insight", terms: []string{
|
|
"insight", "finding", "takeaway", "headline"}},
|
|
{id: "card", phrase: "as a card", terms: []string{
|
|
"card", "panel", "at a glance"}},
|
|
}
|
|
|
|
/* ── Normalization ──────────────────────────────────────────────────────── */
|
|
|
|
// normalize reduces a raw query to the one form everything downstream matches
|
|
// against: lower case, letters and digits only, single-spaced.
|
|
//
|
|
// Every other character — punctuation, quotes, brackets, control characters,
|
|
// emoji, an SQL fragment, a script tag — becomes a space rather than being
|
|
// stripped, so nothing can be glued into a token that was not typed as one.
|
|
// The result is compared against a fixed table of literals and never reaches a
|
|
// query, a template or a log message, so there is no construction to inject
|
|
// into; this is about matching sanely, not about escaping.
|
|
func normalize(raw string) (phrase string, tokens []string, meaningful int) {
|
|
runes := []rune(raw)
|
|
if len(runes) > MaxQueryChars {
|
|
runes = runes[:MaxQueryChars]
|
|
}
|
|
|
|
var b strings.Builder
|
|
b.Grow(len(runes))
|
|
for _, r := range runes {
|
|
switch {
|
|
case unicode.IsLetter(r) || unicode.IsDigit(r):
|
|
b.WriteRune(unicode.ToLower(r))
|
|
meaningful++
|
|
default:
|
|
b.WriteByte(' ')
|
|
}
|
|
}
|
|
|
|
tokens = strings.Fields(b.String())
|
|
return strings.Join(tokens, " "), tokens, meaningful
|
|
}
|
|
|
|
/* ── Scoring ────────────────────────────────────────────────────────────── */
|
|
|
|
// Scores are small integers with a deliberate order:
|
|
//
|
|
// exact a token is the term — the user typed it
|
|
// phrase a multi-word term appears in the query — the most specific hit
|
|
// prefix the term begins with a token — mid-typing: "pipel"
|
|
// extension a token begins with the term — "pipelines"
|
|
// shaped the query named a section type — weakest on its own
|
|
//
|
|
// A shape hit is worth less than any subject hit, so typing "overview" can
|
|
// surface a page's readings but can never outrank a reading the user named.
|
|
const (
|
|
scoreExact = 10
|
|
scorePhrase = 12
|
|
scorePrefix = 6
|
|
scoreExtension = 5
|
|
scoreShaped = 4
|
|
|
|
// minPrefixToken keeps one- and two-letter tokens from matching a term by
|
|
// prefix. "a" begins nothing usefully; "at" would match "attention",
|
|
// "audit" and "activity" at once.
|
|
minPrefixToken = 3
|
|
// minExtensionTerm keeps a short term from being found inside a longer
|
|
// word: without it "list" matches "listen" and "score" matches "scoreboard".
|
|
minExtensionTerm = 4
|
|
)
|
|
|
|
// tokenScore is how well one typed token matches one single-word term.
|
|
func tokenScore(token, term string) int {
|
|
switch {
|
|
case token == term:
|
|
return scoreExact
|
|
case len(token) >= minPrefixToken && strings.HasPrefix(term, token):
|
|
return scorePrefix
|
|
case len(term) >= minExtensionTerm && strings.HasPrefix(token, term):
|
|
return scoreExtension
|
|
default:
|
|
return 0
|
|
}
|
|
}
|
|
|
|
// termsScore ranks a whole term list against the query.
|
|
//
|
|
// Multi-word terms are matched against the phrase, because "how many" means
|
|
// something its two words do not. Single-word terms are scored per TYPED TOKEN,
|
|
// taking that token's best term — so a query is rewarded for how much of what
|
|
// the user typed the intent accounts for, and an intent cannot climb the
|
|
// ranking by listing eight synonyms of one word.
|
|
func termsScore(terms []string, tokens []string, phrase string) int {
|
|
total := 0
|
|
for _, term := range terms {
|
|
if !strings.Contains(term, " ") {
|
|
continue
|
|
}
|
|
if strings.Contains(phrase, term) {
|
|
total += scorePhrase + 2*strings.Count(term, " ")
|
|
}
|
|
}
|
|
for _, token := range tokens {
|
|
best := 0
|
|
for _, term := range terms {
|
|
if strings.Contains(term, " ") {
|
|
continue
|
|
}
|
|
if s := tokenScore(token, term); s > best {
|
|
best = s
|
|
}
|
|
}
|
|
total += best
|
|
}
|
|
return total
|
|
}
|
|
|
|
// matchShape is the section type the query asked for, if it asked for one.
|
|
// Highest scoring wins; ties go to declaration order, which puts `summary`
|
|
// first.
|
|
func matchShape(tokens []string, phrase string) (shape, bool) {
|
|
best, bestScore := shape{}, 0
|
|
for _, s := range shapes {
|
|
if score := termsScore(s.terms, tokens, phrase); score > bestScore {
|
|
best, bestScore = s, score
|
|
}
|
|
}
|
|
return best, bestScore > 0
|
|
}
|
|
|
|
// supportsShape reports whether an intent can be drawn as a section type.
|
|
// `summary` needs only a subject, having no component of its own; every other
|
|
// shape must be one the intent declares.
|
|
func (i Intent) supportsShape(id string) bool {
|
|
if i.Subject == "" {
|
|
return false
|
|
}
|
|
if id == "summary" {
|
|
return true
|
|
}
|
|
for _, s := range i.Shapes {
|
|
if s == id {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
// shaped is the suggestion text for an intent asked for in a given shape.
|
|
func (i Intent) shaped(s shape) string {
|
|
if s.id == "summary" {
|
|
return "Summarize " + i.Subject
|
|
}
|
|
return "Show " + i.Subject + " " + s.phrase
|
|
}
|
|
|
|
/* ── The pipeline ───────────────────────────────────────────────────────── */
|
|
|
|
// filterOnTopic keeps the candidates the query actually named, or nothing if
|
|
// it named none.
|
|
func filterOnTopic(candidates []scored) []scored {
|
|
out := make([]scored, 0, len(candidates))
|
|
for _, c := range candidates {
|
|
if c.onTopic {
|
|
out = append(out, c)
|
|
}
|
|
}
|
|
return out
|
|
}
|
|
|
|
// scored is one candidate on its way through ranking.
|
|
type scored struct {
|
|
suggestion Suggestion
|
|
score int
|
|
order int // declaration index, the tie-break
|
|
|
|
// onTopic records that the query matched this reading's own terms, rather
|
|
// than only naming a section type it happens to support. See Suggest.
|
|
onTopic bool
|
|
}
|
|
|
|
// Suggest ranks the page's catalogue against what the user has typed.
|
|
//
|
|
// The order is fixed and each stage only ever removes: page context, then
|
|
// permission, then relevance, then duplicates, then the cap. Permission comes
|
|
// before relevance so a reading the caller cannot perform is never scored, and
|
|
// therefore cannot be leaked by an ordering bug later.
|
|
//
|
|
// It returns an empty slice, never nil and never a filler suggestion: a query
|
|
// that matches nothing on this page has no answer here, and saying so is more
|
|
// useful than three questions the user did not ask.
|
|
func Suggest(page, query string, role domain.Role) []Suggestion {
|
|
out := []Suggestion{}
|
|
|
|
// Deny by default, as policy.go does. The service resolves the role before
|
|
// calling, so an unrecognised one should be unreachable — but an intent
|
|
// that reads nothing is permitted by every role it is asked about, so
|
|
// without this line a caller whose role failed to parse would be offered
|
|
// the account readings. The check belongs where the answer is decided.
|
|
if _, known := domain.ParseRole(string(role)); !known {
|
|
return out
|
|
}
|
|
|
|
intents, ok := catalogue[page]
|
|
if !ok {
|
|
return out
|
|
}
|
|
|
|
phrase, tokens, meaningful := normalize(query)
|
|
if meaningful < MinQueryChars {
|
|
return out
|
|
}
|
|
|
|
requested, wantsShape := matchShape(tokens, phrase)
|
|
|
|
candidates := make([]scored, 0, len(intents))
|
|
for order, intent := range intents {
|
|
if !intent.permitted(role) {
|
|
continue
|
|
}
|
|
|
|
score := termsScore(intent.Terms, tokens, phrase)
|
|
onTopic := score > 0
|
|
|
|
suggestion := Suggestion{Text: intent.Text, Intent: intent.ID}
|
|
if wantsShape && intent.supportsShape(requested.id) {
|
|
score += scoreShaped
|
|
suggestion.Text = intent.shaped(requested)
|
|
suggestion.Capability = requested.id
|
|
}
|
|
|
|
if score == 0 {
|
|
continue
|
|
}
|
|
candidates = append(candidates, scored{
|
|
suggestion: suggestion, score: score, order: order, onTopic: onTopic,
|
|
})
|
|
}
|
|
|
|
// A shape on its own is a weak signal, and what it means depends on what
|
|
// else matched. "as a table" typed alone is a real request — draw this
|
|
// page's readings that way — but the same words after "hiring activity" are
|
|
// how the user asked for ONE reading, and offering two more that merely
|
|
// support tables is the padding this endpoint is supposed to refuse.
|
|
//
|
|
// So the two are kept in separate tiers: if anything matched the query's
|
|
// subject, only those compete. Shape-only matches answer for the whole page
|
|
// or not at all.
|
|
if onTopic := filterOnTopic(candidates); len(onTopic) > 0 {
|
|
candidates = onTopic
|
|
}
|
|
|
|
// Highest score first; declaration order breaks every tie, so the same
|
|
// request always produces the same three in the same sequence.
|
|
sort.SliceStable(candidates, func(a, b int) bool {
|
|
if candidates[a].score != candidates[b].score {
|
|
return candidates[a].score > candidates[b].score
|
|
}
|
|
return candidates[a].order < candidates[b].order
|
|
})
|
|
|
|
// One suggestion per intent, and no two reading the same. The catalogue is
|
|
// unique per page by construction — TestCatalogueIsWellFormed holds it that
|
|
// way — so this guards the shaped rewrite, which can phrase two intents
|
|
// identically only if two subjects ever collide.
|
|
seenIntent := make(map[string]bool, MaxSuggestions)
|
|
seenText := make(map[string]bool, MaxSuggestions)
|
|
for _, c := range candidates {
|
|
if len(out) == MaxSuggestions {
|
|
break
|
|
}
|
|
key := strings.ToLower(c.suggestion.Text)
|
|
if seenIntent[c.suggestion.Intent] || seenText[key] {
|
|
continue
|
|
}
|
|
seenIntent[c.suggestion.Intent] = true
|
|
seenText[key] = true
|
|
out = append(out, c.suggestion)
|
|
}
|
|
return out
|
|
}
|