package owliver import ( "sort" "strings" "unicode" "github.com/krow/krow-backend/go-api/internal/domain" ) // MaxSuggestions is the most a response may carry. // // Three, because the panel shows them under a composer the user is still typing // into. A fourth line pushes the input off a phone screen, and a ranked list // nobody reads to the bottom is a longer list, not a better one. const MaxSuggestions = 3 // MinQueryChars is the shortest query that is worth ranking. // // Counted in letters and digits after normalization, so " a " and "?!" are // both too short. One character matches a prefix of almost every term in the // catalogue, which would make the first keystroke return three arbitrary // readings and the second replace all three — noise that reads as a bug. const MinQueryChars = 2 // MaxQueryChars bounds the work one request can ask for. The panel sends what // is in the composer, and nothing about a suggestion improves past a couple of // sentences. Beyond this the query is truncated, never rejected: a long paste // should rank on its opening words, not fail. const MaxQueryChars = 200 // Suggestion is one offered question. // // Text is what the user reads. Intent is the frontend capability id the panel // dispatches on. Capability is the section type the answer should be drawn as, // present only when the query asked for one — nothing internal is exposed here: // no terms, no resource names, no policy detail, no scores. type Suggestion struct { Text string `json:"text"` Intent string `json:"intent"` Capability string `json:"capability,omitempty"` } /* ── Shapes ─────────────────────────────────────────────────────────────── */ // shape is one section type an answer can be drawn as. // // Transcribed from OWLIVER_CAPABILITIES in src/lib/skills/surfaces.js: the ids // and the terms are that table's, and `phrase` is how the id reads inside a // sentence. `summary` is first and has no component — it is prose, so it // applies to any intent with a subject. // // Terms here describe the SHAPE and never a subject, which is what keeps a // shape from dragging in another page's readings: "as a flow" belongs here, // "hiring activity" belongs to an intent. type shape struct { id string phrase string terms []string } var shapes = []shape{ {id: "summary", phrase: "", terms: []string{ "summary", "summarise", "summarize", "summarised", "summarized", "summarising", "summarizing", "sum up", "recap", "overview", "brief me", "in short", "tell me about"}}, {id: "flow", phrase: "as a flow", terms: []string{ "flow", "as a flow", "chart", "graph", "diagram", "funnel", "visual", "visualise", "visualize", "step by step"}}, {id: "stats", phrase: "as stats", terms: []string{ "stats", "statistics", "figures", "numbers", "counts"}}, {id: "list", phrase: "as a list", terms: []string{ "list", "which ones", "show me the records"}}, {id: "table", phrase: "as a table", terms: []string{ "table", "as a table", "rows", "grid", "spreadsheet"}}, {id: "timeline", phrase: "as a timeline", terms: []string{ "timeline", "chronology", "over time", "what happened"}}, {id: "progress", phrase: "as progress bars", terms: []string{ "progress", "bars", "completion", "how far"}}, {id: "weights", phrase: "as weights", terms: []string{ "weighting", "weightings", "set the weights", "adjust the weights", "screening weight", "vetting weight"}}, {id: "insight", phrase: "as an insight", terms: []string{ "insight", "finding", "takeaway", "headline"}}, {id: "card", phrase: "as a card", terms: []string{ "card", "panel", "at a glance"}}, } /* ── Normalization ──────────────────────────────────────────────────────── */ // normalize reduces a raw query to the one form everything downstream matches // against: lower case, letters and digits only, single-spaced. // // Every other character — punctuation, quotes, brackets, control characters, // emoji, an SQL fragment, a script tag — becomes a space rather than being // stripped, so nothing can be glued into a token that was not typed as one. // The result is compared against a fixed table of literals and never reaches a // query, a template or a log message, so there is no construction to inject // into; this is about matching sanely, not about escaping. func normalize(raw string) (phrase string, tokens []string, meaningful int) { runes := []rune(raw) if len(runes) > MaxQueryChars { runes = runes[:MaxQueryChars] } var b strings.Builder b.Grow(len(runes)) for _, r := range runes { switch { case unicode.IsLetter(r) || unicode.IsDigit(r): b.WriteRune(unicode.ToLower(r)) meaningful++ default: b.WriteByte(' ') } } tokens = strings.Fields(b.String()) return strings.Join(tokens, " "), tokens, meaningful } /* ── Scoring ────────────────────────────────────────────────────────────── */ // Scores are small integers with a deliberate order: // // exact a token is the term — the user typed it // phrase a multi-word term appears in the query — the most specific hit // prefix the term begins with a token — mid-typing: "pipel" // extension a token begins with the term — "pipelines" // shaped the query named a section type — weakest on its own // // A shape hit is worth less than any subject hit, so typing "overview" can // surface a page's readings but can never outrank a reading the user named. const ( scoreExact = 10 scorePhrase = 12 scorePrefix = 6 scoreExtension = 5 scoreShaped = 4 // minPrefixToken keeps one- and two-letter tokens from matching a term by // prefix. "a" begins nothing usefully; "at" would match "attention", // "audit" and "activity" at once. minPrefixToken = 3 // minExtensionTerm keeps a short term from being found inside a longer // word: without it "list" matches "listen" and "score" matches "scoreboard". minExtensionTerm = 4 ) // tokenScore is how well one typed token matches one single-word term. func tokenScore(token, term string) int { switch { case token == term: return scoreExact case len(token) >= minPrefixToken && strings.HasPrefix(term, token): return scorePrefix case len(term) >= minExtensionTerm && strings.HasPrefix(token, term): return scoreExtension default: return 0 } } // termsScore ranks a whole term list against the query. // // Multi-word terms are matched against the phrase, because "how many" means // something its two words do not. Single-word terms are scored per TYPED TOKEN, // taking that token's best term — so a query is rewarded for how much of what // the user typed the intent accounts for, and an intent cannot climb the // ranking by listing eight synonyms of one word. func termsScore(terms []string, tokens []string, phrase string) int { total := 0 for _, term := range terms { if !strings.Contains(term, " ") { continue } if strings.Contains(phrase, term) { total += scorePhrase + 2*strings.Count(term, " ") } } for _, token := range tokens { best := 0 for _, term := range terms { if strings.Contains(term, " ") { continue } if s := tokenScore(token, term); s > best { best = s } } total += best } return total } // matchShape is the section type the query asked for, if it asked for one. // Highest scoring wins; ties go to declaration order, which puts `summary` // first. func matchShape(tokens []string, phrase string) (shape, bool) { best, bestScore := shape{}, 0 for _, s := range shapes { if score := termsScore(s.terms, tokens, phrase); score > bestScore { best, bestScore = s, score } } return best, bestScore > 0 } // supportsShape reports whether an intent can be drawn as a section type. // `summary` needs only a subject, having no component of its own; every other // shape must be one the intent declares. func (i Intent) supportsShape(id string) bool { if i.Subject == "" { return false } if id == "summary" { return true } for _, s := range i.Shapes { if s == id { return true } } return false } // shaped is the suggestion text for an intent asked for in a given shape. func (i Intent) shaped(s shape) string { if s.id == "summary" { return "Summarize " + i.Subject } return "Show " + i.Subject + " " + s.phrase } /* ── The pipeline ───────────────────────────────────────────────────────── */ // filterOnTopic keeps the candidates the query actually named, or nothing if // it named none. func filterOnTopic(candidates []scored) []scored { out := make([]scored, 0, len(candidates)) for _, c := range candidates { if c.onTopic { out = append(out, c) } } return out } // scored is one candidate on its way through ranking. type scored struct { suggestion Suggestion score int order int // declaration index, the tie-break // onTopic records that the query matched this reading's own terms, rather // than only naming a section type it happens to support. See Suggest. onTopic bool } // Suggest ranks the page's catalogue against what the user has typed. // // The order is fixed and each stage only ever removes: page context, then // permission, then relevance, then duplicates, then the cap. Permission comes // before relevance so a reading the caller cannot perform is never scored, and // therefore cannot be leaked by an ordering bug later. // // It returns an empty slice, never nil and never a filler suggestion: a query // that matches nothing on this page has no answer here, and saying so is more // useful than three questions the user did not ask. func Suggest(page, query string, role domain.Role) []Suggestion { out := []Suggestion{} // Deny by default, as policy.go does. The service resolves the role before // calling, so an unrecognised one should be unreachable — but an intent // that reads nothing is permitted by every role it is asked about, so // without this line a caller whose role failed to parse would be offered // the account readings. The check belongs where the answer is decided. if _, known := domain.ParseRole(string(role)); !known { return out } intents, ok := catalogue[page] if !ok { return out } phrase, tokens, meaningful := normalize(query) if meaningful < MinQueryChars { return out } requested, wantsShape := matchShape(tokens, phrase) candidates := make([]scored, 0, len(intents)) for order, intent := range intents { if !intent.permitted(role) { continue } score := termsScore(intent.Terms, tokens, phrase) onTopic := score > 0 suggestion := Suggestion{Text: intent.Text, Intent: intent.ID} if wantsShape && intent.supportsShape(requested.id) { score += scoreShaped suggestion.Text = intent.shaped(requested) suggestion.Capability = requested.id } if score == 0 { continue } candidates = append(candidates, scored{ suggestion: suggestion, score: score, order: order, onTopic: onTopic, }) } // A shape on its own is a weak signal, and what it means depends on what // else matched. "as a table" typed alone is a real request — draw this // page's readings that way — but the same words after "hiring activity" are // how the user asked for ONE reading, and offering two more that merely // support tables is the padding this endpoint is supposed to refuse. // // So the two are kept in separate tiers: if anything matched the query's // subject, only those compete. Shape-only matches answer for the whole page // or not at all. if onTopic := filterOnTopic(candidates); len(onTopic) > 0 { candidates = onTopic } // Highest score first; declaration order breaks every tie, so the same // request always produces the same three in the same sequence. sort.SliceStable(candidates, func(a, b int) bool { if candidates[a].score != candidates[b].score { return candidates[a].score > candidates[b].score } return candidates[a].order < candidates[b].order }) // One suggestion per intent, and no two reading the same. The catalogue is // unique per page by construction — TestCatalogueIsWellFormed holds it that // way — so this guards the shaped rewrite, which can phrase two intents // identically only if two subjects ever collide. seenIntent := make(map[string]bool, MaxSuggestions) seenText := make(map[string]bool, MaxSuggestions) for _, c := range candidates { if len(out) == MaxSuggestions { break } key := strings.ToLower(c.suggestion.Text) if seenIntent[c.suggestion.Intent] || seenText[key] { continue } seenIntent[c.suggestion.Intent] = true seenText[key] = true out = append(out, c.suggestion) } return out }