Files
krow_backend/go-api/internal/owliver/suggest.go
2026-08-25 16:37:05 +05:30

360 lines
13 KiB
Go

package owliver
import (
"sort"
"strings"
"unicode"
"github.com/krow/krow-backend/go-api/internal/domain"
)
// MaxSuggestions is the most a response may carry.
//
// Three, because the panel shows them under a composer the user is still typing
// into. A fourth line pushes the input off a phone screen, and a ranked list
// nobody reads to the bottom is a longer list, not a better one.
const MaxSuggestions = 3
// MinQueryChars is the shortest query that is worth ranking.
//
// Counted in letters and digits after normalization, so " a " and "?!" are
// both too short. One character matches a prefix of almost every term in the
// catalogue, which would make the first keystroke return three arbitrary
// readings and the second replace all three — noise that reads as a bug.
const MinQueryChars = 2
// MaxQueryChars bounds the work one request can ask for. The panel sends what
// is in the composer, and nothing about a suggestion improves past a couple of
// sentences. Beyond this the query is truncated, never rejected: a long paste
// should rank on its opening words, not fail.
const MaxQueryChars = 200
// Suggestion is one offered question.
//
// Text is what the user reads. Intent is the frontend capability id the panel
// dispatches on. Capability is the section type the answer should be drawn as,
// present only when the query asked for one — nothing internal is exposed here:
// no terms, no resource names, no policy detail, no scores.
type Suggestion struct {
Text string `json:"text"`
Intent string `json:"intent"`
Capability string `json:"capability,omitempty"`
}
/* ── Shapes ─────────────────────────────────────────────────────────────── */
// shape is one section type an answer can be drawn as.
//
// Transcribed from OWLIVER_CAPABILITIES in src/lib/skills/surfaces.js: the ids
// and the terms are that table's, and `phrase` is how the id reads inside a
// sentence. `summary` is first and has no component — it is prose, so it
// applies to any intent with a subject.
//
// Terms here describe the SHAPE and never a subject, which is what keeps a
// shape from dragging in another page's readings: "as a flow" belongs here,
// "hiring activity" belongs to an intent.
type shape struct {
id string
phrase string
terms []string
}
var shapes = []shape{
{id: "summary", phrase: "", terms: []string{
"summary", "summarise", "summarize", "summarised", "summarized",
"summarising", "summarizing", "sum up", "recap", "overview", "brief me",
"in short", "tell me about"}},
{id: "flow", phrase: "as a flow", terms: []string{
"flow", "as a flow", "chart", "graph", "diagram", "funnel", "visual",
"visualise", "visualize", "step by step"}},
{id: "stats", phrase: "as stats", terms: []string{
"stats", "statistics", "figures", "numbers", "counts"}},
{id: "list", phrase: "as a list", terms: []string{
"list", "which ones", "show me the records"}},
{id: "table", phrase: "as a table", terms: []string{
"table", "as a table", "rows", "grid", "spreadsheet"}},
{id: "timeline", phrase: "as a timeline", terms: []string{
"timeline", "chronology", "over time", "what happened"}},
{id: "progress", phrase: "as progress bars", terms: []string{
"progress", "bars", "completion", "how far"}},
{id: "weights", phrase: "as weights", terms: []string{
"weighting", "weightings", "set the weights", "adjust the weights",
"screening weight", "vetting weight"}},
{id: "insight", phrase: "as an insight", terms: []string{
"insight", "finding", "takeaway", "headline"}},
{id: "card", phrase: "as a card", terms: []string{
"card", "panel", "at a glance"}},
}
/* ── Normalization ──────────────────────────────────────────────────────── */
// normalize reduces a raw query to the one form everything downstream matches
// against: lower case, letters and digits only, single-spaced.
//
// Every other character — punctuation, quotes, brackets, control characters,
// emoji, an SQL fragment, a script tag — becomes a space rather than being
// stripped, so nothing can be glued into a token that was not typed as one.
// The result is compared against a fixed table of literals and never reaches a
// query, a template or a log message, so there is no construction to inject
// into; this is about matching sanely, not about escaping.
func normalize(raw string) (phrase string, tokens []string, meaningful int) {
runes := []rune(raw)
if len(runes) > MaxQueryChars {
runes = runes[:MaxQueryChars]
}
var b strings.Builder
b.Grow(len(runes))
for _, r := range runes {
switch {
case unicode.IsLetter(r) || unicode.IsDigit(r):
b.WriteRune(unicode.ToLower(r))
meaningful++
default:
b.WriteByte(' ')
}
}
tokens = strings.Fields(b.String())
return strings.Join(tokens, " "), tokens, meaningful
}
/* ── Scoring ────────────────────────────────────────────────────────────── */
// Scores are small integers with a deliberate order:
//
// exact a token is the term — the user typed it
// phrase a multi-word term appears in the query — the most specific hit
// prefix the term begins with a token — mid-typing: "pipel"
// extension a token begins with the term — "pipelines"
// shaped the query named a section type — weakest on its own
//
// A shape hit is worth less than any subject hit, so typing "overview" can
// surface a page's readings but can never outrank a reading the user named.
const (
scoreExact = 10
scorePhrase = 12
scorePrefix = 6
scoreExtension = 5
scoreShaped = 4
// minPrefixToken keeps one- and two-letter tokens from matching a term by
// prefix. "a" begins nothing usefully; "at" would match "attention",
// "audit" and "activity" at once.
minPrefixToken = 3
// minExtensionTerm keeps a short term from being found inside a longer
// word: without it "list" matches "listen" and "score" matches "scoreboard".
minExtensionTerm = 4
)
// tokenScore is how well one typed token matches one single-word term.
func tokenScore(token, term string) int {
switch {
case token == term:
return scoreExact
case len(token) >= minPrefixToken && strings.HasPrefix(term, token):
return scorePrefix
case len(term) >= minExtensionTerm && strings.HasPrefix(token, term):
return scoreExtension
default:
return 0
}
}
// termsScore ranks a whole term list against the query.
//
// Multi-word terms are matched against the phrase, because "how many" means
// something its two words do not. Single-word terms are scored per TYPED TOKEN,
// taking that token's best term — so a query is rewarded for how much of what
// the user typed the intent accounts for, and an intent cannot climb the
// ranking by listing eight synonyms of one word.
func termsScore(terms []string, tokens []string, phrase string) int {
total := 0
for _, term := range terms {
if !strings.Contains(term, " ") {
continue
}
if strings.Contains(phrase, term) {
total += scorePhrase + 2*strings.Count(term, " ")
}
}
for _, token := range tokens {
best := 0
for _, term := range terms {
if strings.Contains(term, " ") {
continue
}
if s := tokenScore(token, term); s > best {
best = s
}
}
total += best
}
return total
}
// matchShape is the section type the query asked for, if it asked for one.
// Highest scoring wins; ties go to declaration order, which puts `summary`
// first.
func matchShape(tokens []string, phrase string) (shape, bool) {
best, bestScore := shape{}, 0
for _, s := range shapes {
if score := termsScore(s.terms, tokens, phrase); score > bestScore {
best, bestScore = s, score
}
}
return best, bestScore > 0
}
// supportsShape reports whether an intent can be drawn as a section type.
// `summary` needs only a subject, having no component of its own; every other
// shape must be one the intent declares.
func (i Intent) supportsShape(id string) bool {
if i.Subject == "" {
return false
}
if id == "summary" {
return true
}
for _, s := range i.Shapes {
if s == id {
return true
}
}
return false
}
// shaped is the suggestion text for an intent asked for in a given shape.
func (i Intent) shaped(s shape) string {
if s.id == "summary" {
return "Summarize " + i.Subject
}
return "Show " + i.Subject + " " + s.phrase
}
/* ── The pipeline ───────────────────────────────────────────────────────── */
// filterOnTopic keeps the candidates the query actually named, or nothing if
// it named none.
func filterOnTopic(candidates []scored) []scored {
out := make([]scored, 0, len(candidates))
for _, c := range candidates {
if c.onTopic {
out = append(out, c)
}
}
return out
}
// scored is one candidate on its way through ranking.
type scored struct {
suggestion Suggestion
score int
order int // declaration index, the tie-break
// onTopic records that the query matched this reading's own terms, rather
// than only naming a section type it happens to support. See Suggest.
onTopic bool
}
// Suggest ranks the page's catalogue against what the user has typed.
//
// The order is fixed and each stage only ever removes: page context, then
// permission, then relevance, then duplicates, then the cap. Permission comes
// before relevance so a reading the caller cannot perform is never scored, and
// therefore cannot be leaked by an ordering bug later.
//
// It returns an empty slice, never nil and never a filler suggestion: a query
// that matches nothing on this page has no answer here, and saying so is more
// useful than three questions the user did not ask.
func Suggest(page, query string, role domain.Role) []Suggestion {
out := []Suggestion{}
// Deny by default, as policy.go does. The service resolves the role before
// calling, so an unrecognised one should be unreachable — but an intent
// that reads nothing is permitted by every role it is asked about, so
// without this line a caller whose role failed to parse would be offered
// the account readings. The check belongs where the answer is decided.
if _, known := domain.ParseRole(string(role)); !known {
return out
}
intents, ok := catalogue[page]
if !ok {
return out
}
phrase, tokens, meaningful := normalize(query)
if meaningful < MinQueryChars {
return out
}
requested, wantsShape := matchShape(tokens, phrase)
candidates := make([]scored, 0, len(intents))
for order, intent := range intents {
if !intent.permitted(role) {
continue
}
score := termsScore(intent.Terms, tokens, phrase)
onTopic := score > 0
suggestion := Suggestion{Text: intent.Text, Intent: intent.ID}
if wantsShape && intent.supportsShape(requested.id) {
score += scoreShaped
suggestion.Text = intent.shaped(requested)
suggestion.Capability = requested.id
}
if score == 0 {
continue
}
candidates = append(candidates, scored{
suggestion: suggestion, score: score, order: order, onTopic: onTopic,
})
}
// A shape on its own is a weak signal, and what it means depends on what
// else matched. "as a table" typed alone is a real request — draw this
// page's readings that way — but the same words after "hiring activity" are
// how the user asked for ONE reading, and offering two more that merely
// support tables is the padding this endpoint is supposed to refuse.
//
// So the two are kept in separate tiers: if anything matched the query's
// subject, only those compete. Shape-only matches answer for the whole page
// or not at all.
if onTopic := filterOnTopic(candidates); len(onTopic) > 0 {
candidates = onTopic
}
// Highest score first; declaration order breaks every tie, so the same
// request always produces the same three in the same sequence.
sort.SliceStable(candidates, func(a, b int) bool {
if candidates[a].score != candidates[b].score {
return candidates[a].score > candidates[b].score
}
return candidates[a].order < candidates[b].order
})
// One suggestion per intent, and no two reading the same. The catalogue is
// unique per page by construction — TestCatalogueIsWellFormed holds it that
// way — so this guards the shaped rewrite, which can phrase two intents
// identically only if two subjects ever collide.
seenIntent := make(map[string]bool, MaxSuggestions)
seenText := make(map[string]bool, MaxSuggestions)
for _, c := range candidates {
if len(out) == MaxSuggestions {
break
}
key := strings.ToLower(c.suggestion.Text)
if seenIntent[c.suggestion.Intent] || seenText[key] {
continue
}
seenIntent[c.suggestion.Intent] = true
seenText[key] = true
out = append(out, c.suggestion)
}
return out
}