aravind changes
This commit is contained in:
359
go-api/internal/owliver/suggest.go
Normal file
359
go-api/internal/owliver/suggest.go
Normal file
@@ -0,0 +1,359 @@
|
||||
package owliver
|
||||
|
||||
import (
|
||||
"sort"
|
||||
"strings"
|
||||
"unicode"
|
||||
|
||||
"github.com/krow/krow-backend/go-api/internal/domain"
|
||||
)
|
||||
|
||||
// MaxSuggestions is the most a response may carry.
|
||||
//
|
||||
// Three, because the panel shows them under a composer the user is still typing
|
||||
// into. A fourth line pushes the input off a phone screen, and a ranked list
|
||||
// nobody reads to the bottom is a longer list, not a better one.
|
||||
const MaxSuggestions = 3
|
||||
|
||||
// MinQueryChars is the shortest query that is worth ranking.
|
||||
//
|
||||
// Counted in letters and digits after normalization, so " a " and "?!" are
|
||||
// both too short. One character matches a prefix of almost every term in the
|
||||
// catalogue, which would make the first keystroke return three arbitrary
|
||||
// readings and the second replace all three — noise that reads as a bug.
|
||||
const MinQueryChars = 2
|
||||
|
||||
// MaxQueryChars bounds the work one request can ask for. The panel sends what
|
||||
// is in the composer, and nothing about a suggestion improves past a couple of
|
||||
// sentences. Beyond this the query is truncated, never rejected: a long paste
|
||||
// should rank on its opening words, not fail.
|
||||
const MaxQueryChars = 200
|
||||
|
||||
// Suggestion is one offered question.
|
||||
//
|
||||
// Text is what the user reads. Intent is the frontend capability id the panel
|
||||
// dispatches on. Capability is the section type the answer should be drawn as,
|
||||
// present only when the query asked for one — nothing internal is exposed here:
|
||||
// no terms, no resource names, no policy detail, no scores.
|
||||
type Suggestion struct {
|
||||
Text string `json:"text"`
|
||||
Intent string `json:"intent"`
|
||||
Capability string `json:"capability,omitempty"`
|
||||
}
|
||||
|
||||
/* ── Shapes ─────────────────────────────────────────────────────────────── */
|
||||
|
||||
// shape is one section type an answer can be drawn as.
|
||||
//
|
||||
// Transcribed from OWLIVER_CAPABILITIES in src/lib/skills/surfaces.js: the ids
|
||||
// and the terms are that table's, and `phrase` is how the id reads inside a
|
||||
// sentence. `summary` is first and has no component — it is prose, so it
|
||||
// applies to any intent with a subject.
|
||||
//
|
||||
// Terms here describe the SHAPE and never a subject, which is what keeps a
|
||||
// shape from dragging in another page's readings: "as a flow" belongs here,
|
||||
// "hiring activity" belongs to an intent.
|
||||
type shape struct {
|
||||
id string
|
||||
phrase string
|
||||
terms []string
|
||||
}
|
||||
|
||||
var shapes = []shape{
|
||||
{id: "summary", phrase: "", terms: []string{
|
||||
"summary", "summarise", "summarize", "summarised", "summarized",
|
||||
"summarising", "summarizing", "sum up", "recap", "overview", "brief me",
|
||||
"in short", "tell me about"}},
|
||||
{id: "flow", phrase: "as a flow", terms: []string{
|
||||
"flow", "as a flow", "chart", "graph", "diagram", "funnel", "visual",
|
||||
"visualise", "visualize", "step by step"}},
|
||||
{id: "stats", phrase: "as stats", terms: []string{
|
||||
"stats", "statistics", "figures", "numbers", "counts"}},
|
||||
{id: "list", phrase: "as a list", terms: []string{
|
||||
"list", "which ones", "show me the records"}},
|
||||
{id: "table", phrase: "as a table", terms: []string{
|
||||
"table", "as a table", "rows", "grid", "spreadsheet"}},
|
||||
{id: "timeline", phrase: "as a timeline", terms: []string{
|
||||
"timeline", "chronology", "over time", "what happened"}},
|
||||
{id: "progress", phrase: "as progress bars", terms: []string{
|
||||
"progress", "bars", "completion", "how far"}},
|
||||
{id: "weights", phrase: "as weights", terms: []string{
|
||||
"weighting", "weightings", "set the weights", "adjust the weights",
|
||||
"screening weight", "vetting weight"}},
|
||||
{id: "insight", phrase: "as an insight", terms: []string{
|
||||
"insight", "finding", "takeaway", "headline"}},
|
||||
{id: "card", phrase: "as a card", terms: []string{
|
||||
"card", "panel", "at a glance"}},
|
||||
}
|
||||
|
||||
/* ── Normalization ──────────────────────────────────────────────────────── */
|
||||
|
||||
// normalize reduces a raw query to the one form everything downstream matches
|
||||
// against: lower case, letters and digits only, single-spaced.
|
||||
//
|
||||
// Every other character — punctuation, quotes, brackets, control characters,
|
||||
// emoji, an SQL fragment, a script tag — becomes a space rather than being
|
||||
// stripped, so nothing can be glued into a token that was not typed as one.
|
||||
// The result is compared against a fixed table of literals and never reaches a
|
||||
// query, a template or a log message, so there is no construction to inject
|
||||
// into; this is about matching sanely, not about escaping.
|
||||
func normalize(raw string) (phrase string, tokens []string, meaningful int) {
|
||||
runes := []rune(raw)
|
||||
if len(runes) > MaxQueryChars {
|
||||
runes = runes[:MaxQueryChars]
|
||||
}
|
||||
|
||||
var b strings.Builder
|
||||
b.Grow(len(runes))
|
||||
for _, r := range runes {
|
||||
switch {
|
||||
case unicode.IsLetter(r) || unicode.IsDigit(r):
|
||||
b.WriteRune(unicode.ToLower(r))
|
||||
meaningful++
|
||||
default:
|
||||
b.WriteByte(' ')
|
||||
}
|
||||
}
|
||||
|
||||
tokens = strings.Fields(b.String())
|
||||
return strings.Join(tokens, " "), tokens, meaningful
|
||||
}
|
||||
|
||||
/* ── Scoring ────────────────────────────────────────────────────────────── */
|
||||
|
||||
// Scores are small integers with a deliberate order:
|
||||
//
|
||||
// exact a token is the term — the user typed it
|
||||
// phrase a multi-word term appears in the query — the most specific hit
|
||||
// prefix the term begins with a token — mid-typing: "pipel"
|
||||
// extension a token begins with the term — "pipelines"
|
||||
// shaped the query named a section type — weakest on its own
|
||||
//
|
||||
// A shape hit is worth less than any subject hit, so typing "overview" can
|
||||
// surface a page's readings but can never outrank a reading the user named.
|
||||
const (
|
||||
scoreExact = 10
|
||||
scorePhrase = 12
|
||||
scorePrefix = 6
|
||||
scoreExtension = 5
|
||||
scoreShaped = 4
|
||||
|
||||
// minPrefixToken keeps one- and two-letter tokens from matching a term by
|
||||
// prefix. "a" begins nothing usefully; "at" would match "attention",
|
||||
// "audit" and "activity" at once.
|
||||
minPrefixToken = 3
|
||||
// minExtensionTerm keeps a short term from being found inside a longer
|
||||
// word: without it "list" matches "listen" and "score" matches "scoreboard".
|
||||
minExtensionTerm = 4
|
||||
)
|
||||
|
||||
// tokenScore is how well one typed token matches one single-word term.
|
||||
func tokenScore(token, term string) int {
|
||||
switch {
|
||||
case token == term:
|
||||
return scoreExact
|
||||
case len(token) >= minPrefixToken && strings.HasPrefix(term, token):
|
||||
return scorePrefix
|
||||
case len(term) >= minExtensionTerm && strings.HasPrefix(token, term):
|
||||
return scoreExtension
|
||||
default:
|
||||
return 0
|
||||
}
|
||||
}
|
||||
|
||||
// termsScore ranks a whole term list against the query.
|
||||
//
|
||||
// Multi-word terms are matched against the phrase, because "how many" means
|
||||
// something its two words do not. Single-word terms are scored per TYPED TOKEN,
|
||||
// taking that token's best term — so a query is rewarded for how much of what
|
||||
// the user typed the intent accounts for, and an intent cannot climb the
|
||||
// ranking by listing eight synonyms of one word.
|
||||
func termsScore(terms []string, tokens []string, phrase string) int {
|
||||
total := 0
|
||||
for _, term := range terms {
|
||||
if !strings.Contains(term, " ") {
|
||||
continue
|
||||
}
|
||||
if strings.Contains(phrase, term) {
|
||||
total += scorePhrase + 2*strings.Count(term, " ")
|
||||
}
|
||||
}
|
||||
for _, token := range tokens {
|
||||
best := 0
|
||||
for _, term := range terms {
|
||||
if strings.Contains(term, " ") {
|
||||
continue
|
||||
}
|
||||
if s := tokenScore(token, term); s > best {
|
||||
best = s
|
||||
}
|
||||
}
|
||||
total += best
|
||||
}
|
||||
return total
|
||||
}
|
||||
|
||||
// matchShape is the section type the query asked for, if it asked for one.
|
||||
// Highest scoring wins; ties go to declaration order, which puts `summary`
|
||||
// first.
|
||||
func matchShape(tokens []string, phrase string) (shape, bool) {
|
||||
best, bestScore := shape{}, 0
|
||||
for _, s := range shapes {
|
||||
if score := termsScore(s.terms, tokens, phrase); score > bestScore {
|
||||
best, bestScore = s, score
|
||||
}
|
||||
}
|
||||
return best, bestScore > 0
|
||||
}
|
||||
|
||||
// supportsShape reports whether an intent can be drawn as a section type.
|
||||
// `summary` needs only a subject, having no component of its own; every other
|
||||
// shape must be one the intent declares.
|
||||
func (i Intent) supportsShape(id string) bool {
|
||||
if i.Subject == "" {
|
||||
return false
|
||||
}
|
||||
if id == "summary" {
|
||||
return true
|
||||
}
|
||||
for _, s := range i.Shapes {
|
||||
if s == id {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// shaped is the suggestion text for an intent asked for in a given shape.
|
||||
func (i Intent) shaped(s shape) string {
|
||||
if s.id == "summary" {
|
||||
return "Summarize " + i.Subject
|
||||
}
|
||||
return "Show " + i.Subject + " " + s.phrase
|
||||
}
|
||||
|
||||
/* ── The pipeline ───────────────────────────────────────────────────────── */
|
||||
|
||||
// filterOnTopic keeps the candidates the query actually named, or nothing if
|
||||
// it named none.
|
||||
func filterOnTopic(candidates []scored) []scored {
|
||||
out := make([]scored, 0, len(candidates))
|
||||
for _, c := range candidates {
|
||||
if c.onTopic {
|
||||
out = append(out, c)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// scored is one candidate on its way through ranking.
|
||||
type scored struct {
|
||||
suggestion Suggestion
|
||||
score int
|
||||
order int // declaration index, the tie-break
|
||||
|
||||
// onTopic records that the query matched this reading's own terms, rather
|
||||
// than only naming a section type it happens to support. See Suggest.
|
||||
onTopic bool
|
||||
}
|
||||
|
||||
// Suggest ranks the page's catalogue against what the user has typed.
|
||||
//
|
||||
// The order is fixed and each stage only ever removes: page context, then
|
||||
// permission, then relevance, then duplicates, then the cap. Permission comes
|
||||
// before relevance so a reading the caller cannot perform is never scored, and
|
||||
// therefore cannot be leaked by an ordering bug later.
|
||||
//
|
||||
// It returns an empty slice, never nil and never a filler suggestion: a query
|
||||
// that matches nothing on this page has no answer here, and saying so is more
|
||||
// useful than three questions the user did not ask.
|
||||
func Suggest(page, query string, role domain.Role) []Suggestion {
|
||||
out := []Suggestion{}
|
||||
|
||||
// Deny by default, as policy.go does. The service resolves the role before
|
||||
// calling, so an unrecognised one should be unreachable — but an intent
|
||||
// that reads nothing is permitted by every role it is asked about, so
|
||||
// without this line a caller whose role failed to parse would be offered
|
||||
// the account readings. The check belongs where the answer is decided.
|
||||
if _, known := domain.ParseRole(string(role)); !known {
|
||||
return out
|
||||
}
|
||||
|
||||
intents, ok := catalogue[page]
|
||||
if !ok {
|
||||
return out
|
||||
}
|
||||
|
||||
phrase, tokens, meaningful := normalize(query)
|
||||
if meaningful < MinQueryChars {
|
||||
return out
|
||||
}
|
||||
|
||||
requested, wantsShape := matchShape(tokens, phrase)
|
||||
|
||||
candidates := make([]scored, 0, len(intents))
|
||||
for order, intent := range intents {
|
||||
if !intent.permitted(role) {
|
||||
continue
|
||||
}
|
||||
|
||||
score := termsScore(intent.Terms, tokens, phrase)
|
||||
onTopic := score > 0
|
||||
|
||||
suggestion := Suggestion{Text: intent.Text, Intent: intent.ID}
|
||||
if wantsShape && intent.supportsShape(requested.id) {
|
||||
score += scoreShaped
|
||||
suggestion.Text = intent.shaped(requested)
|
||||
suggestion.Capability = requested.id
|
||||
}
|
||||
|
||||
if score == 0 {
|
||||
continue
|
||||
}
|
||||
candidates = append(candidates, scored{
|
||||
suggestion: suggestion, score: score, order: order, onTopic: onTopic,
|
||||
})
|
||||
}
|
||||
|
||||
// A shape on its own is a weak signal, and what it means depends on what
|
||||
// else matched. "as a table" typed alone is a real request — draw this
|
||||
// page's readings that way — but the same words after "hiring activity" are
|
||||
// how the user asked for ONE reading, and offering two more that merely
|
||||
// support tables is the padding this endpoint is supposed to refuse.
|
||||
//
|
||||
// So the two are kept in separate tiers: if anything matched the query's
|
||||
// subject, only those compete. Shape-only matches answer for the whole page
|
||||
// or not at all.
|
||||
if onTopic := filterOnTopic(candidates); len(onTopic) > 0 {
|
||||
candidates = onTopic
|
||||
}
|
||||
|
||||
// Highest score first; declaration order breaks every tie, so the same
|
||||
// request always produces the same three in the same sequence.
|
||||
sort.SliceStable(candidates, func(a, b int) bool {
|
||||
if candidates[a].score != candidates[b].score {
|
||||
return candidates[a].score > candidates[b].score
|
||||
}
|
||||
return candidates[a].order < candidates[b].order
|
||||
})
|
||||
|
||||
// One suggestion per intent, and no two reading the same. The catalogue is
|
||||
// unique per page by construction — TestCatalogueIsWellFormed holds it that
|
||||
// way — so this guards the shaped rewrite, which can phrase two intents
|
||||
// identically only if two subjects ever collide.
|
||||
seenIntent := make(map[string]bool, MaxSuggestions)
|
||||
seenText := make(map[string]bool, MaxSuggestions)
|
||||
for _, c := range candidates {
|
||||
if len(out) == MaxSuggestions {
|
||||
break
|
||||
}
|
||||
key := strings.ToLower(c.suggestion.Text)
|
||||
if seenIntent[c.suggestion.Intent] || seenText[key] {
|
||||
continue
|
||||
}
|
||||
seenIntent[c.suggestion.Intent] = true
|
||||
seenText[key] = true
|
||||
out = append(out, c.suggestion)
|
||||
}
|
||||
return out
|
||||
}
|
||||
Reference in New Issue
Block a user