127 lines
5.5 KiB
Go
127 lines
5.5 KiB
Go
package runtime
|
||
|
||
import "strings"
|
||
|
||
// isSmalltalk reports a message that cannot be answered any better by looking
|
||
// something up.
|
||
//
|
||
// "Hi" used to cost a full operational turn. Retrieval is gated on
|
||
// configuration and never on the question (see retrieve), so a greeting arrived
|
||
// at the model wrapped in eight policy chunks, with the whole tool catalogue
|
||
// attached and the evidence placed BEFORE the question. A model handed that
|
||
// reasonably concludes it was asked for an operational briefing, and answers
|
||
// with one — screening backlog, uncovered shifts, citations and all. Measured
|
||
// against production: 6,174 tokens over two model calls, for the word "hi".
|
||
//
|
||
// The cost is the smaller half. The real damage is that the product appears not
|
||
// to understand being greeted, which is the first thing anybody tries.
|
||
//
|
||
// This is deliberately NOT a per-agent rule, and not an `if agent_key == ...`
|
||
// — §13 lists that as the anti-pattern it is. It is a property of the MESSAGE,
|
||
// applied identically to every spec, so adding an agent still requires no
|
||
// runtime change (I6).
|
||
//
|
||
// Conservative by construction: the normalised message must match a phrase in
|
||
// the set EXACTLY. Nothing substring-matches, so "hi, which shifts are
|
||
// uncovered?" is an operational question and keeps its tools and its evidence.
|
||
// A false negative costs a few thousand tokens; a false positive answers a real
|
||
// question with a greeting, so the set only holds phrases that carry no request
|
||
// at all.
|
||
func isSmalltalk(q string) bool {
|
||
n := normaliseSmalltalk(q)
|
||
if n == "" {
|
||
return false
|
||
}
|
||
_, ok := smalltalkPhrases[n]
|
||
return ok
|
||
}
|
||
|
||
// normaliseSmalltalk reduces a message to lowercase letters and single spaces.
|
||
//
|
||
// Punctuation and emoji are dropped rather than enumerated, so "Hi!", "hi :)"
|
||
// and "HI 👋" all arrive as "hi" without the set needing a row for each. Digits
|
||
// are NOT letters and so are dropped too, which is harmless here: no phrase in
|
||
// the set contains one, and a message that does — "shift 12?" — fails the exact
|
||
// match either way.
|
||
func normaliseSmalltalk(q string) string {
|
||
var b strings.Builder
|
||
b.Grow(len(q))
|
||
space := false
|
||
for _, r := range strings.ToLower(strings.TrimSpace(q)) {
|
||
switch {
|
||
case r >= 'a' && r <= 'z':
|
||
if space && b.Len() > 0 {
|
||
b.WriteByte(' ')
|
||
}
|
||
space = false
|
||
b.WriteRune(r)
|
||
case r == '\'' || r == '’':
|
||
// Dropped outright rather than treated as a separator, so "how's"
|
||
// stays one word. Both the ASCII quote and the curly one a phone
|
||
// keyboard substitutes — the same character to whoever typed it,
|
||
// and not to the first version of this function.
|
||
default:
|
||
// Any other run of non-letters is one separator, so "thank-you"
|
||
// and "thank you" normalise alike.
|
||
space = true
|
||
}
|
||
}
|
||
return b.String()
|
||
}
|
||
|
||
// smalltalkPhrases is the whole rule, as data.
|
||
//
|
||
// Greetings, thanks and farewells only. Each is a complete message that asks
|
||
// for nothing, which is what makes skipping retrieval and tools safe rather
|
||
// than merely cheap. Acknowledgements like "ok" and "cool" are deliberately
|
||
// absent: they are plausible smalltalk but also plausible answers to a
|
||
// question the agent just asked, and the cost of being wrong is higher than
|
||
// the tokens being saved.
|
||
var smalltalkPhrases = map[string]struct{}{
|
||
"hi": {}, "hii": {}, "hiya": {}, "hello": {}, "helo": {}, "hey": {},
|
||
"yo": {}, "howdy": {}, "greetings": {}, "hi there": {},
|
||
"hello there": {}, "hey there": {}, "hi owliver": {},
|
||
"hello owliver": {}, "hey owliver": {},
|
||
|
||
"good morning": {}, "good afternoon": {}, "good evening": {},
|
||
"good day": {}, "morning": {}, "afternoon": {}, "evening": {},
|
||
"gm": {}, "ge": {},
|
||
|
||
"how are you": {}, "how are you doing": {}, "hows it going": {},
|
||
"how is it going": {}, "you there": {}, "are you there": {},
|
||
|
||
"thanks": {}, "thank you": {}, "thanks a lot": {},
|
||
"thank you very much": {}, "thanks very much": {}, "many thanks": {},
|
||
"ty": {}, "cheers": {}, "thank u": {},
|
||
|
||
"bye": {}, "goodbye": {}, "good bye": {}, "see you": {},
|
||
"see ya": {}, "good night": {}, "goodnight": {}, "later": {},
|
||
}
|
||
|
||
// smalltalkDirective is appended to the system prompt for a smalltalk turn.
|
||
//
|
||
// Needed because the agent's own instructions describe an operational analyst,
|
||
// and an operational analyst greeted with "hi" and given no tools will still
|
||
// reach for the longest answer it can justify. Removing the evidence removes
|
||
// the citations; it does not by itself shorten the reply.
|
||
//
|
||
// Appended to the SYSTEM prompt rather than wrapped around the user's message:
|
||
// it is a standing instruction from the platform, not something the person
|
||
// said, and putting words in their mouth is how a transcript stops matching
|
||
// what was typed. I7 is untouched — this is the runtime's own text, not
|
||
// retrieved content, and nothing retrieved can reach here because retrieval did
|
||
// not run.
|
||
const smalltalkDirective = "\n\nThe person has greeted you or said something " +
|
||
"conversational. Reply in one or two short sentences: greet them back and " +
|
||
"offer to help. Do not summarise data, do not list findings or next steps, " +
|
||
"and do not cite sources — you have not looked anything up."
|
||
|
||
// smalltalkMaxOutputTokens caps a greeting's reply.
|
||
//
|
||
// A ceiling the model is not told about truncates mid-sentence rather than
|
||
// winding down, so this sits well above any sane greeting (a sentence or two is
|
||
// well under 100 tokens) and acts only as a backstop for a model that ignores
|
||
// the directive above. The directive does the shortening; this bounds the bill
|
||
// when it does not.
|
||
const smalltalkMaxOutputTokens = 256
|