342 lines
12 KiB
Go
342 lines
12 KiB
Go
package services
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"strings"
|
|
"time"
|
|
|
|
"nearle/services/tools"
|
|
"nearle/utils"
|
|
)
|
|
|
|
// Nearle Buddy's loop.
|
|
//
|
|
// One loop runs every agent. An agent is a name, a tier, a system prompt and an
|
|
// allow-list — data, not a class — so a sixth agent is a config entry rather
|
|
// than a subclass, and the behaviour they all share cannot drift between them.
|
|
//
|
|
// The shape is the ordinary one: ask the model, run any tools it asked for,
|
|
// give it the results, ask again, stop when it answers in words. What matters
|
|
// is what the loop refuses to let the model decide.
|
|
//
|
|
// ── What the model does not get to choose ───────────────────────────────────
|
|
//
|
|
// - Whose data it reads. The caller comes from the verified session and is
|
|
// passed to the registry directly. No tool accepts a tenant argument.
|
|
// - Which tools exist. The allow-list is the agent's, enforced by the
|
|
// registry; a model asking for something else is refused, not obeyed.
|
|
// - When to stop. Steps and tool calls are counted here. A model that keeps
|
|
// calling tools is stopped by arithmetic, not by being asked nicely.
|
|
//
|
|
// ── A refused tool is a message, not an error ───────────────────────────────
|
|
//
|
|
// When the registry refuses a call, the refusal goes back to the model as the
|
|
// tool's result. A model told "that tool needs a tenant" can explain the
|
|
// problem to the person; a model handed a 500 says "something went wrong",
|
|
// which is true and useless. The refusal is still audited either way.
|
|
|
|
// Agent is one assistant, as data.
|
|
type Agent struct {
|
|
Name string
|
|
Tier string
|
|
// System is what the model is told about its job. Rules that MUST hold do
|
|
// not live here — a prompt is a request. This is for tone, scope and the
|
|
// habits that make an answer useful.
|
|
System string
|
|
Tools []string
|
|
// MaxSteps bounds the conversation: one step is one model round trip.
|
|
MaxSteps int
|
|
// MaxToolCalls bounds the work across the whole conversation, because a
|
|
// model can ask for several tools in a single step.
|
|
MaxToolCalls int
|
|
}
|
|
|
|
// AssistantAnswer is what one question produced.
|
|
type AssistantAnswer struct {
|
|
Reply string `json:"reply"`
|
|
// What was actually run, in order. Returned to the console so an answer can
|
|
// show its working — Buddy states a conclusion, and this is how a person
|
|
// sees which numbers it came from.
|
|
Used []AssistantStep `json:"used,omitempty"`
|
|
Model string `json:"model,omitempty"`
|
|
// Where to go and check. Collected from the tools that answered.
|
|
Sources []string `json:"sources,omitempty"`
|
|
// True when the loop stopped on its own limits rather than because the
|
|
// model finished. The reply is still returned — a partial answer beats a
|
|
// spinner — but it is flagged rather than passed off as complete.
|
|
Incomplete bool `json:"incomplete,omitempty"`
|
|
// Set when the assistant resolved a change and is waiting on the person.
|
|
//
|
|
// One card, never a list. A card proposing several actions hides the one
|
|
// they would have refused, so the loop returns the first and stops — the
|
|
// next change is asked for separately.
|
|
Awaiting *tools.Proposal `json:"awaiting,omitempty"`
|
|
}
|
|
|
|
// AssistantStep is one tool call, for the console to render.
|
|
type AssistantStep struct {
|
|
Tool string `json:"tool"`
|
|
// Refused calls are included on purpose. An answer that quietly dropped a
|
|
// refusal would look like the assistant chose not to look.
|
|
Outcome string `json:"outcome"`
|
|
Rows int `json:"rows,omitempty"`
|
|
Detail string `json:"detail,omitempty"`
|
|
Scope string `json:"scope,omitempty"`
|
|
}
|
|
|
|
// AssistantService answers a question.
|
|
type AssistantService interface {
|
|
Ask(ctx context.Context, agentName, question string, caller tools.Caller) (AssistantAnswer, error)
|
|
// Approve performs a change the person has agreed to.
|
|
//
|
|
// Takes no question and involves no model: the card names the action, and
|
|
// the registry re-validates it against the live database. The assistant is
|
|
// not in this call at all, which is the point of splitting it out.
|
|
Approve(ctx context.Context, agentName, card string, caller tools.Caller) (AssistantAnswer, error)
|
|
// Available reports whether typed questions work at all here.
|
|
Available() bool
|
|
// Unavailable says WHY not, or "" when it is available. A disabled composer
|
|
// with no reason is indistinguishable from a misspelled variable, which is
|
|
// how this stayed off without anybody being able to tell.
|
|
Unavailable() string
|
|
}
|
|
|
|
type assistantService struct {
|
|
registry *tools.Registry
|
|
chat utils.Chat
|
|
agents map[string]Agent
|
|
// Why the model is absent, from config. Carried rather than recomputed so
|
|
// the answer the endpoint gives is the one the server actually started with.
|
|
why string
|
|
// How fast one person may ask. Only questions are limited — approving a
|
|
// change the person has already read costs nothing and must not be the call
|
|
// that gets refused.
|
|
limit *askLimiter
|
|
}
|
|
|
|
// NewAssistantService takes its agents already loaded and validated.
|
|
//
|
|
// No fallback to a built-in set: a deployment whose agent files failed to load
|
|
// should refuse to start, not quietly run a different assistant than the one
|
|
// its configuration describes.
|
|
func NewAssistantService(registry *tools.Registry, chat utils.Chat, agents map[string]Agent) AssistantService {
|
|
return &assistantService{registry: registry, chat: chat, agents: agents, limit: newAskLimiter(nil)}
|
|
}
|
|
|
|
func (s *assistantService) Available() bool { return s.chat != nil }
|
|
|
|
func (s *assistantService) Unavailable() string {
|
|
if s.chat != nil {
|
|
return ""
|
|
}
|
|
if s.why != "" {
|
|
return s.why
|
|
}
|
|
return "no assistant model is configured"
|
|
}
|
|
|
|
// SetUnavailableReason records why there is no model, for the status endpoint.
|
|
func (s *assistantService) SetUnavailableReason(why string) { s.why = why }
|
|
|
|
func (s *assistantService) Approve(ctx context.Context, agentName, card string, caller tools.Caller) (AssistantAnswer, error) {
|
|
agent, known := s.agents[agentName]
|
|
if !known {
|
|
return AssistantAnswer{}, fmt.Errorf("no assistant called %q", agentName)
|
|
}
|
|
|
|
// No model is consulted. A deployment with no provider can still approve a
|
|
// card it issued earlier, which matters: the change is the person's
|
|
// decision, and it should not stop being possible because a provider is
|
|
// down.
|
|
result, err := s.registry.Approve(ctx, tools.Agent{Name: agent.Name, Tools: agent.Tools}, card, caller)
|
|
if err != nil {
|
|
return AssistantAnswer{}, err
|
|
}
|
|
|
|
answer := AssistantAnswer{
|
|
Reply: result.Note,
|
|
Used: []AssistantStep{{Tool: "approved", Outcome: tools.OutcomeOK, Rows: result.Count, Scope: result.Scope}},
|
|
}
|
|
if result.Source != "" {
|
|
answer.Sources = []string{result.Source}
|
|
}
|
|
return answer, nil
|
|
}
|
|
|
|
// maxQuestion bounds what a person can send.
|
|
//
|
|
// Not a safety control — it is a cost one. A pasted spreadsheet as a "question"
|
|
// is a large bill and a worse answer.
|
|
const maxQuestion = 4000
|
|
|
|
func (s *assistantService) Ask(ctx context.Context, agentName, question string, caller tools.Caller) (AssistantAnswer, error) {
|
|
question = strings.TrimSpace(question)
|
|
if question == "" {
|
|
return AssistantAnswer{}, fmt.Errorf("ask a question")
|
|
}
|
|
if len(question) > maxQuestion {
|
|
return AssistantAnswer{}, fmt.Errorf("that question is too long; keep it under %d characters", maxQuestion)
|
|
}
|
|
if s.chat == nil {
|
|
return AssistantAnswer{}, utils.ErrChatNotConfigured
|
|
}
|
|
// Checked after the cheap refusals and before the paid one. An empty or
|
|
// oversized question should be told what is wrong with it rather than
|
|
// spending a token from an allowance it never needed.
|
|
if err := s.limit.allow(caller.Userid); err != nil {
|
|
return AssistantAnswer{}, err
|
|
}
|
|
|
|
agent, known := s.agents[agentName]
|
|
if !known {
|
|
return AssistantAnswer{}, fmt.Errorf("no assistant called %q", agentName)
|
|
}
|
|
|
|
allow := tools.Agent{Name: agent.Name, Tools: agent.Tools}
|
|
messages := []utils.Message{
|
|
{Role: utils.RoleSystem, Content: agent.System},
|
|
{Role: utils.RoleUser, Content: question},
|
|
}
|
|
|
|
answer := AssistantAnswer{Model: s.chat.ModelFor(agent.Tier)}
|
|
calls := 0
|
|
seenSource := map[string]bool{}
|
|
|
|
for step := 0; step < agent.MaxSteps; step++ {
|
|
reply, err := s.chat.Complete(ctx, utils.ChatRequest{
|
|
Tier: agent.Tier,
|
|
Messages: messages,
|
|
Tools: s.registry.Definitions(allow),
|
|
})
|
|
if err != nil {
|
|
return AssistantAnswer{}, err
|
|
}
|
|
answer.Model = reply.Model
|
|
|
|
if len(reply.ToolCalls) == 0 {
|
|
answer.Reply = strings.TrimSpace(reply.Content)
|
|
// `length` means the provider cut the reply off mid-sentence. A
|
|
// truncated answer reads exactly like a complete one unless it is
|
|
// flagged here.
|
|
if reply.StopReason == "length" {
|
|
answer.Incomplete = true
|
|
}
|
|
return answer, nil
|
|
}
|
|
|
|
// The assistant turn has to go back verbatim, tool calls and all, or
|
|
// the model has no record of what it asked for and asks again.
|
|
messages = append(messages, utils.Message{
|
|
Role: utils.RoleAssistant,
|
|
Content: reply.Content,
|
|
ToolCalls: reply.ToolCalls,
|
|
})
|
|
|
|
for _, call := range reply.ToolCalls {
|
|
if calls >= agent.MaxToolCalls {
|
|
answer.Incomplete = true
|
|
messages = append(messages, utils.Message{
|
|
Role: utils.RoleTool,
|
|
ToolCallID: call.ID,
|
|
Name: call.Name,
|
|
Content: "Refused: this conversation has already run its maximum number of tool calls. Answer with what you have and say it is partial.",
|
|
})
|
|
continue
|
|
}
|
|
calls++
|
|
|
|
result, err := s.registry.Call(ctx, allow, call.Name, call.Arguments, caller)
|
|
step := AssistantStep{Tool: call.Name, Outcome: tools.OutcomeOK, Rows: result.Count, Scope: result.Scope}
|
|
if err != nil {
|
|
step.Outcome = tools.OutcomeRefused
|
|
step.Detail = err.Error()
|
|
}
|
|
answer.Used = append(answer.Used, step)
|
|
|
|
// A resolved write ends the turn. The model is not asked to carry
|
|
// on planning around a change that has not happened, and it is not
|
|
// given a second chance to propose something else in the same
|
|
// breath.
|
|
if proposal, ok := result.Rows.(tools.Proposal); ok && err == nil {
|
|
answer.Awaiting = &proposal
|
|
}
|
|
|
|
if result.Source != "" && !seenSource[result.Source] {
|
|
seenSource[result.Source] = true
|
|
answer.Sources = append(answer.Sources, result.Source)
|
|
}
|
|
|
|
messages = append(messages, utils.Message{
|
|
Role: utils.RoleTool,
|
|
ToolCallID: call.ID,
|
|
Name: call.Name,
|
|
Content: toolMessage(result, err),
|
|
})
|
|
}
|
|
}
|
|
|
|
// Out of steps with the model still working. Ask once for what it has
|
|
// rather than returning nothing: a partial answer beats a blank panel, and
|
|
// `Incomplete` is what stops it being passed off as the whole story.
|
|
answer.Incomplete = true
|
|
messages = append(messages, utils.Message{
|
|
Role: utils.RoleUser,
|
|
Content: "Answer now with what you already have, and say plainly that you ran out of steps before finishing.",
|
|
})
|
|
reply, err := s.chat.Complete(ctx, utils.ChatRequest{Tier: agent.Tier, Messages: messages})
|
|
if err != nil {
|
|
return answer, err
|
|
}
|
|
answer.Reply = strings.TrimSpace(reply.Content)
|
|
return answer, nil
|
|
}
|
|
|
|
// toolMessage is what the model is told a tool returned.
|
|
//
|
|
// A refusal is reported as text, not as a failure: a model told "that tool
|
|
// needs a tenant" can explain it to the person, where a model handed nothing
|
|
// says "something went wrong".
|
|
//
|
|
// The rows go back as JSON because that is what the model reads most reliably,
|
|
// and `note` rides alongside them rather than inside, so an instruction about
|
|
// truncation cannot be mistaken for data.
|
|
func toolMessage(result tools.Result, err error) string {
|
|
if err != nil {
|
|
return "Refused: " + err.Error()
|
|
}
|
|
|
|
payload := map[string]any{
|
|
"rows": result.Rows,
|
|
"count": result.Count,
|
|
}
|
|
if result.Scope != "" {
|
|
payload["covers"] = result.Scope
|
|
}
|
|
if result.Truncated {
|
|
payload["truncated"] = true
|
|
}
|
|
if result.Note != "" {
|
|
payload["note"] = result.Note
|
|
}
|
|
|
|
encoded, marshalErr := json.Marshal(payload)
|
|
if marshalErr != nil {
|
|
return fmt.Sprintf("Refused: the result could not be encoded: %v", marshalErr)
|
|
}
|
|
return string(encoded)
|
|
}
|
|
|
|
// assistantTimeout bounds one question end to end.
|
|
//
|
|
// Generous, because a deep question makes several round trips, and short enough
|
|
// that a wedged provider does not hold a console connection open all afternoon.
|
|
const assistantTimeout = 90 * time.Second
|
|
|
|
// WithTimeout is the bound the HTTP layer applies. Here rather than in the
|
|
// controller so every caller of Ask — HTTP today, MCP later — gets the same one.
|
|
func WithTimeout(ctx context.Context) (context.Context, context.CancelFunc) {
|
|
return context.WithTimeout(ctx, assistantTimeout)
|
|
}
|