package services import ( "context" "encoding/json" "fmt" "strings" "time" "nearle/services/tools" "nearle/utils" ) // Nearle Buddy's loop. // // One loop runs every agent. An agent is a name, a tier, a system prompt and an // allow-list — data, not a class — so a sixth agent is a config entry rather // than a subclass, and the behaviour they all share cannot drift between them. // // The shape is the ordinary one: ask the model, run any tools it asked for, // give it the results, ask again, stop when it answers in words. What matters // is what the loop refuses to let the model decide. // // ── What the model does not get to choose ─────────────────────────────────── // // - Whose data it reads. The caller comes from the verified session and is // passed to the registry directly. No tool accepts a tenant argument. // - Which tools exist. The allow-list is the agent's, enforced by the // registry; a model asking for something else is refused, not obeyed. // - When to stop. Steps and tool calls are counted here. A model that keeps // calling tools is stopped by arithmetic, not by being asked nicely. // // ── A refused tool is a message, not an error ─────────────────────────────── // // When the registry refuses a call, the refusal goes back to the model as the // tool's result. A model told "that tool needs a tenant" can explain the // problem to the person; a model handed a 500 says "something went wrong", // which is true and useless. The refusal is still audited either way. // Agent is one assistant, as data. type Agent struct { Name string Tier string // System is what the model is told about its job. Rules that MUST hold do // not live here — a prompt is a request. This is for tone, scope and the // habits that make an answer useful. System string Tools []string // MaxSteps bounds the conversation: one step is one model round trip. MaxSteps int // MaxToolCalls bounds the work across the whole conversation, because a // model can ask for several tools in a single step. MaxToolCalls int } // AssistantAnswer is what one question produced. type AssistantAnswer struct { Reply string `json:"reply"` // What was actually run, in order. Returned to the console so an answer can // show its working — Buddy states a conclusion, and this is how a person // sees which numbers it came from. Used []AssistantStep `json:"used,omitempty"` Model string `json:"model,omitempty"` // Where to go and check. Collected from the tools that answered. Sources []string `json:"sources,omitempty"` // True when the loop stopped on its own limits rather than because the // model finished. The reply is still returned — a partial answer beats a // spinner — but it is flagged rather than passed off as complete. Incomplete bool `json:"incomplete,omitempty"` // Set when the assistant resolved a change and is waiting on the person. // // One card, never a list. A card proposing several actions hides the one // they would have refused, so the loop returns the first and stops — the // next change is asked for separately. Awaiting *tools.Proposal `json:"awaiting,omitempty"` } // AssistantStep is one tool call, for the console to render. type AssistantStep struct { Tool string `json:"tool"` // Refused calls are included on purpose. An answer that quietly dropped a // refusal would look like the assistant chose not to look. Outcome string `json:"outcome"` Rows int `json:"rows,omitempty"` Detail string `json:"detail,omitempty"` Scope string `json:"scope,omitempty"` } // AssistantService answers a question. type AssistantService interface { Ask(ctx context.Context, agentName, question string, caller tools.Caller) (AssistantAnswer, error) // Approve performs a change the person has agreed to. // // Takes no question and involves no model: the card names the action, and // the registry re-validates it against the live database. The assistant is // not in this call at all, which is the point of splitting it out. Approve(ctx context.Context, agentName, card string, caller tools.Caller) (AssistantAnswer, error) // Available reports whether typed questions work at all here. Available() bool } type assistantService struct { registry *tools.Registry chat utils.Chat agents map[string]Agent // How fast one person may ask. Only questions are limited — approving a // change the person has already read costs nothing and must not be the call // that gets refused. limit *askLimiter } // NewAssistantService takes its agents already loaded and validated. // // No fallback to a built-in set: a deployment whose agent files failed to load // should refuse to start, not quietly run a different assistant than the one // its configuration describes. func NewAssistantService(registry *tools.Registry, chat utils.Chat, agents map[string]Agent) AssistantService { return &assistantService{registry: registry, chat: chat, agents: agents, limit: newAskLimiter(nil)} } func (s *assistantService) Available() bool { return s.chat != nil } func (s *assistantService) Approve(ctx context.Context, agentName, card string, caller tools.Caller) (AssistantAnswer, error) { agent, known := s.agents[agentName] if !known { return AssistantAnswer{}, fmt.Errorf("no assistant called %q", agentName) } // No model is consulted. A deployment with no provider can still approve a // card it issued earlier, which matters: the change is the person's // decision, and it should not stop being possible because a provider is // down. result, err := s.registry.Approve(ctx, tools.Agent{Name: agent.Name, Tools: agent.Tools}, card, caller) if err != nil { return AssistantAnswer{}, err } answer := AssistantAnswer{ Reply: result.Note, Used: []AssistantStep{{Tool: "approved", Outcome: tools.OutcomeOK, Rows: result.Count, Scope: result.Scope}}, } if result.Source != "" { answer.Sources = []string{result.Source} } return answer, nil } // maxQuestion bounds what a person can send. // // Not a safety control — it is a cost one. A pasted spreadsheet as a "question" // is a large bill and a worse answer. const maxQuestion = 4000 func (s *assistantService) Ask(ctx context.Context, agentName, question string, caller tools.Caller) (AssistantAnswer, error) { question = strings.TrimSpace(question) if question == "" { return AssistantAnswer{}, fmt.Errorf("ask a question") } if len(question) > maxQuestion { return AssistantAnswer{}, fmt.Errorf("that question is too long; keep it under %d characters", maxQuestion) } if s.chat == nil { return AssistantAnswer{}, utils.ErrChatNotConfigured } // Checked after the cheap refusals and before the paid one. An empty or // oversized question should be told what is wrong with it rather than // spending a token from an allowance it never needed. if err := s.limit.allow(caller.Userid); err != nil { return AssistantAnswer{}, err } agent, known := s.agents[agentName] if !known { return AssistantAnswer{}, fmt.Errorf("no assistant called %q", agentName) } allow := tools.Agent{Name: agent.Name, Tools: agent.Tools} messages := []utils.Message{ {Role: utils.RoleSystem, Content: agent.System}, {Role: utils.RoleUser, Content: question}, } answer := AssistantAnswer{Model: s.chat.ModelFor(agent.Tier)} calls := 0 seenSource := map[string]bool{} for step := 0; step < agent.MaxSteps; step++ { reply, err := s.chat.Complete(ctx, utils.ChatRequest{ Tier: agent.Tier, Messages: messages, Tools: s.registry.Definitions(allow), }) if err != nil { return AssistantAnswer{}, err } answer.Model = reply.Model if len(reply.ToolCalls) == 0 { answer.Reply = strings.TrimSpace(reply.Content) // `length` means the provider cut the reply off mid-sentence. A // truncated answer reads exactly like a complete one unless it is // flagged here. if reply.StopReason == "length" { answer.Incomplete = true } return answer, nil } // The assistant turn has to go back verbatim, tool calls and all, or // the model has no record of what it asked for and asks again. messages = append(messages, utils.Message{ Role: utils.RoleAssistant, Content: reply.Content, ToolCalls: reply.ToolCalls, }) for _, call := range reply.ToolCalls { if calls >= agent.MaxToolCalls { answer.Incomplete = true messages = append(messages, utils.Message{ Role: utils.RoleTool, ToolCallID: call.ID, Name: call.Name, Content: "Refused: this conversation has already run its maximum number of tool calls. Answer with what you have and say it is partial.", }) continue } calls++ result, err := s.registry.Call(ctx, allow, call.Name, call.Arguments, caller) step := AssistantStep{Tool: call.Name, Outcome: tools.OutcomeOK, Rows: result.Count, Scope: result.Scope} if err != nil { step.Outcome = tools.OutcomeRefused step.Detail = err.Error() } answer.Used = append(answer.Used, step) // A resolved write ends the turn. The model is not asked to carry // on planning around a change that has not happened, and it is not // given a second chance to propose something else in the same // breath. if proposal, ok := result.Rows.(tools.Proposal); ok && err == nil { answer.Awaiting = &proposal } if result.Source != "" && !seenSource[result.Source] { seenSource[result.Source] = true answer.Sources = append(answer.Sources, result.Source) } messages = append(messages, utils.Message{ Role: utils.RoleTool, ToolCallID: call.ID, Name: call.Name, Content: toolMessage(result, err), }) } } // Out of steps with the model still working. Ask once for what it has // rather than returning nothing: a partial answer beats a blank panel, and // `Incomplete` is what stops it being passed off as the whole story. answer.Incomplete = true messages = append(messages, utils.Message{ Role: utils.RoleUser, Content: "Answer now with what you already have, and say plainly that you ran out of steps before finishing.", }) reply, err := s.chat.Complete(ctx, utils.ChatRequest{Tier: agent.Tier, Messages: messages}) if err != nil { return answer, err } answer.Reply = strings.TrimSpace(reply.Content) return answer, nil } // toolMessage is what the model is told a tool returned. // // A refusal is reported as text, not as a failure: a model told "that tool // needs a tenant" can explain it to the person, where a model handed nothing // says "something went wrong". // // The rows go back as JSON because that is what the model reads most reliably, // and `note` rides alongside them rather than inside, so an instruction about // truncation cannot be mistaken for data. func toolMessage(result tools.Result, err error) string { if err != nil { return "Refused: " + err.Error() } payload := map[string]any{ "rows": result.Rows, "count": result.Count, } if result.Scope != "" { payload["covers"] = result.Scope } if result.Truncated { payload["truncated"] = true } if result.Note != "" { payload["note"] = result.Note } encoded, marshalErr := json.Marshal(payload) if marshalErr != nil { return fmt.Sprintf("Refused: the result could not be encoded: %v", marshalErr) } return string(encoded) } // assistantTimeout bounds one question end to end. // // Generous, because a deep question makes several round trips, and short enough // that a wedged provider does not hold a console connection open all afternoon. const assistantTimeout = 90 * time.Second // WithTimeout is the bound the HTTP layer applies. Here rather than in the // controller so every caller of Ask — HTTP today, MCP later — gets the same one. func WithTimeout(ctx context.Context) (context.Context, context.CancelFunc) { return context.WithTimeout(ctx, assistantTimeout) }