Files
backend_fiesta/services/assistantService.go
2026-09-24 11:01:16 +05:30

322 lines
12 KiB
Go

package services
import (
"context"
"encoding/json"
"fmt"
"strings"
"time"
"nearle/services/tools"
"nearle/utils"
)
// Nearle Buddy's loop.
//
// One loop runs every agent. An agent is a name, a tier, a system prompt and an
// allow-list — data, not a class — so a sixth agent is a config entry rather
// than a subclass, and the behaviour they all share cannot drift between them.
//
// The shape is the ordinary one: ask the model, run any tools it asked for,
// give it the results, ask again, stop when it answers in words. What matters
// is what the loop refuses to let the model decide.
//
// ── What the model does not get to choose ───────────────────────────────────
//
// - Whose data it reads. The caller comes from the verified session and is
// passed to the registry directly. No tool accepts a tenant argument.
// - Which tools exist. The allow-list is the agent's, enforced by the
// registry; a model asking for something else is refused, not obeyed.
// - When to stop. Steps and tool calls are counted here. A model that keeps
// calling tools is stopped by arithmetic, not by being asked nicely.
//
// ── A refused tool is a message, not an error ───────────────────────────────
//
// When the registry refuses a call, the refusal goes back to the model as the
// tool's result. A model told "that tool needs a tenant" can explain the
// problem to the person; a model handed a 500 says "something went wrong",
// which is true and useless. The refusal is still audited either way.
// Agent is one assistant, as data.
type Agent struct {
Name string
Tier string
// System is what the model is told about its job. Rules that MUST hold do
// not live here — a prompt is a request. This is for tone, scope and the
// habits that make an answer useful.
System string
Tools []string
// MaxSteps bounds the conversation: one step is one model round trip.
MaxSteps int
// MaxToolCalls bounds the work across the whole conversation, because a
// model can ask for several tools in a single step.
MaxToolCalls int
}
// AssistantAnswer is what one question produced.
type AssistantAnswer struct {
Reply string `json:"reply"`
// What was actually run, in order. Returned to the console so an answer can
// show its working — Buddy states a conclusion, and this is how a person
// sees which numbers it came from.
Used []AssistantStep `json:"used,omitempty"`
Model string `json:"model,omitempty"`
// Where to go and check. Collected from the tools that answered.
Sources []string `json:"sources,omitempty"`
// True when the loop stopped on its own limits rather than because the
// model finished. The reply is still returned — a partial answer beats a
// spinner — but it is flagged rather than passed off as complete.
Incomplete bool `json:"incomplete,omitempty"`
// Set when the assistant resolved a change and is waiting on the person.
//
// One card, never a list. A card proposing several actions hides the one
// they would have refused, so the loop returns the first and stops — the
// next change is asked for separately.
Awaiting *tools.Proposal `json:"awaiting,omitempty"`
}
// AssistantStep is one tool call, for the console to render.
type AssistantStep struct {
Tool string `json:"tool"`
// Refused calls are included on purpose. An answer that quietly dropped a
// refusal would look like the assistant chose not to look.
Outcome string `json:"outcome"`
Rows int `json:"rows,omitempty"`
Detail string `json:"detail,omitempty"`
Scope string `json:"scope,omitempty"`
}
// AssistantService answers a question.
type AssistantService interface {
Ask(ctx context.Context, agentName, question string, caller tools.Caller) (AssistantAnswer, error)
// Approve performs a change the person has agreed to.
//
// Takes no question and involves no model: the card names the action, and
// the registry re-validates it against the live database. The assistant is
// not in this call at all, which is the point of splitting it out.
Approve(ctx context.Context, agentName, card string, caller tools.Caller) (AssistantAnswer, error)
// Available reports whether typed questions work at all here.
Available() bool
}
type assistantService struct {
registry *tools.Registry
chat utils.Chat
agents map[string]Agent
// How fast one person may ask. Only questions are limited — approving a
// change the person has already read costs nothing and must not be the call
// that gets refused.
limit *askLimiter
}
// NewAssistantService takes its agents already loaded and validated.
//
// No fallback to a built-in set: a deployment whose agent files failed to load
// should refuse to start, not quietly run a different assistant than the one
// its configuration describes.
func NewAssistantService(registry *tools.Registry, chat utils.Chat, agents map[string]Agent) AssistantService {
return &assistantService{registry: registry, chat: chat, agents: agents, limit: newAskLimiter(nil)}
}
func (s *assistantService) Available() bool { return s.chat != nil }
func (s *assistantService) Approve(ctx context.Context, agentName, card string, caller tools.Caller) (AssistantAnswer, error) {
agent, known := s.agents[agentName]
if !known {
return AssistantAnswer{}, fmt.Errorf("no assistant called %q", agentName)
}
// No model is consulted. A deployment with no provider can still approve a
// card it issued earlier, which matters: the change is the person's
// decision, and it should not stop being possible because a provider is
// down.
result, err := s.registry.Approve(ctx, tools.Agent{Name: agent.Name, Tools: agent.Tools}, card, caller)
if err != nil {
return AssistantAnswer{}, err
}
answer := AssistantAnswer{
Reply: result.Note,
Used: []AssistantStep{{Tool: "approved", Outcome: tools.OutcomeOK, Rows: result.Count, Scope: result.Scope}},
}
if result.Source != "" {
answer.Sources = []string{result.Source}
}
return answer, nil
}
// maxQuestion bounds what a person can send.
//
// Not a safety control — it is a cost one. A pasted spreadsheet as a "question"
// is a large bill and a worse answer.
const maxQuestion = 4000
func (s *assistantService) Ask(ctx context.Context, agentName, question string, caller tools.Caller) (AssistantAnswer, error) {
question = strings.TrimSpace(question)
if question == "" {
return AssistantAnswer{}, fmt.Errorf("ask a question")
}
if len(question) > maxQuestion {
return AssistantAnswer{}, fmt.Errorf("that question is too long; keep it under %d characters", maxQuestion)
}
if s.chat == nil {
return AssistantAnswer{}, utils.ErrChatNotConfigured
}
// Checked after the cheap refusals and before the paid one. An empty or
// oversized question should be told what is wrong with it rather than
// spending a token from an allowance it never needed.
if err := s.limit.allow(caller.Userid); err != nil {
return AssistantAnswer{}, err
}
agent, known := s.agents[agentName]
if !known {
return AssistantAnswer{}, fmt.Errorf("no assistant called %q", agentName)
}
allow := tools.Agent{Name: agent.Name, Tools: agent.Tools}
messages := []utils.Message{
{Role: utils.RoleSystem, Content: agent.System},
{Role: utils.RoleUser, Content: question},
}
answer := AssistantAnswer{Model: s.chat.ModelFor(agent.Tier)}
calls := 0
seenSource := map[string]bool{}
for step := 0; step < agent.MaxSteps; step++ {
reply, err := s.chat.Complete(ctx, utils.ChatRequest{
Tier: agent.Tier,
Messages: messages,
Tools: s.registry.Definitions(allow),
})
if err != nil {
return AssistantAnswer{}, err
}
answer.Model = reply.Model
if len(reply.ToolCalls) == 0 {
answer.Reply = strings.TrimSpace(reply.Content)
// `length` means the provider cut the reply off mid-sentence. A
// truncated answer reads exactly like a complete one unless it is
// flagged here.
if reply.StopReason == "length" {
answer.Incomplete = true
}
return answer, nil
}
// The assistant turn has to go back verbatim, tool calls and all, or
// the model has no record of what it asked for and asks again.
messages = append(messages, utils.Message{
Role: utils.RoleAssistant,
Content: reply.Content,
ToolCalls: reply.ToolCalls,
})
for _, call := range reply.ToolCalls {
if calls >= agent.MaxToolCalls {
answer.Incomplete = true
messages = append(messages, utils.Message{
Role: utils.RoleTool,
ToolCallID: call.ID,
Name: call.Name,
Content: "Refused: this conversation has already run its maximum number of tool calls. Answer with what you have and say it is partial.",
})
continue
}
calls++
result, err := s.registry.Call(ctx, allow, call.Name, call.Arguments, caller)
step := AssistantStep{Tool: call.Name, Outcome: tools.OutcomeOK, Rows: result.Count, Scope: result.Scope}
if err != nil {
step.Outcome = tools.OutcomeRefused
step.Detail = err.Error()
}
answer.Used = append(answer.Used, step)
// A resolved write ends the turn. The model is not asked to carry
// on planning around a change that has not happened, and it is not
// given a second chance to propose something else in the same
// breath.
if proposal, ok := result.Rows.(tools.Proposal); ok && err == nil {
answer.Awaiting = &proposal
}
if result.Source != "" && !seenSource[result.Source] {
seenSource[result.Source] = true
answer.Sources = append(answer.Sources, result.Source)
}
messages = append(messages, utils.Message{
Role: utils.RoleTool,
ToolCallID: call.ID,
Name: call.Name,
Content: toolMessage(result, err),
})
}
}
// Out of steps with the model still working. Ask once for what it has
// rather than returning nothing: a partial answer beats a blank panel, and
// `Incomplete` is what stops it being passed off as the whole story.
answer.Incomplete = true
messages = append(messages, utils.Message{
Role: utils.RoleUser,
Content: "Answer now with what you already have, and say plainly that you ran out of steps before finishing.",
})
reply, err := s.chat.Complete(ctx, utils.ChatRequest{Tier: agent.Tier, Messages: messages})
if err != nil {
return answer, err
}
answer.Reply = strings.TrimSpace(reply.Content)
return answer, nil
}
// toolMessage is what the model is told a tool returned.
//
// A refusal is reported as text, not as a failure: a model told "that tool
// needs a tenant" can explain it to the person, where a model handed nothing
// says "something went wrong".
//
// The rows go back as JSON because that is what the model reads most reliably,
// and `note` rides alongside them rather than inside, so an instruction about
// truncation cannot be mistaken for data.
func toolMessage(result tools.Result, err error) string {
if err != nil {
return "Refused: " + err.Error()
}
payload := map[string]any{
"rows": result.Rows,
"count": result.Count,
}
if result.Scope != "" {
payload["covers"] = result.Scope
}
if result.Truncated {
payload["truncated"] = true
}
if result.Note != "" {
payload["note"] = result.Note
}
encoded, marshalErr := json.Marshal(payload)
if marshalErr != nil {
return fmt.Sprintf("Refused: the result could not be encoded: %v", marshalErr)
}
return string(encoded)
}
// assistantTimeout bounds one question end to end.
//
// Generous, because a deep question makes several round trips, and short enough
// that a wedged provider does not hold a console connection open all afternoon.
const assistantTimeout = 90 * time.Second
// WithTimeout is the bound the HTTP layer applies. Here rather than in the
// controller so every caller of Ask — HTTP today, MCP later — gets the same one.
func WithTimeout(ctx context.Context) (context.Context, context.CancelFunc) {
return context.WithTimeout(ctx, assistantTimeout)
}