agent build
This commit is contained in:
224
go-api/internal/tools/all_tools_test.go
Normal file
224
go-api/internal/tools/all_tools_test.go
Normal file
@@ -0,0 +1,224 @@
|
||||
package tools_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/krow/krow-backend/go-api/internal/authctx"
|
||||
"github.com/krow/krow-backend/go-api/internal/repo"
|
||||
"github.com/krow/krow-backend/go-api/internal/testutil"
|
||||
"github.com/krow/krow-backend/go-api/internal/tools"
|
||||
)
|
||||
|
||||
// everyTool is the shipped registry, built the same way the service builds it.
|
||||
//
|
||||
// Duplicated from runtime.DefaultTools deliberately: importing the runtime here
|
||||
// would make the tool package depend on its own caller. The list is asserted
|
||||
// against the registry's own count below, so the two cannot drift silently.
|
||||
func everyTool(db repo.Querier) []tools.Tool {
|
||||
return []tools.Tool{
|
||||
tools.ActivityBreakdown(db), tools.ActivitySignals(db),
|
||||
tools.WorkforceAttendance(db), tools.WorkforceOvertime(db),
|
||||
tools.WorkforceCoverage(db), tools.WorkforceTraining(db),
|
||||
tools.CandidatesQuality(db), tools.HiresRecent(db),
|
||||
tools.HiresPerformance(db), tools.PositionsRisk(db),
|
||||
tools.TalentPool(db), tools.WorkspaceSummary(db), tools.OperationsRisk(db),
|
||||
tools.OpenPositions(db), tools.AvailableWorkers(db), tools.AssignWorker(db),
|
||||
// Retrieval. Nil retriever here: the smoke test drives it with no
|
||||
// corpus, and "there are no documents to search" is the honest answer
|
||||
// for a deployment with no knowledge layer wired.
|
||||
tools.KnowledgeSearch(nil),
|
||||
tools.CandidatesAwaiting(db), tools.MoveApplication(db),
|
||||
}
|
||||
}
|
||||
|
||||
// smokeArgs are the arguments a tool needs before it will do anything.
|
||||
//
|
||||
// Most take none. The two that do are the ones that name a moment rather than a
|
||||
// window, and a required argument is not something to paper over with a default
|
||||
// — a lookup that silently assumed "now" would smoke-test a statement nobody
|
||||
// runs in production.
|
||||
var smokeArgs = map[string][]string{
|
||||
"available_workers": {`{"starts_at":"2030-01-01T09:00:00Z","ends_at":"2030-01-01T17:00:00Z"}`},
|
||||
// A role id and an email that do not exist. The statement still executes,
|
||||
// which is all this test checks; the call is refused on the row not being
|
||||
// found, which is the correct outcome for arguments this made up.
|
||||
"assign_worker": {`{"job_posting_id":"00000000-0000-0000-0000-0000000000ff",` +
|
||||
`"worker_email":"nobody@example.test","starts_at":"2030-01-01T09:00:00Z"}`},
|
||||
// Driven with no retriever wired, so it refuses. Exercised anyway: a tool
|
||||
// registered in the service and never called by any test is a tool whose
|
||||
// schema nobody has looked at.
|
||||
"knowledge_search": {`{"query":"lateness policy"}`},
|
||||
// An application id that does not exist. The statement still executes; the
|
||||
// call is refused on the row not being found, which is correct.
|
||||
"move_application": {`{"application_id":"00000000-0000-0000-0000-0000000000ff","stage":"interview"}`},
|
||||
}
|
||||
|
||||
// TestEveryToolRunsAgainstTheRealSchema is the smoke test that catches a
|
||||
// mistyped column or a status literal that is not in its enum.
|
||||
//
|
||||
// Both fail silently in SQL: a wrong enum member matches no rows and raises
|
||||
// nothing, so a bad guess reads as a confident zero. Only executing the
|
||||
// statement against the real schema finds it, which is why this runs every
|
||||
// tool rather than sampling.
|
||||
func TestEveryToolRunsAgainstTheRealSchema(t *testing.T) {
|
||||
h := testutil.New(t)
|
||||
admin := authctx.Identity{
|
||||
UserID: "00000000-0000-0000-0000-000000000001",
|
||||
OrgID: h.OrgID, Role: "admin", Email: "admin@example.test",
|
||||
}
|
||||
|
||||
reg := tools.NewRegistry()
|
||||
for _, tool := range everyTool(h.Pool) {
|
||||
reg.MustRegister(tool)
|
||||
}
|
||||
|
||||
// Every tool, with no arguments and with a period, so both the windowed
|
||||
// and unwindowed statements are executed.
|
||||
for _, name := range reg.Names() {
|
||||
argSets := []string{`{}`, `{"period":"last-30-days"}`, `{"period":"this-month","limit":3}`}
|
||||
if custom, ok := smokeArgs[name]; ok {
|
||||
argSets = custom
|
||||
}
|
||||
tool, _ := reg.Get(name)
|
||||
|
||||
for _, args := range argSets {
|
||||
res := reg.Dispatch(context.Background(),
|
||||
tools.Context{Principal: admin, RunID: "run_smoke"}, name, json.RawMessage(args))
|
||||
|
||||
// A write never reaches its handler here, because nothing has been
|
||||
// approved. What is being smoke-tested is its RENDERER — which runs
|
||||
// the same resolution queries the handler will, so a mistyped column
|
||||
// in either is caught. The one thing that must not happen is data.
|
||||
if tool.Effect == tools.EffectWrite {
|
||||
if res.Data != nil {
|
||||
t.Errorf("%s wrote without an approval", name)
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
// knowledge_search is wired with no retriever and no corpus here, so
|
||||
// refusing is the correct outcome. Asserted as a refusal rather than
|
||||
// skipped, because the failure worth catching is it answering.
|
||||
if name == "knowledge_search" {
|
||||
if res.Data != nil {
|
||||
t.Errorf("knowledge_search answered with no knowledge layer wired: %+v", res.Data)
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
if res.Error != nil {
|
||||
t.Errorf("%s with %s: %s — %s", name, args, res.Error.Code, res.Error.Message)
|
||||
continue
|
||||
}
|
||||
if res.Data == nil {
|
||||
t.Errorf("%s with %s: returned no data", name, args)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestEveryToolDeniesACallerWithNoTenant(t *testing.T) {
|
||||
// I5, across the whole surface. One tool that forgot would be a
|
||||
// cross-tenant read, so this asserts the property rather than the code.
|
||||
h := testutil.New(t)
|
||||
stranger := authctx.Identity{UserID: "u", Role: "admin", Email: "x@example.test"}
|
||||
|
||||
reg := tools.NewRegistry()
|
||||
for _, tool := range everyTool(h.Pool) {
|
||||
reg.MustRegister(tool)
|
||||
}
|
||||
|
||||
for _, name := range reg.Names() {
|
||||
res := reg.Dispatch(context.Background(),
|
||||
tools.Context{Principal: stranger}, name, json.RawMessage(`{}`))
|
||||
|
||||
// A cross-domain tool reports withheld areas rather than refusing
|
||||
// outright — it has nothing it may read, which is a different answer
|
||||
// from "you may not ask". Either is acceptable; returning data is not.
|
||||
if res.Error != nil {
|
||||
continue
|
||||
}
|
||||
encoded, _ := json.Marshal(res.Data)
|
||||
if !strings.Contains(string(encoded), "withheld") {
|
||||
t.Errorf("%s answered a caller with no tenant: %s", name, encoded)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestEveryToolDeniesAnUnknownRole(t *testing.T) {
|
||||
h := testutil.New(t)
|
||||
stranger := authctx.Identity{
|
||||
UserID: "u", OrgID: h.OrgID, Role: "superuser", Email: "x@example.test",
|
||||
}
|
||||
|
||||
reg := tools.NewRegistry()
|
||||
for _, tool := range everyTool(h.Pool) {
|
||||
reg.MustRegister(tool)
|
||||
}
|
||||
|
||||
for _, name := range reg.Names() {
|
||||
res := reg.Dispatch(context.Background(),
|
||||
tools.Context{Principal: stranger}, name, json.RawMessage(`{}`))
|
||||
if res.Error != nil {
|
||||
continue
|
||||
}
|
||||
encoded, _ := json.Marshal(res.Data)
|
||||
if !strings.Contains(string(encoded), "withheld") {
|
||||
t.Errorf("%s answered an unlisted role: %s", name, encoded)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestEveryToolIsDescribedAndEveryWriteCanExplainItself(t *testing.T) {
|
||||
// This test used to assert that nothing wrote, with a note saying it must
|
||||
// be changed deliberately when the first write landed. assign_worker is
|
||||
// that write, so here is the deliberate change — and the property worth
|
||||
// asserting now is not "no writes" but "every write can say what it does".
|
||||
//
|
||||
// Register() enforces the same thing at boot. Asserted again here because
|
||||
// this list is what a reviewer reads to see the shape of the tool surface,
|
||||
// and a write appearing in it with no renderer should fail loudly next to
|
||||
// its peers rather than only inside a constructor.
|
||||
writes := 0
|
||||
for _, tool := range everyTool(nil) {
|
||||
switch tool.Effect {
|
||||
case tools.EffectRead:
|
||||
if tool.Confirm != nil {
|
||||
t.Errorf("%s is a read with a confirmation renderer that will never run", tool.Name)
|
||||
}
|
||||
case tools.EffectWrite:
|
||||
writes++
|
||||
if tool.Confirm == nil {
|
||||
t.Errorf("%s writes but cannot describe what it would do", tool.Name)
|
||||
}
|
||||
default:
|
||||
t.Errorf("%s declares effect %q, want read or write", tool.Name, tool.Effect)
|
||||
}
|
||||
if len(tool.Description) < 60 {
|
||||
t.Errorf("%s has a %d-character description; the model reads this instead of docs",
|
||||
tool.Name, len(tool.Description))
|
||||
}
|
||||
if tool.InputSchema == nil {
|
||||
t.Errorf("%s has no input schema", tool.Name)
|
||||
}
|
||||
}
|
||||
if writes != 2 {
|
||||
t.Errorf("%d write tools; each one added is a new way for an agent to change the "+
|
||||
"world, so update this count deliberately", writes)
|
||||
}
|
||||
}
|
||||
|
||||
func TestToolCountMatchesTheShippedRegistry(t *testing.T) {
|
||||
// Guards the duplication in everyTool: a tool registered in the service
|
||||
// but missing here would never be smoke-tested.
|
||||
reg := tools.NewRegistry()
|
||||
for _, tool := range everyTool(nil) {
|
||||
reg.MustRegister(tool)
|
||||
}
|
||||
if got := len(reg.Names()); got != 19 {
|
||||
t.Errorf("the registry holds %d tools; update this test and runtime.DefaultTools together", got)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user