225 lines
8.4 KiB
Go
225 lines
8.4 KiB
Go
package tools_test
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"strings"
|
|
"testing"
|
|
|
|
"github.com/krow/krow-backend/go-api/internal/authctx"
|
|
"github.com/krow/krow-backend/go-api/internal/repo"
|
|
"github.com/krow/krow-backend/go-api/internal/testutil"
|
|
"github.com/krow/krow-backend/go-api/internal/tools"
|
|
)
|
|
|
|
// everyTool is the shipped registry, built the same way the service builds it.
|
|
//
|
|
// Duplicated from runtime.DefaultTools deliberately: importing the runtime here
|
|
// would make the tool package depend on its own caller. The list is asserted
|
|
// against the registry's own count below, so the two cannot drift silently.
|
|
func everyTool(db repo.Querier) []tools.Tool {
|
|
return []tools.Tool{
|
|
tools.ActivityBreakdown(db), tools.ActivitySignals(db),
|
|
tools.WorkforceAttendance(db), tools.WorkforceOvertime(db),
|
|
tools.WorkforceCoverage(db), tools.WorkforceTraining(db),
|
|
tools.CandidatesQuality(db), tools.HiresRecent(db),
|
|
tools.HiresPerformance(db), tools.PositionsRisk(db),
|
|
tools.TalentPool(db), tools.WorkspaceSummary(db), tools.OperationsRisk(db),
|
|
tools.OpenPositions(db), tools.AvailableWorkers(db), tools.AssignWorker(db),
|
|
// Retrieval. Nil retriever here: the smoke test drives it with no
|
|
// corpus, and "there are no documents to search" is the honest answer
|
|
// for a deployment with no knowledge layer wired.
|
|
tools.KnowledgeSearch(nil),
|
|
tools.CandidatesAwaiting(db), tools.MoveApplication(db),
|
|
}
|
|
}
|
|
|
|
// smokeArgs are the arguments a tool needs before it will do anything.
|
|
//
|
|
// Most take none. The two that do are the ones that name a moment rather than a
|
|
// window, and a required argument is not something to paper over with a default
|
|
// — a lookup that silently assumed "now" would smoke-test a statement nobody
|
|
// runs in production.
|
|
var smokeArgs = map[string][]string{
|
|
"available_workers": {`{"starts_at":"2030-01-01T09:00:00Z","ends_at":"2030-01-01T17:00:00Z"}`},
|
|
// A role id and an email that do not exist. The statement still executes,
|
|
// which is all this test checks; the call is refused on the row not being
|
|
// found, which is the correct outcome for arguments this made up.
|
|
"assign_worker": {`{"job_posting_id":"00000000-0000-0000-0000-0000000000ff",` +
|
|
`"worker_email":"nobody@example.test","starts_at":"2030-01-01T09:00:00Z"}`},
|
|
// Driven with no retriever wired, so it refuses. Exercised anyway: a tool
|
|
// registered in the service and never called by any test is a tool whose
|
|
// schema nobody has looked at.
|
|
"knowledge_search": {`{"query":"lateness policy"}`},
|
|
// An application id that does not exist. The statement still executes; the
|
|
// call is refused on the row not being found, which is correct.
|
|
"move_application": {`{"application_id":"00000000-0000-0000-0000-0000000000ff","stage":"interview"}`},
|
|
}
|
|
|
|
// TestEveryToolRunsAgainstTheRealSchema is the smoke test that catches a
|
|
// mistyped column or a status literal that is not in its enum.
|
|
//
|
|
// Both fail silently in SQL: a wrong enum member matches no rows and raises
|
|
// nothing, so a bad guess reads as a confident zero. Only executing the
|
|
// statement against the real schema finds it, which is why this runs every
|
|
// tool rather than sampling.
|
|
func TestEveryToolRunsAgainstTheRealSchema(t *testing.T) {
|
|
h := testutil.New(t)
|
|
admin := authctx.Identity{
|
|
UserID: "00000000-0000-0000-0000-000000000001",
|
|
OrgID: h.OrgID, Role: "admin", Email: "admin@example.test",
|
|
}
|
|
|
|
reg := tools.NewRegistry()
|
|
for _, tool := range everyTool(h.Pool) {
|
|
reg.MustRegister(tool)
|
|
}
|
|
|
|
// Every tool, with no arguments and with a period, so both the windowed
|
|
// and unwindowed statements are executed.
|
|
for _, name := range reg.Names() {
|
|
argSets := []string{`{}`, `{"period":"last-30-days"}`, `{"period":"this-month","limit":3}`}
|
|
if custom, ok := smokeArgs[name]; ok {
|
|
argSets = custom
|
|
}
|
|
tool, _ := reg.Get(name)
|
|
|
|
for _, args := range argSets {
|
|
res := reg.Dispatch(context.Background(),
|
|
tools.Context{Principal: admin, RunID: "run_smoke"}, name, json.RawMessage(args))
|
|
|
|
// A write never reaches its handler here, because nothing has been
|
|
// approved. What is being smoke-tested is its RENDERER — which runs
|
|
// the same resolution queries the handler will, so a mistyped column
|
|
// in either is caught. The one thing that must not happen is data.
|
|
if tool.Effect == tools.EffectWrite {
|
|
if res.Data != nil {
|
|
t.Errorf("%s wrote without an approval", name)
|
|
}
|
|
continue
|
|
}
|
|
|
|
// knowledge_search is wired with no retriever and no corpus here, so
|
|
// refusing is the correct outcome. Asserted as a refusal rather than
|
|
// skipped, because the failure worth catching is it answering.
|
|
if name == "knowledge_search" {
|
|
if res.Data != nil {
|
|
t.Errorf("knowledge_search answered with no knowledge layer wired: %+v", res.Data)
|
|
}
|
|
continue
|
|
}
|
|
|
|
if res.Error != nil {
|
|
t.Errorf("%s with %s: %s — %s", name, args, res.Error.Code, res.Error.Message)
|
|
continue
|
|
}
|
|
if res.Data == nil {
|
|
t.Errorf("%s with %s: returned no data", name, args)
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
func TestEveryToolDeniesACallerWithNoTenant(t *testing.T) {
|
|
// I5, across the whole surface. One tool that forgot would be a
|
|
// cross-tenant read, so this asserts the property rather than the code.
|
|
h := testutil.New(t)
|
|
stranger := authctx.Identity{UserID: "u", Role: "admin", Email: "x@example.test"}
|
|
|
|
reg := tools.NewRegistry()
|
|
for _, tool := range everyTool(h.Pool) {
|
|
reg.MustRegister(tool)
|
|
}
|
|
|
|
for _, name := range reg.Names() {
|
|
res := reg.Dispatch(context.Background(),
|
|
tools.Context{Principal: stranger}, name, json.RawMessage(`{}`))
|
|
|
|
// A cross-domain tool reports withheld areas rather than refusing
|
|
// outright — it has nothing it may read, which is a different answer
|
|
// from "you may not ask". Either is acceptable; returning data is not.
|
|
if res.Error != nil {
|
|
continue
|
|
}
|
|
encoded, _ := json.Marshal(res.Data)
|
|
if !strings.Contains(string(encoded), "withheld") {
|
|
t.Errorf("%s answered a caller with no tenant: %s", name, encoded)
|
|
}
|
|
}
|
|
}
|
|
|
|
func TestEveryToolDeniesAnUnknownRole(t *testing.T) {
|
|
h := testutil.New(t)
|
|
stranger := authctx.Identity{
|
|
UserID: "u", OrgID: h.OrgID, Role: "superuser", Email: "x@example.test",
|
|
}
|
|
|
|
reg := tools.NewRegistry()
|
|
for _, tool := range everyTool(h.Pool) {
|
|
reg.MustRegister(tool)
|
|
}
|
|
|
|
for _, name := range reg.Names() {
|
|
res := reg.Dispatch(context.Background(),
|
|
tools.Context{Principal: stranger}, name, json.RawMessage(`{}`))
|
|
if res.Error != nil {
|
|
continue
|
|
}
|
|
encoded, _ := json.Marshal(res.Data)
|
|
if !strings.Contains(string(encoded), "withheld") {
|
|
t.Errorf("%s answered an unlisted role: %s", name, encoded)
|
|
}
|
|
}
|
|
}
|
|
|
|
func TestEveryToolIsDescribedAndEveryWriteCanExplainItself(t *testing.T) {
|
|
// This test used to assert that nothing wrote, with a note saying it must
|
|
// be changed deliberately when the first write landed. assign_worker is
|
|
// that write, so here is the deliberate change — and the property worth
|
|
// asserting now is not "no writes" but "every write can say what it does".
|
|
//
|
|
// Register() enforces the same thing at boot. Asserted again here because
|
|
// this list is what a reviewer reads to see the shape of the tool surface,
|
|
// and a write appearing in it with no renderer should fail loudly next to
|
|
// its peers rather than only inside a constructor.
|
|
writes := 0
|
|
for _, tool := range everyTool(nil) {
|
|
switch tool.Effect {
|
|
case tools.EffectRead:
|
|
if tool.Confirm != nil {
|
|
t.Errorf("%s is a read with a confirmation renderer that will never run", tool.Name)
|
|
}
|
|
case tools.EffectWrite:
|
|
writes++
|
|
if tool.Confirm == nil {
|
|
t.Errorf("%s writes but cannot describe what it would do", tool.Name)
|
|
}
|
|
default:
|
|
t.Errorf("%s declares effect %q, want read or write", tool.Name, tool.Effect)
|
|
}
|
|
if len(tool.Description) < 60 {
|
|
t.Errorf("%s has a %d-character description; the model reads this instead of docs",
|
|
tool.Name, len(tool.Description))
|
|
}
|
|
if tool.InputSchema == nil {
|
|
t.Errorf("%s has no input schema", tool.Name)
|
|
}
|
|
}
|
|
if writes != 2 {
|
|
t.Errorf("%d write tools; each one added is a new way for an agent to change the "+
|
|
"world, so update this count deliberately", writes)
|
|
}
|
|
}
|
|
|
|
func TestToolCountMatchesTheShippedRegistry(t *testing.T) {
|
|
// Guards the duplication in everyTool: a tool registered in the service
|
|
// but missing here would never be smoke-tested.
|
|
reg := tools.NewRegistry()
|
|
for _, tool := range everyTool(nil) {
|
|
reg.MustRegister(tool)
|
|
}
|
|
if got := len(reg.Names()); got != 19 {
|
|
t.Errorf("the registry holds %d tools; update this test and runtime.DefaultTools together", got)
|
|
}
|
|
}
|