Files
doormilxpress_astryx/tests/integration/agentStudio.test.jsx

495 lines
23 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import React from 'react';
import { render, screen, fireEvent, waitFor, within } from '@testing-library/react';
import { QueryClient, QueryClientProvider } from '@tanstack/react-query';
/**
* Settings → Skills & Tools, reading the backend agent registry.
*
* The real page, hooks and adapters run; only the HTTP functions are mocked,
* with responses shaped exactly like /admin/ai/* (see doormile_backend
* internal/ai/registry). What this proves: the page shows what the registry
* says, sends the writes the registry expects, and refuses in the UI what the
* server would refuse. What it does not: that the server accepts them — the
* backend's own route and Postgres tests cover that side.
*/
jest.mock('lucide-react', () =>
new Proxy({}, { get: (_t, prop) => (prop === '__esModule' ? true : (props) => <span data-testid={`icon-${String(prop)}`} {...props} />) })
);
jest.mock('@/api/doormile', () => ({
getAiAgents: jest.fn(),
getAiSkills: jest.fn(),
getAiTools: jest.fn(),
updateAiSkill: jest.fn(),
createAiSkill: jest.fn(),
updateAiAgent: jest.fn(),
getAiInsights: jest.fn(),
getAiStatus: jest.fn(),
getAiDecisions: jest.fn(),
runAiPlayground: jest.fn(),
}));
jest.mock('@/api/doormile/notify', () => ({
OpenToast: jest.fn(),
messageOf: (err, fallback = 'Something went wrong') => err?.response?.data?.message || err?.message || fallback,
}));
let mockUser = { role: 'admin' };
jest.mock('@/lib/AuthContext', () => ({ useAuth: () => ({ user: mockUser }) }));
import * as api from '@/api/doormile';
import AgentStudio from '@/pages/doormile/settings/agentStudio/AgentStudio';
import { playgroundErrorMessage } from '@/pages/doormile/settings/agentStudio/AgentPlayground';
import {
buildStudioModel,
canEditRegistry,
schemaToParameters,
toUiAgent,
} from '@/pages/doormile/settings/agentStudio/registryAdapters';
// ── Registry fixtures (the backend's wire shape) ────────────────────────────
const AGENTS = [
{
agentid: 'EXCEPTION_AGENT', name: 'Exception', runtime: 'engine', classref: 'AI_engine/agents/exception_agent.py:106',
purpose: 'Detects stalled riders.', wakeon: 'TRACKING miler.stalled', status: 'live', llmdecision: 'decide_stall_response',
hasautonomygate: true, autonomous: false, model: '', skillcount: 1, toolcount: 2,
},
{
agentid: 'HUB_AGENT', name: 'Hub', runtime: 'engine', classref: 'AI_engine/agents/hub_agent.py:35',
purpose: 'Hub capacity over 8 fictional hubs.', wakeon: 'Direct task', status: 'simulation', llmdecision: '',
hasautonomygate: false, autonomous: false, model: '', skillcount: 0, toolcount: 0,
},
{
agentid: 'CONSOLE_OPS_AGENT', name: 'Console Ops Agent', runtime: 'console', classref: 'branch',
purpose: 'Ops briefing.', wakeon: 'Console page load', status: 'unmerged', llmdecision: '',
hasautonomygate: false, autonomous: false, model: '', skillcount: 1, toolcount: 1,
},
];
const SKILLS = [
{
skillid: 'stall_response', agentid: 'EXCEPTION_AGENT', title: 'Stalled-rider response', category: 'rider_operations',
description: 'Detect a stalled rider and respond.', sampleprompt: '', source: 'engine', enabled: true, version: 1,
tools: ['reassign_booking', 'nearby_milers'],
thresholds: { stallMinutes: 10 },
thresholdsschema: [{ key: 'stallMinutes', label: 'Stall threshold', unit: 'min', default: 10, min: 5, max: 60, step: 5 }],
},
{
skillid: 'skill_doorstep_stall', agentid: 'CONSOLE_OPS_AGENT', title: 'Doorstep Stall Rescuer', category: 'rider_operations',
description: 'Flags riders stalled at a doorstep.', sampleprompt: '', source: 'console', enabled: true, version: 1,
tools: ['lookup_order'], thresholds: {}, thresholdsschema: [],
},
];
const TOOLS = [
{
toolname: 'reassign_booking', description: 'Reassign a booking whose rider has stalled.', kind: 'write',
target: 'doormile_backend POST /internal/bookings/:id/reassign', implementedat: 'exception_agent.py:466', requiresconfirmation: true,
inputschema: { type: 'object', properties: { booking_id: { type: 'integer', description: 'Booking to reassign' } }, required: ['booking_id'] },
},
{
toolname: 'nearby_milers', description: 'Find riders near a point.', kind: 'read', target: 'Redis GEO', implementedat: 'dispatch_agent.py',
requiresconfirmation: false, inputschema: { type: 'object', properties: {} },
},
{
toolname: 'lookup_order', description: 'Find an order.', kind: 'read', target: 'GET /admin/bookings', implementedat: 'tools.js',
requiresconfirmation: false, inputschema: { type: 'object', properties: { identifier: { type: 'string' } }, required: ['identifier'] },
},
];
// /admin/ai/insights and /admin/ai/decisions, in their wire shape.
const INSIGHTS = {
days: 7,
since: '2026-09-22T19:00:00Z',
receiving: true,
runs: {
total: 42,
failed: 3,
peragent: [{ agentid: 'EXCEPTION_AGENT', runs: 42, failed: 3, avgdurationms: 412, lastrunat: '2026-09-29T18:40:00Z' }],
},
decisions: {
total: 12,
bytype: [{ decisiontype: 'miler_assignment', total: 12, outcomes: { success: 9, pending: 3 } }],
},
live: [{ agentid: 'EXCEPTION_AGENT', status: 'idle', currenttask: null, taskscompleted: 40, tasksfailed: 3, lastseenat: '2026-09-29T18:59:00Z' }],
};
const DECISIONS = [
{ id: 88, decisiontype: 'miler_assignment', bookingid: 501, decision: { miler_id: 8 }, reasoning: 'Nearest idle rider, 1.2 km', outcome: 'success', createdat: '2026-09-29T18:30:00Z' },
];
const renderStudio = () => {
const qc = new QueryClient({ defaultOptions: { queries: { retry: false } } });
return render(
<QueryClientProvider client={qc}>
<AgentStudio />
</QueryClientProvider>
);
};
// Pick an agent from the navigator's switcher.
const switchTo = async (name) => {
const header = screen.getAllByRole('button').find((b) => within(b).queryByText('AI_ENGINE AGENT') || within(b).queryByText('CONSOLE AGENT'));
fireEvent.click(header);
const menu = screen.getByText('Switch Agent Scope').parentElement;
fireEvent.click(within(menu).getByText(name).closest('button'));
};
beforeEach(() => {
localStorage.clear();
mockUser = { role: 'admin' };
jest.clearAllMocks();
api.getAiAgents.mockResolvedValue(AGENTS);
api.getAiSkills.mockResolvedValue(SKILLS);
api.getAiTools.mockResolvedValue(TOOLS);
api.updateAiSkill.mockResolvedValue({ success: true });
api.createAiSkill.mockResolvedValue({ success: true });
api.updateAiAgent.mockResolvedValue({ success: true });
api.getAiInsights.mockResolvedValue(INSIGHTS);
api.getAiDecisions.mockResolvedValue(DECISIONS);
api.getAiStatus.mockResolvedValue({ engine: { readingsettings: false, lastreadat: null }, playground: { configured: false } });
});
// ── Adapters ────────────────────────────────────────────────────────────────
describe('registry adapters', () => {
it('maps an agent and gives every status a readable badge', () => {
const a = toUiAgent(AGENTS[1]);
expect(a).toMatchObject({ id: 'HUB_AGENT', runtime: 'engine', hasAutonomyGate: false });
expect(a.statusMeta.label).toBe('Simulation');
expect(toUiAgent({ ...AGENTS[1], status: 'something_new' }).statusMeta.label).toBe('Something_new');
});
it('turns a JSON-schema into parameter rows, keeping required', () => {
expect(schemaToParameters(TOOLS[0].inputschema)).toEqual([
{ name: 'booking_id', type: 'integer', description: 'Booking to reassign', required: true },
]);
expect(schemaToParameters(undefined)).toEqual([]);
});
it('names the skills that use each tool', () => {
const { tools, skills } = buildStudioModel(AGENTS, SKILLS, TOOLS);
expect(tools.find((t) => t.name === 'reassign_booking').parentSkill).toBe('Stalled-rider response');
expect(skills.find((s) => s.id === 'stall_response').integration).toBe('Exception');
});
it('lets only the admin role edit', () => {
expect(canEditRegistry({ role: 'admin' })).toBe(true);
expect(canEditRegistry({ role: 'Admin' })).toBe(true);
expect(canEditRegistry({ role: 'manager' })).toBe(false);
expect(canEditRegistry(null)).toBe(false);
});
});
// ── The page ────────────────────────────────────────────────────────────────
describe('Agent Studio page', () => {
it('says "All connected" only when the engine reads the registry and the Test tab has a model', async () => {
api.getAiStatus.mockResolvedValue({
engine: { readingsettings: true, lastreadat: new Date().toISOString(), telemetry: true, liveagents: 4 },
playground: { configured: true, model: 'openai/gpt-oss-120b' },
});
renderStudio();
const note = await screen.findByText('All connected.');
const banner = note.closest('[role="note"]');
expect(banner).toHaveTextContent(/AI_engine is following these settings/);
expect(banner).toHaveTextContent(/The Test tab is live \(model openai\/gpt-oss-120b\)/);
expect(banner).not.toHaveTextContent(/Half wired/);
});
it('says what is missing when the engine has not read the registry', async () => {
api.getAiStatus.mockResolvedValue({
engine: { readingsettings: false, lastreadat: null, telemetry: false, liveagents: 0 },
playground: { configured: false },
});
renderStudio();
const banner = (await screen.findByText('Partly connected.')).closest('[role="note"]');
expect(banner).toHaveTextContent(/AI_engine has not read these settings yet/);
expect(banner).toHaveTextContent(/environment defaults/);
expect(banner).toHaveTextContent(/Test tab is off/);
expect(banner).toHaveTextContent(/saved to the Doormile agent registry/);
});
it('says the backend needs deploying when it has no status endpoint', async () => {
api.getAiStatus.mockRejectedValue({ response: { status: 404 } });
renderStudio();
expect(await screen.findByText(/does not report agent status yet/)).toBeInTheDocument();
});
it("shows only the selected agent's skills", async () => {
renderStudio();
expect(await screen.findAllByText('Stalled-rider response')).not.toHaveLength(0);
expect(screen.queryAllByText('Doorstep Stall Rescuer')).toHaveLength(0);
await switchTo('Console Ops Agent');
expect(await screen.findAllByText('Doorstep Stall Rescuer')).not.toHaveLength(0);
expect(screen.queryAllByText('Stalled-rider response')).toHaveLength(0);
});
it('marks a simulated agent as a simulation in the switcher', async () => {
renderStudio();
await screen.findAllByText('Stalled-rider response');
const header = screen.getAllByRole('button').find((b) => within(b).queryByText('AI_ENGINE AGENT'));
fireEvent.click(header);
const hubRow = within(screen.getByText('Switch Agent Scope').parentElement).getByText('Hub').closest('button');
expect(within(hubRow).getByText('Simulation')).toBeInTheDocument();
});
it('sends a skill toggle to the registry', async () => {
renderStudio();
fireEvent.click(await screen.findByText('Details'));
fireEvent.click(screen.getByRole('switch', { name: 'Stalled-rider response enabled' }));
await waitFor(() => expect(api.updateAiSkill).toHaveBeenCalledWith('stall_response', { enabled: false }));
});
it('saves only the thresholds that changed, and blocks an out-of-range value', async () => {
renderStudio();
fireEvent.click(await screen.findByText('Details'));
const input = screen.getByRole('spinbutton', { name: 'Stall threshold' });
const save = screen.getByRole('button', { name: 'Save thresholds' });
fireEvent.change(input, { target: { value: '99' } });
expect(screen.getByText('Between 5 and 60')).toBeInTheDocument();
expect(save).toBeDisabled();
fireEvent.change(input, { target: { value: '12' } });
expect(screen.getByText('In steps of 5')).toBeInTheDocument();
expect(save).toBeDisabled();
fireEvent.change(input, { target: { value: '15' } });
expect(save).toBeEnabled();
fireEvent.click(save);
await waitFor(() => expect(api.updateAiSkill).toHaveBeenCalledWith('stall_response', { thresholds: { stallMinutes: 15 } }));
});
it('shows the server’s refusal in the drawer', async () => {
api.updateAiSkill.mockRejectedValue({ response: { data: { message: 'stallMinutes must be between 5 and 60' } } });
renderStudio();
fireEvent.click(await screen.findByText('Details'));
fireEvent.change(screen.getByRole('spinbutton', { name: 'Stall threshold' }), { target: { value: '20' } });
fireEvent.click(screen.getByRole('button', { name: 'Save thresholds' }));
expect(await screen.findByRole('alert')).toHaveTextContent('stallMinutes must be between 5 and 60');
});
it('registers a new skill on a console agent with the registry payload', async () => {
renderStudio();
await screen.findAllByText('Stalled-rider response');
await switchTo('Console Ops Agent');
fireEvent.click(await screen.findByText(/^New skill$/i));
fireEvent.change(screen.getByLabelText('Skill Title *'), { target: { value: 'Night shift watch' } });
fireEvent.change(screen.getByLabelText('Description *'), { target: { value: 'Watch the night shift.' } });
const dialog = screen.getByRole('dialog');
fireEvent.click(within(dialog).getByText('lookup_order').closest('button'));
fireEvent.click(within(dialog).getByRole('button', { name: 'Add Skill' }));
await waitFor(() =>
expect(api.createAiSkill).toHaveBeenCalledWith({
agentid: 'CONSOLE_OPS_AGENT',
title: 'Night shift watch',
category: 'logistics',
description: 'Watch the night shift.',
sampleprompt: '',
tools: ['lookup_order'],
})
);
await waitFor(() => expect(screen.queryByRole('dialog')).not.toBeInTheDocument());
});
it('does not offer a new skill on an AI_engine agent', async () => {
renderStudio();
await screen.findAllByText('Stalled-rider response');
expect(screen.getByText(/^New skill$/i).closest('button')).toBeDisabled();
});
it('needs the agent id typed before switching autonomy on', async () => {
renderStudio();
await screen.findAllByText('Stalled-rider response');
fireEvent.click(screen.getByText('Configure'));
fireEvent.click(screen.getByRole('switch', { name: 'Autonomous' }));
const save = screen.getByRole('button', { name: /Save changes/ });
expect(save).toBeDisabled();
const confirm = screen.getByText(/to confirm/).closest('label').querySelector('input');
fireEvent.change(confirm, { target: { value: 'exception_agent' } });
expect(save).toBeDisabled();
fireEvent.change(confirm, { target: { value: 'EXCEPTION_AGENT' } });
expect(save).toBeEnabled();
fireEvent.click(save);
await waitFor(() =>
expect(api.updateAiAgent).toHaveBeenCalledWith('EXCEPTION_AGENT', { autonomous: true, confirm: 'EXCEPTION_AGENT' })
);
});
it('offers no autonomy switch on an agent without a gate', async () => {
renderStudio();
await screen.findAllByText('Stalled-rider response');
await switchTo('Hub');
fireEvent.click(screen.getByText('Configure'));
expect(screen.queryByRole('switch', { name: 'Autonomous' })).not.toBeInTheDocument();
expect(screen.getByText(/has no autonomy switch/)).toBeInTheDocument();
});
it('is read-only for a manager', async () => {
mockUser = { role: 'manager' };
renderStudio();
await screen.findAllByText('Stalled-rider response');
expect(screen.getByText('You can view the registry. Only an admin can change it.')).toBeInTheDocument();
expect(screen.getByText(/^New skill$/i).closest('button')).toBeDisabled();
fireEvent.click(screen.getByText('Details'));
expect(screen.getByRole('switch', { name: 'Stalled-rider response enabled' })).toBeDisabled();
expect(screen.getByRole('spinbutton', { name: 'Stall threshold' })).toBeDisabled();
expect(screen.queryByRole('button', { name: 'Save thresholds' })).not.toBeInTheDocument();
});
it('labels read-only tools honestly, not as autonomous', async () => {
renderStudio();
await screen.findAllByText('Stalled-rider response');
fireEvent.click(screen.getAllByText('Tools')[0]);
// Tools render as cards (a header button per tool): kind + safety badges.
// The name also appears in the agent navigator; the tool card is the button carrying its kind badge.
const toolCard = (name) =>
screen.getAllByText(name).map((el) => el.closest('button')).find((btn) => btn && within(btn).queryByText(/^(Read|Write|Notify|General)$/));
const card = toolCard('nearby_milers');
expect(within(card).getByText('Read')).toBeInTheDocument();
expect(within(card).getByText('Open')).toBeInTheDocument();
const gated = toolCard('reassign_booking');
expect(within(gated).getByText('Write')).toBeInTheDocument();
expect(within(gated).getByText('Gated')).toBeInTheDocument();
expect(screen.queryByText('Autonomous')).not.toBeInTheDocument();
});
it('shows real run, failure and decision figures on Insights', async () => {
renderStudio();
await screen.findAllByText('Stalled-rider response');
fireEvent.click(screen.getByText('Insights'));
const runs = (await screen.findByText('Agent runs')).parentElement;
expect(runs).toHaveTextContent('42');
expect(screen.getByText('7% of runs').parentElement).toHaveTextContent('Failed3');
expect(screen.getByText('Agents reporting').parentElement).toHaveTextContent('1 / 2');
const row = screen.getByRole('table', { name: 'Runs per AI_engine agent' });
const exception = within(row).getByText('Exception').closest('tr');
expect(exception).toHaveTextContent('idle');
expect(exception).toHaveTextContent('412 ms');
// The silent agent is shown as silent with zero runs, not left out.
expect(within(row).getByText('Hub').closest('tr')).toHaveTextContent('silent');
expect(screen.getByText('success · 9')).toBeInTheDocument();
expect(await screen.findByText('Nearest idle rider, 1.2 km')).toBeInTheDocument();
expect(screen.queryByText('99.4%')).not.toBeInTheDocument();
});
it('asks the insights endpoint for the chosen window', async () => {
renderStudio();
await screen.findAllByText('Stalled-rider response');
fireEvent.click(screen.getByText('Insights'));
await screen.findByText('Agent runs');
expect(api.getAiInsights).toHaveBeenLastCalledWith(7);
fireEvent.click(screen.getByRole('button', { name: '30 days' }));
await waitFor(() => expect(api.getAiInsights).toHaveBeenLastCalledWith(30));
});
it('says plainly when no telemetry is being received', async () => {
api.getAiInsights.mockResolvedValue({ ...INSIGHTS, receiving: false, runs: { total: 0, failed: 0, peragent: [] }, live: [] });
renderStudio();
await screen.findAllByText('Stalled-rider response');
fireEvent.click(screen.getByText('Insights'));
expect(await screen.findByText(/Not receiving agent telemetry/)).toBeInTheDocument();
expect(screen.getByText('No runs can be recorded until telemetry is received.')).toBeInTheDocument();
});
it('says why when the registry cannot be read', async () => {
api.getAiAgents.mockRejectedValue({ response: { data: { message: 'available to Doormile staff only' } } });
renderStudio();
const alert = await screen.findByRole('alert');
expect(alert).toHaveTextContent('Could not load the agent registry');
expect(alert).toHaveTextContent('available to Doormile staff only');
});
});
// ── Test tab (Phase 6): a real backend run, not a simulation ───────────────
const TRACE = {
agentid: 'EXCEPTION_AGENT', skillid: '', model: 'claude-opus-5-5', turns: 2, inputtokens: 1200, outputtokens: 300,
ms: 4200, stopreason: 'end_turn', final: 'I would reassign it.',
steps: [
{ kind: 'tool', tool: 'nearby_milers', toolkind: 'read', input: { lat: 11, lon: 77 }, outcome: 'executed', result: { count: 2 }, ms: 12 },
{ kind: 'tool', tool: 'reassign_booking', toolkind: 'write', input: { booking_id: 5 }, outcome: 'proposed', result: { executed: false }, ms: 0 },
{ kind: 'text', text: 'I would reassign it.', ms: 900 },
],
};
describe('Agent Studio Test tab', () => {
const openTest = async () => {
renderStudio();
await screen.findAllByText('Stalled-rider response');
// The tab row renders before the skill cards, whose buttons also say Test.
fireEvent.click(screen.getAllByText('Test')[0]);
};
it('runs the prompt on the backend and shows its real trace', async () => {
api.runAiPlayground.mockResolvedValue(TRACE);
await openTest();
fireEvent.change(screen.getByLabelText('Playground prompt'), { target: { value: 'booking 5 is stuck' } });
fireEvent.click(screen.getByText('Run'));
await screen.findByText('I would reassign it.');
expect(api.runAiPlayground).toHaveBeenCalledWith(
{ agentid: 'EXCEPTION_AGENT', skillid: undefined, prompt: 'booking 5 is stuck' },
expect.anything()
);
const tools = screen.getAllByTestId('trace-tool');
expect(tools[0]).toHaveTextContent('nearby_milers()');
expect(tools[0]).toHaveTextContent('executed');
expect(tools[1]).toHaveTextContent('reassign_booking()');
expect(tools[1]).toHaveTextContent('proposal only — not executed');
expect(screen.getByTestId('trace-meta')).toHaveTextContent('2 turns · 1200 in / 300 out tokens · 4.2 s');
expect(screen.getByText('claude-opus-5-5')).toBeInTheDocument();
});
it('sends the chosen skill', async () => {
api.runAiPlayground.mockResolvedValue(TRACE);
await openTest();
fireEvent.change(screen.getByLabelText('Skill:'), { target: { value: 'stall_response' } });
fireEvent.change(screen.getByLabelText('Playground prompt'), { target: { value: 'x' } });
fireEvent.click(screen.getByText('Run'));
await waitFor(() => expect(api.runAiPlayground).toHaveBeenCalled());
expect(api.runAiPlayground.mock.calls[0][0].skillid).toBe('stall_response');
});
it('says plainly when the server has no Claude client', async () => {
api.runAiPlayground.mockRejectedValue({ response: { status: 503, data: { code: 'PLAYGROUND_NOT_CONFIGURED' } } });
await openTest();
fireEvent.change(screen.getByLabelText('Playground prompt'), { target: { value: 'hi' } });
fireEvent.click(screen.getByText('Run'));
const alert = await screen.findByRole('alert');
expect(alert).toHaveTextContent('not switched on for this server yet');
expect(alert).toHaveTextContent('Nothing was run');
});
it('does not let a non-admin run it', async () => {
mockUser = { role: 'manager' };
await openTest();
expect(screen.getByLabelText('Playground prompt')).toBeDisabled();
expect(screen.getByText('Run').closest('button')).toBeDisabled();
expect(screen.getByText('Only an admin can run the Test playground.')).toBeInTheDocument();
expect(api.runAiPlayground).not.toHaveBeenCalled();
});
it('maps server errors to operator messages', () => {
expect(playgroundErrorMessage({ response: { status: 429, data: { message: 'Playground limit reached: 10 runs per 10 minutes.' } } }))
.toMatch(/10 runs per 10 minutes/);
expect(playgroundErrorMessage({ response: { status: 403, data: {} } })).toMatch(/Only an admin/);
expect(playgroundErrorMessage({ response: { status: 502, data: { message: 'The Claude request failed; nothing was changed.' } } }))
.toMatch(/nothing was changed/);
expect(playgroundErrorMessage(new Error('Network Error'))).toMatch(/Nothing was changed/);
});
});