update agents skill design

This commit is contained in:
2026-08-20 18:18:10 +05:30
parent 161b237695
commit b2e6868824
75 changed files with 14586 additions and 98 deletions

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,53 @@
/**
* Writes the Owliver behaviour baseline.
*
* node scripts/owliver-baseline.mjs # report drift, write nothing
* node scripts/owliver-baseline.mjs --write # (re)generate the snapshot
*
* The snapshot is the existing product's behaviour, recorded before the Agent
* layer was built and asserted by `skill-check.mjs` on every run afterwards.
*
* Regenerating is a deliberate act with a reason attached. A baseline rewritten
* to make a red check go green records the regression instead of catching it,
* which is worse than having no baseline at all — it converts a caught bug into
* a documented one.
*/
import { readFileSync, writeFileSync, existsSync, mkdirSync } from 'node:fs';
import { dirname } from 'node:path';
import { createServer } from 'vite';
import { BASELINE_PATH, captureBaseline } from './owliver-capture.mjs';
const ROOT = process.cwd();
const server = await createServer({
root: ROOT,
server: { middlewareMode: true },
appType: 'custom',
logLevel: 'error',
});
const captured = await captureBaseline(server);
await server.close();
const serialized = `${JSON.stringify(captured, null, 2)}\n`;
if (process.argv.includes('--write')) {
mkdirSync(dirname(BASELINE_PATH), { recursive: true });
writeFileSync(BASELINE_PATH, serialized);
const pages = Object.keys(captured.contexts).length;
console.log(`Baseline written — ${pages} contexts, ${captured.skillIds.length} skills, ${Object.keys(captured.routes).length} routes.`);
process.exit(0);
}
if (!existsSync(BASELINE_PATH)) {
console.error('No baseline recorded. Run `node scripts/owliver-baseline.mjs --write`.');
process.exit(1);
}
if (readFileSync(BASELINE_PATH, 'utf8') === serialized) {
console.log('Owliver behaviour matches the baseline.');
process.exit(0);
}
console.error('Owliver behaviour has DRIFTED from the baseline. Run `npm test` for the per-page detail.');
process.exit(1);

158
scripts/owliver-capture.mjs Normal file
View File

@@ -0,0 +1,158 @@
/**
* Today's Owliver behaviour, captured as data.
*
* The Agent layer is additive: opening a Krow page must produce exactly the
* Owliver it produces now. That is a claim about eleven page contexts, their
* skills, their suggestions, their prompts and their intent routing — far more
* than anyone can hold in their head while editing, and all of it silent when
* it breaks. So it is recorded here as a value, committed, and compared on
* every run.
*
* One module, two callers: `owliver-baseline.mjs` writes the snapshot and
* `skill-check.mjs` asserts against it. A second implementation of "what we
* measure" would be a second definition of the contract.
*
* Everything below is pinned. `TODAY` is fixed rather than `new Date()`,
* because `buildFacts` windows records by date and a baseline that drifts
* daily is not a baseline. Lists are sorted; nothing depends on registry
* ordering or on the clock.
*/
import { join } from 'node:path';
/** Where the committed snapshot lives. Named here rather than in the CLI so a
* reader can import the path without also running a Vite server. */
export const BASELINE_PATH = join(process.cwd(), 'scripts/__baseline__/owliver-baseline.json');
/** Fixed so the snapshot means the same thing tomorrow. */
export const TODAY = new Date('2026-08-20T09:00:00.000Z');
/**
* The contexts under contract.
*
* The eight Krow pages the brief protects, plus the three further Admin
* contexts the placement table carries — Create Position, Candidates Analysis
* and Profile. They are page-scoped by exactly the same mechanism, so leaving
* them out would protect most of the boundary and quietly not the rest.
*/
export const CONTEXTS = [
'admin.controlCenter',
'admin.positions',
'admin.createPosition',
'admin.candidatesList',
'admin.candidates',
'admin.hiredHistory',
'admin.talentPool',
'admin.forge',
'admin.analytics',
'admin.activity',
'admin.profile',
];
/**
* What gets asked, per context.
*
* Three kinds on purpose, because they exercise three different branches of
* `resolveIntent` and a change to any one of them is a regression a reader
* would notice: a question the page owns, a question another page owns (which
* must route, not answer), and a question nothing owns (which must decline).
*/
export const QUESTIONS = [
'What needs my attention?',
'Summarize this page',
'Show hiring activity',
'Which positions need attention?',
'Where are candidates dropping off?',
'What are my permissions?',
'What is the weather today?',
];
const sorted = (list) => [...list].sort();
/** Block types only — structure is the contract, wording is not. */
const docShape = (document) =>
Array.isArray(document?.blocks) ? document.blocks.map((b) => b?.type ?? null) : null;
/**
* Loads the real module graph and reads today's behaviour off it.
*
* `ssrLoadModule` rather than a mock, for the reason `skill-check.mjs` gives:
* the pipeline is what is under test, and a mock of it would reproduce none of
* the failures this exists to catch.
*/
export async function captureBaseline(server) {
const registry = await server.ssrLoadModule('/src/lib/skills/registry.js');
const placement = await server.ssrLoadModule('/src/components/ai-assistant/placement.js');
const resolver = await server.ssrLoadModule('/src/lib/skills/owliverResolver.js');
const dynamic = await server.ssrLoadModule('/src/components/ai-assistant/dynamic.js');
const routing = await server.ssrLoadModule('/src/components/ai-assistant/routing.js');
const insights = await server.ssrLoadModule('/src/components/ai-assistant/insights.js');
const seed = await server.ssrLoadModule('/src/api/seed.js');
/* The fact sheet the panel builds, from the shipped seed at a fixed instant. */
const facts = insights.buildFacts({
applications: seed.seedData.JobApplication,
postings: seed.seedData.JobPosting,
interviews: seed.seedData.AIInterview,
staff: seed.seedData.Staff,
profiles: seed.seedData.WorkerProfile,
activity: seed.seedData.UserActivity,
courses: seed.seedData.Course,
profile: null,
user: seed.DEMO_USER,
today: TODAY,
});
const contexts = {};
for (const contextId of CONTEXTS) {
const pageKey = registry.pageKeyForContext(contextId);
contexts[contextId] = {
pageKey,
route: pageKey ? registry.routeForPageKey(pageKey) : null,
/* The page boundary itself: which skills this page carries with no
agent, no account customization and nothing disabled. */
skills: sorted(registry.skillsForContext(contextId, [], []).map((s) => s.id)),
suggestions: sorted(
resolver.owliverSuggestions(contextId, [], [], {}).map((s) => s.label)
),
prompts: (dynamic.buildPrompts(contextId, facts, null) || []).map((p) => p.label),
intents: QUESTIONS.map((question) => {
const matched = registry.matchSkill(question, contextId, [], []);
const intent = routing.resolveIntent({ question, contextId });
return {
question,
matchedSkill: matched?.id ?? null,
kind: intent?.kind ?? null,
skill: intent?.skill?.id ?? null,
capability: intent?.capability ?? null,
action: intent?.action?.name ?? null,
destination: intent?.destination?.contextId ?? null,
doc: docShape(intent?.doc),
};
}),
};
}
/* Route → context, for every placement the product declares. An agent must
never become an input to this. */
const routes = {};
for (const route of Object.keys(placement.PLACEMENT_ROUTES).sort()) {
routes[route] = placement.resolveAssistantContext('admin', route)?.id ?? null;
}
return {
/* Bumped only when the shape of what we measure changes — never to make a
failing comparison pass. */
schema: 1,
today: TODAY.toISOString(),
skillIds: sorted(registry.SKILLS.map((s) => s.id)),
diagnostics: registry.readSkillRegistry([]).diagnostics,
routes,
contexts,
};
}

File diff suppressed because it is too large Load Diff