update owliver agent
This commit is contained in:
181
src/lib/agents/capabilityTest.js
Normal file
181
src/lib/agents/capabilityTest.js
Normal file
@@ -0,0 +1,181 @@
|
||||
import { PLACEMENT_ROUTES } from '@/components/ai-assistant/placement';
|
||||
import { ASSISTANT_CONTEXTS } from '@/components/ai-assistant/contexts';
|
||||
import { allSkills, matchSkill, pageKeyForContext, skillsForContext } from '@/lib/skills/registry';
|
||||
import { canonicalPage, surfaceFor } from '@/lib/skills/surfaces';
|
||||
import { toolsForContext } from '@/lib/skills/tools';
|
||||
import { agentCovers, agentScopedDisabled } from './runtime';
|
||||
|
||||
/**
|
||||
* Trying a capability before committing to it.
|
||||
*
|
||||
* Two things have to be true for this to be worth having, and both are about
|
||||
* refusing to fake something:
|
||||
*
|
||||
* 1. **The evaluation is the live runtime.** `agentCovers`, `skillsForContext`,
|
||||
* `agentScopedDisabled`, `matchSkill` and `toolsForContext` are the same
|
||||
* functions the assistant routes every real question through. Nothing here
|
||||
* re-implements any of them, so a green result cannot mean something
|
||||
* different from what will happen on the page.
|
||||
*
|
||||
* 2. **Testing is not saving.** The agent this reasons about is *hypothetical*
|
||||
* — the draft's fields with the candidate capability added — and it is
|
||||
* never written anywhere. A reader can try six capabilities and leave with
|
||||
* the agent exactly as they found it.
|
||||
*
|
||||
* The other half of the test is the answer itself, and that is deliberately not
|
||||
* here: `scopeFor` returns what `send({ scope })` needs so the question runs in
|
||||
* the *existing* Owliver panel against the target page. One conversation, one
|
||||
* pipeline, no second chat.
|
||||
*/
|
||||
|
||||
/** contextId → the page key it stands for, resolved once. */
|
||||
const CONTEXTS = Object.values(PLACEMENT_ROUTES).map((contextId) => ({
|
||||
contextId,
|
||||
pageKey: pageKeyForContext(contextId),
|
||||
}));
|
||||
|
||||
/** The assistant context for a surface, or null if the page carries no panel. */
|
||||
export function contextForPage(pageKey) {
|
||||
const wanted = canonicalPage(pageKey) || pageKey;
|
||||
const hit = CONTEXTS.find((c) => c.pageKey && (canonicalPage(c.pageKey) || c.pageKey) === wanted);
|
||||
return hit?.contextId || null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The surfaces a capability can actually be tried on, for this agent.
|
||||
*
|
||||
* The intersection of what the agent covers and what the capability declares —
|
||||
* anywhere else the runtime would decline, and offering it as a test target
|
||||
* would be offering a test guaranteed to fail for a reason that is not about
|
||||
* the capability.
|
||||
*/
|
||||
export function testTargets(agentPages = [], skillPages = []) {
|
||||
const skill = new Set((skillPages || []).map((p) => canonicalPage(p) || p));
|
||||
return (agentPages || [])
|
||||
.map((p) => canonicalPage(p) || p)
|
||||
.filter((p) => skill.has(p))
|
||||
.map((pageKey) => ({
|
||||
pageKey,
|
||||
contextId: contextForPage(pageKey),
|
||||
label: surfaceFor(pageKey)?.label || pageKey,
|
||||
}))
|
||||
.filter((t) => t.contextId);
|
||||
}
|
||||
|
||||
/**
|
||||
* The agent as it *would* be with this capability attached.
|
||||
*
|
||||
* Published on purpose: an unpublished draft is refused by the runtime for a
|
||||
* reason that has nothing to do with the capability being tried, and a test
|
||||
* that always says "this agent is not published" answers the wrong question.
|
||||
*/
|
||||
function hypotheticalAgent(fields, skillId) {
|
||||
const skills = skillId && !fields.skills.includes(skillId)
|
||||
? [...fields.skills, skillId]
|
||||
: fields.skills;
|
||||
|
||||
return {
|
||||
id: fields.id || 'draft',
|
||||
name: fields.name || 'This agent',
|
||||
pages: fields.pages,
|
||||
skills,
|
||||
subagents: fields.subagents || [],
|
||||
knowledge: fields.knowledge || [],
|
||||
starters: fields.starters || [],
|
||||
reasoning: fields.reasoning,
|
||||
webSearch: fields.webSearch,
|
||||
status: 'published',
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* What the runtime would do with this question, on this page, with this
|
||||
* capability added.
|
||||
*
|
||||
* `matched` is the honest headline: it is the skill that would actually answer.
|
||||
* When that is the capability under test, the test proves the capability;
|
||||
* when it is a different one, it says so rather than claiming a pass — two
|
||||
* definitions claiming the same phrase is a real condition and the reader
|
||||
* should see it here rather than discover it in production.
|
||||
*/
|
||||
export function evaluateCapability({
|
||||
fields, skillId, contextId, question, customSkills = [], entry = null,
|
||||
}) {
|
||||
const agent = hypotheticalAgent(fields, skillId);
|
||||
const registry = allSkills(customSkills);
|
||||
const disabledSkills = agentScopedDisabled(agent, registry, []);
|
||||
|
||||
const covers = agentCovers(agent, contextId);
|
||||
const reachable = covers ? skillsForContext(contextId, disabledSkills, customSkills) : [];
|
||||
const offered = reachable.some((s) => s.id === skillId);
|
||||
|
||||
/**
|
||||
* There are two ways a question reaches a skill, and conflating them made
|
||||
* this panel lie about its own suggestions.
|
||||
*
|
||||
* A **typed** question is routed by `matchSkill` against declared triggers. A
|
||||
* **suggestion** is not routed at all: it is a chip, it carries the capability
|
||||
* it asks for, and `send` short-circuits intent resolution for exactly that
|
||||
* reason (`capability ? { kind: 'answer' }`). So "What kinds of event are
|
||||
* there?" — a suggestion Activity Analysis declares — matches none of its
|
||||
* triggers and is still answered by it every time.
|
||||
*
|
||||
* Reporting that as "no skill claims this wording" was true of the matcher and
|
||||
* false of the product. Both paths are modelled here, and the one that applies
|
||||
* is named, so the reader is told *how* it would be answered rather than being
|
||||
* shown a warning about a question that works.
|
||||
*/
|
||||
const declared = (entry?.suggestions || []).find(
|
||||
(sug) => (sug.prompt || sug.label) === question
|
||||
);
|
||||
|
||||
const matched = covers && question && !declared
|
||||
? matchSkill(question, contextId, disabledSkills, customSkills)
|
||||
: null;
|
||||
|
||||
return {
|
||||
agent,
|
||||
disabledSkills,
|
||||
covers,
|
||||
/** Is the capability under test offered at all on this page? */
|
||||
reachable: offered,
|
||||
reachableCount: reachable.length,
|
||||
/** The skill that would answer a *typed* question, if any. */
|
||||
matched,
|
||||
/** True when the capability being tested is the one that answers. */
|
||||
claims: Boolean(offered && (declared || (matched && matched.id === skillId))),
|
||||
/** How it would be answered: its own suggestion, or a matched trigger. */
|
||||
route: declared ? 'suggestion' : matched ? 'trigger' : null,
|
||||
/** The capability a suggestion asks for — passed to `send` as a chip would. */
|
||||
capability: declared?.capability || null,
|
||||
tools: covers ? toolsForContext(contextId, disabledSkills, customSkills) : [],
|
||||
pageLabel: ASSISTANT_CONTEXTS[contextId]?.page || contextId,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* What `send({ scope })` needs to run this turn against the target page.
|
||||
*
|
||||
* Deliberately the same `disabledSkills` the evaluation used, so the answer the
|
||||
* reader reads is produced under exactly the conditions the diagnostics above
|
||||
* described.
|
||||
*/
|
||||
export const scopeFor = (evaluation, contextId) => ({
|
||||
contextId,
|
||||
disabledSkills: evaluation.disabledSkills,
|
||||
agent: evaluation.agent,
|
||||
});
|
||||
|
||||
/**
|
||||
* Questions worth trying, taken from the capability itself.
|
||||
*
|
||||
* A definition's `owliver.suggestions` are the questions its author wrote it to
|
||||
* answer, so they are the fairest test of it — and they keep the test from
|
||||
* being a blank box the reader has to guess at. A definition with none falls
|
||||
* back to its own name, which is what its default trigger matches.
|
||||
*/
|
||||
export function testQuestions(entry) {
|
||||
const declared = (entry?.suggestions || []).map((s) => s.prompt || s.label).filter(Boolean);
|
||||
if (declared.length) return declared.slice(0, 3);
|
||||
return entry?.name ? [entry.name] : [];
|
||||
}
|
||||
Reference in New Issue
Block a user