Files
krow_talent_app/src/components/ai-assistant/provider.js
2026-08-28 11:02:02 +05:30

450 lines
17 KiB
JavaScript

/**
* Provider seam for the Owliver dashboard panel.
*
* The UI never generates an answer and never knows where one came from. It calls
* `provider.stream(request)` and renders the snapshots. That single boundary is
* what makes the backend swappable: today a local provider computes answers from
* dashboard data, and later an HTTP provider can call a real service without a
* change to the component.
*
* ── Contract ────────────────────────────────────────────────────────────────
*
* provider.id: string
* provider.stream(request): AsyncIterable<Block[]>
*
* Each yielded value is the response *so far* as an array of blocks (see
* blocks.js). Snapshots rather than deltas because a response is structured:
* a table or a KPI row has no meaningful half-state, and the renderer stays a
* pure function of the latest snapshot.
*
* request = {
* contextId: string, // e.g. 'employer.overview'
* capability: string | null, // capability id, or null for free text
* question: string, // what the user typed, or the chip's prompt
* facts: object, // dashboard fact sheet (see insights.js)
* signal: AbortSignal, // aborts an in-flight response
* }
*/
import { confirmation, note } from './blocks';
/**
* The agent provider: the real runtime, over the real API.
*
* The only provider that answers. Everything that makes that safe lives on the
* server:
*
* - **The principal is the session's.** The body carries a question and
* nothing else about who is asking. The browser cannot name a caller, so it
* cannot ask about records it is not entitled to see.
* - **The tools and corpora are the SPEC's.** Not the request's. An agent
* reads what its published definition says it may read, and no field here
* can widen that.
* - **A write cannot happen from a question.** The backend answers a proposed
* write with a confirmation payload and performs nothing. Approving it is a
* second, explicit call carrying the token — see `confirmation` below.
*
* One snapshot, not a stream. The endpoint is not streaming yet, so this yields
* the finished answer once. The signature is the streaming one because that is
* what the seam is, and switching to real streaming later changes this function
* and nothing else.
*/
export function createAgentProvider({ baseUrl = '/api/v1' } = {}) {
return {
id: 'agent',
async *stream({ question, agent = null, confirmation = null, agentVersion = 0, signal }) {
/* No agent, no run. The panel resolves which agent covers the page before
calling; reaching here without one means the routing layer changed and
this should say so rather than guess at an agent id. */
if (!agent?.id) {
yield [note('No agent is available for this page.')];
return;
}
const response = await fetch(`${baseUrl}/agents/${encodeURIComponent(agent.id)}/runs`, {
method: 'POST',
headers: {
'Content-Type': 'application/json',
/* Ask for a stream. The server answers the same run either way — the
final event carries exactly the body the non-streaming path
returns — so a deployment that cannot stream degrades to one late
snapshot rather than to a broken panel. */
Accept: 'text/event-stream, application/json',
},
/* The session cookie. Without it the API answers 401, which is the
correct answer to a browser that is not signed in. */
credentials: 'include',
body: JSON.stringify({
input: question,
confirmation: confirmation || undefined,
/* Pins the conversation to the version it started with. Every answer
comes back carrying its version; sending it on the next turn is what
stops an edit published mid-thread from silently changing which
agent is answering. */
agentVersion: agentVersion || undefined,
}),
signal,
});
if (!response.ok) {
yield [note(await describeFailure(response))];
return;
}
/* Not a stream after all — a proxy that buffers, or a server answering
JSON. Read it whole. */
if (!isEventStream(response) || !response.body) {
yield toBlocks(await response.json());
return;
}
yield* readRunStream(response, signal);
},
};
}
/**
* Whether the server actually opened a stream, rather than answering JSON.
*
* Defensive about `headers` because the answer to "is this a stream" must be
* NO when anything is unexpected. A response shape this does not recognise gets
* read whole, which works; assuming a stream and finding none would hang.
*/
function isEventStream(response) {
const type = response?.headers?.get?.('Content-Type') || '';
return type.includes('text/event-stream');
}
/**
* Reads the run's event stream, yielding the answer as it grows.
*
* Two kinds of event and they are handled very differently:
*
* - `delta` is a fragment of assistant text. The accumulated text is
* re-parsed into blocks on every one, so a half-written answer renders as
* far as it makes sense to — a table mid-construction stays as text until
* its rows arrive, which reads better than a broken table.
* - `run` is the finished result: the same body the non-streaming path
* returns, carrying confirmations, the termination and the token cost.
* Whatever text streamed is replaced by it, because the final snapshot is
* authoritative and the deltas were a preview of it.
*
* A client that ignored every delta and read only the last event would be in
* exactly the state it would have reached without streaming. That is what keeps
* the two paths honest rather than merely similar.
*/
async function* readRunStream(response, signal) {
const reader = response.body.getReader();
const decoder = new TextDecoder();
let buffer = '';
let text = '';
let final = null;
try {
while (true) {
const { done, value } = await reader.read();
if (done) break;
buffer += decoder.decode(value, { stream: true });
const lines = buffer.split('\n');
/* The last element may be half a line; hold it for the next read. */
buffer = lines.pop() ?? '';
for (const line of lines) {
if (!line.startsWith('data:')) continue;
const payload = line.slice(5).trim();
if (!payload || payload === '[DONE]') continue;
let event;
try {
event = JSON.parse(payload);
} catch {
/* A malformed frame is dropped rather than rendered. It cannot be
assistant text — the server encodes every event as JSON — so
showing it would put transport noise in front of a reader. */
continue;
}
if (typeof event.delta === 'string') {
text += event.delta;
yield markdownToBlocks(text);
continue;
}
if (event.run) final = event.run;
if (event.error) {
yield [note(event.error.message || 'The agent could not be reached.')];
return;
}
}
if (signal?.aborted) break;
}
} finally {
/* Releasing matters on an abort: a reader still holding the body keeps the
connection open, and a user who pressed Stop expects it to stop. */
try { reader.cancel(); } catch { /* already closed */ }
}
if (final) {
yield toBlocks(final);
return;
}
/* The stream ended without a final event — the run was aborted, or the
connection dropped mid-answer. Whatever arrived is kept: a half-read answer
is worth more to the reader than an empty panel. */
if (text) yield markdownToBlocks(text);
}
/**
* Turns a run result into blocks.
*
* The termination decides the shape, and every one of the six produces
* something a person can act on. A run that ended without completing is not an
* error to swallow: it has an answer-so-far worth keeping and a reason worth
* reading.
*/
function toBlocks(run) {
const blocks = [];
if (run.output) blocks.push(...markdownToBlocks(run.output));
/* Pending writes before the trailing note: somebody scrolling to the bottom
should meet the decision, not a footnote about token cost. */
for (const c of run.confirmations || []) {
blocks.push(confirmation(c));
}
/* The surface layer's wording for a run that did not complete. Rendered as a
note rather than as prose, so it reads as the system speaking rather than
as the agent's own words. */
if (run.message) blocks.push(note(run.message));
if (!blocks.length) {
blocks.push(note('The agent finished without saying anything.'));
}
return blocks;
}
/**
* Turns a model's markdown into the block vocabulary the panel already renders.
*
* A real model writes markdown — headings, numbered steps, tables. The local
* simulator never did: it emitted short single paragraphs, so the text renderer
* only ever handled inline bold and italic. Point the panel at a real model and
* `## What I'd do, in order` arrives on screen with the hashes still attached,
* and a comparison table arrives as pipes.
*
* The fix is NOT to render markdown inside a text block. The panel already has
* a heading block, a list block and a table block, all styled with the same
* tokens as the dashboard cards beside them — so the honest move is to parse
* into those, and let a generated answer look like it belongs to Krow rather
* than like a chat window that happens to be embedded in it.
*
* Deliberately a small parser and not a markdown library. Four constructs is
* what a model actually produces in an answer; anything else falls through as a
* paragraph, which reads correctly even when it is not styled richly. A full
* parser would be a large dependency in exchange for handling footnotes nobody
* writes.
*/
export function markdownToBlocks(markdown) {
const lines = String(markdown).replace(/\r\n/g, '\n').split('\n');
const blocks = [];
let paragraph = [];
let listItems = null;
let ordered = false;
const flushParagraph = () => {
const text = paragraph.join(' ').trim();
paragraph = [];
if (text) blocks.push({ type: 'text', text });
};
const flushList = () => {
if (listItems?.length) blocks.push({ type: 'list', items: listItems, ordered });
listItems = null;
};
const flushAll = () => { flushParagraph(); flushList(); };
for (let i = 0; i < lines.length; i += 1) {
const line = lines[i];
const trimmed = line.trim();
if (!trimmed) { flushAll(); continue; }
/* A heading. The level is dropped: this panel has one heading style, and
inventing three would give a 380px column a hierarchy it cannot show. */
const heading = /^(#{1,6})\s+(.+)$/.exec(trimmed);
if (heading) {
flushAll();
blocks.push({ type: 'heading', text: stripInline(heading[2]) });
continue;
}
/* A table: a pipe row followed by a separator row. Checked together,
because a single pipe row is far more likely to be prose. */
if (trimmed.startsWith('|') && isSeparatorRow(lines[i + 1])) {
flushAll();
const { block, next } = parseTable(lines, i);
if (block) { blocks.push(block); i = next; continue; }
}
const bullet = /^[-*]\s+(.+)$/.exec(trimmed);
const numbered = /^\d+[.)]\s+(.+)$/.exec(trimmed);
if (bullet || numbered) {
const wantOrdered = Boolean(numbered);
/* A list that changes kind mid-way is two lists. */
if (listItems && ordered !== wantOrdered) flushList();
flushParagraph();
ordered = wantOrdered;
listItems = listItems || [];
listItems.push((bullet || numbered)[1].trim());
continue;
}
flushList();
paragraph.push(trimmed);
}
flushAll();
return blocks.length ? blocks : [{ type: 'text', text: String(markdown).trim() }];
}
/** `|---|---:|` — the row that makes the one above it a header. */
function isSeparatorRow(line) {
return Boolean(line && /^\s*\|?[\s:|-]+\|[\s:|-]*$/.test(line) && line.includes('-'));
}
function splitRow(line) {
return line.trim().replace(/^\|/, '').replace(/\|$/, '').split('|').map((c) => c.trim());
}
/**
* Reads a markdown table starting at `start`.
*
* Rows with the wrong number of cells are padded or trimmed rather than
* dropped. A model occasionally miscounts a pipe, and losing a whole row of an
* answer over a formatting slip is worse than showing an empty cell — which the
* renderer already draws as an em dash.
*/
function parseTable(lines, start) {
const header = splitRow(lines[start]);
const columns = header.map((label, i) => ({ key: `c${i}`, label: stripInline(label) }));
const rows = [];
let i = start + 2;
for (; i < lines.length; i += 1) {
const line = lines[i];
if (!line.trim().startsWith('|')) break;
const cells = splitRow(line);
const row = {};
columns.forEach((col, n) => { row[col.key] = stripInline(cells[n] ?? ''); });
rows.push(row);
}
if (!rows.length) return { block: null, next: start };
return { block: { type: 'table', columns, rows }, next: i - 1 };
}
/**
* Removes markdown a cell or heading cannot show.
*
* Table cells and headings are rendered as plain strings by their components,
* so `**Maria**` would appear with the asterisks. Paragraphs and list items are
* left alone — those go through `Inline`, which renders bold properly.
*/
function stripInline(value) {
return String(value).replace(/\*\*(.+?)\*\*/g, '$1').replace(/`(.+?)`/g, '$1').trim();
}
/**
* A failed request, in one sentence a person can act on.
*
* The status is what distinguishes the cases that matter, and they are
* genuinely different actions: sign in again, ask someone for access, or wait.
* Flattening them into "something went wrong" makes the user's next move a
* guess.
*/
async function describeFailure(response) {
let detail = '';
try {
const body = await response.json();
detail = body?.error?.message || '';
} catch {
/* A non-JSON error body is a proxy or a gateway, not this API. The status
still says enough. */
}
switch (response.status) {
case 401:
return 'Your session has expired. Sign in again to keep asking.';
case 404:
/* Deliberately the same answer for "no such agent" and "not yours" — the
API refuses to distinguish them, and repeating the distinction here
would undo that. */
return 'That agent is not available on this workspace.';
case 422:
return detail || 'That agent cannot run right now.';
case 429:
return 'Too many requests just now. Try again in a moment.';
default:
return detail || 'The agent could not be reached. Try again in a moment.';
}
}
/**
* The provider the app uses.
*
* There used to be three: a local simulator that computed answers from data
* already in the browser, a streaming HTTP provider for a deployment that had
* one, and the agent. The simulator is gone, and its removal is the point of
* this file's current shape.
*
* WHY IT WENT
*
* Two answering paths behind one avatar meant the same question got different
* answers depending on phrasing — "assign the strongest free worker" matched a
* template and reported nobody was available, while "put the best free worker
* on" reached the agent, which found somebody and proposed them. A user cannot
* be expected to know which sentence talks to which system, and a product where
* the wording decides the answer is a demo with good manners.
*
* WHAT IT COST, SAID PLAINLY
*
* The simulator was instant, free, and could not be wrong about a figure — it
* read the same cache the page rendered from. The agent takes ten to twenty
* seconds and costs tokens. That is a real regression on speed, accepted in
* exchange for answers that can be followed up, cannot be beaten by a synonym,
* and can act on what they find.
*
* AN UNCONFIGURED DEPLOYMENT NOW SAYS SO
*
* With no VITE_AGENT_API there is nothing to fall back to. That state is
* explicit rather than silent: every question is answered with the reason,
* because a panel that quietly does nothing is the worst of the three
* possibilities and the hardest to diagnose.
*/
export function createAssistantProvider() {
const base = import.meta.env?.VITE_AGENT_API;
return base ? createAgentProvider({ baseUrl: base }) : createUnconfiguredProvider();
}
/**
* The provider for a deployment with no agent configured.
*
* Answers every question with the same sentence, which is the honest thing to
* do: nothing here can answer, and pretending otherwise is what the simulator
* was doing.
*/
export function createUnconfiguredProvider() {
return {
id: 'unconfigured',
// eslint-disable-next-line require-yield
async *stream() {
yield [note(
'Owliver is not configured on this deployment. Set VITE_AGENT_API and give the '
+ 'backend a model credential, and this panel will answer from your workspace.'
)];
},
};
}