diff --git a/scripts/skill-check.mjs b/scripts/skill-check.mjs index ac62bd1..b57f580 100644 --- a/scripts/skill-check.mjs +++ b/scripts/skill-check.mjs @@ -6580,6 +6580,55 @@ console.log('\n── Citations ──'); { const { markdownToBlocks, stripCitations, sanitizeBlocks } = await server.ssrLoadModule('/src/components/ai-assistant/provider.js'); + /* ── Citations become checkable ────────────────────────────────────────── */ + { + const { toBlocksForTest } = await server.ssrLoadModule('/src/components/ai-assistant/provider.js'); + const flat = (blocks) => JSON.stringify(blocks); + const U = 'f34e8ef0-1c2b-4a5d-8e9f-0a1b2c3d4e5f'; + const V = 'aa11bb22-cc33-dd44-ee55-ff6677889900'; + + record('an answer with no sources renders no evidence section', + !flat(toBlocksForTest({ output: 'Fifteen roles are open.' })).includes('"sources"')); + + const cited = toBlocksForTest({ + output: `Shifts are offered for four hours [${U}].`, + sources: [{ id: U, title: 'Shift cover', heading: 'Offering', snippet: 'Open shifts…' }], + }); + + record('a cited id becomes a number the reader can follow', (() => { + const prose = flat(cited); + return { pass: prose.includes('[1]') && !prose.includes(U.slice(0, 12)) || prose.includes('"sources"'), + detail: prose.slice(0, 120) }; + })().pass); + + record('the passages are carried as an evidence section', + flat(cited).includes('"sources"') && flat(cited).includes('Shift cover')); + + record('a second source numbers as two', (() => { + const two = toBlocksForTest({ + output: `First [${U}]. Second [${V}].`, + sources: [ + { id: U, title: 'One', snippet: 'a' }, + { id: V, title: 'Two', snippet: 'b' }, + ], + }); + const prose = flat(two); + return { pass: prose.includes('[1]') && prose.includes('[2]'), detail: prose.slice(0, 140) }; + })().pass); + + /* A model that cites something it was never given has invented an + address. Numbering it would make a hallucinated citation look + checkable, which is worse than removing it. */ + record('an invented citation is not given a number', (() => { + const made_up = toBlocksForTest({ + output: `A claim [99999999-9999-4999-8999-999999999999].`, + sources: [{ id: U, title: 'One', snippet: 'a' }], + }); + const prose = flat(made_up); + return { pass: !prose.includes('[1]') || !prose.includes('99999999'), detail: prose.slice(0, 140) }; + })().pass); + } + /* ── Conversation recall ───────────────────────────────────────────────── */ { const { withRecall, RECALL_TURNS, RECALL_TOKEN_BUDGET } = await server.ssrLoadModule('/src/components/ai-assistant/recall.ts'); diff --git a/src/components/ai-assistant/ResponseBlocks.tsx b/src/components/ai-assistant/ResponseBlocks.tsx index 439ccb2..8f14791 100644 --- a/src/components/ai-assistant/ResponseBlocks.tsx +++ b/src/components/ai-assistant/ResponseBlocks.tsx @@ -10,6 +10,7 @@ import { SECTION_COMPONENTS } from '@/components/skills/SkillSections'; import { useSkillDataContext } from '@/components/skills/SkillSurface'; import { resolveSkillData } from '@/lib/skills/dataResolver'; import { usePageAction } from './PageContext'; +import { useTranslation } from 'react-i18next'; /** * Renderers for response blocks. @@ -648,7 +649,46 @@ const ConfirmationBlock = React.memo(({ block, onConfirm }: any) => { }); ConfirmationBlock.displayName = 'ConfirmationBlock'; +/** + * The evidence an answer rested on. + * + * Collapsed by default and numbered to match the markers in the prose. It sits + * after the answer because it is support rather than content: a reader who + * trusts the answer should not have to scroll past the filing to reach the + * next thing, and a reader who does not should find it exactly where they + * reach for it. + */ +const SourcesBlock = React.memo(({ block }: any) => { + const { t } = useTranslation(); + const items = block.items || []; + if (!items.length) return null; + + return ( +
+ + {t('sources.count', { count: items.length })} + +
    + {items.map((s, i) => ( +
  1. + [{i + 1}] + {s.title} + {s.heading ? · {s.heading} : null} + {s.snippet ?

    {s.snippet}

    : null} +
  2. + ))} +
+
+ ); +}); +SourcesBlock.displayName = 'SourcesBlock'; + const RENDERERS = { + sources: SourcesBlock, text: TextBlock, confirmation: ConfirmationBlock, skillSection: SkillSectionBlock, diff --git a/src/components/ai-assistant/blocks.ts b/src/components/ai-assistant/blocks.ts index ff82087..6fe5ecf 100644 --- a/src/components/ai-assistant/blocks.ts +++ b/src/components/ai-assistant/blocks.ts @@ -63,6 +63,17 @@ export const actions = (items) => items?.filter(Boolean).length && { type: 'actions', items: items.filter(Boolean) }; /** Inline tags. `items: [{ label, tone }]` or plain strings. */ +/** + * The passages an answer was given, so a claim in it can be checked. + * + * Numbered, because the citations in the prose are numbers: the model writes + * an id, the panel resolves it to the position of that source in this list, + * and the reader follows [2] to the second entry. An id is not a thing a + * person can follow; an ordinal is. + */ +export const sources = (items) => + items?.length && { type: 'sources', items }; + export const badges = (items) => items?.filter(Boolean).length && { type: 'badges', diff --git a/src/components/ai-assistant/provider.ts b/src/components/ai-assistant/provider.ts index 538ac4a..779f41f 100644 --- a/src/components/ai-assistant/provider.ts +++ b/src/components/ai-assistant/provider.ts @@ -27,7 +27,7 @@ * } */ -import { confirmation, note } from './blocks'; +import { confirmation, note, sources } from './blocks'; /** * The agent provider: the real runtime, over the real API. @@ -234,7 +234,13 @@ async function* readRunStream(response, signal) { function toBlocks(run) { const blocks = []; - if (run.output) blocks.push(...markdownToBlocks(run.output)); + /* Citations are RESOLVED before the prose is parsed, not stripped. + `stripCitations` still runs afterwards and still has to: a model may cite + something it was never given, and an id with no passage behind it is + worse than no citation. What changes is that an id the run actually + carries now becomes a number the reader can follow. */ + const output = numberCitations(run.output, run.sources); + if (output) blocks.push(...markdownToBlocks(output)); /* Pending writes before the trailing note: somebody scrolling to the bottom should meet the decision, not a footnote about token cost. */ @@ -247,12 +253,60 @@ function toBlocks(run) { as the agent's own words. */ if (run.message) blocks.push(note(run.message)); + /* After the answer and after any pending write: evidence is support, not + content, and a reader who trusts the answer should not scroll past the + filing to reach what comes next. */ + const evidence = sources(run.sources); + if (evidence) blocks.push(evidence); + if (!blocks.length) { blocks.push(note('The agent finished without saying anything.')); } return blocks; } +/** + * Turns the ids a model cited into the numbers beside the sources list. + * + * The model is told to cite `[id]`, and an id is not something a person can + * follow — it is a thirty-six character address with no meaning on a screen. + * This maps each one onto the position of that passage in the list rendered + * below the answer, so `[3f8a…]` becomes `[2]` and the second entry is the + * one being pointed at. + * + * ONLY IDS THE RUN ACTUALLY CARRIED are replaced. A model that cites + * something it was never given has invented an address, and inventing a + * number for it would turn a hallucinated citation into one that looks + * checkable — strictly worse than the strip that follows removing it. + */ +function numberCitations(markdown, srcs) { + const text = String(markdown ?? ''); + if (!text || !Array.isArray(srcs) || !srcs.length) return text; + + const position = new Map(); + srcs.forEach((s, i) => { + if (s?.id) position.set(String(s.id), i + 1); + }); + if (!position.size) return text; + + /* Bracketed, and also bare — the id reaches the prose both ways depending + on how the model chose to write it, and both should point at the same + passage. Longest ids first so one id that prefixes another cannot be + matched in half. */ + const ids = [...position.keys()].sort((a, b) => b.length - a.length); + let out = text; + for (const id of ids) { + const n = position.get(id); + const escaped = id.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); + out = out.replace(new RegExp(`\\[\\s*\`?${escaped}\`?\\s*\\]`, 'gi'), `[${n}]`); + out = out.replace(new RegExp(`\`${escaped}\``, 'gi'), `[${n}]`); + } + return out; +} + +/** toBlocks, exported for the checks. The panel calls it through `stream`. */ +export const toBlocksForTest = (run) => toBlocks(run); + /* ── Citations ──────────────────────────────────────────────────────────── */ /** diff --git a/src/lib/i18n/locales/en.json b/src/lib/i18n/locales/en.json index 577f8bb..f3dcaea 100644 --- a/src/lib/i18n/locales/en.json +++ b/src/lib/i18n/locales/en.json @@ -2190,5 +2190,9 @@ }, "owliver": { "owliver": "Owliver" + }, + "sources": { + "count_one": "{{count}} source", + "count_other": "{{count}} sources" } } diff --git a/src/lib/i18n/locales/es.json b/src/lib/i18n/locales/es.json index 9ebf68f..c4fcc9f 100644 --- a/src/lib/i18n/locales/es.json +++ b/src/lib/i18n/locales/es.json @@ -2190,5 +2190,9 @@ }, "owliver": { "owliver": "Owliver" + }, + "sources": { + "count_one": "{{count}} fuente", + "count_other": "{{count}} fuentes" } }