import { agentCovers } from './runtime'; /** * The knowledge seam. * * Boundary 5 of the architecture, established now and deliberately minimal. * There is no vector store, no embedding model and no document corpus in this * phase, and none is faked to make retrieval look implemented. * * What exists is real: an agent's authored `knowledge:` entries, chunked and * matched on terms. Somebody wrote those entries, an answer that quotes one * says where it came from, and an agent with none returns nothing rather than a * plausible paragraph. That is a small capability honestly delivered, not a * stub pretending to be retrieval. * * **The page boundary applies here exactly as it does to skills.** An agent * that does not cover the current page retrieves nothing — otherwise knowledge * would be the one door through which selecting an agent could reach material * the page was not offering, which is the failure the whole design exists to * prevent. Knowledge is scoped by the same rule as data and tools. * * The shape is what makes this replaceable rather than throwaway: passages come * back as `{documentId, chunkId, text, score}`, which is what a retrieval layer * returns. Moving to Postgres and pgvector later means reimplementing * `retrieveKnowledge` behind the same signature — a change of transport, not a * change of format, and the same discipline `base44Client.js` already applies * to the data layer. */ /** Words too common to discriminate between one passage and another. */ const STOP_WORDS = new Set([ 'the', 'a', 'an', 'and', 'or', 'but', 'is', 'are', 'was', 'were', 'be', 'been', 'to', 'of', 'in', 'on', 'at', 'for', 'with', 'by', 'from', 'as', 'that', 'this', 'it', 'its', 'we', 'our', 'us', 'you', 'your', 'i', 'me', 'my', 'what', 'which', 'who', 'when', 'where', 'how', 'why', 'do', 'does', 'did', 'can', 'could', 'should', 'would', 'will', 'about', 'says', 'say', ]); const terms = (text) => String(text || '') .toLowerCase() .split(/[^a-z0-9]+/) .filter((word) => word.length > 2 && !STOP_WORDS.has(word)); /** * One knowledge entry, split into passages. * * By sentence, because a knowledge entry is prose and a sentence is the * smallest piece of it that still means something on its own. A fixed-width * chunker would cut mid-clause and quote half a rule, which is worse than not * answering. */ function chunk(entry) { const body = String(entry.body || '').trim(); if (!body) return []; const sentences = body .split(/(?<=[.!?])\s+/) .map((s) => s.trim()) .filter(Boolean); return (sentences.length ? sentences : [body]).map((text, i) => ({ documentId: entry.id, chunkId: `${entry.id}#${i}`, label: entry.label, kind: entry.kind, url: entry.url || '', text, })); } /** * Passages from this agent's own knowledge that bear on the question. * * Returns `{ passages, source, available, note }`. `available: false` means * there is nothing to search — no agent, no entries, or an agent that does not * cover this page — and the note says which. A caller must render that rather * than treating an empty result as "the documents say nothing". * * @param {Object} options * @param {Object|null} options.agent The active agent. * @param {string|null} options.contextId The page's assistant context. * @param {string} options.question What was asked. * @param {number} options.limit Most passages to return. */ /** @param {any} [options] */ export function retrieveKnowledge({ agent = null, contextId = null, question = '', limit = 3, } = {}) { const empty = (note) => ({ passages: [], source: 'declared', available: false, note, }); if (!agent) return empty('No agent is active, so there is no knowledge to search.'); /* The page boundary. An agent constrained here contributes no knowledge, for the same reason it contributes no skills. */ if (contextId && !agentCovers(agent, contextId)) { return empty(`${agent.name} does not cover this page, so its knowledge is not in reach here.`); } const entries = (agent.knowledge || []).filter((entry) => entry.body || entry.url); if (!entries.length) { return empty(`${agent.name} has no knowledge attached.`); } const wanted = terms(question); if (!wanted.length) { return { passages: [], source: 'declared', available: true, note: 'Ask about something specific and I will check this agent\'s knowledge.', }; } const passages = entries .flatMap(chunk) .map((passage) => { const words = new Set(terms(`${passage.label} ${passage.text}`)); const hits = wanted.filter((word) => words.has(word)); return { ...passage, score: hits.length, matched: hits }; }) /* A passage that matches nothing is not a weak answer, it is a different subject. Returning it would put unrelated prose under a question and let the reader assume it was relevant. */ .filter((passage) => passage.score > 0) .sort((a, b) => b.score - a.score) .slice(0, limit); return { passages, source: 'declared', available: true, note: passages.length ? null : `Nothing in ${agent.name}'s knowledge covers that.`, }; } /** * Whether a question can be answered from knowledge at all. * * Used to decide whether to say "no source is configured" or "the source has * nothing on this" — two different answers, and reporting the second when the * first is true would imply a corpus exists. */ export const hasKnowledge = (agent) => Boolean(agent && (agent.knowledge || []).some((entry) => entry.body || entry.url)); /** The documents an agent carries, for a configuration screen. */ export const knowledgeDocuments = (agent) => (agent?.knowledge || []).map((entry) => ({ id: entry.id, label: entry.label, kind: entry.kind, url: entry.url || '', chunks: chunk(entry).length, }));