Files
doormilxpress_astryx/src/lib/agents/knowledge.js
2026-08-20 18:18:10 +05:30

158 lines
5.8 KiB
JavaScript

import { agentCovers } from './runtime';
/**
* The knowledge seam.
*
* Boundary 5 of the architecture, established now and deliberately minimal.
* There is no vector store, no embedding model and no document corpus in this
* phase, and none is faked to make retrieval look implemented.
*
* What exists is real: an agent's authored `knowledge:` entries, chunked and
* matched on terms. Somebody wrote those entries, an answer that quotes one
* says where it came from, and an agent with none returns nothing rather than a
* plausible paragraph. That is a small capability honestly delivered, not a
* stub pretending to be retrieval.
*
* **The page boundary applies here exactly as it does to skills.** An agent
* that does not cover the current page retrieves nothing — otherwise knowledge
* would be the one door through which selecting an agent could reach material
* the page was not offering, which is the failure the whole design exists to
* prevent. Knowledge is scoped by the same rule as data and tools.
*
* The shape is what makes this replaceable rather than throwaway: passages come
* back as `{documentId, chunkId, text, score}`, which is what a retrieval layer
* returns. Moving to Postgres and pgvector later means reimplementing
* `retrieveKnowledge` behind the same signature — a change of transport, not a
* change of format, and the same discipline `base44Client.js` already applies
* to the data layer.
*/
/** Words too common to discriminate between one passage and another. */
const STOP_WORDS = new Set([
'the', 'a', 'an', 'and', 'or', 'but', 'is', 'are', 'was', 'were', 'be', 'been',
'to', 'of', 'in', 'on', 'at', 'for', 'with', 'by', 'from', 'as', 'that', 'this',
'it', 'its', 'we', 'our', 'us', 'you', 'your', 'i', 'me', 'my', 'what', 'which',
'who', 'when', 'where', 'how', 'why', 'do', 'does', 'did', 'can', 'could',
'should', 'would', 'will', 'about', 'says', 'say',
]);
const terms = (text) =>
String(text || '')
.toLowerCase()
.split(/[^a-z0-9]+/)
.filter((word) => word.length > 2 && !STOP_WORDS.has(word));
/**
* One knowledge entry, split into passages.
*
* By sentence, because a knowledge entry is prose and a sentence is the
* smallest piece of it that still means something on its own. A fixed-width
* chunker would cut mid-clause and quote half a rule, which is worse than not
* answering.
*/
function chunk(entry) {
const body = String(entry.body || '').trim();
if (!body) return [];
const sentences = body
.split(/(?<=[.!?])\s+/)
.map((s) => s.trim())
.filter(Boolean);
return (sentences.length ? sentences : [body]).map((text, i) => ({
documentId: entry.id,
chunkId: `${entry.id}#${i}`,
label: entry.label,
kind: entry.kind,
url: entry.url || '',
text,
}));
}
/**
* Passages from this agent's own knowledge that bear on the question.
*
* Returns `{ passages, source, available, note }`. `available: false` means
* there is nothing to search — no agent, no entries, or an agent that does not
* cover this page — and the note says which. A caller must render that rather
* than treating an empty result as "the documents say nothing".
*
* @param {Object} options
* @param {Object|null} options.agent The active agent.
* @param {string|null} options.contextId The page's assistant context.
* @param {string} options.question What was asked.
* @param {number} options.limit Most passages to return.
*/
/** @param {any} [options] */
export function retrieveKnowledge({
agent = null, contextId = null, question = '', limit = 3,
} = {}) {
const empty = (note) => ({
passages: [], source: 'declared', available: false, note,
});
if (!agent) return empty('No agent is active, so there is no knowledge to search.');
/* The page boundary. An agent constrained here contributes no knowledge, for
the same reason it contributes no skills. */
if (contextId && !agentCovers(agent, contextId)) {
return empty(`${agent.name} does not cover this page, so its knowledge is not in reach here.`);
}
const entries = (agent.knowledge || []).filter((entry) => entry.body || entry.url);
if (!entries.length) {
return empty(`${agent.name} has no knowledge attached.`);
}
const wanted = terms(question);
if (!wanted.length) {
return {
passages: [], source: 'declared', available: true,
note: 'Ask about something specific and I will check this agent\'s knowledge.',
};
}
const passages = entries
.flatMap(chunk)
.map((passage) => {
const words = new Set(terms(`${passage.label} ${passage.text}`));
const hits = wanted.filter((word) => words.has(word));
return { ...passage, score: hits.length, matched: hits };
})
/* A passage that matches nothing is not a weak answer, it is a different
subject. Returning it would put unrelated prose under a question and
let the reader assume it was relevant. */
.filter((passage) => passage.score > 0)
.sort((a, b) => b.score - a.score)
.slice(0, limit);
return {
passages,
source: 'declared',
available: true,
note: passages.length
? null
: `Nothing in ${agent.name}'s knowledge covers that.`,
};
}
/**
* Whether a question can be answered from knowledge at all.
*
* Used to decide whether to say "no source is configured" or "the source has
* nothing on this" — two different answers, and reporting the second when the
* first is true would imply a corpus exists.
*/
export const hasKnowledge = (agent) =>
Boolean(agent && (agent.knowledge || []).some((entry) => entry.body || entry.url));
/** The documents an agent carries, for a configuration screen. */
export const knowledgeDocuments = (agent) =>
(agent?.knowledge || []).map((entry) => ({
id: entry.id,
label: entry.label,
kind: entry.kind,
url: entry.url || '',
chunks: chunk(entry).length,
}));