import { pipeline, env } from '@xenova/transformers'; // ==============================|| Embeddings ||============================== // // // all-MiniLM-L6-v2 running IN PROCESS. No API key, no network call per query, // no per-call cost. // // That is not an incidental choice. assistant/CLAUDE.md §2 blocked LLM work on // exactly one ground: a hosted model needs a secret key and a static CRA build // has nowhere to put one. A local embedding model has no key, so this plan is // outside that blocker. Swapping to a hosted embedding model (text-embedding-3 // -small et al) re-opens §2 and needs its own decision — do not do it quietly. export const MODEL_ID = 'Xenova/all-MiniLM-L6-v2'; export const DIMENSIONS = 384; // Weights are cached on disk (mounted in docker-compose) so a container // restart doesn't re-download 25MB. env.cacheDir = process.env.TRANSFORMERS_CACHE || './.cache'; // Nothing here should reach the network except the one-time model fetch. env.allowRemoteModels = true; let extractor = null; let loading = null; // Loaded once, lazily, and shared. Concurrent callers await the same promise // rather than each triggering their own model load. const getExtractor = async () => { if (extractor) return extractor; if (!loading) { loading = pipeline('feature-extraction', MODEL_ID).then((p) => { extractor = p; return p; }); } return loading; }; export const warmUp = async () => { await getExtractor(); return { model: MODEL_ID, dimensions: DIMENSIONS }; }; // Mean-pooled + L2-normalised sentence embedding. // // Normalisation matters: Chroma's cosine space assumes unit vectors, and // queries must be embedded exactly the same way documents were. A mismatch // doesn't error — it silently degrades every score, which is the worst kind of // bug to chase. `seed` writes the model id into collection metadata so a // mismatch is at least detectable. export const embed = async (text) => { const pipe = await getExtractor(); const output = await pipe(String(text || '').trim(), { pooling: 'mean', normalize: true }); return Array.from(output.data); }; export const embedMany = async (texts) => { const out = []; for (const t of texts) { // Sequential on purpose: batching MiniLM in-process gives no meaningful // speed-up at this corpus size (~700 vectors) and makes memory spikier. // eslint-disable-next-line no-await-in-loop out.push(await embed(t)); } return out; };