67 lines
2.4 KiB
JavaScript
67 lines
2.4 KiB
JavaScript
import { pipeline, env } from '@xenova/transformers';
|
|
|
|
// ==============================|| Embeddings ||============================== //
|
|
//
|
|
// all-MiniLM-L6-v2 running IN PROCESS. No API key, no network call per query,
|
|
// no per-call cost.
|
|
//
|
|
// That is not an incidental choice. assistant/CLAUDE.md §2 blocked LLM work on
|
|
// exactly one ground: a hosted model needs a secret key and a static CRA build
|
|
// has nowhere to put one. A local embedding model has no key, so this plan is
|
|
// outside that blocker. Swapping to a hosted embedding model (text-embedding-3
|
|
// -small et al) re-opens §2 and needs its own decision — do not do it quietly.
|
|
|
|
export const MODEL_ID = 'Xenova/all-MiniLM-L6-v2';
|
|
export const DIMENSIONS = 384;
|
|
|
|
// Weights are cached on disk (mounted in docker-compose) so a container
|
|
// restart doesn't re-download 25MB.
|
|
env.cacheDir = process.env.TRANSFORMERS_CACHE || './.cache';
|
|
// Nothing here should reach the network except the one-time model fetch.
|
|
env.allowRemoteModels = true;
|
|
|
|
let extractor = null;
|
|
let loading = null;
|
|
|
|
// Loaded once, lazily, and shared. Concurrent callers await the same promise
|
|
// rather than each triggering their own model load.
|
|
const getExtractor = async () => {
|
|
if (extractor) return extractor;
|
|
if (!loading) {
|
|
loading = pipeline('feature-extraction', MODEL_ID).then((p) => {
|
|
extractor = p;
|
|
return p;
|
|
});
|
|
}
|
|
return loading;
|
|
};
|
|
|
|
export const warmUp = async () => {
|
|
await getExtractor();
|
|
return { model: MODEL_ID, dimensions: DIMENSIONS };
|
|
};
|
|
|
|
// Mean-pooled + L2-normalised sentence embedding.
|
|
//
|
|
// Normalisation matters: Chroma's cosine space assumes unit vectors, and
|
|
// queries must be embedded exactly the same way documents were. A mismatch
|
|
// doesn't error — it silently degrades every score, which is the worst kind of
|
|
// bug to chase. `seed` writes the model id into collection metadata so a
|
|
// mismatch is at least detectable.
|
|
export const embed = async (text) => {
|
|
const pipe = await getExtractor();
|
|
const output = await pipe(String(text || '').trim(), { pooling: 'mean', normalize: true });
|
|
return Array.from(output.data);
|
|
};
|
|
|
|
export const embedMany = async (texts) => {
|
|
const out = [];
|
|
for (const t of texts) {
|
|
// Sequential on purpose: batching MiniLM in-process gives no meaningful
|
|
// speed-up at this corpus size (~700 vectors) and makes memory spikier.
|
|
// eslint-disable-next-line no-await-in-loop
|
|
out.push(await embed(t));
|
|
}
|
|
return out;
|
|
};
|