implemenation on the bot
This commit is contained in:
66
services/ai/embed.js
Normal file
66
services/ai/embed.js
Normal file
@@ -0,0 +1,66 @@
|
||||
import { pipeline, env } from '@xenova/transformers';
|
||||
|
||||
// ==============================|| Embeddings ||============================== //
|
||||
//
|
||||
// all-MiniLM-L6-v2 running IN PROCESS. No API key, no network call per query,
|
||||
// no per-call cost.
|
||||
//
|
||||
// That is not an incidental choice. assistant/CLAUDE.md §2 blocked LLM work on
|
||||
// exactly one ground: a hosted model needs a secret key and a static CRA build
|
||||
// has nowhere to put one. A local embedding model has no key, so this plan is
|
||||
// outside that blocker. Swapping to a hosted embedding model (text-embedding-3
|
||||
// -small et al) re-opens §2 and needs its own decision — do not do it quietly.
|
||||
|
||||
export const MODEL_ID = 'Xenova/all-MiniLM-L6-v2';
|
||||
export const DIMENSIONS = 384;
|
||||
|
||||
// Weights are cached on disk (mounted in docker-compose) so a container
|
||||
// restart doesn't re-download 25MB.
|
||||
env.cacheDir = process.env.TRANSFORMERS_CACHE || './.cache';
|
||||
// Nothing here should reach the network except the one-time model fetch.
|
||||
env.allowRemoteModels = true;
|
||||
|
||||
let extractor = null;
|
||||
let loading = null;
|
||||
|
||||
// Loaded once, lazily, and shared. Concurrent callers await the same promise
|
||||
// rather than each triggering their own model load.
|
||||
const getExtractor = async () => {
|
||||
if (extractor) return extractor;
|
||||
if (!loading) {
|
||||
loading = pipeline('feature-extraction', MODEL_ID).then((p) => {
|
||||
extractor = p;
|
||||
return p;
|
||||
});
|
||||
}
|
||||
return loading;
|
||||
};
|
||||
|
||||
export const warmUp = async () => {
|
||||
await getExtractor();
|
||||
return { model: MODEL_ID, dimensions: DIMENSIONS };
|
||||
};
|
||||
|
||||
// Mean-pooled + L2-normalised sentence embedding.
|
||||
//
|
||||
// Normalisation matters: Chroma's cosine space assumes unit vectors, and
|
||||
// queries must be embedded exactly the same way documents were. A mismatch
|
||||
// doesn't error — it silently degrades every score, which is the worst kind of
|
||||
// bug to chase. `seed` writes the model id into collection metadata so a
|
||||
// mismatch is at least detectable.
|
||||
export const embed = async (text) => {
|
||||
const pipe = await getExtractor();
|
||||
const output = await pipe(String(text || '').trim(), { pooling: 'mean', normalize: true });
|
||||
return Array.from(output.data);
|
||||
};
|
||||
|
||||
export const embedMany = async (texts) => {
|
||||
const out = [];
|
||||
for (const t of texts) {
|
||||
// Sequential on purpose: batching MiniLM in-process gives no meaningful
|
||||
// speed-up at this corpus size (~700 vectors) and makes memory spikier.
|
||||
// eslint-disable-next-line no-await-in-loop
|
||||
out.push(await embed(t));
|
||||
}
|
||||
return out;
|
||||
};
|
||||
Reference in New Issue
Block a user