149 lines
5.4 KiB
JavaScript
149 lines
5.4 KiB
JavaScript
import fs from 'fs';
|
|
import path from 'path';
|
|
import { fileURLToPath } from 'url';
|
|
import { INTENTS_COLLECTION, DOCS_COLLECTION, resetCollection } from '../collections.js';
|
|
import { embedMany, MODEL_ID } from '../embed.js';
|
|
|
|
// ==============================|| Seeding ||============================== //
|
|
//
|
|
// Rebuilds both collections from scratch. Idempotent — safe to re-run any time,
|
|
// and it MUST be re-run after changing the embedding model or editing any
|
|
// indexed document, or answers drift from the source without any error.
|
|
|
|
const HERE = path.dirname(fileURLToPath(import.meta.url));
|
|
const REPO = path.resolve(HERE, '../../..');
|
|
|
|
// Write intents get a stricter confidence bar in the console (a semantic
|
|
// near-miss must never open a create form), so they are flagged here.
|
|
const WRITE_INTENTS = new Set(['createOrder', 'createCustomer']);
|
|
|
|
const DOMAIN = {
|
|
orders: ['totalOrders', 'weekOrders', 'statusBreakdown', 'orderQuery', 'orderTrend', 'orderRate', 'delayedOrders', 'batchCount', 'revenueTotal', 'comparisonIntent', 'opsSummary', 'orderLookup', 'parcelTrack'],
|
|
riders: ['riderCounts', 'riderLookup', 'riderActivity'],
|
|
tenants: ['tenantList', 'tenantCount', 'tenantDetail'],
|
|
fleet: ['hubStatus', 'hubLookup', 'vehicleStatus', 'vehicleLookup', 'tripsheetStatus', 'exceptionStatus', 'consignmentStatus'],
|
|
admin: ['customerCount', 'appUserCount', 'pricingCount', 'partnerCount', 'competitorBranchCount', 'carrierPricingCount'],
|
|
write: ['createCustomer', 'createOrder']
|
|
};
|
|
|
|
const domainOf = (intentId) => Object.entries(DOMAIN).find(([, ids]) => ids.includes(intentId))?.[0] || 'other';
|
|
|
|
// ---- intent_examples --------------------------------------------------------
|
|
const seedIntents = async () => {
|
|
const raw = JSON.parse(fs.readFileSync(path.join(HERE, 'phrasings.json'), 'utf8'));
|
|
const ids = [];
|
|
const documents = [];
|
|
const metadatas = [];
|
|
|
|
for (const [intentId, phrasings] of Object.entries(raw)) {
|
|
if (intentId.startsWith('_')) continue;
|
|
phrasings.forEach((phrase, i) => {
|
|
ids.push(`${intentId}::${String(i).padStart(2, '0')}`);
|
|
documents.push(phrase);
|
|
metadatas.push({ intentId, domain: domainOf(intentId), isWrite: WRITE_INTENTS.has(intentId) });
|
|
});
|
|
}
|
|
|
|
const collection = await resetCollection(INTENTS_COLLECTION);
|
|
const embeddings = await embedMany(documents);
|
|
await collection.add({ ids, documents, metadatas, embeddings });
|
|
return { intents: new Set(metadatas.map((m) => m.intentId)).size, vectors: ids.length };
|
|
};
|
|
|
|
// ---- console_docs -----------------------------------------------------------
|
|
//
|
|
// Split on markdown headings, then hard-wrap long sections. The heading path is
|
|
// prepended to every chunk so a retrieved passage carries its own context —
|
|
// without it, a chunk reading "Don't do this" is worse than useless.
|
|
const CHUNK_CHARS = 800;
|
|
const OVERLAP = 100;
|
|
|
|
const DOC_SOURCES = [
|
|
'express-console-api.md',
|
|
'CLAUDE.md',
|
|
'src/pages/api/CLAUDE.md',
|
|
'src/pages/nearle/assistant/CLAUDE.md',
|
|
'src/pages/nearle/assistant/ROADMAP.md',
|
|
'src/pages/nearle/assistant/RAG_PLAN.md',
|
|
'src/pages/nearle/dispatch/CLAUDE.md',
|
|
'src/pages/nearle/orders/CLAUDE.md'
|
|
];
|
|
|
|
const chunkMarkdown = (text, source) => {
|
|
const out = [];
|
|
const lines = text.split('\n');
|
|
let heading = source;
|
|
let buffer = [];
|
|
|
|
const flush = () => {
|
|
const body = buffer.join('\n').trim();
|
|
buffer = [];
|
|
if (!body) return;
|
|
for (let i = 0; i < body.length; i += CHUNK_CHARS - OVERLAP) {
|
|
const slice = body.slice(i, i + CHUNK_CHARS).trim();
|
|
if (slice.length > 60) out.push({ heading, text: `${heading}\n\n${slice}` });
|
|
}
|
|
};
|
|
|
|
for (const line of lines) {
|
|
if (/^#{1,4}\s/.test(line)) {
|
|
flush();
|
|
heading = line.replace(/^#+\s*/, '').trim();
|
|
continue;
|
|
}
|
|
buffer.push(line);
|
|
}
|
|
flush();
|
|
return out;
|
|
};
|
|
|
|
const seedDocs = async () => {
|
|
const ids = [];
|
|
const documents = [];
|
|
const metadatas = [];
|
|
|
|
for (const rel of DOC_SOURCES) {
|
|
const abs = path.join(REPO, rel);
|
|
if (!fs.existsSync(abs)) {
|
|
console.warn(`[seed] skipped missing ${rel}`);
|
|
continue;
|
|
}
|
|
const chunks = chunkMarkdown(fs.readFileSync(abs, 'utf8'), rel);
|
|
chunks.forEach((c, i) => {
|
|
ids.push(`${rel}::${i}`);
|
|
documents.push(c.text);
|
|
metadatas.push({ source: rel, heading: c.heading, updatedAt: new Date().toISOString().slice(0, 10) });
|
|
});
|
|
}
|
|
|
|
const collection = await resetCollection(DOCS_COLLECTION);
|
|
const embeddings = await embedMany(documents);
|
|
// Chroma caps how much it will accept in one add; these corpora are small
|
|
// but batch anyway so this doesn't become a surprise later.
|
|
const BATCH = 200;
|
|
for (let i = 0; i < ids.length; i += BATCH) {
|
|
// eslint-disable-next-line no-await-in-loop
|
|
await collection.add({
|
|
ids: ids.slice(i, i + BATCH),
|
|
documents: documents.slice(i, i + BATCH),
|
|
metadatas: metadatas.slice(i, i + BATCH),
|
|
embeddings: embeddings.slice(i, i + BATCH)
|
|
});
|
|
}
|
|
return { sources: DOC_SOURCES.length, vectors: ids.length };
|
|
};
|
|
|
|
const main = async () => {
|
|
console.log(`[seed] model: ${MODEL_ID}`);
|
|
const intents = await seedIntents();
|
|
console.log(`[seed] intent_examples: ${intents.vectors} vectors across ${intents.intents} intents`);
|
|
const docs = await seedDocs();
|
|
console.log(`[seed] console_docs: ${docs.vectors} chunks from ${docs.sources} sources`);
|
|
console.log('[seed] done');
|
|
};
|
|
|
|
main().catch((err) => {
|
|
console.error('[seed] failed:', err.message);
|
|
process.exit(1);
|
|
});
|