Files
2026-08-19 17:08:45 +05:30

149 lines
5.4 KiB
JavaScript

import fs from 'fs';
import path from 'path';
import { fileURLToPath } from 'url';
import { INTENTS_COLLECTION, DOCS_COLLECTION, resetCollection } from '../collections.js';
import { embedMany, MODEL_ID } from '../embed.js';
// ==============================|| Seeding ||============================== //
//
// Rebuilds both collections from scratch. Idempotent — safe to re-run any time,
// and it MUST be re-run after changing the embedding model or editing any
// indexed document, or answers drift from the source without any error.
const HERE = path.dirname(fileURLToPath(import.meta.url));
const REPO = path.resolve(HERE, '../../..');
// Write intents get a stricter confidence bar in the console (a semantic
// near-miss must never open a create form), so they are flagged here.
const WRITE_INTENTS = new Set(['createOrder', 'createCustomer']);
const DOMAIN = {
orders: ['totalOrders', 'weekOrders', 'statusBreakdown', 'orderQuery', 'orderTrend', 'orderRate', 'delayedOrders', 'batchCount', 'revenueTotal', 'comparisonIntent', 'opsSummary', 'orderLookup', 'parcelTrack'],
riders: ['riderCounts', 'riderLookup', 'riderActivity'],
tenants: ['tenantList', 'tenantCount', 'tenantDetail'],
fleet: ['hubStatus', 'hubLookup', 'vehicleStatus', 'vehicleLookup', 'tripsheetStatus', 'exceptionStatus', 'consignmentStatus'],
admin: ['customerCount', 'appUserCount', 'pricingCount', 'partnerCount', 'competitorBranchCount', 'carrierPricingCount'],
write: ['createCustomer', 'createOrder']
};
const domainOf = (intentId) => Object.entries(DOMAIN).find(([, ids]) => ids.includes(intentId))?.[0] || 'other';
// ---- intent_examples --------------------------------------------------------
const seedIntents = async () => {
const raw = JSON.parse(fs.readFileSync(path.join(HERE, 'phrasings.json'), 'utf8'));
const ids = [];
const documents = [];
const metadatas = [];
for (const [intentId, phrasings] of Object.entries(raw)) {
if (intentId.startsWith('_')) continue;
phrasings.forEach((phrase, i) => {
ids.push(`${intentId}::${String(i).padStart(2, '0')}`);
documents.push(phrase);
metadatas.push({ intentId, domain: domainOf(intentId), isWrite: WRITE_INTENTS.has(intentId) });
});
}
const collection = await resetCollection(INTENTS_COLLECTION);
const embeddings = await embedMany(documents);
await collection.add({ ids, documents, metadatas, embeddings });
return { intents: new Set(metadatas.map((m) => m.intentId)).size, vectors: ids.length };
};
// ---- console_docs -----------------------------------------------------------
//
// Split on markdown headings, then hard-wrap long sections. The heading path is
// prepended to every chunk so a retrieved passage carries its own context —
// without it, a chunk reading "Don't do this" is worse than useless.
const CHUNK_CHARS = 800;
const OVERLAP = 100;
const DOC_SOURCES = [
'express-console-api.md',
'CLAUDE.md',
'src/pages/api/CLAUDE.md',
'src/pages/nearle/assistant/CLAUDE.md',
'src/pages/nearle/assistant/ROADMAP.md',
'src/pages/nearle/assistant/RAG_PLAN.md',
'src/pages/nearle/dispatch/CLAUDE.md',
'src/pages/nearle/orders/CLAUDE.md'
];
const chunkMarkdown = (text, source) => {
const out = [];
const lines = text.split('\n');
let heading = source;
let buffer = [];
const flush = () => {
const body = buffer.join('\n').trim();
buffer = [];
if (!body) return;
for (let i = 0; i < body.length; i += CHUNK_CHARS - OVERLAP) {
const slice = body.slice(i, i + CHUNK_CHARS).trim();
if (slice.length > 60) out.push({ heading, text: `${heading}\n\n${slice}` });
}
};
for (const line of lines) {
if (/^#{1,4}\s/.test(line)) {
flush();
heading = line.replace(/^#+\s*/, '').trim();
continue;
}
buffer.push(line);
}
flush();
return out;
};
const seedDocs = async () => {
const ids = [];
const documents = [];
const metadatas = [];
for (const rel of DOC_SOURCES) {
const abs = path.join(REPO, rel);
if (!fs.existsSync(abs)) {
console.warn(`[seed] skipped missing ${rel}`);
continue;
}
const chunks = chunkMarkdown(fs.readFileSync(abs, 'utf8'), rel);
chunks.forEach((c, i) => {
ids.push(`${rel}::${i}`);
documents.push(c.text);
metadatas.push({ source: rel, heading: c.heading, updatedAt: new Date().toISOString().slice(0, 10) });
});
}
const collection = await resetCollection(DOCS_COLLECTION);
const embeddings = await embedMany(documents);
// Chroma caps how much it will accept in one add; these corpora are small
// but batch anyway so this doesn't become a surprise later.
const BATCH = 200;
for (let i = 0; i < ids.length; i += BATCH) {
// eslint-disable-next-line no-await-in-loop
await collection.add({
ids: ids.slice(i, i + BATCH),
documents: documents.slice(i, i + BATCH),
metadatas: metadatas.slice(i, i + BATCH),
embeddings: embeddings.slice(i, i + BATCH)
});
}
return { sources: DOC_SOURCES.length, vectors: ids.length };
};
const main = async () => {
console.log(`[seed] model: ${MODEL_ID}`);
const intents = await seedIntents();
console.log(`[seed] intent_examples: ${intents.vectors} vectors across ${intents.intents} intents`);
const docs = await seedDocs();
console.log(`[seed] console_docs: ${docs.vectors} chunks from ${docs.sources} sources`);
console.log('[seed] done');
};
main().catch((err) => {
console.error('[seed] failed:', err.message);
process.exit(1);
});