import fs from 'fs'; import path from 'path'; import { fileURLToPath } from 'url'; import { INTENTS_COLLECTION, DOCS_COLLECTION, resetCollection } from '../collections.js'; import { embedMany, MODEL_ID } from '../embed.js'; // ==============================|| Seeding ||============================== // // // Rebuilds both collections from scratch. Idempotent — safe to re-run any time, // and it MUST be re-run after changing the embedding model or editing any // indexed document, or answers drift from the source without any error. const HERE = path.dirname(fileURLToPath(import.meta.url)); const REPO = path.resolve(HERE, '../../..'); // Write intents get a stricter confidence bar in the console (a semantic // near-miss must never open a create form), so they are flagged here. const WRITE_INTENTS = new Set(['createOrder', 'createCustomer']); const DOMAIN = { orders: ['totalOrders', 'weekOrders', 'statusBreakdown', 'orderQuery', 'orderTrend', 'orderRate', 'delayedOrders', 'batchCount', 'revenueTotal', 'comparisonIntent', 'opsSummary', 'orderLookup', 'parcelTrack'], riders: ['riderCounts', 'riderLookup', 'riderActivity'], tenants: ['tenantList', 'tenantCount', 'tenantDetail'], fleet: ['hubStatus', 'hubLookup', 'vehicleStatus', 'vehicleLookup', 'tripsheetStatus', 'exceptionStatus', 'consignmentStatus'], admin: ['customerCount', 'appUserCount', 'pricingCount', 'partnerCount', 'competitorBranchCount', 'carrierPricingCount'], write: ['createCustomer', 'createOrder'] }; const domainOf = (intentId) => Object.entries(DOMAIN).find(([, ids]) => ids.includes(intentId))?.[0] || 'other'; // ---- intent_examples -------------------------------------------------------- const seedIntents = async () => { const raw = JSON.parse(fs.readFileSync(path.join(HERE, 'phrasings.json'), 'utf8')); const ids = []; const documents = []; const metadatas = []; for (const [intentId, phrasings] of Object.entries(raw)) { if (intentId.startsWith('_')) continue; phrasings.forEach((phrase, i) => { ids.push(`${intentId}::${String(i).padStart(2, '0')}`); documents.push(phrase); metadatas.push({ intentId, domain: domainOf(intentId), isWrite: WRITE_INTENTS.has(intentId) }); }); } const collection = await resetCollection(INTENTS_COLLECTION); const embeddings = await embedMany(documents); await collection.add({ ids, documents, metadatas, embeddings }); return { intents: new Set(metadatas.map((m) => m.intentId)).size, vectors: ids.length }; }; // ---- console_docs ----------------------------------------------------------- // // Split on markdown headings, then hard-wrap long sections. The heading path is // prepended to every chunk so a retrieved passage carries its own context — // without it, a chunk reading "Don't do this" is worse than useless. const CHUNK_CHARS = 800; const OVERLAP = 100; const DOC_SOURCES = [ 'express-console-api.md', 'CLAUDE.md', 'src/pages/api/CLAUDE.md', 'src/pages/nearle/assistant/CLAUDE.md', 'src/pages/nearle/assistant/ROADMAP.md', 'src/pages/nearle/assistant/RAG_PLAN.md', 'src/pages/nearle/dispatch/CLAUDE.md', 'src/pages/nearle/orders/CLAUDE.md' ]; const chunkMarkdown = (text, source) => { const out = []; const lines = text.split('\n'); let heading = source; let buffer = []; const flush = () => { const body = buffer.join('\n').trim(); buffer = []; if (!body) return; for (let i = 0; i < body.length; i += CHUNK_CHARS - OVERLAP) { const slice = body.slice(i, i + CHUNK_CHARS).trim(); if (slice.length > 60) out.push({ heading, text: `${heading}\n\n${slice}` }); } }; for (const line of lines) { if (/^#{1,4}\s/.test(line)) { flush(); heading = line.replace(/^#+\s*/, '').trim(); continue; } buffer.push(line); } flush(); return out; }; const seedDocs = async () => { const ids = []; const documents = []; const metadatas = []; for (const rel of DOC_SOURCES) { const abs = path.join(REPO, rel); if (!fs.existsSync(abs)) { console.warn(`[seed] skipped missing ${rel}`); continue; } const chunks = chunkMarkdown(fs.readFileSync(abs, 'utf8'), rel); chunks.forEach((c, i) => { ids.push(`${rel}::${i}`); documents.push(c.text); metadatas.push({ source: rel, heading: c.heading, updatedAt: new Date().toISOString().slice(0, 10) }); }); } const collection = await resetCollection(DOCS_COLLECTION); const embeddings = await embedMany(documents); // Chroma caps how much it will accept in one add; these corpora are small // but batch anyway so this doesn't become a surprise later. const BATCH = 200; for (let i = 0; i < ids.length; i += BATCH) { // eslint-disable-next-line no-await-in-loop await collection.add({ ids: ids.slice(i, i + BATCH), documents: documents.slice(i, i + BATCH), metadatas: metadatas.slice(i, i + BATCH), embeddings: embeddings.slice(i, i + BATCH) }); } return { sources: DOC_SOURCES.length, vectors: ids.length }; }; const main = async () => { console.log(`[seed] model: ${MODEL_ID}`); const intents = await seedIntents(); console.log(`[seed] intent_examples: ${intents.vectors} vectors across ${intents.intents} intents`); const docs = await seedDocs(); console.log(`[seed] console_docs: ${docs.vectors} chunks from ${docs.sources} sources`); console.log('[seed] done'); }; main().catch((err) => { console.error('[seed] failed:', err.message); process.exit(1); });