implemenation on the bot
This commit is contained in:
59
services/ai/README.md
Normal file
59
services/ai/README.md
Normal file
@@ -0,0 +1,59 @@
|
||||
# Doormile AI — retrieval sidecar
|
||||
|
||||
Semantic intent routing and document Q&A over a local ChromaDB.
|
||||
|
||||
**This is dev tooling.** The production image is still the static nginx build.
|
||||
The console works with this stack absent — the assistant falls back to its
|
||||
deterministic matcher whenever `REACT_APP_AI_URL` is unset or unreachable.
|
||||
|
||||
## Run it
|
||||
|
||||
```bash
|
||||
docker compose up -d # chroma :8000, sidecar :8787
|
||||
docker compose exec ai-sidecar npm run seed
|
||||
curl localhost:8787/health
|
||||
```
|
||||
|
||||
Then point the console at it:
|
||||
|
||||
```
|
||||
REACT_APP_AI_URL=http://localhost:8787
|
||||
```
|
||||
|
||||
Leave that unset and nothing changes — the bot behaves exactly as it does today.
|
||||
|
||||
## What it does
|
||||
|
||||
| Endpoint | Purpose |
|
||||
|---|---|
|
||||
| `POST /route` | which intent is this question? + confidence |
|
||||
| `POST /ask` | which documentation passages answer this? |
|
||||
| `GET /health` | model + collection counts |
|
||||
|
||||
## What it deliberately does not do
|
||||
|
||||
- **No operational data is embedded.** No bookings, riders or customers. A
|
||||
vector store is a snapshot; this data changes by the minute. Every figure the
|
||||
operator sees still comes from a live API call.
|
||||
- **No generation.** `/ask` returns source passages verbatim with attribution.
|
||||
Summarising needs a hosted model — see `assistant/CLAUDE.md` §2.
|
||||
- **No key.** The embedding model runs in-process. That is what puts this
|
||||
outside §2's blocker.
|
||||
|
||||
## Evaluation
|
||||
|
||||
```bash
|
||||
docker compose exec ai-sidecar npm run eval
|
||||
```
|
||||
|
||||
Runs the **held-out** set in `eval-set.json` — phrasings that appear nowhere in
|
||||
`phrasings.json` and were not used to tune the thresholds. Retrieval always
|
||||
looks good against its own seed data, so this is the only number worth quoting.
|
||||
|
||||
Ship criteria: ≥80% accuracy and **zero false write routes**.
|
||||
|
||||
## After changing anything
|
||||
|
||||
Re-run `npm run seed` after editing `phrasings.json`, any indexed markdown, or
|
||||
the embedding model. Nothing errors if you forget — answers just quietly drift
|
||||
from the source, which is the worst kind of failure to chase.
|
||||
77
services/ai/collections.js
Normal file
77
services/ai/collections.js
Normal file
@@ -0,0 +1,77 @@
|
||||
import { ChromaClient } from 'chromadb';
|
||||
import { MODEL_ID, embed, embedMany } from './embed.js';
|
||||
|
||||
// ==============================|| Chroma collections ||============================== //
|
||||
//
|
||||
// Two collections, deliberately separate:
|
||||
//
|
||||
// intent_examples — ~20 phrasings per intent. Answers "which question is
|
||||
// this?", never "what is the number?".
|
||||
// console_docs — chunked markdown. Genuine document Q&A.
|
||||
//
|
||||
// NO OPERATIONAL DATA IS EMBEDDED. No bookings, riders or customers ever enter
|
||||
// the vector store. A vector store is a snapshot and this data changes by the
|
||||
// minute; every figure still comes from a live API call through the existing
|
||||
// typed functions. That is the whole reason this design is safe to build.
|
||||
|
||||
export const INTENTS_COLLECTION = 'intent_examples';
|
||||
export const DOCS_COLLECTION = 'console_docs';
|
||||
|
||||
const client = new ChromaClient({ path: process.env.CHROMA_URL || 'http://localhost:8000' });
|
||||
|
||||
// Chroma would otherwise call its own default embedding function (which
|
||||
// downloads a different model). We embed ourselves so queries and documents
|
||||
// are guaranteed to come from the same model.
|
||||
const noopEmbeddingFunction = { generate: async (texts) => embedMany(texts) };
|
||||
|
||||
export const getCollection = async (name) =>
|
||||
client.getOrCreateCollection({
|
||||
name,
|
||||
embeddingFunction: noopEmbeddingFunction,
|
||||
metadata: {
|
||||
// Stamped so a model swap is detectable rather than silently degrading
|
||||
// every score. Re-seed after changing it.
|
||||
'hnsw:space': 'cosine',
|
||||
embeddingModel: MODEL_ID
|
||||
}
|
||||
});
|
||||
|
||||
export const resetCollection = async (name) => {
|
||||
try {
|
||||
await client.deleteCollection({ name });
|
||||
} catch {
|
||||
// Not present yet — nothing to delete.
|
||||
}
|
||||
return getCollection(name);
|
||||
};
|
||||
|
||||
// Chroma returns cosine DISTANCE (0 = identical). Similarity is the useful
|
||||
// direction for a human-facing confidence, so convert once, here, rather than
|
||||
// leaving every call site to remember which way round it is.
|
||||
const toSimilarity = (distance) => 1 - Number(distance);
|
||||
|
||||
export const query = async (name, text, topK = 5) => {
|
||||
const collection = await getCollection(name);
|
||||
const vector = await embed(text);
|
||||
const res = await collection.query({ queryEmbeddings: [vector], nResults: topK });
|
||||
|
||||
const ids = res.ids?.[0] || [];
|
||||
return ids.map((id, i) => ({
|
||||
id,
|
||||
document: res.documents?.[0]?.[i] || '',
|
||||
metadata: res.metadatas?.[0]?.[i] || {},
|
||||
score: toSimilarity(res.distances?.[0]?.[i] ?? 1)
|
||||
}));
|
||||
};
|
||||
|
||||
export const health = async () => {
|
||||
await client.heartbeat();
|
||||
const names = (await client.listCollections()).map((c) => c.name ?? c);
|
||||
const counts = {};
|
||||
for (const name of [INTENTS_COLLECTION, DOCS_COLLECTION]) {
|
||||
if (!names.includes(name)) continue;
|
||||
// eslint-disable-next-line no-await-in-loop
|
||||
counts[name] = await (await getCollection(name)).count();
|
||||
}
|
||||
return { chroma: 'up', collections: counts };
|
||||
};
|
||||
36
services/ai/confidence.js
Normal file
36
services/ai/confidence.js
Normal file
@@ -0,0 +1,36 @@
|
||||
// ==============================|| Confidence ||============================== //
|
||||
//
|
||||
// Its own module, not part of index.js, because index.js calls app.listen() at
|
||||
// module scope — importing it from eval.js would boot a second HTTP server as
|
||||
// a side effect of running the evaluation.
|
||||
//
|
||||
// ---- Why margin and not similarity -----------------------------------------
|
||||
//
|
||||
// Cosine similarity is NOT a probability of correctness. A score of 0.87 does
|
||||
// not mean "87% likely right", and showing it as though it does would be the
|
||||
// same false precision this assistant avoids everywhere else.
|
||||
//
|
||||
// What carries information is the MARGIN between the best and second-best
|
||||
// match. A wide margin means the question is unambiguous. A narrow one means it
|
||||
// genuinely could be two things — and that is exactly when the bot should ask
|
||||
// instead of guessing.
|
||||
//
|
||||
// The answer's correctness never comes from this number. It comes from the
|
||||
// intent's deterministic run() hitting a real endpoint.
|
||||
|
||||
// Starting values. Tune from eval.js output, not intuition.
|
||||
export const HIGH_SCORE = 0.75;
|
||||
export const HIGH_MARGIN = 0.1;
|
||||
export const MED_SCORE = 0.6;
|
||||
export const MED_MARGIN = 0.05;
|
||||
|
||||
export const classify = (hits) => {
|
||||
if (!hits.length) return { confidence: 'low', score: 0, margin: 0 };
|
||||
const score = hits[0].score;
|
||||
// With a single hit there is nothing to be ambiguous against, so the margin
|
||||
// is the score itself rather than a fabricated 0.
|
||||
const margin = hits.length > 1 ? score - hits[1].score : score;
|
||||
if (score >= HIGH_SCORE && margin >= HIGH_MARGIN) return { confidence: 'high', score, margin };
|
||||
if (score >= MED_SCORE && margin >= MED_MARGIN) return { confidence: 'medium', score, margin };
|
||||
return { confidence: 'low', score, margin };
|
||||
};
|
||||
66
services/ai/embed.js
Normal file
66
services/ai/embed.js
Normal file
@@ -0,0 +1,66 @@
|
||||
import { pipeline, env } from '@xenova/transformers';
|
||||
|
||||
// ==============================|| Embeddings ||============================== //
|
||||
//
|
||||
// all-MiniLM-L6-v2 running IN PROCESS. No API key, no network call per query,
|
||||
// no per-call cost.
|
||||
//
|
||||
// That is not an incidental choice. assistant/CLAUDE.md §2 blocked LLM work on
|
||||
// exactly one ground: a hosted model needs a secret key and a static CRA build
|
||||
// has nowhere to put one. A local embedding model has no key, so this plan is
|
||||
// outside that blocker. Swapping to a hosted embedding model (text-embedding-3
|
||||
// -small et al) re-opens §2 and needs its own decision — do not do it quietly.
|
||||
|
||||
export const MODEL_ID = 'Xenova/all-MiniLM-L6-v2';
|
||||
export const DIMENSIONS = 384;
|
||||
|
||||
// Weights are cached on disk (mounted in docker-compose) so a container
|
||||
// restart doesn't re-download 25MB.
|
||||
env.cacheDir = process.env.TRANSFORMERS_CACHE || './.cache';
|
||||
// Nothing here should reach the network except the one-time model fetch.
|
||||
env.allowRemoteModels = true;
|
||||
|
||||
let extractor = null;
|
||||
let loading = null;
|
||||
|
||||
// Loaded once, lazily, and shared. Concurrent callers await the same promise
|
||||
// rather than each triggering their own model load.
|
||||
const getExtractor = async () => {
|
||||
if (extractor) return extractor;
|
||||
if (!loading) {
|
||||
loading = pipeline('feature-extraction', MODEL_ID).then((p) => {
|
||||
extractor = p;
|
||||
return p;
|
||||
});
|
||||
}
|
||||
return loading;
|
||||
};
|
||||
|
||||
export const warmUp = async () => {
|
||||
await getExtractor();
|
||||
return { model: MODEL_ID, dimensions: DIMENSIONS };
|
||||
};
|
||||
|
||||
// Mean-pooled + L2-normalised sentence embedding.
|
||||
//
|
||||
// Normalisation matters: Chroma's cosine space assumes unit vectors, and
|
||||
// queries must be embedded exactly the same way documents were. A mismatch
|
||||
// doesn't error — it silently degrades every score, which is the worst kind of
|
||||
// bug to chase. `seed` writes the model id into collection metadata so a
|
||||
// mismatch is at least detectable.
|
||||
export const embed = async (text) => {
|
||||
const pipe = await getExtractor();
|
||||
const output = await pipe(String(text || '').trim(), { pooling: 'mean', normalize: true });
|
||||
return Array.from(output.data);
|
||||
};
|
||||
|
||||
export const embedMany = async (texts) => {
|
||||
const out = [];
|
||||
for (const t of texts) {
|
||||
// Sequential on purpose: batching MiniLM in-process gives no meaningful
|
||||
// speed-up at this corpus size (~700 vectors) and makes memory spikier.
|
||||
// eslint-disable-next-line no-await-in-loop
|
||||
out.push(await embed(t));
|
||||
}
|
||||
return out;
|
||||
};
|
||||
77
services/ai/eval-set.json
Normal file
77
services/ai/eval-set.json
Normal file
@@ -0,0 +1,77 @@
|
||||
{
|
||||
"_comment": "HELD-OUT evaluation set. None of these phrasings appear in phrasings.json, and none were used to choose the confidence thresholds. That is the entire point — retrieval will always score well against its own seed data, so this is the only number worth reporting. Written as an operator would speak, not as the regex matchers were written. If you add a case here, do NOT then add it to phrasings.json; that would quietly turn the held-out set into training data.",
|
||||
|
||||
"cases": [
|
||||
{ "text": "how much did we ship today", "expect": "totalOrders" },
|
||||
{ "text": "did we get many bookings", "expect": "totalOrders" },
|
||||
{ "text": "whats the tally for today", "expect": "totalOrders" },
|
||||
|
||||
{ "text": "seven day totals", "expect": "weekOrders" },
|
||||
{ "text": "how did the week go", "expect": "weekOrders" },
|
||||
|
||||
{ "text": "anything not picked up yet", "expect": "statusBreakdown" },
|
||||
{ "text": "count the scrapped ones", "expect": "statusBreakdown" },
|
||||
{ "text": "how many made it to the customer", "expect": "statusBreakdown" },
|
||||
{ "text": "orders sitting with no rider", "expect": "statusBreakdown" },
|
||||
|
||||
{ "text": "is anything running behind", "expect": "delayedOrders" },
|
||||
{ "text": "show me what missed the window", "expect": "delayedOrders" },
|
||||
{ "text": "are we late anywhere", "expect": "delayedOrders" },
|
||||
{ "text": "which sites are struggling", "expect": "delayedOrders" },
|
||||
|
||||
{ "text": "money in today", "expect": "revenueTotal" },
|
||||
{ "text": "what did we bill this week", "expect": "revenueTotal" },
|
||||
|
||||
{ "text": "how does this week stack up against last", "expect": "comparisonIntent" },
|
||||
{ "text": "better or worse than yesterday", "expect": "comparisonIntent" },
|
||||
|
||||
{ "text": "spread of orders through the day", "expect": "orderTrend" },
|
||||
{ "text": "what time are we busiest", "expect": "orderTrend" },
|
||||
|
||||
{ "text": "what share get scrapped", "expect": "orderRate" },
|
||||
{ "text": "how much of our volume completes", "expect": "orderRate" },
|
||||
|
||||
{ "text": "run me the numbers for today", "expect": "opsSummary" },
|
||||
{ "text": "quick rundown please", "expect": "opsSummary" },
|
||||
|
||||
{ "text": "how many lads are working", "expect": "riderCounts" },
|
||||
{ "text": "anyone free to take a job", "expect": "riderCounts" },
|
||||
{ "text": "whats the driver situation", "expect": "riderCounts" },
|
||||
|
||||
{ "text": "how did Kumar get on", "expect": "riderActivity" },
|
||||
{ "text": "how many drops did Suresh finish", "expect": "riderActivity" },
|
||||
{ "text": "how many jobs did rider Ali turn down", "expect": "riderActivity" },
|
||||
|
||||
{ "text": "which client sends us the most work", "expect": "orderQuery" },
|
||||
{ "text": "Acme deliveries that completed this week", "expect": "orderQuery" },
|
||||
|
||||
{ "text": "who are our clients", "expect": "tenantList" },
|
||||
{ "text": "count the businesses we serve", "expect": "tenantList" },
|
||||
|
||||
{ "text": "what have we got on Acme Foods", "expect": "tenantDetail" },
|
||||
|
||||
{ "text": "are the depots all running", "expect": "hubStatus" },
|
||||
{ "text": "how many depots do we run", "expect": "hubStatus" },
|
||||
|
||||
{ "text": "how many bikes are spare", "expect": "vehicleStatus" },
|
||||
{ "text": "whats sitting idle in the fleet", "expect": "vehicleStatus" },
|
||||
|
||||
{ "text": "any failed drops", "expect": "exceptionStatus" },
|
||||
{ "text": "what problems came up", "expect": "exceptionStatus" },
|
||||
|
||||
{ "text": "how many runs are out", "expect": "tripsheetStatus" },
|
||||
|
||||
{ "text": "how many people are on our books", "expect": "customerCount" },
|
||||
|
||||
{ "text": "whats the state of order 4821", "expect": "orderLookup" },
|
||||
{ "text": "chase up DM-BK-900 for me", "expect": "orderLookup" },
|
||||
|
||||
{ "text": "follow parcel DM-CN-77", "expect": "parcelTrack" },
|
||||
|
||||
{ "text": "I want to add someone new", "expect": "createCustomer", "isWrite": true },
|
||||
{ "text": "put a new client on the system", "expect": "createCustomer", "isWrite": true },
|
||||
|
||||
{ "text": "I need to raise a job", "expect": "createOrder", "isWrite": true },
|
||||
{ "text": "set up a delivery for me", "expect": "createOrder", "isWrite": true }
|
||||
]
|
||||
}
|
||||
79
services/ai/eval.js
Normal file
79
services/ai/eval.js
Normal file
@@ -0,0 +1,79 @@
|
||||
import { INTENTS_COLLECTION, query } from './collections.js';
|
||||
import fs from 'fs';
|
||||
import path from 'path';
|
||||
import { fileURLToPath } from 'url';
|
||||
import { classify } from './confidence.js';
|
||||
|
||||
// Read rather than `import ... assert`: the import-assertion syntax changed
|
||||
// between Node 20 (`assert`) and Node 22 (`with`), and this has to run on both.
|
||||
const HERE = path.dirname(fileURLToPath(import.meta.url));
|
||||
const heldOut = JSON.parse(fs.readFileSync(path.join(HERE, 'eval-set.json'), 'utf8'));
|
||||
|
||||
// ==============================|| Routing evaluation ||============================== //
|
||||
//
|
||||
// The question this answers is NOT "does retrieval work" — it will always look
|
||||
// good against the phrasings it was seeded with. It is:
|
||||
//
|
||||
// does it route phrasings NOBODY tuned it against?
|
||||
//
|
||||
// eval-set.json is deliberately held out: none of these appear in
|
||||
// phrasings.json, and none were used to pick the confidence thresholds.
|
||||
//
|
||||
// Ship criteria (RAG_PLAN.md §7):
|
||||
// • accuracy and coverage beat the regex baseline
|
||||
// • ZERO false write routes — a wrong create is the worst failure here,
|
||||
// the same class as answering "0" when the page shows 19.
|
||||
|
||||
const pct = (n, d) => (d ? `${((n / d) * 100).toFixed(1)}%` : '—');
|
||||
|
||||
const main = async () => {
|
||||
const cases = heldOut.cases || [];
|
||||
let correct = 0;
|
||||
let answered = 0;
|
||||
let falseWrite = 0;
|
||||
const wrong = [];
|
||||
const latencies = [];
|
||||
|
||||
for (const c of cases) {
|
||||
const started = Date.now();
|
||||
// eslint-disable-next-line no-await-in-loop
|
||||
const hits = await query(INTENTS_COLLECTION, c.text, 5);
|
||||
latencies.push(Date.now() - started);
|
||||
|
||||
const { confidence, score, margin } = classify(hits);
|
||||
const top = hits[0]?.metadata || {};
|
||||
const routed = confidence === 'low' ? null : top.intentId;
|
||||
|
||||
if (routed) answered += 1;
|
||||
if (routed === c.expect) correct += 1;
|
||||
else if (routed) wrong.push({ text: c.text, expected: c.expect, got: routed, score: score.toFixed(2), margin: margin.toFixed(2) });
|
||||
|
||||
// A non-write phrasing that routes to a write intent is the failure that
|
||||
// matters most — it would open a create form the operator never asked for.
|
||||
if (top.isWrite && !c.isWrite && confidence === 'high') falseWrite += 1;
|
||||
}
|
||||
|
||||
latencies.sort((a, b) => a - b);
|
||||
const p = (q) => latencies[Math.floor(latencies.length * q)] ?? 0;
|
||||
|
||||
console.log('\n=== held-out routing evaluation ===');
|
||||
console.log(`cases ${cases.length}`);
|
||||
console.log(`accuracy ${correct}/${cases.length} ${pct(correct, cases.length)}`);
|
||||
console.log(`coverage ${answered}/${cases.length} ${pct(answered, cases.length)} (answered at all)`);
|
||||
console.log(`false writes ${falseWrite} ${falseWrite === 0 ? '✓' : '✗ MUST BE ZERO'}`);
|
||||
console.log(`latency p50/p95 ${p(0.5)}ms / ${p(0.95)}ms`);
|
||||
|
||||
if (wrong.length) {
|
||||
console.log('\n--- misroutes ---');
|
||||
wrong.forEach((w) => console.log(` "${w.text}"\n expected ${w.expected}, got ${w.got} (score ${w.score}, margin ${w.margin})`));
|
||||
}
|
||||
|
||||
const ok = falseWrite === 0 && correct / Math.max(cases.length, 1) >= 0.8;
|
||||
console.log(`\n${ok ? 'PASS' : 'FAIL'} — threshold: >=80% accuracy and zero false writes\n`);
|
||||
process.exit(ok ? 0 : 1);
|
||||
};
|
||||
|
||||
main().catch((err) => {
|
||||
console.error('[eval] failed:', err.message);
|
||||
process.exit(1);
|
||||
});
|
||||
97
services/ai/index.js
Normal file
97
services/ai/index.js
Normal file
@@ -0,0 +1,97 @@
|
||||
import express from 'express';
|
||||
import cors from 'cors';
|
||||
import { warmUp } from './embed.js';
|
||||
import { INTENTS_COLLECTION, DOCS_COLLECTION, query, health } from './collections.js';
|
||||
import { classify } from './confidence.js';
|
||||
|
||||
// ==============================|| Doormile AI — retrieval sidecar ||============================== //
|
||||
//
|
||||
// Two jobs, and neither of them is "produce a number":
|
||||
//
|
||||
// POST /route text -> which intent is this? (+ how sure)
|
||||
// POST /ask text -> which documentation passages answer this?
|
||||
//
|
||||
// The console then runs its OWN deterministic intent code to produce every
|
||||
// figure. Nothing this service returns is ever read by an operator as data.
|
||||
|
||||
const app = express();
|
||||
app.use(cors());
|
||||
app.use(express.json({ limit: '64kb' }));
|
||||
|
||||
app.get('/health', async (_req, res) => {
|
||||
try {
|
||||
const model = await warmUp();
|
||||
const store = await health();
|
||||
res.json({ ok: true, model, ...store });
|
||||
} catch (err) {
|
||||
res.status(503).json({ ok: false, error: err.message });
|
||||
}
|
||||
});
|
||||
|
||||
app.post('/route', async (req, res) => {
|
||||
const text = String(req.body?.text || '').trim();
|
||||
if (!text) return res.status(400).json({ error: 'text is required' });
|
||||
|
||||
try {
|
||||
const hits = await query(INTENTS_COLLECTION, text, 5);
|
||||
const { confidence, score, margin } = classify(hits);
|
||||
|
||||
// Distinct intents only — five phrasings of the same intent in the top-5
|
||||
// is a strong signal, not five alternatives to offer the operator.
|
||||
const seen = new Set();
|
||||
const alternatives = [];
|
||||
for (const h of hits) {
|
||||
const id = h.metadata?.intentId;
|
||||
if (!id || seen.has(id)) continue;
|
||||
seen.add(id);
|
||||
alternatives.push({ intentId: id, score: h.score, example: h.document });
|
||||
}
|
||||
|
||||
return res.json({
|
||||
intentId: alternatives[0]?.intentId ?? null,
|
||||
isWrite: Boolean(hits[0]?.metadata?.isWrite),
|
||||
confidence,
|
||||
score: Number(score.toFixed(4)),
|
||||
margin: Number(margin.toFixed(4)),
|
||||
matchedExample: hits[0]?.document ?? null,
|
||||
// Only meaningful when confidence is low — that is when the panel should
|
||||
// offer these as buttons instead of picking one.
|
||||
alternatives: alternatives.slice(1, 3)
|
||||
});
|
||||
} catch (err) {
|
||||
return res.status(503).json({ error: err.message });
|
||||
}
|
||||
});
|
||||
|
||||
app.post('/ask', async (req, res) => {
|
||||
const text = String(req.body?.text || '').trim();
|
||||
if (!text) return res.status(400).json({ error: 'text is required' });
|
||||
|
||||
try {
|
||||
const hits = await query(DOCS_COLLECTION, text, 4);
|
||||
// Passages are returned verbatim with attribution. There is no generation
|
||||
// step: summarising would need a hosted model (CLAUDE.md §2) and would let
|
||||
// a paraphrase drift from what the document actually says.
|
||||
res.json({
|
||||
chunks: hits
|
||||
.filter((h) => h.score > 0.35)
|
||||
.map((h) => ({
|
||||
text: h.document,
|
||||
source: h.metadata?.source,
|
||||
heading: h.metadata?.heading,
|
||||
score: Number(h.score.toFixed(4))
|
||||
}))
|
||||
});
|
||||
} catch (err) {
|
||||
res.status(503).json({ error: err.message });
|
||||
}
|
||||
});
|
||||
|
||||
const port = Number(process.env.PORT) || 8787;
|
||||
app.listen(port, () => {
|
||||
console.log(`[doormile-ai] listening on :${port}`);
|
||||
// Pay the model load at boot rather than on an operator's first question.
|
||||
warmUp()
|
||||
.then((m) => console.log(`[doormile-ai] model ready: ${m.model} (${m.dimensions}d)`))
|
||||
.catch((e) => console.error('[doormile-ai] model failed to load:', e.message));
|
||||
});
|
||||
18
services/ai/package.json
Normal file
18
services/ai/package.json
Normal file
@@ -0,0 +1,18 @@
|
||||
{
|
||||
"name": "doormile-ai-sidecar",
|
||||
"version": "0.1.0",
|
||||
"private": true,
|
||||
"type": "module",
|
||||
"description": "Retrieval sidecar for the Doormile AI assistant — semantic intent routing and document Q&A over a local ChromaDB. Deliberately a separate package: the CRA app's dependency tree and webpack/svgr resolutions must not be touched (root CLAUDE.md §4.3).",
|
||||
"scripts": {
|
||||
"start": "node index.js",
|
||||
"seed": "node seed/run.js",
|
||||
"eval": "node eval.js"
|
||||
},
|
||||
"dependencies": {
|
||||
"@xenova/transformers": "^2.17.2",
|
||||
"chromadb": "^1.9.2",
|
||||
"cors": "^2.8.5",
|
||||
"express": "^4.19.2"
|
||||
}
|
||||
}
|
||||
271
services/ai/seed/phrasings.json
Normal file
271
services/ai/seed/phrasings.json
Normal file
@@ -0,0 +1,271 @@
|
||||
{
|
||||
"_comment": "Example phrasings per intent, used to seed the intent_examples collection. These are deliberately OPERATOR language, not the developer language the regex matchers were written against — the whole point is to cover the phrasings nobody thought to write a pattern for. Do not add a phrasing here that the intent cannot actually answer.",
|
||||
|
||||
"totalOrders": [
|
||||
"how many orders today",
|
||||
"how many bookings do we have today",
|
||||
"order count for today",
|
||||
"what's today's order volume",
|
||||
"how many orders came in",
|
||||
"total orders today",
|
||||
"how many jobs today",
|
||||
"how much work came in today",
|
||||
"give me the order count",
|
||||
"how many did we get today",
|
||||
"orders received today",
|
||||
"number of orders"
|
||||
],
|
||||
|
||||
"weekOrders": [
|
||||
"how many orders this week",
|
||||
"orders for the week",
|
||||
"weekly order count",
|
||||
"how many orders last week",
|
||||
"orders this month",
|
||||
"how many bookings this month",
|
||||
"order volume this week",
|
||||
"what did we do last week",
|
||||
"how many orders in the last 7 days"
|
||||
],
|
||||
|
||||
"statusBreakdown": [
|
||||
"how many delivered orders today",
|
||||
"how many cancelled orders",
|
||||
"how many pending orders",
|
||||
"how many orders are assigned",
|
||||
"count of unassigned orders",
|
||||
"how many orders are out for delivery",
|
||||
"how many got cancelled today",
|
||||
"what's still pending",
|
||||
"how many are waiting to be picked up",
|
||||
"how many orders are in transit",
|
||||
"how many did we deliver",
|
||||
"how many orders got rejected"
|
||||
],
|
||||
|
||||
"orderQuery": [
|
||||
"delivered orders for Acme this week",
|
||||
"cancelled orders for Acme today",
|
||||
"how many orders did rider Suresh deliver today",
|
||||
"morning batch orders for Acme",
|
||||
"top 5 tenants by orders this week",
|
||||
"which tenant has the most orders",
|
||||
"busiest riders this week",
|
||||
"pending orders for Acme Foods",
|
||||
"rank tenants by order volume",
|
||||
"who delivered the most today"
|
||||
],
|
||||
|
||||
"orderTrend": [
|
||||
"orders per day this week",
|
||||
"orders per hour today",
|
||||
"daily order trend",
|
||||
"hourly breakdown of orders",
|
||||
"how are orders spread across the day",
|
||||
"show me the order trend",
|
||||
"when do most orders come in",
|
||||
"orders by day for last week",
|
||||
"what's our busiest hour"
|
||||
],
|
||||
|
||||
"orderRate": [
|
||||
"cancellation rate this week",
|
||||
"what percentage of orders get cancelled",
|
||||
"delivery success rate",
|
||||
"what percent were delivered",
|
||||
"cancellation percentage",
|
||||
"ratio of cancelled orders",
|
||||
"how often do orders get cancelled",
|
||||
"what's our delivery rate"
|
||||
],
|
||||
|
||||
"delayedOrders": [
|
||||
"which orders are delayed",
|
||||
"anything running late",
|
||||
"what's overdue",
|
||||
"how many orders are late",
|
||||
"which hubs are experiencing delays",
|
||||
"anything stuck",
|
||||
"are we behind on anything",
|
||||
"show me late deliveries",
|
||||
"orders past their delivery time",
|
||||
"what's at risk right now",
|
||||
"any SLA breaches",
|
||||
"which orders missed their promise"
|
||||
],
|
||||
|
||||
"batchCount": [
|
||||
"morning batch orders today",
|
||||
"how many in the afternoon batch",
|
||||
"evening batch count",
|
||||
"orders in the morning wave",
|
||||
"what's in the afternoon run",
|
||||
"how many orders in tonight's batch"
|
||||
],
|
||||
|
||||
"revenueTotal": [
|
||||
"total revenue today",
|
||||
"how much did we make today",
|
||||
"revenue this week",
|
||||
"what were the earnings today",
|
||||
"total charges for the week",
|
||||
"how much money came in",
|
||||
"collections today",
|
||||
"what's the total value of today's orders"
|
||||
],
|
||||
|
||||
"comparisonIntent": [
|
||||
"orders today vs yesterday",
|
||||
"compare this week to last week",
|
||||
"revenue this week compared to last week",
|
||||
"how does today compare to yesterday",
|
||||
"are we up or down on last week",
|
||||
"this month versus last month"
|
||||
],
|
||||
|
||||
"opsSummary": [
|
||||
"give me today's operations summary",
|
||||
"how are we doing today",
|
||||
"daily overview",
|
||||
"what's the situation right now",
|
||||
"summarise today",
|
||||
"ops snapshot",
|
||||
"give me the headline numbers",
|
||||
"how's everything looking",
|
||||
"today at a glance"
|
||||
],
|
||||
|
||||
"orderLookup": [
|
||||
"status of order #1234",
|
||||
"where is order DM-BK-123",
|
||||
"what happened to order 4821",
|
||||
"look up booking DM-BK-900",
|
||||
"check order #55",
|
||||
"is order 1234 delivered yet",
|
||||
"who's got order #4821"
|
||||
],
|
||||
|
||||
"parcelTrack": [
|
||||
"track consignment DM-CN-123",
|
||||
"where is parcel DM-CN-99",
|
||||
"track shipment ABC-123",
|
||||
"scan history for DM-CN-123",
|
||||
"what's happened to this consignment"
|
||||
],
|
||||
|
||||
"riderCounts": [
|
||||
"how many riders are active",
|
||||
"how many riders do we have out",
|
||||
"rider availability",
|
||||
"how many are on delivery",
|
||||
"who's available right now",
|
||||
"how many riders are free",
|
||||
"fleet status",
|
||||
"how many riders are offline"
|
||||
],
|
||||
|
||||
"riderLookup": [
|
||||
"where is rider Kumar",
|
||||
"find rider Suresh",
|
||||
"status of rider Ali",
|
||||
"look up rider Priya",
|
||||
"is Kumar online"
|
||||
],
|
||||
|
||||
"riderActivity": [
|
||||
"how is rider Kumar doing today",
|
||||
"rider Suresh performance",
|
||||
"how'd Kumar do today",
|
||||
"how many did rider Ali complete",
|
||||
"rider Priya stats",
|
||||
"how many did Kumar reject",
|
||||
"how far has rider Suresh travelled",
|
||||
"activity for rider Kumar"
|
||||
],
|
||||
|
||||
"tenantList": ["how many tenants do we have", "list all tenants", "how many clients", "show me our tenants", "tenant count"],
|
||||
|
||||
"tenantCount": [
|
||||
"orders for Acme today",
|
||||
"how many orders did Acme place",
|
||||
"Acme Foods order count",
|
||||
"how much work from Beta Kitchens"
|
||||
],
|
||||
|
||||
"tenantDetail": [
|
||||
"tell me about tenant Acme Foods",
|
||||
"details for Acme",
|
||||
"what do we know about Beta Kitchens",
|
||||
"Acme Foods locations",
|
||||
"show me tenant Acme"
|
||||
],
|
||||
|
||||
"hubStatus": [
|
||||
"current hub status",
|
||||
"how many hubs do we have",
|
||||
"are all hubs active",
|
||||
"hub overview",
|
||||
"which hubs are open"
|
||||
],
|
||||
|
||||
"hubLookup": ["status of hub Chennai", "find hub Coimbatore", "look up the Bengaluru hub"],
|
||||
|
||||
"vehicleStatus": [
|
||||
"how many vehicles are available",
|
||||
"vehicle count",
|
||||
"how many bikes do we have",
|
||||
"fleet vehicle status",
|
||||
"what vehicles are free"
|
||||
],
|
||||
|
||||
"vehicleLookup": ["find vehicle TN01AB1234", "status of vehicle TN37XY9", "look up vehicle KA05MM1"],
|
||||
|
||||
"tripsheetStatus": [
|
||||
"how many tripsheets are dispatched",
|
||||
"tripsheet count",
|
||||
"how many trips are running",
|
||||
"open tripsheets"
|
||||
],
|
||||
|
||||
"exceptionStatus": [
|
||||
"how many open exceptions",
|
||||
"any delivery exceptions",
|
||||
"what's gone wrong",
|
||||
"exception count",
|
||||
"failed deliveries"
|
||||
],
|
||||
|
||||
"consignmentStatus": ["how many consignments do we have", "consignment count", "total consignments"],
|
||||
|
||||
"customerCount": ["how many customers do we have", "customer count", "how many people have ordered", "total customers"],
|
||||
|
||||
"appUserCount": ["how many app users do we have", "staff login count", "how many console users"],
|
||||
|
||||
"pricingCount": ["how many pricing rules are configured", "pricing rule count", "how many tariffs do we have"],
|
||||
|
||||
"partnerCount": ["how many partners do we have", "partner count", "list our fleet partners"],
|
||||
|
||||
"competitorBranchCount": ["how many competitor branches are tracked", "competitor branch count", "competitive intel count"],
|
||||
|
||||
"carrierPricingCount": ["how many carrier pricing rules", "carrier tariff count", "carrier pricing entries"],
|
||||
|
||||
"createCustomer": [
|
||||
"create a customer",
|
||||
"add a new customer",
|
||||
"register a client",
|
||||
"I need to add a customer",
|
||||
"new customer please",
|
||||
"set up a customer record",
|
||||
"add customer Ramesh 9876543210"
|
||||
],
|
||||
|
||||
"createOrder": [
|
||||
"create an order",
|
||||
"book a delivery",
|
||||
"place a new order",
|
||||
"I need to create a booking",
|
||||
"new delivery please",
|
||||
"raise an order",
|
||||
"add a booking"
|
||||
]
|
||||
}
|
||||
148
services/ai/seed/run.js
Normal file
148
services/ai/seed/run.js
Normal file
@@ -0,0 +1,148 @@
|
||||
import fs from 'fs';
|
||||
import path from 'path';
|
||||
import { fileURLToPath } from 'url';
|
||||
import { INTENTS_COLLECTION, DOCS_COLLECTION, resetCollection } from '../collections.js';
|
||||
import { embedMany, MODEL_ID } from '../embed.js';
|
||||
|
||||
// ==============================|| Seeding ||============================== //
|
||||
//
|
||||
// Rebuilds both collections from scratch. Idempotent — safe to re-run any time,
|
||||
// and it MUST be re-run after changing the embedding model or editing any
|
||||
// indexed document, or answers drift from the source without any error.
|
||||
|
||||
const HERE = path.dirname(fileURLToPath(import.meta.url));
|
||||
const REPO = path.resolve(HERE, '../../..');
|
||||
|
||||
// Write intents get a stricter confidence bar in the console (a semantic
|
||||
// near-miss must never open a create form), so they are flagged here.
|
||||
const WRITE_INTENTS = new Set(['createOrder', 'createCustomer']);
|
||||
|
||||
const DOMAIN = {
|
||||
orders: ['totalOrders', 'weekOrders', 'statusBreakdown', 'orderQuery', 'orderTrend', 'orderRate', 'delayedOrders', 'batchCount', 'revenueTotal', 'comparisonIntent', 'opsSummary', 'orderLookup', 'parcelTrack'],
|
||||
riders: ['riderCounts', 'riderLookup', 'riderActivity'],
|
||||
tenants: ['tenantList', 'tenantCount', 'tenantDetail'],
|
||||
fleet: ['hubStatus', 'hubLookup', 'vehicleStatus', 'vehicleLookup', 'tripsheetStatus', 'exceptionStatus', 'consignmentStatus'],
|
||||
admin: ['customerCount', 'appUserCount', 'pricingCount', 'partnerCount', 'competitorBranchCount', 'carrierPricingCount'],
|
||||
write: ['createCustomer', 'createOrder']
|
||||
};
|
||||
|
||||
const domainOf = (intentId) => Object.entries(DOMAIN).find(([, ids]) => ids.includes(intentId))?.[0] || 'other';
|
||||
|
||||
// ---- intent_examples --------------------------------------------------------
|
||||
const seedIntents = async () => {
|
||||
const raw = JSON.parse(fs.readFileSync(path.join(HERE, 'phrasings.json'), 'utf8'));
|
||||
const ids = [];
|
||||
const documents = [];
|
||||
const metadatas = [];
|
||||
|
||||
for (const [intentId, phrasings] of Object.entries(raw)) {
|
||||
if (intentId.startsWith('_')) continue;
|
||||
phrasings.forEach((phrase, i) => {
|
||||
ids.push(`${intentId}::${String(i).padStart(2, '0')}`);
|
||||
documents.push(phrase);
|
||||
metadatas.push({ intentId, domain: domainOf(intentId), isWrite: WRITE_INTENTS.has(intentId) });
|
||||
});
|
||||
}
|
||||
|
||||
const collection = await resetCollection(INTENTS_COLLECTION);
|
||||
const embeddings = await embedMany(documents);
|
||||
await collection.add({ ids, documents, metadatas, embeddings });
|
||||
return { intents: new Set(metadatas.map((m) => m.intentId)).size, vectors: ids.length };
|
||||
};
|
||||
|
||||
// ---- console_docs -----------------------------------------------------------
|
||||
//
|
||||
// Split on markdown headings, then hard-wrap long sections. The heading path is
|
||||
// prepended to every chunk so a retrieved passage carries its own context —
|
||||
// without it, a chunk reading "Don't do this" is worse than useless.
|
||||
const CHUNK_CHARS = 800;
|
||||
const OVERLAP = 100;
|
||||
|
||||
const DOC_SOURCES = [
|
||||
'express-console-api.md',
|
||||
'CLAUDE.md',
|
||||
'src/pages/api/CLAUDE.md',
|
||||
'src/pages/nearle/assistant/CLAUDE.md',
|
||||
'src/pages/nearle/assistant/ROADMAP.md',
|
||||
'src/pages/nearle/assistant/RAG_PLAN.md',
|
||||
'src/pages/nearle/dispatch/CLAUDE.md',
|
||||
'src/pages/nearle/orders/CLAUDE.md'
|
||||
];
|
||||
|
||||
const chunkMarkdown = (text, source) => {
|
||||
const out = [];
|
||||
const lines = text.split('\n');
|
||||
let heading = source;
|
||||
let buffer = [];
|
||||
|
||||
const flush = () => {
|
||||
const body = buffer.join('\n').trim();
|
||||
buffer = [];
|
||||
if (!body) return;
|
||||
for (let i = 0; i < body.length; i += CHUNK_CHARS - OVERLAP) {
|
||||
const slice = body.slice(i, i + CHUNK_CHARS).trim();
|
||||
if (slice.length > 60) out.push({ heading, text: `${heading}\n\n${slice}` });
|
||||
}
|
||||
};
|
||||
|
||||
for (const line of lines) {
|
||||
if (/^#{1,4}\s/.test(line)) {
|
||||
flush();
|
||||
heading = line.replace(/^#+\s*/, '').trim();
|
||||
continue;
|
||||
}
|
||||
buffer.push(line);
|
||||
}
|
||||
flush();
|
||||
return out;
|
||||
};
|
||||
|
||||
const seedDocs = async () => {
|
||||
const ids = [];
|
||||
const documents = [];
|
||||
const metadatas = [];
|
||||
|
||||
for (const rel of DOC_SOURCES) {
|
||||
const abs = path.join(REPO, rel);
|
||||
if (!fs.existsSync(abs)) {
|
||||
console.warn(`[seed] skipped missing ${rel}`);
|
||||
continue;
|
||||
}
|
||||
const chunks = chunkMarkdown(fs.readFileSync(abs, 'utf8'), rel);
|
||||
chunks.forEach((c, i) => {
|
||||
ids.push(`${rel}::${i}`);
|
||||
documents.push(c.text);
|
||||
metadatas.push({ source: rel, heading: c.heading, updatedAt: new Date().toISOString().slice(0, 10) });
|
||||
});
|
||||
}
|
||||
|
||||
const collection = await resetCollection(DOCS_COLLECTION);
|
||||
const embeddings = await embedMany(documents);
|
||||
// Chroma caps how much it will accept in one add; these corpora are small
|
||||
// but batch anyway so this doesn't become a surprise later.
|
||||
const BATCH = 200;
|
||||
for (let i = 0; i < ids.length; i += BATCH) {
|
||||
// eslint-disable-next-line no-await-in-loop
|
||||
await collection.add({
|
||||
ids: ids.slice(i, i + BATCH),
|
||||
documents: documents.slice(i, i + BATCH),
|
||||
metadatas: metadatas.slice(i, i + BATCH),
|
||||
embeddings: embeddings.slice(i, i + BATCH)
|
||||
});
|
||||
}
|
||||
return { sources: DOC_SOURCES.length, vectors: ids.length };
|
||||
};
|
||||
|
||||
const main = async () => {
|
||||
console.log(`[seed] model: ${MODEL_ID}`);
|
||||
const intents = await seedIntents();
|
||||
console.log(`[seed] intent_examples: ${intents.vectors} vectors across ${intents.intents} intents`);
|
||||
const docs = await seedDocs();
|
||||
console.log(`[seed] console_docs: ${docs.vectors} chunks from ${docs.sources} sources`);
|
||||
console.log('[seed] done');
|
||||
};
|
||||
|
||||
main().catch((err) => {
|
||||
console.error('[seed] failed:', err.message);
|
||||
process.exit(1);
|
||||
});
|
||||
Reference in New Issue
Block a user