# Embeddings for the knowledge layer. # # Production retrieval was keyword-only: no EMBED_PROVIDER, so every chunk had # a null embedding and a question only matched documents sharing its words. # Ollama is what internal/knowledge/embed.go calls "the default worth reaching # for" — real semantics, no credential, no per-token cost, and no tenant text # leaving the cluster. # # Bounded on purpose. The API pods share this node, so an unbounded model # server is a way to evict them; the limit means the kubelet kills this and # nothing else. The request is what keeps it off the 1.2Gi node, where it # would not fit. apiVersion: v1 kind: PersistentVolumeClaim metadata: { name: ollama-models, namespace: krow } spec: accessModes: [ReadWriteOnce] resources: { requests: { storage: 4Gi } } --- apiVersion: apps/v1 kind: Deployment metadata: { name: ollama, namespace: krow } spec: replicas: 1 selector: { matchLabels: { app: ollama } } strategy: { type: Recreate } # one volume, one writer template: metadata: { labels: { app: ollama } } spec: containers: - name: ollama image: ollama/ollama:0.33.1 ports: [{ containerPort: 11434, name: http }] env: - { name: OLLAMA_HOST, value: "0.0.0.0:11434" } # One model, kept resident: reloading it per request would make # every retrieval pay the load cost. - { name: OLLAMA_KEEP_ALIVE, value: "24h" } - { name: OLLAMA_MAX_LOADED_MODELS, value: "1" } resources: requests: { memory: "1Gi", cpu: "250m" } limits: { memory: "3Gi", cpu: "2" } volumeMounts: [{ name: models, mountPath: /root/.ollama }] readinessProbe: httpGet: { path: /api/version, port: http } initialDelaySeconds: 5 periodSeconds: 10 livenessProbe: httpGet: { path: /api/version, port: http } initialDelaySeconds: 30 periodSeconds: 30 volumes: - name: models persistentVolumeClaim: { claimName: ollama-models } --- apiVersion: v1 kind: Service metadata: { name: ollama, namespace: krow } spec: selector: { app: ollama } ports: [{ port: 11434, targetPort: http, name: http }]