diff --git a/infrastructure/ollama.yaml b/infrastructure/ollama.yaml new file mode 100644 index 0000000..7bbe9dd --- /dev/null +++ b/infrastructure/ollama.yaml @@ -0,0 +1,61 @@ +# Embeddings for the knowledge layer. +# +# Production retrieval was keyword-only: no EMBED_PROVIDER, so every chunk had +# a null embedding and a question only matched documents sharing its words. +# Ollama is what internal/knowledge/embed.go calls "the default worth reaching +# for" — real semantics, no credential, no per-token cost, and no tenant text +# leaving the cluster. +# +# Bounded on purpose. The API pods share this node, so an unbounded model +# server is a way to evict them; the limit means the kubelet kills this and +# nothing else. The request is what keeps it off the 1.2Gi node, where it +# would not fit. +apiVersion: v1 +kind: PersistentVolumeClaim +metadata: { name: ollama-models, namespace: krow } +spec: + accessModes: [ReadWriteOnce] + resources: { requests: { storage: 4Gi } } +--- +apiVersion: apps/v1 +kind: Deployment +metadata: { name: ollama, namespace: krow } +spec: + replicas: 1 + selector: { matchLabels: { app: ollama } } + strategy: { type: Recreate } # one volume, one writer + template: + metadata: { labels: { app: ollama } } + spec: + containers: + - name: ollama + image: ollama/ollama:0.33.1 + ports: [{ containerPort: 11434, name: http }] + env: + - { name: OLLAMA_HOST, value: "0.0.0.0:11434" } + # One model, kept resident: reloading it per request would make + # every retrieval pay the load cost. + - { name: OLLAMA_KEEP_ALIVE, value: "24h" } + - { name: OLLAMA_MAX_LOADED_MODELS, value: "1" } + resources: + requests: { memory: "1Gi", cpu: "250m" } + limits: { memory: "3Gi", cpu: "2" } + volumeMounts: [{ name: models, mountPath: /root/.ollama }] + readinessProbe: + httpGet: { path: /api/version, port: http } + initialDelaySeconds: 5 + periodSeconds: 10 + livenessProbe: + httpGet: { path: /api/version, port: http } + initialDelaySeconds: 30 + periodSeconds: 30 + volumes: + - name: models + persistentVolumeClaim: { claimName: ollama-models } +--- +apiVersion: v1 +kind: Service +metadata: { name: ollama, namespace: krow } +spec: + selector: { app: ollama } + ports: [{ port: 11434, targetPort: http, name: http }]