# services/ai-llm/gpu-deployment.yaml apiVersion: apps/v1 kind: Deployment metadata: name: ollama-gpu namespace: ai spec: replicas: 1 revisionHistoryLimit: 2 strategy: type: Recreate selector: matchLabels: {app: ollama-gpu} template: metadata: labels: {app: ollama-gpu} annotations: fluentbit.io/exclude: "true" spec: automountServiceAccountToken: false nodeSelector: kubernetes.io/hostname: titan-24 runtimeClassName: nvidia securityContext: seccompProfile: {type: RuntimeDefault} containers: - name: ollama image: ollama/ollama@sha256:2c9595c555fd70a28363489ac03bd5bf9e7c5bdf2890373c3a830ffd7252ce6d command: [/bin/bash, -ec] args: - | while ! test -f /models/.pilot-v1-ready || ! test -f /reservation/reserved; do sleep 5; done # Backend parser diagnostics can contain submitted schema text. # Keep request status/timing at the gateway; retain no backend output. exec ollama serve >/dev/null 2>&1 env: - {name: OLLAMA_HOST, value: "0.0.0.0:11434"} - {name: OLLAMA_MODELS, value: /models} - {name: OLLAMA_NO_CLOUD, value: "1"} - {name: OLLAMA_CONTEXT_LENGTH, value: "8192"} - {name: OLLAMA_MAX_LOADED_MODELS, value: "1"} - {name: OLLAMA_NUM_PARALLEL, value: "1"} - {name: OLLAMA_MAX_QUEUE, value: "1"} - {name: OLLAMA_KEEP_ALIVE, value: "20m"} - {name: OLLAMA_LOAD_TIMEOUT, value: "20m"} - {name: OLLAMA_DEBUG, value: "false"} - {name: OLLAMA_FLASH_ATTENTION, value: "1"} - {name: OLLAMA_KV_CACHE_TYPE, value: q8_0} ports: - {name: http, containerPort: 11434} readinessProbe: httpGet: {path: /api/version, port: http} periodSeconds: 10 timeoutSeconds: 3 securityContext: allowPrivilegeEscalation: false readOnlyRootFilesystem: true capabilities: {drop: [ALL]} resources: requests: {cpu: "4", memory: 16Gi, nvidia.com/gpu.shared: 4} limits: {cpu: "12", memory: 24Gi, nvidia.com/gpu.shared: 4} volumeMounts: - {name: models, mountPath: /models, readOnly: true} - {name: reservation, mountPath: /reservation, readOnly: true} - {name: tmp, mountPath: /tmp} - {name: identity, mountPath: /root/.ollama} volumes: - name: reservation hostPath: {path: /var/lib/atlas-maintenance/lan-inference-20260928, type: DirectoryOrCreate} - name: models persistentVolumeClaim: claimName: ollama-gpu-titan24 - name: tmp emptyDir: {sizeLimit: 1Gi} - name: identity emptyDir: {medium: Memory, sizeLimit: 1Mi}