77 lines
2.9 KiB
YAML
77 lines
2.9 KiB
YAML
# services/ai-llm/gpu-deployment.yaml
|
|
apiVersion: apps/v1
|
|
kind: Deployment
|
|
metadata:
|
|
name: ollama-gpu
|
|
namespace: ai
|
|
spec:
|
|
replicas: 1
|
|
revisionHistoryLimit: 2
|
|
strategy:
|
|
type: Recreate
|
|
selector:
|
|
matchLabels: {app: ollama-gpu}
|
|
template:
|
|
metadata:
|
|
labels: {app: ollama-gpu}
|
|
annotations:
|
|
fluentbit.io/exclude: "true"
|
|
spec:
|
|
automountServiceAccountToken: false
|
|
nodeSelector:
|
|
kubernetes.io/hostname: titan-24
|
|
runtimeClassName: nvidia
|
|
securityContext:
|
|
seccompProfile: {type: RuntimeDefault}
|
|
containers:
|
|
- name: ollama
|
|
image: ollama/ollama@sha256:2c9595c555fd70a28363489ac03bd5bf9e7c5bdf2890373c3a830ffd7252ce6d
|
|
command: [/bin/bash, -ec]
|
|
args:
|
|
- |
|
|
while ! test -f /models/.pilot-v1-ready || ! test -f /reservation/reserved; do sleep 5; done
|
|
# Backend parser diagnostics can contain submitted schema text.
|
|
# Keep request status/timing at the gateway; retain no backend output.
|
|
exec ollama serve >/dev/null 2>&1
|
|
env:
|
|
- {name: OLLAMA_HOST, value: "0.0.0.0:11434"}
|
|
- {name: OLLAMA_MODELS, value: /models}
|
|
- {name: OLLAMA_NO_CLOUD, value: "1"}
|
|
- {name: OLLAMA_CONTEXT_LENGTH, value: "8192"}
|
|
- {name: OLLAMA_MAX_LOADED_MODELS, value: "1"}
|
|
- {name: OLLAMA_NUM_PARALLEL, value: "1"}
|
|
- {name: OLLAMA_MAX_QUEUE, value: "1"}
|
|
- {name: OLLAMA_KEEP_ALIVE, value: "20m"}
|
|
- {name: OLLAMA_LOAD_TIMEOUT, value: "20m"}
|
|
- {name: OLLAMA_DEBUG, value: "false"}
|
|
- {name: OLLAMA_FLASH_ATTENTION, value: "1"}
|
|
- {name: OLLAMA_KV_CACHE_TYPE, value: q8_0}
|
|
ports:
|
|
- {name: http, containerPort: 11434}
|
|
readinessProbe:
|
|
httpGet: {path: /api/version, port: http}
|
|
periodSeconds: 10
|
|
timeoutSeconds: 3
|
|
securityContext:
|
|
allowPrivilegeEscalation: false
|
|
readOnlyRootFilesystem: true
|
|
capabilities: {drop: [ALL]}
|
|
resources:
|
|
requests: {cpu: "4", memory: 16Gi, nvidia.com/gpu.shared: 4}
|
|
limits: {cpu: "12", memory: 24Gi, nvidia.com/gpu.shared: 4}
|
|
volumeMounts:
|
|
- {name: models, mountPath: /models, readOnly: true}
|
|
- {name: reservation, mountPath: /reservation, readOnly: true}
|
|
- {name: tmp, mountPath: /tmp}
|
|
- {name: identity, mountPath: /root/.ollama}
|
|
volumes:
|
|
- name: reservation
|
|
hostPath: {path: /var/lib/atlas-maintenance/lan-inference-20260928, type: DirectoryOrCreate}
|
|
- name: models
|
|
persistentVolumeClaim:
|
|
claimName: ollama-gpu-titan24
|
|
- name: tmp
|
|
emptyDir: {sizeLimit: 1Gi}
|
|
- name: identity
|
|
emptyDir: {medium: Memory, sizeLimit: 1Mi}
|