ai: isolate CPU batch inference on titan-23
This commit is contained in:
parent
713cb8f80f
commit
360af7a421
66
services/ai-llm/batch-deployment.yaml
Normal file
66
services/ai-llm/batch-deployment.yaml
Normal file
@ -0,0 +1,66 @@
|
||||
# services/ai-llm/batch-deployment.yaml
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: ollama-batch
|
||||
namespace: ai
|
||||
spec:
|
||||
replicas: 1
|
||||
revisionHistoryLimit: 2
|
||||
strategy:
|
||||
type: Recreate
|
||||
selector:
|
||||
matchLabels: {app: ollama-batch}
|
||||
template:
|
||||
metadata:
|
||||
labels: {app: ollama-batch}
|
||||
spec:
|
||||
automountServiceAccountToken: false
|
||||
nodeSelector:
|
||||
kubernetes.io/hostname: titan-23
|
||||
securityContext:
|
||||
seccompProfile: {type: RuntimeDefault}
|
||||
containers:
|
||||
- name: ollama
|
||||
image: ollama/ollama@sha256:0c0a83210471fb50226bcdc2d6611d20ab13ae87e024cc304c94a6a5765c5e65
|
||||
command: [/bin/bash, -ec]
|
||||
args:
|
||||
- |
|
||||
while ! test -f /models/.pilot-v1-ready; do sleep 5; done
|
||||
exec ollama serve
|
||||
env:
|
||||
- {name: OLLAMA_HOST, value: "0.0.0.0:11434"}
|
||||
- {name: OLLAMA_MODELS, value: /models}
|
||||
- {name: OLLAMA_NO_CLOUD, value: "1"}
|
||||
- {name: OLLAMA_CONTEXT_LENGTH, value: "32768"}
|
||||
- {name: OLLAMA_MAX_LOADED_MODELS, value: "1"}
|
||||
- {name: OLLAMA_NUM_PARALLEL, value: "1"}
|
||||
- {name: OLLAMA_MAX_QUEUE, value: "1"}
|
||||
- {name: OLLAMA_KEEP_ALIVE, value: "10m"}
|
||||
- {name: OLLAMA_LOAD_TIMEOUT, value: "10m"}
|
||||
- {name: OLLAMA_DEBUG, value: "false"}
|
||||
ports:
|
||||
- {name: http, containerPort: 11434}
|
||||
readinessProbe:
|
||||
httpGet: {path: /api/version, port: http}
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 3
|
||||
securityContext:
|
||||
allowPrivilegeEscalation: false
|
||||
readOnlyRootFilesystem: true
|
||||
capabilities: {drop: [ALL]}
|
||||
resources:
|
||||
requests: {cpu: "16", memory: 48Gi}
|
||||
limits: {cpu: "16", memory: 48Gi}
|
||||
volumeMounts:
|
||||
- {name: models, mountPath: /models, readOnly: true}
|
||||
- {name: tmp, mountPath: /tmp}
|
||||
- {name: identity, mountPath: /root/.ollama}
|
||||
volumes:
|
||||
- name: models
|
||||
persistentVolumeClaim:
|
||||
claimName: ollama-batch-titan23
|
||||
- name: tmp
|
||||
emptyDir: {sizeLimit: 1Gi}
|
||||
- name: identity
|
||||
emptyDir: {medium: Memory, sizeLimit: 1Mi}
|
||||
46
services/ai-llm/batch-networkpolicy.yaml
Normal file
46
services/ai-llm/batch-networkpolicy.yaml
Normal file
@ -0,0 +1,46 @@
|
||||
# services/ai-llm/batch-networkpolicy.yaml
|
||||
apiVersion: networking.k8s.io/v1
|
||||
kind: NetworkPolicy
|
||||
metadata:
|
||||
name: ollama-batch-offline
|
||||
namespace: ai
|
||||
spec:
|
||||
podSelector:
|
||||
matchLabels: {app: ollama-batch}
|
||||
policyTypes: [Ingress, Egress]
|
||||
egress: []
|
||||
ingress:
|
||||
- from:
|
||||
- namespaceSelector:
|
||||
matchLabels: {kubernetes.io/metadata.name: hermes}
|
||||
podSelector:
|
||||
matchLabels: {app: hermes-model-gate}
|
||||
ports:
|
||||
- {protocol: TCP, port: 11434}
|
||||
---
|
||||
apiVersion: networking.k8s.io/v1
|
||||
kind: NetworkPolicy
|
||||
metadata:
|
||||
name: ollama-batch-seed
|
||||
namespace: ai
|
||||
spec:
|
||||
podSelector:
|
||||
matchLabels: {app: ollama-batch-seed}
|
||||
policyTypes: [Ingress, Egress]
|
||||
ingress: []
|
||||
egress:
|
||||
- to:
|
||||
- namespaceSelector:
|
||||
matchLabels: {kubernetes.io/metadata.name: kube-system}
|
||||
podSelector:
|
||||
matchLabels: {k8s-app: kube-dns}
|
||||
ports:
|
||||
- {protocol: UDP, port: 53}
|
||||
- {protocol: TCP, port: 53}
|
||||
# This finite download job never receives inference traffic or source data.
|
||||
- to:
|
||||
- ipBlock:
|
||||
cidr: 0.0.0.0/0
|
||||
except: [10.0.0.0/8, 172.16.0.0/12, 192.168.0.0/16, 169.254.0.0/16]
|
||||
ports:
|
||||
- {protocol: TCP, port: 443}
|
||||
12
services/ai-llm/batch-pvc.yaml
Normal file
12
services/ai-llm/batch-pvc.yaml
Normal file
@ -0,0 +1,12 @@
|
||||
# services/ai-llm/batch-pvc.yaml
|
||||
apiVersion: v1
|
||||
kind: PersistentVolumeClaim
|
||||
metadata:
|
||||
name: ollama-batch-titan23
|
||||
namespace: ai
|
||||
spec:
|
||||
accessModes: [ReadWriteOnce]
|
||||
storageClassName: local-path
|
||||
resources:
|
||||
requests:
|
||||
storage: 64Gi
|
||||
54
services/ai-llm/batch-seed-job.yaml
Normal file
54
services/ai-llm/batch-seed-job.yaml
Normal file
@ -0,0 +1,54 @@
|
||||
# services/ai-llm/batch-seed-job.yaml
|
||||
apiVersion: batch/v1
|
||||
kind: Job
|
||||
metadata:
|
||||
name: ollama-batch-seed-v1
|
||||
namespace: ai
|
||||
spec:
|
||||
backoffLimit: 2
|
||||
activeDeadlineSeconds: 7200
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: ollama-batch-seed
|
||||
spec:
|
||||
automountServiceAccountToken: false
|
||||
restartPolicy: Never
|
||||
nodeSelector:
|
||||
kubernetes.io/hostname: titan-23
|
||||
containers:
|
||||
- name: seed
|
||||
image: ollama/ollama@sha256:0c0a83210471fb50226bcdc2d6611d20ab13ae87e024cc304c94a6a5765c5e65
|
||||
env:
|
||||
- {name: OLLAMA_HOST, value: "127.0.0.1:11434"}
|
||||
- {name: OLLAMA_MODELS, value: /models}
|
||||
- {name: OLLAMA_NO_CLOUD, value: "1"}
|
||||
command: [/bin/bash, -ec]
|
||||
args:
|
||||
- |
|
||||
ollama serve >/tmp/ollama.log 2>&1 &
|
||||
server_pid=$!
|
||||
trap 'kill "$server_pid"; wait "$server_pid" || true' EXIT
|
||||
for attempt in $(seq 1 60); do
|
||||
ollama list >/dev/null 2>&1 && break
|
||||
sleep 1
|
||||
done
|
||||
ollama pull qwen3.5:9b >/tmp/pull.log 2>&1 || { tail -c 1000 /tmp/pull.log; exit 1; }
|
||||
ollama pull qwen3.6:27b >/tmp/pull.log 2>&1 || { tail -c 1000 /tmp/pull.log; exit 1; }
|
||||
cd /models/manifests/registry.ollama.ai/library
|
||||
sha256sum -c <<'PINS'
|
||||
6488c96fa5faab64bb65cbd30d4289e20e6130ef535a93ef9a49f42eda893ea7 qwen3.5/9b
|
||||
9d5803d493a991af27b9441c098aa56f2ed7bbd260877f075ec09b575c049bc3 qwen3.6/27b
|
||||
PINS
|
||||
# Runtime mounts this cache read-only, after registry digest checks.
|
||||
chmod -R a+rX /models
|
||||
touch /models/.pilot-v1-ready
|
||||
resources:
|
||||
requests: {cpu: 500m, memory: 512Mi}
|
||||
limits: {cpu: "2", memory: 2Gi}
|
||||
volumeMounts:
|
||||
- {name: models, mountPath: /models}
|
||||
volumes:
|
||||
- name: models
|
||||
persistentVolumeClaim:
|
||||
claimName: ollama-batch-titan23
|
||||
10
services/ai-llm/batch-service.yaml
Normal file
10
services/ai-llm/batch-service.yaml
Normal file
@ -0,0 +1,10 @@
|
||||
# services/ai-llm/batch-service.yaml
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: ollama-batch
|
||||
namespace: ai
|
||||
spec:
|
||||
selector: {app: ollama-batch}
|
||||
ports:
|
||||
- {name: http, port: 11434, targetPort: http}
|
||||
@ -5,5 +5,10 @@ namespace: ai
|
||||
resources:
|
||||
- namespace.yaml
|
||||
- pvc.yaml
|
||||
- batch-pvc.yaml
|
||||
- deployment.yaml
|
||||
- batch-seed-job.yaml
|
||||
- batch-deployment.yaml
|
||||
- service.yaml
|
||||
- batch-service.yaml
|
||||
- batch-networkpolicy.yaml
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user