ai: isolate CPU batch inference on titan-23

This commit is contained in:
jenkins 2026-09-28 17:52:23 -05:00
parent 713cb8f80f
commit 360af7a421
6 changed files with 193 additions and 0 deletions

View File

@ -0,0 +1,66 @@
# services/ai-llm/batch-deployment.yaml
apiVersion: apps/v1
kind: Deployment
metadata:
name: ollama-batch
namespace: ai
spec:
replicas: 1
revisionHistoryLimit: 2
strategy:
type: Recreate
selector:
matchLabels: {app: ollama-batch}
template:
metadata:
labels: {app: ollama-batch}
spec:
automountServiceAccountToken: false
nodeSelector:
kubernetes.io/hostname: titan-23
securityContext:
seccompProfile: {type: RuntimeDefault}
containers:
- name: ollama
image: ollama/ollama@sha256:0c0a83210471fb50226bcdc2d6611d20ab13ae87e024cc304c94a6a5765c5e65
command: [/bin/bash, -ec]
args:
- |
while ! test -f /models/.pilot-v1-ready; do sleep 5; done
exec ollama serve
env:
- {name: OLLAMA_HOST, value: "0.0.0.0:11434"}
- {name: OLLAMA_MODELS, value: /models}
- {name: OLLAMA_NO_CLOUD, value: "1"}
- {name: OLLAMA_CONTEXT_LENGTH, value: "32768"}
- {name: OLLAMA_MAX_LOADED_MODELS, value: "1"}
- {name: OLLAMA_NUM_PARALLEL, value: "1"}
- {name: OLLAMA_MAX_QUEUE, value: "1"}
- {name: OLLAMA_KEEP_ALIVE, value: "10m"}
- {name: OLLAMA_LOAD_TIMEOUT, value: "10m"}
- {name: OLLAMA_DEBUG, value: "false"}
ports:
- {name: http, containerPort: 11434}
readinessProbe:
httpGet: {path: /api/version, port: http}
periodSeconds: 10
timeoutSeconds: 3
securityContext:
allowPrivilegeEscalation: false
readOnlyRootFilesystem: true
capabilities: {drop: [ALL]}
resources:
requests: {cpu: "16", memory: 48Gi}
limits: {cpu: "16", memory: 48Gi}
volumeMounts:
- {name: models, mountPath: /models, readOnly: true}
- {name: tmp, mountPath: /tmp}
- {name: identity, mountPath: /root/.ollama}
volumes:
- name: models
persistentVolumeClaim:
claimName: ollama-batch-titan23
- name: tmp
emptyDir: {sizeLimit: 1Gi}
- name: identity
emptyDir: {medium: Memory, sizeLimit: 1Mi}

View File

@ -0,0 +1,46 @@
# services/ai-llm/batch-networkpolicy.yaml
apiVersion: networking.k8s.io/v1
kind: NetworkPolicy
metadata:
name: ollama-batch-offline
namespace: ai
spec:
podSelector:
matchLabels: {app: ollama-batch}
policyTypes: [Ingress, Egress]
egress: []
ingress:
- from:
- namespaceSelector:
matchLabels: {kubernetes.io/metadata.name: hermes}
podSelector:
matchLabels: {app: hermes-model-gate}
ports:
- {protocol: TCP, port: 11434}
---
apiVersion: networking.k8s.io/v1
kind: NetworkPolicy
metadata:
name: ollama-batch-seed
namespace: ai
spec:
podSelector:
matchLabels: {app: ollama-batch-seed}
policyTypes: [Ingress, Egress]
ingress: []
egress:
- to:
- namespaceSelector:
matchLabels: {kubernetes.io/metadata.name: kube-system}
podSelector:
matchLabels: {k8s-app: kube-dns}
ports:
- {protocol: UDP, port: 53}
- {protocol: TCP, port: 53}
# This finite download job never receives inference traffic or source data.
- to:
- ipBlock:
cidr: 0.0.0.0/0
except: [10.0.0.0/8, 172.16.0.0/12, 192.168.0.0/16, 169.254.0.0/16]
ports:
- {protocol: TCP, port: 443}

View File

@ -0,0 +1,12 @@
# services/ai-llm/batch-pvc.yaml
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: ollama-batch-titan23
namespace: ai
spec:
accessModes: [ReadWriteOnce]
storageClassName: local-path
resources:
requests:
storage: 64Gi

View File

@ -0,0 +1,54 @@
# services/ai-llm/batch-seed-job.yaml
apiVersion: batch/v1
kind: Job
metadata:
name: ollama-batch-seed-v1
namespace: ai
spec:
backoffLimit: 2
activeDeadlineSeconds: 7200
template:
metadata:
labels:
app: ollama-batch-seed
spec:
automountServiceAccountToken: false
restartPolicy: Never
nodeSelector:
kubernetes.io/hostname: titan-23
containers:
- name: seed
image: ollama/ollama@sha256:0c0a83210471fb50226bcdc2d6611d20ab13ae87e024cc304c94a6a5765c5e65
env:
- {name: OLLAMA_HOST, value: "127.0.0.1:11434"}
- {name: OLLAMA_MODELS, value: /models}
- {name: OLLAMA_NO_CLOUD, value: "1"}
command: [/bin/bash, -ec]
args:
- |
ollama serve >/tmp/ollama.log 2>&1 &
server_pid=$!
trap 'kill "$server_pid"; wait "$server_pid" || true' EXIT
for attempt in $(seq 1 60); do
ollama list >/dev/null 2>&1 && break
sleep 1
done
ollama pull qwen3.5:9b >/tmp/pull.log 2>&1 || { tail -c 1000 /tmp/pull.log; exit 1; }
ollama pull qwen3.6:27b >/tmp/pull.log 2>&1 || { tail -c 1000 /tmp/pull.log; exit 1; }
cd /models/manifests/registry.ollama.ai/library
sha256sum -c <<'PINS'
6488c96fa5faab64bb65cbd30d4289e20e6130ef535a93ef9a49f42eda893ea7 qwen3.5/9b
9d5803d493a991af27b9441c098aa56f2ed7bbd260877f075ec09b575c049bc3 qwen3.6/27b
PINS
# Runtime mounts this cache read-only, after registry digest checks.
chmod -R a+rX /models
touch /models/.pilot-v1-ready
resources:
requests: {cpu: 500m, memory: 512Mi}
limits: {cpu: "2", memory: 2Gi}
volumeMounts:
- {name: models, mountPath: /models}
volumes:
- name: models
persistentVolumeClaim:
claimName: ollama-batch-titan23

View File

@ -0,0 +1,10 @@
# services/ai-llm/batch-service.yaml
apiVersion: v1
kind: Service
metadata:
name: ollama-batch
namespace: ai
spec:
selector: {app: ollama-batch}
ports:
- {name: http, port: 11434, targetPort: http}

View File

@ -5,5 +5,10 @@ namespace: ai
resources:
- namespace.yaml
- pvc.yaml
- batch-pvc.yaml
- deployment.yaml
- batch-seed-job.yaml
- batch-deployment.yaml
- service.yaml
- batch-service.yaml
- batch-networkpolicy.yaml