ai: reserve RTX 3080 for the local inference pilot

This commit is contained in:
jenkins 2026-09-28 19:04:14 -05:00
parent 20bf855159
commit 39036a0f9e
14 changed files with 277 additions and 8 deletions

View File

@ -25,7 +25,8 @@ metadata:
namespace: ai
spec:
podSelector:
matchLabels: {app: ollama-batch-seed}
matchExpressions:
- {key: app, operator: In, values: [ollama-batch-seed, ollama-gpu-seed]}
policyTypes: [Ingress, Egress]
ingress: []
egress:

View File

@ -0,0 +1,72 @@
# services/ai-llm/gpu-deployment.yaml
apiVersion: apps/v1
kind: Deployment
metadata:
name: ollama-gpu
namespace: ai
spec:
replicas: 1
revisionHistoryLimit: 2
strategy:
type: Recreate
selector:
matchLabels: {app: ollama-gpu}
template:
metadata:
labels: {app: ollama-gpu}
spec:
automountServiceAccountToken: false
nodeSelector:
kubernetes.io/hostname: titan-24
runtimeClassName: nvidia
securityContext:
seccompProfile: {type: RuntimeDefault}
containers:
- name: ollama
image: ollama/ollama@sha256:2c9595c555fd70a28363489ac03bd5bf9e7c5bdf2890373c3a830ffd7252ce6d
command: [/bin/bash, -ec]
args:
- |
while ! test -f /models/.pilot-v1-ready || ! test -f /reservation/reserved; do sleep 5; done
exec ollama serve
env:
- {name: OLLAMA_HOST, value: "0.0.0.0:11434"}
- {name: OLLAMA_MODELS, value: /models}
- {name: OLLAMA_NO_CLOUD, value: "1"}
- {name: OLLAMA_CONTEXT_LENGTH, value: "8192"}
- {name: OLLAMA_MAX_LOADED_MODELS, value: "1"}
- {name: OLLAMA_NUM_PARALLEL, value: "1"}
- {name: OLLAMA_MAX_QUEUE, value: "1"}
- {name: OLLAMA_KEEP_ALIVE, value: "20m"}
- {name: OLLAMA_LOAD_TIMEOUT, value: "20m"}
- {name: OLLAMA_DEBUG, value: "false"}
- {name: OLLAMA_FLASH_ATTENTION, value: "1"}
- {name: OLLAMA_KV_CACHE_TYPE, value: q8_0}
ports:
- {name: http, containerPort: 11434}
readinessProbe:
httpGet: {path: /api/version, port: http}
periodSeconds: 10
timeoutSeconds: 3
securityContext:
allowPrivilegeEscalation: false
readOnlyRootFilesystem: true
capabilities: {drop: [ALL]}
resources:
requests: {cpu: "4", memory: 16Gi, nvidia.com/gpu.shared: 4}
limits: {cpu: "12", memory: 24Gi, nvidia.com/gpu.shared: 4}
volumeMounts:
- {name: models, mountPath: /models, readOnly: true}
- {name: reservation, mountPath: /reservation, readOnly: true}
- {name: tmp, mountPath: /tmp}
- {name: identity, mountPath: /root/.ollama}
volumes:
- name: reservation
hostPath: {path: /var/lib/atlas-maintenance/lan-inference-20260928, type: DirectoryOrCreate}
- name: models
persistentVolumeClaim:
claimName: ollama-gpu-titan24
- name: tmp
emptyDir: {sizeLimit: 1Gi}
- name: identity
emptyDir: {medium: Memory, sizeLimit: 1Mi}

View File

@ -0,0 +1,19 @@
# services/ai-llm/gpu-networkpolicy.yaml
apiVersion: networking.k8s.io/v1
kind: NetworkPolicy
metadata:
name: ollama-gpu-offline
namespace: ai
spec:
podSelector:
matchLabels: {app: ollama-gpu}
policyTypes: [Ingress, Egress]
egress: []
ingress:
- from:
- namespaceSelector:
matchLabels: {kubernetes.io/metadata.name: hermes}
podSelector:
matchLabels: {app: hermes-model-gate}
ports:
- {protocol: TCP, port: 11434}

View File

@ -0,0 +1,12 @@
# services/ai-llm/gpu-pvc.yaml
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: ollama-gpu-titan24
namespace: ai
spec:
accessModes: [ReadWriteOnce]
storageClassName: local-path
resources:
requests:
storage: 32Gi

View File

@ -0,0 +1,34 @@
# services/ai-llm/gpu-reservation-job.yaml
apiVersion: batch/v1
kind: Job
metadata:
name: ollama-gpu-reserve-20260928
namespace: ai
spec:
backoffLimit: 1
activeDeadlineSeconds: 600
template:
metadata:
labels: {app: ollama-gpu-reservation}
spec:
nodeSelector:
kubernetes.io/hostname: titan-24
automountServiceAccountToken: false
hostPID: true
restartPolicy: Never
containers:
- name: reservation
image: debian@sha256:b6e2a152f22a40ff69d92cb397223c906017e1391a73c952b588e51af8883bf8
command: [/bin/bash, /scripts/gpu_pilot_reservation.sh, reserve]
securityContext: {privileged: true, runAsUser: 0}
resources:
requests: {cpu: 25m, memory: 64Mi}
limits: {cpu: 500m, memory: 128Mi}
volumeMounts:
- {name: host, mountPath: /host}
- {name: script, mountPath: /scripts, readOnly: true}
volumes:
- name: host
hostPath: {path: /, type: Directory}
- name: script
configMap: {name: ollama-gpu-reservation}

View File

@ -0,0 +1,52 @@
# services/ai-llm/gpu-seed-job.yaml
apiVersion: batch/v1
kind: Job
metadata:
name: ollama-gpu-seed-v2
namespace: ai
spec:
backoffLimit: 2
activeDeadlineSeconds: 7200
template:
metadata:
labels:
app: ollama-gpu-seed
spec:
automountServiceAccountToken: false
restartPolicy: Never
nodeSelector:
kubernetes.io/hostname: titan-24
containers:
- name: seed
image: ollama/ollama@sha256:2c9595c555fd70a28363489ac03bd5bf9e7c5bdf2890373c3a830ffd7252ce6d
env:
- {name: OLLAMA_HOST, value: "127.0.0.1:11434"}
- {name: OLLAMA_MODELS, value: /models}
- {name: OLLAMA_NO_CLOUD, value: "1"}
command: [/bin/bash, -ec]
args:
- |
ollama serve >/tmp/ollama.log 2>&1 &
server_pid=$!
trap 'kill "$server_pid"; wait "$server_pid" || true' EXIT
for attempt in $(seq 1 60); do
ollama list >/dev/null 2>&1 && break
sleep 1
done
ollama pull qwen2.5:14b-instruct-q4_0 >/tmp/pull.log 2>&1 || { tail -c 1000 /tmp/pull.log; exit 1; }
cd /models/manifests/registry.ollama.ai/library
sha256sum -c <<'PINS'
5449194ff8035ccb13a6409a5814de6c8f9c39f555f429e383ae0fb7137001bd qwen2.5/14b-instruct-q4_0
PINS
# Runtime mounts this cache read-only, after registry digest checks.
chmod -R a+rX /models
touch /models/.pilot-v1-ready
resources:
requests: {cpu: 500m, memory: 512Mi}
limits: {cpu: "2", memory: 2Gi}
volumeMounts:
- {name: models, mountPath: /models}
volumes:
- name: models
persistentVolumeClaim:
claimName: ollama-gpu-titan24

View File

@ -0,0 +1,10 @@
# services/ai-llm/gpu-service.yaml
apiVersion: v1
kind: Service
metadata:
name: ollama-gpu
namespace: ai
spec:
selector: {app: ollama-gpu}
ports:
- {name: http, port: 11434, targetPort: http}

View File

@ -6,9 +6,22 @@ resources:
- namespace.yaml
- pvc.yaml
- batch-pvc.yaml
- gpu-pvc.yaml
- gpu-reservation-job.yaml
- gpu-seed-job.yaml
- deployment.yaml
- batch-seed-job.yaml
- batch-deployment.yaml
- gpu-deployment.yaml
- service.yaml
- batch-service.yaml
- gpu-service.yaml
- batch-networkpolicy.yaml
- gpu-networkpolicy.yaml
configMapGenerator:
- name: ollama-gpu-reservation
files:
- gpu_pilot_reservation.sh=scripts/gpu_pilot_reservation.sh
options:
disableNameSuffixHash: true

View File

@ -0,0 +1,58 @@
#!/usr/bin/env bash
# Preserve enough host state to restore the temporary RTX 3080 reservation.
set -euo pipefail
state=/host/var/lib/atlas-maintenance/lan-inference-20260928
host() { chroot /host "$@"; }
umask 077
mkdir -p "$state"
case "${1:-}" in
reserve)
# Wolf must be scaled down through Flux before stopping its child containers.
for attempt in $(seq 1 120); do
host pgrep -x wolf >/dev/null || break
sleep 2
done
if host pgrep -x wolf >/dev/null; then
echo 'Wolf still running; reservation refused' >&2
exit 1
fi
if ! test -f "$state/saved"; then
host systemctl show display-manager -p Id --value > "$state/display-unit"
host systemctl is-active display-manager > "$state/display-active" || true
host systemctl is-enabled display-manager > "$state/display-enabled" || true
host docker ps --filter 'name=^/Wolf' --format '{{.ID}}' > "$state/docker-running"
touch "$state/saved"
fi
unit=$(cat "$state/display-unit")
case "$unit" in sddm.service|gdm.service|gdm3.service|lightdm.service) ;;
*) echo 'Unexpected display manager; refusing host mutation' >&2; exit 1 ;;
esac
host systemctl mask --now "$unit"
while read -r container; do
test -z "$container" || host docker stop --time 30 "$container"
done < "$state/docker-running"
touch "$state/reserved"
echo 'GPU desktop and Wolf containers stopped; restoration state saved'
;;
restore)
test -f "$state/saved"
unit=$(cat "$state/display-unit")
case "$unit" in sddm.service|gdm.service|gdm3.service|lightdm.service) ;;
*) exit 1 ;;
esac
case "$(cat "$state/display-enabled")" in
masked*) ;;
*) host systemctl unmask "$unit" ;;
esac
if test "$(cat "$state/display-active")" = active; then
host systemctl start "$unit"
fi
while read -r container; do
test -z "$container" || host docker start "$container"
done < "$state/docker-running"
rm -f "$state/reserved"
echo 'Saved display and Wolf container state restored'
;;
*) echo 'Usage: gpu_pilot_reservation.sh reserve|restore' >&2; exit 2 ;;
esac

View File

@ -8,7 +8,7 @@ metadata:
app: wolf
spec:
serviceName: wolf
replicas: 1
replicas: 0
selector:
matchLabels:
app: wolf

View File

@ -12,8 +12,6 @@ rules:
- titan-24-gpu-owner
verbs:
- get
- patch
- update
---
apiVersion: rbac.authorization.k8s.io/v1
kind: RoleBinding

View File

@ -7,7 +7,7 @@ metadata:
labels:
app: hermes-local-image
spec:
replicas: 1
replicas: 0
revisionHistoryLimit: 2
strategy:
type: Recreate

View File

@ -5,6 +5,6 @@ metadata:
name: titan-24-gpu-owner
namespace: hermes
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
ai.bstein.dev/reservation: "temporary LAN inference pilot; restore via Git"
spec:
holderIdentity: hermes
holderIdentity: lan-inference-pilot

View File

@ -101,7 +101,7 @@ spec:
kubernetes.io/metadata.name: ai
podSelector:
matchLabels:
app: ollama-batch
app: ollama-gpu
ports:
- {protocol: TCP, port: 11434}
# Kubernetes API for the existing image-handoff reads: the ClusterIP