From 39036a0f9e31363ca39e430b15b488730aab3e3a Mon Sep 17 00:00:00 2001 From: jenkins Date: Mon, 28 Sep 2026 19:04:14 -0500 Subject: [PATCH] ai: reserve RTX 3080 for the local inference pilot --- services/ai-llm/batch-networkpolicy.yaml | 3 +- services/ai-llm/gpu-deployment.yaml | 72 +++++++++++++++++++ services/ai-llm/gpu-networkpolicy.yaml | 19 +++++ services/ai-llm/gpu-pvc.yaml | 12 ++++ services/ai-llm/gpu-reservation-job.yaml | 34 +++++++++ services/ai-llm/gpu-seed-job.yaml | 52 ++++++++++++++ services/ai-llm/gpu-service.yaml | 10 +++ services/ai-llm/kustomization.yaml | 13 ++++ .../ai-llm/scripts/gpu_pilot_reservation.sh | 58 +++++++++++++++ services/game-stream/wolf-statefulset.yaml | 2 +- services/hermes/ariadne-handoff-rbac.yaml | 2 - services/hermes/local-image-deployment.yaml | 2 +- services/hermes/model-gate-state.yaml | 4 +- services/hermes/networkpolicy.yaml | 2 +- 14 files changed, 277 insertions(+), 8 deletions(-) create mode 100644 services/ai-llm/gpu-deployment.yaml create mode 100644 services/ai-llm/gpu-networkpolicy.yaml create mode 100644 services/ai-llm/gpu-pvc.yaml create mode 100644 services/ai-llm/gpu-reservation-job.yaml create mode 100644 services/ai-llm/gpu-seed-job.yaml create mode 100644 services/ai-llm/gpu-service.yaml create mode 100755 services/ai-llm/scripts/gpu_pilot_reservation.sh diff --git a/services/ai-llm/batch-networkpolicy.yaml b/services/ai-llm/batch-networkpolicy.yaml index 5994045b..e7f15ef3 100644 --- a/services/ai-llm/batch-networkpolicy.yaml +++ b/services/ai-llm/batch-networkpolicy.yaml @@ -25,7 +25,8 @@ metadata: namespace: ai spec: podSelector: - matchLabels: {app: ollama-batch-seed} + matchExpressions: + - {key: app, operator: In, values: [ollama-batch-seed, ollama-gpu-seed]} policyTypes: [Ingress, Egress] ingress: [] egress: diff --git a/services/ai-llm/gpu-deployment.yaml b/services/ai-llm/gpu-deployment.yaml new file mode 100644 index 00000000..e4079a52 --- /dev/null +++ b/services/ai-llm/gpu-deployment.yaml @@ -0,0 +1,72 @@ +# services/ai-llm/gpu-deployment.yaml +apiVersion: apps/v1 +kind: Deployment +metadata: + name: ollama-gpu + namespace: ai +spec: + replicas: 1 + revisionHistoryLimit: 2 + strategy: + type: Recreate + selector: + matchLabels: {app: ollama-gpu} + template: + metadata: + labels: {app: ollama-gpu} + spec: + automountServiceAccountToken: false + nodeSelector: + kubernetes.io/hostname: titan-24 + runtimeClassName: nvidia + securityContext: + seccompProfile: {type: RuntimeDefault} + containers: + - name: ollama + image: ollama/ollama@sha256:2c9595c555fd70a28363489ac03bd5bf9e7c5bdf2890373c3a830ffd7252ce6d + command: [/bin/bash, -ec] + args: + - | + while ! test -f /models/.pilot-v1-ready || ! test -f /reservation/reserved; do sleep 5; done + exec ollama serve + env: + - {name: OLLAMA_HOST, value: "0.0.0.0:11434"} + - {name: OLLAMA_MODELS, value: /models} + - {name: OLLAMA_NO_CLOUD, value: "1"} + - {name: OLLAMA_CONTEXT_LENGTH, value: "8192"} + - {name: OLLAMA_MAX_LOADED_MODELS, value: "1"} + - {name: OLLAMA_NUM_PARALLEL, value: "1"} + - {name: OLLAMA_MAX_QUEUE, value: "1"} + - {name: OLLAMA_KEEP_ALIVE, value: "20m"} + - {name: OLLAMA_LOAD_TIMEOUT, value: "20m"} + - {name: OLLAMA_DEBUG, value: "false"} + - {name: OLLAMA_FLASH_ATTENTION, value: "1"} + - {name: OLLAMA_KV_CACHE_TYPE, value: q8_0} + ports: + - {name: http, containerPort: 11434} + readinessProbe: + httpGet: {path: /api/version, port: http} + periodSeconds: 10 + timeoutSeconds: 3 + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: {drop: [ALL]} + resources: + requests: {cpu: "4", memory: 16Gi, nvidia.com/gpu.shared: 4} + limits: {cpu: "12", memory: 24Gi, nvidia.com/gpu.shared: 4} + volumeMounts: + - {name: models, mountPath: /models, readOnly: true} + - {name: reservation, mountPath: /reservation, readOnly: true} + - {name: tmp, mountPath: /tmp} + - {name: identity, mountPath: /root/.ollama} + volumes: + - name: reservation + hostPath: {path: /var/lib/atlas-maintenance/lan-inference-20260928, type: DirectoryOrCreate} + - name: models + persistentVolumeClaim: + claimName: ollama-gpu-titan24 + - name: tmp + emptyDir: {sizeLimit: 1Gi} + - name: identity + emptyDir: {medium: Memory, sizeLimit: 1Mi} diff --git a/services/ai-llm/gpu-networkpolicy.yaml b/services/ai-llm/gpu-networkpolicy.yaml new file mode 100644 index 00000000..c0ab478b --- /dev/null +++ b/services/ai-llm/gpu-networkpolicy.yaml @@ -0,0 +1,19 @@ +# services/ai-llm/gpu-networkpolicy.yaml +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: ollama-gpu-offline + namespace: ai +spec: + podSelector: + matchLabels: {app: ollama-gpu} + policyTypes: [Ingress, Egress] + egress: [] + ingress: + - from: + - namespaceSelector: + matchLabels: {kubernetes.io/metadata.name: hermes} + podSelector: + matchLabels: {app: hermes-model-gate} + ports: + - {protocol: TCP, port: 11434} diff --git a/services/ai-llm/gpu-pvc.yaml b/services/ai-llm/gpu-pvc.yaml new file mode 100644 index 00000000..e438ce57 --- /dev/null +++ b/services/ai-llm/gpu-pvc.yaml @@ -0,0 +1,12 @@ +# services/ai-llm/gpu-pvc.yaml +apiVersion: v1 +kind: PersistentVolumeClaim +metadata: + name: ollama-gpu-titan24 + namespace: ai +spec: + accessModes: [ReadWriteOnce] + storageClassName: local-path + resources: + requests: + storage: 32Gi diff --git a/services/ai-llm/gpu-reservation-job.yaml b/services/ai-llm/gpu-reservation-job.yaml new file mode 100644 index 00000000..2a8aa93d --- /dev/null +++ b/services/ai-llm/gpu-reservation-job.yaml @@ -0,0 +1,34 @@ +# services/ai-llm/gpu-reservation-job.yaml +apiVersion: batch/v1 +kind: Job +metadata: + name: ollama-gpu-reserve-20260928 + namespace: ai +spec: + backoffLimit: 1 + activeDeadlineSeconds: 600 + template: + metadata: + labels: {app: ollama-gpu-reservation} + spec: + nodeSelector: + kubernetes.io/hostname: titan-24 + automountServiceAccountToken: false + hostPID: true + restartPolicy: Never + containers: + - name: reservation + image: debian@sha256:b6e2a152f22a40ff69d92cb397223c906017e1391a73c952b588e51af8883bf8 + command: [/bin/bash, /scripts/gpu_pilot_reservation.sh, reserve] + securityContext: {privileged: true, runAsUser: 0} + resources: + requests: {cpu: 25m, memory: 64Mi} + limits: {cpu: 500m, memory: 128Mi} + volumeMounts: + - {name: host, mountPath: /host} + - {name: script, mountPath: /scripts, readOnly: true} + volumes: + - name: host + hostPath: {path: /, type: Directory} + - name: script + configMap: {name: ollama-gpu-reservation} diff --git a/services/ai-llm/gpu-seed-job.yaml b/services/ai-llm/gpu-seed-job.yaml new file mode 100644 index 00000000..238a839d --- /dev/null +++ b/services/ai-llm/gpu-seed-job.yaml @@ -0,0 +1,52 @@ +# services/ai-llm/gpu-seed-job.yaml +apiVersion: batch/v1 +kind: Job +metadata: + name: ollama-gpu-seed-v2 + namespace: ai +spec: + backoffLimit: 2 + activeDeadlineSeconds: 7200 + template: + metadata: + labels: + app: ollama-gpu-seed + spec: + automountServiceAccountToken: false + restartPolicy: Never + nodeSelector: + kubernetes.io/hostname: titan-24 + containers: + - name: seed + image: ollama/ollama@sha256:2c9595c555fd70a28363489ac03bd5bf9e7c5bdf2890373c3a830ffd7252ce6d + env: + - {name: OLLAMA_HOST, value: "127.0.0.1:11434"} + - {name: OLLAMA_MODELS, value: /models} + - {name: OLLAMA_NO_CLOUD, value: "1"} + command: [/bin/bash, -ec] + args: + - | + ollama serve >/tmp/ollama.log 2>&1 & + server_pid=$! + trap 'kill "$server_pid"; wait "$server_pid" || true' EXIT + for attempt in $(seq 1 60); do + ollama list >/dev/null 2>&1 && break + sleep 1 + done + ollama pull qwen2.5:14b-instruct-q4_0 >/tmp/pull.log 2>&1 || { tail -c 1000 /tmp/pull.log; exit 1; } + cd /models/manifests/registry.ollama.ai/library + sha256sum -c <<'PINS' + 5449194ff8035ccb13a6409a5814de6c8f9c39f555f429e383ae0fb7137001bd qwen2.5/14b-instruct-q4_0 + PINS + # Runtime mounts this cache read-only, after registry digest checks. + chmod -R a+rX /models + touch /models/.pilot-v1-ready + resources: + requests: {cpu: 500m, memory: 512Mi} + limits: {cpu: "2", memory: 2Gi} + volumeMounts: + - {name: models, mountPath: /models} + volumes: + - name: models + persistentVolumeClaim: + claimName: ollama-gpu-titan24 diff --git a/services/ai-llm/gpu-service.yaml b/services/ai-llm/gpu-service.yaml new file mode 100644 index 00000000..7f0a9b99 --- /dev/null +++ b/services/ai-llm/gpu-service.yaml @@ -0,0 +1,10 @@ +# services/ai-llm/gpu-service.yaml +apiVersion: v1 +kind: Service +metadata: + name: ollama-gpu + namespace: ai +spec: + selector: {app: ollama-gpu} + ports: + - {name: http, port: 11434, targetPort: http} diff --git a/services/ai-llm/kustomization.yaml b/services/ai-llm/kustomization.yaml index 579ef634..8aaba89e 100644 --- a/services/ai-llm/kustomization.yaml +++ b/services/ai-llm/kustomization.yaml @@ -6,9 +6,22 @@ resources: - namespace.yaml - pvc.yaml - batch-pvc.yaml + - gpu-pvc.yaml + - gpu-reservation-job.yaml + - gpu-seed-job.yaml - deployment.yaml - batch-seed-job.yaml - batch-deployment.yaml + - gpu-deployment.yaml - service.yaml - batch-service.yaml + - gpu-service.yaml - batch-networkpolicy.yaml + - gpu-networkpolicy.yaml + +configMapGenerator: + - name: ollama-gpu-reservation + files: + - gpu_pilot_reservation.sh=scripts/gpu_pilot_reservation.sh + options: + disableNameSuffixHash: true diff --git a/services/ai-llm/scripts/gpu_pilot_reservation.sh b/services/ai-llm/scripts/gpu_pilot_reservation.sh new file mode 100755 index 00000000..15185cb5 --- /dev/null +++ b/services/ai-llm/scripts/gpu_pilot_reservation.sh @@ -0,0 +1,58 @@ +#!/usr/bin/env bash +# Preserve enough host state to restore the temporary RTX 3080 reservation. +set -euo pipefail +state=/host/var/lib/atlas-maintenance/lan-inference-20260928 +host() { chroot /host "$@"; } +umask 077 +mkdir -p "$state" + +case "${1:-}" in + reserve) + # Wolf must be scaled down through Flux before stopping its child containers. + for attempt in $(seq 1 120); do + host pgrep -x wolf >/dev/null || break + sleep 2 + done + if host pgrep -x wolf >/dev/null; then + echo 'Wolf still running; reservation refused' >&2 + exit 1 + fi + if ! test -f "$state/saved"; then + host systemctl show display-manager -p Id --value > "$state/display-unit" + host systemctl is-active display-manager > "$state/display-active" || true + host systemctl is-enabled display-manager > "$state/display-enabled" || true + host docker ps --filter 'name=^/Wolf' --format '{{.ID}}' > "$state/docker-running" + touch "$state/saved" + fi + unit=$(cat "$state/display-unit") + case "$unit" in sddm.service|gdm.service|gdm3.service|lightdm.service) ;; + *) echo 'Unexpected display manager; refusing host mutation' >&2; exit 1 ;; + esac + host systemctl mask --now "$unit" + while read -r container; do + test -z "$container" || host docker stop --time 30 "$container" + done < "$state/docker-running" + touch "$state/reserved" + echo 'GPU desktop and Wolf containers stopped; restoration state saved' + ;; + restore) + test -f "$state/saved" + unit=$(cat "$state/display-unit") + case "$unit" in sddm.service|gdm.service|gdm3.service|lightdm.service) ;; + *) exit 1 ;; + esac + case "$(cat "$state/display-enabled")" in + masked*) ;; + *) host systemctl unmask "$unit" ;; + esac + if test "$(cat "$state/display-active")" = active; then + host systemctl start "$unit" + fi + while read -r container; do + test -z "$container" || host docker start "$container" + done < "$state/docker-running" + rm -f "$state/reserved" + echo 'Saved display and Wolf container state restored' + ;; + *) echo 'Usage: gpu_pilot_reservation.sh reserve|restore' >&2; exit 2 ;; +esac diff --git a/services/game-stream/wolf-statefulset.yaml b/services/game-stream/wolf-statefulset.yaml index f6779e2f..4a117271 100644 --- a/services/game-stream/wolf-statefulset.yaml +++ b/services/game-stream/wolf-statefulset.yaml @@ -8,7 +8,7 @@ metadata: app: wolf spec: serviceName: wolf - replicas: 1 + replicas: 0 selector: matchLabels: app: wolf diff --git a/services/hermes/ariadne-handoff-rbac.yaml b/services/hermes/ariadne-handoff-rbac.yaml index a8da16cc..9863c633 100644 --- a/services/hermes/ariadne-handoff-rbac.yaml +++ b/services/hermes/ariadne-handoff-rbac.yaml @@ -12,8 +12,6 @@ rules: - titan-24-gpu-owner verbs: - get - - patch - - update --- apiVersion: rbac.authorization.k8s.io/v1 kind: RoleBinding diff --git a/services/hermes/local-image-deployment.yaml b/services/hermes/local-image-deployment.yaml index 3dc69e2d..c81aa54e 100644 --- a/services/hermes/local-image-deployment.yaml +++ b/services/hermes/local-image-deployment.yaml @@ -7,7 +7,7 @@ metadata: labels: app: hermes-local-image spec: - replicas: 1 + replicas: 0 revisionHistoryLimit: 2 strategy: type: Recreate diff --git a/services/hermes/model-gate-state.yaml b/services/hermes/model-gate-state.yaml index 75659118..ea98dd0b 100644 --- a/services/hermes/model-gate-state.yaml +++ b/services/hermes/model-gate-state.yaml @@ -5,6 +5,6 @@ metadata: name: titan-24-gpu-owner namespace: hermes annotations: - kustomize.toolkit.fluxcd.io/ssa: IfNotPresent + ai.bstein.dev/reservation: "temporary LAN inference pilot; restore via Git" spec: - holderIdentity: hermes + holderIdentity: lan-inference-pilot diff --git a/services/hermes/networkpolicy.yaml b/services/hermes/networkpolicy.yaml index 0473543a..a0cea4d2 100644 --- a/services/hermes/networkpolicy.yaml +++ b/services/hermes/networkpolicy.yaml @@ -101,7 +101,7 @@ spec: kubernetes.io/metadata.name: ai podSelector: matchLabels: - app: ollama-batch + app: ollama-gpu ports: - {protocol: TCP, port: 11434} # Kubernetes API for the existing image-handoff reads: the ClusterIP