ai: reserve RTX 3080 for the local inference pilot
This commit is contained in:
parent
20bf855159
commit
39036a0f9e
@ -25,7 +25,8 @@ metadata:
|
||||
namespace: ai
|
||||
spec:
|
||||
podSelector:
|
||||
matchLabels: {app: ollama-batch-seed}
|
||||
matchExpressions:
|
||||
- {key: app, operator: In, values: [ollama-batch-seed, ollama-gpu-seed]}
|
||||
policyTypes: [Ingress, Egress]
|
||||
ingress: []
|
||||
egress:
|
||||
|
||||
72
services/ai-llm/gpu-deployment.yaml
Normal file
72
services/ai-llm/gpu-deployment.yaml
Normal file
@ -0,0 +1,72 @@
|
||||
# services/ai-llm/gpu-deployment.yaml
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: ollama-gpu
|
||||
namespace: ai
|
||||
spec:
|
||||
replicas: 1
|
||||
revisionHistoryLimit: 2
|
||||
strategy:
|
||||
type: Recreate
|
||||
selector:
|
||||
matchLabels: {app: ollama-gpu}
|
||||
template:
|
||||
metadata:
|
||||
labels: {app: ollama-gpu}
|
||||
spec:
|
||||
automountServiceAccountToken: false
|
||||
nodeSelector:
|
||||
kubernetes.io/hostname: titan-24
|
||||
runtimeClassName: nvidia
|
||||
securityContext:
|
||||
seccompProfile: {type: RuntimeDefault}
|
||||
containers:
|
||||
- name: ollama
|
||||
image: ollama/ollama@sha256:2c9595c555fd70a28363489ac03bd5bf9e7c5bdf2890373c3a830ffd7252ce6d
|
||||
command: [/bin/bash, -ec]
|
||||
args:
|
||||
- |
|
||||
while ! test -f /models/.pilot-v1-ready || ! test -f /reservation/reserved; do sleep 5; done
|
||||
exec ollama serve
|
||||
env:
|
||||
- {name: OLLAMA_HOST, value: "0.0.0.0:11434"}
|
||||
- {name: OLLAMA_MODELS, value: /models}
|
||||
- {name: OLLAMA_NO_CLOUD, value: "1"}
|
||||
- {name: OLLAMA_CONTEXT_LENGTH, value: "8192"}
|
||||
- {name: OLLAMA_MAX_LOADED_MODELS, value: "1"}
|
||||
- {name: OLLAMA_NUM_PARALLEL, value: "1"}
|
||||
- {name: OLLAMA_MAX_QUEUE, value: "1"}
|
||||
- {name: OLLAMA_KEEP_ALIVE, value: "20m"}
|
||||
- {name: OLLAMA_LOAD_TIMEOUT, value: "20m"}
|
||||
- {name: OLLAMA_DEBUG, value: "false"}
|
||||
- {name: OLLAMA_FLASH_ATTENTION, value: "1"}
|
||||
- {name: OLLAMA_KV_CACHE_TYPE, value: q8_0}
|
||||
ports:
|
||||
- {name: http, containerPort: 11434}
|
||||
readinessProbe:
|
||||
httpGet: {path: /api/version, port: http}
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 3
|
||||
securityContext:
|
||||
allowPrivilegeEscalation: false
|
||||
readOnlyRootFilesystem: true
|
||||
capabilities: {drop: [ALL]}
|
||||
resources:
|
||||
requests: {cpu: "4", memory: 16Gi, nvidia.com/gpu.shared: 4}
|
||||
limits: {cpu: "12", memory: 24Gi, nvidia.com/gpu.shared: 4}
|
||||
volumeMounts:
|
||||
- {name: models, mountPath: /models, readOnly: true}
|
||||
- {name: reservation, mountPath: /reservation, readOnly: true}
|
||||
- {name: tmp, mountPath: /tmp}
|
||||
- {name: identity, mountPath: /root/.ollama}
|
||||
volumes:
|
||||
- name: reservation
|
||||
hostPath: {path: /var/lib/atlas-maintenance/lan-inference-20260928, type: DirectoryOrCreate}
|
||||
- name: models
|
||||
persistentVolumeClaim:
|
||||
claimName: ollama-gpu-titan24
|
||||
- name: tmp
|
||||
emptyDir: {sizeLimit: 1Gi}
|
||||
- name: identity
|
||||
emptyDir: {medium: Memory, sizeLimit: 1Mi}
|
||||
19
services/ai-llm/gpu-networkpolicy.yaml
Normal file
19
services/ai-llm/gpu-networkpolicy.yaml
Normal file
@ -0,0 +1,19 @@
|
||||
# services/ai-llm/gpu-networkpolicy.yaml
|
||||
apiVersion: networking.k8s.io/v1
|
||||
kind: NetworkPolicy
|
||||
metadata:
|
||||
name: ollama-gpu-offline
|
||||
namespace: ai
|
||||
spec:
|
||||
podSelector:
|
||||
matchLabels: {app: ollama-gpu}
|
||||
policyTypes: [Ingress, Egress]
|
||||
egress: []
|
||||
ingress:
|
||||
- from:
|
||||
- namespaceSelector:
|
||||
matchLabels: {kubernetes.io/metadata.name: hermes}
|
||||
podSelector:
|
||||
matchLabels: {app: hermes-model-gate}
|
||||
ports:
|
||||
- {protocol: TCP, port: 11434}
|
||||
12
services/ai-llm/gpu-pvc.yaml
Normal file
12
services/ai-llm/gpu-pvc.yaml
Normal file
@ -0,0 +1,12 @@
|
||||
# services/ai-llm/gpu-pvc.yaml
|
||||
apiVersion: v1
|
||||
kind: PersistentVolumeClaim
|
||||
metadata:
|
||||
name: ollama-gpu-titan24
|
||||
namespace: ai
|
||||
spec:
|
||||
accessModes: [ReadWriteOnce]
|
||||
storageClassName: local-path
|
||||
resources:
|
||||
requests:
|
||||
storage: 32Gi
|
||||
34
services/ai-llm/gpu-reservation-job.yaml
Normal file
34
services/ai-llm/gpu-reservation-job.yaml
Normal file
@ -0,0 +1,34 @@
|
||||
# services/ai-llm/gpu-reservation-job.yaml
|
||||
apiVersion: batch/v1
|
||||
kind: Job
|
||||
metadata:
|
||||
name: ollama-gpu-reserve-20260928
|
||||
namespace: ai
|
||||
spec:
|
||||
backoffLimit: 1
|
||||
activeDeadlineSeconds: 600
|
||||
template:
|
||||
metadata:
|
||||
labels: {app: ollama-gpu-reservation}
|
||||
spec:
|
||||
nodeSelector:
|
||||
kubernetes.io/hostname: titan-24
|
||||
automountServiceAccountToken: false
|
||||
hostPID: true
|
||||
restartPolicy: Never
|
||||
containers:
|
||||
- name: reservation
|
||||
image: debian@sha256:b6e2a152f22a40ff69d92cb397223c906017e1391a73c952b588e51af8883bf8
|
||||
command: [/bin/bash, /scripts/gpu_pilot_reservation.sh, reserve]
|
||||
securityContext: {privileged: true, runAsUser: 0}
|
||||
resources:
|
||||
requests: {cpu: 25m, memory: 64Mi}
|
||||
limits: {cpu: 500m, memory: 128Mi}
|
||||
volumeMounts:
|
||||
- {name: host, mountPath: /host}
|
||||
- {name: script, mountPath: /scripts, readOnly: true}
|
||||
volumes:
|
||||
- name: host
|
||||
hostPath: {path: /, type: Directory}
|
||||
- name: script
|
||||
configMap: {name: ollama-gpu-reservation}
|
||||
52
services/ai-llm/gpu-seed-job.yaml
Normal file
52
services/ai-llm/gpu-seed-job.yaml
Normal file
@ -0,0 +1,52 @@
|
||||
# services/ai-llm/gpu-seed-job.yaml
|
||||
apiVersion: batch/v1
|
||||
kind: Job
|
||||
metadata:
|
||||
name: ollama-gpu-seed-v2
|
||||
namespace: ai
|
||||
spec:
|
||||
backoffLimit: 2
|
||||
activeDeadlineSeconds: 7200
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: ollama-gpu-seed
|
||||
spec:
|
||||
automountServiceAccountToken: false
|
||||
restartPolicy: Never
|
||||
nodeSelector:
|
||||
kubernetes.io/hostname: titan-24
|
||||
containers:
|
||||
- name: seed
|
||||
image: ollama/ollama@sha256:2c9595c555fd70a28363489ac03bd5bf9e7c5bdf2890373c3a830ffd7252ce6d
|
||||
env:
|
||||
- {name: OLLAMA_HOST, value: "127.0.0.1:11434"}
|
||||
- {name: OLLAMA_MODELS, value: /models}
|
||||
- {name: OLLAMA_NO_CLOUD, value: "1"}
|
||||
command: [/bin/bash, -ec]
|
||||
args:
|
||||
- |
|
||||
ollama serve >/tmp/ollama.log 2>&1 &
|
||||
server_pid=$!
|
||||
trap 'kill "$server_pid"; wait "$server_pid" || true' EXIT
|
||||
for attempt in $(seq 1 60); do
|
||||
ollama list >/dev/null 2>&1 && break
|
||||
sleep 1
|
||||
done
|
||||
ollama pull qwen2.5:14b-instruct-q4_0 >/tmp/pull.log 2>&1 || { tail -c 1000 /tmp/pull.log; exit 1; }
|
||||
cd /models/manifests/registry.ollama.ai/library
|
||||
sha256sum -c <<'PINS'
|
||||
5449194ff8035ccb13a6409a5814de6c8f9c39f555f429e383ae0fb7137001bd qwen2.5/14b-instruct-q4_0
|
||||
PINS
|
||||
# Runtime mounts this cache read-only, after registry digest checks.
|
||||
chmod -R a+rX /models
|
||||
touch /models/.pilot-v1-ready
|
||||
resources:
|
||||
requests: {cpu: 500m, memory: 512Mi}
|
||||
limits: {cpu: "2", memory: 2Gi}
|
||||
volumeMounts:
|
||||
- {name: models, mountPath: /models}
|
||||
volumes:
|
||||
- name: models
|
||||
persistentVolumeClaim:
|
||||
claimName: ollama-gpu-titan24
|
||||
10
services/ai-llm/gpu-service.yaml
Normal file
10
services/ai-llm/gpu-service.yaml
Normal file
@ -0,0 +1,10 @@
|
||||
# services/ai-llm/gpu-service.yaml
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: ollama-gpu
|
||||
namespace: ai
|
||||
spec:
|
||||
selector: {app: ollama-gpu}
|
||||
ports:
|
||||
- {name: http, port: 11434, targetPort: http}
|
||||
@ -6,9 +6,22 @@ resources:
|
||||
- namespace.yaml
|
||||
- pvc.yaml
|
||||
- batch-pvc.yaml
|
||||
- gpu-pvc.yaml
|
||||
- gpu-reservation-job.yaml
|
||||
- gpu-seed-job.yaml
|
||||
- deployment.yaml
|
||||
- batch-seed-job.yaml
|
||||
- batch-deployment.yaml
|
||||
- gpu-deployment.yaml
|
||||
- service.yaml
|
||||
- batch-service.yaml
|
||||
- gpu-service.yaml
|
||||
- batch-networkpolicy.yaml
|
||||
- gpu-networkpolicy.yaml
|
||||
|
||||
configMapGenerator:
|
||||
- name: ollama-gpu-reservation
|
||||
files:
|
||||
- gpu_pilot_reservation.sh=scripts/gpu_pilot_reservation.sh
|
||||
options:
|
||||
disableNameSuffixHash: true
|
||||
|
||||
58
services/ai-llm/scripts/gpu_pilot_reservation.sh
Executable file
58
services/ai-llm/scripts/gpu_pilot_reservation.sh
Executable file
@ -0,0 +1,58 @@
|
||||
#!/usr/bin/env bash
|
||||
# Preserve enough host state to restore the temporary RTX 3080 reservation.
|
||||
set -euo pipefail
|
||||
state=/host/var/lib/atlas-maintenance/lan-inference-20260928
|
||||
host() { chroot /host "$@"; }
|
||||
umask 077
|
||||
mkdir -p "$state"
|
||||
|
||||
case "${1:-}" in
|
||||
reserve)
|
||||
# Wolf must be scaled down through Flux before stopping its child containers.
|
||||
for attempt in $(seq 1 120); do
|
||||
host pgrep -x wolf >/dev/null || break
|
||||
sleep 2
|
||||
done
|
||||
if host pgrep -x wolf >/dev/null; then
|
||||
echo 'Wolf still running; reservation refused' >&2
|
||||
exit 1
|
||||
fi
|
||||
if ! test -f "$state/saved"; then
|
||||
host systemctl show display-manager -p Id --value > "$state/display-unit"
|
||||
host systemctl is-active display-manager > "$state/display-active" || true
|
||||
host systemctl is-enabled display-manager > "$state/display-enabled" || true
|
||||
host docker ps --filter 'name=^/Wolf' --format '{{.ID}}' > "$state/docker-running"
|
||||
touch "$state/saved"
|
||||
fi
|
||||
unit=$(cat "$state/display-unit")
|
||||
case "$unit" in sddm.service|gdm.service|gdm3.service|lightdm.service) ;;
|
||||
*) echo 'Unexpected display manager; refusing host mutation' >&2; exit 1 ;;
|
||||
esac
|
||||
host systemctl mask --now "$unit"
|
||||
while read -r container; do
|
||||
test -z "$container" || host docker stop --time 30 "$container"
|
||||
done < "$state/docker-running"
|
||||
touch "$state/reserved"
|
||||
echo 'GPU desktop and Wolf containers stopped; restoration state saved'
|
||||
;;
|
||||
restore)
|
||||
test -f "$state/saved"
|
||||
unit=$(cat "$state/display-unit")
|
||||
case "$unit" in sddm.service|gdm.service|gdm3.service|lightdm.service) ;;
|
||||
*) exit 1 ;;
|
||||
esac
|
||||
case "$(cat "$state/display-enabled")" in
|
||||
masked*) ;;
|
||||
*) host systemctl unmask "$unit" ;;
|
||||
esac
|
||||
if test "$(cat "$state/display-active")" = active; then
|
||||
host systemctl start "$unit"
|
||||
fi
|
||||
while read -r container; do
|
||||
test -z "$container" || host docker start "$container"
|
||||
done < "$state/docker-running"
|
||||
rm -f "$state/reserved"
|
||||
echo 'Saved display and Wolf container state restored'
|
||||
;;
|
||||
*) echo 'Usage: gpu_pilot_reservation.sh reserve|restore' >&2; exit 2 ;;
|
||||
esac
|
||||
@ -8,7 +8,7 @@ metadata:
|
||||
app: wolf
|
||||
spec:
|
||||
serviceName: wolf
|
||||
replicas: 1
|
||||
replicas: 0
|
||||
selector:
|
||||
matchLabels:
|
||||
app: wolf
|
||||
|
||||
@ -12,8 +12,6 @@ rules:
|
||||
- titan-24-gpu-owner
|
||||
verbs:
|
||||
- get
|
||||
- patch
|
||||
- update
|
||||
---
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: RoleBinding
|
||||
|
||||
@ -7,7 +7,7 @@ metadata:
|
||||
labels:
|
||||
app: hermes-local-image
|
||||
spec:
|
||||
replicas: 1
|
||||
replicas: 0
|
||||
revisionHistoryLimit: 2
|
||||
strategy:
|
||||
type: Recreate
|
||||
|
||||
@ -5,6 +5,6 @@ metadata:
|
||||
name: titan-24-gpu-owner
|
||||
namespace: hermes
|
||||
annotations:
|
||||
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
|
||||
ai.bstein.dev/reservation: "temporary LAN inference pilot; restore via Git"
|
||||
spec:
|
||||
holderIdentity: hermes
|
||||
holderIdentity: lan-inference-pilot
|
||||
|
||||
@ -101,7 +101,7 @@ spec:
|
||||
kubernetes.io/metadata.name: ai
|
||||
podSelector:
|
||||
matchLabels:
|
||||
app: ollama-batch
|
||||
app: ollama-gpu
|
||||
ports:
|
||||
- {protocol: TCP, port: 11434}
|
||||
# Kubernetes API for the existing image-handoff reads: the ClusterIP
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user