atlas-iac/services/ai-llm/gpu-reservation-job.yaml

35 lines
1.1 KiB
YAML

# services/ai-llm/gpu-reservation-job.yaml
apiVersion: batch/v1
kind: Job
metadata:
name: ollama-gpu-reserve-20260928
namespace: ai
spec:
backoffLimit: 1
activeDeadlineSeconds: 600
template:
metadata:
labels: {app: ollama-gpu-reservation}
spec:
nodeSelector:
kubernetes.io/hostname: titan-24
automountServiceAccountToken: false
hostPID: true
restartPolicy: Never
containers:
- name: reservation
image: debian@sha256:b6e2a152f22a40ff69d92cb397223c906017e1391a73c952b588e51af8883bf8
command: [/bin/bash, /scripts/gpu_pilot_reservation.sh, reserve]
securityContext: {privileged: true, runAsUser: 0}
resources:
requests: {cpu: 25m, memory: 64Mi}
limits: {cpu: 500m, memory: 128Mi}
volumeMounts:
- {name: host, mountPath: /host}
- {name: script, mountPath: /scripts, readOnly: true}
volumes:
- name: host
hostPath: {path: /, type: Directory}
- name: script
configMap: {name: ollama-gpu-reservation}