core: quarantine workers with sustained runtime storage stalls

This commit is contained in:
jenkins 2026-10-03 01:04:58 -05:00
parent 3355532e00
commit 7fa3bc9c34
2 changed files with 23 additions and 0 deletions

View File

@ -2,6 +2,7 @@
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
- node-maintenance.yaml
- ../modules/base
- ../modules/profiles/atlas-ha
- node-prefer-noschedule-serviceaccount.yaml

View File

@ -0,0 +1,22 @@
# infrastructure/core/node-maintenance.yaml
# Block new application placement while the measured runtime I/O faults remain.
# Existing workloads and storage DaemonSets are not drained by these resources.
apiVersion: v1
kind: Node
metadata:
name: titan-14
annotations:
kustomize.toolkit.fluxcd.io/prune: disabled
atlas.bstein.dev/maintenance-reason: "Runtime USB flash saturation; replace or repair media and verify I/O before uncordoning."
spec:
unschedulable: true
---
apiVersion: v1
kind: Node
metadata:
name: titan-18
annotations:
kustomize.toolkit.fluxcd.io/prune: disabled
atlas.bstein.dev/maintenance-reason: "Sustained runtime I/O stalls prevent Longhorn engine startup; reduce load and verify storage before uncordoning."
spec:
unschedulable: true