core: quarantine workers with sustained runtime storage stalls
This commit is contained in:
parent
3355532e00
commit
7fa3bc9c34
@ -2,6 +2,7 @@
|
||||
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||
kind: Kustomization
|
||||
resources:
|
||||
- node-maintenance.yaml
|
||||
- ../modules/base
|
||||
- ../modules/profiles/atlas-ha
|
||||
- node-prefer-noschedule-serviceaccount.yaml
|
||||
|
||||
22
infrastructure/core/node-maintenance.yaml
Normal file
22
infrastructure/core/node-maintenance.yaml
Normal file
@ -0,0 +1,22 @@
|
||||
# infrastructure/core/node-maintenance.yaml
|
||||
# Block new application placement while the measured runtime I/O faults remain.
|
||||
# Existing workloads and storage DaemonSets are not drained by these resources.
|
||||
apiVersion: v1
|
||||
kind: Node
|
||||
metadata:
|
||||
name: titan-14
|
||||
annotations:
|
||||
kustomize.toolkit.fluxcd.io/prune: disabled
|
||||
atlas.bstein.dev/maintenance-reason: "Runtime USB flash saturation; replace or repair media and verify I/O before uncordoning."
|
||||
spec:
|
||||
unschedulable: true
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Node
|
||||
metadata:
|
||||
name: titan-18
|
||||
annotations:
|
||||
kustomize.toolkit.fluxcd.io/prune: disabled
|
||||
atlas.bstein.dev/maintenance-reason: "Sustained runtime I/O stalls prevent Longhorn engine startup; reduce load and verify storage before uncordoning."
|
||||
spec:
|
||||
unschedulable: true
|
||||
Loading…
x
Reference in New Issue
Block a user