diff --git a/infrastructure/longhorn/core/longhorn-settings-ensure-job.yaml b/infrastructure/longhorn/core/longhorn-settings-ensure-job.yaml index 241459ba..5a98d6ca 100644 --- a/infrastructure/longhorn/core/longhorn-settings-ensure-job.yaml +++ b/infrastructure/longhorn/core/longhorn-settings-ensure-job.yaml @@ -2,7 +2,7 @@ apiVersion: batch/v1 kind: Job metadata: - name: longhorn-settings-ensure-10 + name: longhorn-settings-ensure-11 namespace: longhorn-system spec: backoffLimit: 0 diff --git a/infrastructure/longhorn/core/scripts/longhorn_settings_ensure.sh b/infrastructure/longhorn/core/scripts/longhorn_settings_ensure.sh index c02adc45..e718feb1 100644 --- a/infrastructure/longhorn/core/scripts/longhorn_settings_ensure.sh +++ b/infrastructure/longhorn/core/scripts/longhorn_settings_ensure.sh @@ -57,7 +57,8 @@ update_setting default-instance-manager-image "registry.bstein.dev/infra/longhor update_setting default-backing-image-manager-image "registry.bstein.dev/infra/longhorn-backing-image-manager:v1.8.2" update_setting support-bundle-manager-image "registry.bstein.dev/infra/longhorn-support-bundle-kit:v0.0.56" update_setting taint-toleration "veles.bstein.dev/simulation=true:NoSchedule" -# Keep storage-heavy nodes from getting hammered by rebuild storms and skew. -update_setting replica-auto-balance "best-effort" -update_setting concurrent-replica-rebuild-per-node-limit "2" +# Preserve redundancy without shuffling healthy replicas just to even the spread. +# Serialize rebuilds on the Pi storage nodes to leave I/O for running services. +update_setting replica-auto-balance "least-effort" +update_setting concurrent-replica-rebuild-per-node-limit "1" update_setting node-down-pod-deletion-policy "delete-both-statefulset-and-deployment-pod"