longhorn: recover the guarded stranded tenant engine once
This commit is contained in:
parent
bb05ec9f7a
commit
4296912447
@ -9,12 +9,16 @@ resources:
|
||||
- helmrelease.yaml
|
||||
- cassandra-recurring-jobs.yaml
|
||||
- monerod-recovery-snapshot.yaml
|
||||
- tenant2-stale-engine-recovery-job.yaml
|
||||
- longhorn-settings-ensure-job.yaml
|
||||
- longhorn-csi-toleration-ensure-job.yaml
|
||||
- longhorn-disk-tags-ensure-job.yaml
|
||||
- jellyfin-volume-policy-cronjob.yaml
|
||||
|
||||
configMapGenerator:
|
||||
- name: tenant2-stale-engine-recovery-script
|
||||
files:
|
||||
- longhorn_stale_engine_recovery.sh=scripts/longhorn_stale_engine_recovery.sh
|
||||
- name: longhorn-settings-ensure-script
|
||||
files:
|
||||
- longhorn_settings_ensure.sh=scripts/longhorn_settings_ensure.sh
|
||||
|
||||
@ -0,0 +1,59 @@
|
||||
#!/usr/bin/env bash
|
||||
# One-time recovery of the detached tenant-2 volume's stranded, unwritable engine.
|
||||
# No CR deletion, replica removal, forced volume detach or shared-manager restart.
|
||||
set -euo pipefail
|
||||
|
||||
volume=pvc-02d99a30-3757-4a01-abfb-5b30aef793d6
|
||||
engine="${volume}-e-0"
|
||||
manager=instance-manager-60fcef8ab4aad7407ce6d857ad2340f6
|
||||
|
||||
validate_state() {
|
||||
# Inputs are API JSON files. Only the established stale-engine state is eligible.
|
||||
jq -e --arg id "$volume" '
|
||||
.metadata.name == $id and .status.state == "detached" and
|
||||
.status.currentNodeID == "" and .status.robustness == "unknown"
|
||||
' "$1" >/dev/null &&
|
||||
jq -e --arg id "$engine" --arg manager "$manager" '
|
||||
.metadata.name == $id and
|
||||
.metadata.uid == "5ed29525-baf7-4768-91b4-4a38ac06a163" and
|
||||
.metadata.deletionTimestamp == null and .spec.nodeID == "titan-23" and
|
||||
.status.instanceManagerName == $manager and .status.currentState == "running" and
|
||||
(.status.replicaModeMap | length > 0) and
|
||||
(.status.replicaModeMap | all(. == "ERR"))
|
||||
' "$2" >/dev/null &&
|
||||
jq -e --arg id "$volume" '
|
||||
(.items | length == 3) and
|
||||
(.items | all(.spec.volumeName == $id and .spec.desireState == "stopped" and
|
||||
.status.currentState == "stopped" and .spec.active == true))
|
||||
' "$3" >/dev/null
|
||||
}
|
||||
|
||||
main() {
|
||||
# Local replica copies must exist before this Job is added to Flux.
|
||||
jq -e --arg id "$volume" '
|
||||
.volume == $id and .all_copies_completed == true and
|
||||
(.nodes | sort == ["titan-13", "titan-15", "titan-17"])
|
||||
' /recovery/copy-receipt.json >/dev/null
|
||||
if [ -e /recovery/engine-reset-attempted ]; then
|
||||
echo "Recovery attempt already recorded; no action."
|
||||
return 0
|
||||
fi
|
||||
work=$(mktemp -d)
|
||||
trap 'rm -rf "$work"' EXIT
|
||||
kubectl -n longhorn-system get volumes.longhorn.io "$volume" -o json > "$work/volume.json"
|
||||
kubectl -n longhorn-system get engines.longhorn.io "$engine" -o json > "$work/engine.json"
|
||||
kubectl -n longhorn-system get replicas.longhorn.io -l "longhornvolume=$volume" -o json > "$work/replicas.json"
|
||||
if ! validate_state "$work/volume.json" "$work/engine.json" "$work/replicas.json"; then
|
||||
echo "Recovery preconditions changed; no action."
|
||||
return 0
|
||||
fi
|
||||
# Record before acting so neither Flux nor a failed client can repeat the reset.
|
||||
(set -o noclobber; date -u +%FT%TZ > /recovery/engine-reset-attempted)
|
||||
kubectl -n longhorn-system exec "$manager" -- \
|
||||
longhorn-instance-manager process delete --name "$engine" >/dev/null
|
||||
echo "Requested one native reset of the stranded engine; controllers own recovery."
|
||||
}
|
||||
|
||||
if [[ "${BASH_SOURCE[0]}" == "$0" ]]; then
|
||||
main
|
||||
fi
|
||||
@ -0,0 +1,51 @@
|
||||
# infrastructure/longhorn/core/tenant2-stale-engine-recovery-job.yaml
|
||||
apiVersion: batch/v1
|
||||
kind: Job
|
||||
metadata:
|
||||
name: tenant2-stale-engine-recovery-1
|
||||
namespace: longhorn-system
|
||||
spec:
|
||||
backoffLimit: 0
|
||||
activeDeadlineSeconds: 240
|
||||
# Keep the completed Job; the host receipt also prevents repeated execution.
|
||||
template:
|
||||
spec:
|
||||
serviceAccountName: longhorn-service-account
|
||||
restartPolicy: Never
|
||||
nodeSelector:
|
||||
kubernetes.io/hostname: titan-13
|
||||
securityContext:
|
||||
runAsUser: 0
|
||||
runAsGroup: 0
|
||||
seccompProfile:
|
||||
type: RuntimeDefault
|
||||
containers:
|
||||
- name: recover
|
||||
image: bitnami/kubectl@sha256:554ab88b1858e8424c55de37ad417b16f2a0e65d1607aa0f3fe3ce9b9f10b131
|
||||
command: ["/bin/bash", "/scripts/longhorn_stale_engine_recovery.sh"]
|
||||
securityContext:
|
||||
allowPrivilegeEscalation: false
|
||||
capabilities:
|
||||
drop: ["ALL"]
|
||||
resources:
|
||||
requests:
|
||||
cpu: 10m
|
||||
memory: 32Mi
|
||||
limits:
|
||||
cpu: 100m
|
||||
memory: 128Mi
|
||||
volumeMounts:
|
||||
- name: script
|
||||
mountPath: /scripts
|
||||
readOnly: true
|
||||
- name: recovery
|
||||
mountPath: /recovery
|
||||
volumes:
|
||||
- name: script
|
||||
configMap:
|
||||
name: tenant2-stale-engine-recovery-script
|
||||
defaultMode: 0444
|
||||
- name: recovery
|
||||
hostPath:
|
||||
path: /mnt/astreae/atlas-recovery/tenant2-20261004
|
||||
type: Directory
|
||||
Loading…
x
Reference in New Issue
Block a user