longhorn: recover the guarded stranded tenant engine once
This commit is contained in:
parent
bb05ec9f7a
commit
4296912447
@ -9,12 +9,16 @@ resources:
|
|||||||
- helmrelease.yaml
|
- helmrelease.yaml
|
||||||
- cassandra-recurring-jobs.yaml
|
- cassandra-recurring-jobs.yaml
|
||||||
- monerod-recovery-snapshot.yaml
|
- monerod-recovery-snapshot.yaml
|
||||||
|
- tenant2-stale-engine-recovery-job.yaml
|
||||||
- longhorn-settings-ensure-job.yaml
|
- longhorn-settings-ensure-job.yaml
|
||||||
- longhorn-csi-toleration-ensure-job.yaml
|
- longhorn-csi-toleration-ensure-job.yaml
|
||||||
- longhorn-disk-tags-ensure-job.yaml
|
- longhorn-disk-tags-ensure-job.yaml
|
||||||
- jellyfin-volume-policy-cronjob.yaml
|
- jellyfin-volume-policy-cronjob.yaml
|
||||||
|
|
||||||
configMapGenerator:
|
configMapGenerator:
|
||||||
|
- name: tenant2-stale-engine-recovery-script
|
||||||
|
files:
|
||||||
|
- longhorn_stale_engine_recovery.sh=scripts/longhorn_stale_engine_recovery.sh
|
||||||
- name: longhorn-settings-ensure-script
|
- name: longhorn-settings-ensure-script
|
||||||
files:
|
files:
|
||||||
- longhorn_settings_ensure.sh=scripts/longhorn_settings_ensure.sh
|
- longhorn_settings_ensure.sh=scripts/longhorn_settings_ensure.sh
|
||||||
|
|||||||
@ -0,0 +1,59 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# One-time recovery of the detached tenant-2 volume's stranded, unwritable engine.
|
||||||
|
# No CR deletion, replica removal, forced volume detach or shared-manager restart.
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
volume=pvc-02d99a30-3757-4a01-abfb-5b30aef793d6
|
||||||
|
engine="${volume}-e-0"
|
||||||
|
manager=instance-manager-60fcef8ab4aad7407ce6d857ad2340f6
|
||||||
|
|
||||||
|
validate_state() {
|
||||||
|
# Inputs are API JSON files. Only the established stale-engine state is eligible.
|
||||||
|
jq -e --arg id "$volume" '
|
||||||
|
.metadata.name == $id and .status.state == "detached" and
|
||||||
|
.status.currentNodeID == "" and .status.robustness == "unknown"
|
||||||
|
' "$1" >/dev/null &&
|
||||||
|
jq -e --arg id "$engine" --arg manager "$manager" '
|
||||||
|
.metadata.name == $id and
|
||||||
|
.metadata.uid == "5ed29525-baf7-4768-91b4-4a38ac06a163" and
|
||||||
|
.metadata.deletionTimestamp == null and .spec.nodeID == "titan-23" and
|
||||||
|
.status.instanceManagerName == $manager and .status.currentState == "running" and
|
||||||
|
(.status.replicaModeMap | length > 0) and
|
||||||
|
(.status.replicaModeMap | all(. == "ERR"))
|
||||||
|
' "$2" >/dev/null &&
|
||||||
|
jq -e --arg id "$volume" '
|
||||||
|
(.items | length == 3) and
|
||||||
|
(.items | all(.spec.volumeName == $id and .spec.desireState == "stopped" and
|
||||||
|
.status.currentState == "stopped" and .spec.active == true))
|
||||||
|
' "$3" >/dev/null
|
||||||
|
}
|
||||||
|
|
||||||
|
main() {
|
||||||
|
# Local replica copies must exist before this Job is added to Flux.
|
||||||
|
jq -e --arg id "$volume" '
|
||||||
|
.volume == $id and .all_copies_completed == true and
|
||||||
|
(.nodes | sort == ["titan-13", "titan-15", "titan-17"])
|
||||||
|
' /recovery/copy-receipt.json >/dev/null
|
||||||
|
if [ -e /recovery/engine-reset-attempted ]; then
|
||||||
|
echo "Recovery attempt already recorded; no action."
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
work=$(mktemp -d)
|
||||||
|
trap 'rm -rf "$work"' EXIT
|
||||||
|
kubectl -n longhorn-system get volumes.longhorn.io "$volume" -o json > "$work/volume.json"
|
||||||
|
kubectl -n longhorn-system get engines.longhorn.io "$engine" -o json > "$work/engine.json"
|
||||||
|
kubectl -n longhorn-system get replicas.longhorn.io -l "longhornvolume=$volume" -o json > "$work/replicas.json"
|
||||||
|
if ! validate_state "$work/volume.json" "$work/engine.json" "$work/replicas.json"; then
|
||||||
|
echo "Recovery preconditions changed; no action."
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
# Record before acting so neither Flux nor a failed client can repeat the reset.
|
||||||
|
(set -o noclobber; date -u +%FT%TZ > /recovery/engine-reset-attempted)
|
||||||
|
kubectl -n longhorn-system exec "$manager" -- \
|
||||||
|
longhorn-instance-manager process delete --name "$engine" >/dev/null
|
||||||
|
echo "Requested one native reset of the stranded engine; controllers own recovery."
|
||||||
|
}
|
||||||
|
|
||||||
|
if [[ "${BASH_SOURCE[0]}" == "$0" ]]; then
|
||||||
|
main
|
||||||
|
fi
|
||||||
@ -0,0 +1,51 @@
|
|||||||
|
# infrastructure/longhorn/core/tenant2-stale-engine-recovery-job.yaml
|
||||||
|
apiVersion: batch/v1
|
||||||
|
kind: Job
|
||||||
|
metadata:
|
||||||
|
name: tenant2-stale-engine-recovery-1
|
||||||
|
namespace: longhorn-system
|
||||||
|
spec:
|
||||||
|
backoffLimit: 0
|
||||||
|
activeDeadlineSeconds: 240
|
||||||
|
# Keep the completed Job; the host receipt also prevents repeated execution.
|
||||||
|
template:
|
||||||
|
spec:
|
||||||
|
serviceAccountName: longhorn-service-account
|
||||||
|
restartPolicy: Never
|
||||||
|
nodeSelector:
|
||||||
|
kubernetes.io/hostname: titan-13
|
||||||
|
securityContext:
|
||||||
|
runAsUser: 0
|
||||||
|
runAsGroup: 0
|
||||||
|
seccompProfile:
|
||||||
|
type: RuntimeDefault
|
||||||
|
containers:
|
||||||
|
- name: recover
|
||||||
|
image: bitnami/kubectl@sha256:554ab88b1858e8424c55de37ad417b16f2a0e65d1607aa0f3fe3ce9b9f10b131
|
||||||
|
command: ["/bin/bash", "/scripts/longhorn_stale_engine_recovery.sh"]
|
||||||
|
securityContext:
|
||||||
|
allowPrivilegeEscalation: false
|
||||||
|
capabilities:
|
||||||
|
drop: ["ALL"]
|
||||||
|
resources:
|
||||||
|
requests:
|
||||||
|
cpu: 10m
|
||||||
|
memory: 32Mi
|
||||||
|
limits:
|
||||||
|
cpu: 100m
|
||||||
|
memory: 128Mi
|
||||||
|
volumeMounts:
|
||||||
|
- name: script
|
||||||
|
mountPath: /scripts
|
||||||
|
readOnly: true
|
||||||
|
- name: recovery
|
||||||
|
mountPath: /recovery
|
||||||
|
volumes:
|
||||||
|
- name: script
|
||||||
|
configMap:
|
||||||
|
name: tenant2-stale-engine-recovery-script
|
||||||
|
defaultMode: 0444
|
||||||
|
- name: recovery
|
||||||
|
hostPath:
|
||||||
|
path: /mnt/astreae/atlas-recovery/tenant2-20261004
|
||||||
|
type: Directory
|
||||||
Loading…
x
Reference in New Issue
Block a user