longhorn: recover the guarded stranded tenant engine once

This commit is contained in:
jenkins 2026-10-04 03:49:36 -05:00
parent bb05ec9f7a
commit 4296912447
3 changed files with 114 additions and 0 deletions

View File

@ -9,12 +9,16 @@ resources:
- helmrelease.yaml
- cassandra-recurring-jobs.yaml
- monerod-recovery-snapshot.yaml
- tenant2-stale-engine-recovery-job.yaml
- longhorn-settings-ensure-job.yaml
- longhorn-csi-toleration-ensure-job.yaml
- longhorn-disk-tags-ensure-job.yaml
- jellyfin-volume-policy-cronjob.yaml
configMapGenerator:
- name: tenant2-stale-engine-recovery-script
files:
- longhorn_stale_engine_recovery.sh=scripts/longhorn_stale_engine_recovery.sh
- name: longhorn-settings-ensure-script
files:
- longhorn_settings_ensure.sh=scripts/longhorn_settings_ensure.sh

View File

@ -0,0 +1,59 @@
#!/usr/bin/env bash
# One-time recovery of the detached tenant-2 volume's stranded, unwritable engine.
# No CR deletion, replica removal, forced volume detach or shared-manager restart.
set -euo pipefail
volume=pvc-02d99a30-3757-4a01-abfb-5b30aef793d6
engine="${volume}-e-0"
manager=instance-manager-60fcef8ab4aad7407ce6d857ad2340f6
validate_state() {
# Inputs are API JSON files. Only the established stale-engine state is eligible.
jq -e --arg id "$volume" '
.metadata.name == $id and .status.state == "detached" and
.status.currentNodeID == "" and .status.robustness == "unknown"
' "$1" >/dev/null &&
jq -e --arg id "$engine" --arg manager "$manager" '
.metadata.name == $id and
.metadata.uid == "5ed29525-baf7-4768-91b4-4a38ac06a163" and
.metadata.deletionTimestamp == null and .spec.nodeID == "titan-23" and
.status.instanceManagerName == $manager and .status.currentState == "running" and
(.status.replicaModeMap | length > 0) and
(.status.replicaModeMap | all(. == "ERR"))
' "$2" >/dev/null &&
jq -e --arg id "$volume" '
(.items | length == 3) and
(.items | all(.spec.volumeName == $id and .spec.desireState == "stopped" and
.status.currentState == "stopped" and .spec.active == true))
' "$3" >/dev/null
}
main() {
# Local replica copies must exist before this Job is added to Flux.
jq -e --arg id "$volume" '
.volume == $id and .all_copies_completed == true and
(.nodes | sort == ["titan-13", "titan-15", "titan-17"])
' /recovery/copy-receipt.json >/dev/null
if [ -e /recovery/engine-reset-attempted ]; then
echo "Recovery attempt already recorded; no action."
return 0
fi
work=$(mktemp -d)
trap 'rm -rf "$work"' EXIT
kubectl -n longhorn-system get volumes.longhorn.io "$volume" -o json > "$work/volume.json"
kubectl -n longhorn-system get engines.longhorn.io "$engine" -o json > "$work/engine.json"
kubectl -n longhorn-system get replicas.longhorn.io -l "longhornvolume=$volume" -o json > "$work/replicas.json"
if ! validate_state "$work/volume.json" "$work/engine.json" "$work/replicas.json"; then
echo "Recovery preconditions changed; no action."
return 0
fi
# Record before acting so neither Flux nor a failed client can repeat the reset.
(set -o noclobber; date -u +%FT%TZ > /recovery/engine-reset-attempted)
kubectl -n longhorn-system exec "$manager" -- \
longhorn-instance-manager process delete --name "$engine" >/dev/null
echo "Requested one native reset of the stranded engine; controllers own recovery."
}
if [[ "${BASH_SOURCE[0]}" == "$0" ]]; then
main
fi

View File

@ -0,0 +1,51 @@
# infrastructure/longhorn/core/tenant2-stale-engine-recovery-job.yaml
apiVersion: batch/v1
kind: Job
metadata:
name: tenant2-stale-engine-recovery-1
namespace: longhorn-system
spec:
backoffLimit: 0
activeDeadlineSeconds: 240
# Keep the completed Job; the host receipt also prevents repeated execution.
template:
spec:
serviceAccountName: longhorn-service-account
restartPolicy: Never
nodeSelector:
kubernetes.io/hostname: titan-13
securityContext:
runAsUser: 0
runAsGroup: 0
seccompProfile:
type: RuntimeDefault
containers:
- name: recover
image: bitnami/kubectl@sha256:554ab88b1858e8424c55de37ad417b16f2a0e65d1607aa0f3fe3ce9b9f10b131
command: ["/bin/bash", "/scripts/longhorn_stale_engine_recovery.sh"]
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop: ["ALL"]
resources:
requests:
cpu: 10m
memory: 32Mi
limits:
cpu: 100m
memory: 128Mi
volumeMounts:
- name: script
mountPath: /scripts
readOnly: true
- name: recovery
mountPath: /recovery
volumes:
- name: script
configMap:
name: tenant2-stale-engine-recovery-script
defaultMode: 0444
- name: recovery
hostPath:
path: /mnt/astreae/atlas-recovery/tenant2-20261004
type: Directory