diff --git a/infrastructure/longhorn/core/kustomization.yaml b/infrastructure/longhorn/core/kustomization.yaml index cf91e571..846e04a7 100644 --- a/infrastructure/longhorn/core/kustomization.yaml +++ b/infrastructure/longhorn/core/kustomization.yaml @@ -9,12 +9,16 @@ resources: - helmrelease.yaml - cassandra-recurring-jobs.yaml - monerod-recovery-snapshot.yaml + - tenant2-stale-engine-recovery-job.yaml - longhorn-settings-ensure-job.yaml - longhorn-csi-toleration-ensure-job.yaml - longhorn-disk-tags-ensure-job.yaml - jellyfin-volume-policy-cronjob.yaml configMapGenerator: + - name: tenant2-stale-engine-recovery-script + files: + - longhorn_stale_engine_recovery.sh=scripts/longhorn_stale_engine_recovery.sh - name: longhorn-settings-ensure-script files: - longhorn_settings_ensure.sh=scripts/longhorn_settings_ensure.sh diff --git a/infrastructure/longhorn/core/scripts/longhorn_stale_engine_recovery.sh b/infrastructure/longhorn/core/scripts/longhorn_stale_engine_recovery.sh new file mode 100644 index 00000000..7ba80eed --- /dev/null +++ b/infrastructure/longhorn/core/scripts/longhorn_stale_engine_recovery.sh @@ -0,0 +1,59 @@ +#!/usr/bin/env bash +# One-time recovery of the detached tenant-2 volume's stranded, unwritable engine. +# No CR deletion, replica removal, forced volume detach or shared-manager restart. +set -euo pipefail + +volume=pvc-02d99a30-3757-4a01-abfb-5b30aef793d6 +engine="${volume}-e-0" +manager=instance-manager-60fcef8ab4aad7407ce6d857ad2340f6 + +validate_state() { + # Inputs are API JSON files. Only the established stale-engine state is eligible. + jq -e --arg id "$volume" ' + .metadata.name == $id and .status.state == "detached" and + .status.currentNodeID == "" and .status.robustness == "unknown" + ' "$1" >/dev/null && + jq -e --arg id "$engine" --arg manager "$manager" ' + .metadata.name == $id and + .metadata.uid == "5ed29525-baf7-4768-91b4-4a38ac06a163" and + .metadata.deletionTimestamp == null and .spec.nodeID == "titan-23" and + .status.instanceManagerName == $manager and .status.currentState == "running" and + (.status.replicaModeMap | length > 0) and + (.status.replicaModeMap | all(. == "ERR")) + ' "$2" >/dev/null && + jq -e --arg id "$volume" ' + (.items | length == 3) and + (.items | all(.spec.volumeName == $id and .spec.desireState == "stopped" and + .status.currentState == "stopped" and .spec.active == true)) + ' "$3" >/dev/null +} + +main() { + # Local replica copies must exist before this Job is added to Flux. + jq -e --arg id "$volume" ' + .volume == $id and .all_copies_completed == true and + (.nodes | sort == ["titan-13", "titan-15", "titan-17"]) + ' /recovery/copy-receipt.json >/dev/null + if [ -e /recovery/engine-reset-attempted ]; then + echo "Recovery attempt already recorded; no action." + return 0 + fi + work=$(mktemp -d) + trap 'rm -rf "$work"' EXIT + kubectl -n longhorn-system get volumes.longhorn.io "$volume" -o json > "$work/volume.json" + kubectl -n longhorn-system get engines.longhorn.io "$engine" -o json > "$work/engine.json" + kubectl -n longhorn-system get replicas.longhorn.io -l "longhornvolume=$volume" -o json > "$work/replicas.json" + if ! validate_state "$work/volume.json" "$work/engine.json" "$work/replicas.json"; then + echo "Recovery preconditions changed; no action." + return 0 + fi + # Record before acting so neither Flux nor a failed client can repeat the reset. + (set -o noclobber; date -u +%FT%TZ > /recovery/engine-reset-attempted) + kubectl -n longhorn-system exec "$manager" -- \ + longhorn-instance-manager process delete --name "$engine" >/dev/null + echo "Requested one native reset of the stranded engine; controllers own recovery." +} + +if [[ "${BASH_SOURCE[0]}" == "$0" ]]; then + main +fi diff --git a/infrastructure/longhorn/core/tenant2-stale-engine-recovery-job.yaml b/infrastructure/longhorn/core/tenant2-stale-engine-recovery-job.yaml new file mode 100644 index 00000000..89a16a7e --- /dev/null +++ b/infrastructure/longhorn/core/tenant2-stale-engine-recovery-job.yaml @@ -0,0 +1,51 @@ +# infrastructure/longhorn/core/tenant2-stale-engine-recovery-job.yaml +apiVersion: batch/v1 +kind: Job +metadata: + name: tenant2-stale-engine-recovery-1 + namespace: longhorn-system +spec: + backoffLimit: 0 + activeDeadlineSeconds: 240 + # Keep the completed Job; the host receipt also prevents repeated execution. + template: + spec: + serviceAccountName: longhorn-service-account + restartPolicy: Never + nodeSelector: + kubernetes.io/hostname: titan-13 + securityContext: + runAsUser: 0 + runAsGroup: 0 + seccompProfile: + type: RuntimeDefault + containers: + - name: recover + image: bitnami/kubectl@sha256:554ab88b1858e8424c55de37ad417b16f2a0e65d1607aa0f3fe3ce9b9f10b131 + command: ["/bin/bash", "/scripts/longhorn_stale_engine_recovery.sh"] + securityContext: + allowPrivilegeEscalation: false + capabilities: + drop: ["ALL"] + resources: + requests: + cpu: 10m + memory: 32Mi + limits: + cpu: 100m + memory: 128Mi + volumeMounts: + - name: script + mountPath: /scripts + readOnly: true + - name: recovery + mountPath: /recovery + volumes: + - name: script + configMap: + name: tenant2-stale-engine-recovery-script + defaultMode: 0444 + - name: recovery + hostPath: + path: /mnt/astreae/atlas-recovery/tenant2-20261004 + type: Directory