From 0574eb590cd2a1ddae8a12e3535353be0c3699f5 Mon Sep 17 00:00:00 2001 From: jenkins Date: Wed, 30 Sep 2026 11:05:41 -0500 Subject: [PATCH] hermes: recover suite planner on healthy amd64 node --- .../applications/hermes/kustomization.yaml | 6 ++- docs/hermes_suite_node_recovery_20260930.md | 44 +++++++++++++++++++ services/hermes/suite-planner-deployment.yaml | 24 ++++++++-- 3 files changed, 69 insertions(+), 5 deletions(-) create mode 100644 docs/hermes_suite_node_recovery_20260930.md diff --git a/clusters/atlas/flux-system/applications/hermes/kustomization.yaml b/clusters/atlas/flux-system/applications/hermes/kustomization.yaml index 8f61d2e5..514f2a3b 100644 --- a/clusters/atlas/flux-system/applications/hermes/kustomization.yaml +++ b/clusters/atlas/flux-system/applications/hermes/kustomization.yaml @@ -26,6 +26,10 @@ spec: kind: Deployment name: hermes-switchyard namespace: hermes + - apiVersion: apps/v1 + kind: Deployment + name: hermes-suite-planner + namespace: hermes - apiVersion: apps/v1 kind: Deployment name: hermes-agent @@ -65,5 +69,5 @@ spec: - name: keycloak - name: longhorn - name: vault - - name: jenkins + # CI availability must not block recovery of the inference services. - name: hermes-observer-rbac diff --git a/docs/hermes_suite_node_recovery_20260930.md b/docs/hermes_suite_node_recovery_20260930.md new file mode 100644 index 00000000..5f925941 --- /dev/null +++ b/docs/hermes_suite_node_recovery_20260930.md @@ -0,0 +1,44 @@ +# Suite planner node recovery, 2026-09-30 + +Titan-22 stopped reporting at 15:40:50 UTC and became unreachable at +15:46:05 UTC. The suite planner was evicted and could not reschedule: both +its node selector and its local-path metadata volume required titan-22. +The node was also unreachable over SSH. Its physical fault remains unknown. + +The recovery deployment uses titan-24 with the same pinned amd64 Claude CLI, +Opus 5.5 model, credentials, routing rules, prompts, and execution policy. +It requests 100m CPU and 512Mi memory, with limits of 2 CPU and 2Gi memory. +It requests no GPU and does not move another workload. Titan-24 is the +available compatible host; the simulation host remains reserved. + +The original `hermes-suite-metadata` PVC remains bound to titan-22 and is +protected from Flux pruning. Do not delete it. The recovery deployment uses +the separate `hermes-suite-metadata-recovery` local-path PVC on titan-24. +Historical job status and idempotency records are unavailable until the old +disk can be accessed. Previous results and active checkpoints were in memory +and cannot be recovered by mounting a metadata database alone. + +Use a fresh client job revision and Idempotency-Key for an intentionally new +attempt after the recovery endpoint is verified. The old key cannot provide +deduplication against the inaccessible ledger. No real suite is automatically +resubmitted by this deployment. New jobs retain normal durable idempotency on +the recovery volume. The new volume is still node-local, not replicated storage. + +The HTTP URLs, token retrieval, request/response schema, 7200-second maximum, +disabled estimated-cost guard, configuration revision `suite-v6-20260929`, +prompt revision `implementation-proximity-adaptive-v6-20260930`, execution +revision `suite-adaptive-v11-20260930`, and policy revision +`implementation-five-v1-20260929` are unchanged. + +Deployment annotation: `suite-v6-adaptive-v11-node-recovery-20260930`. +Metadata-store annotation: `titan24-recovery-20260930`. + +Hermes no longer depends on Jenkins readiness for Flux reconciliation. Jenkins +is not needed by the inference request path; its outage had blocked recovery. +The suite planner is now an explicit Hermes Flux health check. + +Rollback must wait for titan-22 to be healthy and all recovery jobs/results to +be collected. Stop new submissions, reconcile the two metadata ledgers with +owner/idempotency conflicts rejected, then restore the node selector and volume +reference through Git. Never switch back to an older ledger while accepting jobs, +and retain both PVCs during recovery. Do not restart an active inference job. diff --git a/services/hermes/suite-planner-deployment.yaml b/services/hermes/suite-planner-deployment.yaml index 16c21e5c..90de269e 100644 --- a/services/hermes/suite-planner-deployment.yaml +++ b/services/hermes/suite-planner-deployment.yaml @@ -11,6 +11,21 @@ kind: PersistentVolumeClaim metadata: name: hermes-suite-metadata namespace: hermes + annotations: + # Preserve titan-22's original ledger until the offline node can be recovered. + kustomize.toolkit.fluxcd.io/prune: disabled +spec: + accessModes: [ReadWriteOnce] + storageClassName: local-path + resources: + requests: + storage: 1Gi +--- +apiVersion: v1 +kind: PersistentVolumeClaim +metadata: + name: hermes-suite-metadata-recovery + namespace: hermes spec: accessModes: [ReadWriteOnce] storageClassName: local-path @@ -36,7 +51,8 @@ spec: app: hermes-suite-planner annotations: fluentbit.io/exclude: "true" - ai.bstein.dev/config-rev: suite-v6-adaptive-v11-20260930 + ai.bstein.dev/config-rev: suite-v6-adaptive-v11-node-recovery-20260930 + ai.bstein.dev/metadata-store: titan24-recovery-20260930 vault.hashicorp.com/agent-inject: "true" vault.hashicorp.com/agent-pre-populate-only: "true" vault.hashicorp.com/agent-init-first: "true" @@ -80,9 +96,9 @@ spec: automountServiceAccountToken: false enableServiceLinks: false terminationGracePeriodSeconds: 15 - # The pinned native CLI is amd64; retain the existing worker placement. + # The pinned CLI needs amd64. Titan-22 is offline; no GPU is requested here. nodeSelector: - kubernetes.io/hostname: titan-22 + kubernetes.io/hostname: titan-24 securityContext: runAsNonRoot: true runAsUser: 10000 @@ -166,7 +182,7 @@ spec: emptyDir: {medium: Memory, sizeLimit: 64Mi} - name: state persistentVolumeClaim: - claimName: hermes-suite-metadata + claimName: hermes-suite-metadata-recovery - name: vault-auth projected: sources: