hermes: recover suite planner on healthy amd64 node
This commit is contained in:
parent
9255a64b23
commit
0574eb590c
@ -26,6 +26,10 @@ spec:
|
||||
kind: Deployment
|
||||
name: hermes-switchyard
|
||||
namespace: hermes
|
||||
- apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
name: hermes-suite-planner
|
||||
namespace: hermes
|
||||
- apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
name: hermes-agent
|
||||
@ -65,5 +69,5 @@ spec:
|
||||
- name: keycloak
|
||||
- name: longhorn
|
||||
- name: vault
|
||||
- name: jenkins
|
||||
# CI availability must not block recovery of the inference services.
|
||||
- name: hermes-observer-rbac
|
||||
|
||||
44
docs/hermes_suite_node_recovery_20260930.md
Normal file
44
docs/hermes_suite_node_recovery_20260930.md
Normal file
@ -0,0 +1,44 @@
|
||||
# Suite planner node recovery, 2026-09-30
|
||||
|
||||
Titan-22 stopped reporting at 15:40:50 UTC and became unreachable at
|
||||
15:46:05 UTC. The suite planner was evicted and could not reschedule: both
|
||||
its node selector and its local-path metadata volume required titan-22.
|
||||
The node was also unreachable over SSH. Its physical fault remains unknown.
|
||||
|
||||
The recovery deployment uses titan-24 with the same pinned amd64 Claude CLI,
|
||||
Opus 5.5 model, credentials, routing rules, prompts, and execution policy.
|
||||
It requests 100m CPU and 512Mi memory, with limits of 2 CPU and 2Gi memory.
|
||||
It requests no GPU and does not move another workload. Titan-24 is the
|
||||
available compatible host; the simulation host remains reserved.
|
||||
|
||||
The original `hermes-suite-metadata` PVC remains bound to titan-22 and is
|
||||
protected from Flux pruning. Do not delete it. The recovery deployment uses
|
||||
the separate `hermes-suite-metadata-recovery` local-path PVC on titan-24.
|
||||
Historical job status and idempotency records are unavailable until the old
|
||||
disk can be accessed. Previous results and active checkpoints were in memory
|
||||
and cannot be recovered by mounting a metadata database alone.
|
||||
|
||||
Use a fresh client job revision and Idempotency-Key for an intentionally new
|
||||
attempt after the recovery endpoint is verified. The old key cannot provide
|
||||
deduplication against the inaccessible ledger. No real suite is automatically
|
||||
resubmitted by this deployment. New jobs retain normal durable idempotency on
|
||||
the recovery volume. The new volume is still node-local, not replicated storage.
|
||||
|
||||
The HTTP URLs, token retrieval, request/response schema, 7200-second maximum,
|
||||
disabled estimated-cost guard, configuration revision `suite-v6-20260929`,
|
||||
prompt revision `implementation-proximity-adaptive-v6-20260930`, execution
|
||||
revision `suite-adaptive-v11-20260930`, and policy revision
|
||||
`implementation-five-v1-20260929` are unchanged.
|
||||
|
||||
Deployment annotation: `suite-v6-adaptive-v11-node-recovery-20260930`.
|
||||
Metadata-store annotation: `titan24-recovery-20260930`.
|
||||
|
||||
Hermes no longer depends on Jenkins readiness for Flux reconciliation. Jenkins
|
||||
is not needed by the inference request path; its outage had blocked recovery.
|
||||
The suite planner is now an explicit Hermes Flux health check.
|
||||
|
||||
Rollback must wait for titan-22 to be healthy and all recovery jobs/results to
|
||||
be collected. Stop new submissions, reconcile the two metadata ledgers with
|
||||
owner/idempotency conflicts rejected, then restore the node selector and volume
|
||||
reference through Git. Never switch back to an older ledger while accepting jobs,
|
||||
and retain both PVCs during recovery. Do not restart an active inference job.
|
||||
@ -11,6 +11,21 @@ kind: PersistentVolumeClaim
|
||||
metadata:
|
||||
name: hermes-suite-metadata
|
||||
namespace: hermes
|
||||
annotations:
|
||||
# Preserve titan-22's original ledger until the offline node can be recovered.
|
||||
kustomize.toolkit.fluxcd.io/prune: disabled
|
||||
spec:
|
||||
accessModes: [ReadWriteOnce]
|
||||
storageClassName: local-path
|
||||
resources:
|
||||
requests:
|
||||
storage: 1Gi
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: PersistentVolumeClaim
|
||||
metadata:
|
||||
name: hermes-suite-metadata-recovery
|
||||
namespace: hermes
|
||||
spec:
|
||||
accessModes: [ReadWriteOnce]
|
||||
storageClassName: local-path
|
||||
@ -36,7 +51,8 @@ spec:
|
||||
app: hermes-suite-planner
|
||||
annotations:
|
||||
fluentbit.io/exclude: "true"
|
||||
ai.bstein.dev/config-rev: suite-v6-adaptive-v11-20260930
|
||||
ai.bstein.dev/config-rev: suite-v6-adaptive-v11-node-recovery-20260930
|
||||
ai.bstein.dev/metadata-store: titan24-recovery-20260930
|
||||
vault.hashicorp.com/agent-inject: "true"
|
||||
vault.hashicorp.com/agent-pre-populate-only: "true"
|
||||
vault.hashicorp.com/agent-init-first: "true"
|
||||
@ -80,9 +96,9 @@ spec:
|
||||
automountServiceAccountToken: false
|
||||
enableServiceLinks: false
|
||||
terminationGracePeriodSeconds: 15
|
||||
# The pinned native CLI is amd64; retain the existing worker placement.
|
||||
# The pinned CLI needs amd64. Titan-22 is offline; no GPU is requested here.
|
||||
nodeSelector:
|
||||
kubernetes.io/hostname: titan-22
|
||||
kubernetes.io/hostname: titan-24
|
||||
securityContext:
|
||||
runAsNonRoot: true
|
||||
runAsUser: 10000
|
||||
@ -166,7 +182,7 @@ spec:
|
||||
emptyDir: {medium: Memory, sizeLimit: 64Mi}
|
||||
- name: state
|
||||
persistentVolumeClaim:
|
||||
claimName: hermes-suite-metadata
|
||||
claimName: hermes-suite-metadata-recovery
|
||||
- name: vault-auth
|
||||
projected:
|
||||
sources:
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user