diff --git a/README.md b/README.md index 4436b98..7e6bd6d 100644 --- a/README.md +++ b/README.md @@ -75,8 +75,9 @@ an existing pod UID whose container or init container is still in ErrImagePull or ImagePullBackOff. Completed and deleting pods are ignored. A helper restart is reserved in the existing run history before execution and may occur at most once per helper in 30 minutes, including failed rollouts. History must remain -writable; otherwise this repair fails closed. It is a single-coordinator control, -not a cross-host distributed lock. Do not run two coordinators against one cluster. +writable; otherwise this repair fails closed. Only the coordinator daemon runs periodic shared-cluster repairs; the peer +continues UPS monitoring and shutdown forwarding. Do not configure two hosts as +coordinators for one cluster. A targeted daemon fix can use the existing installer without rewriting host configuration, UPS settings, or systemd/bootstrap units: diff --git a/internal/service/daemon.go b/internal/service/daemon.go index 3191221..9a0e6b4 100644 --- a/internal/service/daemon.go +++ b/internal/service/daemon.go @@ -217,6 +217,11 @@ func (d *Daemon) Run(ctx context.Context) error { // like a later Vault reseal or stale dead-node deletions without waiting for a // fresh bootstrap run. func (d *Daemon) maybeRunPostStartAutoHeal(ctx context.Context, lastRun *time.Time, anyOnBattery bool) { + // Shared cluster repairs have one owner; peers still monitor their UPS and + // forward shutdown intent through the existing coordination path. + if d.cfg.Coordination.Role == "peer" { + return + } interval := time.Duration(d.cfg.Startup.PostStartAutoHealSeconds) * time.Second if interval <= 0 || anyOnBattery { return diff --git a/internal/service/daemon_poststart_autorepair_test.go b/internal/service/daemon_poststart_autorepair_test.go index 1aa34a8..9737ed1 100644 --- a/internal/service/daemon_poststart_autorepair_test.go +++ b/internal/service/daemon_poststart_autorepair_test.go @@ -26,8 +26,14 @@ func TestDaemonMaybeRunPostStartAutoHeal(t *testing.T) { }, } + d.cfg.Coordination.Role = "peer" var last time.Time d.maybeRunPostStartAutoHeal(context.Background(), &last, false) + if calls != 0 || !last.IsZero() { + t.Fatal("peer must not mutate shared cluster recovery state") + } + d.cfg.Coordination.Role = "coordinator" + d.maybeRunPostStartAutoHeal(context.Background(), &last, false) if calls != 1 { t.Fatalf("expected first auto-heal invocation, got %d", calls) }