recovery: give shared repairs one coordinator
This commit is contained in:
parent
fc0ba95eb7
commit
ec7c2ee053
@ -75,8 +75,9 @@ an existing pod UID whose container or init container is still in ErrImagePull
|
||||
or ImagePullBackOff. Completed and deleting pods are ignored. A helper restart
|
||||
is reserved in the existing run history before execution and may occur at most
|
||||
once per helper in 30 minutes, including failed rollouts. History must remain
|
||||
writable; otherwise this repair fails closed. It is a single-coordinator control,
|
||||
not a cross-host distributed lock. Do not run two coordinators against one cluster.
|
||||
writable; otherwise this repair fails closed. Only the coordinator daemon runs periodic shared-cluster repairs; the peer
|
||||
continues UPS monitoring and shutdown forwarding. Do not configure two hosts as
|
||||
coordinators for one cluster.
|
||||
|
||||
A targeted daemon fix can use the existing installer without rewriting host
|
||||
configuration, UPS settings, or systemd/bootstrap units:
|
||||
|
||||
@ -217,6 +217,11 @@ func (d *Daemon) Run(ctx context.Context) error {
|
||||
// like a later Vault reseal or stale dead-node deletions without waiting for a
|
||||
// fresh bootstrap run.
|
||||
func (d *Daemon) maybeRunPostStartAutoHeal(ctx context.Context, lastRun *time.Time, anyOnBattery bool) {
|
||||
// Shared cluster repairs have one owner; peers still monitor their UPS and
|
||||
// forward shutdown intent through the existing coordination path.
|
||||
if d.cfg.Coordination.Role == "peer" {
|
||||
return
|
||||
}
|
||||
interval := time.Duration(d.cfg.Startup.PostStartAutoHealSeconds) * time.Second
|
||||
if interval <= 0 || anyOnBattery {
|
||||
return
|
||||
|
||||
@ -26,8 +26,14 @@ func TestDaemonMaybeRunPostStartAutoHeal(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
d.cfg.Coordination.Role = "peer"
|
||||
var last time.Time
|
||||
d.maybeRunPostStartAutoHeal(context.Background(), &last, false)
|
||||
if calls != 0 || !last.IsZero() {
|
||||
t.Fatal("peer must not mutate shared cluster recovery state")
|
||||
}
|
||||
d.cfg.Coordination.Role = "coordinator"
|
||||
d.maybeRunPostStartAutoHeal(context.Background(), &last, false)
|
||||
if calls != 1 {
|
||||
t.Fatalf("expected first auto-heal invocation, got %d", calls)
|
||||
}
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user