recovery: give shared repairs one coordinator
This commit is contained in:
parent
fc0ba95eb7
commit
ec7c2ee053
@ -75,8 +75,9 @@ an existing pod UID whose container or init container is still in ErrImagePull
|
|||||||
or ImagePullBackOff. Completed and deleting pods are ignored. A helper restart
|
or ImagePullBackOff. Completed and deleting pods are ignored. A helper restart
|
||||||
is reserved in the existing run history before execution and may occur at most
|
is reserved in the existing run history before execution and may occur at most
|
||||||
once per helper in 30 minutes, including failed rollouts. History must remain
|
once per helper in 30 minutes, including failed rollouts. History must remain
|
||||||
writable; otherwise this repair fails closed. It is a single-coordinator control,
|
writable; otherwise this repair fails closed. Only the coordinator daemon runs periodic shared-cluster repairs; the peer
|
||||||
not a cross-host distributed lock. Do not run two coordinators against one cluster.
|
continues UPS monitoring and shutdown forwarding. Do not configure two hosts as
|
||||||
|
coordinators for one cluster.
|
||||||
|
|
||||||
A targeted daemon fix can use the existing installer without rewriting host
|
A targeted daemon fix can use the existing installer without rewriting host
|
||||||
configuration, UPS settings, or systemd/bootstrap units:
|
configuration, UPS settings, or systemd/bootstrap units:
|
||||||
|
|||||||
@ -217,6 +217,11 @@ func (d *Daemon) Run(ctx context.Context) error {
|
|||||||
// like a later Vault reseal or stale dead-node deletions without waiting for a
|
// like a later Vault reseal or stale dead-node deletions without waiting for a
|
||||||
// fresh bootstrap run.
|
// fresh bootstrap run.
|
||||||
func (d *Daemon) maybeRunPostStartAutoHeal(ctx context.Context, lastRun *time.Time, anyOnBattery bool) {
|
func (d *Daemon) maybeRunPostStartAutoHeal(ctx context.Context, lastRun *time.Time, anyOnBattery bool) {
|
||||||
|
// Shared cluster repairs have one owner; peers still monitor their UPS and
|
||||||
|
// forward shutdown intent through the existing coordination path.
|
||||||
|
if d.cfg.Coordination.Role == "peer" {
|
||||||
|
return
|
||||||
|
}
|
||||||
interval := time.Duration(d.cfg.Startup.PostStartAutoHealSeconds) * time.Second
|
interval := time.Duration(d.cfg.Startup.PostStartAutoHealSeconds) * time.Second
|
||||||
if interval <= 0 || anyOnBattery {
|
if interval <= 0 || anyOnBattery {
|
||||||
return
|
return
|
||||||
|
|||||||
@ -26,8 +26,14 @@ func TestDaemonMaybeRunPostStartAutoHeal(t *testing.T) {
|
|||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
d.cfg.Coordination.Role = "peer"
|
||||||
var last time.Time
|
var last time.Time
|
||||||
d.maybeRunPostStartAutoHeal(context.Background(), &last, false)
|
d.maybeRunPostStartAutoHeal(context.Background(), &last, false)
|
||||||
|
if calls != 0 || !last.IsZero() {
|
||||||
|
t.Fatal("peer must not mutate shared cluster recovery state")
|
||||||
|
}
|
||||||
|
d.cfg.Coordination.Role = "coordinator"
|
||||||
|
d.maybeRunPostStartAutoHeal(context.Background(), &last, false)
|
||||||
if calls != 1 {
|
if calls != 1 {
|
||||||
t.Fatalf("expected first auto-heal invocation, got %d", calls)
|
t.Fatalf("expected first auto-heal invocation, got %d", calls)
|
||||||
}
|
}
|
||||||
|
|||||||
Loading…
x
Reference in New Issue
Block a user