recovery: give shared repairs one coordinator

This commit is contained in:
codex 2026-10-03 02:02:04 -05:00
parent fc0ba95eb7
commit ec7c2ee053
3 changed files with 14 additions and 2 deletions

View File

@ -75,8 +75,9 @@ an existing pod UID whose container or init container is still in ErrImagePull
or ImagePullBackOff. Completed and deleting pods are ignored. A helper restart or ImagePullBackOff. Completed and deleting pods are ignored. A helper restart
is reserved in the existing run history before execution and may occur at most is reserved in the existing run history before execution and may occur at most
once per helper in 30 minutes, including failed rollouts. History must remain once per helper in 30 minutes, including failed rollouts. History must remain
writable; otherwise this repair fails closed. It is a single-coordinator control, writable; otherwise this repair fails closed. Only the coordinator daemon runs periodic shared-cluster repairs; the peer
not a cross-host distributed lock. Do not run two coordinators against one cluster. continues UPS monitoring and shutdown forwarding. Do not configure two hosts as
coordinators for one cluster.
A targeted daemon fix can use the existing installer without rewriting host A targeted daemon fix can use the existing installer without rewriting host
configuration, UPS settings, or systemd/bootstrap units: configuration, UPS settings, or systemd/bootstrap units:

View File

@ -217,6 +217,11 @@ func (d *Daemon) Run(ctx context.Context) error {
// like a later Vault reseal or stale dead-node deletions without waiting for a // like a later Vault reseal or stale dead-node deletions without waiting for a
// fresh bootstrap run. // fresh bootstrap run.
func (d *Daemon) maybeRunPostStartAutoHeal(ctx context.Context, lastRun *time.Time, anyOnBattery bool) { func (d *Daemon) maybeRunPostStartAutoHeal(ctx context.Context, lastRun *time.Time, anyOnBattery bool) {
// Shared cluster repairs have one owner; peers still monitor their UPS and
// forward shutdown intent through the existing coordination path.
if d.cfg.Coordination.Role == "peer" {
return
}
interval := time.Duration(d.cfg.Startup.PostStartAutoHealSeconds) * time.Second interval := time.Duration(d.cfg.Startup.PostStartAutoHealSeconds) * time.Second
if interval <= 0 || anyOnBattery { if interval <= 0 || anyOnBattery {
return return

View File

@ -26,8 +26,14 @@ func TestDaemonMaybeRunPostStartAutoHeal(t *testing.T) {
}, },
} }
d.cfg.Coordination.Role = "peer"
var last time.Time var last time.Time
d.maybeRunPostStartAutoHeal(context.Background(), &last, false) d.maybeRunPostStartAutoHeal(context.Background(), &last, false)
if calls != 0 || !last.IsZero() {
t.Fatal("peer must not mutate shared cluster recovery state")
}
d.cfg.Coordination.Role = "coordinator"
d.maybeRunPostStartAutoHeal(context.Background(), &last, false)
if calls != 1 { if calls != 1 {
t.Fatalf("expected first auto-heal invocation, got %d", calls) t.Fatalf("expected first auto-heal invocation, got %d", calls)
} }