recovery: bound credential repair to current pod failures

This commit is contained in:
codex 2026-10-03 02:00:17 -05:00
parent 76876ef895
commit fc0ba95eb7
10 changed files with 291 additions and 4 deletions

View File

@ -67,3 +67,27 @@ Local testing check before installing:
```
Emergency installs can bypass the gate with `ANANKE_ENFORCE_QUALITY_GATE=0` - try to avoid this. You should be treating failures as an instructive opportunity to improve Ananke.
## Credential recovery and targeted updates
Registry credential recovery requires a warning observed within ten minutes for
an existing pod UID whose container or init container is still in ErrImagePull
or ImagePullBackOff. Completed and deleting pods are ignored. A helper restart
is reserved in the existing run history before execution and may occur at most
once per helper in 30 minutes, including failed rollouts. History must remain
writable; otherwise this repair fails closed. It is a single-coordinator control,
not a cross-host distributed lock. Do not run two coordinators against one cluster.
A targeted daemon fix can use the existing installer without rewriting host
configuration, UPS settings, or systemd/bootstrap units:
```bash
sudo ./scripts/install.sh --binary-only --skip-deps
```
The quality gate still runs. The previous binary is retained at
`/usr/local/lib/ananke/rollback/ananke.previous`; a failed service start restores
it. Verify service health and journals after installation. Automated self-updates
now default to keeping the current binary when their quality gate fails. The
service unit and updater script both set that default; existing hosts need those
two files updated to adopt it.

View File

@ -10,7 +10,7 @@ User=root
Group=root
Environment=ANANKE_UPDATE_LOG_FILE=/var/log/ananke/update.log
Environment=ANANKE_UPDATE_STATE_FILE=/var/lib/ananke/update-last.env
Environment=ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK=1
Environment=ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK=0
ExecStart=/usr/local/lib/ananke/ananke-self-update.sh
TimeoutStartSec=1800
StandardOutput=journal

View File

@ -21,6 +21,7 @@ type Orchestrator struct {
runOverride func(timeoutCtx context.Context, timeout time.Duration, name string, args ...string) (string, error)
runSensitiveOverride func(timeoutCtx context.Context, timeout time.Duration, name string, args ...string) (string, error)
sshInputOverride func(timeoutCtx context.Context, timeout time.Duration, node string, command string, input string) (string, error)
credentialRepairMu sync.Mutex
startupReportMu sync.Mutex
activeStartupReport *startupReport
}

View File

@ -8,8 +8,13 @@ import (
"sort"
"strings"
"time"
"scm.bstein.dev/bstein/ananke/internal/state"
)
const credentialEventMaxAge = 10 * time.Minute
const credentialRepairCooldown = 30 * time.Minute
type imagePullCredentialDeploymentList struct {
Items []struct {
Metadata struct {
@ -37,6 +42,23 @@ func (o *Orchestrator) imagePullCredentialBlockerReasons(ctx context.Context) (m
if err := json.Unmarshal([]byte(eventsOut), &events); err != nil {
return nil, fmt.Errorf("decode events for image-pull credential scan: %w", err)
}
if len(events.Items) == 0 {
return reasons, nil
}
podsOut, err := o.kubectl(ctx, 30*time.Second, "get", "pods", "-A", "-o", "json")
if err != nil {
return nil, fmt.Errorf("query current pods for image-pull credential scan: %w", err)
}
var pods podList
if err := json.Unmarshal([]byte(podsOut), &pods); err != nil {
return nil, fmt.Errorf("decode current pods for image-pull credential scan: %w", err)
}
blocked := map[string]string{}
for _, pod := range pods.Items {
if podHasCurrentImagePullFailure(pod) {
blocked[pod.Metadata.Namespace+"/"+pod.Metadata.Name] = pod.Metadata.UID
}
}
for _, event := range events.Items {
if !strings.EqualFold(strings.TrimSpace(event.Type), "Warning") {
continue
@ -57,6 +79,14 @@ func (o *Orchestrator) imagePullCredentialBlockerReasons(ctx context.Context) (m
if namespace == "" || name == "" {
continue
}
// Retained events outlive pods and successful pulls. Require current state,
// exact pod identity, and recent evidence before changing a sync helper.
uid := blocked[namespace+"/"+name]
observed := eventLastObservedAt(event)
if uid == "" || uid != event.InvolvedObject.UID || observed.IsZero() ||
time.Since(observed) > credentialEventMaxAge || time.Until(observed) > time.Minute {
continue
}
reasons[namespace+"/"+name] = "ImagePullCredentialBlocker:" + imagePullCredentialFailureClass(reason, message)
}
return reasons, nil
@ -114,6 +144,8 @@ func (o *Orchestrator) healImagePullCredentialSync(ctx context.Context) ([]strin
// Why: namespaces using Vault/CSI secret material have tiny sync deployments
// named or labeled vault-sync; restarting those nudges secretObject rotation.
func (o *Orchestrator) restartVaultSyncDeployments(ctx context.Context, namespace string) ([]string, error) {
o.credentialRepairMu.Lock()
defer o.credentialRepairMu.Unlock()
out, err := o.kubectl(ctx, 20*time.Second, "-n", namespace, "get", "deployment", "-o", "json")
if err != nil {
return nil, fmt.Errorf("query deployments in %s for image-pull credential repair: %w", namespace, err)
@ -124,11 +156,20 @@ func (o *Orchestrator) restartVaultSyncDeployments(ctx context.Context, namespac
}
repaired := []string{}
found := false
for _, deployment := range deployments.Items {
name := strings.TrimSpace(deployment.Metadata.Name)
if name == "" || !vaultSyncDeployment(name, deployment.Metadata.Labels) {
continue
}
found = true
allowed, err := o.reserveCredentialRepair(namespace + "/deployment/" + name)
if err != nil {
return repaired, err
}
if !allowed {
continue
}
if _, err := o.kubectl(ctx, 25*time.Second, "-n", namespace, "rollout", "restart", "deployment", name); err != nil {
return repaired, fmt.Errorf("restart %s/deployment/%s for image-pull credential repair: %w", namespace, name, err)
}
@ -137,7 +178,7 @@ func (o *Orchestrator) restartVaultSyncDeployments(ctx context.Context, namespac
}
repaired = append(repaired, namespace+"/deployment/"+name)
}
if len(repaired) == 0 {
if !found {
return nil, fmt.Errorf("image-pull credential blocker in namespace %s but no vault-sync deployment was found", namespace)
}
return repaired, nil
@ -235,3 +276,46 @@ func vaultSyncDeployment(name string, labels map[string]string) bool {
}
return false
}
// podHasCurrentImagePullFailure returns whether a live pod still needs a pull.
// Completed, deleting, and recovered pods cannot justify a credential repair.
// Signature: podHasCurrentImagePullFailure(pod podResource) bool.
// Why: historical warnings must not trigger repairs for recovered pods.
func podHasCurrentImagePullFailure(pod podResource) bool {
if pod.Metadata.DeletionTimestamp != nil || pod.Status.Phase == "Succeeded" || pod.Status.Phase == "Failed" {
return false
}
for _, statuses := range [][]podContainerStatus{pod.Status.InitContainerStatuses, pod.Status.ContainerStatuses} {
for _, status := range statuses {
if wait := status.State.Waiting; wait != nil && (wait.Reason == "ErrImagePull" || wait.Reason == "ImagePullBackOff") {
return true
}
}
}
return false
}
// reserveCredentialRepair records an attempt before mutation and returns false
// during the cooldown. The existing run history preserves it over daemon restarts.
// A failed or timed-out rollout also consumes the cooldown to prevent churn.
// Signature: (o *Orchestrator) reserveCredentialRepair(target string) (bool, error).
// Why: a cooldown must survive daemon restarts and failed rollouts.
func (o *Orchestrator) reserveCredentialRepair(target string) (bool, error) {
if o.store == nil {
return false, errors.New("credential repair requires persistent run history")
}
records, err := o.store.Load()
if err != nil {
return false, fmt.Errorf("read credential repair history: %w", err)
}
for _, record := range records {
if record.Action == "image-pull-credential-repair" && record.Reason == target && time.Since(record.StartedAt) < credentialRepairCooldown {
return false, nil
}
}
now := time.Now()
if err := o.store.Append(state.RunRecord{ID: now.UTC().Format(time.RFC3339Nano), Action: "image-pull-credential-repair", Reason: target, StartedAt: now, EndedAt: now}); err != nil {
return false, fmt.Errorf("record credential repair attempt: %w", err)
}
return true, nil
}

View File

@ -2,8 +2,11 @@ package cluster
import (
"context"
"encoding/json"
"fmt"
"strings"
"testing"
"time"
"scm.bstein.dev/bstein/ananke/internal/config"
)
@ -19,8 +22,10 @@ func TestImagePullCredentialBlockerReasonsClassifiesHarborAuth(t *testing.T) {
`{"metadata":{"namespace":"veles"},"involvedObject":{"kind":"Pod","name":"veles-frontend"},"type":"Warning","reason":"FailedToRetrieveImagePullSecret","message":"Unable to retrieve some image pull secrets (harbor-regcred); attempting to pull the image may not succeed."},` +
`{"involvedObject":{"kind":"Pod","namespace":"logging","name":"oauth2"},"type":"Warning","reason":"Failed","message":"Failed to pull image: lookup registry-1.docker.io: Try again"}` +
`]}`
events, pods := currentCredentialFixture(t, events)
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: events},
{match: matchContains("kubectl", "get", "pods", "-A"), out: pods},
})
reasons, err := orch.imagePullCredentialBlockerReasons(context.Background())
@ -51,8 +56,10 @@ func TestHealImagePullCredentialSyncRestartsVaultSyncDeployment(t *testing.T) {
`]}`
restarted := false
rolledOut := false
events, pods := currentCredentialFixture(t, events)
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: events},
{match: matchContains("kubectl", "get", "pods", "-A"), out: pods},
{match: matchContains("kubectl", "-n", "veles", "get", "deployment", "-o", "json"), out: deployments},
{
match: func(name string, args []string) bool {
@ -85,3 +92,139 @@ func TestHealImagePullCredentialSyncRestartsVaultSyncDeployment(t *testing.T) {
t.Fatalf("expected vault-sync deployment restart and rollout wait, restarted=%v rolledOut=%v", restarted, rolledOut)
}
}
// currentCredentialFixture gives legacy fixtures current pod identities and states.
// Signature: currentCredentialFixture(t *testing.T, raw string) (string, string).
// Why: event-only fixtures cannot demonstrate a currently blocked pull.
func currentCredentialFixture(t *testing.T, raw string) (string, string) {
t.Helper()
var events eventList
if err := json.Unmarshal([]byte(raw), &events); err != nil {
t.Fatal(err)
}
pods := podList{}
for i := range events.Items {
event := &events.Items[i]
event.LastTimestamp = time.Now()
event.InvolvedObject.UID = fmt.Sprintf("pod-%d", i)
pod := podResource{}
pod.Metadata.Namespace = event.InvolvedObject.Namespace
if pod.Metadata.Namespace == "" {
pod.Metadata.Namespace = event.Metadata.Namespace
}
pod.Metadata.Name = event.InvolvedObject.Name
pod.Metadata.UID = event.InvolvedObject.UID
pod.Status.Phase = "Pending"
pod.Status.ContainerStatuses = []podContainerStatus{{State: podContainerState{Waiting: &podContainerWaitingState{Reason: "ImagePullBackOff"}}}}
pods.Items = append(pods.Items, pod)
}
e, _ := json.Marshal(events)
p, _ := json.Marshal(pods)
return string(e), string(p)
}
// TestCredentialRecoveryRequiresCurrentEvidence rejects warnings for healed or replaced pods.
// Signature: TestCredentialRecoveryRequiresCurrentEvidence(t *testing.T).
// Why: only fresh evidence for the current pod may trigger mutation.
func TestCredentialRecoveryRequiresCurrentEvidence(t *testing.T) {
for _, scenario := range []string{"current", "init", "stale", "missing-time", "future", "missing-pod", "replaced", "missing-uid", "running", "deleting", "succeeded", "failed", "query-error", "bad-pods", "empty-events"} {
t.Run(scenario, func(t *testing.T) {
raw, podsRaw := currentCredentialFixture(t, `{"items":[{"type":"Warning","reason":"Failed","message":"unauthorized","involvedObject":{"kind":"Pod","namespace":"apps","name":"test"}}]}`)
var events eventList
var pods podList
_ = json.Unmarshal([]byte(raw), &events)
_ = json.Unmarshal([]byte(podsRaw), &pods)
switch scenario {
case "init":
pods.Items[0].Status.InitContainerStatuses = pods.Items[0].Status.ContainerStatuses
pods.Items[0].Status.ContainerStatuses = nil
case "stale":
events.Items[0].LastTimestamp = time.Now().Add(-11 * time.Minute)
case "missing-time":
events.Items[0].LastTimestamp = time.Time{}
case "future":
events.Items[0].LastTimestamp = time.Now().Add(time.Hour)
case "missing-pod":
pods.Items = nil
case "replaced":
pods.Items[0].Metadata.UID = "replacement"
case "missing-uid":
events.Items[0].InvolvedObject.UID = ""
case "running":
pods.Items[0].Status.ContainerStatuses = nil
pods.Items[0].Status.Phase = "Running"
case "deleting":
now := time.Now()
pods.Items[0].Metadata.DeletionTimestamp = &now
case "succeeded":
pods.Items[0].Status.Phase = "Succeeded"
case "failed":
pods.Items[0].Status.Phase = "Failed"
}
e, _ := json.Marshal(events)
p, _ := json.Marshal(pods)
raw, podsRaw = string(e), string(p)
var queryErr error
if scenario == "query-error" {
queryErr = fmt.Errorf("unavailable")
}
if scenario == "bad-pods" {
podsRaw = "{"
}
if scenario == "empty-events" {
raw = ""
}
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "events", "-A"), out: raw},
{match: matchContains("kubectl", "get", "pods", "-A"), out: podsRaw, err: queryErr},
})
reasons, err := orch.imagePullCredentialBlockerReasons(context.Background())
if scenario == "query-error" || scenario == "bad-pods" {
if err == nil {
t.Fatal("expected error")
}
return
}
if err != nil {
t.Fatal(err)
}
want := 0
if scenario == "current" || scenario == "init" {
want = 1
}
if len(reasons) != want {
t.Fatalf("got %v, want %d blockers", reasons, want)
}
})
}
}
// TestCredentialRepairCooldownSurvivesRestart prevents repeated recovery mutations.
// Signature: TestCredentialRepairCooldownSurvivesRestart(t *testing.T).
// Why: a failing helper must not be restarted every recovery cycle.
func TestCredentialRepairCooldownSurvivesRestart(t *testing.T) {
restarts := 0
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "deployment"), out: `{"items":[{"metadata":{"name":"app-vault-sync"}}]}`},
{match: func(name string, args []string) bool {
if matchContains("kubectl", "rollout", "restart")(name, args) {
restarts++
return true
}
return false
}},
{match: matchContains("kubectl", "rollout", "status"), err: fmt.Errorf("rollout timeout")},
})
if _, err := orch.restartVaultSyncDeployments(context.Background(), "apps"); err == nil {
t.Fatal("expected timeout")
}
replacement := &Orchestrator{cfg: orch.cfg, runner: orch.runner, store: orch.store, log: orch.log, runOverride: orch.runOverride}
repaired, err := replacement.restartVaultSyncDeployments(context.Background(), "apps")
if err != nil || len(repaired) != 0 || restarts != 1 {
t.Fatalf("repeated repair: %v %v %d", repaired, err, restarts)
}
replacement.store = nil
if _, err := replacement.reserveCredentialRepair("apps/other"); err == nil {
t.Fatal("missing state must fail closed")
}
}

View File

@ -231,7 +231,9 @@ func TestImagePullDNSAndCredentialBranches(t *testing.T) {
{"type":"Warning","reason":"FailedPull","metadata":{"namespace":"finance"},"involvedObject":{"kind":"Pod","name":"budget"},"message":"unauthorized: authentication required"},
{"type":"Warning","reason":"FailedToRetrieveImagePullSecret","involvedObject":{"kind":"Pod","namespace":"sso","name":"keycloak"},"message":"unable to retrieve some image pull secrets"}
]}`
eventJSON, podsJSON := currentCredentialFixture(t, eventJSON)
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "pods", "-A"), out: podsJSON},
{match: matchContains("kubectl", "get", "events", "-A"), out: eventJSON},
})
dnsReasons, err := orch.imagePullDNSBlockerReasons(context.Background())

View File

@ -107,12 +107,14 @@ func TestPostStartAutoHealRequestsReconcileAfterImagePullRepair(t *testing.T) {
ctx := context.Background()
eventsJSON := `{"items":[{"type":"Warning","reason":"FailedPull","message":"unauthorized: authentication required","involvedObject":{"kind":"Pod","namespace":"apps","name":"web"}}]}`
deploymentsJSON := `{"items":[{"metadata":{"name":"vault-sync","labels":{"app":"vault-sync"}}}]}`
eventsJSON, credentialPods := currentCredentialFixture(t, eventsJSON)
sawFluxSourceReconcile := false
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "-n", "longhorn-system", "get", "nodes.longhorn.io", "-o", "json"), out: `{"items":[]}`},
{match: matchContains("kubectl", "get", "nodes", "-o", "json"), out: `{"items":[]}`},
{match: matchContains("kubectl", "-n", "vault", "get", "pod", "vault-0"), out: "Pending"},
{match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: eventsJSON},
{match: matchContains("kubectl", "get", "pods", "-A", "-o", "json"), out: credentialPods},
{match: matchContains("kubectl", "-n", "apps", "get", "deployment", "-o", "json"), out: deploymentsJSON},
{match: matchContains("kubectl", "-n", "apps", "rollout", "restart", "deployment", "vault-sync"), out: ""},
{match: matchContains("kubectl", "-n", "apps", "rollout", "status", "deployment/vault-sync"), out: ""},

View File

@ -187,6 +187,7 @@ type eventResource struct {
CreationTimestamp time.Time `json:"creationTimestamp"`
} `json:"metadata"`
InvolvedObject struct {
UID string `json:"uid"`
Kind string `json:"kind"`
Namespace string `json:"namespace"`
Name string `json:"name"`
@ -242,6 +243,7 @@ type podList struct {
type podResource struct {
Metadata struct {
UID string `json:"uid"`
Namespace string `json:"namespace"`
Name string `json:"name"`
Labels map[string]string `json:"labels"`

View File

@ -13,7 +13,7 @@ HOST_SHORT="$(hostname -s 2>/dev/null || hostname)"
LOG_FILE="${ANANKE_UPDATE_LOG_FILE:-/var/log/ananke/update.log}"
STATE_FILE="${ANANKE_UPDATE_STATE_FILE:-/var/lib/ananke/update-last.env}"
LOCK_FILE="${ANANKE_UPDATE_LOCK_FILE:-/var/lock/ananke-update.lock}"
ALLOW_QUALITY_FALLBACK="${ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK:-1}"
ALLOW_QUALITY_FALLBACK="${ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK:-0}"
QUALITY_GATE_MODE="${ANANKE_ENFORCE_QUALITY_GATE:-1}"
mkdir -p "$(dirname "${LOG_FILE}")" "$(dirname "${STATE_FILE}")" "$(dirname "${LOCK_FILE}")"

View File

@ -14,6 +14,7 @@ SYSTEMD_DIR="/etc/systemd/system"
LIB_DIR="/usr/local/lib/ananke"
START_NOW=1
INSTALL_DEPS=1
BINARY_ONLY=0
ENABLE_BOOTSTRAP="${ANANKE_ENABLE_BOOTSTRAP:-auto}"
MANAGE_NUT="${ANANKE_MANAGE_NUT:-1}"
NUT_UPS_NAME="${ANANKE_NUT_UPS_NAME:-}"
@ -27,6 +28,10 @@ ENFORCE_QUALITY_GATE="${ANANKE_ENFORCE_QUALITY_GATE:-1}"
while [[ $# -gt 0 ]]; do
case "$1" in
--binary-only)
BINARY_ONLY=1
shift
;;
--no-start)
START_NOW=0
shift
@ -47,8 +52,14 @@ source "${REPO_DIR}/scripts/install-host-bootstrap.sh"
source "${REPO_DIR}/scripts/install-legacy-migration.sh"
source "${REPO_DIR}/scripts/install-artifacts.sh"
if [[ "${BINARY_ONLY}" == "1" && ! -f "${BIN_DIR}/ananke" ]]; then
echo "[install] binary-only requires an existing installation" >&2
exit 1
fi
ensure_dependencies
migrate_legacy_hecate_install
if [[ "${BINARY_ONLY}" != "1" ]]; then
migrate_legacy_hecate_install
fi
if [[ "${ENFORCE_QUALITY_GATE}" == "1" ]]; then
echo "[install] running quality gate"
@ -69,8 +80,26 @@ go build -o dist/ananke "${BUILD_TARGET}"
echo "[install] installing binary"
install -d -m 0755 "${BIN_DIR}"
if [[ "${BINARY_ONLY}" == "1" && -f "${BIN_DIR}/ananke" ]]; then
install -d -m 0700 "${LIB_DIR}/rollback"
install -m 0700 "${BIN_DIR}/ananke" "${LIB_DIR}/rollback/ananke.previous"
fi
install -m 0755 dist/ananke "${BIN_DIR}/ananke"
# Targeted recovery fixes must not overwrite host config, NUT, or bootstrap units.
if [[ "${BINARY_ONLY}" == "1" ]]; then
if [[ "${START_NOW}" == "1" ]]; then
if ! systemctl restart ananke.service || ! systemctl is-active --quiet ananke.service; then
echo "[install] service failed; restoring previous binary" >&2
install -m 0755 "${LIB_DIR}/rollback/ananke.previous" "${BIN_DIR}/ananke"
systemctl restart ananke.service
exit 1
fi
fi
echo "[install] binary-only update complete; host configuration unchanged"
exit 0
fi
echo "[install] installing config + state dirs"
install -d -m 0750 "${CONF_DIR}"
install -d -m 0750 "${STATE_DIR}"