recovery: bound credential repair to current pod failures

This commit is contained in:
codex 2026-10-03 02:00:17 -05:00
parent 76876ef895
commit fc0ba95eb7
10 changed files with 291 additions and 4 deletions

View File

@ -67,3 +67,27 @@ Local testing check before installing:
``` ```
Emergency installs can bypass the gate with `ANANKE_ENFORCE_QUALITY_GATE=0` - try to avoid this. You should be treating failures as an instructive opportunity to improve Ananke. Emergency installs can bypass the gate with `ANANKE_ENFORCE_QUALITY_GATE=0` - try to avoid this. You should be treating failures as an instructive opportunity to improve Ananke.
## Credential recovery and targeted updates
Registry credential recovery requires a warning observed within ten minutes for
an existing pod UID whose container or init container is still in ErrImagePull
or ImagePullBackOff. Completed and deleting pods are ignored. A helper restart
is reserved in the existing run history before execution and may occur at most
once per helper in 30 minutes, including failed rollouts. History must remain
writable; otherwise this repair fails closed. It is a single-coordinator control,
not a cross-host distributed lock. Do not run two coordinators against one cluster.
A targeted daemon fix can use the existing installer without rewriting host
configuration, UPS settings, or systemd/bootstrap units:
```bash
sudo ./scripts/install.sh --binary-only --skip-deps
```
The quality gate still runs. The previous binary is retained at
`/usr/local/lib/ananke/rollback/ananke.previous`; a failed service start restores
it. Verify service health and journals after installation. Automated self-updates
now default to keeping the current binary when their quality gate fails. The
service unit and updater script both set that default; existing hosts need those
two files updated to adopt it.

View File

@ -10,7 +10,7 @@ User=root
Group=root Group=root
Environment=ANANKE_UPDATE_LOG_FILE=/var/log/ananke/update.log Environment=ANANKE_UPDATE_LOG_FILE=/var/log/ananke/update.log
Environment=ANANKE_UPDATE_STATE_FILE=/var/lib/ananke/update-last.env Environment=ANANKE_UPDATE_STATE_FILE=/var/lib/ananke/update-last.env
Environment=ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK=1 Environment=ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK=0
ExecStart=/usr/local/lib/ananke/ananke-self-update.sh ExecStart=/usr/local/lib/ananke/ananke-self-update.sh
TimeoutStartSec=1800 TimeoutStartSec=1800
StandardOutput=journal StandardOutput=journal

View File

@ -21,6 +21,7 @@ type Orchestrator struct {
runOverride func(timeoutCtx context.Context, timeout time.Duration, name string, args ...string) (string, error) runOverride func(timeoutCtx context.Context, timeout time.Duration, name string, args ...string) (string, error)
runSensitiveOverride func(timeoutCtx context.Context, timeout time.Duration, name string, args ...string) (string, error) runSensitiveOverride func(timeoutCtx context.Context, timeout time.Duration, name string, args ...string) (string, error)
sshInputOverride func(timeoutCtx context.Context, timeout time.Duration, node string, command string, input string) (string, error) sshInputOverride func(timeoutCtx context.Context, timeout time.Duration, node string, command string, input string) (string, error)
credentialRepairMu sync.Mutex
startupReportMu sync.Mutex startupReportMu sync.Mutex
activeStartupReport *startupReport activeStartupReport *startupReport
} }

View File

@ -8,8 +8,13 @@ import (
"sort" "sort"
"strings" "strings"
"time" "time"
"scm.bstein.dev/bstein/ananke/internal/state"
) )
const credentialEventMaxAge = 10 * time.Minute
const credentialRepairCooldown = 30 * time.Minute
type imagePullCredentialDeploymentList struct { type imagePullCredentialDeploymentList struct {
Items []struct { Items []struct {
Metadata struct { Metadata struct {
@ -37,6 +42,23 @@ func (o *Orchestrator) imagePullCredentialBlockerReasons(ctx context.Context) (m
if err := json.Unmarshal([]byte(eventsOut), &events); err != nil { if err := json.Unmarshal([]byte(eventsOut), &events); err != nil {
return nil, fmt.Errorf("decode events for image-pull credential scan: %w", err) return nil, fmt.Errorf("decode events for image-pull credential scan: %w", err)
} }
if len(events.Items) == 0 {
return reasons, nil
}
podsOut, err := o.kubectl(ctx, 30*time.Second, "get", "pods", "-A", "-o", "json")
if err != nil {
return nil, fmt.Errorf("query current pods for image-pull credential scan: %w", err)
}
var pods podList
if err := json.Unmarshal([]byte(podsOut), &pods); err != nil {
return nil, fmt.Errorf("decode current pods for image-pull credential scan: %w", err)
}
blocked := map[string]string{}
for _, pod := range pods.Items {
if podHasCurrentImagePullFailure(pod) {
blocked[pod.Metadata.Namespace+"/"+pod.Metadata.Name] = pod.Metadata.UID
}
}
for _, event := range events.Items { for _, event := range events.Items {
if !strings.EqualFold(strings.TrimSpace(event.Type), "Warning") { if !strings.EqualFold(strings.TrimSpace(event.Type), "Warning") {
continue continue
@ -57,6 +79,14 @@ func (o *Orchestrator) imagePullCredentialBlockerReasons(ctx context.Context) (m
if namespace == "" || name == "" { if namespace == "" || name == "" {
continue continue
} }
// Retained events outlive pods and successful pulls. Require current state,
// exact pod identity, and recent evidence before changing a sync helper.
uid := blocked[namespace+"/"+name]
observed := eventLastObservedAt(event)
if uid == "" || uid != event.InvolvedObject.UID || observed.IsZero() ||
time.Since(observed) > credentialEventMaxAge || time.Until(observed) > time.Minute {
continue
}
reasons[namespace+"/"+name] = "ImagePullCredentialBlocker:" + imagePullCredentialFailureClass(reason, message) reasons[namespace+"/"+name] = "ImagePullCredentialBlocker:" + imagePullCredentialFailureClass(reason, message)
} }
return reasons, nil return reasons, nil
@ -114,6 +144,8 @@ func (o *Orchestrator) healImagePullCredentialSync(ctx context.Context) ([]strin
// Why: namespaces using Vault/CSI secret material have tiny sync deployments // Why: namespaces using Vault/CSI secret material have tiny sync deployments
// named or labeled vault-sync; restarting those nudges secretObject rotation. // named or labeled vault-sync; restarting those nudges secretObject rotation.
func (o *Orchestrator) restartVaultSyncDeployments(ctx context.Context, namespace string) ([]string, error) { func (o *Orchestrator) restartVaultSyncDeployments(ctx context.Context, namespace string) ([]string, error) {
o.credentialRepairMu.Lock()
defer o.credentialRepairMu.Unlock()
out, err := o.kubectl(ctx, 20*time.Second, "-n", namespace, "get", "deployment", "-o", "json") out, err := o.kubectl(ctx, 20*time.Second, "-n", namespace, "get", "deployment", "-o", "json")
if err != nil { if err != nil {
return nil, fmt.Errorf("query deployments in %s for image-pull credential repair: %w", namespace, err) return nil, fmt.Errorf("query deployments in %s for image-pull credential repair: %w", namespace, err)
@ -124,11 +156,20 @@ func (o *Orchestrator) restartVaultSyncDeployments(ctx context.Context, namespac
} }
repaired := []string{} repaired := []string{}
found := false
for _, deployment := range deployments.Items { for _, deployment := range deployments.Items {
name := strings.TrimSpace(deployment.Metadata.Name) name := strings.TrimSpace(deployment.Metadata.Name)
if name == "" || !vaultSyncDeployment(name, deployment.Metadata.Labels) { if name == "" || !vaultSyncDeployment(name, deployment.Metadata.Labels) {
continue continue
} }
found = true
allowed, err := o.reserveCredentialRepair(namespace + "/deployment/" + name)
if err != nil {
return repaired, err
}
if !allowed {
continue
}
if _, err := o.kubectl(ctx, 25*time.Second, "-n", namespace, "rollout", "restart", "deployment", name); err != nil { if _, err := o.kubectl(ctx, 25*time.Second, "-n", namespace, "rollout", "restart", "deployment", name); err != nil {
return repaired, fmt.Errorf("restart %s/deployment/%s for image-pull credential repair: %w", namespace, name, err) return repaired, fmt.Errorf("restart %s/deployment/%s for image-pull credential repair: %w", namespace, name, err)
} }
@ -137,7 +178,7 @@ func (o *Orchestrator) restartVaultSyncDeployments(ctx context.Context, namespac
} }
repaired = append(repaired, namespace+"/deployment/"+name) repaired = append(repaired, namespace+"/deployment/"+name)
} }
if len(repaired) == 0 { if !found {
return nil, fmt.Errorf("image-pull credential blocker in namespace %s but no vault-sync deployment was found", namespace) return nil, fmt.Errorf("image-pull credential blocker in namespace %s but no vault-sync deployment was found", namespace)
} }
return repaired, nil return repaired, nil
@ -235,3 +276,46 @@ func vaultSyncDeployment(name string, labels map[string]string) bool {
} }
return false return false
} }
// podHasCurrentImagePullFailure returns whether a live pod still needs a pull.
// Completed, deleting, and recovered pods cannot justify a credential repair.
// Signature: podHasCurrentImagePullFailure(pod podResource) bool.
// Why: historical warnings must not trigger repairs for recovered pods.
func podHasCurrentImagePullFailure(pod podResource) bool {
if pod.Metadata.DeletionTimestamp != nil || pod.Status.Phase == "Succeeded" || pod.Status.Phase == "Failed" {
return false
}
for _, statuses := range [][]podContainerStatus{pod.Status.InitContainerStatuses, pod.Status.ContainerStatuses} {
for _, status := range statuses {
if wait := status.State.Waiting; wait != nil && (wait.Reason == "ErrImagePull" || wait.Reason == "ImagePullBackOff") {
return true
}
}
}
return false
}
// reserveCredentialRepair records an attempt before mutation and returns false
// during the cooldown. The existing run history preserves it over daemon restarts.
// A failed or timed-out rollout also consumes the cooldown to prevent churn.
// Signature: (o *Orchestrator) reserveCredentialRepair(target string) (bool, error).
// Why: a cooldown must survive daemon restarts and failed rollouts.
func (o *Orchestrator) reserveCredentialRepair(target string) (bool, error) {
if o.store == nil {
return false, errors.New("credential repair requires persistent run history")
}
records, err := o.store.Load()
if err != nil {
return false, fmt.Errorf("read credential repair history: %w", err)
}
for _, record := range records {
if record.Action == "image-pull-credential-repair" && record.Reason == target && time.Since(record.StartedAt) < credentialRepairCooldown {
return false, nil
}
}
now := time.Now()
if err := o.store.Append(state.RunRecord{ID: now.UTC().Format(time.RFC3339Nano), Action: "image-pull-credential-repair", Reason: target, StartedAt: now, EndedAt: now}); err != nil {
return false, fmt.Errorf("record credential repair attempt: %w", err)
}
return true, nil
}

View File

@ -2,8 +2,11 @@ package cluster
import ( import (
"context" "context"
"encoding/json"
"fmt"
"strings" "strings"
"testing" "testing"
"time"
"scm.bstein.dev/bstein/ananke/internal/config" "scm.bstein.dev/bstein/ananke/internal/config"
) )
@ -19,8 +22,10 @@ func TestImagePullCredentialBlockerReasonsClassifiesHarborAuth(t *testing.T) {
`{"metadata":{"namespace":"veles"},"involvedObject":{"kind":"Pod","name":"veles-frontend"},"type":"Warning","reason":"FailedToRetrieveImagePullSecret","message":"Unable to retrieve some image pull secrets (harbor-regcred); attempting to pull the image may not succeed."},` + `{"metadata":{"namespace":"veles"},"involvedObject":{"kind":"Pod","name":"veles-frontend"},"type":"Warning","reason":"FailedToRetrieveImagePullSecret","message":"Unable to retrieve some image pull secrets (harbor-regcred); attempting to pull the image may not succeed."},` +
`{"involvedObject":{"kind":"Pod","namespace":"logging","name":"oauth2"},"type":"Warning","reason":"Failed","message":"Failed to pull image: lookup registry-1.docker.io: Try again"}` + `{"involvedObject":{"kind":"Pod","namespace":"logging","name":"oauth2"},"type":"Warning","reason":"Failed","message":"Failed to pull image: lookup registry-1.docker.io: Try again"}` +
`]}` `]}`
events, pods := currentCredentialFixture(t, events)
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{ orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: events}, {match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: events},
{match: matchContains("kubectl", "get", "pods", "-A"), out: pods},
}) })
reasons, err := orch.imagePullCredentialBlockerReasons(context.Background()) reasons, err := orch.imagePullCredentialBlockerReasons(context.Background())
@ -51,8 +56,10 @@ func TestHealImagePullCredentialSyncRestartsVaultSyncDeployment(t *testing.T) {
`]}` `]}`
restarted := false restarted := false
rolledOut := false rolledOut := false
events, pods := currentCredentialFixture(t, events)
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{ orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: events}, {match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: events},
{match: matchContains("kubectl", "get", "pods", "-A"), out: pods},
{match: matchContains("kubectl", "-n", "veles", "get", "deployment", "-o", "json"), out: deployments}, {match: matchContains("kubectl", "-n", "veles", "get", "deployment", "-o", "json"), out: deployments},
{ {
match: func(name string, args []string) bool { match: func(name string, args []string) bool {
@ -85,3 +92,139 @@ func TestHealImagePullCredentialSyncRestartsVaultSyncDeployment(t *testing.T) {
t.Fatalf("expected vault-sync deployment restart and rollout wait, restarted=%v rolledOut=%v", restarted, rolledOut) t.Fatalf("expected vault-sync deployment restart and rollout wait, restarted=%v rolledOut=%v", restarted, rolledOut)
} }
} }
// currentCredentialFixture gives legacy fixtures current pod identities and states.
// Signature: currentCredentialFixture(t *testing.T, raw string) (string, string).
// Why: event-only fixtures cannot demonstrate a currently blocked pull.
func currentCredentialFixture(t *testing.T, raw string) (string, string) {
t.Helper()
var events eventList
if err := json.Unmarshal([]byte(raw), &events); err != nil {
t.Fatal(err)
}
pods := podList{}
for i := range events.Items {
event := &events.Items[i]
event.LastTimestamp = time.Now()
event.InvolvedObject.UID = fmt.Sprintf("pod-%d", i)
pod := podResource{}
pod.Metadata.Namespace = event.InvolvedObject.Namespace
if pod.Metadata.Namespace == "" {
pod.Metadata.Namespace = event.Metadata.Namespace
}
pod.Metadata.Name = event.InvolvedObject.Name
pod.Metadata.UID = event.InvolvedObject.UID
pod.Status.Phase = "Pending"
pod.Status.ContainerStatuses = []podContainerStatus{{State: podContainerState{Waiting: &podContainerWaitingState{Reason: "ImagePullBackOff"}}}}
pods.Items = append(pods.Items, pod)
}
e, _ := json.Marshal(events)
p, _ := json.Marshal(pods)
return string(e), string(p)
}
// TestCredentialRecoveryRequiresCurrentEvidence rejects warnings for healed or replaced pods.
// Signature: TestCredentialRecoveryRequiresCurrentEvidence(t *testing.T).
// Why: only fresh evidence for the current pod may trigger mutation.
func TestCredentialRecoveryRequiresCurrentEvidence(t *testing.T) {
for _, scenario := range []string{"current", "init", "stale", "missing-time", "future", "missing-pod", "replaced", "missing-uid", "running", "deleting", "succeeded", "failed", "query-error", "bad-pods", "empty-events"} {
t.Run(scenario, func(t *testing.T) {
raw, podsRaw := currentCredentialFixture(t, `{"items":[{"type":"Warning","reason":"Failed","message":"unauthorized","involvedObject":{"kind":"Pod","namespace":"apps","name":"test"}}]}`)
var events eventList
var pods podList
_ = json.Unmarshal([]byte(raw), &events)
_ = json.Unmarshal([]byte(podsRaw), &pods)
switch scenario {
case "init":
pods.Items[0].Status.InitContainerStatuses = pods.Items[0].Status.ContainerStatuses
pods.Items[0].Status.ContainerStatuses = nil
case "stale":
events.Items[0].LastTimestamp = time.Now().Add(-11 * time.Minute)
case "missing-time":
events.Items[0].LastTimestamp = time.Time{}
case "future":
events.Items[0].LastTimestamp = time.Now().Add(time.Hour)
case "missing-pod":
pods.Items = nil
case "replaced":
pods.Items[0].Metadata.UID = "replacement"
case "missing-uid":
events.Items[0].InvolvedObject.UID = ""
case "running":
pods.Items[0].Status.ContainerStatuses = nil
pods.Items[0].Status.Phase = "Running"
case "deleting":
now := time.Now()
pods.Items[0].Metadata.DeletionTimestamp = &now
case "succeeded":
pods.Items[0].Status.Phase = "Succeeded"
case "failed":
pods.Items[0].Status.Phase = "Failed"
}
e, _ := json.Marshal(events)
p, _ := json.Marshal(pods)
raw, podsRaw = string(e), string(p)
var queryErr error
if scenario == "query-error" {
queryErr = fmt.Errorf("unavailable")
}
if scenario == "bad-pods" {
podsRaw = "{"
}
if scenario == "empty-events" {
raw = ""
}
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "events", "-A"), out: raw},
{match: matchContains("kubectl", "get", "pods", "-A"), out: podsRaw, err: queryErr},
})
reasons, err := orch.imagePullCredentialBlockerReasons(context.Background())
if scenario == "query-error" || scenario == "bad-pods" {
if err == nil {
t.Fatal("expected error")
}
return
}
if err != nil {
t.Fatal(err)
}
want := 0
if scenario == "current" || scenario == "init" {
want = 1
}
if len(reasons) != want {
t.Fatalf("got %v, want %d blockers", reasons, want)
}
})
}
}
// TestCredentialRepairCooldownSurvivesRestart prevents repeated recovery mutations.
// Signature: TestCredentialRepairCooldownSurvivesRestart(t *testing.T).
// Why: a failing helper must not be restarted every recovery cycle.
func TestCredentialRepairCooldownSurvivesRestart(t *testing.T) {
restarts := 0
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "deployment"), out: `{"items":[{"metadata":{"name":"app-vault-sync"}}]}`},
{match: func(name string, args []string) bool {
if matchContains("kubectl", "rollout", "restart")(name, args) {
restarts++
return true
}
return false
}},
{match: matchContains("kubectl", "rollout", "status"), err: fmt.Errorf("rollout timeout")},
})
if _, err := orch.restartVaultSyncDeployments(context.Background(), "apps"); err == nil {
t.Fatal("expected timeout")
}
replacement := &Orchestrator{cfg: orch.cfg, runner: orch.runner, store: orch.store, log: orch.log, runOverride: orch.runOverride}
repaired, err := replacement.restartVaultSyncDeployments(context.Background(), "apps")
if err != nil || len(repaired) != 0 || restarts != 1 {
t.Fatalf("repeated repair: %v %v %d", repaired, err, restarts)
}
replacement.store = nil
if _, err := replacement.reserveCredentialRepair("apps/other"); err == nil {
t.Fatal("missing state must fail closed")
}
}

View File

@ -231,7 +231,9 @@ func TestImagePullDNSAndCredentialBranches(t *testing.T) {
{"type":"Warning","reason":"FailedPull","metadata":{"namespace":"finance"},"involvedObject":{"kind":"Pod","name":"budget"},"message":"unauthorized: authentication required"}, {"type":"Warning","reason":"FailedPull","metadata":{"namespace":"finance"},"involvedObject":{"kind":"Pod","name":"budget"},"message":"unauthorized: authentication required"},
{"type":"Warning","reason":"FailedToRetrieveImagePullSecret","involvedObject":{"kind":"Pod","namespace":"sso","name":"keycloak"},"message":"unable to retrieve some image pull secrets"} {"type":"Warning","reason":"FailedToRetrieveImagePullSecret","involvedObject":{"kind":"Pod","namespace":"sso","name":"keycloak"},"message":"unable to retrieve some image pull secrets"}
]}` ]}`
eventJSON, podsJSON := currentCredentialFixture(t, eventJSON)
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{ orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "pods", "-A"), out: podsJSON},
{match: matchContains("kubectl", "get", "events", "-A"), out: eventJSON}, {match: matchContains("kubectl", "get", "events", "-A"), out: eventJSON},
}) })
dnsReasons, err := orch.imagePullDNSBlockerReasons(context.Background()) dnsReasons, err := orch.imagePullDNSBlockerReasons(context.Background())

View File

@ -107,12 +107,14 @@ func TestPostStartAutoHealRequestsReconcileAfterImagePullRepair(t *testing.T) {
ctx := context.Background() ctx := context.Background()
eventsJSON := `{"items":[{"type":"Warning","reason":"FailedPull","message":"unauthorized: authentication required","involvedObject":{"kind":"Pod","namespace":"apps","name":"web"}}]}` eventsJSON := `{"items":[{"type":"Warning","reason":"FailedPull","message":"unauthorized: authentication required","involvedObject":{"kind":"Pod","namespace":"apps","name":"web"}}]}`
deploymentsJSON := `{"items":[{"metadata":{"name":"vault-sync","labels":{"app":"vault-sync"}}}]}` deploymentsJSON := `{"items":[{"metadata":{"name":"vault-sync","labels":{"app":"vault-sync"}}}]}`
eventsJSON, credentialPods := currentCredentialFixture(t, eventsJSON)
sawFluxSourceReconcile := false sawFluxSourceReconcile := false
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{ orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "-n", "longhorn-system", "get", "nodes.longhorn.io", "-o", "json"), out: `{"items":[]}`}, {match: matchContains("kubectl", "-n", "longhorn-system", "get", "nodes.longhorn.io", "-o", "json"), out: `{"items":[]}`},
{match: matchContains("kubectl", "get", "nodes", "-o", "json"), out: `{"items":[]}`}, {match: matchContains("kubectl", "get", "nodes", "-o", "json"), out: `{"items":[]}`},
{match: matchContains("kubectl", "-n", "vault", "get", "pod", "vault-0"), out: "Pending"}, {match: matchContains("kubectl", "-n", "vault", "get", "pod", "vault-0"), out: "Pending"},
{match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: eventsJSON}, {match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: eventsJSON},
{match: matchContains("kubectl", "get", "pods", "-A", "-o", "json"), out: credentialPods},
{match: matchContains("kubectl", "-n", "apps", "get", "deployment", "-o", "json"), out: deploymentsJSON}, {match: matchContains("kubectl", "-n", "apps", "get", "deployment", "-o", "json"), out: deploymentsJSON},
{match: matchContains("kubectl", "-n", "apps", "rollout", "restart", "deployment", "vault-sync"), out: ""}, {match: matchContains("kubectl", "-n", "apps", "rollout", "restart", "deployment", "vault-sync"), out: ""},
{match: matchContains("kubectl", "-n", "apps", "rollout", "status", "deployment/vault-sync"), out: ""}, {match: matchContains("kubectl", "-n", "apps", "rollout", "status", "deployment/vault-sync"), out: ""},

View File

@ -187,6 +187,7 @@ type eventResource struct {
CreationTimestamp time.Time `json:"creationTimestamp"` CreationTimestamp time.Time `json:"creationTimestamp"`
} `json:"metadata"` } `json:"metadata"`
InvolvedObject struct { InvolvedObject struct {
UID string `json:"uid"`
Kind string `json:"kind"` Kind string `json:"kind"`
Namespace string `json:"namespace"` Namespace string `json:"namespace"`
Name string `json:"name"` Name string `json:"name"`
@ -242,6 +243,7 @@ type podList struct {
type podResource struct { type podResource struct {
Metadata struct { Metadata struct {
UID string `json:"uid"`
Namespace string `json:"namespace"` Namespace string `json:"namespace"`
Name string `json:"name"` Name string `json:"name"`
Labels map[string]string `json:"labels"` Labels map[string]string `json:"labels"`

View File

@ -13,7 +13,7 @@ HOST_SHORT="$(hostname -s 2>/dev/null || hostname)"
LOG_FILE="${ANANKE_UPDATE_LOG_FILE:-/var/log/ananke/update.log}" LOG_FILE="${ANANKE_UPDATE_LOG_FILE:-/var/log/ananke/update.log}"
STATE_FILE="${ANANKE_UPDATE_STATE_FILE:-/var/lib/ananke/update-last.env}" STATE_FILE="${ANANKE_UPDATE_STATE_FILE:-/var/lib/ananke/update-last.env}"
LOCK_FILE="${ANANKE_UPDATE_LOCK_FILE:-/var/lock/ananke-update.lock}" LOCK_FILE="${ANANKE_UPDATE_LOCK_FILE:-/var/lock/ananke-update.lock}"
ALLOW_QUALITY_FALLBACK="${ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK:-1}" ALLOW_QUALITY_FALLBACK="${ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK:-0}"
QUALITY_GATE_MODE="${ANANKE_ENFORCE_QUALITY_GATE:-1}" QUALITY_GATE_MODE="${ANANKE_ENFORCE_QUALITY_GATE:-1}"
mkdir -p "$(dirname "${LOG_FILE}")" "$(dirname "${STATE_FILE}")" "$(dirname "${LOCK_FILE}")" mkdir -p "$(dirname "${LOG_FILE}")" "$(dirname "${STATE_FILE}")" "$(dirname "${LOCK_FILE}")"

View File

@ -14,6 +14,7 @@ SYSTEMD_DIR="/etc/systemd/system"
LIB_DIR="/usr/local/lib/ananke" LIB_DIR="/usr/local/lib/ananke"
START_NOW=1 START_NOW=1
INSTALL_DEPS=1 INSTALL_DEPS=1
BINARY_ONLY=0
ENABLE_BOOTSTRAP="${ANANKE_ENABLE_BOOTSTRAP:-auto}" ENABLE_BOOTSTRAP="${ANANKE_ENABLE_BOOTSTRAP:-auto}"
MANAGE_NUT="${ANANKE_MANAGE_NUT:-1}" MANAGE_NUT="${ANANKE_MANAGE_NUT:-1}"
NUT_UPS_NAME="${ANANKE_NUT_UPS_NAME:-}" NUT_UPS_NAME="${ANANKE_NUT_UPS_NAME:-}"
@ -27,6 +28,10 @@ ENFORCE_QUALITY_GATE="${ANANKE_ENFORCE_QUALITY_GATE:-1}"
while [[ $# -gt 0 ]]; do while [[ $# -gt 0 ]]; do
case "$1" in case "$1" in
--binary-only)
BINARY_ONLY=1
shift
;;
--no-start) --no-start)
START_NOW=0 START_NOW=0
shift shift
@ -47,8 +52,14 @@ source "${REPO_DIR}/scripts/install-host-bootstrap.sh"
source "${REPO_DIR}/scripts/install-legacy-migration.sh" source "${REPO_DIR}/scripts/install-legacy-migration.sh"
source "${REPO_DIR}/scripts/install-artifacts.sh" source "${REPO_DIR}/scripts/install-artifacts.sh"
if [[ "${BINARY_ONLY}" == "1" && ! -f "${BIN_DIR}/ananke" ]]; then
echo "[install] binary-only requires an existing installation" >&2
exit 1
fi
ensure_dependencies ensure_dependencies
migrate_legacy_hecate_install if [[ "${BINARY_ONLY}" != "1" ]]; then
migrate_legacy_hecate_install
fi
if [[ "${ENFORCE_QUALITY_GATE}" == "1" ]]; then if [[ "${ENFORCE_QUALITY_GATE}" == "1" ]]; then
echo "[install] running quality gate" echo "[install] running quality gate"
@ -69,8 +80,26 @@ go build -o dist/ananke "${BUILD_TARGET}"
echo "[install] installing binary" echo "[install] installing binary"
install -d -m 0755 "${BIN_DIR}" install -d -m 0755 "${BIN_DIR}"
if [[ "${BINARY_ONLY}" == "1" && -f "${BIN_DIR}/ananke" ]]; then
install -d -m 0700 "${LIB_DIR}/rollback"
install -m 0700 "${BIN_DIR}/ananke" "${LIB_DIR}/rollback/ananke.previous"
fi
install -m 0755 dist/ananke "${BIN_DIR}/ananke" install -m 0755 dist/ananke "${BIN_DIR}/ananke"
# Targeted recovery fixes must not overwrite host config, NUT, or bootstrap units.
if [[ "${BINARY_ONLY}" == "1" ]]; then
if [[ "${START_NOW}" == "1" ]]; then
if ! systemctl restart ananke.service || ! systemctl is-active --quiet ananke.service; then
echo "[install] service failed; restoring previous binary" >&2
install -m 0755 "${LIB_DIR}/rollback/ananke.previous" "${BIN_DIR}/ananke"
systemctl restart ananke.service
exit 1
fi
fi
echo "[install] binary-only update complete; host configuration unchanged"
exit 0
fi
echo "[install] installing config + state dirs" echo "[install] installing config + state dirs"
install -d -m 0750 "${CONF_DIR}" install -d -m 0750 "${CONF_DIR}"
install -d -m 0750 "${STATE_DIR}" install -d -m 0750 "${STATE_DIR}"