Compare commits

..

10 Commits

20 changed files with 450 additions and 19 deletions

28
Jenkinsfile vendored
View File

@ -1,11 +1,16 @@
pipeline {
agent {
kubernetes {
// Keep build I/O off the node's runtime USB drive.
workspaceVolume dynamicPVC(accessModes: 'ReadWriteOnce', requestsSize: '20Gi', storageClassName: 'ci-scratch')
defaultContainer 'go-tester'
yaml """
apiVersion: v1
kind: Pod
spec:
securityContext:
fsGroup: 1000
fsGroupChangePolicy: OnRootMismatch
nodeSelector:
hardware: rpi5
kubernetes.io/arch: arm64
@ -15,11 +20,9 @@ spec:
requiredDuringSchedulingIgnoredDuringExecution:
nodeSelectorTerms:
- matchExpressions:
- key: kubernetes.io/hostname
operator: NotIn
values:
- titan-06
- titan-11
- {key: hardware, operator: In, values: [rpi5, rpi4]}
- {key: node-role.kubernetes.io/worker, operator: In, values: ["true"]}
- {key: kubernetes.io/hostname, operator: NotIn, values: [titan-04, titan-05, titan-06, titan-08, titan-11, titan-12, titan-13, titan-14, titan-15, titan-17, titan-18, titan-19]}
preferredDuringSchedulingIgnoredDuringExecution:
- weight: 100
preference:
@ -60,14 +63,18 @@ spec:
volumeMounts:
- name: workspace-volume
mountPath: /home/jenkins/agent
volumes:
- name: workspace-volume
emptyDir: {}
"""
}
}
environment {
PIP_CACHE_DIR = '/home/jenkins/agent/.cache/pip'
NPM_CONFIG_CACHE = '/home/jenkins/agent/.cache/npm'
SONAR_USER_HOME = '/home/jenkins/agent/.cache/sonar'
TMPDIR = '/home/jenkins/agent/.cache/tmp'
GOCACHE = '/home/jenkins/agent/.cache/go-build'
GOMODCACHE = '/home/jenkins/agent/.cache/go-mod'
GOTMPDIR = '/home/jenkins/agent/.cache/go-tmp'
SUITE_NAME = 'ananke'
PUSHGATEWAY_URL = 'http://platform-quality-gateway.monitoring.svc.cluster.local:9091'
SONARQUBE_HOST_URL = 'http://sonarqube.quality.svc.cluster.local:9000'
@ -90,6 +97,11 @@ spec:
}
stages {
stage('Prepare build scratch') {
steps {
sh 'mkdir -p /home/jenkins/agent/.cache/tmp /home/jenkins/agent/.cache/go-tmp'
}
}
stage('Checkout') {
steps {
checkout scm

View File

@ -67,3 +67,28 @@ Local testing check before installing:
```
Emergency installs can bypass the gate with `ANANKE_ENFORCE_QUALITY_GATE=0` - try to avoid this. You should be treating failures as an instructive opportunity to improve Ananke.
## Credential recovery and targeted updates
Registry credential recovery requires a warning observed within ten minutes for
an existing pod UID whose container or init container is still in ErrImagePull
or ImagePullBackOff. Completed and deleting pods are ignored. A helper restart
is reserved in the existing run history before execution and may occur at most
once per helper in 30 minutes, including failed rollouts. History must remain
writable; otherwise this repair fails closed. Only the coordinator daemon runs periodic shared-cluster repairs; the peer
continues UPS monitoring and shutdown forwarding. Do not configure two hosts as
coordinators for one cluster.
A targeted daemon fix can use the existing installer without rewriting host
configuration, UPS settings, or systemd/bootstrap units:
```bash
sudo ./scripts/install.sh --binary-only --skip-deps
```
The quality gate still runs. The previous binary is retained at
`/usr/local/lib/ananke/rollback/ananke.previous`; a failed service start restores
it. Verify service health and journals after installation. Automated self-updates
now default to keeping the current binary when their quality gate fails. The
service unit and updater script both set that default; existing hosts need those
two files updated to adopt it.

View File

@ -29,7 +29,7 @@ ssh_node_hosts:
titan-22: 192.168.22.22
titan-24: 192.168.22.26
ssh_node_users:
titan-24: atlas
titan-24: tethys
ssh_managed_nodes:
- titan-db
- titan-0a
@ -110,6 +110,10 @@ excluded_namespaces:
- postgres
- maintenance
startup:
# Existing Vault CSI synchronization supplies password-backed host actions.
host_sudo_secret_namespace: maintenance
host_sudo_secret_name_template: "ananke-sudo-{node}"
host_sudo_secret_password_key: password
api_wait_seconds: 1200
api_poll_seconds: 2
shutdown_cooldown_seconds: 45

View File

@ -29,7 +29,7 @@ ssh_node_hosts:
titan-22: 192.168.22.22
titan-24: 192.168.22.26
ssh_node_users:
titan-24: atlas
titan-24: tethys
ssh_managed_nodes:
- titan-db
- titan-0a
@ -110,6 +110,10 @@ excluded_namespaces:
- postgres
- maintenance
startup:
# Existing Vault CSI synchronization supplies password-backed host actions.
host_sudo_secret_namespace: maintenance
host_sudo_secret_name_template: "ananke-sudo-{node}"
host_sudo_secret_password_key: password
api_wait_seconds: 1200
api_poll_seconds: 2
shutdown_cooldown_seconds: 45

View File

@ -10,7 +10,7 @@ User=root
Group=root
Environment=ANANKE_UPDATE_LOG_FILE=/var/log/ananke/update.log
Environment=ANANKE_UPDATE_STATE_FILE=/var/lib/ananke/update-last.env
Environment=ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK=1
Environment=ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK=0
ExecStart=/usr/local/lib/ananke/ananke-self-update.sh
TimeoutStartSec=1800
StandardOutput=journal

View File

@ -21,6 +21,7 @@ type Orchestrator struct {
runOverride func(timeoutCtx context.Context, timeout time.Duration, name string, args ...string) (string, error)
runSensitiveOverride func(timeoutCtx context.Context, timeout time.Duration, name string, args ...string) (string, error)
sshInputOverride func(timeoutCtx context.Context, timeout time.Duration, node string, command string, input string) (string, error)
credentialRepairMu sync.Mutex
startupReportMu sync.Mutex
activeStartupReport *startupReport
}

View File

@ -8,8 +8,13 @@ import (
"sort"
"strings"
"time"
"scm.bstein.dev/bstein/ananke/internal/state"
)
const credentialEventMaxAge = 10 * time.Minute
const credentialRepairCooldown = 30 * time.Minute
type imagePullCredentialDeploymentList struct {
Items []struct {
Metadata struct {
@ -37,6 +42,23 @@ func (o *Orchestrator) imagePullCredentialBlockerReasons(ctx context.Context) (m
if err := json.Unmarshal([]byte(eventsOut), &events); err != nil {
return nil, fmt.Errorf("decode events for image-pull credential scan: %w", err)
}
if len(events.Items) == 0 {
return reasons, nil
}
podsOut, err := o.kubectl(ctx, 30*time.Second, "get", "pods", "-A", "-o", "json")
if err != nil {
return nil, fmt.Errorf("query current pods for image-pull credential scan: %w", err)
}
var pods podList
if err := json.Unmarshal([]byte(podsOut), &pods); err != nil {
return nil, fmt.Errorf("decode current pods for image-pull credential scan: %w", err)
}
blocked := map[string]string{}
for _, pod := range pods.Items {
if podHasCurrentImagePullFailure(pod) {
blocked[pod.Metadata.Namespace+"/"+pod.Metadata.Name] = pod.Metadata.UID
}
}
for _, event := range events.Items {
if !strings.EqualFold(strings.TrimSpace(event.Type), "Warning") {
continue
@ -57,6 +79,14 @@ func (o *Orchestrator) imagePullCredentialBlockerReasons(ctx context.Context) (m
if namespace == "" || name == "" {
continue
}
// Retained events outlive pods and successful pulls. Require current state,
// exact pod identity, and recent evidence before changing a sync helper.
uid := blocked[namespace+"/"+name]
observed := eventLastObservedAt(event)
if uid == "" || uid != event.InvolvedObject.UID || observed.IsZero() ||
time.Since(observed) > credentialEventMaxAge || time.Until(observed) > time.Minute {
continue
}
reasons[namespace+"/"+name] = "ImagePullCredentialBlocker:" + imagePullCredentialFailureClass(reason, message)
}
return reasons, nil
@ -114,6 +144,8 @@ func (o *Orchestrator) healImagePullCredentialSync(ctx context.Context) ([]strin
// Why: namespaces using Vault/CSI secret material have tiny sync deployments
// named or labeled vault-sync; restarting those nudges secretObject rotation.
func (o *Orchestrator) restartVaultSyncDeployments(ctx context.Context, namespace string) ([]string, error) {
o.credentialRepairMu.Lock()
defer o.credentialRepairMu.Unlock()
out, err := o.kubectl(ctx, 20*time.Second, "-n", namespace, "get", "deployment", "-o", "json")
if err != nil {
return nil, fmt.Errorf("query deployments in %s for image-pull credential repair: %w", namespace, err)
@ -124,11 +156,20 @@ func (o *Orchestrator) restartVaultSyncDeployments(ctx context.Context, namespac
}
repaired := []string{}
found := false
for _, deployment := range deployments.Items {
name := strings.TrimSpace(deployment.Metadata.Name)
if name == "" || !vaultSyncDeployment(name, deployment.Metadata.Labels) {
continue
}
found = true
allowed, err := o.reserveCredentialRepair(namespace + "/deployment/" + name)
if err != nil {
return repaired, err
}
if !allowed {
continue
}
if _, err := o.kubectl(ctx, 25*time.Second, "-n", namespace, "rollout", "restart", "deployment", name); err != nil {
return repaired, fmt.Errorf("restart %s/deployment/%s for image-pull credential repair: %w", namespace, name, err)
}
@ -137,7 +178,7 @@ func (o *Orchestrator) restartVaultSyncDeployments(ctx context.Context, namespac
}
repaired = append(repaired, namespace+"/deployment/"+name)
}
if len(repaired) == 0 {
if !found {
return nil, fmt.Errorf("image-pull credential blocker in namespace %s but no vault-sync deployment was found", namespace)
}
return repaired, nil
@ -235,3 +276,46 @@ func vaultSyncDeployment(name string, labels map[string]string) bool {
}
return false
}
// podHasCurrentImagePullFailure returns whether a live pod still needs a pull.
// Completed, deleting, and recovered pods cannot justify a credential repair.
// Signature: podHasCurrentImagePullFailure(pod podResource) bool.
// Why: historical warnings must not trigger repairs for recovered pods.
func podHasCurrentImagePullFailure(pod podResource) bool {
if pod.Metadata.DeletionTimestamp != nil || pod.Status.Phase == "Succeeded" || pod.Status.Phase == "Failed" {
return false
}
for _, statuses := range [][]podContainerStatus{pod.Status.InitContainerStatuses, pod.Status.ContainerStatuses} {
for _, status := range statuses {
if wait := status.State.Waiting; wait != nil && (wait.Reason == "ErrImagePull" || wait.Reason == "ImagePullBackOff") {
return true
}
}
}
return false
}
// reserveCredentialRepair records an attempt before mutation and returns false
// during the cooldown. The existing run history preserves it over daemon restarts.
// A failed or timed-out rollout also consumes the cooldown to prevent churn.
// Signature: (o *Orchestrator) reserveCredentialRepair(target string) (bool, error).
// Why: a cooldown must survive daemon restarts and failed rollouts.
func (o *Orchestrator) reserveCredentialRepair(target string) (bool, error) {
if o.store == nil {
return false, errors.New("credential repair requires persistent run history")
}
records, err := o.store.Load()
if err != nil {
return false, fmt.Errorf("read credential repair history: %w", err)
}
for _, record := range records {
if record.Action == "image-pull-credential-repair" && record.Reason == target && time.Since(record.StartedAt) < credentialRepairCooldown {
return false, nil
}
}
now := time.Now()
if err := o.store.Append(state.RunRecord{ID: now.UTC().Format(time.RFC3339Nano), Action: "image-pull-credential-repair", Reason: target, StartedAt: now, EndedAt: now}); err != nil {
return false, fmt.Errorf("record credential repair attempt: %w", err)
}
return true, nil
}

View File

@ -2,8 +2,11 @@ package cluster
import (
"context"
"encoding/json"
"fmt"
"strings"
"testing"
"time"
"scm.bstein.dev/bstein/ananke/internal/config"
)
@ -19,8 +22,10 @@ func TestImagePullCredentialBlockerReasonsClassifiesHarborAuth(t *testing.T) {
`{"metadata":{"namespace":"veles"},"involvedObject":{"kind":"Pod","name":"veles-frontend"},"type":"Warning","reason":"FailedToRetrieveImagePullSecret","message":"Unable to retrieve some image pull secrets (harbor-regcred); attempting to pull the image may not succeed."},` +
`{"involvedObject":{"kind":"Pod","namespace":"logging","name":"oauth2"},"type":"Warning","reason":"Failed","message":"Failed to pull image: lookup registry-1.docker.io: Try again"}` +
`]}`
events, pods := currentCredentialFixture(t, events)
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: events},
{match: matchContains("kubectl", "get", "pods", "-A"), out: pods},
})
reasons, err := orch.imagePullCredentialBlockerReasons(context.Background())
@ -51,8 +56,10 @@ func TestHealImagePullCredentialSyncRestartsVaultSyncDeployment(t *testing.T) {
`]}`
restarted := false
rolledOut := false
events, pods := currentCredentialFixture(t, events)
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: events},
{match: matchContains("kubectl", "get", "pods", "-A"), out: pods},
{match: matchContains("kubectl", "-n", "veles", "get", "deployment", "-o", "json"), out: deployments},
{
match: func(name string, args []string) bool {
@ -85,3 +92,139 @@ func TestHealImagePullCredentialSyncRestartsVaultSyncDeployment(t *testing.T) {
t.Fatalf("expected vault-sync deployment restart and rollout wait, restarted=%v rolledOut=%v", restarted, rolledOut)
}
}
// currentCredentialFixture gives legacy fixtures current pod identities and states.
// Signature: currentCredentialFixture(t *testing.T, raw string) (string, string).
// Why: event-only fixtures cannot demonstrate a currently blocked pull.
func currentCredentialFixture(t *testing.T, raw string) (string, string) {
t.Helper()
var events eventList
if err := json.Unmarshal([]byte(raw), &events); err != nil {
t.Fatal(err)
}
pods := podList{}
for i := range events.Items {
event := &events.Items[i]
event.LastTimestamp = time.Now()
event.InvolvedObject.UID = fmt.Sprintf("pod-%d", i)
pod := podResource{}
pod.Metadata.Namespace = event.InvolvedObject.Namespace
if pod.Metadata.Namespace == "" {
pod.Metadata.Namespace = event.Metadata.Namespace
}
pod.Metadata.Name = event.InvolvedObject.Name
pod.Metadata.UID = event.InvolvedObject.UID
pod.Status.Phase = "Pending"
pod.Status.ContainerStatuses = []podContainerStatus{{State: podContainerState{Waiting: &podContainerWaitingState{Reason: "ImagePullBackOff"}}}}
pods.Items = append(pods.Items, pod)
}
e, _ := json.Marshal(events)
p, _ := json.Marshal(pods)
return string(e), string(p)
}
// TestCredentialRecoveryRequiresCurrentEvidence rejects warnings for healed or replaced pods.
// Signature: TestCredentialRecoveryRequiresCurrentEvidence(t *testing.T).
// Why: only fresh evidence for the current pod may trigger mutation.
func TestCredentialRecoveryRequiresCurrentEvidence(t *testing.T) {
for _, scenario := range []string{"current", "init", "stale", "missing-time", "future", "missing-pod", "replaced", "missing-uid", "running", "deleting", "succeeded", "failed", "query-error", "bad-pods", "empty-events"} {
t.Run(scenario, func(t *testing.T) {
raw, podsRaw := currentCredentialFixture(t, `{"items":[{"type":"Warning","reason":"Failed","message":"unauthorized","involvedObject":{"kind":"Pod","namespace":"apps","name":"test"}}]}`)
var events eventList
var pods podList
_ = json.Unmarshal([]byte(raw), &events)
_ = json.Unmarshal([]byte(podsRaw), &pods)
switch scenario {
case "init":
pods.Items[0].Status.InitContainerStatuses = pods.Items[0].Status.ContainerStatuses
pods.Items[0].Status.ContainerStatuses = nil
case "stale":
events.Items[0].LastTimestamp = time.Now().Add(-11 * time.Minute)
case "missing-time":
events.Items[0].LastTimestamp = time.Time{}
case "future":
events.Items[0].LastTimestamp = time.Now().Add(time.Hour)
case "missing-pod":
pods.Items = nil
case "replaced":
pods.Items[0].Metadata.UID = "replacement"
case "missing-uid":
events.Items[0].InvolvedObject.UID = ""
case "running":
pods.Items[0].Status.ContainerStatuses = nil
pods.Items[0].Status.Phase = "Running"
case "deleting":
now := time.Now()
pods.Items[0].Metadata.DeletionTimestamp = &now
case "succeeded":
pods.Items[0].Status.Phase = "Succeeded"
case "failed":
pods.Items[0].Status.Phase = "Failed"
}
e, _ := json.Marshal(events)
p, _ := json.Marshal(pods)
raw, podsRaw = string(e), string(p)
var queryErr error
if scenario == "query-error" {
queryErr = fmt.Errorf("unavailable")
}
if scenario == "bad-pods" {
podsRaw = "{"
}
if scenario == "empty-events" {
raw = ""
}
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "events", "-A"), out: raw},
{match: matchContains("kubectl", "get", "pods", "-A"), out: podsRaw, err: queryErr},
})
reasons, err := orch.imagePullCredentialBlockerReasons(context.Background())
if scenario == "query-error" || scenario == "bad-pods" {
if err == nil {
t.Fatal("expected error")
}
return
}
if err != nil {
t.Fatal(err)
}
want := 0
if scenario == "current" || scenario == "init" {
want = 1
}
if len(reasons) != want {
t.Fatalf("got %v, want %d blockers", reasons, want)
}
})
}
}
// TestCredentialRepairCooldownSurvivesRestart prevents repeated recovery mutations.
// Signature: TestCredentialRepairCooldownSurvivesRestart(t *testing.T).
// Why: a failing helper must not be restarted every recovery cycle.
func TestCredentialRepairCooldownSurvivesRestart(t *testing.T) {
restarts := 0
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "deployment"), out: `{"items":[{"metadata":{"name":"app-vault-sync"}}]}`},
{match: func(name string, args []string) bool {
if matchContains("kubectl", "rollout", "restart")(name, args) {
restarts++
return true
}
return false
}},
{match: matchContains("kubectl", "rollout", "status"), err: fmt.Errorf("rollout timeout")},
})
if _, err := orch.restartVaultSyncDeployments(context.Background(), "apps"); err == nil {
t.Fatal("expected timeout")
}
replacement := &Orchestrator{cfg: orch.cfg, runner: orch.runner, store: orch.store, log: orch.log, runOverride: orch.runOverride}
repaired, err := replacement.restartVaultSyncDeployments(context.Background(), "apps")
if err != nil || len(repaired) != 0 || restarts != 1 {
t.Fatalf("repeated repair: %v %v %d", repaired, err, restarts)
}
replacement.store = nil
if _, err := replacement.reserveCredentialRepair("apps/other"); err == nil {
t.Fatal("missing state must fail closed")
}
}

View File

@ -10,8 +10,6 @@ import (
"time"
)
const longhornSystemNamespace = "longhorn-system"
type longhornNodeList struct {
Items []longhornNode `json:"items"`
}

View File

@ -231,7 +231,9 @@ func TestImagePullDNSAndCredentialBranches(t *testing.T) {
{"type":"Warning","reason":"FailedPull","metadata":{"namespace":"finance"},"involvedObject":{"kind":"Pod","name":"budget"},"message":"unauthorized: authentication required"},
{"type":"Warning","reason":"FailedToRetrieveImagePullSecret","involvedObject":{"kind":"Pod","namespace":"sso","name":"keycloak"},"message":"unable to retrieve some image pull secrets"}
]}`
eventJSON, podsJSON := currentCredentialFixture(t, eventJSON)
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "pods", "-A"), out: podsJSON},
{match: matchContains("kubectl", "get", "events", "-A"), out: eventJSON},
})
dnsReasons, err := orch.imagePullDNSBlockerReasons(context.Background())

View File

@ -107,12 +107,14 @@ func TestPostStartAutoHealRequestsReconcileAfterImagePullRepair(t *testing.T) {
ctx := context.Background()
eventsJSON := `{"items":[{"type":"Warning","reason":"FailedPull","message":"unauthorized: authentication required","involvedObject":{"kind":"Pod","namespace":"apps","name":"web"}}]}`
deploymentsJSON := `{"items":[{"metadata":{"name":"vault-sync","labels":{"app":"vault-sync"}}}]}`
eventsJSON, credentialPods := currentCredentialFixture(t, eventsJSON)
sawFluxSourceReconcile := false
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "-n", "longhorn-system", "get", "nodes.longhorn.io", "-o", "json"), out: `{"items":[]}`},
{match: matchContains("kubectl", "get", "nodes", "-o", "json"), out: `{"items":[]}`},
{match: matchContains("kubectl", "-n", "vault", "get", "pod", "vault-0"), out: "Pending"},
{match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: eventsJSON},
{match: matchContains("kubectl", "get", "pods", "-A", "-o", "json"), out: credentialPods},
{match: matchContains("kubectl", "-n", "apps", "get", "deployment", "-o", "json"), out: deploymentsJSON},
{match: matchContains("kubectl", "-n", "apps", "rollout", "restart", "deployment", "vault-sync"), out: ""},
{match: matchContains("kubectl", "-n", "apps", "rollout", "status", "deployment/vault-sync"), out: ""},

View File

@ -187,6 +187,7 @@ type eventResource struct {
CreationTimestamp time.Time `json:"creationTimestamp"`
} `json:"metadata"`
InvolvedObject struct {
UID string `json:"uid"`
Kind string `json:"kind"`
Namespace string `json:"namespace"`
Name string `json:"name"`
@ -242,6 +243,7 @@ type podList struct {
type podResource struct {
Metadata struct {
UID string `json:"uid"`
Namespace string `json:"namespace"`
Name string `json:"name"`
Labels map[string]string `json:"labels"`

View File

@ -135,6 +135,10 @@ func isTimeSynced(raw string) bool {
// Signature: (o *Orchestrator) preflightExternalDatastore(ctx context.Context) error.
// Why: keeps behavior explicit so startup/shutdown workflows remain maintainable as services evolve.
func (o *Orchestrator) preflightExternalDatastore(ctx context.Context) error {
// Cancellation must win even when a local datastore is already reachable.
if err := ctx.Err(); err != nil {
return err
}
if len(o.cfg.ControlPlanes) == 0 {
return nil
}

View File

@ -217,6 +217,11 @@ func (d *Daemon) Run(ctx context.Context) error {
// like a later Vault reseal or stale dead-node deletions without waiting for a
// fresh bootstrap run.
func (d *Daemon) maybeRunPostStartAutoHeal(ctx context.Context, lastRun *time.Time, anyOnBattery bool) {
// Shared cluster repairs have one owner; peers still monitor their UPS and
// forward shutdown intent through the existing coordination path.
if d.cfg.Coordination.Role == "peer" {
return
}
interval := time.Duration(d.cfg.Startup.PostStartAutoHealSeconds) * time.Second
if interval <= 0 || anyOnBattery {
return

View File

@ -1,7 +1,14 @@
package service
import (
"bytes"
"context"
"errors"
"log"
"strings"
"scm.bstein.dev/bstein/ananke/internal/cluster"
"scm.bstein.dev/bstein/ananke/internal/execx"
"testing"
"time"
@ -26,8 +33,14 @@ func TestDaemonMaybeRunPostStartAutoHeal(t *testing.T) {
},
}
d.cfg.Coordination.Role = "peer"
var last time.Time
d.maybeRunPostStartAutoHeal(context.Background(), &last, false)
if calls != 0 || !last.IsZero() {
t.Fatal("peer must not mutate shared cluster recovery state")
}
d.cfg.Coordination.Role = "coordinator"
d.maybeRunPostStartAutoHeal(context.Background(), &last, false)
if calls != 1 {
t.Fatalf("expected first auto-heal invocation, got %d", calls)
}
@ -49,3 +62,28 @@ func TestDaemonMaybeRunPostStartAutoHeal(t *testing.T) {
t.Fatalf("expected second allowed auto-heal call, got %d", calls)
}
}
// TestDaemonRecoveryFailureDoesNotStopMonitoring exercises optional repair isolation.
// Signature: TestDaemonRecoveryFailureDoesNotStopMonitoring(t *testing.T).
// Why: a missing or failed repair helper must not terminate UPS monitoring.
func TestDaemonRecoveryFailureDoesNotStopMonitoring(t *testing.T) {
var messages bytes.Buffer
d := &Daemon{cfg: config.Config{Startup: config.Startup{PostStartAutoHealSeconds: 60}}, log: log.New(&messages, "", 0)}
var last time.Time
d.maybeRunPostStartAutoHeal(context.Background(), &last, false)
if !last.IsZero() {
t.Fatal("missing helper must not consume an interval")
}
if err := d.runPostStartAutoHeal(context.Background()); err != nil {
t.Fatal(err)
}
d.orch = cluster.New(config.Config{}, &execx.Runner{DryRun: true}, nil, d.log)
if err := d.runPostStartAutoHeal(context.Background()); err != nil {
t.Fatal(err)
}
d.postStartAutoHealOverride = func(context.Context) error { return errors.New("synthetic repair failure") }
d.maybeRunPostStartAutoHeal(context.Background(), &last, false)
if last.IsZero() || !strings.Contains(messages.String(), "synthetic repair failure") {
t.Fatal("failed repair must be logged and rate limited")
}
}

View File

@ -13,7 +13,7 @@ HOST_SHORT="$(hostname -s 2>/dev/null || hostname)"
LOG_FILE="${ANANKE_UPDATE_LOG_FILE:-/var/log/ananke/update.log}"
STATE_FILE="${ANANKE_UPDATE_STATE_FILE:-/var/lib/ananke/update-last.env}"
LOCK_FILE="${ANANKE_UPDATE_LOCK_FILE:-/var/lock/ananke-update.lock}"
ALLOW_QUALITY_FALLBACK="${ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK:-1}"
ALLOW_QUALITY_FALLBACK="${ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK:-0}"
QUALITY_GATE_MODE="${ANANKE_ENFORCE_QUALITY_GATE:-1}"
mkdir -p "$(dirname "${LOG_FILE}")" "$(dirname "${STATE_FILE}")" "$(dirname "${LOCK_FILE}")"
@ -103,6 +103,8 @@ if "${REPO_DIR}/scripts/install.sh"; then
write_state "ok" "install-success" "${CURRENT_FROM_REV}" "${TARGET_TO_REV}"
echo "[self-update] completed successfully"
exit 0
else
INSTALL_EXIT_CODE=$?
fi
if [[ "${ALLOW_QUALITY_FALLBACK}" == "1" || "${ALLOW_QUALITY_FALLBACK}" == "true" ]]; then
@ -113,3 +115,8 @@ if [[ "${ALLOW_QUALITY_FALLBACK}" == "1" || "${ALLOW_QUALITY_FALLBACK}" == "true
echo "[self-update] completed via fallback mode"
exit 0
fi
# A failed command used as an if condition does not trigger errexit or ERR.
write_state "failed" "install-failed;rc=${INSTALL_EXIT_CODE}" "${CURRENT_FROM_REV}" "${TARGET_TO_REV}"
echo "[self-update] installer failed rc=${INSTALL_EXIT_CODE}; fallback disabled"
exit "${INSTALL_EXIT_CODE}"

View File

@ -14,6 +14,7 @@ SYSTEMD_DIR="/etc/systemd/system"
LIB_DIR="/usr/local/lib/ananke"
START_NOW=1
INSTALL_DEPS=1
BINARY_ONLY=0
ENABLE_BOOTSTRAP="${ANANKE_ENABLE_BOOTSTRAP:-auto}"
MANAGE_NUT="${ANANKE_MANAGE_NUT:-1}"
NUT_UPS_NAME="${ANANKE_NUT_UPS_NAME:-}"
@ -27,6 +28,10 @@ ENFORCE_QUALITY_GATE="${ANANKE_ENFORCE_QUALITY_GATE:-1}"
while [[ $# -gt 0 ]]; do
case "$1" in
--binary-only)
BINARY_ONLY=1
shift
;;
--no-start)
START_NOW=0
shift
@ -47,8 +52,14 @@ source "${REPO_DIR}/scripts/install-host-bootstrap.sh"
source "${REPO_DIR}/scripts/install-legacy-migration.sh"
source "${REPO_DIR}/scripts/install-artifacts.sh"
if [[ "${BINARY_ONLY}" == "1" && ! -f "${BIN_DIR}/ananke" ]]; then
echo "[install] binary-only requires an existing installation" >&2
exit 1
fi
ensure_dependencies
migrate_legacy_hecate_install
if [[ "${BINARY_ONLY}" != "1" ]]; then
migrate_legacy_hecate_install
fi
if [[ "${ENFORCE_QUALITY_GATE}" == "1" ]]; then
echo "[install] running quality gate"
@ -69,8 +80,26 @@ go build -o dist/ananke "${BUILD_TARGET}"
echo "[install] installing binary"
install -d -m 0755 "${BIN_DIR}"
if [[ "${BINARY_ONLY}" == "1" && -f "${BIN_DIR}/ananke" ]]; then
install -d -m 0700 "${LIB_DIR}/rollback"
install -m 0700 "${BIN_DIR}/ananke" "${LIB_DIR}/rollback/ananke.previous"
fi
install -m 0755 dist/ananke "${BIN_DIR}/ananke"
# Targeted recovery fixes must not overwrite host config, NUT, or bootstrap units.
if [[ "${BINARY_ONLY}" == "1" ]]; then
if [[ "${START_NOW}" == "1" ]]; then
if ! systemctl restart ananke.service || ! systemctl is-active --quiet ananke.service; then
echo "[install] service failed; restoring previous binary" >&2
install -m 0755 "${LIB_DIR}/rollback/ananke.previous" "${BIN_DIR}/ananke"
systemctl restart ananke.service
exit 1
fi
fi
echo "[install] binary-only update complete; host configuration unchanged"
exit 0
fi
echo "[install] installing config + state dirs"
install -d -m 0750 "${CONF_DIR}"
install -d -m 0750 "${STATE_DIR}"

View File

@ -157,6 +157,11 @@ quality_gate_finalize() {
trap 'quality_gate_finalize $?' EXIT
cd "${REPO_DIR}"
# Pin the resolved toolchain for child coverage commands. Auto-switching from an
# older host Go can otherwise look for the removed covdata tool (Go issue 75031).
if [[ "$(go env GOTOOLCHAIN)" == "auto" ]]; then
export GOTOOLCHAIN="$(go env GOVERSION)"
fi
mkdir -p "${BUILD_DIR}"
rm -f "${COVERAGE_PROFILE}" "${COVERAGE_PERCENT_FILE}"
printf 'failed\n' > "${BUILD_DIR}/docs-naming.status"

View File

@ -96,7 +96,7 @@ func TestHookVaultLifecycleBranchMatrix(t *testing.T) {
}
})
t.Run("workload-ready-no-value-and-ensure-error", func(t *testing.T) {
t.Run("workload-ready-no-value-and-ensure-cancel", func(t *testing.T) {
cfg := lifecycleConfig(t)
runNoValue := func(ctx context.Context, timeout time.Duration, name string, args ...string) (string, error) {
command := name + " " + strings.Join(args, " ")
@ -111,6 +111,8 @@ func TestHookVaultLifecycleBranchMatrix(t *testing.T) {
t.Fatalf("expected no-value readiness branch, ready=%v err=%v", ready, err)
}
waitCtx, cancelWait := context.WithCancel(context.Background())
defer cancelWait()
runEnsureErr := func(ctx context.Context, timeout time.Duration, name string, args ...string) (string, error) {
command := name + " " + strings.Join(args, " ")
switch {
@ -119,14 +121,15 @@ func TestHookVaultLifecycleBranchMatrix(t *testing.T) {
case name == "kubectl" && strings.Contains(command, "get pods -o custom-columns"):
return "", nil
case name == "kubectl" && strings.Contains(command, "rollout status statefulset/victoria-metrics-single-server"):
cancelWait()
return "", errors.New("rollout failed")
default:
return lifecycleDispatcher(&commandRecorder{})(ctx, timeout, name, args...)
}
}
orchEnsureErr, _ := newHookOrchestratorWithRunnerMode(t, cfg, false, runEnsureErr, runEnsureErr)
if err := orchEnsureErr.TestHookEnsureCriticalStartupWorkloads(context.Background()); err == nil || !strings.Contains(err.Error(), "rollout failed") {
t.Fatalf("expected ensureCriticalStartupWorkloads wait error branch, got %v", err)
if err := orchEnsureErr.TestHookEnsureCriticalStartupWorkloads(waitCtx); !errors.Is(err, context.Canceled) {
t.Fatalf("expected ensureCriticalStartupWorkloads cancellation after failed rollout, got %v", err)
}
})
@ -359,6 +362,8 @@ func TestHookVaultLifecycleBranchMatrix(t *testing.T) {
cfgFail.Startup.RequireFluxHealth = false
cfgFail.Startup.RequireWorkloadConvergence = false
cfgFail.Startup.RequireCriticalServiceEndpoints = true
cfgFail.Startup.CriticalServiceEndpointWaitSec = 1
cfgFail.Startup.CriticalServiceEndpointPollSec = 1
run := func(ctx context.Context, timeout time.Duration, name string, args ...string) (string, error) {
command := name + " " + strings.Join(args, " ")
if name == "kubectl" && strings.Contains(command, "get endpoints") {

View File

@ -0,0 +1,61 @@
#!/usr/bin/env bash
# Exercise updater exit/status behavior with a local repository and fake installer.
# Run as root; all writes, Git configuration and installer calls stay in a temp dir.
set -euo pipefail
if [[ "${EUID}" -ne 0 ]]; then
echo "Run this isolated updater regression as root" >&2
exit 1
fi
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)"
FIXTURE="$(mktemp -d /tmp/ananke-update-test.XXXXXX)"
trap 'rm -rf "${FIXTURE}"' EXIT
export GIT_CONFIG_GLOBAL="${FIXTURE}/gitconfig"
export GIT_CONFIG_NOSYSTEM=1
git config --global user.name 'Updater regression'
git config --global user.email 'updater-test@example.invalid'
mkdir -p "${FIXTURE}/origin/scripts"
git init -q -b main "${FIXTURE}/origin"
cat > "${FIXTURE}/origin/scripts/install.sh" <<'INSTALLER'
#!/usr/bin/env bash
set -euo pipefail
printf '%s\n' "${ANANKE_ENFORCE_QUALITY_GATE}" >> "${FIXTURE_CALLS}"
if [[ "${ANANKE_ENFORCE_QUALITY_GATE}" == 1 ]]; then
exit "${FIXTURE_INSTALL_RC}"
fi
INSTALLER
chmod 0755 "${FIXTURE}/origin/scripts/install.sh"
git -C "${FIXTURE}/origin" add scripts/install.sh
git -C "${FIXTURE}/origin" commit -qm fixture
# Inputs: case name, installer status, fallback policy, expected exit/state/calls.
# Output: assertion failure or one concise pass line; no host services are invoked.
check_update() {
local name="$1" install_rc="$2" fallback="$3" expected_rc="$4"
local expected_status="$5" expected_calls="$6" actual_rc=0
local directory="${FIXTURE}/${name}"
mkdir -p "${directory}"
env ANANKE_REPO_URL="${FIXTURE}/origin" \
ANANKE_REPO_DIR="${directory}/repo" \
ANANKE_REPO_BRANCH=main \
ANANKE_UPDATE_LOG_FILE="${directory}/update.log" \
ANANKE_UPDATE_STATE_FILE="${directory}/state" \
ANANKE_UPDATE_LOCK_FILE="${directory}/lock" \
ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK="${fallback}" \
ANANKE_ENFORCE_QUALITY_GATE=1 \
FIXTURE_CALLS="${directory}/calls" FIXTURE_INSTALL_RC="${install_rc}" \
bash "${REPO_ROOT}/scripts/ananke-self-update.sh" > "${directory}/output" 2>&1 || actual_rc=$?
[[ "${actual_rc}" == "${expected_rc}" ]]
grep -qx "status=${expected_status}" "${directory}/state"
[[ "$(wc -l < "${directory}/calls")" -eq "${expected_calls}" ]]
if [[ "${expected_status}" == failed ]]; then
grep -qx "detail=install-failed;rc=${install_rc}" "${directory}/state"
fi
printf 'PASS %s: exit=%s state=%s calls=%s\n' \
"${name}" "${actual_rc}" "${expected_status}" "${expected_calls}"
}
check_update strict_failure 42 0 42 failed 1
check_update explicit_legacy_fallback 42 1 0 degraded-ok 2
check_update strict_success 0 0 0 ok 1