Compare commits
1 Commits
main
...
hermes-rep
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8fda40f6f3 |
28
Jenkinsfile
vendored
28
Jenkinsfile
vendored
@ -1,16 +1,11 @@
|
||||
pipeline {
|
||||
agent {
|
||||
kubernetes {
|
||||
// Keep build I/O off the node's runtime USB drive.
|
||||
workspaceVolume dynamicPVC(accessModes: 'ReadWriteOnce', requestsSize: '20Gi', storageClassName: 'ci-scratch')
|
||||
defaultContainer 'go-tester'
|
||||
yaml """
|
||||
apiVersion: v1
|
||||
kind: Pod
|
||||
spec:
|
||||
securityContext:
|
||||
fsGroup: 1000
|
||||
fsGroupChangePolicy: OnRootMismatch
|
||||
nodeSelector:
|
||||
hardware: rpi5
|
||||
kubernetes.io/arch: arm64
|
||||
@ -20,9 +15,11 @@ spec:
|
||||
requiredDuringSchedulingIgnoredDuringExecution:
|
||||
nodeSelectorTerms:
|
||||
- matchExpressions:
|
||||
- {key: hardware, operator: In, values: [rpi5, rpi4]}
|
||||
- {key: node-role.kubernetes.io/worker, operator: In, values: ["true"]}
|
||||
- {key: kubernetes.io/hostname, operator: NotIn, values: [titan-04, titan-05, titan-06, titan-08, titan-11, titan-12, titan-13, titan-14, titan-15, titan-17, titan-18, titan-19]}
|
||||
- key: kubernetes.io/hostname
|
||||
operator: NotIn
|
||||
values:
|
||||
- titan-06
|
||||
- titan-11
|
||||
preferredDuringSchedulingIgnoredDuringExecution:
|
||||
- weight: 100
|
||||
preference:
|
||||
@ -63,18 +60,14 @@ spec:
|
||||
volumeMounts:
|
||||
- name: workspace-volume
|
||||
mountPath: /home/jenkins/agent
|
||||
volumes:
|
||||
- name: workspace-volume
|
||||
emptyDir: {}
|
||||
"""
|
||||
}
|
||||
}
|
||||
|
||||
environment {
|
||||
PIP_CACHE_DIR = '/home/jenkins/agent/.cache/pip'
|
||||
NPM_CONFIG_CACHE = '/home/jenkins/agent/.cache/npm'
|
||||
SONAR_USER_HOME = '/home/jenkins/agent/.cache/sonar'
|
||||
TMPDIR = '/home/jenkins/agent/.cache/tmp'
|
||||
GOCACHE = '/home/jenkins/agent/.cache/go-build'
|
||||
GOMODCACHE = '/home/jenkins/agent/.cache/go-mod'
|
||||
GOTMPDIR = '/home/jenkins/agent/.cache/go-tmp'
|
||||
SUITE_NAME = 'ananke'
|
||||
PUSHGATEWAY_URL = 'http://platform-quality-gateway.monitoring.svc.cluster.local:9091'
|
||||
SONARQUBE_HOST_URL = 'http://sonarqube.quality.svc.cluster.local:9000'
|
||||
@ -97,11 +90,6 @@ spec:
|
||||
}
|
||||
|
||||
stages {
|
||||
stage('Prepare build scratch') {
|
||||
steps {
|
||||
sh 'mkdir -p /home/jenkins/agent/.cache/tmp /home/jenkins/agent/.cache/go-tmp'
|
||||
}
|
||||
}
|
||||
stage('Checkout') {
|
||||
steps {
|
||||
checkout scm
|
||||
|
||||
25
README.md
25
README.md
@ -67,28 +67,3 @@ Local testing check before installing:
|
||||
```
|
||||
|
||||
Emergency installs can bypass the gate with `ANANKE_ENFORCE_QUALITY_GATE=0` - try to avoid this. You should be treating failures as an instructive opportunity to improve Ananke.
|
||||
|
||||
## Credential recovery and targeted updates
|
||||
|
||||
Registry credential recovery requires a warning observed within ten minutes for
|
||||
an existing pod UID whose container or init container is still in ErrImagePull
|
||||
or ImagePullBackOff. Completed and deleting pods are ignored. A helper restart
|
||||
is reserved in the existing run history before execution and may occur at most
|
||||
once per helper in 30 minutes, including failed rollouts. History must remain
|
||||
writable; otherwise this repair fails closed. Only the coordinator daemon runs periodic shared-cluster repairs; the peer
|
||||
continues UPS monitoring and shutdown forwarding. Do not configure two hosts as
|
||||
coordinators for one cluster.
|
||||
|
||||
A targeted daemon fix can use the existing installer without rewriting host
|
||||
configuration, UPS settings, or systemd/bootstrap units:
|
||||
|
||||
```bash
|
||||
sudo ./scripts/install.sh --binary-only --skip-deps
|
||||
```
|
||||
|
||||
The quality gate still runs. The previous binary is retained at
|
||||
`/usr/local/lib/ananke/rollback/ananke.previous`; a failed service start restores
|
||||
it. Verify service health and journals after installation. Automated self-updates
|
||||
now default to keeping the current binary when their quality gate fails. The
|
||||
service unit and updater script both set that default; existing hosts need those
|
||||
two files updated to adopt it.
|
||||
|
||||
@ -29,7 +29,7 @@ ssh_node_hosts:
|
||||
titan-22: 192.168.22.22
|
||||
titan-24: 192.168.22.26
|
||||
ssh_node_users:
|
||||
titan-24: tethys
|
||||
titan-24: atlas
|
||||
ssh_managed_nodes:
|
||||
- titan-db
|
||||
- titan-0a
|
||||
@ -110,10 +110,6 @@ excluded_namespaces:
|
||||
- postgres
|
||||
- maintenance
|
||||
startup:
|
||||
# Existing Vault CSI synchronization supplies password-backed host actions.
|
||||
host_sudo_secret_namespace: maintenance
|
||||
host_sudo_secret_name_template: "ananke-sudo-{node}"
|
||||
host_sudo_secret_password_key: password
|
||||
api_wait_seconds: 1200
|
||||
api_poll_seconds: 2
|
||||
shutdown_cooldown_seconds: 45
|
||||
|
||||
@ -29,7 +29,7 @@ ssh_node_hosts:
|
||||
titan-22: 192.168.22.22
|
||||
titan-24: 192.168.22.26
|
||||
ssh_node_users:
|
||||
titan-24: tethys
|
||||
titan-24: atlas
|
||||
ssh_managed_nodes:
|
||||
- titan-db
|
||||
- titan-0a
|
||||
@ -110,10 +110,6 @@ excluded_namespaces:
|
||||
- postgres
|
||||
- maintenance
|
||||
startup:
|
||||
# Existing Vault CSI synchronization supplies password-backed host actions.
|
||||
host_sudo_secret_namespace: maintenance
|
||||
host_sudo_secret_name_template: "ananke-sudo-{node}"
|
||||
host_sudo_secret_password_key: password
|
||||
api_wait_seconds: 1200
|
||||
api_poll_seconds: 2
|
||||
shutdown_cooldown_seconds: 45
|
||||
|
||||
@ -10,7 +10,7 @@ User=root
|
||||
Group=root
|
||||
Environment=ANANKE_UPDATE_LOG_FILE=/var/log/ananke/update.log
|
||||
Environment=ANANKE_UPDATE_STATE_FILE=/var/lib/ananke/update-last.env
|
||||
Environment=ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK=0
|
||||
Environment=ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK=1
|
||||
ExecStart=/usr/local/lib/ananke/ananke-self-update.sh
|
||||
TimeoutStartSec=1800
|
||||
StandardOutput=journal
|
||||
|
||||
@ -21,7 +21,6 @@ type Orchestrator struct {
|
||||
runOverride func(timeoutCtx context.Context, timeout time.Duration, name string, args ...string) (string, error)
|
||||
runSensitiveOverride func(timeoutCtx context.Context, timeout time.Duration, name string, args ...string) (string, error)
|
||||
sshInputOverride func(timeoutCtx context.Context, timeout time.Duration, node string, command string, input string) (string, error)
|
||||
credentialRepairMu sync.Mutex
|
||||
startupReportMu sync.Mutex
|
||||
activeStartupReport *startupReport
|
||||
}
|
||||
|
||||
@ -8,13 +8,8 @@ import (
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"scm.bstein.dev/bstein/ananke/internal/state"
|
||||
)
|
||||
|
||||
const credentialEventMaxAge = 10 * time.Minute
|
||||
const credentialRepairCooldown = 30 * time.Minute
|
||||
|
||||
type imagePullCredentialDeploymentList struct {
|
||||
Items []struct {
|
||||
Metadata struct {
|
||||
@ -42,23 +37,6 @@ func (o *Orchestrator) imagePullCredentialBlockerReasons(ctx context.Context) (m
|
||||
if err := json.Unmarshal([]byte(eventsOut), &events); err != nil {
|
||||
return nil, fmt.Errorf("decode events for image-pull credential scan: %w", err)
|
||||
}
|
||||
if len(events.Items) == 0 {
|
||||
return reasons, nil
|
||||
}
|
||||
podsOut, err := o.kubectl(ctx, 30*time.Second, "get", "pods", "-A", "-o", "json")
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("query current pods for image-pull credential scan: %w", err)
|
||||
}
|
||||
var pods podList
|
||||
if err := json.Unmarshal([]byte(podsOut), &pods); err != nil {
|
||||
return nil, fmt.Errorf("decode current pods for image-pull credential scan: %w", err)
|
||||
}
|
||||
blocked := map[string]string{}
|
||||
for _, pod := range pods.Items {
|
||||
if podHasCurrentImagePullFailure(pod) {
|
||||
blocked[pod.Metadata.Namespace+"/"+pod.Metadata.Name] = pod.Metadata.UID
|
||||
}
|
||||
}
|
||||
for _, event := range events.Items {
|
||||
if !strings.EqualFold(strings.TrimSpace(event.Type), "Warning") {
|
||||
continue
|
||||
@ -79,14 +57,6 @@ func (o *Orchestrator) imagePullCredentialBlockerReasons(ctx context.Context) (m
|
||||
if namespace == "" || name == "" {
|
||||
continue
|
||||
}
|
||||
// Retained events outlive pods and successful pulls. Require current state,
|
||||
// exact pod identity, and recent evidence before changing a sync helper.
|
||||
uid := blocked[namespace+"/"+name]
|
||||
observed := eventLastObservedAt(event)
|
||||
if uid == "" || uid != event.InvolvedObject.UID || observed.IsZero() ||
|
||||
time.Since(observed) > credentialEventMaxAge || time.Until(observed) > time.Minute {
|
||||
continue
|
||||
}
|
||||
reasons[namespace+"/"+name] = "ImagePullCredentialBlocker:" + imagePullCredentialFailureClass(reason, message)
|
||||
}
|
||||
return reasons, nil
|
||||
@ -144,8 +114,6 @@ func (o *Orchestrator) healImagePullCredentialSync(ctx context.Context) ([]strin
|
||||
// Why: namespaces using Vault/CSI secret material have tiny sync deployments
|
||||
// named or labeled vault-sync; restarting those nudges secretObject rotation.
|
||||
func (o *Orchestrator) restartVaultSyncDeployments(ctx context.Context, namespace string) ([]string, error) {
|
||||
o.credentialRepairMu.Lock()
|
||||
defer o.credentialRepairMu.Unlock()
|
||||
out, err := o.kubectl(ctx, 20*time.Second, "-n", namespace, "get", "deployment", "-o", "json")
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("query deployments in %s for image-pull credential repair: %w", namespace, err)
|
||||
@ -156,20 +124,11 @@ func (o *Orchestrator) restartVaultSyncDeployments(ctx context.Context, namespac
|
||||
}
|
||||
|
||||
repaired := []string{}
|
||||
found := false
|
||||
for _, deployment := range deployments.Items {
|
||||
name := strings.TrimSpace(deployment.Metadata.Name)
|
||||
if name == "" || !vaultSyncDeployment(name, deployment.Metadata.Labels) {
|
||||
continue
|
||||
}
|
||||
found = true
|
||||
allowed, err := o.reserveCredentialRepair(namespace + "/deployment/" + name)
|
||||
if err != nil {
|
||||
return repaired, err
|
||||
}
|
||||
if !allowed {
|
||||
continue
|
||||
}
|
||||
if _, err := o.kubectl(ctx, 25*time.Second, "-n", namespace, "rollout", "restart", "deployment", name); err != nil {
|
||||
return repaired, fmt.Errorf("restart %s/deployment/%s for image-pull credential repair: %w", namespace, name, err)
|
||||
}
|
||||
@ -178,7 +137,7 @@ func (o *Orchestrator) restartVaultSyncDeployments(ctx context.Context, namespac
|
||||
}
|
||||
repaired = append(repaired, namespace+"/deployment/"+name)
|
||||
}
|
||||
if !found {
|
||||
if len(repaired) == 0 {
|
||||
return nil, fmt.Errorf("image-pull credential blocker in namespace %s but no vault-sync deployment was found", namespace)
|
||||
}
|
||||
return repaired, nil
|
||||
@ -276,46 +235,3 @@ func vaultSyncDeployment(name string, labels map[string]string) bool {
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// podHasCurrentImagePullFailure returns whether a live pod still needs a pull.
|
||||
// Completed, deleting, and recovered pods cannot justify a credential repair.
|
||||
// Signature: podHasCurrentImagePullFailure(pod podResource) bool.
|
||||
// Why: historical warnings must not trigger repairs for recovered pods.
|
||||
func podHasCurrentImagePullFailure(pod podResource) bool {
|
||||
if pod.Metadata.DeletionTimestamp != nil || pod.Status.Phase == "Succeeded" || pod.Status.Phase == "Failed" {
|
||||
return false
|
||||
}
|
||||
for _, statuses := range [][]podContainerStatus{pod.Status.InitContainerStatuses, pod.Status.ContainerStatuses} {
|
||||
for _, status := range statuses {
|
||||
if wait := status.State.Waiting; wait != nil && (wait.Reason == "ErrImagePull" || wait.Reason == "ImagePullBackOff") {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// reserveCredentialRepair records an attempt before mutation and returns false
|
||||
// during the cooldown. The existing run history preserves it over daemon restarts.
|
||||
// A failed or timed-out rollout also consumes the cooldown to prevent churn.
|
||||
// Signature: (o *Orchestrator) reserveCredentialRepair(target string) (bool, error).
|
||||
// Why: a cooldown must survive daemon restarts and failed rollouts.
|
||||
func (o *Orchestrator) reserveCredentialRepair(target string) (bool, error) {
|
||||
if o.store == nil {
|
||||
return false, errors.New("credential repair requires persistent run history")
|
||||
}
|
||||
records, err := o.store.Load()
|
||||
if err != nil {
|
||||
return false, fmt.Errorf("read credential repair history: %w", err)
|
||||
}
|
||||
for _, record := range records {
|
||||
if record.Action == "image-pull-credential-repair" && record.Reason == target && time.Since(record.StartedAt) < credentialRepairCooldown {
|
||||
return false, nil
|
||||
}
|
||||
}
|
||||
now := time.Now()
|
||||
if err := o.store.Append(state.RunRecord{ID: now.UTC().Format(time.RFC3339Nano), Action: "image-pull-credential-repair", Reason: target, StartedAt: now, EndedAt: now}); err != nil {
|
||||
return false, fmt.Errorf("record credential repair attempt: %w", err)
|
||||
}
|
||||
return true, nil
|
||||
}
|
||||
|
||||
@ -2,11 +2,8 @@ package cluster
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"scm.bstein.dev/bstein/ananke/internal/config"
|
||||
)
|
||||
@ -22,10 +19,8 @@ func TestImagePullCredentialBlockerReasonsClassifiesHarborAuth(t *testing.T) {
|
||||
`{"metadata":{"namespace":"veles"},"involvedObject":{"kind":"Pod","name":"veles-frontend"},"type":"Warning","reason":"FailedToRetrieveImagePullSecret","message":"Unable to retrieve some image pull secrets (harbor-regcred); attempting to pull the image may not succeed."},` +
|
||||
`{"involvedObject":{"kind":"Pod","namespace":"logging","name":"oauth2"},"type":"Warning","reason":"Failed","message":"Failed to pull image: lookup registry-1.docker.io: Try again"}` +
|
||||
`]}`
|
||||
events, pods := currentCredentialFixture(t, events)
|
||||
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
|
||||
{match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: events},
|
||||
{match: matchContains("kubectl", "get", "pods", "-A"), out: pods},
|
||||
})
|
||||
|
||||
reasons, err := orch.imagePullCredentialBlockerReasons(context.Background())
|
||||
@ -56,10 +51,8 @@ func TestHealImagePullCredentialSyncRestartsVaultSyncDeployment(t *testing.T) {
|
||||
`]}`
|
||||
restarted := false
|
||||
rolledOut := false
|
||||
events, pods := currentCredentialFixture(t, events)
|
||||
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
|
||||
{match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: events},
|
||||
{match: matchContains("kubectl", "get", "pods", "-A"), out: pods},
|
||||
{match: matchContains("kubectl", "-n", "veles", "get", "deployment", "-o", "json"), out: deployments},
|
||||
{
|
||||
match: func(name string, args []string) bool {
|
||||
@ -92,139 +85,3 @@ func TestHealImagePullCredentialSyncRestartsVaultSyncDeployment(t *testing.T) {
|
||||
t.Fatalf("expected vault-sync deployment restart and rollout wait, restarted=%v rolledOut=%v", restarted, rolledOut)
|
||||
}
|
||||
}
|
||||
|
||||
// currentCredentialFixture gives legacy fixtures current pod identities and states.
|
||||
// Signature: currentCredentialFixture(t *testing.T, raw string) (string, string).
|
||||
// Why: event-only fixtures cannot demonstrate a currently blocked pull.
|
||||
func currentCredentialFixture(t *testing.T, raw string) (string, string) {
|
||||
t.Helper()
|
||||
var events eventList
|
||||
if err := json.Unmarshal([]byte(raw), &events); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
pods := podList{}
|
||||
for i := range events.Items {
|
||||
event := &events.Items[i]
|
||||
event.LastTimestamp = time.Now()
|
||||
event.InvolvedObject.UID = fmt.Sprintf("pod-%d", i)
|
||||
pod := podResource{}
|
||||
pod.Metadata.Namespace = event.InvolvedObject.Namespace
|
||||
if pod.Metadata.Namespace == "" {
|
||||
pod.Metadata.Namespace = event.Metadata.Namespace
|
||||
}
|
||||
pod.Metadata.Name = event.InvolvedObject.Name
|
||||
pod.Metadata.UID = event.InvolvedObject.UID
|
||||
pod.Status.Phase = "Pending"
|
||||
pod.Status.ContainerStatuses = []podContainerStatus{{State: podContainerState{Waiting: &podContainerWaitingState{Reason: "ImagePullBackOff"}}}}
|
||||
pods.Items = append(pods.Items, pod)
|
||||
}
|
||||
e, _ := json.Marshal(events)
|
||||
p, _ := json.Marshal(pods)
|
||||
return string(e), string(p)
|
||||
}
|
||||
|
||||
// TestCredentialRecoveryRequiresCurrentEvidence rejects warnings for healed or replaced pods.
|
||||
// Signature: TestCredentialRecoveryRequiresCurrentEvidence(t *testing.T).
|
||||
// Why: only fresh evidence for the current pod may trigger mutation.
|
||||
func TestCredentialRecoveryRequiresCurrentEvidence(t *testing.T) {
|
||||
for _, scenario := range []string{"current", "init", "stale", "missing-time", "future", "missing-pod", "replaced", "missing-uid", "running", "deleting", "succeeded", "failed", "query-error", "bad-pods", "empty-events"} {
|
||||
t.Run(scenario, func(t *testing.T) {
|
||||
raw, podsRaw := currentCredentialFixture(t, `{"items":[{"type":"Warning","reason":"Failed","message":"unauthorized","involvedObject":{"kind":"Pod","namespace":"apps","name":"test"}}]}`)
|
||||
var events eventList
|
||||
var pods podList
|
||||
_ = json.Unmarshal([]byte(raw), &events)
|
||||
_ = json.Unmarshal([]byte(podsRaw), &pods)
|
||||
switch scenario {
|
||||
case "init":
|
||||
pods.Items[0].Status.InitContainerStatuses = pods.Items[0].Status.ContainerStatuses
|
||||
pods.Items[0].Status.ContainerStatuses = nil
|
||||
case "stale":
|
||||
events.Items[0].LastTimestamp = time.Now().Add(-11 * time.Minute)
|
||||
case "missing-time":
|
||||
events.Items[0].LastTimestamp = time.Time{}
|
||||
case "future":
|
||||
events.Items[0].LastTimestamp = time.Now().Add(time.Hour)
|
||||
case "missing-pod":
|
||||
pods.Items = nil
|
||||
case "replaced":
|
||||
pods.Items[0].Metadata.UID = "replacement"
|
||||
case "missing-uid":
|
||||
events.Items[0].InvolvedObject.UID = ""
|
||||
case "running":
|
||||
pods.Items[0].Status.ContainerStatuses = nil
|
||||
pods.Items[0].Status.Phase = "Running"
|
||||
case "deleting":
|
||||
now := time.Now()
|
||||
pods.Items[0].Metadata.DeletionTimestamp = &now
|
||||
case "succeeded":
|
||||
pods.Items[0].Status.Phase = "Succeeded"
|
||||
case "failed":
|
||||
pods.Items[0].Status.Phase = "Failed"
|
||||
}
|
||||
e, _ := json.Marshal(events)
|
||||
p, _ := json.Marshal(pods)
|
||||
raw, podsRaw = string(e), string(p)
|
||||
var queryErr error
|
||||
if scenario == "query-error" {
|
||||
queryErr = fmt.Errorf("unavailable")
|
||||
}
|
||||
if scenario == "bad-pods" {
|
||||
podsRaw = "{"
|
||||
}
|
||||
if scenario == "empty-events" {
|
||||
raw = ""
|
||||
}
|
||||
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
|
||||
{match: matchContains("kubectl", "get", "events", "-A"), out: raw},
|
||||
{match: matchContains("kubectl", "get", "pods", "-A"), out: podsRaw, err: queryErr},
|
||||
})
|
||||
reasons, err := orch.imagePullCredentialBlockerReasons(context.Background())
|
||||
if scenario == "query-error" || scenario == "bad-pods" {
|
||||
if err == nil {
|
||||
t.Fatal("expected error")
|
||||
}
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
want := 0
|
||||
if scenario == "current" || scenario == "init" {
|
||||
want = 1
|
||||
}
|
||||
if len(reasons) != want {
|
||||
t.Fatalf("got %v, want %d blockers", reasons, want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestCredentialRepairCooldownSurvivesRestart prevents repeated recovery mutations.
|
||||
// Signature: TestCredentialRepairCooldownSurvivesRestart(t *testing.T).
|
||||
// Why: a failing helper must not be restarted every recovery cycle.
|
||||
func TestCredentialRepairCooldownSurvivesRestart(t *testing.T) {
|
||||
restarts := 0
|
||||
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
|
||||
{match: matchContains("kubectl", "get", "deployment"), out: `{"items":[{"metadata":{"name":"app-vault-sync"}}]}`},
|
||||
{match: func(name string, args []string) bool {
|
||||
if matchContains("kubectl", "rollout", "restart")(name, args) {
|
||||
restarts++
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}},
|
||||
{match: matchContains("kubectl", "rollout", "status"), err: fmt.Errorf("rollout timeout")},
|
||||
})
|
||||
if _, err := orch.restartVaultSyncDeployments(context.Background(), "apps"); err == nil {
|
||||
t.Fatal("expected timeout")
|
||||
}
|
||||
replacement := &Orchestrator{cfg: orch.cfg, runner: orch.runner, store: orch.store, log: orch.log, runOverride: orch.runOverride}
|
||||
repaired, err := replacement.restartVaultSyncDeployments(context.Background(), "apps")
|
||||
if err != nil || len(repaired) != 0 || restarts != 1 {
|
||||
t.Fatalf("repeated repair: %v %v %d", repaired, err, restarts)
|
||||
}
|
||||
replacement.store = nil
|
||||
if _, err := replacement.reserveCredentialRepair("apps/other"); err == nil {
|
||||
t.Fatal("missing state must fail closed")
|
||||
}
|
||||
}
|
||||
|
||||
@ -10,6 +10,8 @@ import (
|
||||
"time"
|
||||
)
|
||||
|
||||
const longhornSystemNamespace = "longhorn-system"
|
||||
|
||||
type longhornNodeList struct {
|
||||
Items []longhornNode `json:"items"`
|
||||
}
|
||||
|
||||
@ -231,9 +231,7 @@ func TestImagePullDNSAndCredentialBranches(t *testing.T) {
|
||||
{"type":"Warning","reason":"FailedPull","metadata":{"namespace":"finance"},"involvedObject":{"kind":"Pod","name":"budget"},"message":"unauthorized: authentication required"},
|
||||
{"type":"Warning","reason":"FailedToRetrieveImagePullSecret","involvedObject":{"kind":"Pod","namespace":"sso","name":"keycloak"},"message":"unable to retrieve some image pull secrets"}
|
||||
]}`
|
||||
eventJSON, podsJSON := currentCredentialFixture(t, eventJSON)
|
||||
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
|
||||
{match: matchContains("kubectl", "get", "pods", "-A"), out: podsJSON},
|
||||
{match: matchContains("kubectl", "get", "events", "-A"), out: eventJSON},
|
||||
})
|
||||
dnsReasons, err := orch.imagePullDNSBlockerReasons(context.Background())
|
||||
|
||||
@ -107,14 +107,12 @@ func TestPostStartAutoHealRequestsReconcileAfterImagePullRepair(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
eventsJSON := `{"items":[{"type":"Warning","reason":"FailedPull","message":"unauthorized: authentication required","involvedObject":{"kind":"Pod","namespace":"apps","name":"web"}}]}`
|
||||
deploymentsJSON := `{"items":[{"metadata":{"name":"vault-sync","labels":{"app":"vault-sync"}}}]}`
|
||||
eventsJSON, credentialPods := currentCredentialFixture(t, eventsJSON)
|
||||
sawFluxSourceReconcile := false
|
||||
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
|
||||
{match: matchContains("kubectl", "-n", "longhorn-system", "get", "nodes.longhorn.io", "-o", "json"), out: `{"items":[]}`},
|
||||
{match: matchContains("kubectl", "get", "nodes", "-o", "json"), out: `{"items":[]}`},
|
||||
{match: matchContains("kubectl", "-n", "vault", "get", "pod", "vault-0"), out: "Pending"},
|
||||
{match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: eventsJSON},
|
||||
{match: matchContains("kubectl", "get", "pods", "-A", "-o", "json"), out: credentialPods},
|
||||
{match: matchContains("kubectl", "-n", "apps", "get", "deployment", "-o", "json"), out: deploymentsJSON},
|
||||
{match: matchContains("kubectl", "-n", "apps", "rollout", "restart", "deployment", "vault-sync"), out: ""},
|
||||
{match: matchContains("kubectl", "-n", "apps", "rollout", "status", "deployment/vault-sync"), out: ""},
|
||||
|
||||
@ -187,7 +187,6 @@ type eventResource struct {
|
||||
CreationTimestamp time.Time `json:"creationTimestamp"`
|
||||
} `json:"metadata"`
|
||||
InvolvedObject struct {
|
||||
UID string `json:"uid"`
|
||||
Kind string `json:"kind"`
|
||||
Namespace string `json:"namespace"`
|
||||
Name string `json:"name"`
|
||||
@ -243,7 +242,6 @@ type podList struct {
|
||||
|
||||
type podResource struct {
|
||||
Metadata struct {
|
||||
UID string `json:"uid"`
|
||||
Namespace string `json:"namespace"`
|
||||
Name string `json:"name"`
|
||||
Labels map[string]string `json:"labels"`
|
||||
|
||||
@ -135,10 +135,6 @@ func isTimeSynced(raw string) bool {
|
||||
// Signature: (o *Orchestrator) preflightExternalDatastore(ctx context.Context) error.
|
||||
// Why: keeps behavior explicit so startup/shutdown workflows remain maintainable as services evolve.
|
||||
func (o *Orchestrator) preflightExternalDatastore(ctx context.Context) error {
|
||||
// Cancellation must win even when a local datastore is already reachable.
|
||||
if err := ctx.Err(); err != nil {
|
||||
return err
|
||||
}
|
||||
if len(o.cfg.ControlPlanes) == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
@ -217,11 +217,6 @@ func (d *Daemon) Run(ctx context.Context) error {
|
||||
// like a later Vault reseal or stale dead-node deletions without waiting for a
|
||||
// fresh bootstrap run.
|
||||
func (d *Daemon) maybeRunPostStartAutoHeal(ctx context.Context, lastRun *time.Time, anyOnBattery bool) {
|
||||
// Shared cluster repairs have one owner; peers still monitor their UPS and
|
||||
// forward shutdown intent through the existing coordination path.
|
||||
if d.cfg.Coordination.Role == "peer" {
|
||||
return
|
||||
}
|
||||
interval := time.Duration(d.cfg.Startup.PostStartAutoHealSeconds) * time.Second
|
||||
if interval <= 0 || anyOnBattery {
|
||||
return
|
||||
|
||||
@ -1,14 +1,7 @@
|
||||
package service
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"log"
|
||||
"strings"
|
||||
|
||||
"scm.bstein.dev/bstein/ananke/internal/cluster"
|
||||
"scm.bstein.dev/bstein/ananke/internal/execx"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@ -33,14 +26,8 @@ func TestDaemonMaybeRunPostStartAutoHeal(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
d.cfg.Coordination.Role = "peer"
|
||||
var last time.Time
|
||||
d.maybeRunPostStartAutoHeal(context.Background(), &last, false)
|
||||
if calls != 0 || !last.IsZero() {
|
||||
t.Fatal("peer must not mutate shared cluster recovery state")
|
||||
}
|
||||
d.cfg.Coordination.Role = "coordinator"
|
||||
d.maybeRunPostStartAutoHeal(context.Background(), &last, false)
|
||||
if calls != 1 {
|
||||
t.Fatalf("expected first auto-heal invocation, got %d", calls)
|
||||
}
|
||||
@ -62,28 +49,3 @@ func TestDaemonMaybeRunPostStartAutoHeal(t *testing.T) {
|
||||
t.Fatalf("expected second allowed auto-heal call, got %d", calls)
|
||||
}
|
||||
}
|
||||
|
||||
// TestDaemonRecoveryFailureDoesNotStopMonitoring exercises optional repair isolation.
|
||||
// Signature: TestDaemonRecoveryFailureDoesNotStopMonitoring(t *testing.T).
|
||||
// Why: a missing or failed repair helper must not terminate UPS monitoring.
|
||||
func TestDaemonRecoveryFailureDoesNotStopMonitoring(t *testing.T) {
|
||||
var messages bytes.Buffer
|
||||
d := &Daemon{cfg: config.Config{Startup: config.Startup{PostStartAutoHealSeconds: 60}}, log: log.New(&messages, "", 0)}
|
||||
var last time.Time
|
||||
d.maybeRunPostStartAutoHeal(context.Background(), &last, false)
|
||||
if !last.IsZero() {
|
||||
t.Fatal("missing helper must not consume an interval")
|
||||
}
|
||||
if err := d.runPostStartAutoHeal(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
d.orch = cluster.New(config.Config{}, &execx.Runner{DryRun: true}, nil, d.log)
|
||||
if err := d.runPostStartAutoHeal(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
d.postStartAutoHealOverride = func(context.Context) error { return errors.New("synthetic repair failure") }
|
||||
d.maybeRunPostStartAutoHeal(context.Background(), &last, false)
|
||||
if last.IsZero() || !strings.Contains(messages.String(), "synthetic repair failure") {
|
||||
t.Fatal("failed repair must be logged and rate limited")
|
||||
}
|
||||
}
|
||||
|
||||
@ -13,7 +13,7 @@ HOST_SHORT="$(hostname -s 2>/dev/null || hostname)"
|
||||
LOG_FILE="${ANANKE_UPDATE_LOG_FILE:-/var/log/ananke/update.log}"
|
||||
STATE_FILE="${ANANKE_UPDATE_STATE_FILE:-/var/lib/ananke/update-last.env}"
|
||||
LOCK_FILE="${ANANKE_UPDATE_LOCK_FILE:-/var/lock/ananke-update.lock}"
|
||||
ALLOW_QUALITY_FALLBACK="${ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK:-0}"
|
||||
ALLOW_QUALITY_FALLBACK="${ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK:-1}"
|
||||
QUALITY_GATE_MODE="${ANANKE_ENFORCE_QUALITY_GATE:-1}"
|
||||
|
||||
mkdir -p "$(dirname "${LOG_FILE}")" "$(dirname "${STATE_FILE}")" "$(dirname "${LOCK_FILE}")"
|
||||
@ -103,8 +103,6 @@ if "${REPO_DIR}/scripts/install.sh"; then
|
||||
write_state "ok" "install-success" "${CURRENT_FROM_REV}" "${TARGET_TO_REV}"
|
||||
echo "[self-update] completed successfully"
|
||||
exit 0
|
||||
else
|
||||
INSTALL_EXIT_CODE=$?
|
||||
fi
|
||||
|
||||
if [[ "${ALLOW_QUALITY_FALLBACK}" == "1" || "${ALLOW_QUALITY_FALLBACK}" == "true" ]]; then
|
||||
@ -115,8 +113,3 @@ if [[ "${ALLOW_QUALITY_FALLBACK}" == "1" || "${ALLOW_QUALITY_FALLBACK}" == "true
|
||||
echo "[self-update] completed via fallback mode"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# A failed command used as an if condition does not trigger errexit or ERR.
|
||||
write_state "failed" "install-failed;rc=${INSTALL_EXIT_CODE}" "${CURRENT_FROM_REV}" "${TARGET_TO_REV}"
|
||||
echo "[self-update] installer failed rc=${INSTALL_EXIT_CODE}; fallback disabled"
|
||||
exit "${INSTALL_EXIT_CODE}"
|
||||
|
||||
@ -14,7 +14,6 @@ SYSTEMD_DIR="/etc/systemd/system"
|
||||
LIB_DIR="/usr/local/lib/ananke"
|
||||
START_NOW=1
|
||||
INSTALL_DEPS=1
|
||||
BINARY_ONLY=0
|
||||
ENABLE_BOOTSTRAP="${ANANKE_ENABLE_BOOTSTRAP:-auto}"
|
||||
MANAGE_NUT="${ANANKE_MANAGE_NUT:-1}"
|
||||
NUT_UPS_NAME="${ANANKE_NUT_UPS_NAME:-}"
|
||||
@ -28,10 +27,6 @@ ENFORCE_QUALITY_GATE="${ANANKE_ENFORCE_QUALITY_GATE:-1}"
|
||||
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
--binary-only)
|
||||
BINARY_ONLY=1
|
||||
shift
|
||||
;;
|
||||
--no-start)
|
||||
START_NOW=0
|
||||
shift
|
||||
@ -52,14 +47,8 @@ source "${REPO_DIR}/scripts/install-host-bootstrap.sh"
|
||||
source "${REPO_DIR}/scripts/install-legacy-migration.sh"
|
||||
source "${REPO_DIR}/scripts/install-artifacts.sh"
|
||||
|
||||
if [[ "${BINARY_ONLY}" == "1" && ! -f "${BIN_DIR}/ananke" ]]; then
|
||||
echo "[install] binary-only requires an existing installation" >&2
|
||||
exit 1
|
||||
fi
|
||||
ensure_dependencies
|
||||
if [[ "${BINARY_ONLY}" != "1" ]]; then
|
||||
migrate_legacy_hecate_install
|
||||
fi
|
||||
migrate_legacy_hecate_install
|
||||
|
||||
if [[ "${ENFORCE_QUALITY_GATE}" == "1" ]]; then
|
||||
echo "[install] running quality gate"
|
||||
@ -80,26 +69,8 @@ go build -o dist/ananke "${BUILD_TARGET}"
|
||||
|
||||
echo "[install] installing binary"
|
||||
install -d -m 0755 "${BIN_DIR}"
|
||||
if [[ "${BINARY_ONLY}" == "1" && -f "${BIN_DIR}/ananke" ]]; then
|
||||
install -d -m 0700 "${LIB_DIR}/rollback"
|
||||
install -m 0700 "${BIN_DIR}/ananke" "${LIB_DIR}/rollback/ananke.previous"
|
||||
fi
|
||||
install -m 0755 dist/ananke "${BIN_DIR}/ananke"
|
||||
|
||||
# Targeted recovery fixes must not overwrite host config, NUT, or bootstrap units.
|
||||
if [[ "${BINARY_ONLY}" == "1" ]]; then
|
||||
if [[ "${START_NOW}" == "1" ]]; then
|
||||
if ! systemctl restart ananke.service || ! systemctl is-active --quiet ananke.service; then
|
||||
echo "[install] service failed; restoring previous binary" >&2
|
||||
install -m 0755 "${LIB_DIR}/rollback/ananke.previous" "${BIN_DIR}/ananke"
|
||||
systemctl restart ananke.service
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
echo "[install] binary-only update complete; host configuration unchanged"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "[install] installing config + state dirs"
|
||||
install -d -m 0750 "${CONF_DIR}"
|
||||
install -d -m 0750 "${STATE_DIR}"
|
||||
|
||||
@ -157,11 +157,6 @@ quality_gate_finalize() {
|
||||
trap 'quality_gate_finalize $?' EXIT
|
||||
|
||||
cd "${REPO_DIR}"
|
||||
# Pin the resolved toolchain for child coverage commands. Auto-switching from an
|
||||
# older host Go can otherwise look for the removed covdata tool (Go issue 75031).
|
||||
if [[ "$(go env GOTOOLCHAIN)" == "auto" ]]; then
|
||||
export GOTOOLCHAIN="$(go env GOVERSION)"
|
||||
fi
|
||||
mkdir -p "${BUILD_DIR}"
|
||||
rm -f "${COVERAGE_PROFILE}" "${COVERAGE_PERCENT_FILE}"
|
||||
printf 'failed\n' > "${BUILD_DIR}/docs-naming.status"
|
||||
|
||||
@ -96,7 +96,7 @@ func TestHookVaultLifecycleBranchMatrix(t *testing.T) {
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("workload-ready-no-value-and-ensure-cancel", func(t *testing.T) {
|
||||
t.Run("workload-ready-no-value-and-ensure-error", func(t *testing.T) {
|
||||
cfg := lifecycleConfig(t)
|
||||
runNoValue := func(ctx context.Context, timeout time.Duration, name string, args ...string) (string, error) {
|
||||
command := name + " " + strings.Join(args, " ")
|
||||
@ -111,8 +111,6 @@ func TestHookVaultLifecycleBranchMatrix(t *testing.T) {
|
||||
t.Fatalf("expected no-value readiness branch, ready=%v err=%v", ready, err)
|
||||
}
|
||||
|
||||
waitCtx, cancelWait := context.WithCancel(context.Background())
|
||||
defer cancelWait()
|
||||
runEnsureErr := func(ctx context.Context, timeout time.Duration, name string, args ...string) (string, error) {
|
||||
command := name + " " + strings.Join(args, " ")
|
||||
switch {
|
||||
@ -121,15 +119,14 @@ func TestHookVaultLifecycleBranchMatrix(t *testing.T) {
|
||||
case name == "kubectl" && strings.Contains(command, "get pods -o custom-columns"):
|
||||
return "", nil
|
||||
case name == "kubectl" && strings.Contains(command, "rollout status statefulset/victoria-metrics-single-server"):
|
||||
cancelWait()
|
||||
return "", errors.New("rollout failed")
|
||||
default:
|
||||
return lifecycleDispatcher(&commandRecorder{})(ctx, timeout, name, args...)
|
||||
}
|
||||
}
|
||||
orchEnsureErr, _ := newHookOrchestratorWithRunnerMode(t, cfg, false, runEnsureErr, runEnsureErr)
|
||||
if err := orchEnsureErr.TestHookEnsureCriticalStartupWorkloads(waitCtx); !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("expected ensureCriticalStartupWorkloads cancellation after failed rollout, got %v", err)
|
||||
if err := orchEnsureErr.TestHookEnsureCriticalStartupWorkloads(context.Background()); err == nil || !strings.Contains(err.Error(), "rollout failed") {
|
||||
t.Fatalf("expected ensureCriticalStartupWorkloads wait error branch, got %v", err)
|
||||
}
|
||||
})
|
||||
|
||||
@ -362,8 +359,6 @@ func TestHookVaultLifecycleBranchMatrix(t *testing.T) {
|
||||
cfgFail.Startup.RequireFluxHealth = false
|
||||
cfgFail.Startup.RequireWorkloadConvergence = false
|
||||
cfgFail.Startup.RequireCriticalServiceEndpoints = true
|
||||
cfgFail.Startup.CriticalServiceEndpointWaitSec = 1
|
||||
cfgFail.Startup.CriticalServiceEndpointPollSec = 1
|
||||
run := func(ctx context.Context, timeout time.Duration, name string, args ...string) (string, error) {
|
||||
command := name + " " + strings.Join(args, " ")
|
||||
if name == "kubectl" && strings.Contains(command, "get endpoints") {
|
||||
|
||||
@ -1,61 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Exercise updater exit/status behavior with a local repository and fake installer.
|
||||
# Run as root; all writes, Git configuration and installer calls stay in a temp dir.
|
||||
set -euo pipefail
|
||||
|
||||
if [[ "${EUID}" -ne 0 ]]; then
|
||||
echo "Run this isolated updater regression as root" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)"
|
||||
FIXTURE="$(mktemp -d /tmp/ananke-update-test.XXXXXX)"
|
||||
trap 'rm -rf "${FIXTURE}"' EXIT
|
||||
export GIT_CONFIG_GLOBAL="${FIXTURE}/gitconfig"
|
||||
export GIT_CONFIG_NOSYSTEM=1
|
||||
git config --global user.name 'Updater regression'
|
||||
git config --global user.email 'updater-test@example.invalid'
|
||||
mkdir -p "${FIXTURE}/origin/scripts"
|
||||
git init -q -b main "${FIXTURE}/origin"
|
||||
cat > "${FIXTURE}/origin/scripts/install.sh" <<'INSTALLER'
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
printf '%s\n' "${ANANKE_ENFORCE_QUALITY_GATE}" >> "${FIXTURE_CALLS}"
|
||||
if [[ "${ANANKE_ENFORCE_QUALITY_GATE}" == 1 ]]; then
|
||||
exit "${FIXTURE_INSTALL_RC}"
|
||||
fi
|
||||
INSTALLER
|
||||
chmod 0755 "${FIXTURE}/origin/scripts/install.sh"
|
||||
git -C "${FIXTURE}/origin" add scripts/install.sh
|
||||
git -C "${FIXTURE}/origin" commit -qm fixture
|
||||
|
||||
# Inputs: case name, installer status, fallback policy, expected exit/state/calls.
|
||||
# Output: assertion failure or one concise pass line; no host services are invoked.
|
||||
check_update() {
|
||||
local name="$1" install_rc="$2" fallback="$3" expected_rc="$4"
|
||||
local expected_status="$5" expected_calls="$6" actual_rc=0
|
||||
local directory="${FIXTURE}/${name}"
|
||||
mkdir -p "${directory}"
|
||||
env ANANKE_REPO_URL="${FIXTURE}/origin" \
|
||||
ANANKE_REPO_DIR="${directory}/repo" \
|
||||
ANANKE_REPO_BRANCH=main \
|
||||
ANANKE_UPDATE_LOG_FILE="${directory}/update.log" \
|
||||
ANANKE_UPDATE_STATE_FILE="${directory}/state" \
|
||||
ANANKE_UPDATE_LOCK_FILE="${directory}/lock" \
|
||||
ANANKE_UPDATE_ALLOW_QUALITY_FALLBACK="${fallback}" \
|
||||
ANANKE_ENFORCE_QUALITY_GATE=1 \
|
||||
FIXTURE_CALLS="${directory}/calls" FIXTURE_INSTALL_RC="${install_rc}" \
|
||||
bash "${REPO_ROOT}/scripts/ananke-self-update.sh" > "${directory}/output" 2>&1 || actual_rc=$?
|
||||
[[ "${actual_rc}" == "${expected_rc}" ]]
|
||||
grep -qx "status=${expected_status}" "${directory}/state"
|
||||
[[ "$(wc -l < "${directory}/calls")" -eq "${expected_calls}" ]]
|
||||
if [[ "${expected_status}" == failed ]]; then
|
||||
grep -qx "detail=install-failed;rc=${install_rc}" "${directory}/state"
|
||||
fi
|
||||
printf 'PASS %s: exit=%s state=%s calls=%s\n' \
|
||||
"${name}" "${actual_rc}" "${expected_status}" "${expected_calls}"
|
||||
}
|
||||
|
||||
check_update strict_failure 42 0 42 failed 1
|
||||
check_update explicit_legacy_fallback 42 1 0 degraded-ok 2
|
||||
check_update strict_success 0 0 0 ok 1
|
||||
Loading…
x
Reference in New Issue
Block a user