ananke/internal/cluster/orchestrator_image_pull_credentials_test.go

231 lines
9.6 KiB
Go

package cluster
import (
"context"
"encoding/json"
"fmt"
"strings"
"testing"
"time"
"scm.bstein.dev/bstein/ananke/internal/config"
)
// TestImagePullCredentialBlockerReasonsClassifiesHarborAuth runs one
// orchestration or CLI step.
// Signature: TestImagePullCredentialBlockerReasonsClassifiesHarborAuth(t *testing.T).
// Why: Harbor auth failures should drive secret-sync repair instead of pod
// recycling or DNS remediation.
func TestImagePullCredentialBlockerReasonsClassifiesHarborAuth(t *testing.T) {
events := `{"items":[` +
`{"involvedObject":{"kind":"Pod","namespace":"veles","name":"veles-backend"},"type":"Warning","reason":"Failed","message":"Failed to pull image \"registry.bstein.dev/veles/veles-backend:0.5.25\": no basic auth credentials"},` +
`{"metadata":{"namespace":"veles"},"involvedObject":{"kind":"Pod","name":"veles-frontend"},"type":"Warning","reason":"FailedToRetrieveImagePullSecret","message":"Unable to retrieve some image pull secrets (harbor-regcred); attempting to pull the image may not succeed."},` +
`{"involvedObject":{"kind":"Pod","namespace":"logging","name":"oauth2"},"type":"Warning","reason":"Failed","message":"Failed to pull image: lookup registry-1.docker.io: Try again"}` +
`]}`
events, pods := currentCredentialFixture(t, events)
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: events},
{match: matchContains("kubectl", "get", "pods", "-A"), out: pods},
})
reasons, err := orch.imagePullCredentialBlockerReasons(context.Background())
if err != nil {
t.Fatalf("imagePullCredentialBlockerReasons failed: %v", err)
}
if got := reasons["veles/veles-backend"]; got != "ImagePullCredentialBlocker:no-basic-auth" {
t.Fatalf("unexpected backend credential reason %q", got)
}
if got := reasons["veles/veles-frontend"]; got != "ImagePullCredentialBlocker:missing-pull-secret" {
t.Fatalf("unexpected frontend credential reason %q", got)
}
if _, ok := reasons["logging/oauth2"]; ok {
t.Fatalf("DNS image-pull blocker must not be classified as credential failure")
}
}
// TestHealImagePullCredentialSyncRestartsVaultSyncDeployment runs one
// orchestration or CLI step.
// Signature: TestHealImagePullCredentialSyncRestartsVaultSyncDeployment(t *testing.T).
// Why: a namespace with registry credential blockers and a Vault CSI sync helper
// should get that sync helper rolled instead of application pods deleted.
func TestHealImagePullCredentialSyncRestartsVaultSyncDeployment(t *testing.T) {
events := `{"items":[{"involvedObject":{"kind":"Pod","namespace":"veles","name":"veles-backend"},"type":"Warning","reason":"Failed","message":"Failed to pull image \"registry.bstein.dev/veles/veles-backend:0.5.25\": no basic auth credentials"}]}`
deployments := `{"items":[` +
`{"metadata":{"name":"veles-backend","labels":{"app.kubernetes.io/name":"veles-backend"}}},` +
`{"metadata":{"name":"veles-vault-sync","labels":{"app.kubernetes.io/component":"vault-sync"}}}` +
`]}`
restarted := false
rolledOut := false
events, pods := currentCredentialFixture(t, events)
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "events", "-A", "-o", "json"), out: events},
{match: matchContains("kubectl", "get", "pods", "-A"), out: pods},
{match: matchContains("kubectl", "-n", "veles", "get", "deployment", "-o", "json"), out: deployments},
{
match: func(name string, args []string) bool {
if !matchContains("kubectl", "-n", "veles", "rollout", "restart", "deployment", "veles-vault-sync")(name, args) {
return false
}
restarted = true
return true
},
},
{
match: func(name string, args []string) bool {
if !matchContains("kubectl", "-n", "veles", "rollout", "status", "deployment/veles-vault-sync", "--timeout=60s")(name, args) {
return false
}
rolledOut = true
return true
},
},
})
repaired, err := orch.healImagePullCredentialSync(context.Background())
if err != nil {
t.Fatalf("healImagePullCredentialSync failed: %v", err)
}
if strings.Join(repaired, ",") != "veles/deployment/veles-vault-sync" {
t.Fatalf("unexpected image-pull credential repair list: %#v", repaired)
}
if !restarted || !rolledOut {
t.Fatalf("expected vault-sync deployment restart and rollout wait, restarted=%v rolledOut=%v", restarted, rolledOut)
}
}
// currentCredentialFixture gives legacy fixtures current pod identities and states.
// Signature: currentCredentialFixture(t *testing.T, raw string) (string, string).
// Why: event-only fixtures cannot demonstrate a currently blocked pull.
func currentCredentialFixture(t *testing.T, raw string) (string, string) {
t.Helper()
var events eventList
if err := json.Unmarshal([]byte(raw), &events); err != nil {
t.Fatal(err)
}
pods := podList{}
for i := range events.Items {
event := &events.Items[i]
event.LastTimestamp = time.Now()
event.InvolvedObject.UID = fmt.Sprintf("pod-%d", i)
pod := podResource{}
pod.Metadata.Namespace = event.InvolvedObject.Namespace
if pod.Metadata.Namespace == "" {
pod.Metadata.Namespace = event.Metadata.Namespace
}
pod.Metadata.Name = event.InvolvedObject.Name
pod.Metadata.UID = event.InvolvedObject.UID
pod.Status.Phase = "Pending"
pod.Status.ContainerStatuses = []podContainerStatus{{State: podContainerState{Waiting: &podContainerWaitingState{Reason: "ImagePullBackOff"}}}}
pods.Items = append(pods.Items, pod)
}
e, _ := json.Marshal(events)
p, _ := json.Marshal(pods)
return string(e), string(p)
}
// TestCredentialRecoveryRequiresCurrentEvidence rejects warnings for healed or replaced pods.
// Signature: TestCredentialRecoveryRequiresCurrentEvidence(t *testing.T).
// Why: only fresh evidence for the current pod may trigger mutation.
func TestCredentialRecoveryRequiresCurrentEvidence(t *testing.T) {
for _, scenario := range []string{"current", "init", "stale", "missing-time", "future", "missing-pod", "replaced", "missing-uid", "running", "deleting", "succeeded", "failed", "query-error", "bad-pods", "empty-events"} {
t.Run(scenario, func(t *testing.T) {
raw, podsRaw := currentCredentialFixture(t, `{"items":[{"type":"Warning","reason":"Failed","message":"unauthorized","involvedObject":{"kind":"Pod","namespace":"apps","name":"test"}}]}`)
var events eventList
var pods podList
_ = json.Unmarshal([]byte(raw), &events)
_ = json.Unmarshal([]byte(podsRaw), &pods)
switch scenario {
case "init":
pods.Items[0].Status.InitContainerStatuses = pods.Items[0].Status.ContainerStatuses
pods.Items[0].Status.ContainerStatuses = nil
case "stale":
events.Items[0].LastTimestamp = time.Now().Add(-11 * time.Minute)
case "missing-time":
events.Items[0].LastTimestamp = time.Time{}
case "future":
events.Items[0].LastTimestamp = time.Now().Add(time.Hour)
case "missing-pod":
pods.Items = nil
case "replaced":
pods.Items[0].Metadata.UID = "replacement"
case "missing-uid":
events.Items[0].InvolvedObject.UID = ""
case "running":
pods.Items[0].Status.ContainerStatuses = nil
pods.Items[0].Status.Phase = "Running"
case "deleting":
now := time.Now()
pods.Items[0].Metadata.DeletionTimestamp = &now
case "succeeded":
pods.Items[0].Status.Phase = "Succeeded"
case "failed":
pods.Items[0].Status.Phase = "Failed"
}
e, _ := json.Marshal(events)
p, _ := json.Marshal(pods)
raw, podsRaw = string(e), string(p)
var queryErr error
if scenario == "query-error" {
queryErr = fmt.Errorf("unavailable")
}
if scenario == "bad-pods" {
podsRaw = "{"
}
if scenario == "empty-events" {
raw = ""
}
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "events", "-A"), out: raw},
{match: matchContains("kubectl", "get", "pods", "-A"), out: podsRaw, err: queryErr},
})
reasons, err := orch.imagePullCredentialBlockerReasons(context.Background())
if scenario == "query-error" || scenario == "bad-pods" {
if err == nil {
t.Fatal("expected error")
}
return
}
if err != nil {
t.Fatal(err)
}
want := 0
if scenario == "current" || scenario == "init" {
want = 1
}
if len(reasons) != want {
t.Fatalf("got %v, want %d blockers", reasons, want)
}
})
}
}
// TestCredentialRepairCooldownSurvivesRestart prevents repeated recovery mutations.
// Signature: TestCredentialRepairCooldownSurvivesRestart(t *testing.T).
// Why: a failing helper must not be restarted every recovery cycle.
func TestCredentialRepairCooldownSurvivesRestart(t *testing.T) {
restarts := 0
orch := buildOrchestratorWithStubs(t, config.Config{}, []commandStub{
{match: matchContains("kubectl", "get", "deployment"), out: `{"items":[{"metadata":{"name":"app-vault-sync"}}]}`},
{match: func(name string, args []string) bool {
if matchContains("kubectl", "rollout", "restart")(name, args) {
restarts++
return true
}
return false
}},
{match: matchContains("kubectl", "rollout", "status"), err: fmt.Errorf("rollout timeout")},
})
if _, err := orch.restartVaultSyncDeployments(context.Background(), "apps"); err == nil {
t.Fatal("expected timeout")
}
replacement := &Orchestrator{cfg: orch.cfg, runner: orch.runner, store: orch.store, log: orch.log, runOverride: orch.runOverride}
repaired, err := replacement.restartVaultSyncDeployments(context.Background(), "apps")
if err != nil || len(repaired) != 0 || restarts != 1 {
t.Fatalf("repeated repair: %v %v %d", repaired, err, restarts)
}
replacement.store = nil
if _, err := replacement.reserveCredentialRepair("apps/other"); err == nil {
t.Fatal("missing state must fail closed")
}
}