From b0bfb6a9489b4321817f09f9848d21fc5b93b1e5 Mon Sep 17 00:00:00 2001 From: jenkins Date: Sat, 3 Oct 2026 01:34:57 -0500 Subject: [PATCH] backup: schedule application recovery copies and document verified repairs --- docs/CLUSTER_OPERATIONS.md | 27 ++++++++++ docs/CLUSTER_STABILIZATION.md | 41 ++++++++++++--- infrastructure/host-backup/NOTES.md | 39 +++++++++++++++ .../atlas-application-postgres-backup.service | 15 ++++++ .../atlas-application-postgres-backup.timer | 12 +++++ scripts/ops/postgres_application_backup.sh | 10 ++++ scripts/ops/postgres_dump_verify.sh | 31 ++++++++++++ .../monitoring/grafana-alerting-config.yaml | 50 +++++++++++++++++++ 8 files changed, 219 insertions(+), 6 deletions(-) create mode 100644 infrastructure/host-backup/atlas-application-postgres-backup.service create mode 100644 infrastructure/host-backup/atlas-application-postgres-backup.timer create mode 100755 scripts/ops/postgres_dump_verify.sh diff --git a/docs/CLUSTER_OPERATIONS.md b/docs/CLUSTER_OPERATIONS.md index 9611bc9f..9ad6fc7c 100644 --- a/docs/CLUSTER_OPERATIONS.md +++ b/docs/CLUSTER_OPERATIONS.md @@ -148,3 +148,30 @@ to make a coverage dashboard greener. This provides a practical route to ownership; it cannot promise that every hardware, storage or database disaster will be solvable in one day. + +## Physical checks and return to service + +For an offline Pi, record the board identity, supply model and its 5 V rating, +power/activity LEDs, HDMI boot result, Ethernet link and current router address. +Preserve the original SD/USB media. Test with separate known-good spare boot media +before concluding that a board has failed; do not clone a live node identity by +moving another cluster node's boot card into it. + +Pi 5 power must be evaluated at its 5 V output mode, not a charger's headline +wattage. A controlled supply-and-cable substitution isolates the power path. +Record any inline adapters and USB loads. Do not unplug runtime USB storage on +a running node. Titan-11 carries important services; arrange workload movement +before disconnecting it. See the official Raspberry Pi power documentation for +5 V / 5 A capability and cable-loss requirements. + +A repaired node returns only after stable power, clean storage/kernel logs, +reliable networking and representative-load testing. Observe it before restoring +normal placement. Titan-14/18 quarantine is explicitly tracked in +`infrastructure/core/node-maintenance.yaml`; `prune: disabled` protects the Node +objects from deletion when that maintenance declaration is eventually removed. +Clear `spec.unschedulable` through Git after validation, then reconcile core. + +Container runtime media and Longhorn data disks are separate. A Pi with terabytes +of healthy application storage can still stall because containerd, image unpacking +and logs live on a small USB flash device. Include the runtime disk in hardware +repair and capacity checks. diff --git a/docs/CLUSTER_STABILIZATION.md b/docs/CLUSTER_STABILIZATION.md index 785172ec..527a53dd 100644 --- a/docs/CLUSTER_STABILIZATION.md +++ b/docs/CLUSTER_STABILIZATION.md @@ -15,6 +15,12 @@ Use [Cluster operations](CLUSTER_OPERATIONS.md) for the ordinary operator path. | Removed eviction for historical restart counts | Descheduler manifest validates and Flux applied the change | Revert fa9251ae | | Restored GitOps UI Deployment | Helm drift correction recreated weave-gitops; Deployment 1/1 Ready | Revert 6c9398ea to disable ongoing drift correction; this does not remove recovered resources | | Aligned Flux definition ownership | Removed creation-only policy and adopted already-active service state; no service paths or source refs changed | Revert the focused Flux commits; review suspension fields before doing so | +| Backup freshness alerts | Both datastore-copy timestamps are scraped; native Grafana provisioning reload returned HTTP 200 | Revert the focused monitoring change; backups continue independently | +| Quarantined stalled runtime media | Titan-14 and Titan-18 are SchedulingDisabled via `infrastructure/core/node-maintenance.yaml`; no forced storage detach | After repair, change `unschedulable` to false in Git and verify before removing the prune-disabled Node declaration | +| Bounded CI concurrency | Jenkins controller Ready after applying the two-agent cap; existing agents reconnected | Revert 440244f3 and reconcile Jenkins; allow a controller restart | +| Spread Vault injectors | Two healthy replicas on Titan-08 and Titan-22; native anti-affinity and minAvailable=1 PDB | Revert 5f3f3184 | +| Protected application databases | Nineteen logical dumps and globals completed with checksums on Titan-0b; isolated Gitea restore passed in six seconds with 111 tables | Keep copies; stop the daily timer if needed | +| Application PostgreSQL resource/probe repair | Pinned current PostgreSQL 15.18 image, 1 GiB request / 2 GiB limit, native startup/readiness; 2/2 Ready on Titan-17 and healthy attached volume | Revert 103e2064 only after checking destination capacity; another database restart is required | | Protected local-only data policy | Hermes PVCs excluded from Soteria's cloud policy through Flux | Do not broaden this policy without reviewing data authorization | Host backup implementation is in commit 80aff498 and @@ -29,16 +35,18 @@ replay or starting K3s. A complete control-plane disaster drill remains outstand `nodeAffinity` setting. Pi5 then Pi4 are preferred; titan-22 is the explicit last-resort CPU destination. The original Helm operation must converge before the corrected generation can be verified. -* Shared application PostgreSQL: explicit 1 GiB request / 2 GiB limit, startup - and readiness checks, bounded exporter resources. Deployment waits for the - pre-change database recovery copies to finish. * Soteria: source fix 7018d4c in the Soteria repository allows live RWO Longhorn snapshots while preserving the restic mount guard. Manual Longhorn requests also enforce exclusions. All Go package tests pass; normal image publication and a representative backup verification remain required. -* Jenkins: a two-agent concurrency cap is prepared. Its existing configuration - hash triggers a controller rollout, so apply it with ongoing builds accounted - for rather than pretending it is a harmless live-only reload. +* Native application-backup schedule and its freshness alert are being installed + after the initial successful copy and restore check. +* Soteria CI: the coverage report was generated after Sonar analysis, causing a + false zero-coverage gate despite 96.4% measured test coverage. Commit f167d7d + orders tests before analysis and awaits the Sonar gate; Jenkins declarative + validation passed. Release build remains pending; do not claim the runtime + backup defect is repaired until the new image and a real eligible backup pass. + ## Confirmed problems needing further work @@ -51,6 +59,7 @@ replay or starting K3s. A complete control-plane disaster drill remains outstand | titan-08 stale iSCSI session | Pushgateway engine could not log out an obsolete target; replica data remained usable on another host | Repair during a controlled storage maintenance window; do not mass-restart instance managers holding healthy volumes | | Hermes tenant-2 shared workspace | Existing faulted/detached volume, repeated recovery attempts | Preserve replica/recovery evidence and perform component-supported repair; no source content in routine logs | | Backup coverage and restore proof | Old Soteria coverage is insufficient; repairing its scheduler does not instantly create all backups | Confirm eligible data, completed backup objects, achievable schedule and representative restores service by service | +| Database collation drift | Native dumps reported stored collation 2.36 versus runtime 2.41 for several databases | Plan index rebuilds with compatible locale settings before refreshing version metadata; do not merely suppress warnings | | Supported software baseline | Ubuntu 24.10 on titan-db and older Kubernetes cohorts are out of support | Backed-up, staged host/K3s/Longhorn upgrades, one compatible cohort at a time | | Failure capacity | Pi pool is heavily reserved; unused x86 capacity has explicit simulation/GPU roles | Recalculate compatible N+1 capacity after repairs; do not silently take reserved GPUs or simulation capacity | @@ -62,3 +71,23 @@ restore coverage remain visible until addressed. Validate application behavior, backup freshness, actual restore results, and 7-14 days without the recurring failures. No automatic real-suite inference jobs or outage drills are part of this maintenance work. + +## Operational lessons from this maintenance + +The application database move took approximately nine minutes because both +pinned images had to be pulled on the destination. Its volume attached correctly; +PostgreSQL and its exporter became Ready afterward. Preload required images on +an eligible destination before another planned database move. The larger database +dumps also completed much more efficiently through native PostgreSQL than through +`kubectl exec` output streaming. The native recovery script records this path. + +At 06:07 UTC, both Titan-04 and Titan-11 emitted fresh undervoltage messages. +Their instantaneous Pi 5 input samples were 4.9245 V and 4.89904 V; these are not +measurements of the transient minimum. Both had `get_throttled=0x50000` between +events. The fault remains active even when that individual sample reports only +historical bits. The user is checking power/boot-media issues physically. + +Titan-18 remains a blocker: Firefly and OpenSearch pods are waiting for graceful +termination on its stalled runtime. Do not force detach their mounted volumes +while the old host may still have writers. Its replacement/repair is distinct +from Kubernetes scheduling; a cordon alone cannot repair hung I/O. diff --git a/infrastructure/host-backup/NOTES.md b/infrastructure/host-backup/NOTES.md index e06cf6a5..ab1c8c5a 100644 --- a/infrastructure/host-backup/NOTES.md +++ b/infrastructure/host-backup/NOTES.md @@ -105,3 +105,42 @@ Disable the matching timer with `systemctl disable --now atlas-k3s-backup.timer` or `atlas-k3s-replica.timer`. Do not delete existing recovery bundles. If retiring replication, remove only the `atlas-k3s-backup-replica` authorized-key line on titan-db. Disabling these jobs does not restart PostgreSQL or Kubernetes. + +## Application databases + +Titan-0b also runs `atlas-application-postgres-backup.timer` daily at 08:15 UTC +plus up to five minutes. Install PostgreSQL client 16 and the versioned +`postgres_application_backup.sh` as `/usr/local/sbin/atlas-application-postgres-backup`. +It uses the host's existing root K3s access to discover the database pod and read +its mounted password, then connects with native PostgreSQL over the cluster LAN. +The temporary password file is mode 0600 in `/run` and removed on exit. This is +an application recovery copy; it cannot run while the Kubernetes API is down. + +Recovery sets are root-only under `/var/backups/atlas-postgres/run-*`; `latest` +changes only after all databases finish and checksum verification passes. +`databases.txt` maps its ordered entries to `db-1.dump`, `db-2.dump`, and so on. +Each database has its own consistent snapshot; the set is not one transaction +across databases. Globals include sensitive role information. No contents are +sent outside the approved LAN. Old run directories expire after seven days, +but only after a successful new backup. A run is bounded to two hours. + +```bash +ssh titan-0b sudo systemctl status atlas-application-postgres-backup.timer +ssh titan-0b sudo cat /var/backups/atlas-postgres/latest/COMPLETE +ssh titan-0b sudo systemctl start atlas-application-postgres-backup.service +``` + +`errors.log` remains inside the protected bundle. The only exported metric is +`atlas_application_postgres_backup_last_success_timestamp_seconds`. + +`postgres_dump_verify.sh` can verify one trusted dump on a host with PostgreSQL +16 server binaries and a postgres user. It starts a disposable instance without +TCP, restores schema and data without role/ACL replay, counts application tables, +and removes the instance. It does not overwrite a live application database. +On titan-db it is installed as `/usr/local/sbin/atlas-application-verify`. +The initial Gitea restore completed in six seconds with 111 application tables. +This is representative restore evidence, not proof for every application. + +To stop this schedule, disable `atlas-application-postgres-backup.timer` on +Titan-0b. Retain recovery sets. The earlier incomplete `pre-resources-*` directory +is not a valid complete backup; use only a set with `COMPLETE` and valid checksums. diff --git a/infrastructure/host-backup/atlas-application-postgres-backup.service b/infrastructure/host-backup/atlas-application-postgres-backup.service new file mode 100644 index 00000000..9a8e7ab0 --- /dev/null +++ b/infrastructure/host-backup/atlas-application-postgres-backup.service @@ -0,0 +1,15 @@ +# infrastructure/host-backup/atlas-application-postgres-backup.service +[Unit] +Description=Logical application PostgreSQL recovery bundle +After=network-online.target k3s.service +Wants=network-online.target + +[Service] +Type=oneshot +User=root +UMask=0077 +ExecStart=/usr/local/sbin/atlas-application-postgres-backup +TimeoutStartSec=2h +Nice=10 +IOSchedulingClass=best-effort +IOSchedulingPriority=7 diff --git a/infrastructure/host-backup/atlas-application-postgres-backup.timer b/infrastructure/host-backup/atlas-application-postgres-backup.timer new file mode 100644 index 00000000..22338a47 --- /dev/null +++ b/infrastructure/host-backup/atlas-application-postgres-backup.timer @@ -0,0 +1,12 @@ +# infrastructure/host-backup/atlas-application-postgres-backup.timer +[Unit] +Description=Daily application PostgreSQL recovery bundle + +[Timer] +OnCalendar=*-*-* 08:15:00 UTC +RandomizedDelaySec=300 +Persistent=true +Unit=atlas-application-postgres-backup.service + +[Install] +WantedBy=timers.target diff --git a/scripts/ops/postgres_application_backup.sh b/scripts/ops/postgres_application_backup.sh index a675414d..95e37115 100755 --- a/scripts/ops/postgres_application_backup.sh +++ b/scripts/ops/postgres_application_backup.sh @@ -14,6 +14,8 @@ install -d -m 700 "$backup_root" available=$(df -B1 --output=avail "$backup_root" | tail -1) (( available > 10737418240 )) || { echo 'Less than 10 GiB free.' >&2; exit 1; } bundle=$(mktemp -d "$backup_root/run-$(date -u +%Y%m%dT%H%M%SZ)-XXXXXX") +# Database-tool errors can include object names. Keep them with the protected copy. +exec 2>"$bundle/errors.log" passfile=$(mktemp /run/atlas-postgres-password.XXXXXX) trap 'rm -f "$passfile"' EXIT @@ -47,4 +49,12 @@ done < "$bundle/databases.txt" ) ln -sfn "$(basename "$bundle")" "$backup_root/latest.new" mv -Tf "$backup_root/latest.new" "$backup_root/latest" +metric_dir=/var/lib/node_exporter/textfile_collector +install -d -m 755 "$metric_dir" +metric_tmp=$(mktemp "$metric_dir/.application-backup.XXXXXX") +printf 'atlas_application_postgres_backup_last_success_timestamp_seconds %s\n' "$(date +%s)" > "$metric_tmp" +chmod 644 "$metric_tmp" +mv -f "$metric_tmp" "$metric_dir/atlas_application_postgres_backup.prom" +# A failed run never deletes the last good recovery set. +find "$backup_root" -mindepth 1 -maxdepth 1 -type d -name 'run-*' -mtime +7 -exec rm -rf -- {} + printf 'Verified recovery bundle: %s (%d databases).\n' "$bundle" "$index" diff --git a/scripts/ops/postgres_dump_verify.sh b/scripts/ops/postgres_dump_verify.sh new file mode 100755 index 00000000..7d0527cf --- /dev/null +++ b/scripts/ops/postgres_dump_verify.sh @@ -0,0 +1,31 @@ +#!/usr/bin/env bash +# Restore one trusted application dump into a disposable, local-only PG16 server. +set -euo pipefail +umask 077 +[[ $EUID == 0 && $# == 1 && -f $1 ]] || exit 64 +dump=$(readlink -f "$1") +bin=/usr/lib/postgresql/16/bin +stage=$(mktemp -d /var/tmp/atlas-application-restore.XXXXXXXX) +started=$(date +%s) +cleanup() { + if [[ -f $stage/data/postmaster.pid ]]; then + runuser -u postgres -- "$bin/pg_ctl" -D "$stage/data" -m immediate -w stop >/dev/null 2>&1 || true + fi + rm -rf -- "$stage" +} +trap cleanup EXIT +chown postgres:postgres "$stage" +install -o postgres -g postgres -m 0600 "$dump" "$stage/input.dump" +exec 2>"$dump.restore-error.txt" +runuser -u postgres -- "$bin/initdb" -D "$stage/data" -A trust --no-locale >"$stage/init.log" +runuser -u postgres -- "$bin/pg_ctl" -D "$stage/data" -l "$stage/server.log" \ + -o "-k $stage -p 55433 -c listen_addresses=''" -w start >/dev/null +runuser -u postgres -- "$bin/createdb" -h "$stage" -p 55433 restore_check +timeout 1800 runuser -u postgres -- "$bin/pg_restore" --exit-on-error --no-owner --no-privileges \ + -h "$stage" -p 55433 -d restore_check "$stage/input.dump" +tables=$(runuser -u postgres -- "$bin/psql" -XAt -h "$stage" -p 55433 -d restore_check \ + -c "SELECT count(*) FROM pg_tables WHERE schemaname NOT IN ('pg_catalog','information_schema')") +[[ $tables =~ ^[0-9]+$ && $tables -gt 0 ]] +printf 'verified_utc=%s\nelapsed_seconds=%s\napplication_tables=%s\nmethod=isolated_pg16_restore_without_role_acl_replay\n' \ + "$(date -u +%FT%TZ)" "$(( $(date +%s) - started ))" "$tables" >"$dump.RESTORE_CHECK" +cat "$dump.RESTORE_CHECK" diff --git a/services/monitoring/grafana-alerting-config.yaml b/services/monitoring/grafana-alerting-config.yaml index 16434afc..38592a0d 100644 --- a/services/monitoring/grafana-alerting-config.yaml +++ b/services/monitoring/grafana-alerting-config.yaml @@ -1145,3 +1145,53 @@ data: description: Check atlas-k3s-backup on titan-db and atlas-k3s-replica on titan-0b. See infrastructure/host-backup/NOTES.md. labels: severity: critical + - uid: atlas-application-postgres-backup-stale + title: Application PostgreSQL backup stale + condition: C + for: 10m + data: + - refId: A + relativeTimeRange: + from: 600 + to: 0 + datasourceUid: atlas-vm + model: + intervalMs: 60000 + maxDataPoints: 43200 + expr: max(time() - atlas_application_postgres_backup_last_success_timestamp_seconds) or on() vector(1000000000) + legendFormat: backup age seconds + datasource: + type: prometheus + uid: atlas-vm + - refId: B + datasourceUid: __expr__ + model: + expression: A + intervalMs: 60000 + maxDataPoints: 43200 + reducer: last + type: reduce + - refId: C + datasourceUid: __expr__ + model: + expression: B + intervalMs: 60000 + maxDataPoints: 43200 + type: threshold + conditions: + - evaluator: + params: + - 129600 + type: gt + operator: + type: and + reducer: + type: last + type: query + noDataState: Alerting + execErrState: Error + annotations: + summary: Application PostgreSQL backup is missing or older than 36 hours + description: Check atlas-application-postgres-backup on titan-0b. See infrastructure/host-backup/NOTES.md. + labels: + severity: critical