From 80aff498552b1ddd97ffe5a086518440ffa5de49 Mon Sep 17 00:00:00 2001 From: jenkins Date: Sat, 3 Oct 2026 00:19:34 -0500 Subject: [PATCH] recovery: schedule native datastore backups with LAN replica --- infrastructure/host-backup/NOTES.md | 100 ++++++++++++++++++ .../host-backup/atlas-k3s-backup.service | 16 +++ .../host-backup/atlas-k3s-backup.timer | 12 +++ .../host-backup/atlas-k3s-replica.service | 14 +++ .../host-backup/atlas-k3s-replica.timer | 12 +++ scripts/ops/k3s_backup_replica.sh | 32 ++++++ scripts/ops/k3s_backup_verify.sh | 32 ++++++ scripts/ops/k3s_datastore_backup.sh | 39 +++++++ 8 files changed, 257 insertions(+) create mode 100644 infrastructure/host-backup/NOTES.md create mode 100644 infrastructure/host-backup/atlas-k3s-backup.service create mode 100644 infrastructure/host-backup/atlas-k3s-backup.timer create mode 100644 infrastructure/host-backup/atlas-k3s-replica.service create mode 100644 infrastructure/host-backup/atlas-k3s-replica.timer create mode 100755 scripts/ops/k3s_backup_replica.sh create mode 100755 scripts/ops/k3s_backup_verify.sh create mode 100755 scripts/ops/k3s_datastore_backup.sh diff --git a/infrastructure/host-backup/NOTES.md b/infrastructure/host-backup/NOTES.md new file mode 100644 index 00000000..5b9eb461 --- /dev/null +++ b/infrastructure/host-backup/NOTES.md @@ -0,0 +1,100 @@ +# External control-plane datastore recovery + +K3s uses PostgreSQL 16 on `titan-db` (192.168.22.10), not embedded etcd. +These host-side systemd jobs keep recovery independent of the cluster, Vault, +SSO, Harbor and the homegrown controllers. Kubernetes manifests remain Flux-owned. + +## What runs where + +| Host | Timer | Protected directory | Purpose | +| --- | --- | --- | --- | +| titan-db | atlas-k3s-backup.timer, hourly at :00 plus 0-3 minutes | /var/backups/atlas-k3s | Consistent custom-format database dump, PostgreSQL globals, server token, checksums | +| titan-0b | atlas-k3s-replica.timer, hourly at :20 plus 0-3 minutes | /var/backups/atlas-k3s-replica | Second LAN disk with the complete bundle | + +Both keep completed snapshots for seven days, publishing `latest` only after +validation. Partial directories are removed on ordinary failure. Old snapshots +are pruned only after a successful new snapshot. A backup older than 90 minutes +cannot be exported as fresh. Both jobs refuse to start below 5 GiB free space. + +The database snapshot is consistent at dump start; PostgreSQL globals are a +separate dump. Avoid concurrent role or K3s token rotation during backup. +The replica compares the saved token with the local live server token and fails +if they differ. After an intentional token rotation, securely update the root-only +`/etc/atlas-k3s-backup/server-token` on titan-db from a current control plane. + +Files and directories are root-only. SSH encrypts transfer. These copies are +**not encrypted at rest** and are on the same LAN/site; they are not an off-site +or fire/theft recovery guarantee. Do not upload them to a general cloud backup: +the datastore contains credentials and may contain restricted source-derived data. + +## Installation and credentials + +Install `scripts/ops/k3s_datastore_backup.sh` as +`/usr/local/sbin/atlas-k3s-backup` on titan-db, mode 0755. Install +`k3s_backup_replica.sh` as `/usr/local/sbin/atlas-k3s-replica` on titan-0b. +Install the corresponding service/timer files from this directory in +`/etc/systemd/system/`, then run `systemctl daemon-reload`. + +The dedicated private key exists only on titan-0b at +`/root/.ssh/atlas_k3s_backup` (0600). Its pinned host key file is +`/root/.ssh/atlas_k3s_backup_known_hosts` for `[192.168.22.10]:2277`. +Verify replacement host keys out of band. +The public key in titan-db's atlas authorized_keys has `restrict`, a source +restriction to 192.168.22.12, and the forced command +`sudo -n /usr/local/sbin/atlas-k3s-backup export`. It cannot run an arbitrary command +or forward ports. Existing administrative keys remain separate. +No keys, tokens, dumps or PostgreSQL globals belong in this repository. + +## Normal operation + +On the appropriate host (sudo required): + +```bash +systemctl status atlas-k3s-backup.timer atlas-k3s-backup.service +systemctl status atlas-k3s-replica.timer atlas-k3s-replica.service +systemctl list-timers 'atlas-k3s-*' +``` + +Only the matching pair exists on each host. A completed oneshot service is +normally inactive with `Result=success`; the timer must remain active. +Inspect `latest/COMPLETE`, verify `SHA256SUMS`, and check timestamps, not just +whether the directory exists. Routine journal output contains status and byte +counts. Restricted `last-error.txt` is for local diagnosis; do not paste it into +chat or public logs without reviewing it. + +Run a new backup and then replicate it: + +```bash +ssh titan-db sudo systemctl start atlas-k3s-backup.service +ssh titan-0b sudo systemctl start atlas-k3s-replica.service +``` + +## Restore verification + +Install `scripts/ops/k3s_backup_verify.sh` as `/usr/local/sbin/atlas-k3s-verify` +on a host with PostgreSQL 16 binaries and a postgres system user. It restores +into a disposable cluster, using a private Unix socket and no TCP listener. +It never connects to or replaces the running datastore. + +```bash +sudo /usr/local/sbin/atlas-k3s-verify /var/backups/atlas-k3s/latest +``` + +Success records `RESTORE_CHECK` with elapsed time and a nonzero `kine` row count. +This proves archive readability and database/schema restoration. It deliberately +does not replay global roles/ACLs or start K3s, so it is not a full disaster drill. +Temporary database files are removed after the check. + +For actual disaster recovery: fence the old datastore and stop K3s writers first; +preserve any surviving data; provision compatible PostgreSQL; review and restore +globals and the k3s dump; restore the matching server token; verify connectivity +and start one server before the others. Do not overwrite a running database. +See the [K3s recovery requirements](https://docs.k3s.io/datastore/backup-restore) +and [PostgreSQL restore reference](https://www.postgresql.org/docs/16/app-pgrestore.html). + +## Rollback + +Disable the matching timer with `systemctl disable --now atlas-k3s-backup.timer` +or `atlas-k3s-replica.timer`. Do not delete existing recovery bundles. If retiring +replication, remove only the `atlas-k3s-backup-replica` authorized-key line on +titan-db. Disabling these jobs does not restart PostgreSQL or Kubernetes. diff --git a/infrastructure/host-backup/atlas-k3s-backup.service b/infrastructure/host-backup/atlas-k3s-backup.service new file mode 100644 index 00000000..dc0ab2e3 --- /dev/null +++ b/infrastructure/host-backup/atlas-k3s-backup.service @@ -0,0 +1,16 @@ +# infrastructure/host-backup/atlas-k3s-backup.service +[Unit] +Description=Back up the external K3s PostgreSQL datastore +After=postgresql.service + +[Service] +Type=oneshot +ExecStart=/usr/local/sbin/atlas-k3s-backup backup +TimeoutStartSec=20min +UMask=0077 +Nice=10 +IOSchedulingClass=best-effort +IOSchedulingPriority=7 +ProtectSystem=full +ProtectHome=true +PrivateTmp=true diff --git a/infrastructure/host-backup/atlas-k3s-backup.timer b/infrastructure/host-backup/atlas-k3s-backup.timer new file mode 100644 index 00000000..75525dfe --- /dev/null +++ b/infrastructure/host-backup/atlas-k3s-backup.timer @@ -0,0 +1,12 @@ +# infrastructure/host-backup/atlas-k3s-backup.timer +[Unit] +Description=Hourly external K3s datastore backup + +[Timer] +OnCalendar=*-*-* *:00:00 +RandomizedDelaySec=3min +Persistent=true +Unit=atlas-k3s-backup.service + +[Install] +WantedBy=timers.target diff --git a/infrastructure/host-backup/atlas-k3s-replica.service b/infrastructure/host-backup/atlas-k3s-replica.service new file mode 100644 index 00000000..73301eef --- /dev/null +++ b/infrastructure/host-backup/atlas-k3s-replica.service @@ -0,0 +1,14 @@ +# infrastructure/host-backup/atlas-k3s-replica.service +[Unit] +Description=Copy the K3s recovery bundle to an independent LAN disk +After=network-online.target +Wants=network-online.target + +[Service] +Type=oneshot +ExecStart=/usr/local/sbin/atlas-k3s-replica +TimeoutStartSec=20min +UMask=0077 +Nice=10 +ProtectSystem=full +PrivateTmp=true diff --git a/infrastructure/host-backup/atlas-k3s-replica.timer b/infrastructure/host-backup/atlas-k3s-replica.timer new file mode 100644 index 00000000..ff35e42a --- /dev/null +++ b/infrastructure/host-backup/atlas-k3s-replica.timer @@ -0,0 +1,12 @@ +# infrastructure/host-backup/atlas-k3s-replica.timer +[Unit] +Description=Hourly independent LAN copy of the K3s recovery bundle + +[Timer] +OnCalendar=*-*-* *:20:00 +RandomizedDelaySec=3min +Persistent=true +Unit=atlas-k3s-replica.service + +[Install] +WantedBy=timers.target diff --git a/scripts/ops/k3s_backup_replica.sh b/scripts/ops/k3s_backup_replica.sh new file mode 100755 index 00000000..b0fea1d1 --- /dev/null +++ b/scripts/ops/k3s_backup_replica.sh @@ -0,0 +1,32 @@ +#!/usr/bin/env bash +# Keep a second LAN copy and the matching server token on titan-0b. +set -euo pipefail +umask 077 +[[ $EUID == 0 ]] || exit 77 +root=/var/backups/atlas-k3s-replica +install -d -m 0700 "$root" +exec 9>/run/lock/atlas-k3s-replica.lock +flock -n 9 || exit 75 +[[ $(df --output=avail -k "$root" | tail -1) -gt 5242880 ]] || exit 73 +stage=$(mktemp -d "$root/.pending.XXXXXXXX") +trap 'rm -rf -- "$stage"' EXIT +exec 2>"$root/last-error.txt" +timeout 900 ssh -T -p 2277 -o BatchMode=yes -o StrictHostKeyChecking=yes \ + -o UserKnownHostsFile=/root/.ssh/atlas_k3s_backup_known_hosts \ + -o ConnectTimeout=10 -o ServerAliveInterval=30 -o ServerAliveCountMax=3 \ + -o IdentitiesOnly=yes -i /root/.ssh/atlas_k3s_backup \ + atlas@192.168.22.10 export >"$stage/snapshot.tar" +# Extract only the five files provided by the fixed remote command. +tar -xf "$stage/snapshot.tar" -C "$stage" --no-same-owner --no-same-permissions \ + k3s.dump globals.sql server-token SHA256SUMS COMPLETE +rm "$stage/snapshot.tar" +(cd "$stage"; sha256sum --quiet -c SHA256SUMS) +test -s "$stage/k3s.dump" +# Token rotation must update the protected copy on titan-db before success. +cmp -s /var/lib/rancher/k3s/server/token "$stage/server-token" +final="$root/snapshot-$(date -u +%Y%m%dT%H%M%SZ)" +mv "$stage" "$final" +ln -sfn "$(basename "$final")" "$root/.latest" +mv -Tf "$root/.latest" "$root/latest" +find "$root" -mindepth 1 -maxdepth 1 -type d -name 'snapshot-*' -mtime +7 -exec rm -rf -- {} + +printf 'Datastore replica complete; database and server token protected locally\n' diff --git a/scripts/ops/k3s_backup_verify.sh b/scripts/ops/k3s_backup_verify.sh new file mode 100755 index 00000000..df7b264d --- /dev/null +++ b/scripts/ops/k3s_backup_verify.sh @@ -0,0 +1,32 @@ +#!/usr/bin/env bash +# Restore a backup into a disposable PostgreSQL 16 cluster with no TCP listener. +set -euo pipefail +umask 077 +[[ $EUID == 0 && $# == 1 ]] || exit 64 +snapshot=$(readlink -f "$1") +[[ -f $snapshot/COMPLETE && -f $snapshot/k3s.dump ]] || exit 66 +(cd "$snapshot"; sha256sum --quiet -c SHA256SUMS) +bin=/usr/lib/postgresql/16/bin +stage=$(mktemp -d /var/tmp/atlas-k3s-restore.XXXXXXXX) +started=$(date +%s) +cleanup() { + if [[ -f $stage/data/postmaster.pid ]]; then + runuser -u postgres -- "$bin/pg_ctl" -D "$stage/data" -m immediate -w stop >/dev/null 2>&1 || true + fi + rm -rf -- "$stage" +} +trap cleanup EXIT +chown postgres:postgres "$stage" +install -o postgres -g postgres -m 0600 "$snapshot/k3s.dump" "$stage/k3s.dump" +exec 2>"$snapshot/restore-error.txt" +runuser -u postgres -- "$bin/initdb" -D "$stage/data" -A trust --no-locale >"$stage/init.log" +runuser -u postgres -- "$bin/pg_ctl" -D "$stage/data" -l "$stage/server.log" \ + -o "-k $stage -p 55432 -c listen_addresses=''" -w start >/dev/null +runuser -u postgres -- "$bin/createdb" -h "$stage" -p 55432 k3s +timeout 900 runuser -u postgres -- "$bin/pg_restore" --exit-on-error --no-owner --no-privileges \ + -h "$stage" -p 55432 -d k3s "$stage/k3s.dump" +rows=$(runuser -u postgres -- "$bin/psql" -XAt -h "$stage" -p 55432 -d k3s -c 'SELECT count(*) FROM kine') +[[ $rows =~ ^[0-9]+$ && $rows -gt 0 ]] +printf 'verified_utc=%s\nelapsed_seconds=%s\nkine_rows=%s\nmethod=isolated_pg16_restore_without_role_acl_replay\n' \ + "$(date -u +%FT%TZ)" "$(( $(date +%s) - started ))" "$rows" >"$snapshot/RESTORE_CHECK" +cat "$snapshot/RESTORE_CHECK" diff --git a/scripts/ops/k3s_datastore_backup.sh b/scripts/ops/k3s_datastore_backup.sh new file mode 100755 index 00000000..898ef3e4 --- /dev/null +++ b/scripts/ops/k3s_datastore_backup.sh @@ -0,0 +1,39 @@ +#!/usr/bin/env bash +# Native external-datastore backup on titan-db. Never print database contents. +set -euo pipefail +umask 077 +root=/var/backups/atlas-k3s +[[ $EUID == 0 ]] || exit 77 +install -d -m 0700 "$root" +exec 9>/run/lock/atlas-k3s-backup.lock +flock -w 600 9 +if [[ ${1:-backup} == export ]]; then + latest=$(readlink -f "$root/latest") + [[ $latest == "$root"/snapshot-* && -f $latest/COMPLETE ]] || exit 1 + [[ $(( $(date +%s) - $(stat -c %Y "$latest/COMPLETE") )) -lt 5400 ]] || exit 75 + # This is the only command allowed by the replica's restricted SSH key. + cd "$latest" + sha256sum --quiet -c SHA256SUMS + exec tar -cf - k3s.dump globals.sql server-token SHA256SUMS COMPLETE +fi +[[ ${1:-backup} == backup ]] || exit 64 +[[ $(df --output=avail -k "$root" | tail -1) -gt 5242880 ]] || exit 73 +stage=$(mktemp -d "$root/.pending.XXXXXXXX") +trap 'rm -rf -- "$stage"' EXIT +exec 2>"$root/last-error.txt" +started=$(date -u +%FT%TZ) +timeout 900 runuser -u postgres -- pg_dump --format=custom --dbname=k3s >"$stage/k3s.dump" +timeout 120 runuser -u postgres -- pg_dumpall --globals-only >"$stage/globals.sql" +pg_restore --list "$stage/k3s.dump" >/dev/null +install -m 0600 /etc/atlas-k3s-backup/server-token "$stage/server-token" +( + cd "$stage" + sha256sum k3s.dump globals.sql server-token >SHA256SUMS + printf 'started_utc=%s\ncompleted_utc=%s\n' "$started" "$(date -u +%FT%TZ)" >COMPLETE +) +final="$root/snapshot-$(date -u +%Y%m%dT%H%M%SZ)" +mv "$stage" "$final" +ln -sfn "$(basename "$final")" "$root/.latest" +mv -Tf "$root/.latest" "$root/latest" +find "$root" -mindepth 1 -maxdepth 1 -type d -name 'snapshot-*' -mtime +7 -exec rm -rf -- {} + +printf 'Datastore backup complete: %s bytes\n' "$(stat -c %s "$final/k3s.dump")"