diff --git a/infrastructure/host-backup/NOTES.md b/infrastructure/host-backup/NOTES.md index 5b9eb461..e06cf6a5 100644 --- a/infrastructure/host-backup/NOTES.md +++ b/infrastructure/host-backup/NOTES.md @@ -62,6 +62,13 @@ whether the directory exists. Routine journal output contains status and byte counts. Restricted `last-error.txt` is for local diagnosis; do not paste it into chat or public logs without reviewing it. +The scripts atomically publish a non-sensitive success timestamp through the +existing node exporter textfile collector. Titan-db uses the versioned +`node-exporter-textfile.conf` systemd drop-in; titan-0b uses the cluster's +node-exporter DaemonSet. The metric is +`atlas_k3s_backup_last_success_timestamp_seconds`, labelled `copy="datastore"` +or `copy="lan_replica"`. These files contain no paths, keys or database records. + Run a new backup and then replicate it: ```bash diff --git a/infrastructure/host-backup/node-exporter-textfile.conf b/infrastructure/host-backup/node-exporter-textfile.conf new file mode 100644 index 00000000..b6be4012 --- /dev/null +++ b/infrastructure/host-backup/node-exporter-textfile.conf @@ -0,0 +1,5 @@ +# infrastructure/host-backup/node-exporter-textfile.conf +# titan-db's existing node_exporter service, with native textfile collection. +[Service] +ExecStart= +ExecStart=/usr/local/bin/node_exporter --web.listen-address=:9100 --collector.systemd --collector.filesystem.ignored-mount-points=^/(sys|proc|dev|run)($|/) --collector.textfile.directory=/var/lib/node_exporter/textfile_collector diff --git a/scripts/ops/k3s_backup_replica.sh b/scripts/ops/k3s_backup_replica.sh index b0fea1d1..fdaff2b9 100755 --- a/scripts/ops/k3s_backup_replica.sh +++ b/scripts/ops/k3s_backup_replica.sh @@ -29,4 +29,10 @@ mv "$stage" "$final" ln -sfn "$(basename "$final")" "$root/.latest" mv -Tf "$root/.latest" "$root/latest" find "$root" -mindepth 1 -maxdepth 1 -type d -name 'snapshot-*' -mtime +7 -exec rm -rf -- {} + +metrics=/var/lib/node_exporter/textfile_collector +install -d -m 0755 "$metrics" +printf 'atlas_k3s_backup_last_success_timestamp_seconds{copy="lan_replica"} %s\n' \ + "$(date +%s)" >"$metrics/atlas_k3s_replica.prom.tmp" +chmod 0644 "$metrics/atlas_k3s_replica.prom.tmp" +mv "$metrics/atlas_k3s_replica.prom.tmp" "$metrics/atlas_k3s_replica.prom" printf 'Datastore replica complete; database and server token protected locally\n' diff --git a/scripts/ops/k3s_datastore_backup.sh b/scripts/ops/k3s_datastore_backup.sh index 898ef3e4..eff1af27 100755 --- a/scripts/ops/k3s_datastore_backup.sh +++ b/scripts/ops/k3s_datastore_backup.sh @@ -36,4 +36,10 @@ mv "$stage" "$final" ln -sfn "$(basename "$final")" "$root/.latest" mv -Tf "$root/.latest" "$root/latest" find "$root" -mindepth 1 -maxdepth 1 -type d -name 'snapshot-*' -mtime +7 -exec rm -rf -- {} + +metrics=/var/lib/node_exporter/textfile_collector +install -d -m 0755 "$metrics" +printf 'atlas_k3s_backup_last_success_timestamp_seconds{copy="datastore"} %s\n' \ + "$(date +%s)" >"$metrics/atlas_k3s_backup.prom.tmp" +chmod 0644 "$metrics/atlas_k3s_backup.prom.tmp" +mv "$metrics/atlas_k3s_backup.prom.tmp" "$metrics/atlas_k3s_backup.prom" printf 'Datastore backup complete: %s bytes\n' "$(stat -c %s "$final/k3s.dump")" diff --git a/services/monitoring/grafana-alerting-config.yaml b/services/monitoring/grafana-alerting-config.yaml index 091c4ca9..16434afc 100644 --- a/services/monitoring/grafana-alerting-config.yaml +++ b/services/monitoring/grafana-alerting-config.yaml @@ -1089,3 +1089,59 @@ data: summary: "Postmark exporter reports sustained API outage" labels: severity: warning + + - orgId: 1 + name: atlas-datastore-recovery + folder: Alerts + interval: 1m + rules: + - uid: atlas-k3s-backup-stale + title: K3s datastore backup or LAN copy stale + condition: C + for: 10m + data: + - refId: A + relativeTimeRange: + from: 600 + to: 0 + datasourceUid: atlas-vm + model: + intervalMs: 60000 + maxDataPoints: 43200 + expr: (max(time() - atlas_k3s_backup_last_success_timestamp_seconds{copy=~"datastore|lan_replica"}) + 1000000000 * (count(count by (copy) (atlas_k3s_backup_last_success_timestamp_seconds{copy=~"datastore|lan_replica"})) < bool 2)) or on() vector(1000000000) + legendFormat: backup age seconds + datasource: + type: prometheus + uid: atlas-vm + - refId: B + datasourceUid: __expr__ + model: + expression: A + intervalMs: 60000 + maxDataPoints: 43200 + reducer: last + type: reduce + - refId: C + datasourceUid: __expr__ + model: + expression: B + intervalMs: 60000 + maxDataPoints: 43200 + type: threshold + conditions: + - evaluator: + params: + - 7200 + type: gt + operator: + type: and + reducer: + type: last + type: query + noDataState: Alerting + execErrState: Error + annotations: + summary: K3s datastore backup or independent LAN copy is missing or older than two hours + description: Check atlas-k3s-backup on titan-db and atlas-k3s-replica on titan-0b. See infrastructure/host-backup/NOTES.md. + labels: + severity: critical diff --git a/services/monitoring/helmrelease.yaml b/services/monitoring/helmrelease.yaml index c45d513a..2a0bb367 100644 --- a/services/monitoring/helmrelease.yaml +++ b/services/monitoring/helmrelease.yaml @@ -51,6 +51,8 @@ spec: disableWait: true values: hostNetwork: false + extraArgs: + - --collector.textfile.directory=/host/root/var/lib/node_exporter/textfile_collector rbac: pspEnabled: false service: