monitoring: expose datastore backup freshness and missing copies
This commit is contained in:
parent
0b7669f0f6
commit
0780c93f69
@ -62,6 +62,13 @@ whether the directory exists. Routine journal output contains status and byte
|
||||
counts. Restricted `last-error.txt` is for local diagnosis; do not paste it into
|
||||
chat or public logs without reviewing it.
|
||||
|
||||
The scripts atomically publish a non-sensitive success timestamp through the
|
||||
existing node exporter textfile collector. Titan-db uses the versioned
|
||||
`node-exporter-textfile.conf` systemd drop-in; titan-0b uses the cluster's
|
||||
node-exporter DaemonSet. The metric is
|
||||
`atlas_k3s_backup_last_success_timestamp_seconds`, labelled `copy="datastore"`
|
||||
or `copy="lan_replica"`. These files contain no paths, keys or database records.
|
||||
|
||||
Run a new backup and then replicate it:
|
||||
|
||||
```bash
|
||||
|
||||
5
infrastructure/host-backup/node-exporter-textfile.conf
Normal file
5
infrastructure/host-backup/node-exporter-textfile.conf
Normal file
@ -0,0 +1,5 @@
|
||||
# infrastructure/host-backup/node-exporter-textfile.conf
|
||||
# titan-db's existing node_exporter service, with native textfile collection.
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/usr/local/bin/node_exporter --web.listen-address=:9100 --collector.systemd --collector.filesystem.ignored-mount-points=^/(sys|proc|dev|run)($|/) --collector.textfile.directory=/var/lib/node_exporter/textfile_collector
|
||||
@ -29,4 +29,10 @@ mv "$stage" "$final"
|
||||
ln -sfn "$(basename "$final")" "$root/.latest"
|
||||
mv -Tf "$root/.latest" "$root/latest"
|
||||
find "$root" -mindepth 1 -maxdepth 1 -type d -name 'snapshot-*' -mtime +7 -exec rm -rf -- {} +
|
||||
metrics=/var/lib/node_exporter/textfile_collector
|
||||
install -d -m 0755 "$metrics"
|
||||
printf 'atlas_k3s_backup_last_success_timestamp_seconds{copy="lan_replica"} %s\n' \
|
||||
"$(date +%s)" >"$metrics/atlas_k3s_replica.prom.tmp"
|
||||
chmod 0644 "$metrics/atlas_k3s_replica.prom.tmp"
|
||||
mv "$metrics/atlas_k3s_replica.prom.tmp" "$metrics/atlas_k3s_replica.prom"
|
||||
printf 'Datastore replica complete; database and server token protected locally\n'
|
||||
|
||||
@ -36,4 +36,10 @@ mv "$stage" "$final"
|
||||
ln -sfn "$(basename "$final")" "$root/.latest"
|
||||
mv -Tf "$root/.latest" "$root/latest"
|
||||
find "$root" -mindepth 1 -maxdepth 1 -type d -name 'snapshot-*' -mtime +7 -exec rm -rf -- {} +
|
||||
metrics=/var/lib/node_exporter/textfile_collector
|
||||
install -d -m 0755 "$metrics"
|
||||
printf 'atlas_k3s_backup_last_success_timestamp_seconds{copy="datastore"} %s\n' \
|
||||
"$(date +%s)" >"$metrics/atlas_k3s_backup.prom.tmp"
|
||||
chmod 0644 "$metrics/atlas_k3s_backup.prom.tmp"
|
||||
mv "$metrics/atlas_k3s_backup.prom.tmp" "$metrics/atlas_k3s_backup.prom"
|
||||
printf 'Datastore backup complete: %s bytes\n' "$(stat -c %s "$final/k3s.dump")"
|
||||
|
||||
@ -1089,3 +1089,59 @@ data:
|
||||
summary: "Postmark exporter reports sustained API outage"
|
||||
labels:
|
||||
severity: warning
|
||||
|
||||
- orgId: 1
|
||||
name: atlas-datastore-recovery
|
||||
folder: Alerts
|
||||
interval: 1m
|
||||
rules:
|
||||
- uid: atlas-k3s-backup-stale
|
||||
title: K3s datastore backup or LAN copy stale
|
||||
condition: C
|
||||
for: 10m
|
||||
data:
|
||||
- refId: A
|
||||
relativeTimeRange:
|
||||
from: 600
|
||||
to: 0
|
||||
datasourceUid: atlas-vm
|
||||
model:
|
||||
intervalMs: 60000
|
||||
maxDataPoints: 43200
|
||||
expr: (max(time() - atlas_k3s_backup_last_success_timestamp_seconds{copy=~"datastore|lan_replica"}) + 1000000000 * (count(count by (copy) (atlas_k3s_backup_last_success_timestamp_seconds{copy=~"datastore|lan_replica"})) < bool 2)) or on() vector(1000000000)
|
||||
legendFormat: backup age seconds
|
||||
datasource:
|
||||
type: prometheus
|
||||
uid: atlas-vm
|
||||
- refId: B
|
||||
datasourceUid: __expr__
|
||||
model:
|
||||
expression: A
|
||||
intervalMs: 60000
|
||||
maxDataPoints: 43200
|
||||
reducer: last
|
||||
type: reduce
|
||||
- refId: C
|
||||
datasourceUid: __expr__
|
||||
model:
|
||||
expression: B
|
||||
intervalMs: 60000
|
||||
maxDataPoints: 43200
|
||||
type: threshold
|
||||
conditions:
|
||||
- evaluator:
|
||||
params:
|
||||
- 7200
|
||||
type: gt
|
||||
operator:
|
||||
type: and
|
||||
reducer:
|
||||
type: last
|
||||
type: query
|
||||
noDataState: Alerting
|
||||
execErrState: Error
|
||||
annotations:
|
||||
summary: K3s datastore backup or independent LAN copy is missing or older than two hours
|
||||
description: Check atlas-k3s-backup on titan-db and atlas-k3s-replica on titan-0b. See infrastructure/host-backup/NOTES.md.
|
||||
labels:
|
||||
severity: critical
|
||||
|
||||
@ -51,6 +51,8 @@ spec:
|
||||
disableWait: true
|
||||
values:
|
||||
hostNetwork: false
|
||||
extraArgs:
|
||||
- --collector.textfile.directory=/host/root/var/lib/node_exporter/textfile_collector
|
||||
rbac:
|
||||
pspEnabled: false
|
||||
service:
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user