2026-01-13 09:59:39 -03:00
|
|
|
#!/bin/sh
|
|
|
|
|
set -eu
|
|
|
|
|
|
|
|
|
|
ONE_SHOT=${ONE_SHOT:-false}
|
2026-03-30 18:36:53 -03:00
|
|
|
SWEEP_INTERVAL_SEC=${SWEEP_INTERVAL_SEC:-21600}
|
|
|
|
|
HIGH_USAGE_PERCENT=${HIGH_USAGE_PERCENT:-70}
|
|
|
|
|
EMERGENCY_USAGE_PERCENT=${EMERGENCY_USAGE_PERCENT:-85}
|
|
|
|
|
LOG_RETENTION_DAYS=${LOG_RETENTION_DAYS:-7}
|
2026-03-31 00:06:44 -03:00
|
|
|
ORPHAN_POD_RETENTION_DAYS=${ORPHAN_POD_RETENTION_DAYS:-3}
|
2026-03-30 18:36:53 -03:00
|
|
|
JOURNAL_MAX_SIZE=${JOURNAL_MAX_SIZE:-200M}
|
2026-01-13 09:59:39 -03:00
|
|
|
|
2026-03-31 00:06:44 -03:00
|
|
|
sweep_once() {
|
|
|
|
|
usage=$(df -P /host | awk 'NR==2 {gsub(/%/,"",$5); print $5}') || usage=""
|
|
|
|
|
|
|
|
|
|
# crictl image metadata frequently omits createdAt on this cluster; prune by
|
|
|
|
|
# runtime reachability whenever rootfs crosses pressure thresholds.
|
|
|
|
|
if [ -n "${usage}" ] && [ "${usage}" -ge "${HIGH_USAGE_PERCENT}" ]; then
|
|
|
|
|
chroot /host /bin/sh -c "crictl rmi --prune >/dev/null 2>&1 || true"
|
|
|
|
|
fi
|
|
|
|
|
|
2026-10-04 07:13:27 -05:00
|
|
|
# Kubelet owns active logs; retain every active UID, including quiet pods.
|
|
|
|
|
python3 /scripts/node_pod_log_cleanup.py --host-root /host \
|
|
|
|
|
--retention-days "${ORPHAN_POD_RETENTION_DAYS}"
|
2026-03-31 00:06:44 -03:00
|
|
|
|
|
|
|
|
if [ -d /host/var/log.hdd/containers ]; then
|
|
|
|
|
find /host/var/log.hdd/containers -xtype l -print -delete 2>/dev/null || true
|
2026-03-30 18:36:53 -03:00
|
|
|
fi
|
|
|
|
|
|
|
|
|
|
find /host/var/lib/rancher/k3s/agent/images -type f -name "*.tar" -mtime +7 -print -delete 2>/dev/null || true
|
|
|
|
|
find /host/var/lib/rancher/k3s/agent/containerd -maxdepth 1 -type f -mtime +7 -print -delete 2>/dev/null || true
|
|
|
|
|
|
|
|
|
|
if [ -n "${usage}" ] && [ "${usage}" -ge "${EMERGENCY_USAGE_PERCENT}" ]; then
|
|
|
|
|
# Emergency pass for rootfs pressure on SD-backed nodes.
|
2026-03-31 00:06:44 -03:00
|
|
|
chroot /host /bin/sh -c "crictl rmi --prune >/dev/null 2>&1 || true"
|
2026-03-30 18:36:53 -03:00
|
|
|
chroot /host /bin/sh -c "journalctl --vacuum-size='${JOURNAL_MAX_SIZE}' >/dev/null 2>&1 || true"
|
2026-10-04 07:13:27 -05:00
|
|
|
# Pod logs are handled only by the UID-aware check above.
|
|
|
|
|
find /host/var/log -path /host/var/log/pods -prune -o -type f -name "*.gz" -mtime +"${LOG_RETENTION_DAYS}" -print -exec rm -f {} \; 2>/dev/null || true
|
|
|
|
|
find /host/var/log.hdd -path /host/var/log.hdd/pods -prune -o -type f -name "*.gz" -mtime +"${LOG_RETENTION_DAYS}" -print -exec rm -f {} \; 2>/dev/null || true
|
2026-03-30 18:36:53 -03:00
|
|
|
chroot /host /bin/sh -c "if command -v apt-get >/dev/null 2>&1; then apt-get clean >/dev/null 2>&1 || true; fi"
|
|
|
|
|
fi
|
|
|
|
|
}
|
2026-01-13 09:59:39 -03:00
|
|
|
|
2026-03-30 18:36:53 -03:00
|
|
|
sweep_once
|
2026-10-04 07:17:16 -05:00
|
|
|
touch /tmp/initial-sweep-complete
|
2026-01-13 09:59:39 -03:00
|
|
|
|
|
|
|
|
if [ "${ONE_SHOT}" = "true" ]; then
|
|
|
|
|
exit 0
|
|
|
|
|
fi
|
|
|
|
|
|
2026-03-30 18:36:53 -03:00
|
|
|
while true; do
|
|
|
|
|
sleep "${SWEEP_INTERVAL_SEC}"
|
|
|
|
|
sweep_once
|
|
|
|
|
done
|