maintenance: leave image garbage collection to kubelet

This commit is contained in:
jenkins 2026-10-04 07:20:32 -05:00
parent 8ba6ffc471
commit a223b9fa49
2 changed files with 3 additions and 13 deletions

View File

@ -16,7 +16,7 @@ spec:
template:
metadata:
annotations:
atlas.bstein.dev/config-revision: "2026-10-04-preserve-active-pod-logs"
atlas.bstein.dev/config-revision: "2026-10-04-native-image-gc"
labels:
app: node-image-sweeper
spec:
@ -51,8 +51,6 @@ spec:
env:
- name: SWEEP_INTERVAL_SEC
value: "7200"
- name: HIGH_USAGE_PERCENT
value: "70"
- name: EMERGENCY_USAGE_PERCENT
value: "80"
- name: LOG_RETENTION_DAYS

View File

@ -3,7 +3,6 @@ set -eu
ONE_SHOT=${ONE_SHOT:-false}
SWEEP_INTERVAL_SEC=${SWEEP_INTERVAL_SEC:-21600}
HIGH_USAGE_PERCENT=${HIGH_USAGE_PERCENT:-70}
EMERGENCY_USAGE_PERCENT=${EMERGENCY_USAGE_PERCENT:-85}
LOG_RETENTION_DAYS=${LOG_RETENTION_DAYS:-7}
ORPHAN_POD_RETENTION_DAYS=${ORPHAN_POD_RETENTION_DAYS:-3}
@ -12,11 +11,8 @@ JOURNAL_MAX_SIZE=${JOURNAL_MAX_SIZE:-200M}
sweep_once() {
usage=$(df -P /host | awk 'NR==2 {gsub(/%/,"",$5); print $5}') || usage=""
# crictl image metadata frequently omits createdAt on this cluster; prune by
# runtime reachability whenever rootfs crosses pressure thresholds.
if [ -n "${usage}" ] && [ "${usage}" -ge "${HIGH_USAGE_PERCENT}" ]; then
chroot /host /bin/sh -c "crictl rmi --prune >/dev/null 2>&1 || true"
fi
# Kubelet owns container images and runtime state, including garbage collection.
# Keep this legacy-named helper limited to host logs and package-cache cleanup.
# Kubelet owns active logs; retain every active UID, including quiet pods.
python3 /scripts/node_pod_log_cleanup.py --host-root /host \
@ -26,12 +22,8 @@ sweep_once() {
find /host/var/log.hdd/containers -xtype l -print -delete 2>/dev/null || true
fi
find /host/var/lib/rancher/k3s/agent/images -type f -name "*.tar" -mtime +7 -print -delete 2>/dev/null || true
find /host/var/lib/rancher/k3s/agent/containerd -maxdepth 1 -type f -mtime +7 -print -delete 2>/dev/null || true
if [ -n "${usage}" ] && [ "${usage}" -ge "${EMERGENCY_USAGE_PERCENT}" ]; then
# Emergency pass for rootfs pressure on SD-backed nodes.
chroot /host /bin/sh -c "crictl rmi --prune >/dev/null 2>&1 || true"
chroot /host /bin/sh -c "journalctl --vacuum-size='${JOURNAL_MAX_SIZE}' >/dev/null 2>&1 || true"
# Pod logs are handled only by the UID-aware check above.
find /host/var/log -path /host/var/log/pods -prune -o -type f -name "*.gz" -mtime +"${LOG_RETENTION_DAYS}" -print -exec rm -f {} \; 2>/dev/null || true