diff --git a/services/maintenance/node-ops/node-image-sweeper-daemonset.yaml b/services/maintenance/node-ops/node-image-sweeper-daemonset.yaml index 6f029b92..db4b2a9e 100644 --- a/services/maintenance/node-ops/node-image-sweeper-daemonset.yaml +++ b/services/maintenance/node-ops/node-image-sweeper-daemonset.yaml @@ -16,7 +16,7 @@ spec: template: metadata: annotations: - atlas.bstein.dev/config-revision: "2026-10-04-preserve-active-pod-logs" + atlas.bstein.dev/config-revision: "2026-10-04-native-image-gc" labels: app: node-image-sweeper spec: @@ -51,8 +51,6 @@ spec: env: - name: SWEEP_INTERVAL_SEC value: "7200" - - name: HIGH_USAGE_PERCENT - value: "70" - name: EMERGENCY_USAGE_PERCENT value: "80" - name: LOG_RETENTION_DAYS diff --git a/services/maintenance/node-ops/scripts/node_image_sweeper.sh b/services/maintenance/node-ops/scripts/node_image_sweeper.sh index 140560d9..52d389dd 100644 --- a/services/maintenance/node-ops/scripts/node_image_sweeper.sh +++ b/services/maintenance/node-ops/scripts/node_image_sweeper.sh @@ -3,7 +3,6 @@ set -eu ONE_SHOT=${ONE_SHOT:-false} SWEEP_INTERVAL_SEC=${SWEEP_INTERVAL_SEC:-21600} -HIGH_USAGE_PERCENT=${HIGH_USAGE_PERCENT:-70} EMERGENCY_USAGE_PERCENT=${EMERGENCY_USAGE_PERCENT:-85} LOG_RETENTION_DAYS=${LOG_RETENTION_DAYS:-7} ORPHAN_POD_RETENTION_DAYS=${ORPHAN_POD_RETENTION_DAYS:-3} @@ -12,11 +11,8 @@ JOURNAL_MAX_SIZE=${JOURNAL_MAX_SIZE:-200M} sweep_once() { usage=$(df -P /host | awk 'NR==2 {gsub(/%/,"",$5); print $5}') || usage="" - # crictl image metadata frequently omits createdAt on this cluster; prune by - # runtime reachability whenever rootfs crosses pressure thresholds. - if [ -n "${usage}" ] && [ "${usage}" -ge "${HIGH_USAGE_PERCENT}" ]; then - chroot /host /bin/sh -c "crictl rmi --prune >/dev/null 2>&1 || true" - fi + # Kubelet owns container images and runtime state, including garbage collection. + # Keep this legacy-named helper limited to host logs and package-cache cleanup. # Kubelet owns active logs; retain every active UID, including quiet pods. python3 /scripts/node_pod_log_cleanup.py --host-root /host \ @@ -26,12 +22,8 @@ sweep_once() { find /host/var/log.hdd/containers -xtype l -print -delete 2>/dev/null || true fi - find /host/var/lib/rancher/k3s/agent/images -type f -name "*.tar" -mtime +7 -print -delete 2>/dev/null || true - find /host/var/lib/rancher/k3s/agent/containerd -maxdepth 1 -type f -mtime +7 -print -delete 2>/dev/null || true - if [ -n "${usage}" ] && [ "${usage}" -ge "${EMERGENCY_USAGE_PERCENT}" ]; then # Emergency pass for rootfs pressure on SD-backed nodes. - chroot /host /bin/sh -c "crictl rmi --prune >/dev/null 2>&1 || true" chroot /host /bin/sh -c "journalctl --vacuum-size='${JOURNAL_MAX_SIZE}' >/dev/null 2>&1 || true" # Pod logs are handled only by the UID-aware check above. find /host/var/log -path /host/var/log/pods -prune -o -type f -name "*.gz" -mtime +"${LOG_RETENTION_DAYS}" -print -exec rm -f {} \; 2>/dev/null || true