atlas-iac/services/maintenance/node-ops/node-image-sweeper-daemonset.yaml

86 lines
2.4 KiB
YAML

# services/maintenance/node-ops/node-image-sweeper-daemonset.yaml
apiVersion: apps/v1
kind: DaemonSet
metadata:
name: node-image-sweeper
namespace: maintenance
spec:
selector:
matchLabels:
app: node-image-sweeper
updateStrategy:
type: RollingUpdate
rollingUpdate:
# Startup readiness waits for the first bounded cleanup to finish.
maxUnavailable: 1
template:
metadata:
annotations:
atlas.bstein.dev/config-revision: "2026-10-04-native-image-gc"
labels:
app: node-image-sweeper
spec:
serviceAccountName: node-image-sweeper
tolerations:
- key: node-role.kubernetes.io/control-plane
operator: Exists
effect: NoSchedule
- key: node-role.kubernetes.io/master
operator: Exists
effect: NoSchedule
- key: veles.bstein.dev/simulation
operator: Equal
value: "true"
effect: NoSchedule
- key: cassandra.bstein.dev/simulation
operator: Equal
value: "true"
effect: NoSchedule
nodeSelector:
kubernetes.io/os: linux
containers:
- name: node-image-sweeper
image: python:3.12.9-alpine3.20
command: ["/bin/sh", "/scripts/node_image_sweeper.sh"]
startupProbe:
exec:
command: ["/bin/sh", "-c", "test -f /tmp/initial-sweep-complete"]
periodSeconds: 5
timeoutSeconds: 5
failureThreshold: 120
env:
- name: SWEEP_INTERVAL_SEC
value: "7200"
- name: EMERGENCY_USAGE_PERCENT
value: "80"
- name: LOG_RETENTION_DAYS
value: "7"
- name: ORPHAN_POD_RETENTION_DAYS
value: "3"
- name: JOURNAL_MAX_SIZE
value: "200M"
securityContext:
privileged: true
runAsUser: 0
resources:
requests:
cpu: 10m
memory: 32Mi
limits:
cpu: 100m
memory: 256Mi
volumeMounts:
- name: host-root
mountPath: /host
- name: script
mountPath: /scripts
readOnly: true
volumes:
- name: host-root
hostPath:
path: /
- name: script
configMap:
name: node-image-sweeper-script
defaultMode: 0555