atlas-iac/services/monitoring/availability-backfill-v3-job.yaml
2026-08-04 21:02:19 -03:00

93 lines
2.7 KiB
YAML

# services/monitoring/availability-backfill-v3-job.yaml
apiVersion: v1
kind: ConfigMap
metadata:
name: atlas-availability-gateway-v3-backfill-rules
namespace: monitoring
data:
atlas-gateway-history.yaml: |
groups:
- name: atlas.availability.gateway.backfill
interval: 1h
rules:
- record: atlas:availability:gateway_requests_1h
expr: |
sum(increase(
traefik_entrypoint_requests_total{
entrypoint="websecure",
protocol="http",
code=~"[1-5].."
}[1h]
))
labels:
definition: gateway-v3
scope: atlas
rollup: hourly
- record: atlas:availability:gateway_failures_1h
expr: |
sum(increase(
traefik_entrypoint_requests_total{
entrypoint="websecure",
protocol="http",
code=~"502|503|504"
}[1h]
))
labels:
definition: gateway-v3
scope: atlas
rollup: hourly
---
apiVersion: batch/v1
kind: Job
metadata:
name: atlas-availability-gateway-v3-backfill
namespace: monitoring
spec:
backoffLimit: 2
template:
metadata:
labels:
app: atlas-availability-gateway-v3-backfill
spec:
restartPolicy: Never
affinity:
nodeAffinity:
requiredDuringSchedulingIgnoredDuringExecution:
nodeSelectorTerms:
- matchExpressions:
- key: kubernetes.io/hostname
operator: NotIn
values:
- titan-22
- titan-24
containers:
- name: vmalert-replay
image: victoriametrics/vmalert:v1.113.0
args:
- -datasource.url=http://victoria-metrics-single-server:8428
- -remoteWrite.url=http://victoria-metrics-single-server:8428
- -remoteWrite.flushInterval=1s
- -rule=/etc/vmalert/backfill/*.yaml
- -replay.timeFrom=2026-05-01T00:00:00Z
- -replay.timeTo=2026-08-04T23:00:00Z
- -replay.maxDatapointsPerQuery=48
- -replay.rulesDelay=2s
- -replay.disableProgressBar
resources:
requests:
cpu: 100m
memory: 128Mi
limits:
cpu: "1"
memory: 512Mi
volumeMounts:
- name: rules
mountPath: /etc/vmalert/backfill
readOnly: true
volumes:
- name: rules
configMap:
name: atlas-availability-gateway-v3-backfill-rules