monitoring: avoid unstable volume hosts

This commit is contained in:
jenkins 2026-08-04 18:06:43 -03:00
parent 0c108860ab
commit bbab7330ce
7 changed files with 64 additions and 17 deletions

View File

@ -456,7 +456,9 @@ STUCK_TERMINATING_EXPR = (
) )
UPTIME_WINDOW = "365d" UPTIME_WINDOW = "365d"
# vmalert precomputes the expensive long-window rollup so Grafana only reads one compact series. # vmalert precomputes the expensive long-window rollup so Grafana only reads one compact series.
UPTIME_RECORDING_METRIC = f'atlas:availability:ratio_{UPTIME_WINDOW}{{scope="atlas"}}' UPTIME_RECORDING_METRIC = (
f'atlas:availability:ratio_{UPTIME_WINDOW}{{scope="atlas",definition="serving-v2"}}'
)
UPTIME_RECORDING_EXPR = f"last_over_time({UPTIME_RECORDING_METRIC}[24h])" UPTIME_RECORDING_EXPR = f"last_over_time({UPTIME_RECORDING_METRIC}[24h])"
TRAEFIK_READY_EXPR = ( TRAEFIK_READY_EXPR = (
"(" "("
@ -1898,7 +1900,7 @@ OVERVIEW_PANEL_DESCRIPTIONS = {
"Control Plane Ready": "Control-plane nodes currently Ready; full count is good, lower means Kubernetes core capacity is missing.", "Control Plane Ready": "Control-plane nodes currently Ready; full count is good, lower means Kubernetes core capacity is missing.",
"Control Plane Workloads": "Non-core pods running on control-plane nodes; zero is good because control nodes should stay focused.", "Control Plane Workloads": "Non-core pods running on control-plane nodes; zero is good because control nodes should stay focused.",
"Stuck Terminating": "Pods that Kubernetes cannot finish deleting; zero is good, growth means cleanup or storage may be stuck.", "Stuck Terminating": "Pods that Kubernetes cannot finish deleting; zero is good, growth means cleanup or storage may be stuck.",
"Atlas Availability (365d)": "Rolling one-year Atlas availability; higher is better, below target means users saw downtime.", "Atlas Serving Availability": "Observed availability with control-plane quorum and at least one Traefik replica serving; partial replica capacity remains available.",
"Problem Pods": "Current-service pods Pending for more than 15 minutes or in an actionable failed phase. Completed Jobs and retained Veles migration workloads are kept on drill-down dashboards but excluded here.", "Problem Pods": "Current-service pods Pending for more than 15 minutes or in an actionable failed phase. Completed Jobs and retained Veles migration workloads are kept on drill-down dashboards but excluded here.",
"CrashLoop / ImagePull": "Current-service pods stuck in CrashLoopBackOff or ImagePullBackOff for more than 15 minutes. Retained Veles migration workloads remain visible on the Pods dashboard.", "CrashLoop / ImagePull": "Current-service pods stuck in CrashLoopBackOff or ImagePullBackOff for more than 15 minutes. Retained Veles migration workloads remain visible on the Pods dashboard.",
"Workers Ready": "Worker nodes currently Ready; full count is good, lower means less place to run services.", "Workers Ready": "Worker nodes currently Ready; full count is good, lower means less place to run services.",
@ -2131,7 +2133,7 @@ def build_overview():
}, },
{ {
"id": 27, "id": 27,
"title": "Atlas Availability (365d)", "title": "Atlas Serving Availability",
"expr": UPTIME_PERCENT_EXPR, "expr": UPTIME_PERCENT_EXPR,
"kind": "stat", "kind": "stat",
"thresholds": UPTIME_PERCENT_THRESHOLDS, "thresholds": UPTIME_PERCENT_THRESHOLDS,
@ -2139,7 +2141,7 @@ def build_overview():
"decimals": 4, "decimals": 4,
"text_mode": "value", "text_mode": "value",
"instant": True, "instant": True,
"description": "Rolling 365-day availability from vmalert's precomputed atlas:availability:ratio_365d series. Grafana keeps the last successful rollup for up to 24h so one missed long-window evaluation does not render as No data.", "description": "Availability over observed samples, up to 365 days: at least two control-plane nodes Ready and at least one Traefik replica serving. Partial replica capacity remains available; monitoring gaps are excluded. Grafana keeps the last successful rollup for up to 24h.",
}, },
{ {
"id": 4, "id": 4,

View File

@ -57,11 +57,14 @@ def test_overview_availability_panel_uses_recorded_365d_rollup():
dashboard = mod.build_overview() dashboard = mod.build_overview()
panel = next(panel for panel in flatten_panels(dashboard["panels"]) if panel["id"] == 27) panel = next(panel for panel in flatten_panels(dashboard["panels"]) if panel["id"] == 27)
assert panel["title"] == "Atlas Availability (365d)" assert panel["title"] == "Atlas Serving Availability"
assert panel["targets"][0]["expr"] == 'last_over_time(atlas:availability:ratio_365d{scope="atlas"}[24h])' assert panel["targets"][0]["expr"] == (
'last_over_time(atlas:availability:ratio_365d{scope="atlas",definition="serving-v2"}[24h])'
)
assert panel["targets"][0]["instant"] is True assert panel["targets"][0]["instant"] is True
assert "precomputed" in panel["description"] assert "at least one Traefik replica serving" in panel["description"]
assert "last successful rollup for up to 24h" in panel["description"] assert "Partial replica capacity remains available" in panel["description"]
assert "monitoring gaps are excluded" in panel["description"]
def test_overview_uses_readable_quality_power_and_gitops_panels(): def test_overview_uses_readable_quality_power_and_gitops_panels():

View File

@ -31,6 +31,18 @@ def test_victoria_metrics_has_dashboard_burst_headroom() -> None:
assert server["resources"]["limits"]["cpu"] == "2" assert server["resources"]["limits"]["cpu"] == "2"
assert server["resources"]["limits"]["memory"] == "4Gi" assert server["resources"]["limits"]["memory"] == "4Gi"
required_terms = server["affinity"]["nodeAffinity"][
"requiredDuringSchedulingIgnoredDuringExecution"
]["nodeSelectorTerms"]
hostname_rule = next(
expression
for term in required_terms
for expression in term["matchExpressions"]
if expression["key"] == "kubernetes.io/hostname"
)
assert hostname_rule["operator"] == "NotIn"
assert {"titan-14", "titan-18"} <= set(hostname_rule["values"])
def test_yearly_availability_reuses_the_hourly_rollup() -> None: def test_yearly_availability_reuses_the_hourly_rollup() -> None:
"""Prevent the yearly rule from rescanning raw cluster metrics for 365 days.""" """Prevent the yearly rule from rescanning raw cluster metrics for 365 days."""
@ -45,10 +57,29 @@ def test_yearly_availability_reuses_the_hourly_rollup() -> None:
) )
assert "atlas:availability:ratio_1h" in yearly["expr"] assert "atlas:availability:ratio_1h" in yearly["expr"]
assert 'definition="serving-v2"' in yearly["expr"]
assert "kube_node_status_condition" not in yearly["expr"] assert "kube_node_status_condition" not in yearly["expr"]
assert "[365d:1h]" not in yearly["expr"] assert "[365d:1h]" not in yearly["expr"]
def test_availability_counts_serving_state_instead_of_replica_capacity() -> None:
"""Treat quorum and any serving ingress replica as available binary states."""
manifest = _documents(
REPO_ROOT / "services/monitoring/vmalert-atlas-availability.yaml"
)[0]
rules = yaml.safe_load(manifest["data"]["atlas-availability.yaml"])["groups"][0]
hourly = next(
rule
for rule in rules["rules"]
if rule["record"] == "atlas:availability:ratio_1h"
)
assert ">= bool 2" in hourly["expr"]
assert "> bool 0" in hourly["expr"]
assert "/ 3" not in hourly["expr"]
assert hourly["labels"]["definition"] == "serving-v2"
def test_quality_rollups_do_not_run_every_minute() -> None: def test_quality_rollups_do_not_run_every_minute() -> None:
"""Keep high-cardinality quality rollups below the backend saturation cadence.""" """Keep high-cardinality quality rollups below the backend saturation cadence."""
manifest = _documents( manifest = _documents(

View File

@ -216,7 +216,7 @@
{ {
"id": 27, "id": 27,
"type": "stat", "type": "stat",
"title": "Atlas Availability (365d)", "title": "Atlas Serving Availability",
"datasource": { "datasource": {
"type": "prometheus", "type": "prometheus",
"uid": "atlas-vm" "uid": "atlas-vm"
@ -229,7 +229,7 @@
}, },
"targets": [ "targets": [
{ {
"expr": "last_over_time(atlas:availability:ratio_365d{scope=\"atlas\"}[24h])", "expr": "last_over_time(atlas:availability:ratio_365d{scope=\"atlas\",definition=\"serving-v2\"}[24h])",
"refId": "A", "refId": "A",
"instant": true "instant": true
} }
@ -286,7 +286,7 @@
}, },
"textMode": "value" "textMode": "value"
}, },
"description": "Rolling 365-day availability from vmalert's precomputed atlas:availability:ratio_365d series. Grafana keeps the last successful rollup for up to 24h so one missed long-window evaluation does not render as No data." "description": "Availability over observed samples, up to 365 days: at least two control-plane nodes Ready and at least one Traefik replica serving. Partial replica capacity remains available; monitoring gaps are excluded. Grafana keeps the last successful rollup for up to 24h."
}, },
{ {
"id": 4, "id": 4,

View File

@ -225,7 +225,7 @@ data:
{ {
"id": 27, "id": 27,
"type": "stat", "type": "stat",
"title": "Atlas Availability (365d)", "title": "Atlas Serving Availability",
"datasource": { "datasource": {
"type": "prometheus", "type": "prometheus",
"uid": "atlas-vm" "uid": "atlas-vm"
@ -238,7 +238,7 @@ data:
}, },
"targets": [ "targets": [
{ {
"expr": "last_over_time(atlas:availability:ratio_365d{scope=\"atlas\"}[24h])", "expr": "last_over_time(atlas:availability:ratio_365d{scope=\"atlas\",definition=\"serving-v2\"}[24h])",
"refId": "A", "refId": "A",
"instant": true "instant": true
} }
@ -295,7 +295,7 @@ data:
}, },
"textMode": "value" "textMode": "value"
}, },
"description": "Rolling 365-day availability from vmalert's precomputed atlas:availability:ratio_365d series. Grafana keeps the last successful rollup for up to 24h so one missed long-window evaluation does not render as No data." "description": "Availability over observed samples, up to 365 days: at least two control-plane nodes Ready and at least one Traefik replica serving. Partial replica capacity remains available; monitoring gaps are excluded. Grafana keeps the last successful rollup for up to 24h."
}, },
{ {
"id": 4, "id": 4,

View File

@ -111,6 +111,10 @@ spec:
- key: kubernetes.io/hostname - key: kubernetes.io/hostname
operator: NotIn operator: NotIn
values: values:
# Longhorn engine-image probes repeatedly fail on these
# attachment hosts and can surface volume I/O errors.
- titan-14
- titan-18
- titan-20 - titan-20
- titan-21 - titan-21
- titan-22 - titan-22
@ -130,6 +134,11 @@ spec:
operator: In operator: In
values: values:
- rpi4 - rpi4
- weight: 25
preference:
matchExpressions:
- key: node-role.kubernetes.io/storage-backbone
operator: Exists
# Enable built-in Kubernetes scraping # Enable built-in Kubernetes scraping
scrape: scrape:

View File

@ -17,23 +17,25 @@ data:
min( min(
( (
sum(kube_node_status_condition{condition="Ready",status="true",node=~"titan-0a|titan-0b|titan-0c"}) sum(kube_node_status_condition{condition="Ready",status="true",node=~"titan-0a|titan-0b|titan-0c"})
/ 3 >= bool 2
), ),
( (
sum(kube_deployment_status_replicas_available{namespace=~"traefik|kube-system",deployment="traefik"}) sum(kube_deployment_status_replicas_available{namespace=~"traefik|kube-system",deployment="traefik"})
/ clamp_min(sum(kube_deployment_spec_replicas{namespace=~"traefik|kube-system",deployment="traefik"}), 1) > bool 0
) )
) )
)[1h:5m]) )[1h:5m])
labels: labels:
definition: serving-v2
scope: atlas scope: atlas
rollup: hourly rollup: hourly
- record: atlas:availability:ratio_365d - record: atlas:availability:ratio_365d
expr: | expr: |
avg_over_time( avg_over_time(
atlas:availability:ratio_1h{scope="atlas"}[365d] atlas:availability:ratio_1h{scope="atlas",definition="serving-v2"}[365d]
) )
labels: labels:
definition: serving-v2
scope: atlas scope: atlas
rollup: yearly rollup: yearly
platform-quality.yaml: | platform-quality.yaml: |