fix(monitoring): measure Atlas availability honestly across telemetry gaps
The 2026-08-18 metrics-storage outage exposed two defects in the availability pipeline that distorted the figure in opposite directions at once. The Overview panel fell back to a live one-hour Traefik ratio whenever the yearly rollup sample went stale for 48h, and rendered it under the same "365d" title. When the rollup stopped publishing on 2026-08-18 the panel quietly swapped a 365-day measurement for a 60-minute one and read 99.74% instead of the recorded 99.95%. The fallback is removed: a stale rollup now renders no value, and a new atlas-availability-rollup-stale alert pages at 26h, well before the panel goes blank at 48h. The yearly ratio also silently excluded the 34-hour telemetry gap, because missing days contribute zero requests and zero failures. Absent data was read as "nothing happened" — had Atlas genuinely been down in that window, the figure would still have said 99.95%. Availability keeps its measured-days-only definition, which is correct, but coverage is now published alongside it and shown in a new panel, so a telemetry gap lowers disclosed coverage instead of vanishing. The title reads "365d window" to stop implying 365 days of data exist; request-v4 begins 2026-05-01. The rollup job reported healthy runs across a day and a half of lost publishes: a read-only VictoriaMetrics accepts an import and discards it. It now reads each sample back and fails loudly when the write did not survive. Not addressed here: availability is still measured from inside the platform via Traefik counters, so it cannot distinguish "Atlas down" from "telemetry down", and misses failures that never reach Traefik (DNS, TLS, node dead). An external synthetic prober is the real fix and needs a hosting decision.
This commit is contained in:
parent
a45a9b67fe
commit
2862594c62
@ -148,7 +148,7 @@ INFRA_REGEX = f"^({'|'.join(INFRA_PATTERNS)})$"
|
||||
CP_ALLOWED_NS = INFRA_REGEX
|
||||
LONGHORN_NODE_REGEX = "titan-1[2-9]|titan-2[2-4]"
|
||||
ALL_NODE_REGEX = "|".join(CONTROL_ALL + WORKER_NODES)
|
||||
GAUGE_WIDTHS = [4, 3, 3, 4, 3, 3, 4]
|
||||
GAUGE_WIDTHS = [3, 3, 3, 3, 3, 3, 3, 3]
|
||||
CONTROL_WORKLOADS_EXPR = (
|
||||
f'sum(kube_pod_info{{node=~"{CONTROL_REGEX}",namespace!~"{CP_ALLOWED_NS}"}}) or on() vector(0)'
|
||||
)
|
||||
@ -467,21 +467,18 @@ UPTIME_WINDOW = "365d"
|
||||
UPTIME_RECORDING_METRIC = (
|
||||
f'atlas:availability:ratio_{UPTIME_WINDOW}{{scope="atlas",definition="request-v4"}}'
|
||||
)
|
||||
AVAILABILITY_REQUESTS_1H_EXPR = (
|
||||
'sum(increase(traefik_entrypoint_requests_total{'
|
||||
'entrypoint="websecure",protocol="http",code=~"[1-5].."}[1h]))'
|
||||
)
|
||||
AVAILABILITY_FAILURES_1H_EXPR = (
|
||||
'sum(increase(traefik_entrypoint_requests_total{'
|
||||
'entrypoint="websecure",protocol="http",code=~"5.."}[1h]))'
|
||||
)
|
||||
UPTIME_LIVE_FALLBACK_EXPR = (
|
||||
f"(1 - (({AVAILABILITY_FAILURES_1H_EXPR} or on() vector(0)) / "
|
||||
f"clamp_min({AVAILABILITY_REQUESTS_1H_EXPR}, 1)))"
|
||||
)
|
||||
UPTIME_RECORDING_EXPR = (
|
||||
f"(last_over_time({UPTIME_RECORDING_METRIC}[48h]) "
|
||||
f"or on() {UPTIME_LIVE_FALLBACK_EXPR})"
|
||||
# The 2026-08-18 storage outage proved a fallback here lies twice at once:
|
||||
# when the yearly sample went stale the panel silently rendered the last hour
|
||||
# of Traefik traffic under a 365d title, while the unmeasured gap vanished
|
||||
# from the yearly ratio as if it had been observed. A stale rollup must render
|
||||
# no value and page (atlas-availability-rollup-stale), never substitute a
|
||||
# different measurement under the same label.
|
||||
UPTIME_RECORDING_EXPR = f"last_over_time({UPTIME_RECORDING_METRIC}[48h])"
|
||||
# Days of daily availability rollups the yearly figure actually rests on,
|
||||
# published by the same rollup job so freshness matches the ratio sample.
|
||||
UPTIME_COVERAGE_EXPR = (
|
||||
"last_over_time(atlas:availability:coverage_days_365d"
|
||||
'{scope="atlas",definition="request-v4"}[48h])'
|
||||
)
|
||||
|
||||
# Tie-breaker to deterministically pick one node per namespace when shares tie.
|
||||
@ -511,6 +508,18 @@ UPTIME_PERCENT_THRESHOLDS = {
|
||||
{"color": "blue", "value": 0.99999},
|
||||
],
|
||||
}
|
||||
# Coverage is a disclosure, not an SLO: a short history is honest, a shrinking
|
||||
# one means telemetry is being lost. The request-v4 definition was backfilled
|
||||
# to 2026-05-01, so the figure legitimately sits well below a full year.
|
||||
UPTIME_COVERAGE_THRESHOLDS = {
|
||||
"mode": "absolute",
|
||||
"steps": [
|
||||
{"color": "red", "value": None},
|
||||
{"color": "orange", "value": 7},
|
||||
{"color": "yellow", "value": 30},
|
||||
{"color": "green", "value": 90},
|
||||
],
|
||||
}
|
||||
PROBLEM_TABLE_EXPR = (
|
||||
"(time() - kube_pod_created{pod!=\"\"}) "
|
||||
"* on(namespace,pod) group_left(node) kube_pod_info "
|
||||
@ -1910,7 +1919,8 @@ OVERVIEW_PANEL_DESCRIPTIONS = {
|
||||
"Control Plane Ready": "Control-plane nodes currently Ready; full count is good, lower means Kubernetes core capacity is missing.",
|
||||
"Control Plane Workloads": "Non-core pods running on control-plane nodes; zero is good because control nodes should stay focused.",
|
||||
"Stuck Terminating": "Pods that Kubernetes cannot finish deleting; zero is good, growth means cleanup or storage may be stuck.",
|
||||
"Atlas Availability (365d)": "Request-weighted Atlas ingress availability; every server-side 5xx response counts as a failed request.",
|
||||
"Atlas Availability (365d window)": "Request-weighted Atlas ingress availability over measured days only; telemetry gaps are excluded from the ratio, never counted as downtime.",
|
||||
"Availability Coverage (days)": "Days of rollup history behind the availability figure; a telemetry gap lowers this instead of lowering availability.",
|
||||
"Problem Pods": "Current-service pods Pending for more than 15 minutes or in an actionable failed phase. Completed Jobs and retained Veles migration workloads are kept on drill-down dashboards but excluded here.",
|
||||
"CrashLoop / ImagePull": "Current-service pods stuck in CrashLoopBackOff or ImagePullBackOff for more than 15 minutes. Retained Veles migration workloads remain visible on the Pods dashboard.",
|
||||
"Workers Ready": "Worker nodes currently Ready; full count is good, lower means less place to run services.",
|
||||
@ -2143,7 +2153,7 @@ def build_overview():
|
||||
},
|
||||
{
|
||||
"id": 27,
|
||||
"title": "Atlas Availability (365d)",
|
||||
"title": "Atlas Availability (365d window)",
|
||||
"expr": UPTIME_PERCENT_EXPR,
|
||||
"kind": "stat",
|
||||
"thresholds": UPTIME_PERCENT_THRESHOLDS,
|
||||
@ -2151,7 +2161,19 @@ def build_overview():
|
||||
"decimals": 4,
|
||||
"text_mode": "value",
|
||||
"instant": True,
|
||||
"description": "Rolling request-weighted availability at the Atlas HTTPS ingress: responses below 500 divided by all HTTP responses. Every server-side 5xx is a real failed request; client 4xx responses count as served. Replica counts, Grafana health, and monitoring gaps are not inputs, so they cannot lower availability unless user traffic actually receives a 5xx. A daily rollup job publishes one annual sample from deduplicated daily totals; Grafana keeps it for up to 48 hours so one delayed retry cannot cause a fallback, and only uses the same one-hour request SLI before history exists.",
|
||||
"description": "Request-weighted availability at the Atlas HTTPS ingress over the trailing year: responses below 500 divided by all HTTP responses, read from the daily rollup sample. Every server-side 5xx is a real failed request; client 4xx responses count as served. Only measured days enter the ratio: a telemetry gap such as the 2026-08-18 metrics-storage outage is excluded from both sides, never converted into downtime or uptime, and the Availability Coverage panel states how many days the figure actually rests on. If the rollup stops publishing, this panel goes stale and renders no value instead of quietly substituting a one-hour live ratio under a yearly title; the atlas-availability-rollup-stale alert pages first.",
|
||||
},
|
||||
{
|
||||
"id": 36,
|
||||
"title": "Availability Coverage (days)",
|
||||
"expr": UPTIME_COVERAGE_EXPR,
|
||||
"kind": "stat",
|
||||
"thresholds": UPTIME_COVERAGE_THRESHOLDS,
|
||||
"unit": "none",
|
||||
"decimals": 0,
|
||||
"text_mode": "value",
|
||||
"instant": True,
|
||||
"description": "Days of daily rollups the availability figure to the left actually rests on, out of the 365-day window. A telemetry gap such as the 2026-08-18 metrics-storage outage lowers this number instead of moving availability, which is what keeps an observability failure from being reported as Atlas downtime. The request-v4 definition begins on 2026-05-01, so coverage stays below a full year until that history accumulates.",
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
|
||||
82
scripts/tests/test_availability_measurement.py
Normal file
82
scripts/tests/test_availability_measurement.py
Normal file
@ -0,0 +1,82 @@
|
||||
"""Keep the public availability figure honest about what it measured."""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import yaml
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
# On 2026-08-18 the metrics volume filled and the availability pipeline failed
|
||||
# in both directions at once: the stale yearly sample silently fell back to a
|
||||
# one-hour Traefik ratio wearing the 365d title, and the 34-hour telemetry gap
|
||||
# vanished from the yearly ratio as if it had been measured. These guardrails
|
||||
# pin the corrections.
|
||||
|
||||
|
||||
def _overview_panels() -> list[dict]:
|
||||
"""Return the generated Overview dashboard's top-level panels."""
|
||||
dashboard = json.loads(
|
||||
(REPO_ROOT / "services/monitoring/dashboards/atlas-overview.json").read_text()
|
||||
)
|
||||
return dashboard["panels"]
|
||||
|
||||
|
||||
def _panel(title_prefix: str) -> dict:
|
||||
return next(
|
||||
panel
|
||||
for panel in _overview_panels()
|
||||
if str(panel.get("title", "")).startswith(title_prefix)
|
||||
)
|
||||
|
||||
|
||||
def _panel_exprs(panel: dict) -> str:
|
||||
return " ".join(target.get("expr", "") for target in panel.get("targets", []))
|
||||
|
||||
|
||||
def _alert_rules() -> dict[str, dict]:
|
||||
"""Return every provisioned Grafana alert rule keyed by uid."""
|
||||
manifest = next(
|
||||
document
|
||||
for document in yaml.safe_load_all(
|
||||
(REPO_ROOT / "services/monitoring/grafana-alerting-config.yaml").read_text()
|
||||
)
|
||||
if document
|
||||
)
|
||||
groups = yaml.safe_load(manifest["data"]["rules.yaml"])["groups"]
|
||||
return {rule["uid"]: rule for group in groups for rule in group["rules"]}
|
||||
|
||||
|
||||
def test_availability_panel_never_swaps_measurements() -> None:
|
||||
"""A stale yearly rollup must render empty, not a 1h ratio in disguise."""
|
||||
panel = _panel("Atlas Availability")
|
||||
exprs = _panel_exprs(panel)
|
||||
assert "atlas:availability:ratio_365d" in exprs
|
||||
assert "traefik_entrypoint_requests_total" not in exprs
|
||||
|
||||
|
||||
def test_availability_coverage_is_disclosed() -> None:
|
||||
"""The overview must state how many days the yearly figure rests on."""
|
||||
panel = _panel("Availability Coverage")
|
||||
assert "atlas:availability:coverage_days_365d" in _panel_exprs(panel)
|
||||
|
||||
|
||||
def test_stale_availability_rollup_pages() -> None:
|
||||
"""A silent rollup once hid 34 hours of lost publishes; staleness must page."""
|
||||
rules = _alert_rules()
|
||||
stale = rules["atlas-availability-rollup-stale"]
|
||||
assert "atlas:availability:ratio_365d" in stale["data"][0]["model"]["expr"]
|
||||
assert stale["noDataState"] == "Alerting"
|
||||
assert stale["execErrState"] == "Alerting"
|
||||
assert stale["labels"]["severity"] == "warning"
|
||||
|
||||
|
||||
def test_rollup_proves_its_publish_survived() -> None:
|
||||
"""A read-only store accepts imports and drops them; the job must read back."""
|
||||
source = (
|
||||
REPO_ROOT / "services/monitoring/scripts/availability_rollup.py"
|
||||
).read_text()
|
||||
assert "def verify_stored" in source
|
||||
assert "verify_stored(OUTPUT_METRIC" in source
|
||||
assert "atlas:availability:coverage_days_365d" in source
|
||||
@ -43,10 +43,14 @@ def test_calculate_availability_uses_all_server_failures() -> None:
|
||||
|
||||
|
||||
def test_render_metric_publishes_only_the_request_v4_series() -> None:
|
||||
"""Render the single series selected by the Grafana panel."""
|
||||
"""Render the two series selected by the Grafana overview panels."""
|
||||
mod = load_module()
|
||||
|
||||
assert mod.render_metric(0.9995, 1234) == (
|
||||
assert mod.render_metric(mod.OUTPUT_METRIC, 0.9995, 1234) == (
|
||||
'atlas:availability:ratio_365d{definition="request-v4",scope="atlas",'
|
||||
'rollup="yearly"} 0.999500000000 1234\n'
|
||||
)
|
||||
assert mod.render_metric(mod.COVERAGE_METRIC, 109, 1234) == (
|
||||
'atlas:availability:coverage_days_365d{definition="request-v4",scope="atlas",'
|
||||
'rollup="yearly"} 109.000000000000 1234\n'
|
||||
)
|
||||
|
||||
@ -42,16 +42,18 @@ def test_node_filter_and_expr_helpers():
|
||||
def test_overview_availability_panel_uses_recorded_365d_rollup():
|
||||
mod = load_module()
|
||||
dashboard = mod.build_overview()
|
||||
panel = next(panel for panel in flatten_panels(dashboard["panels"]) if panel["id"] == 27)
|
||||
panels_by_id = {panel["id"]: panel for panel in flatten_panels(dashboard["panels"])}
|
||||
panel = panels_by_id[27]
|
||||
|
||||
assert panel["title"] == "Atlas Availability (365d)"
|
||||
assert panel["title"] == "Atlas Availability (365d window)"
|
||||
availability_expr = panel["targets"][0]["expr"]
|
||||
assert (
|
||||
'last_over_time(atlas:availability:ratio_365d{scope="atlas",definition="request-v4"}[48h])'
|
||||
in availability_expr
|
||||
availability_expr
|
||||
== 'last_over_time(atlas:availability:ratio_365d{scope="atlas",definition="request-v4"}[48h])'
|
||||
)
|
||||
assert 'code=~"5.."' in availability_expr
|
||||
assert 'code=~"[1-5].."' in availability_expr
|
||||
# A stale rollup must render nothing rather than silently substituting the
|
||||
# last hour of Traefik traffic under a yearly title, as it did on 2026-08-18.
|
||||
assert "traefik_entrypoint_requests_total" not in availability_expr
|
||||
assert "atlas:availability:failures_1d" not in availability_expr
|
||||
assert "atlas:availability:requests_1d" not in availability_expr
|
||||
assert "sum_over_time" not in availability_expr
|
||||
@ -59,9 +61,15 @@ def test_overview_availability_panel_uses_recorded_365d_rollup():
|
||||
assert "kube_deployment_status_replicas_available" not in availability_expr
|
||||
assert panel["targets"][0]["instant"] is True
|
||||
assert "Every server-side 5xx" in panel["description"]
|
||||
assert "Replica counts, Grafana health" in panel["description"]
|
||||
assert "daily rollup job publishes one annual sample" in panel["description"]
|
||||
assert "keeps it for up to 48 hours" in panel["description"]
|
||||
assert "never converted into downtime or uptime" in panel["description"]
|
||||
|
||||
coverage = panels_by_id[36]
|
||||
assert coverage["title"] == "Availability Coverage (days)"
|
||||
assert (
|
||||
coverage["targets"][0]["expr"]
|
||||
== 'last_over_time(atlas:availability:coverage_days_365d{scope="atlas",definition="request-v4"}[48h])'
|
||||
)
|
||||
assert "lowers this number instead of moving availability" in coverage["description"]
|
||||
|
||||
def test_overview_uses_readable_quality_power_and_gitops_panels():
|
||||
mod = load_module()
|
||||
|
||||
@ -17,7 +17,7 @@
|
||||
},
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 4,
|
||||
"w": 3,
|
||||
"x": 0,
|
||||
"y": 0
|
||||
},
|
||||
@ -72,7 +72,7 @@
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 3,
|
||||
"x": 4,
|
||||
"x": 3,
|
||||
"y": 0
|
||||
},
|
||||
"targets": [
|
||||
@ -148,7 +148,7 @@
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 3,
|
||||
"x": 7,
|
||||
"x": 6,
|
||||
"y": 0
|
||||
},
|
||||
"targets": [
|
||||
@ -216,20 +216,20 @@
|
||||
{
|
||||
"id": 27,
|
||||
"type": "stat",
|
||||
"title": "Atlas Availability (365d)",
|
||||
"title": "Atlas Availability (365d window)",
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "atlas-vm"
|
||||
},
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 4,
|
||||
"x": 10,
|
||||
"w": 3,
|
||||
"x": 9,
|
||||
"y": 0
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"expr": "(last_over_time(atlas:availability:ratio_365d{scope=\"atlas\",definition=\"request-v4\"}[48h]) or on() (1 - ((sum(increase(traefik_entrypoint_requests_total{entrypoint=\"websecure\",protocol=\"http\",code=~\"5..\"}[1h])) or on() vector(0)) / clamp_min(sum(increase(traefik_entrypoint_requests_total{entrypoint=\"websecure\",protocol=\"http\",code=~\"[1-5]..\"}[1h])), 1))))",
|
||||
"expr": "last_over_time(atlas:availability:ratio_365d{scope=\"atlas\",definition=\"request-v4\"}[48h])",
|
||||
"refId": "A",
|
||||
"instant": true
|
||||
}
|
||||
@ -286,7 +286,78 @@
|
||||
},
|
||||
"textMode": "value"
|
||||
},
|
||||
"description": "Rolling request-weighted availability at the Atlas HTTPS ingress: responses below 500 divided by all HTTP responses. Every server-side 5xx is a real failed request; client 4xx responses count as served. Replica counts, Grafana health, and monitoring gaps are not inputs, so they cannot lower availability unless user traffic actually receives a 5xx. A daily rollup job publishes one annual sample from deduplicated daily totals; Grafana keeps it for up to 48 hours so one delayed retry cannot cause a fallback, and only uses the same one-hour request SLI before history exists."
|
||||
"description": "Request-weighted availability at the Atlas HTTPS ingress over the trailing year: responses below 500 divided by all HTTP responses, read from the daily rollup sample. Every server-side 5xx is a real failed request; client 4xx responses count as served. Only measured days enter the ratio: a telemetry gap such as the 2026-08-18 metrics-storage outage is excluded from both sides, never converted into downtime or uptime, and the Availability Coverage panel states how many days the figure actually rests on. If the rollup stops publishing, this panel goes stale and renders no value instead of quietly substituting a one-hour live ratio under a yearly title; the atlas-availability-rollup-stale alert pages first."
|
||||
},
|
||||
{
|
||||
"id": 36,
|
||||
"type": "stat",
|
||||
"title": "Availability Coverage (days)",
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "atlas-vm"
|
||||
},
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 3,
|
||||
"x": 12,
|
||||
"y": 0
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"expr": "last_over_time(atlas:availability:coverage_days_365d{scope=\"atlas\",definition=\"request-v4\"}[48h])",
|
||||
"refId": "A",
|
||||
"instant": true
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"color": {
|
||||
"mode": "thresholds"
|
||||
},
|
||||
"mappings": [],
|
||||
"thresholds": {
|
||||
"mode": "absolute",
|
||||
"steps": [
|
||||
{
|
||||
"color": "dark-red",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"color": "dark-orange",
|
||||
"value": 7
|
||||
},
|
||||
{
|
||||
"color": "dark-yellow",
|
||||
"value": 30
|
||||
},
|
||||
{
|
||||
"color": "dark-green",
|
||||
"value": 90
|
||||
}
|
||||
]
|
||||
},
|
||||
"unit": "none",
|
||||
"custom": {
|
||||
"displayMode": "auto"
|
||||
},
|
||||
"decimals": 0
|
||||
},
|
||||
"overrides": []
|
||||
},
|
||||
"options": {
|
||||
"colorMode": "value",
|
||||
"graphMode": "area",
|
||||
"justifyMode": "center",
|
||||
"reduceOptions": {
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
],
|
||||
"fields": "",
|
||||
"values": false
|
||||
},
|
||||
"textMode": "value"
|
||||
},
|
||||
"description": "Days of daily rollups the availability figure to the left actually rests on, out of the 365-day window. A telemetry gap such as the 2026-08-18 metrics-storage outage lowers this number instead of moving availability, which is what keeps an observability failure from being reported as Atlas downtime. The request-v4 definition begins on 2026-05-01, so coverage stays below a full year until that history accumulates."
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
@ -299,7 +370,7 @@
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 3,
|
||||
"x": 14,
|
||||
"x": 15,
|
||||
"y": 0
|
||||
},
|
||||
"targets": [
|
||||
@ -375,7 +446,7 @@
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 3,
|
||||
"x": 17,
|
||||
"x": 18,
|
||||
"y": 0
|
||||
},
|
||||
"targets": [
|
||||
@ -450,8 +521,8 @@
|
||||
},
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 4,
|
||||
"x": 20,
|
||||
"w": 3,
|
||||
"x": 21,
|
||||
"y": 0
|
||||
},
|
||||
"targets": [
|
||||
|
||||
@ -398,6 +398,58 @@ data:
|
||||
summary: "VictoriaMetrics stored no samples for 15m; every dashboard is about to read empty"
|
||||
labels:
|
||||
severity: critical
|
||||
# The rollup CronJob publishes one yearly availability sample per
|
||||
# day, and the Overview panel refuses to substitute another
|
||||
# measurement when that sample goes stale. Staleness therefore has
|
||||
# to page before the panel goes blank at the 48h lookback.
|
||||
- uid: atlas-availability-rollup-stale
|
||||
title: "Atlas availability rollup is stale (>26h)"
|
||||
condition: C
|
||||
for: "30m"
|
||||
data:
|
||||
- refId: A
|
||||
relativeTimeRange:
|
||||
from: 600
|
||||
to: 0
|
||||
datasourceUid: atlas-vm
|
||||
model:
|
||||
intervalMs: 60000
|
||||
maxDataPoints: 43200
|
||||
expr: (time() - tlast_over_time(atlas:availability:ratio_365d{scope="atlas",definition="request-v4"}[3d])) or on() vector(999999)
|
||||
legendFormat: seconds since publish
|
||||
datasource:
|
||||
type: prometheus
|
||||
uid: atlas-vm
|
||||
- refId: B
|
||||
datasourceUid: __expr__
|
||||
model:
|
||||
expression: A
|
||||
intervalMs: 60000
|
||||
maxDataPoints: 43200
|
||||
reducer: last
|
||||
type: reduce
|
||||
- refId: C
|
||||
datasourceUid: __expr__
|
||||
model:
|
||||
expression: B
|
||||
intervalMs: 60000
|
||||
maxDataPoints: 43200
|
||||
type: threshold
|
||||
conditions:
|
||||
- evaluator:
|
||||
params: [93600]
|
||||
type: gt
|
||||
operator:
|
||||
type: and
|
||||
reducer:
|
||||
type: last
|
||||
type: query
|
||||
noDataState: Alerting
|
||||
execErrState: Alerting
|
||||
annotations:
|
||||
summary: "Atlas availability rollup has not published in >26h; the Overview availability panel goes blank at 48h"
|
||||
labels:
|
||||
severity: warning
|
||||
- orgId: 1
|
||||
name: maintenance
|
||||
folder: Alerts
|
||||
|
||||
@ -26,7 +26,7 @@ data:
|
||||
},
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 4,
|
||||
"w": 3,
|
||||
"x": 0,
|
||||
"y": 0
|
||||
},
|
||||
@ -81,7 +81,7 @@ data:
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 3,
|
||||
"x": 4,
|
||||
"x": 3,
|
||||
"y": 0
|
||||
},
|
||||
"targets": [
|
||||
@ -157,7 +157,7 @@ data:
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 3,
|
||||
"x": 7,
|
||||
"x": 6,
|
||||
"y": 0
|
||||
},
|
||||
"targets": [
|
||||
@ -225,20 +225,20 @@ data:
|
||||
{
|
||||
"id": 27,
|
||||
"type": "stat",
|
||||
"title": "Atlas Availability (365d)",
|
||||
"title": "Atlas Availability (365d window)",
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "atlas-vm"
|
||||
},
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 4,
|
||||
"x": 10,
|
||||
"w": 3,
|
||||
"x": 9,
|
||||
"y": 0
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"expr": "(last_over_time(atlas:availability:ratio_365d{scope=\"atlas\",definition=\"request-v4\"}[48h]) or on() (1 - ((sum(increase(traefik_entrypoint_requests_total{entrypoint=\"websecure\",protocol=\"http\",code=~\"5..\"}[1h])) or on() vector(0)) / clamp_min(sum(increase(traefik_entrypoint_requests_total{entrypoint=\"websecure\",protocol=\"http\",code=~\"[1-5]..\"}[1h])), 1))))",
|
||||
"expr": "last_over_time(atlas:availability:ratio_365d{scope=\"atlas\",definition=\"request-v4\"}[48h])",
|
||||
"refId": "A",
|
||||
"instant": true
|
||||
}
|
||||
@ -295,7 +295,78 @@ data:
|
||||
},
|
||||
"textMode": "value"
|
||||
},
|
||||
"description": "Rolling request-weighted availability at the Atlas HTTPS ingress: responses below 500 divided by all HTTP responses. Every server-side 5xx is a real failed request; client 4xx responses count as served. Replica counts, Grafana health, and monitoring gaps are not inputs, so they cannot lower availability unless user traffic actually receives a 5xx. A daily rollup job publishes one annual sample from deduplicated daily totals; Grafana keeps it for up to 48 hours so one delayed retry cannot cause a fallback, and only uses the same one-hour request SLI before history exists."
|
||||
"description": "Request-weighted availability at the Atlas HTTPS ingress over the trailing year: responses below 500 divided by all HTTP responses, read from the daily rollup sample. Every server-side 5xx is a real failed request; client 4xx responses count as served. Only measured days enter the ratio: a telemetry gap such as the 2026-08-18 metrics-storage outage is excluded from both sides, never converted into downtime or uptime, and the Availability Coverage panel states how many days the figure actually rests on. If the rollup stops publishing, this panel goes stale and renders no value instead of quietly substituting a one-hour live ratio under a yearly title; the atlas-availability-rollup-stale alert pages first."
|
||||
},
|
||||
{
|
||||
"id": 36,
|
||||
"type": "stat",
|
||||
"title": "Availability Coverage (days)",
|
||||
"datasource": {
|
||||
"type": "prometheus",
|
||||
"uid": "atlas-vm"
|
||||
},
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 3,
|
||||
"x": 12,
|
||||
"y": 0
|
||||
},
|
||||
"targets": [
|
||||
{
|
||||
"expr": "last_over_time(atlas:availability:coverage_days_365d{scope=\"atlas\",definition=\"request-v4\"}[48h])",
|
||||
"refId": "A",
|
||||
"instant": true
|
||||
}
|
||||
],
|
||||
"fieldConfig": {
|
||||
"defaults": {
|
||||
"color": {
|
||||
"mode": "thresholds"
|
||||
},
|
||||
"mappings": [],
|
||||
"thresholds": {
|
||||
"mode": "absolute",
|
||||
"steps": [
|
||||
{
|
||||
"color": "dark-red",
|
||||
"value": null
|
||||
},
|
||||
{
|
||||
"color": "dark-orange",
|
||||
"value": 7
|
||||
},
|
||||
{
|
||||
"color": "dark-yellow",
|
||||
"value": 30
|
||||
},
|
||||
{
|
||||
"color": "dark-green",
|
||||
"value": 90
|
||||
}
|
||||
]
|
||||
},
|
||||
"unit": "none",
|
||||
"custom": {
|
||||
"displayMode": "auto"
|
||||
},
|
||||
"decimals": 0
|
||||
},
|
||||
"overrides": []
|
||||
},
|
||||
"options": {
|
||||
"colorMode": "value",
|
||||
"graphMode": "area",
|
||||
"justifyMode": "center",
|
||||
"reduceOptions": {
|
||||
"calcs": [
|
||||
"lastNotNull"
|
||||
],
|
||||
"fields": "",
|
||||
"values": false
|
||||
},
|
||||
"textMode": "value"
|
||||
},
|
||||
"description": "Days of daily rollups the availability figure to the left actually rests on, out of the 365-day window. A telemetry gap such as the 2026-08-18 metrics-storage outage lowers this number instead of moving availability, which is what keeps an observability failure from being reported as Atlas downtime. The request-v4 definition begins on 2026-05-01, so coverage stays below a full year until that history accumulates."
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
@ -308,7 +379,7 @@ data:
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 3,
|
||||
"x": 14,
|
||||
"x": 15,
|
||||
"y": 0
|
||||
},
|
||||
"targets": [
|
||||
@ -384,7 +455,7 @@ data:
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 3,
|
||||
"x": 17,
|
||||
"x": 18,
|
||||
"y": 0
|
||||
},
|
||||
"targets": [
|
||||
@ -459,8 +530,8 @@ data:
|
||||
},
|
||||
"gridPos": {
|
||||
"h": 5,
|
||||
"w": 4,
|
||||
"x": 20,
|
||||
"w": 3,
|
||||
"x": 21,
|
||||
"y": 0
|
||||
},
|
||||
"targets": [
|
||||
|
||||
@ -20,7 +20,12 @@ DEFINITION = "request-v4"
|
||||
REQUESTS_METRIC = "atlas:availability:requests_1d"
|
||||
FAILURES_METRIC = "atlas:availability:failures_1d"
|
||||
OUTPUT_METRIC = "atlas:availability:ratio_365d"
|
||||
COVERAGE_METRIC = "atlas:availability:coverage_days_365d"
|
||||
WINDOW_DAYS = 365
|
||||
# Freshly imported samples sit in the in-memory buffer briefly before they
|
||||
# become searchable, so the read-back check retries instead of failing fast.
|
||||
VERIFY_ATTEMPTS = 6
|
||||
VERIFY_DELAY_SECONDS = 10
|
||||
|
||||
|
||||
def parse_export(lines: Iterable[bytes]) -> dict[int, float]:
|
||||
@ -35,8 +40,8 @@ def parse_export(lines: Iterable[bytes]) -> dict[int, float]:
|
||||
return points
|
||||
|
||||
|
||||
def fetch_rollup(metric: str, start: datetime, end: datetime) -> dict[int, float]:
|
||||
"""Stream one compact rollup series from VictoriaMetrics."""
|
||||
def fetch_series(metric: str, start: datetime, end: datetime) -> dict[int, float]:
|
||||
"""Stream one compact series from VictoriaMetrics."""
|
||||
matcher = (
|
||||
f'{{__name__="{metric}",scope="{SCOPE}",definition="{DEFINITION}"}}'
|
||||
)
|
||||
@ -60,19 +65,19 @@ def calculate_availability(requests: float, failures: float) -> float:
|
||||
return max(0.0, min(1.0, 1.0 - (failures / requests)))
|
||||
|
||||
|
||||
def render_metric(value: float, timestamp_ms: int) -> str:
|
||||
def render_metric(metric: str, value: float, timestamp_ms: int) -> str:
|
||||
"""Render one VictoriaMetrics Prometheus-import sample."""
|
||||
return (
|
||||
f'{OUTPUT_METRIC}{{definition="{DEFINITION}",scope="{SCOPE}",'
|
||||
f'{metric}{{definition="{DEFINITION}",scope="{SCOPE}",'
|
||||
f'rollup="yearly"}} {value:.12f} {timestamp_ms}\n'
|
||||
)
|
||||
|
||||
|
||||
def publish(value: float, timestamp_ms: int) -> None:
|
||||
"""Write the calculated annual ratio to VictoriaMetrics."""
|
||||
def publish(metric: str, value: float, timestamp_ms: int) -> None:
|
||||
"""Write one calculated yearly sample to VictoriaMetrics."""
|
||||
request = Request(
|
||||
f"{VM_URL}/api/v1/import/prometheus",
|
||||
data=render_metric(value, timestamp_ms).encode(),
|
||||
data=render_metric(metric, value, timestamp_ms).encode(),
|
||||
headers={"Content-Type": "text/plain"},
|
||||
method="POST",
|
||||
)
|
||||
@ -81,21 +86,53 @@ def publish(value: float, timestamp_ms: int) -> None:
|
||||
raise RuntimeError(f"VictoriaMetrics import returned HTTP {response.status}")
|
||||
|
||||
|
||||
def verify_stored(metric: str, value: float, timestamp_ms: int) -> None:
|
||||
"""Prove the sample landed; a read-only store accepts writes and drops them.
|
||||
|
||||
During the 2026-08-18 storage outage every import returned success while
|
||||
VictoriaMetrics silently discarded the samples, so this job reported
|
||||
healthy runs across a day and a half of lost publishes. Reading the sample
|
||||
back is the only evidence the write survived.
|
||||
"""
|
||||
when = datetime.fromtimestamp(timestamp_ms / 1000, tz=timezone.utc)
|
||||
for attempt in range(VERIFY_ATTEMPTS):
|
||||
if attempt:
|
||||
time.sleep(VERIFY_DELAY_SECONDS)
|
||||
points = fetch_series(
|
||||
metric, when - timedelta(minutes=5), when + timedelta(minutes=5)
|
||||
)
|
||||
stored = points.get(timestamp_ms)
|
||||
if stored is not None and abs(stored - value) < 1e-9:
|
||||
return
|
||||
raise RuntimeError(
|
||||
f"{metric} sample at {timestamp_ms} is not readable after publish; "
|
||||
"VictoriaMetrics accepted and then discarded the write "
|
||||
"(is /storage read-only?)"
|
||||
)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
"""Rebuild and publish the rolling request-availability sample."""
|
||||
"""Rebuild and publish the rolling request-availability samples."""
|
||||
end = datetime.now(timezone.utc)
|
||||
start = end - timedelta(days=WINDOW_DAYS)
|
||||
requests = sum(fetch_rollup(REQUESTS_METRIC, start, end).values())
|
||||
failures = sum(fetch_rollup(FAILURES_METRIC, start, end).values())
|
||||
availability = calculate_availability(requests, failures)
|
||||
requests = fetch_series(REQUESTS_METRIC, start, end)
|
||||
failures = fetch_series(FAILURES_METRIC, start, end)
|
||||
availability = calculate_availability(
|
||||
sum(requests.values()), sum(failures.values())
|
||||
)
|
||||
coverage_days = float(len(requests))
|
||||
timestamp_ms = time.time_ns() // 1_000_000
|
||||
publish(availability, timestamp_ms)
|
||||
publish(OUTPUT_METRIC, availability, timestamp_ms)
|
||||
publish(COVERAGE_METRIC, coverage_days, timestamp_ms)
|
||||
verify_stored(OUTPUT_METRIC, availability, timestamp_ms)
|
||||
verify_stored(COVERAGE_METRIC, coverage_days, timestamp_ms)
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"requests": requests,
|
||||
"failures": failures,
|
||||
"availability_percent": availability * 100,
|
||||
"coverage_days": coverage_days,
|
||||
"failures": sum(failures.values()),
|
||||
"requests": sum(requests.values()),
|
||||
"timestamp_ms": timestamp_ms,
|
||||
},
|
||||
sort_keys=True,
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user