"description":"Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(atlas_ai_quota_remaining_percent{provider=\"openai\",limit=\"codex\",window=\"seven_day\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"openai\"} == 1)) or on() vector(-1)"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Codex Spark Weekly Remaining",
"panel_id":2,
"panel_type":"stat",
"description":"Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(atlas_ai_quota_remaining_percent{provider=\"openai\",limit=\"gpt-5-3-codex-spark\",window=\"seven_day\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"openai\"} == 1)) or on() vector(-1)"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Claude 5h Remaining",
"panel_id":3,
"panel_type":"stat",
"description":"Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(atlas_ai_quota_remaining_percent{provider=\"anthropic\",window=\"five_hour\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"anthropic\"} == 1)) or on() vector(-1)"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Claude 7d Remaining",
"panel_id":4,
"panel_type":"stat",
"description":"Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(atlas_ai_quota_remaining_percent{provider=\"anthropic\",window=\"seven_day\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"anthropic\"} == 1)) or on() vector(-1)"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Quota Collectors Healthy",
"panel_id":5,
"panel_type":"stat",
"description":"Successful latest quota fetches. Providers are polled independently every five minutes.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sum(atlas_ai_quota_fetch_success) or on() vector(0)"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Oldest Quota Sample",
"panel_id":6,
"panel_type":"stat",
"description":"Age of the stalest successful provider quota snapshot.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"max((time() - atlas_ai_quota_last_success_timestamp_seconds) and (atlas_ai_quota_last_success_timestamp_seconds > 0)) or on() vector(-1)"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Codex Weekly Reset In",
"panel_id":7,
"panel_type":"stat",
"description":"Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(clamp_min(atlas_ai_quota_reset_timestamp_seconds{provider=\"openai\",limit=\"codex\",window=\"seven_day\"} - time(), 0) and on(provider) (atlas_ai_quota_fetch_success{provider=\"openai\"} == 1)) or on() vector(-1)"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Claude 5h Reset In",
"panel_id":8,
"panel_type":"stat",
"description":"Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(clamp_min(atlas_ai_quota_reset_timestamp_seconds{provider=\"anthropic\",window=\"five_hour\"} - time(), 0) and on(provider) (atlas_ai_quota_fetch_success{provider=\"anthropic\"} == 1)) or on() vector(-1)"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Claude 7d Reset In",
"panel_id":9,
"panel_type":"stat",
"description":"Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(clamp_min(atlas_ai_quota_reset_timestamp_seconds{provider=\"anthropic\",window=\"seven_day\"} - time(), 0) and on(provider) (atlas_ai_quota_fetch_success{provider=\"anthropic\"} == 1)) or on() vector(-1)"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Codex Tokens (Latest Day)",
"panel_id":10,
"panel_type":"stat",
"description":"Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(atlas_ai_account_tokens{provider=\"openai\",period=\"latest_day\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"openai\"} == 1)) or on() vector(-1)"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Codex Tokens (7d)",
"panel_id":11,
"panel_type":"stat",
"description":"Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(atlas_ai_account_tokens{provider=\"openai\",period=\"seven_day\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"openai\"} == 1)) or on() vector(-1)"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Switchyard Requests",
"panel_id":12,
"panel_type":"stat",
"description":"Hosted model requests observed by Switchyard in the selected dashboard range.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sum(increase(switchyard_requests_total[$__range])) or on() vector(0)"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Model Selection Rate",
"panel_id":13,
"panel_type":"timeseries",
"description":"AUTO and fixed-route decisions by selected provider, model family, and reasoning effort.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sum by (selected_model) (rate(switchyard_decisions_total[5m]))"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Provider Selections (Range)",
"panel_id":14,
"panel_type":"bargauge",
"description":"Switchyard selections grouped by provider over the selected dashboard range.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort_desc(sum by (provider) (label_replace(increase(switchyard_decisions_total{selected_model=~\"(route|worker)/(codex|claude|local)/.*\"}[$__range]), \"provider\", \"$2\", \"selected_model\", \"^(route|worker)/(codex|claude|local)/.*\")))"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Token Throughput",
"panel_id":15,
"panel_type":"timeseries",
"description":"Prompt, cache, reasoning, and completion token rates reported by hosted Switchyard calls.",
"description":"Local classifier failures that safely fell back to the conservative hosted route.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sum(increase(switchyard_classifier_fail_open_total[$__range])) or on() vector(0)"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Upstream Errors (Range)",
"panel_id":20,
"panel_type":"stat",
"description":"Hosted model attempts that returned errors in the selected range.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sum(increase(switchyard_errors_total[$__range])) or on() vector(0)"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Local Classifier Calls",
"panel_id":21,
"panel_type":"timeseries",
"description":"Local Qwen routing-classifier activity, split by successful and failed calls.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sum by (outcome) (rate(switchyard_llm_calls_total{selected_model=~\"qwen.*\"}[5m]))"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Routing Overhead p95",
"panel_id":22,
"panel_type":"timeseries",
"description":"95th percentile time Switchyard spends selecting a model before the upstream call.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"histogram_quantile(0.95, sum by (le, algorithm) (rate(switchyard_routing_overhead_ms_bucket[5m])))"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Traffic Lanes (Range)",
"panel_id":23,
"panel_type":"bargauge",
"description":"Request volume split between interactive route traffic and durable worker traffic.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort_desc(sum by (lane) (label_replace(increase(switchyard_requests_total{model=~\"(route|worker)/.*\"}[$__range]), \"lane\", \"$1\", \"model\", \"^(route|worker)/.*\")))"
]
},
{
"dashboard":"Atlas AI Operations",
"panel_title":"Hermes Workload CPU (Attribution Proxy)",
"panel_id":25,
"panel_type":"timeseries",
"description":"Compute use by Hermes pod/container. Switchyard currently exposes model and worker-vs-route attribution, but not tenant-slot token labels; CPU is clearly marked as a proxy rather than token usage.",
"tags":[
"atlas",
"ai",
"hermes",
"switchyard"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sum by (pod, container) (rate(container_cpu_usage_seconds_total{namespace=\"hermes\",pod=~\"hermes-(agent|chat-tenant|switchyard|model-gate).*\",container!=\"\",image!=\"\"}[5m]))"
"description":"Proportional share of GPU compute activity observed over the selected time range. Process-aware NVIDIA metrics attribute titan-22/24 work to namespaces and non-pod work to host. Jetson titan-20/21 activity is continuously sampled and assigned by Kubernetes shared-GPU allocations; unallocated activity remains unattributed. The slices total 100% of observed compute; idle appears only when the selected range contains no activity.",
"description":"NVML process-level SM samples mapped to Kubernetes pods through host cgroups; values are per-process activity rather than duplicated whole-device utilization.",
"sum(kube_pod_info{node=~\"titan-0a|titan-0b|titan-0c\",namespace!~\"^(kube-.*|.*-system|traefik|monitoring|logging|cert-manager|maintenance|postgres)$\"}) or on() vector(0)"
"description":"Rolling request-weighted availability at the Atlas HTTPS ingress: responses below 500 divided by all HTTP responses. Every server-side 5xx is a real failed request; client 4xx responses count as served. Replica counts, Grafana health, and monitoring gaps are not inputs, so they cannot lower availability unless user traffic actually receives a 5xx. A daily rollup job publishes one annual sample from deduplicated daily totals; Grafana keeps it for up to 48 hours so one delayed retry cannot cause a fallback, and only uses the same one-hour request SLI before history exists.",
"description":"Current-service pods Pending for more than 15 minutes or in an actionable failed phase. Completed Jobs and retained Veles migration workloads are kept on drill-down dashboards but excluded here.",
"description":"Current-service pods stuck in CrashLoopBackOff or ImagePullBackOff for more than 15 minutes. Retained Veles migration workloads remain visible on the Pods dashboard.",
"description":"Fan intensity lanes on the 0-10 controller scale. Cooler colors are quiet/low intensity; warmer colors mean the enclosure is pushing harder.",
"(avg((min by (suite) (((100 * sum by (suite) (clamp_max(max by (suite, check) ((((sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"ariadne|metis|ananke|atlasbot|pegasus|soteria|titan_iac|bstein_home|data_prepper|lesavka\",branch=~\"main|master|origin/main|origin/master\",check=~\"tests|coverage|loc|style|docs_naming|gate_glue|sonarqube\",status=~\"ok|passed|success|not_applicable|skipped|na|n/a\"})) or (sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"ariadne|metis|ananke|atlasbot|pegasus|soteria|titan_iac|bstein_home|data_prepper|lesavka\",suite=~\"ariadne|atlasbot|bstein_home|data_prepper\",branch=~\"main|master|origin/main|origin/master\",check=\"supply_chain\",status=~\"ok|passed|success|not_applicable|skipped|na|n/a\"})))) > 0), 1) unless on(suite, check) (clamp_max(max by (suite, check) ((((sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"ariadne|metis|ananke|atlasbot|pegasus|soteria|titan_iac|bstein_home|data_prepper|lesavka\",branch=~\"main|master|origin/main|origin/master\",check=~\"tests|coverage|loc|style|docs_naming|gate_glue|sonarqube\",status!~\"ok|passed|success|not_applicable|skipped|na|n/a\"})) or (sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"ariadne|metis|ananke|atlasbot|pegasus|soteria|titan_iac|bstein_home|data_prepper|lesavka\",suite=~\"ariadne|atlasbot|bstein_home|data_prepper\",branch=~\"main|master|origin/main|origin/master\",check=\"supply_chain\",status!~\"ok|passed|success|not_applicable|skipped|na|n/a\"})))) > 0), 1))) / clamp_min(sum by (suite) (clamp_max(max by (suite, check) ((((sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"ariadne|metis|ananke|atlasbot|pegasus|soteria|titan_iac|bstein_home|data_prepper|lesavka\",branch=~\"main|master|origin/main|origin/master\",check=~\"tests|coverage|loc|style|docs_naming|gate_glue|sonarqube\",status!=\"\"})) or (sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"ariadne|metis|ananke|atlasbot|pegasus|soteria|titan_iac|bstein_home|data_prepper|lesavka\",suite=~\"ariadne|atlasbot|bstein_home|data_prepper\",branch=~\"main|master|origin/main|origin/master\",check=\"supply_chain\",status!=\"\"})))) > 0), 1)), 1))) or (min by (suite) (platform_quality:test_category_health_rate:percent_1h{suite=~\"ariadne|metis|ananke|atlasbot|pegasus|soteria|titan_iac|bstein_home|data_prepper|lesavka\",branch!=\"\",branch=~\"main|master|origin/main|origin/master\",category=~\"api|chaos|compatibility|component|contract|e2e|integration|manual|performance|regression|reliability|security|smoke|system|ui|unit\"}))))) or on() vector(0))"
"100 * ((sum(platform_quality:suite_runs:increase_24h{suite=~\"ariadne|metis|ananke|atlasbot|pegasus|pegasus-health|pegasus_health|soteria|titan_iac|titan-iac|bstein_home|bstein-home|data_prepper|data-prepper|lesavka\",status=~\"ok|passed|success\"}) or on() vector(0))) / clamp_min(((sum(platform_quality:suite_runs:increase_24h{suite=~\"ariadne|metis|ananke|atlasbot|pegasus|pegasus-health|pegasus_health|soteria|titan_iac|titan-iac|bstein_home|bstein-home|data_prepper|data-prepper|lesavka\"}) or on() vector(0))), 1)"
]
},
{
"dashboard":"Atlas Overview",
"panel_title":"Failed Runs (24h)",
"panel_id":153,
"panel_type":"stat",
"description":"Published quality-gate runs that failed in 24h; zero is good, any value needs a look.",
"tags":[
"atlas",
"overview"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(sum(platform_quality:suite_runs:increase_24h{suite=~\"ariadne|metis|ananke|atlasbot|pegasus|pegasus-health|pegasus_health|soteria|titan_iac|titan-iac|bstein_home|bstein-home|data_prepper|data-prepper|lesavka\",status!~\"ok|passed|success\"}) or on() vector(0))"
]
},
{
"dashboard":"Atlas Overview",
"panel_title":"Suites With Runs (24h)",
"panel_id":154,
"panel_type":"stat",
"description":"Configured suites with at least one published quality-gate run in 24h; full count means the dashboard is fresh.",
"tags":[
"atlas",
"overview"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sum((sum by (suite) (platform_quality:suite_runs:increase_24h{suite=~\"ariadne|metis|ananke|atlasbot|pegasus|soteria|titan_iac|bstein_home|data_prepper|lesavka\"}) > bool 0)) or on() vector(0)"
]
},
{
"dashboard":"Atlas Overview",
"panel_title":"Avg Coverage",
"panel_id":155,
"panel_type":"stat",
"description":"Average latest line coverage across suites; higher means code is better protected by tests.",
"tags":[
"atlas",
"overview"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(avg((max by (suite) (platform_quality:suite_coverage_percent:latest_1h{suite=~\"ariadne|metis|ananke|atlasbot|pegasus|soteria|titan_iac|bstein_home|data_prepper|lesavka\"}))) or on() vector(0))"
]
},
{
"dashboard":"Atlas Overview",
"panel_title":"GitOps Health",
"panel_id":150,
"panel_type":"state-timeline",
"description":"GitOps readiness and suspension health over time. Blue means perfect; warmer colors mean a readiness or suspension problem appeared.",
"tags":[
"atlas",
"overview"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"label_replace(100 * sum(max by (namespace, name) (ananke_gitops_kustomization_ready{job=\"ananke-power\"})) / clamp_min(count(max by (namespace, name) (ananke_gitops_kustomization_ready{job=\"ananke-power\"})), 1), \"signal\", \"Kustomizations Ready\", \"__name__\", \".*\") or label_replace(100 * sum(max by (namespace, name) (ananke_gitops_helmrelease_ready{job=\"ananke-power\"})) / clamp_min(count(max by (namespace, name) (ananke_gitops_helmrelease_ready{job=\"ananke-power\"})), 1), \"signal\", \"HelmReleases Ready\", \"__name__\", \".*\") or label_replace(100 * (1 - (sum(max by (namespace, name) (ananke_gitops_kustomization_suspended{job=\"ananke-power\"})) or on() vector(0)) / clamp_min((count(max by (namespace, name) (ananke_gitops_kustomization_ready{job=\"ananke-power\"})) or on() vector(0)), 1)), \"signal\", \"Kustomizations Not Suspended\", \"__name__\", \".*\") or label_replace(100 * (1 - (sum(max by (namespace, name) (ananke_gitops_helmrelease_suspended{job=\"ananke-power\"})) or on() vector(0)) / clamp_min((count(max by (namespace, name) (ananke_gitops_helmrelease_ready{job=\"ananke-power\"})) or on() vector(0)), 1)), \"signal\", \"HelmReleases Not Suspended\", \"__name__\", \".*\")"
]
},
{
"dashboard":"Atlas Overview",
"panel_title":"One-off Job Pods (age hours)",
"panel_id":44,
"panel_type":"bargauge",
"description":"Temporary job pods by age; low or empty is good, old pods usually need cleanup.",
"description":"Ariadne automation attempts and failures; attempts show activity, failures show work to investigate.",
"tags":[
"atlas",
"overview"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sum(increase(ariadne_task_runs_total[5m])) or on() vector(0)",
"sum(increase(ariadne_task_runs_total{status=\"error\"}[5m])) or on() vector(0)"
]
},
{
"dashboard":"Atlas Overview",
"panel_title":"Test Category Health",
"panel_id":46,
"panel_type":"state-timeline",
"description":"Health by major test category across all suites over the last 24 hours. Skipped tests are healthy; failures and errors lower the lane.",
"(avg by (category) (platform_quality:test_category_health_rate:percent_1h{suite=~\"ariadne|metis|ananke|atlasbot|pegasus|soteria|titan_iac|bstein_home|data_prepper|lesavka\",branch!=\"\",branch=~\"main|master|origin/main|origin/master\",category=~\"api|chaos|compatibility|component|contract|e2e|integration|performance|regression|reliability|security|smoke|system|ui\"})) or label_set(vector(0), \"category\", \"none\")"
"panel_title":"Jenkins Last Success (h, newest first)",
"panel_id":142,
"panel_type":"stat",
"description":"Top 6 most recent Jenkins successes by age (newest first). Green means last run succeeded; red means last run did not succeed. Use Atlas Jobs for the full list.",
"tags":[
"atlas",
"overview"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort((label_replace((sort(bottomk(6, min by (exported_job,job_url,weather_icon) ((time() - ariadne_jenkins_build_weather_job_last_success_timestamp_seconds) / 3600)))) and on(exported_job,job_url,weather_icon) (max by (exported_job,job_url,weather_icon) (ariadne_jenkins_build_weather_job_last_status) == 1), \"run_state\", \"ok\", \"exported_job\", \".*\")) or (label_replace((sort(bottomk(6, min by (exported_job,job_url,weather_icon) ((time() - ariadne_jenkins_build_weather_job_last_success_timestamp_seconds) / 3600)))) and on(exported_job,job_url,weather_icon) (max by (exported_job,job_url,weather_icon) (ariadne_jenkins_build_weather_job_last_status) != 1), \"run_state\", \"bad\", \"exported_job\", \".*\")))"
]
},
{
"dashboard":"Atlas Overview",
"panel_title":"Jenkins Last Failure (h, newest first)",
"panel_id":243,
"panel_type":"stat",
"description":"Top 6 most recent Jenkins failures by age (newest first). Green means last run succeeded; red means last run did not succeed. Use Atlas Jobs for the full list.",
"tags":[
"atlas",
"overview"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort((label_replace((sort(bottomk(6, min by (exported_job,job_url,weather_icon) ((time() - ariadne_jenkins_build_weather_job_last_failure_timestamp_seconds) / 3600)))) and on(exported_job,job_url,weather_icon) (max by (exported_job,job_url,weather_icon) (ariadne_jenkins_build_weather_job_last_status) == 1), \"run_state\", \"ok\", \"exported_job\", \".*\")) or (label_replace((sort(bottomk(6, min by (exported_job,job_url,weather_icon) ((time() - ariadne_jenkins_build_weather_job_last_failure_timestamp_seconds) / 3600)))) and on(exported_job,job_url,weather_icon) (max by (exported_job,job_url,weather_icon) (ariadne_jenkins_build_weather_job_last_status) != 1), \"run_state\", \"bad\", \"exported_job\", \".*\")))"
]
},
{
"dashboard":"Atlas Overview",
"panel_title":"PVC Backup Health / Age",
"panel_id":47,
"panel_type":"bargauge",
"description":"Backup age in hours computed from last-success timestamps for restic-managed PVCs (nightly target: <=20h green, <40h yellow, <50h orange, >=50h red). PVCs that have backup history but currently no successful backup (missing/no_completed/error) are pinned to 999h for visibility.",
"description":"Proportional share of GPU compute activity observed over the selected time range. Process-aware NVIDIA metrics attribute titan-22/24 work to namespaces and non-pod work to host. Jetson titan-20/21 activity is continuously sampled and assigned by Kubernetes shared-GPU allocations; unallocated activity remains unattributed. The slices total 100% of observed compute; idle appears only when the selected range contains no activity.",
"(sum by (namespace,node) (kube_pod_info{pod!=\"\" , node!=\"\"}) / on(namespace) group_left() clamp_min(sum by (namespace) (kube_pod_info{pod!=\"\"}), 1) * 100) * on(namespace,node) group_left() ((sum by (namespace,node) (kube_pod_info{pod!=\"\" , node!=\"\"}) / on(namespace) group_left() clamp_min(sum by (namespace) (kube_pod_info{pod!=\"\"}), 1) * 100) + on(node) group_left() ((sum by (node) (kube_node_info{node=\"titan-0a\"}) * 0 + 0.001) or (sum by (node) (kube_node_info{node=\"titan-0b\"}) * 0 + 0.002) or (sum by (node) (kube_node_info{node=\"titan-0c\"}) * 0 + 0.003) or (sum by (node) (kube_node_info{node=\"titan-db\"}) * 0 + 0.004) or (sum by (node) (kube_node_info{node=\"titan-jh\"}) * 0 + 0.005) or (sum by (node) (kube_node_info{node=\"titan-04\"}) * 0 + 0.006) or (sum by (node) (kube_node_info{node=\"titan-05\"}) * 0 + 0.007) or (sum by (node) (kube_node_info{node=\"titan-06\"}) * 0 + 0.008) or (sum by (node) (kube_node_info{node=\"titan-07\"}) * 0 + 0.009000000000000001) or (sum by (node) (kube_node_info{node=\"titan-08\"}) * 0 + 0.01) or (sum by (node) (kube_node_info{node=\"titan-11\"}) * 0 + 0.011) or (sum by (node) (kube_node_info{node=\"titan-20\"}) * 0 + 0.012) or (sum by (node) (kube_node_info{node=\"titan-21\"}) * 0 + 0.013000000000000001) or (sum by (node) (kube_node_info{node=\"titan-12\"}) * 0 + 0.014) or (sum by (node) (kube_node_info{node=\"titan-13\"}) * 0 + 0.015) or (sum by (node) (kube_node_info{node=\"titan-14\"}) * 0 + 0.016) or (sum by (node) (kube_node_info{node=\"titan-15\"}) * 0 + 0.017) or (sum by (node) (kube_node_info{node=\"titan-17\"}) * 0 + 0.018000000000000002) or (sum by (node) (kube_node_info{node=\"titan-18\"}) * 0 + 0.019) or (sum by (node) (kube_node_info{node=\"titan-19\"}) * 0 + 0.02) or (sum by (node) (kube_node_info{node=\"titan-22\"}) * 0 + 0.021) or (sum by (node) (kube_node_info{node=\"titan-23\"}) * 0 + 0.022) or (sum by (node) (kube_node_info{node=\"titan-24\"}) * 0 + 0.023)) == bool on(namespace) group_left() (max by (namespace) ((sum by (namespace,node) (kube_pod_info{pod!=\"\" , node!=\"\"}) / on(namespace) group_left() clamp_min(sum by (namespace) (kube_pod_info{pod!=\"\"}), 1) * 100) + on(node) group_left() ((sum by (node) (kube_node_info{node=\"titan-0a\"}) * 0 + 0.001) or (sum by (node) (kube_node_info{node=\"titan-0b\"}) * 0 + 0.002) or (sum by (node) (kube_node_info{node=\"titan-0c\"}) * 0 + 0.003) or (sum by (node) (kube_node_info{node=\"titan-db\"}) * 0 + 0.004) or (sum by (node) (kube_node_info{node=\"titan-jh\"}) * 0 + 0.005) or (sum by (node) (kube_node_info{node=\"titan-04\"}) * 0 + 0.006) or (sum by (node) (kube_node_info{node=\"titan-05\"}) * 0 + 0.007) or (sum by (node) (kube_node_info{node=\"titan-06\"}) * 0 + 0.008) or (sum by (node) (kube_node_info{node=\"titan-07\"}) * 0 + 0.009000000000000001) or (sum by (node) (kube_node_info{node=\"titan-08\"}) * 0 + 0.01) or (sum by (node) (kube_node_info{node=\"titan-11\"}) * 0 + 0.011) or (sum by (node) (kube_node_info{node=\"titan-20\"}) * 0 + 0.012) or (sum by (node) (kube_node_info{node=\"titan-21\"}) * 0 + 0.013000000000000001) or (sum by (node) (kube_node_info{node=\"titan-12\"}) * 0 + 0.014) or (sum by (node) (kube_node_info{node=\"titan-13\"}) * 0 + 0.015) or (sum by (node) (kube_node_info{node=\"titan-14\"}) * 0 + 0.016) or (sum by (node) (kube_node_info{node=\"titan-15\"}) * 0 + 0.017) or (sum by (node) (kube_node_info{node=\"titan-17\"}) * 0 + 0.018000000000000002) or (sum by (node) (kube_node_info{node=\"titan-18\"}) * 0 + 0.019) or (sum by (node) (kube_node_info{node=\"titan-19\"}) * 0 + 0.02) or (sum by (node) (kube_node_info{node=\"titan-22\"}) * 0 + 0.021) or (sum by (node) (kube_node_info{node=\"titan-23\"}) * 0 + 0.022) or (sum by (node) (kube_node_info{node=\"titan-24\"}) * 0 + 0.023)))))"
"max(max without (job,instance,pod,service,endpoint,namespace,node,controller_name,controller_id,port_name,fan_group) (typhon_temperature_celsius != 0)) or on() vector(0)",
"max(max without (job,instance,pod,service,endpoint,namespace,node,controller_name,controller_id,port_name,fan_group) (typhon_vpd_kpa != 0)) or on() vector(0)",
"max(max without (job,instance,pod,service,endpoint,namespace,node,controller_name,controller_id,port_name,fan_group) (typhon_relative_humidity_percent != 0)) or on() vector(0)",
"max((243.12 * (ln(clamp_min((max without (job,instance,pod,service,endpoint,namespace,node,controller_name,controller_id,port_name,fan_group) (typhon_relative_humidity_percent != 0)), 1) / 100) + (17.62 * (max without (job,instance,pod,service,endpoint,namespace,node,controller_name,controller_id,port_name,fan_group) (typhon_temperature_celsius != 0))) / (243.12 + (max without (job,instance,pod,service,endpoint,namespace,node,controller_name,controller_id,port_name,fan_group) (typhon_temperature_celsius != 0))))) / (17.62 - (ln(clamp_min((max without (job,instance,pod,service,endpoint,namespace,node,controller_name,controller_id,port_name,fan_group) (typhon_relative_humidity_percent != 0)), 1) / 100) + (17.62 * (max without (job,instance,pod,service,endpoint,namespace,node,controller_name,controller_id,port_name,fan_group) (typhon_temperature_celsius != 0))) / (243.12 + (max without (job,instance,pod,service,endpoint,namespace,node,controller_name,controller_id,port_name,fan_group) (typhon_temperature_celsius != 0)))))) or on() vector(0)"
"description":"Historical fan activity for all four fan groups (0-10 scale).",
"tags":[
"atlas",
"power",
"climate"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"max without (job,instance,pod,service,endpoint,namespace,node,controller_name,controller_id,port_name,fan_group) (typhon_fan_speed_level{fan_group=\"outlet\"})",
"max without (job,instance,pod,service,endpoint,namespace,node,controller_name,controller_id,port_name,fan_group) (typhon_fan_speed_level{fan_group=\"inside_inlet\"})",
"max without (job,instance,pod,service,endpoint,namespace,node,controller_name,controller_id,port_name,fan_group) (typhon_fan_speed_level{fan_group=\"outside_inlet\"})",
"max without (job,instance,pod,service,endpoint,namespace,node,controller_name,controller_id,port_name,fan_group) (typhon_fan_speed_level{fan_group=\"interior\"})"
"description":"Average latest required gate checks passing across selected suites; this is the current quality state.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(avg((min by (suite) (((100 * (sum by (suite) (((clamp_max(max by (suite, check) ((((sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"tests|coverage|loc|style|docs_naming|gate_glue|sonarqube\",status=~\"ok|passed|success|not_applicable|skipped|na|n/a\"})) or (sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"${suite:regex}\",suite=~\"ariadne|atlasbot|bstein_home|data_prepper\",branch!=\"\",branch=~\"${branch:regex}\",check=\"supply_chain\",status=~\"ok|passed|success|not_applicable|skipped|na|n/a\"})))) > 0), 1)) unless on(suite, check) (clamp_max(max by (suite, check) ((((sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"tests|coverage|loc|style|docs_naming|gate_glue|sonarqube\",status!~\"ok|passed|success|not_applicable|skipped|na|n/a\"})) or (sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"${suite:regex}\",suite=~\"ariadne|atlasbot|bstein_home|data_prepper\",branch!=\"\",branch=~\"${branch:regex}\",check=\"supply_chain\",status!~\"ok|passed|success|not_applicable|skipped|na|n/a\"})))) > 0), 1))))) / clamp_min((sum by (suite) (clamp_max(max by (suite, check) ((((sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"tests|coverage|loc|style|docs_naming|gate_glue|sonarqube\",status!=\"\"})) or (sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"${suite:regex}\",suite=~\"ariadne|atlasbot|bstein_home|data_prepper\",branch!=\"\",branch=~\"${branch:regex}\",check=\"supply_chain\",status!=\"\"})))) > 0), 1))), 1))) or (min by (suite) (platform_quality:test_category_health_rate:percent_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",category=~\"api|chaos|compatibility|component|contract|e2e|integration|manual|performance|regression|reliability|security|smoke|system|ui|unit\"}))))) or on() vector(0))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"CI Run Success Rate (24h)",
"panel_id":2,
"panel_type":"stat",
"description":"Percent of selected quality-gate CI runs that completed successfully in 24h; this is run health, not individual test pass rate.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"100 * ((sum(platform_quality:suite_runs:increase_24h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",status=~\"ok|passed|success\"}) or on() vector(0))) / clamp_min(((sum(platform_quality:suite_runs:increase_24h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\"}) or on() vector(0))), 1)"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"CI Run Success Rate (7d)",
"panel_id":3,
"panel_type":"stat",
"description":"Percent of selected quality-gate CI runs that completed successfully in 7d; higher means more stable automation.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"100 * ((sum(increase((max without(instance, job) (platform_quality_gate_runs_total{suite=~\"${suite:regex}\",exported_job=\"platform-quality-ci\",status=~\"ok|passed|success\"}))[7d:1h])) or on() vector(0))) / clamp_min(((sum(increase((max without(instance, job) (platform_quality_gate_runs_total{suite=~\"${suite:regex}\",exported_job=\"platform-quality-ci\"}))[7d:1h])) or on() vector(0))), 1)"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Failed Runs (24h)",
"panel_id":4,
"panel_type":"stat",
"description":"Selected quality-gate runs that failed in 24h; zero is good and anything else needs a look.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(sum(platform_quality:suite_runs:increase_24h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",status!~\"ok|passed|success\"}) or on() vector(0))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"CI Runs (24h)",
"panel_id":5,
"panel_type":"stat",
"description":"Selected quality-gate CI run count in 24h; zero means the dashboard may be stale.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(sum(platform_quality:suite_runs:increase_24h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\"}) or on() vector(0))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Suite Freshness (24h)",
"panel_id":157,
"panel_type":"stat",
"description":"Percent of selected suites with at least one quality-gate CI run in 24h; 100% means inputs are fresh.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"100 * (sum((sum by (suite) (platform_quality:suite_runs:increase_24h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\"})) > bool 0) or on() vector(0)) / clamp_min(count(((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0))), 1)"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Avg Coverage (%)",
"panel_id":6,
"panel_type":"stat",
"description":"Average latest line coverage for selected suites; higher means better test protection.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(avg((max by (suite) (platform_quality:suite_coverage_percent:latest_1h{suite=~\"${suite:regex}\",branch=~\"${branch:regex}\"}))) or on() vector(0))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Suites with LOC >500",
"panel_id":7,
"panel_type":"stat",
"description":"Selected suites with oversized source files; zero is good for maintainability.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(sum(((max by (suite) (platform_quality:suite_source_lines_over_500_total:latest_1h{suite=~\"${suite:regex}\",branch=~\"${branch:regex}\"})) > bool 0)) or on() vector(0))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Latest Gate Health by Suite",
"panel_id":8,
"panel_type":"bargauge",
"description":"Current health by suite from required gate checks, capped by category-level test health. Skipped and not-applicable results are healthy; failures and errors lower the value.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort(((min by (suite) (((100 * (sum by (suite) (((clamp_max(max by (suite, check) ((((sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"tests|coverage|loc|style|docs_naming|gate_glue|sonarqube\",status=~\"ok|passed|success|not_applicable|skipped|na|n/a\"})) or (sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"${suite:regex}\",suite=~\"ariadne|atlasbot|bstein_home|data_prepper\",branch!=\"\",branch=~\"${branch:regex}\",check=\"supply_chain\",status=~\"ok|passed|success|not_applicable|skipped|na|n/a\"})))) > 0), 1)) unless on(suite, check) (clamp_max(max by (suite, check) ((((sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"tests|coverage|loc|style|docs_naming|gate_glue|sonarqube\",status!~\"ok|passed|success|not_applicable|skipped|na|n/a\"})) or (sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"${suite:regex}\",suite=~\"ariadne|atlasbot|bstein_home|data_prepper\",branch!=\"\",branch=~\"${branch:regex}\",check=\"supply_chain\",status!~\"ok|passed|success|not_applicable|skipped|na|n/a\"})))) > 0), 1))))) / clamp_min((sum by (suite) (clamp_max(max by (suite, check) ((((sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"tests|coverage|loc|style|docs_naming|gate_glue|sonarqube\",status!=\"\"})) or (sum by (suite, branch, check, status) (platform_quality:check_status:present_1h{suite=~\"${suite:regex}\",suite=~\"ariadne|atlasbot|bstein_home|data_prepper\",branch!=\"\",branch=~\"${branch:regex}\",check=\"supply_chain\",status!=\"\"})))) > 0), 1))), 1))) or (min by (suite) (platform_quality:test_category_health_rate:percent_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",category=~\"api|chaos|compatibility|component|contract|e2e|integration|manual|performance|regression|reliability|security|smoke|system|ui|unit\"})))) or on(suite) ((((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))) - 1))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"CI Run Success by Suite (24h)",
"panel_id":9,
"panel_type":"bargauge",
"description":"24h CI run success rate. This is whether automation finished cleanly, so it can stay low after failed or aborted runs even when tests and latest gate checks are green.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort(((100 * (sum by (suite) (platform_quality:suite_runs:increase_24h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",status=~\"ok|passed|success\"})) / clamp_min((sum by (suite) (platform_quality:suite_runs:increase_24h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\"})), 1)) and on(suite) ((sum by (suite) (platform_quality:suite_runs:increase_24h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\"})) > 0)) or on(suite) ((((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))) - 1)))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Coverage by Suite (Latest, gate 95)",
"panel_id":17,
"panel_type":"bargauge",
"description":"Latest suite coverage; 95%+ is acceptable and 100% is strongest.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort((max by (suite) (platform_quality:suite_coverage_percent:latest_1h{suite=~\"${suite:regex}\",branch=~\"${branch:regex}\"})) or on(suite) ((((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))) - 1)))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Files <=500 LOC by Suite (Latest)",
"panel_id":18,
"panel_type":"bargauge",
"description":"Percent of managed LOC-gated files at or under 500 lines. Older suite payloads fall back to 100%/0% until they emit platform_quality_gate_source_files_total.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort(((100 * clamp_min((max by (suite) (platform_quality:suite_source_files_total:latest_1h{suite=~\"${suite:regex}\",branch=~\"${branch:regex}\"})) - (max by (suite) (platform_quality:suite_source_lines_over_500_total:latest_1h{suite=~\"${suite:regex}\",branch=~\"${branch:regex}\"})), 0) / (max by (suite) (platform_quality:suite_source_files_total:latest_1h{suite=~\"${suite:regex}\",branch=~\"${branch:regex}\"}))) and on(suite) ((max by (suite) (platform_quality:suite_source_files_total:latest_1h{suite=~\"${suite:regex}\",branch=~\"${branch:regex}\"})) > 0)) or on(suite) (100 * (1 - clamp_max((max by (suite) (platform_quality:suite_source_lines_over_500_total:latest_1h{suite=~\"${suite:regex}\",branch=~\"${branch:regex}\"})), 1))) or on(suite) ((((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))) - 1)))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"CI Run Success by Suite (7d rolling)",
"panel_id":11,
"panel_type":"state-timeline",
"description":"Seven-day rolling CI run success rate per suite. Each suite gets its own lane, so failed or aborted runs lower the lane color without implying raw test failures.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(100 * sum by (suite) (increase((max without(instance, job) (platform_quality_gate_runs_total{suite=~\"${suite:regex}\",exported_job=\"platform-quality-ci\",status=~\"ok|passed|success\"}))[7d:1h])) / (sum by (suite) (increase((max without(instance, job) (platform_quality_gate_runs_total{suite=~\"${suite:regex}\",exported_job=\"platform-quality-ci\"}))[7d:1h])))) and on(suite) ((sum by (suite) (increase((max without(instance, job) (platform_quality_gate_runs_total{suite=~\"${suite:regex}\",exported_job=\"platform-quality-ci\"}))[7d:1h]))) > 0)"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Test Category Health History",
"panel_id":153,
"panel_type":"state-timeline",
"description":"Health by test category from memoized hourly rollups. Use the Suite filter to focus one project; skipped tests are healthy, while failures and errors lower the lane.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"avg by (category) (platform_quality:test_category_health_rate:percent_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",category=~\"api|chaos|compatibility|component|contract|e2e|integration|manual|performance|regression|reliability|security|smoke|system|ui|unit\"})"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Daily Run Volume (Selected Scope)",
"panel_id":12,
"panel_type":"timeseries",
"description":"Twenty-four-hour rolling quality-gate run counts for the selected suite/branch scope. This is volume, not a pass-rate percentage.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sum(platform_quality:suite_runs:increase_24h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",status=~\"ok|passed|success\"}) or on() vector(0)",
"sum(platform_quality:suite_runs:increase_24h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",status!~\"ok|passed|success\"}) or on() vector(0)"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Coverage History by Suite",
"panel_id":13,
"panel_type":"state-timeline",
"description":"Latest reported line coverage per suite over time. Coverage is separate from LOC compliance so one signal cannot hide the other.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"max by (suite) (platform_quality:suite_coverage_percent:latest_1h{suite=~\"${suite:regex}\",branch=~\"${branch:regex}\"})"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Files <=500 LOC History by Suite",
"panel_id":14,
"panel_type":"state-timeline",
"description":"Percent of LOC-gated source files at or under the 500-line limit. This uses the existing file-count telemetry; longest-file history needs a new publisher metric.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(100 * clamp_min((max by (suite) (platform_quality:suite_source_files_total:latest_1h{suite=~\"${suite:regex}\",branch=~\"${branch:regex}\"})) - (max by (suite) (platform_quality:suite_source_lines_over_500_total:latest_1h{suite=~\"${suite:regex}\",branch=~\"${branch:regex}\"})), 0) / (max by (suite) (platform_quality:suite_source_files_total:latest_1h{suite=~\"${suite:regex}\",branch=~\"${branch:regex}\"}))) and on(suite) ((max by (suite) (platform_quality:suite_source_files_total:latest_1h{suite=~\"${suite:regex}\",branch=~\"${branch:regex}\"})) > 0) or on(suite) (100 * (1 - clamp_max((max by (suite) (platform_quality:suite_source_lines_over_500_total:latest_1h{suite=~\"${suite:regex}\",branch=~\"${branch:regex}\"})), 1)))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Tests Failure Rate",
"panel_id":130,
"panel_type":"state-timeline",
"description":"Latest bad-state percentage for this check family, evaluated over time. Higher means more selected suites/checks are failing in the freshness window; this is not an event-count spike chart.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(((100 * (sum by (suite) (platform_quality:check_failed_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"tests|unit|build\"})) / clamp_min((sum by (suite) (platform_quality:check_seen_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"tests|unit|build\"})), 1))) or on(suite) ((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Coverage Failure Rate",
"panel_id":131,
"panel_type":"state-timeline",
"description":"Latest bad-state percentage for this check family, evaluated over time. Higher means more selected suites/checks are failing in the freshness window; this is not an event-count spike chart.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(((100 * (sum by (suite) (platform_quality:check_failed_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"coverage\"})) / clamp_min((sum by (suite) (platform_quality:check_seen_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"coverage\"})), 1))) or on(suite) ((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"LOC Failure Rate",
"panel_id":132,
"panel_type":"state-timeline",
"description":"Latest bad-state percentage for this check family, evaluated over time. Higher means more selected suites/checks are failing in the freshness window; this is not an event-count spike chart.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(((100 * (sum by (suite) (platform_quality:check_failed_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"loc|smell\"})) / clamp_min((sum by (suite) (platform_quality:check_seen_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"loc|smell\"})), 1))) or on(suite) ((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Style Failure Rate",
"panel_id":133,
"panel_type":"state-timeline",
"description":"Latest bad-state percentage for this check family, evaluated over time. Higher means more selected suites/checks are failing in the freshness window; this is not an event-count spike chart.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(((100 * (sum by (suite) (platform_quality:check_failed_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"docs|naming|hygiene|lint|docs_naming|style\"})) / clamp_min((sum by (suite) (platform_quality:check_seen_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"docs|naming|hygiene|lint|docs_naming|style\"})), 1))) or on(suite) ((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Gate Glue Failure Rate",
"panel_id":134,
"panel_type":"state-timeline",
"description":"Latest bad-state percentage for this check family, evaluated over time. Higher means more selected suites/checks are failing in the freshness window; this is not an event-count spike chart.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(((100 * (sum by (suite) (platform_quality:check_failed_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"gate|glue|gate_glue\"})) / clamp_min((sum by (suite) (platform_quality:check_seen_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"gate|glue|gate_glue\"})), 1))) or on(suite) ((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"SonarQube Failure Rate",
"panel_id":135,
"panel_type":"state-timeline",
"description":"Latest bad-state percentage for this check family, evaluated over time. Higher means more selected suites/checks are failing in the freshness window; this is not an event-count spike chart.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(((100 * (sum by (suite) (platform_quality:check_failed_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"sonarqube|sonar\"})) / clamp_min((sum by (suite) (platform_quality:check_seen_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"sonarqube|sonar\"})), 1))) or on(suite) ((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Supply Chain Failure Rate",
"panel_id":136,
"panel_type":"state-timeline",
"description":"Latest bad-state percentage for this check family, evaluated over time. Higher means more selected suites/checks are failing in the freshness window; this is not an event-count spike chart.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(((100 * (sum by (suite) (platform_quality:check_failed_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"ironbank|supply_chain|image_compliance|artifact_security\"})) / clamp_min((sum by (suite) (platform_quality:check_seen_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"ironbank|supply_chain|image_compliance|artifact_security\"})), 1))) or on(suite) ((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Tests Healthy Rate",
"panel_id":138,
"panel_type":"state-timeline",
"description":"Latest acceptable-state percentage for this check family, evaluated over time. Higher means more selected suites/checks are healthy in the freshness window; gaps mean there was no check evidence.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(((100 * (sum by (suite) (platform_quality:check_healthy_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"tests|unit|build\"})) / clamp_min((sum by (suite) (platform_quality:check_seen_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"tests|unit|build\"})), 1))) or on(suite) ((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Coverage Healthy Rate",
"panel_id":139,
"panel_type":"state-timeline",
"description":"Latest acceptable-state percentage for this check family, evaluated over time. Higher means more selected suites/checks are healthy in the freshness window; gaps mean there was no check evidence.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(((100 * (sum by (suite) (platform_quality:check_healthy_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"coverage\"})) / clamp_min((sum by (suite) (platform_quality:check_seen_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"coverage\"})), 1))) or on(suite) ((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"LOC Healthy Rate",
"panel_id":140,
"panel_type":"state-timeline",
"description":"Latest acceptable-state percentage for this check family, evaluated over time. Higher means more selected suites/checks are healthy in the freshness window; gaps mean there was no check evidence.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(((100 * (sum by (suite) (platform_quality:check_healthy_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"loc|smell\"})) / clamp_min((sum by (suite) (platform_quality:check_seen_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"loc|smell\"})), 1))) or on(suite) ((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Style Healthy Rate",
"panel_id":141,
"panel_type":"state-timeline",
"description":"Latest acceptable-state percentage for this check family, evaluated over time. Higher means more selected suites/checks are healthy in the freshness window; gaps mean there was no check evidence.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(((100 * (sum by (suite) (platform_quality:check_healthy_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"docs|naming|hygiene|lint|docs_naming|style\"})) / clamp_min((sum by (suite) (platform_quality:check_seen_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"docs|naming|hygiene|lint|docs_naming|style\"})), 1))) or on(suite) ((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Gate Glue Healthy Rate",
"panel_id":142,
"panel_type":"state-timeline",
"description":"Latest acceptable-state percentage for this check family, evaluated over time. Higher means more selected suites/checks are healthy in the freshness window; gaps mean there was no check evidence.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(((100 * (sum by (suite) (platform_quality:check_healthy_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"gate|glue|gate_glue\"})) / clamp_min((sum by (suite) (platform_quality:check_seen_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"gate|glue|gate_glue\"})), 1))) or on(suite) ((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"SonarQube Healthy Rate",
"panel_id":143,
"panel_type":"state-timeline",
"description":"Latest acceptable-state percentage for this check family, evaluated over time. Higher means more selected suites/checks are healthy in the freshness window; gaps mean there was no check evidence.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(((100 * (sum by (suite) (platform_quality:check_healthy_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"sonarqube|sonar\"})) / clamp_min((sum by (suite) (platform_quality:check_seen_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"sonarqube|sonar\"})), 1))) or on(suite) ((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Supply Chain Healthy Rate",
"panel_id":144,
"panel_type":"state-timeline",
"description":"Latest acceptable-state percentage for this check family, evaluated over time. Higher means more selected suites/checks are healthy in the freshness window; gaps mean there was no check evidence.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(((100 * (sum by (suite) (platform_quality:check_healthy_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"ironbank|supply_chain|image_compliance|artifact_security\"})) / clamp_min((sum by (suite) (platform_quality:check_seen_flag:present_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",check=~\"ironbank|supply_chain|image_compliance|artifact_security\"})), 1))) or on(suite) ((0 * ((count by (suite) (platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}) >= bool 0)))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Problematic Tests Over Time (Top failures)",
"panel_id":145,
"panel_type":"state-timeline",
"description":"Current outlier tests by rolling 24h failure count. A test needs at least two recent failures to appear, then falls off once it quiets down.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(sum by (suite, test) (sum_over_time(platform_quality:test_case_status:count_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",test!=\"\",test!=\"__no_test_cases__\",status=\"failed\"}[24h:1h]))) and on (suite, test) topk(12, (sum by (suite, test) (sum_over_time(platform_quality:test_case_status:count_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",test!=\"\",test!=\"__no_test_cases__\",status=\"failed\"}[24h:1h] @ end()))) >= 2)"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Most Problematic Test by Suite (7d)",
"panel_id":147,
"panel_type":"bargauge",
"description":"Worst test per suite summed across 7d. This catches repeat offenders while keeping dashboard loads bounded; current hourly top list is quiet.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort_desc(topk by (suite) (1, (sum by (suite, test) (sum_over_time(platform_quality:test_case_status:count_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",test!=\"\",test!=\"__no_test_cases__\",status=\"failed\"}[7d:1h])))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Selected Test Pass/Fail History",
"panel_id":146,
"panel_type":"timeseries",
"description":"Stacked hourly outcome volume for the selected suite/branch/test scope. This uses vmalert rollups only, avoiding expensive raw long-range per-test scans.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(sum(platform_quality:test_case_status:count_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",test!=\"\",test=~\"${test:regex}\",test!=\"__no_test_cases__\",status=\"passed\"}) or on() vector(0))",
"(sum(platform_quality:test_case_status:count_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",test!=\"\",test=~\"${test:regex}\",test!=\"__no_test_cases__\",status=\"failed\"}) or on() vector(0))",
"(sum(platform_quality:test_case_status:count_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",test!=\"\",test=~\"${test:regex}\",test!=\"__no_test_cases__\",status=\"skipped\"}) or on() vector(0))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Selected Test Pass Rate History",
"panel_id":152,
"panel_type":"state-timeline",
"description":"Average pass rate per suite for the selected test filter, using memoized hourly test-case pass-rate rollups instead of raw historical scans.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"avg by (suite) (platform_quality:test_case_pass_rate:percent_1h{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",test!=\"\",test=~\"${test:regex}\",test!=\"__no_test_cases__\"})"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Tests Metrics Present by Suite",
"panel_id":27,
"panel_type":"bargauge",
"description":"Whether suite-level test counts are present; 100% means the suite is reporting.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort((100 * (((label_replace(vector(1), \"suite\", \"ariadne\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"metis\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"ananke\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"atlasbot\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"pegasus\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"soteria\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"titan_iac\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"bstein_home\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"data_prepper\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"lesavka\", \"__name__\", \".*\")) and on(suite) count by (suite) ({__name__=~\".*_quality_gate_tests_total\",exported_job=\"platform-quality-ci\"})))) or on(suite) (0 * (label_replace(vector(1), \"suite\", \"ariadne\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"metis\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"ananke\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"atlasbot\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"pegasus\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"soteria\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"titan_iac\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"bstein_home\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"data_prepper\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"lesavka\", \"__name__\", \".*\"))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Checks Metrics Present by Suite",
"panel_id":28,
"panel_type":"bargauge",
"description":"Whether gate check metrics are present; 100% means health panels have inputs.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort((100 * (((label_replace(vector(1), \"suite\", \"ariadne\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"metis\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"ananke\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"atlasbot\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"pegasus\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"soteria\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"titan_iac\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"bstein_home\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"data_prepper\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"lesavka\", \"__name__\", \".*\")) and on(suite) count by (suite) ({__name__=~\".*_quality_gate_checks_total\",exported_job=\"platform-quality-ci\"})))) or on(suite) (0 * (label_replace(vector(1), \"suite\", \"ariadne\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"metis\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"ananke\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"atlasbot\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"pegasus\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"soteria\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"titan_iac\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"bstein_home\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"data_prepper\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"lesavka\", \"__name__\", \".*\"))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Coverage Metrics Present by Suite",
"panel_id":29,
"panel_type":"bargauge",
"description":"Whether coverage metrics are present; 100% means coverage panels are reliable.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort((100 * (((label_replace(vector(1), \"suite\", \"ariadne\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"metis\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"ananke\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"atlasbot\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"pegasus\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"soteria\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"titan_iac\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"bstein_home\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"data_prepper\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"lesavka\", \"__name__\", \".*\")) and on(suite) count by (suite) (platform_quality_gate_workspace_line_coverage_percent{exported_job=\"platform-quality-ci\"})))) or on(suite) (0 * (label_replace(vector(1), \"suite\", \"ariadne\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"metis\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"ananke\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"atlasbot\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"pegasus\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"soteria\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"titan_iac\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"bstein_home\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"data_prepper\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"lesavka\", \"__name__\", \".*\"))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"LOC Compliance Metrics Present by Suite",
"panel_id":30,
"panel_type":"bargauge",
"description":"Whether LOC metrics are present; 100% means size panels are reliable.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort((100 * (((label_replace(vector(1), \"suite\", \"ariadne\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"metis\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"ananke\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"atlasbot\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"pegasus\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"soteria\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"titan_iac\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"bstein_home\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"data_prepper\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"lesavka\", \"__name__\", \".*\")) and on(suite) count by (suite) (platform_quality_gate_source_lines_over_500_total{exported_job=\"platform-quality-ci\"}) and on(suite) count by (suite) (platform_quality_gate_source_files_total{exported_job=\"platform-quality-ci\"})))) or on(suite) (0 * (label_replace(vector(1), \"suite\", \"ariadne\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"metis\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"ananke\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"atlasbot\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"pegasus\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"soteria\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"titan_iac\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"bstein_home\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"data_prepper\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"lesavka\", \"__name__\", \".*\"))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Test-Case Metrics Present by Suite",
"panel_id":148,
"panel_type":"bargauge",
"description":"Whether per-test metrics are present; 100% enables drilldowns.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort((100 * (((label_replace(vector(1), \"suite\", \"ariadne\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"metis\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"ananke\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"atlasbot\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"pegasus\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"soteria\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"titan_iac\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"bstein_home\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"data_prepper\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"lesavka\", \"__name__\", \".*\")) and on(suite) count by (suite) (platform_quality_gate_test_case_result{exported_job=\"platform-quality-ci\"})))) or on(suite) (0 * (label_replace(vector(1), \"suite\", \"ariadne\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"metis\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"ananke\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"atlasbot\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"pegasus\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"soteria\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"titan_iac\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"bstein_home\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"data_prepper\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"lesavka\", \"__name__\", \".*\"))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Real Test Cases Present by Suite",
"panel_id":151,
"panel_type":"bargauge",
"description":"Whether real test names are present; 100% means not just placeholder telemetry.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort((100 * (((label_replace(vector(1), \"suite\", \"ariadne\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"metis\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"ananke\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"atlasbot\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"pegasus\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"soteria\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"titan_iac\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"bstein_home\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"data_prepper\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"lesavka\", \"__name__\", \".*\")) and on(suite) count by (suite) (platform_quality_gate_test_case_result{exported_job=\"platform-quality-ci\",test!=\"__no_test_cases__\"})))) or on(suite) (0 * (label_replace(vector(1), \"suite\", \"ariadne\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"metis\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"ananke\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"atlasbot\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"pegasus\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"soteria\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"titan_iac\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"bstein_home\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"data_prepper\", \"__name__\", \".*\") or label_replace(vector(1), \"suite\", \"lesavka\", \"__name__\", \".*\"))))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Primary Branch Clean by Suite (7d)",
"panel_id":150,
"panel_type":"bargauge",
"description":"Percent clean of non-primary branch evidence; 100% means only main/master is reporting.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort((100 * (((count by (suite) (max_over_time(platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}[7d:1h]))) > bool 0) unless on(suite) ((count by (suite) (max_over_time(platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\",branch!~\"main|master|origin/main|origin/master|unknown\"}[7d:1h]))) > bool 0))) or on(suite) (0 * ((count by (suite) (max_over_time(platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}[7d:1h]))) > bool 0)))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Recent Branch Evidence by Suite (7d)",
"panel_id":149,
"panel_type":"bargauge",
"description":"Branches with recent CI evidence; unexpected branches can mean drift or stale work.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"sort_desc(count by (suite, branch) (max_over_time(platform_quality_gate_build_info{suite=~\"${suite:regex}\",branch!=\"\",branch=~\"${branch:regex}\",exported_job=\"platform-quality-ci\"}[7d:1h])))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"SonarQube API Up",
"panel_id":31,
"panel_type":"stat",
"description":"Whether the SonarQube exporter can reach SonarQube; 1 is good.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(max(sonarqube_up) or on() vector(0))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Sonar Projects (Selected)",
"panel_id":32,
"panel_type":"stat",
"description":"Selected SonarQube project count; zero means Sonar is not tracking that suite.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(count(max by (project_key) (sonarqube_project_quality_gate_pass{project_key=~\"${suite:regex}\"})) or on() vector(0))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Sonar Gate Fetch Errors",
"panel_id":33,
"panel_type":"stat",
"description":"Sonar exporter fetch errors; zero is good because stale Sonar data misleads.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"(max(sonarqube_quality_gate_fetch_errors_total) or on() vector(0))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Sonar Gate Status Mix (Selected)",
"panel_id":34,
"panel_type":"piechart",
"description":"Mix of Sonar gate states; OK is good and non-OK needs cleanup.",
"tags":[
"atlas",
"testing",
"quality-gate",
"ci"
],
"datasource_uid":"atlas-vm",
"datasource_type":"prometheus",
"exprs":[
"count by (status) (max by (project_key, status) (sonarqube_project_quality_gate_pass{project_key=~\"${suite:regex}\"}))"
]
},
{
"dashboard":"Atlas Testing",
"panel_title":"Sonar Gate Health by Project",
"panel_id":35,
"panel_type":"state-timeline",
"description":"SonarQube gate status over time by project. OK projects render as full healthy lanes; non-OK projects drop to red without disappearing.",
"panel_title":"Triage Escalations Awaiting a Human",
"panel_id":600,
"panel_type":"stat",
"description":"Incidents Hermes diagnosed where Ariadne refused to act automatically. Each one has a Gitea issue when its job is mapped, and fires HermesTriageHumanRequired.",