ops: report preemption and terminating service pods
This commit is contained in:
parent
eb1eb3698a
commit
d85642adc1
@ -25,6 +25,7 @@ QUERIES = {
|
||||
"helm": ["helmreleases.helm.toolkit.fluxcd.io", "-A"],
|
||||
"volumes": ["volumes.longhorn.io", "-n", "longhorn-system"],
|
||||
"warnings": ["events", "-A", "--field-selector=type=Warning"],
|
||||
"preemptions": ["events", "-A", "--field-selector=reason=Preempted"],
|
||||
}
|
||||
|
||||
|
||||
@ -73,6 +74,7 @@ def snapshot() -> dict:
|
||||
"uid": p["metadata"]["uid"], "node": p["spec"].get("nodeName"),
|
||||
"owner": [(o["kind"], o["name"]) for o in owners],
|
||||
"phase": p["status"]["phase"], "ready": conditions(p).get("Ready"),
|
||||
"deleting": p["metadata"].get("deletionTimestamp"),
|
||||
"restarts": {s["name"]: s.get("restartCount", 0) for s in statuses},
|
||||
"waiting": {s["name"]: s["state"]["waiting"].get("reason")
|
||||
for s in statuses if "waiting" in s.get("state", {})},
|
||||
@ -111,7 +113,8 @@ def snapshot() -> dict:
|
||||
"last_observed": e.get("series", {}).get("lastObservedTime")
|
||||
or e.get("lastTimestamp") or e.get("eventTime"),
|
||||
}
|
||||
for e in raw["warnings"]
|
||||
# Scheduler preemption can be a Normal event despite interrupting work.
|
||||
for e in [*raw["warnings"], *raw["preemptions"]]
|
||||
}
|
||||
return result
|
||||
|
||||
@ -130,8 +133,10 @@ def summarize(current: dict, previous: dict | None = None) -> dict:
|
||||
or v.get("rollout_failed", False)
|
||||
or (v.get("observed_generation") or 0) < v["generation"]},
|
||||
"unready_pods_on_reachable_nodes": [n for n, v in current["pods"].items()
|
||||
if v["ready"] != "True" and v["node"] not in offline],
|
||||
"unready_pods_on_offline_nodes": sum(v["ready"] != "True" and v["node"] in offline
|
||||
if (v["ready"] != "True" or v.get("deleting"))
|
||||
and v["node"] not in offline],
|
||||
"unready_pods_on_offline_nodes": sum(bool(v["ready"] != "True" or v.get("deleting"))
|
||||
and v["node"] in offline
|
||||
for v in current["pods"].values()),
|
||||
"unhealthy_attached_volumes": {n: v for n, v in current["volumes"].items()
|
||||
if v["state"] == "attached" and v["robustness"] != "healthy"},
|
||||
|
||||
@ -77,6 +77,15 @@ def test_serving_old_replica_does_not_hide_failed_rollout():
|
||||
assert list(summarize(current)["unready_workloads"]) == ["deployments/app/service"]
|
||||
|
||||
|
||||
def test_terminating_ready_pod_is_not_reported_available():
|
||||
"""A preempted pod can retain Ready while its termination grace runs."""
|
||||
current = baseline()
|
||||
current['pods']['app/worker'] = {
|
||||
'node': 'worker', 'ready': 'True', 'deleting': '2026-10-04T09:00:00Z',
|
||||
}
|
||||
assert summarize(current)['unready_pods_on_reachable_nodes'] == ['app/worker']
|
||||
|
||||
|
||||
def test_snapshot_omits_content_and_short_lived_jobs(monkeypatch):
|
||||
"""Only operational metadata leaves raw API objects; Jobs are not service churn."""
|
||||
import json
|
||||
@ -100,11 +109,17 @@ def test_snapshot_omits_content_and_short_lived_jobs(monkeypatch):
|
||||
'metadata': {'uid': 'event'}, 'involvedObject': {'namespace': 'app', 'name': 'service'},
|
||||
'reason': 'Unhealthy', 'count': 1, 'message': hidden,
|
||||
}]
|
||||
data['preemptions'] = [{
|
||||
'metadata': {'uid': 'scheduler-event'},
|
||||
'involvedObject': {'namespace': 'app', 'name': 'service'},
|
||||
'type': 'Normal', 'reason': 'Preempted', 'count': 1, 'message': hidden,
|
||||
}]
|
||||
monkeypatch.setattr(checker, 'fetch', lambda item: (item[0], data[item[0]]))
|
||||
current = checker.snapshot()
|
||||
assert list(current['pods']) == ['app/service']
|
||||
assert hidden not in json.dumps(current)
|
||||
assert current['warnings']['event']['reason'] == 'Unhealthy'
|
||||
assert current['warnings']['scheduler-event']['reason'] == 'Preempted'
|
||||
|
||||
|
||||
def test_failed_api_read_cannot_become_a_partial_health_report(monkeypatch):
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user