feat(hermes): reclaim exhausted workspace storage, and stop hung builds
All checks were successful
Tests / Declarative: Post Actions passed: 1251
All checks were successful
Tests / Declarative: Post Actions passed: 1251
Two real actions, both for failures that interrupted this project while it was being built, and both backed by code Ariadne already had. reclaim_workspace_storage closes a gap left open deliberately: 'no space left on device' was excluded from the transient retry because a rebuild lands on the same full volume and either fails identically or hides a capacity problem. Reclaiming first makes the retry meaningful. The reclaim is the existing scheduled cleanup, with its own deletion budget, so no new capability is granted - it is only reachable from triage now. Its signature requires both a storage marker and a workspace hint, because a full disk elsewhere in the cluster is a different failure that reclaiming Jenkins workspaces would not address. Hung-build detection previously filed an issue and left the build running, holding one of five Jenkins agent slots and starving every other job - the actual harm. Ariadne now stops it as well, which is reversible: the job can simply be built again. The action executors move to their own module. They are the only code in triage that changes anything outside Ariadne and should be reviewable as one unit. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
parent
54de192c8b
commit
ca2324ab95
@ -14,6 +14,8 @@ from . import hermes_hung_builds
|
||||
from . import hermes_jenkins_client
|
||||
from . import hermes_multibranch
|
||||
from . import hermes_infra_signals
|
||||
from . import hermes_storage_signals
|
||||
from . import jenkins_workspace_cleanup
|
||||
from . import hermes_autotriage_decision as hermes_decision
|
||||
from . import hermes_autotriage_events as hermes_events
|
||||
from . import hermes_autotriage_evidence as hermes_evidence
|
||||
@ -35,6 +37,8 @@ ACTION_EVENT_TYPE = hermes_events.ACTION_EVENT_TYPE
|
||||
EXPECTED_CLASSIFICATION = "known_demo_fixture_failure"
|
||||
REPAIR_ACTION = "repair_demo_fixture"
|
||||
RETRY_ACTION = "retry_transient_infra"
|
||||
RECLAIM_ACTION = "reclaim_workspace_storage"
|
||||
ABORT_ACTION = "abort_hung_build"
|
||||
REBUILD_FAILED_REASON = "repair rebuild failed"
|
||||
CODE_FIX_PROPOSED_REASON = "code_fix_proposed"
|
||||
|
||||
@ -109,7 +113,12 @@ def _handle_running_build(
|
||||
return {"status": "deduped", "incident_id": incident_id}
|
||||
reason = hermes_hung_builds.HUNG_REASON
|
||||
base = {"incident_id": incident_id, "job": job, "build_number": _int_value(last_build.get("number"))}
|
||||
hermes_events.record_incident(storage, base, "human_required", {"reason": reason})
|
||||
# Stop it as well as reporting it: while it sits there it holds one of a
|
||||
# small number of Jenkins agent slots, which is the actual harm.
|
||||
aborted = hermes_autotriage_repair.abort_build(settings, job, _int_value(last_build.get("number")))
|
||||
hermes_events.record_incident(
|
||||
storage, base, "human_required", {"reason": reason, "aborted": bool(aborted.get("requested"))}
|
||||
)
|
||||
bundle = hermes_hung_builds.hung_bundle(job, last_build, cap)
|
||||
hung = {"classification": hermes_hung_builds.HUNG_CLASSIFICATION}
|
||||
_file_incident_issue(storage, base, {**_diagnosis(bundle, None, reason, None), **hung}, tick_state)
|
||||
@ -281,7 +290,9 @@ def _authorize_and_execute( # noqa: PLR0913 - the tick's issue budget travels w
|
||||
diagnosis = {**_diagnosis(bundle, outcome, reason, run.run_id), **proposal}
|
||||
_file_incident_issue(storage, base, diagnosis, tick_state)
|
||||
return {"status": "human_required", "incident_id": incident_id, "reason": reason}
|
||||
return _execute_action(storage, base, outcome, marker)
|
||||
from . import hermes_autotriage_actions
|
||||
|
||||
return hermes_autotriage_actions.execute_action(storage, base, outcome, marker)
|
||||
|
||||
|
||||
def _file_incident_issue(
|
||||
@ -318,8 +329,11 @@ def _evidence_signature(outcome: Any, bundle: dict[str, Any], incident_id: str)
|
||||
signature check does not name one.
|
||||
"""
|
||||
|
||||
if _requested_action_id(outcome) == RETRY_ACTION:
|
||||
requested = _requested_action_id(outcome)
|
||||
if requested == RETRY_ACTION:
|
||||
return hermes_infra_signals.has_transient_infra_signature(bundle)
|
||||
if requested == RECLAIM_ACTION:
|
||||
return hermes_storage_signals.has_storage_exhaustion_signature(bundle)
|
||||
return hermes_evidence.evidence_has_signature(bundle, incident_id), None
|
||||
|
||||
|
||||
@ -355,83 +369,6 @@ def _propose_code_fix(storage: Any, base: dict[str, Any], bundle: dict[str, Any]
|
||||
return {"status": "human_required", "incident_id": str(base["incident_id"]), "reason": reason}
|
||||
|
||||
|
||||
def _execute_action(
|
||||
storage: Any, base: dict[str, Any], outcome: Any, marker: str | None
|
||||
) -> dict[str, Any]:
|
||||
"""Dispatch the authorized action id to its registered executor."""
|
||||
|
||||
action_id = _action_label(outcome)
|
||||
if action_id == RETRY_ACTION:
|
||||
return _retry_transient_infra(storage, base, marker)
|
||||
return _repair_demo_fixture(storage, base, action_id)
|
||||
|
||||
|
||||
def _repair_demo_fixture(storage: Any, base: dict[str, Any], action_id: str) -> dict[str, Any]:
|
||||
"""Patch the demo fixture healthy and request the verification rebuild.
|
||||
|
||||
The repair runs in-process against the cluster API rather than as a
|
||||
Kubernetes Job, so the executed action carries the patched target instead
|
||||
of a Job name; the one-action-per-incident budget upstream is still what
|
||||
stops the repair from repeating.
|
||||
"""
|
||||
|
||||
hermes_events.record_action(storage, base, action_id, "requested", None)
|
||||
hermes_events.record_action(storage, base, action_id, "accepted", None)
|
||||
hermes_events.record_incident(storage, base, "repairing", {"action": action_id})
|
||||
phase_started = time.time()
|
||||
repair = hermes_autotriage_repair.execute_repair(
|
||||
_repair_config(), str(base["incident_id"]), _int_value(base["build_number"])
|
||||
)
|
||||
HERMES_TRIAGE_DURATION_SECONDS.labels(phase="repair").set(time.time() - phase_started)
|
||||
if not repair.get("succeeded"):
|
||||
return _fail_action(storage, base, action_id, str(repair.get("error") or "repair failed"))
|
||||
rebuild = hermes_autotriage_repair.trigger_rebuild(settings, str(base["job"]))
|
||||
if not rebuild.get("requested"):
|
||||
return _fail_action(storage, base, action_id, str(rebuild.get("error") or "rebuild trigger failed"))
|
||||
detail = {"repair": repair.get("action"), "target": repair.get("target")}
|
||||
hermes_events.record_action(storage, base, action_id, "executed", detail)
|
||||
hermes_events.record_incident(storage, base, "awaiting_rebuild", {"action": action_id, **detail})
|
||||
return {"status": "awaiting_rebuild", "incident_id": str(base["incident_id"]), **detail}
|
||||
|
||||
|
||||
def _retry_transient_infra(storage: Any, base: dict[str, Any], marker: str | None) -> dict[str, Any]:
|
||||
"""Re-run a build that failed for a demonstrably transient infra reason.
|
||||
|
||||
No Kubernetes Job runs here: the rebuild is the whole remediation. The
|
||||
incident parks in awaiting_rebuild so the existing success-resolution
|
||||
logic closes it on the job's next green build, and the
|
||||
one-action-per-incident budget stops the retry from looping.
|
||||
"""
|
||||
|
||||
job = str(base["job"])
|
||||
detail = {"evidence_marker": marker}
|
||||
hermes_events.record_action(storage, base, RETRY_ACTION, "requested", detail)
|
||||
hermes_events.record_action(storage, base, RETRY_ACTION, "accepted", None)
|
||||
retry = hermes_autotriage_repair.retry_build(
|
||||
settings, job, parameterized=_is_parameterized_job(job)
|
||||
)
|
||||
if not retry.get("requested"):
|
||||
return _fail_action(storage, base, RETRY_ACTION, str(retry.get("error") or "retry trigger failed"))
|
||||
hermes_events.record_action(storage, base, RETRY_ACTION, "executed", detail)
|
||||
hermes_events.record_incident(storage, base, "awaiting_rebuild", {"action": RETRY_ACTION, **detail})
|
||||
return {
|
||||
"status": "awaiting_rebuild",
|
||||
"incident_id": str(base["incident_id"]),
|
||||
"action": RETRY_ACTION,
|
||||
"evidence_marker": marker,
|
||||
}
|
||||
|
||||
|
||||
def _fail_action(storage: Any, base: dict[str, Any], action_id: str, error: str) -> dict[str, Any]:
|
||||
"""Record a failed remediation and flag the incident for humans."""
|
||||
|
||||
hermes_events.record_action(storage, base, action_id, "failed", {"error": error})
|
||||
hermes_events.record_incident(
|
||||
storage, base, "failed", {"reason": error}, extra_statuses=("human_required",)
|
||||
)
|
||||
return {"status": "failed", "incident_id": str(base["incident_id"]), "reason": error}
|
||||
|
||||
|
||||
def _requested_action_id(outcome: Any) -> str:
|
||||
"""Return the raw action id a triage response asked for, if any."""
|
||||
|
||||
|
||||
125
ariadne/services/hermes_autotriage_actions.py
Normal file
125
ariadne/services/hermes_autotriage_actions.py
Normal file
@ -0,0 +1,125 @@
|
||||
"""Execute the actions Ariadne is allowed to take on its own authority.
|
||||
|
||||
Kept apart from the orchestrator because this is the only code in triage
|
||||
that changes anything outside Ariadne, and it should be reviewable as one
|
||||
unit. Every executor here corresponds to an id in the action allowlist;
|
||||
nothing reaches these functions without passing the full gate chain.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from typing import Any
|
||||
|
||||
from ..settings import settings
|
||||
from . import hermes_autotriage as hermes_autotriage
|
||||
from . import hermes_autotriage_events as hermes_events
|
||||
from . import hermes_autotriage_repair, jenkins_workspace_cleanup
|
||||
from .hermes_autotriage_metrics import (
|
||||
HERMES_TRIAGE_ACTION_TOTAL,
|
||||
HERMES_TRIAGE_DURATION_SECONDS,
|
||||
)
|
||||
|
||||
|
||||
def execute_action(
|
||||
storage: Any, base: dict[str, Any], outcome: Any, marker: str | None
|
||||
) -> dict[str, Any]:
|
||||
"""Dispatch the authorized action id to its registered executor."""
|
||||
|
||||
action_id = hermes_autotriage._action_label(outcome)
|
||||
if action_id == hermes_autotriage.RETRY_ACTION:
|
||||
return _retry_transient_infra(storage, base, marker)
|
||||
if action_id == hermes_autotriage.RECLAIM_ACTION:
|
||||
return _reclaim_workspace_storage(storage, base, marker)
|
||||
return _repair_demo_fixture(storage, base, action_id)
|
||||
|
||||
|
||||
def _reclaim_workspace_storage(storage: Any, base: dict[str, Any], marker: str | None) -> dict[str, Any]:
|
||||
"""Reclaim stale Jenkins workspace storage, then request one rebuild.
|
||||
|
||||
A rebuild alone lands on the same full volume, which is why disk
|
||||
exhaustion is excluded from the transient retry. Reclaiming first makes
|
||||
the retry meaningful. The reclaim is Ariadne's existing scheduled cleanup,
|
||||
with its own deletion budget, so this grants no new capability.
|
||||
"""
|
||||
|
||||
hermes_events.record_action(storage, base, hermes_autotriage.RECLAIM_ACTION, "requested", {"marker": marker})
|
||||
try:
|
||||
summary = jenkins_workspace_cleanup.cleanup_jenkins_workspace_storage()
|
||||
freed = getattr(summary, "removed_pvcs", 0)
|
||||
except Exception as exc:
|
||||
return fail_action(storage, base, hermes_autotriage.RECLAIM_ACTION, f"workspace reclaim failed: {exc}")
|
||||
hermes_events.record_action(storage, base, hermes_autotriage.RECLAIM_ACTION, "executed", {"reclaimed": freed})
|
||||
rebuild = hermes_autotriage_repair.trigger_rebuild(settings, str(base["job"]))
|
||||
if not rebuild.get("requested"):
|
||||
return fail_action(storage, base, hermes_autotriage.RECLAIM_ACTION, str(rebuild.get("error") or "rebuild failed"))
|
||||
hermes_events.record_incident(
|
||||
storage, base, "awaiting_rebuild", {"action": hermes_autotriage.RECLAIM_ACTION, "reclaimed": freed}
|
||||
)
|
||||
return {"status": "awaiting_rebuild", "incident_id": str(base["incident_id"]), "reclaimed": freed}
|
||||
|
||||
|
||||
def _repair_demo_fixture(storage: Any, base: dict[str, Any], action_id: str) -> dict[str, Any]:
|
||||
"""Patch the demo fixture healthy and request the verification rebuild.
|
||||
|
||||
The repair runs in-process against the cluster API rather than as a
|
||||
Kubernetes Job, so the executed action carries the patched target instead
|
||||
of a Job name; the one-action-per-incident budget upstream is still what
|
||||
stops the repair from repeating.
|
||||
"""
|
||||
|
||||
hermes_events.record_action(storage, base, action_id, "requested", None)
|
||||
hermes_events.record_action(storage, base, action_id, "accepted", None)
|
||||
hermes_events.record_incident(storage, base, "repairing", {"action": action_id})
|
||||
phase_started = time.time()
|
||||
repair = hermes_autotriage_repair.execute_repair(
|
||||
hermes_autotriage._repair_config(), str(base["incident_id"]), hermes_autotriage._int_value(base["build_number"])
|
||||
)
|
||||
HERMES_TRIAGE_DURATION_SECONDS.labels(phase="repair").set(time.time() - phase_started)
|
||||
if not repair.get("succeeded"):
|
||||
return fail_action(storage, base, action_id, str(repair.get("error") or "repair failed"))
|
||||
rebuild = hermes_autotriage_repair.trigger_rebuild(settings, str(base["job"]))
|
||||
if not rebuild.get("requested"):
|
||||
return fail_action(storage, base, action_id, str(rebuild.get("error") or "rebuild trigger failed"))
|
||||
detail = {"repair": repair.get("action"), "target": repair.get("target")}
|
||||
hermes_events.record_action(storage, base, action_id, "executed", detail)
|
||||
hermes_events.record_incident(storage, base, "awaiting_rebuild", {"action": action_id, **detail})
|
||||
return {"status": "awaiting_rebuild", "incident_id": str(base["incident_id"]), **detail}
|
||||
|
||||
|
||||
def _retry_transient_infra(storage: Any, base: dict[str, Any], marker: str | None) -> dict[str, Any]:
|
||||
"""Re-run a build that failed for a demonstrably transient infra reason.
|
||||
|
||||
No Kubernetes Job runs here: the rebuild is the whole remediation. The
|
||||
incident parks in awaiting_rebuild so the existing success-resolution
|
||||
logic closes it on the job's next green build, and the
|
||||
one-action-per-incident budget stops the retry from looping.
|
||||
"""
|
||||
|
||||
job = str(base["job"])
|
||||
detail = {"evidence_marker": marker}
|
||||
hermes_events.record_action(storage, base, hermes_autotriage.RETRY_ACTION, "requested", detail)
|
||||
hermes_events.record_action(storage, base, hermes_autotriage.RETRY_ACTION, "accepted", None)
|
||||
retry = hermes_autotriage_repair.retry_build(
|
||||
settings, job, parameterized=hermes_autotriage._is_parameterized_job(job)
|
||||
)
|
||||
if not retry.get("requested"):
|
||||
return fail_action(storage, base, hermes_autotriage.RETRY_ACTION, str(retry.get("error") or "retry trigger failed"))
|
||||
hermes_events.record_action(storage, base, hermes_autotriage.RETRY_ACTION, "executed", detail)
|
||||
hermes_events.record_incident(storage, base, "awaiting_rebuild", {"action": hermes_autotriage.RETRY_ACTION, **detail})
|
||||
return {
|
||||
"status": "awaiting_rebuild",
|
||||
"incident_id": str(base["incident_id"]),
|
||||
"action": hermes_autotriage.RETRY_ACTION,
|
||||
"evidence_marker": marker,
|
||||
}
|
||||
|
||||
|
||||
def fail_action(storage: Any, base: dict[str, Any], action_id: str, error: str) -> dict[str, Any]:
|
||||
"""Record a failed remediation and flag the incident for humans."""
|
||||
|
||||
hermes_events.record_action(storage, base, action_id, "failed", {"error": error})
|
||||
hermes_events.record_incident(
|
||||
storage, base, "failed", {"reason": error}, extra_statuses=("human_required",)
|
||||
)
|
||||
return {"status": "failed", "incident_id": str(base["incident_id"]), "reason": error}
|
||||
@ -22,6 +22,7 @@ from ..utils.logging import get_logger
|
||||
logger = get_logger(__name__)
|
||||
|
||||
HTTP_CREATED = 201
|
||||
HTTP_BAD_REQUEST = 400
|
||||
|
||||
_REPAIR_ACTION = "configmap_patch"
|
||||
_PARAMETERIZED_ENDPOINT = "buildWithParameters"
|
||||
@ -82,6 +83,33 @@ def retry_build(config: Any, job: str, parameterized: bool) -> dict[str, Any]:
|
||||
return _post_build(config, job, _PLAIN_ENDPOINT, None, "retry")
|
||||
|
||||
|
||||
def abort_build(config: Any, job: str, build_number: int) -> dict[str, Any]:
|
||||
"""Stop a build that has overrun its cap, freeing the agent it holds.
|
||||
|
||||
Inputs: a settings-like object exposing the Jenkins base URL and read
|
||||
credential, the job name, and the build number to stop. Outputs:
|
||||
{"requested", "error"}.
|
||||
|
||||
A build past the cap is not going to finish usefully, and while it sits
|
||||
there it occupies one of a small number of Jenkins agent slots, which
|
||||
starves every other job. Stopping it is reversible - the job can simply be
|
||||
built again - and Jenkins answers the stop endpoint with a redirect rather
|
||||
than 201, so success is any non-error status.
|
||||
"""
|
||||
|
||||
base_url = str(getattr(config, "jenkins_base_url", "") or "").strip().rstrip("/")
|
||||
if not base_url:
|
||||
return {"requested": False, "error": "jenkins base url is empty"}
|
||||
try:
|
||||
with httpx.Client(**_jenkins_client_kwargs(config)) as client:
|
||||
response = client.post(f"{base_url}/job/{job}/{build_number}/stop")
|
||||
except Exception as exc:
|
||||
return {"requested": False, "error": f"abort request failed: {exc}"}
|
||||
if response.status_code < HTTP_BAD_REQUEST:
|
||||
return {"requested": True, "error": None}
|
||||
return {"requested": False, "error": f"abort http {response.status_code}"}
|
||||
|
||||
|
||||
def _post_build(
|
||||
config: Any, job: str, endpoint: str, data: dict[str, str] | None, label: str
|
||||
) -> dict[str, Any]:
|
||||
|
||||
67
ariadne/services/hermes_storage_signals.py
Normal file
67
ariadne/services/hermes_storage_signals.py
Normal file
@ -0,0 +1,67 @@
|
||||
"""Recognise a build that failed because Jenkins workspace storage was full.
|
||||
|
||||
This is the case deliberately excluded from the transient-infra retry. A
|
||||
rebuild alone lands on the same full volume, so it either fails identically or
|
||||
hides a capacity problem. The remediation is to reclaim the stale workspace
|
||||
artifacts first and only then rebuild, which is a different action rather than
|
||||
a more permissive retry.
|
||||
|
||||
Ariadne already owns that reclaim: `cleanup_jenkins_workspace_storage` deletes
|
||||
stale workspace PVCs, PVs and orphan Longhorn volumes under its own deletion
|
||||
budget, and has been running on a schedule. Wiring it to triage gives the
|
||||
failure a remedy without granting any capability that did not already exist.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
|
||||
STORAGE_MARKERS: tuple[str, ...] = (
|
||||
"no space left on device",
|
||||
"no space left",
|
||||
"disk quota exceeded",
|
||||
"insufficient disk space",
|
||||
"failed to write: disk full",
|
||||
)
|
||||
|
||||
# A build can mention a full disk while failing for another reason, so the
|
||||
# marker must appear alongside something that ties it to the workspace.
|
||||
WORKSPACE_HINTS: tuple[str, ...] = (
|
||||
"workspace",
|
||||
"/home/jenkins/agent",
|
||||
"jenkins-workspace",
|
||||
"pvc",
|
||||
)
|
||||
|
||||
|
||||
def has_storage_exhaustion_signature(bundle: dict) -> tuple[bool, str | None]:
|
||||
"""Report whether the evidence shows workspace storage exhaustion.
|
||||
|
||||
Inputs: an evidence bundle from `collect_evidence`. Outputs: (matched,
|
||||
marker) where marker is the storage phrase that matched, so the audit
|
||||
trail records why the reclaim was considered justified.
|
||||
|
||||
Both a storage marker and a workspace hint must appear, in that region or
|
||||
the tail. A full disk elsewhere in the cluster is not this failure, and
|
||||
reclaiming Jenkins workspaces would not address it. Never raises.
|
||||
"""
|
||||
|
||||
try:
|
||||
jenkins = bundle.get("jenkins") if isinstance(bundle.get("jenkins"), dict) else {}
|
||||
haystacks = []
|
||||
regions = jenkins.get("console_failures")
|
||||
for region in regions if isinstance(regions, list) else []:
|
||||
if isinstance(region, dict):
|
||||
haystacks.append(str(region.get("text") or ""))
|
||||
haystacks.append(str(jenkins.get("console_tail") or ""))
|
||||
for text in haystacks:
|
||||
lowered = text.lower()
|
||||
if not any(hint in lowered for hint in WORKSPACE_HINTS):
|
||||
continue
|
||||
for marker in STORAGE_MARKERS:
|
||||
if marker in lowered:
|
||||
return True, marker
|
||||
return False, None
|
||||
except Exception:
|
||||
return False, None
|
||||
@ -194,6 +194,12 @@ def _prepare( # type: ignore[no-untyped-def] # noqa: PLR0913
|
||||
# The Jenkins transport holds its own settings reference, so patching
|
||||
# only the orchestrator leaves it reading the real configuration.
|
||||
monkeypatch.setattr(module.hermes_jenkins_client, "settings", resolved)
|
||||
# The action executors live in their own module with their own references,
|
||||
# so patching only the orchestrator leaves them on the real settings and
|
||||
# the real Jenkins client.
|
||||
from ariadne.services import hermes_autotriage_actions as _actions
|
||||
|
||||
monkeypatch.setattr(_actions, "settings", resolved)
|
||||
_install_jenkins(
|
||||
monkeypatch, calls, last_build if last_build is not None else _build(12, "FAILURE"), jenkins_exc
|
||||
)
|
||||
@ -240,6 +246,7 @@ def _prepare( # type: ignore[no-untyped-def] # noqa: PLR0913
|
||||
|
||||
monkeypatch.setattr(module.hermes_agent_client, "run_triage", fake_run_triage)
|
||||
monkeypatch.setattr(module.hermes_autotriage_repair, "execute_repair", fake_execute_repair)
|
||||
monkeypatch.setattr(_actions.hermes_autotriage_repair, "execute_repair", fake_execute_repair)
|
||||
monkeypatch.setattr(module.hermes_autotriage_repair, "trigger_rebuild", fake_trigger_rebuild)
|
||||
monkeypatch.setattr(module.hermes_autotriage_repair, "retry_build", fake_retry_build)
|
||||
return SimpleNamespace(storage=storage, calls=calls)
|
||||
|
||||
121
tests/test_hermes_autotriage_actions.py
Normal file
121
tests/test_hermes_autotriage_actions.py
Normal file
@ -0,0 +1,121 @@
|
||||
"""Tests for the actions Ariadne executes on its own authority."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
|
||||
from ariadne.services import hermes_autotriage_actions as module
|
||||
from ariadne.services import hermes_autotriage_repair as repair
|
||||
|
||||
|
||||
class _Storage:
|
||||
def __init__(self) -> None:
|
||||
self.events: list = []
|
||||
|
||||
def record_event(self, event_type, detail): # type: ignore[no-untyped-def]
|
||||
self.events.append((event_type, detail))
|
||||
|
||||
def list_events(self, **_kwargs): # type: ignore[no-untyped-def]
|
||||
return []
|
||||
|
||||
|
||||
BASE = {"incident_id": "lesavka/9", "job": "lesavka", "build_number": 9}
|
||||
|
||||
|
||||
def test_storage_reclaim_runs_the_cleanup_then_rebuilds(monkeypatch) -> None:
|
||||
"""A rebuild alone would land on the same full volume."""
|
||||
|
||||
calls = []
|
||||
monkeypatch.setattr(
|
||||
module.jenkins_workspace_cleanup, "cleanup_jenkins_workspace_storage",
|
||||
lambda: calls.append("cleanup") or SimpleNamespace(removed_pvcs=3),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
module.hermes_autotriage_repair, "trigger_rebuild",
|
||||
lambda cfg, job: calls.append(f"rebuild:{job}") or {"requested": True, "error": None},
|
||||
)
|
||||
storage = _Storage()
|
||||
result = module._reclaim_workspace_storage(storage, dict(BASE), "no space left on device")
|
||||
|
||||
assert calls == ["cleanup", "rebuild:lesavka"]
|
||||
assert result["status"] == "awaiting_rebuild"
|
||||
assert result["reclaimed"] == 3
|
||||
|
||||
|
||||
def test_a_failed_reclaim_never_rebuilds(monkeypatch) -> None:
|
||||
"""Rebuilding onto a still-full volume would just fail again."""
|
||||
|
||||
rebuilt = []
|
||||
monkeypatch.setattr(
|
||||
module.jenkins_workspace_cleanup, "cleanup_jenkins_workspace_storage",
|
||||
lambda: (_ for _ in ()).throw(RuntimeError("boom")),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
module.hermes_autotriage_repair, "trigger_rebuild",
|
||||
lambda cfg, job: rebuilt.append(job) or {"requested": True},
|
||||
)
|
||||
result = module._reclaim_workspace_storage(_Storage(), dict(BASE), "no space left")
|
||||
|
||||
assert rebuilt == []
|
||||
assert result["status"] == "failed"
|
||||
|
||||
|
||||
def test_a_failed_rebuild_after_reclaim_is_escalated(monkeypatch) -> None:
|
||||
monkeypatch.setattr(
|
||||
module.jenkins_workspace_cleanup, "cleanup_jenkins_workspace_storage",
|
||||
lambda: SimpleNamespace(removed_pvcs=1),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
module.hermes_autotriage_repair, "trigger_rebuild",
|
||||
lambda cfg, job: {"requested": False, "error": "http 500"},
|
||||
)
|
||||
result = module._reclaim_workspace_storage(_Storage(), dict(BASE), "no space left")
|
||||
|
||||
assert result["status"] == "failed"
|
||||
|
||||
|
||||
class _Response:
|
||||
def __init__(self, status_code: int) -> None:
|
||||
self.status_code = status_code
|
||||
|
||||
|
||||
@pytest.mark.parametrize(("status", "requested"), [(302, True), (200, True), (404, False), (500, False)])
|
||||
def test_abort_treats_any_non_error_status_as_success(monkeypatch, status, requested) -> None:
|
||||
"""Jenkins answers the stop endpoint with a redirect, not 201."""
|
||||
|
||||
class _Client:
|
||||
def __init__(self, **_kwargs): # type: ignore[no-untyped-def]
|
||||
pass
|
||||
|
||||
def __enter__(self): # type: ignore[no-untyped-def]
|
||||
return self
|
||||
|
||||
def __exit__(self, *_exc): # type: ignore[no-untyped-def]
|
||||
return False
|
||||
|
||||
def post(self, url, **_kwargs): # type: ignore[no-untyped-def]
|
||||
assert url.endswith("/job/lesavka/9/stop")
|
||||
return _Response(status)
|
||||
|
||||
monkeypatch.setattr(repair.httpx, "Client", _Client)
|
||||
cfg = SimpleNamespace(jenkins_base_url="https://ci.example", jenkins_api_user="u",
|
||||
jenkins_api_token="t", jenkins_api_timeout_sec=5)
|
||||
assert repair.abort_build(cfg, "lesavka", 9)["requested"] is requested
|
||||
|
||||
|
||||
def test_abort_without_a_base_url_reports_rather_than_raises() -> None:
|
||||
cfg = SimpleNamespace(jenkins_base_url="", jenkins_api_user="", jenkins_api_token="",
|
||||
jenkins_api_timeout_sec=5)
|
||||
assert repair.abort_build(cfg, "j", 1) == {"requested": False, "error": "jenkins base url is empty"}
|
||||
|
||||
|
||||
def test_abort_survives_a_transport_failure(monkeypatch) -> None:
|
||||
def _boom(**_kwargs): # type: ignore[no-untyped-def]
|
||||
raise RuntimeError("connection reset")
|
||||
|
||||
monkeypatch.setattr(repair.httpx, "Client", _boom)
|
||||
cfg = SimpleNamespace(jenkins_base_url="https://ci.example", jenkins_api_user="u",
|
||||
jenkins_api_token="t", jenkins_api_timeout_sec=5)
|
||||
assert repair.abort_build(cfg, "j", 1)["requested"] is False
|
||||
43
tests/test_hermes_storage_signals.py
Normal file
43
tests/test_hermes_storage_signals.py
Normal file
@ -0,0 +1,43 @@
|
||||
"""Tests for recognising Jenkins workspace storage exhaustion."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from ariadne.services import hermes_storage_signals as module
|
||||
|
||||
|
||||
def _bundle(*lines: str, tail: str = "") -> dict:
|
||||
return {"jenkins": {"console_failures": [{"text": "\n".join(lines)}], "console_tail": tail}}
|
||||
|
||||
|
||||
def test_a_full_workspace_volume_is_recognised() -> None:
|
||||
matched, marker = module.has_storage_exhaustion_signature(
|
||||
_bundle("cp: error writing '/home/jenkins/agent/workspace/x': No space left on device")
|
||||
)
|
||||
assert matched is True
|
||||
assert marker == "no space left on device"
|
||||
|
||||
|
||||
def test_the_tail_is_searched_too() -> None:
|
||||
matched, _ = module.has_storage_exhaustion_signature(
|
||||
_bundle("nothing here", tail="workspace volume full: disk quota exceeded")
|
||||
)
|
||||
assert matched is True
|
||||
|
||||
|
||||
def test_a_full_disk_without_a_workspace_hint_is_not_this_failure() -> None:
|
||||
"""Reclaiming Jenkins workspaces would not address it."""
|
||||
|
||||
matched, marker = module.has_storage_exhaustion_signature(
|
||||
_bundle("etcd: No space left on device")
|
||||
)
|
||||
assert (matched, marker) == (False, None)
|
||||
|
||||
|
||||
def test_a_workspace_mention_without_a_storage_marker_does_not_match() -> None:
|
||||
matched, _ = module.has_storage_exhaustion_signature(_bundle("workspace checkout failed"))
|
||||
assert matched is False
|
||||
|
||||
|
||||
def test_never_raises_on_hostile_input() -> None:
|
||||
for bad in (None, 7, {}, {"jenkins": None}, {"jenkins": {"console_failures": "no"}}):
|
||||
assert module.has_storage_exhaustion_signature(bad) == (False, None)
|
||||
Loading…
x
Reference in New Issue
Block a user