feat(hermes): reclaim exhausted workspace storage, and stop hung builds
All checks were successful
Tests / Declarative: Post Actions passed: 1251

Two real actions, both for failures that interrupted this project while it was
being built, and both backed by code Ariadne already had.

reclaim_workspace_storage closes a gap left open deliberately: 'no space left
on device' was excluded from the transient retry because a rebuild lands on
the same full volume and either fails identically or hides a capacity problem.
Reclaiming first makes the retry meaningful. The reclaim is the existing
scheduled cleanup, with its own deletion budget, so no new capability is
granted - it is only reachable from triage now. Its signature requires both a
storage marker and a workspace hint, because a full disk elsewhere in the
cluster is a different failure that reclaiming Jenkins workspaces would not
address.

Hung-build detection previously filed an issue and left the build running,
holding one of five Jenkins agent slots and starving every other job - the
actual harm. Ariadne now stops it as well, which is reversible: the job can
simply be built again.

The action executors move to their own module. They are the only code in
triage that changes anything outside Ariadne and should be reviewable as one
unit.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
codex 2026-08-06 20:55:35 -03:00
parent 54de192c8b
commit ca2324ab95
7 changed files with 408 additions and 80 deletions

View File

@ -14,6 +14,8 @@ from . import hermes_hung_builds
from . import hermes_jenkins_client
from . import hermes_multibranch
from . import hermes_infra_signals
from . import hermes_storage_signals
from . import jenkins_workspace_cleanup
from . import hermes_autotriage_decision as hermes_decision
from . import hermes_autotriage_events as hermes_events
from . import hermes_autotriage_evidence as hermes_evidence
@ -35,6 +37,8 @@ ACTION_EVENT_TYPE = hermes_events.ACTION_EVENT_TYPE
EXPECTED_CLASSIFICATION = "known_demo_fixture_failure"
REPAIR_ACTION = "repair_demo_fixture"
RETRY_ACTION = "retry_transient_infra"
RECLAIM_ACTION = "reclaim_workspace_storage"
ABORT_ACTION = "abort_hung_build"
REBUILD_FAILED_REASON = "repair rebuild failed"
CODE_FIX_PROPOSED_REASON = "code_fix_proposed"
@ -109,7 +113,12 @@ def _handle_running_build(
return {"status": "deduped", "incident_id": incident_id}
reason = hermes_hung_builds.HUNG_REASON
base = {"incident_id": incident_id, "job": job, "build_number": _int_value(last_build.get("number"))}
hermes_events.record_incident(storage, base, "human_required", {"reason": reason})
# Stop it as well as reporting it: while it sits there it holds one of a
# small number of Jenkins agent slots, which is the actual harm.
aborted = hermes_autotriage_repair.abort_build(settings, job, _int_value(last_build.get("number")))
hermes_events.record_incident(
storage, base, "human_required", {"reason": reason, "aborted": bool(aborted.get("requested"))}
)
bundle = hermes_hung_builds.hung_bundle(job, last_build, cap)
hung = {"classification": hermes_hung_builds.HUNG_CLASSIFICATION}
_file_incident_issue(storage, base, {**_diagnosis(bundle, None, reason, None), **hung}, tick_state)
@ -281,7 +290,9 @@ def _authorize_and_execute( # noqa: PLR0913 - the tick's issue budget travels w
diagnosis = {**_diagnosis(bundle, outcome, reason, run.run_id), **proposal}
_file_incident_issue(storage, base, diagnosis, tick_state)
return {"status": "human_required", "incident_id": incident_id, "reason": reason}
return _execute_action(storage, base, outcome, marker)
from . import hermes_autotriage_actions
return hermes_autotriage_actions.execute_action(storage, base, outcome, marker)
def _file_incident_issue(
@ -318,8 +329,11 @@ def _evidence_signature(outcome: Any, bundle: dict[str, Any], incident_id: str)
signature check does not name one.
"""
if _requested_action_id(outcome) == RETRY_ACTION:
requested = _requested_action_id(outcome)
if requested == RETRY_ACTION:
return hermes_infra_signals.has_transient_infra_signature(bundle)
if requested == RECLAIM_ACTION:
return hermes_storage_signals.has_storage_exhaustion_signature(bundle)
return hermes_evidence.evidence_has_signature(bundle, incident_id), None
@ -355,83 +369,6 @@ def _propose_code_fix(storage: Any, base: dict[str, Any], bundle: dict[str, Any]
return {"status": "human_required", "incident_id": str(base["incident_id"]), "reason": reason}
def _execute_action(
storage: Any, base: dict[str, Any], outcome: Any, marker: str | None
) -> dict[str, Any]:
"""Dispatch the authorized action id to its registered executor."""
action_id = _action_label(outcome)
if action_id == RETRY_ACTION:
return _retry_transient_infra(storage, base, marker)
return _repair_demo_fixture(storage, base, action_id)
def _repair_demo_fixture(storage: Any, base: dict[str, Any], action_id: str) -> dict[str, Any]:
"""Patch the demo fixture healthy and request the verification rebuild.
The repair runs in-process against the cluster API rather than as a
Kubernetes Job, so the executed action carries the patched target instead
of a Job name; the one-action-per-incident budget upstream is still what
stops the repair from repeating.
"""
hermes_events.record_action(storage, base, action_id, "requested", None)
hermes_events.record_action(storage, base, action_id, "accepted", None)
hermes_events.record_incident(storage, base, "repairing", {"action": action_id})
phase_started = time.time()
repair = hermes_autotriage_repair.execute_repair(
_repair_config(), str(base["incident_id"]), _int_value(base["build_number"])
)
HERMES_TRIAGE_DURATION_SECONDS.labels(phase="repair").set(time.time() - phase_started)
if not repair.get("succeeded"):
return _fail_action(storage, base, action_id, str(repair.get("error") or "repair failed"))
rebuild = hermes_autotriage_repair.trigger_rebuild(settings, str(base["job"]))
if not rebuild.get("requested"):
return _fail_action(storage, base, action_id, str(rebuild.get("error") or "rebuild trigger failed"))
detail = {"repair": repair.get("action"), "target": repair.get("target")}
hermes_events.record_action(storage, base, action_id, "executed", detail)
hermes_events.record_incident(storage, base, "awaiting_rebuild", {"action": action_id, **detail})
return {"status": "awaiting_rebuild", "incident_id": str(base["incident_id"]), **detail}
def _retry_transient_infra(storage: Any, base: dict[str, Any], marker: str | None) -> dict[str, Any]:
"""Re-run a build that failed for a demonstrably transient infra reason.
No Kubernetes Job runs here: the rebuild is the whole remediation. The
incident parks in awaiting_rebuild so the existing success-resolution
logic closes it on the job's next green build, and the
one-action-per-incident budget stops the retry from looping.
"""
job = str(base["job"])
detail = {"evidence_marker": marker}
hermes_events.record_action(storage, base, RETRY_ACTION, "requested", detail)
hermes_events.record_action(storage, base, RETRY_ACTION, "accepted", None)
retry = hermes_autotriage_repair.retry_build(
settings, job, parameterized=_is_parameterized_job(job)
)
if not retry.get("requested"):
return _fail_action(storage, base, RETRY_ACTION, str(retry.get("error") or "retry trigger failed"))
hermes_events.record_action(storage, base, RETRY_ACTION, "executed", detail)
hermes_events.record_incident(storage, base, "awaiting_rebuild", {"action": RETRY_ACTION, **detail})
return {
"status": "awaiting_rebuild",
"incident_id": str(base["incident_id"]),
"action": RETRY_ACTION,
"evidence_marker": marker,
}
def _fail_action(storage: Any, base: dict[str, Any], action_id: str, error: str) -> dict[str, Any]:
"""Record a failed remediation and flag the incident for humans."""
hermes_events.record_action(storage, base, action_id, "failed", {"error": error})
hermes_events.record_incident(
storage, base, "failed", {"reason": error}, extra_statuses=("human_required",)
)
return {"status": "failed", "incident_id": str(base["incident_id"]), "reason": error}
def _requested_action_id(outcome: Any) -> str:
"""Return the raw action id a triage response asked for, if any."""

View File

@ -0,0 +1,125 @@
"""Execute the actions Ariadne is allowed to take on its own authority.
Kept apart from the orchestrator because this is the only code in triage
that changes anything outside Ariadne, and it should be reviewable as one
unit. Every executor here corresponds to an id in the action allowlist;
nothing reaches these functions without passing the full gate chain.
"""
from __future__ import annotations
import time
from typing import Any
from ..settings import settings
from . import hermes_autotriage as hermes_autotriage
from . import hermes_autotriage_events as hermes_events
from . import hermes_autotriage_repair, jenkins_workspace_cleanup
from .hermes_autotriage_metrics import (
HERMES_TRIAGE_ACTION_TOTAL,
HERMES_TRIAGE_DURATION_SECONDS,
)
def execute_action(
storage: Any, base: dict[str, Any], outcome: Any, marker: str | None
) -> dict[str, Any]:
"""Dispatch the authorized action id to its registered executor."""
action_id = hermes_autotriage._action_label(outcome)
if action_id == hermes_autotriage.RETRY_ACTION:
return _retry_transient_infra(storage, base, marker)
if action_id == hermes_autotriage.RECLAIM_ACTION:
return _reclaim_workspace_storage(storage, base, marker)
return _repair_demo_fixture(storage, base, action_id)
def _reclaim_workspace_storage(storage: Any, base: dict[str, Any], marker: str | None) -> dict[str, Any]:
"""Reclaim stale Jenkins workspace storage, then request one rebuild.
A rebuild alone lands on the same full volume, which is why disk
exhaustion is excluded from the transient retry. Reclaiming first makes
the retry meaningful. The reclaim is Ariadne's existing scheduled cleanup,
with its own deletion budget, so this grants no new capability.
"""
hermes_events.record_action(storage, base, hermes_autotriage.RECLAIM_ACTION, "requested", {"marker": marker})
try:
summary = jenkins_workspace_cleanup.cleanup_jenkins_workspace_storage()
freed = getattr(summary, "removed_pvcs", 0)
except Exception as exc:
return fail_action(storage, base, hermes_autotriage.RECLAIM_ACTION, f"workspace reclaim failed: {exc}")
hermes_events.record_action(storage, base, hermes_autotriage.RECLAIM_ACTION, "executed", {"reclaimed": freed})
rebuild = hermes_autotriage_repair.trigger_rebuild(settings, str(base["job"]))
if not rebuild.get("requested"):
return fail_action(storage, base, hermes_autotriage.RECLAIM_ACTION, str(rebuild.get("error") or "rebuild failed"))
hermes_events.record_incident(
storage, base, "awaiting_rebuild", {"action": hermes_autotriage.RECLAIM_ACTION, "reclaimed": freed}
)
return {"status": "awaiting_rebuild", "incident_id": str(base["incident_id"]), "reclaimed": freed}
def _repair_demo_fixture(storage: Any, base: dict[str, Any], action_id: str) -> dict[str, Any]:
"""Patch the demo fixture healthy and request the verification rebuild.
The repair runs in-process against the cluster API rather than as a
Kubernetes Job, so the executed action carries the patched target instead
of a Job name; the one-action-per-incident budget upstream is still what
stops the repair from repeating.
"""
hermes_events.record_action(storage, base, action_id, "requested", None)
hermes_events.record_action(storage, base, action_id, "accepted", None)
hermes_events.record_incident(storage, base, "repairing", {"action": action_id})
phase_started = time.time()
repair = hermes_autotriage_repair.execute_repair(
hermes_autotriage._repair_config(), str(base["incident_id"]), hermes_autotriage._int_value(base["build_number"])
)
HERMES_TRIAGE_DURATION_SECONDS.labels(phase="repair").set(time.time() - phase_started)
if not repair.get("succeeded"):
return fail_action(storage, base, action_id, str(repair.get("error") or "repair failed"))
rebuild = hermes_autotriage_repair.trigger_rebuild(settings, str(base["job"]))
if not rebuild.get("requested"):
return fail_action(storage, base, action_id, str(rebuild.get("error") or "rebuild trigger failed"))
detail = {"repair": repair.get("action"), "target": repair.get("target")}
hermes_events.record_action(storage, base, action_id, "executed", detail)
hermes_events.record_incident(storage, base, "awaiting_rebuild", {"action": action_id, **detail})
return {"status": "awaiting_rebuild", "incident_id": str(base["incident_id"]), **detail}
def _retry_transient_infra(storage: Any, base: dict[str, Any], marker: str | None) -> dict[str, Any]:
"""Re-run a build that failed for a demonstrably transient infra reason.
No Kubernetes Job runs here: the rebuild is the whole remediation. The
incident parks in awaiting_rebuild so the existing success-resolution
logic closes it on the job's next green build, and the
one-action-per-incident budget stops the retry from looping.
"""
job = str(base["job"])
detail = {"evidence_marker": marker}
hermes_events.record_action(storage, base, hermes_autotriage.RETRY_ACTION, "requested", detail)
hermes_events.record_action(storage, base, hermes_autotriage.RETRY_ACTION, "accepted", None)
retry = hermes_autotriage_repair.retry_build(
settings, job, parameterized=hermes_autotriage._is_parameterized_job(job)
)
if not retry.get("requested"):
return fail_action(storage, base, hermes_autotriage.RETRY_ACTION, str(retry.get("error") or "retry trigger failed"))
hermes_events.record_action(storage, base, hermes_autotriage.RETRY_ACTION, "executed", detail)
hermes_events.record_incident(storage, base, "awaiting_rebuild", {"action": hermes_autotriage.RETRY_ACTION, **detail})
return {
"status": "awaiting_rebuild",
"incident_id": str(base["incident_id"]),
"action": hermes_autotriage.RETRY_ACTION,
"evidence_marker": marker,
}
def fail_action(storage: Any, base: dict[str, Any], action_id: str, error: str) -> dict[str, Any]:
"""Record a failed remediation and flag the incident for humans."""
hermes_events.record_action(storage, base, action_id, "failed", {"error": error})
hermes_events.record_incident(
storage, base, "failed", {"reason": error}, extra_statuses=("human_required",)
)
return {"status": "failed", "incident_id": str(base["incident_id"]), "reason": error}

View File

@ -22,6 +22,7 @@ from ..utils.logging import get_logger
logger = get_logger(__name__)
HTTP_CREATED = 201
HTTP_BAD_REQUEST = 400
_REPAIR_ACTION = "configmap_patch"
_PARAMETERIZED_ENDPOINT = "buildWithParameters"
@ -82,6 +83,33 @@ def retry_build(config: Any, job: str, parameterized: bool) -> dict[str, Any]:
return _post_build(config, job, _PLAIN_ENDPOINT, None, "retry")
def abort_build(config: Any, job: str, build_number: int) -> dict[str, Any]:
"""Stop a build that has overrun its cap, freeing the agent it holds.
Inputs: a settings-like object exposing the Jenkins base URL and read
credential, the job name, and the build number to stop. Outputs:
{"requested", "error"}.
A build past the cap is not going to finish usefully, and while it sits
there it occupies one of a small number of Jenkins agent slots, which
starves every other job. Stopping it is reversible - the job can simply be
built again - and Jenkins answers the stop endpoint with a redirect rather
than 201, so success is any non-error status.
"""
base_url = str(getattr(config, "jenkins_base_url", "") or "").strip().rstrip("/")
if not base_url:
return {"requested": False, "error": "jenkins base url is empty"}
try:
with httpx.Client(**_jenkins_client_kwargs(config)) as client:
response = client.post(f"{base_url}/job/{job}/{build_number}/stop")
except Exception as exc:
return {"requested": False, "error": f"abort request failed: {exc}"}
if response.status_code < HTTP_BAD_REQUEST:
return {"requested": True, "error": None}
return {"requested": False, "error": f"abort http {response.status_code}"}
def _post_build(
config: Any, job: str, endpoint: str, data: dict[str, str] | None, label: str
) -> dict[str, Any]:

View File

@ -0,0 +1,67 @@
"""Recognise a build that failed because Jenkins workspace storage was full.
This is the case deliberately excluded from the transient-infra retry. A
rebuild alone lands on the same full volume, so it either fails identically or
hides a capacity problem. The remediation is to reclaim the stale workspace
artifacts first and only then rebuild, which is a different action rather than
a more permissive retry.
Ariadne already owns that reclaim: `cleanup_jenkins_workspace_storage` deletes
stale workspace PVCs, PVs and orphan Longhorn volumes under its own deletion
budget, and has been running on a schedule. Wiring it to triage gives the
failure a remedy without granting any capability that did not already exist.
"""
from __future__ import annotations
from typing import Any
STORAGE_MARKERS: tuple[str, ...] = (
"no space left on device",
"no space left",
"disk quota exceeded",
"insufficient disk space",
"failed to write: disk full",
)
# A build can mention a full disk while failing for another reason, so the
# marker must appear alongside something that ties it to the workspace.
WORKSPACE_HINTS: tuple[str, ...] = (
"workspace",
"/home/jenkins/agent",
"jenkins-workspace",
"pvc",
)
def has_storage_exhaustion_signature(bundle: dict) -> tuple[bool, str | None]:
"""Report whether the evidence shows workspace storage exhaustion.
Inputs: an evidence bundle from `collect_evidence`. Outputs: (matched,
marker) where marker is the storage phrase that matched, so the audit
trail records why the reclaim was considered justified.
Both a storage marker and a workspace hint must appear, in that region or
the tail. A full disk elsewhere in the cluster is not this failure, and
reclaiming Jenkins workspaces would not address it. Never raises.
"""
try:
jenkins = bundle.get("jenkins") if isinstance(bundle.get("jenkins"), dict) else {}
haystacks = []
regions = jenkins.get("console_failures")
for region in regions if isinstance(regions, list) else []:
if isinstance(region, dict):
haystacks.append(str(region.get("text") or ""))
haystacks.append(str(jenkins.get("console_tail") or ""))
for text in haystacks:
lowered = text.lower()
if not any(hint in lowered for hint in WORKSPACE_HINTS):
continue
for marker in STORAGE_MARKERS:
if marker in lowered:
return True, marker
return False, None
except Exception:
return False, None

View File

@ -194,6 +194,12 @@ def _prepare( # type: ignore[no-untyped-def] # noqa: PLR0913
# The Jenkins transport holds its own settings reference, so patching
# only the orchestrator leaves it reading the real configuration.
monkeypatch.setattr(module.hermes_jenkins_client, "settings", resolved)
# The action executors live in their own module with their own references,
# so patching only the orchestrator leaves them on the real settings and
# the real Jenkins client.
from ariadne.services import hermes_autotriage_actions as _actions
monkeypatch.setattr(_actions, "settings", resolved)
_install_jenkins(
monkeypatch, calls, last_build if last_build is not None else _build(12, "FAILURE"), jenkins_exc
)
@ -240,6 +246,7 @@ def _prepare( # type: ignore[no-untyped-def] # noqa: PLR0913
monkeypatch.setattr(module.hermes_agent_client, "run_triage", fake_run_triage)
monkeypatch.setattr(module.hermes_autotriage_repair, "execute_repair", fake_execute_repair)
monkeypatch.setattr(_actions.hermes_autotriage_repair, "execute_repair", fake_execute_repair)
monkeypatch.setattr(module.hermes_autotriage_repair, "trigger_rebuild", fake_trigger_rebuild)
monkeypatch.setattr(module.hermes_autotriage_repair, "retry_build", fake_retry_build)
return SimpleNamespace(storage=storage, calls=calls)

View File

@ -0,0 +1,121 @@
"""Tests for the actions Ariadne executes on its own authority."""
from __future__ import annotations
from types import SimpleNamespace
import pytest
from ariadne.services import hermes_autotriage_actions as module
from ariadne.services import hermes_autotriage_repair as repair
class _Storage:
def __init__(self) -> None:
self.events: list = []
def record_event(self, event_type, detail): # type: ignore[no-untyped-def]
self.events.append((event_type, detail))
def list_events(self, **_kwargs): # type: ignore[no-untyped-def]
return []
BASE = {"incident_id": "lesavka/9", "job": "lesavka", "build_number": 9}
def test_storage_reclaim_runs_the_cleanup_then_rebuilds(monkeypatch) -> None:
"""A rebuild alone would land on the same full volume."""
calls = []
monkeypatch.setattr(
module.jenkins_workspace_cleanup, "cleanup_jenkins_workspace_storage",
lambda: calls.append("cleanup") or SimpleNamespace(removed_pvcs=3),
)
monkeypatch.setattr(
module.hermes_autotriage_repair, "trigger_rebuild",
lambda cfg, job: calls.append(f"rebuild:{job}") or {"requested": True, "error": None},
)
storage = _Storage()
result = module._reclaim_workspace_storage(storage, dict(BASE), "no space left on device")
assert calls == ["cleanup", "rebuild:lesavka"]
assert result["status"] == "awaiting_rebuild"
assert result["reclaimed"] == 3
def test_a_failed_reclaim_never_rebuilds(monkeypatch) -> None:
"""Rebuilding onto a still-full volume would just fail again."""
rebuilt = []
monkeypatch.setattr(
module.jenkins_workspace_cleanup, "cleanup_jenkins_workspace_storage",
lambda: (_ for _ in ()).throw(RuntimeError("boom")),
)
monkeypatch.setattr(
module.hermes_autotriage_repair, "trigger_rebuild",
lambda cfg, job: rebuilt.append(job) or {"requested": True},
)
result = module._reclaim_workspace_storage(_Storage(), dict(BASE), "no space left")
assert rebuilt == []
assert result["status"] == "failed"
def test_a_failed_rebuild_after_reclaim_is_escalated(monkeypatch) -> None:
monkeypatch.setattr(
module.jenkins_workspace_cleanup, "cleanup_jenkins_workspace_storage",
lambda: SimpleNamespace(removed_pvcs=1),
)
monkeypatch.setattr(
module.hermes_autotriage_repair, "trigger_rebuild",
lambda cfg, job: {"requested": False, "error": "http 500"},
)
result = module._reclaim_workspace_storage(_Storage(), dict(BASE), "no space left")
assert result["status"] == "failed"
class _Response:
def __init__(self, status_code: int) -> None:
self.status_code = status_code
@pytest.mark.parametrize(("status", "requested"), [(302, True), (200, True), (404, False), (500, False)])
def test_abort_treats_any_non_error_status_as_success(monkeypatch, status, requested) -> None:
"""Jenkins answers the stop endpoint with a redirect, not 201."""
class _Client:
def __init__(self, **_kwargs): # type: ignore[no-untyped-def]
pass
def __enter__(self): # type: ignore[no-untyped-def]
return self
def __exit__(self, *_exc): # type: ignore[no-untyped-def]
return False
def post(self, url, **_kwargs): # type: ignore[no-untyped-def]
assert url.endswith("/job/lesavka/9/stop")
return _Response(status)
monkeypatch.setattr(repair.httpx, "Client", _Client)
cfg = SimpleNamespace(jenkins_base_url="https://ci.example", jenkins_api_user="u",
jenkins_api_token="t", jenkins_api_timeout_sec=5)
assert repair.abort_build(cfg, "lesavka", 9)["requested"] is requested
def test_abort_without_a_base_url_reports_rather_than_raises() -> None:
cfg = SimpleNamespace(jenkins_base_url="", jenkins_api_user="", jenkins_api_token="",
jenkins_api_timeout_sec=5)
assert repair.abort_build(cfg, "j", 1) == {"requested": False, "error": "jenkins base url is empty"}
def test_abort_survives_a_transport_failure(monkeypatch) -> None:
def _boom(**_kwargs): # type: ignore[no-untyped-def]
raise RuntimeError("connection reset")
monkeypatch.setattr(repair.httpx, "Client", _boom)
cfg = SimpleNamespace(jenkins_base_url="https://ci.example", jenkins_api_user="u",
jenkins_api_token="t", jenkins_api_timeout_sec=5)
assert repair.abort_build(cfg, "j", 1)["requested"] is False

View File

@ -0,0 +1,43 @@
"""Tests for recognising Jenkins workspace storage exhaustion."""
from __future__ import annotations
from ariadne.services import hermes_storage_signals as module
def _bundle(*lines: str, tail: str = "") -> dict:
return {"jenkins": {"console_failures": [{"text": "\n".join(lines)}], "console_tail": tail}}
def test_a_full_workspace_volume_is_recognised() -> None:
matched, marker = module.has_storage_exhaustion_signature(
_bundle("cp: error writing '/home/jenkins/agent/workspace/x': No space left on device")
)
assert matched is True
assert marker == "no space left on device"
def test_the_tail_is_searched_too() -> None:
matched, _ = module.has_storage_exhaustion_signature(
_bundle("nothing here", tail="workspace volume full: disk quota exceeded")
)
assert matched is True
def test_a_full_disk_without_a_workspace_hint_is_not_this_failure() -> None:
"""Reclaiming Jenkins workspaces would not address it."""
matched, marker = module.has_storage_exhaustion_signature(
_bundle("etcd: No space left on device")
)
assert (matched, marker) == (False, None)
def test_a_workspace_mention_without_a_storage_marker_does_not_match() -> None:
matched, _ = module.has_storage_exhaustion_signature(_bundle("workspace checkout failed"))
assert matched is False
def test_never_raises_on_hostile_input() -> None:
for bad in (None, 7, {}, {"jenkins": None}, {"jenkins": {"console_failures": "no"}}):
assert module.has_storage_exhaustion_signature(bad) == (False, None)