diff --git a/ariadne/services/hermes_autotriage.py b/ariadne/services/hermes_autotriage.py index c477f67..7270af8 100644 --- a/ariadne/services/hermes_autotriage.py +++ b/ariadne/services/hermes_autotriage.py @@ -10,6 +10,7 @@ from ..settings import settings from ..utils.logging import get_logger from . import hermes_agent_client, hermes_autotriage_repair, hermes_code_flow, hermes_incident_issue from . import hermes_triage_prompt +from . import hermes_hung_builds from . import hermes_infra_signals from . import hermes_autotriage_decision as hermes_decision from . import hermes_autotriage_events as hermes_events @@ -70,8 +71,10 @@ def _process_job( """Inspect one allowlisted job's last build and advance its incidents.""" last_build = _fetch_last_build(job) - if last_build is None or last_build.get("building") or last_build.get("number") is None: + if last_build is None or last_build.get("number") is None: return {"status": "skipped"} + if last_build.get("building"): + return _handle_running_build(storage, job, last_build, incidents, tick_state) refresh_incident_gauges(job, last_build, incidents) result = str(last_build.get("result") or "").upper() if result == "SUCCESS": @@ -81,6 +84,30 @@ def _process_job( return {"status": "ignored", "result": result} +def _handle_running_build( + storage: Any, job: str, last_build: dict[str, Any], incidents: dict[str, dict], tick_state: dict[str, Any] +) -> dict[str, Any]: + """Escalate a build that has run past the cap; otherwise leave it alone. + + A build that never terminates produces no incident anywhere while holding + an agent slot. No model is consulted: its console is still being written, + so any root cause would be invented. + """ + + cap = float(settings.hermes_hung_build_minutes) + if not hermes_hung_builds.is_hung(last_build, cap): + return {"status": "skipped"} + incident_id = f"{job}/{_int_value(last_build.get('number'))}" + if incidents.get(incident_id) is not None: + return {"status": "deduped", "incident_id": incident_id} + reason = hermes_hung_builds.HUNG_REASON + base = {"incident_id": incident_id, "job": job, "build_number": _int_value(last_build.get("number"))} + hermes_events.record_incident(storage, base, "human_required", {"reason": reason}) + bundle = hermes_hung_builds.hung_bundle(job, last_build, cap) + _file_incident_issue(storage, base, _diagnosis(bundle, None, reason, None), tick_state) + return {"status": "human_required", "incident_id": incident_id, "reason": reason} + + def _fetch_last_build(job: str) -> dict[str, Any] | None: """Fetch the job's lastBuild summary from Jenkins, or None on failure.""" diff --git a/ariadne/services/hermes_hung_builds.py b/ariadne/services/hermes_hung_builds.py new file mode 100644 index 0000000..e50758b --- /dev/null +++ b/ariadne/services/hermes_hung_builds.py @@ -0,0 +1,85 @@ +"""Escalate Jenkins builds that never reach a terminal result. + +Detection elsewhere in auto-triage only considers builds that finished, so a +build wedged on a network call is invisible: it produces no incident, no +issue, and no alert, while holding one of the finite Jenkins agent slots. +Observed on metis build 272, which sat on `apt-get update` for 75 minutes and +starved the demo of an agent without anything anywhere saying so. + +No model call is made here. "This build has been running far longer than the +cap" is a fact Ariadne can establish on its own, and inventing a root cause +for a build whose console is still being written would be guesswork. +""" + +from __future__ import annotations + +import time +from typing import Any + + +HUNG_CLASSIFICATION = "build_exceeded_time_cap" +HUNG_REASON = "hung_build" +_MS_PER_SECOND = 1000.0 +_SECONDS_PER_MINUTE = 60.0 + + +def elapsed_minutes(last_build: dict[str, Any], now: float | None = None) -> float: + """Return how long a build has been running, in minutes. + + Inputs: a Jenkins lastBuild summary carrying `timestamp` in epoch + milliseconds, and an optional current epoch time in seconds for testing. + Outputs: elapsed minutes, or 0.0 when the timestamp is missing or unusable + so an unreadable build is never mistaken for a hung one. + """ + + try: + started = float(last_build.get("timestamp") or 0.0) / _MS_PER_SECOND + except (TypeError, ValueError): + return 0.0 + if started <= 0.0: + return 0.0 + current = time.time() if now is None else now + return max(0.0, (current - started) / _SECONDS_PER_MINUTE) + + +def is_hung(last_build: dict[str, Any], cap_minutes: float, now: float | None = None) -> bool: + """Report whether a running build has exceeded the configured cap. + + Inputs: the lastBuild summary, the cap in minutes, and an optional clock. + Outputs: True only when the build is still running and has been running + longer than the cap. A cap of zero or less disables the check entirely. + """ + + if cap_minutes <= 0 or not last_build.get("building"): + return False + return elapsed_minutes(last_build, now) > cap_minutes + + +def hung_bundle(job: str, last_build: dict[str, Any], cap_minutes: float) -> dict[str, Any]: + """Build the evidence bundle describing a hung build. + + Inputs: the job name, its lastBuild summary, and the configured cap. + Outputs: a bundle shaped like the console-evidence bundles the issue + renderer already accepts, carrying only facts Ariadne observed directly. + """ + + minutes = elapsed_minutes(last_build) + number = last_build.get("number") + return { + "jenkins": { + "job": job, + "build_number": number, + "url": str(last_build.get("url") or ""), + "building": True, + "elapsed_minutes": round(minutes, 1), + "cap_minutes": cap_minutes, + "console_failures": [], + "console_tail": ( + f"Build {job} #{number} has been running for {minutes:.1f} minutes, " + f"past the {cap_minutes:.0f} minute cap, and has not reported a result. " + "Ariadne did not diagnose a cause: the build is still running and its " + "console is incomplete. It is also still holding a Jenkins agent slot." + ), + }, + "log_evidence": {"records": []}, + } diff --git a/ariadne/settings.py b/ariadne/settings.py index 5117f06..cdd6b3d 100644 --- a/ariadne/settings.py +++ b/ariadne/settings.py @@ -189,6 +189,7 @@ class Settings: hermes_action_classifications: dict[str, str] hermes_parameterized_jobs: list[str] hermes_min_confidence: float + hermes_hung_build_minutes: float hermes_max_actions_per_incident: int hermes_api_url: str hermes_api_key: str diff --git a/ariadne/settings_sections.py b/ariadne/settings_sections.py index 769c0e3..ab30ac7 100644 --- a/ariadne/settings_sections.py +++ b/ariadne/settings_sections.py @@ -289,6 +289,7 @@ def _hermes_autotriage_config() -> dict[str, Any]: if item.strip() ], "hermes_min_confidence": _env_float("ARIADNE_HERMES_MIN_CONFIDENCE", 0.85), + "hermes_hung_build_minutes": _env_float("ARIADNE_HERMES_HUNG_BUILD_MINUTES", 45.0), "hermes_max_actions_per_incident": _env_int("ARIADNE_HERMES_MAX_ACTIONS_PER_INCIDENT", 1), "hermes_api_url": _env( "ARIADNE_HERMES_API_URL", diff --git a/tests/hermes_autotriage_harness.py b/tests/hermes_autotriage_harness.py index bb14834..6f4e667 100644 --- a/tests/hermes_autotriage_harness.py +++ b/tests/hermes_autotriage_harness.py @@ -58,6 +58,7 @@ def _settings(**overrides) -> SimpleNamespace: # type: ignore[no-untyped-def] }, "hermes_parameterized_jobs": [JOB], "hermes_min_confidence": 0.85, + "hermes_hung_build_minutes": 45.0, "hermes_max_actions_per_incident": 1, "hermes_api_url": "http://hermes:8642", "hermes_api_key": "key", diff --git a/tests/test_hermes_autotriage.py b/tests/test_hermes_autotriage.py index 8d2dfed..f4e2e16 100644 --- a/tests/test_hermes_autotriage.py +++ b/tests/test_hermes_autotriage.py @@ -1,5 +1,7 @@ from __future__ import annotations +import time + import pytest from ariadne.services import hermes_autotriage as module @@ -273,12 +275,43 @@ def test_jenkins_fetch_failure_skips_job(monkeypatch) -> None: assert env.storage.events == [] -def test_building_build_is_skipped(monkeypatch) -> None: - env = _prepare(monkeypatch, last_build=_build(12, None, building=True)) +def test_building_build_within_the_cap_is_skipped(monkeypatch) -> None: + """A build that is merely still running is not triage's business.""" + + fresh = int(time.time() * 1000) + env = _prepare(monkeypatch, last_build=_build(12, None, building=True, timestamp=fresh)) assert module.run_hermes_autotriage(env.storage)["jobs"][JOB] == {"status": "skipped"} assert env.storage.events == [] +def test_building_build_past_the_cap_escalates(monkeypatch) -> None: + """A build that never terminates must not stay invisible. + + It files no incident, raises no alert, and holds a Jenkins agent slot the + whole time, so the only signal is that it has run too long. + """ + + env = _prepare(monkeypatch, last_build=_build(12, None, building=True)) + summary = module.run_hermes_autotriage(env.storage)["jobs"][JOB] + assert summary == { + "status": "human_required", + "incident_id": INCIDENT_ID, + "reason": module.hermes_hung_builds.HUNG_REASON, + } + assert env.storage.events != [] + + +def test_hung_build_is_not_escalated_twice(monkeypatch) -> None: + """Every tick sees the same running build; only the first may escalate.""" + + env = _prepare(monkeypatch, last_build=_build(12, None, building=True)) + module.run_hermes_autotriage(env.storage) + before = len(env.storage.events) + second = module.run_hermes_autotriage(env.storage)["jobs"][JOB] + assert second == {"status": "deduped", "incident_id": INCIDENT_ID} + assert len(env.storage.events) == before + + def test_empty_jenkins_base_url_skips(monkeypatch) -> None: env = _prepare(monkeypatch, cfg=_settings(jenkins_base_url="")) assert module.run_hermes_autotriage(env.storage)["jobs"][JOB] == {"status": "skipped"}