Some checks failed
Tests / Declarative: Post Actions failed: 2, passed: 142
Both demos went quiet for up to a minute between the build turning red and the monitor reacting, because Ariadne's tick is on cron. The scripts now run that tick immediately over the pod's own loopback - nothing exposed outside the cluster - and print what it saw, so the pause becomes a visible step rather than dead air. Falls back to silence rather than failure: if the request does not land the scheduler still picks the build up within the minute, which is exactly the old behaviour. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
173 lines
7.8 KiB
Bash
173 lines
7.8 KiB
Bash
#!/usr/bin/env bash
|
|
# Shared plumbing for the two Hermes demo drivers.
|
|
#
|
|
# The triage demo and the code demo are separate scripts on purpose: they prove
|
|
# different halves of the Test Automation Diagram, they reset different things,
|
|
# and mixing them behind one command invited exactly the confusion of running
|
|
# the wrong subcommand in front of an audience. What they genuinely share -
|
|
# credentials, Jenkins access, the Ariadne tick reader - lives here, so a fix
|
|
# to any of it applies to both instead of being made twice and drifting.
|
|
#
|
|
# Not executable on its own; both drivers source it.
|
|
|
|
# Local, git-ignored credentials. `hermes_demo.env` is the current name;
|
|
# `hermes_triage_demo.env` is still read so an existing filled-in file keeps
|
|
# working after the split.
|
|
_DEMO_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
DEMO_ENV=""
|
|
for _candidate in "$_DEMO_DIR/hermes_demo.env" "$_DEMO_DIR/hermes_triage_demo.env"; do
|
|
if [ -r "$_candidate" ]; then
|
|
DEMO_ENV="$_candidate"
|
|
# shellcheck disable=SC1090
|
|
. "$_candidate"
|
|
break
|
|
fi
|
|
done
|
|
[ -n "$DEMO_ENV" ] || DEMO_ENV="$_DEMO_DIR/hermes_demo.env"
|
|
|
|
JENKINS_URL="${JENKINS_URL:-https://ci.bstein.dev}"
|
|
GITEA_URL="${GITEA_URL:-https://scm.bstein.dev}"
|
|
FIXTURE_JOB="hermes-triage-demo"
|
|
CODE_JOB="hermes-code-demo"
|
|
DEMO_NS="hermes-triage-demo"
|
|
CODE_REPO_DIR="${CODE_REPO_DIR:-$HOME/Development/hermes-code-demo}"
|
|
|
|
say() { printf '\n\033[1m[%s] %s\033[0m\n' "$(date -u +%H:%M:%S)" "$*"; }
|
|
note() { printf ' %s\n' "$*"; }
|
|
|
|
require_jenkins() {
|
|
if [ -z "${JENKINS_USER:-}" ] || [ -z "${JENKINS_TOKEN:-}" ]; then
|
|
echo "Missing Jenkins credentials." >&2
|
|
echo "Create $DEMO_ENV from hermes_demo.env.example and fill it in." >&2
|
|
exit 1
|
|
fi
|
|
}
|
|
|
|
# Every remote call is time-bounded. A hung curl during a demo is worse than a
|
|
# failed one: a failure says what to do next, a hang says nothing at all.
|
|
jenkins_get() { curl -sk --max-time 25 -u "$JENKINS_USER:$JENKINS_TOKEN" "$JENKINS_URL$1"; }
|
|
jenkins_post() {
|
|
curl -sk --max-time 25 -o /dev/null -w '%{http_code}' \
|
|
-u "$JENKINS_USER:$JENKINS_TOKEN" -X POST "$JENKINS_URL$1"
|
|
}
|
|
gitea_get() { curl -s --max-time 25 -H "Authorization: token ${GITEA_TOKEN:-}" "$GITEA_URL$1"; }
|
|
|
|
# Jenkins tree selectors use square brackets, which some curl builds treat as
|
|
# glob metacharacters and refuse to send - the request never leaves, the body
|
|
# is empty, and the JSON parse dies with a traceback that says nothing about
|
|
# the real cause. Encoded, so the demo does not depend on how curl was built.
|
|
last_build_number() {
|
|
local body
|
|
body="$(jenkins_get "/job/$1/api/json?tree=lastBuild%5Bnumber%5D")"
|
|
if [ -z "$body" ]; then
|
|
echo "Jenkins returned nothing for job $1 (check credentials and $JENKINS_URL)" >&2
|
|
return 1
|
|
fi
|
|
printf '%s' "$body" |
|
|
python3 -c 'import json,sys; print(json.load(sys.stdin)["lastBuild"]["number"])'
|
|
}
|
|
|
|
# Polls before sleeping, not after. Sleeping first meant a build that had
|
|
# already finished still cost a full interval of silence, which on stage reads
|
|
# as the script having missed it. The interval is short for the same reason:
|
|
# the wait is dead air in front of an audience, and a Jenkins status read is
|
|
# cheap.
|
|
BUILD_POLL_SECONDS="${BUILD_POLL_SECONDS:-3}"
|
|
|
|
wait_for_build() { # job number [max_seconds] -> prints result
|
|
local job="$1" num="$2" budget="${3:-1500}"
|
|
local waited=0 body building result
|
|
while [ "$waited" -le "$budget" ]; do
|
|
body="$(jenkins_get "/job/$job/$num/api/json?tree=result,building" || true)"
|
|
building="$(printf '%s' "$body" |
|
|
python3 -c 'import json,sys; print(json.load(sys.stdin).get("building"))' 2>/dev/null || echo unknown)"
|
|
if [ "$building" = "False" ]; then
|
|
result="$(printf '%s' "$body" | python3 -c 'import json,sys; print(json.load(sys.stdin).get("result"))')"
|
|
printf '%s' "$result"
|
|
return 0
|
|
fi
|
|
sleep "$BUILD_POLL_SECONDS"
|
|
waited=$((waited + BUILD_POLL_SECONDS))
|
|
done
|
|
printf 'TIMEOUT'
|
|
}
|
|
|
|
# Ariadne's triage tick is on cron, which cannot fire more often than once a
|
|
# minute. That minute is the largest gap between a build going red and the
|
|
# system visibly reacting, and it is pure dead air on stage. This runs the same
|
|
# tick immediately over the pod's own loopback, so nothing is exposed outside
|
|
# the cluster. The tick is idempotent - incidents dedupe on job and build
|
|
# number - so provoking it can only ever be a no-op, never a second incident.
|
|
poke_ariadne() {
|
|
kubectl -n maintenance exec deploy/ariadne -c ariadne -- python3 -c "
|
|
import json, urllib.request
|
|
req = urllib.request.Request(
|
|
'http://127.0.0.1:8080/api/internal/hermes/autotriage/run', method='POST')
|
|
body = json.load(urllib.request.urlopen(req, timeout=120))
|
|
jobs = body.get('jobs') or {}
|
|
print(body.get('status', 'ok'), '|', ', '.join(
|
|
f\"{name}={info.get('status')}\" for name, info in jobs.items()) or 'no jobs')
|
|
" 2>/dev/null || echo "tick request failed; the scheduler will pick it up within a minute"
|
|
}
|
|
|
|
ariadne_ticks() { # tail the autotriage decisions in human-readable form
|
|
kubectl -n maintenance logs deploy/ariadne -c ariadne --tail="${1:-400}" 2>/dev/null |
|
|
grep 'hermes autotriage tick' |
|
|
python3 -c '
|
|
import sys, json
|
|
for line in sys.stdin:
|
|
try:
|
|
d = json.loads(line)
|
|
except ValueError:
|
|
continue
|
|
print(" ", d["timestamp"][11:19], d.get("jobs"))' | tail -"${2:-5}"
|
|
}
|
|
|
|
# Both drivers narrate the same Test Automation Diagram; the monitor selects
|
|
# which branch of it to follow from MONITOR_JOB.
|
|
run_monitor() { # job
|
|
export MONITOR_JOB="$1"
|
|
exec python3 "$_DEMO_DIR/hermes_triage_monitor.py"
|
|
}
|
|
|
|
# The checks that are true of the lab regardless of which demo is running.
|
|
shared_preflight() {
|
|
note "ariadne image: $(kubectl -n maintenance get deploy ariadne -o jsonpath='{.spec.template.spec.containers[0].image}')"
|
|
note "autoremediation: $(kubectl -n maintenance exec deploy/ariadne -c ariadne -- printenv ARIADNE_HERMES_AUTOREMEDIATION_ENABLED 2>/dev/null)"
|
|
# Printed as a list rather than the raw comma-separated setting: this is the
|
|
# outermost safety boundary, so it is worth being able to read at a glance.
|
|
local allowlist count
|
|
allowlist="$(kubectl -n maintenance exec deploy/ariadne -c ariadne -- printenv ARIADNE_HERMES_AUTOTRIAGE_JOB_ALLOWLIST 2>/dev/null | tr ',' ' ')"
|
|
count=0
|
|
for _job in $allowlist; do count=$((count + 1)); done
|
|
note "jobs Ariadne may triage ($count):"
|
|
for _job in $allowlist; do note " - $_job"; done
|
|
note "hermes: $(kubectl -n hermes get pods -l app=hermes --no-headers | awk '{print $2, $3}')"
|
|
local queued
|
|
queued="$(jenkins_get '/queue/api/json' | python3 -c 'import json,sys; print(len(json.load(sys.stdin)["items"]))')"
|
|
note "jenkins queue depth: $queued (demo is fastest when this is 0)"
|
|
# The Kubernetes cloud caps concurrent agent pods at containerCapStr. When
|
|
# real CI saturates that cap the demo build sits in the queue reporting
|
|
# "all nodes are offline" and the timings in the runbook do not apply.
|
|
local agents cap
|
|
cap="$(kubectl -n jenkins get cm jenkins-jcasc -o jsonpath='{.data.jenkins\.yaml}' 2>/dev/null |
|
|
grep -o 'containerCapStr: "[0-9]*"' | head -1 | grep -o '[0-9]*' || echo 5)"
|
|
agents="$(kubectl -n jenkins get pods --no-headers 2>/dev/null | grep -cE '\-[a-z0-9]{5}-[a-z0-9]{5}-[a-z0-9]{5}' || true)"
|
|
note "jenkins agent pods: ${agents:-0}/${cap:-5} (a full pool stalls the demo — wait for a free slot)"
|
|
}
|
|
|
|
# The alert and incident state both demos are judged by.
|
|
shared_status() {
|
|
say "Incident state (last ticks)"
|
|
ariadne_ticks 600 8
|
|
say "Firing alerts"
|
|
kubectl -n monitoring exec deploy/vmalert-atlas-availability -- wget -qO- localhost:8880/api/v1/alerts 2>/dev/null |
|
|
python3 -c '
|
|
import json,sys
|
|
alerts = json.load(sys.stdin).get("data", {}).get("alerts", [])
|
|
print(" none" if not alerts else "")
|
|
for a in alerts:
|
|
print(" ", a["name"], a["state"], "build", a.get("labels", {}).get("build"))' 2>/dev/null ||
|
|
note "(query vmalert directly if this fails)"
|
|
}
|