#!/usr/bin/env bash # Shared plumbing for the two Hermes demo drivers. # # The triage demo and the code demo are separate scripts on purpose: they prove # different halves of the Test Automation Diagram, they reset different things, # and mixing them behind one command invited exactly the confusion of running # the wrong subcommand in front of an audience. What they genuinely share - # credentials, Jenkins access, the Ariadne tick reader - lives here, so a fix # to any of it applies to both instead of being made twice and drifting. # # Not executable on its own; both drivers source it. # Local, git-ignored credentials. `hermes_demo.env` is the current name; # `hermes_triage_demo.env` is still read so an existing filled-in file keeps # working after the split. _DEMO_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" DEMO_ENV="" for _candidate in "$_DEMO_DIR/hermes_demo.env" "$_DEMO_DIR/hermes_triage_demo.env"; do if [ -r "$_candidate" ]; then DEMO_ENV="$_candidate" # shellcheck disable=SC1090 . "$_candidate" break fi done [ -n "$DEMO_ENV" ] || DEMO_ENV="$_DEMO_DIR/hermes_demo.env" JENKINS_URL="${JENKINS_URL:-https://ci.bstein.dev}" GITEA_URL="${GITEA_URL:-https://scm.bstein.dev}" FIXTURE_JOB="hermes-triage-demo" CODE_JOB="hermes-code-demo" DEMO_NS="hermes-triage-demo" CODE_REPO_DIR="${CODE_REPO_DIR:-$HOME/Development/hermes-code-demo}" say() { printf '\n\033[1m[%s] %s\033[0m\n' "$(date -u +%H:%M:%S)" "$*"; } note() { printf ' %s\n' "$*"; } require_jenkins() { if [ -z "${JENKINS_USER:-}" ] || [ -z "${JENKINS_TOKEN:-}" ]; then echo "Missing Jenkins credentials." >&2 echo "Create $DEMO_ENV from hermes_demo.env.example and fill it in." >&2 exit 1 fi } # Every remote call is time-bounded. A hung curl during a demo is worse than a # failed one: a failure says what to do next, a hang says nothing at all. jenkins_get() { curl -sk --max-time 25 -u "$JENKINS_USER:$JENKINS_TOKEN" "$JENKINS_URL$1"; } jenkins_post() { curl -sk --max-time 25 -o /dev/null -w '%{http_code}' \ -u "$JENKINS_USER:$JENKINS_TOKEN" -X POST "$JENKINS_URL$1" } gitea_get() { curl -s --max-time 25 -H "Authorization: token ${GITEA_TOKEN:-}" "$GITEA_URL$1"; } # Jenkins tree selectors use square brackets, which some curl builds treat as # glob metacharacters and refuse to send - the request never leaves, the body # is empty, and the JSON parse dies with a traceback that says nothing about # the real cause. Encoded, so the demo does not depend on how curl was built. last_build_number() { local body body="$(jenkins_get "/job/$1/api/json?tree=lastBuild%5Bnumber%5D")" if [ -z "$body" ]; then echo "Jenkins returned nothing for job $1 (check credentials and $JENKINS_URL)" >&2 return 1 fi printf '%s' "$body" | python3 -c 'import json,sys; print(json.load(sys.stdin)["lastBuild"]["number"])' } # Polls before sleeping, not after. Sleeping first meant a build that had # already finished still cost a full interval of silence, which on stage reads # as the script having missed it. The interval is short for the same reason: # the wait is dead air in front of an audience, and a Jenkins status read is # cheap. BUILD_POLL_SECONDS="${BUILD_POLL_SECONDS:-3}" wait_for_build() { # job number [max_seconds] -> prints result local job="$1" num="$2" budget="${3:-1500}" local waited=0 body building result while [ "$waited" -le "$budget" ]; do body="$(jenkins_get "/job/$job/$num/api/json?tree=result,building" || true)" building="$(printf '%s' "$body" | python3 -c 'import json,sys; print(json.load(sys.stdin).get("building"))' 2>/dev/null || echo unknown)" if [ "$building" = "False" ]; then result="$(printf '%s' "$body" | python3 -c 'import json,sys; print(json.load(sys.stdin).get("result"))')" printf '%s' "$result" return 0 fi sleep "$BUILD_POLL_SECONDS" waited=$((waited + BUILD_POLL_SECONDS)) done printf 'TIMEOUT' } # Ariadne's triage tick is on cron, which cannot fire more often than once a # minute. That minute is the largest gap between a build going red and the # system visibly reacting, and it is pure dead air on stage. This runs the same # tick immediately over the pod's own loopback, so nothing is exposed outside # the cluster. The tick is idempotent - incidents dedupe on job and build # number - so provoking it can only ever be a no-op, never a second incident. poke_ariadne() { kubectl -n maintenance exec deploy/ariadne -c ariadne -- python3 -c " import json, urllib.request req = urllib.request.Request( 'http://127.0.0.1:8080/api/internal/hermes/autotriage/run', method='POST') body = json.load(urllib.request.urlopen(req, timeout=120)) jobs = body.get('jobs') or {} print(body.get('status', 'ok'), '|', ', '.join( f\"{name}={info.get('status')}\" for name, info in jobs.items()) or 'no jobs') " 2>/dev/null || echo "tick request failed; the scheduler will pick it up within a minute" } ariadne_ticks() { # tail the autotriage decisions in human-readable form kubectl -n maintenance logs deploy/ariadne -c ariadne --tail="${1:-400}" 2>/dev/null | grep 'hermes autotriage tick' | python3 -c ' import sys, json for line in sys.stdin: try: d = json.loads(line) except ValueError: continue print(" ", d["timestamp"][11:19], d.get("jobs"))' | tail -"${2:-5}" } # Both drivers narrate the same Test Automation Diagram; the monitor selects # which branch of it to follow from MONITOR_JOB. run_monitor() { # job export MONITOR_JOB="$1" exec python3 "$_DEMO_DIR/hermes_triage_monitor.py" } # The checks that are true of the lab regardless of which demo is running. shared_preflight() { note "ariadne image: $(kubectl -n maintenance get deploy ariadne -o jsonpath='{.spec.template.spec.containers[0].image}')" note "autoremediation: $(kubectl -n maintenance exec deploy/ariadne -c ariadne -- printenv ARIADNE_HERMES_AUTOREMEDIATION_ENABLED 2>/dev/null)" # Printed as a list rather than the raw comma-separated setting: this is the # outermost safety boundary, so it is worth being able to read at a glance. local allowlist count allowlist="$(kubectl -n maintenance exec deploy/ariadne -c ariadne -- printenv ARIADNE_HERMES_AUTOTRIAGE_JOB_ALLOWLIST 2>/dev/null | tr ',' ' ')" count=0 for _job in $allowlist; do count=$((count + 1)); done note "jobs Ariadne may triage ($count):" for _job in $allowlist; do note " - $_job"; done note "hermes: $(kubectl -n hermes get pods -l app=hermes --no-headers | awk '{print $2, $3}')" local queued queued="$(jenkins_get '/queue/api/json' | python3 -c 'import json,sys; print(len(json.load(sys.stdin)["items"]))')" note "jenkins queue depth: $queued (demo is fastest when this is 0)" # The Kubernetes cloud caps concurrent agent pods at containerCapStr. When # real CI saturates that cap the demo build sits in the queue reporting # "all nodes are offline" and the timings in the runbook do not apply. local agents cap cap="$(kubectl -n jenkins get cm jenkins-jcasc -o jsonpath='{.data.jenkins\.yaml}' 2>/dev/null | grep -o 'containerCapStr: "[0-9]*"' | head -1 | grep -o '[0-9]*' || echo 5)" agents="$(kubectl -n jenkins get pods --no-headers 2>/dev/null | grep -cE '\-[a-z0-9]{5}-[a-z0-9]{5}-[a-z0-9]{5}' || true)" note "jenkins agent pods: ${agents:-0}/${cap:-5} (a full pool stalls the demo — wait for a free slot)" } # The alert and incident state both demos are judged by. shared_status() { say "Incident state (last ticks)" ariadne_ticks 600 8 say "Firing alerts" kubectl -n monitoring exec deploy/vmalert-atlas-availability -- wget -qO- localhost:8880/api/v1/alerts 2>/dev/null | python3 -c ' import json,sys alerts = json.load(sys.stdin).get("data", {}).get("alerts", []) print(" none" if not alerts else "") for a in alerts: print(" ", a["name"], a["state"], "build", a.get("labels", {}).get("build"))' 2>/dev/null || note "(query vmalert directly if this fails)" }