titan-iac/scripts/ops/hermes_demo_lib.sh
jenkins bd6c0678d8
Some checks failed
Tests / Declarative: Post Actions failed: 2, passed: 142
feat(demo): provoke the triage tick instead of waiting a minute for it
Both demos went quiet for up to a minute between the build turning red and
the monitor reacting, because Ariadne's tick is on cron. The scripts now run
that tick immediately over the pod's own loopback - nothing exposed outside
the cluster - and print what it saw, so the pause becomes a visible step
rather than dead air.

Falls back to silence rather than failure: if the request does not land the
scheduler still picks the build up within the minute, which is exactly the
old behaviour.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-07 01:44:03 -03:00

173 lines
7.8 KiB
Bash

#!/usr/bin/env bash
# Shared plumbing for the two Hermes demo drivers.
#
# The triage demo and the code demo are separate scripts on purpose: they prove
# different halves of the Test Automation Diagram, they reset different things,
# and mixing them behind one command invited exactly the confusion of running
# the wrong subcommand in front of an audience. What they genuinely share -
# credentials, Jenkins access, the Ariadne tick reader - lives here, so a fix
# to any of it applies to both instead of being made twice and drifting.
#
# Not executable on its own; both drivers source it.
# Local, git-ignored credentials. `hermes_demo.env` is the current name;
# `hermes_triage_demo.env` is still read so an existing filled-in file keeps
# working after the split.
_DEMO_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
DEMO_ENV=""
for _candidate in "$_DEMO_DIR/hermes_demo.env" "$_DEMO_DIR/hermes_triage_demo.env"; do
if [ -r "$_candidate" ]; then
DEMO_ENV="$_candidate"
# shellcheck disable=SC1090
. "$_candidate"
break
fi
done
[ -n "$DEMO_ENV" ] || DEMO_ENV="$_DEMO_DIR/hermes_demo.env"
JENKINS_URL="${JENKINS_URL:-https://ci.bstein.dev}"
GITEA_URL="${GITEA_URL:-https://scm.bstein.dev}"
FIXTURE_JOB="hermes-triage-demo"
CODE_JOB="hermes-code-demo"
DEMO_NS="hermes-triage-demo"
CODE_REPO_DIR="${CODE_REPO_DIR:-$HOME/Development/hermes-code-demo}"
say() { printf '\n\033[1m[%s] %s\033[0m\n' "$(date -u +%H:%M:%S)" "$*"; }
note() { printf ' %s\n' "$*"; }
require_jenkins() {
if [ -z "${JENKINS_USER:-}" ] || [ -z "${JENKINS_TOKEN:-}" ]; then
echo "Missing Jenkins credentials." >&2
echo "Create $DEMO_ENV from hermes_demo.env.example and fill it in." >&2
exit 1
fi
}
# Every remote call is time-bounded. A hung curl during a demo is worse than a
# failed one: a failure says what to do next, a hang says nothing at all.
jenkins_get() { curl -sk --max-time 25 -u "$JENKINS_USER:$JENKINS_TOKEN" "$JENKINS_URL$1"; }
jenkins_post() {
curl -sk --max-time 25 -o /dev/null -w '%{http_code}' \
-u "$JENKINS_USER:$JENKINS_TOKEN" -X POST "$JENKINS_URL$1"
}
gitea_get() { curl -s --max-time 25 -H "Authorization: token ${GITEA_TOKEN:-}" "$GITEA_URL$1"; }
# Jenkins tree selectors use square brackets, which some curl builds treat as
# glob metacharacters and refuse to send - the request never leaves, the body
# is empty, and the JSON parse dies with a traceback that says nothing about
# the real cause. Encoded, so the demo does not depend on how curl was built.
last_build_number() {
local body
body="$(jenkins_get "/job/$1/api/json?tree=lastBuild%5Bnumber%5D")"
if [ -z "$body" ]; then
echo "Jenkins returned nothing for job $1 (check credentials and $JENKINS_URL)" >&2
return 1
fi
printf '%s' "$body" |
python3 -c 'import json,sys; print(json.load(sys.stdin)["lastBuild"]["number"])'
}
# Polls before sleeping, not after. Sleeping first meant a build that had
# already finished still cost a full interval of silence, which on stage reads
# as the script having missed it. The interval is short for the same reason:
# the wait is dead air in front of an audience, and a Jenkins status read is
# cheap.
BUILD_POLL_SECONDS="${BUILD_POLL_SECONDS:-3}"
wait_for_build() { # job number [max_seconds] -> prints result
local job="$1" num="$2" budget="${3:-1500}"
local waited=0 body building result
while [ "$waited" -le "$budget" ]; do
body="$(jenkins_get "/job/$job/$num/api/json?tree=result,building" || true)"
building="$(printf '%s' "$body" |
python3 -c 'import json,sys; print(json.load(sys.stdin).get("building"))' 2>/dev/null || echo unknown)"
if [ "$building" = "False" ]; then
result="$(printf '%s' "$body" | python3 -c 'import json,sys; print(json.load(sys.stdin).get("result"))')"
printf '%s' "$result"
return 0
fi
sleep "$BUILD_POLL_SECONDS"
waited=$((waited + BUILD_POLL_SECONDS))
done
printf 'TIMEOUT'
}
# Ariadne's triage tick is on cron, which cannot fire more often than once a
# minute. That minute is the largest gap between a build going red and the
# system visibly reacting, and it is pure dead air on stage. This runs the same
# tick immediately over the pod's own loopback, so nothing is exposed outside
# the cluster. The tick is idempotent - incidents dedupe on job and build
# number - so provoking it can only ever be a no-op, never a second incident.
poke_ariadne() {
kubectl -n maintenance exec deploy/ariadne -c ariadne -- python3 -c "
import json, urllib.request
req = urllib.request.Request(
'http://127.0.0.1:8080/api/internal/hermes/autotriage/run', method='POST')
body = json.load(urllib.request.urlopen(req, timeout=120))
jobs = body.get('jobs') or {}
print(body.get('status', 'ok'), '|', ', '.join(
f\"{name}={info.get('status')}\" for name, info in jobs.items()) or 'no jobs')
" 2>/dev/null || echo "tick request failed; the scheduler will pick it up within a minute"
}
ariadne_ticks() { # tail the autotriage decisions in human-readable form
kubectl -n maintenance logs deploy/ariadne -c ariadne --tail="${1:-400}" 2>/dev/null |
grep 'hermes autotriage tick' |
python3 -c '
import sys, json
for line in sys.stdin:
try:
d = json.loads(line)
except ValueError:
continue
print(" ", d["timestamp"][11:19], d.get("jobs"))' | tail -"${2:-5}"
}
# Both drivers narrate the same Test Automation Diagram; the monitor selects
# which branch of it to follow from MONITOR_JOB.
run_monitor() { # job
export MONITOR_JOB="$1"
exec python3 "$_DEMO_DIR/hermes_triage_monitor.py"
}
# The checks that are true of the lab regardless of which demo is running.
shared_preflight() {
note "ariadne image: $(kubectl -n maintenance get deploy ariadne -o jsonpath='{.spec.template.spec.containers[0].image}')"
note "autoremediation: $(kubectl -n maintenance exec deploy/ariadne -c ariadne -- printenv ARIADNE_HERMES_AUTOREMEDIATION_ENABLED 2>/dev/null)"
# Printed as a list rather than the raw comma-separated setting: this is the
# outermost safety boundary, so it is worth being able to read at a glance.
local allowlist count
allowlist="$(kubectl -n maintenance exec deploy/ariadne -c ariadne -- printenv ARIADNE_HERMES_AUTOTRIAGE_JOB_ALLOWLIST 2>/dev/null | tr ',' ' ')"
count=0
for _job in $allowlist; do count=$((count + 1)); done
note "jobs Ariadne may triage ($count):"
for _job in $allowlist; do note " - $_job"; done
note "hermes: $(kubectl -n hermes get pods -l app=hermes --no-headers | awk '{print $2, $3}')"
local queued
queued="$(jenkins_get '/queue/api/json' | python3 -c 'import json,sys; print(len(json.load(sys.stdin)["items"]))')"
note "jenkins queue depth: $queued (demo is fastest when this is 0)"
# The Kubernetes cloud caps concurrent agent pods at containerCapStr. When
# real CI saturates that cap the demo build sits in the queue reporting
# "all nodes are offline" and the timings in the runbook do not apply.
local agents cap
cap="$(kubectl -n jenkins get cm jenkins-jcasc -o jsonpath='{.data.jenkins\.yaml}' 2>/dev/null |
grep -o 'containerCapStr: "[0-9]*"' | head -1 | grep -o '[0-9]*' || echo 5)"
agents="$(kubectl -n jenkins get pods --no-headers 2>/dev/null | grep -cE '\-[a-z0-9]{5}-[a-z0-9]{5}-[a-z0-9]{5}' || true)"
note "jenkins agent pods: ${agents:-0}/${cap:-5} (a full pool stalls the demo — wait for a free slot)"
}
# The alert and incident state both demos are judged by.
shared_status() {
say "Incident state (last ticks)"
ariadne_ticks 600 8
say "Firing alerts"
kubectl -n monitoring exec deploy/vmalert-atlas-availability -- wget -qO- localhost:8880/api/v1/alerts 2>/dev/null |
python3 -c '
import json,sys
alerts = json.load(sys.stdin).get("data", {}).get("alerts", [])
print(" none" if not alerts else "")
for a in alerts:
print(" ", a["name"], a["state"], "build", a.get("labels", {}).get("build"))' 2>/dev/null ||
note "(query vmalert directly if this fails)"
}