From 2a08a8cd8e83045bea883ddea944c638c38231e4 Mon Sep 17 00:00:00 2001 From: Niklas Ye Date: Fri, 2 Oct 2026 21:23:44 +0200 Subject: [PATCH] examples/demo: add run-demo.sh, an automated kind-cluster demo One script, two modes (run-demo.sh / run-demo.sh --teardown), that takes a fresh empty kind cluster all the way to a working demo: creates the cluster if needed, helm-installs this chart, applies every CRD kind in this directory, waits for all nine objects to go Ready, then does what the README's own first-login section cannot (see niklas/terdut-server#23 and niklas/terdut-operator#3 -- no service-account credential this operator holds can ever call /api/admin/settings or POST /api/users) by reaching into the demo's own throwaway Postgres directly: flips signup_mode to open, signs alice up for real over the ordinary signup endpoint, and joins her to both Platform and Payments (open signup always creates its own new team, never joins an existing one by name, so without this she'd have a working login that can't see a single incident this demo fires -- /api/incidents and /api/alerts are both scoped to the caller's own team memberships). Finishes by port-forwarding the service and firing fire-alerts.sh at both teams, so a fresh run already has visible incidents waiting in the web UI. Verified end to end against a real kind cluster, including a second, genuinely-fresh run that hit niklas/terdut-operator#3 live (terdutteam- platform wedged in the 403 retry loop that issue describes) -- confirmed the script itself fails cleanly on that (clear FAILED message, correct exit code, no orphaned port-forward) rather than hanging or leaving a mess, which is the most this script can do about a bug in the operator it's driving. --- examples/demo/run-demo.sh | 324 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 324 insertions(+) create mode 100755 examples/demo/run-demo.sh diff --git a/examples/demo/run-demo.sh b/examples/demo/run-demo.sh new file mode 100755 index 0000000..f19795f --- /dev/null +++ b/examples/demo/run-demo.sh @@ -0,0 +1,324 @@ +#!/usr/bin/env bash +# Stands up the complete terdut demo (terdut-operator + every CRD kind this +# repo ships + a working local login + a few synthetic incidents) on a kind +# cluster, fully automated. Password login only -- this demo kit's own +# 01-server.yaml carries no oidc: block at all, so there is nothing to +# disable; OIDC is simply absent. +# +# Usage: +# ./run-demo.sh deploy the whole demo (idempotent: safe to +# re-run against a cluster that already has it) +# ./run-demo.sh --teardown delete the demo namespace (and, if +# TEARDOWN_CLUSTER=true, the kind cluster too) +# ./run-demo.sh --help +# +# All of the defaults below are overridable as environment variables. +set -euo pipefail + +CLUSTER_NAME="${CLUSTER_NAME:-terdut-demo}" +NAMESPACE="${NAMESPACE:-terdut-operator-demo}" +OPERATOR_NAMESPACE="${OPERATOR_NAMESPACE:-terdut-operator-system}" +RELEASE_NAME="${RELEASE_NAME:-terdut-operator}" + +ALICE_USERNAME="${ALICE_USERNAME:-alice}" +ALICE_EMAIL="${ALICE_EMAIL:-alice@terdut-demo.local}" +ALICE_TEAM_NAME="${ALICE_TEAM_NAME:-alice-demo}" +DEMO_PASSWORD="${DEMO_PASSWORD:-terdut-demo-1234}" + +BASE_URL="${BASE_URL:-http://localhost:8080}" +LOCAL_PORT="${LOCAL_PORT:-8080}" + +HELM_TIMEOUT="${HELM_TIMEOUT:-180s}" +WAIT_TIMEOUT="${WAIT_TIMEOUT:-180s}" + +TEARDOWN_CLUSTER="${TEARDOWN_CLUSTER:-false}" + +PF_PIDFILE="${PF_PIDFILE:-/tmp/terdut-demo-port-forward.pid}" +PF_LOGFILE="${PF_LOGFILE:-/tmp/terdut-demo-port-forward.log}" + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +CHART_DIR="$(cd "$SCRIPT_DIR/../.." && pwd)/charts/terdut-operator" + +# --------------------------------------------------------------------------- +# helpers +# --------------------------------------------------------------------------- + +log() { printf '[run-demo] %s\n' "$*" >&2; } +die() { printf '[run-demo] FAILED: %s\n' "$*" >&2; exit 1; } + +usage() { + cat </dev/null || true + fi + exit "$rc" +} +trap cleanup_on_failure EXIT + +# --------------------------------------------------------------------------- +# steps +# --------------------------------------------------------------------------- + +preflight() { + local missing=() + for bin in kubectl kind helm jq curl; do + command -v "$bin" >/dev/null 2>&1 || missing+=("$bin") + done + if [ "${#missing[@]}" -gt 0 ]; then + die "missing required tools on PATH: ${missing[*]}" + fi +} + +ensure_kind_cluster() { + case "$(kind get clusters 2>/dev/null)" in + *"$CLUSTER_NAME"*) + log "kind cluster '$CLUSTER_NAME' already exists, skipping creation" ;; + *) + log "creating kind cluster '$CLUSTER_NAME'" + kind create cluster --name "$CLUSTER_NAME" ;; + esac + kubectl config use-context "kind-${CLUSTER_NAME}" >/dev/null +} + +install_operator() { + log "installing terdut-operator into namespace $OPERATOR_NAMESPACE" + helm upgrade --install "$RELEASE_NAME" "$CHART_DIR" \ + --namespace "$OPERATOR_NAMESPACE" --create-namespace \ + --wait --timeout "$HELM_TIMEOUT" \ + || die "helm install of terdut-operator did not become ready" +} + +apply_demo() { + log "creating namespace $NAMESPACE" + kubectl create namespace "$NAMESPACE" --dry-run=client -o yaml | kubectl apply -f - >/dev/null + log "applying demo CRs (00-09) into $NAMESPACE" + kubectl apply -n "$NAMESPACE" -k "$SCRIPT_DIR" >/dev/null +} + +wait_for_ready() { + local objects=( + "terdutserver/terdut-operator-demo" + "terdutteam/terdutteam-platform" + "terdutteam/terdutteam-payments" + "terdutescalationrule/terdutescalationrule-platform" + "terdutescalationrule/terdutescalationrule-payments" + "terdutdeadmanswitch/terdutdeadmanswitch-platform" + "terdutdeadmanswitch/terdutdeadmanswitch-payments" + "terdutalertsource/terdutalertsource-platform" + "terdutalertsource/terdutalertsource-payments" + ) + local obj + for obj in "${objects[@]}"; do + log "waiting for $obj to become Ready" + kubectl wait --for=condition=Ready --timeout "$WAIT_TIMEOUT" -n "$NAMESPACE" "$obj" >/dev/null \ + || die "timed out waiting for $obj -- try: kubectl describe -n $NAMESPACE $obj" + done +} + +start_port_forward() { + # A stale pidfile from an earlier run would otherwise collide with us on + # $LOCAL_PORT -- if that pid is still alive, stop it first. + if [ -f "$PF_PIDFILE" ]; then + local old_pid + old_pid="$(cat "$PF_PIDFILE" 2>/dev/null || true)" + if [ -n "$old_pid" ] && kill -0 "$old_pid" 2>/dev/null; then + log "stopping stale port-forward from a previous run (pid $old_pid)" + kill "$old_pid" 2>/dev/null || true + sleep 1 + fi + rm -f "$PF_PIDFILE" + fi + + log "starting port-forward svc/terdut-operator-demo ${LOCAL_PORT}:8080" + kubectl -n "$NAMESPACE" port-forward svc/terdut-operator-demo "${LOCAL_PORT}:8080" \ + >"$PF_LOGFILE" 2>&1 & + STARTED_PF_PID=$! + echo "$STARTED_PF_PID" > "$PF_PIDFILE" + + local tries=0 + until curl -sf -o /dev/null "http://localhost:${LOCAL_PORT}/healthz"; do + tries=$((tries + 1)) + if [ "$tries" -ge 30 ]; then + die "port-forward never became ready -- see $PF_LOGFILE" + fi + sleep 1 + done + log "port-forward ready (pid $STARTED_PF_PID, log $PF_LOGFILE)" +} + +# Flips signup_mode to 'open' directly in Postgres, then uses the ordinary +# unauthenticated signup endpoint to create a real local account with a +# server-computed bcrypt hash. See this repo's run-demo.sh header / the +# plan this was built from for why this bypasses the demo README's own +# (currently broken) admin-token approach: the operator's service-account +# credential cannot call /api/admin/settings or POST /api/users -- AdminOnly +# only recognizes a human session/API-key caller, confirmed against +# terdut-server's internal/api/middleware.go. +bootstrap_login() { + log "flipping signup_mode to open directly in Postgres" + local pg_pod + pg_pod="$(kubectl -n "$NAMESPACE" get pod -l app=terdut-operator-demo-postgres \ + -o jsonpath='{.items[0].metadata.name}')" + [ -n "$pg_pod" ] || die "could not find the demo Postgres pod in $NAMESPACE" + + kubectl -n "$NAMESPACE" exec "$pg_pod" -- psql -U terdut -d terdut -c \ + "INSERT INTO settings (key, value) VALUES ('signup_mode', 'open') ON CONFLICT (key) DO UPDATE SET value = EXCLUDED.value;" \ + >/dev/null || die "could not set signup_mode=open in Postgres" + + log "signing up ${ALICE_USERNAME}" + local body resp_file code + body="$(jq -n \ + --arg u "$ALICE_USERNAME" --arg e "$ALICE_EMAIL" \ + --arg p "$DEMO_PASSWORD" --arg t "$ALICE_TEAM_NAME" \ + '{username: $u, email: $e, password: $p, team_name: $t}')" + resp_file="$(mktemp)" + code="$(curl -sS -o "$resp_file" -w '%{http_code}' \ + -X POST "${BASE_URL}/api/signup" -H 'Content-Type: application/json' -d "$body")" + + case "$code" in + 201) log "created local account ${ALICE_USERNAME}" ;; + 409) log "account ${ALICE_USERNAME} already exists, skipping (re-run detected)" ;; + *) die "signup failed (HTTP $code): $(cat "$resp_file")" ;; + esac + rm -f "$resp_file" +} + +# Open signup always creates a brand-new team owned by the signer (it never +# joins an existing team by name -- confirmed against terdut-server's +# internal/api/signup.go), so without this step alice would have a working +# login that can't see a single incident this demo fires: /api/incidents +# and /api/alerts are scoped to the caller's own team memberships. Join her +# to both demo teams directly, the same way bootstrap_login already reaches +# into Postgres for signup_mode -- there is no API path for this either +# (adding a member to a team you don't already belong to is an owner/admin +# action, and alice is neither of those for Platform/Payments). +join_demo_teams() { + log "adding ${ALICE_USERNAME} to the Platform and Payments teams" + local platform_id payments_id pg_pod + platform_id="$(kubectl -n "$NAMESPACE" get terdutteam terdutteam-platform -o jsonpath='{.status.teamID}')" + payments_id="$(kubectl -n "$NAMESPACE" get terdutteam terdutteam-payments -o jsonpath='{.status.teamID}')" + [ -n "$platform_id" ] && [ -n "$payments_id" ] \ + || die "could not read status.teamID off terdutteam-platform/terdutteam-payments" + + pg_pod="$(kubectl -n "$NAMESPACE" get pod -l app=terdut-operator-demo-postgres \ + -o jsonpath='{.items[0].metadata.name}')" + kubectl -n "$NAMESPACE" exec "$pg_pod" -- psql -U terdut -d terdut -c " + INSERT INTO team_members (team_id, user_id, role) + VALUES + (${platform_id}, (SELECT id FROM users WHERE username = '${ALICE_USERNAME}'), 'member'), + (${payments_id}, (SELECT id FROM users WHERE username = '${ALICE_USERNAME}'), 'member') + ON CONFLICT (team_id, user_id) DO NOTHING;" \ + >/dev/null || die "could not add ${ALICE_USERNAME} to the demo teams" +} + +fire_demo_alerts() { + log "firing representative demo alerts" + NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform high-cpu + NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform disk-full + NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" payments pod-crash +} + +print_summary() { + cat </dev/null || echo unknown)} (log: ${PF_LOGFILE}) + stop it with: kill \$(cat ${PF_PIDFILE}) + + Fire more alerts: + export NAMESPACE=${NAMESPACE} BASE_URL=${BASE_URL} + ./fire-alerts.sh platform high-cpu + ./fire-alerts.sh platform high-cpu resolve + ./fire-alerts.sh payments heartbeat # send repeatedly (e.g. every minute) + # to keep a dead man's switch alive; + # stop sending it and, 15 minutes + # later, terdut-server opens a + # critical incident on its own. + + Tear down: + $0 --teardown + # add TEARDOWN_CLUSTER=true to also delete the kind cluster itself +EOF +} + +teardown() { + if [ -f "$PF_PIDFILE" ]; then + local pid + pid="$(cat "$PF_PIDFILE" 2>/dev/null || true)" + if [ -n "$pid" ] && kill -0 "$pid" 2>/dev/null; then + log "stopping port-forward (pid $pid)" + kill "$pid" 2>/dev/null || true + fi + rm -f "$PF_PIDFILE" + fi + + log "deleting namespace $NAMESPACE" + kubectl delete namespace "$NAMESPACE" --ignore-not-found --wait=true --timeout "$WAIT_TIMEOUT" + + if [ "$TEARDOWN_CLUSTER" = "true" ]; then + log "deleting kind cluster $CLUSTER_NAME" + kind delete cluster --name "$CLUSTER_NAME" + else + log "leaving kind cluster '$CLUSTER_NAME' and the operator install in place" \ + "(set TEARDOWN_CLUSTER=true to also delete the cluster)" + fi +} + +# --------------------------------------------------------------------------- +# main +# --------------------------------------------------------------------------- + +main() { + case "${1:-}" in + --teardown) + preflight + teardown + trap - EXIT + exit 0 + ;; + --help|-h) + usage + trap - EXIT + exit 0 + ;; + "") ;; + *) + usage + die "unknown argument: $1" + ;; + esac + + preflight + ensure_kind_cluster + install_operator + apply_demo + wait_for_ready + start_port_forward + bootstrap_login + join_demo_teams + fire_demo_alerts + print_summary + + # Success: leave the port-forward running, don't let the EXIT trap kill it. + trap - EXIT +} + +main "$@"