Files
terdut-operator/examples/demo/run-demo.sh
T
Niklas Ye 822c80dda6 examples/demo: split the Ready wait so escalation rules wait on alice too
wait_for_ready waited for every demo object at once, including
terdutescalationrule-platform, which names alice as a level-1 target --
but alice does not exist yet at that point in main(): she is created by
redeem_platform_invite, which ran after wait_for_ready. terdut-server
resolves every named username at reconcile time, not just when an
escalation actually fires, so that CR could never reach Ready before
alice did, and main() had no step in between to create her.

Split into wait_for_objects (the shared loop, now taking its object list
as arguments) plus two callers: wait_for_teams_ready, covering just the
server and the two teams redeem_platform_invite/join_payments_team
need, run before alice exists; wait_for_remaining_ready, covering the
escalation rules, dead man's switches and alert sources, run after.

Co-authored-by: Claude <noreply@anthropic.com>
2026-10-03 18:15:42 +02:00

350 lines
13 KiB
Bash
Executable File

#!/usr/bin/env bash
# Stands up the complete terdut demo (terdut-operator + every CRD kind this
# repo ships + a working local login + a few synthetic incidents) on a kind
# cluster, fully automated. Password login only -- this demo kit's own
# 01-server.yaml carries no oidc: block at all, so there is nothing to
# disable; OIDC is simply absent.
#
# Usage:
# ./run-demo.sh deploy the whole demo (idempotent: safe to
# re-run against a cluster that already has it)
# ./run-demo.sh --teardown delete the demo namespace (and, if
# TEARDOWN_CLUSTER=true, the kind cluster too)
# ./run-demo.sh --help
#
# All of the defaults below are overridable as environment variables.
set -euo pipefail
CLUSTER_NAME="${CLUSTER_NAME:-terdut-demo}"
NAMESPACE="${NAMESPACE:-terdut-operator-demo}"
OPERATOR_NAMESPACE="${OPERATOR_NAMESPACE:-terdut-operator-system}"
RELEASE_NAME="${RELEASE_NAME:-terdut-operator}"
ALICE_USERNAME="${ALICE_USERNAME:-alice}"
ALICE_EMAIL="${ALICE_EMAIL:-alice@terdut-demo.local}"
DEMO_PASSWORD="${DEMO_PASSWORD:-terdut-demo-1234}"
BASE_URL="${BASE_URL:-http://localhost:8080}"
LOCAL_PORT="${LOCAL_PORT:-8080}"
HELM_TIMEOUT="${HELM_TIMEOUT:-180s}"
WAIT_TIMEOUT="${WAIT_TIMEOUT:-180s}"
TEARDOWN_CLUSTER="${TEARDOWN_CLUSTER:-false}"
PF_PIDFILE="${PF_PIDFILE:-/tmp/terdut-demo-port-forward.pid}"
PF_LOGFILE="${PF_LOGFILE:-/tmp/terdut-demo-port-forward.log}"
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
CHART_DIR="$(cd "$SCRIPT_DIR/../.." && pwd)/charts/terdut-operator"
# ---------------------------------------------------------------------------
# helpers
# ---------------------------------------------------------------------------
log() { printf '[run-demo] %s\n' "$*" >&2; }
die() { printf '[run-demo] FAILED: %s\n' "$*" >&2; exit 1; }
usage() {
cat <<EOF
Usage: $0 [--teardown|--help]
Deploys (or tears down) the complete terdut demo on a kind cluster.
See the top of this file for every overridable environment variable.
EOF
}
# Only kills a port-forward THIS invocation started, so a successful run
# never has its background job reaped by its own exit trap.
STARTED_PF_PID=""
cleanup_on_failure() {
local rc=$?
if [ "$rc" -ne 0 ] && [ -n "$STARTED_PF_PID" ]; then
log "run failed -- stopping the port-forward it started (pid $STARTED_PF_PID)"
kill "$STARTED_PF_PID" 2>/dev/null || true
fi
exit "$rc"
}
trap cleanup_on_failure EXIT
# ---------------------------------------------------------------------------
# steps
# ---------------------------------------------------------------------------
preflight() {
local missing=()
for bin in kubectl kind helm jq curl; do
command -v "$bin" >/dev/null 2>&1 || missing+=("$bin")
done
if [ "${#missing[@]}" -gt 0 ]; then
die "missing required tools on PATH: ${missing[*]}"
fi
}
ensure_kind_cluster() {
case "$(kind get clusters 2>/dev/null)" in
*"$CLUSTER_NAME"*)
log "kind cluster '$CLUSTER_NAME' already exists, skipping creation" ;;
*)
log "creating kind cluster '$CLUSTER_NAME'"
kind create cluster --name "$CLUSTER_NAME" ;;
esac
kubectl config use-context "kind-${CLUSTER_NAME}" >/dev/null
}
install_operator() {
log "installing terdut-operator into namespace $OPERATOR_NAMESPACE"
helm upgrade --install "$RELEASE_NAME" "$CHART_DIR" \
--namespace "$OPERATOR_NAMESPACE" --create-namespace \
--wait --timeout "$HELM_TIMEOUT" \
|| die "helm install of terdut-operator did not become ready"
}
apply_demo() {
log "creating namespace $NAMESPACE"
kubectl create namespace "$NAMESPACE" --dry-run=client -o yaml | kubectl apply -f - >/dev/null
log "applying demo CRs (00-09) into $NAMESPACE"
kubectl apply -n "$NAMESPACE" -k "$SCRIPT_DIR" >/dev/null
}
wait_for_objects() {
local obj
for obj in "$@"; do
log "waiting for $obj to become Ready"
kubectl wait --for=condition=Ready --timeout "$WAIT_TIMEOUT" -n "$NAMESPACE" "$obj" >/dev/null \
|| die "timed out waiting for $obj -- try: kubectl describe -n $NAMESPACE $obj"
done
}
# Just the server and the two teams -- everything redeem_platform_invite and
# join_payments_team need. Deliberately NOT the escalation rules here: this
# demo kit's own terdutescalationrule-platform names alice as a level-1
# target, and that CR cannot reach Ready until alice actually exists
# (terdut-server resolves every named username at reconcile time, not just
# at escalation time) -- a real dependency this script has to satisfy by
# creating her first, not something kubectl wait can be told to ignore.
wait_for_teams_ready() {
wait_for_objects \
"terdutserver/terdut-operator-demo" \
"terdutteam/terdutteam-platform" \
"terdutteam/terdutteam-payments"
}
# Everything that was waiting on alice (or just on the teams above, now
# already satisfied) to exist.
wait_for_remaining_ready() {
wait_for_objects \
"terdutescalationrule/terdutescalationrule-platform" \
"terdutescalationrule/terdutescalationrule-payments" \
"terdutdeadmanswitch/terdutdeadmanswitch-platform" \
"terdutdeadmanswitch/terdutdeadmanswitch-payments" \
"terdutalertsource/terdutalertsource-platform" \
"terdutalertsource/terdutalertsource-payments"
}
start_port_forward() {
# A stale pidfile from an earlier run would otherwise collide with us on
# $LOCAL_PORT -- if that pid is still alive, stop it first.
if [ -f "$PF_PIDFILE" ]; then
local old_pid
old_pid="$(cat "$PF_PIDFILE" 2>/dev/null || true)"
if [ -n "$old_pid" ] && kill -0 "$old_pid" 2>/dev/null; then
log "stopping stale port-forward from a previous run (pid $old_pid)"
kill "$old_pid" 2>/dev/null || true
sleep 1
fi
rm -f "$PF_PIDFILE"
fi
log "starting port-forward svc/terdut-operator-demo ${LOCAL_PORT}:8080"
kubectl -n "$NAMESPACE" port-forward svc/terdut-operator-demo "${LOCAL_PORT}:8080" \
>"$PF_LOGFILE" 2>&1 &
STARTED_PF_PID=$!
echo "$STARTED_PF_PID" > "$PF_PIDFILE"
local tries=0
until curl -sf -o /dev/null "http://localhost:${LOCAL_PORT}/healthz"; do
tries=$((tries + 1))
if [ "$tries" -ge 30 ]; then
die "port-forward never became ready -- see $PF_LOGFILE"
fi
sleep 1
done
log "port-forward ready (pid $STARTED_PF_PID, log $PF_LOGFILE)"
}
# Redeems Platform's own invite link -- minted by its TerdutTeam
# (02-team-platform.yaml's spec.invite.enabled, reconciled through that
# team's own already-working team-scoped credential, which requireTeamOwner
# already treats as owner-equivalent for /invites -- ratified, not a
# workaround, in terdut-server's SERVICE-ACCOUNTS.md) and redeemed through
# the ordinary signup endpoint. Invite redemption bypasses signup_mode
# entirely (terdut-server's internal/api/signup.go), so this needs no admin
# credential, no signup_mode flip, and no direct Postgres access at all --
# unlike an earlier version of this script, before terdut-operator grew
# this feature (see niklas/terdut-server#23).
redeem_platform_invite() {
log "reading Platform's invite link"
local secret_name invite_url invite_token tries=0
until secret_name="$(kubectl -n "$NAMESPACE" get terdutteam terdutteam-platform \
-o jsonpath='{.status.inviteSecretRef.name}' 2>/dev/null)" && [ -n "$secret_name" ]; do
tries=$((tries + 1))
[ "$tries" -lt 30 ] || die "terdutteam-platform never reported status.inviteSecretRef -- check spec.invite.enabled and kubectl describe it"
sleep 1
done
invite_url="$(kubectl -n "$NAMESPACE" get secret "$secret_name" -o jsonpath='{.data.url}' | base64 -d)"
invite_token="${invite_url##*invite=}"
[ -n "$invite_token" ] || die "could not parse an invite token out of $invite_url"
log "signing up ${ALICE_USERNAME} via Platform's invite"
local body resp_file code
body="$(jq -n \
--arg u "$ALICE_USERNAME" --arg e "$ALICE_EMAIL" \
--arg p "$DEMO_PASSWORD" --arg i "$invite_token" \
'{username: $u, email: $e, password: $p, invite: $i}')"
resp_file="$(mktemp)"
code="$(curl -sS -o "$resp_file" -w '%{http_code}' \
-X POST "${BASE_URL}/api/signup" -H 'Content-Type: application/json' -d "$body")"
case "$code" in
201) log "created local account ${ALICE_USERNAME} (joined Platform)" ;;
409) log "account ${ALICE_USERNAME} already exists, skipping (re-run detected)" ;;
*) die "signup failed (HTTP $code): $(cat "$resp_file")" ;;
esac
rm -f "$resp_file"
}
# redeem_platform_invite's signup already used up alice's one signup -- a
# second POST /api/signup would just 409 on the taken username, it doesn't
# join an existing account to another team. Payments is joined through the
# ordinary team-scoped member endpoint instead, using that team's own
# already-working team-scoped credential (owner-equivalent, same reach the
# invite route above relies on) and alice's user id resolved via
# GET /api/users -- the same lookup tdclient.GetUserByUsername does
# operator-side. The endpoint upserts on (team_id, user_id), so this is
# naturally idempotent across re-runs with no separate conflict handling
# needed.
join_payments_team() {
log "adding ${ALICE_USERNAME} to Payments"
local payments_id payments_secret payments_key alice_id resp_file code
payments_id="$(kubectl -n "$NAMESPACE" get terdutteam terdutteam-payments -o jsonpath='{.status.teamID}')"
[ -n "$payments_id" ] || die "could not read status.teamID off terdutteam-payments"
payments_secret="$(kubectl -n "$NAMESPACE" get terdutteam terdutteam-payments \
-o jsonpath='{.status.credentialsSecretRef.name}')"
[ -n "$payments_secret" ] || die "terdutteam-payments has no status.credentialsSecretRef yet"
payments_key="$(kubectl -n "$OPERATOR_NAMESPACE" get secret "$payments_secret" -o jsonpath='{.data.token}' | base64 -d)"
alice_id="$(curl -sS -H "Authorization: Bearer $payments_key" "${BASE_URL}/api/users" \
| jq -r --arg u "$ALICE_USERNAME" '.[] | select(.username == $u) | .id')"
[ -n "$alice_id" ] || die "could not resolve ${ALICE_USERNAME}'s user id via GET /api/users"
resp_file="$(mktemp)"
code="$(curl -sS -o "$resp_file" -w '%{http_code}' \
-X POST "${BASE_URL}/api/teams/${payments_id}/members" \
-H "Authorization: Bearer $payments_key" -H 'Content-Type: application/json' \
-d "$(jq -n --argjson id "$alice_id" '{user_id: $id, role: "member"}')")"
[ "$code" = "204" ] || die "adding ${ALICE_USERNAME} to Payments failed (HTTP $code): $(cat "$resp_file")"
log "added ${ALICE_USERNAME} to Payments"
rm -f "$resp_file"
}
fire_demo_alerts() {
log "firing representative demo alerts"
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform high-cpu
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform disk-full
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" payments pod-crash
}
print_summary() {
cat <<EOF
terdut demo is up.
Web UI: http://localhost:${LOCAL_PORT}
Login: ${ALICE_USERNAME} / ${DEMO_PASSWORD}
Cluster: kind-${CLUSTER_NAME}
Namespace: ${NAMESPACE}
Port-forward pid: ${STARTED_PF_PID:-$(cat "$PF_PIDFILE" 2>/dev/null || echo unknown)} (log: ${PF_LOGFILE})
stop it with: kill \$(cat ${PF_PIDFILE})
Fire more alerts:
export NAMESPACE=${NAMESPACE} BASE_URL=${BASE_URL}
./fire-alerts.sh platform high-cpu
./fire-alerts.sh platform high-cpu resolve
./fire-alerts.sh payments heartbeat # send repeatedly (e.g. every minute)
# to keep a dead man's switch alive;
# stop sending it and, 15 minutes
# later, terdut-server opens a
# critical incident on its own.
Tear down:
$0 --teardown
# add TEARDOWN_CLUSTER=true to also delete the kind cluster itself
EOF
}
teardown() {
if [ -f "$PF_PIDFILE" ]; then
local pid
pid="$(cat "$PF_PIDFILE" 2>/dev/null || true)"
if [ -n "$pid" ] && kill -0 "$pid" 2>/dev/null; then
log "stopping port-forward (pid $pid)"
kill "$pid" 2>/dev/null || true
fi
rm -f "$PF_PIDFILE"
fi
log "deleting namespace $NAMESPACE"
kubectl delete namespace "$NAMESPACE" --ignore-not-found --wait=true --timeout "$WAIT_TIMEOUT"
if [ "$TEARDOWN_CLUSTER" = "true" ]; then
log "deleting kind cluster $CLUSTER_NAME"
kind delete cluster --name "$CLUSTER_NAME"
else
log "leaving kind cluster '$CLUSTER_NAME' and the operator install in place" \
"(set TEARDOWN_CLUSTER=true to also delete the cluster)"
fi
}
# ---------------------------------------------------------------------------
# main
# ---------------------------------------------------------------------------
main() {
case "${1:-}" in
--teardown)
preflight
teardown
trap - EXIT
exit 0
;;
--help|-h)
usage
trap - EXIT
exit 0
;;
"") ;;
*)
usage
die "unknown argument: $1"
;;
esac
preflight
ensure_kind_cluster
install_operator
apply_demo
wait_for_teams_ready
start_port_forward
redeem_platform_invite
join_payments_team
wait_for_remaining_ready
fire_demo_alerts
print_summary
# Success: leave the port-forward running, don't let the EXIT trap kill it.
trap - EXIT
}
main "$@"