diff --git a/examples/demo/run-demo.sh b/examples/demo/run-demo.sh new file mode 100755 index 0000000..f19795f --- /dev/null +++ b/examples/demo/run-demo.sh @@ -0,0 +1,324 @@ +#!/usr/bin/env bash +# Stands up the complete terdut demo (terdut-operator + every CRD kind this +# repo ships + a working local login + a few synthetic incidents) on a kind +# cluster, fully automated. Password login only -- this demo kit's own +# 01-server.yaml carries no oidc: block at all, so there is nothing to +# disable; OIDC is simply absent. +# +# Usage: +# ./run-demo.sh deploy the whole demo (idempotent: safe to +# re-run against a cluster that already has it) +# ./run-demo.sh --teardown delete the demo namespace (and, if +# TEARDOWN_CLUSTER=true, the kind cluster too) +# ./run-demo.sh --help +# +# All of the defaults below are overridable as environment variables. +set -euo pipefail + +CLUSTER_NAME="${CLUSTER_NAME:-terdut-demo}" +NAMESPACE="${NAMESPACE:-terdut-operator-demo}" +OPERATOR_NAMESPACE="${OPERATOR_NAMESPACE:-terdut-operator-system}" +RELEASE_NAME="${RELEASE_NAME:-terdut-operator}" + +ALICE_USERNAME="${ALICE_USERNAME:-alice}" +ALICE_EMAIL="${ALICE_EMAIL:-alice@terdut-demo.local}" +ALICE_TEAM_NAME="${ALICE_TEAM_NAME:-alice-demo}" +DEMO_PASSWORD="${DEMO_PASSWORD:-terdut-demo-1234}" + +BASE_URL="${BASE_URL:-http://localhost:8080}" +LOCAL_PORT="${LOCAL_PORT:-8080}" + +HELM_TIMEOUT="${HELM_TIMEOUT:-180s}" +WAIT_TIMEOUT="${WAIT_TIMEOUT:-180s}" + +TEARDOWN_CLUSTER="${TEARDOWN_CLUSTER:-false}" + +PF_PIDFILE="${PF_PIDFILE:-/tmp/terdut-demo-port-forward.pid}" +PF_LOGFILE="${PF_LOGFILE:-/tmp/terdut-demo-port-forward.log}" + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +CHART_DIR="$(cd "$SCRIPT_DIR/../.." && pwd)/charts/terdut-operator" + +# --------------------------------------------------------------------------- +# helpers +# --------------------------------------------------------------------------- + +log() { printf '[run-demo] %s\n' "$*" >&2; } +die() { printf '[run-demo] FAILED: %s\n' "$*" >&2; exit 1; } + +usage() { + cat </dev/null || true + fi + exit "$rc" +} +trap cleanup_on_failure EXIT + +# --------------------------------------------------------------------------- +# steps +# --------------------------------------------------------------------------- + +preflight() { + local missing=() + for bin in kubectl kind helm jq curl; do + command -v "$bin" >/dev/null 2>&1 || missing+=("$bin") + done + if [ "${#missing[@]}" -gt 0 ]; then + die "missing required tools on PATH: ${missing[*]}" + fi +} + +ensure_kind_cluster() { + case "$(kind get clusters 2>/dev/null)" in + *"$CLUSTER_NAME"*) + log "kind cluster '$CLUSTER_NAME' already exists, skipping creation" ;; + *) + log "creating kind cluster '$CLUSTER_NAME'" + kind create cluster --name "$CLUSTER_NAME" ;; + esac + kubectl config use-context "kind-${CLUSTER_NAME}" >/dev/null +} + +install_operator() { + log "installing terdut-operator into namespace $OPERATOR_NAMESPACE" + helm upgrade --install "$RELEASE_NAME" "$CHART_DIR" \ + --namespace "$OPERATOR_NAMESPACE" --create-namespace \ + --wait --timeout "$HELM_TIMEOUT" \ + || die "helm install of terdut-operator did not become ready" +} + +apply_demo() { + log "creating namespace $NAMESPACE" + kubectl create namespace "$NAMESPACE" --dry-run=client -o yaml | kubectl apply -f - >/dev/null + log "applying demo CRs (00-09) into $NAMESPACE" + kubectl apply -n "$NAMESPACE" -k "$SCRIPT_DIR" >/dev/null +} + +wait_for_ready() { + local objects=( + "terdutserver/terdut-operator-demo" + "terdutteam/terdutteam-platform" + "terdutteam/terdutteam-payments" + "terdutescalationrule/terdutescalationrule-platform" + "terdutescalationrule/terdutescalationrule-payments" + "terdutdeadmanswitch/terdutdeadmanswitch-platform" + "terdutdeadmanswitch/terdutdeadmanswitch-payments" + "terdutalertsource/terdutalertsource-platform" + "terdutalertsource/terdutalertsource-payments" + ) + local obj + for obj in "${objects[@]}"; do + log "waiting for $obj to become Ready" + kubectl wait --for=condition=Ready --timeout "$WAIT_TIMEOUT" -n "$NAMESPACE" "$obj" >/dev/null \ + || die "timed out waiting for $obj -- try: kubectl describe -n $NAMESPACE $obj" + done +} + +start_port_forward() { + # A stale pidfile from an earlier run would otherwise collide with us on + # $LOCAL_PORT -- if that pid is still alive, stop it first. + if [ -f "$PF_PIDFILE" ]; then + local old_pid + old_pid="$(cat "$PF_PIDFILE" 2>/dev/null || true)" + if [ -n "$old_pid" ] && kill -0 "$old_pid" 2>/dev/null; then + log "stopping stale port-forward from a previous run (pid $old_pid)" + kill "$old_pid" 2>/dev/null || true + sleep 1 + fi + rm -f "$PF_PIDFILE" + fi + + log "starting port-forward svc/terdut-operator-demo ${LOCAL_PORT}:8080" + kubectl -n "$NAMESPACE" port-forward svc/terdut-operator-demo "${LOCAL_PORT}:8080" \ + >"$PF_LOGFILE" 2>&1 & + STARTED_PF_PID=$! + echo "$STARTED_PF_PID" > "$PF_PIDFILE" + + local tries=0 + until curl -sf -o /dev/null "http://localhost:${LOCAL_PORT}/healthz"; do + tries=$((tries + 1)) + if [ "$tries" -ge 30 ]; then + die "port-forward never became ready -- see $PF_LOGFILE" + fi + sleep 1 + done + log "port-forward ready (pid $STARTED_PF_PID, log $PF_LOGFILE)" +} + +# Flips signup_mode to 'open' directly in Postgres, then uses the ordinary +# unauthenticated signup endpoint to create a real local account with a +# server-computed bcrypt hash. See this repo's run-demo.sh header / the +# plan this was built from for why this bypasses the demo README's own +# (currently broken) admin-token approach: the operator's service-account +# credential cannot call /api/admin/settings or POST /api/users -- AdminOnly +# only recognizes a human session/API-key caller, confirmed against +# terdut-server's internal/api/middleware.go. +bootstrap_login() { + log "flipping signup_mode to open directly in Postgres" + local pg_pod + pg_pod="$(kubectl -n "$NAMESPACE" get pod -l app=terdut-operator-demo-postgres \ + -o jsonpath='{.items[0].metadata.name}')" + [ -n "$pg_pod" ] || die "could not find the demo Postgres pod in $NAMESPACE" + + kubectl -n "$NAMESPACE" exec "$pg_pod" -- psql -U terdut -d terdut -c \ + "INSERT INTO settings (key, value) VALUES ('signup_mode', 'open') ON CONFLICT (key) DO UPDATE SET value = EXCLUDED.value;" \ + >/dev/null || die "could not set signup_mode=open in Postgres" + + log "signing up ${ALICE_USERNAME}" + local body resp_file code + body="$(jq -n \ + --arg u "$ALICE_USERNAME" --arg e "$ALICE_EMAIL" \ + --arg p "$DEMO_PASSWORD" --arg t "$ALICE_TEAM_NAME" \ + '{username: $u, email: $e, password: $p, team_name: $t}')" + resp_file="$(mktemp)" + code="$(curl -sS -o "$resp_file" -w '%{http_code}' \ + -X POST "${BASE_URL}/api/signup" -H 'Content-Type: application/json' -d "$body")" + + case "$code" in + 201) log "created local account ${ALICE_USERNAME}" ;; + 409) log "account ${ALICE_USERNAME} already exists, skipping (re-run detected)" ;; + *) die "signup failed (HTTP $code): $(cat "$resp_file")" ;; + esac + rm -f "$resp_file" +} + +# Open signup always creates a brand-new team owned by the signer (it never +# joins an existing team by name -- confirmed against terdut-server's +# internal/api/signup.go), so without this step alice would have a working +# login that can't see a single incident this demo fires: /api/incidents +# and /api/alerts are scoped to the caller's own team memberships. Join her +# to both demo teams directly, the same way bootstrap_login already reaches +# into Postgres for signup_mode -- there is no API path for this either +# (adding a member to a team you don't already belong to is an owner/admin +# action, and alice is neither of those for Platform/Payments). +join_demo_teams() { + log "adding ${ALICE_USERNAME} to the Platform and Payments teams" + local platform_id payments_id pg_pod + platform_id="$(kubectl -n "$NAMESPACE" get terdutteam terdutteam-platform -o jsonpath='{.status.teamID}')" + payments_id="$(kubectl -n "$NAMESPACE" get terdutteam terdutteam-payments -o jsonpath='{.status.teamID}')" + [ -n "$platform_id" ] && [ -n "$payments_id" ] \ + || die "could not read status.teamID off terdutteam-platform/terdutteam-payments" + + pg_pod="$(kubectl -n "$NAMESPACE" get pod -l app=terdut-operator-demo-postgres \ + -o jsonpath='{.items[0].metadata.name}')" + kubectl -n "$NAMESPACE" exec "$pg_pod" -- psql -U terdut -d terdut -c " + INSERT INTO team_members (team_id, user_id, role) + VALUES + (${platform_id}, (SELECT id FROM users WHERE username = '${ALICE_USERNAME}'), 'member'), + (${payments_id}, (SELECT id FROM users WHERE username = '${ALICE_USERNAME}'), 'member') + ON CONFLICT (team_id, user_id) DO NOTHING;" \ + >/dev/null || die "could not add ${ALICE_USERNAME} to the demo teams" +} + +fire_demo_alerts() { + log "firing representative demo alerts" + NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform high-cpu + NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform disk-full + NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" payments pod-crash +} + +print_summary() { + cat </dev/null || echo unknown)} (log: ${PF_LOGFILE}) + stop it with: kill \$(cat ${PF_PIDFILE}) + + Fire more alerts: + export NAMESPACE=${NAMESPACE} BASE_URL=${BASE_URL} + ./fire-alerts.sh platform high-cpu + ./fire-alerts.sh platform high-cpu resolve + ./fire-alerts.sh payments heartbeat # send repeatedly (e.g. every minute) + # to keep a dead man's switch alive; + # stop sending it and, 15 minutes + # later, terdut-server opens a + # critical incident on its own. + + Tear down: + $0 --teardown + # add TEARDOWN_CLUSTER=true to also delete the kind cluster itself +EOF +} + +teardown() { + if [ -f "$PF_PIDFILE" ]; then + local pid + pid="$(cat "$PF_PIDFILE" 2>/dev/null || true)" + if [ -n "$pid" ] && kill -0 "$pid" 2>/dev/null; then + log "stopping port-forward (pid $pid)" + kill "$pid" 2>/dev/null || true + fi + rm -f "$PF_PIDFILE" + fi + + log "deleting namespace $NAMESPACE" + kubectl delete namespace "$NAMESPACE" --ignore-not-found --wait=true --timeout "$WAIT_TIMEOUT" + + if [ "$TEARDOWN_CLUSTER" = "true" ]; then + log "deleting kind cluster $CLUSTER_NAME" + kind delete cluster --name "$CLUSTER_NAME" + else + log "leaving kind cluster '$CLUSTER_NAME' and the operator install in place" \ + "(set TEARDOWN_CLUSTER=true to also delete the cluster)" + fi +} + +# --------------------------------------------------------------------------- +# main +# --------------------------------------------------------------------------- + +main() { + case "${1:-}" in + --teardown) + preflight + teardown + trap - EXIT + exit 0 + ;; + --help|-h) + usage + trap - EXIT + exit 0 + ;; + "") ;; + *) + usage + die "unknown argument: $1" + ;; + esac + + preflight + ensure_kind_cluster + install_operator + apply_demo + wait_for_ready + start_port_forward + bootstrap_login + join_demo_teams + fire_demo_alerts + print_summary + + # Success: leave the port-forward running, don't let the EXIT trap kill it. + trap - EXIT +} + +main "$@"