822c80dda6
wait_for_ready waited for every demo object at once, including terdutescalationrule-platform, which names alice as a level-1 target -- but alice does not exist yet at that point in main(): she is created by redeem_platform_invite, which ran after wait_for_ready. terdut-server resolves every named username at reconcile time, not just when an escalation actually fires, so that CR could never reach Ready before alice did, and main() had no step in between to create her. Split into wait_for_objects (the shared loop, now taking its object list as arguments) plus two callers: wait_for_teams_ready, covering just the server and the two teams redeem_platform_invite/join_payments_team need, run before alice exists; wait_for_remaining_ready, covering the escalation rules, dead man's switches and alert sources, run after. Co-authored-by: Claude <noreply@anthropic.com>
350 lines
13 KiB
Bash
Executable File
350 lines
13 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# Stands up the complete terdut demo (terdut-operator + every CRD kind this
|
|
# repo ships + a working local login + a few synthetic incidents) on a kind
|
|
# cluster, fully automated. Password login only -- this demo kit's own
|
|
# 01-server.yaml carries no oidc: block at all, so there is nothing to
|
|
# disable; OIDC is simply absent.
|
|
#
|
|
# Usage:
|
|
# ./run-demo.sh deploy the whole demo (idempotent: safe to
|
|
# re-run against a cluster that already has it)
|
|
# ./run-demo.sh --teardown delete the demo namespace (and, if
|
|
# TEARDOWN_CLUSTER=true, the kind cluster too)
|
|
# ./run-demo.sh --help
|
|
#
|
|
# All of the defaults below are overridable as environment variables.
|
|
set -euo pipefail
|
|
|
|
CLUSTER_NAME="${CLUSTER_NAME:-terdut-demo}"
|
|
NAMESPACE="${NAMESPACE:-terdut-operator-demo}"
|
|
OPERATOR_NAMESPACE="${OPERATOR_NAMESPACE:-terdut-operator-system}"
|
|
RELEASE_NAME="${RELEASE_NAME:-terdut-operator}"
|
|
|
|
ALICE_USERNAME="${ALICE_USERNAME:-alice}"
|
|
ALICE_EMAIL="${ALICE_EMAIL:-alice@terdut-demo.local}"
|
|
DEMO_PASSWORD="${DEMO_PASSWORD:-terdut-demo-1234}"
|
|
|
|
BASE_URL="${BASE_URL:-http://localhost:8080}"
|
|
LOCAL_PORT="${LOCAL_PORT:-8080}"
|
|
|
|
HELM_TIMEOUT="${HELM_TIMEOUT:-180s}"
|
|
WAIT_TIMEOUT="${WAIT_TIMEOUT:-180s}"
|
|
|
|
TEARDOWN_CLUSTER="${TEARDOWN_CLUSTER:-false}"
|
|
|
|
PF_PIDFILE="${PF_PIDFILE:-/tmp/terdut-demo-port-forward.pid}"
|
|
PF_LOGFILE="${PF_LOGFILE:-/tmp/terdut-demo-port-forward.log}"
|
|
|
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
CHART_DIR="$(cd "$SCRIPT_DIR/../.." && pwd)/charts/terdut-operator"
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# helpers
|
|
# ---------------------------------------------------------------------------
|
|
|
|
log() { printf '[run-demo] %s\n' "$*" >&2; }
|
|
die() { printf '[run-demo] FAILED: %s\n' "$*" >&2; exit 1; }
|
|
|
|
usage() {
|
|
cat <<EOF
|
|
Usage: $0 [--teardown|--help]
|
|
|
|
Deploys (or tears down) the complete terdut demo on a kind cluster.
|
|
See the top of this file for every overridable environment variable.
|
|
EOF
|
|
}
|
|
|
|
# Only kills a port-forward THIS invocation started, so a successful run
|
|
# never has its background job reaped by its own exit trap.
|
|
STARTED_PF_PID=""
|
|
cleanup_on_failure() {
|
|
local rc=$?
|
|
if [ "$rc" -ne 0 ] && [ -n "$STARTED_PF_PID" ]; then
|
|
log "run failed -- stopping the port-forward it started (pid $STARTED_PF_PID)"
|
|
kill "$STARTED_PF_PID" 2>/dev/null || true
|
|
fi
|
|
exit "$rc"
|
|
}
|
|
trap cleanup_on_failure EXIT
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# steps
|
|
# ---------------------------------------------------------------------------
|
|
|
|
preflight() {
|
|
local missing=()
|
|
for bin in kubectl kind helm jq curl; do
|
|
command -v "$bin" >/dev/null 2>&1 || missing+=("$bin")
|
|
done
|
|
if [ "${#missing[@]}" -gt 0 ]; then
|
|
die "missing required tools on PATH: ${missing[*]}"
|
|
fi
|
|
}
|
|
|
|
ensure_kind_cluster() {
|
|
case "$(kind get clusters 2>/dev/null)" in
|
|
*"$CLUSTER_NAME"*)
|
|
log "kind cluster '$CLUSTER_NAME' already exists, skipping creation" ;;
|
|
*)
|
|
log "creating kind cluster '$CLUSTER_NAME'"
|
|
kind create cluster --name "$CLUSTER_NAME" ;;
|
|
esac
|
|
kubectl config use-context "kind-${CLUSTER_NAME}" >/dev/null
|
|
}
|
|
|
|
install_operator() {
|
|
log "installing terdut-operator into namespace $OPERATOR_NAMESPACE"
|
|
helm upgrade --install "$RELEASE_NAME" "$CHART_DIR" \
|
|
--namespace "$OPERATOR_NAMESPACE" --create-namespace \
|
|
--wait --timeout "$HELM_TIMEOUT" \
|
|
|| die "helm install of terdut-operator did not become ready"
|
|
}
|
|
|
|
apply_demo() {
|
|
log "creating namespace $NAMESPACE"
|
|
kubectl create namespace "$NAMESPACE" --dry-run=client -o yaml | kubectl apply -f - >/dev/null
|
|
log "applying demo CRs (00-09) into $NAMESPACE"
|
|
kubectl apply -n "$NAMESPACE" -k "$SCRIPT_DIR" >/dev/null
|
|
}
|
|
|
|
wait_for_objects() {
|
|
local obj
|
|
for obj in "$@"; do
|
|
log "waiting for $obj to become Ready"
|
|
kubectl wait --for=condition=Ready --timeout "$WAIT_TIMEOUT" -n "$NAMESPACE" "$obj" >/dev/null \
|
|
|| die "timed out waiting for $obj -- try: kubectl describe -n $NAMESPACE $obj"
|
|
done
|
|
}
|
|
|
|
# Just the server and the two teams -- everything redeem_platform_invite and
|
|
# join_payments_team need. Deliberately NOT the escalation rules here: this
|
|
# demo kit's own terdutescalationrule-platform names alice as a level-1
|
|
# target, and that CR cannot reach Ready until alice actually exists
|
|
# (terdut-server resolves every named username at reconcile time, not just
|
|
# at escalation time) -- a real dependency this script has to satisfy by
|
|
# creating her first, not something kubectl wait can be told to ignore.
|
|
wait_for_teams_ready() {
|
|
wait_for_objects \
|
|
"terdutserver/terdut-operator-demo" \
|
|
"terdutteam/terdutteam-platform" \
|
|
"terdutteam/terdutteam-payments"
|
|
}
|
|
|
|
# Everything that was waiting on alice (or just on the teams above, now
|
|
# already satisfied) to exist.
|
|
wait_for_remaining_ready() {
|
|
wait_for_objects \
|
|
"terdutescalationrule/terdutescalationrule-platform" \
|
|
"terdutescalationrule/terdutescalationrule-payments" \
|
|
"terdutdeadmanswitch/terdutdeadmanswitch-platform" \
|
|
"terdutdeadmanswitch/terdutdeadmanswitch-payments" \
|
|
"terdutalertsource/terdutalertsource-platform" \
|
|
"terdutalertsource/terdutalertsource-payments"
|
|
}
|
|
|
|
start_port_forward() {
|
|
# A stale pidfile from an earlier run would otherwise collide with us on
|
|
# $LOCAL_PORT -- if that pid is still alive, stop it first.
|
|
if [ -f "$PF_PIDFILE" ]; then
|
|
local old_pid
|
|
old_pid="$(cat "$PF_PIDFILE" 2>/dev/null || true)"
|
|
if [ -n "$old_pid" ] && kill -0 "$old_pid" 2>/dev/null; then
|
|
log "stopping stale port-forward from a previous run (pid $old_pid)"
|
|
kill "$old_pid" 2>/dev/null || true
|
|
sleep 1
|
|
fi
|
|
rm -f "$PF_PIDFILE"
|
|
fi
|
|
|
|
log "starting port-forward svc/terdut-operator-demo ${LOCAL_PORT}:8080"
|
|
kubectl -n "$NAMESPACE" port-forward svc/terdut-operator-demo "${LOCAL_PORT}:8080" \
|
|
>"$PF_LOGFILE" 2>&1 &
|
|
STARTED_PF_PID=$!
|
|
echo "$STARTED_PF_PID" > "$PF_PIDFILE"
|
|
|
|
local tries=0
|
|
until curl -sf -o /dev/null "http://localhost:${LOCAL_PORT}/healthz"; do
|
|
tries=$((tries + 1))
|
|
if [ "$tries" -ge 30 ]; then
|
|
die "port-forward never became ready -- see $PF_LOGFILE"
|
|
fi
|
|
sleep 1
|
|
done
|
|
log "port-forward ready (pid $STARTED_PF_PID, log $PF_LOGFILE)"
|
|
}
|
|
|
|
# Redeems Platform's own invite link -- minted by its TerdutTeam
|
|
# (02-team-platform.yaml's spec.invite.enabled, reconciled through that
|
|
# team's own already-working team-scoped credential, which requireTeamOwner
|
|
# already treats as owner-equivalent for /invites -- ratified, not a
|
|
# workaround, in terdut-server's SERVICE-ACCOUNTS.md) and redeemed through
|
|
# the ordinary signup endpoint. Invite redemption bypasses signup_mode
|
|
# entirely (terdut-server's internal/api/signup.go), so this needs no admin
|
|
# credential, no signup_mode flip, and no direct Postgres access at all --
|
|
# unlike an earlier version of this script, before terdut-operator grew
|
|
# this feature (see niklas/terdut-server#23).
|
|
redeem_platform_invite() {
|
|
log "reading Platform's invite link"
|
|
local secret_name invite_url invite_token tries=0
|
|
until secret_name="$(kubectl -n "$NAMESPACE" get terdutteam terdutteam-platform \
|
|
-o jsonpath='{.status.inviteSecretRef.name}' 2>/dev/null)" && [ -n "$secret_name" ]; do
|
|
tries=$((tries + 1))
|
|
[ "$tries" -lt 30 ] || die "terdutteam-platform never reported status.inviteSecretRef -- check spec.invite.enabled and kubectl describe it"
|
|
sleep 1
|
|
done
|
|
invite_url="$(kubectl -n "$NAMESPACE" get secret "$secret_name" -o jsonpath='{.data.url}' | base64 -d)"
|
|
invite_token="${invite_url##*invite=}"
|
|
[ -n "$invite_token" ] || die "could not parse an invite token out of $invite_url"
|
|
|
|
log "signing up ${ALICE_USERNAME} via Platform's invite"
|
|
local body resp_file code
|
|
body="$(jq -n \
|
|
--arg u "$ALICE_USERNAME" --arg e "$ALICE_EMAIL" \
|
|
--arg p "$DEMO_PASSWORD" --arg i "$invite_token" \
|
|
'{username: $u, email: $e, password: $p, invite: $i}')"
|
|
resp_file="$(mktemp)"
|
|
code="$(curl -sS -o "$resp_file" -w '%{http_code}' \
|
|
-X POST "${BASE_URL}/api/signup" -H 'Content-Type: application/json' -d "$body")"
|
|
|
|
case "$code" in
|
|
201) log "created local account ${ALICE_USERNAME} (joined Platform)" ;;
|
|
409) log "account ${ALICE_USERNAME} already exists, skipping (re-run detected)" ;;
|
|
*) die "signup failed (HTTP $code): $(cat "$resp_file")" ;;
|
|
esac
|
|
rm -f "$resp_file"
|
|
}
|
|
|
|
# redeem_platform_invite's signup already used up alice's one signup -- a
|
|
# second POST /api/signup would just 409 on the taken username, it doesn't
|
|
# join an existing account to another team. Payments is joined through the
|
|
# ordinary team-scoped member endpoint instead, using that team's own
|
|
# already-working team-scoped credential (owner-equivalent, same reach the
|
|
# invite route above relies on) and alice's user id resolved via
|
|
# GET /api/users -- the same lookup tdclient.GetUserByUsername does
|
|
# operator-side. The endpoint upserts on (team_id, user_id), so this is
|
|
# naturally idempotent across re-runs with no separate conflict handling
|
|
# needed.
|
|
join_payments_team() {
|
|
log "adding ${ALICE_USERNAME} to Payments"
|
|
local payments_id payments_secret payments_key alice_id resp_file code
|
|
payments_id="$(kubectl -n "$NAMESPACE" get terdutteam terdutteam-payments -o jsonpath='{.status.teamID}')"
|
|
[ -n "$payments_id" ] || die "could not read status.teamID off terdutteam-payments"
|
|
payments_secret="$(kubectl -n "$NAMESPACE" get terdutteam terdutteam-payments \
|
|
-o jsonpath='{.status.credentialsSecretRef.name}')"
|
|
[ -n "$payments_secret" ] || die "terdutteam-payments has no status.credentialsSecretRef yet"
|
|
payments_key="$(kubectl -n "$OPERATOR_NAMESPACE" get secret "$payments_secret" -o jsonpath='{.data.token}' | base64 -d)"
|
|
|
|
alice_id="$(curl -sS -H "Authorization: Bearer $payments_key" "${BASE_URL}/api/users" \
|
|
| jq -r --arg u "$ALICE_USERNAME" '.[] | select(.username == $u) | .id')"
|
|
[ -n "$alice_id" ] || die "could not resolve ${ALICE_USERNAME}'s user id via GET /api/users"
|
|
|
|
resp_file="$(mktemp)"
|
|
code="$(curl -sS -o "$resp_file" -w '%{http_code}' \
|
|
-X POST "${BASE_URL}/api/teams/${payments_id}/members" \
|
|
-H "Authorization: Bearer $payments_key" -H 'Content-Type: application/json' \
|
|
-d "$(jq -n --argjson id "$alice_id" '{user_id: $id, role: "member"}')")"
|
|
[ "$code" = "204" ] || die "adding ${ALICE_USERNAME} to Payments failed (HTTP $code): $(cat "$resp_file")"
|
|
log "added ${ALICE_USERNAME} to Payments"
|
|
rm -f "$resp_file"
|
|
}
|
|
|
|
fire_demo_alerts() {
|
|
log "firing representative demo alerts"
|
|
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform high-cpu
|
|
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform disk-full
|
|
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" payments pod-crash
|
|
}
|
|
|
|
print_summary() {
|
|
cat <<EOF
|
|
|
|
terdut demo is up.
|
|
|
|
Web UI: http://localhost:${LOCAL_PORT}
|
|
Login: ${ALICE_USERNAME} / ${DEMO_PASSWORD}
|
|
Cluster: kind-${CLUSTER_NAME}
|
|
Namespace: ${NAMESPACE}
|
|
Port-forward pid: ${STARTED_PF_PID:-$(cat "$PF_PIDFILE" 2>/dev/null || echo unknown)} (log: ${PF_LOGFILE})
|
|
stop it with: kill \$(cat ${PF_PIDFILE})
|
|
|
|
Fire more alerts:
|
|
export NAMESPACE=${NAMESPACE} BASE_URL=${BASE_URL}
|
|
./fire-alerts.sh platform high-cpu
|
|
./fire-alerts.sh platform high-cpu resolve
|
|
./fire-alerts.sh payments heartbeat # send repeatedly (e.g. every minute)
|
|
# to keep a dead man's switch alive;
|
|
# stop sending it and, 15 minutes
|
|
# later, terdut-server opens a
|
|
# critical incident on its own.
|
|
|
|
Tear down:
|
|
$0 --teardown
|
|
# add TEARDOWN_CLUSTER=true to also delete the kind cluster itself
|
|
EOF
|
|
}
|
|
|
|
teardown() {
|
|
if [ -f "$PF_PIDFILE" ]; then
|
|
local pid
|
|
pid="$(cat "$PF_PIDFILE" 2>/dev/null || true)"
|
|
if [ -n "$pid" ] && kill -0 "$pid" 2>/dev/null; then
|
|
log "stopping port-forward (pid $pid)"
|
|
kill "$pid" 2>/dev/null || true
|
|
fi
|
|
rm -f "$PF_PIDFILE"
|
|
fi
|
|
|
|
log "deleting namespace $NAMESPACE"
|
|
kubectl delete namespace "$NAMESPACE" --ignore-not-found --wait=true --timeout "$WAIT_TIMEOUT"
|
|
|
|
if [ "$TEARDOWN_CLUSTER" = "true" ]; then
|
|
log "deleting kind cluster $CLUSTER_NAME"
|
|
kind delete cluster --name "$CLUSTER_NAME"
|
|
else
|
|
log "leaving kind cluster '$CLUSTER_NAME' and the operator install in place" \
|
|
"(set TEARDOWN_CLUSTER=true to also delete the cluster)"
|
|
fi
|
|
}
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# main
|
|
# ---------------------------------------------------------------------------
|
|
|
|
main() {
|
|
case "${1:-}" in
|
|
--teardown)
|
|
preflight
|
|
teardown
|
|
trap - EXIT
|
|
exit 0
|
|
;;
|
|
--help|-h)
|
|
usage
|
|
trap - EXIT
|
|
exit 0
|
|
;;
|
|
"") ;;
|
|
*)
|
|
usage
|
|
die "unknown argument: $1"
|
|
;;
|
|
esac
|
|
|
|
preflight
|
|
ensure_kind_cluster
|
|
install_operator
|
|
apply_demo
|
|
wait_for_teams_ready
|
|
start_port_forward
|
|
redeem_platform_invite
|
|
join_payments_team
|
|
wait_for_remaining_ready
|
|
fire_demo_alerts
|
|
print_summary
|
|
|
|
# Success: leave the port-forward running, don't let the EXIT trap kill it.
|
|
trap - EXIT
|
|
}
|
|
|
|
main "$@"
|