50ce5bcec0
The demo pinned v0.36.0, the floor for replicas: 2, and so showed none of the web UI since: the queue and incident layouts, the rota and escalation pages, the theme toggle, and the cluster chip, filter and page titles (v0.42.0-v0.43.0). It pins v0.43.0 now; the comment keeps v0.36.0 as the floor, which is what the replicas setting actually depends on. fire-alerts.sh takes an optional CLUSTER, standing in for a Prometheus external label plus `cluster` in Alertmanager's group_by (terdut-server's README, "Several clusters, one team"). It goes on the alert's labels and groupLabels, and into the group key and the fingerprint, so the same alert in two clusters is two incidents and not one. Unset, the payload is exactly what it was. run-demo.sh fires its alerts across prod-eu and prod-us, high-cpu in both, so the queue has a chip and a filter to show. run-demo.sh also failed on its second run, though it says it is safe to re-run: it expected HTTP 409 when alice already exists, but a spent invite is answered with 403 "invite link is not usable" before the username is ever checked. It now tries to log alice in first and skips the signup if that works. Checked on the kind cluster: the server rolled to v0.43.0, every CR became Ready and Adopted (server, both teams, both escalation rules, both dead man's switches, both alert sources), and /api/incidents/clusters, /api/incidents?cluster=prod-us and the incident titles came back as expected. No operator code changed, so this needs no operator release. Co-authored-by: Claude <noreply@anthropic.com>
366 lines
14 KiB
Bash
Executable File
366 lines
14 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# Stands up the complete terdut demo (terdut-operator + every CRD kind this
|
|
# repo ships + a working local login + a few synthetic incidents) on a kind
|
|
# cluster, fully automated. Password login only -- this demo kit's own
|
|
# 01-server.yaml carries no oidc: block at all, so there is nothing to
|
|
# disable; OIDC is simply absent.
|
|
#
|
|
# Usage:
|
|
# ./run-demo.sh deploy the whole demo (idempotent: safe to
|
|
# re-run against a cluster that already has it)
|
|
# ./run-demo.sh --teardown delete the demo namespace (and, if
|
|
# TEARDOWN_CLUSTER=true, the kind cluster too)
|
|
# ./run-demo.sh --help
|
|
#
|
|
# All of the defaults below are overridable as environment variables.
|
|
set -euo pipefail
|
|
|
|
CLUSTER_NAME="${CLUSTER_NAME:-terdut-demo}"
|
|
NAMESPACE="${NAMESPACE:-terdut-operator-demo}"
|
|
OPERATOR_NAMESPACE="${OPERATOR_NAMESPACE:-terdut-operator-system}"
|
|
RELEASE_NAME="${RELEASE_NAME:-terdut-operator}"
|
|
|
|
ALICE_USERNAME="${ALICE_USERNAME:-alice}"
|
|
ALICE_EMAIL="${ALICE_EMAIL:-alice@terdut-demo.local}"
|
|
DEMO_PASSWORD="${DEMO_PASSWORD:-terdut-demo-1234}"
|
|
|
|
BASE_URL="${BASE_URL:-http://localhost:8080}"
|
|
LOCAL_PORT="${LOCAL_PORT:-8080}"
|
|
|
|
HELM_TIMEOUT="${HELM_TIMEOUT:-180s}"
|
|
WAIT_TIMEOUT="${WAIT_TIMEOUT:-180s}"
|
|
|
|
TEARDOWN_CLUSTER="${TEARDOWN_CLUSTER:-false}"
|
|
|
|
PF_PIDFILE="${PF_PIDFILE:-/tmp/terdut-demo-port-forward.pid}"
|
|
PF_LOGFILE="${PF_LOGFILE:-/tmp/terdut-demo-port-forward.log}"
|
|
|
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
CHART_DIR="$(cd "$SCRIPT_DIR/../.." && pwd)/charts/terdut-operator"
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# helpers
|
|
# ---------------------------------------------------------------------------
|
|
|
|
log() { printf '[run-demo] %s\n' "$*" >&2; }
|
|
die() { printf '[run-demo] FAILED: %s\n' "$*" >&2; exit 1; }
|
|
|
|
usage() {
|
|
cat <<EOF
|
|
Usage: $0 [--teardown|--help]
|
|
|
|
Deploys (or tears down) the complete terdut demo on a kind cluster.
|
|
See the top of this file for every overridable environment variable.
|
|
EOF
|
|
}
|
|
|
|
# Only kills a port-forward THIS invocation started, so a successful run
|
|
# never has its background job reaped by its own exit trap.
|
|
STARTED_PF_PID=""
|
|
cleanup_on_failure() {
|
|
local rc=$?
|
|
if [ "$rc" -ne 0 ] && [ -n "$STARTED_PF_PID" ]; then
|
|
log "run failed -- stopping the port-forward it started (pid $STARTED_PF_PID)"
|
|
kill "$STARTED_PF_PID" 2>/dev/null || true
|
|
fi
|
|
exit "$rc"
|
|
}
|
|
trap cleanup_on_failure EXIT
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# steps
|
|
# ---------------------------------------------------------------------------
|
|
|
|
preflight() {
|
|
local missing=()
|
|
for bin in kubectl kind helm jq curl; do
|
|
command -v "$bin" >/dev/null 2>&1 || missing+=("$bin")
|
|
done
|
|
if [ "${#missing[@]}" -gt 0 ]; then
|
|
die "missing required tools on PATH: ${missing[*]}"
|
|
fi
|
|
}
|
|
|
|
ensure_kind_cluster() {
|
|
case "$(kind get clusters 2>/dev/null)" in
|
|
*"$CLUSTER_NAME"*)
|
|
log "kind cluster '$CLUSTER_NAME' already exists, skipping creation" ;;
|
|
*)
|
|
log "creating kind cluster '$CLUSTER_NAME'"
|
|
kind create cluster --name "$CLUSTER_NAME" ;;
|
|
esac
|
|
kubectl config use-context "kind-${CLUSTER_NAME}" >/dev/null
|
|
}
|
|
|
|
install_operator() {
|
|
log "installing terdut-operator into namespace $OPERATOR_NAMESPACE"
|
|
helm upgrade --install "$RELEASE_NAME" "$CHART_DIR" \
|
|
--namespace "$OPERATOR_NAMESPACE" --create-namespace \
|
|
--wait --timeout "$HELM_TIMEOUT" \
|
|
|| die "helm install of terdut-operator did not become ready"
|
|
}
|
|
|
|
apply_demo() {
|
|
log "creating namespace $NAMESPACE"
|
|
kubectl create namespace "$NAMESPACE" --dry-run=client -o yaml | kubectl apply -f - >/dev/null
|
|
log "applying demo CRs (00-09) into $NAMESPACE"
|
|
kubectl apply -n "$NAMESPACE" -k "$SCRIPT_DIR" >/dev/null
|
|
}
|
|
|
|
wait_for_objects() {
|
|
local obj
|
|
for obj in "$@"; do
|
|
log "waiting for $obj to become Ready"
|
|
kubectl wait --for=condition=Ready --timeout "$WAIT_TIMEOUT" -n "$NAMESPACE" "$obj" >/dev/null \
|
|
|| die "timed out waiting for $obj -- try: kubectl describe -n $NAMESPACE $obj"
|
|
done
|
|
}
|
|
|
|
# Just the server and the two teams -- everything redeem_platform_invite and
|
|
# join_payments_team need. Deliberately NOT the escalation rules here: this
|
|
# demo kit's own terdutescalationrule-platform names alice as a level-1
|
|
# target, and that CR cannot reach Ready until alice actually exists
|
|
# (terdut-server resolves every named username at reconcile time, not just
|
|
# at escalation time) -- a real dependency this script has to satisfy by
|
|
# creating her first, not something kubectl wait can be told to ignore.
|
|
wait_for_teams_ready() {
|
|
wait_for_objects \
|
|
"terdutserver/terdut-operator-demo" \
|
|
"terdutteam/terdutteam-platform" \
|
|
"terdutteam/terdutteam-payments"
|
|
}
|
|
|
|
# Everything that was waiting on alice (or just on the teams above, now
|
|
# already satisfied) to exist.
|
|
wait_for_remaining_ready() {
|
|
wait_for_objects \
|
|
"terdutescalationrule/terdutescalationrule-platform" \
|
|
"terdutescalationrule/terdutescalationrule-payments" \
|
|
"terdutdeadmanswitch/terdutdeadmanswitch-platform" \
|
|
"terdutdeadmanswitch/terdutdeadmanswitch-payments" \
|
|
"terdutalertsource/terdutalertsource-platform" \
|
|
"terdutalertsource/terdutalertsource-payments"
|
|
}
|
|
|
|
start_port_forward() {
|
|
# A stale pidfile from an earlier run would otherwise collide with us on
|
|
# $LOCAL_PORT -- if that pid is still alive, stop it first.
|
|
if [ -f "$PF_PIDFILE" ]; then
|
|
local old_pid
|
|
old_pid="$(cat "$PF_PIDFILE" 2>/dev/null || true)"
|
|
if [ -n "$old_pid" ] && kill -0 "$old_pid" 2>/dev/null; then
|
|
log "stopping stale port-forward from a previous run (pid $old_pid)"
|
|
kill "$old_pid" 2>/dev/null || true
|
|
sleep 1
|
|
fi
|
|
rm -f "$PF_PIDFILE"
|
|
fi
|
|
|
|
log "starting port-forward svc/terdut-operator-demo ${LOCAL_PORT}:8080"
|
|
kubectl -n "$NAMESPACE" port-forward svc/terdut-operator-demo "${LOCAL_PORT}:8080" \
|
|
>"$PF_LOGFILE" 2>&1 &
|
|
STARTED_PF_PID=$!
|
|
echo "$STARTED_PF_PID" > "$PF_PIDFILE"
|
|
|
|
local tries=0
|
|
until curl -sf -o /dev/null "http://localhost:${LOCAL_PORT}/healthz"; do
|
|
tries=$((tries + 1))
|
|
if [ "$tries" -ge 30 ]; then
|
|
die "port-forward never became ready -- see $PF_LOGFILE"
|
|
fi
|
|
sleep 1
|
|
done
|
|
log "port-forward ready (pid $STARTED_PF_PID, log $PF_LOGFILE)"
|
|
}
|
|
|
|
# Redeems Platform's own invite link -- minted by its TerdutTeam
|
|
# (02-team-platform.yaml's spec.invite.enabled, reconciled through that
|
|
# team's own already-working team-scoped credential, which requireTeamOwner
|
|
# already treats as owner-equivalent for /invites -- ratified, not a
|
|
# workaround, in terdut-server's SERVICE-ACCOUNTS.md) and redeemed through
|
|
# the ordinary signup endpoint. Invite redemption bypasses signup_mode
|
|
# entirely (terdut-server's internal/api/signup.go), so this needs no admin
|
|
# credential, no signup_mode flip, and no direct Postgres access at all --
|
|
# unlike an earlier version of this script, before terdut-operator grew
|
|
# this feature (see niklas/terdut-server#23).
|
|
redeem_platform_invite() {
|
|
log "reading Platform's invite link"
|
|
local secret_name invite_url invite_token tries=0
|
|
until secret_name="$(kubectl -n "$NAMESPACE" get terdutteam terdutteam-platform \
|
|
-o jsonpath='{.status.inviteSecretRef.name}' 2>/dev/null)" && [ -n "$secret_name" ]; do
|
|
tries=$((tries + 1))
|
|
[ "$tries" -lt 30 ] || die "terdutteam-platform never reported status.inviteSecretRef -- check spec.invite.enabled and kubectl describe it"
|
|
sleep 1
|
|
done
|
|
invite_url="$(kubectl -n "$NAMESPACE" get secret "$secret_name" -o jsonpath='{.data.url}' | base64 -d)"
|
|
invite_token="${invite_url##*invite=}"
|
|
[ -n "$invite_token" ] || die "could not parse an invite token out of $invite_url"
|
|
|
|
# A re-run: the invite was spent by the first run, and the server answers a
|
|
# spent invite with 403 before it ever looks at the username, so the 409
|
|
# handled below never arrives. If alice can already sign in, she exists.
|
|
local login_code
|
|
login_code="$(curl -sS -o /dev/null -w '%{http_code}' \
|
|
-X POST "${BASE_URL}/api/login" -H 'Content-Type: application/json' \
|
|
-d "$(jq -n --arg u "$ALICE_USERNAME" --arg p "$DEMO_PASSWORD" '{username: $u, password: $p}')")"
|
|
if [ "$login_code" = "200" ]; then
|
|
log "account ${ALICE_USERNAME} already exists and can sign in, skipping signup (re-run detected)"
|
|
return 0
|
|
fi
|
|
|
|
log "signing up ${ALICE_USERNAME} via Platform's invite"
|
|
local body resp_file code
|
|
body="$(jq -n \
|
|
--arg u "$ALICE_USERNAME" --arg e "$ALICE_EMAIL" \
|
|
--arg p "$DEMO_PASSWORD" --arg i "$invite_token" \
|
|
'{username: $u, email: $e, password: $p, invite: $i}')"
|
|
resp_file="$(mktemp)"
|
|
code="$(curl -sS -o "$resp_file" -w '%{http_code}' \
|
|
-X POST "${BASE_URL}/api/signup" -H 'Content-Type: application/json' -d "$body")"
|
|
|
|
case "$code" in
|
|
201) log "created local account ${ALICE_USERNAME} (joined Platform)" ;;
|
|
409) log "account ${ALICE_USERNAME} already exists, skipping (re-run detected)" ;;
|
|
*) die "signup failed (HTTP $code): $(cat "$resp_file")" ;;
|
|
esac
|
|
rm -f "$resp_file"
|
|
}
|
|
|
|
# redeem_platform_invite's signup already used up alice's one signup -- a
|
|
# second POST /api/signup would just 409 on the taken username, it doesn't
|
|
# join an existing account to another team. Payments is joined through the
|
|
# ordinary team-scoped member endpoint instead, using that team's own
|
|
# already-working team-scoped credential (owner-equivalent, same reach the
|
|
# invite route above relies on) and alice's user id resolved via
|
|
# GET /api/users -- the same lookup tdclient.GetUserByUsername does
|
|
# operator-side. The endpoint upserts on (team_id, user_id), so this is
|
|
# naturally idempotent across re-runs with no separate conflict handling
|
|
# needed.
|
|
join_payments_team() {
|
|
log "adding ${ALICE_USERNAME} to Payments"
|
|
local payments_id payments_secret payments_key alice_id resp_file code
|
|
payments_id="$(kubectl -n "$NAMESPACE" get terdutteam terdutteam-payments -o jsonpath='{.status.teamID}')"
|
|
[ -n "$payments_id" ] || die "could not read status.teamID off terdutteam-payments"
|
|
payments_secret="$(kubectl -n "$NAMESPACE" get terdutteam terdutteam-payments \
|
|
-o jsonpath='{.status.credentialsSecretRef.name}')"
|
|
[ -n "$payments_secret" ] || die "terdutteam-payments has no status.credentialsSecretRef yet"
|
|
payments_key="$(kubectl -n "$OPERATOR_NAMESPACE" get secret "$payments_secret" -o jsonpath='{.data.token}' | base64 -d)"
|
|
|
|
alice_id="$(curl -sS -H "Authorization: Bearer $payments_key" "${BASE_URL}/api/users" \
|
|
| jq -r --arg u "$ALICE_USERNAME" '.[] | select(.username == $u) | .id')"
|
|
[ -n "$alice_id" ] || die "could not resolve ${ALICE_USERNAME}'s user id via GET /api/users"
|
|
|
|
resp_file="$(mktemp)"
|
|
code="$(curl -sS -o "$resp_file" -w '%{http_code}' \
|
|
-X POST "${BASE_URL}/api/teams/${payments_id}/members" \
|
|
-H "Authorization: Bearer $payments_key" -H 'Content-Type: application/json' \
|
|
-d "$(jq -n --argjson id "$alice_id" '{user_id: $id, role: "member"}')")"
|
|
[ "$code" = "204" ] || die "adding ${ALICE_USERNAME} to Payments failed (HTTP $code): $(cat "$resp_file")"
|
|
log "added ${ALICE_USERNAME} to Payments"
|
|
rm -f "$resp_file"
|
|
}
|
|
|
|
fire_demo_alerts() {
|
|
log "firing representative demo alerts"
|
|
# Two clusters, so the queue shows the cluster chip and offers its filter.
|
|
# high-cpu fires in both: the same alert in two clusters is two incidents.
|
|
local fire="$SCRIPT_DIR/fire-alerts.sh"
|
|
CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu
|
|
CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu
|
|
CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform disk-full
|
|
CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" payments pod-crash
|
|
}
|
|
|
|
print_summary() {
|
|
cat <<EOF
|
|
|
|
terdut demo is up.
|
|
|
|
Web UI: http://localhost:${LOCAL_PORT}
|
|
Login: ${ALICE_USERNAME} / ${DEMO_PASSWORD}
|
|
Cluster: kind-${CLUSTER_NAME}
|
|
Namespace: ${NAMESPACE}
|
|
Port-forward pid: ${STARTED_PF_PID:-$(cat "$PF_PIDFILE" 2>/dev/null || echo unknown)} (log: ${PF_LOGFILE})
|
|
stop it with: kill \$(cat ${PF_PIDFILE})
|
|
|
|
Fire more alerts:
|
|
export NAMESPACE=${NAMESPACE} BASE_URL=${BASE_URL}
|
|
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu
|
|
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu resolve
|
|
./fire-alerts.sh payments heartbeat # send repeatedly (e.g. every minute)
|
|
# to keep a dead man's switch alive;
|
|
# stop sending it and, 15 minutes
|
|
# later, terdut-server opens a
|
|
# critical incident on its own.
|
|
|
|
Tear down:
|
|
$0 --teardown
|
|
# add TEARDOWN_CLUSTER=true to also delete the kind cluster itself
|
|
EOF
|
|
}
|
|
|
|
teardown() {
|
|
if [ -f "$PF_PIDFILE" ]; then
|
|
local pid
|
|
pid="$(cat "$PF_PIDFILE" 2>/dev/null || true)"
|
|
if [ -n "$pid" ] && kill -0 "$pid" 2>/dev/null; then
|
|
log "stopping port-forward (pid $pid)"
|
|
kill "$pid" 2>/dev/null || true
|
|
fi
|
|
rm -f "$PF_PIDFILE"
|
|
fi
|
|
|
|
log "deleting namespace $NAMESPACE"
|
|
kubectl delete namespace "$NAMESPACE" --ignore-not-found --wait=true --timeout "$WAIT_TIMEOUT"
|
|
|
|
if [ "$TEARDOWN_CLUSTER" = "true" ]; then
|
|
log "deleting kind cluster $CLUSTER_NAME"
|
|
kind delete cluster --name "$CLUSTER_NAME"
|
|
else
|
|
log "leaving kind cluster '$CLUSTER_NAME' and the operator install in place" \
|
|
"(set TEARDOWN_CLUSTER=true to also delete the cluster)"
|
|
fi
|
|
}
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# main
|
|
# ---------------------------------------------------------------------------
|
|
|
|
main() {
|
|
case "${1:-}" in
|
|
--teardown)
|
|
preflight
|
|
teardown
|
|
trap - EXIT
|
|
exit 0
|
|
;;
|
|
--help|-h)
|
|
usage
|
|
trap - EXIT
|
|
exit 0
|
|
;;
|
|
"") ;;
|
|
*)
|
|
usage
|
|
die "unknown argument: $1"
|
|
;;
|
|
esac
|
|
|
|
preflight
|
|
ensure_kind_cluster
|
|
install_operator
|
|
apply_demo
|
|
wait_for_teams_ready
|
|
start_port_forward
|
|
redeem_platform_invite
|
|
join_payments_team
|
|
wait_for_remaining_ready
|
|
fire_demo_alerts
|
|
print_summary
|
|
|
|
# Success: leave the port-forward running, don't let the EXIT trap kill it.
|
|
trap - EXIT
|
|
}
|
|
|
|
main "$@"
|