e1103f2b7d
Credentials: the TerdutServer controller generates <name>-operator-key in the server's own namespace (owned by it) and hands it to the pods as TERDUT_OPERATOR_KEY; the server creates its instance-scoped account from it at every start. A replaced Secret rolls the pods. The bootstrap handshake, the checkpoint Secret, per-team service accounts and credentials Secrets, BootstrapStateLost and credentials.deletionPolicy are gone. CRDs: TerdutServer, TerdutTeam and TerdutAlertSource. TerdutEscalationRule and TerdutDeadmanSwitch become spec.escalation and spec.deadmanSwitches[] on the team (matched by name, extras removed); team invites are removed. A team is created under the identity <namespace>/<name> (external_id), so a retry, a lost status or a deleted team heal by repeating the same call, and a display name owned by another team is TeamNameTaken instead of an adoption. The server resolves escalation usernames (UnknownUser condition). OIDC claim names and trustEmail are spec fields. Fixes: query values are URL-escaped; every delete treats 404 as success; deleting a team no longer depends on allowedTeams consent; a switch or integration deleted on the server is recreated; unnamed switches take the CR's name. Cleanup: scaffold e2e test, AGENTS.md, devcontainer, unused config/ pieces and Client.Version() removed; DESIGN.md, README, ROADMAP and the demo (run-demo.sh, manifests) rewritten for the new design. Secret RBAC stays cluster-wide, now stated in DESIGN.md section 9. Claude-Session: https://claude.ai/code/session_016mBLURvJoMuUEr9cB2RpUN
318 lines
11 KiB
Bash
Executable File
318 lines
11 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# Stands up the complete terdut demo (terdut-operator + every CRD kind this
|
|
# repo ships + a working local login + a few synthetic incidents) on a kind
|
|
# cluster, fully automated. Password login only -- this demo kit's own
|
|
# 01-server.yaml carries no oidc: block at all, so there is nothing to
|
|
# disable; OIDC is simply absent.
|
|
#
|
|
# Usage:
|
|
# ./run-demo.sh deploy the whole demo (idempotent: safe to
|
|
# re-run against a cluster that already has it)
|
|
# ./run-demo.sh --teardown delete the demo namespace (and, if
|
|
# TEARDOWN_CLUSTER=true, the kind cluster too)
|
|
# ./run-demo.sh --help
|
|
#
|
|
# All of the defaults below are overridable as environment variables.
|
|
set -euo pipefail
|
|
|
|
CLUSTER_NAME="${CLUSTER_NAME:-terdut-demo}"
|
|
NAMESPACE="${NAMESPACE:-terdut-operator-demo}"
|
|
OPERATOR_NAMESPACE="${OPERATOR_NAMESPACE:-terdut-operator-system}"
|
|
RELEASE_NAME="${RELEASE_NAME:-terdut-operator}"
|
|
|
|
ALICE_USERNAME="${ALICE_USERNAME:-alice}"
|
|
ALICE_EMAIL="${ALICE_EMAIL:-alice@terdut-demo.local}"
|
|
DEMO_PASSWORD="${DEMO_PASSWORD:-terdut-demo-1234}"
|
|
|
|
BASE_URL="${BASE_URL:-http://localhost:8080}"
|
|
LOCAL_PORT="${LOCAL_PORT:-8080}"
|
|
|
|
HELM_TIMEOUT="${HELM_TIMEOUT:-180s}"
|
|
WAIT_TIMEOUT="${WAIT_TIMEOUT:-180s}"
|
|
|
|
TEARDOWN_CLUSTER="${TEARDOWN_CLUSTER:-false}"
|
|
|
|
PF_PIDFILE="${PF_PIDFILE:-/tmp/terdut-demo-port-forward.pid}"
|
|
PF_LOGFILE="${PF_LOGFILE:-/tmp/terdut-demo-port-forward.log}"
|
|
|
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
CHART_DIR="$(cd "$SCRIPT_DIR/../.." && pwd)/charts/terdut-operator"
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# helpers
|
|
# ---------------------------------------------------------------------------
|
|
|
|
log() { printf '[run-demo] %s\n' "$*" >&2; }
|
|
die() { printf '[run-demo] FAILED: %s\n' "$*" >&2; exit 1; }
|
|
|
|
usage() {
|
|
cat <<EOF
|
|
Usage: $0 [--teardown|--help]
|
|
|
|
Deploys (or tears down) the complete terdut demo on a kind cluster.
|
|
See the top of this file for every overridable environment variable.
|
|
EOF
|
|
}
|
|
|
|
# Only kills a port-forward THIS invocation started, so a successful run
|
|
# never has its background job reaped by its own exit trap.
|
|
STARTED_PF_PID=""
|
|
cleanup_on_failure() {
|
|
local rc=$?
|
|
if [ "$rc" -ne 0 ] && [ -n "$STARTED_PF_PID" ]; then
|
|
log "run failed -- stopping the port-forward it started (pid $STARTED_PF_PID)"
|
|
kill "$STARTED_PF_PID" 2>/dev/null || true
|
|
fi
|
|
exit "$rc"
|
|
}
|
|
trap cleanup_on_failure EXIT
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# steps
|
|
# ---------------------------------------------------------------------------
|
|
|
|
preflight() {
|
|
local missing=()
|
|
for bin in kubectl kind helm jq curl; do
|
|
command -v "$bin" >/dev/null 2>&1 || missing+=("$bin")
|
|
done
|
|
if [ "${#missing[@]}" -gt 0 ]; then
|
|
die "missing required tools on PATH: ${missing[*]}"
|
|
fi
|
|
}
|
|
|
|
ensure_kind_cluster() {
|
|
case "$(kind get clusters 2>/dev/null)" in
|
|
*"$CLUSTER_NAME"*)
|
|
log "kind cluster '$CLUSTER_NAME' already exists, skipping creation" ;;
|
|
*)
|
|
log "creating kind cluster '$CLUSTER_NAME'"
|
|
kind create cluster --name "$CLUSTER_NAME" ;;
|
|
esac
|
|
kubectl config use-context "kind-${CLUSTER_NAME}" >/dev/null
|
|
}
|
|
|
|
install_operator() {
|
|
log "installing terdut-operator into namespace $OPERATOR_NAMESPACE"
|
|
helm upgrade --install "$RELEASE_NAME" "$CHART_DIR" \
|
|
--namespace "$OPERATOR_NAMESPACE" --create-namespace \
|
|
--wait --timeout "$HELM_TIMEOUT" \
|
|
|| die "helm install of terdut-operator did not become ready"
|
|
}
|
|
|
|
apply_demo() {
|
|
log "creating namespace $NAMESPACE"
|
|
kubectl create namespace "$NAMESPACE" --dry-run=client -o yaml | kubectl apply -f - >/dev/null
|
|
log "applying demo CRs (00-05) into $NAMESPACE"
|
|
kubectl apply -n "$NAMESPACE" -k "$SCRIPT_DIR" >/dev/null
|
|
}
|
|
|
|
wait_for_objects() {
|
|
local obj
|
|
for obj in "$@"; do
|
|
log "waiting for $obj to become Ready"
|
|
kubectl wait --for=condition=Ready --timeout "$WAIT_TIMEOUT" -n "$NAMESPACE" "$obj" >/dev/null \
|
|
|| die "timed out waiting for $obj -- try: kubectl describe -n $NAMESPACE $obj"
|
|
done
|
|
}
|
|
|
|
# Only the server: the teams cannot reach Ready until alice exists (their
|
|
# escalation ladders name her, and terdut-server resolves usernames when the
|
|
# ladder is applied), and alice is created below, through the server itself.
|
|
wait_for_server_ready() {
|
|
wait_for_objects "terdutserver/terdut-operator-demo"
|
|
}
|
|
|
|
# Everything that was waiting on alice to exist.
|
|
wait_for_remaining_ready() {
|
|
wait_for_objects \
|
|
"terdutteam/terdutteam-platform" \
|
|
"terdutteam/terdutteam-payments" \
|
|
"terdutalertsource/terdutalertsource-platform" \
|
|
"terdutalertsource/terdutalertsource-payments"
|
|
}
|
|
|
|
start_port_forward() {
|
|
# A stale pidfile from an earlier run would otherwise collide with us on
|
|
# $LOCAL_PORT -- if that pid is still alive, stop it first.
|
|
if [ -f "$PF_PIDFILE" ]; then
|
|
local old_pid
|
|
old_pid="$(cat "$PF_PIDFILE" 2>/dev/null || true)"
|
|
if [ -n "$old_pid" ] && kill -0 "$old_pid" 2>/dev/null; then
|
|
log "stopping stale port-forward from a previous run (pid $old_pid)"
|
|
kill "$old_pid" 2>/dev/null || true
|
|
sleep 1
|
|
fi
|
|
rm -f "$PF_PIDFILE"
|
|
fi
|
|
|
|
log "starting port-forward svc/terdut-operator-demo ${LOCAL_PORT}:8080"
|
|
kubectl -n "$NAMESPACE" port-forward svc/terdut-operator-demo "${LOCAL_PORT}:8080" \
|
|
>"$PF_LOGFILE" 2>&1 &
|
|
STARTED_PF_PID=$!
|
|
echo "$STARTED_PF_PID" > "$PF_PIDFILE"
|
|
|
|
local tries=0
|
|
until curl -sf -o /dev/null "http://localhost:${LOCAL_PORT}/healthz"; do
|
|
tries=$((tries + 1))
|
|
if [ "$tries" -ge 30 ]; then
|
|
die "port-forward never became ready -- see $PF_LOGFILE"
|
|
fi
|
|
sleep 1
|
|
done
|
|
log "port-forward ready (pid $STARTED_PF_PID, log $PF_LOGFILE)"
|
|
}
|
|
|
|
# Creates alice as the install's first user and administrator through
|
|
# /api/bootstrap, which is open until a first user exists -- the operator
|
|
# authenticates with its own seeded key and never uses it, so this is how a
|
|
# person gets in. An administrator may manage any team, so alice's own key is
|
|
# enough to add her to both; the teams' own status.teamID is set as soon as
|
|
# the operator has created each team, even while they still wait on her.
|
|
create_alice_and_join_teams() {
|
|
# A re-run: she exists already (and was added to the teams the first time).
|
|
local login_code
|
|
login_code="$(curl -sS -o /dev/null -w '%{http_code}' \
|
|
-X POST "${BASE_URL}/api/login" -H 'Content-Type: application/json' \
|
|
-d "$(jq -n --arg u "$ALICE_USERNAME" --arg p "$DEMO_PASSWORD" '{username: $u, password: $p}')")"
|
|
if [ "$login_code" = "200" ]; then
|
|
log "account ${ALICE_USERNAME} already exists and can sign in, skipping creation (re-run detected)"
|
|
return 0
|
|
fi
|
|
|
|
log "creating ${ALICE_USERNAME} as the first user (administrator)"
|
|
local resp_file code alice_key
|
|
resp_file="$(mktemp)"
|
|
code="$(curl -sS -o "$resp_file" -w '%{http_code}' \
|
|
-X POST "${BASE_URL}/api/bootstrap" -H 'Content-Type: application/json' \
|
|
-d "$(jq -n --arg u "$ALICE_USERNAME" --arg e "$ALICE_EMAIL" --arg p "$DEMO_PASSWORD" \
|
|
'{username: $u, email: $e, password: $p}')")"
|
|
[ "$code" = "201" ] || die "bootstrap failed (HTTP $code): $(cat "$resp_file")"
|
|
alice_key="$(jq -r '.api_key.key' "$resp_file")"
|
|
rm -f "$resp_file"
|
|
|
|
local team team_id
|
|
for team in terdutteam-platform terdutteam-payments; do
|
|
team_id=""
|
|
local tries=0
|
|
until team_id="$(kubectl -n "$NAMESPACE" get terdutteam "$team" -o jsonpath='{.status.teamID}' 2>/dev/null)" \
|
|
&& [ -n "$team_id" ]; do
|
|
tries=$((tries + 1))
|
|
[ "$tries" -lt 60 ] || die "$team never reported status.teamID -- kubectl describe it"
|
|
sleep 1
|
|
done
|
|
local alice_id
|
|
alice_id="$(curl -sS -H "Authorization: Bearer $alice_key" "${BASE_URL}/api/users" \
|
|
| jq -r --arg u "$ALICE_USERNAME" '.[] | select(.username == $u) | .id')"
|
|
code="$(curl -sS -o /dev/null -w '%{http_code}' \
|
|
-X POST "${BASE_URL}/api/teams/${team_id}/members" \
|
|
-H "Authorization: Bearer $alice_key" -H 'Content-Type: application/json' \
|
|
-d "$(jq -n --argjson id "$alice_id" '{user_id: $id, role: "owner"}')")"
|
|
[ "$code" = "204" ] || die "adding ${ALICE_USERNAME} to ${team} failed (HTTP $code)"
|
|
log "added ${ALICE_USERNAME} to ${team}"
|
|
done
|
|
}
|
|
|
|
fire_demo_alerts() {
|
|
log "firing representative demo alerts"
|
|
# Two clusters, so the queue shows the cluster chip and offers its filter.
|
|
# high-cpu fires in both: the same alert in two clusters is two incidents.
|
|
local fire="$SCRIPT_DIR/fire-alerts.sh"
|
|
CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu
|
|
CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu
|
|
CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform disk-full
|
|
CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" payments pod-crash
|
|
}
|
|
|
|
print_summary() {
|
|
cat <<EOF
|
|
|
|
terdut demo is up.
|
|
|
|
Web UI: http://localhost:${LOCAL_PORT}
|
|
Login: ${ALICE_USERNAME} / ${DEMO_PASSWORD}
|
|
Cluster: kind-${CLUSTER_NAME}
|
|
Namespace: ${NAMESPACE}
|
|
Port-forward pid: ${STARTED_PF_PID:-$(cat "$PF_PIDFILE" 2>/dev/null || echo unknown)} (log: ${PF_LOGFILE})
|
|
stop it with: kill \$(cat ${PF_PIDFILE})
|
|
|
|
Fire more alerts:
|
|
export NAMESPACE=${NAMESPACE} BASE_URL=${BASE_URL}
|
|
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu
|
|
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu resolve
|
|
./fire-alerts.sh payments heartbeat # send repeatedly (e.g. every minute)
|
|
# to keep a dead man's switch alive;
|
|
# stop sending it and, 15 minutes
|
|
# later, terdut-server opens a
|
|
# critical incident on its own.
|
|
|
|
Tear down:
|
|
$0 --teardown
|
|
# add TEARDOWN_CLUSTER=true to also delete the kind cluster itself
|
|
EOF
|
|
}
|
|
|
|
teardown() {
|
|
if [ -f "$PF_PIDFILE" ]; then
|
|
local pid
|
|
pid="$(cat "$PF_PIDFILE" 2>/dev/null || true)"
|
|
if [ -n "$pid" ] && kill -0 "$pid" 2>/dev/null; then
|
|
log "stopping port-forward (pid $pid)"
|
|
kill "$pid" 2>/dev/null || true
|
|
fi
|
|
rm -f "$PF_PIDFILE"
|
|
fi
|
|
|
|
log "deleting namespace $NAMESPACE"
|
|
kubectl delete namespace "$NAMESPACE" --ignore-not-found --wait=true --timeout "$WAIT_TIMEOUT"
|
|
|
|
if [ "$TEARDOWN_CLUSTER" = "true" ]; then
|
|
log "deleting kind cluster $CLUSTER_NAME"
|
|
kind delete cluster --name "$CLUSTER_NAME"
|
|
else
|
|
log "leaving kind cluster '$CLUSTER_NAME' and the operator install in place" \
|
|
"(set TEARDOWN_CLUSTER=true to also delete the cluster)"
|
|
fi
|
|
}
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# main
|
|
# ---------------------------------------------------------------------------
|
|
|
|
main() {
|
|
case "${1:-}" in
|
|
--teardown)
|
|
preflight
|
|
teardown
|
|
trap - EXIT
|
|
exit 0
|
|
;;
|
|
--help|-h)
|
|
usage
|
|
trap - EXIT
|
|
exit 0
|
|
;;
|
|
"") ;;
|
|
*)
|
|
usage
|
|
die "unknown argument: $1"
|
|
;;
|
|
esac
|
|
|
|
preflight
|
|
ensure_kind_cluster
|
|
install_operator
|
|
apply_demo
|
|
wait_for_server_ready
|
|
start_port_forward
|
|
create_alice_and_join_teams
|
|
wait_for_remaining_ready
|
|
fire_demo_alerts
|
|
print_summary
|
|
|
|
# Success: leave the port-forward running, don't let the EXIT trap kill it.
|
|
trap - EXIT
|
|
}
|
|
|
|
main "$@"
|