diff --git a/examples/demo/01-server.yaml b/examples/demo/01-server.yaml index 901784d..af28d0f 100644 --- a/examples/demo/01-server.yaml +++ b/examples/demo/01-server.yaml @@ -14,13 +14,22 @@ metadata: spec: image: repository: git.ryuvia.com/niklas/terdut-server - # v0.34.0: fixes callerMayManageServiceAccount so an instance-scoped - # service account can adopt/rotate a key on a team-scoped account it - # didn't just create in the same call -- without this, terdutteam-* - # can wedge permanently on exactly the crash-window race this demo - # hit live (niklas/terdut-operator#3). - tag: v0.34.0 - replicas: 1 + # v0.36.0 is the floor now that replicas below is 2 (this demo pins + # the current release, v0.43.0, so it shows the current web UI too): that + # release put the sweeper, the notifier and the migration runner each + # behind a Postgres advisory lock, and gave incident creation its own + # conflict resolution, which is what makes a second replica safe + # instead of racing the first. (Still carries v0.34.0's fix too -- + # callerMayManageServiceAccount, so an instance-scoped service account + # can adopt/rotate a key on a team-scoped account it didn't just create + # in the same call -- without which terdutteam-* can wedge permanently + # on the crash-window race this demo hit live, niklas/terdut-operator#3.) + tag: v0.43.0 + # Matches this CRD's own spec.replicas default (v0.4.0) -- stated + # explicitly, like every other field in this file, rather than left to + # the default. RollingUpdate follows automatically; this operator does + # not expose Strategy as a spec field. + replicas: 2 networking: hostname: terdut-operator-demo.example servicePort: 8080 diff --git a/examples/demo/README.md b/examples/demo/README.md index 0829e5f..c91e83b 100644 --- a/examples/demo/README.md +++ b/examples/demo/README.md @@ -133,6 +133,20 @@ Each `(team, scenario)` pair is one stable fingerprint, so firing the same one twice updates the same alert (a real re-fire) and `resolve` closes exactly that one. +Set `CLUSTER` to send the alert as if it came from one of several clusters: + +```sh +CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu +CLUSTER=prod-us ./fire-alerts.sh platform high-cpu # a second incident, not a join +CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu resolve +``` + +It stands in for a Prometheus external label plus `cluster` in Alertmanager's +`group_by` (terdut-server's README, "Several clusters, one team"): the web UI +then shows the cluster chip on each incident and a cluster filter in the +queue. `CLUSTER` is part of the fingerprint, so resolve with the same value you +fired with. `./run-demo.sh` fires its alerts across `prod-eu` and `prod-us`. + ### Dead man's switches `06-deadman-platform.yaml` / `07-deadman-payments.yaml` expect a heartbeat diff --git a/examples/demo/fire-alerts.sh b/examples/demo/fire-alerts.sh index b8a8ac9..7af76b7 100755 --- a/examples/demo/fire-alerts.sh +++ b/examples/demo/fire-alerts.sh @@ -16,7 +16,15 @@ # is the one part of that URL still usable here. # # Usage: -# ./fire-alerts.sh [resolve] +# [CLUSTER=prod-eu] ./fire-alerts.sh [resolve] +# +# CLUSTER stands in for a Prometheus externalLabel plus `cluster` in +# Alertmanager's group_by (terdut-server's README, "Several clusters, one +# team"): it is put on the alert's labels and on groupLabels, so the incident +# carries it and the web UI shows the cluster chip and the queue's cluster +# filter. It is also part of the group key and the fingerprint, which is what +# keeps the same alert in two clusters from joining one incident. Unset, the +# alert is sent exactly as before. # # Prerequisites: kubectl context pointed at the demo namespace, jq, curl, # and (in another terminal) a running: @@ -24,6 +32,7 @@ set -euo pipefail NAMESPACE="${NAMESPACE:-}" +CLUSTER="${CLUSTER:-}" BASE_URL="${BASE_URL:-http://localhost:8080}" usage() { @@ -45,6 +54,8 @@ scenarios: env vars: NAMESPACE kubectl -n for reading the webhook Secret (required) BASE_URL where the port-forwarded terdut-server is (default http://localhost:8080) + CLUSTER optional cluster name, e.g. prod-eu: sent as a `cluster` label and + group label, so the UI shows where the incident came from EOF exit 1 } @@ -90,7 +101,7 @@ key="$(kubectl -n "$NAMESPACE" get secret "$secret_name" -o jsonpath='{.data.key # created: terdut-server correlates on (team_id, fingerprint), not on # anything else in the payload. Real Alertmanager computes this from the # alert's label set; a fixed string plays the same role here. -fingerprint="demo-${team}-${scenario}" +fingerprint="demo-${team}-${scenario}${CLUSTER:+-$CLUSTER}" now="$(date -u +%Y-%m-%dT%H:%M:%SZ)" if [ "$status" = firing ]; then @@ -101,7 +112,8 @@ fi payload="$(jq -n \ --arg status "$status" \ - --arg groupKey "demo:${team}:${scenario}" \ + --arg groupKey "demo:${team}:${scenario}${CLUSTER:+:$CLUSTER}" \ + --arg cluster "$CLUSTER" \ --arg alertname "$alertname" \ --arg team "$team" \ --arg severity "$severity" \ @@ -113,10 +125,10 @@ payload="$(jq -n \ version: "4", status: $status, groupKey: $groupKey, - groupLabels: { alertname: $alertname, team: $team }, + groupLabels: ({ alertname: $alertname, team: $team } + (if $cluster != "" then { cluster: $cluster } else {} end)), alerts: [{ status: $status, - labels: { alertname: $alertname, severity: $severity, team: $team, instance: "demo" }, + labels: ({ alertname: $alertname, severity: $severity, team: $team, instance: "demo" } + (if $cluster != "" then { cluster: $cluster } else {} end)), annotations: { summary: $summary }, startsAt: $startsAt, endsAt: $endsAt, @@ -126,7 +138,7 @@ payload="$(jq -n \ }')" url="${BASE_URL}/api/integrations/${key}/alertmanager" -echo "POST $url (team=$team scenario=$scenario status=$status)" >&2 +echo "POST $url (team=$team scenario=$scenario status=$status${CLUSTER:+ cluster=$CLUSTER})" >&2 code="$(curl -sS -o /tmp/fire-alerts-response.json -w '%{http_code}' \ -X POST "$url" -H 'Content-Type: application/json' -d "$payload")" echo "-> HTTP $code" >&2 diff --git a/examples/demo/run-demo.sh b/examples/demo/run-demo.sh index e589f13..017bb7e 100755 --- a/examples/demo/run-demo.sh +++ b/examples/demo/run-demo.sh @@ -107,26 +107,41 @@ apply_demo() { kubectl apply -n "$NAMESPACE" -k "$SCRIPT_DIR" >/dev/null } -wait_for_ready() { - local objects=( - "terdutserver/terdut-operator-demo" - "terdutteam/terdutteam-platform" - "terdutteam/terdutteam-payments" - "terdutescalationrule/terdutescalationrule-platform" - "terdutescalationrule/terdutescalationrule-payments" - "terdutdeadmanswitch/terdutdeadmanswitch-platform" - "terdutdeadmanswitch/terdutdeadmanswitch-payments" - "terdutalertsource/terdutalertsource-platform" - "terdutalertsource/terdutalertsource-payments" - ) +wait_for_objects() { local obj - for obj in "${objects[@]}"; do + for obj in "$@"; do log "waiting for $obj to become Ready" kubectl wait --for=condition=Ready --timeout "$WAIT_TIMEOUT" -n "$NAMESPACE" "$obj" >/dev/null \ || die "timed out waiting for $obj -- try: kubectl describe -n $NAMESPACE $obj" done } +# Just the server and the two teams -- everything redeem_platform_invite and +# join_payments_team need. Deliberately NOT the escalation rules here: this +# demo kit's own terdutescalationrule-platform names alice as a level-1 +# target, and that CR cannot reach Ready until alice actually exists +# (terdut-server resolves every named username at reconcile time, not just +# at escalation time) -- a real dependency this script has to satisfy by +# creating her first, not something kubectl wait can be told to ignore. +wait_for_teams_ready() { + wait_for_objects \ + "terdutserver/terdut-operator-demo" \ + "terdutteam/terdutteam-platform" \ + "terdutteam/terdutteam-payments" +} + +# Everything that was waiting on alice (or just on the teams above, now +# already satisfied) to exist. +wait_for_remaining_ready() { + wait_for_objects \ + "terdutescalationrule/terdutescalationrule-platform" \ + "terdutescalationrule/terdutescalationrule-payments" \ + "terdutdeadmanswitch/terdutdeadmanswitch-platform" \ + "terdutdeadmanswitch/terdutdeadmanswitch-payments" \ + "terdutalertsource/terdutalertsource-platform" \ + "terdutalertsource/terdutalertsource-payments" +} + start_port_forward() { # A stale pidfile from an earlier run would otherwise collide with us on # $LOCAL_PORT -- if that pid is still alive, stop it first. @@ -181,6 +196,18 @@ redeem_platform_invite() { invite_token="${invite_url##*invite=}" [ -n "$invite_token" ] || die "could not parse an invite token out of $invite_url" + # A re-run: the invite was spent by the first run, and the server answers a + # spent invite with 403 before it ever looks at the username, so the 409 + # handled below never arrives. If alice can already sign in, she exists. + local login_code + login_code="$(curl -sS -o /dev/null -w '%{http_code}' \ + -X POST "${BASE_URL}/api/login" -H 'Content-Type: application/json' \ + -d "$(jq -n --arg u "$ALICE_USERNAME" --arg p "$DEMO_PASSWORD" '{username: $u, password: $p}')")" + if [ "$login_code" = "200" ]; then + log "account ${ALICE_USERNAME} already exists and can sign in, skipping signup (re-run detected)" + return 0 + fi + log "signing up ${ALICE_USERNAME} via Platform's invite" local body resp_file code body="$(jq -n \ @@ -235,9 +262,13 @@ join_payments_team() { fire_demo_alerts() { log "firing representative demo alerts" - NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform high-cpu - NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform disk-full - NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" payments pod-crash + # Two clusters, so the queue shows the cluster chip and offers its filter. + # high-cpu fires in both: the same alert in two clusters is two incidents. + local fire="$SCRIPT_DIR/fire-alerts.sh" + CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu + CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu + CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform disk-full + CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" payments pod-crash } print_summary() { @@ -254,8 +285,8 @@ terdut demo is up. Fire more alerts: export NAMESPACE=${NAMESPACE} BASE_URL=${BASE_URL} - ./fire-alerts.sh platform high-cpu - ./fire-alerts.sh platform high-cpu resolve + CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu + CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu resolve ./fire-alerts.sh payments heartbeat # send repeatedly (e.g. every minute) # to keep a dead man's switch alive; # stop sending it and, 15 minutes @@ -319,10 +350,11 @@ main() { ensure_kind_cluster install_operator apply_demo - wait_for_ready + wait_for_teams_ready start_port_forward redeem_platform_invite join_payments_team + wait_for_remaining_ready fire_demo_alerts print_summary