diff --git a/examples/demo/01-server.yaml b/examples/demo/01-server.yaml index e877a31..af28d0f 100644 --- a/examples/demo/01-server.yaml +++ b/examples/demo/01-server.yaml @@ -14,7 +14,8 @@ metadata: spec: image: repository: git.ryuvia.com/niklas/terdut-server - # v0.36.0 or newer is required now that replicas below is 2: that + # v0.36.0 is the floor now that replicas below is 2 (this demo pins + # the current release, v0.43.0, so it shows the current web UI too): that # release put the sweeper, the notifier and the migration runner each # behind a Postgres advisory lock, and gave incident creation its own # conflict resolution, which is what makes a second replica safe @@ -23,7 +24,7 @@ spec: # can adopt/rotate a key on a team-scoped account it didn't just create # in the same call -- without which terdutteam-* can wedge permanently # on the crash-window race this demo hit live, niklas/terdut-operator#3.) - tag: v0.36.0 + tag: v0.43.0 # Matches this CRD's own spec.replicas default (v0.4.0) -- stated # explicitly, like every other field in this file, rather than left to # the default. RollingUpdate follows automatically; this operator does diff --git a/examples/demo/README.md b/examples/demo/README.md index 0829e5f..c91e83b 100644 --- a/examples/demo/README.md +++ b/examples/demo/README.md @@ -133,6 +133,20 @@ Each `(team, scenario)` pair is one stable fingerprint, so firing the same one twice updates the same alert (a real re-fire) and `resolve` closes exactly that one. +Set `CLUSTER` to send the alert as if it came from one of several clusters: + +```sh +CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu +CLUSTER=prod-us ./fire-alerts.sh platform high-cpu # a second incident, not a join +CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu resolve +``` + +It stands in for a Prometheus external label plus `cluster` in Alertmanager's +`group_by` (terdut-server's README, "Several clusters, one team"): the web UI +then shows the cluster chip on each incident and a cluster filter in the +queue. `CLUSTER` is part of the fingerprint, so resolve with the same value you +fired with. `./run-demo.sh` fires its alerts across `prod-eu` and `prod-us`. + ### Dead man's switches `06-deadman-platform.yaml` / `07-deadman-payments.yaml` expect a heartbeat diff --git a/examples/demo/fire-alerts.sh b/examples/demo/fire-alerts.sh index b8a8ac9..7af76b7 100755 --- a/examples/demo/fire-alerts.sh +++ b/examples/demo/fire-alerts.sh @@ -16,7 +16,15 @@ # is the one part of that URL still usable here. # # Usage: -# ./fire-alerts.sh [resolve] +# [CLUSTER=prod-eu] ./fire-alerts.sh [resolve] +# +# CLUSTER stands in for a Prometheus externalLabel plus `cluster` in +# Alertmanager's group_by (terdut-server's README, "Several clusters, one +# team"): it is put on the alert's labels and on groupLabels, so the incident +# carries it and the web UI shows the cluster chip and the queue's cluster +# filter. It is also part of the group key and the fingerprint, which is what +# keeps the same alert in two clusters from joining one incident. Unset, the +# alert is sent exactly as before. # # Prerequisites: kubectl context pointed at the demo namespace, jq, curl, # and (in another terminal) a running: @@ -24,6 +32,7 @@ set -euo pipefail NAMESPACE="${NAMESPACE:-}" +CLUSTER="${CLUSTER:-}" BASE_URL="${BASE_URL:-http://localhost:8080}" usage() { @@ -45,6 +54,8 @@ scenarios: env vars: NAMESPACE kubectl -n for reading the webhook Secret (required) BASE_URL where the port-forwarded terdut-server is (default http://localhost:8080) + CLUSTER optional cluster name, e.g. prod-eu: sent as a `cluster` label and + group label, so the UI shows where the incident came from EOF exit 1 } @@ -90,7 +101,7 @@ key="$(kubectl -n "$NAMESPACE" get secret "$secret_name" -o jsonpath='{.data.key # created: terdut-server correlates on (team_id, fingerprint), not on # anything else in the payload. Real Alertmanager computes this from the # alert's label set; a fixed string plays the same role here. -fingerprint="demo-${team}-${scenario}" +fingerprint="demo-${team}-${scenario}${CLUSTER:+-$CLUSTER}" now="$(date -u +%Y-%m-%dT%H:%M:%SZ)" if [ "$status" = firing ]; then @@ -101,7 +112,8 @@ fi payload="$(jq -n \ --arg status "$status" \ - --arg groupKey "demo:${team}:${scenario}" \ + --arg groupKey "demo:${team}:${scenario}${CLUSTER:+:$CLUSTER}" \ + --arg cluster "$CLUSTER" \ --arg alertname "$alertname" \ --arg team "$team" \ --arg severity "$severity" \ @@ -113,10 +125,10 @@ payload="$(jq -n \ version: "4", status: $status, groupKey: $groupKey, - groupLabels: { alertname: $alertname, team: $team }, + groupLabels: ({ alertname: $alertname, team: $team } + (if $cluster != "" then { cluster: $cluster } else {} end)), alerts: [{ status: $status, - labels: { alertname: $alertname, severity: $severity, team: $team, instance: "demo" }, + labels: ({ alertname: $alertname, severity: $severity, team: $team, instance: "demo" } + (if $cluster != "" then { cluster: $cluster } else {} end)), annotations: { summary: $summary }, startsAt: $startsAt, endsAt: $endsAt, @@ -126,7 +138,7 @@ payload="$(jq -n \ }')" url="${BASE_URL}/api/integrations/${key}/alertmanager" -echo "POST $url (team=$team scenario=$scenario status=$status)" >&2 +echo "POST $url (team=$team scenario=$scenario status=$status${CLUSTER:+ cluster=$CLUSTER})" >&2 code="$(curl -sS -o /tmp/fire-alerts-response.json -w '%{http_code}' \ -X POST "$url" -H 'Content-Type: application/json' -d "$payload")" echo "-> HTTP $code" >&2 diff --git a/examples/demo/run-demo.sh b/examples/demo/run-demo.sh index 013bf80..017bb7e 100755 --- a/examples/demo/run-demo.sh +++ b/examples/demo/run-demo.sh @@ -196,6 +196,18 @@ redeem_platform_invite() { invite_token="${invite_url##*invite=}" [ -n "$invite_token" ] || die "could not parse an invite token out of $invite_url" + # A re-run: the invite was spent by the first run, and the server answers a + # spent invite with 403 before it ever looks at the username, so the 409 + # handled below never arrives. If alice can already sign in, she exists. + local login_code + login_code="$(curl -sS -o /dev/null -w '%{http_code}' \ + -X POST "${BASE_URL}/api/login" -H 'Content-Type: application/json' \ + -d "$(jq -n --arg u "$ALICE_USERNAME" --arg p "$DEMO_PASSWORD" '{username: $u, password: $p}')")" + if [ "$login_code" = "200" ]; then + log "account ${ALICE_USERNAME} already exists and can sign in, skipping signup (re-run detected)" + return 0 + fi + log "signing up ${ALICE_USERNAME} via Platform's invite" local body resp_file code body="$(jq -n \ @@ -250,9 +262,13 @@ join_payments_team() { fire_demo_alerts() { log "firing representative demo alerts" - NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform high-cpu - NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform disk-full - NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" payments pod-crash + # Two clusters, so the queue shows the cluster chip and offers its filter. + # high-cpu fires in both: the same alert in two clusters is two incidents. + local fire="$SCRIPT_DIR/fire-alerts.sh" + CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu + CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu + CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform disk-full + CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" payments pod-crash } print_summary() { @@ -269,8 +285,8 @@ terdut demo is up. Fire more alerts: export NAMESPACE=${NAMESPACE} BASE_URL=${BASE_URL} - ./fire-alerts.sh platform high-cpu - ./fire-alerts.sh platform high-cpu resolve + CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu + CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu resolve ./fire-alerts.sh payments heartbeat # send repeatedly (e.g. every minute) # to keep a dead man's switch alive; # stop sending it and, 15 minutes