From 0ee7ede648ca83a6b45829d042e2839d065f7d1c Mon Sep 17 00:00:00 2001 From: Niklas Ye Date: Sat, 3 Oct 2026 18:13:09 +0200 Subject: [PATCH 1/3] examples/demo: match v0.4.0's new spec.replicas default and bump terdut-server replicas: 1 and tag: v0.34.0 were both correct when written, but the CRD's own default moved to 2 in v0.4.0 (same release this demo is meant to show off), and v0.34.0 predates v0.36.0's advisory locks that make a second replica safe instead of racing the first. Left as-is, the demo would have been the one place in this repo demonstrating the exact unsafe combination the CRD's own doc comment warns against: more than one replica against an image that doesn't guard the sweeper/notifier/migration-runner singletons. replicas is now stated explicitly as 2 rather than dropped to pick up the default silently, matching every other field in this file's own habit of spelling out what it depends on. tag moves to v0.36.0 specifically -- the first version where the lock landed -- with the comment keeping v0.34.0's original reasoning (the service-account race fix) alongside the new one, since v0.36.0 still carries that fix forward. Co-authored-by: Claude --- examples/demo/01-server.yaml | 22 +++++++++++++++------- 1 file changed, 15 insertions(+), 7 deletions(-) diff --git a/examples/demo/01-server.yaml b/examples/demo/01-server.yaml index 901784d..e877a31 100644 --- a/examples/demo/01-server.yaml +++ b/examples/demo/01-server.yaml @@ -14,13 +14,21 @@ metadata: spec: image: repository: git.ryuvia.com/niklas/terdut-server - # v0.34.0: fixes callerMayManageServiceAccount so an instance-scoped - # service account can adopt/rotate a key on a team-scoped account it - # didn't just create in the same call -- without this, terdutteam-* - # can wedge permanently on exactly the crash-window race this demo - # hit live (niklas/terdut-operator#3). - tag: v0.34.0 - replicas: 1 + # v0.36.0 or newer is required now that replicas below is 2: that + # release put the sweeper, the notifier and the migration runner each + # behind a Postgres advisory lock, and gave incident creation its own + # conflict resolution, which is what makes a second replica safe + # instead of racing the first. (Still carries v0.34.0's fix too -- + # callerMayManageServiceAccount, so an instance-scoped service account + # can adopt/rotate a key on a team-scoped account it didn't just create + # in the same call -- without which terdutteam-* can wedge permanently + # on the crash-window race this demo hit live, niklas/terdut-operator#3.) + tag: v0.36.0 + # Matches this CRD's own spec.replicas default (v0.4.0) -- stated + # explicitly, like every other field in this file, rather than left to + # the default. RollingUpdate follows automatically; this operator does + # not expose Strategy as a spec field. + replicas: 2 networking: hostname: terdut-operator-demo.example servicePort: 8080 From 822c80dda61103ac2cfe019bbd9999d0fbfbdae2 Mon Sep 17 00:00:00 2001 From: Niklas Ye Date: Sat, 3 Oct 2026 18:15:42 +0200 Subject: [PATCH 2/3] examples/demo: split the Ready wait so escalation rules wait on alice too wait_for_ready waited for every demo object at once, including terdutescalationrule-platform, which names alice as a level-1 target -- but alice does not exist yet at that point in main(): she is created by redeem_platform_invite, which ran after wait_for_ready. terdut-server resolves every named username at reconcile time, not just when an escalation actually fires, so that CR could never reach Ready before alice did, and main() had no step in between to create her. Split into wait_for_objects (the shared loop, now taking its object list as arguments) plus two callers: wait_for_teams_ready, covering just the server and the two teams redeem_platform_invite/join_payments_team need, run before alice exists; wait_for_remaining_ready, covering the escalation rules, dead man's switches and alert sources, run after. Co-authored-by: Claude --- examples/demo/run-demo.sh | 44 ++++++++++++++++++++++++++------------- 1 file changed, 30 insertions(+), 14 deletions(-) diff --git a/examples/demo/run-demo.sh b/examples/demo/run-demo.sh index e589f13..013bf80 100755 --- a/examples/demo/run-demo.sh +++ b/examples/demo/run-demo.sh @@ -107,26 +107,41 @@ apply_demo() { kubectl apply -n "$NAMESPACE" -k "$SCRIPT_DIR" >/dev/null } -wait_for_ready() { - local objects=( - "terdutserver/terdut-operator-demo" - "terdutteam/terdutteam-platform" - "terdutteam/terdutteam-payments" - "terdutescalationrule/terdutescalationrule-platform" - "terdutescalationrule/terdutescalationrule-payments" - "terdutdeadmanswitch/terdutdeadmanswitch-platform" - "terdutdeadmanswitch/terdutdeadmanswitch-payments" - "terdutalertsource/terdutalertsource-platform" - "terdutalertsource/terdutalertsource-payments" - ) +wait_for_objects() { local obj - for obj in "${objects[@]}"; do + for obj in "$@"; do log "waiting for $obj to become Ready" kubectl wait --for=condition=Ready --timeout "$WAIT_TIMEOUT" -n "$NAMESPACE" "$obj" >/dev/null \ || die "timed out waiting for $obj -- try: kubectl describe -n $NAMESPACE $obj" done } +# Just the server and the two teams -- everything redeem_platform_invite and +# join_payments_team need. Deliberately NOT the escalation rules here: this +# demo kit's own terdutescalationrule-platform names alice as a level-1 +# target, and that CR cannot reach Ready until alice actually exists +# (terdut-server resolves every named username at reconcile time, not just +# at escalation time) -- a real dependency this script has to satisfy by +# creating her first, not something kubectl wait can be told to ignore. +wait_for_teams_ready() { + wait_for_objects \ + "terdutserver/terdut-operator-demo" \ + "terdutteam/terdutteam-platform" \ + "terdutteam/terdutteam-payments" +} + +# Everything that was waiting on alice (or just on the teams above, now +# already satisfied) to exist. +wait_for_remaining_ready() { + wait_for_objects \ + "terdutescalationrule/terdutescalationrule-platform" \ + "terdutescalationrule/terdutescalationrule-payments" \ + "terdutdeadmanswitch/terdutdeadmanswitch-platform" \ + "terdutdeadmanswitch/terdutdeadmanswitch-payments" \ + "terdutalertsource/terdutalertsource-platform" \ + "terdutalertsource/terdutalertsource-payments" +} + start_port_forward() { # A stale pidfile from an earlier run would otherwise collide with us on # $LOCAL_PORT -- if that pid is still alive, stop it first. @@ -319,10 +334,11 @@ main() { ensure_kind_cluster install_operator apply_demo - wait_for_ready + wait_for_teams_ready start_port_forward redeem_platform_invite join_payments_team + wait_for_remaining_ready fire_demo_alerts print_summary From 50ce5bcec007d0965d761a158630b11d524868d5 Mon Sep 17 00:00:00 2001 From: Niklas Ye Date: Thu, 8 Oct 2026 18:55:02 +0200 Subject: [PATCH 3/3] examples/demo: run terdut-server v0.43.0, with alerts from two clusters The demo pinned v0.36.0, the floor for replicas: 2, and so showed none of the web UI since: the queue and incident layouts, the rota and escalation pages, the theme toggle, and the cluster chip, filter and page titles (v0.42.0-v0.43.0). It pins v0.43.0 now; the comment keeps v0.36.0 as the floor, which is what the replicas setting actually depends on. fire-alerts.sh takes an optional CLUSTER, standing in for a Prometheus external label plus `cluster` in Alertmanager's group_by (terdut-server's README, "Several clusters, one team"). It goes on the alert's labels and groupLabels, and into the group key and the fingerprint, so the same alert in two clusters is two incidents and not one. Unset, the payload is exactly what it was. run-demo.sh fires its alerts across prod-eu and prod-us, high-cpu in both, so the queue has a chip and a filter to show. run-demo.sh also failed on its second run, though it says it is safe to re-run: it expected HTTP 409 when alice already exists, but a spent invite is answered with 403 "invite link is not usable" before the username is ever checked. It now tries to log alice in first and skips the signup if that works. Checked on the kind cluster: the server rolled to v0.43.0, every CR became Ready and Adopted (server, both teams, both escalation rules, both dead man's switches, both alert sources), and /api/incidents/clusters, /api/incidents?cluster=prod-us and the incident titles came back as expected. No operator code changed, so this needs no operator release. Co-authored-by: Claude --- examples/demo/01-server.yaml | 5 +++-- examples/demo/README.md | 14 ++++++++++++++ examples/demo/fire-alerts.sh | 24 ++++++++++++++++++------ examples/demo/run-demo.sh | 26 +++++++++++++++++++++----- 4 files changed, 56 insertions(+), 13 deletions(-) diff --git a/examples/demo/01-server.yaml b/examples/demo/01-server.yaml index e877a31..af28d0f 100644 --- a/examples/demo/01-server.yaml +++ b/examples/demo/01-server.yaml @@ -14,7 +14,8 @@ metadata: spec: image: repository: git.ryuvia.com/niklas/terdut-server - # v0.36.0 or newer is required now that replicas below is 2: that + # v0.36.0 is the floor now that replicas below is 2 (this demo pins + # the current release, v0.43.0, so it shows the current web UI too): that # release put the sweeper, the notifier and the migration runner each # behind a Postgres advisory lock, and gave incident creation its own # conflict resolution, which is what makes a second replica safe @@ -23,7 +24,7 @@ spec: # can adopt/rotate a key on a team-scoped account it didn't just create # in the same call -- without which terdutteam-* can wedge permanently # on the crash-window race this demo hit live, niklas/terdut-operator#3.) - tag: v0.36.0 + tag: v0.43.0 # Matches this CRD's own spec.replicas default (v0.4.0) -- stated # explicitly, like every other field in this file, rather than left to # the default. RollingUpdate follows automatically; this operator does diff --git a/examples/demo/README.md b/examples/demo/README.md index 0829e5f..c91e83b 100644 --- a/examples/demo/README.md +++ b/examples/demo/README.md @@ -133,6 +133,20 @@ Each `(team, scenario)` pair is one stable fingerprint, so firing the same one twice updates the same alert (a real re-fire) and `resolve` closes exactly that one. +Set `CLUSTER` to send the alert as if it came from one of several clusters: + +```sh +CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu +CLUSTER=prod-us ./fire-alerts.sh platform high-cpu # a second incident, not a join +CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu resolve +``` + +It stands in for a Prometheus external label plus `cluster` in Alertmanager's +`group_by` (terdut-server's README, "Several clusters, one team"): the web UI +then shows the cluster chip on each incident and a cluster filter in the +queue. `CLUSTER` is part of the fingerprint, so resolve with the same value you +fired with. `./run-demo.sh` fires its alerts across `prod-eu` and `prod-us`. + ### Dead man's switches `06-deadman-platform.yaml` / `07-deadman-payments.yaml` expect a heartbeat diff --git a/examples/demo/fire-alerts.sh b/examples/demo/fire-alerts.sh index b8a8ac9..7af76b7 100755 --- a/examples/demo/fire-alerts.sh +++ b/examples/demo/fire-alerts.sh @@ -16,7 +16,15 @@ # is the one part of that URL still usable here. # # Usage: -# ./fire-alerts.sh [resolve] +# [CLUSTER=prod-eu] ./fire-alerts.sh [resolve] +# +# CLUSTER stands in for a Prometheus externalLabel plus `cluster` in +# Alertmanager's group_by (terdut-server's README, "Several clusters, one +# team"): it is put on the alert's labels and on groupLabels, so the incident +# carries it and the web UI shows the cluster chip and the queue's cluster +# filter. It is also part of the group key and the fingerprint, which is what +# keeps the same alert in two clusters from joining one incident. Unset, the +# alert is sent exactly as before. # # Prerequisites: kubectl context pointed at the demo namespace, jq, curl, # and (in another terminal) a running: @@ -24,6 +32,7 @@ set -euo pipefail NAMESPACE="${NAMESPACE:-}" +CLUSTER="${CLUSTER:-}" BASE_URL="${BASE_URL:-http://localhost:8080}" usage() { @@ -45,6 +54,8 @@ scenarios: env vars: NAMESPACE kubectl -n for reading the webhook Secret (required) BASE_URL where the port-forwarded terdut-server is (default http://localhost:8080) + CLUSTER optional cluster name, e.g. prod-eu: sent as a `cluster` label and + group label, so the UI shows where the incident came from EOF exit 1 } @@ -90,7 +101,7 @@ key="$(kubectl -n "$NAMESPACE" get secret "$secret_name" -o jsonpath='{.data.key # created: terdut-server correlates on (team_id, fingerprint), not on # anything else in the payload. Real Alertmanager computes this from the # alert's label set; a fixed string plays the same role here. -fingerprint="demo-${team}-${scenario}" +fingerprint="demo-${team}-${scenario}${CLUSTER:+-$CLUSTER}" now="$(date -u +%Y-%m-%dT%H:%M:%SZ)" if [ "$status" = firing ]; then @@ -101,7 +112,8 @@ fi payload="$(jq -n \ --arg status "$status" \ - --arg groupKey "demo:${team}:${scenario}" \ + --arg groupKey "demo:${team}:${scenario}${CLUSTER:+:$CLUSTER}" \ + --arg cluster "$CLUSTER" \ --arg alertname "$alertname" \ --arg team "$team" \ --arg severity "$severity" \ @@ -113,10 +125,10 @@ payload="$(jq -n \ version: "4", status: $status, groupKey: $groupKey, - groupLabels: { alertname: $alertname, team: $team }, + groupLabels: ({ alertname: $alertname, team: $team } + (if $cluster != "" then { cluster: $cluster } else {} end)), alerts: [{ status: $status, - labels: { alertname: $alertname, severity: $severity, team: $team, instance: "demo" }, + labels: ({ alertname: $alertname, severity: $severity, team: $team, instance: "demo" } + (if $cluster != "" then { cluster: $cluster } else {} end)), annotations: { summary: $summary }, startsAt: $startsAt, endsAt: $endsAt, @@ -126,7 +138,7 @@ payload="$(jq -n \ }')" url="${BASE_URL}/api/integrations/${key}/alertmanager" -echo "POST $url (team=$team scenario=$scenario status=$status)" >&2 +echo "POST $url (team=$team scenario=$scenario status=$status${CLUSTER:+ cluster=$CLUSTER})" >&2 code="$(curl -sS -o /tmp/fire-alerts-response.json -w '%{http_code}' \ -X POST "$url" -H 'Content-Type: application/json' -d "$payload")" echo "-> HTTP $code" >&2 diff --git a/examples/demo/run-demo.sh b/examples/demo/run-demo.sh index 013bf80..017bb7e 100755 --- a/examples/demo/run-demo.sh +++ b/examples/demo/run-demo.sh @@ -196,6 +196,18 @@ redeem_platform_invite() { invite_token="${invite_url##*invite=}" [ -n "$invite_token" ] || die "could not parse an invite token out of $invite_url" + # A re-run: the invite was spent by the first run, and the server answers a + # spent invite with 403 before it ever looks at the username, so the 409 + # handled below never arrives. If alice can already sign in, she exists. + local login_code + login_code="$(curl -sS -o /dev/null -w '%{http_code}' \ + -X POST "${BASE_URL}/api/login" -H 'Content-Type: application/json' \ + -d "$(jq -n --arg u "$ALICE_USERNAME" --arg p "$DEMO_PASSWORD" '{username: $u, password: $p}')")" + if [ "$login_code" = "200" ]; then + log "account ${ALICE_USERNAME} already exists and can sign in, skipping signup (re-run detected)" + return 0 + fi + log "signing up ${ALICE_USERNAME} via Platform's invite" local body resp_file code body="$(jq -n \ @@ -250,9 +262,13 @@ join_payments_team() { fire_demo_alerts() { log "firing representative demo alerts" - NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform high-cpu - NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform disk-full - NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" payments pod-crash + # Two clusters, so the queue shows the cluster chip and offers its filter. + # high-cpu fires in both: the same alert in two clusters is two incidents. + local fire="$SCRIPT_DIR/fire-alerts.sh" + CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu + CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu + CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform disk-full + CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" payments pod-crash } print_summary() { @@ -269,8 +285,8 @@ terdut demo is up. Fire more alerts: export NAMESPACE=${NAMESPACE} BASE_URL=${BASE_URL} - ./fire-alerts.sh platform high-cpu - ./fire-alerts.sh platform high-cpu resolve + CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu + CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu resolve ./fire-alerts.sh payments heartbeat # send repeatedly (e.g. every minute) # to keep a dead man's switch alive; # stop sending it and, 15 minutes