Merge pull request 'examples/demo: run terdut-server v0.43.0, with alerts from two clusters' (#6) from demo-matches-v0.4.0-crd into main
CI / chart (push) Successful in 2s
CI / security (push) Successful in 51s
CI / test (push) Successful in 2m26s

Reviewed-on: #6
This commit was merged in pull request #6.
This commit is contained in:
2026-10-08 17:10:06 +00:00
4 changed files with 99 additions and 32 deletions
+16 -7
View File
@@ -14,13 +14,22 @@ metadata:
spec:
image:
repository: git.ryuvia.com/niklas/terdut-server
# v0.34.0: fixes callerMayManageServiceAccount so an instance-scoped
# service account can adopt/rotate a key on a team-scoped account it
# didn't just create in the same call -- without this, terdutteam-*
# can wedge permanently on exactly the crash-window race this demo
# hit live (niklas/terdut-operator#3).
tag: v0.34.0
replicas: 1
# v0.36.0 is the floor now that replicas below is 2 (this demo pins
# the current release, v0.43.0, so it shows the current web UI too): that
# release put the sweeper, the notifier and the migration runner each
# behind a Postgres advisory lock, and gave incident creation its own
# conflict resolution, which is what makes a second replica safe
# instead of racing the first. (Still carries v0.34.0's fix too --
# callerMayManageServiceAccount, so an instance-scoped service account
# can adopt/rotate a key on a team-scoped account it didn't just create
# in the same call -- without which terdutteam-* can wedge permanently
# on the crash-window race this demo hit live, niklas/terdut-operator#3.)
tag: v0.43.0
# Matches this CRD's own spec.replicas default (v0.4.0) -- stated
# explicitly, like every other field in this file, rather than left to
# the default. RollingUpdate follows automatically; this operator does
# not expose Strategy as a spec field.
replicas: 2
networking:
hostname: terdut-operator-demo.example
servicePort: 8080
+14
View File
@@ -133,6 +133,20 @@ Each `(team, scenario)` pair is one stable fingerprint, so firing the same
one twice updates the same alert (a real re-fire) and `resolve` closes
exactly that one.
Set `CLUSTER` to send the alert as if it came from one of several clusters:
```sh
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu
CLUSTER=prod-us ./fire-alerts.sh platform high-cpu # a second incident, not a join
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu resolve
```
It stands in for a Prometheus external label plus `cluster` in Alertmanager's
`group_by` (terdut-server's README, "Several clusters, one team"): the web UI
then shows the cluster chip on each incident and a cluster filter in the
queue. `CLUSTER` is part of the fingerprint, so resolve with the same value you
fired with. `./run-demo.sh` fires its alerts across `prod-eu` and `prod-us`.
### Dead man's switches
`06-deadman-platform.yaml` / `07-deadman-payments.yaml` expect a heartbeat
+18 -6
View File
@@ -16,7 +16,15 @@
# is the one part of that URL still usable here.
#
# Usage:
# ./fire-alerts.sh <platform|payments> <high-cpu|disk-full|pod-crash|heartbeat> [resolve]
# [CLUSTER=prod-eu] ./fire-alerts.sh <platform|payments> <high-cpu|disk-full|pod-crash|heartbeat> [resolve]
#
# CLUSTER stands in for a Prometheus externalLabel plus `cluster` in
# Alertmanager's group_by (terdut-server's README, "Several clusters, one
# team"): it is put on the alert's labels and on groupLabels, so the incident
# carries it and the web UI shows the cluster chip and the queue's cluster
# filter. It is also part of the group key and the fingerprint, which is what
# keeps the same alert in two clusters from joining one incident. Unset, the
# alert is sent exactly as before.
#
# Prerequisites: kubectl context pointed at the demo namespace, jq, curl,
# and (in another terminal) a running:
@@ -24,6 +32,7 @@
set -euo pipefail
NAMESPACE="${NAMESPACE:-}"
CLUSTER="${CLUSTER:-}"
BASE_URL="${BASE_URL:-http://localhost:8080}"
usage() {
@@ -45,6 +54,8 @@ scenarios:
env vars:
NAMESPACE kubectl -n for reading the webhook Secret (required)
BASE_URL where the port-forwarded terdut-server is (default http://localhost:8080)
CLUSTER optional cluster name, e.g. prod-eu: sent as a `cluster` label and
group label, so the UI shows where the incident came from
EOF
exit 1
}
@@ -90,7 +101,7 @@ key="$(kubectl -n "$NAMESPACE" get secret "$secret_name" -o jsonpath='{.data.key
# created: terdut-server correlates on (team_id, fingerprint), not on
# anything else in the payload. Real Alertmanager computes this from the
# alert's label set; a fixed string plays the same role here.
fingerprint="demo-${team}-${scenario}"
fingerprint="demo-${team}-${scenario}${CLUSTER:+-$CLUSTER}"
now="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
if [ "$status" = firing ]; then
@@ -101,7 +112,8 @@ fi
payload="$(jq -n \
--arg status "$status" \
--arg groupKey "demo:${team}:${scenario}" \
--arg groupKey "demo:${team}:${scenario}${CLUSTER:+:$CLUSTER}" \
--arg cluster "$CLUSTER" \
--arg alertname "$alertname" \
--arg team "$team" \
--arg severity "$severity" \
@@ -113,10 +125,10 @@ payload="$(jq -n \
version: "4",
status: $status,
groupKey: $groupKey,
groupLabels: { alertname: $alertname, team: $team },
groupLabels: ({ alertname: $alertname, team: $team } + (if $cluster != "" then { cluster: $cluster } else {} end)),
alerts: [{
status: $status,
labels: { alertname: $alertname, severity: $severity, team: $team, instance: "demo" },
labels: ({ alertname: $alertname, severity: $severity, team: $team, instance: "demo" } + (if $cluster != "" then { cluster: $cluster } else {} end)),
annotations: { summary: $summary },
startsAt: $startsAt,
endsAt: $endsAt,
@@ -126,7 +138,7 @@ payload="$(jq -n \
}')"
url="${BASE_URL}/api/integrations/${key}/alertmanager"
echo "POST $url (team=$team scenario=$scenario status=$status)" >&2
echo "POST $url (team=$team scenario=$scenario status=$status${CLUSTER:+ cluster=$CLUSTER})" >&2
code="$(curl -sS -o /tmp/fire-alerts-response.json -w '%{http_code}' \
-X POST "$url" -H 'Content-Type: application/json' -d "$payload")"
echo "-> HTTP $code" >&2
+51 -19
View File
@@ -107,26 +107,41 @@ apply_demo() {
kubectl apply -n "$NAMESPACE" -k "$SCRIPT_DIR" >/dev/null
}
wait_for_ready() {
local objects=(
"terdutserver/terdut-operator-demo"
"terdutteam/terdutteam-platform"
"terdutteam/terdutteam-payments"
"terdutescalationrule/terdutescalationrule-platform"
"terdutescalationrule/terdutescalationrule-payments"
"terdutdeadmanswitch/terdutdeadmanswitch-platform"
"terdutdeadmanswitch/terdutdeadmanswitch-payments"
"terdutalertsource/terdutalertsource-platform"
"terdutalertsource/terdutalertsource-payments"
)
wait_for_objects() {
local obj
for obj in "${objects[@]}"; do
for obj in "$@"; do
log "waiting for $obj to become Ready"
kubectl wait --for=condition=Ready --timeout "$WAIT_TIMEOUT" -n "$NAMESPACE" "$obj" >/dev/null \
|| die "timed out waiting for $obj -- try: kubectl describe -n $NAMESPACE $obj"
done
}
# Just the server and the two teams -- everything redeem_platform_invite and
# join_payments_team need. Deliberately NOT the escalation rules here: this
# demo kit's own terdutescalationrule-platform names alice as a level-1
# target, and that CR cannot reach Ready until alice actually exists
# (terdut-server resolves every named username at reconcile time, not just
# at escalation time) -- a real dependency this script has to satisfy by
# creating her first, not something kubectl wait can be told to ignore.
wait_for_teams_ready() {
wait_for_objects \
"terdutserver/terdut-operator-demo" \
"terdutteam/terdutteam-platform" \
"terdutteam/terdutteam-payments"
}
# Everything that was waiting on alice (or just on the teams above, now
# already satisfied) to exist.
wait_for_remaining_ready() {
wait_for_objects \
"terdutescalationrule/terdutescalationrule-platform" \
"terdutescalationrule/terdutescalationrule-payments" \
"terdutdeadmanswitch/terdutdeadmanswitch-platform" \
"terdutdeadmanswitch/terdutdeadmanswitch-payments" \
"terdutalertsource/terdutalertsource-platform" \
"terdutalertsource/terdutalertsource-payments"
}
start_port_forward() {
# A stale pidfile from an earlier run would otherwise collide with us on
# $LOCAL_PORT -- if that pid is still alive, stop it first.
@@ -181,6 +196,18 @@ redeem_platform_invite() {
invite_token="${invite_url##*invite=}"
[ -n "$invite_token" ] || die "could not parse an invite token out of $invite_url"
# A re-run: the invite was spent by the first run, and the server answers a
# spent invite with 403 before it ever looks at the username, so the 409
# handled below never arrives. If alice can already sign in, she exists.
local login_code
login_code="$(curl -sS -o /dev/null -w '%{http_code}' \
-X POST "${BASE_URL}/api/login" -H 'Content-Type: application/json' \
-d "$(jq -n --arg u "$ALICE_USERNAME" --arg p "$DEMO_PASSWORD" '{username: $u, password: $p}')")"
if [ "$login_code" = "200" ]; then
log "account ${ALICE_USERNAME} already exists and can sign in, skipping signup (re-run detected)"
return 0
fi
log "signing up ${ALICE_USERNAME} via Platform's invite"
local body resp_file code
body="$(jq -n \
@@ -235,9 +262,13 @@ join_payments_team() {
fire_demo_alerts() {
log "firing representative demo alerts"
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform high-cpu
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform disk-full
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" payments pod-crash
# Two clusters, so the queue shows the cluster chip and offers its filter.
# high-cpu fires in both: the same alert in two clusters is two incidents.
local fire="$SCRIPT_DIR/fire-alerts.sh"
CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu
CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu
CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform disk-full
CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" payments pod-crash
}
print_summary() {
@@ -254,8 +285,8 @@ terdut demo is up.
Fire more alerts:
export NAMESPACE=${NAMESPACE} BASE_URL=${BASE_URL}
./fire-alerts.sh platform high-cpu
./fire-alerts.sh platform high-cpu resolve
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu resolve
./fire-alerts.sh payments heartbeat # send repeatedly (e.g. every minute)
# to keep a dead man's switch alive;
# stop sending it and, 15 minutes
@@ -319,10 +350,11 @@ main() {
ensure_kind_cluster
install_operator
apply_demo
wait_for_ready
wait_for_teams_ready
start_port_forward
redeem_platform_invite
join_payments_team
wait_for_remaining_ready
fire_demo_alerts
print_summary