Compare commits
3 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 50ce5bcec0 | |||
| 822c80dda6 | |||
| 0ee7ede648 |
@@ -14,13 +14,22 @@ metadata:
|
||||
spec:
|
||||
image:
|
||||
repository: git.ryuvia.com/niklas/terdut-server
|
||||
# v0.34.0: fixes callerMayManageServiceAccount so an instance-scoped
|
||||
# service account can adopt/rotate a key on a team-scoped account it
|
||||
# didn't just create in the same call -- without this, terdutteam-*
|
||||
# can wedge permanently on exactly the crash-window race this demo
|
||||
# hit live (niklas/terdut-operator#3).
|
||||
tag: v0.34.0
|
||||
replicas: 1
|
||||
# v0.36.0 is the floor now that replicas below is 2 (this demo pins
|
||||
# the current release, v0.43.0, so it shows the current web UI too): that
|
||||
# release put the sweeper, the notifier and the migration runner each
|
||||
# behind a Postgres advisory lock, and gave incident creation its own
|
||||
# conflict resolution, which is what makes a second replica safe
|
||||
# instead of racing the first. (Still carries v0.34.0's fix too --
|
||||
# callerMayManageServiceAccount, so an instance-scoped service account
|
||||
# can adopt/rotate a key on a team-scoped account it didn't just create
|
||||
# in the same call -- without which terdutteam-* can wedge permanently
|
||||
# on the crash-window race this demo hit live, niklas/terdut-operator#3.)
|
||||
tag: v0.43.0
|
||||
# Matches this CRD's own spec.replicas default (v0.4.0) -- stated
|
||||
# explicitly, like every other field in this file, rather than left to
|
||||
# the default. RollingUpdate follows automatically; this operator does
|
||||
# not expose Strategy as a spec field.
|
||||
replicas: 2
|
||||
networking:
|
||||
hostname: terdut-operator-demo.example
|
||||
servicePort: 8080
|
||||
|
||||
@@ -133,6 +133,20 @@ Each `(team, scenario)` pair is one stable fingerprint, so firing the same
|
||||
one twice updates the same alert (a real re-fire) and `resolve` closes
|
||||
exactly that one.
|
||||
|
||||
Set `CLUSTER` to send the alert as if it came from one of several clusters:
|
||||
|
||||
```sh
|
||||
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu
|
||||
CLUSTER=prod-us ./fire-alerts.sh platform high-cpu # a second incident, not a join
|
||||
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu resolve
|
||||
```
|
||||
|
||||
It stands in for a Prometheus external label plus `cluster` in Alertmanager's
|
||||
`group_by` (terdut-server's README, "Several clusters, one team"): the web UI
|
||||
then shows the cluster chip on each incident and a cluster filter in the
|
||||
queue. `CLUSTER` is part of the fingerprint, so resolve with the same value you
|
||||
fired with. `./run-demo.sh` fires its alerts across `prod-eu` and `prod-us`.
|
||||
|
||||
### Dead man's switches
|
||||
|
||||
`06-deadman-platform.yaml` / `07-deadman-payments.yaml` expect a heartbeat
|
||||
|
||||
@@ -16,7 +16,15 @@
|
||||
# is the one part of that URL still usable here.
|
||||
#
|
||||
# Usage:
|
||||
# ./fire-alerts.sh <platform|payments> <high-cpu|disk-full|pod-crash|heartbeat> [resolve]
|
||||
# [CLUSTER=prod-eu] ./fire-alerts.sh <platform|payments> <high-cpu|disk-full|pod-crash|heartbeat> [resolve]
|
||||
#
|
||||
# CLUSTER stands in for a Prometheus externalLabel plus `cluster` in
|
||||
# Alertmanager's group_by (terdut-server's README, "Several clusters, one
|
||||
# team"): it is put on the alert's labels and on groupLabels, so the incident
|
||||
# carries it and the web UI shows the cluster chip and the queue's cluster
|
||||
# filter. It is also part of the group key and the fingerprint, which is what
|
||||
# keeps the same alert in two clusters from joining one incident. Unset, the
|
||||
# alert is sent exactly as before.
|
||||
#
|
||||
# Prerequisites: kubectl context pointed at the demo namespace, jq, curl,
|
||||
# and (in another terminal) a running:
|
||||
@@ -24,6 +32,7 @@
|
||||
set -euo pipefail
|
||||
|
||||
NAMESPACE="${NAMESPACE:-}"
|
||||
CLUSTER="${CLUSTER:-}"
|
||||
BASE_URL="${BASE_URL:-http://localhost:8080}"
|
||||
|
||||
usage() {
|
||||
@@ -45,6 +54,8 @@ scenarios:
|
||||
env vars:
|
||||
NAMESPACE kubectl -n for reading the webhook Secret (required)
|
||||
BASE_URL where the port-forwarded terdut-server is (default http://localhost:8080)
|
||||
CLUSTER optional cluster name, e.g. prod-eu: sent as a `cluster` label and
|
||||
group label, so the UI shows where the incident came from
|
||||
EOF
|
||||
exit 1
|
||||
}
|
||||
@@ -90,7 +101,7 @@ key="$(kubectl -n "$NAMESPACE" get secret "$secret_name" -o jsonpath='{.data.key
|
||||
# created: terdut-server correlates on (team_id, fingerprint), not on
|
||||
# anything else in the payload. Real Alertmanager computes this from the
|
||||
# alert's label set; a fixed string plays the same role here.
|
||||
fingerprint="demo-${team}-${scenario}"
|
||||
fingerprint="demo-${team}-${scenario}${CLUSTER:+-$CLUSTER}"
|
||||
|
||||
now="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
|
||||
if [ "$status" = firing ]; then
|
||||
@@ -101,7 +112,8 @@ fi
|
||||
|
||||
payload="$(jq -n \
|
||||
--arg status "$status" \
|
||||
--arg groupKey "demo:${team}:${scenario}" \
|
||||
--arg groupKey "demo:${team}:${scenario}${CLUSTER:+:$CLUSTER}" \
|
||||
--arg cluster "$CLUSTER" \
|
||||
--arg alertname "$alertname" \
|
||||
--arg team "$team" \
|
||||
--arg severity "$severity" \
|
||||
@@ -113,10 +125,10 @@ payload="$(jq -n \
|
||||
version: "4",
|
||||
status: $status,
|
||||
groupKey: $groupKey,
|
||||
groupLabels: { alertname: $alertname, team: $team },
|
||||
groupLabels: ({ alertname: $alertname, team: $team } + (if $cluster != "" then { cluster: $cluster } else {} end)),
|
||||
alerts: [{
|
||||
status: $status,
|
||||
labels: { alertname: $alertname, severity: $severity, team: $team, instance: "demo" },
|
||||
labels: ({ alertname: $alertname, severity: $severity, team: $team, instance: "demo" } + (if $cluster != "" then { cluster: $cluster } else {} end)),
|
||||
annotations: { summary: $summary },
|
||||
startsAt: $startsAt,
|
||||
endsAt: $endsAt,
|
||||
@@ -126,7 +138,7 @@ payload="$(jq -n \
|
||||
}')"
|
||||
|
||||
url="${BASE_URL}/api/integrations/${key}/alertmanager"
|
||||
echo "POST $url (team=$team scenario=$scenario status=$status)" >&2
|
||||
echo "POST $url (team=$team scenario=$scenario status=$status${CLUSTER:+ cluster=$CLUSTER})" >&2
|
||||
code="$(curl -sS -o /tmp/fire-alerts-response.json -w '%{http_code}' \
|
||||
-X POST "$url" -H 'Content-Type: application/json' -d "$payload")"
|
||||
echo "-> HTTP $code" >&2
|
||||
|
||||
+51
-19
@@ -107,26 +107,41 @@ apply_demo() {
|
||||
kubectl apply -n "$NAMESPACE" -k "$SCRIPT_DIR" >/dev/null
|
||||
}
|
||||
|
||||
wait_for_ready() {
|
||||
local objects=(
|
||||
"terdutserver/terdut-operator-demo"
|
||||
"terdutteam/terdutteam-platform"
|
||||
"terdutteam/terdutteam-payments"
|
||||
"terdutescalationrule/terdutescalationrule-platform"
|
||||
"terdutescalationrule/terdutescalationrule-payments"
|
||||
"terdutdeadmanswitch/terdutdeadmanswitch-platform"
|
||||
"terdutdeadmanswitch/terdutdeadmanswitch-payments"
|
||||
"terdutalertsource/terdutalertsource-platform"
|
||||
"terdutalertsource/terdutalertsource-payments"
|
||||
)
|
||||
wait_for_objects() {
|
||||
local obj
|
||||
for obj in "${objects[@]}"; do
|
||||
for obj in "$@"; do
|
||||
log "waiting for $obj to become Ready"
|
||||
kubectl wait --for=condition=Ready --timeout "$WAIT_TIMEOUT" -n "$NAMESPACE" "$obj" >/dev/null \
|
||||
|| die "timed out waiting for $obj -- try: kubectl describe -n $NAMESPACE $obj"
|
||||
done
|
||||
}
|
||||
|
||||
# Just the server and the two teams -- everything redeem_platform_invite and
|
||||
# join_payments_team need. Deliberately NOT the escalation rules here: this
|
||||
# demo kit's own terdutescalationrule-platform names alice as a level-1
|
||||
# target, and that CR cannot reach Ready until alice actually exists
|
||||
# (terdut-server resolves every named username at reconcile time, not just
|
||||
# at escalation time) -- a real dependency this script has to satisfy by
|
||||
# creating her first, not something kubectl wait can be told to ignore.
|
||||
wait_for_teams_ready() {
|
||||
wait_for_objects \
|
||||
"terdutserver/terdut-operator-demo" \
|
||||
"terdutteam/terdutteam-platform" \
|
||||
"terdutteam/terdutteam-payments"
|
||||
}
|
||||
|
||||
# Everything that was waiting on alice (or just on the teams above, now
|
||||
# already satisfied) to exist.
|
||||
wait_for_remaining_ready() {
|
||||
wait_for_objects \
|
||||
"terdutescalationrule/terdutescalationrule-platform" \
|
||||
"terdutescalationrule/terdutescalationrule-payments" \
|
||||
"terdutdeadmanswitch/terdutdeadmanswitch-platform" \
|
||||
"terdutdeadmanswitch/terdutdeadmanswitch-payments" \
|
||||
"terdutalertsource/terdutalertsource-platform" \
|
||||
"terdutalertsource/terdutalertsource-payments"
|
||||
}
|
||||
|
||||
start_port_forward() {
|
||||
# A stale pidfile from an earlier run would otherwise collide with us on
|
||||
# $LOCAL_PORT -- if that pid is still alive, stop it first.
|
||||
@@ -181,6 +196,18 @@ redeem_platform_invite() {
|
||||
invite_token="${invite_url##*invite=}"
|
||||
[ -n "$invite_token" ] || die "could not parse an invite token out of $invite_url"
|
||||
|
||||
# A re-run: the invite was spent by the first run, and the server answers a
|
||||
# spent invite with 403 before it ever looks at the username, so the 409
|
||||
# handled below never arrives. If alice can already sign in, she exists.
|
||||
local login_code
|
||||
login_code="$(curl -sS -o /dev/null -w '%{http_code}' \
|
||||
-X POST "${BASE_URL}/api/login" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -n --arg u "$ALICE_USERNAME" --arg p "$DEMO_PASSWORD" '{username: $u, password: $p}')")"
|
||||
if [ "$login_code" = "200" ]; then
|
||||
log "account ${ALICE_USERNAME} already exists and can sign in, skipping signup (re-run detected)"
|
||||
return 0
|
||||
fi
|
||||
|
||||
log "signing up ${ALICE_USERNAME} via Platform's invite"
|
||||
local body resp_file code
|
||||
body="$(jq -n \
|
||||
@@ -235,9 +262,13 @@ join_payments_team() {
|
||||
|
||||
fire_demo_alerts() {
|
||||
log "firing representative demo alerts"
|
||||
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform high-cpu
|
||||
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform disk-full
|
||||
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" payments pod-crash
|
||||
# Two clusters, so the queue shows the cluster chip and offers its filter.
|
||||
# high-cpu fires in both: the same alert in two clusters is two incidents.
|
||||
local fire="$SCRIPT_DIR/fire-alerts.sh"
|
||||
CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu
|
||||
CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu
|
||||
CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform disk-full
|
||||
CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" payments pod-crash
|
||||
}
|
||||
|
||||
print_summary() {
|
||||
@@ -254,8 +285,8 @@ terdut demo is up.
|
||||
|
||||
Fire more alerts:
|
||||
export NAMESPACE=${NAMESPACE} BASE_URL=${BASE_URL}
|
||||
./fire-alerts.sh platform high-cpu
|
||||
./fire-alerts.sh platform high-cpu resolve
|
||||
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu
|
||||
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu resolve
|
||||
./fire-alerts.sh payments heartbeat # send repeatedly (e.g. every minute)
|
||||
# to keep a dead man's switch alive;
|
||||
# stop sending it and, 15 minutes
|
||||
@@ -319,10 +350,11 @@ main() {
|
||||
ensure_kind_cluster
|
||||
install_operator
|
||||
apply_demo
|
||||
wait_for_ready
|
||||
wait_for_teams_ready
|
||||
start_port_forward
|
||||
redeem_platform_invite
|
||||
join_payments_team
|
||||
wait_for_remaining_ready
|
||||
fire_demo_alerts
|
||||
print_summary
|
||||
|
||||
|
||||
Reference in New Issue
Block a user