ca2cd2c645
A self-contained demo kit: a TerdutServer against a throwaway, bare Postgres (bring-your-own DSN -- simplest path to stand up from nothing, ROADMAP.md Stage 1's own note), two TerdutTeams, and each team's own TerdutEscalationRule/TerdutDeadmanSwitch/TerdutAlertSource, so every CRD this operator manages is exercised together rather than in isolation the way config/samples' one-of-each already does. fire-alerts.sh sends terdut-server's own amPayload/amAlert shape (read from internal/api/alertmanager.go in that repo, not guessed from its docs) at whichever TerdutAlertSource's generated webhook Secret it reads the key out of -- high-cpu/disk-full/pod-crash scenarios to open and resolve incidents, and a heartbeat scenario matching each team's dead man's switch matcher, so stopping it demonstrates the switch noticing silence on its own. Verified server-side (kubectl apply --dry-run=server -k examples/demo) against this operator's own dev cluster, which already has these CRDs installed: every object validates. The one warning that cluster's "restricted" PodSecurity raises (postgres:17-alpine's entrypoint needs to start as root before it drops privileges itself) is noted inline in 00-postgres.yaml rather than worked around -- not a real production pattern, and this Postgres exists only to be thrown away with the rest of the demo namespace. README.md walks through: applying, watching status, why a few early CrashLoopBackOff restarts on terdut-demo itself are expected (this operator's Deployment template has no wait-for-postgres init container yet, unlike charts/terdut-server's chart as of v0.33.2), reaching the web UI (port-forward -- spec.networking.hostname is accepted but nothing creates an HTTPRoute for it yet), turning on open signup with the operator's own generated admin token since the bootstrap-created account has no password, firing alerts, and tearing down.
139 lines
5.2 KiB
Bash
Executable File
139 lines
5.2 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# Sends a synthetic Alertmanager v4 webhook payload at the "platform" or
|
|
# "payments" demo team's TerdutAlertSource, so terdut-server opens (or
|
|
# resolves) an incident exactly the way it would for a real Alertmanager.
|
|
#
|
|
# The payload shape here is amPayload/amAlert, read straight out of
|
|
# terdut-server's own internal/api/alertmanager.go rather than guessed from
|
|
# its docs -- version/status/groupKey/groupLabels, and alerts[] carrying
|
|
# status/labels/annotations/startsAt/endsAt/generatorURL/fingerprint.
|
|
#
|
|
# Why this reads the webhook key out of a kubectl Secret instead of using
|
|
# the "url" key already in it: that URL is built from spec.networking.hostname
|
|
# (TERDUT_PUBLIC_URL), and nothing in this demo stands up real ingress for
|
|
# it (01-server.yaml's own comment) -- so it resolves nowhere. The key
|
|
# alone, against whatever you've actually port-forwarded BASE_URL to below,
|
|
# is the one part of that URL still usable here.
|
|
#
|
|
# Usage:
|
|
# ./fire-alerts.sh <platform|payments> <high-cpu|disk-full|pod-crash|heartbeat> [resolve]
|
|
#
|
|
# Prerequisites: kubectl context pointed at the demo namespace, jq, curl,
|
|
# and (in another terminal) a running:
|
|
# kubectl port-forward svc/terdut-demo 8080:8080
|
|
set -euo pipefail
|
|
|
|
NAMESPACE="${NAMESPACE:-}"
|
|
BASE_URL="${BASE_URL:-http://localhost:8080}"
|
|
|
|
usage() {
|
|
cat >&2 <<'EOF'
|
|
usage: fire-alerts.sh <platform|payments> <scenario> [resolve]
|
|
|
|
scenarios:
|
|
high-cpu warning -- CPU usage above 90% for 10 minutes
|
|
disk-full critical -- disk usage above 95%
|
|
pod-crash error -- a pod crash-looping
|
|
heartbeat critical -- the team's dead man's switch heartbeat
|
|
(matches the matcher in 06/07-deadman-*.yaml -- send this
|
|
repeatedly to keep the switch alive, or stop sending it and
|
|
watch terdut-server open an incident on its own once
|
|
`timeout` passes with no heartbeat. "resolve" is not a valid
|
|
third argument for this scenario: a heartbeat is only ever
|
|
firing.)
|
|
|
|
env vars:
|
|
NAMESPACE kubectl -n for reading the webhook Secret (required)
|
|
BASE_URL where the port-forwarded terdut-server is (default http://localhost:8080)
|
|
EOF
|
|
exit 1
|
|
}
|
|
|
|
[ $# -ge 2 ] || usage
|
|
team="$1" scenario="$2" verb="${3:-fire}"
|
|
[ -n "$NAMESPACE" ] || { echo "fire-alerts.sh: set NAMESPACE" >&2; exit 1; }
|
|
|
|
case "$team" in
|
|
platform|payments) ;;
|
|
*) usage ;;
|
|
esac
|
|
|
|
case "$scenario" in
|
|
high-cpu) alertname=TerdutDemoHighCPU severity=warning summary="CPU usage above 90% for 10 minutes" ;;
|
|
disk-full) alertname=TerdutDemoDiskFull severity=critical summary="Disk usage above 95% on /data" ;;
|
|
pod-crash) alertname=TerdutDemoPodCrashLooping severity=error summary="Pod web-7f8b9 is crash-looping (5 restarts in 10m)" ;;
|
|
heartbeat)
|
|
# Must match 06-deadman-platform.yaml / 07-deadman-payments.yaml's own
|
|
# matcher exactly -- that's what makes this a heartbeat rather than a
|
|
# third ordinary alert.
|
|
case "$team" in
|
|
platform) alertname=PlatformWatchdog ;;
|
|
payments) alertname=PaymentsWatchdog ;;
|
|
esac
|
|
severity=critical summary="demo heartbeat"
|
|
[ "$verb" = fire ] || { echo "fire-alerts.sh: heartbeat is only ever fired, never resolved -- just stop sending it" >&2; exit 1; }
|
|
;;
|
|
*) usage ;;
|
|
esac
|
|
|
|
case "$verb" in
|
|
fire) status=firing ;;
|
|
resolve) status=resolved ;;
|
|
*) usage ;;
|
|
esac
|
|
|
|
secret_name="terdutalertsource-${team}-terdut-webhook"
|
|
key="$(kubectl -n "$NAMESPACE" get secret "$secret_name" -o jsonpath='{.data.key}' | base64 -d)"
|
|
[ -n "$key" ] || { echo "fire-alerts.sh: empty key read from Secret $secret_name -- has 08/09-alertsource-*.yaml reconciled yet?" >&2; exit 1; }
|
|
|
|
# Stable per (team, scenario) so a resolve targets the same alert a fire
|
|
# created: terdut-server correlates on (team_id, fingerprint), not on
|
|
# anything else in the payload. Real Alertmanager computes this from the
|
|
# alert's label set; a fixed string plays the same role here.
|
|
fingerprint="demo-${team}-${scenario}"
|
|
|
|
now="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
|
|
if [ "$status" = firing ]; then
|
|
ends_at="0001-01-01T00:00:00Z" # Alertmanager's own "not resolved" zero value
|
|
else
|
|
ends_at="$now"
|
|
fi
|
|
|
|
payload="$(jq -n \
|
|
--arg status "$status" \
|
|
--arg groupKey "demo:${team}:${scenario}" \
|
|
--arg alertname "$alertname" \
|
|
--arg team "$team" \
|
|
--arg severity "$severity" \
|
|
--arg summary "$summary" \
|
|
--arg startsAt "$now" \
|
|
--arg endsAt "$ends_at" \
|
|
--arg fingerprint "$fingerprint" \
|
|
'{
|
|
version: "4",
|
|
status: $status,
|
|
groupKey: $groupKey,
|
|
groupLabels: { alertname: $alertname, team: $team },
|
|
alerts: [{
|
|
status: $status,
|
|
labels: { alertname: $alertname, severity: $severity, team: $team, instance: "demo" },
|
|
annotations: { summary: $summary },
|
|
startsAt: $startsAt,
|
|
endsAt: $endsAt,
|
|
generatorURL: "https://example.com/demo",
|
|
fingerprint: $fingerprint
|
|
}]
|
|
}')"
|
|
|
|
url="${BASE_URL}/api/integrations/${key}/alertmanager"
|
|
echo "POST $url (team=$team scenario=$scenario status=$status)" >&2
|
|
code="$(curl -sS -o /tmp/fire-alerts-response.json -w '%{http_code}' \
|
|
-X POST "$url" -H 'Content-Type: application/json' -d "$payload")"
|
|
echo "-> HTTP $code" >&2
|
|
cat /tmp/fire-alerts-response.json >&2
|
|
echo >&2
|
|
|
|
if [ "$code" != "200" ]; then
|
|
exit 1
|
|
fi
|