d9315322fc
1. Renamed every object this demo creates (TerdutServer, Postgres
Secret/Deployment/Service) from terdut-demo[-postgres] to
terdut-operator-demo[-postgres]. The user applied this kit into the
already-live "terdut-demo" namespace -- the real operator exercise
from earlier in this repo's own history -- and this demo's own
TerdutServer/Postgres objects shared that exact name. The TerdutServer
apply was rejected outright (DatabaseSpec's own CEL rule: adding dsn
while the live object already had postgresClusterRef violates "exactly
one of" and the API server refused it), and the real Postgres Service
was never touched (confirmed live: still Zalando's own spilo selector,
endpoint still the real StatefulSet pod) -- but the Postgres Secret and
Deployment, having no such protection, were created as brand new,
extra, crash-looping objects sitting right next to the real ones.
Prefixing every name this demo creates means a repeat of this exact
mistake no longer collides with anything, documented directly in
README.md now.
2. The actual crash itself, independent of (1): capabilities.drop: ["ALL"]
(added responding to a PodSecurity "restricted" warning) took
CAP_CHOWN/CAP_FOWNER away from the root user postgres:17-alpine's own
entrypoint needs to chown/chmod the data directory before it drops
privileges itself -- confirmed in a real crashed pod's logs: `chmod:
/var/run/postgresql: Operation not permitted`. kubectl apply
--dry-run=server, which is as far as this got verified before, only
checks admission policy; it was never actually booted. Removed the
capability drop and verified for real this time: applied just
00-postgres.yaml alone into a disposable namespace, waited for the pod
to go Ready, read its logs ("database system is ready to accept
connections"), then deleted that namespace.
139 lines
5.2 KiB
Bash
Executable File
139 lines
5.2 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# Sends a synthetic Alertmanager v4 webhook payload at the "platform" or
|
|
# "payments" demo team's TerdutAlertSource, so terdut-server opens (or
|
|
# resolves) an incident exactly the way it would for a real Alertmanager.
|
|
#
|
|
# The payload shape here is amPayload/amAlert, read straight out of
|
|
# terdut-server's own internal/api/alertmanager.go rather than guessed from
|
|
# its docs -- version/status/groupKey/groupLabels, and alerts[] carrying
|
|
# status/labels/annotations/startsAt/endsAt/generatorURL/fingerprint.
|
|
#
|
|
# Why this reads the webhook key out of a kubectl Secret instead of using
|
|
# the "url" key already in it: that URL is built from spec.networking.hostname
|
|
# (TERDUT_PUBLIC_URL), and nothing in this demo stands up real ingress for
|
|
# it (01-server.yaml's own comment) -- so it resolves nowhere. The key
|
|
# alone, against whatever you've actually port-forwarded BASE_URL to below,
|
|
# is the one part of that URL still usable here.
|
|
#
|
|
# Usage:
|
|
# ./fire-alerts.sh <platform|payments> <high-cpu|disk-full|pod-crash|heartbeat> [resolve]
|
|
#
|
|
# Prerequisites: kubectl context pointed at the demo namespace, jq, curl,
|
|
# and (in another terminal) a running:
|
|
# kubectl port-forward svc/terdut-operator-demo 8080:8080
|
|
set -euo pipefail
|
|
|
|
NAMESPACE="${NAMESPACE:-}"
|
|
BASE_URL="${BASE_URL:-http://localhost:8080}"
|
|
|
|
usage() {
|
|
cat >&2 <<'EOF'
|
|
usage: fire-alerts.sh <platform|payments> <scenario> [resolve]
|
|
|
|
scenarios:
|
|
high-cpu warning -- CPU usage above 90% for 10 minutes
|
|
disk-full critical -- disk usage above 95%
|
|
pod-crash error -- a pod crash-looping
|
|
heartbeat critical -- the team's dead man's switch heartbeat
|
|
(matches the matcher in 06/07-deadman-*.yaml -- send this
|
|
repeatedly to keep the switch alive, or stop sending it and
|
|
watch terdut-server open an incident on its own once
|
|
`timeout` passes with no heartbeat. "resolve" is not a valid
|
|
third argument for this scenario: a heartbeat is only ever
|
|
firing.)
|
|
|
|
env vars:
|
|
NAMESPACE kubectl -n for reading the webhook Secret (required)
|
|
BASE_URL where the port-forwarded terdut-server is (default http://localhost:8080)
|
|
EOF
|
|
exit 1
|
|
}
|
|
|
|
[ $# -ge 2 ] || usage
|
|
team="$1" scenario="$2" verb="${3:-fire}"
|
|
[ -n "$NAMESPACE" ] || { echo "fire-alerts.sh: set NAMESPACE" >&2; exit 1; }
|
|
|
|
case "$team" in
|
|
platform|payments) ;;
|
|
*) usage ;;
|
|
esac
|
|
|
|
case "$scenario" in
|
|
high-cpu) alertname=TerdutDemoHighCPU severity=warning summary="CPU usage above 90% for 10 minutes" ;;
|
|
disk-full) alertname=TerdutDemoDiskFull severity=critical summary="Disk usage above 95% on /data" ;;
|
|
pod-crash) alertname=TerdutDemoPodCrashLooping severity=error summary="Pod web-7f8b9 is crash-looping (5 restarts in 10m)" ;;
|
|
heartbeat)
|
|
# Must match 06-deadman-platform.yaml / 07-deadman-payments.yaml's own
|
|
# matcher exactly -- that's what makes this a heartbeat rather than a
|
|
# third ordinary alert.
|
|
case "$team" in
|
|
platform) alertname=PlatformWatchdog ;;
|
|
payments) alertname=PaymentsWatchdog ;;
|
|
esac
|
|
severity=critical summary="demo heartbeat"
|
|
[ "$verb" = fire ] || { echo "fire-alerts.sh: heartbeat is only ever fired, never resolved -- just stop sending it" >&2; exit 1; }
|
|
;;
|
|
*) usage ;;
|
|
esac
|
|
|
|
case "$verb" in
|
|
fire) status=firing ;;
|
|
resolve) status=resolved ;;
|
|
*) usage ;;
|
|
esac
|
|
|
|
secret_name="terdutalertsource-${team}-terdut-webhook"
|
|
key="$(kubectl -n "$NAMESPACE" get secret "$secret_name" -o jsonpath='{.data.key}' | base64 -d)"
|
|
[ -n "$key" ] || { echo "fire-alerts.sh: empty key read from Secret $secret_name -- has 08/09-alertsource-*.yaml reconciled yet?" >&2; exit 1; }
|
|
|
|
# Stable per (team, scenario) so a resolve targets the same alert a fire
|
|
# created: terdut-server correlates on (team_id, fingerprint), not on
|
|
# anything else in the payload. Real Alertmanager computes this from the
|
|
# alert's label set; a fixed string plays the same role here.
|
|
fingerprint="demo-${team}-${scenario}"
|
|
|
|
now="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
|
|
if [ "$status" = firing ]; then
|
|
ends_at="0001-01-01T00:00:00Z" # Alertmanager's own "not resolved" zero value
|
|
else
|
|
ends_at="$now"
|
|
fi
|
|
|
|
payload="$(jq -n \
|
|
--arg status "$status" \
|
|
--arg groupKey "demo:${team}:${scenario}" \
|
|
--arg alertname "$alertname" \
|
|
--arg team "$team" \
|
|
--arg severity "$severity" \
|
|
--arg summary "$summary" \
|
|
--arg startsAt "$now" \
|
|
--arg endsAt "$ends_at" \
|
|
--arg fingerprint "$fingerprint" \
|
|
'{
|
|
version: "4",
|
|
status: $status,
|
|
groupKey: $groupKey,
|
|
groupLabels: { alertname: $alertname, team: $team },
|
|
alerts: [{
|
|
status: $status,
|
|
labels: { alertname: $alertname, severity: $severity, team: $team, instance: "demo" },
|
|
annotations: { summary: $summary },
|
|
startsAt: $startsAt,
|
|
endsAt: $endsAt,
|
|
generatorURL: "https://example.com/demo",
|
|
fingerprint: $fingerprint
|
|
}]
|
|
}')"
|
|
|
|
url="${BASE_URL}/api/integrations/${key}/alertmanager"
|
|
echo "POST $url (team=$team scenario=$scenario status=$status)" >&2
|
|
code="$(curl -sS -o /tmp/fire-alerts-response.json -w '%{http_code}' \
|
|
-X POST "$url" -H 'Content-Type: application/json' -d "$payload")"
|
|
echo "-> HTTP $code" >&2
|
|
cat /tmp/fire-alerts-response.json >&2
|
|
echo >&2
|
|
|
|
if [ "$code" != "200" ]; then
|
|
exit 1
|
|
fi
|