Add examples/demo: one of every CRD, plus a script to fire alerts at it
A self-contained demo kit: a TerdutServer against a throwaway, bare Postgres (bring-your-own DSN -- simplest path to stand up from nothing, ROADMAP.md Stage 1's own note), two TerdutTeams, and each team's own TerdutEscalationRule/TerdutDeadmanSwitch/TerdutAlertSource, so every CRD this operator manages is exercised together rather than in isolation the way config/samples' one-of-each already does. fire-alerts.sh sends terdut-server's own amPayload/amAlert shape (read from internal/api/alertmanager.go in that repo, not guessed from its docs) at whichever TerdutAlertSource's generated webhook Secret it reads the key out of -- high-cpu/disk-full/pod-crash scenarios to open and resolve incidents, and a heartbeat scenario matching each team's dead man's switch matcher, so stopping it demonstrates the switch noticing silence on its own. Verified server-side (kubectl apply --dry-run=server -k examples/demo) against this operator's own dev cluster, which already has these CRDs installed: every object validates. The one warning that cluster's "restricted" PodSecurity raises (postgres:17-alpine's entrypoint needs to start as root before it drops privileges itself) is noted inline in 00-postgres.yaml rather than worked around -- not a real production pattern, and this Postgres exists only to be thrown away with the rest of the demo namespace. README.md walks through: applying, watching status, why a few early CrashLoopBackOff restarts on terdut-demo itself are expected (this operator's Deployment template has no wait-for-postgres init container yet, unlike charts/terdut-server's chart as of v0.33.2), reaching the web UI (port-forward -- spec.networking.hostname is accepted but nothing creates an HTTPRoute for it yet), turning on open signup with the operator's own generated admin token since the bootstrap-created account has no password, firing alerts, and tearing down.
This commit is contained in:
@@ -0,0 +1,91 @@
|
||||
# Demo-only Postgres: a bare Deployment+Service+Secret, not the Zalando
|
||||
# postgres-operator path (DatabaseSpec.postgresClusterRef, DESIGN.md §8).
|
||||
# Bring-your-own DSN is the simpler of the two paths to stand up from
|
||||
# nothing (ROADMAP.md Stage 1's own note), which is all this needs to be.
|
||||
#
|
||||
# emptyDir, one replica, a password sitting in a plaintext Secret below --
|
||||
# none of that is how you'd run Postgres for real. It exists only so
|
||||
# 01-server.yaml has something to talk to. Throw the whole demo namespace
|
||||
# away when you're done; nothing here is meant to survive that.
|
||||
apiVersion: v1
|
||||
kind: Secret
|
||||
metadata:
|
||||
name: terdut-demo-postgres
|
||||
type: Opaque
|
||||
stringData:
|
||||
password: demo-not-a-real-password
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: terdut-demo-postgres
|
||||
labels:
|
||||
app: terdut-demo-postgres
|
||||
spec:
|
||||
replicas: 1
|
||||
# Recreate, not RollingUpdate: emptyDir means a new pod starts with an
|
||||
# empty database anyway, and two Postgres pods would never agree on one
|
||||
# emptyDir each.
|
||||
strategy:
|
||||
type: Recreate
|
||||
selector:
|
||||
matchLabels:
|
||||
app: terdut-demo-postgres
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: terdut-demo-postgres
|
||||
spec:
|
||||
containers:
|
||||
- name: postgres
|
||||
image: postgres:17-alpine
|
||||
# Partial, deliberately: the official image's entrypoint needs to
|
||||
# start as root to chown the data directory before it drops
|
||||
# privileges itself (gosu, to the postgres user) -- forcing
|
||||
# runAsNonRoot here would just refuse to start the container. A
|
||||
# "restricted" PodSecurity namespace warns on that gap rather
|
||||
# than blocking (confirmed server-side against this operator's
|
||||
# own dev cluster), which is an acceptable tradeoff for Postgres
|
||||
# that exists only to be thrown away with the rest of this demo.
|
||||
securityContext:
|
||||
allowPrivilegeEscalation: false
|
||||
capabilities:
|
||||
drop: ["ALL"]
|
||||
seccompProfile:
|
||||
type: RuntimeDefault
|
||||
ports:
|
||||
- name: postgres
|
||||
containerPort: 5432
|
||||
env:
|
||||
- name: POSTGRES_USER
|
||||
value: terdut
|
||||
- name: POSTGRES_DB
|
||||
value: terdut
|
||||
- name: POSTGRES_PASSWORD
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: terdut-demo-postgres
|
||||
key: password
|
||||
volumeMounts:
|
||||
- name: data
|
||||
mountPath: /var/lib/postgresql/data
|
||||
subPath: pgdata
|
||||
readinessProbe:
|
||||
exec:
|
||||
command: ["pg_isready", "-U", "terdut"]
|
||||
initialDelaySeconds: 5
|
||||
volumes:
|
||||
- name: data
|
||||
emptyDir: {}
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: terdut-demo-postgres
|
||||
spec:
|
||||
selector:
|
||||
app: terdut-demo-postgres
|
||||
ports:
|
||||
- name: postgres
|
||||
port: 5432
|
||||
targetPort: postgres
|
||||
@@ -0,0 +1,32 @@
|
||||
# The one TerdutServer this whole demo runs against. Everything else in
|
||||
# this directory (teams, escalation rules, dead man's switches, alert
|
||||
# sources) references it by name.
|
||||
#
|
||||
# networking.hostname is accepted but not yet acted on: creating the
|
||||
# HTTPRoute for it isn't implemented yet (api/v1alpha1/terdutserver_types.go,
|
||||
# NetworkingSpec's own doc comment) -- this TerdutServer is reachable from
|
||||
# outside the cluster only by port-forwarding its Service, same name as
|
||||
# this object (see README.md).
|
||||
apiVersion: terdut.ryuvia.com/v1alpha1
|
||||
kind: TerdutServer
|
||||
metadata:
|
||||
name: terdut-demo
|
||||
spec:
|
||||
image:
|
||||
repository: git.ryuvia.com/niklas/terdut-server
|
||||
tag: v0.33.2
|
||||
replicas: 1
|
||||
networking:
|
||||
hostname: terdut-demo.example
|
||||
servicePort: 8080
|
||||
database:
|
||||
dsn: "postgres://terdut@terdut-demo-postgres:5432/terdut?sslmode=disable"
|
||||
passwordSecretRef:
|
||||
name: terdut-demo-postgres
|
||||
key: password
|
||||
sweeper:
|
||||
staleAfter: 6h
|
||||
archiveAfter: 168h
|
||||
# No oidc block: password login only, so there's nothing external to
|
||||
# register a redirect URI with before this demo can sign in.
|
||||
passwordLogin: true
|
||||
@@ -0,0 +1,17 @@
|
||||
# Two teams (this one and 03-team-payments.yaml) so the demo shows
|
||||
# per-team isolation -- separate incident lists, separate escalation
|
||||
# ladders, separate alert sources -- rather than one team standing in for
|
||||
# everything.
|
||||
apiVersion: terdut.ryuvia.com/v1alpha1
|
||||
kind: TerdutTeam
|
||||
metadata:
|
||||
name: terdutteam-platform
|
||||
spec:
|
||||
# serverRef.namespace omitted: both this and terdut-demo (01-server.yaml)
|
||||
# live in whatever namespace you apply this directory into, which is the
|
||||
# common case and needs no allowedTeams consent on the TerdutServer side
|
||||
# (DESIGN.md §4.1, §4.6).
|
||||
serverRef:
|
||||
name: terdut-demo
|
||||
displayName: Platform
|
||||
# No oidc block: this demo is password-login only (01-server.yaml).
|
||||
@@ -0,0 +1,8 @@
|
||||
apiVersion: terdut.ryuvia.com/v1alpha1
|
||||
kind: TerdutTeam
|
||||
metadata:
|
||||
name: terdutteam-payments
|
||||
spec:
|
||||
serverRef:
|
||||
name: terdut-demo
|
||||
displayName: Payments
|
||||
@@ -0,0 +1,24 @@
|
||||
# One per team is the rule (DESIGN.md §4.3) -- a second TerdutEscalationRule
|
||||
# naming the same teamRef would just clobber this one on the next reconcile.
|
||||
apiVersion: terdut.ryuvia.com/v1alpha1
|
||||
kind: TerdutEscalationRule
|
||||
metadata:
|
||||
name: terdutescalationrule-platform
|
||||
spec:
|
||||
teamRef:
|
||||
name: terdutteam-platform
|
||||
repeatCount: 2
|
||||
fallbackTopic: platform-fallback
|
||||
levels:
|
||||
# username is required iff kind is "user", rejected otherwise -- CEL
|
||||
# validation at apply time (api/v1alpha1/terdutescalationrule_types.go).
|
||||
# alice won't exist on a fresh demo install -- see README.md for
|
||||
# creating a real user if you want this level to mean something, or
|
||||
# just watch it fall through to oncall after 5m.
|
||||
- timeout: 5m
|
||||
targets:
|
||||
- kind: user
|
||||
username: alice
|
||||
- timeout: 10m
|
||||
targets:
|
||||
- kind: oncall
|
||||
@@ -0,0 +1,13 @@
|
||||
apiVersion: terdut.ryuvia.com/v1alpha1
|
||||
kind: TerdutEscalationRule
|
||||
metadata:
|
||||
name: terdutescalationrule-payments
|
||||
spec:
|
||||
teamRef:
|
||||
name: terdutteam-payments
|
||||
repeatCount: 1
|
||||
fallbackTopic: payments-fallback
|
||||
levels:
|
||||
- timeout: 5m
|
||||
targets:
|
||||
- kind: oncall
|
||||
@@ -0,0 +1,17 @@
|
||||
# A per-team dead man's switch (DESIGN.md §4.4) -- a different thing from
|
||||
# 01-server.yaml's spec.deadman, which this demo leaves unset so this CRD
|
||||
# is what you're actually seeing reconcile. fire-alerts.sh's "heartbeat"
|
||||
# scenario sends a matching alert; stop sending it and terdut-server
|
||||
# itself opens an incident once `timeout` passes with no heartbeat.
|
||||
apiVersion: terdut.ryuvia.com/v1alpha1
|
||||
kind: TerdutDeadmanSwitch
|
||||
metadata:
|
||||
name: terdutdeadmanswitch-platform
|
||||
spec:
|
||||
teamRef:
|
||||
name: terdutteam-platform
|
||||
# name omitted -- terdut-server derives one from the matcher's own
|
||||
# canonical form (DESIGN.md §4.4).
|
||||
matcher: "alertname=PlatformWatchdog"
|
||||
timeout: 15m
|
||||
severity: critical
|
||||
@@ -0,0 +1,10 @@
|
||||
apiVersion: terdut.ryuvia.com/v1alpha1
|
||||
kind: TerdutDeadmanSwitch
|
||||
metadata:
|
||||
name: terdutdeadmanswitch-payments
|
||||
spec:
|
||||
teamRef:
|
||||
name: terdutteam-payments
|
||||
matcher: "alertname=PaymentsWatchdog"
|
||||
timeout: 15m
|
||||
severity: critical
|
||||
@@ -0,0 +1,18 @@
|
||||
# The webhook URL/key fire-alerts.sh sends to, for the Platform team.
|
||||
# terdut-server shows the key exactly once, at creation, and never again
|
||||
# (DESIGN.md §4.5) -- this object's status.webhookURLSecretRef names the
|
||||
# generated Secret holding it (keys "url" and "key"), which is what
|
||||
# fire-alerts.sh reads. See README.md before applying this: it's the one
|
||||
# object in this directory whose Secret you can't just re-read if you
|
||||
# miss it.
|
||||
apiVersion: terdut.ryuvia.com/v1alpha1
|
||||
kind: TerdutAlertSource
|
||||
metadata:
|
||||
name: terdutalertsource-platform
|
||||
spec:
|
||||
teamRef:
|
||||
name: terdutteam-platform
|
||||
# kind defaults to "alertmanager" -- the only value terdut-server
|
||||
# supports today.
|
||||
kind: alertmanager
|
||||
name: platform-demo-alertmanager
|
||||
@@ -0,0 +1,9 @@
|
||||
apiVersion: terdut.ryuvia.com/v1alpha1
|
||||
kind: TerdutAlertSource
|
||||
metadata:
|
||||
name: terdutalertsource-payments
|
||||
spec:
|
||||
teamRef:
|
||||
name: terdutteam-payments
|
||||
kind: alertmanager
|
||||
name: payments-demo-alertmanager
|
||||
@@ -0,0 +1,139 @@
|
||||
# Demo
|
||||
|
||||
Every CRD this operator reconciles, wired into one working install: one
|
||||
`TerdutServer`, two `TerdutTeam`s (Platform and Payments) each with their
|
||||
own `TerdutEscalationRule`, `TerdutDeadmanSwitch` and `TerdutAlertSource`,
|
||||
plus a script that fires synthetic Alertmanager webhooks at it so you can
|
||||
watch real incidents appear, escalate and resolve.
|
||||
|
||||
This is a demo kit, not a reference deployment: `00-postgres.yaml` runs
|
||||
Postgres with `emptyDir` storage and a password committed in this
|
||||
directory. Throw the whole namespace away when you're done.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- The operator and its CRDs installed and running (`make install
|
||||
deploy IMG=...`, or `charts/terdut-operator` — see this repo's own
|
||||
README.md/DESIGN.md), pointed at a cluster you're fine creating
|
||||
throwaway resources in. A `kind` cluster is the easy choice.
|
||||
- `kubectl`, `jq`, `curl` on your path.
|
||||
|
||||
## Apply it
|
||||
|
||||
```sh
|
||||
kubectl create namespace terdut-operator-demo
|
||||
kubectl apply -n terdut-operator-demo -k .
|
||||
```
|
||||
|
||||
Listed and applied in dependency order (server → team → everything that
|
||||
`teamRef`s it), but you don't have to preserve that order yourself:
|
||||
every controller here re-queues and waits rather than failing when a ref
|
||||
isn't resolvable yet (`kubectl describe` shows `Reason: TeamRefNotFound` /
|
||||
`WaitingForTeam` while that settles).
|
||||
|
||||
Watch it converge:
|
||||
|
||||
```sh
|
||||
kubectl get terdutservers,terdutteams,terdutescalationrules,terdutdeadmanswitches,terdutalertsources \
|
||||
-n terdut-operator-demo
|
||||
```
|
||||
|
||||
**A few early `CrashLoopBackOff` restarts on the `terdut-demo` pod are
|
||||
expected**, not a sign anything is wrong: this is a brand-new Postgres
|
||||
doing its very first boot, and the operator's own Deployment template has
|
||||
no wait-for-postgres step yet (unlike `charts/terdut-server`'s chart as of
|
||||
v0.33.2 — see that repo's `CLAUDE.md`/release notes for why) — the pod
|
||||
restarts a couple of times until Postgres is actually accepting
|
||||
connections, then stays up. Carrying that chart's fix into this operator's
|
||||
own Deployment template is still open.
|
||||
|
||||
Once `terdut-demo`'s own `Ready` condition is `True`, everything downstream
|
||||
of it should settle within a reconcile interval or two.
|
||||
|
||||
## See the web UI
|
||||
|
||||
The operator doesn't create any external exposure yet
|
||||
(`NetworkingSpec`'s own doc comment in `api/v1alpha1/terdutserver_types.go`
|
||||
— `spec.networking.hostname` is accepted but nothing acts on it), so:
|
||||
|
||||
```sh
|
||||
kubectl -n terdut-operator-demo port-forward svc/terdut-demo 8080:8080
|
||||
```
|
||||
|
||||
and open http://localhost:8080.
|
||||
|
||||
### First login
|
||||
|
||||
The operator's own bootstrap (DESIGN.md §6) creates the first user through
|
||||
`/api/bootstrap` and immediately mints itself a service-account token from
|
||||
it — that account has no password, so there's nothing to sign in with yet.
|
||||
`signup_mode` also defaults to `invite_only`, so open signup needs turning
|
||||
on first, using the admin token the operator generated for itself:
|
||||
|
||||
```sh
|
||||
# Which namespace the operator itself runs in:
|
||||
kubectl get deploy -A -l control-plane=controller-manager
|
||||
|
||||
# The Secret holding the operator's own admin token for this TerdutServer
|
||||
# (cross-namespace from terdut-operator-demo, per DESIGN.md §7):
|
||||
secretname=$(kubectl -n terdut-operator-demo get terdutserver terdut-demo \
|
||||
-o jsonpath='{.status.credentialsSecretRef.name}')
|
||||
token=$(kubectl -n <operator-namespace-from-above> get secret "$secretname" \
|
||||
-o jsonpath='{.data.token}' | base64 -d)
|
||||
|
||||
curl -X PUT http://localhost:8080/api/admin/settings \
|
||||
-H "Authorization: Bearer $token" -H 'Content-Type: application/json' \
|
||||
-d '{"signup_mode":"open"}'
|
||||
```
|
||||
|
||||
Then sign up through the UI as a normal human account. `04-escalation-platform.yaml`
|
||||
names a user `alice` at its first escalation level — sign up as `alice` if
|
||||
you want that level to mean something rather than falling through to
|
||||
on-call after 5 minutes.
|
||||
|
||||
## Fire some alerts
|
||||
|
||||
In another terminal, with the port-forward above still running:
|
||||
|
||||
```sh
|
||||
export NAMESPACE=terdut-operator-demo
|
||||
|
||||
./fire-alerts.sh platform high-cpu
|
||||
./fire-alerts.sh platform disk-full
|
||||
./fire-alerts.sh payments pod-crash
|
||||
|
||||
# Watch it open an incident, escalate per 04/05-escalation-*.yaml's
|
||||
# levels, and show up on the Platform/Payments team's own incident list.
|
||||
|
||||
./fire-alerts.sh platform high-cpu resolve
|
||||
```
|
||||
|
||||
`fire-alerts.sh -h` (or any bad argument) prints the full scenario list.
|
||||
Each `(team, scenario)` pair is one stable fingerprint, so firing the same
|
||||
one twice updates the same alert (a real re-fire) and `resolve` closes
|
||||
exactly that one.
|
||||
|
||||
### Dead man's switches
|
||||
|
||||
`06-deadman-platform.yaml` / `07-deadman-payments.yaml` expect a heartbeat
|
||||
alert on a 15-minute timeout:
|
||||
|
||||
```sh
|
||||
./fire-alerts.sh platform heartbeat
|
||||
```
|
||||
|
||||
Keep sending that (e.g. a `watch -n 60`) and nothing happens — that's the
|
||||
point. Stop sending it and, 15 minutes after the last one, terdut-server
|
||||
opens a `critical` incident on its own, with no webhook involved: proof
|
||||
the switch is watching for silence, not for a signal.
|
||||
|
||||
## Tear down
|
||||
|
||||
```sh
|
||||
kubectl delete namespace terdut-operator-demo
|
||||
```
|
||||
|
||||
The operator's own finalizers clean up everything cross-namespace
|
||||
(credentials Secrets in the operator's namespace, server-side team/rule/
|
||||
integration rows) before this namespace's objects actually disappear —
|
||||
give it a few seconds past the `kubectl delete` returning.
|
||||
Executable
+138
@@ -0,0 +1,138 @@
|
||||
#!/usr/bin/env bash
|
||||
# Sends a synthetic Alertmanager v4 webhook payload at the "platform" or
|
||||
# "payments" demo team's TerdutAlertSource, so terdut-server opens (or
|
||||
# resolves) an incident exactly the way it would for a real Alertmanager.
|
||||
#
|
||||
# The payload shape here is amPayload/amAlert, read straight out of
|
||||
# terdut-server's own internal/api/alertmanager.go rather than guessed from
|
||||
# its docs -- version/status/groupKey/groupLabels, and alerts[] carrying
|
||||
# status/labels/annotations/startsAt/endsAt/generatorURL/fingerprint.
|
||||
#
|
||||
# Why this reads the webhook key out of a kubectl Secret instead of using
|
||||
# the "url" key already in it: that URL is built from spec.networking.hostname
|
||||
# (TERDUT_PUBLIC_URL), and nothing in this demo stands up real ingress for
|
||||
# it (01-server.yaml's own comment) -- so it resolves nowhere. The key
|
||||
# alone, against whatever you've actually port-forwarded BASE_URL to below,
|
||||
# is the one part of that URL still usable here.
|
||||
#
|
||||
# Usage:
|
||||
# ./fire-alerts.sh <platform|payments> <high-cpu|disk-full|pod-crash|heartbeat> [resolve]
|
||||
#
|
||||
# Prerequisites: kubectl context pointed at the demo namespace, jq, curl,
|
||||
# and (in another terminal) a running:
|
||||
# kubectl port-forward svc/terdut-demo 8080:8080
|
||||
set -euo pipefail
|
||||
|
||||
NAMESPACE="${NAMESPACE:-}"
|
||||
BASE_URL="${BASE_URL:-http://localhost:8080}"
|
||||
|
||||
usage() {
|
||||
cat >&2 <<'EOF'
|
||||
usage: fire-alerts.sh <platform|payments> <scenario> [resolve]
|
||||
|
||||
scenarios:
|
||||
high-cpu warning -- CPU usage above 90% for 10 minutes
|
||||
disk-full critical -- disk usage above 95%
|
||||
pod-crash error -- a pod crash-looping
|
||||
heartbeat critical -- the team's dead man's switch heartbeat
|
||||
(matches the matcher in 06/07-deadman-*.yaml -- send this
|
||||
repeatedly to keep the switch alive, or stop sending it and
|
||||
watch terdut-server open an incident on its own once
|
||||
`timeout` passes with no heartbeat. "resolve" is not a valid
|
||||
third argument for this scenario: a heartbeat is only ever
|
||||
firing.)
|
||||
|
||||
env vars:
|
||||
NAMESPACE kubectl -n for reading the webhook Secret (required)
|
||||
BASE_URL where the port-forwarded terdut-server is (default http://localhost:8080)
|
||||
EOF
|
||||
exit 1
|
||||
}
|
||||
|
||||
[ $# -ge 2 ] || usage
|
||||
team="$1" scenario="$2" verb="${3:-fire}"
|
||||
[ -n "$NAMESPACE" ] || { echo "fire-alerts.sh: set NAMESPACE" >&2; exit 1; }
|
||||
|
||||
case "$team" in
|
||||
platform|payments) ;;
|
||||
*) usage ;;
|
||||
esac
|
||||
|
||||
case "$scenario" in
|
||||
high-cpu) alertname=TerdutDemoHighCPU severity=warning summary="CPU usage above 90% for 10 minutes" ;;
|
||||
disk-full) alertname=TerdutDemoDiskFull severity=critical summary="Disk usage above 95% on /data" ;;
|
||||
pod-crash) alertname=TerdutDemoPodCrashLooping severity=error summary="Pod web-7f8b9 is crash-looping (5 restarts in 10m)" ;;
|
||||
heartbeat)
|
||||
# Must match 06-deadman-platform.yaml / 07-deadman-payments.yaml's own
|
||||
# matcher exactly -- that's what makes this a heartbeat rather than a
|
||||
# third ordinary alert.
|
||||
case "$team" in
|
||||
platform) alertname=PlatformWatchdog ;;
|
||||
payments) alertname=PaymentsWatchdog ;;
|
||||
esac
|
||||
severity=critical summary="demo heartbeat"
|
||||
[ "$verb" = fire ] || { echo "fire-alerts.sh: heartbeat is only ever fired, never resolved -- just stop sending it" >&2; exit 1; }
|
||||
;;
|
||||
*) usage ;;
|
||||
esac
|
||||
|
||||
case "$verb" in
|
||||
fire) status=firing ;;
|
||||
resolve) status=resolved ;;
|
||||
*) usage ;;
|
||||
esac
|
||||
|
||||
secret_name="terdutalertsource-${team}-terdut-webhook"
|
||||
key="$(kubectl -n "$NAMESPACE" get secret "$secret_name" -o jsonpath='{.data.key}' | base64 -d)"
|
||||
[ -n "$key" ] || { echo "fire-alerts.sh: empty key read from Secret $secret_name -- has 08/09-alertsource-*.yaml reconciled yet?" >&2; exit 1; }
|
||||
|
||||
# Stable per (team, scenario) so a resolve targets the same alert a fire
|
||||
# created: terdut-server correlates on (team_id, fingerprint), not on
|
||||
# anything else in the payload. Real Alertmanager computes this from the
|
||||
# alert's label set; a fixed string plays the same role here.
|
||||
fingerprint="demo-${team}-${scenario}"
|
||||
|
||||
now="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
|
||||
if [ "$status" = firing ]; then
|
||||
ends_at="0001-01-01T00:00:00Z" # Alertmanager's own "not resolved" zero value
|
||||
else
|
||||
ends_at="$now"
|
||||
fi
|
||||
|
||||
payload="$(jq -n \
|
||||
--arg status "$status" \
|
||||
--arg groupKey "demo:${team}:${scenario}" \
|
||||
--arg alertname "$alertname" \
|
||||
--arg team "$team" \
|
||||
--arg severity "$severity" \
|
||||
--arg summary "$summary" \
|
||||
--arg startsAt "$now" \
|
||||
--arg endsAt "$ends_at" \
|
||||
--arg fingerprint "$fingerprint" \
|
||||
'{
|
||||
version: "4",
|
||||
status: $status,
|
||||
groupKey: $groupKey,
|
||||
groupLabels: { alertname: $alertname, team: $team },
|
||||
alerts: [{
|
||||
status: $status,
|
||||
labels: { alertname: $alertname, severity: $severity, team: $team, instance: "demo" },
|
||||
annotations: { summary: $summary },
|
||||
startsAt: $startsAt,
|
||||
endsAt: $endsAt,
|
||||
generatorURL: "https://example.com/demo",
|
||||
fingerprint: $fingerprint
|
||||
}]
|
||||
}')"
|
||||
|
||||
url="${BASE_URL}/api/integrations/${key}/alertmanager"
|
||||
echo "POST $url (team=$team scenario=$scenario status=$status)" >&2
|
||||
code="$(curl -sS -o /tmp/fire-alerts-response.json -w '%{http_code}' \
|
||||
-X POST "$url" -H 'Content-Type: application/json' -d "$payload")"
|
||||
echo "-> HTTP $code" >&2
|
||||
cat /tmp/fire-alerts-response.json >&2
|
||||
echo >&2
|
||||
|
||||
if [ "$code" != "200" ]; then
|
||||
exit 1
|
||||
fi
|
||||
@@ -0,0 +1,17 @@
|
||||
## kubectl apply -n <your-demo-namespace> -k examples/demo
|
||||
##
|
||||
## Listed in apply order even though kustomize itself doesn't need that --
|
||||
## a human reading this file top-to-bottom should see the same dependency
|
||||
## order the controllers themselves require (server before team, team
|
||||
## before everything that teamRefs it).
|
||||
resources:
|
||||
- 00-postgres.yaml
|
||||
- 01-server.yaml
|
||||
- 02-team-platform.yaml
|
||||
- 03-team-payments.yaml
|
||||
- 04-escalation-platform.yaml
|
||||
- 05-escalation-payments.yaml
|
||||
- 06-deadman-platform.yaml
|
||||
- 07-deadman-payments.yaml
|
||||
- 08-alertsource-platform.yaml
|
||||
- 09-alertsource-payments.yaml
|
||||
Reference in New Issue
Block a user