Compare commits
5 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 50ce5bcec0 | |||
| 822c80dda6 | |||
| 0ee7ede648 | |||
| d50248531c | |||
| 4007f54279 |
@@ -153,7 +153,7 @@ spec:
|
|||||||
image:
|
image:
|
||||||
repository: git.ryuvia.com/niklas/terdut-server
|
repository: git.ryuvia.com/niklas/terdut-server
|
||||||
tag: v0.9.3
|
tag: v0.9.3
|
||||||
replicas: 1 # terdut-server is not horizontally-scale-tested; keep the field, default 1
|
replicas: 2 # default since terdut-server v0.36.0's advisory locks; see TerdutServerSpec.Replicas
|
||||||
networking:
|
networking:
|
||||||
hostname: terdut.example.com
|
hostname: terdut.example.com
|
||||||
servicePort: 8080
|
servicePort: 8080
|
||||||
@@ -244,10 +244,10 @@ no custom wrapper buys anything for any of these, matching how
|
|||||||
CloudNativePG and the Zalando postgres-operator both expose the same
|
CloudNativePG and the Zalando postgres-operator both expose the same
|
||||||
knobs. `affinity` is pure user-supplied passthrough, not a
|
knobs. `affinity` is pure user-supplied passthrough, not a
|
||||||
toggle-plus-generated-default the way a multi-replica-aware operator's
|
toggle-plus-generated-default the way a multi-replica-aware operator's
|
||||||
pod anti-affinity typically is: this operator never auto-generates
|
pod anti-affinity typically is: even though `replicas` now defaults to 2
|
||||||
affinity of its own, since `replicas` above 1 isn't a supported topology
|
(terdut-server v0.36.0's advisory locks made that safe, §4.1's own
|
||||||
(the sweeper/notifier singleton constraint, §4.1's own illustrative YAML
|
illustrative YAML comment), this operator still never auto-generates
|
||||||
comment). `spec.pod.disruptionBudget` is the one field here that isn't a
|
affinity of its own. `spec.pod.disruptionBudget` is the one field here that isn't a
|
||||||
straight PodTemplateSpec knob — when set, the controller reconciles a
|
straight PodTemplateSpec knob — when set, the controller reconciles a
|
||||||
`PodDisruptionBudget` selecting this `TerdutServer`'s pods; clearing it
|
`PodDisruptionBudget` selecting this `TerdutServer`'s pods; clearing it
|
||||||
deletes any it previously created (§7). `minAvailable`/`maxUnavailable`
|
deletes any it previously created (§7). `minAvailable`/`maxUnavailable`
|
||||||
@@ -832,10 +832,12 @@ what it was, a separate install, until someone deletes it.
|
|||||||
- Automatic Deployment restart on upstream Postgres credential rotation.
|
- Automatic Deployment restart on upstream Postgres credential rotation.
|
||||||
- `spec.pod.priorityClassName`, pod-label passthrough beyond
|
- `spec.pod.priorityClassName`, pod-label passthrough beyond
|
||||||
`spec.pod.annotations`, and a HorizontalPodAutoscaler for `TerdutServer`
|
`spec.pod.annotations`, and a HorizontalPodAutoscaler for `TerdutServer`
|
||||||
— all considered alongside §4.1's `spec.pod` and explicitly left out of
|
— all considered alongside §4.1's `spec.pod` and left out of that round.
|
||||||
that round: an HPA in particular would actively contradict
|
An HPA no longer contradicts anything now that `spec.replicas` defaults
|
||||||
`spec.replicas`'s own stance that this operator doesn't support more
|
to 2 (terdut-server v0.36.0's advisory locks), but it is still a
|
||||||
than one replica (the sweeper/notifier singleton constraint).
|
separate, not-yet-made decision: a fixed replica count has no scaling
|
||||||
|
metric, min/max bounds, or cooldown behaviour to get right, and nobody
|
||||||
|
has asked for it yet.
|
||||||
- Admission webhooks / CEL-only validation limits (e.g. verifying a
|
- Admission webhooks / CEL-only validation limits (e.g. verifying a
|
||||||
`teamRef` exists at admission time rather than surfacing it as a status
|
`teamRef` exists at admission time rather than surfacing it as a status
|
||||||
condition after the fact).
|
condition after the fact).
|
||||||
|
|||||||
@@ -229,11 +229,11 @@ type PodSpec struct {
|
|||||||
Tolerations []corev1.Toleration `json:"tolerations,omitempty"`
|
Tolerations []corev1.Toleration `json:"tolerations,omitempty"`
|
||||||
|
|
||||||
// affinity covers node affinity, pod affinity and pod anti-affinity in
|
// affinity covers node affinity, pod affinity and pod anti-affinity in
|
||||||
// one field -- unlike a multi-replica-aware operator, this one never
|
// one field -- even though replicas now defaults to 2 (see
|
||||||
// generates a default anti-affinity itself (replicas above 1 isn't a
|
// TerdutServerSpec.Replicas's own doc comment), this operator still
|
||||||
// supported topology, see TerdutServerSpec.Replicas's own doc comment),
|
// never generates a default anti-affinity of its own the way a
|
||||||
// so this is pure user-supplied passthrough, not a toggle-plus-generated-
|
// multi-replica-aware operator typically would, so this stays pure
|
||||||
// default.
|
// user-supplied passthrough, not a toggle-plus-generated-default.
|
||||||
// +optional
|
// +optional
|
||||||
Affinity *corev1.Affinity `json:"affinity,omitempty"`
|
Affinity *corev1.Affinity `json:"affinity,omitempty"`
|
||||||
|
|
||||||
@@ -301,10 +301,14 @@ type TerdutServerSpec struct {
|
|||||||
// +required
|
// +required
|
||||||
Image ImageSpec `json:"image"`
|
Image ImageSpec `json:"image"`
|
||||||
|
|
||||||
// replicas. terdut-server is not horizontally-scale-tested; keep this
|
// replicas. Defaults to 2: terdut-server v0.36.0 put the sweeper, the
|
||||||
// at its default of 1 unless you've verified otherwise -- the sweeper
|
// notifier and the migration runner each behind a Postgres advisory
|
||||||
// and the notifier are unsynchronised singletons.
|
// lock, and gave incident creation its own conflict resolution, so
|
||||||
// +kubebuilder:default=1
|
// more than one replica no longer double-pages, races a migration, or
|
||||||
|
// drops a webhook payload. image.tag must be v0.36.0 or newer for
|
||||||
|
// that to hold -- an older terdut-server has none of these guards,
|
||||||
|
// and this field does not check the tag for you.
|
||||||
|
// +kubebuilder:default=2
|
||||||
// +optional
|
// +optional
|
||||||
Replicas int32 `json:"replicas,omitempty"`
|
Replicas int32 `json:"replicas,omitempty"`
|
||||||
|
|
||||||
|
|||||||
@@ -6,8 +6,8 @@ type: application
|
|||||||
# These fields decide nothing: `make helm-package` passes --version and
|
# These fields decide nothing: `make helm-package` passes --version and
|
||||||
# --app-version from the release tag (same reasoning as terdut-server's own
|
# --app-version from the release tag (same reasoning as terdut-server's own
|
||||||
# chart). They're for whoever reads the tree before a tag exists.
|
# chart). They're for whoever reads the tree before a tag exists.
|
||||||
version: 0.3.0
|
version: 0.4.0
|
||||||
appVersion: "v0.3.0"
|
appVersion: "v0.4.0"
|
||||||
|
|
||||||
keywords:
|
keywords:
|
||||||
- kubernetes
|
- kubernetes
|
||||||
|
|||||||
@@ -327,11 +327,11 @@ spec:
|
|||||||
affinity:
|
affinity:
|
||||||
description: |-
|
description: |-
|
||||||
affinity covers node affinity, pod affinity and pod anti-affinity in
|
affinity covers node affinity, pod affinity and pod anti-affinity in
|
||||||
one field -- unlike a multi-replica-aware operator, this one never
|
one field -- even though replicas now defaults to 2 (see
|
||||||
generates a default anti-affinity itself (replicas above 1 isn't a
|
TerdutServerSpec.Replicas's own doc comment), this operator still
|
||||||
supported topology, see TerdutServerSpec.Replicas's own doc comment),
|
never generates a default anti-affinity of its own the way a
|
||||||
so this is pure user-supplied passthrough, not a toggle-plus-generated-
|
multi-replica-aware operator typically would, so this stays pure
|
||||||
default.
|
user-supplied passthrough, not a toggle-plus-generated-default.
|
||||||
properties:
|
properties:
|
||||||
nodeAffinity:
|
nodeAffinity:
|
||||||
description: Describes node affinity scheduling rules for
|
description: Describes node affinity scheduling rules for
|
||||||
@@ -4304,11 +4304,15 @@ spec:
|
|||||||
type: array
|
type: array
|
||||||
type: object
|
type: object
|
||||||
replicas:
|
replicas:
|
||||||
default: 1
|
default: 2
|
||||||
description: |-
|
description: |-
|
||||||
replicas. terdut-server is not horizontally-scale-tested; keep this
|
replicas. Defaults to 2: terdut-server v0.36.0 put the sweeper, the
|
||||||
at its default of 1 unless you've verified otherwise -- the sweeper
|
notifier and the migration runner each behind a Postgres advisory
|
||||||
and the notifier are unsynchronised singletons.
|
lock, and gave incident creation its own conflict resolution, so
|
||||||
|
more than one replica no longer double-pages, races a migration, or
|
||||||
|
drops a webhook payload. image.tag must be v0.36.0 or newer for
|
||||||
|
that to hold -- an older terdut-server has none of these guards,
|
||||||
|
and this field does not check the tag for you.
|
||||||
format: int32
|
format: int32
|
||||||
type: integer
|
type: integer
|
||||||
sweeper:
|
sweeper:
|
||||||
|
|||||||
@@ -233,7 +233,11 @@ terdutServer:
|
|||||||
## Required when terdutServer.enabled.
|
## Required when terdutServer.enabled.
|
||||||
# tag: ""
|
# tag: ""
|
||||||
|
|
||||||
replicas: 1
|
## Safe above 1 since terdut-server v0.36.0 (image.tag above must be that or
|
||||||
|
## newer): the sweeper, notifier and migration runner are each behind a
|
||||||
|
## Postgres advisory lock, and incident creation resolves its own insert
|
||||||
|
## conflict, matching this CRD's own spec.replicas default.
|
||||||
|
replicas: 2
|
||||||
|
|
||||||
networking:
|
networking:
|
||||||
## Required when terdutServer.enabled -- terdut-server's own public
|
## Required when terdutServer.enabled -- terdut-server's own public
|
||||||
|
|||||||
@@ -324,11 +324,11 @@ spec:
|
|||||||
affinity:
|
affinity:
|
||||||
description: |-
|
description: |-
|
||||||
affinity covers node affinity, pod affinity and pod anti-affinity in
|
affinity covers node affinity, pod affinity and pod anti-affinity in
|
||||||
one field -- unlike a multi-replica-aware operator, this one never
|
one field -- even though replicas now defaults to 2 (see
|
||||||
generates a default anti-affinity itself (replicas above 1 isn't a
|
TerdutServerSpec.Replicas's own doc comment), this operator still
|
||||||
supported topology, see TerdutServerSpec.Replicas's own doc comment),
|
never generates a default anti-affinity of its own the way a
|
||||||
so this is pure user-supplied passthrough, not a toggle-plus-generated-
|
multi-replica-aware operator typically would, so this stays pure
|
||||||
default.
|
user-supplied passthrough, not a toggle-plus-generated-default.
|
||||||
properties:
|
properties:
|
||||||
nodeAffinity:
|
nodeAffinity:
|
||||||
description: Describes node affinity scheduling rules for
|
description: Describes node affinity scheduling rules for
|
||||||
@@ -4301,11 +4301,15 @@ spec:
|
|||||||
type: array
|
type: array
|
||||||
type: object
|
type: object
|
||||||
replicas:
|
replicas:
|
||||||
default: 1
|
default: 2
|
||||||
description: |-
|
description: |-
|
||||||
replicas. terdut-server is not horizontally-scale-tested; keep this
|
replicas. Defaults to 2: terdut-server v0.36.0 put the sweeper, the
|
||||||
at its default of 1 unless you've verified otherwise -- the sweeper
|
notifier and the migration runner each behind a Postgres advisory
|
||||||
and the notifier are unsynchronised singletons.
|
lock, and gave incident creation its own conflict resolution, so
|
||||||
|
more than one replica no longer double-pages, races a migration, or
|
||||||
|
drops a webhook payload. image.tag must be v0.36.0 or newer for
|
||||||
|
that to hold -- an older terdut-server has none of these guards,
|
||||||
|
and this field does not check the tag for you.
|
||||||
format: int32
|
format: int32
|
||||||
type: integer
|
type: integer
|
||||||
sweeper:
|
sweeper:
|
||||||
|
|||||||
@@ -14,13 +14,22 @@ metadata:
|
|||||||
spec:
|
spec:
|
||||||
image:
|
image:
|
||||||
repository: git.ryuvia.com/niklas/terdut-server
|
repository: git.ryuvia.com/niklas/terdut-server
|
||||||
# v0.34.0: fixes callerMayManageServiceAccount so an instance-scoped
|
# v0.36.0 is the floor now that replicas below is 2 (this demo pins
|
||||||
# service account can adopt/rotate a key on a team-scoped account it
|
# the current release, v0.43.0, so it shows the current web UI too): that
|
||||||
# didn't just create in the same call -- without this, terdutteam-*
|
# release put the sweeper, the notifier and the migration runner each
|
||||||
# can wedge permanently on exactly the crash-window race this demo
|
# behind a Postgres advisory lock, and gave incident creation its own
|
||||||
# hit live (niklas/terdut-operator#3).
|
# conflict resolution, which is what makes a second replica safe
|
||||||
tag: v0.34.0
|
# instead of racing the first. (Still carries v0.34.0's fix too --
|
||||||
replicas: 1
|
# callerMayManageServiceAccount, so an instance-scoped service account
|
||||||
|
# can adopt/rotate a key on a team-scoped account it didn't just create
|
||||||
|
# in the same call -- without which terdutteam-* can wedge permanently
|
||||||
|
# on the crash-window race this demo hit live, niklas/terdut-operator#3.)
|
||||||
|
tag: v0.43.0
|
||||||
|
# Matches this CRD's own spec.replicas default (v0.4.0) -- stated
|
||||||
|
# explicitly, like every other field in this file, rather than left to
|
||||||
|
# the default. RollingUpdate follows automatically; this operator does
|
||||||
|
# not expose Strategy as a spec field.
|
||||||
|
replicas: 2
|
||||||
networking:
|
networking:
|
||||||
hostname: terdut-operator-demo.example
|
hostname: terdut-operator-demo.example
|
||||||
servicePort: 8080
|
servicePort: 8080
|
||||||
|
|||||||
@@ -133,6 +133,20 @@ Each `(team, scenario)` pair is one stable fingerprint, so firing the same
|
|||||||
one twice updates the same alert (a real re-fire) and `resolve` closes
|
one twice updates the same alert (a real re-fire) and `resolve` closes
|
||||||
exactly that one.
|
exactly that one.
|
||||||
|
|
||||||
|
Set `CLUSTER` to send the alert as if it came from one of several clusters:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu
|
||||||
|
CLUSTER=prod-us ./fire-alerts.sh platform high-cpu # a second incident, not a join
|
||||||
|
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu resolve
|
||||||
|
```
|
||||||
|
|
||||||
|
It stands in for a Prometheus external label plus `cluster` in Alertmanager's
|
||||||
|
`group_by` (terdut-server's README, "Several clusters, one team"): the web UI
|
||||||
|
then shows the cluster chip on each incident and a cluster filter in the
|
||||||
|
queue. `CLUSTER` is part of the fingerprint, so resolve with the same value you
|
||||||
|
fired with. `./run-demo.sh` fires its alerts across `prod-eu` and `prod-us`.
|
||||||
|
|
||||||
### Dead man's switches
|
### Dead man's switches
|
||||||
|
|
||||||
`06-deadman-platform.yaml` / `07-deadman-payments.yaml` expect a heartbeat
|
`06-deadman-platform.yaml` / `07-deadman-payments.yaml` expect a heartbeat
|
||||||
|
|||||||
@@ -16,7 +16,15 @@
|
|||||||
# is the one part of that URL still usable here.
|
# is the one part of that URL still usable here.
|
||||||
#
|
#
|
||||||
# Usage:
|
# Usage:
|
||||||
# ./fire-alerts.sh <platform|payments> <high-cpu|disk-full|pod-crash|heartbeat> [resolve]
|
# [CLUSTER=prod-eu] ./fire-alerts.sh <platform|payments> <high-cpu|disk-full|pod-crash|heartbeat> [resolve]
|
||||||
|
#
|
||||||
|
# CLUSTER stands in for a Prometheus externalLabel plus `cluster` in
|
||||||
|
# Alertmanager's group_by (terdut-server's README, "Several clusters, one
|
||||||
|
# team"): it is put on the alert's labels and on groupLabels, so the incident
|
||||||
|
# carries it and the web UI shows the cluster chip and the queue's cluster
|
||||||
|
# filter. It is also part of the group key and the fingerprint, which is what
|
||||||
|
# keeps the same alert in two clusters from joining one incident. Unset, the
|
||||||
|
# alert is sent exactly as before.
|
||||||
#
|
#
|
||||||
# Prerequisites: kubectl context pointed at the demo namespace, jq, curl,
|
# Prerequisites: kubectl context pointed at the demo namespace, jq, curl,
|
||||||
# and (in another terminal) a running:
|
# and (in another terminal) a running:
|
||||||
@@ -24,6 +32,7 @@
|
|||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
NAMESPACE="${NAMESPACE:-}"
|
NAMESPACE="${NAMESPACE:-}"
|
||||||
|
CLUSTER="${CLUSTER:-}"
|
||||||
BASE_URL="${BASE_URL:-http://localhost:8080}"
|
BASE_URL="${BASE_URL:-http://localhost:8080}"
|
||||||
|
|
||||||
usage() {
|
usage() {
|
||||||
@@ -45,6 +54,8 @@ scenarios:
|
|||||||
env vars:
|
env vars:
|
||||||
NAMESPACE kubectl -n for reading the webhook Secret (required)
|
NAMESPACE kubectl -n for reading the webhook Secret (required)
|
||||||
BASE_URL where the port-forwarded terdut-server is (default http://localhost:8080)
|
BASE_URL where the port-forwarded terdut-server is (default http://localhost:8080)
|
||||||
|
CLUSTER optional cluster name, e.g. prod-eu: sent as a `cluster` label and
|
||||||
|
group label, so the UI shows where the incident came from
|
||||||
EOF
|
EOF
|
||||||
exit 1
|
exit 1
|
||||||
}
|
}
|
||||||
@@ -90,7 +101,7 @@ key="$(kubectl -n "$NAMESPACE" get secret "$secret_name" -o jsonpath='{.data.key
|
|||||||
# created: terdut-server correlates on (team_id, fingerprint), not on
|
# created: terdut-server correlates on (team_id, fingerprint), not on
|
||||||
# anything else in the payload. Real Alertmanager computes this from the
|
# anything else in the payload. Real Alertmanager computes this from the
|
||||||
# alert's label set; a fixed string plays the same role here.
|
# alert's label set; a fixed string plays the same role here.
|
||||||
fingerprint="demo-${team}-${scenario}"
|
fingerprint="demo-${team}-${scenario}${CLUSTER:+-$CLUSTER}"
|
||||||
|
|
||||||
now="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
|
now="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
|
||||||
if [ "$status" = firing ]; then
|
if [ "$status" = firing ]; then
|
||||||
@@ -101,7 +112,8 @@ fi
|
|||||||
|
|
||||||
payload="$(jq -n \
|
payload="$(jq -n \
|
||||||
--arg status "$status" \
|
--arg status "$status" \
|
||||||
--arg groupKey "demo:${team}:${scenario}" \
|
--arg groupKey "demo:${team}:${scenario}${CLUSTER:+:$CLUSTER}" \
|
||||||
|
--arg cluster "$CLUSTER" \
|
||||||
--arg alertname "$alertname" \
|
--arg alertname "$alertname" \
|
||||||
--arg team "$team" \
|
--arg team "$team" \
|
||||||
--arg severity "$severity" \
|
--arg severity "$severity" \
|
||||||
@@ -113,10 +125,10 @@ payload="$(jq -n \
|
|||||||
version: "4",
|
version: "4",
|
||||||
status: $status,
|
status: $status,
|
||||||
groupKey: $groupKey,
|
groupKey: $groupKey,
|
||||||
groupLabels: { alertname: $alertname, team: $team },
|
groupLabels: ({ alertname: $alertname, team: $team } + (if $cluster != "" then { cluster: $cluster } else {} end)),
|
||||||
alerts: [{
|
alerts: [{
|
||||||
status: $status,
|
status: $status,
|
||||||
labels: { alertname: $alertname, severity: $severity, team: $team, instance: "demo" },
|
labels: ({ alertname: $alertname, severity: $severity, team: $team, instance: "demo" } + (if $cluster != "" then { cluster: $cluster } else {} end)),
|
||||||
annotations: { summary: $summary },
|
annotations: { summary: $summary },
|
||||||
startsAt: $startsAt,
|
startsAt: $startsAt,
|
||||||
endsAt: $endsAt,
|
endsAt: $endsAt,
|
||||||
@@ -126,7 +138,7 @@ payload="$(jq -n \
|
|||||||
}')"
|
}')"
|
||||||
|
|
||||||
url="${BASE_URL}/api/integrations/${key}/alertmanager"
|
url="${BASE_URL}/api/integrations/${key}/alertmanager"
|
||||||
echo "POST $url (team=$team scenario=$scenario status=$status)" >&2
|
echo "POST $url (team=$team scenario=$scenario status=$status${CLUSTER:+ cluster=$CLUSTER})" >&2
|
||||||
code="$(curl -sS -o /tmp/fire-alerts-response.json -w '%{http_code}' \
|
code="$(curl -sS -o /tmp/fire-alerts-response.json -w '%{http_code}' \
|
||||||
-X POST "$url" -H 'Content-Type: application/json' -d "$payload")"
|
-X POST "$url" -H 'Content-Type: application/json' -d "$payload")"
|
||||||
echo "-> HTTP $code" >&2
|
echo "-> HTTP $code" >&2
|
||||||
|
|||||||
+51
-19
@@ -107,26 +107,41 @@ apply_demo() {
|
|||||||
kubectl apply -n "$NAMESPACE" -k "$SCRIPT_DIR" >/dev/null
|
kubectl apply -n "$NAMESPACE" -k "$SCRIPT_DIR" >/dev/null
|
||||||
}
|
}
|
||||||
|
|
||||||
wait_for_ready() {
|
wait_for_objects() {
|
||||||
local objects=(
|
|
||||||
"terdutserver/terdut-operator-demo"
|
|
||||||
"terdutteam/terdutteam-platform"
|
|
||||||
"terdutteam/terdutteam-payments"
|
|
||||||
"terdutescalationrule/terdutescalationrule-platform"
|
|
||||||
"terdutescalationrule/terdutescalationrule-payments"
|
|
||||||
"terdutdeadmanswitch/terdutdeadmanswitch-platform"
|
|
||||||
"terdutdeadmanswitch/terdutdeadmanswitch-payments"
|
|
||||||
"terdutalertsource/terdutalertsource-platform"
|
|
||||||
"terdutalertsource/terdutalertsource-payments"
|
|
||||||
)
|
|
||||||
local obj
|
local obj
|
||||||
for obj in "${objects[@]}"; do
|
for obj in "$@"; do
|
||||||
log "waiting for $obj to become Ready"
|
log "waiting for $obj to become Ready"
|
||||||
kubectl wait --for=condition=Ready --timeout "$WAIT_TIMEOUT" -n "$NAMESPACE" "$obj" >/dev/null \
|
kubectl wait --for=condition=Ready --timeout "$WAIT_TIMEOUT" -n "$NAMESPACE" "$obj" >/dev/null \
|
||||||
|| die "timed out waiting for $obj -- try: kubectl describe -n $NAMESPACE $obj"
|
|| die "timed out waiting for $obj -- try: kubectl describe -n $NAMESPACE $obj"
|
||||||
done
|
done
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# Just the server and the two teams -- everything redeem_platform_invite and
|
||||||
|
# join_payments_team need. Deliberately NOT the escalation rules here: this
|
||||||
|
# demo kit's own terdutescalationrule-platform names alice as a level-1
|
||||||
|
# target, and that CR cannot reach Ready until alice actually exists
|
||||||
|
# (terdut-server resolves every named username at reconcile time, not just
|
||||||
|
# at escalation time) -- a real dependency this script has to satisfy by
|
||||||
|
# creating her first, not something kubectl wait can be told to ignore.
|
||||||
|
wait_for_teams_ready() {
|
||||||
|
wait_for_objects \
|
||||||
|
"terdutserver/terdut-operator-demo" \
|
||||||
|
"terdutteam/terdutteam-platform" \
|
||||||
|
"terdutteam/terdutteam-payments"
|
||||||
|
}
|
||||||
|
|
||||||
|
# Everything that was waiting on alice (or just on the teams above, now
|
||||||
|
# already satisfied) to exist.
|
||||||
|
wait_for_remaining_ready() {
|
||||||
|
wait_for_objects \
|
||||||
|
"terdutescalationrule/terdutescalationrule-platform" \
|
||||||
|
"terdutescalationrule/terdutescalationrule-payments" \
|
||||||
|
"terdutdeadmanswitch/terdutdeadmanswitch-platform" \
|
||||||
|
"terdutdeadmanswitch/terdutdeadmanswitch-payments" \
|
||||||
|
"terdutalertsource/terdutalertsource-platform" \
|
||||||
|
"terdutalertsource/terdutalertsource-payments"
|
||||||
|
}
|
||||||
|
|
||||||
start_port_forward() {
|
start_port_forward() {
|
||||||
# A stale pidfile from an earlier run would otherwise collide with us on
|
# A stale pidfile from an earlier run would otherwise collide with us on
|
||||||
# $LOCAL_PORT -- if that pid is still alive, stop it first.
|
# $LOCAL_PORT -- if that pid is still alive, stop it first.
|
||||||
@@ -181,6 +196,18 @@ redeem_platform_invite() {
|
|||||||
invite_token="${invite_url##*invite=}"
|
invite_token="${invite_url##*invite=}"
|
||||||
[ -n "$invite_token" ] || die "could not parse an invite token out of $invite_url"
|
[ -n "$invite_token" ] || die "could not parse an invite token out of $invite_url"
|
||||||
|
|
||||||
|
# A re-run: the invite was spent by the first run, and the server answers a
|
||||||
|
# spent invite with 403 before it ever looks at the username, so the 409
|
||||||
|
# handled below never arrives. If alice can already sign in, she exists.
|
||||||
|
local login_code
|
||||||
|
login_code="$(curl -sS -o /dev/null -w '%{http_code}' \
|
||||||
|
-X POST "${BASE_URL}/api/login" -H 'Content-Type: application/json' \
|
||||||
|
-d "$(jq -n --arg u "$ALICE_USERNAME" --arg p "$DEMO_PASSWORD" '{username: $u, password: $p}')")"
|
||||||
|
if [ "$login_code" = "200" ]; then
|
||||||
|
log "account ${ALICE_USERNAME} already exists and can sign in, skipping signup (re-run detected)"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
|
||||||
log "signing up ${ALICE_USERNAME} via Platform's invite"
|
log "signing up ${ALICE_USERNAME} via Platform's invite"
|
||||||
local body resp_file code
|
local body resp_file code
|
||||||
body="$(jq -n \
|
body="$(jq -n \
|
||||||
@@ -235,9 +262,13 @@ join_payments_team() {
|
|||||||
|
|
||||||
fire_demo_alerts() {
|
fire_demo_alerts() {
|
||||||
log "firing representative demo alerts"
|
log "firing representative demo alerts"
|
||||||
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform high-cpu
|
# Two clusters, so the queue shows the cluster chip and offers its filter.
|
||||||
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform disk-full
|
# high-cpu fires in both: the same alert in two clusters is two incidents.
|
||||||
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" payments pod-crash
|
local fire="$SCRIPT_DIR/fire-alerts.sh"
|
||||||
|
CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu
|
||||||
|
CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu
|
||||||
|
CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform disk-full
|
||||||
|
CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" payments pod-crash
|
||||||
}
|
}
|
||||||
|
|
||||||
print_summary() {
|
print_summary() {
|
||||||
@@ -254,8 +285,8 @@ terdut demo is up.
|
|||||||
|
|
||||||
Fire more alerts:
|
Fire more alerts:
|
||||||
export NAMESPACE=${NAMESPACE} BASE_URL=${BASE_URL}
|
export NAMESPACE=${NAMESPACE} BASE_URL=${BASE_URL}
|
||||||
./fire-alerts.sh platform high-cpu
|
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu
|
||||||
./fire-alerts.sh platform high-cpu resolve
|
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu resolve
|
||||||
./fire-alerts.sh payments heartbeat # send repeatedly (e.g. every minute)
|
./fire-alerts.sh payments heartbeat # send repeatedly (e.g. every minute)
|
||||||
# to keep a dead man's switch alive;
|
# to keep a dead man's switch alive;
|
||||||
# stop sending it and, 15 minutes
|
# stop sending it and, 15 minutes
|
||||||
@@ -319,10 +350,11 @@ main() {
|
|||||||
ensure_kind_cluster
|
ensure_kind_cluster
|
||||||
install_operator
|
install_operator
|
||||||
apply_demo
|
apply_demo
|
||||||
wait_for_ready
|
wait_for_teams_ready
|
||||||
start_port_forward
|
start_port_forward
|
||||||
redeem_platform_invite
|
redeem_platform_invite
|
||||||
join_payments_team
|
join_payments_team
|
||||||
|
wait_for_remaining_ready
|
||||||
fire_demo_alerts
|
fire_demo_alerts
|
||||||
print_summary
|
print_summary
|
||||||
|
|
||||||
|
|||||||
@@ -37,17 +37,27 @@ func (r *TerdutServerReconciler) reconcileDeployment(
|
|||||||
_, err := controllerutil.CreateOrUpdate(ctx, r.Client, deploy, func() error {
|
_, err := controllerutil.CreateOrUpdate(ctx, r.Client, deploy, func() error {
|
||||||
replicas := srv.Spec.Replicas
|
replicas := srv.Spec.Replicas
|
||||||
if replicas == 0 {
|
if replicas == 0 {
|
||||||
replicas = 1
|
// Only reachable for a TerdutServer stored before the
|
||||||
|
// +kubebuilder:default=2 marker existed -- the API server's own
|
||||||
|
// CRD defaulting fills this in for anything created or updated
|
||||||
|
// through it, so a fresh zero value here means a pre-existing
|
||||||
|
// object that predates the default, not a deliberate "none"
|
||||||
|
// (there is no way to request zero replicas).
|
||||||
|
replicas = 2
|
||||||
}
|
}
|
||||||
labels := labelsFor(srv)
|
labels := labelsFor(srv)
|
||||||
|
|
||||||
deploy.Spec.Replicas = &replicas
|
deploy.Spec.Replicas = &replicas
|
||||||
deploy.Spec.Selector = &metav1.LabelSelector{MatchLabels: labels}
|
deploy.Spec.Selector = &metav1.LabelSelector{MatchLabels: labels}
|
||||||
// Recreate, not RollingUpdate: the sweeper and the notifier are
|
// RollingUpdate, not Recreate: terdut-server v0.36.0 put the sweeper,
|
||||||
// unsynchronised singletons inside terdut-server, and two replicas
|
// the notifier and the migration runner each behind a Postgres
|
||||||
// overlapping during a rollout would both page for the same
|
// advisory lock, and gave incident creation its own conflict
|
||||||
// incident (matches the chart's own deployment.yaml comment).
|
// resolution, so two replicas overlapping during a rollout no longer
|
||||||
deploy.Spec.Strategy = appsv1.DeploymentStrategy{Type: appsv1.RecreateDeploymentStrategyType}
|
// double-page, race a migration, or drop a webhook payload (matches
|
||||||
|
// the chart's own deployment.yaml comment). No explicit
|
||||||
|
// maxUnavailable/maxSurge: left at the 25%/25% default, which rounds
|
||||||
|
// to 0/1 at the default replicas: 2 -- already zero-downtime.
|
||||||
|
deploy.Spec.Strategy = appsv1.DeploymentStrategy{Type: appsv1.RollingUpdateDeploymentStrategyType}
|
||||||
pod := srv.Spec.Pod
|
pod := srv.Spec.Pod
|
||||||
deploy.Spec.Template = corev1.PodTemplateSpec{
|
deploy.Spec.Template = corev1.PodTemplateSpec{
|
||||||
// pod.Annotations is assigned directly, not merged -- nothing
|
// pod.Annotations is assigned directly, not merged -- nothing
|
||||||
|
|||||||
Reference in New Issue
Block a user