Compare commits
5 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 50ce5bcec0 | |||
| 822c80dda6 | |||
| 0ee7ede648 | |||
| d50248531c | |||
| 4007f54279 |
@@ -153,7 +153,7 @@ spec:
|
||||
image:
|
||||
repository: git.ryuvia.com/niklas/terdut-server
|
||||
tag: v0.9.3
|
||||
replicas: 1 # terdut-server is not horizontally-scale-tested; keep the field, default 1
|
||||
replicas: 2 # default since terdut-server v0.36.0's advisory locks; see TerdutServerSpec.Replicas
|
||||
networking:
|
||||
hostname: terdut.example.com
|
||||
servicePort: 8080
|
||||
@@ -244,10 +244,10 @@ no custom wrapper buys anything for any of these, matching how
|
||||
CloudNativePG and the Zalando postgres-operator both expose the same
|
||||
knobs. `affinity` is pure user-supplied passthrough, not a
|
||||
toggle-plus-generated-default the way a multi-replica-aware operator's
|
||||
pod anti-affinity typically is: this operator never auto-generates
|
||||
affinity of its own, since `replicas` above 1 isn't a supported topology
|
||||
(the sweeper/notifier singleton constraint, §4.1's own illustrative YAML
|
||||
comment). `spec.pod.disruptionBudget` is the one field here that isn't a
|
||||
pod anti-affinity typically is: even though `replicas` now defaults to 2
|
||||
(terdut-server v0.36.0's advisory locks made that safe, §4.1's own
|
||||
illustrative YAML comment), this operator still never auto-generates
|
||||
affinity of its own. `spec.pod.disruptionBudget` is the one field here that isn't a
|
||||
straight PodTemplateSpec knob — when set, the controller reconciles a
|
||||
`PodDisruptionBudget` selecting this `TerdutServer`'s pods; clearing it
|
||||
deletes any it previously created (§7). `minAvailable`/`maxUnavailable`
|
||||
@@ -832,10 +832,12 @@ what it was, a separate install, until someone deletes it.
|
||||
- Automatic Deployment restart on upstream Postgres credential rotation.
|
||||
- `spec.pod.priorityClassName`, pod-label passthrough beyond
|
||||
`spec.pod.annotations`, and a HorizontalPodAutoscaler for `TerdutServer`
|
||||
— all considered alongside §4.1's `spec.pod` and explicitly left out of
|
||||
that round: an HPA in particular would actively contradict
|
||||
`spec.replicas`'s own stance that this operator doesn't support more
|
||||
than one replica (the sweeper/notifier singleton constraint).
|
||||
— all considered alongside §4.1's `spec.pod` and left out of that round.
|
||||
An HPA no longer contradicts anything now that `spec.replicas` defaults
|
||||
to 2 (terdut-server v0.36.0's advisory locks), but it is still a
|
||||
separate, not-yet-made decision: a fixed replica count has no scaling
|
||||
metric, min/max bounds, or cooldown behaviour to get right, and nobody
|
||||
has asked for it yet.
|
||||
- Admission webhooks / CEL-only validation limits (e.g. verifying a
|
||||
`teamRef` exists at admission time rather than surfacing it as a status
|
||||
condition after the fact).
|
||||
|
||||
@@ -229,11 +229,11 @@ type PodSpec struct {
|
||||
Tolerations []corev1.Toleration `json:"tolerations,omitempty"`
|
||||
|
||||
// affinity covers node affinity, pod affinity and pod anti-affinity in
|
||||
// one field -- unlike a multi-replica-aware operator, this one never
|
||||
// generates a default anti-affinity itself (replicas above 1 isn't a
|
||||
// supported topology, see TerdutServerSpec.Replicas's own doc comment),
|
||||
// so this is pure user-supplied passthrough, not a toggle-plus-generated-
|
||||
// default.
|
||||
// one field -- even though replicas now defaults to 2 (see
|
||||
// TerdutServerSpec.Replicas's own doc comment), this operator still
|
||||
// never generates a default anti-affinity of its own the way a
|
||||
// multi-replica-aware operator typically would, so this stays pure
|
||||
// user-supplied passthrough, not a toggle-plus-generated-default.
|
||||
// +optional
|
||||
Affinity *corev1.Affinity `json:"affinity,omitempty"`
|
||||
|
||||
@@ -301,10 +301,14 @@ type TerdutServerSpec struct {
|
||||
// +required
|
||||
Image ImageSpec `json:"image"`
|
||||
|
||||
// replicas. terdut-server is not horizontally-scale-tested; keep this
|
||||
// at its default of 1 unless you've verified otherwise -- the sweeper
|
||||
// and the notifier are unsynchronised singletons.
|
||||
// +kubebuilder:default=1
|
||||
// replicas. Defaults to 2: terdut-server v0.36.0 put the sweeper, the
|
||||
// notifier and the migration runner each behind a Postgres advisory
|
||||
// lock, and gave incident creation its own conflict resolution, so
|
||||
// more than one replica no longer double-pages, races a migration, or
|
||||
// drops a webhook payload. image.tag must be v0.36.0 or newer for
|
||||
// that to hold -- an older terdut-server has none of these guards,
|
||||
// and this field does not check the tag for you.
|
||||
// +kubebuilder:default=2
|
||||
// +optional
|
||||
Replicas int32 `json:"replicas,omitempty"`
|
||||
|
||||
|
||||
@@ -6,8 +6,8 @@ type: application
|
||||
# These fields decide nothing: `make helm-package` passes --version and
|
||||
# --app-version from the release tag (same reasoning as terdut-server's own
|
||||
# chart). They're for whoever reads the tree before a tag exists.
|
||||
version: 0.3.0
|
||||
appVersion: "v0.3.0"
|
||||
version: 0.4.0
|
||||
appVersion: "v0.4.0"
|
||||
|
||||
keywords:
|
||||
- kubernetes
|
||||
|
||||
@@ -327,11 +327,11 @@ spec:
|
||||
affinity:
|
||||
description: |-
|
||||
affinity covers node affinity, pod affinity and pod anti-affinity in
|
||||
one field -- unlike a multi-replica-aware operator, this one never
|
||||
generates a default anti-affinity itself (replicas above 1 isn't a
|
||||
supported topology, see TerdutServerSpec.Replicas's own doc comment),
|
||||
so this is pure user-supplied passthrough, not a toggle-plus-generated-
|
||||
default.
|
||||
one field -- even though replicas now defaults to 2 (see
|
||||
TerdutServerSpec.Replicas's own doc comment), this operator still
|
||||
never generates a default anti-affinity of its own the way a
|
||||
multi-replica-aware operator typically would, so this stays pure
|
||||
user-supplied passthrough, not a toggle-plus-generated-default.
|
||||
properties:
|
||||
nodeAffinity:
|
||||
description: Describes node affinity scheduling rules for
|
||||
@@ -4304,11 +4304,15 @@ spec:
|
||||
type: array
|
||||
type: object
|
||||
replicas:
|
||||
default: 1
|
||||
default: 2
|
||||
description: |-
|
||||
replicas. terdut-server is not horizontally-scale-tested; keep this
|
||||
at its default of 1 unless you've verified otherwise -- the sweeper
|
||||
and the notifier are unsynchronised singletons.
|
||||
replicas. Defaults to 2: terdut-server v0.36.0 put the sweeper, the
|
||||
notifier and the migration runner each behind a Postgres advisory
|
||||
lock, and gave incident creation its own conflict resolution, so
|
||||
more than one replica no longer double-pages, races a migration, or
|
||||
drops a webhook payload. image.tag must be v0.36.0 or newer for
|
||||
that to hold -- an older terdut-server has none of these guards,
|
||||
and this field does not check the tag for you.
|
||||
format: int32
|
||||
type: integer
|
||||
sweeper:
|
||||
|
||||
@@ -233,7 +233,11 @@ terdutServer:
|
||||
## Required when terdutServer.enabled.
|
||||
# tag: ""
|
||||
|
||||
replicas: 1
|
||||
## Safe above 1 since terdut-server v0.36.0 (image.tag above must be that or
|
||||
## newer): the sweeper, notifier and migration runner are each behind a
|
||||
## Postgres advisory lock, and incident creation resolves its own insert
|
||||
## conflict, matching this CRD's own spec.replicas default.
|
||||
replicas: 2
|
||||
|
||||
networking:
|
||||
## Required when terdutServer.enabled -- terdut-server's own public
|
||||
|
||||
@@ -324,11 +324,11 @@ spec:
|
||||
affinity:
|
||||
description: |-
|
||||
affinity covers node affinity, pod affinity and pod anti-affinity in
|
||||
one field -- unlike a multi-replica-aware operator, this one never
|
||||
generates a default anti-affinity itself (replicas above 1 isn't a
|
||||
supported topology, see TerdutServerSpec.Replicas's own doc comment),
|
||||
so this is pure user-supplied passthrough, not a toggle-plus-generated-
|
||||
default.
|
||||
one field -- even though replicas now defaults to 2 (see
|
||||
TerdutServerSpec.Replicas's own doc comment), this operator still
|
||||
never generates a default anti-affinity of its own the way a
|
||||
multi-replica-aware operator typically would, so this stays pure
|
||||
user-supplied passthrough, not a toggle-plus-generated-default.
|
||||
properties:
|
||||
nodeAffinity:
|
||||
description: Describes node affinity scheduling rules for
|
||||
@@ -4301,11 +4301,15 @@ spec:
|
||||
type: array
|
||||
type: object
|
||||
replicas:
|
||||
default: 1
|
||||
default: 2
|
||||
description: |-
|
||||
replicas. terdut-server is not horizontally-scale-tested; keep this
|
||||
at its default of 1 unless you've verified otherwise -- the sweeper
|
||||
and the notifier are unsynchronised singletons.
|
||||
replicas. Defaults to 2: terdut-server v0.36.0 put the sweeper, the
|
||||
notifier and the migration runner each behind a Postgres advisory
|
||||
lock, and gave incident creation its own conflict resolution, so
|
||||
more than one replica no longer double-pages, races a migration, or
|
||||
drops a webhook payload. image.tag must be v0.36.0 or newer for
|
||||
that to hold -- an older terdut-server has none of these guards,
|
||||
and this field does not check the tag for you.
|
||||
format: int32
|
||||
type: integer
|
||||
sweeper:
|
||||
|
||||
@@ -14,13 +14,22 @@ metadata:
|
||||
spec:
|
||||
image:
|
||||
repository: git.ryuvia.com/niklas/terdut-server
|
||||
# v0.34.0: fixes callerMayManageServiceAccount so an instance-scoped
|
||||
# service account can adopt/rotate a key on a team-scoped account it
|
||||
# didn't just create in the same call -- without this, terdutteam-*
|
||||
# can wedge permanently on exactly the crash-window race this demo
|
||||
# hit live (niklas/terdut-operator#3).
|
||||
tag: v0.34.0
|
||||
replicas: 1
|
||||
# v0.36.0 is the floor now that replicas below is 2 (this demo pins
|
||||
# the current release, v0.43.0, so it shows the current web UI too): that
|
||||
# release put the sweeper, the notifier and the migration runner each
|
||||
# behind a Postgres advisory lock, and gave incident creation its own
|
||||
# conflict resolution, which is what makes a second replica safe
|
||||
# instead of racing the first. (Still carries v0.34.0's fix too --
|
||||
# callerMayManageServiceAccount, so an instance-scoped service account
|
||||
# can adopt/rotate a key on a team-scoped account it didn't just create
|
||||
# in the same call -- without which terdutteam-* can wedge permanently
|
||||
# on the crash-window race this demo hit live, niklas/terdut-operator#3.)
|
||||
tag: v0.43.0
|
||||
# Matches this CRD's own spec.replicas default (v0.4.0) -- stated
|
||||
# explicitly, like every other field in this file, rather than left to
|
||||
# the default. RollingUpdate follows automatically; this operator does
|
||||
# not expose Strategy as a spec field.
|
||||
replicas: 2
|
||||
networking:
|
||||
hostname: terdut-operator-demo.example
|
||||
servicePort: 8080
|
||||
|
||||
@@ -133,6 +133,20 @@ Each `(team, scenario)` pair is one stable fingerprint, so firing the same
|
||||
one twice updates the same alert (a real re-fire) and `resolve` closes
|
||||
exactly that one.
|
||||
|
||||
Set `CLUSTER` to send the alert as if it came from one of several clusters:
|
||||
|
||||
```sh
|
||||
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu
|
||||
CLUSTER=prod-us ./fire-alerts.sh platform high-cpu # a second incident, not a join
|
||||
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu resolve
|
||||
```
|
||||
|
||||
It stands in for a Prometheus external label plus `cluster` in Alertmanager's
|
||||
`group_by` (terdut-server's README, "Several clusters, one team"): the web UI
|
||||
then shows the cluster chip on each incident and a cluster filter in the
|
||||
queue. `CLUSTER` is part of the fingerprint, so resolve with the same value you
|
||||
fired with. `./run-demo.sh` fires its alerts across `prod-eu` and `prod-us`.
|
||||
|
||||
### Dead man's switches
|
||||
|
||||
`06-deadman-platform.yaml` / `07-deadman-payments.yaml` expect a heartbeat
|
||||
|
||||
@@ -16,7 +16,15 @@
|
||||
# is the one part of that URL still usable here.
|
||||
#
|
||||
# Usage:
|
||||
# ./fire-alerts.sh <platform|payments> <high-cpu|disk-full|pod-crash|heartbeat> [resolve]
|
||||
# [CLUSTER=prod-eu] ./fire-alerts.sh <platform|payments> <high-cpu|disk-full|pod-crash|heartbeat> [resolve]
|
||||
#
|
||||
# CLUSTER stands in for a Prometheus externalLabel plus `cluster` in
|
||||
# Alertmanager's group_by (terdut-server's README, "Several clusters, one
|
||||
# team"): it is put on the alert's labels and on groupLabels, so the incident
|
||||
# carries it and the web UI shows the cluster chip and the queue's cluster
|
||||
# filter. It is also part of the group key and the fingerprint, which is what
|
||||
# keeps the same alert in two clusters from joining one incident. Unset, the
|
||||
# alert is sent exactly as before.
|
||||
#
|
||||
# Prerequisites: kubectl context pointed at the demo namespace, jq, curl,
|
||||
# and (in another terminal) a running:
|
||||
@@ -24,6 +32,7 @@
|
||||
set -euo pipefail
|
||||
|
||||
NAMESPACE="${NAMESPACE:-}"
|
||||
CLUSTER="${CLUSTER:-}"
|
||||
BASE_URL="${BASE_URL:-http://localhost:8080}"
|
||||
|
||||
usage() {
|
||||
@@ -45,6 +54,8 @@ scenarios:
|
||||
env vars:
|
||||
NAMESPACE kubectl -n for reading the webhook Secret (required)
|
||||
BASE_URL where the port-forwarded terdut-server is (default http://localhost:8080)
|
||||
CLUSTER optional cluster name, e.g. prod-eu: sent as a `cluster` label and
|
||||
group label, so the UI shows where the incident came from
|
||||
EOF
|
||||
exit 1
|
||||
}
|
||||
@@ -90,7 +101,7 @@ key="$(kubectl -n "$NAMESPACE" get secret "$secret_name" -o jsonpath='{.data.key
|
||||
# created: terdut-server correlates on (team_id, fingerprint), not on
|
||||
# anything else in the payload. Real Alertmanager computes this from the
|
||||
# alert's label set; a fixed string plays the same role here.
|
||||
fingerprint="demo-${team}-${scenario}"
|
||||
fingerprint="demo-${team}-${scenario}${CLUSTER:+-$CLUSTER}"
|
||||
|
||||
now="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
|
||||
if [ "$status" = firing ]; then
|
||||
@@ -101,7 +112,8 @@ fi
|
||||
|
||||
payload="$(jq -n \
|
||||
--arg status "$status" \
|
||||
--arg groupKey "demo:${team}:${scenario}" \
|
||||
--arg groupKey "demo:${team}:${scenario}${CLUSTER:+:$CLUSTER}" \
|
||||
--arg cluster "$CLUSTER" \
|
||||
--arg alertname "$alertname" \
|
||||
--arg team "$team" \
|
||||
--arg severity "$severity" \
|
||||
@@ -113,10 +125,10 @@ payload="$(jq -n \
|
||||
version: "4",
|
||||
status: $status,
|
||||
groupKey: $groupKey,
|
||||
groupLabels: { alertname: $alertname, team: $team },
|
||||
groupLabels: ({ alertname: $alertname, team: $team } + (if $cluster != "" then { cluster: $cluster } else {} end)),
|
||||
alerts: [{
|
||||
status: $status,
|
||||
labels: { alertname: $alertname, severity: $severity, team: $team, instance: "demo" },
|
||||
labels: ({ alertname: $alertname, severity: $severity, team: $team, instance: "demo" } + (if $cluster != "" then { cluster: $cluster } else {} end)),
|
||||
annotations: { summary: $summary },
|
||||
startsAt: $startsAt,
|
||||
endsAt: $endsAt,
|
||||
@@ -126,7 +138,7 @@ payload="$(jq -n \
|
||||
}')"
|
||||
|
||||
url="${BASE_URL}/api/integrations/${key}/alertmanager"
|
||||
echo "POST $url (team=$team scenario=$scenario status=$status)" >&2
|
||||
echo "POST $url (team=$team scenario=$scenario status=$status${CLUSTER:+ cluster=$CLUSTER})" >&2
|
||||
code="$(curl -sS -o /tmp/fire-alerts-response.json -w '%{http_code}' \
|
||||
-X POST "$url" -H 'Content-Type: application/json' -d "$payload")"
|
||||
echo "-> HTTP $code" >&2
|
||||
|
||||
+51
-19
@@ -107,26 +107,41 @@ apply_demo() {
|
||||
kubectl apply -n "$NAMESPACE" -k "$SCRIPT_DIR" >/dev/null
|
||||
}
|
||||
|
||||
wait_for_ready() {
|
||||
local objects=(
|
||||
"terdutserver/terdut-operator-demo"
|
||||
"terdutteam/terdutteam-platform"
|
||||
"terdutteam/terdutteam-payments"
|
||||
"terdutescalationrule/terdutescalationrule-platform"
|
||||
"terdutescalationrule/terdutescalationrule-payments"
|
||||
"terdutdeadmanswitch/terdutdeadmanswitch-platform"
|
||||
"terdutdeadmanswitch/terdutdeadmanswitch-payments"
|
||||
"terdutalertsource/terdutalertsource-platform"
|
||||
"terdutalertsource/terdutalertsource-payments"
|
||||
)
|
||||
wait_for_objects() {
|
||||
local obj
|
||||
for obj in "${objects[@]}"; do
|
||||
for obj in "$@"; do
|
||||
log "waiting for $obj to become Ready"
|
||||
kubectl wait --for=condition=Ready --timeout "$WAIT_TIMEOUT" -n "$NAMESPACE" "$obj" >/dev/null \
|
||||
|| die "timed out waiting for $obj -- try: kubectl describe -n $NAMESPACE $obj"
|
||||
done
|
||||
}
|
||||
|
||||
# Just the server and the two teams -- everything redeem_platform_invite and
|
||||
# join_payments_team need. Deliberately NOT the escalation rules here: this
|
||||
# demo kit's own terdutescalationrule-platform names alice as a level-1
|
||||
# target, and that CR cannot reach Ready until alice actually exists
|
||||
# (terdut-server resolves every named username at reconcile time, not just
|
||||
# at escalation time) -- a real dependency this script has to satisfy by
|
||||
# creating her first, not something kubectl wait can be told to ignore.
|
||||
wait_for_teams_ready() {
|
||||
wait_for_objects \
|
||||
"terdutserver/terdut-operator-demo" \
|
||||
"terdutteam/terdutteam-platform" \
|
||||
"terdutteam/terdutteam-payments"
|
||||
}
|
||||
|
||||
# Everything that was waiting on alice (or just on the teams above, now
|
||||
# already satisfied) to exist.
|
||||
wait_for_remaining_ready() {
|
||||
wait_for_objects \
|
||||
"terdutescalationrule/terdutescalationrule-platform" \
|
||||
"terdutescalationrule/terdutescalationrule-payments" \
|
||||
"terdutdeadmanswitch/terdutdeadmanswitch-platform" \
|
||||
"terdutdeadmanswitch/terdutdeadmanswitch-payments" \
|
||||
"terdutalertsource/terdutalertsource-platform" \
|
||||
"terdutalertsource/terdutalertsource-payments"
|
||||
}
|
||||
|
||||
start_port_forward() {
|
||||
# A stale pidfile from an earlier run would otherwise collide with us on
|
||||
# $LOCAL_PORT -- if that pid is still alive, stop it first.
|
||||
@@ -181,6 +196,18 @@ redeem_platform_invite() {
|
||||
invite_token="${invite_url##*invite=}"
|
||||
[ -n "$invite_token" ] || die "could not parse an invite token out of $invite_url"
|
||||
|
||||
# A re-run: the invite was spent by the first run, and the server answers a
|
||||
# spent invite with 403 before it ever looks at the username, so the 409
|
||||
# handled below never arrives. If alice can already sign in, she exists.
|
||||
local login_code
|
||||
login_code="$(curl -sS -o /dev/null -w '%{http_code}' \
|
||||
-X POST "${BASE_URL}/api/login" -H 'Content-Type: application/json' \
|
||||
-d "$(jq -n --arg u "$ALICE_USERNAME" --arg p "$DEMO_PASSWORD" '{username: $u, password: $p}')")"
|
||||
if [ "$login_code" = "200" ]; then
|
||||
log "account ${ALICE_USERNAME} already exists and can sign in, skipping signup (re-run detected)"
|
||||
return 0
|
||||
fi
|
||||
|
||||
log "signing up ${ALICE_USERNAME} via Platform's invite"
|
||||
local body resp_file code
|
||||
body="$(jq -n \
|
||||
@@ -235,9 +262,13 @@ join_payments_team() {
|
||||
|
||||
fire_demo_alerts() {
|
||||
log "firing representative demo alerts"
|
||||
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform high-cpu
|
||||
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" platform disk-full
|
||||
NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$SCRIPT_DIR/fire-alerts.sh" payments pod-crash
|
||||
# Two clusters, so the queue shows the cluster chip and offers its filter.
|
||||
# high-cpu fires in both: the same alert in two clusters is two incidents.
|
||||
local fire="$SCRIPT_DIR/fire-alerts.sh"
|
||||
CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu
|
||||
CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform high-cpu
|
||||
CLUSTER=prod-us NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" platform disk-full
|
||||
CLUSTER=prod-eu NAMESPACE="$NAMESPACE" BASE_URL="$BASE_URL" "$fire" payments pod-crash
|
||||
}
|
||||
|
||||
print_summary() {
|
||||
@@ -254,8 +285,8 @@ terdut demo is up.
|
||||
|
||||
Fire more alerts:
|
||||
export NAMESPACE=${NAMESPACE} BASE_URL=${BASE_URL}
|
||||
./fire-alerts.sh platform high-cpu
|
||||
./fire-alerts.sh platform high-cpu resolve
|
||||
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu
|
||||
CLUSTER=prod-eu ./fire-alerts.sh platform high-cpu resolve
|
||||
./fire-alerts.sh payments heartbeat # send repeatedly (e.g. every minute)
|
||||
# to keep a dead man's switch alive;
|
||||
# stop sending it and, 15 minutes
|
||||
@@ -319,10 +350,11 @@ main() {
|
||||
ensure_kind_cluster
|
||||
install_operator
|
||||
apply_demo
|
||||
wait_for_ready
|
||||
wait_for_teams_ready
|
||||
start_port_forward
|
||||
redeem_platform_invite
|
||||
join_payments_team
|
||||
wait_for_remaining_ready
|
||||
fire_demo_alerts
|
||||
print_summary
|
||||
|
||||
|
||||
@@ -37,17 +37,27 @@ func (r *TerdutServerReconciler) reconcileDeployment(
|
||||
_, err := controllerutil.CreateOrUpdate(ctx, r.Client, deploy, func() error {
|
||||
replicas := srv.Spec.Replicas
|
||||
if replicas == 0 {
|
||||
replicas = 1
|
||||
// Only reachable for a TerdutServer stored before the
|
||||
// +kubebuilder:default=2 marker existed -- the API server's own
|
||||
// CRD defaulting fills this in for anything created or updated
|
||||
// through it, so a fresh zero value here means a pre-existing
|
||||
// object that predates the default, not a deliberate "none"
|
||||
// (there is no way to request zero replicas).
|
||||
replicas = 2
|
||||
}
|
||||
labels := labelsFor(srv)
|
||||
|
||||
deploy.Spec.Replicas = &replicas
|
||||
deploy.Spec.Selector = &metav1.LabelSelector{MatchLabels: labels}
|
||||
// Recreate, not RollingUpdate: the sweeper and the notifier are
|
||||
// unsynchronised singletons inside terdut-server, and two replicas
|
||||
// overlapping during a rollout would both page for the same
|
||||
// incident (matches the chart's own deployment.yaml comment).
|
||||
deploy.Spec.Strategy = appsv1.DeploymentStrategy{Type: appsv1.RecreateDeploymentStrategyType}
|
||||
// RollingUpdate, not Recreate: terdut-server v0.36.0 put the sweeper,
|
||||
// the notifier and the migration runner each behind a Postgres
|
||||
// advisory lock, and gave incident creation its own conflict
|
||||
// resolution, so two replicas overlapping during a rollout no longer
|
||||
// double-page, race a migration, or drop a webhook payload (matches
|
||||
// the chart's own deployment.yaml comment). No explicit
|
||||
// maxUnavailable/maxSurge: left at the 25%/25% default, which rounds
|
||||
// to 0/1 at the default replicas: 2 -- already zero-downtime.
|
||||
deploy.Spec.Strategy = appsv1.DeploymentStrategy{Type: appsv1.RollingUpdateDeploymentStrategyType}
|
||||
pod := srv.Spec.Pod
|
||||
deploy.Spec.Template = corev1.PodTemplateSpec{
|
||||
// pod.Annotations is assigned directly, not merged -- nothing
|
||||
|
||||
Reference in New Issue
Block a user