diff --git a/examples/demo/README.md b/examples/demo/README.md index 143087a..c11e782 100644 --- a/examples/demo/README.md +++ b/examples/demo/README.md @@ -38,14 +38,10 @@ kubectl get terdutservers,terdutteams,terdutescalationrules,terdutdeadmanswitche -n terdut-operator-demo ``` -**A few early `CrashLoopBackOff` restarts on the `terdut-demo` pod are -expected**, not a sign anything is wrong: this is a brand-new Postgres -doing its very first boot, and the operator's own Deployment template has -no wait-for-postgres step yet (unlike `charts/terdut-server`'s chart as of -v0.33.2 — see that repo's `CLAUDE.md`/release notes for why) — the pod -restarts a couple of times until Postgres is actually accepting -connections, then stays up. Carrying that chart's fix into this operator's -own Deployment template is still open. +The operator's own Deployment template now carries a `wait-for-postgres` +init container (same fix as `charts/terdut-server`'s chart as of v0.33.2), +so `terdut-demo`'s pod should come up clean even against this brand-new +Postgres doing its very first boot — no `CrashLoopBackOff` expected here. Once `terdut-demo`'s own `Ready` condition is `True`, everything downstream of it should settle within a reconcile interval or two. diff --git a/internal/controller/terdutserver_deployment.go b/internal/controller/terdutserver_deployment.go index ea3094e..9f99c7f 100644 --- a/internal/controller/terdutserver_deployment.go +++ b/internal/controller/terdutserver_deployment.go @@ -52,6 +52,7 @@ func (r *TerdutServerReconciler) reconcileDeployment( ObjectMeta: metav1.ObjectMeta{Labels: labels}, Spec: corev1.PodSpec{ EnableServiceLinks: new(false), + InitContainers: []corev1.Container{waitForPostgresContainer(dbEnv)}, Containers: []corev1.Container{{ Name: "terdut-server", Image: fmt.Sprintf("%s:%s", srv.Spec.Image.Repository, srv.Spec.Image.Tag), @@ -104,6 +105,36 @@ func servicePort(srv *terdutv1alpha1.TerdutServer) int32 { return srv.Spec.Networking.ServicePort } +// waitForPostgresContainer blocks the main container from starting until +// Postgres accepts connections, matching charts/terdut-server's own +// deployment.yaml template as of v0.33.2 (that repo's CLAUDE.md/release +// notes) -- that chart grew this the moment this exact Deployment, created +// by this controller, crash-looped a few times against a from-scratch +// postgres-operator cluster still doing initdb and Patroni leader election: +// terdut-server's own ping-retry budget on startup (internal/db/db.go) is +// sized for a much shorter, different race (NetworkPolicy propagation, a +// few seconds), not for genuine first-time cluster creation, so it +// exhausted and the process exited before ever binding its HTTP port -- a +// startupProbe cannot help there, since the crash happens before there is +// anything to probe. +// +// Reuses dbEnv unchanged: both of resolveDatabaseEnv's paths put +// TERDUT_DB_DSN first (terdutserver_database.go), so it's already exactly +// what pg_isready needs, and pg_isready needs no credentials -- it reports +// PQPING_OK on anything that amounts to a Postgres backend answering, +// including an auth challenge -- so including dbEnv's optional PGPASSWORD +// here too is harmless rather than load-bearing. +func waitForPostgresContainer(dbEnv []corev1.EnvVar) corev1.Container { + return corev1.Container{ + Name: "wait-for-postgres", + Image: "postgres:17-alpine", + Env: dbEnv, + Command: []string{"sh", "-c", + `until pg_isready -d "$TERDUT_DB_DSN"; do echo "wait-for-postgres: not ready yet, retrying in 2s"; sleep 2; done`, + }, + } +} + func healthzProbe() *corev1.Probe { return &corev1.Probe{ ProbeHandler: corev1.ProbeHandler{