Files
terdut-operator/internal/controller/terdutserver_deployment.go
T
Niklas Ye 6a699d4341 Let TerdutServer customize its pod, and never manage its own ingress
spec.pod (api/v1alpha1/terdutserver_types.go): annotations, nodeSelector,
tolerations, affinity, topologySpreadConstraints, resources, pod and
container securityContext, serviceAccountName, extraEnv/extraEnvFrom,
extraVolumes/extraVolumeMounts, imagePullSecrets, and an optional
disruptionBudget. All direct corev1 passthrough -- no wrapper types buy
anything for any of these, matching how CloudNativePG and the Zalando
postgres-operator both expose the same knobs, and matching this repo's
own SweeperSpec precedent ("wrap only when a round-trip through a
different type buys something"). affinity is pure user-supplied
passthrough, not a toggle-plus-generated-default the way a multi-replica
cluster operator's pod anti-affinity usually is: this operator never
auto-generates one, since spec.replicas above 1 isn't a supported
topology (the sweeper/notifier singleton constraint). Considered and
declined for this round: priorityClassName, pod labels beyond
annotations, and a HorizontalPodAutoscaler -- the last of those would
directly contradict the singleton constraint above.

disruptionBudget is the one field here that isn't a plain PodTemplateSpec
knob: when set, the controller now reconciles a PodDisruptionBudget
selecting the TerdutServer's own pods (new terdutserver_pdb.go); clearing
it deletes any it previously created. New RBAC marker on
poddisruptionbudgets to match.

Driven by a public-release pass: looking past this project's own use case
at what a mature, general-purpose operator CRD exposes here (researched
against Zalando postgres-operator and CloudNativePG specifically), not
just the fields this install happened to need.

Separately, and found while answering a question about exposing
TerdutServer through Istio instead of Gateway API: spec.networking's own
doc comment quietly promised a Gateway API HTTPRoute this operator would
build eventually ("a near-term follow-up, not deferred"). That promise is
wrong for a public release -- an operator managing someone's ingress
mechanism for them is a worse default than not touching it at all, and a
surprise HTTPRoute appearing once that follow-up eventually landed would
have been exactly backwards for an Istio (or plain-Ingress, or
intentionally-unexposed) install. Made the non-goal explicit and
permanent instead (DESIGN.md §1), removed the dead `gatewayListener`
field it was the only consumer of (zero runtime call sites anywhere --
setting it already had no effect, so this is a schema cleanup, not a
behavior change), and corrected ROADMAP.md's framing. hostname/servicePort
stay: both are live (TERDUT_PUBLIC_URL, container/Service port), this
operator just never acts on hostname for exposure. Added
examples/networking (Gateway API HTTPRoute, Istio VirtualService) showing
how to expose the plain ClusterIP Service the operator already creates --
outside the operator itself, as illustrations, not as something
examples/demo applies automatically.

No new terdut-server version requirement: both changes are CRD/controller-
only, nothing about the API this operator's bootstrap flow depends on
changed.
2026-10-02 18:50:38 +02:00

273 lines
11 KiB
Go

package controller
import (
"context"
"fmt"
"strings"
appsv1 "k8s.io/api/apps/v1"
corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/util/intstr"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/controller-runtime/pkg/controller/controllerutil"
terdutv1alpha1 "git.ryuvia.com/niklas/terdut-operator/api/v1alpha1"
)
// labelsFor returns the selector labels for srv's Deployment/Service —
// fixed once a Deployment exists (its selector is immutable), so this must
// never depend on anything in srv.Spec that could change later.
func labelsFor(srv *terdutv1alpha1.TerdutServer) map[string]string {
return map[string]string{
"app.kubernetes.io/name": "terdut-server",
"app.kubernetes.io/instance": srv.Name,
}
}
// reconcileDeployment creates or updates the Deployment running
// terdut-server, owned by srv (DESIGN.md §7: same-namespace generated
// objects carry a plain OwnerReference, no finalizer). Returns the live
// object (not just the desired one) so the caller can check
// status.readyReplicas.
func (r *TerdutServerReconciler) reconcileDeployment(
ctx context.Context, srv *terdutv1alpha1.TerdutServer, dbEnv []corev1.EnvVar,
) (*appsv1.Deployment, error) {
deploy := &appsv1.Deployment{ObjectMeta: metav1.ObjectMeta{Name: srv.Name, Namespace: srv.Namespace}}
_, err := controllerutil.CreateOrUpdate(ctx, r.Client, deploy, func() error {
replicas := srv.Spec.Replicas
if replicas == 0 {
replicas = 1
}
labels := labelsFor(srv)
deploy.Spec.Replicas = &replicas
deploy.Spec.Selector = &metav1.LabelSelector{MatchLabels: labels}
// Recreate, not RollingUpdate: the sweeper and the notifier are
// unsynchronised singletons inside terdut-server, and two replicas
// overlapping during a rollout would both page for the same
// incident (matches the chart's own deployment.yaml comment).
deploy.Spec.Strategy = appsv1.DeploymentStrategy{Type: appsv1.RecreateDeploymentStrategyType}
pod := srv.Spec.Pod
deploy.Spec.Template = corev1.PodTemplateSpec{
// pod.Annotations is assigned directly, not merged -- nothing
// else sets pod-template annotations today. If a future change
// needs the controller to set one of its own (e.g. a
// Prometheus-scrape annotation), this needs to become a real
// map merge with a stated precedence rather than silently
// clobbering one side.
ObjectMeta: metav1.ObjectMeta{Labels: labels, Annotations: pod.Annotations},
Spec: corev1.PodSpec{
EnableServiceLinks: new(false),
NodeSelector: pod.NodeSelector,
Tolerations: pod.Tolerations,
Affinity: pod.Affinity,
TopologySpreadConstraints: pod.TopologySpreadConstraints,
SecurityContext: pod.SecurityContext,
ServiceAccountName: pod.ServiceAccountName,
ImagePullSecrets: pod.ImagePullSecrets,
InitContainers: []corev1.Container{waitForPostgresContainer(dbEnv)},
Volumes: pod.ExtraVolumes,
Containers: []corev1.Container{{
Name: "terdut-server",
Image: fmt.Sprintf("%s:%s", srv.Spec.Image.Repository, srv.Spec.Image.Tag),
Ports: []corev1.ContainerPort{{
Name: "http",
ContainerPort: servicePort(srv),
Protocol: corev1.ProtocolTCP,
}},
Env: buildEnv(srv, dbEnv),
EnvFrom: pod.ExtraEnvFrom,
VolumeMounts: pod.ExtraVolumeMounts,
Resources: pod.Resources,
SecurityContext: pod.ContainerSecurityContext,
LivenessProbe: healthzProbe(),
ReadinessProbe: healthzProbe(),
}},
},
}
return controllerutil.SetControllerReference(srv, deploy, r.Scheme)
})
if err != nil {
return nil, err
}
// CreateOrUpdate's own object is desired-state-only on a fresh create
// (no .status yet); re-fetch so the ready-replica check the caller does
// next sees the real, live object.
if err := r.Get(ctx, client.ObjectKeyFromObject(deploy), deploy); err != nil {
return nil, err
}
return deploy, nil
}
// reconcileService creates or updates the ClusterIP Service in front of the
// Deployment, owned by srv.
func (r *TerdutServerReconciler) reconcileService(ctx context.Context, srv *terdutv1alpha1.TerdutServer) error {
svc := &corev1.Service{ObjectMeta: metav1.ObjectMeta{Name: srv.Name, Namespace: srv.Namespace}}
_, err := controllerutil.CreateOrUpdate(ctx, r.Client, svc, func() error {
svc.Spec.Selector = labelsFor(srv)
svc.Spec.Ports = []corev1.ServicePort{{
Name: "http",
Port: servicePort(srv),
TargetPort: intstr.FromString("http"),
Protocol: corev1.ProtocolTCP,
}}
return controllerutil.SetControllerReference(srv, svc, r.Scheme)
})
return err
}
func servicePort(srv *terdutv1alpha1.TerdutServer) int32 {
if srv.Spec.Networking.ServicePort == 0 {
return 8080
}
return srv.Spec.Networking.ServicePort
}
// waitForPostgresContainer blocks the main container from starting until
// Postgres accepts connections, matching charts/terdut-server's own
// deployment.yaml template as of v0.33.2 (that repo's CLAUDE.md/release
// notes) -- that chart grew this the moment this exact Deployment, created
// by this controller, crash-looped a few times against a from-scratch
// postgres-operator cluster still doing initdb and Patroni leader election:
// terdut-server's own ping-retry budget on startup (internal/db/db.go) is
// sized for a much shorter, different race (NetworkPolicy propagation, a
// few seconds), not for genuine first-time cluster creation, so it
// exhausted and the process exited before ever binding its HTTP port -- a
// startupProbe cannot help there, since the crash happens before there is
// anything to probe.
//
// Reuses dbEnv unchanged: both of resolveDatabaseEnv's paths put
// TERDUT_DB_DSN first (terdutserver_database.go), so it's already exactly
// what pg_isready needs, and pg_isready needs no credentials -- it reports
// PQPING_OK on anything that amounts to a Postgres backend answering,
// including an auth challenge -- so including dbEnv's optional PGPASSWORD
// here too is harmless rather than load-bearing.
func waitForPostgresContainer(dbEnv []corev1.EnvVar) corev1.Container {
return corev1.Container{
Name: "wait-for-postgres",
Image: "postgres:17-alpine",
Env: dbEnv,
Command: []string{"sh", "-c",
`until pg_isready -d "$TERDUT_DB_DSN"; do echo "wait-for-postgres: not ready yet, retrying in 2s"; sleep 2; done`,
},
}
}
func healthzProbe() *corev1.Probe {
return &corev1.Probe{
ProbeHandler: corev1.ProbeHandler{
HTTPGet: &corev1.HTTPGetAction{
Path: "/healthz",
Port: intstr.FromString("http"),
},
},
InitialDelaySeconds: 5,
}
}
// buildEnv mirrors charts/terdut-server's own deployment.yaml env block
// field-for-field (confirmed against that source, not reconstructed from
// DESIGN.md's illustrative YAML alone) — dbEnv (TERDUT_DB_DSN, optionally
// PGPASSWORD) comes from resolveDatabaseEnv, since which of §8's two paths
// produced it doesn't matter past this point. spec.pod.extraEnv is appended
// last, after every fixed var -- this is the one place that owns "what env
// this container gets," so the escape hatch lives here rather than being
// appended separately in reconcileDeployment.
func buildEnv(srv *terdutv1alpha1.TerdutServer, dbEnv []corev1.EnvVar) []corev1.EnvVar {
env := []corev1.EnvVar{{Name: "TERDUT_ADDR", Value: fmt.Sprintf(":%d", servicePort(srv))}}
env = append(env, dbEnv...)
sweeper := srv.Spec.Sweeper
env = append(env,
corev1.EnvVar{Name: "TERDUT_STALE_AFTER", Value: sweeper.StaleAfter},
corev1.EnvVar{Name: "TERDUT_ARCHIVE_AFTER", Value: sweeper.ArchiveAfter},
)
deadman := srv.Spec.Deadman
env = append(env,
corev1.EnvVar{Name: "TERDUT_DEADMAN_MATCHERS", Value: deadman.Matchers},
corev1.EnvVar{Name: "TERDUT_DEADMAN_TIMEOUT", Value: deadman.Timeout},
corev1.EnvVar{Name: "TERDUT_DEADMAN_SEVERITY", Value: deadman.Severity},
)
notify := srv.Spec.Notify
if notify.NtfyURL != "" {
env = append(env,
corev1.EnvVar{Name: "TERDUT_NTFY_URL", Value: notify.NtfyURL},
corev1.EnvVar{Name: "TERDUT_NTFY_FALLBACK_TOPIC", Value: notify.FallbackTopic},
corev1.EnvVar{Name: "TERDUT_NOTIFY_REPEAT", Value: notify.RepeatEvery},
)
if notify.TokenSecretRef != nil {
env = append(env, corev1.EnvVar{
Name: "TERDUT_NTFY_TOKEN",
ValueFrom: secretEnvSource(notify.TokenSecretRef),
})
}
}
publicURL := fmt.Sprintf("https://%s", srv.Spec.Networking.Hostname)
env = append(env,
corev1.EnvVar{Name: "TERDUT_PUBLIC_URL", Value: publicURL},
corev1.EnvVar{Name: "TERDUT_PASSWORD_LOGIN", Value: boolString(srv.Spec.PasswordLogin)},
// Always on, unlike the chart's own default-off operatorMode: every
// write this operator's own TerdutTeam/EscalationRule/etc.
// controllers make goes through a service account already, and the
// whole reason to run this operator is gitops-managed config, not a
// human editing an operator-managed install's teams/policies by
// hand (chart's values.yaml comment: "a statement that something
// like terdut-operator... owns this install's configuration from
// here on" -- which is unconditionally true for anything this
// operator creates).
corev1.EnvVar{Name: "TERDUT_OPERATOR_MODE", Value: "true"},
)
if oidc := srv.Spec.OIDC; oidc.Enabled {
env = append(env,
corev1.EnvVar{Name: "TERDUT_OIDC_ISSUER", Value: oidc.Issuer},
corev1.EnvVar{Name: "TERDUT_OIDC_CLIENT_ID", Value: oidc.ClientID},
corev1.EnvVar{Name: "TERDUT_OIDC_NAME", Value: oidc.Name},
corev1.EnvVar{Name: "TERDUT_OIDC_SCOPES", Value: oidc.Scopes},
// terdut-server's own defaults for the claims/trust-email knobs
// the chart exposes but DESIGN.md's spec doesn't (§4.1's doc
// comment on OIDCSpec) -- not configurable here, not an
// oversight.
corev1.EnvVar{Name: "TERDUT_OIDC_USERNAME_CLAIM", Value: "preferred_username"},
corev1.EnvVar{Name: "TERDUT_OIDC_EMAIL_CLAIM", Value: "email"},
corev1.EnvVar{Name: "TERDUT_OIDC_GROUPS_CLAIM", Value: "groups"},
corev1.EnvVar{Name: "TERDUT_OIDC_TRUST_EMAIL", Value: "false"},
corev1.EnvVar{Name: "TERDUT_OIDC_SESSION_MAX_AGE", Value: oidc.SessionMaxAge},
)
if oidc.ClientSecretRef != nil {
env = append(env, corev1.EnvVar{
Name: "TERDUT_OIDC_CLIENT_SECRET",
ValueFrom: secretEnvSource(oidc.ClientSecretRef),
})
}
if len(oidc.AllowedGroups) > 0 {
env = append(env, corev1.EnvVar{Name: "TERDUT_OIDC_ALLOWED_GROUPS", Value: strings.Join(oidc.AllowedGroups, ",")})
}
if oidc.AdminGroup != "" {
env = append(env, corev1.EnvVar{Name: "TERDUT_OIDC_ADMIN_GROUP", Value: oidc.AdminGroup})
}
}
return append(env, srv.Spec.Pod.ExtraEnv...)
}
func secretEnvSource(ref *terdutv1alpha1.SecretKeyRef) *corev1.EnvVarSource {
return &corev1.EnvVarSource{
SecretKeyRef: &corev1.SecretKeySelector{
LocalObjectReference: corev1.LocalObjectReference{Name: ref.Name},
Key: ref.Key,
},
}
}
func boolString(b bool) string {
if b {
return "true"
}
return "false"
}