Add service accounts and operator mode

Service accounts (SERVICE-ACCOUNTS.md) are a scoped, non-human credential:
not a users row, so they never touch OIDC sync, login or the is_admin flag.
Instance scope can create a team and mint a team-scoped account for it;
team scope is owner-equivalent for that one team and nothing else. This is
what unblocks terdut-operator's DESIGN.md §6 — no more impersonating a
human admin, and a real rotation story instead of the unworkable
delete-and-re-bootstrap /api/bootstrap can't actually do.

- migration 014: service_accounts + service_account_keys
- POST /api/service-accounts, POST/DELETE .../keys, GET ?name= self-lookup
- AuthMiddleware resolves a tdsa_-prefixed key to a distinct principal;
  a team-scoped account gets a synthetic single membership so
  requireTeamMember/requireTeamOwner work on it unmodified
- handleCreateTeam accepts an instance-scoped caller; the team it creates
  has no human owner, which is the expected shape for one an operator is
  about to hand a team-scoped credential to

Operator mode (TERDUT_OPERATOR_MODE / values.operatorMode) declares an
install gitops-managed: session and user-API-key writes to teams,
escalation policies, dead man's switches and integrations get 403
reason=operator_managed, while a service account's writes still go
through. Team membership/invites and the schedule are deliberately left
out — never gitops-managed by design, and still human day-to-day work.
/api/auth/config reports operator_mode so the web UI can grey these
sections out from the start rather than only after a write fails.

Also: GET /api/version (both terdut-tui and terdut-operator currently
detect server capability by route-probing; this gives them a real answer),
and a PUT for dead man's switches so a reconciler can update one in place
instead of deleting and recreating it.
This commit is contained in:
Niklas Ye
2026-09-29 21:25:17 +02:00
parent b5573fbca2
commit a4dd60f6b8
15 changed files with 1111 additions and 45 deletions
+122 -7
View File
@@ -116,6 +116,15 @@ func handleUserTeams(db *sql.DB) http.HandlerFunc {
// handleCreateTeam creates a team and makes its creator the first owner. A team
// with no owner would need an administrator to repair before anybody could use
// it, so the two happen in one transaction.
//
// An instance-scoped service account may also create a team (SERVICE-ACCOUNTS.md:
// it acts with the same reach system administration has over teams), but it
// is not a users row and cannot become an owner the way a person does. The
// team it creates starts with no human owner at all — not a bug, the expected
// shape for one terdut-operator is about to provision: a system administrator
// can always act as owner to repair or hand it off (requireTeamOwner), and
// the account that created it mints itself a team-scoped credential for it
// next, via POST /api/service-accounts.
func handleCreateTeam(db *sql.DB) http.HandlerFunc {
return func(w http.ResponseWriter, r *http.Request) {
var req struct {
@@ -131,7 +140,14 @@ func handleCreateTeam(db *sql.DB) http.HandlerFunc {
return
}
caller, _ := userFromContext(r.Context())
caller, isUser := userFromContext(r.Context())
if !isUser && !isInstanceServiceAccount(r.Context()) {
// A team-scoped service account authenticates as owner of exactly
// one team already (see serveAsServiceAccount); letting it create
// another would reach outside that boundary.
respond(w, http.StatusForbidden, errResp("instance-scoped service account or user access required"))
return
}
tx, err := db.BeginTx(r.Context(), nil)
if err != nil {
@@ -152,11 +168,13 @@ func handleCreateTeam(db *sql.DB) http.HandlerFunc {
respond(w, http.StatusInternalServerError, errResp("internal error"))
return
}
if _, err := tx.ExecContext(r.Context(),
"INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3)",
team.ID, caller.ID, models.RoleOwner); err != nil {
respond(w, http.StatusInternalServerError, errResp("internal error"))
return
if isUser {
if _, err := tx.ExecContext(r.Context(),
"INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3)",
team.ID, caller.ID, models.RoleOwner); err != nil {
respond(w, http.StatusInternalServerError, errResp("internal error"))
return
}
}
if err := tx.Commit(); err != nil {
respond(w, http.StatusInternalServerError, errResp("internal error"))
@@ -164,7 +182,9 @@ func handleCreateTeam(db *sql.DB) http.HandlerFunc {
}
team.CreatedAt = time.Unix(created, 0).UTC()
team.Role = models.RoleOwner
if isUser {
team.Role = models.RoleOwner
}
respond(w, http.StatusCreated, team)
}
}
@@ -861,6 +881,101 @@ func handleCreateTeamDeadman(db *sql.DB) http.HandlerFunc {
}
}
// handleUpdateTeamDeadman replaces one switch's configuration in place.
// Added alongside create/delete so an automated caller (terdut-operator) can
// reconcile a spec change without deleting and recreating the switch, which
// would otherwise be the only option and would needlessly rotate its id for
// no reason a reconciler's diff should ever manufacture.
func handleUpdateTeamDeadman(db *sql.DB) http.HandlerFunc {
return func(w http.ResponseWriter, r *http.Request) {
teamID, ok := teamParam(w, r)
if !ok {
return
}
if !requireTeamOwner(w, r, teamID) {
return
}
switchID, err := strconv.ParseInt(chi.URLParam(r, "switchID"), 10, 64)
if err != nil {
respond(w, http.StatusBadRequest, errResp("invalid switch id"))
return
}
var req deadmanSwitchRequest
if err := decodeJSON(r, &req); err != nil {
respond(w, http.StatusBadRequest, errResp("invalid request body"))
return
}
req.Matcher = strings.TrimSpace(req.Matcher)
req.Name = strings.TrimSpace(req.Name)
if req.Severity == "" {
req.Severity = "critical"
}
if !deadmanSeverities[req.Severity] {
respond(w, http.StatusBadRequest, errResp("severity must be critical, error, warning or info"))
return
}
if req.TimeoutSeconds <= 0 {
respond(w, http.StatusBadRequest, errResp("timeout_seconds must be positive"))
return
}
if strings.Contains(req.Matcher, ";") {
respond(w, http.StatusBadRequest, errResp("one matcher per switch: add another switch instead of separating with ;"))
return
}
m, err := parseDeadmanMatcher(req.Matcher)
if err != nil {
respond(w, http.StatusBadRequest, errResp(
"unusable matcher ("+err.Error()+"): each must name an alertname, as in alertname=Watchdog,cluster=prod"))
return
}
if req.Name == "" {
req.Name = m.config()
}
if len(req.Name) > 100 {
respond(w, http.StatusBadRequest, errResp("name is too long"))
return
}
res, err := db.ExecContext(r.Context(), `
UPDATE deadman_switches
SET name = $1, matcher = $2, timeout_seconds = $3, severity = $4
WHERE id = $5 AND team_id = $6`,
req.Name, m.config(), req.TimeoutSeconds, req.Severity, switchID, teamID)
if err != nil {
respond(w, http.StatusInternalServerError, errResp("internal error"))
return
}
if n, _ := res.RowsAffected(); n == 0 {
respond(w, http.StatusNotFound, errResp("switch not found"))
return
}
// The full status, not the bare request echoed back: an update can
// change whether the switch is dormant, alive or dead (a longer
// timeout can revive one that just went dead), and a caller
// reconciling against status deserves the same view
// handleListTeamDeadman would give it.
set, err := deadmanSetForTeam(r.Context(), db, teamID)
if err != nil {
respond(w, http.StatusInternalServerError, errResp("internal error"))
return
}
statuses, err := deadmanStatuses(r.Context(), db, teamID, set, time.Now())
if err != nil {
respond(w, http.StatusInternalServerError, errResp("internal error"))
return
}
for _, s := range statuses {
if s.ID == switchID {
respond(w, http.StatusOK, s)
return
}
}
respond(w, http.StatusInternalServerError, errResp("internal error"))
}
}
// handleDeleteTeamDeadman removes a switch. An incident it already opened stays
// open until somebody resolves it: deleting the switch says "stop watching", not
// "the problem is gone".