From 9029d48584fc86b0c0fbc1e897616ddc6baacde4 Mon Sep 17 00:00:00 2001 From: Niklas Ye Date: Fri, 9 Oct 2026 14:56:13 +0200 Subject: [PATCH] Let an operator authenticate with a seeded key, and reset the schema - TERDUT_OPERATOR_KEY creates or re-keys the instance-scoped service account "terdut-operator" at every start, so terdut-operator needs no bootstrap handshake. An instance-scoped account now acts as owner of every team's configuration, but is not a member of any team. - POST /api/teams takes an external_id (instance service accounts only) and is idempotent on it, so automation finds its own team again after a crash instead of adopting by display name. GET /api/teams?name= is removed. - Integration and dead man's switch names are unique per team (409). The escalation PUT accepts usernames and resolves them itself. - The 18 migrations are squashed into 001_schema.sql, with no Default team. TERDUT_DEADMAN_* and the env seeding of switches are removed: teams carry their own. Existing development databases must be recreated. Security and robustness: - GET /api/users no longer returns other people's email or ntfy topic to non-admins. - The access log records the route pattern, so integration keys and ack tokens in the path are not written to the log. Server errors are logged. - Rate limits take the client address TERDUT_TRUSTED_PROXIES hops from the right of X-Forwarded-For instead of trusting the first, forgeable entry. - /api/bootstrap runs in a transaction under an advisory lock, so two concurrent calls cannot both create an administrator. - API key last_used_at is written at most every five minutes. Cleanup: remove GET /api/incidents/{id}/alerts, unused exports, SQLite remnants in comments and config. Claude-Session: https://claude.ai/code/session_016mBLURvJoMuUEr9cB2RpUN --- .gitea/workflows/ci.yaml | 4 +- .gitignore | 2 - Makefile | 2 +- .../terdut-server/templates/deployment.yaml | 8 +- charts/terdut-server/values.yaml | 47 +- cmd/terdut/main.go | 13 +- internal/api/alertmanager.go | 2 +- internal/api/alerts.go | 6 +- internal/api/api_test.go | 43 +- internal/api/archiver.go | 4 +- internal/api/auth.go | 56 +- internal/api/caller.go | 21 +- internal/api/deadman.go | 112 ---- internal/api/deadman_test.go | 31 +- internal/api/device.go | 22 +- internal/api/escalation.go | 42 +- internal/api/escalation_test.go | 42 ++ internal/api/export_test.go | 31 + internal/api/helpers.go | 29 +- internal/api/incidents.go | 74 +-- internal/api/incidents_test.go | 3 +- internal/api/middleware.go | 80 ++- internal/api/notifier.go | 2 +- internal/api/notify_ack.go | 4 +- internal/api/oidc.go | 6 +- internal/api/oidc_teams.go | 4 +- internal/api/operator_key.go | 72 +++ internal/api/operator_key_test.go | 134 ++++ internal/api/router.go | 14 +- internal/api/schedule.go | 20 +- internal/api/service_accounts.go | 19 +- internal/api/settings.go | 28 +- internal/api/signup.go | 32 +- internal/api/similar.go | 2 +- internal/api/stats.go | 25 +- internal/api/teams.go | 172 ++--- internal/api/teams_test.go | 115 ++-- internal/api/testdb_test.go | 5 +- internal/api/users.go | 93 ++- internal/config/config.go | 71 ++- internal/db/db.go | 8 +- internal/db/migrations/001_schema.sql | 587 ++++++++++++++++++ internal/models/alert.go | 4 +- internal/models/team.go | 4 + internal/oidc/grants.go | 3 - internal/web/static/js/format.js | 2 +- 46 files changed, 1445 insertions(+), 655 deletions(-) create mode 100644 internal/api/export_test.go create mode 100644 internal/api/operator_key.go create mode 100644 internal/api/operator_key_test.go create mode 100644 internal/db/migrations/001_schema.sql diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 693ad74..3111081 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -46,8 +46,8 @@ jobs: - go-build-cache:/root/.cache/go-build - gobin-cache:/go/bin - # The suite needs a real Postgres -- there is no in-memory Postgres the way there was - # an in-memory SQLite, so each test gets its own schema on a shared server instead. + # The suite needs a real Postgres -- there is no in-memory Postgres, + # so each test gets its own schema on a shared server instead. # The job and the service share the dind bridge, so the service is reachable by its # name rather than on localhost. services: diff --git a/.gitignore b/.gitignore index aa08c62..31709d7 100644 --- a/.gitignore +++ b/.gitignore @@ -7,8 +7,6 @@ # one (which has an unreachable entry) cannot abort a release /.helm-repos.yaml -# SQLite database files -*.db *.db-shm *.db-wal diff --git a/Makefile b/Makefile index 3f9b660..d284c30 100644 --- a/Makefile +++ b/Makefile @@ -28,7 +28,7 @@ help: ## Show this help # -race below. Both need a Postgres to test against; see test-db. # The suite needs a Postgres, because the server does: there is no in-memory -# Postgres the way there was an in-memory SQLite. TERDUT_TEST_DSN says where, and +# Postgres. TERDUT_TEST_DSN says where, and # the tests fail rather than skip without it — a suite that quietly tests nothing # is worse than one that does not run. `make test-db` starts a local one; # ci.yaml runs the same thing as a service container. diff --git a/charts/terdut-server/templates/deployment.yaml b/charts/terdut-server/templates/deployment.yaml index 15557d6..49acee1 100644 --- a/charts/terdut-server/templates/deployment.yaml +++ b/charts/terdut-server/templates/deployment.yaml @@ -95,12 +95,6 @@ spec: value: "{{ .Values.sweeper.staleAfter }}" - name: TERDUT_ARCHIVE_AFTER value: "{{ .Values.sweeper.archiveAfter }}" - - name: TERDUT_DEADMAN_MATCHERS - value: "{{ .Values.deadman.matchers }}" - - name: TERDUT_DEADMAN_TIMEOUT - value: "{{ .Values.deadman.timeout }}" - - name: TERDUT_DEADMAN_SEVERITY - value: "{{ .Values.deadman.severity }}" {{- if .Values.notify.ntfyUrl }} - name: TERDUT_NTFY_URL value: "{{ .Values.notify.ntfyUrl }}" @@ -122,6 +116,8 @@ spec: value: "{{ .Values.notify.publicUrl | default (printf "https://%s" .Values.networking.hostname) }}" - name: TERDUT_PASSWORD_LOGIN value: {{ .Values.passwordLogin | quote }} + - name: TERDUT_TRUSTED_PROXIES + value: {{ .Values.trustedProxies | quote }} - name: TERDUT_OPERATOR_MODE value: {{ .Values.operatorMode | quote }} {{- if .Values.oidc.enabled }} diff --git a/charts/terdut-server/values.yaml b/charts/terdut-server/values.yaml index 64d3de4..e084c23 100644 --- a/charts/terdut-server/values.yaml +++ b/charts/terdut-server/values.yaml @@ -72,43 +72,10 @@ sweeper: # How long a resolved alert stays in the default list before auto-archiving. archiveAfter: 168h -# Alerts treated as dead man's switches: receiving one opens no incident, and -# the absence of one does. The Watchdog alert kube-prometheus-stack ships is -# exactly this — an always-firing alert whose only value is something noticing -# when it stops. -deadman: - # Which alerts to treat as heartbeats. ";" separates matchers, "," separates - # the label conditions within one, "=" is exact equality. Every matcher must - # name an alertname: - # alertname=Watchdog,cluster=prod; alertname=EdgeHeartbeat - # Each distinct label set is watched independently, so two clusters sending - # the same alertname are two switches and a live one cannot mask a dead one. - matchers: "alertname=Watchdog" - # How long a heartbeat may go unheard before its switch is declared dead. - # - # This must be SHORTER than the Alertmanager repeat_interval of the route - # carrying the heartbeat — the opposite of sweeper.staleAfter. The default - # repeat_interval of 4h (12h in many setups) makes for a useless dead man's - # switch, so give the heartbeat a route of its own: - # - # - matchers: [ 'alertname = "Watchdog"' ] - # receiver: terdut - # group_wait: 0s - # group_interval: 1m - # repeat_interval: 1m - # - # That delivers every 2m rather than every 1m: a group is only reconsidered - # each group_interval, and at exactly one elapsed interval repeat_interval has - # not quite passed, so equal values give 2x. Fine against 15m; use - # group_interval: 30s if you want a true 1m. - # - # Set to 0 to disable dead man's switch handling entirely. - timeout: 15m - # Severity a dead man's switch incident opens at. These incidents have no - # member alerts to derive one from, and the heartbeat's own severity label is - # meaningless — Watchdog ships as "none". Only "critical" maps to the ntfy - # priority that overrides a phone's quiet hours. - severity: critical +# How many reverse proxies in front of the server append to X-Forwarded-For. +# The per-address login/sign-up rate limits take the client address that many +# entries from the right. 0 ignores the header. +trustedProxies: 1 notify: # ntfy server that push notifications are published to, e.g. @@ -190,10 +157,8 @@ oidc: # Hard ceiling on a session made by a single sign-on login. sessionMaxAge: 12h -# Backups are no longer this chart's business. The SQLite database lived on a PVC -# beside the app, so it needed a sidecar with a sqlite3 module for k8up to exec a -# dump in; Postgres is backed up where it runs, through a k8up.io/backupcommand -# pg_dump annotation on the database pod itself. +# Backups are not this chart's business: Postgres is backed up where it runs, +# through a k8up.io/backupcommand pg_dump annotation on the database pod itself. bootstrap: enabled: true diff --git a/cmd/terdut/main.go b/cmd/terdut/main.go index 2997f5c..08d8661 100644 --- a/cmd/terdut/main.go +++ b/cmd/terdut/main.go @@ -39,21 +39,16 @@ func main() { RepeatEvery: cfg.NotifyRepeat, } - // Dead man's switches live per team now. The environment variables are the - // defaults a team starts from: every team without a configuration of its - // own gets one from them here, and an owner's later edit is never - // overwritten by a redeploy. - deadman := api.ParseDeadmanConfig(cfg.DeadmanMatchers, cfg.DeadmanTimeout, cfg.DeadmanSeverity) - if err := api.SeedDeadmanConfigs(context.Background(), database, deadman); err != nil { - log.Fatalf("seed dead man's switch defaults: %v", err) - } - // The behaviour knobs move into the database on first start, after which an // administrator owns them and a redeploy leaves them alone. if err := api.SeedSettings(context.Background(), database, cfg); err != nil { log.Fatalf("seed settings: %v", err) } + if err := api.SeedOperatorKey(context.Background(), database, cfg.OperatorKey); err != nil { + log.Fatalf("%v", err) + } + router := api.NewRouter(database, notify, cfg, version) srv := &http.Server{ diff --git a/internal/api/alertmanager.go b/internal/api/alertmanager.go index 071684b..7e3cc39 100644 --- a/internal/api/alertmanager.go +++ b/internal/api/alertmanager.go @@ -87,7 +87,7 @@ func handleIntegrationWebhook(db *sql.DB, notify NotifyConfig) http.HandlerFunc respond(w, http.StatusUnauthorized, errResp("unknown integration key")) return } - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } receiveWebhook(w, r, db, notify, src) diff --git a/internal/api/alerts.go b/internal/api/alerts.go index a2b0b2b..5bdecf1 100644 --- a/internal/api/alerts.go +++ b/internal/api/alerts.go @@ -90,7 +90,7 @@ func handleListAlerts(db *sql.DB) http.HandlerFunc { fmt.Sprintf("%s WHERE %s ORDER BY a.received_at DESC LIMIT %s", alertSelectFrom, clause, args.add(limit)), args.all()...) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -99,7 +99,7 @@ func handleListAlerts(db *sql.DB) http.HandlerFunc { for rows.Next() { a, err := scanAlert(rows) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } alerts = append(alerts, a) @@ -121,7 +121,7 @@ func handleGetAlert(db *sql.DB) http.HandlerFunc { return } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, a) diff --git a/internal/api/api_test.go b/internal/api/api_test.go index ec26d42..bae9c57 100644 --- a/internal/api/api_test.go +++ b/internal/api/api_test.go @@ -78,6 +78,10 @@ func newTSWith(t *testing.T, deadman api.DeadmanConfig, cfg api.NotifyConfig, co s := &ts{Server: srv, key: key, db: database, notify: cfg, deadman: deadman} + // A fresh install has no team, so the tests that want "the" team make it + // here: it is id 1, owned by the admin, which is what defaultTeam names. + decode(t, s.req(t, http.MethodPost, "/api/teams", map[string]string{"name": "Default"}), &struct{}{}) + var integration struct { Key string `json:"key"` } @@ -768,7 +772,7 @@ func TestWebhook_IgnoresOutOfOrderRetry(t *testing.T) { // An expiry resolve writes ends_at as an upper bound, not an observed end: an // Alertmanager watermark already on the row is preserved, and a row that never // carried one is stamped at sweep time. Clients are told to read it that way — -// see "resolution_source says how much to trust ends_at" in the README. +// see "resolution_source says how much to trust ends_at" in docs/api.md. func TestExpiry_EndsAtIsUpperBound(t *testing.T) { s := newTS(t) @@ -813,7 +817,7 @@ func TestExpiry_EndsAtIsUpperBound(t *testing.T) { // // received_at is documented as a public liveness signal, so these lock the // behaviour clients are told they may rely on. See "received_at is a liveness -// heartbeat" in the README and the comment on models.Alert.ReceivedAt. +// heartbeat" in docs/api.md and the comment on models.Alert.ReceivedAt. // --------------------------------------------------------------------------- // The heartbeat itself: an unchanged firing notification — what Alertmanager @@ -875,3 +879,38 @@ func TestStats_ByDayReturnsSevenSlots(t *testing.T) { t.Errorf("expected 7 day slots, got %d", len(slots)) } } + +// Two simultaneous bootstraps on an empty install must not both win. +func TestBootstrap_ConcurrentCallsCreateOneAdmin(t *testing.T) { + database := newTestDB(t) + srv := httptest.NewServer(api.NewRouter(database, api.NotifyConfig{}, testConfig(), "test")) + t.Cleanup(srv.Close) + + const n = 8 + codes := make(chan int, n) + for i := 0; i < n; i++ { + go func(i int) { + body, _ := json.Marshal(map[string]string{"username": fmt.Sprintf("u%d", i), "email": fmt.Sprintf("u%d@x.com", i)}) + resp, err := http.Post(srv.URL+"/api/bootstrap", "application/json", bytes.NewReader(body)) + if err != nil { + codes <- 0 + return + } + resp.Body.Close() + codes <- resp.StatusCode + }(i) + } + created := 0 + for i := 0; i < n; i++ { + if <-codes == http.StatusCreated { + created++ + } + } + var users int + if err := database.QueryRow("SELECT COUNT(*) FROM users").Scan(&users); err != nil { + t.Fatal(err) + } + if created != 1 || users != 1 { + t.Errorf("expected exactly one bootstrap to win, got %d created and %d users", created, users) + } +} diff --git a/internal/api/archiver.go b/internal/api/archiver.go index 6759596..b99df7c 100644 --- a/internal/api/archiver.go +++ b/internal/api/archiver.go @@ -154,9 +154,7 @@ func expireStale(ctx context.Context, db *sql.DB, staleAfter time.Duration, skip } // staleAlertIDs reads the ids in one go and closes the cursor before the caller -// writes. Under SQLite's single connection an open read would have blocked the -// update outright; with a pool it is no longer a deadlock, but reading the set -// first still keeps the write off a cursor the same transaction is walking. +// writes, which keeps the write off a cursor the same transaction is walking. func staleAlertIDs(ctx context.Context, db *sql.DB, now time.Time, staleAfter time.Duration) ([]int64, error) { rows, err := db.QueryContext(ctx, ` SELECT id FROM alerts diff --git a/internal/api/auth.go b/internal/api/auth.go index 804f4d5..100acfb 100644 --- a/internal/api/auth.go +++ b/internal/api/auth.go @@ -10,6 +10,7 @@ import ( "strconv" "strings" "sync" + "sync/atomic" "time" "github.com/go-chi/chi/v5" @@ -28,6 +29,9 @@ const ( // sessionTouchEvery bounds how often a request may slide the expiry. sessionTouchEvery = time.Hour + // keyTouchEvery is the same bound for an API key's last_used_at. + keyTouchEvery = 5 * time.Minute + minPasswordLen = 10 // maxPasswordLen is bcrypt's limit; it rejects longer input outright. maxPasswordLen = 72 @@ -126,14 +130,32 @@ func purgeRateLimits(ctx context.Context, db *sql.DB) { } } +// trustedProxies is how many X-Forwarded-For hops clientAddr trusts. Set once +// by NewRouter from config. +var trustedProxies atomic.Int32 + // clientAddr is the address a login is counted against. Behind the gateway -// RemoteAddr is the gateway itself, so the first X-Forwarded-For hop is used -// when present. It can be forged, but only to dodge the address limit; the -// per-username limit does not depend on it. +// RemoteAddr is the gateway itself, so the client address is read from +// X-Forwarded-For, counting trustedProxies entries from the right: each trusted +// proxy appends the address it saw, so the entries to the left of those are +// client-supplied and could be forged to dodge the limit. func clientAddr(r *http.Request) string { - if xff := r.Header.Get("X-Forwarded-For"); xff != "" { - first, _, _ := strings.Cut(xff, ",") - return strings.TrimSpace(first) + if n := int(trustedProxies.Load()); n > 0 { + var hops []string + for _, v := range r.Header.Values("X-Forwarded-For") { + for _, h := range strings.Split(v, ",") { + if h = strings.TrimSpace(h); h != "" { + hops = append(hops, h) + } + } + } + if len(hops) > 0 { + i := len(hops) - n + if i < 0 { + i = 0 + } + return hops[i] + } } host, _, err := net.SplitHostPort(r.RemoteAddr) if err != nil { @@ -241,7 +263,7 @@ func handleLogin(db *sql.DB, limiter *loginLimiter, publicURL string) http.Handl "SELECT id, password_hash FROM users WHERE username = $1", username, ).Scan(&userID, &hash) if err != nil && !errors.Is(err, sql.ErrNoRows) { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -258,13 +280,13 @@ func handleLogin(db *sql.DB, limiter *loginLimiter, publicURL string) http.Handl limiter.clear(r.Context(), userKey) if err := startSession(w, r, db, userID, publicURL); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } user, err := fetchUser(r.Context(), db, userID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, meResponse{User: user, HasPassword: true}) @@ -320,7 +342,7 @@ func handleMe(db *sql.DB) http.HandlerFunc { } user, err := fetchUser(r.Context(), db, caller.ID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } var hash sql.NullString @@ -332,7 +354,7 @@ func handleMe(db *sql.DB) http.HandlerFunc { // transient database problem, not a missing user — worth a 500 // rather than silently answering "no password, not dismissed", // which a client would otherwise take at face value. - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, meResponse{ @@ -384,7 +406,7 @@ func handleSetPassword(db *sql.DB) http.HandlerFunc { return } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -397,30 +419,30 @@ func handleSetPassword(db *sql.DB) http.HandlerFunc { hash, err := hashPassword(req.Password) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } tx, err := db.BeginTx(r.Context(), nil) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer tx.Rollback() if _, err := tx.ExecContext(r.Context(), "UPDATE users SET password_hash = $1 WHERE id = $2", hash, id); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } keep, _ := sessionFromContext(r.Context()) // zero when changed with an API key if _, err := tx.ExecContext(r.Context(), "DELETE FROM sessions WHERE user_id = $1 AND id != $2", id, keep); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if err := tx.Commit(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } w.WriteHeader(http.StatusNoContent) diff --git a/internal/api/caller.go b/internal/api/caller.go index 95501de..a656445 100644 --- a/internal/api/caller.go +++ b/internal/api/caller.go @@ -2,7 +2,6 @@ package api import ( "context" - "fmt" "git.ryuvia.com/niklas/terdut-server/internal/models" ) @@ -61,7 +60,7 @@ func (c Caller) IsAdmin() bool { // by being a human (and becomes its owner as a side effect), an // instance-scoped service account creates one with no human owner at all; // the two paths are not interchangeable, so this predicate must not also -// admit a human admin the way MayActAsInstanceAdmin deliberately does. +// admit a human admin the way an administrator check would. func (c Caller) IsInstanceServiceAccount() bool { return c.sa != nil && c.sa.scope == models.ServiceAccountScopeInstance } @@ -110,24 +109,6 @@ func (c Caller) ServiceAccountName() (string, bool) { return c.sa.name, true } -// Identity is a stable, log/audit-facing string distinguishing a human -// caller from a service account — "user:42" or "service-account:7". Not -// wired into any database column — incidents.go's acknowledged_by/ -// incident_events.user_id use AsHuman()/ServiceAccountID() directly against -// the parallel *_service_account_id columns (migration 015) instead, since a -// column needs the id, not this rendered string. assigned_to stays -// human-only and out of scope (terdut-server#25's follow-up). -func (c Caller) Identity() string { - switch { - case c.user != nil: - return fmt.Sprintf("user:%d", c.user.ID) - case c.sa != nil: - return fmt.Sprintf("service-account:%d", c.sa.id) - default: - return "unknown" - } -} - func callerFromContext(ctx context.Context) (Caller, bool) { c, ok := ctx.Value(ctxCaller).(Caller) return c, ok diff --git a/internal/api/deadman.go b/internal/api/deadman.go index 70bbd01..6b813f0 100644 --- a/internal/api/deadman.go +++ b/internal/api/deadman.go @@ -65,28 +65,6 @@ func (m DeadmanMatcher) matches(labels map[string]string) bool { return true } -// DeadmanConfig is the server-wide default a team's switches are seeded from: -// the environment's matchers, timeout and severity. Switches themselves are rows -// of a team's own — see DeadmanSwitch — and this is only how a fresh install -// starts out. -type DeadmanConfig struct { - Matchers []DeadmanMatcher - - // Timeout is how long a matched alert may go without a refreshing webhook - // before it is declared dead. It must be shorter than Alertmanager's - // repeat_interval for the heartbeat's route, which is what refreshes it. - // Zero disables dead man's switch handling entirely. - Timeout time.Duration - - // Severity is the severity every dead man's switch incident opens at. These - // incidents have no member alerts to derive one from, and the heartbeat's - // own severity label is meaningless — Watchdog ships as "none". - Severity string -} - -// enabled reports whether there is anything to watch. -func (c DeadmanConfig) enabled() bool { return c.Timeout > 0 && len(c.Matchers) > 0 } - // DeadmanSwitch inverts the handling of the alerts it matches: receiving one // opens nothing, and the absence of one opens an incident. // @@ -160,46 +138,6 @@ func parseDeadmanMatcher(entry string) (DeadmanMatcher, error) { return m, nil } -// ParseDeadmanConfig reads the matcher list from its configured form: -// ";" separates matchers, and each is parsed as parseDeadmanMatcher does. -// -// A malformed or alertname-less entry is dropped rather than fatal, following -// config.duration's rule that one bad tuning knob should not take the server -// down. Silence would be worse here than elsewhere, though — a typo that -// disarms the switch is exactly the failure this feature exists to catch — so -// the matchers that survived are logged. -func ParseDeadmanConfig(matchers string, timeout time.Duration, severity string) DeadmanConfig { - cfg := DeadmanConfig{Timeout: timeout, Severity: severity} - - for _, entry := range strings.Split(matchers, ";") { - entry = strings.TrimSpace(entry) - if entry == "" { - continue - } - m, err := parseDeadmanMatcher(entry) - if err != nil { - log.Printf("deadman: ignoring matcher %q: %v", entry, err) - continue - } - cfg.Matchers = append(cfg.Matchers, m) - } - - switch { - case timeout <= 0: - log.Print("deadman: disabled (timeout is zero)") - case len(cfg.Matchers) == 0: - log.Print("deadman: disabled (no usable matchers)") - default: - rendered := make([]string, 0, len(cfg.Matchers)) - for _, m := range cfg.Matchers { - rendered = append(rendered, m.String()) - } - log.Printf("deadman: default for new teams: %s, timeout %s, severity %s", - strings.Join(rendered, "; "), timeout, severity) - } - return cfg -} - // deadmanAlert is one heartbeat: the alert row carrying its last sighting, and // the switch that claimed it. type deadmanAlert struct { @@ -490,56 +428,6 @@ func deadmanSets(ctx context.Context, db *sql.DB) (map[int64]deadmanSet, error) return scanDeadmanSwitches(rows) } -// deadmanSeededKey is the settings row that records the environment defaults -// were handed out. Without it, a team that deleted its last switch would get -// the default back on the next restart. -const deadmanSeededKey = "deadman_seeded" - -// SeedDeadmanConfigs gives every team the server's environment defaults as -// switches, exactly once per install, so a fresh install watches Watchdog -// without anybody setting it up. -// -// Once seeded it never runs again: a team's switches are its own, and a redeploy -// must not quietly put the environment's value back over an owner's edit or -// deletion. Installs that upgraded from per-team configuration were already -// seeded, which migration 009 records. -// -// A team created after that gets none and watches nothing until its owner says -// otherwise. That is deliberate: inheriting an install-wide heartbeat would page -// a new team about a source it has never heard of, and a switch nobody chose is -// the kind that gets muted rather than fixed. -func SeedDeadmanConfigs(ctx context.Context, db *sql.DB, cfg DeadmanConfig) error { - if !cfg.enabled() { - return nil - } - - tx, err := db.BeginTx(ctx, nil) - if err != nil { - return err - } - defer tx.Rollback() //nolint:errcheck - - res, err := tx.ExecContext(ctx, - "INSERT INTO settings (key, value) VALUES ($1, '1') ON CONFLICT (key) DO NOTHING", - deadmanSeededKey) - if err != nil { - return err - } - if n, _ := res.RowsAffected(); n == 0 { - return nil - } - - for _, m := range cfg.Matchers { - if _, err := tx.ExecContext(ctx, ` - INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity) - SELECT id, $1, $1, $2, $3 FROM teams`, - m.config(), int64(cfg.Timeout.Seconds()), cfg.Severity); err != nil { - return err - } - } - return tx.Commit() -} - // --------------------------------------------------------------------------- // Status // --------------------------------------------------------------------------- diff --git a/internal/api/deadman_test.go b/internal/api/deadman_test.go index 4d6d172..485be77 100644 --- a/internal/api/deadman_test.go +++ b/internal/api/deadman_test.go @@ -1,7 +1,6 @@ package api_test import ( - "context" "net/http" "strings" "testing" @@ -122,8 +121,11 @@ func TestDeadman_MixedGroupExcludesHeartbeat(t *testing.T) { t.Fatalf("expected 1 incident for the real alert, got %d", got) } - var alerts []map[string]any - decode(t, s.req(t, http.MethodGet, "/api/incidents/1/alerts", nil), &alerts) + var incident struct { + Alerts []map[string]any `json:"alerts"` + } + decode(t, s.req(t, http.MethodGet, "/api/incidents/1", nil), &incident) + alerts := incident.Alerts if len(alerts) != 1 { t.Fatalf("expected 1 member alert, got %d", len(alerts)) } @@ -717,26 +719,3 @@ func TestDeadman_DeleteIsScopedToTheTeam(t *testing.T) { t.Errorf("expected no switches, got %d", got) } } - -// The environment's defaults are handed out once and then belong to the teams. -func TestDeadman_SeedRunsOnce(t *testing.T) { - s := newTS(t) - cfg := api.ParseDeadmanConfig("alertname=Watchdog", time.Hour, "critical") - - if err := api.SeedDeadmanConfigs(context.Background(), s.db, cfg); err != nil { - t.Fatalf("seed: %v", err) - } - if got := len(listSwitches(t, s)); got != 1 { - t.Fatalf("the first seed should add the default, got %d switches", got) - } - - // The owner deletes it; a restart must not put it back. - id := int64(listSwitches(t, s)[0]["id"].(float64)) - s.req(t, http.MethodDelete, "/api/teams/"+defaultTeam+"/deadman/switches/"+id64(id), nil).Body.Close() - if err := api.SeedDeadmanConfigs(context.Background(), s.db, cfg); err != nil { - t.Fatalf("seed again: %v", err) - } - if got := len(listSwitches(t, s)); got != 0 { - t.Errorf("a second seed resurrected %d switch(es)", got) - } -} diff --git a/internal/api/device.go b/internal/api/device.go index d694b50..120dad9 100644 --- a/internal/api/device.go +++ b/internal/api/device.go @@ -81,7 +81,7 @@ func handleDeviceStart(db *sql.DB, limiter *loginLimiter, publicURL string) http deviceCode, deviceHash, err := randomToken() if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -105,7 +105,7 @@ func handleDeviceStart(db *sql.DB, limiter *loginLimiter, publicURL string) http } if err != nil { log.Printf("device login: start: %v", err) - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -160,7 +160,7 @@ func handleDeviceDecision(db *sql.DB, approve bool) http.HandlerFunc { WHERE user_code = $3 AND status = 'pending' AND expires_at > $4`, status, caller.ID, code, time.Now().Unix()) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -190,7 +190,7 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht tx, err := db.BeginTx(r.Context(), nil) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer tx.Rollback() //nolint:errcheck @@ -206,7 +206,7 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht return } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -226,11 +226,11 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht } if _, err := tx.ExecContext(r.Context(), "UPDATE device_logins SET last_polled_at = $1 WHERE device_hash = $2", now.Unix(), hash); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if err := tx.Commit(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusAccepted, map[string]string{"status": "pending"}) @@ -240,7 +240,7 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht // Approved. Single use: the row goes before the session is made, so two // racing polls cannot both be given one. if _, err := tx.ExecContext(r.Context(), "DELETE FROM device_logins WHERE device_hash = $1", hash); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } var disabled, sso bool @@ -252,7 +252,7 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht return } if err := tx.Commit(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if disabled { @@ -269,12 +269,12 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht } if err := startSessionCapped(w, r, db, userID.Int64, publicURL, maxAge); err != nil { log.Printf("device login: start session: %v", err) - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } user, err := fetchUser(r.Context(), db, userID.Int64) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, meResponse{User: user, HasPassword: false}) diff --git a/internal/api/escalation.go b/internal/api/escalation.go index 59c3a84..ed3275c 100644 --- a/internal/api/escalation.go +++ b/internal/api/escalation.go @@ -3,6 +3,7 @@ package api import ( "context" "database/sql" + "errors" "log" "net/http" "strconv" @@ -343,12 +344,12 @@ func handleGetEscalation(db *sql.DB) http.HandlerFunc { policy, err := loadEscalationPolicy(r.Context(), db, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } view, err := escalationStatus(r.Context(), db, teamID, policy) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, view) @@ -543,6 +544,10 @@ type escalationLevelJSON struct { type escalationTargetJSON struct { Kind string `json:"kind"` UserID *int64 `json:"user_id,omitempty"` + // Username is accepted in place of user_id on a PUT, and resolved to the + // id before anything is stored. It is never returned: the stored form is + // the id, which survives a rename. + Username string `json:"username,omitempty"` } type escalationJSON struct { @@ -600,6 +605,27 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc { return } for i, l := range req.Levels { + for j, t := range l.Targets { + if t.Username == "" { + continue + } + if t.Kind != "user" || t.UserID != nil { + respond(w, http.StatusBadRequest, errResp("username belongs on a user target, instead of user_id")) + return + } + var id int64 + err := db.QueryRowContext(r.Context(), "SELECT id FROM users WHERE username = $1", t.Username).Scan(&id) + if errors.Is(err, sql.ErrNoRows) { + respond(w, http.StatusBadRequest, errResp("unknown user "+strconv.Quote(t.Username))) + return + } + if err != nil { + serverError(w, r, err) + return + } + req.Levels[i].Targets[j].UserID = &id + req.Levels[i].Targets[j].Username = "" + } if l.TimeoutSeconds <= 0 { respond(w, http.StatusBadRequest, errResp("every level needs a timeout")) return @@ -632,7 +658,7 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc { tx, err := db.BeginTx(r.Context(), nil) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer tx.Rollback() //nolint:errcheck @@ -645,13 +671,13 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc { fallback_topic = excluded.fallback_topic, updated_at = excluded.updated_at`, teamID, req.RepeatCount, req.FallbackTopic); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } // The levels are replaced, not merged; the cascade takes the targets. if _, err := tx.ExecContext(r.Context(), "DELETE FROM escalation_levels WHERE team_id = $1", teamID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -661,7 +687,7 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc { INSERT INTO escalation_levels (team_id, position, timeout_seconds) VALUES ($1, $2, $3) RETURNING id`, teamID, int64(i+1), l.TimeoutSeconds).Scan(&levelID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } for _, t := range l.Targets { @@ -676,13 +702,13 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc { } if err := tx.Commit(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } policy, err := loadEscalationPolicy(r.Context(), db, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, escalationResponse(policy, teamID)) diff --git a/internal/api/escalation_test.go b/internal/api/escalation_test.go index 39a428a..e36f96f 100644 --- a/internal/api/escalation_test.go +++ b/internal/api/escalation_test.go @@ -502,3 +502,45 @@ func TestEscalation_StatusWithoutALadder(t *testing.T) { t.Errorf("a team with no ladder should read as empty, got %+v", v) } } + +// A user target may name the person instead of carrying an id; the server +// resolves it and stores the id. +func TestEscalation_UserTargetByUsername(t *testing.T) { + s := newTS(t) + id := teamUser(t, s, "alice", "") + + put := func(username string) *http.Response { + return s.req(t, http.MethodPut, "/api/teams/"+defaultTeam+"/escalation", map[string]any{ + "repeat_count": 0, + "levels": []map[string]any{{ + "timeout_seconds": 300, + "targets": []map[string]any{{"kind": "user", "username": username}}, + }}, + }) + } + + resp := put("alice") + resp.Body.Close() + if resp.StatusCode >= 300 { + t.Fatalf("PUT by username: %d", resp.StatusCode) + } + var got struct { + Levels []struct { + Targets []struct { + UserID *int64 `json:"user_id"` + Username string `json:"username"` + } `json:"targets"` + } `json:"levels"` + } + decode(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/escalation", nil), &got) + if len(got.Levels) != 1 || len(got.Levels[0].Targets) != 1 || + got.Levels[0].Targets[0].UserID == nil || *got.Levels[0].Targets[0].UserID != id { + t.Errorf("expected the target stored as user %d, got %+v", id, got) + } + + bad := put("nobody") + bad.Body.Close() + if bad.StatusCode != http.StatusBadRequest { + t.Errorf("unknown username should be a 400, got %d", bad.StatusCode) + } +} diff --git a/internal/api/export_test.go b/internal/api/export_test.go new file mode 100644 index 0000000..9928297 --- /dev/null +++ b/internal/api/export_test.go @@ -0,0 +1,31 @@ +package api + +import ( + "strings" + "time" +) + +// DeadmanConfig is a test fixture only: a set of matchers with one timeout and +// severity, turned into switches over the API by the test helpers. Production +// has no server-wide default any more -- switches belong to teams. +type DeadmanConfig struct { + Matchers []DeadmanMatcher + Timeout time.Duration + Severity string +} + +// ParseDeadmanConfig reads a ";"-separated matcher list the way the removed +// environment variable did, dropping malformed entries. +func ParseDeadmanConfig(matchers string, timeout time.Duration, severity string) DeadmanConfig { + cfg := DeadmanConfig{Timeout: timeout, Severity: severity} + for _, entry := range strings.Split(matchers, ";") { + entry = strings.TrimSpace(entry) + if entry == "" { + continue + } + if m, err := parseDeadmanMatcher(entry); err == nil { + cfg.Matchers = append(cfg.Matchers, m) + } + } + return cfg +} diff --git a/internal/api/helpers.go b/internal/api/helpers.go index 4ea7dcd..1b73b51 100644 --- a/internal/api/helpers.go +++ b/internal/api/helpers.go @@ -5,6 +5,7 @@ import ( "database/sql" "encoding/json" "errors" + "github.com/go-chi/chi/v5" "log" "net/http" "strconv" @@ -17,8 +18,7 @@ import ( // sqlArgs accumulates query arguments and hands back the placeholder for each. // // Postgres numbers its placeholders, so a dynamically assembled WHERE clause has -// to keep its $1, $2, … in step with the order of the values — which SQLite's -// positional `?` did for free. Handing out the placeholder and storing the value +// to keep its $1, $2, … in step with the order of the values. Handing out the placeholder and storing the value // in one call is what keeps them in step: a filter can be added, removed or // reordered without renumbering anything by hand. type sqlArgs struct{ vals []any } @@ -31,8 +31,7 @@ func (a *sqlArgs) add(v any) string { // addList stores every value and returns their placeholders as "$1, $2, …", // ready to drop into an IN (…) clause. Returns an empty string for no values, -// which no caller should reach: `IN ()` is a syntax error in Postgres as it was -// in SQLite, so callers check for an empty set before building the query. +// which no caller should reach: `IN ()` is a syntax error in Postgres, so callers check for an empty set before building the query. func (a *sqlArgs) addList(vs []any) string { parts := make([]string, len(vs)) for i, v := range vs { @@ -45,7 +44,7 @@ func (a *sqlArgs) addList(vs []any) string { func (a *sqlArgs) all() []any { return a.vals } // nowEpoch is the SQL expression for "now, as unix seconds", matching how every -// timestamp in this schema is stored. SQLite spelled it unixepoch(). +// timestamp in this schema is stored. // // FLOOR, not a bare cast: EXTRACT returns fractional seconds and casting to // bigint rounds half up, so a row written at .6 of a second would claim a @@ -56,11 +55,9 @@ const nowEpoch = "FLOOR(EXTRACT(EPOCH FROM now()))::bigint" // isUniqueViolation reports whether err is a broken unique constraint, which // callers turn into 409 Conflict rather than 500. // -// Postgres reports it as SQLSTATE 23505 on a typed error; the SQLite driver this -// replaced only put "UNIQUE constraint failed" in the message, which is why the -// check used to be a substring match. Matching the code means a renamed -// constraint or a translated message cannot quietly turn a conflict back into a -// 500. +// Postgres reports it as SQLSTATE 23505 on a typed error. Matching the code +// means a renamed constraint or a translated message cannot quietly turn a +// conflict back into a 500. func isUniqueViolation(err error) bool { var pgErr *pgconn.PgError return errors.As(err, &pgErr) && pgErr.Code == pgerrcode.UniqueViolation @@ -72,6 +69,18 @@ func respond(w http.ResponseWriter, status int, v any) { json.NewEncoder(w).Encode(v) } +// serverError answers 500 and logs why. The response stays opaque, so the log +// line is the only record of what failed. +func serverError(w http.ResponseWriter, r *http.Request, err error) { + // The route pattern, not the path: two routes carry a credential in it. + route := r.URL.Path + if rc := chi.RouteContext(r.Context()); rc != nil && rc.RoutePattern() != "" { + route = rc.RoutePattern() + } + log.Printf("%s %s: %v", r.Method, route, err) + respond(w, http.StatusInternalServerError, errResp("internal error")) +} + // maxBodyBytes caps an ordinary JSON request body. 1 MiB is far more than any // endpoint below needs — it exists so an unauthenticated caller (signup, // login, bootstrap) can't make the server buffer an arbitrarily large body diff --git a/internal/api/incidents.go b/internal/api/incidents.go index 19c12a1..50cc9ac 100644 --- a/internal/api/incidents.go +++ b/internal/api/incidents.go @@ -37,7 +37,7 @@ func handleListClusters(db *sql.DB) http.HandlerFunc { label, strings.Join(where, " AND ")), args.all()...) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -46,13 +46,13 @@ func handleListClusters(db *sql.DB) http.HandlerFunc { for rows.Next() { var v string if err := rows.Scan(&v); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } clusters = append(clusters, v) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, clusters) @@ -138,7 +138,7 @@ func handleListIncidents(db *sql.DB) http.HandlerFunc { incidentSelectFrom, strings.Join(where, " AND "), order, args.add(limit)), args.all()...) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -147,13 +147,13 @@ func handleListIncidents(db *sql.DB) http.HandlerFunc { for rows.Next() { i, err := scanIncident(rows) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } incidents = append(incidents, i) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, incidents) @@ -172,35 +172,17 @@ func handleGetIncident(db *sql.DB) http.HandlerFunc { return } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if inc.Alerts, err = incidentAlerts(r, db, id); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, inc) } } -func handleIncidentAlerts(db *sql.DB) http.HandlerFunc { - return func(w http.ResponseWriter, r *http.Request) { - id, ok := incidentIDParam(w, r, db) - if !ok { - return - } - if !incidentExists(w, r, db, id) { - return - } - alerts, err := incidentAlerts(r, db, id) - if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) - return - } - respond(w, http.StatusOK, alerts) - } -} - func handleIncidentTimeline(db *sql.DB) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { id, ok := incidentIDParam(w, r, db) @@ -225,7 +207,7 @@ func handleIncidentTimeline(db *sql.DB) http.HandlerFunc { WHERE e.incident_id = $1 ORDER BY e.created_at ASC, e.id ASC`, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -239,14 +221,14 @@ func handleIncidentTimeline(db *sql.DB) http.HandlerFunc { &e.ActorUserID, &e.ActorUsername, &e.ActorServiceAccountID, &e.ActorServiceAccountName, &e.AlertID, &e.Detail, &ts); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } e.CreatedAt = time.Unix(ts, 0).UTC() events = append(events, e) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, events) @@ -262,7 +244,7 @@ func handleIncidentAcknowledge(db *sql.DB) http.HandlerFunc { userID, saID := callerActorIDs(r.Context()) acked, err := acknowledgeIncidentAs(r.Context(), db, id, userID, saID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if !acked { @@ -272,7 +254,7 @@ func handleIncidentAcknowledge(db *sql.DB) http.HandlerFunc { // (acknowledged) already holds. inc, err := fetchIncident(r.Context(), db, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if inc.Status == "resolved" { @@ -300,7 +282,7 @@ func handleIncidentUnacknowledge(db *sql.DB) http.HandlerFunc { return } if err := logEvent(r.Context(), db, id, evUnacknowledged, userID, saID, nil, nil); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } w.WriteHeader(http.StatusNoContent) @@ -335,16 +317,16 @@ func handleIncidentResolve(db *sql.DB) http.HandlerFunc { } // A person closing an incident is the clearest possible "I have this". if err := stopEscalation(r.Context(), db, id); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if err := logEvent(r.Context(), db, id, evResolved, userID, saID, nil, nil); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if req.Resolution != "" { if err := logEvent(r.Context(), db, id, evResolutionNote, userID, saID, nil, &req.Resolution); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } } @@ -385,7 +367,7 @@ func handleIncidentAssign(db *sql.DB) http.HandlerFunc { // the actor_* columns. actorUserID, actorSAID := callerActorIDs(r.Context()) if err := logAssignedEvent(r.Context(), db, id, req.UserID, actorUserID, actorSAID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respondIncident(w, r, db, id) @@ -443,7 +425,7 @@ func handleIncidentSnooze(db *sql.DB) http.HandlerFunc { } detail := until.UTC().Format(time.RFC3339) if err := logEvent(r.Context(), db, id, evSnoozed, userID, saID, nil, &detail); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respondIncident(w, r, db, id) @@ -462,7 +444,7 @@ func handleIncidentUnsnooze(db *sql.DB) http.HandlerFunc { return } if err := logEvent(r.Context(), db, id, evUnsnoozed, userID, saID, nil, nil); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } w.WriteHeader(http.StatusNoContent) @@ -478,7 +460,7 @@ func handleIncidentArchive(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "UPDATE incidents SET archived_at = "+nowEpoch+" WHERE id = $1", id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -487,7 +469,7 @@ func handleIncidentArchive(db *sql.DB) http.HandlerFunc { } userID, saID := callerActorIDs(r.Context()) if err := logEvent(r.Context(), db, id, evArchived, userID, saID, nil, nil); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respondIncident(w, r, db, id) @@ -503,7 +485,7 @@ func handleIncidentUnarchive(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "UPDATE incidents SET archived_at = NULL WHERE id = $1", id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -512,7 +494,7 @@ func handleIncidentUnarchive(db *sql.DB) http.HandlerFunc { } userID, saID := callerActorIDs(r.Context()) if err := logEvent(r.Context(), db, id, evUnarchived, userID, saID, nil, nil); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } w.WriteHeader(http.StatusNoContent) @@ -557,7 +539,7 @@ func handleCreateNote(db *sql.DB) http.HandlerFunc { VALUES ($1, $2, $3, $4, $5, $6) RETURNING id`, id, noteType, userID, saID, req.Content, now.Unix()).Scan(&eventID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -598,7 +580,7 @@ func handleDeleteNote(db *sql.DB) http.HandlerFunc { AND (user_id = $5 OR service_account_id = $6)`, eventID, id, evNote, evResolutionNote, userID, saID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -652,7 +634,7 @@ func incidentExists(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64 func updateOpenIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64, query string, args ...any) bool { res, err := db.ExecContext(r.Context(), query, args...) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return false } if n, _ := res.RowsAffected(); n > 0 { @@ -668,7 +650,7 @@ func updateOpenIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id i func respondIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64) { inc, err := fetchIncident(r.Context(), db, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, inc) diff --git a/internal/api/incidents_test.go b/internal/api/incidents_test.go index 6fbd466..4dc23d7 100644 --- a/internal/api/incidents_test.go +++ b/internal/api/incidents_test.go @@ -794,8 +794,7 @@ func TestStats_Incidents(t *testing.T) { } // An empty window is a report of zero, not a failure. SUM over no rows is NULL -// in Postgres as it was in SQLite, and that used to come back as a 500 the -// moment every incident was archived — the state a quiet installation settles +// in Postgres, and that used to come back as a 500 the moment every incident was archived — the state a quiet installation settles // into. func TestStats_IncidentsEmptyWindowIsZeroNotAnError(t *testing.T) { s := newTS(t) diff --git a/internal/api/middleware.go b/internal/api/middleware.go index ae0aee7..a9e0fc6 100644 --- a/internal/api/middleware.go +++ b/internal/api/middleware.go @@ -5,10 +5,14 @@ import ( "crypto/sha256" "database/sql" "encoding/hex" + "log" "net/http" "strings" "time" + "github.com/go-chi/chi/v5" + "github.com/go-chi/chi/v5/middleware" + "git.ryuvia.com/niklas/terdut-server/internal/config" "git.ryuvia.com/niklas/terdut-server/internal/models" ) @@ -106,7 +110,7 @@ func securityHeaders(publicURL string) func(http.Handler) http.Handler { // // 403 and not 404: the route exists and the caller is authenticated, they are // simply not allowed. Hiding the endpoint would buy nothing — every one of them -// is in the README. +// is in docs/api.md. func AdminOnly(next http.Handler) http.Handler { return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { caller, ok := userFromContext(r.Context()) @@ -141,19 +145,23 @@ func requireSelfOrAdmin(w http.ResponseWriter, r *http.Request, targetID int64) // resolve and has to be caught afterwards. func apiKeyUser(ctx context.Context, db *sql.DB, token string) (int64, bool) { var keyID, userID int64 + var lastUsed sql.NullInt64 err := db.QueryRowContext(ctx, - `SELECT id, user_id FROM api_keys + `SELECT id, user_id, last_used_at FROM api_keys WHERE key_hash = $1 AND (expires_at IS NULL OR expires_at > $2)`, hashToken(token), time.Now().Unix(), - ).Scan(&keyID, &userID) + ).Scan(&keyID, &userID, &lastUsed) if err != nil { return 0, false } - // best-effort; don't fail the request if this update fails - db.ExecContext(ctx, - "UPDATE api_keys SET last_used_at = $1 WHERE id = $2", - time.Now().Unix(), keyID) + // best-effort; don't fail the request if this update fails. Throttled like + // the session expiry, so a polling client does not write a row per request. + if now := time.Now(); !lastUsed.Valid || now.Sub(time.Unix(lastUsed.Int64, 0)) > keyTouchEvery { + db.ExecContext(ctx, + "UPDATE api_keys SET last_used_at = $1 WHERE id = $2", + now.Unix(), keyID) + } return userID, true } @@ -206,7 +214,7 @@ func serveAs(w http.ResponseWriter, r *http.Request, next http.Handler, db *sql. // table with one row per membership. teams, err := callerMemberships(r.Context(), db, userID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -222,10 +230,8 @@ func hashToken(token string) string { return hex.EncodeToString(h[:]) } -// userFromContext is a thin compatibility wrapper over Caller.AsHuman(), so -// every call site written before the Caller abstraction (alerts.go, -// incidents.go, schedule.go, stats.go, and more) needs no change and keeps -// its exact existing behavior. +// userFromContext returns the human behind the request, or false for a service +// account: Caller.AsHuman() on the request's Caller. func userFromContext(ctx context.Context) (models.User, bool) { c, _ := callerFromContext(ctx) return c.AsHuman() @@ -245,13 +251,13 @@ type serviceAccountPrincipal struct { func serviceAccountFor(ctx context.Context, db *sql.DB, token string) (serviceAccountPrincipal, bool) { var sa serviceAccountPrincipal var keyID int64 - var teamID sql.NullInt64 + var teamID, lastUsed sql.NullInt64 err := db.QueryRowContext(ctx, ` - SELECT k.id, a.id, a.name, a.scope, a.team_id + SELECT k.id, a.id, a.name, a.scope, a.team_id, k.last_used_at FROM service_account_keys k JOIN service_accounts a ON a.id = k.service_account_id WHERE k.key_hash = $1`, hashToken(token), - ).Scan(&keyID, &sa.id, &sa.name, &sa.scope, &teamID) + ).Scan(&keyID, &sa.id, &sa.name, &sa.scope, &teamID, &lastUsed) if err != nil { return serviceAccountPrincipal{}, false } @@ -260,9 +266,11 @@ func serviceAccountFor(ctx context.Context, db *sql.DB, token string) (serviceAc } // best-effort; don't fail the request if this update fails - db.ExecContext(ctx, - "UPDATE service_account_keys SET last_used_at = $1 WHERE id = $2", - time.Now().Unix(), keyID) + if now := time.Now(); !lastUsed.Valid || now.Sub(time.Unix(lastUsed.Int64, 0)) > keyTouchEvery { + db.ExecContext(ctx, + "UPDATE service_account_keys SET last_used_at = $1 WHERE id = $2", + now.Unix(), keyID) + } return sa, true } @@ -286,10 +294,8 @@ func serveAsServiceAccount(w http.ResponseWriter, r *http.Request, next http.Han next.ServeHTTP(w, r.WithContext(ctx)) } -// isInstanceServiceAccount is a thin compatibility wrapper over -// Caller.IsInstanceServiceAccount(), for call sites outside this package's -// core predicates (handleCreateTeam, handleCreateServiceAccount) that -// needed this exact, narrow check before the Caller abstraction existed. +// isInstanceServiceAccount is Caller.IsInstanceServiceAccount() on the +// request's Caller. func isInstanceServiceAccount(ctx context.Context) bool { c, _ := callerFromContext(ctx) return c.IsInstanceServiceAccount() @@ -397,6 +403,13 @@ func requireTeamOwner(w http.ResponseWriter, r *http.Request, teamID int64) bool if ok && role == models.RoleOwner { return true } + // The operator's instance-scoped account manages every team's + // configuration, which is what lets it use one credential instead of + // minting one per team. This is owner reach only: it does not make the + // account a member, so it still reads no team's incidents. + if c, _ := callerFromContext(r.Context()); c.IsInstanceServiceAccount() { + return true + } if caller, _ := userFromContext(r.Context()); caller.IsAdmin { return true } @@ -414,3 +427,26 @@ func sessionFromContext(ctx context.Context) (int64, bool) { id, ok := ctx.Value(ctxSession).(int64) return id, ok } + +// requestLogger logs one line per request with the matched route pattern in +// place of the URL path. Two routes carry a credential in the path (the +// integration key and the ack token), and chi's stock logger would write it to +// the log verbatim. +func requestLogger(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + start := time.Now() + ww := middleware.NewWrapResponseWriter(w, r.ProtoMajor) + next.ServeHTTP(ww, r) + route := "unmatched" + if rc := chi.RouteContext(r.Context()); rc != nil { + if p := rc.RoutePattern(); p != "" { + route = p + } + } + status := ww.Status() + if status == 0 { + status = http.StatusOK + } + log.Printf("%s %s %d %dB %s", r.Method, route, status, ww.BytesWritten(), time.Since(start).Round(time.Millisecond)) + }) +} diff --git a/internal/api/notifier.go b/internal/api/notifier.go index 8f5b621..610e79b 100644 --- a/internal/api/notifier.go +++ b/internal/api/notifier.go @@ -449,7 +449,7 @@ func renderNotification(inc models.Incident, n outboxRow, firing int, cfg Notify // originLabel is the label that says where an alert came from, for a team with // several Kubernetes clusters behind it. It comes from Prometheus's // externalLabels and reaches an incident through Alertmanager's group_by; the -// web UI reads the same label, and the README ("Several clusters, one team") +// web UI reads the same label, and docs/incidents.md ("Several clusters, one team") // explains how to set it up. const originLabel = "cluster" diff --git a/internal/api/notify_ack.go b/internal/api/notify_ack.go index 43c4507..86e9e63 100644 --- a/internal/api/notify_ack.go +++ b/internal/api/notify_ack.go @@ -62,7 +62,7 @@ func handleNotifyAck(db *sql.DB) http.HandlerFunc { acked, err := acknowledgeIncident(r.Context(), db, incidentID, userID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if !acked { @@ -73,7 +73,7 @@ func handleNotifyAck(db *sql.DB) http.HandlerFunc { // ntfy show a success toast rather than a failure. inc, err := fetchIncident(r.Context(), db, incidentID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, map[string]any{ diff --git a/internal/api/oidc.go b/internal/api/oidc.go index fe4baa7..5221cd7 100644 --- a/internal/api/oidc.go +++ b/internal/api/oidc.go @@ -130,12 +130,12 @@ func handleOIDCLogin(db *sql.DB, prov *oidc.Provider, limiter *loginLimiter, pub state, stateHash, err := randomToken() if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } nonce, _, err := randomToken() if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } verifier := oidc.NewVerifier() @@ -149,7 +149,7 @@ func handleOIDCLogin(db *sql.DB, prov *oidc.Provider, limiter *loginLimiter, pub INSERT INTO oidc_logins (state_hash, nonce, pkce_verifier, next, expires_at) VALUES ($1, $2, $3, $4, $5)`, stateHash, nonce, verifier, next, now.Add(oidcLoginTTL).Unix()); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } diff --git a/internal/api/oidc_teams.go b/internal/api/oidc_teams.go index 7c8fb18..0bdc8ef 100644 --- a/internal/api/oidc_teams.go +++ b/internal/api/oidc_teams.go @@ -31,7 +31,7 @@ func handleGetTeamOIDCGroups(db *sql.DB) http.HandlerFunc { "SELECT COALESCE(oidc_member_group, ''), COALESCE(oidc_owner_group, '') FROM teams WHERE id = $1", teamID).Scan(&g.MemberGroup, &g.OwnerGroup) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, g) @@ -66,7 +66,7 @@ func handleSetTeamOIDCGroups(db *sql.DB) http.HandlerFunc { oidc_owner_group = NULLIF($2, '') WHERE id = $3`, req.MemberGroup, req.OwnerGroup, teamID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } w.WriteHeader(http.StatusNoContent) diff --git a/internal/api/operator_key.go b/internal/api/operator_key.go new file mode 100644 index 0000000..952a0ec --- /dev/null +++ b/internal/api/operator_key.go @@ -0,0 +1,72 @@ +package api + +import ( + "context" + "database/sql" + "fmt" + + "git.ryuvia.com/niklas/terdut-server/internal/models" +) + +const ( + // operatorAccountName is the instance-scoped service account the operator + // key belongs to. + operatorAccountName = "terdut-operator" + + // operatorKeyName names the one key SeedOperatorKey manages on it, so a + // rotation replaces that key and leaves any others alone. + operatorKeyName = "seed" +) + +// SeedOperatorKey makes key the operator account's credential: it creates the +// instance-scoped service account if needed and replaces its "seed" key with +// this one. Idempotent, so every replica can run it at every start, and a +// rotated key simply wins on the next restart. +// +// The key is hashed like any other, so only the caller that generated it ever +// holds the raw value. An empty key does nothing. +func SeedOperatorKey(ctx context.Context, db *sql.DB, key string) error { + if key == "" { + return nil + } + tx, err := db.BeginTx(ctx, nil) + if err != nil { + return fmt.Errorf("seed operator key: %w", err) + } + defer tx.Rollback() //nolint:errcheck + + // Serialise replicas starting together; transaction-scoped, so it needs no + // explicit release. + if _, err := tx.ExecContext(ctx, "SELECT pg_advisory_xact_lock($1)", operatorKeyLockKey); err != nil { + return fmt.Errorf("seed operator key: lock: %w", err) + } + + var accountID int64 + err = tx.QueryRowContext(ctx, + "SELECT id FROM service_accounts WHERE name = $1 AND scope = $2", + operatorAccountName, models.ServiceAccountScopeInstance).Scan(&accountID) + if err == sql.ErrNoRows { + err = tx.QueryRowContext(ctx, + "INSERT INTO service_accounts (name, scope) VALUES ($1, $2) RETURNING id", + operatorAccountName, models.ServiceAccountScopeInstance).Scan(&accountID) + } + if err != nil { + return fmt.Errorf("seed operator key: account: %w", err) + } + + if _, err := tx.ExecContext(ctx, + "DELETE FROM service_account_keys WHERE service_account_id = $1 AND name = $2", + accountID, operatorKeyName); err != nil { + return fmt.Errorf("seed operator key: drop old key: %w", err) + } + if _, err := tx.ExecContext(ctx, + "INSERT INTO service_account_keys (service_account_id, key_hash, name) VALUES ($1, $2, $3)", + accountID, hashToken(key), operatorKeyName); err != nil { + return fmt.Errorf("seed operator key: store key: %w", err) + } + return tx.Commit() +} + +// operatorKeyLockKey is the transaction-scoped advisory lock SeedOperatorKey +// holds; distinct from the other lock keys in this package. +const operatorKeyLockKey int64 = 7265_0010 diff --git a/internal/api/operator_key_test.go b/internal/api/operator_key_test.go new file mode 100644 index 0000000..2481842 --- /dev/null +++ b/internal/api/operator_key_test.go @@ -0,0 +1,134 @@ +package api_test + +import ( + "context" + "encoding/json" + "net/http" + "testing" + + "git.ryuvia.com/niklas/terdut-server/internal/api" +) + +const ( + operatorKeyOne = "tdsa_operator-key-number-one-0123456789" + operatorKeyTwo = "tdsa_operator-key-number-two-0123456789" +) + +func TestSeedOperatorKey_AuthenticatesAsInstanceAccount(t *testing.T) { + s := newTS(t) + if err := api.SeedOperatorKey(context.Background(), s.db, operatorKeyOne); err != nil { + t.Fatal(err) + } + + resp := s.reqAs(t, operatorKeyOne, http.MethodPost, "/api/teams", map[string]string{"name": "seeded"}) + resp.Body.Close() + if resp.StatusCode != http.StatusCreated { + t.Fatalf("seeded key should create a team, got %d", resp.StatusCode) + } +} + +func TestSeedOperatorKey_RotationReplacesAndIsIdempotent(t *testing.T) { + s := newTS(t) + ctx := context.Background() + for _, key := range []string{operatorKeyOne, operatorKeyOne, operatorKeyTwo} { + if err := api.SeedOperatorKey(ctx, s.db, key); err != nil { + t.Fatal(err) + } + } + + old := s.reqAs(t, operatorKeyOne, http.MethodGet, "/api/teams", nil) + old.Body.Close() + if old.StatusCode != http.StatusUnauthorized { + t.Errorf("rotated-out key should be refused, got %d", old.StatusCode) + } + cur := s.reqAs(t, operatorKeyTwo, http.MethodGet, "/api/teams", nil) + cur.Body.Close() + if cur.StatusCode != http.StatusOK { + t.Errorf("current key should work, got %d", cur.StatusCode) + } + + var accounts, keys int + s.db.QueryRow("SELECT COUNT(*) FROM service_accounts WHERE name = 'terdut-operator'").Scan(&accounts) + s.db.QueryRow("SELECT COUNT(*) FROM service_account_keys").Scan(&keys) + if accounts != 1 || keys != 1 { + t.Errorf("expected one account and one key, got %d and %d", accounts, keys) + } +} + +func TestSeedOperatorKey_EmptyKeyDoesNothing(t *testing.T) { + s := newTS(t) + if err := api.SeedOperatorKey(context.Background(), s.db, ""); err != nil { + t.Fatal(err) + } + var n int + s.db.QueryRow("SELECT COUNT(*) FROM service_accounts").Scan(&n) + if n != 0 { + t.Errorf("expected no service account, got %d", n) + } +} + +// The instance account configures a team it did not create, which is what lets +// the operator hold one credential instead of one per team, yet it is not a +// member and so reads none of the team's incidents. +func TestInstanceAccount_ActsAsOwnerOfAnyTeamButIsNoMember(t *testing.T) { + s := newTS(t) + other := newTeam(t, s, "other") + if err := api.SeedOperatorKey(context.Background(), s.db, operatorKeyOne); err != nil { + t.Fatal(err) + } + + rename := s.reqAs(t, operatorKeyOne, http.MethodPut, "/api/teams/"+id64(other.id), map[string]string{"name": "renamed"}) + rename.Body.Close() + if rename.StatusCode >= 300 { + t.Errorf("instance account should rename any team, got %d", rename.StatusCode) + } + + // Not a member: the team's queue is not visible to it. + var queue []map[string]any + decode(t, s.reqAs(t, operatorKeyOne, http.MethodGet, "/api/incidents", nil), &queue) + if len(queue) != 0 { + t.Errorf("instance account should see no incidents, got %v", queue) + } +} + +// external_id lets automation find its own team again after a crash, without +// trusting a display name. +func TestCreateTeam_ExternalIDIsIdempotentAndInstanceOnly(t *testing.T) { + s := newTS(t) + if err := api.SeedOperatorKey(context.Background(), s.db, operatorKeyOne); err != nil { + t.Fatal(err) + } + create := func(name string) (int, map[string]any) { + resp := s.reqAs(t, operatorKeyOne, http.MethodPost, "/api/teams", + map[string]string{"name": name, "external_id": "ns/platform"}) + var out map[string]any + _ = json.NewDecoder(resp.Body).Decode(&out) + resp.Body.Close() + return resp.StatusCode, out + } + + code, first := create("Platform") + if code != http.StatusCreated { + t.Fatalf("first create: %d", code) + } + // Same identity, even under a new display name: the same team comes back. + code, again := create("Platform renamed") + if code != http.StatusOK || again["id"] != first["id"] { + t.Errorf("repeat with the same external_id: want 200 and team %v, got %d %v", first["id"], code, again) + } + + // A different identity cannot take the name. + resp := s.reqAs(t, operatorKeyOne, http.MethodPost, "/api/teams", + map[string]string{"name": "Platform", "external_id": "other/platform"}) + resp.Body.Close() + if resp.StatusCode != http.StatusConflict { + t.Errorf("taken name under another external_id: want 409, got %d", resp.StatusCode) + } + + // A person cannot set one. + resp = s.req(t, http.MethodPost, "/api/teams", map[string]string{"name": "Mine", "external_id": "x/y"}) + resp.Body.Close() + if resp.StatusCode != http.StatusForbidden { + t.Errorf("a user setting external_id: want 403, got %d", resp.StatusCode) + } +} diff --git a/internal/api/router.go b/internal/api/router.go index 1b24f08..14ff537 100644 --- a/internal/api/router.go +++ b/internal/api/router.go @@ -23,12 +23,13 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version strin // One limiter each, both process-wide for the life of the router: login // counts failed passwords, sign-up counts account creation, and mixing the // two would let a burst of sign-ups lock somebody out of logging in. + trustedProxies.Store(int32(cfg.TrustedProxies)) loginLimit := newLoginLimiter(db) signupLimiter := newLoginLimiter(db) oidcLimit := newLoginLimiter(db) r := chi.NewRouter() - r.Use(middleware.Logger) + r.Use(requestLogger) r.Use(middleware.Recoverer) r.Use(securityHeaders(notify.PublicURL)) @@ -48,11 +49,7 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version strin // Alert ingestion. The key in the path says both that the sender may post // and which team the alerts belong to, which is why it needs no session. - // - // This is the only way in. The pre-teams /api/alertmanager/webhook, which - // took no credential at all, was removed in v0.13.0 once the cluster's - // Alertmanager had moved onto a key; a sender still posting there gets the - // JSON 404 every unknown /api path gets. + // This is the only way in. r.Post("/api/integrations/{key}/alertmanager", handleIntegrationWebhook(db, notify)) // Signing up. Both are unauthenticated by necessity: the caller has no @@ -116,9 +113,7 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version strin r.Post("/api/users/{id}/api-keys", handleCreateAPIKey(db)) r.Delete("/api/users/{id}/api-keys/{keyID}", handleDeleteAPIKey(db)) - // Administration: who exists, and who is an administrator. Until #3 - // these were open to any authenticated caller, which meant every user - // could delete every other one. + // Administration: who exists, and who is an administrator. r.Group(func(r chi.Router) { r.Use(AdminOnly) @@ -147,7 +142,6 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version strin r.Get("/api/incidents", handleListIncidents(db)) r.Get("/api/incidents/clusters", handleListClusters(db)) r.Get("/api/incidents/{id}", handleGetIncident(db)) - r.Get("/api/incidents/{id}/alerts", handleIncidentAlerts(db)) r.Get("/api/incidents/{id}/timeline", handleIncidentTimeline(db)) r.Get("/api/incidents/{id}/similar", handleIncidentSimilar(db)) r.Post("/api/incidents/{id}/acknowledge", handleIncidentAcknowledge(db)) diff --git a/internal/api/schedule.go b/internal/api/schedule.go index 089800c..7535218 100644 --- a/internal/api/schedule.go +++ b/internal/api/schedule.go @@ -69,7 +69,7 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc { // be, so the delete and the insert share one transaction. tx, err := db.BeginTx(r.Context(), nil) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer tx.Rollback() @@ -79,7 +79,7 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc { if _, err := tx.ExecContext(r.Context(), "DELETE FROM schedule_entries WHERE team_id = $1 AND date = $2", teamID, d); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } } @@ -91,12 +91,12 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc { errResp("date already assigned: "+d+" (pass replace to take it)")) return } - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } } if err := tx.Commit(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -107,7 +107,7 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc { } all, err := scheduleRange(r.Context(), db, teamID, req.Dates[0], req.Dates[len(req.Dates)-1]) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } created := []models.ScheduleEntry{} @@ -147,7 +147,7 @@ func handleListSchedule(db *sql.DB) http.HandlerFunc { entries, err := scheduleRange(r.Context(), db, teamID, from, to) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, entries) @@ -171,7 +171,7 @@ func handleDeleteSchedule(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "DELETE FROM schedule_entries WHERE id = $1 AND team_id = $2", id, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -197,7 +197,7 @@ func handleCurrentSchedule(db *sql.DB) http.HandlerFunc { WHERE s.date = $1 AND s.team_id = ANY($2) ORDER BY t.name`, today, callerTeamIDs(r.Context())) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -207,14 +207,14 @@ func handleCurrentSchedule(db *sql.DB) http.HandlerFunc { var e models.ScheduleEntry var ts int64 if err := rows.Scan(&e.ID, &e.TeamID, &e.TeamName, &e.UserID, &e.Username, &e.Date, &ts); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } e.CreatedAt = time.Unix(ts, 0).UTC() entries = append(entries, e) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, entries) diff --git a/internal/api/service_accounts.go b/internal/api/service_accounts.go index c03bfcd..b95cb3e 100644 --- a/internal/api/service_accounts.go +++ b/internal/api/service_accounts.go @@ -50,6 +50,9 @@ func callerIsAdmin(ctx context.Context) bool { // writing a response: callers here need to combine it with other ways of // being allowed, not stop at the first no. func callerOwnsTeam(ctx context.Context, teamID int64) bool { + if c, _ := callerFromContext(ctx); c.IsInstanceServiceAccount() { + return true + } role, ok := callerRole(ctx, teamID) return ok && role == models.RoleOwner } @@ -132,7 +135,7 @@ func handleCreateServiceAccount(db *sql.DB) http.HandlerFunc { key, err := mintServiceAccountKey(r.Context(), db, sa.ID, "initial") if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusCreated, map[string]any{"service_account": sa, "key": key}) @@ -233,7 +236,7 @@ func handleCreateServiceAccountKey(db *sql.DB) http.HandlerFunc { return } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if !callerMayManageServiceAccount(r.Context(), sa) { @@ -255,7 +258,7 @@ func handleCreateServiceAccountKey(db *sql.DB) http.HandlerFunc { key, err := mintServiceAccountKey(r.Context(), db, sa.ID, req.Name) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusCreated, key) @@ -274,7 +277,7 @@ func handleDeleteServiceAccountKey(db *sql.DB) http.HandlerFunc { return } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if !callerMayManageServiceAccount(r.Context(), sa) { @@ -290,7 +293,7 @@ func handleDeleteServiceAccountKey(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "DELETE FROM service_account_keys WHERE id = $1 AND service_account_id = $2", keyID, sa.ID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -325,7 +328,7 @@ func handleListServiceAccounts(db *sql.DB) http.HandlerFunc { rows, err := db.QueryContext(r.Context(), query, args...) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -335,14 +338,14 @@ func handleListServiceAccounts(db *sql.DB) http.HandlerFunc { var sa models.ServiceAccount var created int64 if err := rows.Scan(&sa.ID, &sa.Name, &sa.Scope, &sa.TeamID, &sa.CreatedBy, &created); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } sa.CreatedAt = time.Unix(created, 0).UTC() accounts = append(accounts, sa) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, accounts) diff --git a/internal/api/settings.go b/internal/api/settings.go index f736c5f..fb22c44 100644 --- a/internal/api/settings.go +++ b/internal/api/settings.go @@ -203,7 +203,7 @@ func handleSetSettings(db *sql.DB) http.HandlerFunc { tx, err := db.BeginTx(r.Context(), nil) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer tx.Rollback() //nolint:errcheck @@ -215,12 +215,12 @@ func handleSetSettings(db *sql.DB) http.HandlerFunc { ON CONFLICT (key) DO UPDATE SET value = excluded.value, updated_at = excluded.updated_at`, key, value); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } } if err := tx.Commit(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -261,7 +261,7 @@ func handleAdminListTeams(db *sql.DB) http.HandlerFunc { FROM teams t ORDER BY t.name`) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -272,14 +272,14 @@ func handleAdminListTeams(db *sql.DB) http.HandlerFunc { var created int64 if err := rows.Scan(&t.ID, &t.Name, &created, &t.Members, &t.OpenIncidents, &t.OIDCMemberGroup, &t.OIDCOwnerGroup); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } t.CreatedAt = time.Unix(created, 0).UTC() teams = append(teams, t) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, teams) @@ -319,7 +319,7 @@ func handleAdminGetTeam(db *sql.DB) http.HandlerFunc { return } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } t.CreatedAt = time.Unix(created, 0).UTC() @@ -333,7 +333,7 @@ func handleAdminGetTeam(db *sql.DB) http.HandlerFunc { WHERE m.team_id = $1 ORDER BY u.username`, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -343,14 +343,14 @@ func handleAdminGetTeam(db *sql.DB) http.HandlerFunc { var m models.TeamMember var joined int64 if err := rows.Scan(&m.TeamID, &m.UserID, &m.Username, &m.Role, &joined, &m.Source); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } m.JoinedAt = time.Unix(joined, 0).UTC() members = append(members, m) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -396,7 +396,7 @@ func handleRenameTeam(db *sql.DB) http.HandlerFunc { respond(w, http.StatusConflict, errResp("a team with that name already exists")) return } - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -435,7 +435,7 @@ func handleSetUserDisabled(db *sql.DB) http.HandlerFunc { } last, err := isLastAdmin(r.Context(), db, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if last { @@ -453,7 +453,7 @@ func handleSetUserDisabled(db *sql.DB) http.HandlerFunc { "UPDATE users SET disabled_at = NULL WHERE id = $1", id) } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -475,7 +475,7 @@ func handleSetUserDisabled(db *sql.DB) http.HandlerFunc { user, err := fetchUser(r.Context(), db, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, user) diff --git a/internal/api/signup.go b/internal/api/signup.go index 9b32e83..0d2a3f8 100644 --- a/internal/api/signup.go +++ b/internal/api/signup.go @@ -170,13 +170,13 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand hash, err := hashPassword(req.Password) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } tx, err := db.BeginTx(r.Context(), nil) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer tx.Rollback() //nolint:errcheck @@ -194,7 +194,7 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand respond(w, http.StatusConflict, errResp("username or email already exists")) return } - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -207,7 +207,7 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand respond(w, http.StatusConflict, errResp("a team with that name already exists")) return } - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } role = models.RoleOwner @@ -216,7 +216,7 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand if _, err := tx.ExecContext(r.Context(), "INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3)", teamID, userID, role); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -226,7 +226,7 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand res, err := tx.ExecContext(r.Context(), "UPDATE invites SET uses = uses + 1 WHERE id = $1 AND uses < max_uses", inv.id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -236,14 +236,14 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand } if err := tx.Commit(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } // Signed in immediately: the alternative is a form that says "now go // and log in", which is the same credential typed twice. if err := startSession(w, r, db, userID, publicURL); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } user, _ := fetchUser(r.Context(), db, userID) @@ -291,7 +291,7 @@ func handleListInvites(db *sql.DB) http.HandlerFunc { WHERE team_id = $1 ORDER BY id DESC`, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -303,7 +303,7 @@ func handleListInvites(db *sql.DB) http.HandlerFunc { var revoked *int64 if err := rows.Scan(&i.ID, &i.TeamID, &i.Role, &created, &expires, &i.MaxUses, &i.Uses, &revoked); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } i.CreatedAt = time.Unix(created, 0).UTC() @@ -312,7 +312,7 @@ func handleListInvites(db *sql.DB) http.HandlerFunc { out = append(out, i) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, out) @@ -356,7 +356,7 @@ func handleCreateInvite(db *sql.DB, publicURL string) http.HandlerFunc { raw, hash, err := randomToken() if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } // created_by is nullable (ON DELETE SET NULL) for exactly this @@ -382,7 +382,7 @@ func handleCreateInvite(db *sql.DB, publicURL string) http.HandlerFunc { RETURNING id, team_id, role, created_at, expires_at, max_uses, uses`, hash, teamID, req.Role, createdBy, expires.Unix(), req.MaxUses). Scan(&out.ID, &out.TeamID, &out.Role, &created, &expiresAt, &out.MaxUses, &out.Uses); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } out.CreatedAt = time.Unix(created, 0).UTC() @@ -412,7 +412,7 @@ func handleRevokeInvite(db *sql.DB) http.HandlerFunc { "UPDATE invites SET revoked_at = "+nowEpoch+ " WHERE id = $1 AND team_id = $2 AND revoked_at IS NULL", id, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -450,7 +450,7 @@ func handleTestNotification(cfg NotifyConfig, db *sql.DB) http.HandlerFunc { var topic *string if err := db.QueryRowContext(r.Context(), "SELECT ntfy_topic FROM users WHERE id = $1", caller.ID).Scan(&topic); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if topic == nil || *topic == "" { @@ -501,7 +501,7 @@ func handleDismissOnboarding(db *sql.DB) http.HandlerFunc { "UPDATE users SET onboarding_dismissed_at = NULL WHERE id = $1", caller.ID) } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } w.WriteHeader(http.StatusNoContent) diff --git a/internal/api/similar.go b/internal/api/similar.go index b316289..c13f400 100644 --- a/internal/api/similar.go +++ b/internal/api/similar.go @@ -37,7 +37,7 @@ func handleIncidentSimilar(db *sql.DB) http.HandlerFunc { out, err := similarIncidents(r.Context(), db, id, limit) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, out) diff --git a/internal/api/stats.go b/internal/api/stats.go index ee30af9..a9ab4da 100644 --- a/internal/api/stats.go +++ b/internal/api/stats.go @@ -24,7 +24,7 @@ func handleStatsAlerts(db *sql.DB) http.HandlerFunc { FROM alerts WHERE %s`, where), args.all()..., ).Scan(&total, &firing, &resolved) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, map[string]int64{ @@ -55,7 +55,7 @@ func handleStatsTop(db *sql.DB) http.HandlerFunc { ORDER BY cnt DESC LIMIT %s`, where, args.add(limit)), args.all()...) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -68,13 +68,13 @@ func handleStatsTop(db *sql.DB) http.HandlerFunc { for rows.Next() { var e entry if err := rows.Scan(&e.Name, &e.Count); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } result = append(result, e) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, result) @@ -93,7 +93,7 @@ func handleStatsByHour(db *sql.DB) http.HandlerFunc { GROUP BY hr ORDER BY hr ASC`, where), args.all()...) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -103,13 +103,13 @@ func handleStatsByHour(db *sql.DB) http.HandlerFunc { var hr int var cnt int64 if err := rows.Scan(&hr, &cnt); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } counts[hr] = cnt } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -129,8 +129,7 @@ func handleStatsByDay(db *sql.DB) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { where, args := statsFilter(r.URL.Query(), "received_at", callerTeamIDs(r.Context())) - // Postgres EXTRACT(DOW …) → 0=Sunday … 6=Saturday, the same numbering - // SQLite's strftime('%w') returned, so the frontend needs no change. + // Postgres EXTRACT(DOW …) → 0=Sunday … 6=Saturday. rows, err := db.QueryContext(r.Context(), fmt.Sprintf(` SELECT EXTRACT(DOW FROM to_timestamp(received_at) AT TIME ZONE 'UTC')::int AS dow, COUNT(*) AS cnt @@ -139,7 +138,7 @@ func handleStatsByDay(db *sql.DB) http.HandlerFunc { GROUP BY dow ORDER BY dow ASC`, where), args.all()...) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -149,13 +148,13 @@ func handleStatsByDay(db *sql.DB) http.HandlerFunc { var dow int var cnt int64 if err := rows.Scan(&dow, &cnt); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } counts[dow] = cnt } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -198,7 +197,7 @@ func handleStatsIncidents(db *sql.DB) http.HandlerFunc { FROM incidents WHERE %s`, where), args.all()..., ).Scan(&total, &triggered, &acknowledged, &resolved, &mtta, &mttr) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } diff --git a/internal/api/teams.go b/internal/api/teams.go index e2b17c7..e92d7d9 100644 --- a/internal/api/teams.go +++ b/internal/api/teams.go @@ -13,19 +13,12 @@ import ( "github.com/go-chi/chi/v5" ) -// handleListTeams lists the caller's own teams, each with their role in it, -// or — with ?name= — looks up one team by exact name regardless of caller -// identity (TEAM-LOOKUP.md). An administrator listing every team goes -// through the admin endpoint instead: the no-name case here answers "what am -// I part of", which is what the UI's team filter and the combined queue are -// built from. +// handleListTeams lists the caller's own teams, each with their role in it. +// An administrator listing every team goes through the admin endpoint instead: +// this answers "what am I part of", which is what the UI's team filter and the +// combined queue are built from. func handleListTeams(db *sql.DB) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { - if name := strings.TrimSpace(r.URL.Query().Get("name")); name != "" { - handleListTeamsByName(db, w, r, name) - return - } - caller, _ := userFromContext(r.Context()) rows, err := db.QueryContext(r.Context(), ` SELECT t.id, t.name, t.created_at, m.role, m.source @@ -34,7 +27,7 @@ func handleListTeams(db *sql.DB) http.HandlerFunc { WHERE m.user_id = $1 ORDER BY t.name`, caller.ID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -44,45 +37,20 @@ func handleListTeams(db *sql.DB) http.HandlerFunc { var t models.Team var created int64 if err := rows.Scan(&t.ID, &t.Name, &created, &t.Role, &t.Source); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } t.CreatedAt = time.Unix(created, 0).UTC() teams = append(teams, t) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, teams) } } -// handleListTeamsByName answers "is there a team named exactly this", open to -// any authenticated caller including a service account (TEAM-LOOKUP.md) — -// mirrors handleListServiceAccounts' own ?name= lookup: a one-or-zero-length -// array, never an error on no match, and no caller-identity filtering at -// all, since what it discloses (a name is taken, nothing about who's in it -// or any of its data) is the same low sensitivity that lookup already -// accepts for service-account names. -func handleListTeamsByName(db *sql.DB, w http.ResponseWriter, r *http.Request, name string) { - var t models.Team - var created int64 - err := db.QueryRowContext(r.Context(), - "SELECT id, name, created_at FROM teams WHERE name = $1", name, - ).Scan(&t.ID, &t.Name, &created) - if errors.Is(err, sql.ErrNoRows) { - respond(w, http.StatusOK, []models.Team{}) - return - } - if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) - return - } - t.CreatedAt = time.Unix(created, 0).UTC() - respond(w, http.StatusOK, []models.Team{t}) -} - // handleUserTeams lists one user's teams, for the admin page's per-user view: // "what is this person in", which /api/teams cannot answer because it is always // about the caller. @@ -106,7 +74,7 @@ func handleUserTeams(db *sql.DB) http.HandlerFunc { var exists bool if err := db.QueryRowContext(r.Context(), "SELECT EXISTS (SELECT 1 FROM users WHERE id = $1)", id).Scan(&exists); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if !exists { @@ -121,7 +89,7 @@ func handleUserTeams(db *sql.DB) http.HandlerFunc { WHERE m.user_id = $1 ORDER BY t.name`, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -131,14 +99,14 @@ func handleUserTeams(db *sql.DB) http.HandlerFunc { var t models.Team var created int64 if err := rows.Scan(&t.ID, &t.Name, &created, &t.Role, &t.Source); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } t.CreatedAt = time.Unix(created, 0).UTC() teams = append(teams, t) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, teams) @@ -161,6 +129,12 @@ func handleCreateTeam(db *sql.DB) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { var req struct { Name string `json:"name"` + // ExternalID makes the call idempotent for automation: a team + // already carrying it is returned as-is (200) instead of created, so + // a client that crashed between the POST and recording the id finds + // its own team again. Instance-scoped service accounts only; a name + // that belongs to a different team is still a 409. + ExternalID string `json:"external_id"` } if err := decodeJSON(r, &req); err != nil { respond(w, http.StatusBadRequest, errResp("invalid request body")) @@ -173,6 +147,10 @@ func handleCreateTeam(db *sql.DB) http.HandlerFunc { } caller, isUser := userFromContext(r.Context()) + if req.ExternalID != "" && !isInstanceServiceAccount(r.Context()) { + respond(w, http.StatusForbidden, errResp("external_id is for instance-scoped service accounts")) + return + } if !isUser && !isInstanceServiceAccount(r.Context()) { // A team-scoped service account authenticates as owner of exactly // one team already (see serveAsServiceAccount); letting it create @@ -183,37 +161,57 @@ func handleCreateTeam(db *sql.DB) http.HandlerFunc { tx, err := db.BeginTx(r.Context(), nil) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer tx.Rollback() //nolint:errcheck var team models.Team var created int64 + if req.ExternalID != "" { + err := tx.QueryRowContext(r.Context(), + "SELECT id, name, created_at FROM teams WHERE external_id = $1", req.ExternalID). + Scan(&team.ID, &team.Name, &created) + if err == nil { + team.ExternalID = &req.ExternalID + team.CreatedAt = time.Unix(created, 0).UTC() + respond(w, http.StatusOK, team) + return + } + if !errors.Is(err, sql.ErrNoRows) { + serverError(w, r, err) + return + } + } + var externalID *string + if req.ExternalID != "" { + externalID = &req.ExternalID + } if err := tx.QueryRowContext(r.Context(), - "INSERT INTO teams (name) VALUES ($1) RETURNING id, name, created_at", - req.Name).Scan(&team.ID, &team.Name, &created); err != nil { + "INSERT INTO teams (name, external_id) VALUES ($1, $2) RETURNING id, name, created_at", + req.Name, externalID).Scan(&team.ID, &team.Name, &created); err != nil { if isUniqueViolation(err) { respond(w, http.StatusConflict, errResp("a team with that name already exists")) return } - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if isUser { if _, err := tx.ExecContext(r.Context(), "INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3)", team.ID, caller.ID, models.RoleOwner); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } } if err := tx.Commit(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } team.CreatedAt = time.Unix(created, 0).UTC() + team.ExternalID = externalID if isUser { team.Role = models.RoleOwner } @@ -241,7 +239,7 @@ func handleDeleteTeam(db *sql.DB) http.HandlerFunc { if err := db.QueryRowContext(r.Context(), "SELECT COUNT(*) FROM incidents WHERE team_id = $1 AND resolved_at IS NULL", teamID). Scan(&open); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if open > 0 { @@ -251,7 +249,7 @@ func handleDeleteTeam(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "DELETE FROM teams WHERE id = $1", teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -322,7 +320,7 @@ func handleListTeamMembers(db *sql.DB) http.HandlerFunc { WHERE m.team_id = $1 ORDER BY u.username`, teamID, todayUTC()) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -334,7 +332,7 @@ func handleListTeamMembers(db *sql.DB) http.HandlerFunc { var hasTopic, disabled bool if err := rows.Scan(&m.TeamID, &m.UserID, &m.Username, &m.Role, &joined, &m.Source, &hasTopic, &disabled, &lastActive, &m.OnCall, &m.NextShift); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } m.JoinedAt = time.Unix(joined, 0).UTC() @@ -360,7 +358,7 @@ func handleListTeamMembers(db *sql.DB) http.HandlerFunc { members = append(members, m) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, members) @@ -396,7 +394,7 @@ func handleAddTeamMember(db *sql.DB) http.HandlerFunc { } if managed, err := isSSOManagedMember(r.Context(), db, teamID, req.UserID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } else if managed { respond(w, http.StatusConflict, errResp(ssoManagedMsg)) @@ -408,7 +406,7 @@ func handleAddTeamMember(db *sql.DB) http.HandlerFunc { if req.Role == models.RoleMember { last, err := isLastTeamOwner(r.Context(), db, teamID, req.UserID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if last { @@ -453,7 +451,7 @@ func handleRemoveTeamMember(db *sql.DB) http.HandlerFunc { } if managed, err := isSSOManagedMember(r.Context(), db, teamID, userID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } else if managed { respond(w, http.StatusConflict, errResp(ssoManagedMsg)) @@ -462,7 +460,7 @@ func handleRemoveTeamMember(db *sql.DB) http.HandlerFunc { last, err := isLastTeamOwner(r.Context(), db, teamID, userID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if last { @@ -473,7 +471,7 @@ func handleRemoveTeamMember(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "DELETE FROM team_members WHERE team_id = $1 AND user_id = $2", teamID, userID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -572,7 +570,7 @@ func handleListIntegrations(db *sql.DB) http.HandlerFunc { WHERE i.team_id = $1 ORDER BY i.id`, teamID, now.Add(-sourceQuietAfter).Unix()) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -584,7 +582,7 @@ func handleListIntegrations(db *sql.DB) http.HandlerFunc { var lastUsed, lastAlert *int64 if err := rows.Scan(&i.ID, &i.TeamID, &i.Kind, &i.Name, &created, &lastUsed, &lastAlert, &i.Alerts24h); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } i.CreatedAt = time.Unix(created, 0).UTC() @@ -601,7 +599,7 @@ func handleListIntegrations(db *sql.DB) http.HandlerFunc { integrations = append(integrations, i) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, integrations) @@ -644,7 +642,7 @@ func handleCreateIntegration(db *sql.DB, publicURL string) http.HandlerFunc { raw, hash, err := randomToken() if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -656,7 +654,11 @@ func handleCreateIntegration(db *sql.DB, publicURL string) http.HandlerFunc { RETURNING id, team_id, kind, name, created_at`, teamID, req.Kind, req.Name, hash). Scan(&i.ID, &i.TeamID, &i.Kind, &i.Name, &created); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + if isUniqueViolation(err) { + respond(w, http.StatusConflict, errResp("an integration with that name already exists in this team")) + return + } + serverError(w, r, err) return } i.CreatedAt = time.Unix(created, 0).UTC() @@ -703,7 +705,11 @@ func handleRenameIntegration(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "UPDATE integrations SET name = $1 WHERE id = $2 AND team_id = $3", req.Name, id, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + if isUniqueViolation(err) { + respond(w, http.StatusConflict, errResp("an integration with that name already exists in this team")) + return + } + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -732,7 +738,7 @@ func handleDeleteIntegration(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "DELETE FROM integrations WHERE id = $1 AND team_id = $2", id, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -790,16 +796,6 @@ func teamParam(w http.ResponseWriter, r *http.Request) (int64, bool) { return id, true } -// defaultTeamID is the oldest team, which on an upgraded install is the -// "Default" team every pre-teams row was moved into and on a fresh one is the -// team migration 003 creates. Bootstrap puts the first user in it, so somebody -// signing in to a new server lands somewhere rather than in no team at all. -func defaultTeamID(ctx context.Context, db *sql.DB) (int64, error) { - var id int64 - err := db.QueryRowContext(ctx, "SELECT id FROM teams ORDER BY id LIMIT 1").Scan(&id) - return id, err -} - // --------------------------------------------------------------------------- // A team's dead man's switches // --------------------------------------------------------------------------- @@ -833,12 +829,12 @@ func handleListTeamDeadman(db *sql.DB) http.HandlerFunc { set, err := deadmanSetForTeam(r.Context(), db, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } out, err := deadmanStatuses(r.Context(), db, teamID, set, time.Now()) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, out) @@ -901,7 +897,11 @@ func handleCreateTeamDeadman(db *sql.DB) http.HandlerFunc { INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity) VALUES ($1, $2, $3, $4, $5) RETURNING id`, teamID, req.Name, m.config(), req.TimeoutSeconds, req.Severity).Scan(&id); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + if isUniqueViolation(err) { + respond(w, http.StatusConflict, errResp("a switch with that name already exists in this team")) + return + } + serverError(w, r, err) return } @@ -975,7 +975,11 @@ func handleUpdateTeamDeadman(db *sql.DB) http.HandlerFunc { WHERE id = $5 AND team_id = $6`, req.Name, m.config(), req.TimeoutSeconds, req.Severity, switchID, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + if isUniqueViolation(err) { + respond(w, http.StatusConflict, errResp("a switch with that name already exists in this team")) + return + } + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -990,12 +994,12 @@ func handleUpdateTeamDeadman(db *sql.DB) http.HandlerFunc { // handleListTeamDeadman would give it. set, err := deadmanSetForTeam(r.Context(), db, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } statuses, err := deadmanStatuses(r.Context(), db, teamID, set, time.Now()) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } for _, s := range statuses { @@ -1004,7 +1008,7 @@ func handleUpdateTeamDeadman(db *sql.DB) http.HandlerFunc { return } } - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) } } @@ -1029,7 +1033,7 @@ func handleDeleteTeamDeadman(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "DELETE FROM deadman_switches WHERE id = $1 AND team_id = $2", switchID, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { diff --git a/internal/api/teams_test.go b/internal/api/teams_test.go index 8cebe1a..85879f9 100644 --- a/internal/api/teams_test.go +++ b/internal/api/teams_test.go @@ -6,8 +6,6 @@ import ( "io" "net/http" "testing" - - "git.ryuvia.com/niklas/terdut-server/internal/models" ) // The whole point of #4: two teams sharing one server must not see each other's @@ -146,7 +144,6 @@ func TestTeams_IncidentsAreScopedToTheReceivingTeam(t *testing.T) { otherID := int64(blueIncidents[0]["id"].(float64)) for _, path := range []string{ "/api/incidents/" + id64(otherID), - "/api/incidents/" + id64(otherID) + "/alerts", "/api/incidents/" + id64(otherID) + "/timeline", } { resp := red.call(http.MethodGet, path, nil) @@ -461,70 +458,58 @@ func TestTeams_OutsiderSeesNothing(t *testing.T) { // GET /api/teams?name= (TEAM-LOOKUP.md) // --------------------------------------------------------------------------- -func TestListTeamsByName_FindsExactMatch(t *testing.T) { - s := newTS(t) - instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0) - teamID := createTeamAs(t, s, instanceKey, "platform") - - teams := list(t, s.reqAs(t, instanceKey, http.MethodGet, "/api/teams?name=platform", nil)) - if len(teams) != 1 { - t.Fatalf("expected exactly one match for ?name=platform, got %d: %v", len(teams), teams) - } - if int64(teams[0]["id"].(float64)) != teamID { - t.Errorf("id = %v, want %d", teams[0]["id"], teamID) - } - // No membership, so no role to report (models.Team's own doc comment: - // "empty when nobody in particular is asking"). - if _, has := teams[0]["role"]; has { - t.Errorf("expected no role on a name-lookup match, got %v", teams[0]["role"]) - } -} - -func TestListTeamsByName_NoMatchIsAnEmptyArrayNotAnError(t *testing.T) { - s := newTS(t) - instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0) - - resp := s.reqAs(t, instanceKey, http.MethodGet, "/api/teams?name=does-not-exist", nil) - if resp.StatusCode != http.StatusOK { - t.Fatalf("expected 200 on no match, got %d", resp.StatusCode) - } - teams := list(t, resp) - if len(teams) != 0 { - t.Errorf("expected an empty array, got %v", teams) - } -} - -// The actual motivating scenario (TEAM-LOOKUP.md): a service account that -// already created a team, interrupted before it could remember the id, -// recovers it via ?name= on the same name its own POST 409s on. -func TestListTeamsByName_RecoversAfterCreateConflict(t *testing.T) { - s := newTS(t) - instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0) - original := createTeamAs(t, s, instanceKey, "recovered") - - conflict := s.reqAs(t, instanceKey, http.MethodPost, "/api/teams", map[string]string{"name": "recovered"}) - if conflict.StatusCode != http.StatusConflict { - t.Fatalf("expected 409 recreating the same name, got %d", conflict.StatusCode) - } - conflict.Body.Close() - - teams := list(t, s.reqAs(t, instanceKey, http.MethodGet, "/api/teams?name=recovered", nil)) - if len(teams) != 1 || int64(teams[0]["id"].(float64)) != original { - t.Fatalf("expected to recover the original team %d via ?name=, got %v", original, teams) - } -} - -// Not gated by isInstanceServiceAccount or AdminOnly (TEAM-LOOKUP.md): any -// authenticated caller may ask whether a name is taken, the same low -// sensitivity GET /api/service-accounts?name= already accepts. -func TestListTeamsByName_OpenToAnyAuthenticatedCaller(t *testing.T) { +// Everybody signed in can list users to name them, but only an admin (or the +// row's owner) sees an email or an ntfy topic, which is a publish secret. +func TestListUsers_RedactsEmailAndTopicForOthers(t *testing.T) { s := newTS(t) red := newTeam(t, s, "red") - _ = createTeamAs(t, s, s.key, "blue-target") + _ = newTeam(t, s, "blue") + s.exec(t, "UPDATE users SET ntfy_topic = 'secret-topic'") - // red's own member, not a member of "blue-target", still gets a match. - teams := list(t, red.call(http.MethodGet, "/api/teams?name=blue-target", nil)) - if len(teams) != 1 || teams[0]["name"] != "blue-target" { - t.Errorf("expected a non-member caller to still find the team by name, got %v", teams) + var asAdmin []map[string]any + decode(t, s.req(t, http.MethodGet, "/api/users", nil), &asAdmin) + for _, u := range asAdmin { + if u["email"] == "" || u["ntfy_topic"] != "secret-topic" { + t.Errorf("admin should see everything, got %v", u) + } + } + + asMember := list(t, red.call(http.MethodGet, "/api/users", nil)) + if len(asMember) < 3 { + t.Fatalf("expected the whole user list, got %v", asMember) + } + for _, u := range asMember { + own := u["username"] == "red-user" + if own != (u["email"] != "") || own != (u["ntfy_topic"] != nil) { + t.Errorf("only red-user's own row should keep email and topic, got %v", u) + } + } +} + +// Names identify integrations and switches within a team. +func TestTeamNames_AreUniquePerTeam(t *testing.T) { + s := newTS(t) // creates one integration named "test" in the default team + + dup := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/integrations", map[string]string{"name": "test"}) + dup.Body.Close() + if dup.StatusCode != http.StatusConflict { + t.Errorf("duplicate integration name: expected 409, got %d", dup.StatusCode) + } + + body := map[string]any{"matcher": "alertname=Watchdog", "timeout_seconds": 60, "severity": "critical"} + first := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches", body) + first.Body.Close() + second := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches", body) + second.Body.Close() + if first.StatusCode != http.StatusCreated || second.StatusCode != http.StatusConflict { + t.Errorf("duplicate switch name: expected 201 then 409, got %d then %d", first.StatusCode, second.StatusCode) + } + + // The same name in another team is fine. + other := newTeam(t, s, "elsewhere") + ok := s.req(t, http.MethodPost, "/api/teams/"+id64(other.id)+"/integrations", map[string]string{"name": "test"}) + ok.Body.Close() + if ok.StatusCode != http.StatusCreated { + t.Errorf("same name in another team: expected 201, got %d", ok.StatusCode) } } diff --git a/internal/api/testdb_test.go b/internal/api/testdb_test.go index 33d6294..25852e3 100644 --- a/internal/api/testdb_test.go +++ b/internal/api/testdb_test.go @@ -13,9 +13,8 @@ import ( "git.ryuvia.com/niklas/terdut-server/internal/db" ) -// Tests run against a real Postgres, because the server does. SQLite's -// ":memory:" gave every test a private database for free; Postgres has no -// equivalent, so isolation is bought with a schema per test. +// Tests run against a real Postgres, because the server does. Isolation is +// bought with a schema per test. // // A schema rather than a database: CREATE DATABASE copies a template on disk and // costs a hundred milliseconds or so each time, while CREATE SCHEMA plus the one diff --git a/internal/api/users.go b/internal/api/users.go index aa457a2..ba3f7f7 100644 --- a/internal/api/users.go +++ b/internal/api/users.go @@ -15,6 +15,10 @@ import ( "github.com/go-chi/chi/v5" ) +// bootstrapLockKey is the transaction-scoped advisory lock handleBootstrap +// holds; distinct from the migration and notifier keys. +const bootstrapLockKey = 0x7465726475744254 + func handleBootstrap(db *sql.DB) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { var req struct { @@ -40,15 +44,30 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc { } h, err := hashPassword(req.Password) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } passwordHash = &h } + // Check-then-insert has to be one atomic step: two concurrent calls on + // an empty install would otherwise both see zero users and both create + // an admin. The transaction-scoped lock serialises them, and the loser + // sees the winner's row. + tx, err := db.BeginTx(r.Context(), nil) + if err != nil { + serverError(w, r, err) + return + } + defer tx.Rollback() //nolint:errcheck + if _, err := tx.ExecContext(r.Context(), "SELECT pg_advisory_xact_lock($1)", bootstrapLockKey); err != nil { + serverError(w, r, err) + return + } + var count int - if err := db.QueryRowContext(r.Context(), "SELECT COUNT(*) FROM users").Scan(&count); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + if err := tx.QueryRowContext(r.Context(), "SELECT COUNT(*) FROM users").Scan(&count); err != nil { + serverError(w, r, err) return } if count > 0 { @@ -57,34 +76,29 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc { } var userID int64 - if err := db.QueryRowContext(r.Context(), + if err := tx.QueryRowContext(r.Context(), "INSERT INTO users (username, email, password_hash, is_admin) VALUES ($1, $2, $3, true) RETURNING id", req.Username, req.Email, passwordHash).Scan(&userID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } raw, hash, err := randomToken() if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } var keyID int64 - if err := db.QueryRowContext(r.Context(), + if err := tx.QueryRowContext(r.Context(), "INSERT INTO api_keys (user_id, key_hash, name) VALUES ($1, $2, $3) RETURNING id", userID, hash, "bootstrap").Scan(&keyID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } - // The default team exists from migration 003, on a fresh install too. - // Without a membership the first user signs in to a working server with - // no queue, no schedule and nowhere for an integration to hang off. - if teamID, err := defaultTeamID(r.Context(), db); err == nil { - db.ExecContext(r.Context(), //nolint:errcheck - "INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3) "+ - "ON CONFLICT (team_id, user_id) DO NOTHING", - teamID, userID, models.RoleOwner) + if err := tx.Commit(); err != nil { + serverError(w, r, err) + return } user, _ := fetchUser(r.Context(), db, userID) @@ -93,12 +107,19 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc { } } +// handleListUsers is readable by anyone signed in, because the assignment +// control and the schedule need to name people. What it returns about other +// people is therefore only what naming them takes: email and ntfy_topic are +// blanked unless the caller is an admin or the row is their own. The topic in +// particular is a publish secret. func handleListUsers(db *sql.DB) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { + caller, _ := userFromContext(r.Context()) + seeAll := caller.IsAdmin rows, err := db.QueryContext(r.Context(), "SELECT id, username, email, created_at, ntfy_topic, is_admin, admin_source, disabled_at FROM users ORDER BY id") if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -109,15 +130,19 @@ func handleListUsers(db *sql.DB) http.HandlerFunc { var ts int64 var disabled *int64 if err := rows.Scan(&u.ID, &u.Username, &u.Email, &ts, &u.NtfyTopic, &u.IsAdmin, &u.AdminSource, &disabled); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } u.CreatedAt = time.Unix(ts, 0).UTC() u.DisabledAt = unixPtr(disabled) + if !seeAll && u.ID != caller.ID { + u.Email = "" + u.NtfyTopic = nil + } users = append(users, u) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, users) @@ -147,7 +172,7 @@ func handleCreateUser(db *sql.DB) http.HandlerFunc { respond(w, http.StatusConflict, errResp("username or email already exists")) return } - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } user, _ := fetchUser(r.Context(), db, id) @@ -185,7 +210,7 @@ func handleSetNotifyTarget(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "UPDATE users SET ntfy_topic = $1 WHERE id = $2", topic, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -195,7 +220,7 @@ func handleSetNotifyTarget(db *sql.DB) http.HandlerFunc { user, err := fetchUser(r.Context(), db, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, user) @@ -217,7 +242,7 @@ func handleDeleteUser(db *sql.DB) http.HandlerFunc { return } if last, err := isLastAdmin(r.Context(), db, id); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } else if last { respond(w, http.StatusConflict, errResp("cannot delete the last administrator")) @@ -226,7 +251,7 @@ func handleDeleteUser(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "DELETE FROM users WHERE id = $1", id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } n, _ := res.RowsAffected() @@ -283,7 +308,7 @@ func handleCreateAPIKey(db *sql.DB) http.HandlerFunc { raw, hash, err := randomToken() if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } var expiresAt *int64 @@ -298,7 +323,7 @@ func handleCreateAPIKey(db *sql.DB) http.HandlerFunc { if err := db.QueryRowContext(r.Context(), "INSERT INTO api_keys (user_id, key_hash, name, expires_at) VALUES ($1, $2, $3, $4) RETURNING id", userID, hash, req.Name, expiresAt).Scan(&keyID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } key := models.APIKey{ @@ -329,7 +354,7 @@ func handleListAPIKeys(db *sql.DB) http.HandlerFunc { `SELECT id, name, created_at, last_used_at, expires_at FROM api_keys WHERE user_id = $1 ORDER BY created_at DESC`, userID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -340,7 +365,7 @@ func handleListAPIKeys(db *sql.DB) http.HandlerFunc { var created int64 var lastUsed, expires *int64 if err := rows.Scan(&k.ID, &k.Name, &created, &lastUsed, &expires); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } k.UserID = userID @@ -350,7 +375,7 @@ func handleListAPIKeys(db *sql.DB) http.HandlerFunc { keys = append(keys, k) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, keys) @@ -376,7 +401,7 @@ func handleDeleteAPIKey(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "DELETE FROM api_keys WHERE id = $1 AND user_id = $2", keyID, userID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } n, _ := res.RowsAffected() @@ -442,7 +467,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc { if err := db.QueryRowContext(r.Context(), "SELECT EXISTS (SELECT 1 FROM users WHERE id = $1 AND is_admin AND admin_source = 'oidc')", id).Scan(&managed); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if managed { @@ -456,7 +481,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc { return } if last, err := isLastAdmin(r.Context(), db, id); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } else if last { respond(w, http.StatusConflict, errResp("cannot revoke the last administrator")) @@ -467,7 +492,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "UPDATE users SET is_admin = $1 WHERE id = $2", *req.IsAdmin, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -477,7 +502,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc { user, err := fetchUser(r.Context(), db, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, user) diff --git a/internal/config/config.go b/internal/config/config.go index 565638c..581bbd4 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -5,17 +5,21 @@ import ( "fmt" "net/url" "os" + "strconv" "strings" "time" ) +// MinOperatorKeyLength is the shortest TERDUT_OPERATOR_KEY accepted: it is a +// bearer credential with instance reach, so a short one is refused outright. +const MinOperatorKeyLength = 32 + type Config struct { Addr string // DSN is the Postgres connection string, e.g. // postgres://terdut:secret@host:5432/terdut?sslmode=require. Required: - // unlike the SQLite path it replaced there is no sensible default, and a - // server that silently came up against the wrong database would be worse + // there is no sensible default, and a server that silently came up against the wrong database would be worse // than one that refuses to start. DSN string @@ -26,25 +30,6 @@ type Config struct { // repeat_interval (default 4h), which is what refreshes the alert. StaleAfter time.Duration - // DeadmanMatchers selects the alerts that are heartbeats rather than - // problems: receiving one opens no incident, and the absence of one does. - // - // ";" separates matchers, "," the label conditions within one, "=" is exact - // equality — `alertname=Watchdog,cluster=prod; alertname=Heartbeat`. Every - // matcher must name an alertname. See api.ParseDeadmanConfig. - DeadmanMatchers string - - // DeadmanTimeout is how long a heartbeat may go unheard before its switch is - // declared dead. It must be *shorter* than the Alertmanager repeat_interval - // of the route carrying the heartbeat — the opposite of StaleAfter, and the - // reason a dead man's switch usually wants a route of its own. Zero disables - // dead man's switch handling entirely. - DeadmanTimeout time.Duration - - // DeadmanSeverity is the severity a dead man's switch incident opens at. - // These incidents have no member alerts to derive one from. - DeadmanSeverity string - // NtfyURL is the ntfy server push notifications are published to. Empty // disables notifications entirely. NtfyURL string @@ -70,6 +55,20 @@ type Config struct { // Config, which is what a test or a new caller builds, keeps passwords working. DisablePasswordLogin bool + // OperatorKey, when set, is the credential of the instance-scoped service + // account "terdut-operator", created or re-keyed at every start. It is how + // terdut-operator gets in without a bootstrap handshake: the operator + // generates the key, hands it to the server here, and uses it as its bearer + // token. Empty means no such account is managed. + OperatorKey string + + // TrustedProxies is how many reverse proxies sit in front of the server and + // append to X-Forwarded-For. The per-address rate limits take the client + // address that many entries from the right, because everything further left + // is whatever the client chose to send. 0 ignores the header and uses the + // connection's own address. + TrustedProxies int + // OIDC configures single sign-on. The zero value, with no Issuer, is off. OIDC OIDC @@ -133,24 +132,12 @@ func Load() Config { if addr == "" { addr = ":8080" } - deadmanMatchers := os.Getenv("TERDUT_DEADMAN_MATCHERS") - if deadmanMatchers == "" { - deadmanMatchers = "alertname=Watchdog" - } - deadmanSeverity := os.Getenv("TERDUT_DEADMAN_SEVERITY") - if deadmanSeverity == "" { - deadmanSeverity = "critical" - } return Config{ Addr: addr, DSN: os.Getenv("TERDUT_DB_DSN"), ArchiveAfter: duration("TERDUT_ARCHIVE_AFTER", 7*24*time.Hour), StaleAfter: duration("TERDUT_STALE_AFTER", 6*time.Hour), - DeadmanMatchers: deadmanMatchers, - DeadmanTimeout: duration("TERDUT_DEADMAN_TIMEOUT", 15*time.Minute), - DeadmanSeverity: deadmanSeverity, - NtfyURL: os.Getenv("TERDUT_NTFY_URL"), NtfyToken: os.Getenv("TERDUT_NTFY_TOKEN"), NtfyFallbackTopic: os.Getenv("TERDUT_NTFY_FALLBACK_TOPIC"), @@ -158,7 +145,10 @@ func Load() Config { NotifyRepeat: duration("TERDUT_NOTIFY_REPEAT", 15*time.Minute), DisablePasswordLogin: !boolean("TERDUT_PASSWORD_LOGIN", true), - OIDC: loadOIDC(), + + OperatorKey: strings.TrimSpace(os.Getenv("TERDUT_OPERATOR_KEY")), + TrustedProxies: integer("TERDUT_TRUSTED_PROXIES", 1), + OIDC: loadOIDC(), OperatorMode: boolean("TERDUT_OPERATOR_MODE", false), } @@ -187,6 +177,9 @@ func loadOIDC() OIDC { // provider would come up and then fail every login, which is harder to notice // than not starting. func (c Config) Validate() error { + if c.OperatorKey != "" && len(c.OperatorKey) < MinOperatorKeyLength { + return fmt.Errorf("TERDUT_OPERATOR_KEY must be at least %d characters", MinOperatorKeyLength) + } o := c.OIDC if !o.Enabled() { if c.DisablePasswordLogin { @@ -226,6 +219,16 @@ func str(env, def string) string { return def } +// integer reads a non-negative int env var; anything else takes the default. +func integer(env string, def int) int { + if s := os.Getenv(env); s != "" { + if n, err := strconv.Atoi(strings.TrimSpace(s)); err == nil && n >= 0 { + return n + } + } + return def +} + // list reads a comma- or space-separated env var. func list(env, def string) []string { s := os.Getenv(env) diff --git a/internal/db/db.go b/internal/db/db.go index c43bce8..5b14708 100644 --- a/internal/db/db.go +++ b/internal/db/db.go @@ -35,8 +35,7 @@ const ( // // The pool is modest on purpose: this server's concurrency comes from a handful // of HTTP handlers plus two background loops, and a cloud-native-pg instance -// sized for it has a low max_connections. It is still a pool, unlike the single -// connection SQLite forced, so the notifier no longer blocks a webhook. +// sized for it has a low max_connections. func Open(dsn string) (*sql.DB, error) { if dsn == "" { return nil, fmt.Errorf("empty DSN: set TERDUT_DB_DSN") @@ -76,9 +75,8 @@ const migrationLockKey int64 = 7265_0003 // Migrate applies every embedded migration that has not been applied yet, in // filename order, recording each in schema_migrations. // -// Each file runs inside a transaction, which SQLite's version did not do: a -// migration that failed half way used to leave the schema in whatever state it -// had reached. Postgres has transactional DDL, so the rollback is real. +// Each file runs inside a transaction, so a migration that fails half way +// leaves the schema as it was: Postgres has transactional DDL. func Migrate(db *sql.DB) error { ctx := context.Background() conn, err := db.Conn(ctx) diff --git a/internal/db/migrations/001_schema.sql b/internal/db/migrations/001_schema.sql new file mode 100644 index 0000000..a7cb016 --- /dev/null +++ b/internal/db/migrations/001_schema.sql @@ -0,0 +1,587 @@ +-- Terdut Server schema. One baseline: the project has not shipped, so the +-- migration history that led here (SQLite import, a Default team, per-team +-- deadman configs later replaced by switches) is not carried. Later changes are +-- new numbered files after this one. +-- +-- Timestamps are Unix epoch seconds in BIGINT columns throughout. + +CREATE TABLE alerts ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + fingerprint text NOT NULL, + name text NOT NULL, + status text NOT NULL, + labels jsonb DEFAULT '{}'::jsonb NOT NULL, + annotations jsonb DEFAULT '{}'::jsonb NOT NULL, + starts_at bigint NOT NULL, + ends_at bigint, + generator_url text DEFAULT ''::text NOT NULL, + received_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + archived_at bigint, + resolution_source text, + team_id bigint NOT NULL, + integration_id bigint, + CONSTRAINT alerts_status_check CHECK ((status = ANY (ARRAY['firing'::text, 'resolved'::text]))) +); + +CREATE TABLE api_keys ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + user_id bigint NOT NULL, + key_hash text NOT NULL, + name text NOT NULL, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + last_used_at bigint, + expires_at bigint +); + +CREATE TABLE deadman_switches ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + team_id bigint NOT NULL, + name text NOT NULL, + matcher text NOT NULL, + timeout_seconds bigint NOT NULL, + severity text DEFAULT 'critical'::text NOT NULL, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + CONSTRAINT deadman_switches_timeout_seconds_check CHECK ((timeout_seconds > 0)) +); + +CREATE TABLE device_logins ( + device_hash text NOT NULL, + user_code text NOT NULL, + status text DEFAULT 'pending'::text NOT NULL, + user_id bigint, + expires_at bigint NOT NULL, + last_polled_at bigint DEFAULT 0 NOT NULL, + CONSTRAINT device_logins_status_check CHECK ((status = ANY (ARRAY['pending'::text, 'approved'::text, 'denied'::text]))) +); + +CREATE TABLE escalation_levels ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + team_id bigint NOT NULL, + "position" bigint NOT NULL, + timeout_seconds bigint NOT NULL, + CONSTRAINT escalation_levels_timeout_seconds_check CHECK ((timeout_seconds > 0)) +); + +CREATE TABLE escalation_policies ( + team_id bigint NOT NULL, + repeat_count bigint DEFAULT 0 NOT NULL, + fallback_topic text DEFAULT ''::text NOT NULL, + updated_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + CONSTRAINT escalation_policies_repeat_count_check CHECK (((repeat_count >= 0) AND (repeat_count <= 10))) +); + +CREATE TABLE escalation_targets ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + level_id bigint NOT NULL, + kind text NOT NULL, + user_id bigint, + CONSTRAINT escalation_targets_check CHECK ((((kind = 'user'::text) AND (user_id IS NOT NULL)) OR ((kind = 'oncall'::text) AND (user_id IS NULL)))), + CONSTRAINT escalation_targets_kind_check CHECK ((kind = ANY (ARRAY['user'::text, 'oncall'::text]))) +); + +CREATE TABLE incident_ack_tokens ( + token_hash text NOT NULL, + incident_id bigint NOT NULL, + user_id bigint NOT NULL, + created_at bigint NOT NULL, + expires_at bigint NOT NULL +); + +CREATE TABLE incident_alerts ( + incident_id bigint NOT NULL, + alert_id bigint NOT NULL, + added_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL +); + +CREATE TABLE incident_events ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + incident_id bigint NOT NULL, + type text NOT NULL, + user_id bigint, + alert_id bigint, + detail text, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + service_account_id bigint, + actor_user_id bigint, + actor_service_account_id bigint, + CONSTRAINT incident_events_actor_xor_chk CHECK (((user_id IS NULL) OR (service_account_id IS NULL))), + CONSTRAINT incident_events_assign_actor_xor_chk CHECK (((actor_user_id IS NULL) OR (actor_service_account_id IS NULL))) +); + +CREATE TABLE incidents ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + group_key text NOT NULL, + title text NOT NULL, + group_labels jsonb DEFAULT '{}'::jsonb NOT NULL, + status text NOT NULL, + severity text, + triggered_at bigint NOT NULL, + acknowledged_by bigint, + acknowledged_at bigint, + assigned_to bigint, + snoozed_until bigint, + resolved_at bigint, + resolution_source text, + archived_at bigint, + team_id bigint NOT NULL, + escalation_level bigint DEFAULT 0 NOT NULL, + escalation_level_at bigint, + escalation_round bigint DEFAULT 0 NOT NULL, + signature text DEFAULT ''::text NOT NULL, + acknowledged_by_service_account_id bigint, + CONSTRAINT incidents_ack_actor_xor_chk CHECK (((acknowledged_by IS NULL) OR (acknowledged_by_service_account_id IS NULL))), + CONSTRAINT incidents_status_check CHECK ((status = ANY (ARRAY['triggered'::text, 'acknowledged'::text, 'resolved'::text]))) +); + +CREATE TABLE integrations ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + team_id bigint NOT NULL, + kind text NOT NULL, + name text NOT NULL, + key_hash text NOT NULL, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + last_used_at bigint, + CONSTRAINT integrations_kind_check CHECK ((kind = 'alertmanager'::text)) +); + +CREATE TABLE invites ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + token_hash text NOT NULL, + team_id bigint NOT NULL, + role text NOT NULL, + created_by bigint, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + expires_at bigint NOT NULL, + max_uses bigint DEFAULT 1 NOT NULL, + uses bigint DEFAULT 0 NOT NULL, + revoked_at bigint, + CONSTRAINT invites_max_uses_check CHECK (((max_uses > 0) AND (max_uses <= 100))), + CONSTRAINT invites_role_check CHECK ((role = ANY (ARRAY['owner'::text, 'member'::text]))) +); + +CREATE TABLE notifications ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + incident_id bigint NOT NULL, + user_id bigint, + topic text NOT NULL, + kind text NOT NULL, + created_at bigint NOT NULL, + send_after bigint NOT NULL, + attempts bigint DEFAULT 0 NOT NULL, + sent_at bigint, + last_error text, + CONSTRAINT notifications_kind_check CHECK ((kind = ANY (ARRAY['triggered'::text, 'reminder'::text, 'resolved'::text, 'escalated'::text]))) +); + +CREATE TABLE oidc_logins ( + state_hash text NOT NULL, + nonce text NOT NULL, + pkce_verifier text NOT NULL, + expires_at bigint NOT NULL, + next text DEFAULT '/'::text NOT NULL +); + +CREATE TABLE rate_limit_counters ( + key text NOT NULL, + window_start bigint NOT NULL, + count integer NOT NULL +); + +CREATE TABLE schedule_entries ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + user_id bigint NOT NULL, + date text NOT NULL, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + team_id bigint NOT NULL +); + +CREATE TABLE service_account_keys ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + service_account_id bigint NOT NULL, + key_hash text NOT NULL, + name text NOT NULL, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + last_used_at bigint +); + +CREATE TABLE service_accounts ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + name text NOT NULL, + scope text NOT NULL, + team_id bigint, + created_by bigint, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + CONSTRAINT service_accounts_scope_check CHECK ((scope = ANY (ARRAY['instance'::text, 'team'::text]))), + CONSTRAINT service_accounts_scope_team_id_chk CHECK ((((scope = 'team'::text) AND (team_id IS NOT NULL)) OR ((scope = 'instance'::text) AND (team_id IS NULL)))) +); + +CREATE TABLE sessions ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + token_hash text NOT NULL, + user_id bigint NOT NULL, + created_at bigint NOT NULL, + last_seen_at bigint NOT NULL, + expires_at bigint NOT NULL, + user_agent text, + max_expires_at bigint +); + +CREATE TABLE settings ( + key text NOT NULL, + value text NOT NULL, + updated_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL +); + +CREATE TABLE team_members ( + team_id bigint NOT NULL, + user_id bigint NOT NULL, + role text NOT NULL, + joined_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + source text DEFAULT 'manual'::text NOT NULL, + CONSTRAINT team_members_role_check CHECK ((role = ANY (ARRAY['owner'::text, 'member'::text]))), + CONSTRAINT team_members_source_check CHECK ((source = ANY (ARRAY['manual'::text, 'oidc'::text]))) +); + +CREATE TABLE teams ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + name text NOT NULL, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + oidc_member_group text, + oidc_owner_group text, + -- A stable identity for a team managed by automation (terdut-operator: + -- "/" of its TerdutTeam), so it can find or recreate its own + -- team without trusting a display name. NULL for a team a person made. + external_id text +); + +CREATE TABLE user_identities ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + user_id bigint NOT NULL, + issuer text NOT NULL, + subject text NOT NULL, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + last_login_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL +); + +CREATE TABLE users ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + username text NOT NULL, + email text NOT NULL, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + ntfy_topic text, + password_hash text, + is_admin boolean DEFAULT false NOT NULL, + disabled_at bigint, + invited_via bigint, + onboarding_dismissed_at bigint, + admin_source text DEFAULT 'manual'::text NOT NULL, + CONSTRAINT users_admin_source_check CHECK ((admin_source = ANY (ARRAY['manual'::text, 'oidc'::text]))) +); + +ALTER TABLE alerts + ADD CONSTRAINT alerts_pkey PRIMARY KEY (id); + +ALTER TABLE api_keys + ADD CONSTRAINT api_keys_key_hash_key UNIQUE (key_hash); + +ALTER TABLE api_keys + ADD CONSTRAINT api_keys_pkey PRIMARY KEY (id); + +ALTER TABLE deadman_switches + ADD CONSTRAINT deadman_switches_pkey PRIMARY KEY (id); + +ALTER TABLE device_logins + ADD CONSTRAINT device_logins_pkey PRIMARY KEY (device_hash); + +ALTER TABLE device_logins + ADD CONSTRAINT device_logins_user_code_key UNIQUE (user_code); + +ALTER TABLE escalation_levels + ADD CONSTRAINT escalation_levels_pkey PRIMARY KEY (id); + +ALTER TABLE escalation_levels + ADD CONSTRAINT escalation_levels_team_id_position_key UNIQUE (team_id, "position"); + +ALTER TABLE escalation_policies + ADD CONSTRAINT escalation_policies_pkey PRIMARY KEY (team_id); + +ALTER TABLE escalation_targets + ADD CONSTRAINT escalation_targets_pkey PRIMARY KEY (id); + +ALTER TABLE incident_ack_tokens + ADD CONSTRAINT incident_ack_tokens_pkey PRIMARY KEY (token_hash); + +ALTER TABLE incident_alerts + ADD CONSTRAINT incident_alerts_pkey PRIMARY KEY (incident_id, alert_id); + +ALTER TABLE incident_events + ADD CONSTRAINT incident_events_pkey PRIMARY KEY (id); + +ALTER TABLE incidents + ADD CONSTRAINT incidents_pkey PRIMARY KEY (id); + +ALTER TABLE integrations + ADD CONSTRAINT integrations_key_hash_key UNIQUE (key_hash); + +ALTER TABLE integrations + ADD CONSTRAINT integrations_pkey PRIMARY KEY (id); + +ALTER TABLE invites + ADD CONSTRAINT invites_pkey PRIMARY KEY (id); + +ALTER TABLE invites + ADD CONSTRAINT invites_token_hash_key UNIQUE (token_hash); + +ALTER TABLE notifications + ADD CONSTRAINT notifications_pkey PRIMARY KEY (id); + +ALTER TABLE oidc_logins + ADD CONSTRAINT oidc_logins_pkey PRIMARY KEY (state_hash); + +ALTER TABLE rate_limit_counters + ADD CONSTRAINT rate_limit_counters_pkey PRIMARY KEY (key); + +ALTER TABLE schedule_entries + ADD CONSTRAINT schedule_entries_pkey PRIMARY KEY (id); + +ALTER TABLE service_account_keys + ADD CONSTRAINT service_account_keys_key_hash_key UNIQUE (key_hash); + +ALTER TABLE service_account_keys + ADD CONSTRAINT service_account_keys_pkey PRIMARY KEY (id); + +ALTER TABLE service_accounts + ADD CONSTRAINT service_accounts_name_key UNIQUE (name); + +ALTER TABLE service_accounts + ADD CONSTRAINT service_accounts_pkey PRIMARY KEY (id); + +ALTER TABLE sessions + ADD CONSTRAINT sessions_pkey PRIMARY KEY (id); + +ALTER TABLE sessions + ADD CONSTRAINT sessions_token_hash_key UNIQUE (token_hash); + +ALTER TABLE settings + ADD CONSTRAINT settings_pkey PRIMARY KEY (key); + +ALTER TABLE team_members + ADD CONSTRAINT team_members_pkey PRIMARY KEY (team_id, user_id); + +ALTER TABLE teams + ADD CONSTRAINT teams_name_key UNIQUE (name); + +ALTER TABLE teams + ADD CONSTRAINT teams_external_id_key UNIQUE (external_id); + +ALTER TABLE teams + ADD CONSTRAINT teams_pkey PRIMARY KEY (id); + +ALTER TABLE user_identities + ADD CONSTRAINT user_identities_issuer_subject_key UNIQUE (issuer, subject); + +ALTER TABLE user_identities + ADD CONSTRAINT user_identities_pkey PRIMARY KEY (id); + +ALTER TABLE users + ADD CONSTRAINT users_email_key UNIQUE (email); + +ALTER TABLE users + ADD CONSTRAINT users_pkey PRIMARY KEY (id); + +ALTER TABLE users + ADD CONSTRAINT users_username_key UNIQUE (username); + +CREATE INDEX alerts_archived_at_idx ON alerts USING btree (archived_at); + +CREATE INDEX alerts_integration_idx ON alerts USING btree (integration_id, received_at) WHERE (integration_id IS NOT NULL); + +CREATE INDEX alerts_name_idx ON alerts USING btree (name); + +CREATE INDEX alerts_received_at_idx ON alerts USING btree (received_at DESC); + +CREATE INDEX alerts_status_idx ON alerts USING btree (status); + +CREATE UNIQUE INDEX alerts_team_fingerprint_idx ON alerts USING btree (team_id, fingerprint); + +CREATE INDEX alerts_team_received_idx ON alerts USING btree (team_id, received_at DESC); + +CREATE INDEX deadman_switches_team_idx ON deadman_switches USING btree (team_id); + +CREATE INDEX device_logins_expires_idx ON device_logins USING btree (expires_at); + +CREATE INDEX escalation_targets_level_idx ON escalation_targets USING btree (level_id); + +CREATE INDEX idx_sessions_user ON sessions USING btree (user_id); + +CREATE INDEX incident_ack_tokens_expires_idx ON incident_ack_tokens USING btree (expires_at); + +CREATE INDEX incident_alerts_alert_id_idx ON incident_alerts USING btree (alert_id); + +CREATE INDEX incident_events_actor_service_account_id_idx ON incident_events USING btree (actor_service_account_id); + +CREATE INDEX incident_events_actor_user_id_idx ON incident_events USING btree (actor_user_id); + +CREATE INDEX incident_events_incident_idx ON incident_events USING btree (incident_id, created_at); + +CREATE INDEX incident_events_service_account_id_idx ON incident_events USING btree (service_account_id); + +CREATE INDEX incidents_acknowledged_by_service_account_id_idx ON incidents USING btree (acknowledged_by_service_account_id); + +CREATE INDEX incidents_archived_at_idx ON incidents USING btree (archived_at); + +CREATE INDEX incidents_escalation_idx ON incidents USING btree (escalation_level_at) WHERE ((resolved_at IS NULL) AND (status = 'triggered'::text)); + +CREATE UNIQUE INDEX incidents_open_group_key_idx ON incidents USING btree (team_id, group_key) WHERE (resolved_at IS NULL); + +CREATE INDEX incidents_signature_idx ON incidents USING btree (team_id, signature, triggered_at DESC); + +CREATE INDEX incidents_status_idx ON incidents USING btree (status); + +CREATE INDEX incidents_team_triggered_idx ON incidents USING btree (team_id, triggered_at DESC); + +CREATE INDEX incidents_triggered_at_idx ON incidents USING btree (triggered_at DESC); + +CREATE INDEX integrations_team_idx ON integrations USING btree (team_id); + +-- A name identifies an integration (and a switch) within its team, so a client +-- that manages them declaratively can look one up by name instead of listing +-- and matching. +CREATE UNIQUE INDEX integrations_team_name_key ON integrations (team_id, name); +CREATE UNIQUE INDEX deadman_switches_team_name_key ON deadman_switches (team_id, name); + +CREATE INDEX invites_team_idx ON invites USING btree (team_id); + +CREATE INDEX notifications_incident_idx ON notifications USING btree (incident_id, id DESC); + +CREATE INDEX notifications_pending_idx ON notifications USING btree (send_after) WHERE (sent_at IS NULL); + +CREATE INDEX oidc_logins_expires_idx ON oidc_logins USING btree (expires_at); + +CREATE INDEX schedule_entries_date_idx ON schedule_entries USING btree (date); + +CREATE UNIQUE INDEX schedule_entries_team_date_idx ON schedule_entries USING btree (team_id, date); + +CREATE INDEX service_account_keys_service_account_id_idx ON service_account_keys USING btree (service_account_id); + +CREATE INDEX service_accounts_team_id_idx ON service_accounts USING btree (team_id); + +CREATE INDEX team_members_user_idx ON team_members USING btree (user_id); + +CREATE INDEX user_identities_user_idx ON user_identities USING btree (user_id); + +CREATE INDEX users_is_admin_idx ON users USING btree (is_admin) WHERE is_admin; + +ALTER TABLE alerts + ADD CONSTRAINT alerts_integration_id_fkey FOREIGN KEY (integration_id) REFERENCES integrations(id) ON DELETE SET NULL; + +ALTER TABLE alerts + ADD CONSTRAINT alerts_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE api_keys + ADD CONSTRAINT api_keys_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE; + +ALTER TABLE deadman_switches + ADD CONSTRAINT deadman_switches_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE device_logins + ADD CONSTRAINT device_logins_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE; + +ALTER TABLE escalation_levels + ADD CONSTRAINT escalation_levels_team_id_fkey FOREIGN KEY (team_id) REFERENCES escalation_policies(team_id) ON DELETE CASCADE; + +ALTER TABLE escalation_policies + ADD CONSTRAINT escalation_policies_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE escalation_targets + ADD CONSTRAINT escalation_targets_level_id_fkey FOREIGN KEY (level_id) REFERENCES escalation_levels(id) ON DELETE CASCADE; + +ALTER TABLE escalation_targets + ADD CONSTRAINT escalation_targets_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE; + +ALTER TABLE incident_ack_tokens + ADD CONSTRAINT incident_ack_tokens_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE; + +ALTER TABLE incident_ack_tokens + ADD CONSTRAINT incident_ack_tokens_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE; + +ALTER TABLE incident_alerts + ADD CONSTRAINT incident_alerts_alert_id_fkey FOREIGN KEY (alert_id) REFERENCES alerts(id) ON DELETE CASCADE; + +ALTER TABLE incident_alerts + ADD CONSTRAINT incident_alerts_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE; + +ALTER TABLE incident_events + ADD CONSTRAINT incident_events_actor_service_account_id_fkey FOREIGN KEY (actor_service_account_id) REFERENCES service_accounts(id) ON DELETE SET NULL; + +ALTER TABLE incident_events + ADD CONSTRAINT incident_events_actor_user_id_fkey FOREIGN KEY (actor_user_id) REFERENCES users(id) ON DELETE SET NULL; + +ALTER TABLE incident_events + ADD CONSTRAINT incident_events_alert_id_fkey FOREIGN KEY (alert_id) REFERENCES alerts(id) ON DELETE SET NULL; + +ALTER TABLE incident_events + ADD CONSTRAINT incident_events_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE; + +ALTER TABLE incident_events + ADD CONSTRAINT incident_events_service_account_id_fkey FOREIGN KEY (service_account_id) REFERENCES service_accounts(id) ON DELETE SET NULL; + +ALTER TABLE incident_events + ADD CONSTRAINT incident_events_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE SET NULL; + +ALTER TABLE incidents + ADD CONSTRAINT incidents_acknowledged_by_fkey FOREIGN KEY (acknowledged_by) REFERENCES users(id) ON DELETE SET NULL; + +ALTER TABLE incidents + ADD CONSTRAINT incidents_acknowledged_by_service_account_id_fkey FOREIGN KEY (acknowledged_by_service_account_id) REFERENCES service_accounts(id) ON DELETE SET NULL; + +ALTER TABLE incidents + ADD CONSTRAINT incidents_assigned_to_fkey FOREIGN KEY (assigned_to) REFERENCES users(id) ON DELETE SET NULL; + +ALTER TABLE incidents + ADD CONSTRAINT incidents_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE integrations + ADD CONSTRAINT integrations_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE invites + ADD CONSTRAINT invites_created_by_fkey FOREIGN KEY (created_by) REFERENCES users(id) ON DELETE SET NULL; + +ALTER TABLE invites + ADD CONSTRAINT invites_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE notifications + ADD CONSTRAINT notifications_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE; + +ALTER TABLE notifications + ADD CONSTRAINT notifications_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE SET NULL; + +ALTER TABLE schedule_entries + ADD CONSTRAINT schedule_entries_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE schedule_entries + ADD CONSTRAINT schedule_entries_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE; + +ALTER TABLE service_account_keys + ADD CONSTRAINT service_account_keys_service_account_id_fkey FOREIGN KEY (service_account_id) REFERENCES service_accounts(id) ON DELETE CASCADE; + +ALTER TABLE service_accounts + ADD CONSTRAINT service_accounts_created_by_fkey FOREIGN KEY (created_by) REFERENCES users(id) ON DELETE SET NULL; + +ALTER TABLE service_accounts + ADD CONSTRAINT service_accounts_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE sessions + ADD CONSTRAINT sessions_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE; + +ALTER TABLE team_members + ADD CONSTRAINT team_members_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE team_members + ADD CONSTRAINT team_members_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE; + +ALTER TABLE user_identities + ADD CONSTRAINT user_identities_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE; + +ALTER TABLE users + ADD CONSTRAINT users_invited_via_fkey FOREIGN KEY (invited_via) REFERENCES invites(id) ON DELETE SET NULL; diff --git a/internal/models/alert.go b/internal/models/alert.go index 01047e8..c6e0356 100644 --- a/internal/models/alert.go +++ b/internal/models/alert.go @@ -33,7 +33,7 @@ type Alert struct { // refreshed. The sweeper stale-dates against it (see expireStale), API // clients render it, and GET /api/alerts is ordered by it. Anything that // stops the webhook handler from advancing it on a re-send is a breaking - // change — see "received_at is a liveness heartbeat" in the README and + // change — see "received_at is a liveness heartbeat" in docs/api.md and // TestWebhook_ResendBumpsReceivedAt. ReceivedAt time.Time `json:"received_at"` @@ -52,7 +52,7 @@ type Alert struct { // inferred. Under "expiry" nothing ever reported an end, so EndsAt is only // an upper bound (see expireStale) and ReceivedAt is the more truthful // signal. Treat the value set as open — see "resolution_source says how much - // to trust ends_at" in the README, and TestWebhook_ResolvedSetsSource / + // to trust ends_at" in docs/api.md, and TestWebhook_ResolvedSetsSource / // TestExpiry_StaleFiringAlert. ResolutionSource *string `json:"resolution_source,omitempty"` diff --git a/internal/models/team.go b/internal/models/team.go index 8bedf25..0ed6365 100644 --- a/internal/models/team.go +++ b/internal/models/team.go @@ -9,6 +9,10 @@ type Team struct { Name string `json:"name"` CreatedAt time.Time `json:"created_at"` + // ExternalID identifies a team managed by automation; see handleCreateTeam. + // Shown to instance service accounts and admins only. + ExternalID *string `json:"external_id,omitempty"` + // Role is the caller's own role in this team, populated when a team is // listed for a particular person. Empty when nobody in particular is // asking, as in the admin listing. diff --git a/internal/oidc/grants.go b/internal/oidc/grants.go index 2a67a65..4de53a1 100644 --- a/internal/oidc/grants.go +++ b/internal/oidc/grants.go @@ -102,6 +102,3 @@ func rank(role string) int { } return 0 } - -// HigherRole reports whether role a outranks role b. -func HigherRole(a, b string) bool { return rank(a) > rank(b) } diff --git a/internal/web/static/js/format.js b/internal/web/static/js/format.js index 9bf994f..98be815 100644 --- a/internal/web/static/js/format.js +++ b/internal/web/static/js/format.js @@ -123,7 +123,7 @@ export function initial(name) { // convention. It comes from Prometheus's externalLabels, so it is on every // alert; an incident carries it only when it is in Alertmanager's group_by, // which is also what keeps two clusters' identical alerts from merging into one -// incident (see the README, "Several clusters, one team"). +// incident (see docs/incidents.md, "Several clusters, one team"). export const ORIGIN_LABEL = 'cluster'; export function originOf(labels) {