diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 693ad74..09cd03d 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -35,7 +35,7 @@ jobs: # Runs inside the toolchain image rather than installing Go per job. Note this puts # the job on the dind bridge, which cannot reach github.com or get.helm.sh -- # proxy.golang.org and git.ryuvia.com are reachable, which is all this job needs. - image: golang:1.26.6-bookworm + image: golang:1.26.9-bookworm # act_runner destroys a job's own volumes when it finishes, so without these every # run re-downloads the whole module graph. The names must appear in the runner's # container.valid_volumes allowlist (charts/act-runner in the k8s repo); unlisted @@ -46,8 +46,8 @@ jobs: - go-build-cache:/root/.cache/go-build - gobin-cache:/go/bin - # The suite needs a real Postgres -- there is no in-memory Postgres the way there was - # an in-memory SQLite, so each test gets its own schema on a shared server instead. + # The suite needs a real Postgres -- there is no in-memory Postgres, + # so each test gets its own schema on a shared server instead. # The job and the service share the dind bridge, so the service is reachable by its # name rather than on localhost. services: @@ -101,7 +101,7 @@ jobs: security: runs-on: ubuntu-latest container: - image: golang:1.26.6-bookworm + image: golang:1.26.9-bookworm volumes: - go-mod-cache:/go/pkg/mod - go-build-cache:/root/.cache/go-build diff --git a/.gitea/workflows/release.yaml b/.gitea/workflows/release.yaml index 50bd5ff..c4ded87 100644 --- a/.gitea/workflows/release.yaml +++ b/.gitea/workflows/release.yaml @@ -30,7 +30,7 @@ jobs: test: runs-on: ubuntu-latest container: - image: golang:1.26.6-bookworm + image: golang:1.26.9-bookworm volumes: - go-mod-cache:/go/pkg/mod - go-build-cache:/root/.cache/go-build @@ -71,7 +71,7 @@ jobs: needs: test runs-on: ubuntu-latest container: - image: golang:1.26.6-bookworm + image: golang:1.26.9-bookworm volumes: - go-mod-cache:/go/pkg/mod - go-build-cache:/root/.cache/go-build diff --git a/.gitignore b/.gitignore index aa08c62..31709d7 100644 --- a/.gitignore +++ b/.gitignore @@ -7,8 +7,6 @@ # one (which has an unreachable entry) cannot abort a release /.helm-repos.yaml -# SQLite database files -*.db *.db-shm *.db-wal diff --git a/CLAUDE.md b/CLAUDE.md index 278c78c..ef79927 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -12,7 +12,7 @@ Preconditions and the plan, without side effects: ``` Config is `.release.conf` here plus `make release-vars`. The process itself lives in -`~/.claude/skills/release/`; why it is shaped this way is in README.md §Releasing. +`~/.claude/skills/release/`; why it is shaped this way is in docs/development.md (Releasing). Three things about this repo specifically: diff --git a/Makefile b/Makefile index 3f9b660..d284c30 100644 --- a/Makefile +++ b/Makefile @@ -28,7 +28,7 @@ help: ## Show this help # -race below. Both need a Postgres to test against; see test-db. # The suite needs a Postgres, because the server does: there is no in-memory -# Postgres the way there was an in-memory SQLite. TERDUT_TEST_DSN says where, and +# Postgres. TERDUT_TEST_DSN says where, and # the tests fail rather than skip without it — a suite that quietly tests nothing # is worse than one that does not run. `make test-db` starts a local one; # ci.yaml runs the same thing as a service container. diff --git a/README.md b/README.md index da37499..fc57704 100644 --- a/README.md +++ b/README.md @@ -1,31 +1,72 @@ # Terminal Duty (terdut-server) -Incident management server for teams using Prometheus Alertmanager. +Incident management for teams that already run Prometheus Alertmanager. Point Alertmanager at it, +and alerts become incidents that get assigned to whoever is on call, paged, escalated when nobody +answers, and tracked to resolution. One binary, one Postgres. -- Receives Alertmanager webhooks directly — no adapter needed -- Turns alerts into **incidents**, correlated by Alertmanager's own `groupKey` -- Incident workflow: acknowledge, assign, snooze, note, resolve, with a full timeline -- On-call schedule management, with new incidents auto-assigned to whoever is on call -- Alert and incident statistics, including MTTA and MTTR -- Web UI for phones and desktops, served by the same binary -- REST API with per-user API key authentication -- Single binary plus a Postgres — straightforward to self-host + ---- +## Highlights + +- **Alertmanager-native.** Receives Alertmanager webhooks directly, with no adapter, and groups + alerts into incidents by Alertmanager's own `groupKey`. An incident opens on a new occurrence, + not on every re-send. [Details](./docs/incidents.md) +- **A real incident workflow.** Acknowledge, assign, snooze, add notes, resolve and archive, with a + full timeline of who did what and when. Several clusters can feed one team without + cross-talk. [Details](./docs/incidents.md) +- **On-call rota.** Each team keeps its own rota, and new incidents go to whoever is on call. + [Web UI](./docs/web-ui.md) +- **Pages that escalate.** Push notifications through [ntfy](https://ntfy.sh), with an + Acknowledge button right in the notification, and escalation ladders that move on to the next + level when nobody answers. [Notifications](./docs/notifications.md) · + [Escalation](./docs/escalation.md) +- **Notices when the alerts stop.** Dead man's switches turn the absence of a heartbeat such as + `Watchdog` into an incident. [Details](./docs/dead-mans-switch.md) +- **Teams.** Every team owns its queue, rota, escalation, alert sources and switches; people see + only the teams they belong to. +- **Stats.** Incident counts, mean time to acknowledge and resolve, and alert frequency by name, + hour and day. [Web UI](./docs/web-ui.md) +- **Single sign-on.** OpenID Connect with group-to-team and administrator mapping, and a device + flow so a terminal client can sign in through the browser. [Details](./docs/single-sign-on.md) +- **Built for the phone first.** The web UI is served by the same binary, follows the system's + dark mode, and can be added to the home screen. +- **An API for everything.** A REST API with API keys and service accounts for automation. + [Reference](./docs/api.md) +- **Easy to run.** A scratch container image, a Helm chart, and + [terdut-operator](https://git.ryuvia.com/niklas/terdut-operator) if you want teams and + escalation as Kubernetes objects. [Deployment](./docs/deployment.md) + +## Screenshots + +| | | +|---|---| +|  |  | +| **Work an incident**: acknowledge, assign, snooze, note, resolve | **Dark mode**, following the system | +|  |  | +| **On call**: who holds the pager now, and the week ahead | **Escalation**: who is paged next, and when | +|  |  | +| **Stats**: MTTA, MTTR and what fires most | **Alerts**: the raw feed behind the incidents | + +On a phone the queue and the incident page are the same interface, with a sticky action bar: + +
+
+
+
` — one of `denied`, `expired`, `failed`, `unavailable`, `not_allowed`, `no_email`, `email_conflict`, `disabled`, `not_bootstrapped` (no user exists on this install yet — sign in again once something has called `/api/bootstrap`) |
-| `POST` | `/api/logout` | Ends the session and clears the cookie |
-| `GET` | `/api/me` | The caller: `{user, has_password}` |
-
-### Users
-
-| Method | Path | Description |
-|---|---|---|
-**admin** marks an endpoint that requires the administrator flag; **self or
-admin** marks one you may use on your own account and an administrator may use
-on anybody's.
-
-| Method | Path | Who | Description |
-|---|---|---|---|
-| `GET` | `/api/signup` | — | Whether sign-up is open, and whether `?invite=` is usable. No session needed: the caller has no account yet |
-| `POST` | `/api/signup` | — | Create an account `{"username","email","password","invite"?,"team_name"?}` and sign in. `403` without a usable invite when the mode is invite-only |
-| `POST` | `/api/bootstrap` | — | Create first user + API key `{"username","email","password"?}` (only works on empty DB). The user is an administrator |
-| `GET` | `/api/users` | any | List users. Open to everybody: the queue's assignment control and the schedule both have to name people |
-| `GET` | `/api/users/{id}/teams` | self or admin | The teams that user is in, each with their role. `/api/teams` is always about the caller; this one answers it about somebody else, for the admin page's per-user view. `404` for a user who does not exist, so "no teams" and "no such person" are distinguishable |
-| `POST` | `/api/users` | **admin** | Create user `{"username","email"}`. Not an administrator |
-| `DELETE` | `/api/users/{id}` | **admin** | Delete user (cascades to keys). `409` for yourself or the last administrator |
-| `PUT` | `/api/users/{id}/admin` | **admin** | Grant or revoke the administrator flag `{"is_admin"}`. `409` for yourself, the last administrator, or an administrator granted by single sign-on |
-| `PUT` | `/api/users/{id}/disabled` | **admin** | Take an account out of use, or put it back `{"disabled"}`. `409` for yourself or the last administrator |
-| `PUT` | `/api/users/{id}/notify` | self or admin | Set push notification target `{"ntfy_topic"}` — empty string clears it |
-| `PUT` | `/api/users/{id}/password` | self or admin | Set web UI password `{"password","current_password"}`. `current_password` is required only when changing your own existing password. Ends the user's other sessions |
-| `POST` | `/api/users/{id}/api-keys` | self or admin | Issue API key `{"name"}` — key shown once |
-| `DELETE` | `/api/users/{id}/api-keys/{keyID}` | self or admin | Revoke API key |
-
-### Administration
-
-| Method | Path | Who | Description |
-|---|---|---|---|
-| `GET` | `/api/admin/teams` | **admin** | Every team on the server, with its member and open-incident counts. `/api/teams` answers "what am I in"; this answers "what is there" |
-| `GET` | `/api/admin/teams/{teamID}` | **admin** | One team and who is in it: `{"team", "members"}`. `404` for a team that does not exist. `GET /api/teams/{teamID}/members` is **member**-only and still `404`s an administrator from outside the team — reading a team's shape and reading its work are different questions, so they are different endpoints |
-| `GET` | `/api/admin/settings` | **admin** | The editable settings with their bounds, plus the environment-configured ones, read-only. Never credentials |
-| `PUT` | `/api/admin/settings` | **admin** | Change one or more `{"key": seconds}`, or `{"signup_mode": "open"\|"invite_only"}`. `400` for an unknown key or a value outside its bounds |
-
-### Service accounts
-
-A service account is a scoped, non-human credential for automation — not a
-`users` row, so it never signs in, is never a team member, and never carries
-the administrator flag. Two scopes:
-
-- **instance** — the same reach system administration has over teams: create
- one, and mint a **team**-scoped account against any of them. There is no
- cap on how many instance-scoped accounts exist, but ordinarily there is one,
- belonging to whatever is provisioning this install end to end.
-- **team** — owner-equivalent for that one team, and nothing else: every
- **owner**-gated endpoint under [Teams](#teams), membership and invites
- included. Nothing narrower is enforced server-side; what actually keeps
- membership out of automation's hands is that no operator built against this
- scope should ever call those two endpoints — see
- [operator mode](#authentication) and `SERVICE-ACCOUNTS.md`'s note on this.
-
-A key is shown once, at creation or rotation, and only its hash is stored —
-the same handling as a user's API key. Losing it means minting a new one;
-there is no way to recover a raw key from the server.
-
-| Method | Path | Who | Description |
-|---|---|---|---|
-| `GET` | `/api/service-accounts` | **admin** | Every service account. Pass `?name=` instead to look one up by its exact name — open to **any** authenticated caller (human or service account), since it returns no key material and is how an account finds its own id |
-| `POST` | `/api/service-accounts` | owner\* | Create one and mint its first key `{"name","scope","team_id"?}` (`team_id` required for `scope:"team"`, absent for `scope:"instance"`). Returns `{"service_account", "key"}` — `key.key` shown once |
-| `POST` | `/api/service-accounts/{id}/keys` | owner\* | Mint an additional key `{"name"}` — rotation without recreating the account. Shown once |
-| `DELETE` | `/api/service-accounts/{id}/keys/{keyID}` | owner\* | Revoke one key |
-
-\* For an **instance**-scoped account: a system administrator only. For a
-**team**-scoped account: a system administrator, that team's own human owner,
-an instance-scoped service account (minting a narrower credential for a team
-it just created), or — for the two key endpoints only — the account rotating
-or revoking its own key, which is not a privilege escalation, the same
-reasoning a user's own API keys rest on.
-
-### Alert ingestion
-
-Alerts arrive on a team's integration key. The key is both the credential and the
-routing: it says that the sender may post, and which team the alerts belong to.
-Create one with `POST /api/teams/{teamID}/integrations`, which returns the key
-and the full URL once and stores only a SHA-256 hash.
-
-| Method | Path | Description |
-|---|---|---|
-| `POST` | `/api/integrations/{key}/alertmanager` | Alertmanager v4 webhook receiver for the key's team. `401` for an unknown key |
-
-This is the only way in. The pre-teams `POST /api/alertmanager/webhook` took no
-credential at all — anything able to reach the port could open an incident —
-and was removed in v0.13.0 once senders had moved onto keys.
-
-### Teams
-
-**owner** below means an owner of that team, a system administrator (who
-passes every one of these without being a member), or that team's own
-team-scoped [service account](#service-accounts) — including membership and
-invites, technically, though no automation this scope was designed for
-(a Kubernetes operator's CRDs, see `SERVICE-ACCOUNTS.md`) ever models team
-membership or would call those two. See [Authentication](#authentication).
-**member** means membership and nothing else: an administrator who is not in
-the team gets the same `404` as anybody else.
-
-| Method | Path | Who | Description |
-|---|---|---|---|
-| `GET` | `/api/teams` | any | The caller's own teams, each with their role |
-| `POST` | `/api/teams` | any | Create a team `{"name"}`; a human creator becomes its first owner. An instance-scoped [service account](#service-accounts) may also create one, and it gets no owner at all — expected for a team an operator is about to hand a team-scoped credential to, not an orphaned team a human made |
-| `PUT` | `/api/teams/{teamID}` | **owner** | Rename it `{"name"}`. `409` if the name is taken |
-| `DELETE` | `/api/teams/{teamID}` | **owner** | Delete a team and everything under it. `409` while it has open incidents |
-| `GET` | `/api/teams/{teamID}/members` | member | Who is in the team, with `status` (`oncall` if the rota has them today, `unpageable` when a page to them would go nowhere — even if they are on call — else `reachable`), `on_call`, `next_shift` (first rota day after today), `pageable` and `problem` (`has no ntfy topic` / `account is disabled`; never the topic itself) and `last_active_at` (their newest session or API-key use). Every member sees the same list |
-| `POST` | `/api/teams/{teamID}/members` | **owner** | Add a member, or change their role `{"user_id","role"}`. `409` when it would demote the last owner, or the membership is managed by single sign-on |
-| `DELETE` | `/api/teams/{teamID}/members/{userID}` | **owner** | Remove a member. `409` for the last owner, or a membership managed by single sign-on |
-| `GET` | `/api/teams/{teamID}/oidc-groups` | member | Which groups control this team's membership: `{"member_group","owner_group"}`. An empty string means no group grants that role here |
-| `PUT` | `/api/teams/{teamID}/oidc-groups` | **owner** | Set them. An empty string clears a binding |
-| `GET` | `/api/teams/{teamID}/integrations` | member | List integrations. Never returns keys. Each carries `status` (`active` if its key posted within 24h, `quiet` if it has but not lately, `never`), `last_used_at` (last webhook, usable or not), `last_alert_at` (when an alert last arrived on it) and `alerts_24h` (distinct alerts it refreshed in the last day). Alerts delivered before the source was recorded (migration 010) have none, so the last two fill in as Alertmanager re-sends them |
-| `PATCH` | `/api/teams/{teamID}/integrations/{integrationID}` | **owner** | Rename `{"name"}`. The key does not change |
-| `POST` | `/api/teams/{teamID}/integrations` | **owner** | Mint an integration `{"name","kind"}` — key and URL shown once |
-| `DELETE` | `/api/teams/{teamID}/integrations/{integrationID}` | **owner** | Revoke an integration. Alerts it delivered stay, unattributed |
-| `GET` | `/api/teams/{teamID}/invites` | **owner** | The team's invite links, with their uses and expiry. Never the tokens |
-| `POST` | `/api/teams/{teamID}/invites` | **owner** | Mint one `{"role","max_uses"}` — the full URL is returned once |
-| `DELETE` | `/api/teams/{teamID}/invites/{inviteID}` | **owner** | Revoke a link before it expires |
-| `GET` | `/api/teams/{teamID}/escalation` | member | The team's [escalation ladder](#escalation) `{repeat_count, fallback_topic, levels[], last_escalated_at?, last_escalated_incident_id?}`. Empty levels means the team has none. Each level also carries `status` (`ready`, `escalating` when an unanswered incident has climbed to it, `unreachable` when nobody on it could be woken), `waiting` (ids of the open incidents on it) and, per target, `username` (who it means today — the person on call, for a rota target), `reachable` and `problem`. The extra fields are output only; `PUT` takes the plain shape |
-| `PUT` | `/api/teams/{teamID}/escalation` | **owner** | Replace it wholesale. `400` for a level with no targets or no timeout — a rung that pages nobody is a silence with a number on it |
-| `GET` | `/api/teams/{teamID}/deadman/switches` | member | The team's [dead man's switches](#dead-mans-switch), each `{id, name, matcher, timeout_seconds, severity, status, last_heartbeat_at, last_triggered_at, open_incident_id, sources[]}`. `status` is `healthy`, `dead` or `dormant`; `sources` has one entry per heartbeat fingerprint. Empty when the team watches nothing |
-| `POST` | `/api/teams/{teamID}/deadman/switches` | **owner** | Add one: `{name?, matcher, timeout_seconds, severity?}`. `400` when the matcher names no `alertname` or holds several, or the timeout is not positive — a switch that silently watches nothing is the failure this feature exists to prevent |
-| `PUT` | `/api/teams/{teamID}/deadman/switches/{switchID}` | **owner** | Replace one in place, same body and validation as create. Its id is unchanged — for an automated caller reconciling a spec change, unlike delete-and-recreate |
-| `DELETE` | `/api/teams/{teamID}/deadman/switches/{switchID}` | **owner** | Stop watching. An incident it opened stays open. `404` for a switch of another team |
-
-### Notifications
-
-| Method | Path | Description |
-|---|---|---|
-| `POST` | `/api/notify/ack/{token}` | Acknowledge an incident from a push notification's Acknowledge button. No auth: the token in the path is the credential — one incident, one action, 24 hours, idempotent. Must stay publicly reachable |
-
-### Incidents
-
-| Method | Path | Description |
-|---|---|---|
-| `GET` | `/api/incidents` | List incidents. Filters: `?status=triggered\|acknowledged\|resolved`, `?severity=`, `?assigned_to=`, `?archived=true`, `?snoozed=true`, `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD`, `?sort=severity`, `?cluster=`, `?limit=` (default 50, max 500) |
-| `GET` | `/api/incidents/clusters` | The distinct `cluster` values on the caller's incidents from the last 90 days, sorted (`?team_id=` narrows it). An empty array when nothing carries the label |
-| `GET` | `/api/incidents/{id}` | Get single incident, with its alerts inline |
-| `GET` | `/api/incidents/{id}/alerts` | Alerts under this incident |
-| `GET` | `/api/incidents/{id}/timeline` | Full event history, chronological |
-| `POST` | `/api/incidents/{id}/acknowledge` | Acknowledge (stamps authed user + time) |
-| `DELETE` | `/api/incidents/{id}/acknowledge` | Clear acknowledgement, back to `triggered` |
-| `POST` | `/api/incidents/{id}/resolve` | Close by hand — **terminal**, see above |
-| `POST` | `/api/incidents/{id}/assign` | Reassign `{"user_id"}` |
-| `POST` | `/api/incidents/{id}/snooze` | Hide until `{"until": RFC3339}` or `{"duration": "2h"}` |
-| `DELETE` | `/api/incidents/{id}/snooze` | Un-snooze |
-| `POST` | `/api/incidents/{id}/archive` | Archive (hides from the default list) |
-| `DELETE` | `/api/incidents/{id}/archive` | Un-archive |
-| `POST` | `/api/incidents/{id}/notes` | Add a note `{"content"}` |
-| `DELETE` | `/api/incidents/{id}/notes/{eventID}` | Delete own note |
-
-With no `?status=` filter, `GET /api/incidents` returns **open** incidents only —
-the queue an on-call person wants. Currently snoozed and archived incidents are
-excluded unless asked for. Actions that only make sense on an open incident
-return `409` once it is resolved.
-
-Notes are ordinary timeline events of type `note`; only they are deletable, and
-only by their author. The rest of the timeline is a record of what happened.
-
-#### The incident object
-
-| Field | Type | Notes |
-|---|---|---|
-| `id` | integer | Server-assigned |
-| `group_key` | string | Alertmanager's `groupKey` — opaque, treat as an identifier |
-| `title` | string | Rendered from `groupLabels` |
-| `group_labels` | object | String→string, as sent by Alertmanager |
-| `status` | string | `"triggered"`, `"acknowledged"` or `"resolved"` |
-| `severity` | string | *optional* — high-water mark across the incident's alerts; never lowered |
-| `triggered_at` | timestamp | When the incident opened |
-| `acknowledged_by_id` / `acknowledged_by` / `acknowledged_at` | | *optional* — user id, username, time |
-| `assigned_to_id` / `assigned_to` | | *optional* — user id, username |
-| `snoozed_until` | timestamp | *optional* — a value in the past reads as not snoozed |
-| `resolved_at` | timestamp | *optional* |
-| `resolution_source` | string | *optional* — `"alerts"`, `"manual"` or `"recovered"` |
-| `archived_at` | timestamp | *optional* |
-| `alerts` | array | Only on `GET /api/incidents/{id}` |
-
-Treat `resolution_source` as an open set, as with the alert field of the same
-name: degrade unknown values to "resolved, reason unknown".
-
-#### The timeline event object
-
-| Field | Type | Notes |
-|---|---|---|
-| `id` | integer | |
-| `incident_id` | integer | |
-| `type` | string | See below — treat as an open set |
-| `user_id` / `username` | | *optional* — absent when the server acted rather than a person |
-| `alert_id` | integer | *optional* — the alert an `alert_added` / `alert_resolved` event refers to |
-| `detail` | string | *optional* — the note body, the snooze deadline, etc. |
-| `created_at` | timestamp | |
-
-Types written today: `triggered`, `alert_added`, `alert_resolved`,
-`acknowledged`, `unacknowledged`, `assigned`, `archived`, `unarchived`, `snoozed`,
-`unsnoozed`, `resolved`, `note`, `notified`, `notify_failed`, `deadman_silent`. On an
-`assigned` event `user_id` is the **assignee**, not the actor; the actor is in
-`actor_user_id`/`actor_username` or `actor_service_account_id`/`actor_service_account_name`
-(absent on assignments made before they were recorded). New types may be added; render
-unknown ones generically rather than dropping them.
-
-On `notified` and `notify_failed`, `detail` carries the notification kind
-(`triggered` | `reminder` | `resolved`), and on a failure the reason after it.
-`user_id` is who was paged — absent means the page went to the shared fallback
-topic and so belongs to nobody. The topic itself is never written to the
-timeline: it is a shared secret with the ntfy server, and every API key can read
-this.
-
-### Alerts
-
-Alerts are read-only. Everything a person does happens on the incident.
-
-| Method | Path | Description |
-|---|---|---|
-| `GET` | `/api/alerts` | List alerts. Filters: `?status=firing\|resolved`, `?name=`, `?incident_id=`, `?archived=true`, `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD`, `?limit=` (default 50, max 500) |
-| `GET` | `/api/alerts/{id}` | Get single alert |
-
-Archived alerts are hidden from `GET /api/alerts` unless `?archived=true` is
-passed; alert archiving is automatic housekeeping by the sweeper, not a user
-action. Resolved alerts carry `resolution_source`: `"alertmanager"` for a real
-resolved webhook, `"expiry"` when the sweeper inferred it (see
-[Stale alert expiry](#stale-alert-expiry)), `"deadman"` for a heartbeat declared
-dead (see [Dead man's switch](#dead-mans-switch)).
-
-#### The alert object
-
-Returned by `GET /api/alerts` (as an array) and `GET /api/alerts/{id}`.
-Timestamps are RFC 3339 in UTC. Fields marked *optional* are omitted entirely
-when unset, so clients must treat them as nullable.
-
-| Field | Type | Notes |
-|---|---|---|
-| `id` | integer | Server-assigned; stable for the life of the row |
-| `fingerprint` | string | Alertmanager's fingerprint — the upsert key |
-| `name` | string | From the `alertname` label |
-| `status` | string | `"firing"` or `"resolved"` |
-| `labels` | object | String→string, as sent by Alertmanager |
-| `annotations` | object | String→string, as sent by Alertmanager |
-| `starts_at` | timestamp | When the alert instance began, **per Prometheus** |
-| `ends_at` | timestamp | *optional* — absent while no end is known |
-| `generator_url` | string | Link back to the originating Prometheus |
-| `received_at` | timestamp | When the server last accepted a webhook for this alert — see below |
-| `incident_id` | integer | *optional* — the most recent incident this alert belongs to |
-| `resolution_source` | string | *optional* — `"alertmanager"`, `"expiry"` or `"deadman"` |
-| `archived_at` | timestamp | *optional* — set while archived |
-
-##### `received_at` is a liveness heartbeat
-
-`starts_at` comes from Prometheus and **never changes** for the lifetime of an
-alert instance. It says when the problem began, not whether it is still
-happening — an alert that started twelve days ago looks identical whether
-Alertmanager refreshed it a minute ago or went silent a week ago.
-
-`received_at` is the field that answers "is this still live". It is set to the
-server's clock on **every accepted webhook** for that fingerprint, including the
-unchanged firing notifications Alertmanager re-sends every `repeat_interval`.
-Clients may rely on this:
-
-- **A firing alert whose `received_at` is advancing is still being refreshed.**
- Stale-dating it against `repeat_interval` is a valid liveness check, and it is
- what the built-in sweeper does (see
- [Stale alert expiry](#stale-alert-expiry)).
-- **`received_at` tracks accepted payloads, not delivery attempts.** A retry
- that describes an older instance than the stored one is discarded, and a
- discarded payload does not move `received_at`.
-- **It stops advancing once the alert resolves,** because Alertmanager stops
- re-sending. On an alert resolved by the sweeper
- (`"resolution_source": "expiry"`) it therefore marks the last time
- Alertmanager was actually heard from, which is earlier than `ends_at`.
-
-`GET /api/alerts` is ordered by `received_at` descending — most recently
-refreshed first — and the `?from=` / `?to=` filters on both the alert and stats
-endpoints select on `received_at`, not `starts_at`.
-
-##### `resolution_source` says how much to trust `ends_at`
-
-An alert can leave the firing state two ways, and `resolution_source` records
-which happened. Clients may rely on this:
-
-- **Absent while firing.** It is set only on resolve, and a re-fire under the
- same fingerprint clears it again, so its presence always agrees with
- `"status": "resolved"`.
-- **`"alertmanager"` — a real resolved webhook arrived.** `ends_at` is the end
- time Alertmanager reported. It is an observed value and can be displayed as
- fact.
-- **`"expiry"` — the sweeper inferred the resolve** because Alertmanager stopped
- refreshing the alert (see [Stale alert expiry](#stale-alert-expiry)). Nothing
- ever reported an end, so **`ends_at` is approximate**: it is either the stale
- `endsAt` watermark from the last notification, or — when that notification
- carried none — the time the sweep ran, which lags the last real contact by up
- to `TERDUT_STALE_AFTER` plus a sweep interval. Treat it as "no later than",
- not as when the problem stopped.
-
- On these alerts `received_at` is the more truthful signal: it marks the last
- time Alertmanager was actually heard from. Surfacing the distinction is
- worthwhile, since `"expiry"` can also mean the alert is still firing and the
- notification path broke.
-
-- **`"deadman"` — a heartbeat was declared dead** (see
- [Dead man's switch](#dead-mans-switch)). Like `"expiry"`, an inference from
- silence rather than an observed end, so `ends_at` is approximate — but a much
- tighter one, bounded by `TERDUT_DEADMAN_TIMEOUT`. It is also the one resolution
- a re-fire under the same `starts_at` can undo, since the switch coming back is
- exactly the evidence that the inference was wrong.
-
-Treat the value as an open set and tolerate ones you do not recognise — new
-sources may be added, and unknown values should degrade to "resolved, reason
-unknown" rather than being rejected.
-
-### On-call schedule
-
-| Method | Path | Description |
-|---|---|---|
-Each team keeps its own rota, so two teams can have two different people on call
-on the same day. The person taking a shift has to be in the team — paging
-somebody who cannot open the incident is worse than paging nobody.
-
-| Method | Path | Who | Description |
-|---|---|---|---|
-| `POST` | `/api/teams/{teamID}/schedule` | **owner** | Assign user to dates `{"user_id", "dates":["YYYY-MM-DD",...], "replace"}` — all-or-nothing |
-| `GET` | `/api/teams/{teamID}/schedule` | member | List entries. Filters: `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD` |
-| `DELETE` | `/api/teams/{teamID}/schedule/{id}` | **owner** | Remove schedule entry |
-| `GET` | `/api/schedule/current` | any | Who is on call today (UTC) in **every** team the caller is in — one entry per team, `[]` when nobody anywhere |
-
-### Statistics
-
-Every figure counts the caller's own teams only: a report that counted other
-teams' incidents would leak their volume, and their alert names through the
-top-alerts list, and would not be a number about the reader's work anyway.
-
-All stat endpoints accept optional `?from=YYYY-MM-DD` and `?to=YYYY-MM-DD`, and exclude archived rows to match the default list views. Alert stats filter on `received_at`; incident stats filter on `triggered_at`.
-
-| Method | Path | Description |
-|---|---|---|
-| `GET` | `/api/stats/incidents` | `{total, triggered, acknowledged, resolved, mtta_seconds, mttr_seconds}` |
-| `GET` | `/api/stats/alerts` | `{total, firing, resolved}` counts |
-| `GET` | `/api/stats/alerts/top` | Most frequent alert names. `?limit=` (default 10, max 100) |
-| `GET` | `/api/stats/alerts/by-hour` | Count per hour-of-day (UTC), all 24 slots returned |
-| `GET` | `/api/stats/alerts/by-day` | Count per day-of-week, all 7 slots with names returned |
-
-`mtta_seconds` (time to acknowledge) and `mttr_seconds` (time to resolve) are
-averages over incidents that have actually been acknowledged or resolved, and are
-**null** until there are any — null means "no data", not zero.
-
----
-
-## Upgrading to teams
-
-Everything that existed before teams moves into one team called **Default**, and
-every existing user becomes an owner of it. The upgrade is a no-op for the
-people using it: the same queue, the same schedule, the same incidents, with a
-name on them.
-
-What changes, and will need attention:
-
-- **Alert ingestion moved.** Mint a key with
- `POST /api/teams/{teamID}/integrations` and point Alertmanager at the URL it
- returns. In v0.12.0 the old `POST /api/alertmanager/webhook` still worked,
- deprecated, routing everything to the oldest team; **v0.13.0 removes it**, so
- upgrade straight from v0.11.x to v0.13.0 only after the senders are moved.
-- **The schedule endpoints moved** under `/api/teams/{teamID}/schedule`, and
- editing the rota is now an owner's job. `GET /api/schedule/current` stayed
- where it was but now returns an **array** — one entry per team with somebody
- on call — instead of a single object or a 404. This is a breaking API change
- for anything that reads it, terdut-tui included.
-- **Uniqueness is per team now.** Two teams can legitimately see the same alert
- fingerprint, the same Alertmanager groupKey, and put somebody on call on the
- same date.
-
-**Dead man's switches moved too.** `TERDUT_DEADMAN_MATCHERS`, `_TIMEOUT` and
-`_SEVERITY` are no longer the setting; they are the default each existing team
-is seeded with at startup, after which an owner manages them per team through
-`/api/teams/{teamID}/deadman/switches` and a redeploy never overwrites that.
-
-Nothing else about an incident changes, and incidents never move between teams:
-an alert belongs to whichever team's key it arrived on.
-
-## Upgrading to roles
-
-Before this release every authenticated caller could create and delete users,
-set anybody's password and mint anybody's API keys. That is now the
-administrator flag, and the migration **makes every existing user an
-administrator** — they already held those powers, so nobody's access changes on
-upgrade and demotion is a deliberate act afterwards. Promoting only the first
-user would have silently stripped the rest, and could leave an install whose
-only administrator is an account nobody has a password for.
-
-Users created after the upgrade are not administrators. Hand the flag out with:
-
-```bash
-curl -X PUT https://terdut.example.com/api/users/7/admin \
- -H "Authorization: Bearer $TERDUT_API_KEY" \
- -H 'Content-Type: application/json' \
- -d '{"is_admin": true}'
-```
-
-Nothing in the API changed shape, so terdut-tui needs no new version — but a
-non-administrator now gets `403` where a `200` used to come back.
-
-## Upgrading from SQLite
-
-Versions up to v0.10.2 stored everything in a SQLite file. From v0.11.1 the server needs
-`TERDUT_DB_DSN` and keeps nothing on disk.
-
-The copy was done by `scripts/sqlite-to-postgres.go`, which **was deleted in v0.13.0** along
-with the SQLite driver it was the last user of. It is still in the history — check out the
-`v0.12.0` tag to get it:
-
-```bash
-git show v0.12.0:scripts/sqlite-to-postgres.go > sqlite-to-postgres.go
-```
-
-The cutover is ordered, and the server must not be running while the copy happens: stop the
-old version, let the new binary build the schema against an empty Postgres, run the script
-with `-sqlite` and `-dsn`, then start the new version for good. On Kubernetes step three runs
-as a Job with the same image against the PVC before it is removed.
-
-The copy preserves every id, so incidents keep their numbers and the timeline, alert
-membership, outbox and ack tokens all still point where they did. It refuses a target that
-already has rows, so a second run cannot double-insert.
-
-## Upgrading to incidents
-
-The incidents release moves the workflow off alerts, which is a **breaking API
-change**. These endpoints are gone:
-
-| Removed | Replacement |
+| | |
|---|---|
-| `POST`/`DELETE` `/api/alerts/{id}/acknowledge` | `POST`/`DELETE` `/api/incidents/{id}/acknowledge` |
-| `POST`/`DELETE` `/api/alerts/{id}/archive` | `POST`/`DELETE` `/api/incidents/{id}/archive` (alert archiving is now sweeper-only) |
-| `GET`/`POST` `/api/alerts/{id}/comments` | `GET /api/incidents/{id}/timeline`, `POST /api/incidents/{id}/notes` |
-| `DELETE /api/alerts/{id}/comments/{commentID}` | `DELETE /api/incidents/{id}/notes/{eventID}` |
+| [Deployment](./docs/deployment.md) | Docker, Helm chart, database, backups, the operator |
+| [Configuration](./docs/configuration.md) | Environment variables and settings |
+| [Alertmanager configuration](./docs/alertmanager.md) | Routes, keys, webhooks |
+| [Alerts and incidents](./docs/incidents.md) | Correlation, lifecycle, on-call assignment |
+| [Single sign-on](./docs/single-sign-on.md) | OIDC and the terminal device flow |
+| [API reference](./docs/api.md) | Every endpoint |
+| [Development and releasing](./docs/development.md) | Tests, CI gate, release pipeline |
-The alert object also drops `acknowledged_by_id`, `acknowledged_by` and
-`acknowledged_at`, and gains `incident_id`.
+## Related
-Migration `008_incidents.sql` runs automatically on start and preserves existing
-data: every alert gets a backfilled incident carrying its acknowledgement, and
-comments become timeline notes. Backfilled incidents have a `group_key` of
-`backfill:` — there is no historical `groupKey` to correlate on, so
-they are one-per-alert rather than grouped.
+- [terdut-tui](https://git.ryuvia.com/niklas/terdut-tui): a terminal client for the same server.
+- [terdut-operator](https://git.ryuvia.com/niklas/terdut-operator): a Kubernetes operator that
+ runs the server and manages teams, escalation, switches and alert sources as objects.
-Nothing about the two documented alert contracts changes: `received_at` is still
-advanced on every accepted webhook, and `resolution_source` still means what it
-did.
+## License
-## Upgrading to dead man's switches
-
-Dead man's switch handling is **on by default**, watching `alertname=Watchdog`
-with a 15 minute timeout. If you already route `Watchdog` to this server, the
-behaviour of that alert changes on upgrade, in both directions:
-
-- it stops opening incidents when it arrives, and
-- it starts opening one when it stops arriving.
-
-**Check your `repeat_interval` before upgrading.** The switch pages whenever a
-heartbeat has not been refreshed within `TERDUT_DEADMAN_TIMEOUT`, so a `Watchdog`
-route inheriting a 4h or 12h `repeat_interval` will page constantly against the
-15 minute default. Either give the heartbeat
-[its own fast route](#alertmanager-configuration) — the point of the feature — or
-set `TERDUT_DEADMAN_TIMEOUT` above your current `repeat_interval` until you have.
-`TERDUT_DEADMAN_TIMEOUT=0` turns the whole thing off.
-
-There is no migration and no schema change. An existing open incident from a
-`Watchdog` that arrived under the old behaviour is unaffected; resolve it by hand.
-
----
-
-## Development
-
-```bash
-make test-db # start a local Postgres for the tests (podman or docker)
-make test # run all tests
-go build ./... # compile all packages
-go run ./cmd/terdut # run locally (needs TERDUT_DB_DSN)
-```
-
-The tests need a real Postgres, because the server does — there is no in-memory Postgres the
-way there was an in-memory SQLite. `TERDUT_TEST_DSN` says where it is, `make test-db` starts
-one on port 5433 and prints the DSN, and `make test-db-stop` removes it. Each test gets its
-own schema on that server, so tests cannot see each other's rows. An unset `TERDUT_TEST_DSN`
-fails the suite rather than skipping it: a run that quietly tests nothing is worse than one
-that does not run.
-
-`make fmt lint test helm-lint` is the gate. It mirrors `.gitea/workflows/ci.yaml` step for
-step, so a green run here means a green pipeline — with one deliberate exception: `make test`
-adds `-race`, which CI does not. The sweeper, the notifier goroutine and the dead man's switch
-sweep all run concurrently against the same database, and a race between them would surface as
-a flaky incident in production rather than as a red build.
-
-The web UI lives in `internal/web/static/` as plain HTML, CSS and ES modules,
-embedded into the binary with `go:embed`. It has no build step and no npm, so
-editing a file and restarting the server is the whole loop.
-
-## Releasing
-
-```
-push or PR → ci.yaml gofmt, go vet, go test -race
- govulncheck, gitleaks
- helm lint + render
-push tag vX.Y.Z → release.yaml the same gate, then publish:
- git.ryuvia.com/niklas/terdut-server:vX.Y.Z
- oci://git.ryuvia.com/niklas/terdut-server X.Y.Z
- then trivy-scan the pushed image
-PR to Ryuvia/charts → bump the wrapper chart to X.Y.Z; on merge
- Flux reconciles and the release rolls out
-```
-
-Both artifacts go to the **personal** Gitea namespace rather than `ryuvia`, because Gitea
-scopes package visibility to the owner with no per-package override — so `ryuvia/*` is private
-because the org is. Publishing to `niklas` keeps them anonymously pullable, which is why no
-pull secret is needed in the cluster. Same reasoning, and the same choice, as riksdata and
-rd-web.
-
-Saying **"Release"** runs all three rows: the `release` skill commits, pushes, tags, waits for
-the pipeline, and opens the `Ryuvia/charts` PR, stopping before the merge. See
-`~/.claude/skills/release/`, or `.release.conf` here for this repo's part of it.
-
-The chart is published **only** from the tag, by the `chart` job. There used to be a second
-publisher on every `charts/**` push to main, and the two raced for the same chart version with
-different answers — chart 0.9.0 went out reading `appVersion: "latest"` that way. One
-publisher, triggered by the tag (`766f439`). The cost is that a chart-only change has no
-version of its own and rides the next app tag.
-
-Both workflows are thin drivers over the Makefile: `ci.yaml` runs `make fmt lint test` and
-`make helm-lint`, `release.yaml` adds `make binaries`, `make push`, `make helm-package` and
-`make helm-push`. That is deliberate — it is what makes a green local gate and a green
-pipeline the same code rather than two descriptions of it, and it is how riksdata and rd-web
-have always worked.
-
-`make push` builds and pushes in one step, unlike those two, because the image is
-`linux/amd64,linux/arm64` and buildx cannot load a multi-platform result into the local image
-store. `make build` stays single-platform and local-only. Both refuse `VERSION=dev`:
-publishing is one command, so it is also one command to run by accident. Publishing happens
-by pushing a tag.
-
-Two things the release process needs to know about this repo:
-
-- **The image scan runs after publishing**, like riksdata's and rd-web's: trivy cannot read
- a locally built image on this runner, so it pulls the pushed one. A red `scan-image` means
- do not bump the wrapper chart to that version — it cannot unpublish anything. The image is
- `FROM scratch`, so trivy sees exactly one target, the Go binary and its module graph.
-- **The wrapper chart's `values.yaml` has two `tag:` lines** — the app image and the python
- backup sidecar — so `chart-bump` is given `--image` to say which one moves. The sidecar is
- on its way out with SQLite: once the wrapper chart drops it and declares a `postgresql` CR
- instead, there is one `tag:` line again, and `--image` becomes belt and braces.
-
-The wrapper chart must have **its own `version:` bumped in the same commit**. Flux reconciles
-with `reconcileStrategy: ChartVersion`, so a chart whose version did not change produces no
-new artifact and the change is never deployed — with no error anywhere.
+See [LICENSE](./LICENSE).
diff --git a/SERVICE-ACCOUNTS.md b/SERVICE-ACCOUNTS.md
index 2d264c0..ff768a9 100644
--- a/SERVICE-ACCOUNTS.md
+++ b/SERVICE-ACCOUNTS.md
@@ -1,263 +1,60 @@
-# Service accounts: a scoped, non-human credential type
+# Service accounts
-This is a design note for a feature, not an implementation plan — it exists to
-propose the shape before writing code. It's raised directly by `terdut-operator`
-(a separate repo, no shared code — see its `DESIGN.md` §6, §9, §13), which needs
-a credential for unattended, repeatable API access and currently has no good one
-available. Anything automating terdut-server long-term (this operator, CI, future
-integrations) hits the same gap, so this is written as a general primitive, not
-operator-specific.
+A non-human credential for automation (terdut-operator, CI, scripts). It is not a
+`users` row: no password, no `is_admin`, no OIDC identity, so it can never be
+pulled into login or group sync, and it is never mistaken for a person in an
+audit trail. The bearer token has the same shape as an API key (SHA-256 hash
+stored, raw value shown once), prefixed `tdsa_`.
-## The problem
+## Scopes
-terdut-server has two credential types today, and neither fits "an unattended
-process that manages teams/schedules/policies on someone's behalf":
+- **instance** — acts as owner of every team's *configuration* (rename, OIDC
+ groups, escalation, dead man's switches, integrations, members, delete) and may
+ create teams. It is not a member of any team, so it reads no incidents or
+ queue. It is never an administrator: user management and
+ `/api/admin/settings` stay human-only.
+- **team** — acts as owner of exactly one team, through a single synthetic
+ membership. It may also mint another service account for its own team.
-- **User API keys** (`api_keys`, `internal/api/users.go`) are always tied to a
- real `users` row and carry that user's full rights — every team they're a
- member of, their admin flag if set. There's no `kind`/`service` marker
- distinguishing "a human's personal automation key" from "a login session," and
- no way to mint one scoped to less than the full user.
-- **Integration keys** (`integrations`, `internal/api/*teams*.go`) are team-scoped,
- but narrowly: they authenticate exactly one inbound Alertmanager webhook call
- (`POST /api/integrations/{key}/alertmanager`) and nothing else. They're not a
- general management-API credential and shouldn't become one — overloading a
- narrow, one-way ingestion credential with broad read/write access would weaken
- the one property that makes it safe to embed in an Alertmanager config today.
+An account has many keys, so rotating is "mint a new key, revoke the old one"
+without losing the account's identity or history.
-The result: any automation that needs to create teams, set escalation policies,
-manage dead-man switches, or rotate integration keys has to hold a real human
-admin's or team owner's API key. That key is exactly as powerful as that person
-logging in — full team access, and full instance access if they're an admin.
-`terdut-operator`'s design ran directly into this (its DESIGN.md §6): its
-described bootstrap/rotation flow assumed a repeatable, identity-scoped way to
-get a credential, and `/api/bootstrap`'s actual behavior (single-shot per
-install, gated on `COUNT(*) FROM users`, confirmed via `internal/api/users.go`
-and `charts/terdut-server/templates/bootstrap-job.yaml`) doesn't provide one —
-it mints exactly one founding admin, once, ever.
+## Endpoints
-## Goals
+- `POST /api/service-accounts` `{name, scope, team_id}` — returns the account and
+ its first key. An instance-scoped account is granted by a human administrator;
+ a team-scoped one by an administrator, that team's owner, or an instance-scoped
+ account.
+- `GET /api/service-accounts?name=` — look one up by name.
+- `POST /api/service-accounts/{id}/keys`, `DELETE .../keys/{keyID}` — mint or
+ revoke a key. An instance-scoped account may manage any team-scoped account's
+ keys, and any account may manage its own.
-- A credential type that isn't a human: doesn't touch OIDC group sync, login,
- session, or the `is_admin`/account-management semantics that come with a real
- `users` row.
-- Two scopes matching the two shapes automation actually needs: instance-wide
- (create/list teams — what a server-owning controller needs) and team-scoped
- (manage one team's escalation policy, dead-man switches, integrations,
- schedule, OIDC group bindings — what a per-team controller or integration
- needs).
-- Repeatable issuance and rotation — unlike `/api/bootstrap`, callable more than
- once, by anything that already holds admin rights, without destroying and
- recreating state to get a fresh credential.
-- Visibly distinct from a human in every place identity shows up (audit trails,
- timeline entries, UI attribution) — a service account acting on a team should
- never be indistinguishable from a person.
+## Seeding the operator's account
-## Non-goals
+`TERDUT_OPERATOR_KEY` (at least 32 characters) creates the instance-scoped account
+`terdut-operator` if missing and replaces its `seed` key with this value at every
+start (`internal/api/operator_key.go`). The deployer generates the key and
+nothing has to call `/api/bootstrap` for it; rotating is a restart with a new
+value. With `TERDUT_OPERATOR_MODE` on, configuration writes by humans are refused
+and a service account of either scope passes.
-- Not a general OAuth2/OIDC client-credentials flow — this is a bearer-token
- primitive matching the shape `api_keys` already uses (SHA-256 hash stored,
- raw key shown once at creation), not a new auth protocol.
-- Not replacing integration keys — those stay as the narrow, one-way webhook
- credential they are today.
-- Not modeling per-endpoint or per-verb permissions within a scope — `instance`
- and `team` are the only two scopes for now; finer-grained scoping is future
- work if a real need shows up.
+## How it is enforced
-## Proposed shape
+Every request resolves to one `Caller` (`internal/api/caller.go`): a human
+(session or API key) or a service account.
-### Schema
+- `Caller.IsAdmin()` is true only for a human administrator. `AdminOnly` and
+ `requireSelfOrAdmin` key on it alone; do not widen them — each time a gap came up
+ the fix was a narrower purpose-built capability instead.
+- `Caller.IsInstanceServiceAccount()` is true only for an instance-scoped account,
+ never for a human. `requireTeamOwner` and `callerOwnsTeam` admit it for any team.
+- `Caller.Role(teamID)`/`TeamIDs()` are a human's memberships or a team-scoped
+ account's single owner membership; instance scope has none.
+- `Caller.AsHuman()` is what a handler must call when it needs a real `user_id`;
+ handlers meant for people answer 403 to a service account instead of writing a
+ zero id.
-```sql
-CREATE TABLE service_accounts (
- id BIGSERIAL PRIMARY KEY,
- name TEXT NOT NULL UNIQUE, -- e.g. "terdut-operator"
- scope TEXT NOT NULL CHECK (scope IN ('instance', 'team')),
- team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE,
- -- team_id required iff scope = 'team'; NULL iff scope = 'instance'
- created_by BIGINT REFERENCES users(id),
- created_at TIMESTAMPTZ NOT NULL DEFAULT now()
-);
-
-CREATE TABLE service_account_keys (
- id BIGSERIAL PRIMARY KEY,
- service_account_id BIGINT NOT NULL REFERENCES service_accounts(id) ON DELETE CASCADE,
- key_hash TEXT NOT NULL UNIQUE,
- name TEXT NOT NULL, -- e.g. "initial", "2026-Q4-rotation"
- created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
- last_used_at TIMESTAMPTZ
-);
-```
-
-Deliberately not a `users` row: no `password_hash`, no `is_admin`, no
-`user_identities` linkage, so it's structurally impossible for a service account
-to be pulled into OIDC group sync or password login. Multiple keys per account
-(mirroring `api_keys`' existing one-user-many-keys shape) so rotation is "mint a
-new key, revoke the old one," not "recreate the account."
-
-### Endpoints
-
-- `POST /api/service-accounts` — instance-scope/admin-only. Body:
- `{"name": ..., "scope": "instance"|"team", "teamID": ... }` (teamID required
- iff scope=team, and caller must be that team's owner or a system admin).
- Returns the account plus its first raw key (shown once, same pattern as
- `POST /api/users/{id}/api-keys`). Safe to call again with the same `name` —
- see "idempotent lookup" below — unlike `/api/bootstrap`, which is inherently
- one-shot by design (it's answering "does any user exist yet," a question with
- no analogue once one already does).
-- `POST /api/service-accounts/{id}/keys` — mint an additional key on an existing
- account (self-service-equivalent: instance admin for `instance` scope, team
- owner or system admin for `team` scope). Enables rotation without recreating
- the account or losing its identity/audit history.
-- `DELETE /api/service-accounts/{id}/keys/{keyID}` — revoke one key, mirroring
- `DELETE /api/users/{id}/api-keys/{keyID}`.
-- `GET /api/service-accounts?name=` — look up an existing account by name.
- This is what turns "I tried to create my account and got a conflict" into a
- normal flow instead of an error: a controller that expects to have already
- registered itself calls this first, and only falls through to `POST` if
- nothing comes back.
-
-### Auth middleware
-
-**Revised** (this section originally described an aspiration that didn't
-match what shipped — `TEAM-LOOKUP.md` already caught one instance of that,
-and a fuller audit found three more; this is the corrected, as-built
-description, not the original proposal).
-
-`internal/api/middleware.go`'s dual resolution (`Authorization: Bearer` →
-`apiKeyUser()`, or session cookie → `sessionUser()`) and the service-account
-path (`serviceAccountFor()`) both resolve into one `Caller` type
-(`internal/api/caller.go`), not two parallel, un-unified context
-representations the way an earlier version of this server kept them. Every
-authorization predicate reads `Caller`'s methods:
-
-- `Caller.IsAdmin()` — true **only** for a human system administrator, never
- for a service account of either scope, under any circumstance. `AdminOnly`
- and `requireSelfOrAdmin` key on this alone — user management
- (`POST /api/users`, `PUT /api/users/{id}/admin`, etc.) and
- `GET/PUT /api/admin/settings` stay human-only, forever. The original text
- here claimed an instance-scoped service account satisfies `AdminOnly` "for
- team-creation/listing purposes" — that was never true of the shipped code
- (`TEAM-LOOKUP.md` caught the listing half; the creation half was always a
- separate, bespoke check in `handleCreateTeam`, not `AdminOnly` itself) and
- is not being made true now. Don't widen `AdminOnly`: every time this has
- come up, the fix has been a narrower, purpose-built capability instead
- (`?name=` lookups for teams and service accounts; now
- `terdut-operator`'s own invite-minting feature for the one real gap this
- boundary left — how a human ever gets a first login on a no-OIDC,
- operator-managed install. See the bottom of "What this unblocks.")
-- `Caller.IsInstanceServiceAccount()` — true only for an instance-scoped
- service account, never for a human (including a human admin).
- `handleCreateTeam` uses exactly this: a human creates a team by being a
- human (and becomes its owner); an instance-scoped service account creates
- one with no human owner at all. The two paths are not interchangeable, so
- this predicate deliberately does not also admit a human admin.
-- `Caller.Role(teamID)`/`TeamIDs()` — a human's real `team_members` rows, or
- a team-scoped service account's single synthetic owner membership
- (`serveAsServiceAccount`). This is what makes `requireTeamMember`/
- `requireTeamOwner` treat a team-scoped service account as owner-equivalent
- for that one team, with no separate branch needed in either function.
-- `Caller.ServiceAccountID()` — used by `OperatorModeBlock` ("any service
- account passes") and by `callerMayManageServiceAccount`'s self-rotation
- check.
-- `Caller.AsHuman()` — the accessor every handler that needs a real
- `user_id` to act on behalf of must call and check, instead of reading a
- user off context unconditionally. Before the `Caller` type existed, four
- handlers did the latter and silently misbehaved for a service-account
- caller: `handleMe` and `handleTestNotification` 500'd (a zero-value user id
- that matches no row), `handleDismissOnboarding` silently no-op'd (`UPDATE
- ... WHERE id = 0` affects nothing, still returns 204), and
- `handleCreateInvite` wrote that same zero value into `invites.created_by`
- — a real foreign-key violation, not just a wrong answer, since that column
- is nullable but was never passed as `nil`. All four now call `AsHuman()`
- and return an explicit 403 ("this endpoint is for human accounts only")
- or, for the invite case, leave `created_by` `NULL` the same way
- `handleCreateServiceAccount` already did for the analogous situation.
-
-**Team scope is owner-equivalent for every `requireTeamOwner` endpoint,
-membership and invites included — by design, not by an unclosed gap.** An
-earlier version of this document flagged this as "acknowledged rather than
-closed," kept in check only by the social convention that nobody *builds*
-automation against those two routes. That convention is retired:
-`terdut-operator`'s `TerdutTeam` controller now mints and revokes its own
-team's invite link through exactly this capability (its existing
-team-scoped credential, `POST`/`DELETE /api/teams/{teamID}/invites`), which
-is the real fix for the human-onboarding gap below — not a narrower
-carve-out of this capability. `service_accounts_test.go`'s
-`TestServiceAccount_TeamScopeManagesItsOwnInvites` pins it.
-
-**A team-scoped account can also mint another service account scoped to its
-own team** (`handleCreateServiceAccount`'s `callerOwnsTeam` branch, which a
-team-scoped caller already satisfies for its own team via the synthetic
-membership above). Kept, not restricted, for the same reason: a team-scoped
-credential is that team's owner's reach, full stop — carving this one
-capability out while leaving membership/invites alone would be an arbitrary
-asymmetry. Pinned by
-`TestServiceAccount_TeamScopeCanMintAnotherAccountForItsOwnTeam`.
-
-**`callerMayManageServiceAccount` gained the one load-bearing fix this
-redesign exists for:** an instance-scoped service account may manage
-(mint/revoke a key on) *any* team-scoped account, not only one admin, that
-team's human owner, or the account itself. `handleCreateServiceAccount`
-already let an instance-scoped caller *create* a team-scoped account for
-any team; this closes the gap where adopting or rotating one it didn't just
-create in the same call — exactly `terdut-operator`'s documented
-adopt-on-409 crash-window recovery (its own `DESIGN.md` §5) — 403'd forever
-instead of succeeding (`terdut-operator#3`). Pinned by
-`TestServiceAccount_InstanceScopeAdoptsAnExistingTeamScopedAccountsKey`.
-
-Anywhere identity is recorded for a human (incident timeline
-`acknowledged_by`/`assigned_to`, audit-relevant fields), a service-account
-caller is still coerced into a bare `user_id` of `0` today — `Caller`'s new
-`Identity()` accessor exists for exactly this follow-up, but wiring it in
-needs a schema migration (an actor-attribution column distinct from
-`user_id`) and is deliberately out of scope here. Tracked separately, not by
-this document.
-
-## What this unblocks
-
-Directly resolves `terdut-operator` DESIGN.md §6's two broken assumptions:
-1. **Bootstrap becomes single-purpose again.** `/api/bootstrap` mints exactly
- the founding human admin, once. The operator's actual first-reconcile flow:
- call `/api/bootstrap` only on a genuinely empty install; otherwise (or
- immediately after, if it won the bootstrap race) call
- `GET /api/service-accounts?name=terdut-operator`, and `POST` one if it
- doesn't exist yet. From then on the operator never touches `/api/bootstrap`
- again.
-2. **Rotation becomes real.** `POST /api/service-accounts/{id}/keys` + revoke the
- old one — no destructive DB-level workaround, no re-triggering a single-shot
- endpoint that can't fire twice.
-3. **Cross-namespace credential mirroring is no longer needed at all.**
- `terdut-operator`'s current design holds every credential — instance- and
- team-scoped alike — privately in the operator's own namespace, never in
- the namespace of the CR each one authenticates for; reconciliation happens
- entirely inside the operator's controller loop, so no CR owner ever needs
- read access to a terdut-server credential regardless of same- or
- cross-namespace `serverRef`. Team scoping is still what bounds the blast
- radius of any individual credential: a leaked team-scoped key exposes
- exactly one team's resources, never the whole server, which is what makes
- holding many credentials in one place (the operator's namespace) an
- acceptable trade rather than reintroducing the mirrored design's
- server-admin-equivalent-everywhere problem.
-4. **A human can get a first login on a no-OIDC, operator-managed install —
- without ever touching `AdminOnly` or `/api/admin/settings`.** This was
- filed as `terdut-server#23` ("no API path to create a human login after
- bootstrap") and diagnosed, at the time, as this server needing to let a
- service account through `AdminOnly`. It doesn't: the fix lives entirely
- in `terdut-operator`, because a team-scoped credential was *already*
- owner-equivalent for `POST /api/teams/{teamID}/invites`, and invite
- redemption (`POST /api/signup` with an `invite` token) bypasses
- `signup_mode` entirely — `terdut-operator` just never grew a feature to
- use either fact. Its `TerdutTeam` controller now mints and surfaces one
- via its own existing team-scoped credential (`spec.invite`,
- `status.inviteSecretRef`, see that repo's own docs), so a human joins a
- CRD-managed team by a real invite link, the same way anyone else would.
- `terdut-server#23` is closed with this note once that feature ships — its
- named routes stay human-only, correctly, not a gap.
-
-## Suggested sequencing
-
-Land this before `terdut-operator` implements any bootstrap/credential-handling
-code — that code would otherwise be written against the current one-shot,
-user-only credential model as a known-temporary workaround, which is wasted
-effort on a repo that currently has zero implementation to begin with.
+Where a service account acts on an incident (acknowledge, resolve), the
+timeline and `acknowledged_by` record it through parallel `*_service_account_id`
+columns, never as a user.
diff --git a/TEAM-LOOKUP.md b/TEAM-LOOKUP.md
deleted file mode 100644
index 67419cd..0000000
--- a/TEAM-LOOKUP.md
+++ /dev/null
@@ -1,73 +0,0 @@
-# Team lookup for service accounts: closing terdut-operator's create-path crash window
-
-This is a design note for a feature, not an implementation plan — same posture as
-`SERVICE-ACCOUNTS.md`, and raised for the same reason: `terdut-operator`'s `TerdutTeam`
-controller (ROADMAP.md Stage 2, a separate repo, no shared code) hit a gap this server has
-no answer for yet.
-
-## The problem
-
-`POST /api/teams` (`handleCreateTeam`, confirmed against `internal/api/teams.go`) lets an
-instance-scoped service account create a team — it has its own explicit
-`isInstanceServiceAccount(...)` branch alongside the human-user path, not gated by
-`AdminOnly`. If that call succeeds server-side but the caller (`TerdutTeam`'s controller)
-crashes before persisting the resulting team ID locally, a retry's `POST` 409s on the name's
-unique constraint (confirmed: the `isUniqueViolation` branch in the same handler).
-
-Recovering from that 409 means looking the team up by name, and nothing today permits that
-for a service account:
-
-- `GET /api/teams` (`handleListTeams`) answers "what teams does the *caller* belong to", via
- a `team_members` join keyed on `userFromContext`'s `caller.ID` — confirmed against source.
- A service account is never a member of anything, so this always returns empty for one,
- regardless of what exists.
-- `GET /api/admin/teams` (`handleAdminListTeams`) is gated by `AdminOnly`, and `AdminOnly`'s
- actual code (`internal/api/middleware.go`) checks only `userFromContext(...).IsAdmin` — no
- branch for a service account at all, confirmed against source. This contradicts
- `SERVICE-ACCOUNTS.md`'s own text, which claims "an instance-scoped [service account
- satisfies] `AdminOnly` for team-creation/listing purposes" — that claim doesn't match this
- endpoint's actual, shipped code. (Team *creation* is fine: `handleCreateTeam` isn't behind
- `AdminOnly` at all, it has its own check. Only the listing half of that sentence is wrong.)
-
-This is exactly the shape of gap `SERVICE-ACCOUNTS.md`'s own `GET /api/service-accounts?name=`
-closed for service accounts themselves (confirmed: that endpoint's own comment —
-"the name lookup is open to any authenticated caller... what lets a service account find its
-own account on the 403 that follows a second POST"). Teams never got the equivalent, because
-nothing needed it until an operator started creating them unattended.
-
-## Goals
-
-- A service-account-accessible way to look up one team by exact name, mirroring
- `GET /api/service-accounts?name=` as closely as possible — same shape, same reasoning,
- same low sensitivity of what it discloses.
-- No change to today's behavior for an empty/no-name request.
-
-## Proposed shape
-
-Extend `GET /api/teams` itself, the same way `handleListServiceAccounts` already branches on
-a `?name=` query param, rather than adding a new route:
-
-- `name` unset (today's behavior, unchanged): the caller's own teams, via `team_members`.
-- `name=` set: look up that one team by exact name — a one-or-zero-length array, not
- an error on no match, mirroring `GET /api/service-accounts?name=`'s own response shape and
- status codes exactly. Deliberately **not** gated by `isInstanceServiceAccount` or
- `AdminOnly`: a human caller who's already a member sees this same information in their own
- team list regardless, and a non-member learning only that a name is taken — not who's in
- the team, not any of its data — is the same low-sensitivity disclosure
- `GET /api/service-accounts?name=` already accepts for service-account names.
-
-## What this unblocks
-
-Directly resolves the crash-window gap in `terdut-operator`'s `TerdutTeam` controller: on a
-409 from `POST /api/teams`, `GET /api/teams?name=` — authenticated with the
-same instance-scoped credential that just got the 409 — finds the id, and the controller
-proceeds as if its own create had returned it directly. The same adopt-on-409 pattern already
-proven for service accounts (that repo's `DESIGN.md` §6 point 1, §5's general rule), not a
-new one.
-
-## Suggested sequencing
-
-Land this before `TerdutTeam`'s create path is implemented — the same reasoning
-`SERVICE-ACCOUNTS.md` gave for its own sequencing: writing that code against today's gap as a
-"known-temporary workaround" is wasted effort when the fix is this small and this
-well-precedented.
diff --git a/charts/terdut-server/templates/deployment.yaml b/charts/terdut-server/templates/deployment.yaml
index 15557d6..49acee1 100644
--- a/charts/terdut-server/templates/deployment.yaml
+++ b/charts/terdut-server/templates/deployment.yaml
@@ -95,12 +95,6 @@ spec:
value: "{{ .Values.sweeper.staleAfter }}"
- name: TERDUT_ARCHIVE_AFTER
value: "{{ .Values.sweeper.archiveAfter }}"
- - name: TERDUT_DEADMAN_MATCHERS
- value: "{{ .Values.deadman.matchers }}"
- - name: TERDUT_DEADMAN_TIMEOUT
- value: "{{ .Values.deadman.timeout }}"
- - name: TERDUT_DEADMAN_SEVERITY
- value: "{{ .Values.deadman.severity }}"
{{- if .Values.notify.ntfyUrl }}
- name: TERDUT_NTFY_URL
value: "{{ .Values.notify.ntfyUrl }}"
@@ -122,6 +116,8 @@ spec:
value: "{{ .Values.notify.publicUrl | default (printf "https://%s" .Values.networking.hostname) }}"
- name: TERDUT_PASSWORD_LOGIN
value: {{ .Values.passwordLogin | quote }}
+ - name: TERDUT_TRUSTED_PROXIES
+ value: {{ .Values.trustedProxies | quote }}
- name: TERDUT_OPERATOR_MODE
value: {{ .Values.operatorMode | quote }}
{{- if .Values.oidc.enabled }}
diff --git a/charts/terdut-server/values.yaml b/charts/terdut-server/values.yaml
index 64d3de4..e084c23 100644
--- a/charts/terdut-server/values.yaml
+++ b/charts/terdut-server/values.yaml
@@ -72,43 +72,10 @@ sweeper:
# How long a resolved alert stays in the default list before auto-archiving.
archiveAfter: 168h
-# Alerts treated as dead man's switches: receiving one opens no incident, and
-# the absence of one does. The Watchdog alert kube-prometheus-stack ships is
-# exactly this — an always-firing alert whose only value is something noticing
-# when it stops.
-deadman:
- # Which alerts to treat as heartbeats. ";" separates matchers, "," separates
- # the label conditions within one, "=" is exact equality. Every matcher must
- # name an alertname:
- # alertname=Watchdog,cluster=prod; alertname=EdgeHeartbeat
- # Each distinct label set is watched independently, so two clusters sending
- # the same alertname are two switches and a live one cannot mask a dead one.
- matchers: "alertname=Watchdog"
- # How long a heartbeat may go unheard before its switch is declared dead.
- #
- # This must be SHORTER than the Alertmanager repeat_interval of the route
- # carrying the heartbeat — the opposite of sweeper.staleAfter. The default
- # repeat_interval of 4h (12h in many setups) makes for a useless dead man's
- # switch, so give the heartbeat a route of its own:
- #
- # - matchers: [ 'alertname = "Watchdog"' ]
- # receiver: terdut
- # group_wait: 0s
- # group_interval: 1m
- # repeat_interval: 1m
- #
- # That delivers every 2m rather than every 1m: a group is only reconsidered
- # each group_interval, and at exactly one elapsed interval repeat_interval has
- # not quite passed, so equal values give 2x. Fine against 15m; use
- # group_interval: 30s if you want a true 1m.
- #
- # Set to 0 to disable dead man's switch handling entirely.
- timeout: 15m
- # Severity a dead man's switch incident opens at. These incidents have no
- # member alerts to derive one from, and the heartbeat's own severity label is
- # meaningless — Watchdog ships as "none". Only "critical" maps to the ntfy
- # priority that overrides a phone's quiet hours.
- severity: critical
+# How many reverse proxies in front of the server append to X-Forwarded-For.
+# The per-address login/sign-up rate limits take the client address that many
+# entries from the right. 0 ignores the header.
+trustedProxies: 1
notify:
# ntfy server that push notifications are published to, e.g.
@@ -190,10 +157,8 @@ oidc:
# Hard ceiling on a session made by a single sign-on login.
sessionMaxAge: 12h
-# Backups are no longer this chart's business. The SQLite database lived on a PVC
-# beside the app, so it needed a sidecar with a sqlite3 module for k8up to exec a
-# dump in; Postgres is backed up where it runs, through a k8up.io/backupcommand
-# pg_dump annotation on the database pod itself.
+# Backups are not this chart's business: Postgres is backed up where it runs,
+# through a k8up.io/backupcommand pg_dump annotation on the database pod itself.
bootstrap:
enabled: true
diff --git a/cmd/terdut/main.go b/cmd/terdut/main.go
index 2997f5c..08d8661 100644
--- a/cmd/terdut/main.go
+++ b/cmd/terdut/main.go
@@ -39,21 +39,16 @@ func main() {
RepeatEvery: cfg.NotifyRepeat,
}
- // Dead man's switches live per team now. The environment variables are the
- // defaults a team starts from: every team without a configuration of its
- // own gets one from them here, and an owner's later edit is never
- // overwritten by a redeploy.
- deadman := api.ParseDeadmanConfig(cfg.DeadmanMatchers, cfg.DeadmanTimeout, cfg.DeadmanSeverity)
- if err := api.SeedDeadmanConfigs(context.Background(), database, deadman); err != nil {
- log.Fatalf("seed dead man's switch defaults: %v", err)
- }
-
// The behaviour knobs move into the database on first start, after which an
// administrator owns them and a redeploy leaves them alone.
if err := api.SeedSettings(context.Background(), database, cfg); err != nil {
log.Fatalf("seed settings: %v", err)
}
+ if err := api.SeedOperatorKey(context.Background(), database, cfg.OperatorKey); err != nil {
+ log.Fatalf("%v", err)
+ }
+
router := api.NewRouter(database, notify, cfg, version)
srv := &http.Server{
diff --git a/docs/README.md b/docs/README.md
new file mode 100644
index 0000000..c59199e
--- /dev/null
+++ b/docs/README.md
@@ -0,0 +1,21 @@
+# Terminal Duty documentation
+
+The [README](../README.md) is the short tour. These pages hold the detail.
+
+**Running it**
+- [Deployment](./deployment.md): Docker, the Helm chart, the database, backups, and the operator.
+- [Configuration](./configuration.md): environment variables and settings.
+- [Single sign-on](./single-sign-on.md): OIDC, group mapping, the terminal device flow.
+
+**Using it**
+- [The web UI](./web-ui.md): sessions, the Team and Admin tabs.
+- [Alertmanager configuration](./alertmanager.md): routes, integration keys and webhooks.
+- [Alerts and incidents](./incidents.md): correlation, lifecycle, on-call assignment, stale-alert expiry.
+- [Push notifications](./notifications.md): ntfy pages and acknowledging from them.
+- [Escalation](./escalation.md): ladders, repeats and the fallback topic.
+- [Dead man's switches](./dead-mans-switch.md): noticing that alerts stopped arriving.
+
+**Integrating and contributing**
+- [API reference](./api.md): every endpoint, authentication and error shape.
+- [Service accounts](../SERVICE-ACCOUNTS.md): non-human credentials for automation.
+- [Development and releasing](./development.md): tests, the CI gate, the release pipeline.
diff --git a/docs/alertmanager.md b/docs/alertmanager.md
new file mode 100644
index 0000000..a153e72
--- /dev/null
+++ b/docs/alertmanager.md
@@ -0,0 +1,62 @@
+# Alertmanager configuration
+
+_Pointing Alertmanager at the server._ Back to the [README](../README.md) and the [documentation index](./README.md).
+
+Alerts arrive on a team's **integration key**, which says both that the sender
+may post and which team the alerts belong to. Mint one as an owner of the team:
+
+```bash
+curl -X POST https://terdut.example.com/api/teams/1/integrations \
+ -H "Authorization: Bearer $TERDUT_API_KEY" \
+ -H 'Content-Type: application/json' \
+ -d '{"name":"prod alertmanager"}'
+```
+
+The response carries the key and the full URL **once**; only a SHA-256 hash is
+stored. Put it in your `alertmanager.yml`:
+
+```yaml
+receivers:
+ - name: terdut
+ webhook_configs:
+ - url: http://terdut-server:8080/api/integrations//alertmanager
+ send_resolved: true
+
+route:
+ receiver: terdut
+```
+
+The whole URL is a credential, so treat it like one. Alertmanager 0.26 and
+later can read it from a file with `url_file:` instead, which keeps it out of
+your configuration repository:
+
+```yaml
+ - url_file: /etc/alertmanager/secrets/terdut-webhook-url/url
+ send_resolved: true
+```
+
+The webhook endpoint requires no authentication.
+
+If you use the [dead man's switch](./dead-mans-switch.md) — and the default configuration does — give
+the heartbeat a route of its own, because the deadline is only as tight as the interval feeding it:
+
+```yaml
+route:
+ receiver: terdut
+ repeat_interval: 4h
+ routes:
+ - matchers: [ 'alertname = "Watchdog"' ]
+ receiver: terdut
+ group_wait: 0s
+ group_interval: 1m
+ repeat_interval: 1m
+```
+
+That delivers a heartbeat every **2 minutes**, not every minute. Alertmanager only reconsiders a
+group every `group_interval`, and at exactly one elapsed interval `repeat_interval` has not *quite*
+passed, so the send slips to the next tick — equal values give 2×. Two minutes against the 15 minute
+default is seven heartbeats per window, which is the point; use `group_interval: 30s` if you want
+the numbers to mean what they say.
+
+kube-prometheus-stack users get the `Watchdog` alert (`expr: vector(1)`) for free; it just needs
+routing to terdut rather than to `null`.
diff --git a/docs/api.md b/docs/api.md
new file mode 100644
index 0000000..6af26a0
--- /dev/null
+++ b/docs/api.md
@@ -0,0 +1,436 @@
+# API reference
+
+_The REST API._ Back to the [README](../README.md) and the [documentation index](./README.md).
+
+## Authentication
+
+All endpoints except `/api/bootstrap`, `/api/integrations/{key}/alertmanager`,
+`/api/notify/ack/{token}`, `/api/login`, `/api/logout`, `/api/auth/config`,
+`/api/version`, `/api/oidc/login`, `/api/oidc/callback`, `/api/oidc/device` and
+`/api/oidc/device/token` require either an API key:
+
+```
+Authorization: Bearer
+```
+
+or the web UI's session cookie. A request that carries an `Authorization` header
+is judged on that header alone.
+
+Two kinds of user exist. An **administrator** manages accounts: creating and
+deleting users, setting anybody's password, minting keys for anybody, and
+granting the flag itself. Everybody else works incidents — acknowledging,
+assigning, snoozing, resolving, noting — and manages their own account and
+nobody else's. An API key carries exactly the rights of the user it belongs to.
+
+A third principal, the **service account**, exists for automation (a
+Kubernetes operator, most likely) that needs to manage teams, escalation
+policies, dead man's switches and integrations without impersonating a human.
+It is not a user — it never signs in, never appears in a team's member list,
+and never holds the administrator flag — and its key is prefixed `tdsa_` so it
+reads as one at a glance in a log line. See [Service accounts](#service-accounts).
+
+**Getting an account.** The first one comes from `/api/bootstrap`. After that
+it depends on `signup_mode`, an administrator setting:
+
+- `invite_only` (the default) — a team owner mints a link with
+ `POST /api/teams/{teamID}/invites`, and the person who opens it picks a
+ username and password and lands in that team with the role the link carries.
+ Links are single-use unless told otherwise, expire after seven days, and can
+ be revoked before that.
+- `open` — anybody who can reach the server can create an account, and must
+ name a team, which they then own.
+
+Invites are **links, not email**: this server has no SMTP, and adding it to send
+one message would be a subsystem to run, secure and monitor. Send the link
+however you already talk to the person.
+
+A domain-restricted third mode was considered and dropped: with no email there
+is nothing to verify an address against, so it would only check the domain of a
+string somebody typed.
+
+The first user, from `/api/bootstrap`, is an administrator. Users created
+afterwards are not, until an administrator says so. An install always keeps at
+least one: the last administrator can be neither deleted nor demoted, and
+nobody can delete or demote themselves.
+
+Endpoints that require the flag answer `403` with
+`{"error":"administrator access required"}`.
+
+**Teams** are the unit of tenancy, and are a separate axis from the administrator
+flag. A team owns its incidents, alerts, schedule and integrations, and a user
+sees exactly the teams they belong to. Within a team an **owner** configures it
+(schedule, integrations, membership) and a **member** works its incidents.
+
+An administrator crosses that line in one direction only. They **configure any
+team** without being in it — every owner-only endpoint accepts the flag, because
+otherwise a team whose last owner left could never be repaired. They do **not
+read any team**: the queue, the alerts and the incidents are filtered by real
+membership, so an administrator sees a team's work only by joining it, which is
+a membership change and shows up as one. Administration is about accounts and
+the shape of a team, not about reading other people's incidents.
+
+Anything belonging to a team you are not in answers `404`, not `403`: whether an
+incident exists is itself something only its team should learn.
+
+**Operator mode** (`TERDUT_OPERATOR_MODE`, see [Configuration](./configuration.md#configuration))
+declares this install gitops-managed. When it is on, a session or a user's own
+API key gets `403 {"error": "...", "reason": "operator_managed"}` on every
+write this page marks **owner**-gated under Teams below (creating, renaming
+or deleting a team; its OIDC group binding; its escalation ladder; its dead
+man's switches; its integrations) — a service account's writes are unaffected.
+Team membership and invites are deliberately excluded: they are never
+gitops-managed, in operator mode or out of it. `GET /api/auth/config` reports
+`operator_mode` so a client can grey those sections out before a write is ever
+attempted.
+
+| Method | Path | Description |
+|---|---|---|
+| `GET` | `/api/auth/config` | How to sign in: `{"password_login", "oidc": {"enabled","name"}, "device_login", "operator_mode"}`. No session needed |
+| `GET` | `/api/version` | `{"version"}` — this build's version string. No session needed, the same as `/healthz` |
+| `POST` | `/api/login` | `{"username","password"}` → sets the session cookie, returns `{user, has_password}`. `429` after too many failures; `403` when `TERDUT_PASSWORD_LOGIN=false` |
+| `GET` | `/api/oidc/login` | Starts a single sign-on sign-in: redirects the browser to the provider. `?next=/path` is where to land afterwards; only a path on this server is honoured. Only exists when SSO is configured |
+| `POST` | `/api/oidc/device` | Starts a device login: returns `{device_code, user_code, verification_url, interval, expires_in}`. Only exists when SSO is configured |
+| `POST` | `/api/oidc/device/token` | `{"device_code"}` → `202 {"status":"pending"}`, then `200` with the session cookie once approved (once only). `410` with `{"error":"expired"}` or `{"error":"denied"}`; `429 {"error":"slow_down"}` if polled faster than `interval` |
+| `POST` | `/api/oidc/device/approve` | **session** — `{"user_code"}`. Approves a pending device login as the caller. `403` for an API key; `404` for an unknown, expired or already decided code |
+| `POST` | `/api/oidc/device/deny` | **session** — `{"user_code"}`. Refuses it |
+| `GET` | `/api/oidc/callback` | Where the provider sends the browser back. Sets the session cookie and redirects to `/`, or to `/?sso_error=` — one of `denied`, `expired`, `failed`, `unavailable`, `not_allowed`, `no_email`, `email_conflict`, `disabled`, `not_bootstrapped` (no user exists on this install yet — sign in again once something has called `/api/bootstrap`) |
+| `POST` | `/api/logout` | Ends the session and clears the cookie |
+| `GET` | `/api/me` | The caller: `{user, has_password}` |
+
+## Users
+
+| Method | Path | Description |
+|---|---|---|
+**admin** marks an endpoint that requires the administrator flag; **self or
+admin** marks one you may use on your own account and an administrator may use
+on anybody's.
+
+| Method | Path | Who | Description |
+|---|---|---|---|
+| `GET` | `/api/signup` | — | Whether sign-up is open, and whether `?invite=` is usable. No session needed: the caller has no account yet |
+| `POST` | `/api/signup` | — | Create an account `{"username","email","password","invite"?,"team_name"?}` and sign in. `403` without a usable invite when the mode is invite-only |
+| `POST` | `/api/bootstrap` | — | Create first user + API key `{"username","email","password"?}` (only works on empty DB). The user is an administrator |
+| `GET` | `/api/users` | any | List users. Open to everybody: the queue's assignment control and the schedule both have to name people |
+| `GET` | `/api/users/{id}/teams` | self or admin | The teams that user is in, each with their role. `/api/teams` is always about the caller; this one answers it about somebody else, for the admin page's per-user view. `404` for a user who does not exist, so "no teams" and "no such person" are distinguishable |
+| `POST` | `/api/users` | **admin** | Create user `{"username","email"}`. Not an administrator |
+| `DELETE` | `/api/users/{id}` | **admin** | Delete user (cascades to keys). `409` for yourself or the last administrator |
+| `PUT` | `/api/users/{id}/admin` | **admin** | Grant or revoke the administrator flag `{"is_admin"}`. `409` for yourself, the last administrator, or an administrator granted by single sign-on |
+| `PUT` | `/api/users/{id}/disabled` | **admin** | Take an account out of use, or put it back `{"disabled"}`. `409` for yourself or the last administrator |
+| `PUT` | `/api/users/{id}/notify` | self or admin | Set push notification target `{"ntfy_topic"}` — empty string clears it |
+| `PUT` | `/api/users/{id}/password` | self or admin | Set web UI password `{"password","current_password"}`. `current_password` is required only when changing your own existing password. Ends the user's other sessions |
+| `POST` | `/api/users/{id}/api-keys` | self or admin | Issue API key `{"name"}` — key shown once |
+| `DELETE` | `/api/users/{id}/api-keys/{keyID}` | self or admin | Revoke API key |
+
+## Administration
+
+| Method | Path | Who | Description |
+|---|---|---|---|
+| `GET` | `/api/admin/teams` | **admin** | Every team on the server, with its member and open-incident counts. `/api/teams` answers "what am I in"; this answers "what is there" |
+| `GET` | `/api/admin/teams/{teamID}` | **admin** | One team and who is in it: `{"team", "members"}`. `404` for a team that does not exist. `GET /api/teams/{teamID}/members` is **member**-only and still `404`s an administrator from outside the team — reading a team's shape and reading its work are different questions, so they are different endpoints |
+| `GET` | `/api/admin/settings` | **admin** | The editable settings with their bounds, plus the environment-configured ones, read-only. Never credentials |
+| `PUT` | `/api/admin/settings` | **admin** | Change one or more `{"key": seconds}`, or `{"signup_mode": "open"\|"invite_only"}`. `400` for an unknown key or a value outside its bounds |
+
+## Service accounts
+
+A service account is a scoped, non-human credential for automation — not a
+`users` row, so it never signs in, is never a team member, and never carries
+the administrator flag. Two scopes:
+
+- **instance** — the same reach system administration has over teams: create
+ one, and mint a **team**-scoped account against any of them. There is no
+ cap on how many instance-scoped accounts exist, but ordinarily there is one,
+ belonging to whatever is provisioning this install end to end.
+- **team** — owner-equivalent for that one team, and nothing else: every
+ **owner**-gated endpoint under [Teams](#teams), membership and invites
+ included. Nothing narrower is enforced server-side; what actually keeps
+ membership out of automation's hands is that no operator built against this
+ scope should ever call those two endpoints — see
+ [operator mode](#authentication) and [`SERVICE-ACCOUNTS.md`](../SERVICE-ACCOUNTS.md)'s note on this.
+
+A key is shown once, at creation or rotation, and only its hash is stored —
+the same handling as a user's API key. Losing it means minting a new one;
+there is no way to recover a raw key from the server.
+
+| Method | Path | Who | Description |
+|---|---|---|---|
+| `GET` | `/api/service-accounts` | **admin** | Every service account. Pass `?name=` instead to look one up by its exact name — open to **any** authenticated caller (human or service account), since it returns no key material and is how an account finds its own id |
+| `POST` | `/api/service-accounts` | owner\* | Create one and mint its first key `{"name","scope","team_id"?}` (`team_id` required for `scope:"team"`, absent for `scope:"instance"`). Returns `{"service_account", "key"}` — `key.key` shown once |
+| `POST` | `/api/service-accounts/{id}/keys` | owner\* | Mint an additional key `{"name"}` — rotation without recreating the account. Shown once |
+| `DELETE` | `/api/service-accounts/{id}/keys/{keyID}` | owner\* | Revoke one key |
+
+\* For an **instance**-scoped account: a system administrator only. For a
+**team**-scoped account: a system administrator, that team's own human owner,
+an instance-scoped service account (minting a narrower credential for a team
+it just created), or — for the two key endpoints only — the account rotating
+or revoking its own key, which is not a privilege escalation, the same
+reasoning a user's own API keys rest on.
+
+## Alert ingestion
+
+Alerts arrive on a team's integration key. The key is both the credential and the
+routing: it says that the sender may post, and which team the alerts belong to.
+Create one with `POST /api/teams/{teamID}/integrations`, which returns the key
+and the full URL once and stores only a SHA-256 hash.
+
+| Method | Path | Description |
+|---|---|---|
+| `POST` | `/api/integrations/{key}/alertmanager` | Alertmanager v4 webhook receiver for the key's team. `401` for an unknown key |
+
+This is the only way in. The pre-teams `POST /api/alertmanager/webhook` took no
+credential at all — anything able to reach the port could open an incident —
+and was removed in v0.13.0 once senders had moved onto keys.
+
+## Teams
+
+**owner** below means an owner of that team, a system administrator (who
+passes every one of these without being a member), or that team's own
+team-scoped [service account](#service-accounts) — including membership and
+invites, technically, though no automation this scope was designed for
+(a Kubernetes operator's CRDs, see [`SERVICE-ACCOUNTS.md`](../SERVICE-ACCOUNTS.md)) ever models team
+membership or would call those two. See [Authentication](#authentication).
+**member** means membership and nothing else: an administrator who is not in
+the team gets the same `404` as anybody else.
+
+| Method | Path | Who | Description |
+|---|---|---|---|
+| `GET` | `/api/teams` | any | The caller's own teams, each with their role |
+| `POST` | `/api/teams` | any | Create a team `{"name"}`; a human creator becomes its first owner. An instance-scoped [service account](#service-accounts) may also create one, and it gets no owner at all — expected for a team an operator is about to hand a team-scoped credential to, not an orphaned team a human made |
+| `PUT` | `/api/teams/{teamID}` | **owner** | Rename it `{"name"}`. `409` if the name is taken |
+| `DELETE` | `/api/teams/{teamID}` | **owner** | Delete a team and everything under it. `409` while it has open incidents |
+| `GET` | `/api/teams/{teamID}/members` | member | Who is in the team, with `status` (`oncall` if the rota has them today, `unpageable` when a page to them would go nowhere — even if they are on call — else `reachable`), `on_call`, `next_shift` (first rota day after today), `pageable` and `problem` (`has no ntfy topic` / `account is disabled`; never the topic itself) and `last_active_at` (their newest session or API-key use). Every member sees the same list |
+| `POST` | `/api/teams/{teamID}/members` | **owner** | Add a member, or change their role `{"user_id","role"}`. `409` when it would demote the last owner, or the membership is managed by single sign-on |
+| `DELETE` | `/api/teams/{teamID}/members/{userID}` | **owner** | Remove a member. `409` for the last owner, or a membership managed by single sign-on |
+| `GET` | `/api/teams/{teamID}/oidc-groups` | member | Which groups control this team's membership: `{"member_group","owner_group"}`. An empty string means no group grants that role here |
+| `PUT` | `/api/teams/{teamID}/oidc-groups` | **owner** | Set them. An empty string clears a binding |
+| `GET` | `/api/teams/{teamID}/integrations` | member | List integrations. Never returns keys. Each carries `status` (`active` if its key posted within 24h, `quiet` if it has but not lately, `never`), `last_used_at` (last webhook, usable or not), `last_alert_at` (when an alert last arrived on it) and `alerts_24h` (distinct alerts it refreshed in the last day). Alerts delivered before the source was recorded (migration 010) have none, so the last two fill in as Alertmanager re-sends them |
+| `PATCH` | `/api/teams/{teamID}/integrations/{integrationID}` | **owner** | Rename `{"name"}`. The key does not change |
+| `POST` | `/api/teams/{teamID}/integrations` | **owner** | Mint an integration `{"name","kind"}` — key and URL shown once |
+| `DELETE` | `/api/teams/{teamID}/integrations/{integrationID}` | **owner** | Revoke an integration. Alerts it delivered stay, unattributed |
+| `GET` | `/api/teams/{teamID}/invites` | **owner** | The team's invite links, with their uses and expiry. Never the tokens |
+| `POST` | `/api/teams/{teamID}/invites` | **owner** | Mint one `{"role","max_uses"}` — the full URL is returned once |
+| `DELETE` | `/api/teams/{teamID}/invites/{inviteID}` | **owner** | Revoke a link before it expires |
+| `GET` | `/api/teams/{teamID}/escalation` | member | The team's [escalation ladder](./escalation.md#escalation) `{repeat_count, fallback_topic, levels[], last_escalated_at?, last_escalated_incident_id?}`. Empty levels means the team has none. Each level also carries `status` (`ready`, `escalating` when an unanswered incident has climbed to it, `unreachable` when nobody on it could be woken), `waiting` (ids of the open incidents on it) and, per target, `username` (who it means today — the person on call, for a rota target), `reachable` and `problem`. The extra fields are output only; `PUT` takes the plain shape |
+| `PUT` | `/api/teams/{teamID}/escalation` | **owner** | Replace it wholesale. `400` for a level with no targets or no timeout — a rung that pages nobody is a silence with a number on it |
+| `GET` | `/api/teams/{teamID}/deadman/switches` | member | The team's [dead man's switches](./dead-mans-switch.md), each `{id, name, matcher, timeout_seconds, severity, status, last_heartbeat_at, last_triggered_at, open_incident_id, sources[]}`. `status` is `healthy`, `dead` or `dormant`; `sources` has one entry per heartbeat fingerprint. Empty when the team watches nothing |
+| `POST` | `/api/teams/{teamID}/deadman/switches` | **owner** | Add one: `{name?, matcher, timeout_seconds, severity?}`. `400` when the matcher names no `alertname` or holds several, or the timeout is not positive — a switch that silently watches nothing is the failure this feature exists to prevent |
+| `PUT` | `/api/teams/{teamID}/deadman/switches/{switchID}` | **owner** | Replace one in place, same body and validation as create. Its id is unchanged — for an automated caller reconciling a spec change, unlike delete-and-recreate |
+| `DELETE` | `/api/teams/{teamID}/deadman/switches/{switchID}` | **owner** | Stop watching. An incident it opened stays open. `404` for a switch of another team |
+
+## Notifications
+
+| Method | Path | Description |
+|---|---|---|
+| `POST` | `/api/notify/ack/{token}` | Acknowledge an incident from a push notification's Acknowledge button. No auth: the token in the path is the credential — one incident, one action, 24 hours, idempotent. Must stay publicly reachable |
+
+## Incidents
+
+| Method | Path | Description |
+|---|---|---|
+| `GET` | `/api/incidents` | List incidents. Filters: `?status=triggered\|acknowledged\|resolved`, `?severity=`, `?assigned_to=`, `?archived=true`, `?snoozed=true`, `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD`, `?sort=severity`, `?cluster=`, `?limit=` (default 50, max 500) |
+| `GET` | `/api/incidents/clusters` | The distinct `cluster` values on the caller's incidents from the last 90 days, sorted (`?team_id=` narrows it). An empty array when nothing carries the label |
+| `GET` | `/api/incidents/{id}` | Get single incident, with its alerts inline |
+| `GET` | `/api/incidents/{id}/alerts` | Alerts under this incident |
+| `GET` | `/api/incidents/{id}/timeline` | Full event history, chronological |
+| `POST` | `/api/incidents/{id}/acknowledge` | Acknowledge (stamps authed user + time) |
+| `DELETE` | `/api/incidents/{id}/acknowledge` | Clear acknowledgement, back to `triggered` |
+| `POST` | `/api/incidents/{id}/resolve` | Close by hand — **terminal**, see above |
+| `POST` | `/api/incidents/{id}/assign` | Reassign `{"user_id"}` |
+| `POST` | `/api/incidents/{id}/snooze` | Hide until `{"until": RFC3339}` or `{"duration": "2h"}` |
+| `DELETE` | `/api/incidents/{id}/snooze` | Un-snooze |
+| `POST` | `/api/incidents/{id}/archive` | Archive (hides from the default list) |
+| `DELETE` | `/api/incidents/{id}/archive` | Un-archive |
+| `POST` | `/api/incidents/{id}/notes` | Add a note `{"content"}` |
+| `DELETE` | `/api/incidents/{id}/notes/{eventID}` | Delete own note |
+
+With no `?status=` filter, `GET /api/incidents` returns **open** incidents only —
+the queue an on-call person wants. Currently snoozed and archived incidents are
+excluded unless asked for. Actions that only make sense on an open incident
+return `409` once it is resolved.
+
+Notes are ordinary timeline events of type `note`; only they are deletable, and
+only by their author. The rest of the timeline is a record of what happened.
+
+### The incident object
+
+| Field | Type | Notes |
+|---|---|---|
+| `id` | integer | Server-assigned |
+| `group_key` | string | Alertmanager's `groupKey` — opaque, treat as an identifier |
+| `title` | string | Rendered from `groupLabels` |
+| `group_labels` | object | String→string, as sent by Alertmanager |
+| `status` | string | `"triggered"`, `"acknowledged"` or `"resolved"` |
+| `severity` | string | *optional* — high-water mark across the incident's alerts; never lowered |
+| `triggered_at` | timestamp | When the incident opened |
+| `acknowledged_by_id` / `acknowledged_by` / `acknowledged_at` | | *optional* — user id, username, time |
+| `assigned_to_id` / `assigned_to` | | *optional* — user id, username |
+| `snoozed_until` | timestamp | *optional* — a value in the past reads as not snoozed |
+| `resolved_at` | timestamp | *optional* |
+| `resolution_source` | string | *optional* — `"alerts"`, `"manual"` or `"recovered"` |
+| `archived_at` | timestamp | *optional* |
+| `alerts` | array | Only on `GET /api/incidents/{id}` |
+
+Treat `resolution_source` as an open set, as with the alert field of the same
+name: degrade unknown values to "resolved, reason unknown".
+
+### The timeline event object
+
+| Field | Type | Notes |
+|---|---|---|
+| `id` | integer | |
+| `incident_id` | integer | |
+| `type` | string | See below — treat as an open set |
+| `user_id` / `username` | | *optional* — absent when the server acted rather than a person |
+| `alert_id` | integer | *optional* — the alert an `alert_added` / `alert_resolved` event refers to |
+| `detail` | string | *optional* — the note body, the snooze deadline, etc. |
+| `created_at` | timestamp | |
+
+Types written today: `triggered`, `alert_added`, `alert_resolved`,
+`acknowledged`, `unacknowledged`, `assigned`, `archived`, `unarchived`, `snoozed`,
+`unsnoozed`, `resolved`, `note`, `notified`, `notify_failed`, `deadman_silent`. On an
+`assigned` event `user_id` is the **assignee**, not the actor; the actor is in
+`actor_user_id`/`actor_username` or `actor_service_account_id`/`actor_service_account_name`
+(absent on assignments made before they were recorded). New types may be added; render
+unknown ones generically rather than dropping them.
+
+On `notified` and `notify_failed`, `detail` carries the notification kind
+(`triggered` | `reminder` | `resolved`), and on a failure the reason after it.
+`user_id` is who was paged — absent means the page went to the shared fallback
+topic and so belongs to nobody. The topic itself is never written to the
+timeline: it is a shared secret with the ntfy server, and every API key can read
+this.
+
+## Alerts
+
+Alerts are read-only. Everything a person does happens on the incident.
+
+| Method | Path | Description |
+|---|---|---|
+| `GET` | `/api/alerts` | List alerts. Filters: `?status=firing\|resolved`, `?name=`, `?incident_id=`, `?archived=true`, `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD`, `?limit=` (default 50, max 500) |
+| `GET` | `/api/alerts/{id}` | Get single alert |
+
+Archived alerts are hidden from `GET /api/alerts` unless `?archived=true` is
+passed; alert archiving is automatic housekeeping by the sweeper, not a user
+action. Resolved alerts carry `resolution_source`: `"alertmanager"` for a real
+resolved webhook, `"expiry"` when the sweeper inferred it (see
+[Stale alert expiry](./incidents.md#stale-alert-expiry)), `"deadman"` for a heartbeat declared
+dead (see [Dead man's switch](./dead-mans-switch.md)).
+
+### The alert object
+
+Returned by `GET /api/alerts` (as an array) and `GET /api/alerts/{id}`.
+Timestamps are RFC 3339 in UTC. Fields marked *optional* are omitted entirely
+when unset, so clients must treat them as nullable.
+
+| Field | Type | Notes |
+|---|---|---|
+| `id` | integer | Server-assigned; stable for the life of the row |
+| `fingerprint` | string | Alertmanager's fingerprint — the upsert key |
+| `name` | string | From the `alertname` label |
+| `status` | string | `"firing"` or `"resolved"` |
+| `labels` | object | String→string, as sent by Alertmanager |
+| `annotations` | object | String→string, as sent by Alertmanager |
+| `starts_at` | timestamp | When the alert instance began, **per Prometheus** |
+| `ends_at` | timestamp | *optional* — absent while no end is known |
+| `generator_url` | string | Link back to the originating Prometheus |
+| `received_at` | timestamp | When the server last accepted a webhook for this alert — see below |
+| `incident_id` | integer | *optional* — the most recent incident this alert belongs to |
+| `resolution_source` | string | *optional* — `"alertmanager"`, `"expiry"` or `"deadman"` |
+| `archived_at` | timestamp | *optional* — set while archived |
+
+#### `received_at` is a liveness heartbeat
+
+`starts_at` comes from Prometheus and **never changes** for the lifetime of an
+alert instance. It says when the problem began, not whether it is still
+happening — an alert that started twelve days ago looks identical whether
+Alertmanager refreshed it a minute ago or went silent a week ago.
+
+`received_at` is the field that answers "is this still live". It is set to the
+server's clock on **every accepted webhook** for that fingerprint, including the
+unchanged firing notifications Alertmanager re-sends every `repeat_interval`.
+Clients may rely on this:
+
+- **A firing alert whose `received_at` is advancing is still being refreshed.**
+ Stale-dating it against `repeat_interval` is a valid liveness check, and it is
+ what the built-in sweeper does (see
+ [Stale alert expiry](./incidents.md#stale-alert-expiry)).
+- **`received_at` tracks accepted payloads, not delivery attempts.** A retry
+ that describes an older instance than the stored one is discarded, and a
+ discarded payload does not move `received_at`.
+- **It stops advancing once the alert resolves,** because Alertmanager stops
+ re-sending. On an alert resolved by the sweeper
+ (`"resolution_source": "expiry"`) it therefore marks the last time
+ Alertmanager was actually heard from, which is earlier than `ends_at`.
+
+`GET /api/alerts` is ordered by `received_at` descending — most recently
+refreshed first — and the `?from=` / `?to=` filters on both the alert and stats
+endpoints select on `received_at`, not `starts_at`.
+
+#### `resolution_source` says how much to trust `ends_at`
+
+An alert can leave the firing state two ways, and `resolution_source` records
+which happened. Clients may rely on this:
+
+- **Absent while firing.** It is set only on resolve, and a re-fire under the
+ same fingerprint clears it again, so its presence always agrees with
+ `"status": "resolved"`.
+- **`"alertmanager"` — a real resolved webhook arrived.** `ends_at` is the end
+ time Alertmanager reported. It is an observed value and can be displayed as
+ fact.
+- **`"expiry"` — the sweeper inferred the resolve** because Alertmanager stopped
+ refreshing the alert (see [Stale alert expiry](./incidents.md#stale-alert-expiry)). Nothing
+ ever reported an end, so **`ends_at` is approximate**: it is either the stale
+ `endsAt` watermark from the last notification, or — when that notification
+ carried none — the time the sweep ran, which lags the last real contact by up
+ to `TERDUT_STALE_AFTER` plus a sweep interval. Treat it as "no later than",
+ not as when the problem stopped.
+
+ On these alerts `received_at` is the more truthful signal: it marks the last
+ time Alertmanager was actually heard from. Surfacing the distinction is
+ worthwhile, since `"expiry"` can also mean the alert is still firing and the
+ notification path broke.
+
+- **`"deadman"` — a heartbeat was declared dead** (see
+ [Dead man's switch](./dead-mans-switch.md)). Like `"expiry"`, an inference from
+ silence rather than an observed end, so `ends_at` is approximate — but a much
+ tighter one, bounded by the switch's timeout. It is also the one resolution
+ a re-fire under the same `starts_at` can undo, since the switch coming back is
+ exactly the evidence that the inference was wrong.
+
+Treat the value as an open set and tolerate ones you do not recognise — new
+sources may be added, and unknown values should degrade to "resolved, reason
+unknown" rather than being rejected.
+
+## On-call schedule
+
+| Method | Path | Description |
+|---|---|---|
+Each team keeps its own rota, so two teams can have two different people on call
+on the same day. The person taking a shift has to be in the team — paging
+somebody who cannot open the incident is worse than paging nobody.
+
+| Method | Path | Who | Description |
+|---|---|---|---|
+| `POST` | `/api/teams/{teamID}/schedule` | **owner** | Assign user to dates `{"user_id", "dates":["YYYY-MM-DD",...], "replace"}` — all-or-nothing |
+| `GET` | `/api/teams/{teamID}/schedule` | member | List entries. Filters: `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD` |
+| `DELETE` | `/api/teams/{teamID}/schedule/{id}` | **owner** | Remove schedule entry |
+| `GET` | `/api/schedule/current` | any | Who is on call today (UTC) in **every** team the caller is in — one entry per team, `[]` when nobody anywhere |
+
+## Statistics
+
+Every figure counts the caller's own teams only: a report that counted other
+teams' incidents would leak their volume, and their alert names through the
+top-alerts list, and would not be a number about the reader's work anyway.
+
+All stat endpoints accept optional `?from=YYYY-MM-DD` and `?to=YYYY-MM-DD`, and exclude archived rows to match the default list views. Alert stats filter on `received_at`; incident stats filter on `triggered_at`.
+
+| Method | Path | Description |
+|---|---|---|
+| `GET` | `/api/stats/incidents` | `{total, triggered, acknowledged, resolved, mtta_seconds, mttr_seconds}` |
+| `GET` | `/api/stats/alerts` | `{total, firing, resolved}` counts |
+| `GET` | `/api/stats/alerts/top` | Most frequent alert names. `?limit=` (default 10, max 100) |
+| `GET` | `/api/stats/alerts/by-hour` | Count per hour-of-day (UTC), all 24 slots returned |
+| `GET` | `/api/stats/alerts/by-day` | Count per day-of-week, all 7 slots with names returned |
+
+`mtta_seconds` (time to acknowledge) and `mttr_seconds` (time to resolve) are
+averages over incidents that have actually been acknowledged or resolved, and are
+**null** until there are any — null means "no data", not zero.
diff --git a/docs/configuration.md b/docs/configuration.md
new file mode 100644
index 0000000..b89a49a
--- /dev/null
+++ b/docs/configuration.md
@@ -0,0 +1,49 @@
+# Configuration
+
+_Environment variables and settings._ Back to the [README](../README.md) and the [documentation index](./README.md).
+
+Two kinds of setting, split by who changes them and how often.
+
+**Where the server is plugged in** stays in the environment: the listen address,
+the database DSN, the ntfy URL and token, the public URL. They are needed before
+the database is open, and two of them are credentials.
+
+**How the server behaves** lives in the database and is edited by an
+administrator in the web UI or through `PUT /api/admin/settings`, taking effect
+on the next sweep rather than at the next restart. The variables below marked
+**seed** are the value each of those starts from: written once, on first start,
+and never overwritten afterwards — a redeploy cannot put a chart's default back
+over an administrator's edit.
+
+| Variable | Default | Description |
+|---|---|---|
+| `TERDUT_ADDR` | `:8080` | TCP address to listen on |
+| `TERDUT_DB_DSN` | — | **Required.** Postgres connection string, e.g. `postgres://terdut:secret@localhost:5432/terdut?sslmode=require` |
+| `TERDUT_ARCHIVE_AFTER` | `168h` (7d) | **seed.** How long a resolved alert or incident stays in the default list before being auto-archived |
+| `TERDUT_STALE_AFTER` | `6h` | **seed.** How long a firing alert may go without a refreshing webhook before it is treated as resolved — **must exceed your Alertmanager `repeat_interval`** |
+| `TERDUT_NTFY_URL` | — | ntfy server to publish push notifications to. Empty disables notifications entirely |
+| `TERDUT_NTFY_TOKEN` | — | Bearer token for an access-controlled ntfy |
+| `TERDUT_NTFY_FALLBACK_TOPIC` | — | Topic used when nobody is on call |
+| `TERDUT_PUBLIC_URL` | — | Base URL a phone uses to reach this server: the notification's link into the web UI, its Acknowledge button, and whether the session cookie is `Secure` |
+| `TERDUT_NOTIFY_REPEAT` | `15m` | **seed.** How long an incident may sit unacknowledged before it is paged again. `0` notifies once and never repeats |
+| `TERDUT_PASSWORD_LOGIN` | `true` | `false` refuses password login and password sign-up (`403`), leaving single sign-on the only way in. Refused at startup unless SSO is configured |
+| `TERDUT_TRUSTED_PROXIES` | `1` | How many reverse proxies in front of the server append to `X-Forwarded-For`; the per-address rate limits use the entry that many hops from the right. `0` ignores the header |
+| `TERDUT_OPERATOR_KEY` | — | At least 32 characters. When set, the instance-scoped service account `terdut-operator` is created if missing and its `seed` key replaced with this value at every start — how terdut-operator authenticates without a bootstrap handshake. An instance-scoped account acts as owner of every team (team configuration) but is not a member of any, so it reads no incidents |
+| `TERDUT_OPERATOR_MODE` | `false` | Declares this install gitops-managed: a session's or a user's own API key's writes to teams, escalation policies, dead man's switches and integrations are refused (`403 reason:"operator_managed"`); a [service account](./api.md#service-accounts)'s are not. Team membership and the schedule stay editable regardless |
+| `TERDUT_OIDC_ISSUER` | — | Turns single sign-on on. The provider's issuer URL; discovery is read from `/.well-known/openid-configuration`. See [Single sign-on](./single-sign-on.md#single-sign-on-oidc) |
+| `TERDUT_OIDC_CLIENT_ID` / `TERDUT_OIDC_CLIENT_SECRET` | — | **Required with an issuer.** The confidential client registered at the provider. Keep the secret in a Secret, not in values |
+| `TERDUT_OIDC_NAME` | `SSO` | What the sign-in button calls the provider |
+| `TERDUT_OIDC_SCOPES` | `openid profile email` | Scopes requested, comma or space separated. Authentik puts `groups` behind `profile` |
+| `TERDUT_OIDC_USERNAME_CLAIM` / `_EMAIL_CLAIM` / `_GROUPS_CLAIM` | `preferred_username` / `email` / `groups` | ID token claims read for the username, email and groups |
+| `TERDUT_OIDC_TRUST_EMAIL` | `false` | Link a first sign-in to an existing local user by email even if the provider does not mark the address verified |
+| `TERDUT_OIDC_ALLOWED_GROUPS` | — | Comma-separated. Only people in one of these may sign in. Empty admits everybody the provider authenticates |
+| `TERDUT_OIDC_ADMIN_GROUP` | — | Members are system administrators |
+| `TERDUT_OIDC_SESSION_MAX_AGE` | `12h` | Hard ceiling on a session made by an SSO sign-in |
+
+Durations use Go syntax (`30m`, `12h`, `168h`). An unparseable value falls back to the default.
+
+Note that `TERDUT_STALE_AFTER` and a dead man's switch timeout point in opposite directions. Staleness
+is a generous grace period around a `repeat_interval` you do not control; a dead man's switch is a
+deadline you set deliberately, and the heartbeat's route is configured to beat faster than it.
+
+In the Helm chart the two sweeper durations are set via `sweeper.staleAfter` and `sweeper.archiveAfter`, notifications via the `notify.*` values, single sign-on via `oidc.*` and `passwordLogin`, and operator mode via `operatorMode`.
diff --git a/docs/dead-mans-switch.md b/docs/dead-mans-switch.md
new file mode 100644
index 0000000..bd2a43b
--- /dev/null
+++ b/docs/dead-mans-switch.md
@@ -0,0 +1,81 @@
+# Dead man's switches
+
+_Detecting that alerts have stopped arriving._ Back to the [README](../README.md) and the [documentation index](./README.md).
+
+Everything above assumes alerts arrive. If Prometheus stops evaluating, or
+Alertmanager cannot reach this server, nothing arrives — and silence looks
+exactly like everything being fine. A dead man's switch inverts the handling for
+one designated alert so that silence is the signal:
+
+- **receiving** it opens no incident, and
+- the **absence** of it does.
+
+kube-prometheus-stack already ships the alert for this. `Watchdog` is
+`expr: vector(1)`, so it fires permanently and is re-sent forever; it is worth
+nothing unless something downstream notices it stop. It is the usual first switch.
+
+**Switches belong to a team**, which decides which of its own alerts are
+heartbeats and how long a silence has to last. Each **switch** is a row of its
+own — a name, one matcher, a timeout and a severity — so switches in one team
+can have different deadlines. An owner adds and removes them on **Team →
+Switches**, which lists each with a status (**healthy**, **dead**, or
+**dormant** until its first heartbeat), when it was last heard from, and when it
+last opened an incident; a matcher that several clusters satisfy is broken down
+per cluster. The API is `POST`/`DELETE /api/teams/{teamID}/deadman/switches`. A
+missed heartbeat opens an incident in the team whose integration received it.
+Removing a switch stops the watching; an incident it already opened stays open
+until somebody resolves it.
+
+A new team watches nothing until its owner (or terdut-operator, from a
+`TerdutTeam`) adds a switch: inheriting an install-wide heartbeat would page a
+new team about a source it has never heard of.
+
+A matcher is a set of exact label conditions, one of which must be the
+`alertname`, , one matcher per switch, `,` between the label conditions:
+
+```
+alertname=Watchdog,cluster=prod
+```
+
+**The unit of monitoring is the fingerprint, not the alert name.** Two clusters
+sending the same `Watchdog` are two independent switches, so a healthy one can
+never mask a dead one.
+
+## The lifecycle
+
+A switch is **dormant** until its first heartbeat arrives. A configured matcher
+that has never been heard from opens nothing, so a fresh deploy or a restored
+database does not page. It also means a matcher that never matches anything is
+silently inert.
+
+Once armed, the sweeper declares it **dead** when either the heartbeat has not
+been refreshed within the switch's `timeout_seconds`, or Alertmanager explicitly
+resolved it — the sender saying the heartbeat stopped needs no further waiting.
+That opens an incident at the switch's `severity`, assigned and paged like any
+other, and marks the heartbeat alert `"resolution_source": "deadman"` so the
+alert list stops claiming a dead switch is firing.
+
+It **recovers** when the heartbeat starts arriving again: the incident resolves
+with `"resolution_source": "recovered"` and the all-clear goes to whoever was
+paged.
+
+Resolving the incident by hand sticks, the same way it does for an alert-backed
+one. While the switch stays silent nothing new opens — so a decommissioned
+source is a one-time page rather than a nag. The switch **re-arms** on the next
+heartbeat: come back and die again, and that is a new incident.
+
+## Two things to know
+
+A switch's timeout must be **shorter** than the `repeat_interval` of the
+route carrying the heartbeat, which is the exact opposite of
+`TERDUT_STALE_AFTER`. Inheriting a default `repeat_interval` of 4h gives you a
+switch that takes four hours to notice anything, so give the heartbeat
+[its own route](./alertmanager.md#alertmanager-configuration). Matched alerts are exempt from
+stale-alert expiry — a heartbeat answers to its own timeout and nothing else.
+
+A dead man's switch incident has **no member alerts**:
+`GET /api/incidents/{id}/alerts` returns an empty list. There is no alert
+describing the problem, because the problem is that no alert arrived. What
+happened is on the timeline instead, as a `deadman_silent` event carrying the age
+of the last heartbeat, and the heartbeat's labels are on the incident's
+`group_labels`.
diff --git a/docs/deployment.md b/docs/deployment.md
new file mode 100644
index 0000000..20e1ae9
--- /dev/null
+++ b/docs/deployment.md
@@ -0,0 +1,74 @@
+# Deployment
+
+_Running the server in a container and on Kubernetes with the Helm chart._ Back to the [README](../README.md) and the [documentation index](./README.md).
+
+## Docker
+
+```bash
+docker build -t terdut-server .
+docker run -p 8080:8080 \
+ -e TERDUT_DB_DSN='postgres://terdut:secret@host.docker.internal:5432/terdut?sslmode=disable' \
+ terdut-server
+```
+
+The server creates its own schema on startup and needs a reachable Postgres; it stores nothing on
+disk, so there is no volume to mount.
+
+## Kubernetes
+
+A Helm chart is published from this repository as an OCI artifact, versioned in lockstep
+with the app — chart `x.y.z` is always app `vx.y.z`:
+
+```bash
+helm upgrade --install terdut-server oci://git.ryuvia.com/niklas/terdut-server \
+ --version 0.9.2 \
+ --namespace terdut-server --create-namespace \
+ --set networking.hostname=terdut.example.com
+```
+
+The chart expects a [Gateway API](https://gateway-api.sigs.k8s.io/) Gateway named `envoy-main` in
+the `envoy-gateway-system` namespace to already exist — it renders an `HTTPRoute` against it rather
+than an `Ingress`. TLS is terminated at the gateway, so the server itself never sees a certificate.
+
+| Value | Default | Description |
+|---|---|---|
+| `networking.hostname` | `terdut.example.com` | Hostname the `HTTPRoute` serves |
+| `networking.listener` | `""` | Gateway listener (`sectionName`) to bind to. Empty attaches to every matching listener, **including plaintext HTTP** — set it to the HTTPS listener's name to serve TLS only |
+| `networking.servicePort` | `8080` | Port the route forwards to; keep in sync with `service.port` |
+| `bootstrap.enabled` | `true` | Runs a post-install hook that creates the first user and stores its API key in the `-admin-key` Secret. Already-bootstrapped servers are left alone |
+| `database.dsn` | `""` | **Required.** Postgres DSN, with no password in it. The chart provisions no database |
+| `database.passwordSecret.name` | `""` | Secret supplying `PGPASSWORD`. With the Zalando postgres operator, the Secret it generates for the role |
+| `database.passwordSecret.key` | `password` | Key within that Secret |
+
+The API key travels in an `Authorization: Bearer` header, so set `networking.listener` whenever the
+hostname is reachable outside a trusted network.
+
+### The database
+
+The chart provisions no database: it takes a DSN and expects a Postgres that already exists. In this
+cluster the wrapper chart declares an `acid.zalan.do/v1 postgresql` CR; anywhere else, any reachable
+Postgres 14+ will do.
+
+The DSN carries no password. pgx falls back to libpq's environment variables for whatever the DSN
+leaves out, so the password arrives as `PGPASSWORD` from a Secret and never appears in values, in
+the rendered manifest or in `kubectl describe pod`. With the postgres operator that Secret is the
+one it generates for the role, so a rebuild mints a new password with nothing to keep in sync —
+the same wiring miniflux uses.
+
+The server migrates its own schema on startup, so a new database only has to exist and be writable.
+
+### Backups
+
+Postgres is backed up where it runs, not from here. The database pod carries a
+[k8up](https://k8up.io/) `k8up.io/backupcommand` annotation that streams a `pg_dump`, the same way
+gitea and immich do in this cluster.
+
+## On Kubernetes with the operator
+
+[terdut-operator](https://git.ryuvia.com/niklas/terdut-operator) runs a server for you from a
+`TerdutServer` object and manages its teams, escalation ladders, dead man's switches and alert
+sources as Kubernetes objects. It hands the server a generated key through `TERDUT_OPERATOR_KEY`
+(see [Configuration](./configuration.md) and [`SERVICE-ACCOUNTS.md`](../SERVICE-ACCOUNTS.md)), and
+the server then treats configuration as operator-managed (`TERDUT_OPERATOR_MODE`), refusing edits
+made by hand in the web UI. Use the Helm chart above for a plain install, the operator when you
+want that configuration in gitops.
diff --git a/docs/development.md b/docs/development.md
new file mode 100644
index 0000000..ae4b154
--- /dev/null
+++ b/docs/development.md
@@ -0,0 +1,93 @@
+# Development and releasing
+
+_Building, testing and releasing the server._ Back to the [README](../README.md) and the [documentation index](./README.md).
+
+## Upgrading
+
+The schema is a single baseline (`internal/db/migrations/001_schema.sql`) and no
+release has shipped yet, so there is no upgrade path from earlier development
+databases: start from an empty one. Changes after the first release arrive as
+new numbered migrations.
+
+## Development
+
+```bash
+make test-db # start a local Postgres for the tests (podman or docker)
+make test # run all tests
+go build ./... # compile all packages
+go run ./cmd/terdut # run locally (needs TERDUT_DB_DSN)
+```
+
+The tests need a real Postgres, because the server does — there is no in-memory Postgres.
+`TERDUT_TEST_DSN` says where it is, `make test-db` starts
+one on port 5433 and prints the DSN, and `make test-db-stop` removes it. Each test gets its
+own schema on that server, so tests cannot see each other's rows. An unset `TERDUT_TEST_DSN`
+fails the suite rather than skipping it: a run that quietly tests nothing is worse than one
+that does not run.
+
+`make fmt lint test helm-lint` is the gate. It mirrors `.gitea/workflows/ci.yaml` step for
+step, so a green run here means a green pipeline — with one deliberate exception: `make test`
+adds `-race`, which CI does not. The sweeper, the notifier goroutine and the dead man's switch
+sweep all run concurrently against the same database, and a race between them would surface as
+a flaky incident in production rather than as a red build.
+
+The web UI lives in `internal/web/static/` as plain HTML, CSS and ES modules,
+embedded into the binary with `go:embed`. It has no build step and no npm, so
+editing a file and restarting the server is the whole loop.
+
+## Releasing
+
+```
+push or PR → ci.yaml gofmt, go vet, go test -race
+ govulncheck, gitleaks
+ helm lint + render
+push tag vX.Y.Z → release.yaml the same gate, then publish:
+ git.ryuvia.com/niklas/terdut-server:vX.Y.Z
+ oci://git.ryuvia.com/niklas/terdut-server X.Y.Z
+ then trivy-scan the pushed image
+PR to Ryuvia/charts → bump the wrapper chart to X.Y.Z; on merge
+ Flux reconciles and the release rolls out
+```
+
+Both artifacts go to the **personal** Gitea namespace rather than `ryuvia`, because Gitea
+scopes package visibility to the owner with no per-package override — so `ryuvia/*` is private
+because the org is. Publishing to `niklas` keeps them anonymously pullable, which is why no
+pull secret is needed in the cluster. Same reasoning, and the same choice, as riksdata and
+rd-web.
+
+Saying **"Release"** runs all three rows: the `release` skill commits, pushes, tags, waits for
+the pipeline, and opens the `Ryuvia/charts` PR, stopping before the merge. See
+`~/.claude/skills/release/`, or `.release.conf` here for this repo's part of it.
+
+The chart is published **only** from the tag, by the `chart` job. There used to be a second
+publisher on every `charts/**` push to main, and the two raced for the same chart version with
+different answers — chart 0.9.0 went out reading `appVersion: "latest"` that way. One
+publisher, triggered by the tag (`766f439`). The cost is that a chart-only change has no
+version of its own and rides the next app tag.
+
+Both workflows are thin drivers over the Makefile: `ci.yaml` runs `make fmt lint test` and
+`make helm-lint`, `release.yaml` adds `make binaries`, `make push`, `make helm-package` and
+`make helm-push`. That is deliberate — it is what makes a green local gate and a green
+pipeline the same code rather than two descriptions of it, and it is how riksdata and rd-web
+have always worked.
+
+`make push` builds and pushes in one step, unlike those two, because the image is
+`linux/amd64,linux/arm64` and buildx cannot load a multi-platform result into the local image
+store. `make build` stays single-platform and local-only. Both refuse `VERSION=dev`:
+publishing is one command, so it is also one command to run by accident. Publishing happens
+by pushing a tag.
+
+Two things the release process needs to know about this repo:
+
+- **The image scan runs after publishing**, like riksdata's and rd-web's: trivy cannot read
+ a locally built image on this runner, so it pulls the pushed one. A red `scan-image` means
+ do not bump the wrapper chart to that version — it cannot unpublish anything. The image is
+ `FROM scratch`, so trivy sees exactly one target, the Go binary and its module graph.
+- **The wrapper chart's `values.yaml` has two `tag:` lines** — the app image and the python
+ backup sidecar — so `chart-bump` is given `--image` to say which one moves. Once the wrapper
+ chart drops the sidecar and declares a `postgresql` CR instead, there is one `tag:` line
+ again, and `--image` becomes belt and braces.
+
+The wrapper chart must have **its own `version:` bumped in the same commit**. Flux reconciles
+with `reconcileStrategy: ChartVersion`, so a chart whose version did not change produces no
+new artifact and the change is never deployed — with no error anywhere.
diff --git a/docs/escalation.md b/docs/escalation.md
new file mode 100644
index 0000000..583aa2b
--- /dev/null
+++ b/docs/escalation.md
@@ -0,0 +1,45 @@
+# Escalation
+
+_Escalation ladders: who is paged next when nobody acknowledges._ Back to the [README](../README.md) and the [documentation index](./README.md).
+
+Without a ladder, an unacknowledged incident re-pages the same topic every
+`notify_repeat` forever. That is a louder version of the same silence: if the
+person on call is asleep, out of signal, or has left, nothing else happens.
+
+A team can configure an ordered ladder instead. Each level has a timeout and a
+set of targets, and a target is either a named person or **whoever the team's
+rota says is on call today** — the target that keeps working when the rota
+changes and nobody remembers to edit the policy.
+
+```
+level 1 5m oncall the rota gets first refusal
+level 2 5m user:bob then a named second
+ then repeat_count more rounds
+ then the team's fallback topic, once
+```
+
+When a level's timeout passes with the incident still `triggered`, the next
+level is paged. Off the end of the ladder the whole thing runs again
+`repeat_count` times, and after that the team's `fallback_topic` is paged once
+as the end of the line. The incident stays open throughout: running out of
+people to wake is not the same as somebody answering.
+
+**Acknowledging or resolving stops it**, which is the point — continuing to wake
+people after somebody has said "I have this" is how a tool teaches people to
+mute it. **Snoozing pauses it**: a deliberate "not now" holds the ladder where
+it is, and it resumes when the snooze runs out.
+
+Every step is on the incident's timeline with the level and the names it woke,
+so somebody reading it afterwards can tell why their phone rang at 04:00. A
+level whose targets are all unreachable — no ntfy topic, a disabled account, an
+empty rota — is recorded as `nobody reachable` and the ladder moves on rather
+than stalling on a rung that cannot ring.
+
+**Reminders and escalation never both run.** A team with a ladder gets
+escalation; a team without keeps the reminder behaviour exactly as it was. Two
+pages for one silence is the surest way to get a tool muted.
+
+The ladder's `fallback_topic` is per team, unlike `TERDUT_NTFY_FALLBACK_TOPIC`,
+which is the install-wide topic used when an incident opens with nobody on call.
+They answer different questions: one is "nobody was scheduled", the other is
+"everybody scheduled has been tried".
diff --git a/docs/images/admin.png b/docs/images/admin.png
new file mode 100644
index 0000000..ea088af
Binary files /dev/null and b/docs/images/admin.png differ
diff --git a/docs/images/alerts.png b/docs/images/alerts.png
new file mode 100644
index 0000000..f556dcd
Binary files /dev/null and b/docs/images/alerts.png differ
diff --git a/docs/images/incident-note.png b/docs/images/incident-note.png
new file mode 100644
index 0000000..0b4cb16
Binary files /dev/null and b/docs/images/incident-note.png differ
diff --git a/docs/images/mobile-incident.png b/docs/images/mobile-incident.png
new file mode 100644
index 0000000..4762343
Binary files /dev/null and b/docs/images/mobile-incident.png differ
diff --git a/docs/images/mobile-queue.png b/docs/images/mobile-queue.png
new file mode 100644
index 0000000..b9fb72a
Binary files /dev/null and b/docs/images/mobile-queue.png differ
diff --git a/docs/images/oncall.png b/docs/images/oncall.png
new file mode 100644
index 0000000..0a2a667
Binary files /dev/null and b/docs/images/oncall.png differ
diff --git a/docs/images/queue-dark.png b/docs/images/queue-dark.png
new file mode 100644
index 0000000..75d231b
Binary files /dev/null and b/docs/images/queue-dark.png differ
diff --git a/docs/images/queue-light.png b/docs/images/queue-light.png
new file mode 100644
index 0000000..8c05295
Binary files /dev/null and b/docs/images/queue-light.png differ
diff --git a/docs/images/stats.png b/docs/images/stats.png
new file mode 100644
index 0000000..68816a7
Binary files /dev/null and b/docs/images/stats.png differ
diff --git a/docs/images/team-deadman.png b/docs/images/team-deadman.png
new file mode 100644
index 0000000..e958cb4
Binary files /dev/null and b/docs/images/team-deadman.png differ
diff --git a/docs/images/team-escalation.png b/docs/images/team-escalation.png
new file mode 100644
index 0000000..45d88ca
Binary files /dev/null and b/docs/images/team-escalation.png differ
diff --git a/docs/incidents.md b/docs/incidents.md
new file mode 100644
index 0000000..77cf82d
--- /dev/null
+++ b/docs/incidents.md
@@ -0,0 +1,113 @@
+# Alerts and incidents
+
+_How alerts become incidents and how incidents are worked, notified, escalated and expired._ Back to the [README](../README.md) and the [documentation index](./README.md).
+
+There are two objects, and the difference between them is the whole design.
+
+**An alert is Alertmanager's record.** It has two states, `firing` and
+`resolved`, one row per fingerprint, and no human ever writes to it. The API
+exposes alerts read-only.
+
+**An incident is the work item.** It goes `triggered → acknowledged → resolved`,
+carries an assignee, a snooze, notes and a timeline, and is the only thing people
+act on. Many alerts belong to one incident.
+
+## Correlation uses Alertmanager's `groupKey`
+
+Alertmanager has already grouped alerts according to the `group_by` routing tree
+you configured, and it sends the resulting `groupKey` and `groupLabels` on every
+webhook. Incidents adopt that answer rather than re-grouping alerts a second
+time — if you want different correlation, change `group_by` in
+`alertmanager.yml` and terdut follows.
+
+At most one incident is open per `groupKey` at a time. Alerts firing in a group
+that already has an open incident join it. The incident's `severity` is a
+high-water mark — the highest `severity` label any of its alerts has carried — so
+an incident that hit `critical` still reads as critical after the critical alert
+clears.
+
+## Several clusters, one team
+
+A team with one Alertmanager per Kubernetes cluster, each posting to its own
+source, needs two settings or the clusters run together.
+
+1. Give every alert a `cluster` label at the source. In Prometheus that is
+ `externalLabels: {cluster: prod-eu}` (kube-prometheus-stack:
+ `prometheus.prometheusSpec.externalLabels`).
+2. Add `cluster` to `group_by` in `alertmanager.yml`.
+
+The second one is the one that matters. Incidents are matched on the team and
+Alertmanager's `groupKey`, and the `groupKey` does not include external labels:
+without `cluster` in `group_by`, the same alert in two clusters has the same
+key and joins one incident. With it, each cluster gets its own, `cluster` is in
+the incident's `group_labels`, and the web UI shows it as a coloured chip on the
+queue, the incident and the alert list, instead of leaving it in the title.
+An alert that is not grouped by `cluster` still shows the chip on the alert
+list, which reads the label from the alert itself.
+
+The queue has a cluster dropdown once there are two or more values to choose
+between. It filters on the incident's `cluster` group label
+(`GET /api/incidents?cluster=...`), so it only sees incidents grouped by it.
+
+## An incident opens only on a new occurrence
+
+An incident opens when an alert **transitions into firing**: a fingerprint that
+was never seen, an alert with a newer `startsAt`, or a resolved alert that
+started again. The unchanged firing notifications Alertmanager re-sends every
+`repeat_interval` are none of those, and open nothing.
+
+This is what makes closing an incident by hand mean something. Without the rule,
+`POST /api/incidents/{id}/resolve` would be undone by the next re-send of an
+alert that never stopped firing.
+
+## Leaving the open state
+
+- **Automatically**, once every alert under the incident has stopped firing —
+ whether by a resolved webhook or by the sweeper's
+ [stale-alert expiry](#stale-alert-expiry). The incident gets
+ `"resolution_source": "alerts"`.
+- **By hand**, via `POST /api/incidents/{id}/resolve`
+ (`"resolution_source": "manual"`). This is **terminal**: a later occurrence in
+ that group opens a *new* incident rather than reopening this one. If the alert
+ underneath never stops firing, the incident stays closed — that is what
+ resolving by hand asserts.
+- **On recovery**, for a [dead man's switch](./dead-mans-switch.md) incident whose
+ heartbeat started arriving again (`"resolution_source": "recovered"`). These
+ incidents have no member alerts, so the automatic cascade above cannot reach
+ them.
+
+To quieten an incident you expect to come back, snooze it instead
+(`POST /api/incidents/{id}/snooze`). A snooze hides the incident from the default
+list without closing it, and expires by simply falling into the past.
+
+## On-call assignment
+
+A new incident is assigned to whoever holds today's schedule entry at the moment
+it opens (`GET /api/schedule/current`). If nobody is scheduled it opens
+unassigned. Reassign with `POST /api/incidents/{id}/assign`.
+
+One person holds a given day, so `POST /api/schedule` refuses a date somebody
+already has: taking a shift off the person expecting to be paged for it should
+not be something a plain call does by accident. Pass `"replace": true` to take
+them anyway. Either way the whole request is one transaction — a week where some
+days are free and some are taken moves as a unit, and a failure leaves the rota
+exactly as it was rather than with a hole in it.
+
+## Stale alert expiry
+
+A resolved webhook is the only signal that an alert has stopped firing, so a
+notification that is dropped, silenced, or lost to a restart would otherwise pin
+that alert as firing forever. A background sweeper resolves firing alerts that
+Alertmanager has stopped refreshing, using either signal:
+
+- the `endsAt` watermark on the last notification has passed, or
+- no webhook has refreshed the alert within `TERDUT_STALE_AFTER`.
+
+Alertmanager re-sends firing notifications every `repeat_interval`, which is what
+keeps a live alert fresh — so `TERDUT_STALE_AFTER` must be comfortably larger
+than your `repeat_interval` (default 4h), or live alerts will be resolved
+prematurely. Alerts resolved this way are marked `"resolution_source": "expiry"`
+to distinguish them from a real Alertmanager resolve (`"alertmanager"`).
+
+An expiry cascades: once it leaves an incident with nothing firing under it, the
+incident resolves too, in the same sweep.
diff --git a/docs/notifications.md b/docs/notifications.md
new file mode 100644
index 0000000..d455c4a
--- /dev/null
+++ b/docs/notifications.md
@@ -0,0 +1,62 @@
+# Push notifications
+
+_Pages through ntfy, who gets them and how to acknowledge from the notification._ Back to the [README](../README.md) and the [documentation index](./README.md).
+
+With `TERDUT_NTFY_URL` set, an incident that opens is pushed to the on-call
+person's phone through [ntfy](https://ntfy.sh). Everybody sets their own topic
+under *Account* in the web UI, where a **Send a test push** button proves it
+before an incident has to; `PUT /api/users/{id}/notify` is the same thing over
+the API, and an administrator may set somebody else's. A user with no topic
+falls back to `TERDUT_NTFY_FALLBACK_TOPIC`, as does an incident that opens with
+nobody on call. If neither yields a topic, nothing is queued.
+
+The **server** is the install's one ntfy, from `TERDUT_NTFY_URL`, and is not
+something a user picks. Only the topic is per-person.
+
+A topic is a shared secret with the ntfy server: anyone who knows it can both
+read the pages and publish to it, so an unguessable one is worth the trouble.
+That is also why the topic never appears in an incident's timeline, which every
+API key can read.
+
+Three things get pushed:
+
+- **triggered** — an incident opened. Priority follows severity (`critical` maps
+ to ntfy's max priority, the one that overrides the phone's quiet settings).
+- **reminder** — the incident is still `triggered` after `TERDUT_NOTIFY_REPEAT`.
+ Repeats until somebody acts. Acknowledging, snoozing, resolving or archiving
+ all stop it — snooze is the mute button.
+- **resolved** — every alert under the incident stopped firing. Only sent to
+ whoever was paged in the first place, and only for the automatic cascade:
+ resolving by hand pushes nothing, since the person who did it already knows.
+
+Notifications carry an **Acknowledge** button that acknowledges the incident
+without opening anything. It POSTs to `/api/notify/ack/{token}`, an
+unauthenticated route authorised by the 256-bit token in its path — minted fresh
+per notification, scoped to one incident and one action, and valid for 24 hours.
+A real API key is never put in a notification, because the message is stored on
+the ntfy server and cached on the device.
+
+The token is **not** consumed by use. Acknowledging is idempotent, so a token
+stays valid for its full 24 hours and a second tap is a no-op that reports the
+incident's current state rather than an error — which is what you want when a
+tap is retried on a flaky mobile connection. What bounds it is scope, not a use
+count: one incident, one action, one day. Expired tokens are purged by the
+sweeper.
+
+Two consequences worth planning for:
+
+- `/api/notify/ack/{token}` **must stay publicly reachable**, or the button will
+ not work when the responder is off your network.
+- Notifications sent to the fallback topic carry **no** Acknowledge button. The
+ topic is shared, and a button on it would let any subscriber acknowledge as
+ somebody else.
+
+Delivery is a queue, not an inline call: the webhook writes a row and a
+background notifier sends it within 30 seconds, retrying with exponential
+backoff up to 8 attempts. Nothing about ingestion blocks on ntfy being reachable.
+
+Every delivery is recorded on the incident's timeline: a `notified` event once
+ntfy accepts the publish, and a `notify_failed` event when a notification
+exhausts its retries. Written from the result rather than at enqueue, so the
+timeline says what actually happened — and a page that never landed is visible
+instead of looking the same as one that did.
diff --git a/docs/single-sign-on.md b/docs/single-sign-on.md
new file mode 100644
index 0000000..6432c0c
--- /dev/null
+++ b/docs/single-sign-on.md
@@ -0,0 +1,105 @@
+# Single sign-on (OIDC)
+
+_Signing in through an OpenID Connect provider, and mapping its groups to teams and administrators._ Back to the [README](../README.md) and the [documentation index](./README.md).
+
+terdut can sign people in through any OpenID Connect provider; the examples use
+[Authentik](https://goauthentik.io/). Groups at the provider decide who may sign
+in, which teams they belong to and whether they administer the install, much as
+Grafana's OAuth role and org mapping does. Password login keeps working alongside
+it unless you turn it off.
+
+**At the provider**, create an OAuth2/OpenID provider and an application for it:
+a *confidential* client, redirect URI `/api/oidc/callback`, and
+the `openid`, `profile` and `email` scopes. The issuer is the application's, e.g.
+`https://auth.example.com/application/o/terdut/`. Then set:
+
+```sh
+TERDUT_PUBLIC_URL=https://terdut.example.com
+TERDUT_OIDC_ISSUER=https://auth.example.com/application/o/terdut/
+TERDUT_OIDC_CLIENT_ID=terdut
+TERDUT_OIDC_CLIENT_SECRET=...
+TERDUT_OIDC_ALLOWED_GROUPS=terdut-users,terdut-admins
+TERDUT_OIDC_ADMIN_GROUP=terdut-admins
+```
+
+Which team a group grants is not server-wide config: each team names its own
+group(s), set by that team's own owner (or an administrator) from its Members
+tab, or `PUT /api/teams/{teamID}/oidc-groups {"member_group":"sre","owner_group":"sre-leads"}`.
+A team must already exist before a group can grant access to it — the sync
+never creates one.
+
+The web UI's sign-in page shows a "Sign in with " button (a plain link to
+`/api/oidc/login`) above the password form, or instead of it when
+`TERDUT_PASSWORD_LOGIN=false`; it asks `GET /api/auth/config` what the server offers
+(`password_login`, `oidc.enabled`, `oidc.name`). A refused sign-in comes back to that
+page with the reason spelled out. Access the groups grant is badged **SSO** on the
+Team, Admin and per-user pages, with its edit and remove controls disabled, and the
+Account page does not offer to set a password nobody could use.
+
+**What a sign-in does**
+
+1. *Who.* The provider's `(issuer, subject)` is the identity. The first time, a
+ user is found by email — only when the provider marks it verified, or
+ `TERDUT_OIDC_TRUST_EMAIL` is set — or created with no password. A username taken
+ by somebody else gets a numeric suffix (`alice-2`). Username and email follow the
+ provider at each sign-in. Authentik reports `email_verified` as false unless
+ configured otherwise, so linking existing users usually needs
+ `TERDUT_OIDC_TRUST_EMAIL=true`.
+2. *Whether.* With `TERDUT_OIDC_ALLOWED_GROUPS` set, somebody in none of them is
+ refused and nothing is created.
+3. *What.* The administrator flag follows `TERDUT_OIDC_ADMIN_GROUP`. Team roles
+ follow each team's own `oidc_member_group`/`oidc_owner_group`; where both of a
+ team's groups match, the owner group wins.
+
+**Managed access.** What the sync grants is marked as managed by single sign-on,
+and only that is ever changed by it. It is added at sign-in, and removed at the
+next sign-in after the group is gone, even if that leaves a team without an owner
+(an administrator can always repair a team) — the provider is the source of truth
+for what it grants, so the last-owner and last-administrator guards do not apply.
+Memberships and administrators added by hand are left alone; the exception is a
+hand-added member whose team's own group grants a *higher* role, who is raised and
+from then on managed. Editing managed access by hand (`POST` or `DELETE` on a
+team's members, revoking an SSO-granted administrator) is refused with `409`, since
+the next sign-in would undo it.
+
+> **Upgrading past migration 013: reconfigure every team's groups.**
+> `TERDUT_OIDC_GROUP_MAPPINGS` is gone, and the sync no longer creates a team by
+> name. Group-to-team-role mapping is now each team's own setting — an owner sets
+> it from the Members tab, or `PUT /api/teams/{teamID}/oidc-groups`. Until a team's
+> owner does that, an OIDC-sourced membership in it is dropped at that user's next
+> SSO sign-in, the same as any other loss of group access. Set every team's groups
+> before affected users next sign in, to avoid a visible gap in access.
+
+**How fast changes arrive.** Groups are read only at sign-in. A session made by an
+SSO sign-in has a hard ceiling (`TERDUT_OIDC_SESSION_MAX_AGE`, default 12h) that
+sliding never extends, so a change at the provider reaches terdut within that time.
+Password sessions are unaffected.
+
+> **API keys are not revoked when somebody is removed at the provider.** terdut
+> holds no refresh token and never asks the provider again, so a person removed
+> from every allowed group loses their sessions within `TERDUT_OIDC_SESSION_MAX_AGE`
+> and cannot sign in again, but keeps any API key they made (the TUI and scripts use
+> them) until an administrator disables the user in terdut.
+
+**Signing in from a terminal.** A client with no browser of its own, such as the
+TUI over SSH, signs in with a device code, run by terdut itself so the terminal
+never talks to the provider:
+
+1. The terminal calls `POST /api/oidc/device` and shows the person a link
+ (`/device?code=XXXX-XXXX`) and the code.
+2. On any device the person opens the link, signs in (by the provider or by
+ password, whatever the login page offers), sees the code and the account, and
+ presses **Approve**. Only a browser session can approve; an API key cannot.
+3. The terminal polls `POST /api/oidc/device/token` every 5 seconds and is given the
+ ordinary `terdut_session` cookie once. A person who signs in through the provider
+ gets the same `TERDUT_OIDC_SESSION_MAX_AGE` ceiling on the terminal's session as
+ on their browser's.
+
+A login expires after 10 minutes. `GET /api/auth/config` reports `device_login`.
+
+**If the provider is down**, terdut still starts (discovery is fetched on first
+use) and password login is the way in. With `TERDUT_PASSWORD_LOGIN=false` that way
+is closed: set it back to `true`. The first administrator comes from the bootstrap
+endpoint, and stays a manual administrator that no group can revoke; on an SSO-only
+install set `bootstrap.enabled: false` in the chart if you don't want that account,
+or keep it and never give it a password.
diff --git a/docs/web-ui.md b/docs/web-ui.md
new file mode 100644
index 0000000..5c20c3f
--- /dev/null
+++ b/docs/web-ui.md
@@ -0,0 +1,90 @@
+# The web UI
+
+_What the web UI offers, how sign-in and sessions work, and the Team and Admin tabs._ Back to the [README](../README.md) and the [documentation index](./README.md).
+
+The server serves a web UI at `/`: the incident queue, each incident's alerts
+and timeline with every action (acknowledge, assign, snooze, note, resolve,
+archive), who is on call, the alert feed, and an *Account* tab for your own
+password and the ntfy topic your pages go to. It is built for a phone first. On a phone
+it navigates through a hamburger menu and has a sticky action bar, it follows the
+system's dark mode, and it can be added to the home screen. From 900px wide it switches
+to a sidebar with the queue and the incident side by side. The Stats page shows
+incident counts, MTTA and MTTR, and alert frequency by name, hour and day over a
+chosen range.
+
+You sign in with a username and password. Users have no password until one is
+set, and a user without one can only use API keys:
+
+```bash
+# an admin sets someone's first password with their API key
+curl -X PUT http://localhost:8080/api/users/2/password \
+ -H "Authorization: Bearer $KEY" -H "Content-Type: application/json" \
+ -d '{"password": ""}'
+```
+
+After that, users change it themselves under *Account*. Changing your own
+password requires the current one.
+
+How a browser stays signed in:
+
+- A successful login sets an `HttpOnly`, `SameSite=Lax` session cookie. It lasts
+ 30 days and slides forward while it is used, so an on-call phone stays signed
+ in.
+- The cookie is marked `Secure` when `TERDUT_PUBLIC_URL` starts with `https://`,
+ so set it to the HTTPS address. TLS terminates at the gateway and the server
+ itself only ever sees plain HTTP.
+- Requests authenticated by the cookie are checked for cross-origin use (Go's
+ `http.CrossOriginProtection`). That is the CSRF guard. Bearer-key clients are
+ not affected.
+- Setting a password signs that user out everywhere else.
+- Ten failed logins for one username within 15 minutes lock that username for
+ the rest of the window.
+
+With `TERDUT_PUBLIC_URL` set, tapping a push notification opens the incident in
+the web UI (`/incidents/{id}`).
+
+A **Team** tab holds everything a team owns, in five sub-sections with a URL
+each and a strip across the top to move between them: the on-call rota
+(`/team/rota`), the membership (`/team/members`), the escalation ladder
+(`/team/escalation`), the alert sources with their keys (`/team/sources`) and
+the dead man's switches (`/team/deadman`). `/team` itself is an overview — who
+is on call today, how many members and owners, how many ladder levels, how many
+keys and how many switches — so a page fetches only what it shows. An owner
+edits it; a member sees the same pages read-only, because the server refuses
+their writes anyway. Somebody in more than one team picks between them above
+the strip, since the choice changes the subject of all five.
+
+The rota is a month at a time, one coloured initial per day with a legend
+underneath, and it says how many days are left uncovered — the question a rota
+is read for is who holds which stretch, and a run of one colour answers it
+where a list of dates does not. An owner taps a day to hand it to somebody or
+empty it, and fills a whole shift from the range form folded in below.
+
+The **Admin** tab appears only for a system administrator, and holds what
+belongs to the whole server rather than to one team. It has three sub-sections,
+each with a URL of its own and a strip across the top to move between them:
+every team (`/admin/teams`), every user (`/admin/users`), and the settings that
+used to be environment variables (`/admin/settings`). `/admin` itself is an
+overview — how many of each, and what each section is for. Adding somebody is
+minting them an invite link into a team, rather than creating a bare account:
+the person who accepts it picks their own password, so one never passes through
+an administrator, and the link carries the team, so they land somewhere with a
+queue in it. That happens on the team's own page, since an invite is a fact
+about a team; the user list points there rather than asking which team beside a
+form.
+
+A name in the team list opens **that team's page**, at `/admin/teams/{id}`: when it
+was created, how many are in it and how much is open, a field to rename it, the
+members with their roles, the invites into it, and deletion. The member list is the
+one thing there that needed a new endpoint — `GET /api/teams/{id}/members` is
+member-only and answers `404` to an administrator who is not in the team, which is
+the rule and not an oversight, so the page reads `GET /api/admin/teams/{id}` instead.
+An administrator still sees none of that team's incidents, alerts or rota.
+
+A name in the user list opens **that person's page**, at `/admin/users/{id}`: their
+email and when they joined, where their notifications go, whether they are an
+administrator, whether the account is disabled, the teams they are in with their
+role in each, a password field for a first or forgotten one, and deletion. It is
+the one place membership is edited from the person's side — the Team tab answers
+"who is in this team", and answering "which teams is this person in" there means
+visiting each team in turn.
diff --git a/internal/api/alertmanager.go b/internal/api/alertmanager.go
index 071684b..7e3cc39 100644
--- a/internal/api/alertmanager.go
+++ b/internal/api/alertmanager.go
@@ -87,7 +87,7 @@ func handleIntegrationWebhook(db *sql.DB, notify NotifyConfig) http.HandlerFunc
respond(w, http.StatusUnauthorized, errResp("unknown integration key"))
return
}
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
receiveWebhook(w, r, db, notify, src)
diff --git a/internal/api/alerts.go b/internal/api/alerts.go
index a2b0b2b..5bdecf1 100644
--- a/internal/api/alerts.go
+++ b/internal/api/alerts.go
@@ -90,7 +90,7 @@ func handleListAlerts(db *sql.DB) http.HandlerFunc {
fmt.Sprintf("%s WHERE %s ORDER BY a.received_at DESC LIMIT %s", alertSelectFrom, clause, args.add(limit)),
args.all()...)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -99,7 +99,7 @@ func handleListAlerts(db *sql.DB) http.HandlerFunc {
for rows.Next() {
a, err := scanAlert(rows)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
alerts = append(alerts, a)
@@ -121,7 +121,7 @@ func handleGetAlert(db *sql.DB) http.HandlerFunc {
return
}
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, a)
diff --git a/internal/api/api_test.go b/internal/api/api_test.go
index ec26d42..bae9c57 100644
--- a/internal/api/api_test.go
+++ b/internal/api/api_test.go
@@ -78,6 +78,10 @@ func newTSWith(t *testing.T, deadman api.DeadmanConfig, cfg api.NotifyConfig, co
s := &ts{Server: srv, key: key, db: database, notify: cfg, deadman: deadman}
+ // A fresh install has no team, so the tests that want "the" team make it
+ // here: it is id 1, owned by the admin, which is what defaultTeam names.
+ decode(t, s.req(t, http.MethodPost, "/api/teams", map[string]string{"name": "Default"}), &struct{}{})
+
var integration struct {
Key string `json:"key"`
}
@@ -768,7 +772,7 @@ func TestWebhook_IgnoresOutOfOrderRetry(t *testing.T) {
// An expiry resolve writes ends_at as an upper bound, not an observed end: an
// Alertmanager watermark already on the row is preserved, and a row that never
// carried one is stamped at sweep time. Clients are told to read it that way —
-// see "resolution_source says how much to trust ends_at" in the README.
+// see "resolution_source says how much to trust ends_at" in docs/api.md.
func TestExpiry_EndsAtIsUpperBound(t *testing.T) {
s := newTS(t)
@@ -813,7 +817,7 @@ func TestExpiry_EndsAtIsUpperBound(t *testing.T) {
//
// received_at is documented as a public liveness signal, so these lock the
// behaviour clients are told they may rely on. See "received_at is a liveness
-// heartbeat" in the README and the comment on models.Alert.ReceivedAt.
+// heartbeat" in docs/api.md and the comment on models.Alert.ReceivedAt.
// ---------------------------------------------------------------------------
// The heartbeat itself: an unchanged firing notification — what Alertmanager
@@ -875,3 +879,38 @@ func TestStats_ByDayReturnsSevenSlots(t *testing.T) {
t.Errorf("expected 7 day slots, got %d", len(slots))
}
}
+
+// Two simultaneous bootstraps on an empty install must not both win.
+func TestBootstrap_ConcurrentCallsCreateOneAdmin(t *testing.T) {
+ database := newTestDB(t)
+ srv := httptest.NewServer(api.NewRouter(database, api.NotifyConfig{}, testConfig(), "test"))
+ t.Cleanup(srv.Close)
+
+ const n = 8
+ codes := make(chan int, n)
+ for i := 0; i < n; i++ {
+ go func(i int) {
+ body, _ := json.Marshal(map[string]string{"username": fmt.Sprintf("u%d", i), "email": fmt.Sprintf("u%d@x.com", i)})
+ resp, err := http.Post(srv.URL+"/api/bootstrap", "application/json", bytes.NewReader(body))
+ if err != nil {
+ codes <- 0
+ return
+ }
+ resp.Body.Close()
+ codes <- resp.StatusCode
+ }(i)
+ }
+ created := 0
+ for i := 0; i < n; i++ {
+ if <-codes == http.StatusCreated {
+ created++
+ }
+ }
+ var users int
+ if err := database.QueryRow("SELECT COUNT(*) FROM users").Scan(&users); err != nil {
+ t.Fatal(err)
+ }
+ if created != 1 || users != 1 {
+ t.Errorf("expected exactly one bootstrap to win, got %d created and %d users", created, users)
+ }
+}
diff --git a/internal/api/archiver.go b/internal/api/archiver.go
index 6759596..b99df7c 100644
--- a/internal/api/archiver.go
+++ b/internal/api/archiver.go
@@ -154,9 +154,7 @@ func expireStale(ctx context.Context, db *sql.DB, staleAfter time.Duration, skip
}
// staleAlertIDs reads the ids in one go and closes the cursor before the caller
-// writes. Under SQLite's single connection an open read would have blocked the
-// update outright; with a pool it is no longer a deadlock, but reading the set
-// first still keeps the write off a cursor the same transaction is walking.
+// writes, which keeps the write off a cursor the same transaction is walking.
func staleAlertIDs(ctx context.Context, db *sql.DB, now time.Time, staleAfter time.Duration) ([]int64, error) {
rows, err := db.QueryContext(ctx, `
SELECT id FROM alerts
diff --git a/internal/api/auth.go b/internal/api/auth.go
index 804f4d5..38feff9 100644
--- a/internal/api/auth.go
+++ b/internal/api/auth.go
@@ -10,6 +10,7 @@ import (
"strconv"
"strings"
"sync"
+ "sync/atomic"
"time"
"github.com/go-chi/chi/v5"
@@ -28,6 +29,9 @@ const (
// sessionTouchEvery bounds how often a request may slide the expiry.
sessionTouchEvery = time.Hour
+ // keyTouchEvery is the same bound for an API key's last_used_at.
+ keyTouchEvery = 5 * time.Minute
+
minPasswordLen = 10
// maxPasswordLen is bcrypt's limit; it rejects longer input outright.
maxPasswordLen = 72
@@ -126,14 +130,32 @@ func purgeRateLimits(ctx context.Context, db *sql.DB) {
}
}
+// trustedProxies is how many X-Forwarded-For hops clientAddr trusts. Set once
+// by NewRouter from config.
+var trustedProxies atomic.Int64
+
// clientAddr is the address a login is counted against. Behind the gateway
-// RemoteAddr is the gateway itself, so the first X-Forwarded-For hop is used
-// when present. It can be forged, but only to dodge the address limit; the
-// per-username limit does not depend on it.
+// RemoteAddr is the gateway itself, so the client address is read from
+// X-Forwarded-For, counting trustedProxies entries from the right: each trusted
+// proxy appends the address it saw, so the entries to the left of those are
+// client-supplied and could be forged to dodge the limit.
func clientAddr(r *http.Request) string {
- if xff := r.Header.Get("X-Forwarded-For"); xff != "" {
- first, _, _ := strings.Cut(xff, ",")
- return strings.TrimSpace(first)
+ if n := int(trustedProxies.Load()); n > 0 {
+ var hops []string
+ for _, v := range r.Header.Values("X-Forwarded-For") {
+ for _, h := range strings.Split(v, ",") {
+ if h = strings.TrimSpace(h); h != "" {
+ hops = append(hops, h)
+ }
+ }
+ }
+ if len(hops) > 0 {
+ i := len(hops) - n
+ if i < 0 {
+ i = 0
+ }
+ return hops[i]
+ }
}
host, _, err := net.SplitHostPort(r.RemoteAddr)
if err != nil {
@@ -241,7 +263,7 @@ func handleLogin(db *sql.DB, limiter *loginLimiter, publicURL string) http.Handl
"SELECT id, password_hash FROM users WHERE username = $1", username,
).Scan(&userID, &hash)
if err != nil && !errors.Is(err, sql.ErrNoRows) {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
@@ -258,13 +280,13 @@ func handleLogin(db *sql.DB, limiter *loginLimiter, publicURL string) http.Handl
limiter.clear(r.Context(), userKey)
if err := startSession(w, r, db, userID, publicURL); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
user, err := fetchUser(r.Context(), db, userID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, meResponse{User: user, HasPassword: true})
@@ -320,7 +342,7 @@ func handleMe(db *sql.DB) http.HandlerFunc {
}
user, err := fetchUser(r.Context(), db, caller.ID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
var hash sql.NullString
@@ -332,7 +354,7 @@ func handleMe(db *sql.DB) http.HandlerFunc {
// transient database problem, not a missing user — worth a 500
// rather than silently answering "no password, not dismissed",
// which a client would otherwise take at face value.
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, meResponse{
@@ -384,7 +406,7 @@ func handleSetPassword(db *sql.DB) http.HandlerFunc {
return
}
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
@@ -397,30 +419,30 @@ func handleSetPassword(db *sql.DB) http.HandlerFunc {
hash, err := hashPassword(req.Password)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
tx, err := db.BeginTx(r.Context(), nil)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer tx.Rollback()
if _, err := tx.ExecContext(r.Context(),
"UPDATE users SET password_hash = $1 WHERE id = $2", hash, id); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
keep, _ := sessionFromContext(r.Context()) // zero when changed with an API key
if _, err := tx.ExecContext(r.Context(),
"DELETE FROM sessions WHERE user_id = $1 AND id != $2", id, keep); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if err := tx.Commit(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
w.WriteHeader(http.StatusNoContent)
diff --git a/internal/api/caller.go b/internal/api/caller.go
index 95501de..a656445 100644
--- a/internal/api/caller.go
+++ b/internal/api/caller.go
@@ -2,7 +2,6 @@ package api
import (
"context"
- "fmt"
"git.ryuvia.com/niklas/terdut-server/internal/models"
)
@@ -61,7 +60,7 @@ func (c Caller) IsAdmin() bool {
// by being a human (and becomes its owner as a side effect), an
// instance-scoped service account creates one with no human owner at all;
// the two paths are not interchangeable, so this predicate must not also
-// admit a human admin the way MayActAsInstanceAdmin deliberately does.
+// admit a human admin the way an administrator check would.
func (c Caller) IsInstanceServiceAccount() bool {
return c.sa != nil && c.sa.scope == models.ServiceAccountScopeInstance
}
@@ -110,24 +109,6 @@ func (c Caller) ServiceAccountName() (string, bool) {
return c.sa.name, true
}
-// Identity is a stable, log/audit-facing string distinguishing a human
-// caller from a service account — "user:42" or "service-account:7". Not
-// wired into any database column — incidents.go's acknowledged_by/
-// incident_events.user_id use AsHuman()/ServiceAccountID() directly against
-// the parallel *_service_account_id columns (migration 015) instead, since a
-// column needs the id, not this rendered string. assigned_to stays
-// human-only and out of scope (terdut-server#25's follow-up).
-func (c Caller) Identity() string {
- switch {
- case c.user != nil:
- return fmt.Sprintf("user:%d", c.user.ID)
- case c.sa != nil:
- return fmt.Sprintf("service-account:%d", c.sa.id)
- default:
- return "unknown"
- }
-}
-
func callerFromContext(ctx context.Context) (Caller, bool) {
c, ok := ctx.Value(ctxCaller).(Caller)
return c, ok
diff --git a/internal/api/deadman.go b/internal/api/deadman.go
index 70bbd01..6b813f0 100644
--- a/internal/api/deadman.go
+++ b/internal/api/deadman.go
@@ -65,28 +65,6 @@ func (m DeadmanMatcher) matches(labels map[string]string) bool {
return true
}
-// DeadmanConfig is the server-wide default a team's switches are seeded from:
-// the environment's matchers, timeout and severity. Switches themselves are rows
-// of a team's own — see DeadmanSwitch — and this is only how a fresh install
-// starts out.
-type DeadmanConfig struct {
- Matchers []DeadmanMatcher
-
- // Timeout is how long a matched alert may go without a refreshing webhook
- // before it is declared dead. It must be shorter than Alertmanager's
- // repeat_interval for the heartbeat's route, which is what refreshes it.
- // Zero disables dead man's switch handling entirely.
- Timeout time.Duration
-
- // Severity is the severity every dead man's switch incident opens at. These
- // incidents have no member alerts to derive one from, and the heartbeat's
- // own severity label is meaningless — Watchdog ships as "none".
- Severity string
-}
-
-// enabled reports whether there is anything to watch.
-func (c DeadmanConfig) enabled() bool { return c.Timeout > 0 && len(c.Matchers) > 0 }
-
// DeadmanSwitch inverts the handling of the alerts it matches: receiving one
// opens nothing, and the absence of one opens an incident.
//
@@ -160,46 +138,6 @@ func parseDeadmanMatcher(entry string) (DeadmanMatcher, error) {
return m, nil
}
-// ParseDeadmanConfig reads the matcher list from its configured form:
-// ";" separates matchers, and each is parsed as parseDeadmanMatcher does.
-//
-// A malformed or alertname-less entry is dropped rather than fatal, following
-// config.duration's rule that one bad tuning knob should not take the server
-// down. Silence would be worse here than elsewhere, though — a typo that
-// disarms the switch is exactly the failure this feature exists to catch — so
-// the matchers that survived are logged.
-func ParseDeadmanConfig(matchers string, timeout time.Duration, severity string) DeadmanConfig {
- cfg := DeadmanConfig{Timeout: timeout, Severity: severity}
-
- for _, entry := range strings.Split(matchers, ";") {
- entry = strings.TrimSpace(entry)
- if entry == "" {
- continue
- }
- m, err := parseDeadmanMatcher(entry)
- if err != nil {
- log.Printf("deadman: ignoring matcher %q: %v", entry, err)
- continue
- }
- cfg.Matchers = append(cfg.Matchers, m)
- }
-
- switch {
- case timeout <= 0:
- log.Print("deadman: disabled (timeout is zero)")
- case len(cfg.Matchers) == 0:
- log.Print("deadman: disabled (no usable matchers)")
- default:
- rendered := make([]string, 0, len(cfg.Matchers))
- for _, m := range cfg.Matchers {
- rendered = append(rendered, m.String())
- }
- log.Printf("deadman: default for new teams: %s, timeout %s, severity %s",
- strings.Join(rendered, "; "), timeout, severity)
- }
- return cfg
-}
-
// deadmanAlert is one heartbeat: the alert row carrying its last sighting, and
// the switch that claimed it.
type deadmanAlert struct {
@@ -490,56 +428,6 @@ func deadmanSets(ctx context.Context, db *sql.DB) (map[int64]deadmanSet, error)
return scanDeadmanSwitches(rows)
}
-// deadmanSeededKey is the settings row that records the environment defaults
-// were handed out. Without it, a team that deleted its last switch would get
-// the default back on the next restart.
-const deadmanSeededKey = "deadman_seeded"
-
-// SeedDeadmanConfigs gives every team the server's environment defaults as
-// switches, exactly once per install, so a fresh install watches Watchdog
-// without anybody setting it up.
-//
-// Once seeded it never runs again: a team's switches are its own, and a redeploy
-// must not quietly put the environment's value back over an owner's edit or
-// deletion. Installs that upgraded from per-team configuration were already
-// seeded, which migration 009 records.
-//
-// A team created after that gets none and watches nothing until its owner says
-// otherwise. That is deliberate: inheriting an install-wide heartbeat would page
-// a new team about a source it has never heard of, and a switch nobody chose is
-// the kind that gets muted rather than fixed.
-func SeedDeadmanConfigs(ctx context.Context, db *sql.DB, cfg DeadmanConfig) error {
- if !cfg.enabled() {
- return nil
- }
-
- tx, err := db.BeginTx(ctx, nil)
- if err != nil {
- return err
- }
- defer tx.Rollback() //nolint:errcheck
-
- res, err := tx.ExecContext(ctx,
- "INSERT INTO settings (key, value) VALUES ($1, '1') ON CONFLICT (key) DO NOTHING",
- deadmanSeededKey)
- if err != nil {
- return err
- }
- if n, _ := res.RowsAffected(); n == 0 {
- return nil
- }
-
- for _, m := range cfg.Matchers {
- if _, err := tx.ExecContext(ctx, `
- INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity)
- SELECT id, $1, $1, $2, $3 FROM teams`,
- m.config(), int64(cfg.Timeout.Seconds()), cfg.Severity); err != nil {
- return err
- }
- }
- return tx.Commit()
-}
-
// ---------------------------------------------------------------------------
// Status
// ---------------------------------------------------------------------------
diff --git a/internal/api/deadman_test.go b/internal/api/deadman_test.go
index 4d6d172..485be77 100644
--- a/internal/api/deadman_test.go
+++ b/internal/api/deadman_test.go
@@ -1,7 +1,6 @@
package api_test
import (
- "context"
"net/http"
"strings"
"testing"
@@ -122,8 +121,11 @@ func TestDeadman_MixedGroupExcludesHeartbeat(t *testing.T) {
t.Fatalf("expected 1 incident for the real alert, got %d", got)
}
- var alerts []map[string]any
- decode(t, s.req(t, http.MethodGet, "/api/incidents/1/alerts", nil), &alerts)
+ var incident struct {
+ Alerts []map[string]any `json:"alerts"`
+ }
+ decode(t, s.req(t, http.MethodGet, "/api/incidents/1", nil), &incident)
+ alerts := incident.Alerts
if len(alerts) != 1 {
t.Fatalf("expected 1 member alert, got %d", len(alerts))
}
@@ -717,26 +719,3 @@ func TestDeadman_DeleteIsScopedToTheTeam(t *testing.T) {
t.Errorf("expected no switches, got %d", got)
}
}
-
-// The environment's defaults are handed out once and then belong to the teams.
-func TestDeadman_SeedRunsOnce(t *testing.T) {
- s := newTS(t)
- cfg := api.ParseDeadmanConfig("alertname=Watchdog", time.Hour, "critical")
-
- if err := api.SeedDeadmanConfigs(context.Background(), s.db, cfg); err != nil {
- t.Fatalf("seed: %v", err)
- }
- if got := len(listSwitches(t, s)); got != 1 {
- t.Fatalf("the first seed should add the default, got %d switches", got)
- }
-
- // The owner deletes it; a restart must not put it back.
- id := int64(listSwitches(t, s)[0]["id"].(float64))
- s.req(t, http.MethodDelete, "/api/teams/"+defaultTeam+"/deadman/switches/"+id64(id), nil).Body.Close()
- if err := api.SeedDeadmanConfigs(context.Background(), s.db, cfg); err != nil {
- t.Fatalf("seed again: %v", err)
- }
- if got := len(listSwitches(t, s)); got != 0 {
- t.Errorf("a second seed resurrected %d switch(es)", got)
- }
-}
diff --git a/internal/api/device.go b/internal/api/device.go
index d694b50..120dad9 100644
--- a/internal/api/device.go
+++ b/internal/api/device.go
@@ -81,7 +81,7 @@ func handleDeviceStart(db *sql.DB, limiter *loginLimiter, publicURL string) http
deviceCode, deviceHash, err := randomToken()
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
@@ -105,7 +105,7 @@ func handleDeviceStart(db *sql.DB, limiter *loginLimiter, publicURL string) http
}
if err != nil {
log.Printf("device login: start: %v", err)
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
@@ -160,7 +160,7 @@ func handleDeviceDecision(db *sql.DB, approve bool) http.HandlerFunc {
WHERE user_code = $3 AND status = 'pending' AND expires_at > $4`,
status, caller.ID, code, time.Now().Unix())
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -190,7 +190,7 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht
tx, err := db.BeginTx(r.Context(), nil)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer tx.Rollback() //nolint:errcheck
@@ -206,7 +206,7 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht
return
}
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
@@ -226,11 +226,11 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht
}
if _, err := tx.ExecContext(r.Context(),
"UPDATE device_logins SET last_polled_at = $1 WHERE device_hash = $2", now.Unix(), hash); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if err := tx.Commit(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusAccepted, map[string]string{"status": "pending"})
@@ -240,7 +240,7 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht
// Approved. Single use: the row goes before the session is made, so two
// racing polls cannot both be given one.
if _, err := tx.ExecContext(r.Context(), "DELETE FROM device_logins WHERE device_hash = $1", hash); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
var disabled, sso bool
@@ -252,7 +252,7 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht
return
}
if err := tx.Commit(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if disabled {
@@ -269,12 +269,12 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht
}
if err := startSessionCapped(w, r, db, userID.Int64, publicURL, maxAge); err != nil {
log.Printf("device login: start session: %v", err)
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
user, err := fetchUser(r.Context(), db, userID.Int64)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, meResponse{User: user, HasPassword: false})
diff --git a/internal/api/escalation.go b/internal/api/escalation.go
index 59c3a84..ed3275c 100644
--- a/internal/api/escalation.go
+++ b/internal/api/escalation.go
@@ -3,6 +3,7 @@ package api
import (
"context"
"database/sql"
+ "errors"
"log"
"net/http"
"strconv"
@@ -343,12 +344,12 @@ func handleGetEscalation(db *sql.DB) http.HandlerFunc {
policy, err := loadEscalationPolicy(r.Context(), db, teamID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
view, err := escalationStatus(r.Context(), db, teamID, policy)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, view)
@@ -543,6 +544,10 @@ type escalationLevelJSON struct {
type escalationTargetJSON struct {
Kind string `json:"kind"`
UserID *int64 `json:"user_id,omitempty"`
+ // Username is accepted in place of user_id on a PUT, and resolved to the
+ // id before anything is stored. It is never returned: the stored form is
+ // the id, which survives a rename.
+ Username string `json:"username,omitempty"`
}
type escalationJSON struct {
@@ -600,6 +605,27 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc {
return
}
for i, l := range req.Levels {
+ for j, t := range l.Targets {
+ if t.Username == "" {
+ continue
+ }
+ if t.Kind != "user" || t.UserID != nil {
+ respond(w, http.StatusBadRequest, errResp("username belongs on a user target, instead of user_id"))
+ return
+ }
+ var id int64
+ err := db.QueryRowContext(r.Context(), "SELECT id FROM users WHERE username = $1", t.Username).Scan(&id)
+ if errors.Is(err, sql.ErrNoRows) {
+ respond(w, http.StatusBadRequest, errResp("unknown user "+strconv.Quote(t.Username)))
+ return
+ }
+ if err != nil {
+ serverError(w, r, err)
+ return
+ }
+ req.Levels[i].Targets[j].UserID = &id
+ req.Levels[i].Targets[j].Username = ""
+ }
if l.TimeoutSeconds <= 0 {
respond(w, http.StatusBadRequest, errResp("every level needs a timeout"))
return
@@ -632,7 +658,7 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc {
tx, err := db.BeginTx(r.Context(), nil)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer tx.Rollback() //nolint:errcheck
@@ -645,13 +671,13 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc {
fallback_topic = excluded.fallback_topic,
updated_at = excluded.updated_at`,
teamID, req.RepeatCount, req.FallbackTopic); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
// The levels are replaced, not merged; the cascade takes the targets.
if _, err := tx.ExecContext(r.Context(),
"DELETE FROM escalation_levels WHERE team_id = $1", teamID); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
@@ -661,7 +687,7 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc {
INSERT INTO escalation_levels (team_id, position, timeout_seconds)
VALUES ($1, $2, $3) RETURNING id`,
teamID, int64(i+1), l.TimeoutSeconds).Scan(&levelID); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
for _, t := range l.Targets {
@@ -676,13 +702,13 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc {
}
if err := tx.Commit(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
policy, err := loadEscalationPolicy(r.Context(), db, teamID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, escalationResponse(policy, teamID))
diff --git a/internal/api/escalation_test.go b/internal/api/escalation_test.go
index 39a428a..e36f96f 100644
--- a/internal/api/escalation_test.go
+++ b/internal/api/escalation_test.go
@@ -502,3 +502,45 @@ func TestEscalation_StatusWithoutALadder(t *testing.T) {
t.Errorf("a team with no ladder should read as empty, got %+v", v)
}
}
+
+// A user target may name the person instead of carrying an id; the server
+// resolves it and stores the id.
+func TestEscalation_UserTargetByUsername(t *testing.T) {
+ s := newTS(t)
+ id := teamUser(t, s, "alice", "")
+
+ put := func(username string) *http.Response {
+ return s.req(t, http.MethodPut, "/api/teams/"+defaultTeam+"/escalation", map[string]any{
+ "repeat_count": 0,
+ "levels": []map[string]any{{
+ "timeout_seconds": 300,
+ "targets": []map[string]any{{"kind": "user", "username": username}},
+ }},
+ })
+ }
+
+ resp := put("alice")
+ resp.Body.Close()
+ if resp.StatusCode >= 300 {
+ t.Fatalf("PUT by username: %d", resp.StatusCode)
+ }
+ var got struct {
+ Levels []struct {
+ Targets []struct {
+ UserID *int64 `json:"user_id"`
+ Username string `json:"username"`
+ } `json:"targets"`
+ } `json:"levels"`
+ }
+ decode(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/escalation", nil), &got)
+ if len(got.Levels) != 1 || len(got.Levels[0].Targets) != 1 ||
+ got.Levels[0].Targets[0].UserID == nil || *got.Levels[0].Targets[0].UserID != id {
+ t.Errorf("expected the target stored as user %d, got %+v", id, got)
+ }
+
+ bad := put("nobody")
+ bad.Body.Close()
+ if bad.StatusCode != http.StatusBadRequest {
+ t.Errorf("unknown username should be a 400, got %d", bad.StatusCode)
+ }
+}
diff --git a/internal/api/export_test.go b/internal/api/export_test.go
new file mode 100644
index 0000000..9928297
--- /dev/null
+++ b/internal/api/export_test.go
@@ -0,0 +1,31 @@
+package api
+
+import (
+ "strings"
+ "time"
+)
+
+// DeadmanConfig is a test fixture only: a set of matchers with one timeout and
+// severity, turned into switches over the API by the test helpers. Production
+// has no server-wide default any more -- switches belong to teams.
+type DeadmanConfig struct {
+ Matchers []DeadmanMatcher
+ Timeout time.Duration
+ Severity string
+}
+
+// ParseDeadmanConfig reads a ";"-separated matcher list the way the removed
+// environment variable did, dropping malformed entries.
+func ParseDeadmanConfig(matchers string, timeout time.Duration, severity string) DeadmanConfig {
+ cfg := DeadmanConfig{Timeout: timeout, Severity: severity}
+ for _, entry := range strings.Split(matchers, ";") {
+ entry = strings.TrimSpace(entry)
+ if entry == "" {
+ continue
+ }
+ if m, err := parseDeadmanMatcher(entry); err == nil {
+ cfg.Matchers = append(cfg.Matchers, m)
+ }
+ }
+ return cfg
+}
diff --git a/internal/api/helpers.go b/internal/api/helpers.go
index 4ea7dcd..2ae751e 100644
--- a/internal/api/helpers.go
+++ b/internal/api/helpers.go
@@ -5,6 +5,7 @@ import (
"database/sql"
"encoding/json"
"errors"
+ "github.com/go-chi/chi/v5"
"log"
"net/http"
"strconv"
@@ -17,8 +18,7 @@ import (
// sqlArgs accumulates query arguments and hands back the placeholder for each.
//
// Postgres numbers its placeholders, so a dynamically assembled WHERE clause has
-// to keep its $1, $2, … in step with the order of the values — which SQLite's
-// positional `?` did for free. Handing out the placeholder and storing the value
+// to keep its $1, $2, … in step with the order of the values. Handing out the placeholder and storing the value
// in one call is what keeps them in step: a filter can be added, removed or
// reordered without renumbering anything by hand.
type sqlArgs struct{ vals []any }
@@ -31,8 +31,7 @@ func (a *sqlArgs) add(v any) string {
// addList stores every value and returns their placeholders as "$1, $2, …",
// ready to drop into an IN (…) clause. Returns an empty string for no values,
-// which no caller should reach: `IN ()` is a syntax error in Postgres as it was
-// in SQLite, so callers check for an empty set before building the query.
+// which no caller should reach: `IN ()` is a syntax error in Postgres, so callers check for an empty set before building the query.
func (a *sqlArgs) addList(vs []any) string {
parts := make([]string, len(vs))
for i, v := range vs {
@@ -45,7 +44,7 @@ func (a *sqlArgs) addList(vs []any) string {
func (a *sqlArgs) all() []any { return a.vals }
// nowEpoch is the SQL expression for "now, as unix seconds", matching how every
-// timestamp in this schema is stored. SQLite spelled it unixepoch().
+// timestamp in this schema is stored.
//
// FLOOR, not a bare cast: EXTRACT returns fractional seconds and casting to
// bigint rounds half up, so a row written at .6 of a second would claim a
@@ -56,11 +55,9 @@ const nowEpoch = "FLOOR(EXTRACT(EPOCH FROM now()))::bigint"
// isUniqueViolation reports whether err is a broken unique constraint, which
// callers turn into 409 Conflict rather than 500.
//
-// Postgres reports it as SQLSTATE 23505 on a typed error; the SQLite driver this
-// replaced only put "UNIQUE constraint failed" in the message, which is why the
-// check used to be a substring match. Matching the code means a renamed
-// constraint or a translated message cannot quietly turn a conflict back into a
-// 500.
+// Postgres reports it as SQLSTATE 23505 on a typed error. Matching the code
+// means a renamed constraint or a translated message cannot quietly turn a
+// conflict back into a 500.
func isUniqueViolation(err error) bool {
var pgErr *pgconn.PgError
return errors.As(err, &pgErr) && pgErr.Code == pgerrcode.UniqueViolation
@@ -72,6 +69,18 @@ func respond(w http.ResponseWriter, status int, v any) {
json.NewEncoder(w).Encode(v)
}
+// serverError answers 500 and logs why. The response stays opaque, so the log
+// line is the only record of what failed.
+func serverError(w http.ResponseWriter, r *http.Request, err error) {
+ // The route pattern, not the path: two routes carry a credential in it.
+ route := r.URL.Path
+ if rc := chi.RouteContext(r.Context()); rc != nil && rc.RoutePattern() != "" {
+ route = rc.RoutePattern()
+ }
+ log.Printf("%s %s: %v", strconv.Quote(r.Method), strconv.Quote(route), err)
+ respond(w, http.StatusInternalServerError, errResp("internal error"))
+}
+
// maxBodyBytes caps an ordinary JSON request body. 1 MiB is far more than any
// endpoint below needs — it exists so an unauthenticated caller (signup,
// login, bootstrap) can't make the server buffer an arbitrarily large body
diff --git a/internal/api/incidents.go b/internal/api/incidents.go
index 19c12a1..50cc9ac 100644
--- a/internal/api/incidents.go
+++ b/internal/api/incidents.go
@@ -37,7 +37,7 @@ func handleListClusters(db *sql.DB) http.HandlerFunc {
label, strings.Join(where, " AND ")),
args.all()...)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -46,13 +46,13 @@ func handleListClusters(db *sql.DB) http.HandlerFunc {
for rows.Next() {
var v string
if err := rows.Scan(&v); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
clusters = append(clusters, v)
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, clusters)
@@ -138,7 +138,7 @@ func handleListIncidents(db *sql.DB) http.HandlerFunc {
incidentSelectFrom, strings.Join(where, " AND "), order, args.add(limit)),
args.all()...)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -147,13 +147,13 @@ func handleListIncidents(db *sql.DB) http.HandlerFunc {
for rows.Next() {
i, err := scanIncident(rows)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
incidents = append(incidents, i)
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, incidents)
@@ -172,35 +172,17 @@ func handleGetIncident(db *sql.DB) http.HandlerFunc {
return
}
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if inc.Alerts, err = incidentAlerts(r, db, id); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, inc)
}
}
-func handleIncidentAlerts(db *sql.DB) http.HandlerFunc {
- return func(w http.ResponseWriter, r *http.Request) {
- id, ok := incidentIDParam(w, r, db)
- if !ok {
- return
- }
- if !incidentExists(w, r, db, id) {
- return
- }
- alerts, err := incidentAlerts(r, db, id)
- if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
- return
- }
- respond(w, http.StatusOK, alerts)
- }
-}
-
func handleIncidentTimeline(db *sql.DB) http.HandlerFunc {
return func(w http.ResponseWriter, r *http.Request) {
id, ok := incidentIDParam(w, r, db)
@@ -225,7 +207,7 @@ func handleIncidentTimeline(db *sql.DB) http.HandlerFunc {
WHERE e.incident_id = $1
ORDER BY e.created_at ASC, e.id ASC`, id)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -239,14 +221,14 @@ func handleIncidentTimeline(db *sql.DB) http.HandlerFunc {
&e.ActorUserID, &e.ActorUsername,
&e.ActorServiceAccountID, &e.ActorServiceAccountName,
&e.AlertID, &e.Detail, &ts); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
e.CreatedAt = time.Unix(ts, 0).UTC()
events = append(events, e)
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, events)
@@ -262,7 +244,7 @@ func handleIncidentAcknowledge(db *sql.DB) http.HandlerFunc {
userID, saID := callerActorIDs(r.Context())
acked, err := acknowledgeIncidentAs(r.Context(), db, id, userID, saID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if !acked {
@@ -272,7 +254,7 @@ func handleIncidentAcknowledge(db *sql.DB) http.HandlerFunc {
// (acknowledged) already holds.
inc, err := fetchIncident(r.Context(), db, id)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if inc.Status == "resolved" {
@@ -300,7 +282,7 @@ func handleIncidentUnacknowledge(db *sql.DB) http.HandlerFunc {
return
}
if err := logEvent(r.Context(), db, id, evUnacknowledged, userID, saID, nil, nil); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
w.WriteHeader(http.StatusNoContent)
@@ -335,16 +317,16 @@ func handleIncidentResolve(db *sql.DB) http.HandlerFunc {
}
// A person closing an incident is the clearest possible "I have this".
if err := stopEscalation(r.Context(), db, id); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if err := logEvent(r.Context(), db, id, evResolved, userID, saID, nil, nil); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if req.Resolution != "" {
if err := logEvent(r.Context(), db, id, evResolutionNote, userID, saID, nil, &req.Resolution); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
}
@@ -385,7 +367,7 @@ func handleIncidentAssign(db *sql.DB) http.HandlerFunc {
// the actor_* columns.
actorUserID, actorSAID := callerActorIDs(r.Context())
if err := logAssignedEvent(r.Context(), db, id, req.UserID, actorUserID, actorSAID); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respondIncident(w, r, db, id)
@@ -443,7 +425,7 @@ func handleIncidentSnooze(db *sql.DB) http.HandlerFunc {
}
detail := until.UTC().Format(time.RFC3339)
if err := logEvent(r.Context(), db, id, evSnoozed, userID, saID, nil, &detail); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respondIncident(w, r, db, id)
@@ -462,7 +444,7 @@ func handleIncidentUnsnooze(db *sql.DB) http.HandlerFunc {
return
}
if err := logEvent(r.Context(), db, id, evUnsnoozed, userID, saID, nil, nil); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
w.WriteHeader(http.StatusNoContent)
@@ -478,7 +460,7 @@ func handleIncidentArchive(db *sql.DB) http.HandlerFunc {
res, err := db.ExecContext(r.Context(),
"UPDATE incidents SET archived_at = "+nowEpoch+" WHERE id = $1", id)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -487,7 +469,7 @@ func handleIncidentArchive(db *sql.DB) http.HandlerFunc {
}
userID, saID := callerActorIDs(r.Context())
if err := logEvent(r.Context(), db, id, evArchived, userID, saID, nil, nil); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respondIncident(w, r, db, id)
@@ -503,7 +485,7 @@ func handleIncidentUnarchive(db *sql.DB) http.HandlerFunc {
res, err := db.ExecContext(r.Context(),
"UPDATE incidents SET archived_at = NULL WHERE id = $1", id)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -512,7 +494,7 @@ func handleIncidentUnarchive(db *sql.DB) http.HandlerFunc {
}
userID, saID := callerActorIDs(r.Context())
if err := logEvent(r.Context(), db, id, evUnarchived, userID, saID, nil, nil); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
w.WriteHeader(http.StatusNoContent)
@@ -557,7 +539,7 @@ func handleCreateNote(db *sql.DB) http.HandlerFunc {
VALUES ($1, $2, $3, $4, $5, $6)
RETURNING id`, id, noteType, userID, saID, req.Content, now.Unix()).Scan(&eventID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
@@ -598,7 +580,7 @@ func handleDeleteNote(db *sql.DB) http.HandlerFunc {
AND (user_id = $5 OR service_account_id = $6)`,
eventID, id, evNote, evResolutionNote, userID, saID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -652,7 +634,7 @@ func incidentExists(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64
func updateOpenIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64, query string, args ...any) bool {
res, err := db.ExecContext(r.Context(), query, args...)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return false
}
if n, _ := res.RowsAffected(); n > 0 {
@@ -668,7 +650,7 @@ func updateOpenIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id i
func respondIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64) {
inc, err := fetchIncident(r.Context(), db, id)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, inc)
diff --git a/internal/api/incidents_test.go b/internal/api/incidents_test.go
index 6fbd466..4dc23d7 100644
--- a/internal/api/incidents_test.go
+++ b/internal/api/incidents_test.go
@@ -794,8 +794,7 @@ func TestStats_Incidents(t *testing.T) {
}
// An empty window is a report of zero, not a failure. SUM over no rows is NULL
-// in Postgres as it was in SQLite, and that used to come back as a 500 the
-// moment every incident was archived — the state a quiet installation settles
+// in Postgres, and that used to come back as a 500 the moment every incident was archived — the state a quiet installation settles
// into.
func TestStats_IncidentsEmptyWindowIsZeroNotAnError(t *testing.T) {
s := newTS(t)
diff --git a/internal/api/middleware.go b/internal/api/middleware.go
index ae0aee7..462b386 100644
--- a/internal/api/middleware.go
+++ b/internal/api/middleware.go
@@ -5,10 +5,14 @@ import (
"crypto/sha256"
"database/sql"
"encoding/hex"
+ "log"
"net/http"
"strings"
"time"
+ "github.com/go-chi/chi/v5"
+ "github.com/go-chi/chi/v5/middleware"
+
"git.ryuvia.com/niklas/terdut-server/internal/config"
"git.ryuvia.com/niklas/terdut-server/internal/models"
)
@@ -106,7 +110,7 @@ func securityHeaders(publicURL string) func(http.Handler) http.Handler {
//
// 403 and not 404: the route exists and the caller is authenticated, they are
// simply not allowed. Hiding the endpoint would buy nothing — every one of them
-// is in the README.
+// is in docs/api.md.
func AdminOnly(next http.Handler) http.Handler {
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
caller, ok := userFromContext(r.Context())
@@ -141,19 +145,23 @@ func requireSelfOrAdmin(w http.ResponseWriter, r *http.Request, targetID int64)
// resolve and has to be caught afterwards.
func apiKeyUser(ctx context.Context, db *sql.DB, token string) (int64, bool) {
var keyID, userID int64
+ var lastUsed sql.NullInt64
err := db.QueryRowContext(ctx,
- `SELECT id, user_id FROM api_keys
+ `SELECT id, user_id, last_used_at FROM api_keys
WHERE key_hash = $1 AND (expires_at IS NULL OR expires_at > $2)`,
hashToken(token), time.Now().Unix(),
- ).Scan(&keyID, &userID)
+ ).Scan(&keyID, &userID, &lastUsed)
if err != nil {
return 0, false
}
- // best-effort; don't fail the request if this update fails
- db.ExecContext(ctx,
- "UPDATE api_keys SET last_used_at = $1 WHERE id = $2",
- time.Now().Unix(), keyID)
+ // best-effort; don't fail the request if this update fails. Throttled like
+ // the session expiry, so a polling client does not write a row per request.
+ if now := time.Now(); !lastUsed.Valid || now.Sub(time.Unix(lastUsed.Int64, 0)) > keyTouchEvery {
+ db.ExecContext(ctx,
+ "UPDATE api_keys SET last_used_at = $1 WHERE id = $2",
+ now.Unix(), keyID)
+ }
return userID, true
}
@@ -206,7 +214,7 @@ func serveAs(w http.ResponseWriter, r *http.Request, next http.Handler, db *sql.
// table with one row per membership.
teams, err := callerMemberships(r.Context(), db, userID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
@@ -222,10 +230,8 @@ func hashToken(token string) string {
return hex.EncodeToString(h[:])
}
-// userFromContext is a thin compatibility wrapper over Caller.AsHuman(), so
-// every call site written before the Caller abstraction (alerts.go,
-// incidents.go, schedule.go, stats.go, and more) needs no change and keeps
-// its exact existing behavior.
+// userFromContext returns the human behind the request, or false for a service
+// account: Caller.AsHuman() on the request's Caller.
func userFromContext(ctx context.Context) (models.User, bool) {
c, _ := callerFromContext(ctx)
return c.AsHuman()
@@ -245,13 +251,13 @@ type serviceAccountPrincipal struct {
func serviceAccountFor(ctx context.Context, db *sql.DB, token string) (serviceAccountPrincipal, bool) {
var sa serviceAccountPrincipal
var keyID int64
- var teamID sql.NullInt64
+ var teamID, lastUsed sql.NullInt64
err := db.QueryRowContext(ctx, `
- SELECT k.id, a.id, a.name, a.scope, a.team_id
+ SELECT k.id, a.id, a.name, a.scope, a.team_id, k.last_used_at
FROM service_account_keys k
JOIN service_accounts a ON a.id = k.service_account_id
WHERE k.key_hash = $1`, hashToken(token),
- ).Scan(&keyID, &sa.id, &sa.name, &sa.scope, &teamID)
+ ).Scan(&keyID, &sa.id, &sa.name, &sa.scope, &teamID, &lastUsed)
if err != nil {
return serviceAccountPrincipal{}, false
}
@@ -260,9 +266,11 @@ func serviceAccountFor(ctx context.Context, db *sql.DB, token string) (serviceAc
}
// best-effort; don't fail the request if this update fails
- db.ExecContext(ctx,
- "UPDATE service_account_keys SET last_used_at = $1 WHERE id = $2",
- time.Now().Unix(), keyID)
+ if now := time.Now(); !lastUsed.Valid || now.Sub(time.Unix(lastUsed.Int64, 0)) > keyTouchEvery {
+ db.ExecContext(ctx,
+ "UPDATE service_account_keys SET last_used_at = $1 WHERE id = $2",
+ now.Unix(), keyID)
+ }
return sa, true
}
@@ -286,10 +294,8 @@ func serveAsServiceAccount(w http.ResponseWriter, r *http.Request, next http.Han
next.ServeHTTP(w, r.WithContext(ctx))
}
-// isInstanceServiceAccount is a thin compatibility wrapper over
-// Caller.IsInstanceServiceAccount(), for call sites outside this package's
-// core predicates (handleCreateTeam, handleCreateServiceAccount) that
-// needed this exact, narrow check before the Caller abstraction existed.
+// isInstanceServiceAccount is Caller.IsInstanceServiceAccount() on the
+// request's Caller.
func isInstanceServiceAccount(ctx context.Context) bool {
c, _ := callerFromContext(ctx)
return c.IsInstanceServiceAccount()
@@ -397,6 +403,13 @@ func requireTeamOwner(w http.ResponseWriter, r *http.Request, teamID int64) bool
if ok && role == models.RoleOwner {
return true
}
+ // The operator's instance-scoped account manages every team's
+ // configuration, which is what lets it use one credential instead of
+ // minting one per team. This is owner reach only: it does not make the
+ // account a member, so it still reads no team's incidents.
+ if c, _ := callerFromContext(r.Context()); c.IsInstanceServiceAccount() {
+ return true
+ }
if caller, _ := userFromContext(r.Context()); caller.IsAdmin {
return true
}
@@ -414,3 +427,26 @@ func sessionFromContext(ctx context.Context) (int64, bool) {
id, ok := ctx.Value(ctxSession).(int64)
return id, ok
}
+
+// requestLogger logs one line per request with the matched route pattern in
+// place of the URL path. Two routes carry a credential in the path (the
+// integration key and the ack token), and chi's stock logger would write it to
+// the log verbatim.
+func requestLogger(next http.Handler) http.Handler {
+ return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
+ start := time.Now()
+ ww := middleware.NewWrapResponseWriter(w, r.ProtoMajor)
+ next.ServeHTTP(ww, r)
+ route := "unmatched"
+ if rc := chi.RouteContext(r.Context()); rc != nil {
+ if p := rc.RoutePattern(); p != "" {
+ route = p
+ }
+ }
+ status := ww.Status()
+ if status == 0 {
+ status = http.StatusOK
+ }
+ log.Printf("%q %q %d %dB %s", r.Method, route, status, ww.BytesWritten(), time.Since(start).Round(time.Millisecond)) // #nosec G706 -- method and route are %q-quoted, the route is a registered pattern, the rest are numbers
+ })
+}
diff --git a/internal/api/notifier.go b/internal/api/notifier.go
index 8f5b621..610e79b 100644
--- a/internal/api/notifier.go
+++ b/internal/api/notifier.go
@@ -449,7 +449,7 @@ func renderNotification(inc models.Incident, n outboxRow, firing int, cfg Notify
// originLabel is the label that says where an alert came from, for a team with
// several Kubernetes clusters behind it. It comes from Prometheus's
// externalLabels and reaches an incident through Alertmanager's group_by; the
-// web UI reads the same label, and the README ("Several clusters, one team")
+// web UI reads the same label, and docs/incidents.md ("Several clusters, one team")
// explains how to set it up.
const originLabel = "cluster"
diff --git a/internal/api/notify_ack.go b/internal/api/notify_ack.go
index 43c4507..86e9e63 100644
--- a/internal/api/notify_ack.go
+++ b/internal/api/notify_ack.go
@@ -62,7 +62,7 @@ func handleNotifyAck(db *sql.DB) http.HandlerFunc {
acked, err := acknowledgeIncident(r.Context(), db, incidentID, userID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if !acked {
@@ -73,7 +73,7 @@ func handleNotifyAck(db *sql.DB) http.HandlerFunc {
// ntfy show a success toast rather than a failure.
inc, err := fetchIncident(r.Context(), db, incidentID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, map[string]any{
diff --git a/internal/api/oidc.go b/internal/api/oidc.go
index fe4baa7..5221cd7 100644
--- a/internal/api/oidc.go
+++ b/internal/api/oidc.go
@@ -130,12 +130,12 @@ func handleOIDCLogin(db *sql.DB, prov *oidc.Provider, limiter *loginLimiter, pub
state, stateHash, err := randomToken()
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
nonce, _, err := randomToken()
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
verifier := oidc.NewVerifier()
@@ -149,7 +149,7 @@ func handleOIDCLogin(db *sql.DB, prov *oidc.Provider, limiter *loginLimiter, pub
INSERT INTO oidc_logins (state_hash, nonce, pkce_verifier, next, expires_at)
VALUES ($1, $2, $3, $4, $5)`,
stateHash, nonce, verifier, next, now.Add(oidcLoginTTL).Unix()); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
diff --git a/internal/api/oidc_teams.go b/internal/api/oidc_teams.go
index 7c8fb18..0bdc8ef 100644
--- a/internal/api/oidc_teams.go
+++ b/internal/api/oidc_teams.go
@@ -31,7 +31,7 @@ func handleGetTeamOIDCGroups(db *sql.DB) http.HandlerFunc {
"SELECT COALESCE(oidc_member_group, ''), COALESCE(oidc_owner_group, '') FROM teams WHERE id = $1",
teamID).Scan(&g.MemberGroup, &g.OwnerGroup)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, g)
@@ -66,7 +66,7 @@ func handleSetTeamOIDCGroups(db *sql.DB) http.HandlerFunc {
oidc_owner_group = NULLIF($2, '')
WHERE id = $3`,
req.MemberGroup, req.OwnerGroup, teamID); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
w.WriteHeader(http.StatusNoContent)
diff --git a/internal/api/operator_key.go b/internal/api/operator_key.go
new file mode 100644
index 0000000..952a0ec
--- /dev/null
+++ b/internal/api/operator_key.go
@@ -0,0 +1,72 @@
+package api
+
+import (
+ "context"
+ "database/sql"
+ "fmt"
+
+ "git.ryuvia.com/niklas/terdut-server/internal/models"
+)
+
+const (
+ // operatorAccountName is the instance-scoped service account the operator
+ // key belongs to.
+ operatorAccountName = "terdut-operator"
+
+ // operatorKeyName names the one key SeedOperatorKey manages on it, so a
+ // rotation replaces that key and leaves any others alone.
+ operatorKeyName = "seed"
+)
+
+// SeedOperatorKey makes key the operator account's credential: it creates the
+// instance-scoped service account if needed and replaces its "seed" key with
+// this one. Idempotent, so every replica can run it at every start, and a
+// rotated key simply wins on the next restart.
+//
+// The key is hashed like any other, so only the caller that generated it ever
+// holds the raw value. An empty key does nothing.
+func SeedOperatorKey(ctx context.Context, db *sql.DB, key string) error {
+ if key == "" {
+ return nil
+ }
+ tx, err := db.BeginTx(ctx, nil)
+ if err != nil {
+ return fmt.Errorf("seed operator key: %w", err)
+ }
+ defer tx.Rollback() //nolint:errcheck
+
+ // Serialise replicas starting together; transaction-scoped, so it needs no
+ // explicit release.
+ if _, err := tx.ExecContext(ctx, "SELECT pg_advisory_xact_lock($1)", operatorKeyLockKey); err != nil {
+ return fmt.Errorf("seed operator key: lock: %w", err)
+ }
+
+ var accountID int64
+ err = tx.QueryRowContext(ctx,
+ "SELECT id FROM service_accounts WHERE name = $1 AND scope = $2",
+ operatorAccountName, models.ServiceAccountScopeInstance).Scan(&accountID)
+ if err == sql.ErrNoRows {
+ err = tx.QueryRowContext(ctx,
+ "INSERT INTO service_accounts (name, scope) VALUES ($1, $2) RETURNING id",
+ operatorAccountName, models.ServiceAccountScopeInstance).Scan(&accountID)
+ }
+ if err != nil {
+ return fmt.Errorf("seed operator key: account: %w", err)
+ }
+
+ if _, err := tx.ExecContext(ctx,
+ "DELETE FROM service_account_keys WHERE service_account_id = $1 AND name = $2",
+ accountID, operatorKeyName); err != nil {
+ return fmt.Errorf("seed operator key: drop old key: %w", err)
+ }
+ if _, err := tx.ExecContext(ctx,
+ "INSERT INTO service_account_keys (service_account_id, key_hash, name) VALUES ($1, $2, $3)",
+ accountID, hashToken(key), operatorKeyName); err != nil {
+ return fmt.Errorf("seed operator key: store key: %w", err)
+ }
+ return tx.Commit()
+}
+
+// operatorKeyLockKey is the transaction-scoped advisory lock SeedOperatorKey
+// holds; distinct from the other lock keys in this package.
+const operatorKeyLockKey int64 = 7265_0010
diff --git a/internal/api/operator_key_test.go b/internal/api/operator_key_test.go
new file mode 100644
index 0000000..2481842
--- /dev/null
+++ b/internal/api/operator_key_test.go
@@ -0,0 +1,134 @@
+package api_test
+
+import (
+ "context"
+ "encoding/json"
+ "net/http"
+ "testing"
+
+ "git.ryuvia.com/niklas/terdut-server/internal/api"
+)
+
+const (
+ operatorKeyOne = "tdsa_operator-key-number-one-0123456789"
+ operatorKeyTwo = "tdsa_operator-key-number-two-0123456789"
+)
+
+func TestSeedOperatorKey_AuthenticatesAsInstanceAccount(t *testing.T) {
+ s := newTS(t)
+ if err := api.SeedOperatorKey(context.Background(), s.db, operatorKeyOne); err != nil {
+ t.Fatal(err)
+ }
+
+ resp := s.reqAs(t, operatorKeyOne, http.MethodPost, "/api/teams", map[string]string{"name": "seeded"})
+ resp.Body.Close()
+ if resp.StatusCode != http.StatusCreated {
+ t.Fatalf("seeded key should create a team, got %d", resp.StatusCode)
+ }
+}
+
+func TestSeedOperatorKey_RotationReplacesAndIsIdempotent(t *testing.T) {
+ s := newTS(t)
+ ctx := context.Background()
+ for _, key := range []string{operatorKeyOne, operatorKeyOne, operatorKeyTwo} {
+ if err := api.SeedOperatorKey(ctx, s.db, key); err != nil {
+ t.Fatal(err)
+ }
+ }
+
+ old := s.reqAs(t, operatorKeyOne, http.MethodGet, "/api/teams", nil)
+ old.Body.Close()
+ if old.StatusCode != http.StatusUnauthorized {
+ t.Errorf("rotated-out key should be refused, got %d", old.StatusCode)
+ }
+ cur := s.reqAs(t, operatorKeyTwo, http.MethodGet, "/api/teams", nil)
+ cur.Body.Close()
+ if cur.StatusCode != http.StatusOK {
+ t.Errorf("current key should work, got %d", cur.StatusCode)
+ }
+
+ var accounts, keys int
+ s.db.QueryRow("SELECT COUNT(*) FROM service_accounts WHERE name = 'terdut-operator'").Scan(&accounts)
+ s.db.QueryRow("SELECT COUNT(*) FROM service_account_keys").Scan(&keys)
+ if accounts != 1 || keys != 1 {
+ t.Errorf("expected one account and one key, got %d and %d", accounts, keys)
+ }
+}
+
+func TestSeedOperatorKey_EmptyKeyDoesNothing(t *testing.T) {
+ s := newTS(t)
+ if err := api.SeedOperatorKey(context.Background(), s.db, ""); err != nil {
+ t.Fatal(err)
+ }
+ var n int
+ s.db.QueryRow("SELECT COUNT(*) FROM service_accounts").Scan(&n)
+ if n != 0 {
+ t.Errorf("expected no service account, got %d", n)
+ }
+}
+
+// The instance account configures a team it did not create, which is what lets
+// the operator hold one credential instead of one per team, yet it is not a
+// member and so reads none of the team's incidents.
+func TestInstanceAccount_ActsAsOwnerOfAnyTeamButIsNoMember(t *testing.T) {
+ s := newTS(t)
+ other := newTeam(t, s, "other")
+ if err := api.SeedOperatorKey(context.Background(), s.db, operatorKeyOne); err != nil {
+ t.Fatal(err)
+ }
+
+ rename := s.reqAs(t, operatorKeyOne, http.MethodPut, "/api/teams/"+id64(other.id), map[string]string{"name": "renamed"})
+ rename.Body.Close()
+ if rename.StatusCode >= 300 {
+ t.Errorf("instance account should rename any team, got %d", rename.StatusCode)
+ }
+
+ // Not a member: the team's queue is not visible to it.
+ var queue []map[string]any
+ decode(t, s.reqAs(t, operatorKeyOne, http.MethodGet, "/api/incidents", nil), &queue)
+ if len(queue) != 0 {
+ t.Errorf("instance account should see no incidents, got %v", queue)
+ }
+}
+
+// external_id lets automation find its own team again after a crash, without
+// trusting a display name.
+func TestCreateTeam_ExternalIDIsIdempotentAndInstanceOnly(t *testing.T) {
+ s := newTS(t)
+ if err := api.SeedOperatorKey(context.Background(), s.db, operatorKeyOne); err != nil {
+ t.Fatal(err)
+ }
+ create := func(name string) (int, map[string]any) {
+ resp := s.reqAs(t, operatorKeyOne, http.MethodPost, "/api/teams",
+ map[string]string{"name": name, "external_id": "ns/platform"})
+ var out map[string]any
+ _ = json.NewDecoder(resp.Body).Decode(&out)
+ resp.Body.Close()
+ return resp.StatusCode, out
+ }
+
+ code, first := create("Platform")
+ if code != http.StatusCreated {
+ t.Fatalf("first create: %d", code)
+ }
+ // Same identity, even under a new display name: the same team comes back.
+ code, again := create("Platform renamed")
+ if code != http.StatusOK || again["id"] != first["id"] {
+ t.Errorf("repeat with the same external_id: want 200 and team %v, got %d %v", first["id"], code, again)
+ }
+
+ // A different identity cannot take the name.
+ resp := s.reqAs(t, operatorKeyOne, http.MethodPost, "/api/teams",
+ map[string]string{"name": "Platform", "external_id": "other/platform"})
+ resp.Body.Close()
+ if resp.StatusCode != http.StatusConflict {
+ t.Errorf("taken name under another external_id: want 409, got %d", resp.StatusCode)
+ }
+
+ // A person cannot set one.
+ resp = s.req(t, http.MethodPost, "/api/teams", map[string]string{"name": "Mine", "external_id": "x/y"})
+ resp.Body.Close()
+ if resp.StatusCode != http.StatusForbidden {
+ t.Errorf("a user setting external_id: want 403, got %d", resp.StatusCode)
+ }
+}
diff --git a/internal/api/router.go b/internal/api/router.go
index 1b24f08..aceecd1 100644
--- a/internal/api/router.go
+++ b/internal/api/router.go
@@ -23,12 +23,13 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version strin
// One limiter each, both process-wide for the life of the router: login
// counts failed passwords, sign-up counts account creation, and mixing the
// two would let a burst of sign-ups lock somebody out of logging in.
+ trustedProxies.Store(int64(cfg.TrustedProxies))
loginLimit := newLoginLimiter(db)
signupLimiter := newLoginLimiter(db)
oidcLimit := newLoginLimiter(db)
r := chi.NewRouter()
- r.Use(middleware.Logger)
+ r.Use(requestLogger)
r.Use(middleware.Recoverer)
r.Use(securityHeaders(notify.PublicURL))
@@ -48,11 +49,7 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version strin
// Alert ingestion. The key in the path says both that the sender may post
// and which team the alerts belong to, which is why it needs no session.
- //
- // This is the only way in. The pre-teams /api/alertmanager/webhook, which
- // took no credential at all, was removed in v0.13.0 once the cluster's
- // Alertmanager had moved onto a key; a sender still posting there gets the
- // JSON 404 every unknown /api path gets.
+ // This is the only way in.
r.Post("/api/integrations/{key}/alertmanager", handleIntegrationWebhook(db, notify))
// Signing up. Both are unauthenticated by necessity: the caller has no
@@ -116,9 +113,7 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version strin
r.Post("/api/users/{id}/api-keys", handleCreateAPIKey(db))
r.Delete("/api/users/{id}/api-keys/{keyID}", handleDeleteAPIKey(db))
- // Administration: who exists, and who is an administrator. Until #3
- // these were open to any authenticated caller, which meant every user
- // could delete every other one.
+ // Administration: who exists, and who is an administrator.
r.Group(func(r chi.Router) {
r.Use(AdminOnly)
@@ -147,7 +142,6 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version strin
r.Get("/api/incidents", handleListIncidents(db))
r.Get("/api/incidents/clusters", handleListClusters(db))
r.Get("/api/incidents/{id}", handleGetIncident(db))
- r.Get("/api/incidents/{id}/alerts", handleIncidentAlerts(db))
r.Get("/api/incidents/{id}/timeline", handleIncidentTimeline(db))
r.Get("/api/incidents/{id}/similar", handleIncidentSimilar(db))
r.Post("/api/incidents/{id}/acknowledge", handleIncidentAcknowledge(db))
diff --git a/internal/api/schedule.go b/internal/api/schedule.go
index 089800c..7535218 100644
--- a/internal/api/schedule.go
+++ b/internal/api/schedule.go
@@ -69,7 +69,7 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
// be, so the delete and the insert share one transaction.
tx, err := db.BeginTx(r.Context(), nil)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer tx.Rollback()
@@ -79,7 +79,7 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
if _, err := tx.ExecContext(r.Context(),
"DELETE FROM schedule_entries WHERE team_id = $1 AND date = $2",
teamID, d); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
}
@@ -91,12 +91,12 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
errResp("date already assigned: "+d+" (pass replace to take it)"))
return
}
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
}
if err := tx.Commit(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
@@ -107,7 +107,7 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
}
all, err := scheduleRange(r.Context(), db, teamID, req.Dates[0], req.Dates[len(req.Dates)-1])
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
created := []models.ScheduleEntry{}
@@ -147,7 +147,7 @@ func handleListSchedule(db *sql.DB) http.HandlerFunc {
entries, err := scheduleRange(r.Context(), db, teamID, from, to)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, entries)
@@ -171,7 +171,7 @@ func handleDeleteSchedule(db *sql.DB) http.HandlerFunc {
res, err := db.ExecContext(r.Context(),
"DELETE FROM schedule_entries WHERE id = $1 AND team_id = $2", id, teamID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -197,7 +197,7 @@ func handleCurrentSchedule(db *sql.DB) http.HandlerFunc {
WHERE s.date = $1 AND s.team_id = ANY($2)
ORDER BY t.name`, today, callerTeamIDs(r.Context()))
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -207,14 +207,14 @@ func handleCurrentSchedule(db *sql.DB) http.HandlerFunc {
var e models.ScheduleEntry
var ts int64
if err := rows.Scan(&e.ID, &e.TeamID, &e.TeamName, &e.UserID, &e.Username, &e.Date, &ts); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
e.CreatedAt = time.Unix(ts, 0).UTC()
entries = append(entries, e)
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, entries)
diff --git a/internal/api/service_accounts.go b/internal/api/service_accounts.go
index c03bfcd..b95cb3e 100644
--- a/internal/api/service_accounts.go
+++ b/internal/api/service_accounts.go
@@ -50,6 +50,9 @@ func callerIsAdmin(ctx context.Context) bool {
// writing a response: callers here need to combine it with other ways of
// being allowed, not stop at the first no.
func callerOwnsTeam(ctx context.Context, teamID int64) bool {
+ if c, _ := callerFromContext(ctx); c.IsInstanceServiceAccount() {
+ return true
+ }
role, ok := callerRole(ctx, teamID)
return ok && role == models.RoleOwner
}
@@ -132,7 +135,7 @@ func handleCreateServiceAccount(db *sql.DB) http.HandlerFunc {
key, err := mintServiceAccountKey(r.Context(), db, sa.ID, "initial")
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusCreated, map[string]any{"service_account": sa, "key": key})
@@ -233,7 +236,7 @@ func handleCreateServiceAccountKey(db *sql.DB) http.HandlerFunc {
return
}
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if !callerMayManageServiceAccount(r.Context(), sa) {
@@ -255,7 +258,7 @@ func handleCreateServiceAccountKey(db *sql.DB) http.HandlerFunc {
key, err := mintServiceAccountKey(r.Context(), db, sa.ID, req.Name)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusCreated, key)
@@ -274,7 +277,7 @@ func handleDeleteServiceAccountKey(db *sql.DB) http.HandlerFunc {
return
}
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if !callerMayManageServiceAccount(r.Context(), sa) {
@@ -290,7 +293,7 @@ func handleDeleteServiceAccountKey(db *sql.DB) http.HandlerFunc {
res, err := db.ExecContext(r.Context(),
"DELETE FROM service_account_keys WHERE id = $1 AND service_account_id = $2", keyID, sa.ID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -325,7 +328,7 @@ func handleListServiceAccounts(db *sql.DB) http.HandlerFunc {
rows, err := db.QueryContext(r.Context(), query, args...)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -335,14 +338,14 @@ func handleListServiceAccounts(db *sql.DB) http.HandlerFunc {
var sa models.ServiceAccount
var created int64
if err := rows.Scan(&sa.ID, &sa.Name, &sa.Scope, &sa.TeamID, &sa.CreatedBy, &created); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
sa.CreatedAt = time.Unix(created, 0).UTC()
accounts = append(accounts, sa)
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, accounts)
diff --git a/internal/api/settings.go b/internal/api/settings.go
index f736c5f..fb22c44 100644
--- a/internal/api/settings.go
+++ b/internal/api/settings.go
@@ -203,7 +203,7 @@ func handleSetSettings(db *sql.DB) http.HandlerFunc {
tx, err := db.BeginTx(r.Context(), nil)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer tx.Rollback() //nolint:errcheck
@@ -215,12 +215,12 @@ func handleSetSettings(db *sql.DB) http.HandlerFunc {
ON CONFLICT (key) DO UPDATE SET
value = excluded.value, updated_at = excluded.updated_at`,
key, value); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
}
if err := tx.Commit(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
@@ -261,7 +261,7 @@ func handleAdminListTeams(db *sql.DB) http.HandlerFunc {
FROM teams t
ORDER BY t.name`)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -272,14 +272,14 @@ func handleAdminListTeams(db *sql.DB) http.HandlerFunc {
var created int64
if err := rows.Scan(&t.ID, &t.Name, &created, &t.Members, &t.OpenIncidents,
&t.OIDCMemberGroup, &t.OIDCOwnerGroup); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
t.CreatedAt = time.Unix(created, 0).UTC()
teams = append(teams, t)
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, teams)
@@ -319,7 +319,7 @@ func handleAdminGetTeam(db *sql.DB) http.HandlerFunc {
return
}
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
t.CreatedAt = time.Unix(created, 0).UTC()
@@ -333,7 +333,7 @@ func handleAdminGetTeam(db *sql.DB) http.HandlerFunc {
WHERE m.team_id = $1
ORDER BY u.username`, teamID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -343,14 +343,14 @@ func handleAdminGetTeam(db *sql.DB) http.HandlerFunc {
var m models.TeamMember
var joined int64
if err := rows.Scan(&m.TeamID, &m.UserID, &m.Username, &m.Role, &joined, &m.Source); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
m.JoinedAt = time.Unix(joined, 0).UTC()
members = append(members, m)
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
@@ -396,7 +396,7 @@ func handleRenameTeam(db *sql.DB) http.HandlerFunc {
respond(w, http.StatusConflict, errResp("a team with that name already exists"))
return
}
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -435,7 +435,7 @@ func handleSetUserDisabled(db *sql.DB) http.HandlerFunc {
}
last, err := isLastAdmin(r.Context(), db, id)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if last {
@@ -453,7 +453,7 @@ func handleSetUserDisabled(db *sql.DB) http.HandlerFunc {
"UPDATE users SET disabled_at = NULL WHERE id = $1", id)
}
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -475,7 +475,7 @@ func handleSetUserDisabled(db *sql.DB) http.HandlerFunc {
user, err := fetchUser(r.Context(), db, id)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, user)
diff --git a/internal/api/signup.go b/internal/api/signup.go
index 9b32e83..0d2a3f8 100644
--- a/internal/api/signup.go
+++ b/internal/api/signup.go
@@ -170,13 +170,13 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand
hash, err := hashPassword(req.Password)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
tx, err := db.BeginTx(r.Context(), nil)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer tx.Rollback() //nolint:errcheck
@@ -194,7 +194,7 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand
respond(w, http.StatusConflict, errResp("username or email already exists"))
return
}
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
@@ -207,7 +207,7 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand
respond(w, http.StatusConflict, errResp("a team with that name already exists"))
return
}
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
role = models.RoleOwner
@@ -216,7 +216,7 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand
if _, err := tx.ExecContext(r.Context(),
"INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3)",
teamID, userID, role); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
@@ -226,7 +226,7 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand
res, err := tx.ExecContext(r.Context(),
"UPDATE invites SET uses = uses + 1 WHERE id = $1 AND uses < max_uses", inv.id)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -236,14 +236,14 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand
}
if err := tx.Commit(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
// Signed in immediately: the alternative is a form that says "now go
// and log in", which is the same credential typed twice.
if err := startSession(w, r, db, userID, publicURL); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
user, _ := fetchUser(r.Context(), db, userID)
@@ -291,7 +291,7 @@ func handleListInvites(db *sql.DB) http.HandlerFunc {
WHERE team_id = $1
ORDER BY id DESC`, teamID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -303,7 +303,7 @@ func handleListInvites(db *sql.DB) http.HandlerFunc {
var revoked *int64
if err := rows.Scan(&i.ID, &i.TeamID, &i.Role, &created, &expires,
&i.MaxUses, &i.Uses, &revoked); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
i.CreatedAt = time.Unix(created, 0).UTC()
@@ -312,7 +312,7 @@ func handleListInvites(db *sql.DB) http.HandlerFunc {
out = append(out, i)
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, out)
@@ -356,7 +356,7 @@ func handleCreateInvite(db *sql.DB, publicURL string) http.HandlerFunc {
raw, hash, err := randomToken()
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
// created_by is nullable (ON DELETE SET NULL) for exactly this
@@ -382,7 +382,7 @@ func handleCreateInvite(db *sql.DB, publicURL string) http.HandlerFunc {
RETURNING id, team_id, role, created_at, expires_at, max_uses, uses`,
hash, teamID, req.Role, createdBy, expires.Unix(), req.MaxUses).
Scan(&out.ID, &out.TeamID, &out.Role, &created, &expiresAt, &out.MaxUses, &out.Uses); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
out.CreatedAt = time.Unix(created, 0).UTC()
@@ -412,7 +412,7 @@ func handleRevokeInvite(db *sql.DB) http.HandlerFunc {
"UPDATE invites SET revoked_at = "+nowEpoch+
" WHERE id = $1 AND team_id = $2 AND revoked_at IS NULL", id, teamID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -450,7 +450,7 @@ func handleTestNotification(cfg NotifyConfig, db *sql.DB) http.HandlerFunc {
var topic *string
if err := db.QueryRowContext(r.Context(),
"SELECT ntfy_topic FROM users WHERE id = $1", caller.ID).Scan(&topic); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if topic == nil || *topic == "" {
@@ -501,7 +501,7 @@ func handleDismissOnboarding(db *sql.DB) http.HandlerFunc {
"UPDATE users SET onboarding_dismissed_at = NULL WHERE id = $1", caller.ID)
}
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
w.WriteHeader(http.StatusNoContent)
diff --git a/internal/api/similar.go b/internal/api/similar.go
index b316289..c13f400 100644
--- a/internal/api/similar.go
+++ b/internal/api/similar.go
@@ -37,7 +37,7 @@ func handleIncidentSimilar(db *sql.DB) http.HandlerFunc {
out, err := similarIncidents(r.Context(), db, id, limit)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, out)
diff --git a/internal/api/stats.go b/internal/api/stats.go
index ee30af9..a9ab4da 100644
--- a/internal/api/stats.go
+++ b/internal/api/stats.go
@@ -24,7 +24,7 @@ func handleStatsAlerts(db *sql.DB) http.HandlerFunc {
FROM alerts WHERE %s`, where), args.all()...,
).Scan(&total, &firing, &resolved)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, map[string]int64{
@@ -55,7 +55,7 @@ func handleStatsTop(db *sql.DB) http.HandlerFunc {
ORDER BY cnt DESC
LIMIT %s`, where, args.add(limit)), args.all()...)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -68,13 +68,13 @@ func handleStatsTop(db *sql.DB) http.HandlerFunc {
for rows.Next() {
var e entry
if err := rows.Scan(&e.Name, &e.Count); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
result = append(result, e)
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, result)
@@ -93,7 +93,7 @@ func handleStatsByHour(db *sql.DB) http.HandlerFunc {
GROUP BY hr
ORDER BY hr ASC`, where), args.all()...)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -103,13 +103,13 @@ func handleStatsByHour(db *sql.DB) http.HandlerFunc {
var hr int
var cnt int64
if err := rows.Scan(&hr, &cnt); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
counts[hr] = cnt
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
@@ -129,8 +129,7 @@ func handleStatsByDay(db *sql.DB) http.HandlerFunc {
return func(w http.ResponseWriter, r *http.Request) {
where, args := statsFilter(r.URL.Query(), "received_at", callerTeamIDs(r.Context()))
- // Postgres EXTRACT(DOW …) → 0=Sunday … 6=Saturday, the same numbering
- // SQLite's strftime('%w') returned, so the frontend needs no change.
+ // Postgres EXTRACT(DOW …) → 0=Sunday … 6=Saturday.
rows, err := db.QueryContext(r.Context(), fmt.Sprintf(`
SELECT EXTRACT(DOW FROM to_timestamp(received_at) AT TIME ZONE 'UTC')::int AS dow,
COUNT(*) AS cnt
@@ -139,7 +138,7 @@ func handleStatsByDay(db *sql.DB) http.HandlerFunc {
GROUP BY dow
ORDER BY dow ASC`, where), args.all()...)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -149,13 +148,13 @@ func handleStatsByDay(db *sql.DB) http.HandlerFunc {
var dow int
var cnt int64
if err := rows.Scan(&dow, &cnt); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
counts[dow] = cnt
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
@@ -198,7 +197,7 @@ func handleStatsIncidents(db *sql.DB) http.HandlerFunc {
FROM incidents WHERE %s`, where), args.all()...,
).Scan(&total, &triggered, &acknowledged, &resolved, &mtta, &mttr)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
diff --git a/internal/api/teams.go b/internal/api/teams.go
index e2b17c7..e92d7d9 100644
--- a/internal/api/teams.go
+++ b/internal/api/teams.go
@@ -13,19 +13,12 @@ import (
"github.com/go-chi/chi/v5"
)
-// handleListTeams lists the caller's own teams, each with their role in it,
-// or — with ?name= — looks up one team by exact name regardless of caller
-// identity (TEAM-LOOKUP.md). An administrator listing every team goes
-// through the admin endpoint instead: the no-name case here answers "what am
-// I part of", which is what the UI's team filter and the combined queue are
-// built from.
+// handleListTeams lists the caller's own teams, each with their role in it.
+// An administrator listing every team goes through the admin endpoint instead:
+// this answers "what am I part of", which is what the UI's team filter and the
+// combined queue are built from.
func handleListTeams(db *sql.DB) http.HandlerFunc {
return func(w http.ResponseWriter, r *http.Request) {
- if name := strings.TrimSpace(r.URL.Query().Get("name")); name != "" {
- handleListTeamsByName(db, w, r, name)
- return
- }
-
caller, _ := userFromContext(r.Context())
rows, err := db.QueryContext(r.Context(), `
SELECT t.id, t.name, t.created_at, m.role, m.source
@@ -34,7 +27,7 @@ func handleListTeams(db *sql.DB) http.HandlerFunc {
WHERE m.user_id = $1
ORDER BY t.name`, caller.ID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -44,45 +37,20 @@ func handleListTeams(db *sql.DB) http.HandlerFunc {
var t models.Team
var created int64
if err := rows.Scan(&t.ID, &t.Name, &created, &t.Role, &t.Source); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
t.CreatedAt = time.Unix(created, 0).UTC()
teams = append(teams, t)
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, teams)
}
}
-// handleListTeamsByName answers "is there a team named exactly this", open to
-// any authenticated caller including a service account (TEAM-LOOKUP.md) —
-// mirrors handleListServiceAccounts' own ?name= lookup: a one-or-zero-length
-// array, never an error on no match, and no caller-identity filtering at
-// all, since what it discloses (a name is taken, nothing about who's in it
-// or any of its data) is the same low sensitivity that lookup already
-// accepts for service-account names.
-func handleListTeamsByName(db *sql.DB, w http.ResponseWriter, r *http.Request, name string) {
- var t models.Team
- var created int64
- err := db.QueryRowContext(r.Context(),
- "SELECT id, name, created_at FROM teams WHERE name = $1", name,
- ).Scan(&t.ID, &t.Name, &created)
- if errors.Is(err, sql.ErrNoRows) {
- respond(w, http.StatusOK, []models.Team{})
- return
- }
- if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
- return
- }
- t.CreatedAt = time.Unix(created, 0).UTC()
- respond(w, http.StatusOK, []models.Team{t})
-}
-
// handleUserTeams lists one user's teams, for the admin page's per-user view:
// "what is this person in", which /api/teams cannot answer because it is always
// about the caller.
@@ -106,7 +74,7 @@ func handleUserTeams(db *sql.DB) http.HandlerFunc {
var exists bool
if err := db.QueryRowContext(r.Context(),
"SELECT EXISTS (SELECT 1 FROM users WHERE id = $1)", id).Scan(&exists); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if !exists {
@@ -121,7 +89,7 @@ func handleUserTeams(db *sql.DB) http.HandlerFunc {
WHERE m.user_id = $1
ORDER BY t.name`, id)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -131,14 +99,14 @@ func handleUserTeams(db *sql.DB) http.HandlerFunc {
var t models.Team
var created int64
if err := rows.Scan(&t.ID, &t.Name, &created, &t.Role, &t.Source); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
t.CreatedAt = time.Unix(created, 0).UTC()
teams = append(teams, t)
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, teams)
@@ -161,6 +129,12 @@ func handleCreateTeam(db *sql.DB) http.HandlerFunc {
return func(w http.ResponseWriter, r *http.Request) {
var req struct {
Name string `json:"name"`
+ // ExternalID makes the call idempotent for automation: a team
+ // already carrying it is returned as-is (200) instead of created, so
+ // a client that crashed between the POST and recording the id finds
+ // its own team again. Instance-scoped service accounts only; a name
+ // that belongs to a different team is still a 409.
+ ExternalID string `json:"external_id"`
}
if err := decodeJSON(r, &req); err != nil {
respond(w, http.StatusBadRequest, errResp("invalid request body"))
@@ -173,6 +147,10 @@ func handleCreateTeam(db *sql.DB) http.HandlerFunc {
}
caller, isUser := userFromContext(r.Context())
+ if req.ExternalID != "" && !isInstanceServiceAccount(r.Context()) {
+ respond(w, http.StatusForbidden, errResp("external_id is for instance-scoped service accounts"))
+ return
+ }
if !isUser && !isInstanceServiceAccount(r.Context()) {
// A team-scoped service account authenticates as owner of exactly
// one team already (see serveAsServiceAccount); letting it create
@@ -183,37 +161,57 @@ func handleCreateTeam(db *sql.DB) http.HandlerFunc {
tx, err := db.BeginTx(r.Context(), nil)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer tx.Rollback() //nolint:errcheck
var team models.Team
var created int64
+ if req.ExternalID != "" {
+ err := tx.QueryRowContext(r.Context(),
+ "SELECT id, name, created_at FROM teams WHERE external_id = $1", req.ExternalID).
+ Scan(&team.ID, &team.Name, &created)
+ if err == nil {
+ team.ExternalID = &req.ExternalID
+ team.CreatedAt = time.Unix(created, 0).UTC()
+ respond(w, http.StatusOK, team)
+ return
+ }
+ if !errors.Is(err, sql.ErrNoRows) {
+ serverError(w, r, err)
+ return
+ }
+ }
+ var externalID *string
+ if req.ExternalID != "" {
+ externalID = &req.ExternalID
+ }
if err := tx.QueryRowContext(r.Context(),
- "INSERT INTO teams (name) VALUES ($1) RETURNING id, name, created_at",
- req.Name).Scan(&team.ID, &team.Name, &created); err != nil {
+ "INSERT INTO teams (name, external_id) VALUES ($1, $2) RETURNING id, name, created_at",
+ req.Name, externalID).Scan(&team.ID, &team.Name, &created); err != nil {
if isUniqueViolation(err) {
respond(w, http.StatusConflict, errResp("a team with that name already exists"))
return
}
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if isUser {
if _, err := tx.ExecContext(r.Context(),
"INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3)",
team.ID, caller.ID, models.RoleOwner); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
}
if err := tx.Commit(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
team.CreatedAt = time.Unix(created, 0).UTC()
+ team.ExternalID = externalID
if isUser {
team.Role = models.RoleOwner
}
@@ -241,7 +239,7 @@ func handleDeleteTeam(db *sql.DB) http.HandlerFunc {
if err := db.QueryRowContext(r.Context(),
"SELECT COUNT(*) FROM incidents WHERE team_id = $1 AND resolved_at IS NULL", teamID).
Scan(&open); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if open > 0 {
@@ -251,7 +249,7 @@ func handleDeleteTeam(db *sql.DB) http.HandlerFunc {
res, err := db.ExecContext(r.Context(), "DELETE FROM teams WHERE id = $1", teamID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -322,7 +320,7 @@ func handleListTeamMembers(db *sql.DB) http.HandlerFunc {
WHERE m.team_id = $1
ORDER BY u.username`, teamID, todayUTC())
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -334,7 +332,7 @@ func handleListTeamMembers(db *sql.DB) http.HandlerFunc {
var hasTopic, disabled bool
if err := rows.Scan(&m.TeamID, &m.UserID, &m.Username, &m.Role, &joined, &m.Source,
&hasTopic, &disabled, &lastActive, &m.OnCall, &m.NextShift); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
m.JoinedAt = time.Unix(joined, 0).UTC()
@@ -360,7 +358,7 @@ func handleListTeamMembers(db *sql.DB) http.HandlerFunc {
members = append(members, m)
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, members)
@@ -396,7 +394,7 @@ func handleAddTeamMember(db *sql.DB) http.HandlerFunc {
}
if managed, err := isSSOManagedMember(r.Context(), db, teamID, req.UserID); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
} else if managed {
respond(w, http.StatusConflict, errResp(ssoManagedMsg))
@@ -408,7 +406,7 @@ func handleAddTeamMember(db *sql.DB) http.HandlerFunc {
if req.Role == models.RoleMember {
last, err := isLastTeamOwner(r.Context(), db, teamID, req.UserID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if last {
@@ -453,7 +451,7 @@ func handleRemoveTeamMember(db *sql.DB) http.HandlerFunc {
}
if managed, err := isSSOManagedMember(r.Context(), db, teamID, userID); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
} else if managed {
respond(w, http.StatusConflict, errResp(ssoManagedMsg))
@@ -462,7 +460,7 @@ func handleRemoveTeamMember(db *sql.DB) http.HandlerFunc {
last, err := isLastTeamOwner(r.Context(), db, teamID, userID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if last {
@@ -473,7 +471,7 @@ func handleRemoveTeamMember(db *sql.DB) http.HandlerFunc {
res, err := db.ExecContext(r.Context(),
"DELETE FROM team_members WHERE team_id = $1 AND user_id = $2", teamID, userID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -572,7 +570,7 @@ func handleListIntegrations(db *sql.DB) http.HandlerFunc {
WHERE i.team_id = $1
ORDER BY i.id`, teamID, now.Add(-sourceQuietAfter).Unix())
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -584,7 +582,7 @@ func handleListIntegrations(db *sql.DB) http.HandlerFunc {
var lastUsed, lastAlert *int64
if err := rows.Scan(&i.ID, &i.TeamID, &i.Kind, &i.Name, &created, &lastUsed,
&lastAlert, &i.Alerts24h); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
i.CreatedAt = time.Unix(created, 0).UTC()
@@ -601,7 +599,7 @@ func handleListIntegrations(db *sql.DB) http.HandlerFunc {
integrations = append(integrations, i)
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, integrations)
@@ -644,7 +642,7 @@ func handleCreateIntegration(db *sql.DB, publicURL string) http.HandlerFunc {
raw, hash, err := randomToken()
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
@@ -656,7 +654,11 @@ func handleCreateIntegration(db *sql.DB, publicURL string) http.HandlerFunc {
RETURNING id, team_id, kind, name, created_at`,
teamID, req.Kind, req.Name, hash).
Scan(&i.ID, &i.TeamID, &i.Kind, &i.Name, &created); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ if isUniqueViolation(err) {
+ respond(w, http.StatusConflict, errResp("an integration with that name already exists in this team"))
+ return
+ }
+ serverError(w, r, err)
return
}
i.CreatedAt = time.Unix(created, 0).UTC()
@@ -703,7 +705,11 @@ func handleRenameIntegration(db *sql.DB) http.HandlerFunc {
res, err := db.ExecContext(r.Context(),
"UPDATE integrations SET name = $1 WHERE id = $2 AND team_id = $3", req.Name, id, teamID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ if isUniqueViolation(err) {
+ respond(w, http.StatusConflict, errResp("an integration with that name already exists in this team"))
+ return
+ }
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -732,7 +738,7 @@ func handleDeleteIntegration(db *sql.DB) http.HandlerFunc {
res, err := db.ExecContext(r.Context(),
"DELETE FROM integrations WHERE id = $1 AND team_id = $2", id, teamID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -790,16 +796,6 @@ func teamParam(w http.ResponseWriter, r *http.Request) (int64, bool) {
return id, true
}
-// defaultTeamID is the oldest team, which on an upgraded install is the
-// "Default" team every pre-teams row was moved into and on a fresh one is the
-// team migration 003 creates. Bootstrap puts the first user in it, so somebody
-// signing in to a new server lands somewhere rather than in no team at all.
-func defaultTeamID(ctx context.Context, db *sql.DB) (int64, error) {
- var id int64
- err := db.QueryRowContext(ctx, "SELECT id FROM teams ORDER BY id LIMIT 1").Scan(&id)
- return id, err
-}
-
// ---------------------------------------------------------------------------
// A team's dead man's switches
// ---------------------------------------------------------------------------
@@ -833,12 +829,12 @@ func handleListTeamDeadman(db *sql.DB) http.HandlerFunc {
set, err := deadmanSetForTeam(r.Context(), db, teamID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
out, err := deadmanStatuses(r.Context(), db, teamID, set, time.Now())
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, out)
@@ -901,7 +897,11 @@ func handleCreateTeamDeadman(db *sql.DB) http.HandlerFunc {
INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity)
VALUES ($1, $2, $3, $4, $5) RETURNING id`,
teamID, req.Name, m.config(), req.TimeoutSeconds, req.Severity).Scan(&id); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ if isUniqueViolation(err) {
+ respond(w, http.StatusConflict, errResp("a switch with that name already exists in this team"))
+ return
+ }
+ serverError(w, r, err)
return
}
@@ -975,7 +975,11 @@ func handleUpdateTeamDeadman(db *sql.DB) http.HandlerFunc {
WHERE id = $5 AND team_id = $6`,
req.Name, m.config(), req.TimeoutSeconds, req.Severity, switchID, teamID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ if isUniqueViolation(err) {
+ respond(w, http.StatusConflict, errResp("a switch with that name already exists in this team"))
+ return
+ }
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -990,12 +994,12 @@ func handleUpdateTeamDeadman(db *sql.DB) http.HandlerFunc {
// handleListTeamDeadman would give it.
set, err := deadmanSetForTeam(r.Context(), db, teamID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
statuses, err := deadmanStatuses(r.Context(), db, teamID, set, time.Now())
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
for _, s := range statuses {
@@ -1004,7 +1008,7 @@ func handleUpdateTeamDeadman(db *sql.DB) http.HandlerFunc {
return
}
}
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
}
}
@@ -1029,7 +1033,7 @@ func handleDeleteTeamDeadman(db *sql.DB) http.HandlerFunc {
res, err := db.ExecContext(r.Context(),
"DELETE FROM deadman_switches WHERE id = $1 AND team_id = $2", switchID, teamID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
diff --git a/internal/api/teams_test.go b/internal/api/teams_test.go
index 8cebe1a..85879f9 100644
--- a/internal/api/teams_test.go
+++ b/internal/api/teams_test.go
@@ -6,8 +6,6 @@ import (
"io"
"net/http"
"testing"
-
- "git.ryuvia.com/niklas/terdut-server/internal/models"
)
// The whole point of #4: two teams sharing one server must not see each other's
@@ -146,7 +144,6 @@ func TestTeams_IncidentsAreScopedToTheReceivingTeam(t *testing.T) {
otherID := int64(blueIncidents[0]["id"].(float64))
for _, path := range []string{
"/api/incidents/" + id64(otherID),
- "/api/incidents/" + id64(otherID) + "/alerts",
"/api/incidents/" + id64(otherID) + "/timeline",
} {
resp := red.call(http.MethodGet, path, nil)
@@ -461,70 +458,58 @@ func TestTeams_OutsiderSeesNothing(t *testing.T) {
// GET /api/teams?name= (TEAM-LOOKUP.md)
// ---------------------------------------------------------------------------
-func TestListTeamsByName_FindsExactMatch(t *testing.T) {
- s := newTS(t)
- instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
- teamID := createTeamAs(t, s, instanceKey, "platform")
-
- teams := list(t, s.reqAs(t, instanceKey, http.MethodGet, "/api/teams?name=platform", nil))
- if len(teams) != 1 {
- t.Fatalf("expected exactly one match for ?name=platform, got %d: %v", len(teams), teams)
- }
- if int64(teams[0]["id"].(float64)) != teamID {
- t.Errorf("id = %v, want %d", teams[0]["id"], teamID)
- }
- // No membership, so no role to report (models.Team's own doc comment:
- // "empty when nobody in particular is asking").
- if _, has := teams[0]["role"]; has {
- t.Errorf("expected no role on a name-lookup match, got %v", teams[0]["role"])
- }
-}
-
-func TestListTeamsByName_NoMatchIsAnEmptyArrayNotAnError(t *testing.T) {
- s := newTS(t)
- instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
-
- resp := s.reqAs(t, instanceKey, http.MethodGet, "/api/teams?name=does-not-exist", nil)
- if resp.StatusCode != http.StatusOK {
- t.Fatalf("expected 200 on no match, got %d", resp.StatusCode)
- }
- teams := list(t, resp)
- if len(teams) != 0 {
- t.Errorf("expected an empty array, got %v", teams)
- }
-}
-
-// The actual motivating scenario (TEAM-LOOKUP.md): a service account that
-// already created a team, interrupted before it could remember the id,
-// recovers it via ?name= on the same name its own POST 409s on.
-func TestListTeamsByName_RecoversAfterCreateConflict(t *testing.T) {
- s := newTS(t)
- instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
- original := createTeamAs(t, s, instanceKey, "recovered")
-
- conflict := s.reqAs(t, instanceKey, http.MethodPost, "/api/teams", map[string]string{"name": "recovered"})
- if conflict.StatusCode != http.StatusConflict {
- t.Fatalf("expected 409 recreating the same name, got %d", conflict.StatusCode)
- }
- conflict.Body.Close()
-
- teams := list(t, s.reqAs(t, instanceKey, http.MethodGet, "/api/teams?name=recovered", nil))
- if len(teams) != 1 || int64(teams[0]["id"].(float64)) != original {
- t.Fatalf("expected to recover the original team %d via ?name=, got %v", original, teams)
- }
-}
-
-// Not gated by isInstanceServiceAccount or AdminOnly (TEAM-LOOKUP.md): any
-// authenticated caller may ask whether a name is taken, the same low
-// sensitivity GET /api/service-accounts?name= already accepts.
-func TestListTeamsByName_OpenToAnyAuthenticatedCaller(t *testing.T) {
+// Everybody signed in can list users to name them, but only an admin (or the
+// row's owner) sees an email or an ntfy topic, which is a publish secret.
+func TestListUsers_RedactsEmailAndTopicForOthers(t *testing.T) {
s := newTS(t)
red := newTeam(t, s, "red")
- _ = createTeamAs(t, s, s.key, "blue-target")
+ _ = newTeam(t, s, "blue")
+ s.exec(t, "UPDATE users SET ntfy_topic = 'secret-topic'")
- // red's own member, not a member of "blue-target", still gets a match.
- teams := list(t, red.call(http.MethodGet, "/api/teams?name=blue-target", nil))
- if len(teams) != 1 || teams[0]["name"] != "blue-target" {
- t.Errorf("expected a non-member caller to still find the team by name, got %v", teams)
+ var asAdmin []map[string]any
+ decode(t, s.req(t, http.MethodGet, "/api/users", nil), &asAdmin)
+ for _, u := range asAdmin {
+ if u["email"] == "" || u["ntfy_topic"] != "secret-topic" {
+ t.Errorf("admin should see everything, got %v", u)
+ }
+ }
+
+ asMember := list(t, red.call(http.MethodGet, "/api/users", nil))
+ if len(asMember) < 3 {
+ t.Fatalf("expected the whole user list, got %v", asMember)
+ }
+ for _, u := range asMember {
+ own := u["username"] == "red-user"
+ if own != (u["email"] != "") || own != (u["ntfy_topic"] != nil) {
+ t.Errorf("only red-user's own row should keep email and topic, got %v", u)
+ }
+ }
+}
+
+// Names identify integrations and switches within a team.
+func TestTeamNames_AreUniquePerTeam(t *testing.T) {
+ s := newTS(t) // creates one integration named "test" in the default team
+
+ dup := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/integrations", map[string]string{"name": "test"})
+ dup.Body.Close()
+ if dup.StatusCode != http.StatusConflict {
+ t.Errorf("duplicate integration name: expected 409, got %d", dup.StatusCode)
+ }
+
+ body := map[string]any{"matcher": "alertname=Watchdog", "timeout_seconds": 60, "severity": "critical"}
+ first := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches", body)
+ first.Body.Close()
+ second := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches", body)
+ second.Body.Close()
+ if first.StatusCode != http.StatusCreated || second.StatusCode != http.StatusConflict {
+ t.Errorf("duplicate switch name: expected 201 then 409, got %d then %d", first.StatusCode, second.StatusCode)
+ }
+
+ // The same name in another team is fine.
+ other := newTeam(t, s, "elsewhere")
+ ok := s.req(t, http.MethodPost, "/api/teams/"+id64(other.id)+"/integrations", map[string]string{"name": "test"})
+ ok.Body.Close()
+ if ok.StatusCode != http.StatusCreated {
+ t.Errorf("same name in another team: expected 201, got %d", ok.StatusCode)
}
}
diff --git a/internal/api/testdb_test.go b/internal/api/testdb_test.go
index 33d6294..25852e3 100644
--- a/internal/api/testdb_test.go
+++ b/internal/api/testdb_test.go
@@ -13,9 +13,8 @@ import (
"git.ryuvia.com/niklas/terdut-server/internal/db"
)
-// Tests run against a real Postgres, because the server does. SQLite's
-// ":memory:" gave every test a private database for free; Postgres has no
-// equivalent, so isolation is bought with a schema per test.
+// Tests run against a real Postgres, because the server does. Isolation is
+// bought with a schema per test.
//
// A schema rather than a database: CREATE DATABASE copies a template on disk and
// costs a hundred milliseconds or so each time, while CREATE SCHEMA plus the one
diff --git a/internal/api/users.go b/internal/api/users.go
index aa457a2..ba3f7f7 100644
--- a/internal/api/users.go
+++ b/internal/api/users.go
@@ -15,6 +15,10 @@ import (
"github.com/go-chi/chi/v5"
)
+// bootstrapLockKey is the transaction-scoped advisory lock handleBootstrap
+// holds; distinct from the migration and notifier keys.
+const bootstrapLockKey = 0x7465726475744254
+
func handleBootstrap(db *sql.DB) http.HandlerFunc {
return func(w http.ResponseWriter, r *http.Request) {
var req struct {
@@ -40,15 +44,30 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc {
}
h, err := hashPassword(req.Password)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
passwordHash = &h
}
+ // Check-then-insert has to be one atomic step: two concurrent calls on
+ // an empty install would otherwise both see zero users and both create
+ // an admin. The transaction-scoped lock serialises them, and the loser
+ // sees the winner's row.
+ tx, err := db.BeginTx(r.Context(), nil)
+ if err != nil {
+ serverError(w, r, err)
+ return
+ }
+ defer tx.Rollback() //nolint:errcheck
+ if _, err := tx.ExecContext(r.Context(), "SELECT pg_advisory_xact_lock($1)", bootstrapLockKey); err != nil {
+ serverError(w, r, err)
+ return
+ }
+
var count int
- if err := db.QueryRowContext(r.Context(), "SELECT COUNT(*) FROM users").Scan(&count); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ if err := tx.QueryRowContext(r.Context(), "SELECT COUNT(*) FROM users").Scan(&count); err != nil {
+ serverError(w, r, err)
return
}
if count > 0 {
@@ -57,34 +76,29 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc {
}
var userID int64
- if err := db.QueryRowContext(r.Context(),
+ if err := tx.QueryRowContext(r.Context(),
"INSERT INTO users (username, email, password_hash, is_admin) VALUES ($1, $2, $3, true) RETURNING id",
req.Username, req.Email, passwordHash).Scan(&userID); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
raw, hash, err := randomToken()
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
var keyID int64
- if err := db.QueryRowContext(r.Context(),
+ if err := tx.QueryRowContext(r.Context(),
"INSERT INTO api_keys (user_id, key_hash, name) VALUES ($1, $2, $3) RETURNING id",
userID, hash, "bootstrap").Scan(&keyID); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
- // The default team exists from migration 003, on a fresh install too.
- // Without a membership the first user signs in to a working server with
- // no queue, no schedule and nowhere for an integration to hang off.
- if teamID, err := defaultTeamID(r.Context(), db); err == nil {
- db.ExecContext(r.Context(), //nolint:errcheck
- "INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3) "+
- "ON CONFLICT (team_id, user_id) DO NOTHING",
- teamID, userID, models.RoleOwner)
+ if err := tx.Commit(); err != nil {
+ serverError(w, r, err)
+ return
}
user, _ := fetchUser(r.Context(), db, userID)
@@ -93,12 +107,19 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc {
}
}
+// handleListUsers is readable by anyone signed in, because the assignment
+// control and the schedule need to name people. What it returns about other
+// people is therefore only what naming them takes: email and ntfy_topic are
+// blanked unless the caller is an admin or the row is their own. The topic in
+// particular is a publish secret.
func handleListUsers(db *sql.DB) http.HandlerFunc {
return func(w http.ResponseWriter, r *http.Request) {
+ caller, _ := userFromContext(r.Context())
+ seeAll := caller.IsAdmin
rows, err := db.QueryContext(r.Context(),
"SELECT id, username, email, created_at, ntfy_topic, is_admin, admin_source, disabled_at FROM users ORDER BY id")
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -109,15 +130,19 @@ func handleListUsers(db *sql.DB) http.HandlerFunc {
var ts int64
var disabled *int64
if err := rows.Scan(&u.ID, &u.Username, &u.Email, &ts, &u.NtfyTopic, &u.IsAdmin, &u.AdminSource, &disabled); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
u.CreatedAt = time.Unix(ts, 0).UTC()
u.DisabledAt = unixPtr(disabled)
+ if !seeAll && u.ID != caller.ID {
+ u.Email = ""
+ u.NtfyTopic = nil
+ }
users = append(users, u)
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, users)
@@ -147,7 +172,7 @@ func handleCreateUser(db *sql.DB) http.HandlerFunc {
respond(w, http.StatusConflict, errResp("username or email already exists"))
return
}
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
user, _ := fetchUser(r.Context(), db, id)
@@ -185,7 +210,7 @@ func handleSetNotifyTarget(db *sql.DB) http.HandlerFunc {
res, err := db.ExecContext(r.Context(),
"UPDATE users SET ntfy_topic = $1 WHERE id = $2", topic, id)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -195,7 +220,7 @@ func handleSetNotifyTarget(db *sql.DB) http.HandlerFunc {
user, err := fetchUser(r.Context(), db, id)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, user)
@@ -217,7 +242,7 @@ func handleDeleteUser(db *sql.DB) http.HandlerFunc {
return
}
if last, err := isLastAdmin(r.Context(), db, id); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
} else if last {
respond(w, http.StatusConflict, errResp("cannot delete the last administrator"))
@@ -226,7 +251,7 @@ func handleDeleteUser(db *sql.DB) http.HandlerFunc {
res, err := db.ExecContext(r.Context(), "DELETE FROM users WHERE id = $1", id)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
n, _ := res.RowsAffected()
@@ -283,7 +308,7 @@ func handleCreateAPIKey(db *sql.DB) http.HandlerFunc {
raw, hash, err := randomToken()
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
var expiresAt *int64
@@ -298,7 +323,7 @@ func handleCreateAPIKey(db *sql.DB) http.HandlerFunc {
if err := db.QueryRowContext(r.Context(),
"INSERT INTO api_keys (user_id, key_hash, name, expires_at) VALUES ($1, $2, $3, $4) RETURNING id",
userID, hash, req.Name, expiresAt).Scan(&keyID); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
key := models.APIKey{
@@ -329,7 +354,7 @@ func handleListAPIKeys(db *sql.DB) http.HandlerFunc {
`SELECT id, name, created_at, last_used_at, expires_at
FROM api_keys WHERE user_id = $1 ORDER BY created_at DESC`, userID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
defer rows.Close()
@@ -340,7 +365,7 @@ func handleListAPIKeys(db *sql.DB) http.HandlerFunc {
var created int64
var lastUsed, expires *int64
if err := rows.Scan(&k.ID, &k.Name, &created, &lastUsed, &expires); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
k.UserID = userID
@@ -350,7 +375,7 @@ func handleListAPIKeys(db *sql.DB) http.HandlerFunc {
keys = append(keys, k)
}
if err := rows.Err(); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, keys)
@@ -376,7 +401,7 @@ func handleDeleteAPIKey(db *sql.DB) http.HandlerFunc {
res, err := db.ExecContext(r.Context(),
"DELETE FROM api_keys WHERE id = $1 AND user_id = $2", keyID, userID)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
n, _ := res.RowsAffected()
@@ -442,7 +467,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc {
if err := db.QueryRowContext(r.Context(),
"SELECT EXISTS (SELECT 1 FROM users WHERE id = $1 AND is_admin AND admin_source = 'oidc')",
id).Scan(&managed); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if managed {
@@ -456,7 +481,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc {
return
}
if last, err := isLastAdmin(r.Context(), db, id); err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
} else if last {
respond(w, http.StatusConflict, errResp("cannot revoke the last administrator"))
@@ -467,7 +492,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc {
res, err := db.ExecContext(r.Context(),
"UPDATE users SET is_admin = $1 WHERE id = $2", *req.IsAdmin, id)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
if n, _ := res.RowsAffected(); n == 0 {
@@ -477,7 +502,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc {
user, err := fetchUser(r.Context(), db, id)
if err != nil {
- respond(w, http.StatusInternalServerError, errResp("internal error"))
+ serverError(w, r, err)
return
}
respond(w, http.StatusOK, user)
diff --git a/internal/config/config.go b/internal/config/config.go
index 565638c..581bbd4 100644
--- a/internal/config/config.go
+++ b/internal/config/config.go
@@ -5,17 +5,21 @@ import (
"fmt"
"net/url"
"os"
+ "strconv"
"strings"
"time"
)
+// MinOperatorKeyLength is the shortest TERDUT_OPERATOR_KEY accepted: it is a
+// bearer credential with instance reach, so a short one is refused outright.
+const MinOperatorKeyLength = 32
+
type Config struct {
Addr string
// DSN is the Postgres connection string, e.g.
// postgres://terdut:secret@host:5432/terdut?sslmode=require. Required:
- // unlike the SQLite path it replaced there is no sensible default, and a
- // server that silently came up against the wrong database would be worse
+ // there is no sensible default, and a server that silently came up against the wrong database would be worse
// than one that refuses to start.
DSN string
@@ -26,25 +30,6 @@ type Config struct {
// repeat_interval (default 4h), which is what refreshes the alert.
StaleAfter time.Duration
- // DeadmanMatchers selects the alerts that are heartbeats rather than
- // problems: receiving one opens no incident, and the absence of one does.
- //
- // ";" separates matchers, "," the label conditions within one, "=" is exact
- // equality — `alertname=Watchdog,cluster=prod; alertname=Heartbeat`. Every
- // matcher must name an alertname. See api.ParseDeadmanConfig.
- DeadmanMatchers string
-
- // DeadmanTimeout is how long a heartbeat may go unheard before its switch is
- // declared dead. It must be *shorter* than the Alertmanager repeat_interval
- // of the route carrying the heartbeat — the opposite of StaleAfter, and the
- // reason a dead man's switch usually wants a route of its own. Zero disables
- // dead man's switch handling entirely.
- DeadmanTimeout time.Duration
-
- // DeadmanSeverity is the severity a dead man's switch incident opens at.
- // These incidents have no member alerts to derive one from.
- DeadmanSeverity string
-
// NtfyURL is the ntfy server push notifications are published to. Empty
// disables notifications entirely.
NtfyURL string
@@ -70,6 +55,20 @@ type Config struct {
// Config, which is what a test or a new caller builds, keeps passwords working.
DisablePasswordLogin bool
+ // OperatorKey, when set, is the credential of the instance-scoped service
+ // account "terdut-operator", created or re-keyed at every start. It is how
+ // terdut-operator gets in without a bootstrap handshake: the operator
+ // generates the key, hands it to the server here, and uses it as its bearer
+ // token. Empty means no such account is managed.
+ OperatorKey string
+
+ // TrustedProxies is how many reverse proxies sit in front of the server and
+ // append to X-Forwarded-For. The per-address rate limits take the client
+ // address that many entries from the right, because everything further left
+ // is whatever the client chose to send. 0 ignores the header and uses the
+ // connection's own address.
+ TrustedProxies int
+
// OIDC configures single sign-on. The zero value, with no Issuer, is off.
OIDC OIDC
@@ -133,24 +132,12 @@ func Load() Config {
if addr == "" {
addr = ":8080"
}
- deadmanMatchers := os.Getenv("TERDUT_DEADMAN_MATCHERS")
- if deadmanMatchers == "" {
- deadmanMatchers = "alertname=Watchdog"
- }
- deadmanSeverity := os.Getenv("TERDUT_DEADMAN_SEVERITY")
- if deadmanSeverity == "" {
- deadmanSeverity = "critical"
- }
return Config{
Addr: addr,
DSN: os.Getenv("TERDUT_DB_DSN"),
ArchiveAfter: duration("TERDUT_ARCHIVE_AFTER", 7*24*time.Hour),
StaleAfter: duration("TERDUT_STALE_AFTER", 6*time.Hour),
- DeadmanMatchers: deadmanMatchers,
- DeadmanTimeout: duration("TERDUT_DEADMAN_TIMEOUT", 15*time.Minute),
- DeadmanSeverity: deadmanSeverity,
-
NtfyURL: os.Getenv("TERDUT_NTFY_URL"),
NtfyToken: os.Getenv("TERDUT_NTFY_TOKEN"),
NtfyFallbackTopic: os.Getenv("TERDUT_NTFY_FALLBACK_TOPIC"),
@@ -158,7 +145,10 @@ func Load() Config {
NotifyRepeat: duration("TERDUT_NOTIFY_REPEAT", 15*time.Minute),
DisablePasswordLogin: !boolean("TERDUT_PASSWORD_LOGIN", true),
- OIDC: loadOIDC(),
+
+ OperatorKey: strings.TrimSpace(os.Getenv("TERDUT_OPERATOR_KEY")),
+ TrustedProxies: integer("TERDUT_TRUSTED_PROXIES", 1),
+ OIDC: loadOIDC(),
OperatorMode: boolean("TERDUT_OPERATOR_MODE", false),
}
@@ -187,6 +177,9 @@ func loadOIDC() OIDC {
// provider would come up and then fail every login, which is harder to notice
// than not starting.
func (c Config) Validate() error {
+ if c.OperatorKey != "" && len(c.OperatorKey) < MinOperatorKeyLength {
+ return fmt.Errorf("TERDUT_OPERATOR_KEY must be at least %d characters", MinOperatorKeyLength)
+ }
o := c.OIDC
if !o.Enabled() {
if c.DisablePasswordLogin {
@@ -226,6 +219,16 @@ func str(env, def string) string {
return def
}
+// integer reads a non-negative int env var; anything else takes the default.
+func integer(env string, def int) int {
+ if s := os.Getenv(env); s != "" {
+ if n, err := strconv.Atoi(strings.TrimSpace(s)); err == nil && n >= 0 {
+ return n
+ }
+ }
+ return def
+}
+
// list reads a comma- or space-separated env var.
func list(env, def string) []string {
s := os.Getenv(env)
diff --git a/internal/db/db.go b/internal/db/db.go
index c43bce8..5b14708 100644
--- a/internal/db/db.go
+++ b/internal/db/db.go
@@ -35,8 +35,7 @@ const (
//
// The pool is modest on purpose: this server's concurrency comes from a handful
// of HTTP handlers plus two background loops, and a cloud-native-pg instance
-// sized for it has a low max_connections. It is still a pool, unlike the single
-// connection SQLite forced, so the notifier no longer blocks a webhook.
+// sized for it has a low max_connections.
func Open(dsn string) (*sql.DB, error) {
if dsn == "" {
return nil, fmt.Errorf("empty DSN: set TERDUT_DB_DSN")
@@ -76,9 +75,8 @@ const migrationLockKey int64 = 7265_0003
// Migrate applies every embedded migration that has not been applied yet, in
// filename order, recording each in schema_migrations.
//
-// Each file runs inside a transaction, which SQLite's version did not do: a
-// migration that failed half way used to leave the schema in whatever state it
-// had reached. Postgres has transactional DDL, so the rollback is real.
+// Each file runs inside a transaction, so a migration that fails half way
+// leaves the schema as it was: Postgres has transactional DDL.
func Migrate(db *sql.DB) error {
ctx := context.Background()
conn, err := db.Conn(ctx)
diff --git a/internal/db/migrations/001_baseline.sql b/internal/db/migrations/001_baseline.sql
deleted file mode 100644
index 2cfb54e..0000000
--- a/internal/db/migrations/001_baseline.sql
+++ /dev/null
@@ -1,182 +0,0 @@
--- The Postgres baseline: the schema as it stood at the end of the SQLite line,
--- in one file rather than ten.
---
--- The ten SQLite migrations are in git history up to the commit that introduced
--- this one, and they replay against nothing here: their shape was incremental
--- (columns added, then dropped again in 008) and 008's backfill rewrote data
--- that a Postgres install never had. An existing SQLite database is carried over
--- by scripts/sqlite-to-postgres.go, which copies rows into this schema.
---
--- Two conventions inherited deliberately:
---
--- * Timestamps are BIGINT unix seconds, not timestamptz. Everything in Go
--- already speaks epochs, and converting was a second change riding along
--- with the port. Worth revisiting on its own.
---
--- * Ids are GENERATED BY DEFAULT, not ALWAYS, so the migration script can
--- insert rows with their original ids and keep every foreign key intact.
--- setval at the end of the copy puts the sequences past them.
-
-CREATE TABLE users (
- id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
- username TEXT NOT NULL UNIQUE,
- email TEXT NOT NULL UNIQUE,
- created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
- -- Where this user's notifications go. NULL means they get none; incidents
- -- assigned to them fall back to the configured fallback topic.
- ntfy_topic TEXT,
- -- NULL means the user has no password and can only use API keys.
- password_hash TEXT
-);
-
-CREATE TABLE api_keys (
- id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
- user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
- key_hash TEXT NOT NULL UNIQUE,
- name TEXT NOT NULL,
- created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
- last_used_at BIGINT
-);
-
--- A session is a browser's credential, the cookie counterpart of an API key:
--- only the hash of the token is stored. expires_at slides forward while the
--- session is in use, so an on-call phone stays signed in.
-CREATE TABLE sessions (
- id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
- token_hash TEXT NOT NULL UNIQUE,
- user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
- created_at BIGINT NOT NULL,
- last_seen_at BIGINT NOT NULL,
- expires_at BIGINT NOT NULL,
- user_agent TEXT
-);
-
-CREATE INDEX idx_sessions_user ON sessions(user_id);
-
--- The machine-owned signal record: what Alertmanager says is true right now.
--- Workflow state lives on incidents, never here, because the webhook upsert owns
--- these rows and would overwrite it.
-CREATE TABLE alerts (
- id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
- fingerprint TEXT NOT NULL UNIQUE,
- name TEXT NOT NULL,
- status TEXT NOT NULL CHECK (status IN ('firing', 'resolved')),
- labels JSONB NOT NULL DEFAULT '{}'::jsonb,
- annotations JSONB NOT NULL DEFAULT '{}'::jsonb,
- starts_at BIGINT NOT NULL,
- ends_at BIGINT,
- generator_url TEXT NOT NULL DEFAULT '',
- received_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
- archived_at BIGINT,
- -- Why the alert left the firing state: 'alertmanager' when a resolved
- -- webhook set it, 'expiry' when the sweeper inferred it from staleness.
- resolution_source TEXT
-);
-
-CREATE INDEX alerts_status_idx ON alerts(status);
-CREATE INDEX alerts_name_idx ON alerts(name);
-CREATE INDEX alerts_received_at_idx ON alerts(received_at DESC);
-CREATE INDEX alerts_archived_at_idx ON alerts(archived_at);
-
-CREATE TABLE schedule_entries (
- id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
- user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
- date TEXT NOT NULL UNIQUE, -- YYYY-MM-DD; one person per day
- created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
-);
-
-CREATE INDEX schedule_entries_date_idx ON schedule_entries(date);
-
--- The human work item: what people acknowledge, assign, snooze, discuss and
--- resolve. Correlation uses Alertmanager's own groupKey, so incidents follow the
--- group_by routing tree the operator already tuned.
-CREATE TABLE incidents (
- id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
- group_key TEXT NOT NULL, -- Alertmanager groupKey, opaque
- title TEXT NOT NULL, -- rendered from group_labels
- group_labels JSONB NOT NULL DEFAULT '{}'::jsonb,
- status TEXT NOT NULL CHECK (status IN ('triggered', 'acknowledged', 'resolved')),
- severity TEXT, -- highest `severity` label across firing members
- triggered_at BIGINT NOT NULL,
- acknowledged_by BIGINT REFERENCES users(id) ON DELETE SET NULL,
- acknowledged_at BIGINT,
- assigned_to BIGINT REFERENCES users(id) ON DELETE SET NULL,
- snoozed_until BIGINT,
- resolved_at BIGINT,
- resolution_source TEXT, -- 'alerts' | 'manual'
- archived_at BIGINT
-);
-
--- Load-bearing: at most one OPEN incident per group_key. This is what makes
--- "resolved incident + a new alert occurrence = a new incident" work, and it is
--- the constraint the webhook's find-or-open lookup relies on.
-CREATE UNIQUE INDEX incidents_open_group_key_idx ON incidents(group_key) WHERE resolved_at IS NULL;
-CREATE INDEX incidents_status_idx ON incidents(status);
-CREATE INDEX incidents_triggered_at_idx ON incidents(triggered_at DESC);
-CREATE INDEX incidents_archived_at_idx ON incidents(archived_at);
-
--- Membership is historical, not a pointer on alerts: one alert row (one
--- fingerprint) resolves and re-fires over time and belongs to a different
--- incident each occurrence.
-CREATE TABLE incident_alerts (
- incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
- alert_id BIGINT NOT NULL REFERENCES alerts(id) ON DELETE CASCADE,
- added_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
- PRIMARY KEY (incident_id, alert_id)
-);
-
-CREATE INDEX incident_alerts_alert_id_idx ON incident_alerts(alert_id);
-
--- The timeline. Append-only, and the only history this server keeps: alert rows
--- are mutated in place, so without this there is no record that anything
--- happened. Notes are events too, so one query renders the whole story.
-CREATE TABLE incident_events (
- id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
- incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
- -- triggered | alert_added | alert_resolved | acknowledged | unacknowledged
- -- | assigned | snoozed | unsnoozed | resolved | note | notified | notify_failed
- type TEXT NOT NULL,
- user_id BIGINT REFERENCES users(id) ON DELETE SET NULL, -- NULL = the server acted
- alert_id BIGINT REFERENCES alerts(id) ON DELETE SET NULL,
- detail TEXT,
- created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
-);
-
-CREATE INDEX incident_events_incident_idx ON incident_events(incident_id, created_at);
-
--- Delivery is an outbox rather than an inline HTTP call: a POST made while
--- holding the webhook's transaction would hold a connection open across a
--- network round trip. The webhook inserts a row; the notifier goroutine
--- delivers it.
-CREATE TABLE notifications (
- id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
- incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
- -- Nullable: a notification sent to the fallback topic belongs to nobody,
- -- because nobody was on call when the incident opened.
- user_id BIGINT REFERENCES users(id) ON DELETE SET NULL,
- topic TEXT NOT NULL, -- resolved at enqueue: who was on call then
- kind TEXT NOT NULL CHECK (kind IN ('triggered', 'reminder', 'resolved')),
- created_at BIGINT NOT NULL,
- send_after BIGINT NOT NULL, -- retry backoff watermark
- attempts BIGINT NOT NULL DEFAULT 0,
- sent_at BIGINT,
- last_error TEXT -- kept after the last attempt, for debugging
-);
-
--- The delivery loop's only query: what is due and still unsent.
-CREATE INDEX notifications_pending_idx ON notifications(send_after) WHERE sent_at IS NULL;
--- Reminders and resolved notices both look up an incident's newest row.
-CREATE INDEX notifications_incident_idx ON notifications(incident_id, id DESC);
-
--- A notification body is stored on the ntfy server and cached on the device, so
--- a real API key must never appear in one. Each delivery mints its own token
--- instead: one incident, one action, one day.
-CREATE TABLE incident_ack_tokens (
- token_hash TEXT PRIMARY KEY, -- SHA-256 of the raw token, as with api_keys
- incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
- user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
- created_at BIGINT NOT NULL,
- expires_at BIGINT NOT NULL
-);
-
-CREATE INDEX incident_ack_tokens_expires_idx ON incident_ack_tokens(expires_at);
diff --git a/internal/db/migrations/001_schema.sql b/internal/db/migrations/001_schema.sql
new file mode 100644
index 0000000..a7cb016
--- /dev/null
+++ b/internal/db/migrations/001_schema.sql
@@ -0,0 +1,587 @@
+-- Terdut Server schema. One baseline: the project has not shipped, so the
+-- migration history that led here (SQLite import, a Default team, per-team
+-- deadman configs later replaced by switches) is not carried. Later changes are
+-- new numbered files after this one.
+--
+-- Timestamps are Unix epoch seconds in BIGINT columns throughout.
+
+CREATE TABLE alerts (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ fingerprint text NOT NULL,
+ name text NOT NULL,
+ status text NOT NULL,
+ labels jsonb DEFAULT '{}'::jsonb NOT NULL,
+ annotations jsonb DEFAULT '{}'::jsonb NOT NULL,
+ starts_at bigint NOT NULL,
+ ends_at bigint,
+ generator_url text DEFAULT ''::text NOT NULL,
+ received_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
+ archived_at bigint,
+ resolution_source text,
+ team_id bigint NOT NULL,
+ integration_id bigint,
+ CONSTRAINT alerts_status_check CHECK ((status = ANY (ARRAY['firing'::text, 'resolved'::text])))
+);
+
+CREATE TABLE api_keys (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ user_id bigint NOT NULL,
+ key_hash text NOT NULL,
+ name text NOT NULL,
+ created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
+ last_used_at bigint,
+ expires_at bigint
+);
+
+CREATE TABLE deadman_switches (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ team_id bigint NOT NULL,
+ name text NOT NULL,
+ matcher text NOT NULL,
+ timeout_seconds bigint NOT NULL,
+ severity text DEFAULT 'critical'::text NOT NULL,
+ created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
+ CONSTRAINT deadman_switches_timeout_seconds_check CHECK ((timeout_seconds > 0))
+);
+
+CREATE TABLE device_logins (
+ device_hash text NOT NULL,
+ user_code text NOT NULL,
+ status text DEFAULT 'pending'::text NOT NULL,
+ user_id bigint,
+ expires_at bigint NOT NULL,
+ last_polled_at bigint DEFAULT 0 NOT NULL,
+ CONSTRAINT device_logins_status_check CHECK ((status = ANY (ARRAY['pending'::text, 'approved'::text, 'denied'::text])))
+);
+
+CREATE TABLE escalation_levels (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ team_id bigint NOT NULL,
+ "position" bigint NOT NULL,
+ timeout_seconds bigint NOT NULL,
+ CONSTRAINT escalation_levels_timeout_seconds_check CHECK ((timeout_seconds > 0))
+);
+
+CREATE TABLE escalation_policies (
+ team_id bigint NOT NULL,
+ repeat_count bigint DEFAULT 0 NOT NULL,
+ fallback_topic text DEFAULT ''::text NOT NULL,
+ updated_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
+ CONSTRAINT escalation_policies_repeat_count_check CHECK (((repeat_count >= 0) AND (repeat_count <= 10)))
+);
+
+CREATE TABLE escalation_targets (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ level_id bigint NOT NULL,
+ kind text NOT NULL,
+ user_id bigint,
+ CONSTRAINT escalation_targets_check CHECK ((((kind = 'user'::text) AND (user_id IS NOT NULL)) OR ((kind = 'oncall'::text) AND (user_id IS NULL)))),
+ CONSTRAINT escalation_targets_kind_check CHECK ((kind = ANY (ARRAY['user'::text, 'oncall'::text])))
+);
+
+CREATE TABLE incident_ack_tokens (
+ token_hash text NOT NULL,
+ incident_id bigint NOT NULL,
+ user_id bigint NOT NULL,
+ created_at bigint NOT NULL,
+ expires_at bigint NOT NULL
+);
+
+CREATE TABLE incident_alerts (
+ incident_id bigint NOT NULL,
+ alert_id bigint NOT NULL,
+ added_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL
+);
+
+CREATE TABLE incident_events (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ incident_id bigint NOT NULL,
+ type text NOT NULL,
+ user_id bigint,
+ alert_id bigint,
+ detail text,
+ created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
+ service_account_id bigint,
+ actor_user_id bigint,
+ actor_service_account_id bigint,
+ CONSTRAINT incident_events_actor_xor_chk CHECK (((user_id IS NULL) OR (service_account_id IS NULL))),
+ CONSTRAINT incident_events_assign_actor_xor_chk CHECK (((actor_user_id IS NULL) OR (actor_service_account_id IS NULL)))
+);
+
+CREATE TABLE incidents (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ group_key text NOT NULL,
+ title text NOT NULL,
+ group_labels jsonb DEFAULT '{}'::jsonb NOT NULL,
+ status text NOT NULL,
+ severity text,
+ triggered_at bigint NOT NULL,
+ acknowledged_by bigint,
+ acknowledged_at bigint,
+ assigned_to bigint,
+ snoozed_until bigint,
+ resolved_at bigint,
+ resolution_source text,
+ archived_at bigint,
+ team_id bigint NOT NULL,
+ escalation_level bigint DEFAULT 0 NOT NULL,
+ escalation_level_at bigint,
+ escalation_round bigint DEFAULT 0 NOT NULL,
+ signature text DEFAULT ''::text NOT NULL,
+ acknowledged_by_service_account_id bigint,
+ CONSTRAINT incidents_ack_actor_xor_chk CHECK (((acknowledged_by IS NULL) OR (acknowledged_by_service_account_id IS NULL))),
+ CONSTRAINT incidents_status_check CHECK ((status = ANY (ARRAY['triggered'::text, 'acknowledged'::text, 'resolved'::text])))
+);
+
+CREATE TABLE integrations (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ team_id bigint NOT NULL,
+ kind text NOT NULL,
+ name text NOT NULL,
+ key_hash text NOT NULL,
+ created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
+ last_used_at bigint,
+ CONSTRAINT integrations_kind_check CHECK ((kind = 'alertmanager'::text))
+);
+
+CREATE TABLE invites (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ token_hash text NOT NULL,
+ team_id bigint NOT NULL,
+ role text NOT NULL,
+ created_by bigint,
+ created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
+ expires_at bigint NOT NULL,
+ max_uses bigint DEFAULT 1 NOT NULL,
+ uses bigint DEFAULT 0 NOT NULL,
+ revoked_at bigint,
+ CONSTRAINT invites_max_uses_check CHECK (((max_uses > 0) AND (max_uses <= 100))),
+ CONSTRAINT invites_role_check CHECK ((role = ANY (ARRAY['owner'::text, 'member'::text])))
+);
+
+CREATE TABLE notifications (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ incident_id bigint NOT NULL,
+ user_id bigint,
+ topic text NOT NULL,
+ kind text NOT NULL,
+ created_at bigint NOT NULL,
+ send_after bigint NOT NULL,
+ attempts bigint DEFAULT 0 NOT NULL,
+ sent_at bigint,
+ last_error text,
+ CONSTRAINT notifications_kind_check CHECK ((kind = ANY (ARRAY['triggered'::text, 'reminder'::text, 'resolved'::text, 'escalated'::text])))
+);
+
+CREATE TABLE oidc_logins (
+ state_hash text NOT NULL,
+ nonce text NOT NULL,
+ pkce_verifier text NOT NULL,
+ expires_at bigint NOT NULL,
+ next text DEFAULT '/'::text NOT NULL
+);
+
+CREATE TABLE rate_limit_counters (
+ key text NOT NULL,
+ window_start bigint NOT NULL,
+ count integer NOT NULL
+);
+
+CREATE TABLE schedule_entries (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ user_id bigint NOT NULL,
+ date text NOT NULL,
+ created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
+ team_id bigint NOT NULL
+);
+
+CREATE TABLE service_account_keys (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ service_account_id bigint NOT NULL,
+ key_hash text NOT NULL,
+ name text NOT NULL,
+ created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
+ last_used_at bigint
+);
+
+CREATE TABLE service_accounts (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ name text NOT NULL,
+ scope text NOT NULL,
+ team_id bigint,
+ created_by bigint,
+ created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
+ CONSTRAINT service_accounts_scope_check CHECK ((scope = ANY (ARRAY['instance'::text, 'team'::text]))),
+ CONSTRAINT service_accounts_scope_team_id_chk CHECK ((((scope = 'team'::text) AND (team_id IS NOT NULL)) OR ((scope = 'instance'::text) AND (team_id IS NULL))))
+);
+
+CREATE TABLE sessions (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ token_hash text NOT NULL,
+ user_id bigint NOT NULL,
+ created_at bigint NOT NULL,
+ last_seen_at bigint NOT NULL,
+ expires_at bigint NOT NULL,
+ user_agent text,
+ max_expires_at bigint
+);
+
+CREATE TABLE settings (
+ key text NOT NULL,
+ value text NOT NULL,
+ updated_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL
+);
+
+CREATE TABLE team_members (
+ team_id bigint NOT NULL,
+ user_id bigint NOT NULL,
+ role text NOT NULL,
+ joined_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
+ source text DEFAULT 'manual'::text NOT NULL,
+ CONSTRAINT team_members_role_check CHECK ((role = ANY (ARRAY['owner'::text, 'member'::text]))),
+ CONSTRAINT team_members_source_check CHECK ((source = ANY (ARRAY['manual'::text, 'oidc'::text])))
+);
+
+CREATE TABLE teams (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ name text NOT NULL,
+ created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
+ oidc_member_group text,
+ oidc_owner_group text,
+ -- A stable identity for a team managed by automation (terdut-operator:
+ -- "/" of its TerdutTeam), so it can find or recreate its own
+ -- team without trusting a display name. NULL for a team a person made.
+ external_id text
+);
+
+CREATE TABLE user_identities (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ user_id bigint NOT NULL,
+ issuer text NOT NULL,
+ subject text NOT NULL,
+ created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
+ last_login_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL
+);
+
+CREATE TABLE users (
+ id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
+ username text NOT NULL,
+ email text NOT NULL,
+ created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
+ ntfy_topic text,
+ password_hash text,
+ is_admin boolean DEFAULT false NOT NULL,
+ disabled_at bigint,
+ invited_via bigint,
+ onboarding_dismissed_at bigint,
+ admin_source text DEFAULT 'manual'::text NOT NULL,
+ CONSTRAINT users_admin_source_check CHECK ((admin_source = ANY (ARRAY['manual'::text, 'oidc'::text])))
+);
+
+ALTER TABLE alerts
+ ADD CONSTRAINT alerts_pkey PRIMARY KEY (id);
+
+ALTER TABLE api_keys
+ ADD CONSTRAINT api_keys_key_hash_key UNIQUE (key_hash);
+
+ALTER TABLE api_keys
+ ADD CONSTRAINT api_keys_pkey PRIMARY KEY (id);
+
+ALTER TABLE deadman_switches
+ ADD CONSTRAINT deadman_switches_pkey PRIMARY KEY (id);
+
+ALTER TABLE device_logins
+ ADD CONSTRAINT device_logins_pkey PRIMARY KEY (device_hash);
+
+ALTER TABLE device_logins
+ ADD CONSTRAINT device_logins_user_code_key UNIQUE (user_code);
+
+ALTER TABLE escalation_levels
+ ADD CONSTRAINT escalation_levels_pkey PRIMARY KEY (id);
+
+ALTER TABLE escalation_levels
+ ADD CONSTRAINT escalation_levels_team_id_position_key UNIQUE (team_id, "position");
+
+ALTER TABLE escalation_policies
+ ADD CONSTRAINT escalation_policies_pkey PRIMARY KEY (team_id);
+
+ALTER TABLE escalation_targets
+ ADD CONSTRAINT escalation_targets_pkey PRIMARY KEY (id);
+
+ALTER TABLE incident_ack_tokens
+ ADD CONSTRAINT incident_ack_tokens_pkey PRIMARY KEY (token_hash);
+
+ALTER TABLE incident_alerts
+ ADD CONSTRAINT incident_alerts_pkey PRIMARY KEY (incident_id, alert_id);
+
+ALTER TABLE incident_events
+ ADD CONSTRAINT incident_events_pkey PRIMARY KEY (id);
+
+ALTER TABLE incidents
+ ADD CONSTRAINT incidents_pkey PRIMARY KEY (id);
+
+ALTER TABLE integrations
+ ADD CONSTRAINT integrations_key_hash_key UNIQUE (key_hash);
+
+ALTER TABLE integrations
+ ADD CONSTRAINT integrations_pkey PRIMARY KEY (id);
+
+ALTER TABLE invites
+ ADD CONSTRAINT invites_pkey PRIMARY KEY (id);
+
+ALTER TABLE invites
+ ADD CONSTRAINT invites_token_hash_key UNIQUE (token_hash);
+
+ALTER TABLE notifications
+ ADD CONSTRAINT notifications_pkey PRIMARY KEY (id);
+
+ALTER TABLE oidc_logins
+ ADD CONSTRAINT oidc_logins_pkey PRIMARY KEY (state_hash);
+
+ALTER TABLE rate_limit_counters
+ ADD CONSTRAINT rate_limit_counters_pkey PRIMARY KEY (key);
+
+ALTER TABLE schedule_entries
+ ADD CONSTRAINT schedule_entries_pkey PRIMARY KEY (id);
+
+ALTER TABLE service_account_keys
+ ADD CONSTRAINT service_account_keys_key_hash_key UNIQUE (key_hash);
+
+ALTER TABLE service_account_keys
+ ADD CONSTRAINT service_account_keys_pkey PRIMARY KEY (id);
+
+ALTER TABLE service_accounts
+ ADD CONSTRAINT service_accounts_name_key UNIQUE (name);
+
+ALTER TABLE service_accounts
+ ADD CONSTRAINT service_accounts_pkey PRIMARY KEY (id);
+
+ALTER TABLE sessions
+ ADD CONSTRAINT sessions_pkey PRIMARY KEY (id);
+
+ALTER TABLE sessions
+ ADD CONSTRAINT sessions_token_hash_key UNIQUE (token_hash);
+
+ALTER TABLE settings
+ ADD CONSTRAINT settings_pkey PRIMARY KEY (key);
+
+ALTER TABLE team_members
+ ADD CONSTRAINT team_members_pkey PRIMARY KEY (team_id, user_id);
+
+ALTER TABLE teams
+ ADD CONSTRAINT teams_name_key UNIQUE (name);
+
+ALTER TABLE teams
+ ADD CONSTRAINT teams_external_id_key UNIQUE (external_id);
+
+ALTER TABLE teams
+ ADD CONSTRAINT teams_pkey PRIMARY KEY (id);
+
+ALTER TABLE user_identities
+ ADD CONSTRAINT user_identities_issuer_subject_key UNIQUE (issuer, subject);
+
+ALTER TABLE user_identities
+ ADD CONSTRAINT user_identities_pkey PRIMARY KEY (id);
+
+ALTER TABLE users
+ ADD CONSTRAINT users_email_key UNIQUE (email);
+
+ALTER TABLE users
+ ADD CONSTRAINT users_pkey PRIMARY KEY (id);
+
+ALTER TABLE users
+ ADD CONSTRAINT users_username_key UNIQUE (username);
+
+CREATE INDEX alerts_archived_at_idx ON alerts USING btree (archived_at);
+
+CREATE INDEX alerts_integration_idx ON alerts USING btree (integration_id, received_at) WHERE (integration_id IS NOT NULL);
+
+CREATE INDEX alerts_name_idx ON alerts USING btree (name);
+
+CREATE INDEX alerts_received_at_idx ON alerts USING btree (received_at DESC);
+
+CREATE INDEX alerts_status_idx ON alerts USING btree (status);
+
+CREATE UNIQUE INDEX alerts_team_fingerprint_idx ON alerts USING btree (team_id, fingerprint);
+
+CREATE INDEX alerts_team_received_idx ON alerts USING btree (team_id, received_at DESC);
+
+CREATE INDEX deadman_switches_team_idx ON deadman_switches USING btree (team_id);
+
+CREATE INDEX device_logins_expires_idx ON device_logins USING btree (expires_at);
+
+CREATE INDEX escalation_targets_level_idx ON escalation_targets USING btree (level_id);
+
+CREATE INDEX idx_sessions_user ON sessions USING btree (user_id);
+
+CREATE INDEX incident_ack_tokens_expires_idx ON incident_ack_tokens USING btree (expires_at);
+
+CREATE INDEX incident_alerts_alert_id_idx ON incident_alerts USING btree (alert_id);
+
+CREATE INDEX incident_events_actor_service_account_id_idx ON incident_events USING btree (actor_service_account_id);
+
+CREATE INDEX incident_events_actor_user_id_idx ON incident_events USING btree (actor_user_id);
+
+CREATE INDEX incident_events_incident_idx ON incident_events USING btree (incident_id, created_at);
+
+CREATE INDEX incident_events_service_account_id_idx ON incident_events USING btree (service_account_id);
+
+CREATE INDEX incidents_acknowledged_by_service_account_id_idx ON incidents USING btree (acknowledged_by_service_account_id);
+
+CREATE INDEX incidents_archived_at_idx ON incidents USING btree (archived_at);
+
+CREATE INDEX incidents_escalation_idx ON incidents USING btree (escalation_level_at) WHERE ((resolved_at IS NULL) AND (status = 'triggered'::text));
+
+CREATE UNIQUE INDEX incidents_open_group_key_idx ON incidents USING btree (team_id, group_key) WHERE (resolved_at IS NULL);
+
+CREATE INDEX incidents_signature_idx ON incidents USING btree (team_id, signature, triggered_at DESC);
+
+CREATE INDEX incidents_status_idx ON incidents USING btree (status);
+
+CREATE INDEX incidents_team_triggered_idx ON incidents USING btree (team_id, triggered_at DESC);
+
+CREATE INDEX incidents_triggered_at_idx ON incidents USING btree (triggered_at DESC);
+
+CREATE INDEX integrations_team_idx ON integrations USING btree (team_id);
+
+-- A name identifies an integration (and a switch) within its team, so a client
+-- that manages them declaratively can look one up by name instead of listing
+-- and matching.
+CREATE UNIQUE INDEX integrations_team_name_key ON integrations (team_id, name);
+CREATE UNIQUE INDEX deadman_switches_team_name_key ON deadman_switches (team_id, name);
+
+CREATE INDEX invites_team_idx ON invites USING btree (team_id);
+
+CREATE INDEX notifications_incident_idx ON notifications USING btree (incident_id, id DESC);
+
+CREATE INDEX notifications_pending_idx ON notifications USING btree (send_after) WHERE (sent_at IS NULL);
+
+CREATE INDEX oidc_logins_expires_idx ON oidc_logins USING btree (expires_at);
+
+CREATE INDEX schedule_entries_date_idx ON schedule_entries USING btree (date);
+
+CREATE UNIQUE INDEX schedule_entries_team_date_idx ON schedule_entries USING btree (team_id, date);
+
+CREATE INDEX service_account_keys_service_account_id_idx ON service_account_keys USING btree (service_account_id);
+
+CREATE INDEX service_accounts_team_id_idx ON service_accounts USING btree (team_id);
+
+CREATE INDEX team_members_user_idx ON team_members USING btree (user_id);
+
+CREATE INDEX user_identities_user_idx ON user_identities USING btree (user_id);
+
+CREATE INDEX users_is_admin_idx ON users USING btree (is_admin) WHERE is_admin;
+
+ALTER TABLE alerts
+ ADD CONSTRAINT alerts_integration_id_fkey FOREIGN KEY (integration_id) REFERENCES integrations(id) ON DELETE SET NULL;
+
+ALTER TABLE alerts
+ ADD CONSTRAINT alerts_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
+
+ALTER TABLE api_keys
+ ADD CONSTRAINT api_keys_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
+
+ALTER TABLE deadman_switches
+ ADD CONSTRAINT deadman_switches_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
+
+ALTER TABLE device_logins
+ ADD CONSTRAINT device_logins_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
+
+ALTER TABLE escalation_levels
+ ADD CONSTRAINT escalation_levels_team_id_fkey FOREIGN KEY (team_id) REFERENCES escalation_policies(team_id) ON DELETE CASCADE;
+
+ALTER TABLE escalation_policies
+ ADD CONSTRAINT escalation_policies_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
+
+ALTER TABLE escalation_targets
+ ADD CONSTRAINT escalation_targets_level_id_fkey FOREIGN KEY (level_id) REFERENCES escalation_levels(id) ON DELETE CASCADE;
+
+ALTER TABLE escalation_targets
+ ADD CONSTRAINT escalation_targets_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
+
+ALTER TABLE incident_ack_tokens
+ ADD CONSTRAINT incident_ack_tokens_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE;
+
+ALTER TABLE incident_ack_tokens
+ ADD CONSTRAINT incident_ack_tokens_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
+
+ALTER TABLE incident_alerts
+ ADD CONSTRAINT incident_alerts_alert_id_fkey FOREIGN KEY (alert_id) REFERENCES alerts(id) ON DELETE CASCADE;
+
+ALTER TABLE incident_alerts
+ ADD CONSTRAINT incident_alerts_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE;
+
+ALTER TABLE incident_events
+ ADD CONSTRAINT incident_events_actor_service_account_id_fkey FOREIGN KEY (actor_service_account_id) REFERENCES service_accounts(id) ON DELETE SET NULL;
+
+ALTER TABLE incident_events
+ ADD CONSTRAINT incident_events_actor_user_id_fkey FOREIGN KEY (actor_user_id) REFERENCES users(id) ON DELETE SET NULL;
+
+ALTER TABLE incident_events
+ ADD CONSTRAINT incident_events_alert_id_fkey FOREIGN KEY (alert_id) REFERENCES alerts(id) ON DELETE SET NULL;
+
+ALTER TABLE incident_events
+ ADD CONSTRAINT incident_events_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE;
+
+ALTER TABLE incident_events
+ ADD CONSTRAINT incident_events_service_account_id_fkey FOREIGN KEY (service_account_id) REFERENCES service_accounts(id) ON DELETE SET NULL;
+
+ALTER TABLE incident_events
+ ADD CONSTRAINT incident_events_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE SET NULL;
+
+ALTER TABLE incidents
+ ADD CONSTRAINT incidents_acknowledged_by_fkey FOREIGN KEY (acknowledged_by) REFERENCES users(id) ON DELETE SET NULL;
+
+ALTER TABLE incidents
+ ADD CONSTRAINT incidents_acknowledged_by_service_account_id_fkey FOREIGN KEY (acknowledged_by_service_account_id) REFERENCES service_accounts(id) ON DELETE SET NULL;
+
+ALTER TABLE incidents
+ ADD CONSTRAINT incidents_assigned_to_fkey FOREIGN KEY (assigned_to) REFERENCES users(id) ON DELETE SET NULL;
+
+ALTER TABLE incidents
+ ADD CONSTRAINT incidents_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
+
+ALTER TABLE integrations
+ ADD CONSTRAINT integrations_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
+
+ALTER TABLE invites
+ ADD CONSTRAINT invites_created_by_fkey FOREIGN KEY (created_by) REFERENCES users(id) ON DELETE SET NULL;
+
+ALTER TABLE invites
+ ADD CONSTRAINT invites_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
+
+ALTER TABLE notifications
+ ADD CONSTRAINT notifications_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE;
+
+ALTER TABLE notifications
+ ADD CONSTRAINT notifications_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE SET NULL;
+
+ALTER TABLE schedule_entries
+ ADD CONSTRAINT schedule_entries_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
+
+ALTER TABLE schedule_entries
+ ADD CONSTRAINT schedule_entries_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
+
+ALTER TABLE service_account_keys
+ ADD CONSTRAINT service_account_keys_service_account_id_fkey FOREIGN KEY (service_account_id) REFERENCES service_accounts(id) ON DELETE CASCADE;
+
+ALTER TABLE service_accounts
+ ADD CONSTRAINT service_accounts_created_by_fkey FOREIGN KEY (created_by) REFERENCES users(id) ON DELETE SET NULL;
+
+ALTER TABLE service_accounts
+ ADD CONSTRAINT service_accounts_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
+
+ALTER TABLE sessions
+ ADD CONSTRAINT sessions_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
+
+ALTER TABLE team_members
+ ADD CONSTRAINT team_members_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
+
+ALTER TABLE team_members
+ ADD CONSTRAINT team_members_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
+
+ALTER TABLE user_identities
+ ADD CONSTRAINT user_identities_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
+
+ALTER TABLE users
+ ADD CONSTRAINT users_invited_via_fkey FOREIGN KEY (invited_via) REFERENCES invites(id) ON DELETE SET NULL;
diff --git a/internal/db/migrations/002_admin_role.sql b/internal/db/migrations/002_admin_role.sql
deleted file mode 100644
index 99cc095..0000000
--- a/internal/db/migrations/002_admin_role.sql
+++ /dev/null
@@ -1,25 +0,0 @@
--- A system administrator role, and the first thing in this server that one user
--- can do and another cannot.
---
--- Until now every authenticated caller could create and delete users, set
--- anybody's password and mint API keys for anybody — auth.go said so in a
--- comment. That was defensible with one operator and a hand-made account; it is
--- not once people sign themselves up (see #7).
---
--- EVERY EXISTING USER BECOMES AN ADMIN. They already hold these powers, so
--- this migration changes nobody's access: it names what is already true, and
--- leaves demotion as a deliberate act somebody performs afterwards. The
--- alternative — promoting only user 1 — would silently strip the others, and
--- could leave an install whose only admin is an account nobody has a password
--- for.
---
--- New users are not admins: the column defaults to false, and the only ways to
--- become one are this backfill, the bootstrap endpoint, or an existing admin
--- granting it.
-ALTER TABLE users ADD COLUMN is_admin BOOLEAN NOT NULL DEFAULT false;
-
-UPDATE users SET is_admin = true;
-
--- The queue's assignment dropdown and the on-call schedule read every user, and
--- the admin screens in #5 will filter on this.
-CREATE INDEX users_is_admin_idx ON users(is_admin) WHERE is_admin;
diff --git a/internal/db/migrations/003_teams.sql b/internal/db/migrations/003_teams.sql
deleted file mode 100644
index ea1358d..0000000
--- a/internal/db/migrations/003_teams.sql
+++ /dev/null
@@ -1,103 +0,0 @@
--- Teams: the unit of tenancy. Everything a person works on now belongs to one.
---
--- Until this migration the install was one shared space — every user saw every
--- alert and every incident, and the Alertmanager webhook was unauthenticated, so
--- anything that could reach the port could open an incident for everybody.
---
--- The shape, in one paragraph: a team owns its incidents, alerts, schedule and
--- integrations. A user belongs to as many teams as they like, with a role in
--- each: an `owner` configures the team, a `member` works its incidents. An
--- integration key is what an alert arrives on, and the key is what says which
--- team the alert belongs to.
---
--- EVERYTHING EXISTING MOVES INTO ONE DEFAULT TEAM, and every existing user
--- becomes an owner of it. That keeps an upgrade a no-op for the people using it:
--- the same queue, the same schedule, the same incidents, with a name on them.
-
-CREATE TABLE teams (
- id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
- name TEXT NOT NULL UNIQUE,
- created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
-);
-
--- role is free text with a CHECK rather than an enum, so adding a third role
--- later is a migration and not a type rewrite.
-CREATE TABLE team_members (
- team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
- user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
- role TEXT NOT NULL CHECK (role IN ('owner', 'member')),
- joined_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
- PRIMARY KEY (team_id, user_id)
-);
-
-CREATE INDEX team_members_user_idx ON team_members(user_id);
-
--- How alerts get in, and the only thing that says which team they belong to.
--- The key is stored as a SHA-256 hash, like api_keys and the ack tokens: a
--- leaked database gives nobody the ability to post alerts.
-CREATE TABLE integrations (
- id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
- team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
- kind TEXT NOT NULL CHECK (kind IN ('alertmanager')),
- name TEXT NOT NULL,
- key_hash TEXT NOT NULL UNIQUE,
- created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
- last_used_at BIGINT
-);
-
-CREATE INDEX integrations_team_idx ON integrations(team_id);
-
--- ---------------------------------------------------------------------------
--- The default team, and everything that already exists moving into it.
---
--- Created unconditionally, even on an empty install, so there is always a team
--- for the bootstrap user to land in and for the first integration to hang off.
--- ---------------------------------------------------------------------------
-
-INSERT INTO teams (name) VALUES ('Default');
-
-INSERT INTO team_members (team_id, user_id, role)
-SELECT (SELECT id FROM teams WHERE name = 'Default'), id, 'owner' FROM users;
-
--- ---------------------------------------------------------------------------
--- team_id on everything a team owns.
---
--- Added nullable, backfilled, then made NOT NULL: adding a NOT NULL column with
--- no default to a table with rows is rejected, and a DEFAULT pointing at the
--- default team would quietly keep working after the default team is gone.
--- ---------------------------------------------------------------------------
-
-ALTER TABLE alerts ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
-ALTER TABLE incidents ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
-ALTER TABLE schedule_entries ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
-
-UPDATE alerts SET team_id = (SELECT id FROM teams WHERE name = 'Default');
-UPDATE incidents SET team_id = (SELECT id FROM teams WHERE name = 'Default');
-UPDATE schedule_entries SET team_id = (SELECT id FROM teams WHERE name = 'Default');
-
-ALTER TABLE alerts ALTER COLUMN team_id SET NOT NULL;
-ALTER TABLE incidents ALTER COLUMN team_id SET NOT NULL;
-ALTER TABLE schedule_entries ALTER COLUMN team_id SET NOT NULL;
-
--- ---------------------------------------------------------------------------
--- The uniqueness rules were all written for one tenant, and every one of them
--- is wrong now: two teams monitoring two clusters legitimately see the same
--- fingerprint, the same groupKey, and want somebody on call on the same day.
--- ---------------------------------------------------------------------------
-
-ALTER TABLE alerts DROP CONSTRAINT alerts_fingerprint_key;
-CREATE UNIQUE INDEX alerts_team_fingerprint_idx ON alerts(team_id, fingerprint);
-
-DROP INDEX incidents_open_group_key_idx;
--- Still load-bearing, now per team: at most one OPEN incident per group_key
--- within a team. This is what makes "resolved incident + a new alert occurrence
--- = a new incident" work, and what the webhook's find-or-open lookup relies on.
-CREATE UNIQUE INDEX incidents_open_group_key_idx
- ON incidents(team_id, group_key) WHERE resolved_at IS NULL;
-
-ALTER TABLE schedule_entries DROP CONSTRAINT schedule_entries_date_key;
-CREATE UNIQUE INDEX schedule_entries_team_date_idx ON schedule_entries(team_id, date);
-
--- The list views all filter by team first.
-CREATE INDEX alerts_team_received_idx ON alerts(team_id, received_at DESC);
-CREATE INDEX incidents_team_triggered_idx ON incidents(team_id, triggered_at DESC);
diff --git a/internal/db/migrations/004_team_deadman.sql b/internal/db/migrations/004_team_deadman.sql
deleted file mode 100644
index 3a066fd..0000000
--- a/internal/db/migrations/004_team_deadman.sql
+++ /dev/null
@@ -1,39 +0,0 @@
--- Dead man's switches become a team's own configuration.
---
--- They were three environment variables — TERDUT_DEADMAN_MATCHERS, _TIMEOUT and
--- _SEVERITY — which made them one setting for the whole install. That was the
--- last piece of the alerting path a team could not control: a team could take
--- its own alerts on its own key and still not say which of them were
--- heartbeats, or how long a silence had to last before somebody was paged.
---
--- One row per team rather than one row per switch. The unit of monitoring is
--- still the fingerprint, as it always was — two clusters sending the same
--- heartbeat alertname are two independent switches — and the matcher string
--- keeps the format the environment variable used, so a value can be moved from
--- one to the other unchanged.
---
--- No rows are seeded here: a migration cannot read the environment. The server
--- inserts a row per team at startup from its own configuration, and the same
--- values therefore carry forward into the first team's row without anybody
--- retyping them. See seedDeadmanConfigs.
-CREATE TABLE deadman_configs (
- team_id BIGINT PRIMARY KEY REFERENCES teams(id) ON DELETE CASCADE,
-
- -- ";" separates matchers, "," the label conditions within one, "=" is exact
- -- equality: `alertname=Watchdog,cluster=prod; alertname=EdgeHeartbeat`.
- -- Every matcher must name an alertname. Empty watches nothing.
- matchers TEXT NOT NULL DEFAULT '',
-
- -- Seconds rather than a Go duration string: the column is compared and
- -- arithmetic is done on it, and a value that has to be parsed before it can
- -- be believed is a value that can be stored unparseable. Zero disables the
- -- team's switches entirely.
- timeout_seconds BIGINT NOT NULL DEFAULT 0,
-
- -- The severity these incidents open at. They have no member alerts to
- -- derive one from, and a heartbeat's own severity label is meaningless —
- -- Watchdog ships as "none".
- severity TEXT NOT NULL DEFAULT 'critical',
-
- updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
-);
diff --git a/internal/db/migrations/005_settings.sql b/internal/db/migrations/005_settings.sql
deleted file mode 100644
index c9e060c..0000000
--- a/internal/db/migrations/005_settings.sql
+++ /dev/null
@@ -1,35 +0,0 @@
--- Settings that an administrator can change without a redeploy, and the flag
--- that takes an account out of use without deleting it.
---
--- Three of the server's tunables were environment variables, which meant
--- changing how long an incident waits before it is paged again required editing
--- a chart, merging it, and waiting for a reconcile. They are behaviour, not
--- infrastructure, and the difference is who needs to change them and how often.
---
--- What stays in the environment: the ntfy URL and token, the database DSN, the
--- listen address and the public URL. Those are where the server is plugged in
--- rather than how it behaves, they are needed before the database is open, and
--- two of them are credentials.
---
--- Key/value rather than a column per setting. A settings table with one row and
--- a column per knob needs a migration for every new knob, and #6 and #7 will
--- both add some. The cost is that values are text and the accessor has to say
--- what type it wanted; settings.go does that in one place.
---
--- No rows are seeded here: a migration cannot read the environment. The server
--- inserts each key from its own configuration at startup, once, so an install
--- that upgrades keeps exactly the behaviour it had. See SeedSettings.
-CREATE TABLE settings (
- key TEXT PRIMARY KEY,
- value TEXT NOT NULL,
- updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
-);
-
--- Disabling an account rather than deleting it: the person has left, or the
--- credential is suspect, and their incidents, acknowledgements and timeline
--- entries must stay exactly where they are. Deleting a user nulls their
--- acknowledged_by and assigned_to, which quietly rewrites history.
---
--- A disabled user cannot sign in and their API keys stop working, but they are
--- still a name the timeline can show and still a member of their teams.
-ALTER TABLE users ADD COLUMN disabled_at BIGINT;
diff --git a/internal/db/migrations/006_escalation.sql b/internal/db/migrations/006_escalation.sql
deleted file mode 100644
index 245567d..0000000
--- a/internal/db/migrations/006_escalation.sql
+++ /dev/null
@@ -1,95 +0,0 @@
--- Escalation: page somebody else when the first person does not answer.
---
--- This is the gap the whole multi-tenancy line of work was opened to close.
--- Until now an unacknowledged incident re-paged the same topic every
--- notify_repeat forever, which is a louder version of the same silence: if the
--- person on call is asleep, has no signal, or has left, nothing else happens.
---
--- Shape: one policy per team, an ordered list of levels, each level with a
--- timeout and a set of targets. When a level's timeout passes and the incident
--- is still triggered, the next level is paged. When the last level passes, the
--- chain repeats repeat_count times, and then the team's fallback topic is paged
--- once as the end of the line.
---
--- A team WITHOUT a policy keeps exactly today's behaviour: page the assignee,
--- then remind on the same topic. Escalation is opt-in per team, and the two
--- never both run for one incident -- see enqueueReminders.
-CREATE TABLE escalation_policies (
- -- One per team for now, hence the team as the key rather than an id with a
- -- unique index: routing different alerts to different chains needs the
- -- alert to carry something to route ON, which is a separate question.
- team_id BIGINT PRIMARY KEY REFERENCES teams(id) ON DELETE CASCADE,
-
- -- How many extra times to run the whole chain after it has been walked
- -- once. 0 means walk it once and stop at the fallback.
- repeat_count BIGINT NOT NULL DEFAULT 0 CHECK (repeat_count >= 0 AND repeat_count <= 10),
-
- -- Where the last page goes when every level has been tried. Per team now:
- -- TERDUT_NTFY_FALLBACK_TOPIC was one topic for the whole install, which in
- -- a multi-team server pages the wrong people. Empty means the chain simply
- -- ends.
- fallback_topic TEXT NOT NULL DEFAULT '',
-
- updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
-);
-
-CREATE TABLE escalation_levels (
- id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
- team_id BIGINT NOT NULL REFERENCES escalation_policies(team_id) ON DELETE CASCADE,
- -- 1-based, dense. The API rewrites the whole ladder on every edit rather
- -- than patching one rung, so there is no way to leave a gap.
- position BIGINT NOT NULL,
- -- How long this level has to produce an acknowledgement before the next one
- -- is paged. Seconds, like every other duration in this schema.
- timeout_seconds BIGINT NOT NULL CHECK (timeout_seconds > 0),
-
- UNIQUE (team_id, position)
-);
-
--- Who a level pages. Either a named person, or whoever the team's rota says is
--- on call today -- which is the target that keeps working when the rota
--- changes and nobody remembers to edit the policy.
-CREATE TABLE escalation_targets (
- id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
- level_id BIGINT NOT NULL REFERENCES escalation_levels(id) ON DELETE CASCADE,
- kind TEXT NOT NULL CHECK (kind IN ('user', 'oncall')),
- -- Set for kind='user', NULL for kind='oncall'.
- user_id BIGINT REFERENCES users(id) ON DELETE CASCADE,
-
- CHECK ((kind = 'user' AND user_id IS NOT NULL) OR (kind = 'oncall' AND user_id IS NULL))
-);
-
-CREATE INDEX escalation_targets_level_idx ON escalation_targets(level_id);
-
--- ---------------------------------------------------------------------------
--- Where an incident is in its chain.
---
--- On the incident rather than in a side table: it is read on every notifier
--- tick alongside the incident's status, and one row per incident is exactly
--- what the state is.
--- ---------------------------------------------------------------------------
-
--- 0 means no level has been paged yet, which is the state of every incident
--- that existed before escalation and of every incident in a team with no
--- policy. 1 is the first level.
-ALTER TABLE incidents ADD COLUMN escalation_level BIGINT NOT NULL DEFAULT 0;
-
--- When the current level was entered, and therefore what its timeout is
--- measured from. NULL while escalation_level is 0.
-ALTER TABLE incidents ADD COLUMN escalation_level_at BIGINT;
-
--- How many times the chain has been walked in full. Compared against the
--- policy's repeat_count.
-ALTER TABLE incidents ADD COLUMN escalation_round BIGINT NOT NULL DEFAULT 0;
-
--- The notifier's escalation query: incidents still waiting, oldest level first.
-CREATE INDEX incidents_escalation_idx
- ON incidents(escalation_level_at)
- WHERE resolved_at IS NULL AND status = 'triggered';
-
--- 'escalated' joins the outbox kinds: a page that went out because nobody
--- answered the last one, which is worth telling apart from the first page and
--- from a reminder when reading the timeline or debugging a delivery.
-ALTER TABLE notifications DROP CONSTRAINT notifications_kind_check;
-ALTER TABLE notifications ADD CONSTRAINT notifications_kind_check
- CHECK (kind IN ('triggered', 'reminder', 'resolved', 'escalated'));
diff --git a/internal/db/migrations/007_signup_invites.sql b/internal/db/migrations/007_signup_invites.sql
deleted file mode 100644
index c615353..0000000
--- a/internal/db/migrations/007_signup_invites.sql
+++ /dev/null
@@ -1,49 +0,0 @@
--- Self-service sign-up, and the invite links that make it useful.
---
--- Until now the only way to get an account was for somebody who already had one
--- to create it, and the login page told people to "ask an admin". That is a
--- workable arrangement for one operator and an impossible one for a team.
---
--- An invite is a link, not an email: this server has no SMTP and adding it to
--- send one message would be a new subsystem to run, secure and monitor. The
--- person inviting sends the link however they already talk to the person they
--- are inviting.
-CREATE TABLE invites (
- id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
-
- -- SHA-256 of the raw token, like api_keys, the integration keys and the
- -- acknowledgement tokens. A leaked database hands nobody an account.
- token_hash TEXT NOT NULL UNIQUE,
-
- -- Which team the invitee lands in, and as what. An invite always names a
- -- team: an account in no team sees an empty queue and can be paged by
- -- nobody, which is not a state to invite somebody into.
- team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
- role TEXT NOT NULL CHECK (role IN ('owner', 'member')),
-
- created_by BIGINT REFERENCES users(id) ON DELETE SET NULL,
- created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
-
- -- Invites expire. A link that works forever is a credential nobody
- -- remembers issuing, sitting in a chat log.
- expires_at BIGINT NOT NULL,
-
- -- Single-use by default: max_uses 1. A team onboarding six people at once
- -- can raise it rather than minting six links.
- max_uses BIGINT NOT NULL DEFAULT 1 CHECK (max_uses > 0 AND max_uses <= 100),
- uses BIGINT NOT NULL DEFAULT 0,
-
- -- Revoked by hand, separately from expiry, so "this link is no longer
- -- wanted" and "this link timed out" stay distinguishable in the listing.
- revoked_at BIGINT
-);
-
-CREATE INDEX invites_team_idx ON invites(team_id);
-
--- Who redeemed which invite. Kept after the invite is gone — the answer to "how
--- did this account get here" should outlive the link that made it.
-ALTER TABLE users ADD COLUMN invited_via BIGINT REFERENCES invites(id) ON DELETE SET NULL;
-
--- Where a person is in the first-run checklist, so it can be resumed and
--- dismissed rather than nagging forever. One row per user, created on demand.
-ALTER TABLE users ADD COLUMN onboarding_dismissed_at BIGINT;
diff --git a/internal/db/migrations/008_incident_signature.sql b/internal/db/migrations/008_incident_signature.sql
deleted file mode 100644
index 3e63b31..0000000
--- a/internal/db/migrations/008_incident_signature.sql
+++ /dev/null
@@ -1,23 +0,0 @@
--- Similar incidents: a signature per incident, so "has this happened before"
--- is an indexed equality instead of a search.
---
--- The signature is the alert name plus the group labels that identify WHAT is
--- broken, minus the ones that only say WHERE it happened to run this time
--- (instance, pod, ...). Two incidents with the same signature in the same team
--- are the same problem for a responder's purposes.
---
--- Computed in Go for new incidents (incidentSignature in incident_store.go).
--- The backfill below MUST produce the same string; keep the volatile list in
--- both places in step.
-ALTER TABLE incidents ADD COLUMN signature TEXT NOT NULL DEFAULT '';
-
-UPDATE incidents SET signature =
- COALESCE(NULLIF(group_labels->>'alertname', ''), title) || '|' ||
- COALESCE((
- SELECT string_agg(e.k || '=' || e.v, ',' ORDER BY e.k)
- FROM jsonb_each_text(incidents.group_labels) AS e(k, v)
- WHERE e.k <> 'alertname'
- AND e.k NOT IN ('instance', 'pod', 'pod_name', 'pod_ip', 'container', 'container_name', 'endpoint')
- ), '');
-
-CREATE INDEX incidents_signature_idx ON incidents(team_id, signature, triggered_at DESC);
diff --git a/internal/db/migrations/009_deadman_switches.sql b/internal/db/migrations/009_deadman_switches.sql
deleted file mode 100644
index e693d46..0000000
--- a/internal/db/migrations/009_deadman_switches.sql
+++ /dev/null
@@ -1,54 +0,0 @@
--- Dead man's switches become rows of their own.
---
--- 004 kept a team's switches in one string with one timeout and one severity,
--- which was enough to configure them and not enough to show them: there was no
--- thing to list, nothing to hang a status on, and every switch in a team had to
--- share a deadline. A row per switch gives each its own name, matcher, timeout
--- and severity, and gives the Team → Switches page something to be a list of.
---
--- The matcher keeps the syntax the string used, one matcher per row:
--- `alertname=Watchdog,cluster=prod`. The unit of monitoring is still the
--- fingerprint, so a matcher that many clusters satisfy is still one switch row
--- watching several independent heartbeats.
-CREATE TABLE deadman_switches (
- id BIGSERIAL PRIMARY KEY,
- team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
-
- -- What the owner calls it. Defaults to the matcher when they do not say.
- name TEXT NOT NULL,
-
- -- "," separates the label conditions, "=" is exact equality, and alertname is
- -- mandatory: it is what keeps the sweeper's candidate query on an index.
- matcher TEXT NOT NULL,
-
- -- Seconds of silence before the switch is declared dead. Never zero: a switch
- -- that cannot fire is deleted, not disabled.
- timeout_seconds BIGINT NOT NULL CHECK (timeout_seconds > 0),
-
- -- The severity its incidents open at. See 004 for why they carry their own.
- severity TEXT NOT NULL DEFAULT 'critical',
-
- created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
-);
-
-CREATE INDEX deadman_switches_team_idx ON deadman_switches (team_id);
-
--- Carry every team's configuration over, one row per matcher. A team whose
--- timeout was zero had switches turned off, which is now "no rows".
-INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity)
-SELECT c.team_id, btrim(m), btrim(m), c.timeout_seconds, c.severity
-FROM deadman_configs c,
- LATERAL regexp_split_to_table(c.matchers, ';') AS m
-WHERE c.timeout_seconds > 0
- AND btrim(m) <> ''
-ORDER BY c.team_id;
-
--- The server seeds environment defaults into teams once, and remembers that it
--- did. An install that had a row per team was already seeded; without this
--- marker the first start after upgrading would seed teams that had switched
--- theirs off.
-INSERT INTO settings (key, value)
-SELECT 'deadman_seeded', '1'
-WHERE EXISTS (SELECT 1 FROM deadman_configs);
-
-DROP TABLE deadman_configs;
diff --git a/internal/db/migrations/010_alert_source.sql b/internal/db/migrations/010_alert_source.sql
deleted file mode 100644
index f397fad..0000000
--- a/internal/db/migrations/010_alert_source.sql
+++ /dev/null
@@ -1,21 +0,0 @@
--- Which alert source an alert last arrived on.
---
--- Team -> Sources shows when each source last posted, which integrations
--- already knew (last_used_at, stamped on every webhook). What it could not say
--- was what a source delivered: an alert never recorded the key it came in on, so
--- "prod alertmanager" and "staging alertmanager" were indistinguishable once
--- inside. This column is that link, and lets the page show each source's last
--- alert and how many alerts it has kept fresh over the past day.
---
--- Last sender wins: every accepted payload restamps it, the way it advances
--- received_at. Two sources posting the same fingerprint into one team is
--- already one alert, and it is attributed to whichever spoke last.
---
--- Nullable, and not backfilled. Alerts that arrived before this migration have
--- no source, and NULL says so honestly rather than guessing. It heals by itself:
--- Alertmanager re-sends every alert each repeat_interval, and each re-send is an
--- accepted payload. Deleting a source keeps its alerts, unattributed.
-ALTER TABLE alerts ADD COLUMN integration_id BIGINT REFERENCES integrations(id) ON DELETE SET NULL;
-
-CREATE INDEX alerts_integration_idx ON alerts (integration_id, received_at)
- WHERE integration_id IS NOT NULL;
diff --git a/internal/db/migrations/011_oidc.sql b/internal/db/migrations/011_oidc.sql
deleted file mode 100644
index 557d72c..0000000
--- a/internal/db/migrations/011_oidc.sql
+++ /dev/null
@@ -1,60 +0,0 @@
--- Single sign-on through an OpenID Connect provider (Authentik, and anything
--- else that speaks OIDC).
---
--- Four things change, and none of them touches a password user: every new column
--- has a default that says "this is how it has always worked".
---
--- 1. user_identities says which provider account a user is. It is keyed on
--- (issuer, subject), never on email or username: those are mutable at the
--- provider, and a recycled address must not inherit somebody's account. A
--- user can have several identities (a second provider later), and none at all
--- (a local, password-only user), which is why this is a table and not two
--- columns on users.
---
--- 2. team_members.source and users.admin_source record who granted a role. 'oidc'
--- rows are owned by the group sync: it adds them when a group grants access
--- and removes them when it stops, and nothing else may edit them. 'manual' rows
--- are everything that existed before this migration, and are never touched by
--- the sync. Without the marker the sync could not tell a membership it created
--- from one an owner added by hand, and would have to either leave stale access
--- behind or delete people it had no business deleting.
---
--- 3. sessions.max_expires_at is a hard ceiling on a session's life. Ordinary
--- sessions slide for as long as they are used; a session made by an SSO login
--- must not, because the login is the only moment the groups are re-read.
--- Capping the session is what makes "removed from the group in the provider"
--- take effect within a bounded time. NULL means no ceiling.
---
--- 4. oidc_logins holds a login that has been started and not yet finished: the
--- state, nonce and PKCE verifier the callback must see again. A row rather
--- than a signed cookie, so it survives a restart and needs no signing key.
--- Only the hash of the state is stored, like every other token here; the
--- nonce and verifier are useless without the state that names the row.
-CREATE TABLE user_identities (
- id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
- user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
- issuer TEXT NOT NULL,
- subject TEXT NOT NULL,
- created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
- last_login_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
- UNIQUE (issuer, subject)
-);
-
-CREATE INDEX user_identities_user_idx ON user_identities (user_id);
-
-ALTER TABLE team_members
- ADD COLUMN source TEXT NOT NULL DEFAULT 'manual' CHECK (source IN ('manual', 'oidc'));
-
-ALTER TABLE users
- ADD COLUMN admin_source TEXT NOT NULL DEFAULT 'manual' CHECK (admin_source IN ('manual', 'oidc'));
-
-ALTER TABLE sessions ADD COLUMN max_expires_at BIGINT;
-
-CREATE TABLE oidc_logins (
- state_hash TEXT PRIMARY KEY,
- nonce TEXT NOT NULL,
- pkce_verifier TEXT NOT NULL,
- expires_at BIGINT NOT NULL
-);
-
-CREATE INDEX oidc_logins_expires_idx ON oidc_logins (expires_at);
diff --git a/internal/db/migrations/012_device_login.sql b/internal/db/migrations/012_device_login.sql
deleted file mode 100644
index 1764e62..0000000
--- a/internal/db/migrations/012_device_login.sql
+++ /dev/null
@@ -1,40 +0,0 @@
--- Signing in from a terminal, for clients that cannot open a browser on the
--- machine they run on (the TUI over SSH is the reason).
---
--- The flow is the OAuth device authorization grant, run by terdut itself rather
--- than the identity provider, so the terminal never talks to the provider and
--- the server issues its ordinary session at the end:
---
--- 1. The terminal asks for a login and gets two secrets: a device code it
--- keeps and polls with, and a short user code it shows the person.
--- 2. The person opens the verification URL on any device, signs in by whatever
--- means the server offers, sees the user code, and approves it.
--- 3. The terminal's next poll finds the row approved and is given a session.
---
--- Only the hash of the device code is stored, like every other token here: the
--- device code is what earns a session, so a database read must not yield one.
--- The user code is shown on screens and typed by people, so it is stored as is;
--- on its own it can only be approved, never redeemed.
---
--- user_id is the person who approved. It is empty until then, and the session
--- is minted at redemption, not at approval: an approval nobody collects must not
--- leave a live session lying about.
---
--- last_polled_at lets the server refuse a client that polls faster than the
--- interval it was told.
-CREATE TABLE device_logins (
- device_hash TEXT PRIMARY KEY,
- user_code TEXT NOT NULL UNIQUE,
- status TEXT NOT NULL DEFAULT 'pending' CHECK (status IN ('pending', 'approved', 'denied')),
- user_id BIGINT REFERENCES users(id) ON DELETE CASCADE,
- expires_at BIGINT NOT NULL,
- last_polled_at BIGINT NOT NULL DEFAULT 0
-);
-
-CREATE INDEX device_logins_expires_idx ON device_logins (expires_at);
-
--- Where to send the browser once a single sign-on login completes. A person who
--- opens /device?code=... without a session has to sign in first and then come
--- back to it, and the same is true of any other deep link. Validated when it is
--- stored: only a path on this server is ever kept.
-ALTER TABLE oidc_logins ADD COLUMN next TEXT NOT NULL DEFAULT '/';
diff --git a/internal/db/migrations/013_oidc_team_groups.sql b/internal/db/migrations/013_oidc_team_groups.sql
deleted file mode 100644
index d269402..0000000
--- a/internal/db/migrations/013_oidc_team_groups.sql
+++ /dev/null
@@ -1,26 +0,0 @@
--- Per-team OIDC group configuration, replacing the global
--- TERDUT_OIDC_GROUP_MAPPINGS env var.
---
--- Group -> team -> role used to be one global list an operator set for the
--- whole install, matched against a team by name, and the sync would create
--- the team if no team by that name existed yet. That put the decision of
--- which group controls a team in the server's environment rather than the
--- team's own hands, meant changing it needed an env var edit and a restart,
--- and let a typo in a team name silently create a stray team.
---
--- Each team now names, itself, which group grants membership and which
--- grants ownership. Nullable: most teams need neither. No uniqueness
--- constraint on either column — two teams may legitimately watch the same
--- provider group (a broad team and a narrower one both keyed off overlapping
--- groups is a choice for their owners to make, not one the schema should
--- refuse).
---
--- BREAKING CHANGE, deliberately not auto-migrated: TERDUT_OIDC_GROUP_MAPPINGS
--- stops being read as of this version, and the sync no longer creates a team
--- by name. Every team's group binding must be set again through
--- PUT /api/teams/{teamID}/oidc-groups. Until an owner does that, an
--- OIDC-sourced membership in that team is dropped at that user's next SSO
--- sign-in, the same way any other loss of group access is handled. See the
--- README's OIDC section.
-ALTER TABLE teams ADD COLUMN oidc_member_group TEXT;
-ALTER TABLE teams ADD COLUMN oidc_owner_group TEXT;
diff --git a/internal/db/migrations/014_service_accounts.sql b/internal/db/migrations/014_service_accounts.sql
deleted file mode 100644
index bd93644..0000000
--- a/internal/db/migrations/014_service_accounts.sql
+++ /dev/null
@@ -1,43 +0,0 @@
--- Service accounts: a scoped, non-human credential for automation (e.g.
--- terdut-operator) that needs to manage teams, escalation policies, dead
--- man's switches, integrations and OIDC group bindings without impersonating
--- a human user. See SERVICE-ACCOUNTS.md for the design this implements.
---
--- Deliberately not a users row: no password_hash, no is_admin, no
--- user_identities linkage, so a service account can never be pulled into
--- OIDC group sync or password login, and is never mistaken for a human in an
--- audit trail.
---
--- scope is 'instance' (acts with the same reach system administration has
--- over teams: create one, list them, mint a 'team'-scoped account against
--- any of them) or 'team' (acts as that one team's owner, and nothing else).
--- The CHECK ties team_id's presence to scope directly, rather than leaving it
--- to application code to keep the two consistent.
-CREATE TABLE service_accounts (
- id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
- name TEXT NOT NULL UNIQUE,
- scope TEXT NOT NULL CHECK (scope IN ('instance', 'team')),
- team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE,
- created_by BIGINT REFERENCES users(id) ON DELETE SET NULL,
- created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
- CONSTRAINT service_accounts_scope_team_id_chk CHECK (
- (scope = 'team' AND team_id IS NOT NULL) OR
- (scope = 'instance' AND team_id IS NULL)
- )
-);
-
-CREATE INDEX service_accounts_team_id_idx ON service_accounts(team_id);
-
--- One account, many keys: rotation is minting a new one and revoking the
--- old, the same shape api_keys already has, so an account's identity and
--- audit history survive a rotation instead of being recreated by it.
-CREATE TABLE service_account_keys (
- id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
- service_account_id BIGINT NOT NULL REFERENCES service_accounts(id) ON DELETE CASCADE,
- key_hash TEXT NOT NULL UNIQUE,
- name TEXT NOT NULL,
- created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
- last_used_at BIGINT
-);
-
-CREATE INDEX service_account_keys_service_account_id_idx ON service_account_keys(service_account_id);
diff --git a/internal/db/migrations/015_incident_service_account_actors.sql b/internal/db/migrations/015_incident_service_account_actors.sql
deleted file mode 100644
index b225f34..0000000
--- a/internal/db/migrations/015_incident_service_account_actors.sql
+++ /dev/null
@@ -1,39 +0,0 @@
--- Service-account actors on incident mutations (terdut-server#25). A
--- team-scoped service account acknowledging/resolving/snoozing/noting an
--- incident is not a users row, so it cannot be written into
--- acknowledged_by/incident_events.user_id — doing so either violates the
--- users(id) FK (new rows) or, for incident_events.user_id, silently matches
--- zero rows on delete. These columns are the service-account-shaped parallel
--- to the existing human ones: nullable, mutually exclusive with their human
--- counterpart, ON DELETE SET NULL so a deleted service account doesn't take
--- the incident history with it.
-ALTER TABLE incidents
- ADD COLUMN acknowledged_by_service_account_id BIGINT
- REFERENCES service_accounts(id) ON DELETE SET NULL;
-
-ALTER TABLE incident_events
- ADD COLUMN service_account_id BIGINT
- REFERENCES service_accounts(id) ON DELETE SET NULL;
-
--- At most one actor kind per row: both NULL ("the server acted") is valid,
--- exactly one set is valid, both set is a bug this constraint refuses to
--- store rather than silently accepting.
-ALTER TABLE incidents
- ADD CONSTRAINT incidents_ack_actor_xor_chk CHECK (
- acknowledged_by IS NULL OR acknowledged_by_service_account_id IS NULL
- );
-
-ALTER TABLE incident_events
- ADD CONSTRAINT incident_events_actor_xor_chk CHECK (
- user_id IS NULL OR service_account_id IS NULL
- );
-
-CREATE INDEX incidents_acknowledged_by_service_account_id_idx
- ON incidents(acknowledged_by_service_account_id);
-CREATE INDEX incident_events_service_account_id_idx
- ON incident_events(service_account_id);
-
--- assigned_to_service_account_id is deliberately not added here: it would sit
--- unpopulated until handleIncidentAssign itself tracks an actor, which is a
--- separate, pre-existing gap (it records the assignee today, never the
--- actor, for humans either) tracked in its own follow-up issue.
diff --git a/internal/db/migrations/016_rate_limit_counters.sql b/internal/db/migrations/016_rate_limit_counters.sql
deleted file mode 100644
index 5553401..0000000
--- a/internal/db/migrations/016_rate_limit_counters.sql
+++ /dev/null
@@ -1,16 +0,0 @@
--- Backs the rate limiters (failed logins, sign-ups, OIDC/device start) with
--- Postgres instead of an in-memory map, now that the server runs more than
--- one replica in production (v0.37.0): a counter that only ever sees its own
--- pod's traffic quietly let every one of these limits through multiplied by
--- the replica count.
---
--- window_start is the start of the current fixed window for key, in the same
--- "unix seconds" shape every other timestamp in this schema uses. The window
--- resets rather than slides, matching the in-memory limiter it replaces:
--- once a key's window is older than the limiter's window length, the next
--- failure starts a fresh one instead of extending the stale one.
-CREATE TABLE rate_limit_counters (
- key TEXT PRIMARY KEY,
- window_start BIGINT NOT NULL,
- count INT NOT NULL
-);
diff --git a/internal/db/migrations/017_api_key_expiry.sql b/internal/db/migrations/017_api_key_expiry.sql
deleted file mode 100644
index 17f8ab5..0000000
--- a/internal/db/migrations/017_api_key_expiry.sql
+++ /dev/null
@@ -1,7 +0,0 @@
--- Optional expiry on a user's own API keys. NULL (the existing default for
--- every row already in this table) means "never expires" -- the same
--- behavior these keys have always had, so no existing integration breaks.
--- Service account keys are deliberately NOT touched: they are a different
--- table, managed by automation, and already distinguished by their own
--- "tdsa_" prefix.
-ALTER TABLE api_keys ADD COLUMN expires_at BIGINT;
diff --git a/internal/db/migrations/018_incident_event_actor.sql b/internal/db/migrations/018_incident_event_actor.sql
deleted file mode 100644
index cfee52f..0000000
--- a/internal/db/migrations/018_incident_event_actor.sql
+++ /dev/null
@@ -1,21 +0,0 @@
--- Who performed an assignment (terdut-server#35). On an 'assigned' event
--- incident_events.user_id is the assignee, so the actor needs columns of its
--- own. Only populated for 'assigned' events; every other event type keeps
--- using user_id/service_account_id for the actor. Older 'assigned' rows stay
--- NULL (the actor was never recorded). Same shape as migration 015: nullable,
--- mutually exclusive, ON DELETE SET NULL.
---
--- assigned_to_service_account_id is still deliberately not added: making
--- service accounts assignable is a separate change (request body, assignee
--- picker, notifier, filters).
-ALTER TABLE incident_events
- ADD COLUMN actor_user_id BIGINT REFERENCES users(id) ON DELETE SET NULL,
- ADD COLUMN actor_service_account_id BIGINT REFERENCES service_accounts(id) ON DELETE SET NULL;
-
-ALTER TABLE incident_events
- ADD CONSTRAINT incident_events_assign_actor_xor_chk CHECK (
- actor_user_id IS NULL OR actor_service_account_id IS NULL
- );
-
-CREATE INDEX incident_events_actor_user_id_idx ON incident_events(actor_user_id);
-CREATE INDEX incident_events_actor_service_account_id_idx ON incident_events(actor_service_account_id);
diff --git a/internal/models/alert.go b/internal/models/alert.go
index 01047e8..c6e0356 100644
--- a/internal/models/alert.go
+++ b/internal/models/alert.go
@@ -33,7 +33,7 @@ type Alert struct {
// refreshed. The sweeper stale-dates against it (see expireStale), API
// clients render it, and GET /api/alerts is ordered by it. Anything that
// stops the webhook handler from advancing it on a re-send is a breaking
- // change — see "received_at is a liveness heartbeat" in the README and
+ // change — see "received_at is a liveness heartbeat" in docs/api.md and
// TestWebhook_ResendBumpsReceivedAt.
ReceivedAt time.Time `json:"received_at"`
@@ -52,7 +52,7 @@ type Alert struct {
// inferred. Under "expiry" nothing ever reported an end, so EndsAt is only
// an upper bound (see expireStale) and ReceivedAt is the more truthful
// signal. Treat the value set as open — see "resolution_source says how much
- // to trust ends_at" in the README, and TestWebhook_ResolvedSetsSource /
+ // to trust ends_at" in docs/api.md, and TestWebhook_ResolvedSetsSource /
// TestExpiry_StaleFiringAlert.
ResolutionSource *string `json:"resolution_source,omitempty"`
diff --git a/internal/models/team.go b/internal/models/team.go
index 8bedf25..0ed6365 100644
--- a/internal/models/team.go
+++ b/internal/models/team.go
@@ -9,6 +9,10 @@ type Team struct {
Name string `json:"name"`
CreatedAt time.Time `json:"created_at"`
+ // ExternalID identifies a team managed by automation; see handleCreateTeam.
+ // Shown to instance service accounts and admins only.
+ ExternalID *string `json:"external_id,omitempty"`
+
// Role is the caller's own role in this team, populated when a team is
// listed for a particular person. Empty when nobody in particular is
// asking, as in the admin listing.
diff --git a/internal/oidc/grants.go b/internal/oidc/grants.go
index 2a67a65..4de53a1 100644
--- a/internal/oidc/grants.go
+++ b/internal/oidc/grants.go
@@ -102,6 +102,3 @@ func rank(role string) int {
}
return 0
}
-
-// HigherRole reports whether role a outranks role b.
-func HigherRole(a, b string) bool { return rank(a) > rank(b) }
diff --git a/internal/web/static/js/format.js b/internal/web/static/js/format.js
index 9bf994f..98be815 100644
--- a/internal/web/static/js/format.js
+++ b/internal/web/static/js/format.js
@@ -123,7 +123,7 @@ export function initial(name) {
// convention. It comes from Prometheus's externalLabels, so it is on every
// alert; an incident carries it only when it is in Alertmanager's group_by,
// which is also what keeps two clusters' identical alerts from merging into one
-// incident (see the README, "Several clusters, one team").
+// incident (see docs/incidents.md, "Several clusters, one team").
export const ORIGIN_LABEL = 'cluster';
export function originOf(labels) {