diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml index 693ad74..09cd03d 100644 --- a/.gitea/workflows/ci.yaml +++ b/.gitea/workflows/ci.yaml @@ -35,7 +35,7 @@ jobs: # Runs inside the toolchain image rather than installing Go per job. Note this puts # the job on the dind bridge, which cannot reach github.com or get.helm.sh -- # proxy.golang.org and git.ryuvia.com are reachable, which is all this job needs. - image: golang:1.26.6-bookworm + image: golang:1.26.9-bookworm # act_runner destroys a job's own volumes when it finishes, so without these every # run re-downloads the whole module graph. The names must appear in the runner's # container.valid_volumes allowlist (charts/act-runner in the k8s repo); unlisted @@ -46,8 +46,8 @@ jobs: - go-build-cache:/root/.cache/go-build - gobin-cache:/go/bin - # The suite needs a real Postgres -- there is no in-memory Postgres the way there was - # an in-memory SQLite, so each test gets its own schema on a shared server instead. + # The suite needs a real Postgres -- there is no in-memory Postgres, + # so each test gets its own schema on a shared server instead. # The job and the service share the dind bridge, so the service is reachable by its # name rather than on localhost. services: @@ -101,7 +101,7 @@ jobs: security: runs-on: ubuntu-latest container: - image: golang:1.26.6-bookworm + image: golang:1.26.9-bookworm volumes: - go-mod-cache:/go/pkg/mod - go-build-cache:/root/.cache/go-build diff --git a/.gitea/workflows/release.yaml b/.gitea/workflows/release.yaml index 50bd5ff..c4ded87 100644 --- a/.gitea/workflows/release.yaml +++ b/.gitea/workflows/release.yaml @@ -30,7 +30,7 @@ jobs: test: runs-on: ubuntu-latest container: - image: golang:1.26.6-bookworm + image: golang:1.26.9-bookworm volumes: - go-mod-cache:/go/pkg/mod - go-build-cache:/root/.cache/go-build @@ -71,7 +71,7 @@ jobs: needs: test runs-on: ubuntu-latest container: - image: golang:1.26.6-bookworm + image: golang:1.26.9-bookworm volumes: - go-mod-cache:/go/pkg/mod - go-build-cache:/root/.cache/go-build diff --git a/.gitignore b/.gitignore index aa08c62..31709d7 100644 --- a/.gitignore +++ b/.gitignore @@ -7,8 +7,6 @@ # one (which has an unreachable entry) cannot abort a release /.helm-repos.yaml -# SQLite database files -*.db *.db-shm *.db-wal diff --git a/CLAUDE.md b/CLAUDE.md index 278c78c..ef79927 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -12,7 +12,7 @@ Preconditions and the plan, without side effects: ``` Config is `.release.conf` here plus `make release-vars`. The process itself lives in -`~/.claude/skills/release/`; why it is shaped this way is in README.md §Releasing. +`~/.claude/skills/release/`; why it is shaped this way is in docs/development.md (Releasing). Three things about this repo specifically: diff --git a/Makefile b/Makefile index 3f9b660..d284c30 100644 --- a/Makefile +++ b/Makefile @@ -28,7 +28,7 @@ help: ## Show this help # -race below. Both need a Postgres to test against; see test-db. # The suite needs a Postgres, because the server does: there is no in-memory -# Postgres the way there was an in-memory SQLite. TERDUT_TEST_DSN says where, and +# Postgres. TERDUT_TEST_DSN says where, and # the tests fail rather than skip without it — a suite that quietly tests nothing # is worse than one that does not run. `make test-db` starts a local one; # ci.yaml runs the same thing as a service container. diff --git a/README.md b/README.md index da37499..fc57704 100644 --- a/README.md +++ b/README.md @@ -1,31 +1,72 @@ # Terminal Duty (terdut-server) -Incident management server for teams using Prometheus Alertmanager. +Incident management for teams that already run Prometheus Alertmanager. Point Alertmanager at it, +and alerts become incidents that get assigned to whoever is on call, paged, escalated when nobody +answers, and tracked to resolution. One binary, one Postgres. -- Receives Alertmanager webhooks directly — no adapter needed -- Turns alerts into **incidents**, correlated by Alertmanager's own `groupKey` -- Incident workflow: acknowledge, assign, snooze, note, resolve, with a full timeline -- On-call schedule management, with new incidents auto-assigned to whoever is on call -- Alert and incident statistics, including MTTA and MTTR -- Web UI for phones and desktops, served by the same binary -- REST API with per-user API key authentication -- Single binary plus a Postgres — straightforward to self-host +![The incident queue with an incident open](./docs/images/queue-light.png) ---- +## Highlights + +- **Alertmanager-native.** Receives Alertmanager webhooks directly, with no adapter, and groups + alerts into incidents by Alertmanager's own `groupKey`. An incident opens on a new occurrence, + not on every re-send. [Details](./docs/incidents.md) +- **A real incident workflow.** Acknowledge, assign, snooze, add notes, resolve and archive, with a + full timeline of who did what and when. Several clusters can feed one team without + cross-talk. [Details](./docs/incidents.md) +- **On-call rota.** Each team keeps its own rota, and new incidents go to whoever is on call. + [Web UI](./docs/web-ui.md) +- **Pages that escalate.** Push notifications through [ntfy](https://ntfy.sh), with an + Acknowledge button right in the notification, and escalation ladders that move on to the next + level when nobody answers. [Notifications](./docs/notifications.md) · + [Escalation](./docs/escalation.md) +- **Notices when the alerts stop.** Dead man's switches turn the absence of a heartbeat such as + `Watchdog` into an incident. [Details](./docs/dead-mans-switch.md) +- **Teams.** Every team owns its queue, rota, escalation, alert sources and switches; people see + only the teams they belong to. +- **Stats.** Incident counts, mean time to acknowledge and resolve, and alert frequency by name, + hour and day. [Web UI](./docs/web-ui.md) +- **Single sign-on.** OpenID Connect with group-to-team and administrator mapping, and a device + flow so a terminal client can sign in through the browser. [Details](./docs/single-sign-on.md) +- **Built for the phone first.** The web UI is served by the same binary, follows the system's + dark mode, and can be added to the home screen. +- **An API for everything.** A REST API with API keys and service accounts for automation. + [Reference](./docs/api.md) +- **Easy to run.** A scratch container image, a Helm chart, and + [terdut-operator](https://git.ryuvia.com/niklas/terdut-operator) if you want teams and + escalation as Kubernetes objects. [Deployment](./docs/deployment.md) + +## Screenshots + +| | | +|---|---| +| ![An incident with a note on its timeline](./docs/images/incident-note.png) | ![The queue in dark mode](./docs/images/queue-dark.png) | +| **Work an incident**: acknowledge, assign, snooze, note, resolve | **Dark mode**, following the system | +| ![The on-call page](./docs/images/oncall.png) | ![An escalation ladder](./docs/images/team-escalation.png) | +| **On call**: who holds the pager now, and the week ahead | **Escalation**: who is paged next, and when | +| ![Statistics](./docs/images/stats.png) | ![The alert feed](./docs/images/alerts.png) | +| **Stats**: MTTA, MTTR and what fires most | **Alerts**: the raw feed behind the incidents | + +On a phone the queue and the incident page are the same interface, with a sticky action bar: + +

+ The queue on a phone + An incident on a phone +

## Quick start -**Prerequisites:** Go 1.21+ +You need Go 1.25+ and a Postgres 14+ the server can reach. ```bash git clone https://git.ryuvia.com/niklas/terdut-server cd terdut-server +export TERDUT_DB_DSN='postgres://terdut:secret@localhost:5432/terdut?sslmode=disable' go run ./cmd/terdut ``` -The server starts on `:8080` with a `terdut.db` file in the working directory. - -### Create the first user +The server creates its schema on startup and listens on `:8080`. Create the first user, an +administrator, while no user exists yet: ```bash curl -X POST http://localhost:8080/api/bootstrap \ @@ -33,1334 +74,36 @@ curl -X POST http://localhost:8080/api/bootstrap \ -d '{"username": "admin", "email": "admin@example.com", "password": ""}' ``` -Save the `api_key.key` value from the response — it is shown **once only**. The -`password` is optional and is what signs you in to the [web UI](#web-ui). - -Use it as a bearer token for all subsequent requests: +Sign in at with that username and password. The response also carries an +API key, shown **once**, for scripts: ```bash -export KEY= curl -H "Authorization: Bearer $KEY" http://localhost:8080/api/users ``` -### Web UI +Then create a team, add an alert source to get a webhook URL, and point Alertmanager's webhook +receiver at it: see [Alertmanager configuration](./docs/alertmanager.md). -The server serves a web UI at `/`: the incident queue, each incident's alerts -and timeline with every action (acknowledge, assign, snooze, note, resolve, -archive), who is on call, the alert feed, and an *Account* tab for your own -password and the ntfy topic your pages go to. It is built for a phone first. On a phone -it navigates through a hamburger menu and has a sticky action bar, it follows the -system's dark mode, and it can be added to the home screen. From 900px wide it switches -to a sidebar with the queue and the incident side by side. The Stats page shows -incident counts, MTTA and MTTR, and alert frequency by name, hour and day over a -chosen range. +## Documentation -You sign in with a username and password. Users have no password until one is -set, and a user without one can only use API keys: +The [documentation index](./docs/README.md) lists everything. The main pages: -```bash -# an admin sets someone's first password with their API key -curl -X PUT http://localhost:8080/api/users/2/password \ - -H "Authorization: Bearer $KEY" -H "Content-Type: application/json" \ - -d '{"password": ""}' -``` - -After that, users change it themselves under *Account*. Changing your own -password requires the current one. - -How a browser stays signed in: - -- A successful login sets an `HttpOnly`, `SameSite=Lax` session cookie. It lasts - 30 days and slides forward while it is used, so an on-call phone stays signed - in. -- The cookie is marked `Secure` when `TERDUT_PUBLIC_URL` starts with `https://`, - so set it to the HTTPS address. TLS terminates at the gateway and the server - itself only ever sees plain HTTP. -- Requests authenticated by the cookie are checked for cross-origin use (Go's - `http.CrossOriginProtection`). That is the CSRF guard. Bearer-key clients are - not affected. -- Setting a password signs that user out everywhere else. -- Ten failed logins for one username within 15 minutes lock that username for - the rest of the window. - -With `TERDUT_PUBLIC_URL` set, tapping a push notification opens the incident in -the web UI (`/incidents/{id}`). - -A **Team** tab holds everything a team owns, in five sub-sections with a URL -each and a strip across the top to move between them: the on-call rota -(`/team/rota`), the membership (`/team/members`), the escalation ladder -(`/team/escalation`), the alert sources with their keys (`/team/sources`) and -the dead man's switches (`/team/deadman`). `/team` itself is an overview — who -is on call today, how many members and owners, how many ladder levels, how many -keys and how many switches — so a page fetches only what it shows. An owner -edits it; a member sees the same pages read-only, because the server refuses -their writes anyway. Somebody in more than one team picks between them above -the strip, since the choice changes the subject of all five. - -The rota is a month at a time, one coloured initial per day with a legend -underneath, and it says how many days are left uncovered — the question a rota -is read for is who holds which stretch, and a run of one colour answers it -where a list of dates does not. An owner taps a day to hand it to somebody or -empty it, and fills a whole shift from the range form folded in below. - -The **Admin** tab appears only for a system administrator, and holds what -belongs to the whole server rather than to one team. It has three sub-sections, -each with a URL of its own and a strip across the top to move between them: -every team (`/admin/teams`), every user (`/admin/users`), and the settings that -used to be environment variables (`/admin/settings`). `/admin` itself is an -overview — how many of each, and what each section is for. Adding somebody is -minting them an invite link into a team, rather than creating a bare account: -the person who accepts it picks their own password, so one never passes through -an administrator, and the link carries the team, so they land somewhere with a -queue in it. That happens on the team's own page, since an invite is a fact -about a team; the user list points there rather than asking which team beside a -form. - -A name in the team list opens **that team's page**, at `/admin/teams/{id}`: when it -was created, how many are in it and how much is open, a field to rename it, the -members with their roles, the invites into it, and deletion. The member list is the -one thing there that needed a new endpoint — `GET /api/teams/{id}/members` is -member-only and answers `404` to an administrator who is not in the team, which is -the rule and not an oversight, so the page reads `GET /api/admin/teams/{id}` instead. -An administrator still sees none of that team's incidents, alerts or rota. - -A name in the user list opens **that person's page**, at `/admin/users/{id}`: their -email and when they joined, where their notifications go, whether they are an -administrator, whether the account is disabled, the teams they are in with their -role in each, a password field for a first or forgotten one, and deletion. It is -the one place membership is edited from the person's side — the Team tab answers -"who is in this team", and answering "which teams is this person in" there means -visiting each team in turn. - -### Single sign-on (OIDC) - -terdut can sign people in through any OpenID Connect provider; the examples use -[Authentik](https://goauthentik.io/). Groups at the provider decide who may sign -in, which teams they belong to and whether they administer the install, much as -Grafana's OAuth role and org mapping does. Password login keeps working alongside -it unless you turn it off. - -**At the provider**, create an OAuth2/OpenID provider and an application for it: -a *confidential* client, redirect URI `/api/oidc/callback`, and -the `openid`, `profile` and `email` scopes. The issuer is the application's, e.g. -`https://auth.example.com/application/o/terdut/`. Then set: - -```sh -TERDUT_PUBLIC_URL=https://terdut.example.com -TERDUT_OIDC_ISSUER=https://auth.example.com/application/o/terdut/ -TERDUT_OIDC_CLIENT_ID=terdut -TERDUT_OIDC_CLIENT_SECRET=... -TERDUT_OIDC_ALLOWED_GROUPS=terdut-users,terdut-admins -TERDUT_OIDC_ADMIN_GROUP=terdut-admins -``` - -Which team a group grants is not server-wide config: each team names its own -group(s), set by that team's own owner (or an administrator) from its Members -tab, or `PUT /api/teams/{teamID}/oidc-groups {"member_group":"sre","owner_group":"sre-leads"}`. -A team must already exist before a group can grant access to it — the sync -never creates one. - -The web UI's sign-in page shows a "Sign in with " button (a plain link to -`/api/oidc/login`) above the password form, or instead of it when -`TERDUT_PASSWORD_LOGIN=false`; it asks `GET /api/auth/config` what the server offers -(`password_login`, `oidc.enabled`, `oidc.name`). A refused sign-in comes back to that -page with the reason spelled out. Access the groups grant is badged **SSO** on the -Team, Admin and per-user pages, with its edit and remove controls disabled, and the -Account page does not offer to set a password nobody could use. - -**What a sign-in does** - -1. *Who.* The provider's `(issuer, subject)` is the identity. The first time, a - user is found by email — only when the provider marks it verified, or - `TERDUT_OIDC_TRUST_EMAIL` is set — or created with no password. A username taken - by somebody else gets a numeric suffix (`alice-2`). Username and email follow the - provider at each sign-in. Authentik reports `email_verified` as false unless - configured otherwise, so linking existing users usually needs - `TERDUT_OIDC_TRUST_EMAIL=true`. -2. *Whether.* With `TERDUT_OIDC_ALLOWED_GROUPS` set, somebody in none of them is - refused and nothing is created. -3. *What.* The administrator flag follows `TERDUT_OIDC_ADMIN_GROUP`. Team roles - follow each team's own `oidc_member_group`/`oidc_owner_group`; where both of a - team's groups match, the owner group wins. - -**Managed access.** What the sync grants is marked as managed by single sign-on, -and only that is ever changed by it. It is added at sign-in, and removed at the -next sign-in after the group is gone, even if that leaves a team without an owner -(an administrator can always repair a team) — the provider is the source of truth -for what it grants, so the last-owner and last-administrator guards do not apply. -Memberships and administrators added by hand are left alone; the exception is a -hand-added member whose team's own group grants a *higher* role, who is raised and -from then on managed. Editing managed access by hand (`POST` or `DELETE` on a -team's members, revoking an SSO-granted administrator) is refused with `409`, since -the next sign-in would undo it. - -> **Upgrading past migration 013: reconfigure every team's groups.** -> `TERDUT_OIDC_GROUP_MAPPINGS` is gone, and the sync no longer creates a team by -> name. Group-to-team-role mapping is now each team's own setting — an owner sets -> it from the Members tab, or `PUT /api/teams/{teamID}/oidc-groups`. Until a team's -> owner does that, an OIDC-sourced membership in it is dropped at that user's next -> SSO sign-in, the same as any other loss of group access. Set every team's groups -> before affected users next sign in, to avoid a visible gap in access. - -**How fast changes arrive.** Groups are read only at sign-in. A session made by an -SSO sign-in has a hard ceiling (`TERDUT_OIDC_SESSION_MAX_AGE`, default 12h) that -sliding never extends, so a change at the provider reaches terdut within that time. -Password sessions are unaffected. - -> **API keys are not revoked when somebody is removed at the provider.** terdut -> holds no refresh token and never asks the provider again, so a person removed -> from every allowed group loses their sessions within `TERDUT_OIDC_SESSION_MAX_AGE` -> and cannot sign in again, but keeps any API key they made (the TUI and scripts use -> them) until an administrator disables the user in terdut. - -**Signing in from a terminal.** A client with no browser of its own, such as the -TUI over SSH, signs in with a device code, run by terdut itself so the terminal -never talks to the provider: - -1. The terminal calls `POST /api/oidc/device` and shows the person a link - (`/device?code=XXXX-XXXX`) and the code. -2. On any device the person opens the link, signs in (by the provider or by - password, whatever the login page offers), sees the code and the account, and - presses **Approve**. Only a browser session can approve; an API key cannot. -3. The terminal polls `POST /api/oidc/device/token` every 5 seconds and is given the - ordinary `terdut_session` cookie once. A person who signs in through the provider - gets the same `TERDUT_OIDC_SESSION_MAX_AGE` ceiling on the terminal's session as - on their browser's. - -A login expires after 10 minutes. `GET /api/auth/config` reports `device_login`. - -**If the provider is down**, terdut still starts (discovery is fetched on first -use) and password login is the way in. With `TERDUT_PASSWORD_LOGIN=false` that way -is closed: set it back to `true`. The first administrator comes from the bootstrap -endpoint, and stays a manual administrator that no group can revoke; on an SSO-only -install set `bootstrap.enabled: false` in the chart if you don't want that account, -or keep it and never give it a password. - -### Docker - -```bash -docker build -t terdut-server . -docker run -p 8080:8080 \ - -e TERDUT_DB_DSN='postgres://terdut:secret@host.docker.internal:5432/terdut?sslmode=disable' \ - terdut-server -``` - -The server creates its own schema on startup and needs a reachable Postgres; it stores nothing on -disk, so there is no volume to mount. - -### Kubernetes - -A Helm chart is published from this repository as an OCI artifact, versioned in lockstep -with the app — chart `x.y.z` is always app `vx.y.z`: - -```bash -helm upgrade --install terdut-server oci://git.ryuvia.com/niklas/terdut-server \ - --version 0.9.2 \ - --namespace terdut-server --create-namespace \ - --set networking.hostname=terdut.example.com -``` - -The chart expects a [Gateway API](https://gateway-api.sigs.k8s.io/) Gateway named `envoy-main` in -the `envoy-gateway-system` namespace to already exist — it renders an `HTTPRoute` against it rather -than an `Ingress`. TLS is terminated at the gateway, so the server itself never sees a certificate. - -| Value | Default | Description | -|---|---|---| -| `networking.hostname` | `terdut.example.com` | Hostname the `HTTPRoute` serves | -| `networking.listener` | `""` | Gateway listener (`sectionName`) to bind to. Empty attaches to every matching listener, **including plaintext HTTP** — set it to the HTTPS listener's name to serve TLS only | -| `networking.servicePort` | `8080` | Port the route forwards to; keep in sync with `service.port` | -| `bootstrap.enabled` | `true` | Runs a post-install hook that creates the first user and stores its API key in the `-admin-key` Secret. Already-bootstrapped servers are left alone | -| `database.dsn` | `""` | **Required.** Postgres DSN, with no password in it. The chart provisions no database | -| `database.passwordSecret.name` | `""` | Secret supplying `PGPASSWORD`. With the Zalando postgres operator, the Secret it generates for the role | -| `database.passwordSecret.key` | `password` | Key within that Secret | - -The API key travels in an `Authorization: Bearer` header, so set `networking.listener` whenever the -hostname is reachable outside a trusted network. - -#### The database - -The chart provisions no database: it takes a DSN and expects a Postgres that already exists. In this -cluster the wrapper chart declares an `acid.zalan.do/v1 postgresql` CR; anywhere else, any reachable -Postgres 14+ will do. - -The DSN carries no password. pgx falls back to libpq's environment variables for whatever the DSN -leaves out, so the password arrives as `PGPASSWORD` from a Secret and never appears in values, in -the rendered manifest or in `kubectl describe pod`. With the postgres operator that Secret is the -one it generates for the role, so a rebuild mints a new password with nothing to keep in sync — -the same wiring miniflux uses. - -The server migrates its own schema on startup, so a new database only has to exist and be writable. - -#### Backups - -Postgres is backed up where it runs, not from here. The database pod carries a -[k8up](https://k8up.io/) `k8up.io/backupcommand` annotation that streams a `pg_dump`, the same way -gitea and immich do in this cluster. - -This used to be the app's problem: the SQLite database lived on a PVC beside the server, the image -is `FROM scratch` with no interpreter to dump it, and WAL mode makes a file-level copy of the volume -non-crash-consistent — so the chart shipped an idle `python:*-alpine` sidecar purely to give k8up -somewhere to exec. The sidecar, the PVC and the `backupSidecar` values are all gone. - ---- - -## Configuration - -Two kinds of setting, split by who changes them and how often. - -**Where the server is plugged in** stays in the environment: the listen address, -the database DSN, the ntfy URL and token, the public URL. They are needed before -the database is open, and two of them are credentials. - -**How the server behaves** lives in the database and is edited by an -administrator in the web UI or through `PUT /api/admin/settings`, taking effect -on the next sweep rather than at the next restart. The variables below marked -**seed** are the value each of those starts from: written once, on first start, -and never overwritten afterwards — a redeploy cannot put a chart's default back -over an administrator's edit. - -| Variable | Default | Description | -|---|---|---| -| `TERDUT_ADDR` | `:8080` | TCP address to listen on | -| `TERDUT_DB_DSN` | — | **Required.** Postgres connection string, e.g. `postgres://terdut:secret@localhost:5432/terdut?sslmode=require` | -| `TERDUT_ARCHIVE_AFTER` | `168h` (7d) | **seed.** How long a resolved alert or incident stays in the default list before being auto-archived | -| `TERDUT_STALE_AFTER` | `6h` | **seed.** How long a firing alert may go without a refreshing webhook before it is treated as resolved — **must exceed your Alertmanager `repeat_interval`** | -| `TERDUT_DEADMAN_MATCHERS` | `alertname=Watchdog` | The **default** matchers a team starts with — switches are per team now, and this seeds teams that have no configuration of their own. `;` separates matchers, `,` the label conditions within one, `=` is exact equality. Every matcher must name an `alertname` | -| `TERDUT_DEADMAN_TIMEOUT` | `15m` | How long a heartbeat may go unheard before its switch is declared dead — **must be shorter than the `repeat_interval` of the route carrying it**. `0` disables dead man's switch handling | -| `TERDUT_DEADMAN_SEVERITY` | `critical` | Severity a dead man's switch incident opens at | -| `TERDUT_NTFY_URL` | — | ntfy server to publish push notifications to. Empty disables notifications entirely | -| `TERDUT_NTFY_TOKEN` | — | Bearer token for an access-controlled ntfy | -| `TERDUT_NTFY_FALLBACK_TOPIC` | — | Topic used when nobody is on call | -| `TERDUT_PUBLIC_URL` | — | Base URL a phone uses to reach this server: the notification's link into the web UI, its Acknowledge button, and whether the session cookie is `Secure` | -| `TERDUT_NOTIFY_REPEAT` | `15m` | **seed.** How long an incident may sit unacknowledged before it is paged again. `0` notifies once and never repeats | -| `TERDUT_PASSWORD_LOGIN` | `true` | `false` refuses password login and password sign-up (`403`), leaving single sign-on the only way in. Refused at startup unless SSO is configured | -| `TERDUT_OPERATOR_MODE` | `false` | Declares this install gitops-managed: a session's or a user's own API key's writes to teams, escalation policies, dead man's switches and integrations are refused (`403 reason:"operator_managed"`); a [service account](#service-accounts)'s are not. Team membership and the schedule stay editable regardless | -| `TERDUT_OIDC_ISSUER` | — | Turns single sign-on on. The provider's issuer URL; discovery is read from `/.well-known/openid-configuration`. See [Single sign-on](#single-sign-on-oidc) | -| `TERDUT_OIDC_CLIENT_ID` / `TERDUT_OIDC_CLIENT_SECRET` | — | **Required with an issuer.** The confidential client registered at the provider. Keep the secret in a Secret, not in values | -| `TERDUT_OIDC_NAME` | `SSO` | What the sign-in button calls the provider | -| `TERDUT_OIDC_SCOPES` | `openid profile email` | Scopes requested, comma or space separated. Authentik puts `groups` behind `profile` | -| `TERDUT_OIDC_USERNAME_CLAIM` / `_EMAIL_CLAIM` / `_GROUPS_CLAIM` | `preferred_username` / `email` / `groups` | ID token claims read for the username, email and groups | -| `TERDUT_OIDC_TRUST_EMAIL` | `false` | Link a first sign-in to an existing local user by email even if the provider does not mark the address verified | -| `TERDUT_OIDC_ALLOWED_GROUPS` | — | Comma-separated. Only people in one of these may sign in. Empty admits everybody the provider authenticates | -| `TERDUT_OIDC_ADMIN_GROUP` | — | Members are system administrators | -| `TERDUT_OIDC_SESSION_MAX_AGE` | `12h` | Hard ceiling on a session made by an SSO sign-in | - -Durations use Go syntax (`30m`, `12h`, `168h`). An unparseable value falls back to the default. - -Note that `TERDUT_STALE_AFTER` and `TERDUT_DEADMAN_TIMEOUT` point in opposite directions. Staleness -is a generous grace period around a `repeat_interval` you do not control; a dead man's switch is a -deadline you set deliberately, and the heartbeat's route is configured to beat faster than it. - -In the Helm chart the two sweeper durations are set via `sweeper.staleAfter` and `sweeper.archiveAfter`, dead man's switches via the `deadman.*` values, notifications via the `notify.*` values, single sign-on via `oidc.*` and `passwordLogin`, and operator mode via `operatorMode`. - ---- - -## Alertmanager configuration - -Alerts arrive on a team's **integration key**, which says both that the sender -may post and which team the alerts belong to. Mint one as an owner of the team: - -```bash -curl -X POST https://terdut.example.com/api/teams/1/integrations \ - -H "Authorization: Bearer $TERDUT_API_KEY" \ - -H 'Content-Type: application/json' \ - -d '{"name":"prod alertmanager"}' -``` - -The response carries the key and the full URL **once**; only a SHA-256 hash is -stored. Put it in your `alertmanager.yml`: - -```yaml -receivers: - - name: terdut - webhook_configs: - - url: http://terdut-server:8080/api/integrations//alertmanager - send_resolved: true - -route: - receiver: terdut -``` - -The whole URL is a credential, so treat it like one. Alertmanager 0.26 and -later can read it from a file with `url_file:` instead, which keeps it out of -your configuration repository: - -```yaml - - url_file: /etc/alertmanager/secrets/terdut-webhook-url/url - send_resolved: true -``` - -The webhook endpoint requires no authentication. - -If you use the [dead man's switch](#dead-mans-switch) — and the default configuration does — give -the heartbeat a route of its own, because the deadline is only as tight as the interval feeding it: - -```yaml -route: - receiver: terdut - repeat_interval: 4h - routes: - - matchers: [ 'alertname = "Watchdog"' ] - receiver: terdut - group_wait: 0s - group_interval: 1m - repeat_interval: 1m -``` - -That delivers a heartbeat every **2 minutes**, not every minute. Alertmanager only reconsiders a -group every `group_interval`, and at exactly one elapsed interval `repeat_interval` has not *quite* -passed, so the send slips to the next tick — equal values give 2×. Two minutes against the 15 minute -default is seven heartbeats per window, which is the point; use `group_interval: 30s` if you want -the numbers to mean what they say. - -kube-prometheus-stack users get the `Watchdog` alert (`expr: vector(1)`) for free; it just needs -routing to terdut rather than to `null`. - ---- - -## Alerts and incidents - -There are two objects, and the difference between them is the whole design. - -**An alert is Alertmanager's record.** It has two states, `firing` and -`resolved`, one row per fingerprint, and no human ever writes to it. The API -exposes alerts read-only. - -**An incident is the work item.** It goes `triggered → acknowledged → resolved`, -carries an assignee, a snooze, notes and a timeline, and is the only thing people -act on. Many alerts belong to one incident. - -### Correlation uses Alertmanager's `groupKey` - -Alertmanager has already grouped alerts according to the `group_by` routing tree -you configured, and it sends the resulting `groupKey` and `groupLabels` on every -webhook. Incidents adopt that answer rather than re-grouping alerts a second -time — if you want different correlation, change `group_by` in -`alertmanager.yml` and terdut follows. - -At most one incident is open per `groupKey` at a time. Alerts firing in a group -that already has an open incident join it. The incident's `severity` is a -high-water mark — the highest `severity` label any of its alerts has carried — so -an incident that hit `critical` still reads as critical after the critical alert -clears. - -### Several clusters, one team - -A team with one Alertmanager per Kubernetes cluster, each posting to its own -source, needs two settings or the clusters run together. - -1. Give every alert a `cluster` label at the source. In Prometheus that is - `externalLabels: {cluster: prod-eu}` (kube-prometheus-stack: - `prometheus.prometheusSpec.externalLabels`). -2. Add `cluster` to `group_by` in `alertmanager.yml`. - -The second one is the one that matters. Incidents are matched on the team and -Alertmanager's `groupKey`, and the `groupKey` does not include external labels: -without `cluster` in `group_by`, the same alert in two clusters has the same -key and joins one incident. With it, each cluster gets its own, `cluster` is in -the incident's `group_labels`, and the web UI shows it as a coloured chip on the -queue, the incident and the alert list, instead of leaving it in the title. -An alert that is not grouped by `cluster` still shows the chip on the alert -list, which reads the label from the alert itself. - -The queue has a cluster dropdown once there are two or more values to choose -between. It filters on the incident's `cluster` group label -(`GET /api/incidents?cluster=...`), so it only sees incidents grouped by it. - -### An incident opens only on a new occurrence - -An incident opens when an alert **transitions into firing**: a fingerprint that -was never seen, an alert with a newer `startsAt`, or a resolved alert that -started again. The unchanged firing notifications Alertmanager re-sends every -`repeat_interval` are none of those, and open nothing. - -This is what makes closing an incident by hand mean something. Without the rule, -`POST /api/incidents/{id}/resolve` would be undone by the next re-send of an -alert that never stopped firing. - -### Leaving the open state - -- **Automatically**, once every alert under the incident has stopped firing — - whether by a resolved webhook or by the sweeper's - [stale-alert expiry](#stale-alert-expiry). The incident gets - `"resolution_source": "alerts"`. -- **By hand**, via `POST /api/incidents/{id}/resolve` - (`"resolution_source": "manual"`). This is **terminal**: a later occurrence in - that group opens a *new* incident rather than reopening this one. If the alert - underneath never stops firing, the incident stays closed — that is what - resolving by hand asserts. -- **On recovery**, for a [dead man's switch](#dead-mans-switch) incident whose - heartbeat started arriving again (`"resolution_source": "recovered"`). These - incidents have no member alerts, so the automatic cascade above cannot reach - them. - -To quieten an incident you expect to come back, snooze it instead -(`POST /api/incidents/{id}/snooze`). A snooze hides the incident from the default -list without closing it, and expires by simply falling into the past. - -### On-call assignment - -A new incident is assigned to whoever holds today's schedule entry at the moment -it opens (`GET /api/schedule/current`). If nobody is scheduled it opens -unassigned. Reassign with `POST /api/incidents/{id}/assign`. - -One person holds a given day, so `POST /api/schedule` refuses a date somebody -already has: taking a shift off the person expecting to be paged for it should -not be something a plain call does by accident. Pass `"replace": true` to take -them anyway. Either way the whole request is one transaction — a week where some -days are free and some are taken moves as a unit, and a failure leaves the rota -exactly as it was rather than with a hole in it. - -### Push notifications - -With `TERDUT_NTFY_URL` set, an incident that opens is pushed to the on-call -person's phone through [ntfy](https://ntfy.sh). Everybody sets their own topic -under *Account* in the web UI, where a **Send a test push** button proves it -before an incident has to; `PUT /api/users/{id}/notify` is the same thing over -the API, and an administrator may set somebody else's. A user with no topic -falls back to `TERDUT_NTFY_FALLBACK_TOPIC`, as does an incident that opens with -nobody on call. If neither yields a topic, nothing is queued. - -The **server** is the install's one ntfy, from `TERDUT_NTFY_URL`, and is not -something a user picks. Only the topic is per-person. - -A topic is a shared secret with the ntfy server: anyone who knows it can both -read the pages and publish to it, so an unguessable one is worth the trouble. -That is also why the topic never appears in an incident's timeline, which every -API key can read. - -Three things get pushed: - -- **triggered** — an incident opened. Priority follows severity (`critical` maps - to ntfy's max priority, the one that overrides the phone's quiet settings). -- **reminder** — the incident is still `triggered` after `TERDUT_NOTIFY_REPEAT`. - Repeats until somebody acts. Acknowledging, snoozing, resolving or archiving - all stop it — snooze is the mute button. -- **resolved** — every alert under the incident stopped firing. Only sent to - whoever was paged in the first place, and only for the automatic cascade: - resolving by hand pushes nothing, since the person who did it already knows. - -Notifications carry an **Acknowledge** button that acknowledges the incident -without opening anything. It POSTs to `/api/notify/ack/{token}`, an -unauthenticated route authorised by the 256-bit token in its path — minted fresh -per notification, scoped to one incident and one action, and valid for 24 hours. -A real API key is never put in a notification, because the message is stored on -the ntfy server and cached on the device. - -The token is **not** consumed by use. Acknowledging is idempotent, so a token -stays valid for its full 24 hours and a second tap is a no-op that reports the -incident's current state rather than an error — which is what you want when a -tap is retried on a flaky mobile connection. What bounds it is scope, not a use -count: one incident, one action, one day. Expired tokens are purged by the -sweeper. - -Two consequences worth planning for: - -- `/api/notify/ack/{token}` **must stay publicly reachable**, or the button will - not work when the responder is off your network. -- Notifications sent to the fallback topic carry **no** Acknowledge button. The - topic is shared, and a button on it would let any subscriber acknowledge as - somebody else. - -Delivery is a queue, not an inline call: the webhook writes a row and a -background notifier sends it within 30 seconds, retrying with exponential -backoff up to 8 attempts. Nothing about ingestion blocks on ntfy being reachable. - -Every delivery is recorded on the incident's timeline: a `notified` event once -ntfy accepts the publish, and a `notify_failed` event when a notification -exhausts its retries. Written from the result rather than at enqueue, so the -timeline says what actually happened — and a page that never landed is visible -instead of looking the same as one that did. - -### Escalation - -Without a ladder, an unacknowledged incident re-pages the same topic every -`notify_repeat` forever. That is a louder version of the same silence: if the -person on call is asleep, out of signal, or has left, nothing else happens. - -A team can configure an ordered ladder instead. Each level has a timeout and a -set of targets, and a target is either a named person or **whoever the team's -rota says is on call today** — the target that keeps working when the rota -changes and nobody remembers to edit the policy. - -``` -level 1 5m oncall the rota gets first refusal -level 2 5m user:bob then a named second - then repeat_count more rounds - then the team's fallback topic, once -``` - -When a level's timeout passes with the incident still `triggered`, the next -level is paged. Off the end of the ladder the whole thing runs again -`repeat_count` times, and after that the team's `fallback_topic` is paged once -as the end of the line. The incident stays open throughout: running out of -people to wake is not the same as somebody answering. - -**Acknowledging or resolving stops it**, which is the point — continuing to wake -people after somebody has said "I have this" is how a tool teaches people to -mute it. **Snoozing pauses it**: a deliberate "not now" holds the ladder where -it is, and it resumes when the snooze runs out. - -Every step is on the incident's timeline with the level and the names it woke, -so somebody reading it afterwards can tell why their phone rang at 04:00. A -level whose targets are all unreachable — no ntfy topic, a disabled account, an -empty rota — is recorded as `nobody reachable` and the ladder moves on rather -than stalling on a rung that cannot ring. - -**Reminders and escalation never both run.** A team with a ladder gets -escalation; a team without keeps the reminder behaviour exactly as it was. Two -pages for one silence is the surest way to get a tool muted. - -The ladder's `fallback_topic` is per team, unlike `TERDUT_NTFY_FALLBACK_TOPIC`, -which is the install-wide topic used when an incident opens with nobody on call. -They answer different questions: one is "nobody was scheduled", the other is -"everybody scheduled has been tried". - -### Stale alert expiry - -A resolved webhook is the only signal that an alert has stopped firing, so a -notification that is dropped, silenced, or lost to a restart would otherwise pin -that alert as firing forever. A background sweeper resolves firing alerts that -Alertmanager has stopped refreshing, using either signal: - -- the `endsAt` watermark on the last notification has passed, or -- no webhook has refreshed the alert within `TERDUT_STALE_AFTER`. - -Alertmanager re-sends firing notifications every `repeat_interval`, which is what -keeps a live alert fresh — so `TERDUT_STALE_AFTER` must be comfortably larger -than your `repeat_interval` (default 4h), or live alerts will be resolved -prematurely. Alerts resolved this way are marked `"resolution_source": "expiry"` -to distinguish them from a real Alertmanager resolve (`"alertmanager"`). - -An expiry cascades: once it leaves an incident with nothing firing under it, the -incident resolves too, in the same sweep. - -### Dead man's switch - -Everything above assumes alerts arrive. If Prometheus stops evaluating, or -Alertmanager cannot reach this server, nothing arrives — and silence looks -exactly like everything being fine. A dead man's switch inverts the handling for -one designated alert so that silence is the signal: - -- **receiving** it opens no incident, and -- the **absence** of it does. - -kube-prometheus-stack already ships the alert for this. `Watchdog` is -`expr: vector(1)`, so it fires permanently and is re-sent forever; it is worth -nothing unless something downstream notices it stop. That is what -`TERDUT_DEADMAN_MATCHERS` defaults to. - -**Switches belong to a team**, which decides which of its own alerts are -heartbeats and how long a silence has to last. Each **switch** is a row of its -own — a name, one matcher, a timeout and a severity — so switches in one team -can have different deadlines. An owner adds and removes them on **Team → -Switches**, which lists each with a status (**healthy**, **dead**, or -**dormant** until its first heartbeat), when it was last heard from, and when it -last opened an incident; a matcher that several clusters satisfy is broken down -per cluster. The API is `POST`/`DELETE /api/teams/{teamID}/deadman/switches`. A -missed heartbeat opens an incident in the team whose integration received it. -Removing a switch stops the watching; an incident it already opened stays open -until somebody resolves it. - -The environment variables are the starting point, not the setting: the **first** -time the server starts, every team is given a switch per default matcher from -them, once. After that a team's switches are its own — an owner's edit or -deletion is never put back by a redeploy. A team created later starts watching -nothing until its owner says otherwise — inheriting an install-wide heartbeat -would page a new team about a source it has never heard of. - -A matcher is a set of exact label conditions, one of which must be the -`alertname`, in the format the environment variable uses (one matcher per switch; the -variable takes several, separated by `;`): - -``` -alertname=Watchdog,cluster=prod; alertname=EdgeHeartbeat -``` - -**The unit of monitoring is the fingerprint, not the alert name.** Two clusters -sending the same `Watchdog` are two independent switches, so a healthy one can -never mask a dead one. - -#### The lifecycle - -A switch is **dormant** until its first heartbeat arrives. A configured matcher -that has never been heard from opens nothing, so a fresh deploy or a restored -database does not page. It also means a matcher that never matches anything is -silently inert — check the startup log line, which lists the matchers that -survived parsing. - -Once armed, the sweeper declares it **dead** when either the heartbeat has not -been refreshed within `TERDUT_DEADMAN_TIMEOUT`, or Alertmanager explicitly -resolved it — the sender saying the heartbeat stopped needs no further waiting. -That opens an incident at `TERDUT_DEADMAN_SEVERITY`, assigned and paged like any -other, and marks the heartbeat alert `"resolution_source": "deadman"` so the -alert list stops claiming a dead switch is firing. - -It **recovers** when the heartbeat starts arriving again: the incident resolves -with `"resolution_source": "recovered"` and the all-clear goes to whoever was -paged. - -Resolving the incident by hand sticks, the same way it does for an alert-backed -one. While the switch stays silent nothing new opens — so a decommissioned -source is a one-time page rather than a nag. The switch **re-arms** on the next -heartbeat: come back and die again, and that is a new incident. - -#### Two things to know - -`TERDUT_DEADMAN_TIMEOUT` must be **shorter** than the `repeat_interval` of the -route carrying the heartbeat, which is the exact opposite of -`TERDUT_STALE_AFTER`. Inheriting a default `repeat_interval` of 4h gives you a -switch that takes four hours to notice anything, so give the heartbeat -[its own route](#alertmanager-configuration). Matched alerts are exempt from -stale-alert expiry — a heartbeat answers to its own timeout and nothing else. - -A dead man's switch incident has **no member alerts**: -`GET /api/incidents/{id}/alerts` returns an empty list. There is no alert -describing the problem, because the problem is that no alert arrived. What -happened is on the timeline instead, as a `deadman_silent` event carrying the age -of the last heartbeat, and the heartbeat's labels are on the incident's -`group_labels`. - ---- - -## API reference - -### Authentication - -All endpoints except `/api/bootstrap`, `/api/integrations/{key}/alertmanager`, -`/api/notify/ack/{token}`, `/api/login`, `/api/logout`, `/api/auth/config`, -`/api/version`, `/api/oidc/login`, `/api/oidc/callback`, `/api/oidc/device` and -`/api/oidc/device/token` require either an API key: - -``` -Authorization: Bearer -``` - -or the web UI's session cookie. A request that carries an `Authorization` header -is judged on that header alone. - -Two kinds of user exist. An **administrator** manages accounts: creating and -deleting users, setting anybody's password, minting keys for anybody, and -granting the flag itself. Everybody else works incidents — acknowledging, -assigning, snoozing, resolving, noting — and manages their own account and -nobody else's. An API key carries exactly the rights of the user it belongs to. - -A third principal, the **service account**, exists for automation (a -Kubernetes operator, most likely) that needs to manage teams, escalation -policies, dead man's switches and integrations without impersonating a human. -It is not a user — it never signs in, never appears in a team's member list, -and never holds the administrator flag — and its key is prefixed `tdsa_` so it -reads as one at a glance in a log line. See [Service accounts](#service-accounts). - -**Getting an account.** The first one comes from `/api/bootstrap`. After that -it depends on `signup_mode`, an administrator setting: - -- `invite_only` (the default) — a team owner mints a link with - `POST /api/teams/{teamID}/invites`, and the person who opens it picks a - username and password and lands in that team with the role the link carries. - Links are single-use unless told otherwise, expire after seven days, and can - be revoked before that. -- `open` — anybody who can reach the server can create an account, and must - name a team, which they then own. - -Invites are **links, not email**: this server has no SMTP, and adding it to send -one message would be a subsystem to run, secure and monitor. Send the link -however you already talk to the person. - -A domain-restricted third mode was considered and dropped: with no email there -is nothing to verify an address against, so it would only check the domain of a -string somebody typed. - -The first user, from `/api/bootstrap`, is an administrator. Users created -afterwards are not, until an administrator says so. An install always keeps at -least one: the last administrator can be neither deleted nor demoted, and -nobody can delete or demote themselves. - -Endpoints that require the flag answer `403` with -`{"error":"administrator access required"}`. - -**Teams** are the unit of tenancy, and are a separate axis from the administrator -flag. A team owns its incidents, alerts, schedule and integrations, and a user -sees exactly the teams they belong to. Within a team an **owner** configures it -(schedule, integrations, membership) and a **member** works its incidents. - -An administrator crosses that line in one direction only. They **configure any -team** without being in it — every owner-only endpoint accepts the flag, because -otherwise a team whose last owner left could never be repaired. They do **not -read any team**: the queue, the alerts and the incidents are filtered by real -membership, so an administrator sees a team's work only by joining it, which is -a membership change and shows up as one. Administration is about accounts and -the shape of a team, not about reading other people's incidents. - -Anything belonging to a team you are not in answers `404`, not `403`: whether an -incident exists is itself something only its team should learn. - -**Operator mode** (`TERDUT_OPERATOR_MODE`, see [Configuration](#configuration)) -declares this install gitops-managed. When it is on, a session or a user's own -API key gets `403 {"error": "...", "reason": "operator_managed"}` on every -write this README marks **owner**-gated under Teams below (creating, renaming -or deleting a team; its OIDC group binding; its escalation ladder; its dead -man's switches; its integrations) — a service account's writes are unaffected. -Team membership and invites are deliberately excluded: they are never -gitops-managed, in operator mode or out of it. `GET /api/auth/config` reports -`operator_mode` so a client can grey those sections out before a write is ever -attempted. - -| Method | Path | Description | -|---|---|---| -| `GET` | `/api/auth/config` | How to sign in: `{"password_login", "oidc": {"enabled","name"}, "device_login", "operator_mode"}`. No session needed | -| `GET` | `/api/version` | `{"version"}` — this build's version string. No session needed, the same as `/healthz` | -| `POST` | `/api/login` | `{"username","password"}` → sets the session cookie, returns `{user, has_password}`. `429` after too many failures; `403` when `TERDUT_PASSWORD_LOGIN=false` | -| `GET` | `/api/oidc/login` | Starts a single sign-on sign-in: redirects the browser to the provider. `?next=/path` is where to land afterwards; only a path on this server is honoured. Only exists when SSO is configured | -| `POST` | `/api/oidc/device` | Starts a device login: returns `{device_code, user_code, verification_url, interval, expires_in}`. Only exists when SSO is configured | -| `POST` | `/api/oidc/device/token` | `{"device_code"}` → `202 {"status":"pending"}`, then `200` with the session cookie once approved (once only). `410` with `{"error":"expired"}` or `{"error":"denied"}`; `429 {"error":"slow_down"}` if polled faster than `interval` | -| `POST` | `/api/oidc/device/approve` | **session** — `{"user_code"}`. Approves a pending device login as the caller. `403` for an API key; `404` for an unknown, expired or already decided code | -| `POST` | `/api/oidc/device/deny` | **session** — `{"user_code"}`. Refuses it | -| `GET` | `/api/oidc/callback` | Where the provider sends the browser back. Sets the session cookie and redirects to `/`, or to `/?sso_error=` — one of `denied`, `expired`, `failed`, `unavailable`, `not_allowed`, `no_email`, `email_conflict`, `disabled`, `not_bootstrapped` (no user exists on this install yet — sign in again once something has called `/api/bootstrap`) | -| `POST` | `/api/logout` | Ends the session and clears the cookie | -| `GET` | `/api/me` | The caller: `{user, has_password}` | - -### Users - -| Method | Path | Description | -|---|---|---| -**admin** marks an endpoint that requires the administrator flag; **self or -admin** marks one you may use on your own account and an administrator may use -on anybody's. - -| Method | Path | Who | Description | -|---|---|---|---| -| `GET` | `/api/signup` | — | Whether sign-up is open, and whether `?invite=` is usable. No session needed: the caller has no account yet | -| `POST` | `/api/signup` | — | Create an account `{"username","email","password","invite"?,"team_name"?}` and sign in. `403` without a usable invite when the mode is invite-only | -| `POST` | `/api/bootstrap` | — | Create first user + API key `{"username","email","password"?}` (only works on empty DB). The user is an administrator | -| `GET` | `/api/users` | any | List users. Open to everybody: the queue's assignment control and the schedule both have to name people | -| `GET` | `/api/users/{id}/teams` | self or admin | The teams that user is in, each with their role. `/api/teams` is always about the caller; this one answers it about somebody else, for the admin page's per-user view. `404` for a user who does not exist, so "no teams" and "no such person" are distinguishable | -| `POST` | `/api/users` | **admin** | Create user `{"username","email"}`. Not an administrator | -| `DELETE` | `/api/users/{id}` | **admin** | Delete user (cascades to keys). `409` for yourself or the last administrator | -| `PUT` | `/api/users/{id}/admin` | **admin** | Grant or revoke the administrator flag `{"is_admin"}`. `409` for yourself, the last administrator, or an administrator granted by single sign-on | -| `PUT` | `/api/users/{id}/disabled` | **admin** | Take an account out of use, or put it back `{"disabled"}`. `409` for yourself or the last administrator | -| `PUT` | `/api/users/{id}/notify` | self or admin | Set push notification target `{"ntfy_topic"}` — empty string clears it | -| `PUT` | `/api/users/{id}/password` | self or admin | Set web UI password `{"password","current_password"}`. `current_password` is required only when changing your own existing password. Ends the user's other sessions | -| `POST` | `/api/users/{id}/api-keys` | self or admin | Issue API key `{"name"}` — key shown once | -| `DELETE` | `/api/users/{id}/api-keys/{keyID}` | self or admin | Revoke API key | - -### Administration - -| Method | Path | Who | Description | -|---|---|---|---| -| `GET` | `/api/admin/teams` | **admin** | Every team on the server, with its member and open-incident counts. `/api/teams` answers "what am I in"; this answers "what is there" | -| `GET` | `/api/admin/teams/{teamID}` | **admin** | One team and who is in it: `{"team", "members"}`. `404` for a team that does not exist. `GET /api/teams/{teamID}/members` is **member**-only and still `404`s an administrator from outside the team — reading a team's shape and reading its work are different questions, so they are different endpoints | -| `GET` | `/api/admin/settings` | **admin** | The editable settings with their bounds, plus the environment-configured ones, read-only. Never credentials | -| `PUT` | `/api/admin/settings` | **admin** | Change one or more `{"key": seconds}`, or `{"signup_mode": "open"\|"invite_only"}`. `400` for an unknown key or a value outside its bounds | - -### Service accounts - -A service account is a scoped, non-human credential for automation — not a -`users` row, so it never signs in, is never a team member, and never carries -the administrator flag. Two scopes: - -- **instance** — the same reach system administration has over teams: create - one, and mint a **team**-scoped account against any of them. There is no - cap on how many instance-scoped accounts exist, but ordinarily there is one, - belonging to whatever is provisioning this install end to end. -- **team** — owner-equivalent for that one team, and nothing else: every - **owner**-gated endpoint under [Teams](#teams), membership and invites - included. Nothing narrower is enforced server-side; what actually keeps - membership out of automation's hands is that no operator built against this - scope should ever call those two endpoints — see - [operator mode](#authentication) and `SERVICE-ACCOUNTS.md`'s note on this. - -A key is shown once, at creation or rotation, and only its hash is stored — -the same handling as a user's API key. Losing it means minting a new one; -there is no way to recover a raw key from the server. - -| Method | Path | Who | Description | -|---|---|---|---| -| `GET` | `/api/service-accounts` | **admin** | Every service account. Pass `?name=` instead to look one up by its exact name — open to **any** authenticated caller (human or service account), since it returns no key material and is how an account finds its own id | -| `POST` | `/api/service-accounts` | owner\* | Create one and mint its first key `{"name","scope","team_id"?}` (`team_id` required for `scope:"team"`, absent for `scope:"instance"`). Returns `{"service_account", "key"}` — `key.key` shown once | -| `POST` | `/api/service-accounts/{id}/keys` | owner\* | Mint an additional key `{"name"}` — rotation without recreating the account. Shown once | -| `DELETE` | `/api/service-accounts/{id}/keys/{keyID}` | owner\* | Revoke one key | - -\* For an **instance**-scoped account: a system administrator only. For a -**team**-scoped account: a system administrator, that team's own human owner, -an instance-scoped service account (minting a narrower credential for a team -it just created), or — for the two key endpoints only — the account rotating -or revoking its own key, which is not a privilege escalation, the same -reasoning a user's own API keys rest on. - -### Alert ingestion - -Alerts arrive on a team's integration key. The key is both the credential and the -routing: it says that the sender may post, and which team the alerts belong to. -Create one with `POST /api/teams/{teamID}/integrations`, which returns the key -and the full URL once and stores only a SHA-256 hash. - -| Method | Path | Description | -|---|---|---| -| `POST` | `/api/integrations/{key}/alertmanager` | Alertmanager v4 webhook receiver for the key's team. `401` for an unknown key | - -This is the only way in. The pre-teams `POST /api/alertmanager/webhook` took no -credential at all — anything able to reach the port could open an incident — -and was removed in v0.13.0 once senders had moved onto keys. - -### Teams - -**owner** below means an owner of that team, a system administrator (who -passes every one of these without being a member), or that team's own -team-scoped [service account](#service-accounts) — including membership and -invites, technically, though no automation this scope was designed for -(a Kubernetes operator's CRDs, see `SERVICE-ACCOUNTS.md`) ever models team -membership or would call those two. See [Authentication](#authentication). -**member** means membership and nothing else: an administrator who is not in -the team gets the same `404` as anybody else. - -| Method | Path | Who | Description | -|---|---|---|---| -| `GET` | `/api/teams` | any | The caller's own teams, each with their role | -| `POST` | `/api/teams` | any | Create a team `{"name"}`; a human creator becomes its first owner. An instance-scoped [service account](#service-accounts) may also create one, and it gets no owner at all — expected for a team an operator is about to hand a team-scoped credential to, not an orphaned team a human made | -| `PUT` | `/api/teams/{teamID}` | **owner** | Rename it `{"name"}`. `409` if the name is taken | -| `DELETE` | `/api/teams/{teamID}` | **owner** | Delete a team and everything under it. `409` while it has open incidents | -| `GET` | `/api/teams/{teamID}/members` | member | Who is in the team, with `status` (`oncall` if the rota has them today, `unpageable` when a page to them would go nowhere — even if they are on call — else `reachable`), `on_call`, `next_shift` (first rota day after today), `pageable` and `problem` (`has no ntfy topic` / `account is disabled`; never the topic itself) and `last_active_at` (their newest session or API-key use). Every member sees the same list | -| `POST` | `/api/teams/{teamID}/members` | **owner** | Add a member, or change their role `{"user_id","role"}`. `409` when it would demote the last owner, or the membership is managed by single sign-on | -| `DELETE` | `/api/teams/{teamID}/members/{userID}` | **owner** | Remove a member. `409` for the last owner, or a membership managed by single sign-on | -| `GET` | `/api/teams/{teamID}/oidc-groups` | member | Which groups control this team's membership: `{"member_group","owner_group"}`. An empty string means no group grants that role here | -| `PUT` | `/api/teams/{teamID}/oidc-groups` | **owner** | Set them. An empty string clears a binding | -| `GET` | `/api/teams/{teamID}/integrations` | member | List integrations. Never returns keys. Each carries `status` (`active` if its key posted within 24h, `quiet` if it has but not lately, `never`), `last_used_at` (last webhook, usable or not), `last_alert_at` (when an alert last arrived on it) and `alerts_24h` (distinct alerts it refreshed in the last day). Alerts delivered before the source was recorded (migration 010) have none, so the last two fill in as Alertmanager re-sends them | -| `PATCH` | `/api/teams/{teamID}/integrations/{integrationID}` | **owner** | Rename `{"name"}`. The key does not change | -| `POST` | `/api/teams/{teamID}/integrations` | **owner** | Mint an integration `{"name","kind"}` — key and URL shown once | -| `DELETE` | `/api/teams/{teamID}/integrations/{integrationID}` | **owner** | Revoke an integration. Alerts it delivered stay, unattributed | -| `GET` | `/api/teams/{teamID}/invites` | **owner** | The team's invite links, with their uses and expiry. Never the tokens | -| `POST` | `/api/teams/{teamID}/invites` | **owner** | Mint one `{"role","max_uses"}` — the full URL is returned once | -| `DELETE` | `/api/teams/{teamID}/invites/{inviteID}` | **owner** | Revoke a link before it expires | -| `GET` | `/api/teams/{teamID}/escalation` | member | The team's [escalation ladder](#escalation) `{repeat_count, fallback_topic, levels[], last_escalated_at?, last_escalated_incident_id?}`. Empty levels means the team has none. Each level also carries `status` (`ready`, `escalating` when an unanswered incident has climbed to it, `unreachable` when nobody on it could be woken), `waiting` (ids of the open incidents on it) and, per target, `username` (who it means today — the person on call, for a rota target), `reachable` and `problem`. The extra fields are output only; `PUT` takes the plain shape | -| `PUT` | `/api/teams/{teamID}/escalation` | **owner** | Replace it wholesale. `400` for a level with no targets or no timeout — a rung that pages nobody is a silence with a number on it | -| `GET` | `/api/teams/{teamID}/deadman/switches` | member | The team's [dead man's switches](#dead-mans-switch), each `{id, name, matcher, timeout_seconds, severity, status, last_heartbeat_at, last_triggered_at, open_incident_id, sources[]}`. `status` is `healthy`, `dead` or `dormant`; `sources` has one entry per heartbeat fingerprint. Empty when the team watches nothing | -| `POST` | `/api/teams/{teamID}/deadman/switches` | **owner** | Add one: `{name?, matcher, timeout_seconds, severity?}`. `400` when the matcher names no `alertname` or holds several, or the timeout is not positive — a switch that silently watches nothing is the failure this feature exists to prevent | -| `PUT` | `/api/teams/{teamID}/deadman/switches/{switchID}` | **owner** | Replace one in place, same body and validation as create. Its id is unchanged — for an automated caller reconciling a spec change, unlike delete-and-recreate | -| `DELETE` | `/api/teams/{teamID}/deadman/switches/{switchID}` | **owner** | Stop watching. An incident it opened stays open. `404` for a switch of another team | - -### Notifications - -| Method | Path | Description | -|---|---|---| -| `POST` | `/api/notify/ack/{token}` | Acknowledge an incident from a push notification's Acknowledge button. No auth: the token in the path is the credential — one incident, one action, 24 hours, idempotent. Must stay publicly reachable | - -### Incidents - -| Method | Path | Description | -|---|---|---| -| `GET` | `/api/incidents` | List incidents. Filters: `?status=triggered\|acknowledged\|resolved`, `?severity=`, `?assigned_to=`, `?archived=true`, `?snoozed=true`, `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD`, `?sort=severity`, `?cluster=`, `?limit=` (default 50, max 500) | -| `GET` | `/api/incidents/clusters` | The distinct `cluster` values on the caller's incidents from the last 90 days, sorted (`?team_id=` narrows it). An empty array when nothing carries the label | -| `GET` | `/api/incidents/{id}` | Get single incident, with its alerts inline | -| `GET` | `/api/incidents/{id}/alerts` | Alerts under this incident | -| `GET` | `/api/incidents/{id}/timeline` | Full event history, chronological | -| `POST` | `/api/incidents/{id}/acknowledge` | Acknowledge (stamps authed user + time) | -| `DELETE` | `/api/incidents/{id}/acknowledge` | Clear acknowledgement, back to `triggered` | -| `POST` | `/api/incidents/{id}/resolve` | Close by hand — **terminal**, see above | -| `POST` | `/api/incidents/{id}/assign` | Reassign `{"user_id"}` | -| `POST` | `/api/incidents/{id}/snooze` | Hide until `{"until": RFC3339}` or `{"duration": "2h"}` | -| `DELETE` | `/api/incidents/{id}/snooze` | Un-snooze | -| `POST` | `/api/incidents/{id}/archive` | Archive (hides from the default list) | -| `DELETE` | `/api/incidents/{id}/archive` | Un-archive | -| `POST` | `/api/incidents/{id}/notes` | Add a note `{"content"}` | -| `DELETE` | `/api/incidents/{id}/notes/{eventID}` | Delete own note | - -With no `?status=` filter, `GET /api/incidents` returns **open** incidents only — -the queue an on-call person wants. Currently snoozed and archived incidents are -excluded unless asked for. Actions that only make sense on an open incident -return `409` once it is resolved. - -Notes are ordinary timeline events of type `note`; only they are deletable, and -only by their author. The rest of the timeline is a record of what happened. - -#### The incident object - -| Field | Type | Notes | -|---|---|---| -| `id` | integer | Server-assigned | -| `group_key` | string | Alertmanager's `groupKey` — opaque, treat as an identifier | -| `title` | string | Rendered from `groupLabels` | -| `group_labels` | object | String→string, as sent by Alertmanager | -| `status` | string | `"triggered"`, `"acknowledged"` or `"resolved"` | -| `severity` | string | *optional* — high-water mark across the incident's alerts; never lowered | -| `triggered_at` | timestamp | When the incident opened | -| `acknowledged_by_id` / `acknowledged_by` / `acknowledged_at` | | *optional* — user id, username, time | -| `assigned_to_id` / `assigned_to` | | *optional* — user id, username | -| `snoozed_until` | timestamp | *optional* — a value in the past reads as not snoozed | -| `resolved_at` | timestamp | *optional* | -| `resolution_source` | string | *optional* — `"alerts"`, `"manual"` or `"recovered"` | -| `archived_at` | timestamp | *optional* | -| `alerts` | array | Only on `GET /api/incidents/{id}` | - -Treat `resolution_source` as an open set, as with the alert field of the same -name: degrade unknown values to "resolved, reason unknown". - -#### The timeline event object - -| Field | Type | Notes | -|---|---|---| -| `id` | integer | | -| `incident_id` | integer | | -| `type` | string | See below — treat as an open set | -| `user_id` / `username` | | *optional* — absent when the server acted rather than a person | -| `alert_id` | integer | *optional* — the alert an `alert_added` / `alert_resolved` event refers to | -| `detail` | string | *optional* — the note body, the snooze deadline, etc. | -| `created_at` | timestamp | | - -Types written today: `triggered`, `alert_added`, `alert_resolved`, -`acknowledged`, `unacknowledged`, `assigned`, `archived`, `unarchived`, `snoozed`, -`unsnoozed`, `resolved`, `note`, `notified`, `notify_failed`, `deadman_silent`. On an -`assigned` event `user_id` is the **assignee**, not the actor; the actor is in -`actor_user_id`/`actor_username` or `actor_service_account_id`/`actor_service_account_name` -(absent on assignments made before they were recorded). New types may be added; render -unknown ones generically rather than dropping them. - -On `notified` and `notify_failed`, `detail` carries the notification kind -(`triggered` | `reminder` | `resolved`), and on a failure the reason after it. -`user_id` is who was paged — absent means the page went to the shared fallback -topic and so belongs to nobody. The topic itself is never written to the -timeline: it is a shared secret with the ntfy server, and every API key can read -this. - -### Alerts - -Alerts are read-only. Everything a person does happens on the incident. - -| Method | Path | Description | -|---|---|---| -| `GET` | `/api/alerts` | List alerts. Filters: `?status=firing\|resolved`, `?name=`, `?incident_id=`, `?archived=true`, `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD`, `?limit=` (default 50, max 500) | -| `GET` | `/api/alerts/{id}` | Get single alert | - -Archived alerts are hidden from `GET /api/alerts` unless `?archived=true` is -passed; alert archiving is automatic housekeeping by the sweeper, not a user -action. Resolved alerts carry `resolution_source`: `"alertmanager"` for a real -resolved webhook, `"expiry"` when the sweeper inferred it (see -[Stale alert expiry](#stale-alert-expiry)), `"deadman"` for a heartbeat declared -dead (see [Dead man's switch](#dead-mans-switch)). - -#### The alert object - -Returned by `GET /api/alerts` (as an array) and `GET /api/alerts/{id}`. -Timestamps are RFC 3339 in UTC. Fields marked *optional* are omitted entirely -when unset, so clients must treat them as nullable. - -| Field | Type | Notes | -|---|---|---| -| `id` | integer | Server-assigned; stable for the life of the row | -| `fingerprint` | string | Alertmanager's fingerprint — the upsert key | -| `name` | string | From the `alertname` label | -| `status` | string | `"firing"` or `"resolved"` | -| `labels` | object | String→string, as sent by Alertmanager | -| `annotations` | object | String→string, as sent by Alertmanager | -| `starts_at` | timestamp | When the alert instance began, **per Prometheus** | -| `ends_at` | timestamp | *optional* — absent while no end is known | -| `generator_url` | string | Link back to the originating Prometheus | -| `received_at` | timestamp | When the server last accepted a webhook for this alert — see below | -| `incident_id` | integer | *optional* — the most recent incident this alert belongs to | -| `resolution_source` | string | *optional* — `"alertmanager"`, `"expiry"` or `"deadman"` | -| `archived_at` | timestamp | *optional* — set while archived | - -##### `received_at` is a liveness heartbeat - -`starts_at` comes from Prometheus and **never changes** for the lifetime of an -alert instance. It says when the problem began, not whether it is still -happening — an alert that started twelve days ago looks identical whether -Alertmanager refreshed it a minute ago or went silent a week ago. - -`received_at` is the field that answers "is this still live". It is set to the -server's clock on **every accepted webhook** for that fingerprint, including the -unchanged firing notifications Alertmanager re-sends every `repeat_interval`. -Clients may rely on this: - -- **A firing alert whose `received_at` is advancing is still being refreshed.** - Stale-dating it against `repeat_interval` is a valid liveness check, and it is - what the built-in sweeper does (see - [Stale alert expiry](#stale-alert-expiry)). -- **`received_at` tracks accepted payloads, not delivery attempts.** A retry - that describes an older instance than the stored one is discarded, and a - discarded payload does not move `received_at`. -- **It stops advancing once the alert resolves,** because Alertmanager stops - re-sending. On an alert resolved by the sweeper - (`"resolution_source": "expiry"`) it therefore marks the last time - Alertmanager was actually heard from, which is earlier than `ends_at`. - -`GET /api/alerts` is ordered by `received_at` descending — most recently -refreshed first — and the `?from=` / `?to=` filters on both the alert and stats -endpoints select on `received_at`, not `starts_at`. - -##### `resolution_source` says how much to trust `ends_at` - -An alert can leave the firing state two ways, and `resolution_source` records -which happened. Clients may rely on this: - -- **Absent while firing.** It is set only on resolve, and a re-fire under the - same fingerprint clears it again, so its presence always agrees with - `"status": "resolved"`. -- **`"alertmanager"` — a real resolved webhook arrived.** `ends_at` is the end - time Alertmanager reported. It is an observed value and can be displayed as - fact. -- **`"expiry"` — the sweeper inferred the resolve** because Alertmanager stopped - refreshing the alert (see [Stale alert expiry](#stale-alert-expiry)). Nothing - ever reported an end, so **`ends_at` is approximate**: it is either the stale - `endsAt` watermark from the last notification, or — when that notification - carried none — the time the sweep ran, which lags the last real contact by up - to `TERDUT_STALE_AFTER` plus a sweep interval. Treat it as "no later than", - not as when the problem stopped. - - On these alerts `received_at` is the more truthful signal: it marks the last - time Alertmanager was actually heard from. Surfacing the distinction is - worthwhile, since `"expiry"` can also mean the alert is still firing and the - notification path broke. - -- **`"deadman"` — a heartbeat was declared dead** (see - [Dead man's switch](#dead-mans-switch)). Like `"expiry"`, an inference from - silence rather than an observed end, so `ends_at` is approximate — but a much - tighter one, bounded by `TERDUT_DEADMAN_TIMEOUT`. It is also the one resolution - a re-fire under the same `starts_at` can undo, since the switch coming back is - exactly the evidence that the inference was wrong. - -Treat the value as an open set and tolerate ones you do not recognise — new -sources may be added, and unknown values should degrade to "resolved, reason -unknown" rather than being rejected. - -### On-call schedule - -| Method | Path | Description | -|---|---|---| -Each team keeps its own rota, so two teams can have two different people on call -on the same day. The person taking a shift has to be in the team — paging -somebody who cannot open the incident is worse than paging nobody. - -| Method | Path | Who | Description | -|---|---|---|---| -| `POST` | `/api/teams/{teamID}/schedule` | **owner** | Assign user to dates `{"user_id", "dates":["YYYY-MM-DD",...], "replace"}` — all-or-nothing | -| `GET` | `/api/teams/{teamID}/schedule` | member | List entries. Filters: `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD` | -| `DELETE` | `/api/teams/{teamID}/schedule/{id}` | **owner** | Remove schedule entry | -| `GET` | `/api/schedule/current` | any | Who is on call today (UTC) in **every** team the caller is in — one entry per team, `[]` when nobody anywhere | - -### Statistics - -Every figure counts the caller's own teams only: a report that counted other -teams' incidents would leak their volume, and their alert names through the -top-alerts list, and would not be a number about the reader's work anyway. - -All stat endpoints accept optional `?from=YYYY-MM-DD` and `?to=YYYY-MM-DD`, and exclude archived rows to match the default list views. Alert stats filter on `received_at`; incident stats filter on `triggered_at`. - -| Method | Path | Description | -|---|---|---| -| `GET` | `/api/stats/incidents` | `{total, triggered, acknowledged, resolved, mtta_seconds, mttr_seconds}` | -| `GET` | `/api/stats/alerts` | `{total, firing, resolved}` counts | -| `GET` | `/api/stats/alerts/top` | Most frequent alert names. `?limit=` (default 10, max 100) | -| `GET` | `/api/stats/alerts/by-hour` | Count per hour-of-day (UTC), all 24 slots returned | -| `GET` | `/api/stats/alerts/by-day` | Count per day-of-week, all 7 slots with names returned | - -`mtta_seconds` (time to acknowledge) and `mttr_seconds` (time to resolve) are -averages over incidents that have actually been acknowledged or resolved, and are -**null** until there are any — null means "no data", not zero. - ---- - -## Upgrading to teams - -Everything that existed before teams moves into one team called **Default**, and -every existing user becomes an owner of it. The upgrade is a no-op for the -people using it: the same queue, the same schedule, the same incidents, with a -name on them. - -What changes, and will need attention: - -- **Alert ingestion moved.** Mint a key with - `POST /api/teams/{teamID}/integrations` and point Alertmanager at the URL it - returns. In v0.12.0 the old `POST /api/alertmanager/webhook` still worked, - deprecated, routing everything to the oldest team; **v0.13.0 removes it**, so - upgrade straight from v0.11.x to v0.13.0 only after the senders are moved. -- **The schedule endpoints moved** under `/api/teams/{teamID}/schedule`, and - editing the rota is now an owner's job. `GET /api/schedule/current` stayed - where it was but now returns an **array** — one entry per team with somebody - on call — instead of a single object or a 404. This is a breaking API change - for anything that reads it, terdut-tui included. -- **Uniqueness is per team now.** Two teams can legitimately see the same alert - fingerprint, the same Alertmanager groupKey, and put somebody on call on the - same date. - -**Dead man's switches moved too.** `TERDUT_DEADMAN_MATCHERS`, `_TIMEOUT` and -`_SEVERITY` are no longer the setting; they are the default each existing team -is seeded with at startup, after which an owner manages them per team through -`/api/teams/{teamID}/deadman/switches` and a redeploy never overwrites that. - -Nothing else about an incident changes, and incidents never move between teams: -an alert belongs to whichever team's key it arrived on. - -## Upgrading to roles - -Before this release every authenticated caller could create and delete users, -set anybody's password and mint anybody's API keys. That is now the -administrator flag, and the migration **makes every existing user an -administrator** — they already held those powers, so nobody's access changes on -upgrade and demotion is a deliberate act afterwards. Promoting only the first -user would have silently stripped the rest, and could leave an install whose -only administrator is an account nobody has a password for. - -Users created after the upgrade are not administrators. Hand the flag out with: - -```bash -curl -X PUT https://terdut.example.com/api/users/7/admin \ - -H "Authorization: Bearer $TERDUT_API_KEY" \ - -H 'Content-Type: application/json' \ - -d '{"is_admin": true}' -``` - -Nothing in the API changed shape, so terdut-tui needs no new version — but a -non-administrator now gets `403` where a `200` used to come back. - -## Upgrading from SQLite - -Versions up to v0.10.2 stored everything in a SQLite file. From v0.11.1 the server needs -`TERDUT_DB_DSN` and keeps nothing on disk. - -The copy was done by `scripts/sqlite-to-postgres.go`, which **was deleted in v0.13.0** along -with the SQLite driver it was the last user of. It is still in the history — check out the -`v0.12.0` tag to get it: - -```bash -git show v0.12.0:scripts/sqlite-to-postgres.go > sqlite-to-postgres.go -``` - -The cutover is ordered, and the server must not be running while the copy happens: stop the -old version, let the new binary build the schema against an empty Postgres, run the script -with `-sqlite` and `-dsn`, then start the new version for good. On Kubernetes step three runs -as a Job with the same image against the PVC before it is removed. - -The copy preserves every id, so incidents keep their numbers and the timeline, alert -membership, outbox and ack tokens all still point where they did. It refuses a target that -already has rows, so a second run cannot double-insert. - -## Upgrading to incidents - -The incidents release moves the workflow off alerts, which is a **breaking API -change**. These endpoints are gone: - -| Removed | Replacement | +| | | |---|---| -| `POST`/`DELETE` `/api/alerts/{id}/acknowledge` | `POST`/`DELETE` `/api/incidents/{id}/acknowledge` | -| `POST`/`DELETE` `/api/alerts/{id}/archive` | `POST`/`DELETE` `/api/incidents/{id}/archive` (alert archiving is now sweeper-only) | -| `GET`/`POST` `/api/alerts/{id}/comments` | `GET /api/incidents/{id}/timeline`, `POST /api/incidents/{id}/notes` | -| `DELETE /api/alerts/{id}/comments/{commentID}` | `DELETE /api/incidents/{id}/notes/{eventID}` | +| [Deployment](./docs/deployment.md) | Docker, Helm chart, database, backups, the operator | +| [Configuration](./docs/configuration.md) | Environment variables and settings | +| [Alertmanager configuration](./docs/alertmanager.md) | Routes, keys, webhooks | +| [Alerts and incidents](./docs/incidents.md) | Correlation, lifecycle, on-call assignment | +| [Single sign-on](./docs/single-sign-on.md) | OIDC and the terminal device flow | +| [API reference](./docs/api.md) | Every endpoint | +| [Development and releasing](./docs/development.md) | Tests, CI gate, release pipeline | -The alert object also drops `acknowledged_by_id`, `acknowledged_by` and -`acknowledged_at`, and gains `incident_id`. +## Related -Migration `008_incidents.sql` runs automatically on start and preserves existing -data: every alert gets a backfilled incident carrying its acknowledgement, and -comments become timeline notes. Backfilled incidents have a `group_key` of -`backfill:` — there is no historical `groupKey` to correlate on, so -they are one-per-alert rather than grouped. +- [terdut-tui](https://git.ryuvia.com/niklas/terdut-tui): a terminal client for the same server. +- [terdut-operator](https://git.ryuvia.com/niklas/terdut-operator): a Kubernetes operator that + runs the server and manages teams, escalation, switches and alert sources as objects. -Nothing about the two documented alert contracts changes: `received_at` is still -advanced on every accepted webhook, and `resolution_source` still means what it -did. +## License -## Upgrading to dead man's switches - -Dead man's switch handling is **on by default**, watching `alertname=Watchdog` -with a 15 minute timeout. If you already route `Watchdog` to this server, the -behaviour of that alert changes on upgrade, in both directions: - -- it stops opening incidents when it arrives, and -- it starts opening one when it stops arriving. - -**Check your `repeat_interval` before upgrading.** The switch pages whenever a -heartbeat has not been refreshed within `TERDUT_DEADMAN_TIMEOUT`, so a `Watchdog` -route inheriting a 4h or 12h `repeat_interval` will page constantly against the -15 minute default. Either give the heartbeat -[its own fast route](#alertmanager-configuration) — the point of the feature — or -set `TERDUT_DEADMAN_TIMEOUT` above your current `repeat_interval` until you have. -`TERDUT_DEADMAN_TIMEOUT=0` turns the whole thing off. - -There is no migration and no schema change. An existing open incident from a -`Watchdog` that arrived under the old behaviour is unaffected; resolve it by hand. - ---- - -## Development - -```bash -make test-db # start a local Postgres for the tests (podman or docker) -make test # run all tests -go build ./... # compile all packages -go run ./cmd/terdut # run locally (needs TERDUT_DB_DSN) -``` - -The tests need a real Postgres, because the server does — there is no in-memory Postgres the -way there was an in-memory SQLite. `TERDUT_TEST_DSN` says where it is, `make test-db` starts -one on port 5433 and prints the DSN, and `make test-db-stop` removes it. Each test gets its -own schema on that server, so tests cannot see each other's rows. An unset `TERDUT_TEST_DSN` -fails the suite rather than skipping it: a run that quietly tests nothing is worse than one -that does not run. - -`make fmt lint test helm-lint` is the gate. It mirrors `.gitea/workflows/ci.yaml` step for -step, so a green run here means a green pipeline — with one deliberate exception: `make test` -adds `-race`, which CI does not. The sweeper, the notifier goroutine and the dead man's switch -sweep all run concurrently against the same database, and a race between them would surface as -a flaky incident in production rather than as a red build. - -The web UI lives in `internal/web/static/` as plain HTML, CSS and ES modules, -embedded into the binary with `go:embed`. It has no build step and no npm, so -editing a file and restarting the server is the whole loop. - -## Releasing - -``` -push or PR → ci.yaml gofmt, go vet, go test -race - govulncheck, gitleaks - helm lint + render -push tag vX.Y.Z → release.yaml the same gate, then publish: - git.ryuvia.com/niklas/terdut-server:vX.Y.Z - oci://git.ryuvia.com/niklas/terdut-server X.Y.Z - then trivy-scan the pushed image -PR to Ryuvia/charts → bump the wrapper chart to X.Y.Z; on merge - Flux reconciles and the release rolls out -``` - -Both artifacts go to the **personal** Gitea namespace rather than `ryuvia`, because Gitea -scopes package visibility to the owner with no per-package override — so `ryuvia/*` is private -because the org is. Publishing to `niklas` keeps them anonymously pullable, which is why no -pull secret is needed in the cluster. Same reasoning, and the same choice, as riksdata and -rd-web. - -Saying **"Release"** runs all three rows: the `release` skill commits, pushes, tags, waits for -the pipeline, and opens the `Ryuvia/charts` PR, stopping before the merge. See -`~/.claude/skills/release/`, or `.release.conf` here for this repo's part of it. - -The chart is published **only** from the tag, by the `chart` job. There used to be a second -publisher on every `charts/**` push to main, and the two raced for the same chart version with -different answers — chart 0.9.0 went out reading `appVersion: "latest"` that way. One -publisher, triggered by the tag (`766f439`). The cost is that a chart-only change has no -version of its own and rides the next app tag. - -Both workflows are thin drivers over the Makefile: `ci.yaml` runs `make fmt lint test` and -`make helm-lint`, `release.yaml` adds `make binaries`, `make push`, `make helm-package` and -`make helm-push`. That is deliberate — it is what makes a green local gate and a green -pipeline the same code rather than two descriptions of it, and it is how riksdata and rd-web -have always worked. - -`make push` builds and pushes in one step, unlike those two, because the image is -`linux/amd64,linux/arm64` and buildx cannot load a multi-platform result into the local image -store. `make build` stays single-platform and local-only. Both refuse `VERSION=dev`: -publishing is one command, so it is also one command to run by accident. Publishing happens -by pushing a tag. - -Two things the release process needs to know about this repo: - -- **The image scan runs after publishing**, like riksdata's and rd-web's: trivy cannot read - a locally built image on this runner, so it pulls the pushed one. A red `scan-image` means - do not bump the wrapper chart to that version — it cannot unpublish anything. The image is - `FROM scratch`, so trivy sees exactly one target, the Go binary and its module graph. -- **The wrapper chart's `values.yaml` has two `tag:` lines** — the app image and the python - backup sidecar — so `chart-bump` is given `--image` to say which one moves. The sidecar is - on its way out with SQLite: once the wrapper chart drops it and declares a `postgresql` CR - instead, there is one `tag:` line again, and `--image` becomes belt and braces. - -The wrapper chart must have **its own `version:` bumped in the same commit**. Flux reconciles -with `reconcileStrategy: ChartVersion`, so a chart whose version did not change produces no -new artifact and the change is never deployed — with no error anywhere. +See [LICENSE](./LICENSE). diff --git a/SERVICE-ACCOUNTS.md b/SERVICE-ACCOUNTS.md index 2d264c0..ff768a9 100644 --- a/SERVICE-ACCOUNTS.md +++ b/SERVICE-ACCOUNTS.md @@ -1,263 +1,60 @@ -# Service accounts: a scoped, non-human credential type +# Service accounts -This is a design note for a feature, not an implementation plan — it exists to -propose the shape before writing code. It's raised directly by `terdut-operator` -(a separate repo, no shared code — see its `DESIGN.md` §6, §9, §13), which needs -a credential for unattended, repeatable API access and currently has no good one -available. Anything automating terdut-server long-term (this operator, CI, future -integrations) hits the same gap, so this is written as a general primitive, not -operator-specific. +A non-human credential for automation (terdut-operator, CI, scripts). It is not a +`users` row: no password, no `is_admin`, no OIDC identity, so it can never be +pulled into login or group sync, and it is never mistaken for a person in an +audit trail. The bearer token has the same shape as an API key (SHA-256 hash +stored, raw value shown once), prefixed `tdsa_`. -## The problem +## Scopes -terdut-server has two credential types today, and neither fits "an unattended -process that manages teams/schedules/policies on someone's behalf": +- **instance** — acts as owner of every team's *configuration* (rename, OIDC + groups, escalation, dead man's switches, integrations, members, delete) and may + create teams. It is not a member of any team, so it reads no incidents or + queue. It is never an administrator: user management and + `/api/admin/settings` stay human-only. +- **team** — acts as owner of exactly one team, through a single synthetic + membership. It may also mint another service account for its own team. -- **User API keys** (`api_keys`, `internal/api/users.go`) are always tied to a - real `users` row and carry that user's full rights — every team they're a - member of, their admin flag if set. There's no `kind`/`service` marker - distinguishing "a human's personal automation key" from "a login session," and - no way to mint one scoped to less than the full user. -- **Integration keys** (`integrations`, `internal/api/*teams*.go`) are team-scoped, - but narrowly: they authenticate exactly one inbound Alertmanager webhook call - (`POST /api/integrations/{key}/alertmanager`) and nothing else. They're not a - general management-API credential and shouldn't become one — overloading a - narrow, one-way ingestion credential with broad read/write access would weaken - the one property that makes it safe to embed in an Alertmanager config today. +An account has many keys, so rotating is "mint a new key, revoke the old one" +without losing the account's identity or history. -The result: any automation that needs to create teams, set escalation policies, -manage dead-man switches, or rotate integration keys has to hold a real human -admin's or team owner's API key. That key is exactly as powerful as that person -logging in — full team access, and full instance access if they're an admin. -`terdut-operator`'s design ran directly into this (its DESIGN.md §6): its -described bootstrap/rotation flow assumed a repeatable, identity-scoped way to -get a credential, and `/api/bootstrap`'s actual behavior (single-shot per -install, gated on `COUNT(*) FROM users`, confirmed via `internal/api/users.go` -and `charts/terdut-server/templates/bootstrap-job.yaml`) doesn't provide one — -it mints exactly one founding admin, once, ever. +## Endpoints -## Goals +- `POST /api/service-accounts` `{name, scope, team_id}` — returns the account and + its first key. An instance-scoped account is granted by a human administrator; + a team-scoped one by an administrator, that team's owner, or an instance-scoped + account. +- `GET /api/service-accounts?name=` — look one up by name. +- `POST /api/service-accounts/{id}/keys`, `DELETE .../keys/{keyID}` — mint or + revoke a key. An instance-scoped account may manage any team-scoped account's + keys, and any account may manage its own. -- A credential type that isn't a human: doesn't touch OIDC group sync, login, - session, or the `is_admin`/account-management semantics that come with a real - `users` row. -- Two scopes matching the two shapes automation actually needs: instance-wide - (create/list teams — what a server-owning controller needs) and team-scoped - (manage one team's escalation policy, dead-man switches, integrations, - schedule, OIDC group bindings — what a per-team controller or integration - needs). -- Repeatable issuance and rotation — unlike `/api/bootstrap`, callable more than - once, by anything that already holds admin rights, without destroying and - recreating state to get a fresh credential. -- Visibly distinct from a human in every place identity shows up (audit trails, - timeline entries, UI attribution) — a service account acting on a team should - never be indistinguishable from a person. +## Seeding the operator's account -## Non-goals +`TERDUT_OPERATOR_KEY` (at least 32 characters) creates the instance-scoped account +`terdut-operator` if missing and replaces its `seed` key with this value at every +start (`internal/api/operator_key.go`). The deployer generates the key and +nothing has to call `/api/bootstrap` for it; rotating is a restart with a new +value. With `TERDUT_OPERATOR_MODE` on, configuration writes by humans are refused +and a service account of either scope passes. -- Not a general OAuth2/OIDC client-credentials flow — this is a bearer-token - primitive matching the shape `api_keys` already uses (SHA-256 hash stored, - raw key shown once at creation), not a new auth protocol. -- Not replacing integration keys — those stay as the narrow, one-way webhook - credential they are today. -- Not modeling per-endpoint or per-verb permissions within a scope — `instance` - and `team` are the only two scopes for now; finer-grained scoping is future - work if a real need shows up. +## How it is enforced -## Proposed shape +Every request resolves to one `Caller` (`internal/api/caller.go`): a human +(session or API key) or a service account. -### Schema +- `Caller.IsAdmin()` is true only for a human administrator. `AdminOnly` and + `requireSelfOrAdmin` key on it alone; do not widen them — each time a gap came up + the fix was a narrower purpose-built capability instead. +- `Caller.IsInstanceServiceAccount()` is true only for an instance-scoped account, + never for a human. `requireTeamOwner` and `callerOwnsTeam` admit it for any team. +- `Caller.Role(teamID)`/`TeamIDs()` are a human's memberships or a team-scoped + account's single owner membership; instance scope has none. +- `Caller.AsHuman()` is what a handler must call when it needs a real `user_id`; + handlers meant for people answer 403 to a service account instead of writing a + zero id. -```sql -CREATE TABLE service_accounts ( - id BIGSERIAL PRIMARY KEY, - name TEXT NOT NULL UNIQUE, -- e.g. "terdut-operator" - scope TEXT NOT NULL CHECK (scope IN ('instance', 'team')), - team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE, - -- team_id required iff scope = 'team'; NULL iff scope = 'instance' - created_by BIGINT REFERENCES users(id), - created_at TIMESTAMPTZ NOT NULL DEFAULT now() -); - -CREATE TABLE service_account_keys ( - id BIGSERIAL PRIMARY KEY, - service_account_id BIGINT NOT NULL REFERENCES service_accounts(id) ON DELETE CASCADE, - key_hash TEXT NOT NULL UNIQUE, - name TEXT NOT NULL, -- e.g. "initial", "2026-Q4-rotation" - created_at TIMESTAMPTZ NOT NULL DEFAULT now(), - last_used_at TIMESTAMPTZ -); -``` - -Deliberately not a `users` row: no `password_hash`, no `is_admin`, no -`user_identities` linkage, so it's structurally impossible for a service account -to be pulled into OIDC group sync or password login. Multiple keys per account -(mirroring `api_keys`' existing one-user-many-keys shape) so rotation is "mint a -new key, revoke the old one," not "recreate the account." - -### Endpoints - -- `POST /api/service-accounts` — instance-scope/admin-only. Body: - `{"name": ..., "scope": "instance"|"team", "teamID": ... }` (teamID required - iff scope=team, and caller must be that team's owner or a system admin). - Returns the account plus its first raw key (shown once, same pattern as - `POST /api/users/{id}/api-keys`). Safe to call again with the same `name` — - see "idempotent lookup" below — unlike `/api/bootstrap`, which is inherently - one-shot by design (it's answering "does any user exist yet," a question with - no analogue once one already does). -- `POST /api/service-accounts/{id}/keys` — mint an additional key on an existing - account (self-service-equivalent: instance admin for `instance` scope, team - owner or system admin for `team` scope). Enables rotation without recreating - the account or losing its identity/audit history. -- `DELETE /api/service-accounts/{id}/keys/{keyID}` — revoke one key, mirroring - `DELETE /api/users/{id}/api-keys/{keyID}`. -- `GET /api/service-accounts?name=` — look up an existing account by name. - This is what turns "I tried to create my account and got a conflict" into a - normal flow instead of an error: a controller that expects to have already - registered itself calls this first, and only falls through to `POST` if - nothing comes back. - -### Auth middleware - -**Revised** (this section originally described an aspiration that didn't -match what shipped — `TEAM-LOOKUP.md` already caught one instance of that, -and a fuller audit found three more; this is the corrected, as-built -description, not the original proposal). - -`internal/api/middleware.go`'s dual resolution (`Authorization: Bearer` → -`apiKeyUser()`, or session cookie → `sessionUser()`) and the service-account -path (`serviceAccountFor()`) both resolve into one `Caller` type -(`internal/api/caller.go`), not two parallel, un-unified context -representations the way an earlier version of this server kept them. Every -authorization predicate reads `Caller`'s methods: - -- `Caller.IsAdmin()` — true **only** for a human system administrator, never - for a service account of either scope, under any circumstance. `AdminOnly` - and `requireSelfOrAdmin` key on this alone — user management - (`POST /api/users`, `PUT /api/users/{id}/admin`, etc.) and - `GET/PUT /api/admin/settings` stay human-only, forever. The original text - here claimed an instance-scoped service account satisfies `AdminOnly` "for - team-creation/listing purposes" — that was never true of the shipped code - (`TEAM-LOOKUP.md` caught the listing half; the creation half was always a - separate, bespoke check in `handleCreateTeam`, not `AdminOnly` itself) and - is not being made true now. Don't widen `AdminOnly`: every time this has - come up, the fix has been a narrower, purpose-built capability instead - (`?name=` lookups for teams and service accounts; now - `terdut-operator`'s own invite-minting feature for the one real gap this - boundary left — how a human ever gets a first login on a no-OIDC, - operator-managed install. See the bottom of "What this unblocks.") -- `Caller.IsInstanceServiceAccount()` — true only for an instance-scoped - service account, never for a human (including a human admin). - `handleCreateTeam` uses exactly this: a human creates a team by being a - human (and becomes its owner); an instance-scoped service account creates - one with no human owner at all. The two paths are not interchangeable, so - this predicate deliberately does not also admit a human admin. -- `Caller.Role(teamID)`/`TeamIDs()` — a human's real `team_members` rows, or - a team-scoped service account's single synthetic owner membership - (`serveAsServiceAccount`). This is what makes `requireTeamMember`/ - `requireTeamOwner` treat a team-scoped service account as owner-equivalent - for that one team, with no separate branch needed in either function. -- `Caller.ServiceAccountID()` — used by `OperatorModeBlock` ("any service - account passes") and by `callerMayManageServiceAccount`'s self-rotation - check. -- `Caller.AsHuman()` — the accessor every handler that needs a real - `user_id` to act on behalf of must call and check, instead of reading a - user off context unconditionally. Before the `Caller` type existed, four - handlers did the latter and silently misbehaved for a service-account - caller: `handleMe` and `handleTestNotification` 500'd (a zero-value user id - that matches no row), `handleDismissOnboarding` silently no-op'd (`UPDATE - ... WHERE id = 0` affects nothing, still returns 204), and - `handleCreateInvite` wrote that same zero value into `invites.created_by` - — a real foreign-key violation, not just a wrong answer, since that column - is nullable but was never passed as `nil`. All four now call `AsHuman()` - and return an explicit 403 ("this endpoint is for human accounts only") - or, for the invite case, leave `created_by` `NULL` the same way - `handleCreateServiceAccount` already did for the analogous situation. - -**Team scope is owner-equivalent for every `requireTeamOwner` endpoint, -membership and invites included — by design, not by an unclosed gap.** An -earlier version of this document flagged this as "acknowledged rather than -closed," kept in check only by the social convention that nobody *builds* -automation against those two routes. That convention is retired: -`terdut-operator`'s `TerdutTeam` controller now mints and revokes its own -team's invite link through exactly this capability (its existing -team-scoped credential, `POST`/`DELETE /api/teams/{teamID}/invites`), which -is the real fix for the human-onboarding gap below — not a narrower -carve-out of this capability. `service_accounts_test.go`'s -`TestServiceAccount_TeamScopeManagesItsOwnInvites` pins it. - -**A team-scoped account can also mint another service account scoped to its -own team** (`handleCreateServiceAccount`'s `callerOwnsTeam` branch, which a -team-scoped caller already satisfies for its own team via the synthetic -membership above). Kept, not restricted, for the same reason: a team-scoped -credential is that team's owner's reach, full stop — carving this one -capability out while leaving membership/invites alone would be an arbitrary -asymmetry. Pinned by -`TestServiceAccount_TeamScopeCanMintAnotherAccountForItsOwnTeam`. - -**`callerMayManageServiceAccount` gained the one load-bearing fix this -redesign exists for:** an instance-scoped service account may manage -(mint/revoke a key on) *any* team-scoped account, not only one admin, that -team's human owner, or the account itself. `handleCreateServiceAccount` -already let an instance-scoped caller *create* a team-scoped account for -any team; this closes the gap where adopting or rotating one it didn't just -create in the same call — exactly `terdut-operator`'s documented -adopt-on-409 crash-window recovery (its own `DESIGN.md` §5) — 403'd forever -instead of succeeding (`terdut-operator#3`). Pinned by -`TestServiceAccount_InstanceScopeAdoptsAnExistingTeamScopedAccountsKey`. - -Anywhere identity is recorded for a human (incident timeline -`acknowledged_by`/`assigned_to`, audit-relevant fields), a service-account -caller is still coerced into a bare `user_id` of `0` today — `Caller`'s new -`Identity()` accessor exists for exactly this follow-up, but wiring it in -needs a schema migration (an actor-attribution column distinct from -`user_id`) and is deliberately out of scope here. Tracked separately, not by -this document. - -## What this unblocks - -Directly resolves `terdut-operator` DESIGN.md §6's two broken assumptions: -1. **Bootstrap becomes single-purpose again.** `/api/bootstrap` mints exactly - the founding human admin, once. The operator's actual first-reconcile flow: - call `/api/bootstrap` only on a genuinely empty install; otherwise (or - immediately after, if it won the bootstrap race) call - `GET /api/service-accounts?name=terdut-operator`, and `POST` one if it - doesn't exist yet. From then on the operator never touches `/api/bootstrap` - again. -2. **Rotation becomes real.** `POST /api/service-accounts/{id}/keys` + revoke the - old one — no destructive DB-level workaround, no re-triggering a single-shot - endpoint that can't fire twice. -3. **Cross-namespace credential mirroring is no longer needed at all.** - `terdut-operator`'s current design holds every credential — instance- and - team-scoped alike — privately in the operator's own namespace, never in - the namespace of the CR each one authenticates for; reconciliation happens - entirely inside the operator's controller loop, so no CR owner ever needs - read access to a terdut-server credential regardless of same- or - cross-namespace `serverRef`. Team scoping is still what bounds the blast - radius of any individual credential: a leaked team-scoped key exposes - exactly one team's resources, never the whole server, which is what makes - holding many credentials in one place (the operator's namespace) an - acceptable trade rather than reintroducing the mirrored design's - server-admin-equivalent-everywhere problem. -4. **A human can get a first login on a no-OIDC, operator-managed install — - without ever touching `AdminOnly` or `/api/admin/settings`.** This was - filed as `terdut-server#23` ("no API path to create a human login after - bootstrap") and diagnosed, at the time, as this server needing to let a - service account through `AdminOnly`. It doesn't: the fix lives entirely - in `terdut-operator`, because a team-scoped credential was *already* - owner-equivalent for `POST /api/teams/{teamID}/invites`, and invite - redemption (`POST /api/signup` with an `invite` token) bypasses - `signup_mode` entirely — `terdut-operator` just never grew a feature to - use either fact. Its `TerdutTeam` controller now mints and surfaces one - via its own existing team-scoped credential (`spec.invite`, - `status.inviteSecretRef`, see that repo's own docs), so a human joins a - CRD-managed team by a real invite link, the same way anyone else would. - `terdut-server#23` is closed with this note once that feature ships — its - named routes stay human-only, correctly, not a gap. - -## Suggested sequencing - -Land this before `terdut-operator` implements any bootstrap/credential-handling -code — that code would otherwise be written against the current one-shot, -user-only credential model as a known-temporary workaround, which is wasted -effort on a repo that currently has zero implementation to begin with. +Where a service account acts on an incident (acknowledge, resolve), the +timeline and `acknowledged_by` record it through parallel `*_service_account_id` +columns, never as a user. diff --git a/TEAM-LOOKUP.md b/TEAM-LOOKUP.md deleted file mode 100644 index 67419cd..0000000 --- a/TEAM-LOOKUP.md +++ /dev/null @@ -1,73 +0,0 @@ -# Team lookup for service accounts: closing terdut-operator's create-path crash window - -This is a design note for a feature, not an implementation plan — same posture as -`SERVICE-ACCOUNTS.md`, and raised for the same reason: `terdut-operator`'s `TerdutTeam` -controller (ROADMAP.md Stage 2, a separate repo, no shared code) hit a gap this server has -no answer for yet. - -## The problem - -`POST /api/teams` (`handleCreateTeam`, confirmed against `internal/api/teams.go`) lets an -instance-scoped service account create a team — it has its own explicit -`isInstanceServiceAccount(...)` branch alongside the human-user path, not gated by -`AdminOnly`. If that call succeeds server-side but the caller (`TerdutTeam`'s controller) -crashes before persisting the resulting team ID locally, a retry's `POST` 409s on the name's -unique constraint (confirmed: the `isUniqueViolation` branch in the same handler). - -Recovering from that 409 means looking the team up by name, and nothing today permits that -for a service account: - -- `GET /api/teams` (`handleListTeams`) answers "what teams does the *caller* belong to", via - a `team_members` join keyed on `userFromContext`'s `caller.ID` — confirmed against source. - A service account is never a member of anything, so this always returns empty for one, - regardless of what exists. -- `GET /api/admin/teams` (`handleAdminListTeams`) is gated by `AdminOnly`, and `AdminOnly`'s - actual code (`internal/api/middleware.go`) checks only `userFromContext(...).IsAdmin` — no - branch for a service account at all, confirmed against source. This contradicts - `SERVICE-ACCOUNTS.md`'s own text, which claims "an instance-scoped [service account - satisfies] `AdminOnly` for team-creation/listing purposes" — that claim doesn't match this - endpoint's actual, shipped code. (Team *creation* is fine: `handleCreateTeam` isn't behind - `AdminOnly` at all, it has its own check. Only the listing half of that sentence is wrong.) - -This is exactly the shape of gap `SERVICE-ACCOUNTS.md`'s own `GET /api/service-accounts?name=` -closed for service accounts themselves (confirmed: that endpoint's own comment — -"the name lookup is open to any authenticated caller... what lets a service account find its -own account on the 403 that follows a second POST"). Teams never got the equivalent, because -nothing needed it until an operator started creating them unattended. - -## Goals - -- A service-account-accessible way to look up one team by exact name, mirroring - `GET /api/service-accounts?name=` as closely as possible — same shape, same reasoning, - same low sensitivity of what it discloses. -- No change to today's behavior for an empty/no-name request. - -## Proposed shape - -Extend `GET /api/teams` itself, the same way `handleListServiceAccounts` already branches on -a `?name=` query param, rather than adding a new route: - -- `name` unset (today's behavior, unchanged): the caller's own teams, via `team_members`. -- `name=` set: look up that one team by exact name — a one-or-zero-length array, not - an error on no match, mirroring `GET /api/service-accounts?name=`'s own response shape and - status codes exactly. Deliberately **not** gated by `isInstanceServiceAccount` or - `AdminOnly`: a human caller who's already a member sees this same information in their own - team list regardless, and a non-member learning only that a name is taken — not who's in - the team, not any of its data — is the same low-sensitivity disclosure - `GET /api/service-accounts?name=` already accepts for service-account names. - -## What this unblocks - -Directly resolves the crash-window gap in `terdut-operator`'s `TerdutTeam` controller: on a -409 from `POST /api/teams`, `GET /api/teams?name=` — authenticated with the -same instance-scoped credential that just got the 409 — finds the id, and the controller -proceeds as if its own create had returned it directly. The same adopt-on-409 pattern already -proven for service accounts (that repo's `DESIGN.md` §6 point 1, §5's general rule), not a -new one. - -## Suggested sequencing - -Land this before `TerdutTeam`'s create path is implemented — the same reasoning -`SERVICE-ACCOUNTS.md` gave for its own sequencing: writing that code against today's gap as a -"known-temporary workaround" is wasted effort when the fix is this small and this -well-precedented. diff --git a/charts/terdut-server/templates/deployment.yaml b/charts/terdut-server/templates/deployment.yaml index 15557d6..49acee1 100644 --- a/charts/terdut-server/templates/deployment.yaml +++ b/charts/terdut-server/templates/deployment.yaml @@ -95,12 +95,6 @@ spec: value: "{{ .Values.sweeper.staleAfter }}" - name: TERDUT_ARCHIVE_AFTER value: "{{ .Values.sweeper.archiveAfter }}" - - name: TERDUT_DEADMAN_MATCHERS - value: "{{ .Values.deadman.matchers }}" - - name: TERDUT_DEADMAN_TIMEOUT - value: "{{ .Values.deadman.timeout }}" - - name: TERDUT_DEADMAN_SEVERITY - value: "{{ .Values.deadman.severity }}" {{- if .Values.notify.ntfyUrl }} - name: TERDUT_NTFY_URL value: "{{ .Values.notify.ntfyUrl }}" @@ -122,6 +116,8 @@ spec: value: "{{ .Values.notify.publicUrl | default (printf "https://%s" .Values.networking.hostname) }}" - name: TERDUT_PASSWORD_LOGIN value: {{ .Values.passwordLogin | quote }} + - name: TERDUT_TRUSTED_PROXIES + value: {{ .Values.trustedProxies | quote }} - name: TERDUT_OPERATOR_MODE value: {{ .Values.operatorMode | quote }} {{- if .Values.oidc.enabled }} diff --git a/charts/terdut-server/values.yaml b/charts/terdut-server/values.yaml index 64d3de4..e084c23 100644 --- a/charts/terdut-server/values.yaml +++ b/charts/terdut-server/values.yaml @@ -72,43 +72,10 @@ sweeper: # How long a resolved alert stays in the default list before auto-archiving. archiveAfter: 168h -# Alerts treated as dead man's switches: receiving one opens no incident, and -# the absence of one does. The Watchdog alert kube-prometheus-stack ships is -# exactly this — an always-firing alert whose only value is something noticing -# when it stops. -deadman: - # Which alerts to treat as heartbeats. ";" separates matchers, "," separates - # the label conditions within one, "=" is exact equality. Every matcher must - # name an alertname: - # alertname=Watchdog,cluster=prod; alertname=EdgeHeartbeat - # Each distinct label set is watched independently, so two clusters sending - # the same alertname are two switches and a live one cannot mask a dead one. - matchers: "alertname=Watchdog" - # How long a heartbeat may go unheard before its switch is declared dead. - # - # This must be SHORTER than the Alertmanager repeat_interval of the route - # carrying the heartbeat — the opposite of sweeper.staleAfter. The default - # repeat_interval of 4h (12h in many setups) makes for a useless dead man's - # switch, so give the heartbeat a route of its own: - # - # - matchers: [ 'alertname = "Watchdog"' ] - # receiver: terdut - # group_wait: 0s - # group_interval: 1m - # repeat_interval: 1m - # - # That delivers every 2m rather than every 1m: a group is only reconsidered - # each group_interval, and at exactly one elapsed interval repeat_interval has - # not quite passed, so equal values give 2x. Fine against 15m; use - # group_interval: 30s if you want a true 1m. - # - # Set to 0 to disable dead man's switch handling entirely. - timeout: 15m - # Severity a dead man's switch incident opens at. These incidents have no - # member alerts to derive one from, and the heartbeat's own severity label is - # meaningless — Watchdog ships as "none". Only "critical" maps to the ntfy - # priority that overrides a phone's quiet hours. - severity: critical +# How many reverse proxies in front of the server append to X-Forwarded-For. +# The per-address login/sign-up rate limits take the client address that many +# entries from the right. 0 ignores the header. +trustedProxies: 1 notify: # ntfy server that push notifications are published to, e.g. @@ -190,10 +157,8 @@ oidc: # Hard ceiling on a session made by a single sign-on login. sessionMaxAge: 12h -# Backups are no longer this chart's business. The SQLite database lived on a PVC -# beside the app, so it needed a sidecar with a sqlite3 module for k8up to exec a -# dump in; Postgres is backed up where it runs, through a k8up.io/backupcommand -# pg_dump annotation on the database pod itself. +# Backups are not this chart's business: Postgres is backed up where it runs, +# through a k8up.io/backupcommand pg_dump annotation on the database pod itself. bootstrap: enabled: true diff --git a/cmd/terdut/main.go b/cmd/terdut/main.go index 2997f5c..08d8661 100644 --- a/cmd/terdut/main.go +++ b/cmd/terdut/main.go @@ -39,21 +39,16 @@ func main() { RepeatEvery: cfg.NotifyRepeat, } - // Dead man's switches live per team now. The environment variables are the - // defaults a team starts from: every team without a configuration of its - // own gets one from them here, and an owner's later edit is never - // overwritten by a redeploy. - deadman := api.ParseDeadmanConfig(cfg.DeadmanMatchers, cfg.DeadmanTimeout, cfg.DeadmanSeverity) - if err := api.SeedDeadmanConfigs(context.Background(), database, deadman); err != nil { - log.Fatalf("seed dead man's switch defaults: %v", err) - } - // The behaviour knobs move into the database on first start, after which an // administrator owns them and a redeploy leaves them alone. if err := api.SeedSettings(context.Background(), database, cfg); err != nil { log.Fatalf("seed settings: %v", err) } + if err := api.SeedOperatorKey(context.Background(), database, cfg.OperatorKey); err != nil { + log.Fatalf("%v", err) + } + router := api.NewRouter(database, notify, cfg, version) srv := &http.Server{ diff --git a/docs/README.md b/docs/README.md new file mode 100644 index 0000000..c59199e --- /dev/null +++ b/docs/README.md @@ -0,0 +1,21 @@ +# Terminal Duty documentation + +The [README](../README.md) is the short tour. These pages hold the detail. + +**Running it** +- [Deployment](./deployment.md): Docker, the Helm chart, the database, backups, and the operator. +- [Configuration](./configuration.md): environment variables and settings. +- [Single sign-on](./single-sign-on.md): OIDC, group mapping, the terminal device flow. + +**Using it** +- [The web UI](./web-ui.md): sessions, the Team and Admin tabs. +- [Alertmanager configuration](./alertmanager.md): routes, integration keys and webhooks. +- [Alerts and incidents](./incidents.md): correlation, lifecycle, on-call assignment, stale-alert expiry. +- [Push notifications](./notifications.md): ntfy pages and acknowledging from them. +- [Escalation](./escalation.md): ladders, repeats and the fallback topic. +- [Dead man's switches](./dead-mans-switch.md): noticing that alerts stopped arriving. + +**Integrating and contributing** +- [API reference](./api.md): every endpoint, authentication and error shape. +- [Service accounts](../SERVICE-ACCOUNTS.md): non-human credentials for automation. +- [Development and releasing](./development.md): tests, the CI gate, the release pipeline. diff --git a/docs/alertmanager.md b/docs/alertmanager.md new file mode 100644 index 0000000..a153e72 --- /dev/null +++ b/docs/alertmanager.md @@ -0,0 +1,62 @@ +# Alertmanager configuration + +_Pointing Alertmanager at the server._ Back to the [README](../README.md) and the [documentation index](./README.md). + +Alerts arrive on a team's **integration key**, which says both that the sender +may post and which team the alerts belong to. Mint one as an owner of the team: + +```bash +curl -X POST https://terdut.example.com/api/teams/1/integrations \ + -H "Authorization: Bearer $TERDUT_API_KEY" \ + -H 'Content-Type: application/json' \ + -d '{"name":"prod alertmanager"}' +``` + +The response carries the key and the full URL **once**; only a SHA-256 hash is +stored. Put it in your `alertmanager.yml`: + +```yaml +receivers: + - name: terdut + webhook_configs: + - url: http://terdut-server:8080/api/integrations//alertmanager + send_resolved: true + +route: + receiver: terdut +``` + +The whole URL is a credential, so treat it like one. Alertmanager 0.26 and +later can read it from a file with `url_file:` instead, which keeps it out of +your configuration repository: + +```yaml + - url_file: /etc/alertmanager/secrets/terdut-webhook-url/url + send_resolved: true +``` + +The webhook endpoint requires no authentication. + +If you use the [dead man's switch](./dead-mans-switch.md) — and the default configuration does — give +the heartbeat a route of its own, because the deadline is only as tight as the interval feeding it: + +```yaml +route: + receiver: terdut + repeat_interval: 4h + routes: + - matchers: [ 'alertname = "Watchdog"' ] + receiver: terdut + group_wait: 0s + group_interval: 1m + repeat_interval: 1m +``` + +That delivers a heartbeat every **2 minutes**, not every minute. Alertmanager only reconsiders a +group every `group_interval`, and at exactly one elapsed interval `repeat_interval` has not *quite* +passed, so the send slips to the next tick — equal values give 2×. Two minutes against the 15 minute +default is seven heartbeats per window, which is the point; use `group_interval: 30s` if you want +the numbers to mean what they say. + +kube-prometheus-stack users get the `Watchdog` alert (`expr: vector(1)`) for free; it just needs +routing to terdut rather than to `null`. diff --git a/docs/api.md b/docs/api.md new file mode 100644 index 0000000..6af26a0 --- /dev/null +++ b/docs/api.md @@ -0,0 +1,436 @@ +# API reference + +_The REST API._ Back to the [README](../README.md) and the [documentation index](./README.md). + +## Authentication + +All endpoints except `/api/bootstrap`, `/api/integrations/{key}/alertmanager`, +`/api/notify/ack/{token}`, `/api/login`, `/api/logout`, `/api/auth/config`, +`/api/version`, `/api/oidc/login`, `/api/oidc/callback`, `/api/oidc/device` and +`/api/oidc/device/token` require either an API key: + +``` +Authorization: Bearer +``` + +or the web UI's session cookie. A request that carries an `Authorization` header +is judged on that header alone. + +Two kinds of user exist. An **administrator** manages accounts: creating and +deleting users, setting anybody's password, minting keys for anybody, and +granting the flag itself. Everybody else works incidents — acknowledging, +assigning, snoozing, resolving, noting — and manages their own account and +nobody else's. An API key carries exactly the rights of the user it belongs to. + +A third principal, the **service account**, exists for automation (a +Kubernetes operator, most likely) that needs to manage teams, escalation +policies, dead man's switches and integrations without impersonating a human. +It is not a user — it never signs in, never appears in a team's member list, +and never holds the administrator flag — and its key is prefixed `tdsa_` so it +reads as one at a glance in a log line. See [Service accounts](#service-accounts). + +**Getting an account.** The first one comes from `/api/bootstrap`. After that +it depends on `signup_mode`, an administrator setting: + +- `invite_only` (the default) — a team owner mints a link with + `POST /api/teams/{teamID}/invites`, and the person who opens it picks a + username and password and lands in that team with the role the link carries. + Links are single-use unless told otherwise, expire after seven days, and can + be revoked before that. +- `open` — anybody who can reach the server can create an account, and must + name a team, which they then own. + +Invites are **links, not email**: this server has no SMTP, and adding it to send +one message would be a subsystem to run, secure and monitor. Send the link +however you already talk to the person. + +A domain-restricted third mode was considered and dropped: with no email there +is nothing to verify an address against, so it would only check the domain of a +string somebody typed. + +The first user, from `/api/bootstrap`, is an administrator. Users created +afterwards are not, until an administrator says so. An install always keeps at +least one: the last administrator can be neither deleted nor demoted, and +nobody can delete or demote themselves. + +Endpoints that require the flag answer `403` with +`{"error":"administrator access required"}`. + +**Teams** are the unit of tenancy, and are a separate axis from the administrator +flag. A team owns its incidents, alerts, schedule and integrations, and a user +sees exactly the teams they belong to. Within a team an **owner** configures it +(schedule, integrations, membership) and a **member** works its incidents. + +An administrator crosses that line in one direction only. They **configure any +team** without being in it — every owner-only endpoint accepts the flag, because +otherwise a team whose last owner left could never be repaired. They do **not +read any team**: the queue, the alerts and the incidents are filtered by real +membership, so an administrator sees a team's work only by joining it, which is +a membership change and shows up as one. Administration is about accounts and +the shape of a team, not about reading other people's incidents. + +Anything belonging to a team you are not in answers `404`, not `403`: whether an +incident exists is itself something only its team should learn. + +**Operator mode** (`TERDUT_OPERATOR_MODE`, see [Configuration](./configuration.md#configuration)) +declares this install gitops-managed. When it is on, a session or a user's own +API key gets `403 {"error": "...", "reason": "operator_managed"}` on every +write this page marks **owner**-gated under Teams below (creating, renaming +or deleting a team; its OIDC group binding; its escalation ladder; its dead +man's switches; its integrations) — a service account's writes are unaffected. +Team membership and invites are deliberately excluded: they are never +gitops-managed, in operator mode or out of it. `GET /api/auth/config` reports +`operator_mode` so a client can grey those sections out before a write is ever +attempted. + +| Method | Path | Description | +|---|---|---| +| `GET` | `/api/auth/config` | How to sign in: `{"password_login", "oidc": {"enabled","name"}, "device_login", "operator_mode"}`. No session needed | +| `GET` | `/api/version` | `{"version"}` — this build's version string. No session needed, the same as `/healthz` | +| `POST` | `/api/login` | `{"username","password"}` → sets the session cookie, returns `{user, has_password}`. `429` after too many failures; `403` when `TERDUT_PASSWORD_LOGIN=false` | +| `GET` | `/api/oidc/login` | Starts a single sign-on sign-in: redirects the browser to the provider. `?next=/path` is where to land afterwards; only a path on this server is honoured. Only exists when SSO is configured | +| `POST` | `/api/oidc/device` | Starts a device login: returns `{device_code, user_code, verification_url, interval, expires_in}`. Only exists when SSO is configured | +| `POST` | `/api/oidc/device/token` | `{"device_code"}` → `202 {"status":"pending"}`, then `200` with the session cookie once approved (once only). `410` with `{"error":"expired"}` or `{"error":"denied"}`; `429 {"error":"slow_down"}` if polled faster than `interval` | +| `POST` | `/api/oidc/device/approve` | **session** — `{"user_code"}`. Approves a pending device login as the caller. `403` for an API key; `404` for an unknown, expired or already decided code | +| `POST` | `/api/oidc/device/deny` | **session** — `{"user_code"}`. Refuses it | +| `GET` | `/api/oidc/callback` | Where the provider sends the browser back. Sets the session cookie and redirects to `/`, or to `/?sso_error=` — one of `denied`, `expired`, `failed`, `unavailable`, `not_allowed`, `no_email`, `email_conflict`, `disabled`, `not_bootstrapped` (no user exists on this install yet — sign in again once something has called `/api/bootstrap`) | +| `POST` | `/api/logout` | Ends the session and clears the cookie | +| `GET` | `/api/me` | The caller: `{user, has_password}` | + +## Users + +| Method | Path | Description | +|---|---|---| +**admin** marks an endpoint that requires the administrator flag; **self or +admin** marks one you may use on your own account and an administrator may use +on anybody's. + +| Method | Path | Who | Description | +|---|---|---|---| +| `GET` | `/api/signup` | — | Whether sign-up is open, and whether `?invite=` is usable. No session needed: the caller has no account yet | +| `POST` | `/api/signup` | — | Create an account `{"username","email","password","invite"?,"team_name"?}` and sign in. `403` without a usable invite when the mode is invite-only | +| `POST` | `/api/bootstrap` | — | Create first user + API key `{"username","email","password"?}` (only works on empty DB). The user is an administrator | +| `GET` | `/api/users` | any | List users. Open to everybody: the queue's assignment control and the schedule both have to name people | +| `GET` | `/api/users/{id}/teams` | self or admin | The teams that user is in, each with their role. `/api/teams` is always about the caller; this one answers it about somebody else, for the admin page's per-user view. `404` for a user who does not exist, so "no teams" and "no such person" are distinguishable | +| `POST` | `/api/users` | **admin** | Create user `{"username","email"}`. Not an administrator | +| `DELETE` | `/api/users/{id}` | **admin** | Delete user (cascades to keys). `409` for yourself or the last administrator | +| `PUT` | `/api/users/{id}/admin` | **admin** | Grant or revoke the administrator flag `{"is_admin"}`. `409` for yourself, the last administrator, or an administrator granted by single sign-on | +| `PUT` | `/api/users/{id}/disabled` | **admin** | Take an account out of use, or put it back `{"disabled"}`. `409` for yourself or the last administrator | +| `PUT` | `/api/users/{id}/notify` | self or admin | Set push notification target `{"ntfy_topic"}` — empty string clears it | +| `PUT` | `/api/users/{id}/password` | self or admin | Set web UI password `{"password","current_password"}`. `current_password` is required only when changing your own existing password. Ends the user's other sessions | +| `POST` | `/api/users/{id}/api-keys` | self or admin | Issue API key `{"name"}` — key shown once | +| `DELETE` | `/api/users/{id}/api-keys/{keyID}` | self or admin | Revoke API key | + +## Administration + +| Method | Path | Who | Description | +|---|---|---|---| +| `GET` | `/api/admin/teams` | **admin** | Every team on the server, with its member and open-incident counts. `/api/teams` answers "what am I in"; this answers "what is there" | +| `GET` | `/api/admin/teams/{teamID}` | **admin** | One team and who is in it: `{"team", "members"}`. `404` for a team that does not exist. `GET /api/teams/{teamID}/members` is **member**-only and still `404`s an administrator from outside the team — reading a team's shape and reading its work are different questions, so they are different endpoints | +| `GET` | `/api/admin/settings` | **admin** | The editable settings with their bounds, plus the environment-configured ones, read-only. Never credentials | +| `PUT` | `/api/admin/settings` | **admin** | Change one or more `{"key": seconds}`, or `{"signup_mode": "open"\|"invite_only"}`. `400` for an unknown key or a value outside its bounds | + +## Service accounts + +A service account is a scoped, non-human credential for automation — not a +`users` row, so it never signs in, is never a team member, and never carries +the administrator flag. Two scopes: + +- **instance** — the same reach system administration has over teams: create + one, and mint a **team**-scoped account against any of them. There is no + cap on how many instance-scoped accounts exist, but ordinarily there is one, + belonging to whatever is provisioning this install end to end. +- **team** — owner-equivalent for that one team, and nothing else: every + **owner**-gated endpoint under [Teams](#teams), membership and invites + included. Nothing narrower is enforced server-side; what actually keeps + membership out of automation's hands is that no operator built against this + scope should ever call those two endpoints — see + [operator mode](#authentication) and [`SERVICE-ACCOUNTS.md`](../SERVICE-ACCOUNTS.md)'s note on this. + +A key is shown once, at creation or rotation, and only its hash is stored — +the same handling as a user's API key. Losing it means minting a new one; +there is no way to recover a raw key from the server. + +| Method | Path | Who | Description | +|---|---|---|---| +| `GET` | `/api/service-accounts` | **admin** | Every service account. Pass `?name=` instead to look one up by its exact name — open to **any** authenticated caller (human or service account), since it returns no key material and is how an account finds its own id | +| `POST` | `/api/service-accounts` | owner\* | Create one and mint its first key `{"name","scope","team_id"?}` (`team_id` required for `scope:"team"`, absent for `scope:"instance"`). Returns `{"service_account", "key"}` — `key.key` shown once | +| `POST` | `/api/service-accounts/{id}/keys` | owner\* | Mint an additional key `{"name"}` — rotation without recreating the account. Shown once | +| `DELETE` | `/api/service-accounts/{id}/keys/{keyID}` | owner\* | Revoke one key | + +\* For an **instance**-scoped account: a system administrator only. For a +**team**-scoped account: a system administrator, that team's own human owner, +an instance-scoped service account (minting a narrower credential for a team +it just created), or — for the two key endpoints only — the account rotating +or revoking its own key, which is not a privilege escalation, the same +reasoning a user's own API keys rest on. + +## Alert ingestion + +Alerts arrive on a team's integration key. The key is both the credential and the +routing: it says that the sender may post, and which team the alerts belong to. +Create one with `POST /api/teams/{teamID}/integrations`, which returns the key +and the full URL once and stores only a SHA-256 hash. + +| Method | Path | Description | +|---|---|---| +| `POST` | `/api/integrations/{key}/alertmanager` | Alertmanager v4 webhook receiver for the key's team. `401` for an unknown key | + +This is the only way in. The pre-teams `POST /api/alertmanager/webhook` took no +credential at all — anything able to reach the port could open an incident — +and was removed in v0.13.0 once senders had moved onto keys. + +## Teams + +**owner** below means an owner of that team, a system administrator (who +passes every one of these without being a member), or that team's own +team-scoped [service account](#service-accounts) — including membership and +invites, technically, though no automation this scope was designed for +(a Kubernetes operator's CRDs, see [`SERVICE-ACCOUNTS.md`](../SERVICE-ACCOUNTS.md)) ever models team +membership or would call those two. See [Authentication](#authentication). +**member** means membership and nothing else: an administrator who is not in +the team gets the same `404` as anybody else. + +| Method | Path | Who | Description | +|---|---|---|---| +| `GET` | `/api/teams` | any | The caller's own teams, each with their role | +| `POST` | `/api/teams` | any | Create a team `{"name"}`; a human creator becomes its first owner. An instance-scoped [service account](#service-accounts) may also create one, and it gets no owner at all — expected for a team an operator is about to hand a team-scoped credential to, not an orphaned team a human made | +| `PUT` | `/api/teams/{teamID}` | **owner** | Rename it `{"name"}`. `409` if the name is taken | +| `DELETE` | `/api/teams/{teamID}` | **owner** | Delete a team and everything under it. `409` while it has open incidents | +| `GET` | `/api/teams/{teamID}/members` | member | Who is in the team, with `status` (`oncall` if the rota has them today, `unpageable` when a page to them would go nowhere — even if they are on call — else `reachable`), `on_call`, `next_shift` (first rota day after today), `pageable` and `problem` (`has no ntfy topic` / `account is disabled`; never the topic itself) and `last_active_at` (their newest session or API-key use). Every member sees the same list | +| `POST` | `/api/teams/{teamID}/members` | **owner** | Add a member, or change their role `{"user_id","role"}`. `409` when it would demote the last owner, or the membership is managed by single sign-on | +| `DELETE` | `/api/teams/{teamID}/members/{userID}` | **owner** | Remove a member. `409` for the last owner, or a membership managed by single sign-on | +| `GET` | `/api/teams/{teamID}/oidc-groups` | member | Which groups control this team's membership: `{"member_group","owner_group"}`. An empty string means no group grants that role here | +| `PUT` | `/api/teams/{teamID}/oidc-groups` | **owner** | Set them. An empty string clears a binding | +| `GET` | `/api/teams/{teamID}/integrations` | member | List integrations. Never returns keys. Each carries `status` (`active` if its key posted within 24h, `quiet` if it has but not lately, `never`), `last_used_at` (last webhook, usable or not), `last_alert_at` (when an alert last arrived on it) and `alerts_24h` (distinct alerts it refreshed in the last day). Alerts delivered before the source was recorded (migration 010) have none, so the last two fill in as Alertmanager re-sends them | +| `PATCH` | `/api/teams/{teamID}/integrations/{integrationID}` | **owner** | Rename `{"name"}`. The key does not change | +| `POST` | `/api/teams/{teamID}/integrations` | **owner** | Mint an integration `{"name","kind"}` — key and URL shown once | +| `DELETE` | `/api/teams/{teamID}/integrations/{integrationID}` | **owner** | Revoke an integration. Alerts it delivered stay, unattributed | +| `GET` | `/api/teams/{teamID}/invites` | **owner** | The team's invite links, with their uses and expiry. Never the tokens | +| `POST` | `/api/teams/{teamID}/invites` | **owner** | Mint one `{"role","max_uses"}` — the full URL is returned once | +| `DELETE` | `/api/teams/{teamID}/invites/{inviteID}` | **owner** | Revoke a link before it expires | +| `GET` | `/api/teams/{teamID}/escalation` | member | The team's [escalation ladder](./escalation.md#escalation) `{repeat_count, fallback_topic, levels[], last_escalated_at?, last_escalated_incident_id?}`. Empty levels means the team has none. Each level also carries `status` (`ready`, `escalating` when an unanswered incident has climbed to it, `unreachable` when nobody on it could be woken), `waiting` (ids of the open incidents on it) and, per target, `username` (who it means today — the person on call, for a rota target), `reachable` and `problem`. The extra fields are output only; `PUT` takes the plain shape | +| `PUT` | `/api/teams/{teamID}/escalation` | **owner** | Replace it wholesale. `400` for a level with no targets or no timeout — a rung that pages nobody is a silence with a number on it | +| `GET` | `/api/teams/{teamID}/deadman/switches` | member | The team's [dead man's switches](./dead-mans-switch.md), each `{id, name, matcher, timeout_seconds, severity, status, last_heartbeat_at, last_triggered_at, open_incident_id, sources[]}`. `status` is `healthy`, `dead` or `dormant`; `sources` has one entry per heartbeat fingerprint. Empty when the team watches nothing | +| `POST` | `/api/teams/{teamID}/deadman/switches` | **owner** | Add one: `{name?, matcher, timeout_seconds, severity?}`. `400` when the matcher names no `alertname` or holds several, or the timeout is not positive — a switch that silently watches nothing is the failure this feature exists to prevent | +| `PUT` | `/api/teams/{teamID}/deadman/switches/{switchID}` | **owner** | Replace one in place, same body and validation as create. Its id is unchanged — for an automated caller reconciling a spec change, unlike delete-and-recreate | +| `DELETE` | `/api/teams/{teamID}/deadman/switches/{switchID}` | **owner** | Stop watching. An incident it opened stays open. `404` for a switch of another team | + +## Notifications + +| Method | Path | Description | +|---|---|---| +| `POST` | `/api/notify/ack/{token}` | Acknowledge an incident from a push notification's Acknowledge button. No auth: the token in the path is the credential — one incident, one action, 24 hours, idempotent. Must stay publicly reachable | + +## Incidents + +| Method | Path | Description | +|---|---|---| +| `GET` | `/api/incidents` | List incidents. Filters: `?status=triggered\|acknowledged\|resolved`, `?severity=`, `?assigned_to=`, `?archived=true`, `?snoozed=true`, `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD`, `?sort=severity`, `?cluster=`, `?limit=` (default 50, max 500) | +| `GET` | `/api/incidents/clusters` | The distinct `cluster` values on the caller's incidents from the last 90 days, sorted (`?team_id=` narrows it). An empty array when nothing carries the label | +| `GET` | `/api/incidents/{id}` | Get single incident, with its alerts inline | +| `GET` | `/api/incidents/{id}/alerts` | Alerts under this incident | +| `GET` | `/api/incidents/{id}/timeline` | Full event history, chronological | +| `POST` | `/api/incidents/{id}/acknowledge` | Acknowledge (stamps authed user + time) | +| `DELETE` | `/api/incidents/{id}/acknowledge` | Clear acknowledgement, back to `triggered` | +| `POST` | `/api/incidents/{id}/resolve` | Close by hand — **terminal**, see above | +| `POST` | `/api/incidents/{id}/assign` | Reassign `{"user_id"}` | +| `POST` | `/api/incidents/{id}/snooze` | Hide until `{"until": RFC3339}` or `{"duration": "2h"}` | +| `DELETE` | `/api/incidents/{id}/snooze` | Un-snooze | +| `POST` | `/api/incidents/{id}/archive` | Archive (hides from the default list) | +| `DELETE` | `/api/incidents/{id}/archive` | Un-archive | +| `POST` | `/api/incidents/{id}/notes` | Add a note `{"content"}` | +| `DELETE` | `/api/incidents/{id}/notes/{eventID}` | Delete own note | + +With no `?status=` filter, `GET /api/incidents` returns **open** incidents only — +the queue an on-call person wants. Currently snoozed and archived incidents are +excluded unless asked for. Actions that only make sense on an open incident +return `409` once it is resolved. + +Notes are ordinary timeline events of type `note`; only they are deletable, and +only by their author. The rest of the timeline is a record of what happened. + +### The incident object + +| Field | Type | Notes | +|---|---|---| +| `id` | integer | Server-assigned | +| `group_key` | string | Alertmanager's `groupKey` — opaque, treat as an identifier | +| `title` | string | Rendered from `groupLabels` | +| `group_labels` | object | String→string, as sent by Alertmanager | +| `status` | string | `"triggered"`, `"acknowledged"` or `"resolved"` | +| `severity` | string | *optional* — high-water mark across the incident's alerts; never lowered | +| `triggered_at` | timestamp | When the incident opened | +| `acknowledged_by_id` / `acknowledged_by` / `acknowledged_at` | | *optional* — user id, username, time | +| `assigned_to_id` / `assigned_to` | | *optional* — user id, username | +| `snoozed_until` | timestamp | *optional* — a value in the past reads as not snoozed | +| `resolved_at` | timestamp | *optional* | +| `resolution_source` | string | *optional* — `"alerts"`, `"manual"` or `"recovered"` | +| `archived_at` | timestamp | *optional* | +| `alerts` | array | Only on `GET /api/incidents/{id}` | + +Treat `resolution_source` as an open set, as with the alert field of the same +name: degrade unknown values to "resolved, reason unknown". + +### The timeline event object + +| Field | Type | Notes | +|---|---|---| +| `id` | integer | | +| `incident_id` | integer | | +| `type` | string | See below — treat as an open set | +| `user_id` / `username` | | *optional* — absent when the server acted rather than a person | +| `alert_id` | integer | *optional* — the alert an `alert_added` / `alert_resolved` event refers to | +| `detail` | string | *optional* — the note body, the snooze deadline, etc. | +| `created_at` | timestamp | | + +Types written today: `triggered`, `alert_added`, `alert_resolved`, +`acknowledged`, `unacknowledged`, `assigned`, `archived`, `unarchived`, `snoozed`, +`unsnoozed`, `resolved`, `note`, `notified`, `notify_failed`, `deadman_silent`. On an +`assigned` event `user_id` is the **assignee**, not the actor; the actor is in +`actor_user_id`/`actor_username` or `actor_service_account_id`/`actor_service_account_name` +(absent on assignments made before they were recorded). New types may be added; render +unknown ones generically rather than dropping them. + +On `notified` and `notify_failed`, `detail` carries the notification kind +(`triggered` | `reminder` | `resolved`), and on a failure the reason after it. +`user_id` is who was paged — absent means the page went to the shared fallback +topic and so belongs to nobody. The topic itself is never written to the +timeline: it is a shared secret with the ntfy server, and every API key can read +this. + +## Alerts + +Alerts are read-only. Everything a person does happens on the incident. + +| Method | Path | Description | +|---|---|---| +| `GET` | `/api/alerts` | List alerts. Filters: `?status=firing\|resolved`, `?name=`, `?incident_id=`, `?archived=true`, `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD`, `?limit=` (default 50, max 500) | +| `GET` | `/api/alerts/{id}` | Get single alert | + +Archived alerts are hidden from `GET /api/alerts` unless `?archived=true` is +passed; alert archiving is automatic housekeeping by the sweeper, not a user +action. Resolved alerts carry `resolution_source`: `"alertmanager"` for a real +resolved webhook, `"expiry"` when the sweeper inferred it (see +[Stale alert expiry](./incidents.md#stale-alert-expiry)), `"deadman"` for a heartbeat declared +dead (see [Dead man's switch](./dead-mans-switch.md)). + +### The alert object + +Returned by `GET /api/alerts` (as an array) and `GET /api/alerts/{id}`. +Timestamps are RFC 3339 in UTC. Fields marked *optional* are omitted entirely +when unset, so clients must treat them as nullable. + +| Field | Type | Notes | +|---|---|---| +| `id` | integer | Server-assigned; stable for the life of the row | +| `fingerprint` | string | Alertmanager's fingerprint — the upsert key | +| `name` | string | From the `alertname` label | +| `status` | string | `"firing"` or `"resolved"` | +| `labels` | object | String→string, as sent by Alertmanager | +| `annotations` | object | String→string, as sent by Alertmanager | +| `starts_at` | timestamp | When the alert instance began, **per Prometheus** | +| `ends_at` | timestamp | *optional* — absent while no end is known | +| `generator_url` | string | Link back to the originating Prometheus | +| `received_at` | timestamp | When the server last accepted a webhook for this alert — see below | +| `incident_id` | integer | *optional* — the most recent incident this alert belongs to | +| `resolution_source` | string | *optional* — `"alertmanager"`, `"expiry"` or `"deadman"` | +| `archived_at` | timestamp | *optional* — set while archived | + +#### `received_at` is a liveness heartbeat + +`starts_at` comes from Prometheus and **never changes** for the lifetime of an +alert instance. It says when the problem began, not whether it is still +happening — an alert that started twelve days ago looks identical whether +Alertmanager refreshed it a minute ago or went silent a week ago. + +`received_at` is the field that answers "is this still live". It is set to the +server's clock on **every accepted webhook** for that fingerprint, including the +unchanged firing notifications Alertmanager re-sends every `repeat_interval`. +Clients may rely on this: + +- **A firing alert whose `received_at` is advancing is still being refreshed.** + Stale-dating it against `repeat_interval` is a valid liveness check, and it is + what the built-in sweeper does (see + [Stale alert expiry](./incidents.md#stale-alert-expiry)). +- **`received_at` tracks accepted payloads, not delivery attempts.** A retry + that describes an older instance than the stored one is discarded, and a + discarded payload does not move `received_at`. +- **It stops advancing once the alert resolves,** because Alertmanager stops + re-sending. On an alert resolved by the sweeper + (`"resolution_source": "expiry"`) it therefore marks the last time + Alertmanager was actually heard from, which is earlier than `ends_at`. + +`GET /api/alerts` is ordered by `received_at` descending — most recently +refreshed first — and the `?from=` / `?to=` filters on both the alert and stats +endpoints select on `received_at`, not `starts_at`. + +#### `resolution_source` says how much to trust `ends_at` + +An alert can leave the firing state two ways, and `resolution_source` records +which happened. Clients may rely on this: + +- **Absent while firing.** It is set only on resolve, and a re-fire under the + same fingerprint clears it again, so its presence always agrees with + `"status": "resolved"`. +- **`"alertmanager"` — a real resolved webhook arrived.** `ends_at` is the end + time Alertmanager reported. It is an observed value and can be displayed as + fact. +- **`"expiry"` — the sweeper inferred the resolve** because Alertmanager stopped + refreshing the alert (see [Stale alert expiry](./incidents.md#stale-alert-expiry)). Nothing + ever reported an end, so **`ends_at` is approximate**: it is either the stale + `endsAt` watermark from the last notification, or — when that notification + carried none — the time the sweep ran, which lags the last real contact by up + to `TERDUT_STALE_AFTER` plus a sweep interval. Treat it as "no later than", + not as when the problem stopped. + + On these alerts `received_at` is the more truthful signal: it marks the last + time Alertmanager was actually heard from. Surfacing the distinction is + worthwhile, since `"expiry"` can also mean the alert is still firing and the + notification path broke. + +- **`"deadman"` — a heartbeat was declared dead** (see + [Dead man's switch](./dead-mans-switch.md)). Like `"expiry"`, an inference from + silence rather than an observed end, so `ends_at` is approximate — but a much + tighter one, bounded by the switch's timeout. It is also the one resolution + a re-fire under the same `starts_at` can undo, since the switch coming back is + exactly the evidence that the inference was wrong. + +Treat the value as an open set and tolerate ones you do not recognise — new +sources may be added, and unknown values should degrade to "resolved, reason +unknown" rather than being rejected. + +## On-call schedule + +| Method | Path | Description | +|---|---|---| +Each team keeps its own rota, so two teams can have two different people on call +on the same day. The person taking a shift has to be in the team — paging +somebody who cannot open the incident is worse than paging nobody. + +| Method | Path | Who | Description | +|---|---|---|---| +| `POST` | `/api/teams/{teamID}/schedule` | **owner** | Assign user to dates `{"user_id", "dates":["YYYY-MM-DD",...], "replace"}` — all-or-nothing | +| `GET` | `/api/teams/{teamID}/schedule` | member | List entries. Filters: `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD` | +| `DELETE` | `/api/teams/{teamID}/schedule/{id}` | **owner** | Remove schedule entry | +| `GET` | `/api/schedule/current` | any | Who is on call today (UTC) in **every** team the caller is in — one entry per team, `[]` when nobody anywhere | + +## Statistics + +Every figure counts the caller's own teams only: a report that counted other +teams' incidents would leak their volume, and their alert names through the +top-alerts list, and would not be a number about the reader's work anyway. + +All stat endpoints accept optional `?from=YYYY-MM-DD` and `?to=YYYY-MM-DD`, and exclude archived rows to match the default list views. Alert stats filter on `received_at`; incident stats filter on `triggered_at`. + +| Method | Path | Description | +|---|---|---| +| `GET` | `/api/stats/incidents` | `{total, triggered, acknowledged, resolved, mtta_seconds, mttr_seconds}` | +| `GET` | `/api/stats/alerts` | `{total, firing, resolved}` counts | +| `GET` | `/api/stats/alerts/top` | Most frequent alert names. `?limit=` (default 10, max 100) | +| `GET` | `/api/stats/alerts/by-hour` | Count per hour-of-day (UTC), all 24 slots returned | +| `GET` | `/api/stats/alerts/by-day` | Count per day-of-week, all 7 slots with names returned | + +`mtta_seconds` (time to acknowledge) and `mttr_seconds` (time to resolve) are +averages over incidents that have actually been acknowledged or resolved, and are +**null** until there are any — null means "no data", not zero. diff --git a/docs/configuration.md b/docs/configuration.md new file mode 100644 index 0000000..b89a49a --- /dev/null +++ b/docs/configuration.md @@ -0,0 +1,49 @@ +# Configuration + +_Environment variables and settings._ Back to the [README](../README.md) and the [documentation index](./README.md). + +Two kinds of setting, split by who changes them and how often. + +**Where the server is plugged in** stays in the environment: the listen address, +the database DSN, the ntfy URL and token, the public URL. They are needed before +the database is open, and two of them are credentials. + +**How the server behaves** lives in the database and is edited by an +administrator in the web UI or through `PUT /api/admin/settings`, taking effect +on the next sweep rather than at the next restart. The variables below marked +**seed** are the value each of those starts from: written once, on first start, +and never overwritten afterwards — a redeploy cannot put a chart's default back +over an administrator's edit. + +| Variable | Default | Description | +|---|---|---| +| `TERDUT_ADDR` | `:8080` | TCP address to listen on | +| `TERDUT_DB_DSN` | — | **Required.** Postgres connection string, e.g. `postgres://terdut:secret@localhost:5432/terdut?sslmode=require` | +| `TERDUT_ARCHIVE_AFTER` | `168h` (7d) | **seed.** How long a resolved alert or incident stays in the default list before being auto-archived | +| `TERDUT_STALE_AFTER` | `6h` | **seed.** How long a firing alert may go without a refreshing webhook before it is treated as resolved — **must exceed your Alertmanager `repeat_interval`** | +| `TERDUT_NTFY_URL` | — | ntfy server to publish push notifications to. Empty disables notifications entirely | +| `TERDUT_NTFY_TOKEN` | — | Bearer token for an access-controlled ntfy | +| `TERDUT_NTFY_FALLBACK_TOPIC` | — | Topic used when nobody is on call | +| `TERDUT_PUBLIC_URL` | — | Base URL a phone uses to reach this server: the notification's link into the web UI, its Acknowledge button, and whether the session cookie is `Secure` | +| `TERDUT_NOTIFY_REPEAT` | `15m` | **seed.** How long an incident may sit unacknowledged before it is paged again. `0` notifies once and never repeats | +| `TERDUT_PASSWORD_LOGIN` | `true` | `false` refuses password login and password sign-up (`403`), leaving single sign-on the only way in. Refused at startup unless SSO is configured | +| `TERDUT_TRUSTED_PROXIES` | `1` | How many reverse proxies in front of the server append to `X-Forwarded-For`; the per-address rate limits use the entry that many hops from the right. `0` ignores the header | +| `TERDUT_OPERATOR_KEY` | — | At least 32 characters. When set, the instance-scoped service account `terdut-operator` is created if missing and its `seed` key replaced with this value at every start — how terdut-operator authenticates without a bootstrap handshake. An instance-scoped account acts as owner of every team (team configuration) but is not a member of any, so it reads no incidents | +| `TERDUT_OPERATOR_MODE` | `false` | Declares this install gitops-managed: a session's or a user's own API key's writes to teams, escalation policies, dead man's switches and integrations are refused (`403 reason:"operator_managed"`); a [service account](./api.md#service-accounts)'s are not. Team membership and the schedule stay editable regardless | +| `TERDUT_OIDC_ISSUER` | — | Turns single sign-on on. The provider's issuer URL; discovery is read from `/.well-known/openid-configuration`. See [Single sign-on](./single-sign-on.md#single-sign-on-oidc) | +| `TERDUT_OIDC_CLIENT_ID` / `TERDUT_OIDC_CLIENT_SECRET` | — | **Required with an issuer.** The confidential client registered at the provider. Keep the secret in a Secret, not in values | +| `TERDUT_OIDC_NAME` | `SSO` | What the sign-in button calls the provider | +| `TERDUT_OIDC_SCOPES` | `openid profile email` | Scopes requested, comma or space separated. Authentik puts `groups` behind `profile` | +| `TERDUT_OIDC_USERNAME_CLAIM` / `_EMAIL_CLAIM` / `_GROUPS_CLAIM` | `preferred_username` / `email` / `groups` | ID token claims read for the username, email and groups | +| `TERDUT_OIDC_TRUST_EMAIL` | `false` | Link a first sign-in to an existing local user by email even if the provider does not mark the address verified | +| `TERDUT_OIDC_ALLOWED_GROUPS` | — | Comma-separated. Only people in one of these may sign in. Empty admits everybody the provider authenticates | +| `TERDUT_OIDC_ADMIN_GROUP` | — | Members are system administrators | +| `TERDUT_OIDC_SESSION_MAX_AGE` | `12h` | Hard ceiling on a session made by an SSO sign-in | + +Durations use Go syntax (`30m`, `12h`, `168h`). An unparseable value falls back to the default. + +Note that `TERDUT_STALE_AFTER` and a dead man's switch timeout point in opposite directions. Staleness +is a generous grace period around a `repeat_interval` you do not control; a dead man's switch is a +deadline you set deliberately, and the heartbeat's route is configured to beat faster than it. + +In the Helm chart the two sweeper durations are set via `sweeper.staleAfter` and `sweeper.archiveAfter`, notifications via the `notify.*` values, single sign-on via `oidc.*` and `passwordLogin`, and operator mode via `operatorMode`. diff --git a/docs/dead-mans-switch.md b/docs/dead-mans-switch.md new file mode 100644 index 0000000..bd2a43b --- /dev/null +++ b/docs/dead-mans-switch.md @@ -0,0 +1,81 @@ +# Dead man's switches + +_Detecting that alerts have stopped arriving._ Back to the [README](../README.md) and the [documentation index](./README.md). + +Everything above assumes alerts arrive. If Prometheus stops evaluating, or +Alertmanager cannot reach this server, nothing arrives — and silence looks +exactly like everything being fine. A dead man's switch inverts the handling for +one designated alert so that silence is the signal: + +- **receiving** it opens no incident, and +- the **absence** of it does. + +kube-prometheus-stack already ships the alert for this. `Watchdog` is +`expr: vector(1)`, so it fires permanently and is re-sent forever; it is worth +nothing unless something downstream notices it stop. It is the usual first switch. + +**Switches belong to a team**, which decides which of its own alerts are +heartbeats and how long a silence has to last. Each **switch** is a row of its +own — a name, one matcher, a timeout and a severity — so switches in one team +can have different deadlines. An owner adds and removes them on **Team → +Switches**, which lists each with a status (**healthy**, **dead**, or +**dormant** until its first heartbeat), when it was last heard from, and when it +last opened an incident; a matcher that several clusters satisfy is broken down +per cluster. The API is `POST`/`DELETE /api/teams/{teamID}/deadman/switches`. A +missed heartbeat opens an incident in the team whose integration received it. +Removing a switch stops the watching; an incident it already opened stays open +until somebody resolves it. + +A new team watches nothing until its owner (or terdut-operator, from a +`TerdutTeam`) adds a switch: inheriting an install-wide heartbeat would page a +new team about a source it has never heard of. + +A matcher is a set of exact label conditions, one of which must be the +`alertname`, , one matcher per switch, `,` between the label conditions: + +``` +alertname=Watchdog,cluster=prod +``` + +**The unit of monitoring is the fingerprint, not the alert name.** Two clusters +sending the same `Watchdog` are two independent switches, so a healthy one can +never mask a dead one. + +## The lifecycle + +A switch is **dormant** until its first heartbeat arrives. A configured matcher +that has never been heard from opens nothing, so a fresh deploy or a restored +database does not page. It also means a matcher that never matches anything is +silently inert. + +Once armed, the sweeper declares it **dead** when either the heartbeat has not +been refreshed within the switch's `timeout_seconds`, or Alertmanager explicitly +resolved it — the sender saying the heartbeat stopped needs no further waiting. +That opens an incident at the switch's `severity`, assigned and paged like any +other, and marks the heartbeat alert `"resolution_source": "deadman"` so the +alert list stops claiming a dead switch is firing. + +It **recovers** when the heartbeat starts arriving again: the incident resolves +with `"resolution_source": "recovered"` and the all-clear goes to whoever was +paged. + +Resolving the incident by hand sticks, the same way it does for an alert-backed +one. While the switch stays silent nothing new opens — so a decommissioned +source is a one-time page rather than a nag. The switch **re-arms** on the next +heartbeat: come back and die again, and that is a new incident. + +## Two things to know + +A switch's timeout must be **shorter** than the `repeat_interval` of the +route carrying the heartbeat, which is the exact opposite of +`TERDUT_STALE_AFTER`. Inheriting a default `repeat_interval` of 4h gives you a +switch that takes four hours to notice anything, so give the heartbeat +[its own route](./alertmanager.md#alertmanager-configuration). Matched alerts are exempt from +stale-alert expiry — a heartbeat answers to its own timeout and nothing else. + +A dead man's switch incident has **no member alerts**: +`GET /api/incidents/{id}/alerts` returns an empty list. There is no alert +describing the problem, because the problem is that no alert arrived. What +happened is on the timeline instead, as a `deadman_silent` event carrying the age +of the last heartbeat, and the heartbeat's labels are on the incident's +`group_labels`. diff --git a/docs/deployment.md b/docs/deployment.md new file mode 100644 index 0000000..20e1ae9 --- /dev/null +++ b/docs/deployment.md @@ -0,0 +1,74 @@ +# Deployment + +_Running the server in a container and on Kubernetes with the Helm chart._ Back to the [README](../README.md) and the [documentation index](./README.md). + +## Docker + +```bash +docker build -t terdut-server . +docker run -p 8080:8080 \ + -e TERDUT_DB_DSN='postgres://terdut:secret@host.docker.internal:5432/terdut?sslmode=disable' \ + terdut-server +``` + +The server creates its own schema on startup and needs a reachable Postgres; it stores nothing on +disk, so there is no volume to mount. + +## Kubernetes + +A Helm chart is published from this repository as an OCI artifact, versioned in lockstep +with the app — chart `x.y.z` is always app `vx.y.z`: + +```bash +helm upgrade --install terdut-server oci://git.ryuvia.com/niklas/terdut-server \ + --version 0.9.2 \ + --namespace terdut-server --create-namespace \ + --set networking.hostname=terdut.example.com +``` + +The chart expects a [Gateway API](https://gateway-api.sigs.k8s.io/) Gateway named `envoy-main` in +the `envoy-gateway-system` namespace to already exist — it renders an `HTTPRoute` against it rather +than an `Ingress`. TLS is terminated at the gateway, so the server itself never sees a certificate. + +| Value | Default | Description | +|---|---|---| +| `networking.hostname` | `terdut.example.com` | Hostname the `HTTPRoute` serves | +| `networking.listener` | `""` | Gateway listener (`sectionName`) to bind to. Empty attaches to every matching listener, **including plaintext HTTP** — set it to the HTTPS listener's name to serve TLS only | +| `networking.servicePort` | `8080` | Port the route forwards to; keep in sync with `service.port` | +| `bootstrap.enabled` | `true` | Runs a post-install hook that creates the first user and stores its API key in the `-admin-key` Secret. Already-bootstrapped servers are left alone | +| `database.dsn` | `""` | **Required.** Postgres DSN, with no password in it. The chart provisions no database | +| `database.passwordSecret.name` | `""` | Secret supplying `PGPASSWORD`. With the Zalando postgres operator, the Secret it generates for the role | +| `database.passwordSecret.key` | `password` | Key within that Secret | + +The API key travels in an `Authorization: Bearer` header, so set `networking.listener` whenever the +hostname is reachable outside a trusted network. + +### The database + +The chart provisions no database: it takes a DSN and expects a Postgres that already exists. In this +cluster the wrapper chart declares an `acid.zalan.do/v1 postgresql` CR; anywhere else, any reachable +Postgres 14+ will do. + +The DSN carries no password. pgx falls back to libpq's environment variables for whatever the DSN +leaves out, so the password arrives as `PGPASSWORD` from a Secret and never appears in values, in +the rendered manifest or in `kubectl describe pod`. With the postgres operator that Secret is the +one it generates for the role, so a rebuild mints a new password with nothing to keep in sync — +the same wiring miniflux uses. + +The server migrates its own schema on startup, so a new database only has to exist and be writable. + +### Backups + +Postgres is backed up where it runs, not from here. The database pod carries a +[k8up](https://k8up.io/) `k8up.io/backupcommand` annotation that streams a `pg_dump`, the same way +gitea and immich do in this cluster. + +## On Kubernetes with the operator + +[terdut-operator](https://git.ryuvia.com/niklas/terdut-operator) runs a server for you from a +`TerdutServer` object and manages its teams, escalation ladders, dead man's switches and alert +sources as Kubernetes objects. It hands the server a generated key through `TERDUT_OPERATOR_KEY` +(see [Configuration](./configuration.md) and [`SERVICE-ACCOUNTS.md`](../SERVICE-ACCOUNTS.md)), and +the server then treats configuration as operator-managed (`TERDUT_OPERATOR_MODE`), refusing edits +made by hand in the web UI. Use the Helm chart above for a plain install, the operator when you +want that configuration in gitops. diff --git a/docs/development.md b/docs/development.md new file mode 100644 index 0000000..ae4b154 --- /dev/null +++ b/docs/development.md @@ -0,0 +1,93 @@ +# Development and releasing + +_Building, testing and releasing the server._ Back to the [README](../README.md) and the [documentation index](./README.md). + +## Upgrading + +The schema is a single baseline (`internal/db/migrations/001_schema.sql`) and no +release has shipped yet, so there is no upgrade path from earlier development +databases: start from an empty one. Changes after the first release arrive as +new numbered migrations. + +## Development + +```bash +make test-db # start a local Postgres for the tests (podman or docker) +make test # run all tests +go build ./... # compile all packages +go run ./cmd/terdut # run locally (needs TERDUT_DB_DSN) +``` + +The tests need a real Postgres, because the server does — there is no in-memory Postgres. +`TERDUT_TEST_DSN` says where it is, `make test-db` starts +one on port 5433 and prints the DSN, and `make test-db-stop` removes it. Each test gets its +own schema on that server, so tests cannot see each other's rows. An unset `TERDUT_TEST_DSN` +fails the suite rather than skipping it: a run that quietly tests nothing is worse than one +that does not run. + +`make fmt lint test helm-lint` is the gate. It mirrors `.gitea/workflows/ci.yaml` step for +step, so a green run here means a green pipeline — with one deliberate exception: `make test` +adds `-race`, which CI does not. The sweeper, the notifier goroutine and the dead man's switch +sweep all run concurrently against the same database, and a race between them would surface as +a flaky incident in production rather than as a red build. + +The web UI lives in `internal/web/static/` as plain HTML, CSS and ES modules, +embedded into the binary with `go:embed`. It has no build step and no npm, so +editing a file and restarting the server is the whole loop. + +## Releasing + +``` +push or PR → ci.yaml gofmt, go vet, go test -race + govulncheck, gitleaks + helm lint + render +push tag vX.Y.Z → release.yaml the same gate, then publish: + git.ryuvia.com/niklas/terdut-server:vX.Y.Z + oci://git.ryuvia.com/niklas/terdut-server X.Y.Z + then trivy-scan the pushed image +PR to Ryuvia/charts → bump the wrapper chart to X.Y.Z; on merge + Flux reconciles and the release rolls out +``` + +Both artifacts go to the **personal** Gitea namespace rather than `ryuvia`, because Gitea +scopes package visibility to the owner with no per-package override — so `ryuvia/*` is private +because the org is. Publishing to `niklas` keeps them anonymously pullable, which is why no +pull secret is needed in the cluster. Same reasoning, and the same choice, as riksdata and +rd-web. + +Saying **"Release"** runs all three rows: the `release` skill commits, pushes, tags, waits for +the pipeline, and opens the `Ryuvia/charts` PR, stopping before the merge. See +`~/.claude/skills/release/`, or `.release.conf` here for this repo's part of it. + +The chart is published **only** from the tag, by the `chart` job. There used to be a second +publisher on every `charts/**` push to main, and the two raced for the same chart version with +different answers — chart 0.9.0 went out reading `appVersion: "latest"` that way. One +publisher, triggered by the tag (`766f439`). The cost is that a chart-only change has no +version of its own and rides the next app tag. + +Both workflows are thin drivers over the Makefile: `ci.yaml` runs `make fmt lint test` and +`make helm-lint`, `release.yaml` adds `make binaries`, `make push`, `make helm-package` and +`make helm-push`. That is deliberate — it is what makes a green local gate and a green +pipeline the same code rather than two descriptions of it, and it is how riksdata and rd-web +have always worked. + +`make push` builds and pushes in one step, unlike those two, because the image is +`linux/amd64,linux/arm64` and buildx cannot load a multi-platform result into the local image +store. `make build` stays single-platform and local-only. Both refuse `VERSION=dev`: +publishing is one command, so it is also one command to run by accident. Publishing happens +by pushing a tag. + +Two things the release process needs to know about this repo: + +- **The image scan runs after publishing**, like riksdata's and rd-web's: trivy cannot read + a locally built image on this runner, so it pulls the pushed one. A red `scan-image` means + do not bump the wrapper chart to that version — it cannot unpublish anything. The image is + `FROM scratch`, so trivy sees exactly one target, the Go binary and its module graph. +- **The wrapper chart's `values.yaml` has two `tag:` lines** — the app image and the python + backup sidecar — so `chart-bump` is given `--image` to say which one moves. Once the wrapper + chart drops the sidecar and declares a `postgresql` CR instead, there is one `tag:` line + again, and `--image` becomes belt and braces. + +The wrapper chart must have **its own `version:` bumped in the same commit**. Flux reconciles +with `reconcileStrategy: ChartVersion`, so a chart whose version did not change produces no +new artifact and the change is never deployed — with no error anywhere. diff --git a/docs/escalation.md b/docs/escalation.md new file mode 100644 index 0000000..583aa2b --- /dev/null +++ b/docs/escalation.md @@ -0,0 +1,45 @@ +# Escalation + +_Escalation ladders: who is paged next when nobody acknowledges._ Back to the [README](../README.md) and the [documentation index](./README.md). + +Without a ladder, an unacknowledged incident re-pages the same topic every +`notify_repeat` forever. That is a louder version of the same silence: if the +person on call is asleep, out of signal, or has left, nothing else happens. + +A team can configure an ordered ladder instead. Each level has a timeout and a +set of targets, and a target is either a named person or **whoever the team's +rota says is on call today** — the target that keeps working when the rota +changes and nobody remembers to edit the policy. + +``` +level 1 5m oncall the rota gets first refusal +level 2 5m user:bob then a named second + then repeat_count more rounds + then the team's fallback topic, once +``` + +When a level's timeout passes with the incident still `triggered`, the next +level is paged. Off the end of the ladder the whole thing runs again +`repeat_count` times, and after that the team's `fallback_topic` is paged once +as the end of the line. The incident stays open throughout: running out of +people to wake is not the same as somebody answering. + +**Acknowledging or resolving stops it**, which is the point — continuing to wake +people after somebody has said "I have this" is how a tool teaches people to +mute it. **Snoozing pauses it**: a deliberate "not now" holds the ladder where +it is, and it resumes when the snooze runs out. + +Every step is on the incident's timeline with the level and the names it woke, +so somebody reading it afterwards can tell why their phone rang at 04:00. A +level whose targets are all unreachable — no ntfy topic, a disabled account, an +empty rota — is recorded as `nobody reachable` and the ladder moves on rather +than stalling on a rung that cannot ring. + +**Reminders and escalation never both run.** A team with a ladder gets +escalation; a team without keeps the reminder behaviour exactly as it was. Two +pages for one silence is the surest way to get a tool muted. + +The ladder's `fallback_topic` is per team, unlike `TERDUT_NTFY_FALLBACK_TOPIC`, +which is the install-wide topic used when an incident opens with nobody on call. +They answer different questions: one is "nobody was scheduled", the other is +"everybody scheduled has been tried". diff --git a/docs/images/admin.png b/docs/images/admin.png new file mode 100644 index 0000000..ea088af Binary files /dev/null and b/docs/images/admin.png differ diff --git a/docs/images/alerts.png b/docs/images/alerts.png new file mode 100644 index 0000000..f556dcd Binary files /dev/null and b/docs/images/alerts.png differ diff --git a/docs/images/incident-note.png b/docs/images/incident-note.png new file mode 100644 index 0000000..0b4cb16 Binary files /dev/null and b/docs/images/incident-note.png differ diff --git a/docs/images/mobile-incident.png b/docs/images/mobile-incident.png new file mode 100644 index 0000000..4762343 Binary files /dev/null and b/docs/images/mobile-incident.png differ diff --git a/docs/images/mobile-queue.png b/docs/images/mobile-queue.png new file mode 100644 index 0000000..b9fb72a Binary files /dev/null and b/docs/images/mobile-queue.png differ diff --git a/docs/images/oncall.png b/docs/images/oncall.png new file mode 100644 index 0000000..0a2a667 Binary files /dev/null and b/docs/images/oncall.png differ diff --git a/docs/images/queue-dark.png b/docs/images/queue-dark.png new file mode 100644 index 0000000..75d231b Binary files /dev/null and b/docs/images/queue-dark.png differ diff --git a/docs/images/queue-light.png b/docs/images/queue-light.png new file mode 100644 index 0000000..8c05295 Binary files /dev/null and b/docs/images/queue-light.png differ diff --git a/docs/images/stats.png b/docs/images/stats.png new file mode 100644 index 0000000..68816a7 Binary files /dev/null and b/docs/images/stats.png differ diff --git a/docs/images/team-deadman.png b/docs/images/team-deadman.png new file mode 100644 index 0000000..e958cb4 Binary files /dev/null and b/docs/images/team-deadman.png differ diff --git a/docs/images/team-escalation.png b/docs/images/team-escalation.png new file mode 100644 index 0000000..45d88ca Binary files /dev/null and b/docs/images/team-escalation.png differ diff --git a/docs/incidents.md b/docs/incidents.md new file mode 100644 index 0000000..77cf82d --- /dev/null +++ b/docs/incidents.md @@ -0,0 +1,113 @@ +# Alerts and incidents + +_How alerts become incidents and how incidents are worked, notified, escalated and expired._ Back to the [README](../README.md) and the [documentation index](./README.md). + +There are two objects, and the difference between them is the whole design. + +**An alert is Alertmanager's record.** It has two states, `firing` and +`resolved`, one row per fingerprint, and no human ever writes to it. The API +exposes alerts read-only. + +**An incident is the work item.** It goes `triggered → acknowledged → resolved`, +carries an assignee, a snooze, notes and a timeline, and is the only thing people +act on. Many alerts belong to one incident. + +## Correlation uses Alertmanager's `groupKey` + +Alertmanager has already grouped alerts according to the `group_by` routing tree +you configured, and it sends the resulting `groupKey` and `groupLabels` on every +webhook. Incidents adopt that answer rather than re-grouping alerts a second +time — if you want different correlation, change `group_by` in +`alertmanager.yml` and terdut follows. + +At most one incident is open per `groupKey` at a time. Alerts firing in a group +that already has an open incident join it. The incident's `severity` is a +high-water mark — the highest `severity` label any of its alerts has carried — so +an incident that hit `critical` still reads as critical after the critical alert +clears. + +## Several clusters, one team + +A team with one Alertmanager per Kubernetes cluster, each posting to its own +source, needs two settings or the clusters run together. + +1. Give every alert a `cluster` label at the source. In Prometheus that is + `externalLabels: {cluster: prod-eu}` (kube-prometheus-stack: + `prometheus.prometheusSpec.externalLabels`). +2. Add `cluster` to `group_by` in `alertmanager.yml`. + +The second one is the one that matters. Incidents are matched on the team and +Alertmanager's `groupKey`, and the `groupKey` does not include external labels: +without `cluster` in `group_by`, the same alert in two clusters has the same +key and joins one incident. With it, each cluster gets its own, `cluster` is in +the incident's `group_labels`, and the web UI shows it as a coloured chip on the +queue, the incident and the alert list, instead of leaving it in the title. +An alert that is not grouped by `cluster` still shows the chip on the alert +list, which reads the label from the alert itself. + +The queue has a cluster dropdown once there are two or more values to choose +between. It filters on the incident's `cluster` group label +(`GET /api/incidents?cluster=...`), so it only sees incidents grouped by it. + +## An incident opens only on a new occurrence + +An incident opens when an alert **transitions into firing**: a fingerprint that +was never seen, an alert with a newer `startsAt`, or a resolved alert that +started again. The unchanged firing notifications Alertmanager re-sends every +`repeat_interval` are none of those, and open nothing. + +This is what makes closing an incident by hand mean something. Without the rule, +`POST /api/incidents/{id}/resolve` would be undone by the next re-send of an +alert that never stopped firing. + +## Leaving the open state + +- **Automatically**, once every alert under the incident has stopped firing — + whether by a resolved webhook or by the sweeper's + [stale-alert expiry](#stale-alert-expiry). The incident gets + `"resolution_source": "alerts"`. +- **By hand**, via `POST /api/incidents/{id}/resolve` + (`"resolution_source": "manual"`). This is **terminal**: a later occurrence in + that group opens a *new* incident rather than reopening this one. If the alert + underneath never stops firing, the incident stays closed — that is what + resolving by hand asserts. +- **On recovery**, for a [dead man's switch](./dead-mans-switch.md) incident whose + heartbeat started arriving again (`"resolution_source": "recovered"`). These + incidents have no member alerts, so the automatic cascade above cannot reach + them. + +To quieten an incident you expect to come back, snooze it instead +(`POST /api/incidents/{id}/snooze`). A snooze hides the incident from the default +list without closing it, and expires by simply falling into the past. + +## On-call assignment + +A new incident is assigned to whoever holds today's schedule entry at the moment +it opens (`GET /api/schedule/current`). If nobody is scheduled it opens +unassigned. Reassign with `POST /api/incidents/{id}/assign`. + +One person holds a given day, so `POST /api/schedule` refuses a date somebody +already has: taking a shift off the person expecting to be paged for it should +not be something a plain call does by accident. Pass `"replace": true` to take +them anyway. Either way the whole request is one transaction — a week where some +days are free and some are taken moves as a unit, and a failure leaves the rota +exactly as it was rather than with a hole in it. + +## Stale alert expiry + +A resolved webhook is the only signal that an alert has stopped firing, so a +notification that is dropped, silenced, or lost to a restart would otherwise pin +that alert as firing forever. A background sweeper resolves firing alerts that +Alertmanager has stopped refreshing, using either signal: + +- the `endsAt` watermark on the last notification has passed, or +- no webhook has refreshed the alert within `TERDUT_STALE_AFTER`. + +Alertmanager re-sends firing notifications every `repeat_interval`, which is what +keeps a live alert fresh — so `TERDUT_STALE_AFTER` must be comfortably larger +than your `repeat_interval` (default 4h), or live alerts will be resolved +prematurely. Alerts resolved this way are marked `"resolution_source": "expiry"` +to distinguish them from a real Alertmanager resolve (`"alertmanager"`). + +An expiry cascades: once it leaves an incident with nothing firing under it, the +incident resolves too, in the same sweep. diff --git a/docs/notifications.md b/docs/notifications.md new file mode 100644 index 0000000..d455c4a --- /dev/null +++ b/docs/notifications.md @@ -0,0 +1,62 @@ +# Push notifications + +_Pages through ntfy, who gets them and how to acknowledge from the notification._ Back to the [README](../README.md) and the [documentation index](./README.md). + +With `TERDUT_NTFY_URL` set, an incident that opens is pushed to the on-call +person's phone through [ntfy](https://ntfy.sh). Everybody sets their own topic +under *Account* in the web UI, where a **Send a test push** button proves it +before an incident has to; `PUT /api/users/{id}/notify` is the same thing over +the API, and an administrator may set somebody else's. A user with no topic +falls back to `TERDUT_NTFY_FALLBACK_TOPIC`, as does an incident that opens with +nobody on call. If neither yields a topic, nothing is queued. + +The **server** is the install's one ntfy, from `TERDUT_NTFY_URL`, and is not +something a user picks. Only the topic is per-person. + +A topic is a shared secret with the ntfy server: anyone who knows it can both +read the pages and publish to it, so an unguessable one is worth the trouble. +That is also why the topic never appears in an incident's timeline, which every +API key can read. + +Three things get pushed: + +- **triggered** — an incident opened. Priority follows severity (`critical` maps + to ntfy's max priority, the one that overrides the phone's quiet settings). +- **reminder** — the incident is still `triggered` after `TERDUT_NOTIFY_REPEAT`. + Repeats until somebody acts. Acknowledging, snoozing, resolving or archiving + all stop it — snooze is the mute button. +- **resolved** — every alert under the incident stopped firing. Only sent to + whoever was paged in the first place, and only for the automatic cascade: + resolving by hand pushes nothing, since the person who did it already knows. + +Notifications carry an **Acknowledge** button that acknowledges the incident +without opening anything. It POSTs to `/api/notify/ack/{token}`, an +unauthenticated route authorised by the 256-bit token in its path — minted fresh +per notification, scoped to one incident and one action, and valid for 24 hours. +A real API key is never put in a notification, because the message is stored on +the ntfy server and cached on the device. + +The token is **not** consumed by use. Acknowledging is idempotent, so a token +stays valid for its full 24 hours and a second tap is a no-op that reports the +incident's current state rather than an error — which is what you want when a +tap is retried on a flaky mobile connection. What bounds it is scope, not a use +count: one incident, one action, one day. Expired tokens are purged by the +sweeper. + +Two consequences worth planning for: + +- `/api/notify/ack/{token}` **must stay publicly reachable**, or the button will + not work when the responder is off your network. +- Notifications sent to the fallback topic carry **no** Acknowledge button. The + topic is shared, and a button on it would let any subscriber acknowledge as + somebody else. + +Delivery is a queue, not an inline call: the webhook writes a row and a +background notifier sends it within 30 seconds, retrying with exponential +backoff up to 8 attempts. Nothing about ingestion blocks on ntfy being reachable. + +Every delivery is recorded on the incident's timeline: a `notified` event once +ntfy accepts the publish, and a `notify_failed` event when a notification +exhausts its retries. Written from the result rather than at enqueue, so the +timeline says what actually happened — and a page that never landed is visible +instead of looking the same as one that did. diff --git a/docs/single-sign-on.md b/docs/single-sign-on.md new file mode 100644 index 0000000..6432c0c --- /dev/null +++ b/docs/single-sign-on.md @@ -0,0 +1,105 @@ +# Single sign-on (OIDC) + +_Signing in through an OpenID Connect provider, and mapping its groups to teams and administrators._ Back to the [README](../README.md) and the [documentation index](./README.md). + +terdut can sign people in through any OpenID Connect provider; the examples use +[Authentik](https://goauthentik.io/). Groups at the provider decide who may sign +in, which teams they belong to and whether they administer the install, much as +Grafana's OAuth role and org mapping does. Password login keeps working alongside +it unless you turn it off. + +**At the provider**, create an OAuth2/OpenID provider and an application for it: +a *confidential* client, redirect URI `/api/oidc/callback`, and +the `openid`, `profile` and `email` scopes. The issuer is the application's, e.g. +`https://auth.example.com/application/o/terdut/`. Then set: + +```sh +TERDUT_PUBLIC_URL=https://terdut.example.com +TERDUT_OIDC_ISSUER=https://auth.example.com/application/o/terdut/ +TERDUT_OIDC_CLIENT_ID=terdut +TERDUT_OIDC_CLIENT_SECRET=... +TERDUT_OIDC_ALLOWED_GROUPS=terdut-users,terdut-admins +TERDUT_OIDC_ADMIN_GROUP=terdut-admins +``` + +Which team a group grants is not server-wide config: each team names its own +group(s), set by that team's own owner (or an administrator) from its Members +tab, or `PUT /api/teams/{teamID}/oidc-groups {"member_group":"sre","owner_group":"sre-leads"}`. +A team must already exist before a group can grant access to it — the sync +never creates one. + +The web UI's sign-in page shows a "Sign in with " button (a plain link to +`/api/oidc/login`) above the password form, or instead of it when +`TERDUT_PASSWORD_LOGIN=false`; it asks `GET /api/auth/config` what the server offers +(`password_login`, `oidc.enabled`, `oidc.name`). A refused sign-in comes back to that +page with the reason spelled out. Access the groups grant is badged **SSO** on the +Team, Admin and per-user pages, with its edit and remove controls disabled, and the +Account page does not offer to set a password nobody could use. + +**What a sign-in does** + +1. *Who.* The provider's `(issuer, subject)` is the identity. The first time, a + user is found by email — only when the provider marks it verified, or + `TERDUT_OIDC_TRUST_EMAIL` is set — or created with no password. A username taken + by somebody else gets a numeric suffix (`alice-2`). Username and email follow the + provider at each sign-in. Authentik reports `email_verified` as false unless + configured otherwise, so linking existing users usually needs + `TERDUT_OIDC_TRUST_EMAIL=true`. +2. *Whether.* With `TERDUT_OIDC_ALLOWED_GROUPS` set, somebody in none of them is + refused and nothing is created. +3. *What.* The administrator flag follows `TERDUT_OIDC_ADMIN_GROUP`. Team roles + follow each team's own `oidc_member_group`/`oidc_owner_group`; where both of a + team's groups match, the owner group wins. + +**Managed access.** What the sync grants is marked as managed by single sign-on, +and only that is ever changed by it. It is added at sign-in, and removed at the +next sign-in after the group is gone, even if that leaves a team without an owner +(an administrator can always repair a team) — the provider is the source of truth +for what it grants, so the last-owner and last-administrator guards do not apply. +Memberships and administrators added by hand are left alone; the exception is a +hand-added member whose team's own group grants a *higher* role, who is raised and +from then on managed. Editing managed access by hand (`POST` or `DELETE` on a +team's members, revoking an SSO-granted administrator) is refused with `409`, since +the next sign-in would undo it. + +> **Upgrading past migration 013: reconfigure every team's groups.** +> `TERDUT_OIDC_GROUP_MAPPINGS` is gone, and the sync no longer creates a team by +> name. Group-to-team-role mapping is now each team's own setting — an owner sets +> it from the Members tab, or `PUT /api/teams/{teamID}/oidc-groups`. Until a team's +> owner does that, an OIDC-sourced membership in it is dropped at that user's next +> SSO sign-in, the same as any other loss of group access. Set every team's groups +> before affected users next sign in, to avoid a visible gap in access. + +**How fast changes arrive.** Groups are read only at sign-in. A session made by an +SSO sign-in has a hard ceiling (`TERDUT_OIDC_SESSION_MAX_AGE`, default 12h) that +sliding never extends, so a change at the provider reaches terdut within that time. +Password sessions are unaffected. + +> **API keys are not revoked when somebody is removed at the provider.** terdut +> holds no refresh token and never asks the provider again, so a person removed +> from every allowed group loses their sessions within `TERDUT_OIDC_SESSION_MAX_AGE` +> and cannot sign in again, but keeps any API key they made (the TUI and scripts use +> them) until an administrator disables the user in terdut. + +**Signing in from a terminal.** A client with no browser of its own, such as the +TUI over SSH, signs in with a device code, run by terdut itself so the terminal +never talks to the provider: + +1. The terminal calls `POST /api/oidc/device` and shows the person a link + (`/device?code=XXXX-XXXX`) and the code. +2. On any device the person opens the link, signs in (by the provider or by + password, whatever the login page offers), sees the code and the account, and + presses **Approve**. Only a browser session can approve; an API key cannot. +3. The terminal polls `POST /api/oidc/device/token` every 5 seconds and is given the + ordinary `terdut_session` cookie once. A person who signs in through the provider + gets the same `TERDUT_OIDC_SESSION_MAX_AGE` ceiling on the terminal's session as + on their browser's. + +A login expires after 10 minutes. `GET /api/auth/config` reports `device_login`. + +**If the provider is down**, terdut still starts (discovery is fetched on first +use) and password login is the way in. With `TERDUT_PASSWORD_LOGIN=false` that way +is closed: set it back to `true`. The first administrator comes from the bootstrap +endpoint, and stays a manual administrator that no group can revoke; on an SSO-only +install set `bootstrap.enabled: false` in the chart if you don't want that account, +or keep it and never give it a password. diff --git a/docs/web-ui.md b/docs/web-ui.md new file mode 100644 index 0000000..5c20c3f --- /dev/null +++ b/docs/web-ui.md @@ -0,0 +1,90 @@ +# The web UI + +_What the web UI offers, how sign-in and sessions work, and the Team and Admin tabs._ Back to the [README](../README.md) and the [documentation index](./README.md). + +The server serves a web UI at `/`: the incident queue, each incident's alerts +and timeline with every action (acknowledge, assign, snooze, note, resolve, +archive), who is on call, the alert feed, and an *Account* tab for your own +password and the ntfy topic your pages go to. It is built for a phone first. On a phone +it navigates through a hamburger menu and has a sticky action bar, it follows the +system's dark mode, and it can be added to the home screen. From 900px wide it switches +to a sidebar with the queue and the incident side by side. The Stats page shows +incident counts, MTTA and MTTR, and alert frequency by name, hour and day over a +chosen range. + +You sign in with a username and password. Users have no password until one is +set, and a user without one can only use API keys: + +```bash +# an admin sets someone's first password with their API key +curl -X PUT http://localhost:8080/api/users/2/password \ + -H "Authorization: Bearer $KEY" -H "Content-Type: application/json" \ + -d '{"password": ""}' +``` + +After that, users change it themselves under *Account*. Changing your own +password requires the current one. + +How a browser stays signed in: + +- A successful login sets an `HttpOnly`, `SameSite=Lax` session cookie. It lasts + 30 days and slides forward while it is used, so an on-call phone stays signed + in. +- The cookie is marked `Secure` when `TERDUT_PUBLIC_URL` starts with `https://`, + so set it to the HTTPS address. TLS terminates at the gateway and the server + itself only ever sees plain HTTP. +- Requests authenticated by the cookie are checked for cross-origin use (Go's + `http.CrossOriginProtection`). That is the CSRF guard. Bearer-key clients are + not affected. +- Setting a password signs that user out everywhere else. +- Ten failed logins for one username within 15 minutes lock that username for + the rest of the window. + +With `TERDUT_PUBLIC_URL` set, tapping a push notification opens the incident in +the web UI (`/incidents/{id}`). + +A **Team** tab holds everything a team owns, in five sub-sections with a URL +each and a strip across the top to move between them: the on-call rota +(`/team/rota`), the membership (`/team/members`), the escalation ladder +(`/team/escalation`), the alert sources with their keys (`/team/sources`) and +the dead man's switches (`/team/deadman`). `/team` itself is an overview — who +is on call today, how many members and owners, how many ladder levels, how many +keys and how many switches — so a page fetches only what it shows. An owner +edits it; a member sees the same pages read-only, because the server refuses +their writes anyway. Somebody in more than one team picks between them above +the strip, since the choice changes the subject of all five. + +The rota is a month at a time, one coloured initial per day with a legend +underneath, and it says how many days are left uncovered — the question a rota +is read for is who holds which stretch, and a run of one colour answers it +where a list of dates does not. An owner taps a day to hand it to somebody or +empty it, and fills a whole shift from the range form folded in below. + +The **Admin** tab appears only for a system administrator, and holds what +belongs to the whole server rather than to one team. It has three sub-sections, +each with a URL of its own and a strip across the top to move between them: +every team (`/admin/teams`), every user (`/admin/users`), and the settings that +used to be environment variables (`/admin/settings`). `/admin` itself is an +overview — how many of each, and what each section is for. Adding somebody is +minting them an invite link into a team, rather than creating a bare account: +the person who accepts it picks their own password, so one never passes through +an administrator, and the link carries the team, so they land somewhere with a +queue in it. That happens on the team's own page, since an invite is a fact +about a team; the user list points there rather than asking which team beside a +form. + +A name in the team list opens **that team's page**, at `/admin/teams/{id}`: when it +was created, how many are in it and how much is open, a field to rename it, the +members with their roles, the invites into it, and deletion. The member list is the +one thing there that needed a new endpoint — `GET /api/teams/{id}/members` is +member-only and answers `404` to an administrator who is not in the team, which is +the rule and not an oversight, so the page reads `GET /api/admin/teams/{id}` instead. +An administrator still sees none of that team's incidents, alerts or rota. + +A name in the user list opens **that person's page**, at `/admin/users/{id}`: their +email and when they joined, where their notifications go, whether they are an +administrator, whether the account is disabled, the teams they are in with their +role in each, a password field for a first or forgotten one, and deletion. It is +the one place membership is edited from the person's side — the Team tab answers +"who is in this team", and answering "which teams is this person in" there means +visiting each team in turn. diff --git a/internal/api/alertmanager.go b/internal/api/alertmanager.go index 071684b..7e3cc39 100644 --- a/internal/api/alertmanager.go +++ b/internal/api/alertmanager.go @@ -87,7 +87,7 @@ func handleIntegrationWebhook(db *sql.DB, notify NotifyConfig) http.HandlerFunc respond(w, http.StatusUnauthorized, errResp("unknown integration key")) return } - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } receiveWebhook(w, r, db, notify, src) diff --git a/internal/api/alerts.go b/internal/api/alerts.go index a2b0b2b..5bdecf1 100644 --- a/internal/api/alerts.go +++ b/internal/api/alerts.go @@ -90,7 +90,7 @@ func handleListAlerts(db *sql.DB) http.HandlerFunc { fmt.Sprintf("%s WHERE %s ORDER BY a.received_at DESC LIMIT %s", alertSelectFrom, clause, args.add(limit)), args.all()...) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -99,7 +99,7 @@ func handleListAlerts(db *sql.DB) http.HandlerFunc { for rows.Next() { a, err := scanAlert(rows) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } alerts = append(alerts, a) @@ -121,7 +121,7 @@ func handleGetAlert(db *sql.DB) http.HandlerFunc { return } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, a) diff --git a/internal/api/api_test.go b/internal/api/api_test.go index ec26d42..bae9c57 100644 --- a/internal/api/api_test.go +++ b/internal/api/api_test.go @@ -78,6 +78,10 @@ func newTSWith(t *testing.T, deadman api.DeadmanConfig, cfg api.NotifyConfig, co s := &ts{Server: srv, key: key, db: database, notify: cfg, deadman: deadman} + // A fresh install has no team, so the tests that want "the" team make it + // here: it is id 1, owned by the admin, which is what defaultTeam names. + decode(t, s.req(t, http.MethodPost, "/api/teams", map[string]string{"name": "Default"}), &struct{}{}) + var integration struct { Key string `json:"key"` } @@ -768,7 +772,7 @@ func TestWebhook_IgnoresOutOfOrderRetry(t *testing.T) { // An expiry resolve writes ends_at as an upper bound, not an observed end: an // Alertmanager watermark already on the row is preserved, and a row that never // carried one is stamped at sweep time. Clients are told to read it that way — -// see "resolution_source says how much to trust ends_at" in the README. +// see "resolution_source says how much to trust ends_at" in docs/api.md. func TestExpiry_EndsAtIsUpperBound(t *testing.T) { s := newTS(t) @@ -813,7 +817,7 @@ func TestExpiry_EndsAtIsUpperBound(t *testing.T) { // // received_at is documented as a public liveness signal, so these lock the // behaviour clients are told they may rely on. See "received_at is a liveness -// heartbeat" in the README and the comment on models.Alert.ReceivedAt. +// heartbeat" in docs/api.md and the comment on models.Alert.ReceivedAt. // --------------------------------------------------------------------------- // The heartbeat itself: an unchanged firing notification — what Alertmanager @@ -875,3 +879,38 @@ func TestStats_ByDayReturnsSevenSlots(t *testing.T) { t.Errorf("expected 7 day slots, got %d", len(slots)) } } + +// Two simultaneous bootstraps on an empty install must not both win. +func TestBootstrap_ConcurrentCallsCreateOneAdmin(t *testing.T) { + database := newTestDB(t) + srv := httptest.NewServer(api.NewRouter(database, api.NotifyConfig{}, testConfig(), "test")) + t.Cleanup(srv.Close) + + const n = 8 + codes := make(chan int, n) + for i := 0; i < n; i++ { + go func(i int) { + body, _ := json.Marshal(map[string]string{"username": fmt.Sprintf("u%d", i), "email": fmt.Sprintf("u%d@x.com", i)}) + resp, err := http.Post(srv.URL+"/api/bootstrap", "application/json", bytes.NewReader(body)) + if err != nil { + codes <- 0 + return + } + resp.Body.Close() + codes <- resp.StatusCode + }(i) + } + created := 0 + for i := 0; i < n; i++ { + if <-codes == http.StatusCreated { + created++ + } + } + var users int + if err := database.QueryRow("SELECT COUNT(*) FROM users").Scan(&users); err != nil { + t.Fatal(err) + } + if created != 1 || users != 1 { + t.Errorf("expected exactly one bootstrap to win, got %d created and %d users", created, users) + } +} diff --git a/internal/api/archiver.go b/internal/api/archiver.go index 6759596..b99df7c 100644 --- a/internal/api/archiver.go +++ b/internal/api/archiver.go @@ -154,9 +154,7 @@ func expireStale(ctx context.Context, db *sql.DB, staleAfter time.Duration, skip } // staleAlertIDs reads the ids in one go and closes the cursor before the caller -// writes. Under SQLite's single connection an open read would have blocked the -// update outright; with a pool it is no longer a deadlock, but reading the set -// first still keeps the write off a cursor the same transaction is walking. +// writes, which keeps the write off a cursor the same transaction is walking. func staleAlertIDs(ctx context.Context, db *sql.DB, now time.Time, staleAfter time.Duration) ([]int64, error) { rows, err := db.QueryContext(ctx, ` SELECT id FROM alerts diff --git a/internal/api/auth.go b/internal/api/auth.go index 804f4d5..38feff9 100644 --- a/internal/api/auth.go +++ b/internal/api/auth.go @@ -10,6 +10,7 @@ import ( "strconv" "strings" "sync" + "sync/atomic" "time" "github.com/go-chi/chi/v5" @@ -28,6 +29,9 @@ const ( // sessionTouchEvery bounds how often a request may slide the expiry. sessionTouchEvery = time.Hour + // keyTouchEvery is the same bound for an API key's last_used_at. + keyTouchEvery = 5 * time.Minute + minPasswordLen = 10 // maxPasswordLen is bcrypt's limit; it rejects longer input outright. maxPasswordLen = 72 @@ -126,14 +130,32 @@ func purgeRateLimits(ctx context.Context, db *sql.DB) { } } +// trustedProxies is how many X-Forwarded-For hops clientAddr trusts. Set once +// by NewRouter from config. +var trustedProxies atomic.Int64 + // clientAddr is the address a login is counted against. Behind the gateway -// RemoteAddr is the gateway itself, so the first X-Forwarded-For hop is used -// when present. It can be forged, but only to dodge the address limit; the -// per-username limit does not depend on it. +// RemoteAddr is the gateway itself, so the client address is read from +// X-Forwarded-For, counting trustedProxies entries from the right: each trusted +// proxy appends the address it saw, so the entries to the left of those are +// client-supplied and could be forged to dodge the limit. func clientAddr(r *http.Request) string { - if xff := r.Header.Get("X-Forwarded-For"); xff != "" { - first, _, _ := strings.Cut(xff, ",") - return strings.TrimSpace(first) + if n := int(trustedProxies.Load()); n > 0 { + var hops []string + for _, v := range r.Header.Values("X-Forwarded-For") { + for _, h := range strings.Split(v, ",") { + if h = strings.TrimSpace(h); h != "" { + hops = append(hops, h) + } + } + } + if len(hops) > 0 { + i := len(hops) - n + if i < 0 { + i = 0 + } + return hops[i] + } } host, _, err := net.SplitHostPort(r.RemoteAddr) if err != nil { @@ -241,7 +263,7 @@ func handleLogin(db *sql.DB, limiter *loginLimiter, publicURL string) http.Handl "SELECT id, password_hash FROM users WHERE username = $1", username, ).Scan(&userID, &hash) if err != nil && !errors.Is(err, sql.ErrNoRows) { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -258,13 +280,13 @@ func handleLogin(db *sql.DB, limiter *loginLimiter, publicURL string) http.Handl limiter.clear(r.Context(), userKey) if err := startSession(w, r, db, userID, publicURL); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } user, err := fetchUser(r.Context(), db, userID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, meResponse{User: user, HasPassword: true}) @@ -320,7 +342,7 @@ func handleMe(db *sql.DB) http.HandlerFunc { } user, err := fetchUser(r.Context(), db, caller.ID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } var hash sql.NullString @@ -332,7 +354,7 @@ func handleMe(db *sql.DB) http.HandlerFunc { // transient database problem, not a missing user — worth a 500 // rather than silently answering "no password, not dismissed", // which a client would otherwise take at face value. - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, meResponse{ @@ -384,7 +406,7 @@ func handleSetPassword(db *sql.DB) http.HandlerFunc { return } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -397,30 +419,30 @@ func handleSetPassword(db *sql.DB) http.HandlerFunc { hash, err := hashPassword(req.Password) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } tx, err := db.BeginTx(r.Context(), nil) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer tx.Rollback() if _, err := tx.ExecContext(r.Context(), "UPDATE users SET password_hash = $1 WHERE id = $2", hash, id); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } keep, _ := sessionFromContext(r.Context()) // zero when changed with an API key if _, err := tx.ExecContext(r.Context(), "DELETE FROM sessions WHERE user_id = $1 AND id != $2", id, keep); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if err := tx.Commit(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } w.WriteHeader(http.StatusNoContent) diff --git a/internal/api/caller.go b/internal/api/caller.go index 95501de..a656445 100644 --- a/internal/api/caller.go +++ b/internal/api/caller.go @@ -2,7 +2,6 @@ package api import ( "context" - "fmt" "git.ryuvia.com/niklas/terdut-server/internal/models" ) @@ -61,7 +60,7 @@ func (c Caller) IsAdmin() bool { // by being a human (and becomes its owner as a side effect), an // instance-scoped service account creates one with no human owner at all; // the two paths are not interchangeable, so this predicate must not also -// admit a human admin the way MayActAsInstanceAdmin deliberately does. +// admit a human admin the way an administrator check would. func (c Caller) IsInstanceServiceAccount() bool { return c.sa != nil && c.sa.scope == models.ServiceAccountScopeInstance } @@ -110,24 +109,6 @@ func (c Caller) ServiceAccountName() (string, bool) { return c.sa.name, true } -// Identity is a stable, log/audit-facing string distinguishing a human -// caller from a service account — "user:42" or "service-account:7". Not -// wired into any database column — incidents.go's acknowledged_by/ -// incident_events.user_id use AsHuman()/ServiceAccountID() directly against -// the parallel *_service_account_id columns (migration 015) instead, since a -// column needs the id, not this rendered string. assigned_to stays -// human-only and out of scope (terdut-server#25's follow-up). -func (c Caller) Identity() string { - switch { - case c.user != nil: - return fmt.Sprintf("user:%d", c.user.ID) - case c.sa != nil: - return fmt.Sprintf("service-account:%d", c.sa.id) - default: - return "unknown" - } -} - func callerFromContext(ctx context.Context) (Caller, bool) { c, ok := ctx.Value(ctxCaller).(Caller) return c, ok diff --git a/internal/api/deadman.go b/internal/api/deadman.go index 70bbd01..6b813f0 100644 --- a/internal/api/deadman.go +++ b/internal/api/deadman.go @@ -65,28 +65,6 @@ func (m DeadmanMatcher) matches(labels map[string]string) bool { return true } -// DeadmanConfig is the server-wide default a team's switches are seeded from: -// the environment's matchers, timeout and severity. Switches themselves are rows -// of a team's own — see DeadmanSwitch — and this is only how a fresh install -// starts out. -type DeadmanConfig struct { - Matchers []DeadmanMatcher - - // Timeout is how long a matched alert may go without a refreshing webhook - // before it is declared dead. It must be shorter than Alertmanager's - // repeat_interval for the heartbeat's route, which is what refreshes it. - // Zero disables dead man's switch handling entirely. - Timeout time.Duration - - // Severity is the severity every dead man's switch incident opens at. These - // incidents have no member alerts to derive one from, and the heartbeat's - // own severity label is meaningless — Watchdog ships as "none". - Severity string -} - -// enabled reports whether there is anything to watch. -func (c DeadmanConfig) enabled() bool { return c.Timeout > 0 && len(c.Matchers) > 0 } - // DeadmanSwitch inverts the handling of the alerts it matches: receiving one // opens nothing, and the absence of one opens an incident. // @@ -160,46 +138,6 @@ func parseDeadmanMatcher(entry string) (DeadmanMatcher, error) { return m, nil } -// ParseDeadmanConfig reads the matcher list from its configured form: -// ";" separates matchers, and each is parsed as parseDeadmanMatcher does. -// -// A malformed or alertname-less entry is dropped rather than fatal, following -// config.duration's rule that one bad tuning knob should not take the server -// down. Silence would be worse here than elsewhere, though — a typo that -// disarms the switch is exactly the failure this feature exists to catch — so -// the matchers that survived are logged. -func ParseDeadmanConfig(matchers string, timeout time.Duration, severity string) DeadmanConfig { - cfg := DeadmanConfig{Timeout: timeout, Severity: severity} - - for _, entry := range strings.Split(matchers, ";") { - entry = strings.TrimSpace(entry) - if entry == "" { - continue - } - m, err := parseDeadmanMatcher(entry) - if err != nil { - log.Printf("deadman: ignoring matcher %q: %v", entry, err) - continue - } - cfg.Matchers = append(cfg.Matchers, m) - } - - switch { - case timeout <= 0: - log.Print("deadman: disabled (timeout is zero)") - case len(cfg.Matchers) == 0: - log.Print("deadman: disabled (no usable matchers)") - default: - rendered := make([]string, 0, len(cfg.Matchers)) - for _, m := range cfg.Matchers { - rendered = append(rendered, m.String()) - } - log.Printf("deadman: default for new teams: %s, timeout %s, severity %s", - strings.Join(rendered, "; "), timeout, severity) - } - return cfg -} - // deadmanAlert is one heartbeat: the alert row carrying its last sighting, and // the switch that claimed it. type deadmanAlert struct { @@ -490,56 +428,6 @@ func deadmanSets(ctx context.Context, db *sql.DB) (map[int64]deadmanSet, error) return scanDeadmanSwitches(rows) } -// deadmanSeededKey is the settings row that records the environment defaults -// were handed out. Without it, a team that deleted its last switch would get -// the default back on the next restart. -const deadmanSeededKey = "deadman_seeded" - -// SeedDeadmanConfigs gives every team the server's environment defaults as -// switches, exactly once per install, so a fresh install watches Watchdog -// without anybody setting it up. -// -// Once seeded it never runs again: a team's switches are its own, and a redeploy -// must not quietly put the environment's value back over an owner's edit or -// deletion. Installs that upgraded from per-team configuration were already -// seeded, which migration 009 records. -// -// A team created after that gets none and watches nothing until its owner says -// otherwise. That is deliberate: inheriting an install-wide heartbeat would page -// a new team about a source it has never heard of, and a switch nobody chose is -// the kind that gets muted rather than fixed. -func SeedDeadmanConfigs(ctx context.Context, db *sql.DB, cfg DeadmanConfig) error { - if !cfg.enabled() { - return nil - } - - tx, err := db.BeginTx(ctx, nil) - if err != nil { - return err - } - defer tx.Rollback() //nolint:errcheck - - res, err := tx.ExecContext(ctx, - "INSERT INTO settings (key, value) VALUES ($1, '1') ON CONFLICT (key) DO NOTHING", - deadmanSeededKey) - if err != nil { - return err - } - if n, _ := res.RowsAffected(); n == 0 { - return nil - } - - for _, m := range cfg.Matchers { - if _, err := tx.ExecContext(ctx, ` - INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity) - SELECT id, $1, $1, $2, $3 FROM teams`, - m.config(), int64(cfg.Timeout.Seconds()), cfg.Severity); err != nil { - return err - } - } - return tx.Commit() -} - // --------------------------------------------------------------------------- // Status // --------------------------------------------------------------------------- diff --git a/internal/api/deadman_test.go b/internal/api/deadman_test.go index 4d6d172..485be77 100644 --- a/internal/api/deadman_test.go +++ b/internal/api/deadman_test.go @@ -1,7 +1,6 @@ package api_test import ( - "context" "net/http" "strings" "testing" @@ -122,8 +121,11 @@ func TestDeadman_MixedGroupExcludesHeartbeat(t *testing.T) { t.Fatalf("expected 1 incident for the real alert, got %d", got) } - var alerts []map[string]any - decode(t, s.req(t, http.MethodGet, "/api/incidents/1/alerts", nil), &alerts) + var incident struct { + Alerts []map[string]any `json:"alerts"` + } + decode(t, s.req(t, http.MethodGet, "/api/incidents/1", nil), &incident) + alerts := incident.Alerts if len(alerts) != 1 { t.Fatalf("expected 1 member alert, got %d", len(alerts)) } @@ -717,26 +719,3 @@ func TestDeadman_DeleteIsScopedToTheTeam(t *testing.T) { t.Errorf("expected no switches, got %d", got) } } - -// The environment's defaults are handed out once and then belong to the teams. -func TestDeadman_SeedRunsOnce(t *testing.T) { - s := newTS(t) - cfg := api.ParseDeadmanConfig("alertname=Watchdog", time.Hour, "critical") - - if err := api.SeedDeadmanConfigs(context.Background(), s.db, cfg); err != nil { - t.Fatalf("seed: %v", err) - } - if got := len(listSwitches(t, s)); got != 1 { - t.Fatalf("the first seed should add the default, got %d switches", got) - } - - // The owner deletes it; a restart must not put it back. - id := int64(listSwitches(t, s)[0]["id"].(float64)) - s.req(t, http.MethodDelete, "/api/teams/"+defaultTeam+"/deadman/switches/"+id64(id), nil).Body.Close() - if err := api.SeedDeadmanConfigs(context.Background(), s.db, cfg); err != nil { - t.Fatalf("seed again: %v", err) - } - if got := len(listSwitches(t, s)); got != 0 { - t.Errorf("a second seed resurrected %d switch(es)", got) - } -} diff --git a/internal/api/device.go b/internal/api/device.go index d694b50..120dad9 100644 --- a/internal/api/device.go +++ b/internal/api/device.go @@ -81,7 +81,7 @@ func handleDeviceStart(db *sql.DB, limiter *loginLimiter, publicURL string) http deviceCode, deviceHash, err := randomToken() if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -105,7 +105,7 @@ func handleDeviceStart(db *sql.DB, limiter *loginLimiter, publicURL string) http } if err != nil { log.Printf("device login: start: %v", err) - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -160,7 +160,7 @@ func handleDeviceDecision(db *sql.DB, approve bool) http.HandlerFunc { WHERE user_code = $3 AND status = 'pending' AND expires_at > $4`, status, caller.ID, code, time.Now().Unix()) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -190,7 +190,7 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht tx, err := db.BeginTx(r.Context(), nil) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer tx.Rollback() //nolint:errcheck @@ -206,7 +206,7 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht return } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -226,11 +226,11 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht } if _, err := tx.ExecContext(r.Context(), "UPDATE device_logins SET last_polled_at = $1 WHERE device_hash = $2", now.Unix(), hash); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if err := tx.Commit(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusAccepted, map[string]string{"status": "pending"}) @@ -240,7 +240,7 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht // Approved. Single use: the row goes before the session is made, so two // racing polls cannot both be given one. if _, err := tx.ExecContext(r.Context(), "DELETE FROM device_logins WHERE device_hash = $1", hash); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } var disabled, sso bool @@ -252,7 +252,7 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht return } if err := tx.Commit(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if disabled { @@ -269,12 +269,12 @@ func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) ht } if err := startSessionCapped(w, r, db, userID.Int64, publicURL, maxAge); err != nil { log.Printf("device login: start session: %v", err) - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } user, err := fetchUser(r.Context(), db, userID.Int64) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, meResponse{User: user, HasPassword: false}) diff --git a/internal/api/escalation.go b/internal/api/escalation.go index 59c3a84..ed3275c 100644 --- a/internal/api/escalation.go +++ b/internal/api/escalation.go @@ -3,6 +3,7 @@ package api import ( "context" "database/sql" + "errors" "log" "net/http" "strconv" @@ -343,12 +344,12 @@ func handleGetEscalation(db *sql.DB) http.HandlerFunc { policy, err := loadEscalationPolicy(r.Context(), db, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } view, err := escalationStatus(r.Context(), db, teamID, policy) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, view) @@ -543,6 +544,10 @@ type escalationLevelJSON struct { type escalationTargetJSON struct { Kind string `json:"kind"` UserID *int64 `json:"user_id,omitempty"` + // Username is accepted in place of user_id on a PUT, and resolved to the + // id before anything is stored. It is never returned: the stored form is + // the id, which survives a rename. + Username string `json:"username,omitempty"` } type escalationJSON struct { @@ -600,6 +605,27 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc { return } for i, l := range req.Levels { + for j, t := range l.Targets { + if t.Username == "" { + continue + } + if t.Kind != "user" || t.UserID != nil { + respond(w, http.StatusBadRequest, errResp("username belongs on a user target, instead of user_id")) + return + } + var id int64 + err := db.QueryRowContext(r.Context(), "SELECT id FROM users WHERE username = $1", t.Username).Scan(&id) + if errors.Is(err, sql.ErrNoRows) { + respond(w, http.StatusBadRequest, errResp("unknown user "+strconv.Quote(t.Username))) + return + } + if err != nil { + serverError(w, r, err) + return + } + req.Levels[i].Targets[j].UserID = &id + req.Levels[i].Targets[j].Username = "" + } if l.TimeoutSeconds <= 0 { respond(w, http.StatusBadRequest, errResp("every level needs a timeout")) return @@ -632,7 +658,7 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc { tx, err := db.BeginTx(r.Context(), nil) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer tx.Rollback() //nolint:errcheck @@ -645,13 +671,13 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc { fallback_topic = excluded.fallback_topic, updated_at = excluded.updated_at`, teamID, req.RepeatCount, req.FallbackTopic); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } // The levels are replaced, not merged; the cascade takes the targets. if _, err := tx.ExecContext(r.Context(), "DELETE FROM escalation_levels WHERE team_id = $1", teamID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -661,7 +687,7 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc { INSERT INTO escalation_levels (team_id, position, timeout_seconds) VALUES ($1, $2, $3) RETURNING id`, teamID, int64(i+1), l.TimeoutSeconds).Scan(&levelID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } for _, t := range l.Targets { @@ -676,13 +702,13 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc { } if err := tx.Commit(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } policy, err := loadEscalationPolicy(r.Context(), db, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, escalationResponse(policy, teamID)) diff --git a/internal/api/escalation_test.go b/internal/api/escalation_test.go index 39a428a..e36f96f 100644 --- a/internal/api/escalation_test.go +++ b/internal/api/escalation_test.go @@ -502,3 +502,45 @@ func TestEscalation_StatusWithoutALadder(t *testing.T) { t.Errorf("a team with no ladder should read as empty, got %+v", v) } } + +// A user target may name the person instead of carrying an id; the server +// resolves it and stores the id. +func TestEscalation_UserTargetByUsername(t *testing.T) { + s := newTS(t) + id := teamUser(t, s, "alice", "") + + put := func(username string) *http.Response { + return s.req(t, http.MethodPut, "/api/teams/"+defaultTeam+"/escalation", map[string]any{ + "repeat_count": 0, + "levels": []map[string]any{{ + "timeout_seconds": 300, + "targets": []map[string]any{{"kind": "user", "username": username}}, + }}, + }) + } + + resp := put("alice") + resp.Body.Close() + if resp.StatusCode >= 300 { + t.Fatalf("PUT by username: %d", resp.StatusCode) + } + var got struct { + Levels []struct { + Targets []struct { + UserID *int64 `json:"user_id"` + Username string `json:"username"` + } `json:"targets"` + } `json:"levels"` + } + decode(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/escalation", nil), &got) + if len(got.Levels) != 1 || len(got.Levels[0].Targets) != 1 || + got.Levels[0].Targets[0].UserID == nil || *got.Levels[0].Targets[0].UserID != id { + t.Errorf("expected the target stored as user %d, got %+v", id, got) + } + + bad := put("nobody") + bad.Body.Close() + if bad.StatusCode != http.StatusBadRequest { + t.Errorf("unknown username should be a 400, got %d", bad.StatusCode) + } +} diff --git a/internal/api/export_test.go b/internal/api/export_test.go new file mode 100644 index 0000000..9928297 --- /dev/null +++ b/internal/api/export_test.go @@ -0,0 +1,31 @@ +package api + +import ( + "strings" + "time" +) + +// DeadmanConfig is a test fixture only: a set of matchers with one timeout and +// severity, turned into switches over the API by the test helpers. Production +// has no server-wide default any more -- switches belong to teams. +type DeadmanConfig struct { + Matchers []DeadmanMatcher + Timeout time.Duration + Severity string +} + +// ParseDeadmanConfig reads a ";"-separated matcher list the way the removed +// environment variable did, dropping malformed entries. +func ParseDeadmanConfig(matchers string, timeout time.Duration, severity string) DeadmanConfig { + cfg := DeadmanConfig{Timeout: timeout, Severity: severity} + for _, entry := range strings.Split(matchers, ";") { + entry = strings.TrimSpace(entry) + if entry == "" { + continue + } + if m, err := parseDeadmanMatcher(entry); err == nil { + cfg.Matchers = append(cfg.Matchers, m) + } + } + return cfg +} diff --git a/internal/api/helpers.go b/internal/api/helpers.go index 4ea7dcd..2ae751e 100644 --- a/internal/api/helpers.go +++ b/internal/api/helpers.go @@ -5,6 +5,7 @@ import ( "database/sql" "encoding/json" "errors" + "github.com/go-chi/chi/v5" "log" "net/http" "strconv" @@ -17,8 +18,7 @@ import ( // sqlArgs accumulates query arguments and hands back the placeholder for each. // // Postgres numbers its placeholders, so a dynamically assembled WHERE clause has -// to keep its $1, $2, … in step with the order of the values — which SQLite's -// positional `?` did for free. Handing out the placeholder and storing the value +// to keep its $1, $2, … in step with the order of the values. Handing out the placeholder and storing the value // in one call is what keeps them in step: a filter can be added, removed or // reordered without renumbering anything by hand. type sqlArgs struct{ vals []any } @@ -31,8 +31,7 @@ func (a *sqlArgs) add(v any) string { // addList stores every value and returns their placeholders as "$1, $2, …", // ready to drop into an IN (…) clause. Returns an empty string for no values, -// which no caller should reach: `IN ()` is a syntax error in Postgres as it was -// in SQLite, so callers check for an empty set before building the query. +// which no caller should reach: `IN ()` is a syntax error in Postgres, so callers check for an empty set before building the query. func (a *sqlArgs) addList(vs []any) string { parts := make([]string, len(vs)) for i, v := range vs { @@ -45,7 +44,7 @@ func (a *sqlArgs) addList(vs []any) string { func (a *sqlArgs) all() []any { return a.vals } // nowEpoch is the SQL expression for "now, as unix seconds", matching how every -// timestamp in this schema is stored. SQLite spelled it unixepoch(). +// timestamp in this schema is stored. // // FLOOR, not a bare cast: EXTRACT returns fractional seconds and casting to // bigint rounds half up, so a row written at .6 of a second would claim a @@ -56,11 +55,9 @@ const nowEpoch = "FLOOR(EXTRACT(EPOCH FROM now()))::bigint" // isUniqueViolation reports whether err is a broken unique constraint, which // callers turn into 409 Conflict rather than 500. // -// Postgres reports it as SQLSTATE 23505 on a typed error; the SQLite driver this -// replaced only put "UNIQUE constraint failed" in the message, which is why the -// check used to be a substring match. Matching the code means a renamed -// constraint or a translated message cannot quietly turn a conflict back into a -// 500. +// Postgres reports it as SQLSTATE 23505 on a typed error. Matching the code +// means a renamed constraint or a translated message cannot quietly turn a +// conflict back into a 500. func isUniqueViolation(err error) bool { var pgErr *pgconn.PgError return errors.As(err, &pgErr) && pgErr.Code == pgerrcode.UniqueViolation @@ -72,6 +69,18 @@ func respond(w http.ResponseWriter, status int, v any) { json.NewEncoder(w).Encode(v) } +// serverError answers 500 and logs why. The response stays opaque, so the log +// line is the only record of what failed. +func serverError(w http.ResponseWriter, r *http.Request, err error) { + // The route pattern, not the path: two routes carry a credential in it. + route := r.URL.Path + if rc := chi.RouteContext(r.Context()); rc != nil && rc.RoutePattern() != "" { + route = rc.RoutePattern() + } + log.Printf("%s %s: %v", strconv.Quote(r.Method), strconv.Quote(route), err) + respond(w, http.StatusInternalServerError, errResp("internal error")) +} + // maxBodyBytes caps an ordinary JSON request body. 1 MiB is far more than any // endpoint below needs — it exists so an unauthenticated caller (signup, // login, bootstrap) can't make the server buffer an arbitrarily large body diff --git a/internal/api/incidents.go b/internal/api/incidents.go index 19c12a1..50cc9ac 100644 --- a/internal/api/incidents.go +++ b/internal/api/incidents.go @@ -37,7 +37,7 @@ func handleListClusters(db *sql.DB) http.HandlerFunc { label, strings.Join(where, " AND ")), args.all()...) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -46,13 +46,13 @@ func handleListClusters(db *sql.DB) http.HandlerFunc { for rows.Next() { var v string if err := rows.Scan(&v); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } clusters = append(clusters, v) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, clusters) @@ -138,7 +138,7 @@ func handleListIncidents(db *sql.DB) http.HandlerFunc { incidentSelectFrom, strings.Join(where, " AND "), order, args.add(limit)), args.all()...) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -147,13 +147,13 @@ func handleListIncidents(db *sql.DB) http.HandlerFunc { for rows.Next() { i, err := scanIncident(rows) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } incidents = append(incidents, i) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, incidents) @@ -172,35 +172,17 @@ func handleGetIncident(db *sql.DB) http.HandlerFunc { return } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if inc.Alerts, err = incidentAlerts(r, db, id); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, inc) } } -func handleIncidentAlerts(db *sql.DB) http.HandlerFunc { - return func(w http.ResponseWriter, r *http.Request) { - id, ok := incidentIDParam(w, r, db) - if !ok { - return - } - if !incidentExists(w, r, db, id) { - return - } - alerts, err := incidentAlerts(r, db, id) - if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) - return - } - respond(w, http.StatusOK, alerts) - } -} - func handleIncidentTimeline(db *sql.DB) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { id, ok := incidentIDParam(w, r, db) @@ -225,7 +207,7 @@ func handleIncidentTimeline(db *sql.DB) http.HandlerFunc { WHERE e.incident_id = $1 ORDER BY e.created_at ASC, e.id ASC`, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -239,14 +221,14 @@ func handleIncidentTimeline(db *sql.DB) http.HandlerFunc { &e.ActorUserID, &e.ActorUsername, &e.ActorServiceAccountID, &e.ActorServiceAccountName, &e.AlertID, &e.Detail, &ts); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } e.CreatedAt = time.Unix(ts, 0).UTC() events = append(events, e) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, events) @@ -262,7 +244,7 @@ func handleIncidentAcknowledge(db *sql.DB) http.HandlerFunc { userID, saID := callerActorIDs(r.Context()) acked, err := acknowledgeIncidentAs(r.Context(), db, id, userID, saID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if !acked { @@ -272,7 +254,7 @@ func handleIncidentAcknowledge(db *sql.DB) http.HandlerFunc { // (acknowledged) already holds. inc, err := fetchIncident(r.Context(), db, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if inc.Status == "resolved" { @@ -300,7 +282,7 @@ func handleIncidentUnacknowledge(db *sql.DB) http.HandlerFunc { return } if err := logEvent(r.Context(), db, id, evUnacknowledged, userID, saID, nil, nil); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } w.WriteHeader(http.StatusNoContent) @@ -335,16 +317,16 @@ func handleIncidentResolve(db *sql.DB) http.HandlerFunc { } // A person closing an incident is the clearest possible "I have this". if err := stopEscalation(r.Context(), db, id); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if err := logEvent(r.Context(), db, id, evResolved, userID, saID, nil, nil); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if req.Resolution != "" { if err := logEvent(r.Context(), db, id, evResolutionNote, userID, saID, nil, &req.Resolution); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } } @@ -385,7 +367,7 @@ func handleIncidentAssign(db *sql.DB) http.HandlerFunc { // the actor_* columns. actorUserID, actorSAID := callerActorIDs(r.Context()) if err := logAssignedEvent(r.Context(), db, id, req.UserID, actorUserID, actorSAID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respondIncident(w, r, db, id) @@ -443,7 +425,7 @@ func handleIncidentSnooze(db *sql.DB) http.HandlerFunc { } detail := until.UTC().Format(time.RFC3339) if err := logEvent(r.Context(), db, id, evSnoozed, userID, saID, nil, &detail); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respondIncident(w, r, db, id) @@ -462,7 +444,7 @@ func handleIncidentUnsnooze(db *sql.DB) http.HandlerFunc { return } if err := logEvent(r.Context(), db, id, evUnsnoozed, userID, saID, nil, nil); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } w.WriteHeader(http.StatusNoContent) @@ -478,7 +460,7 @@ func handleIncidentArchive(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "UPDATE incidents SET archived_at = "+nowEpoch+" WHERE id = $1", id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -487,7 +469,7 @@ func handleIncidentArchive(db *sql.DB) http.HandlerFunc { } userID, saID := callerActorIDs(r.Context()) if err := logEvent(r.Context(), db, id, evArchived, userID, saID, nil, nil); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respondIncident(w, r, db, id) @@ -503,7 +485,7 @@ func handleIncidentUnarchive(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "UPDATE incidents SET archived_at = NULL WHERE id = $1", id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -512,7 +494,7 @@ func handleIncidentUnarchive(db *sql.DB) http.HandlerFunc { } userID, saID := callerActorIDs(r.Context()) if err := logEvent(r.Context(), db, id, evUnarchived, userID, saID, nil, nil); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } w.WriteHeader(http.StatusNoContent) @@ -557,7 +539,7 @@ func handleCreateNote(db *sql.DB) http.HandlerFunc { VALUES ($1, $2, $3, $4, $5, $6) RETURNING id`, id, noteType, userID, saID, req.Content, now.Unix()).Scan(&eventID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -598,7 +580,7 @@ func handleDeleteNote(db *sql.DB) http.HandlerFunc { AND (user_id = $5 OR service_account_id = $6)`, eventID, id, evNote, evResolutionNote, userID, saID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -652,7 +634,7 @@ func incidentExists(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64 func updateOpenIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64, query string, args ...any) bool { res, err := db.ExecContext(r.Context(), query, args...) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return false } if n, _ := res.RowsAffected(); n > 0 { @@ -668,7 +650,7 @@ func updateOpenIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id i func respondIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64) { inc, err := fetchIncident(r.Context(), db, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, inc) diff --git a/internal/api/incidents_test.go b/internal/api/incidents_test.go index 6fbd466..4dc23d7 100644 --- a/internal/api/incidents_test.go +++ b/internal/api/incidents_test.go @@ -794,8 +794,7 @@ func TestStats_Incidents(t *testing.T) { } // An empty window is a report of zero, not a failure. SUM over no rows is NULL -// in Postgres as it was in SQLite, and that used to come back as a 500 the -// moment every incident was archived — the state a quiet installation settles +// in Postgres, and that used to come back as a 500 the moment every incident was archived — the state a quiet installation settles // into. func TestStats_IncidentsEmptyWindowIsZeroNotAnError(t *testing.T) { s := newTS(t) diff --git a/internal/api/middleware.go b/internal/api/middleware.go index ae0aee7..462b386 100644 --- a/internal/api/middleware.go +++ b/internal/api/middleware.go @@ -5,10 +5,14 @@ import ( "crypto/sha256" "database/sql" "encoding/hex" + "log" "net/http" "strings" "time" + "github.com/go-chi/chi/v5" + "github.com/go-chi/chi/v5/middleware" + "git.ryuvia.com/niklas/terdut-server/internal/config" "git.ryuvia.com/niklas/terdut-server/internal/models" ) @@ -106,7 +110,7 @@ func securityHeaders(publicURL string) func(http.Handler) http.Handler { // // 403 and not 404: the route exists and the caller is authenticated, they are // simply not allowed. Hiding the endpoint would buy nothing — every one of them -// is in the README. +// is in docs/api.md. func AdminOnly(next http.Handler) http.Handler { return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { caller, ok := userFromContext(r.Context()) @@ -141,19 +145,23 @@ func requireSelfOrAdmin(w http.ResponseWriter, r *http.Request, targetID int64) // resolve and has to be caught afterwards. func apiKeyUser(ctx context.Context, db *sql.DB, token string) (int64, bool) { var keyID, userID int64 + var lastUsed sql.NullInt64 err := db.QueryRowContext(ctx, - `SELECT id, user_id FROM api_keys + `SELECT id, user_id, last_used_at FROM api_keys WHERE key_hash = $1 AND (expires_at IS NULL OR expires_at > $2)`, hashToken(token), time.Now().Unix(), - ).Scan(&keyID, &userID) + ).Scan(&keyID, &userID, &lastUsed) if err != nil { return 0, false } - // best-effort; don't fail the request if this update fails - db.ExecContext(ctx, - "UPDATE api_keys SET last_used_at = $1 WHERE id = $2", - time.Now().Unix(), keyID) + // best-effort; don't fail the request if this update fails. Throttled like + // the session expiry, so a polling client does not write a row per request. + if now := time.Now(); !lastUsed.Valid || now.Sub(time.Unix(lastUsed.Int64, 0)) > keyTouchEvery { + db.ExecContext(ctx, + "UPDATE api_keys SET last_used_at = $1 WHERE id = $2", + now.Unix(), keyID) + } return userID, true } @@ -206,7 +214,7 @@ func serveAs(w http.ResponseWriter, r *http.Request, next http.Handler, db *sql. // table with one row per membership. teams, err := callerMemberships(r.Context(), db, userID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -222,10 +230,8 @@ func hashToken(token string) string { return hex.EncodeToString(h[:]) } -// userFromContext is a thin compatibility wrapper over Caller.AsHuman(), so -// every call site written before the Caller abstraction (alerts.go, -// incidents.go, schedule.go, stats.go, and more) needs no change and keeps -// its exact existing behavior. +// userFromContext returns the human behind the request, or false for a service +// account: Caller.AsHuman() on the request's Caller. func userFromContext(ctx context.Context) (models.User, bool) { c, _ := callerFromContext(ctx) return c.AsHuman() @@ -245,13 +251,13 @@ type serviceAccountPrincipal struct { func serviceAccountFor(ctx context.Context, db *sql.DB, token string) (serviceAccountPrincipal, bool) { var sa serviceAccountPrincipal var keyID int64 - var teamID sql.NullInt64 + var teamID, lastUsed sql.NullInt64 err := db.QueryRowContext(ctx, ` - SELECT k.id, a.id, a.name, a.scope, a.team_id + SELECT k.id, a.id, a.name, a.scope, a.team_id, k.last_used_at FROM service_account_keys k JOIN service_accounts a ON a.id = k.service_account_id WHERE k.key_hash = $1`, hashToken(token), - ).Scan(&keyID, &sa.id, &sa.name, &sa.scope, &teamID) + ).Scan(&keyID, &sa.id, &sa.name, &sa.scope, &teamID, &lastUsed) if err != nil { return serviceAccountPrincipal{}, false } @@ -260,9 +266,11 @@ func serviceAccountFor(ctx context.Context, db *sql.DB, token string) (serviceAc } // best-effort; don't fail the request if this update fails - db.ExecContext(ctx, - "UPDATE service_account_keys SET last_used_at = $1 WHERE id = $2", - time.Now().Unix(), keyID) + if now := time.Now(); !lastUsed.Valid || now.Sub(time.Unix(lastUsed.Int64, 0)) > keyTouchEvery { + db.ExecContext(ctx, + "UPDATE service_account_keys SET last_used_at = $1 WHERE id = $2", + now.Unix(), keyID) + } return sa, true } @@ -286,10 +294,8 @@ func serveAsServiceAccount(w http.ResponseWriter, r *http.Request, next http.Han next.ServeHTTP(w, r.WithContext(ctx)) } -// isInstanceServiceAccount is a thin compatibility wrapper over -// Caller.IsInstanceServiceAccount(), for call sites outside this package's -// core predicates (handleCreateTeam, handleCreateServiceAccount) that -// needed this exact, narrow check before the Caller abstraction existed. +// isInstanceServiceAccount is Caller.IsInstanceServiceAccount() on the +// request's Caller. func isInstanceServiceAccount(ctx context.Context) bool { c, _ := callerFromContext(ctx) return c.IsInstanceServiceAccount() @@ -397,6 +403,13 @@ func requireTeamOwner(w http.ResponseWriter, r *http.Request, teamID int64) bool if ok && role == models.RoleOwner { return true } + // The operator's instance-scoped account manages every team's + // configuration, which is what lets it use one credential instead of + // minting one per team. This is owner reach only: it does not make the + // account a member, so it still reads no team's incidents. + if c, _ := callerFromContext(r.Context()); c.IsInstanceServiceAccount() { + return true + } if caller, _ := userFromContext(r.Context()); caller.IsAdmin { return true } @@ -414,3 +427,26 @@ func sessionFromContext(ctx context.Context) (int64, bool) { id, ok := ctx.Value(ctxSession).(int64) return id, ok } + +// requestLogger logs one line per request with the matched route pattern in +// place of the URL path. Two routes carry a credential in the path (the +// integration key and the ack token), and chi's stock logger would write it to +// the log verbatim. +func requestLogger(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + start := time.Now() + ww := middleware.NewWrapResponseWriter(w, r.ProtoMajor) + next.ServeHTTP(ww, r) + route := "unmatched" + if rc := chi.RouteContext(r.Context()); rc != nil { + if p := rc.RoutePattern(); p != "" { + route = p + } + } + status := ww.Status() + if status == 0 { + status = http.StatusOK + } + log.Printf("%q %q %d %dB %s", r.Method, route, status, ww.BytesWritten(), time.Since(start).Round(time.Millisecond)) // #nosec G706 -- method and route are %q-quoted, the route is a registered pattern, the rest are numbers + }) +} diff --git a/internal/api/notifier.go b/internal/api/notifier.go index 8f5b621..610e79b 100644 --- a/internal/api/notifier.go +++ b/internal/api/notifier.go @@ -449,7 +449,7 @@ func renderNotification(inc models.Incident, n outboxRow, firing int, cfg Notify // originLabel is the label that says where an alert came from, for a team with // several Kubernetes clusters behind it. It comes from Prometheus's // externalLabels and reaches an incident through Alertmanager's group_by; the -// web UI reads the same label, and the README ("Several clusters, one team") +// web UI reads the same label, and docs/incidents.md ("Several clusters, one team") // explains how to set it up. const originLabel = "cluster" diff --git a/internal/api/notify_ack.go b/internal/api/notify_ack.go index 43c4507..86e9e63 100644 --- a/internal/api/notify_ack.go +++ b/internal/api/notify_ack.go @@ -62,7 +62,7 @@ func handleNotifyAck(db *sql.DB) http.HandlerFunc { acked, err := acknowledgeIncident(r.Context(), db, incidentID, userID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if !acked { @@ -73,7 +73,7 @@ func handleNotifyAck(db *sql.DB) http.HandlerFunc { // ntfy show a success toast rather than a failure. inc, err := fetchIncident(r.Context(), db, incidentID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, map[string]any{ diff --git a/internal/api/oidc.go b/internal/api/oidc.go index fe4baa7..5221cd7 100644 --- a/internal/api/oidc.go +++ b/internal/api/oidc.go @@ -130,12 +130,12 @@ func handleOIDCLogin(db *sql.DB, prov *oidc.Provider, limiter *loginLimiter, pub state, stateHash, err := randomToken() if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } nonce, _, err := randomToken() if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } verifier := oidc.NewVerifier() @@ -149,7 +149,7 @@ func handleOIDCLogin(db *sql.DB, prov *oidc.Provider, limiter *loginLimiter, pub INSERT INTO oidc_logins (state_hash, nonce, pkce_verifier, next, expires_at) VALUES ($1, $2, $3, $4, $5)`, stateHash, nonce, verifier, next, now.Add(oidcLoginTTL).Unix()); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } diff --git a/internal/api/oidc_teams.go b/internal/api/oidc_teams.go index 7c8fb18..0bdc8ef 100644 --- a/internal/api/oidc_teams.go +++ b/internal/api/oidc_teams.go @@ -31,7 +31,7 @@ func handleGetTeamOIDCGroups(db *sql.DB) http.HandlerFunc { "SELECT COALESCE(oidc_member_group, ''), COALESCE(oidc_owner_group, '') FROM teams WHERE id = $1", teamID).Scan(&g.MemberGroup, &g.OwnerGroup) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, g) @@ -66,7 +66,7 @@ func handleSetTeamOIDCGroups(db *sql.DB) http.HandlerFunc { oidc_owner_group = NULLIF($2, '') WHERE id = $3`, req.MemberGroup, req.OwnerGroup, teamID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } w.WriteHeader(http.StatusNoContent) diff --git a/internal/api/operator_key.go b/internal/api/operator_key.go new file mode 100644 index 0000000..952a0ec --- /dev/null +++ b/internal/api/operator_key.go @@ -0,0 +1,72 @@ +package api + +import ( + "context" + "database/sql" + "fmt" + + "git.ryuvia.com/niklas/terdut-server/internal/models" +) + +const ( + // operatorAccountName is the instance-scoped service account the operator + // key belongs to. + operatorAccountName = "terdut-operator" + + // operatorKeyName names the one key SeedOperatorKey manages on it, so a + // rotation replaces that key and leaves any others alone. + operatorKeyName = "seed" +) + +// SeedOperatorKey makes key the operator account's credential: it creates the +// instance-scoped service account if needed and replaces its "seed" key with +// this one. Idempotent, so every replica can run it at every start, and a +// rotated key simply wins on the next restart. +// +// The key is hashed like any other, so only the caller that generated it ever +// holds the raw value. An empty key does nothing. +func SeedOperatorKey(ctx context.Context, db *sql.DB, key string) error { + if key == "" { + return nil + } + tx, err := db.BeginTx(ctx, nil) + if err != nil { + return fmt.Errorf("seed operator key: %w", err) + } + defer tx.Rollback() //nolint:errcheck + + // Serialise replicas starting together; transaction-scoped, so it needs no + // explicit release. + if _, err := tx.ExecContext(ctx, "SELECT pg_advisory_xact_lock($1)", operatorKeyLockKey); err != nil { + return fmt.Errorf("seed operator key: lock: %w", err) + } + + var accountID int64 + err = tx.QueryRowContext(ctx, + "SELECT id FROM service_accounts WHERE name = $1 AND scope = $2", + operatorAccountName, models.ServiceAccountScopeInstance).Scan(&accountID) + if err == sql.ErrNoRows { + err = tx.QueryRowContext(ctx, + "INSERT INTO service_accounts (name, scope) VALUES ($1, $2) RETURNING id", + operatorAccountName, models.ServiceAccountScopeInstance).Scan(&accountID) + } + if err != nil { + return fmt.Errorf("seed operator key: account: %w", err) + } + + if _, err := tx.ExecContext(ctx, + "DELETE FROM service_account_keys WHERE service_account_id = $1 AND name = $2", + accountID, operatorKeyName); err != nil { + return fmt.Errorf("seed operator key: drop old key: %w", err) + } + if _, err := tx.ExecContext(ctx, + "INSERT INTO service_account_keys (service_account_id, key_hash, name) VALUES ($1, $2, $3)", + accountID, hashToken(key), operatorKeyName); err != nil { + return fmt.Errorf("seed operator key: store key: %w", err) + } + return tx.Commit() +} + +// operatorKeyLockKey is the transaction-scoped advisory lock SeedOperatorKey +// holds; distinct from the other lock keys in this package. +const operatorKeyLockKey int64 = 7265_0010 diff --git a/internal/api/operator_key_test.go b/internal/api/operator_key_test.go new file mode 100644 index 0000000..2481842 --- /dev/null +++ b/internal/api/operator_key_test.go @@ -0,0 +1,134 @@ +package api_test + +import ( + "context" + "encoding/json" + "net/http" + "testing" + + "git.ryuvia.com/niklas/terdut-server/internal/api" +) + +const ( + operatorKeyOne = "tdsa_operator-key-number-one-0123456789" + operatorKeyTwo = "tdsa_operator-key-number-two-0123456789" +) + +func TestSeedOperatorKey_AuthenticatesAsInstanceAccount(t *testing.T) { + s := newTS(t) + if err := api.SeedOperatorKey(context.Background(), s.db, operatorKeyOne); err != nil { + t.Fatal(err) + } + + resp := s.reqAs(t, operatorKeyOne, http.MethodPost, "/api/teams", map[string]string{"name": "seeded"}) + resp.Body.Close() + if resp.StatusCode != http.StatusCreated { + t.Fatalf("seeded key should create a team, got %d", resp.StatusCode) + } +} + +func TestSeedOperatorKey_RotationReplacesAndIsIdempotent(t *testing.T) { + s := newTS(t) + ctx := context.Background() + for _, key := range []string{operatorKeyOne, operatorKeyOne, operatorKeyTwo} { + if err := api.SeedOperatorKey(ctx, s.db, key); err != nil { + t.Fatal(err) + } + } + + old := s.reqAs(t, operatorKeyOne, http.MethodGet, "/api/teams", nil) + old.Body.Close() + if old.StatusCode != http.StatusUnauthorized { + t.Errorf("rotated-out key should be refused, got %d", old.StatusCode) + } + cur := s.reqAs(t, operatorKeyTwo, http.MethodGet, "/api/teams", nil) + cur.Body.Close() + if cur.StatusCode != http.StatusOK { + t.Errorf("current key should work, got %d", cur.StatusCode) + } + + var accounts, keys int + s.db.QueryRow("SELECT COUNT(*) FROM service_accounts WHERE name = 'terdut-operator'").Scan(&accounts) + s.db.QueryRow("SELECT COUNT(*) FROM service_account_keys").Scan(&keys) + if accounts != 1 || keys != 1 { + t.Errorf("expected one account and one key, got %d and %d", accounts, keys) + } +} + +func TestSeedOperatorKey_EmptyKeyDoesNothing(t *testing.T) { + s := newTS(t) + if err := api.SeedOperatorKey(context.Background(), s.db, ""); err != nil { + t.Fatal(err) + } + var n int + s.db.QueryRow("SELECT COUNT(*) FROM service_accounts").Scan(&n) + if n != 0 { + t.Errorf("expected no service account, got %d", n) + } +} + +// The instance account configures a team it did not create, which is what lets +// the operator hold one credential instead of one per team, yet it is not a +// member and so reads none of the team's incidents. +func TestInstanceAccount_ActsAsOwnerOfAnyTeamButIsNoMember(t *testing.T) { + s := newTS(t) + other := newTeam(t, s, "other") + if err := api.SeedOperatorKey(context.Background(), s.db, operatorKeyOne); err != nil { + t.Fatal(err) + } + + rename := s.reqAs(t, operatorKeyOne, http.MethodPut, "/api/teams/"+id64(other.id), map[string]string{"name": "renamed"}) + rename.Body.Close() + if rename.StatusCode >= 300 { + t.Errorf("instance account should rename any team, got %d", rename.StatusCode) + } + + // Not a member: the team's queue is not visible to it. + var queue []map[string]any + decode(t, s.reqAs(t, operatorKeyOne, http.MethodGet, "/api/incidents", nil), &queue) + if len(queue) != 0 { + t.Errorf("instance account should see no incidents, got %v", queue) + } +} + +// external_id lets automation find its own team again after a crash, without +// trusting a display name. +func TestCreateTeam_ExternalIDIsIdempotentAndInstanceOnly(t *testing.T) { + s := newTS(t) + if err := api.SeedOperatorKey(context.Background(), s.db, operatorKeyOne); err != nil { + t.Fatal(err) + } + create := func(name string) (int, map[string]any) { + resp := s.reqAs(t, operatorKeyOne, http.MethodPost, "/api/teams", + map[string]string{"name": name, "external_id": "ns/platform"}) + var out map[string]any + _ = json.NewDecoder(resp.Body).Decode(&out) + resp.Body.Close() + return resp.StatusCode, out + } + + code, first := create("Platform") + if code != http.StatusCreated { + t.Fatalf("first create: %d", code) + } + // Same identity, even under a new display name: the same team comes back. + code, again := create("Platform renamed") + if code != http.StatusOK || again["id"] != first["id"] { + t.Errorf("repeat with the same external_id: want 200 and team %v, got %d %v", first["id"], code, again) + } + + // A different identity cannot take the name. + resp := s.reqAs(t, operatorKeyOne, http.MethodPost, "/api/teams", + map[string]string{"name": "Platform", "external_id": "other/platform"}) + resp.Body.Close() + if resp.StatusCode != http.StatusConflict { + t.Errorf("taken name under another external_id: want 409, got %d", resp.StatusCode) + } + + // A person cannot set one. + resp = s.req(t, http.MethodPost, "/api/teams", map[string]string{"name": "Mine", "external_id": "x/y"}) + resp.Body.Close() + if resp.StatusCode != http.StatusForbidden { + t.Errorf("a user setting external_id: want 403, got %d", resp.StatusCode) + } +} diff --git a/internal/api/router.go b/internal/api/router.go index 1b24f08..aceecd1 100644 --- a/internal/api/router.go +++ b/internal/api/router.go @@ -23,12 +23,13 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version strin // One limiter each, both process-wide for the life of the router: login // counts failed passwords, sign-up counts account creation, and mixing the // two would let a burst of sign-ups lock somebody out of logging in. + trustedProxies.Store(int64(cfg.TrustedProxies)) loginLimit := newLoginLimiter(db) signupLimiter := newLoginLimiter(db) oidcLimit := newLoginLimiter(db) r := chi.NewRouter() - r.Use(middleware.Logger) + r.Use(requestLogger) r.Use(middleware.Recoverer) r.Use(securityHeaders(notify.PublicURL)) @@ -48,11 +49,7 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version strin // Alert ingestion. The key in the path says both that the sender may post // and which team the alerts belong to, which is why it needs no session. - // - // This is the only way in. The pre-teams /api/alertmanager/webhook, which - // took no credential at all, was removed in v0.13.0 once the cluster's - // Alertmanager had moved onto a key; a sender still posting there gets the - // JSON 404 every unknown /api path gets. + // This is the only way in. r.Post("/api/integrations/{key}/alertmanager", handleIntegrationWebhook(db, notify)) // Signing up. Both are unauthenticated by necessity: the caller has no @@ -116,9 +113,7 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version strin r.Post("/api/users/{id}/api-keys", handleCreateAPIKey(db)) r.Delete("/api/users/{id}/api-keys/{keyID}", handleDeleteAPIKey(db)) - // Administration: who exists, and who is an administrator. Until #3 - // these were open to any authenticated caller, which meant every user - // could delete every other one. + // Administration: who exists, and who is an administrator. r.Group(func(r chi.Router) { r.Use(AdminOnly) @@ -147,7 +142,6 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version strin r.Get("/api/incidents", handleListIncidents(db)) r.Get("/api/incidents/clusters", handleListClusters(db)) r.Get("/api/incidents/{id}", handleGetIncident(db)) - r.Get("/api/incidents/{id}/alerts", handleIncidentAlerts(db)) r.Get("/api/incidents/{id}/timeline", handleIncidentTimeline(db)) r.Get("/api/incidents/{id}/similar", handleIncidentSimilar(db)) r.Post("/api/incidents/{id}/acknowledge", handleIncidentAcknowledge(db)) diff --git a/internal/api/schedule.go b/internal/api/schedule.go index 089800c..7535218 100644 --- a/internal/api/schedule.go +++ b/internal/api/schedule.go @@ -69,7 +69,7 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc { // be, so the delete and the insert share one transaction. tx, err := db.BeginTx(r.Context(), nil) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer tx.Rollback() @@ -79,7 +79,7 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc { if _, err := tx.ExecContext(r.Context(), "DELETE FROM schedule_entries WHERE team_id = $1 AND date = $2", teamID, d); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } } @@ -91,12 +91,12 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc { errResp("date already assigned: "+d+" (pass replace to take it)")) return } - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } } if err := tx.Commit(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -107,7 +107,7 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc { } all, err := scheduleRange(r.Context(), db, teamID, req.Dates[0], req.Dates[len(req.Dates)-1]) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } created := []models.ScheduleEntry{} @@ -147,7 +147,7 @@ func handleListSchedule(db *sql.DB) http.HandlerFunc { entries, err := scheduleRange(r.Context(), db, teamID, from, to) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, entries) @@ -171,7 +171,7 @@ func handleDeleteSchedule(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "DELETE FROM schedule_entries WHERE id = $1 AND team_id = $2", id, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -197,7 +197,7 @@ func handleCurrentSchedule(db *sql.DB) http.HandlerFunc { WHERE s.date = $1 AND s.team_id = ANY($2) ORDER BY t.name`, today, callerTeamIDs(r.Context())) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -207,14 +207,14 @@ func handleCurrentSchedule(db *sql.DB) http.HandlerFunc { var e models.ScheduleEntry var ts int64 if err := rows.Scan(&e.ID, &e.TeamID, &e.TeamName, &e.UserID, &e.Username, &e.Date, &ts); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } e.CreatedAt = time.Unix(ts, 0).UTC() entries = append(entries, e) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, entries) diff --git a/internal/api/service_accounts.go b/internal/api/service_accounts.go index c03bfcd..b95cb3e 100644 --- a/internal/api/service_accounts.go +++ b/internal/api/service_accounts.go @@ -50,6 +50,9 @@ func callerIsAdmin(ctx context.Context) bool { // writing a response: callers here need to combine it with other ways of // being allowed, not stop at the first no. func callerOwnsTeam(ctx context.Context, teamID int64) bool { + if c, _ := callerFromContext(ctx); c.IsInstanceServiceAccount() { + return true + } role, ok := callerRole(ctx, teamID) return ok && role == models.RoleOwner } @@ -132,7 +135,7 @@ func handleCreateServiceAccount(db *sql.DB) http.HandlerFunc { key, err := mintServiceAccountKey(r.Context(), db, sa.ID, "initial") if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusCreated, map[string]any{"service_account": sa, "key": key}) @@ -233,7 +236,7 @@ func handleCreateServiceAccountKey(db *sql.DB) http.HandlerFunc { return } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if !callerMayManageServiceAccount(r.Context(), sa) { @@ -255,7 +258,7 @@ func handleCreateServiceAccountKey(db *sql.DB) http.HandlerFunc { key, err := mintServiceAccountKey(r.Context(), db, sa.ID, req.Name) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusCreated, key) @@ -274,7 +277,7 @@ func handleDeleteServiceAccountKey(db *sql.DB) http.HandlerFunc { return } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if !callerMayManageServiceAccount(r.Context(), sa) { @@ -290,7 +293,7 @@ func handleDeleteServiceAccountKey(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "DELETE FROM service_account_keys WHERE id = $1 AND service_account_id = $2", keyID, sa.ID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -325,7 +328,7 @@ func handleListServiceAccounts(db *sql.DB) http.HandlerFunc { rows, err := db.QueryContext(r.Context(), query, args...) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -335,14 +338,14 @@ func handleListServiceAccounts(db *sql.DB) http.HandlerFunc { var sa models.ServiceAccount var created int64 if err := rows.Scan(&sa.ID, &sa.Name, &sa.Scope, &sa.TeamID, &sa.CreatedBy, &created); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } sa.CreatedAt = time.Unix(created, 0).UTC() accounts = append(accounts, sa) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, accounts) diff --git a/internal/api/settings.go b/internal/api/settings.go index f736c5f..fb22c44 100644 --- a/internal/api/settings.go +++ b/internal/api/settings.go @@ -203,7 +203,7 @@ func handleSetSettings(db *sql.DB) http.HandlerFunc { tx, err := db.BeginTx(r.Context(), nil) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer tx.Rollback() //nolint:errcheck @@ -215,12 +215,12 @@ func handleSetSettings(db *sql.DB) http.HandlerFunc { ON CONFLICT (key) DO UPDATE SET value = excluded.value, updated_at = excluded.updated_at`, key, value); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } } if err := tx.Commit(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -261,7 +261,7 @@ func handleAdminListTeams(db *sql.DB) http.HandlerFunc { FROM teams t ORDER BY t.name`) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -272,14 +272,14 @@ func handleAdminListTeams(db *sql.DB) http.HandlerFunc { var created int64 if err := rows.Scan(&t.ID, &t.Name, &created, &t.Members, &t.OpenIncidents, &t.OIDCMemberGroup, &t.OIDCOwnerGroup); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } t.CreatedAt = time.Unix(created, 0).UTC() teams = append(teams, t) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, teams) @@ -319,7 +319,7 @@ func handleAdminGetTeam(db *sql.DB) http.HandlerFunc { return } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } t.CreatedAt = time.Unix(created, 0).UTC() @@ -333,7 +333,7 @@ func handleAdminGetTeam(db *sql.DB) http.HandlerFunc { WHERE m.team_id = $1 ORDER BY u.username`, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -343,14 +343,14 @@ func handleAdminGetTeam(db *sql.DB) http.HandlerFunc { var m models.TeamMember var joined int64 if err := rows.Scan(&m.TeamID, &m.UserID, &m.Username, &m.Role, &joined, &m.Source); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } m.JoinedAt = time.Unix(joined, 0).UTC() members = append(members, m) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -396,7 +396,7 @@ func handleRenameTeam(db *sql.DB) http.HandlerFunc { respond(w, http.StatusConflict, errResp("a team with that name already exists")) return } - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -435,7 +435,7 @@ func handleSetUserDisabled(db *sql.DB) http.HandlerFunc { } last, err := isLastAdmin(r.Context(), db, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if last { @@ -453,7 +453,7 @@ func handleSetUserDisabled(db *sql.DB) http.HandlerFunc { "UPDATE users SET disabled_at = NULL WHERE id = $1", id) } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -475,7 +475,7 @@ func handleSetUserDisabled(db *sql.DB) http.HandlerFunc { user, err := fetchUser(r.Context(), db, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, user) diff --git a/internal/api/signup.go b/internal/api/signup.go index 9b32e83..0d2a3f8 100644 --- a/internal/api/signup.go +++ b/internal/api/signup.go @@ -170,13 +170,13 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand hash, err := hashPassword(req.Password) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } tx, err := db.BeginTx(r.Context(), nil) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer tx.Rollback() //nolint:errcheck @@ -194,7 +194,7 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand respond(w, http.StatusConflict, errResp("username or email already exists")) return } - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -207,7 +207,7 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand respond(w, http.StatusConflict, errResp("a team with that name already exists")) return } - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } role = models.RoleOwner @@ -216,7 +216,7 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand if _, err := tx.ExecContext(r.Context(), "INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3)", teamID, userID, role); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -226,7 +226,7 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand res, err := tx.ExecContext(r.Context(), "UPDATE invites SET uses = uses + 1 WHERE id = $1 AND uses < max_uses", inv.id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -236,14 +236,14 @@ func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.Hand } if err := tx.Commit(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } // Signed in immediately: the alternative is a form that says "now go // and log in", which is the same credential typed twice. if err := startSession(w, r, db, userID, publicURL); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } user, _ := fetchUser(r.Context(), db, userID) @@ -291,7 +291,7 @@ func handleListInvites(db *sql.DB) http.HandlerFunc { WHERE team_id = $1 ORDER BY id DESC`, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -303,7 +303,7 @@ func handleListInvites(db *sql.DB) http.HandlerFunc { var revoked *int64 if err := rows.Scan(&i.ID, &i.TeamID, &i.Role, &created, &expires, &i.MaxUses, &i.Uses, &revoked); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } i.CreatedAt = time.Unix(created, 0).UTC() @@ -312,7 +312,7 @@ func handleListInvites(db *sql.DB) http.HandlerFunc { out = append(out, i) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, out) @@ -356,7 +356,7 @@ func handleCreateInvite(db *sql.DB, publicURL string) http.HandlerFunc { raw, hash, err := randomToken() if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } // created_by is nullable (ON DELETE SET NULL) for exactly this @@ -382,7 +382,7 @@ func handleCreateInvite(db *sql.DB, publicURL string) http.HandlerFunc { RETURNING id, team_id, role, created_at, expires_at, max_uses, uses`, hash, teamID, req.Role, createdBy, expires.Unix(), req.MaxUses). Scan(&out.ID, &out.TeamID, &out.Role, &created, &expiresAt, &out.MaxUses, &out.Uses); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } out.CreatedAt = time.Unix(created, 0).UTC() @@ -412,7 +412,7 @@ func handleRevokeInvite(db *sql.DB) http.HandlerFunc { "UPDATE invites SET revoked_at = "+nowEpoch+ " WHERE id = $1 AND team_id = $2 AND revoked_at IS NULL", id, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -450,7 +450,7 @@ func handleTestNotification(cfg NotifyConfig, db *sql.DB) http.HandlerFunc { var topic *string if err := db.QueryRowContext(r.Context(), "SELECT ntfy_topic FROM users WHERE id = $1", caller.ID).Scan(&topic); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if topic == nil || *topic == "" { @@ -501,7 +501,7 @@ func handleDismissOnboarding(db *sql.DB) http.HandlerFunc { "UPDATE users SET onboarding_dismissed_at = NULL WHERE id = $1", caller.ID) } if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } w.WriteHeader(http.StatusNoContent) diff --git a/internal/api/similar.go b/internal/api/similar.go index b316289..c13f400 100644 --- a/internal/api/similar.go +++ b/internal/api/similar.go @@ -37,7 +37,7 @@ func handleIncidentSimilar(db *sql.DB) http.HandlerFunc { out, err := similarIncidents(r.Context(), db, id, limit) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, out) diff --git a/internal/api/stats.go b/internal/api/stats.go index ee30af9..a9ab4da 100644 --- a/internal/api/stats.go +++ b/internal/api/stats.go @@ -24,7 +24,7 @@ func handleStatsAlerts(db *sql.DB) http.HandlerFunc { FROM alerts WHERE %s`, where), args.all()..., ).Scan(&total, &firing, &resolved) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, map[string]int64{ @@ -55,7 +55,7 @@ func handleStatsTop(db *sql.DB) http.HandlerFunc { ORDER BY cnt DESC LIMIT %s`, where, args.add(limit)), args.all()...) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -68,13 +68,13 @@ func handleStatsTop(db *sql.DB) http.HandlerFunc { for rows.Next() { var e entry if err := rows.Scan(&e.Name, &e.Count); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } result = append(result, e) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, result) @@ -93,7 +93,7 @@ func handleStatsByHour(db *sql.DB) http.HandlerFunc { GROUP BY hr ORDER BY hr ASC`, where), args.all()...) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -103,13 +103,13 @@ func handleStatsByHour(db *sql.DB) http.HandlerFunc { var hr int var cnt int64 if err := rows.Scan(&hr, &cnt); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } counts[hr] = cnt } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -129,8 +129,7 @@ func handleStatsByDay(db *sql.DB) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { where, args := statsFilter(r.URL.Query(), "received_at", callerTeamIDs(r.Context())) - // Postgres EXTRACT(DOW …) → 0=Sunday … 6=Saturday, the same numbering - // SQLite's strftime('%w') returned, so the frontend needs no change. + // Postgres EXTRACT(DOW …) → 0=Sunday … 6=Saturday. rows, err := db.QueryContext(r.Context(), fmt.Sprintf(` SELECT EXTRACT(DOW FROM to_timestamp(received_at) AT TIME ZONE 'UTC')::int AS dow, COUNT(*) AS cnt @@ -139,7 +138,7 @@ func handleStatsByDay(db *sql.DB) http.HandlerFunc { GROUP BY dow ORDER BY dow ASC`, where), args.all()...) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -149,13 +148,13 @@ func handleStatsByDay(db *sql.DB) http.HandlerFunc { var dow int var cnt int64 if err := rows.Scan(&dow, &cnt); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } counts[dow] = cnt } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -198,7 +197,7 @@ func handleStatsIncidents(db *sql.DB) http.HandlerFunc { FROM incidents WHERE %s`, where), args.all()..., ).Scan(&total, &triggered, &acknowledged, &resolved, &mtta, &mttr) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } diff --git a/internal/api/teams.go b/internal/api/teams.go index e2b17c7..e92d7d9 100644 --- a/internal/api/teams.go +++ b/internal/api/teams.go @@ -13,19 +13,12 @@ import ( "github.com/go-chi/chi/v5" ) -// handleListTeams lists the caller's own teams, each with their role in it, -// or — with ?name= — looks up one team by exact name regardless of caller -// identity (TEAM-LOOKUP.md). An administrator listing every team goes -// through the admin endpoint instead: the no-name case here answers "what am -// I part of", which is what the UI's team filter and the combined queue are -// built from. +// handleListTeams lists the caller's own teams, each with their role in it. +// An administrator listing every team goes through the admin endpoint instead: +// this answers "what am I part of", which is what the UI's team filter and the +// combined queue are built from. func handleListTeams(db *sql.DB) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { - if name := strings.TrimSpace(r.URL.Query().Get("name")); name != "" { - handleListTeamsByName(db, w, r, name) - return - } - caller, _ := userFromContext(r.Context()) rows, err := db.QueryContext(r.Context(), ` SELECT t.id, t.name, t.created_at, m.role, m.source @@ -34,7 +27,7 @@ func handleListTeams(db *sql.DB) http.HandlerFunc { WHERE m.user_id = $1 ORDER BY t.name`, caller.ID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -44,45 +37,20 @@ func handleListTeams(db *sql.DB) http.HandlerFunc { var t models.Team var created int64 if err := rows.Scan(&t.ID, &t.Name, &created, &t.Role, &t.Source); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } t.CreatedAt = time.Unix(created, 0).UTC() teams = append(teams, t) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, teams) } } -// handleListTeamsByName answers "is there a team named exactly this", open to -// any authenticated caller including a service account (TEAM-LOOKUP.md) — -// mirrors handleListServiceAccounts' own ?name= lookup: a one-or-zero-length -// array, never an error on no match, and no caller-identity filtering at -// all, since what it discloses (a name is taken, nothing about who's in it -// or any of its data) is the same low sensitivity that lookup already -// accepts for service-account names. -func handleListTeamsByName(db *sql.DB, w http.ResponseWriter, r *http.Request, name string) { - var t models.Team - var created int64 - err := db.QueryRowContext(r.Context(), - "SELECT id, name, created_at FROM teams WHERE name = $1", name, - ).Scan(&t.ID, &t.Name, &created) - if errors.Is(err, sql.ErrNoRows) { - respond(w, http.StatusOK, []models.Team{}) - return - } - if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) - return - } - t.CreatedAt = time.Unix(created, 0).UTC() - respond(w, http.StatusOK, []models.Team{t}) -} - // handleUserTeams lists one user's teams, for the admin page's per-user view: // "what is this person in", which /api/teams cannot answer because it is always // about the caller. @@ -106,7 +74,7 @@ func handleUserTeams(db *sql.DB) http.HandlerFunc { var exists bool if err := db.QueryRowContext(r.Context(), "SELECT EXISTS (SELECT 1 FROM users WHERE id = $1)", id).Scan(&exists); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if !exists { @@ -121,7 +89,7 @@ func handleUserTeams(db *sql.DB) http.HandlerFunc { WHERE m.user_id = $1 ORDER BY t.name`, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -131,14 +99,14 @@ func handleUserTeams(db *sql.DB) http.HandlerFunc { var t models.Team var created int64 if err := rows.Scan(&t.ID, &t.Name, &created, &t.Role, &t.Source); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } t.CreatedAt = time.Unix(created, 0).UTC() teams = append(teams, t) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, teams) @@ -161,6 +129,12 @@ func handleCreateTeam(db *sql.DB) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { var req struct { Name string `json:"name"` + // ExternalID makes the call idempotent for automation: a team + // already carrying it is returned as-is (200) instead of created, so + // a client that crashed between the POST and recording the id finds + // its own team again. Instance-scoped service accounts only; a name + // that belongs to a different team is still a 409. + ExternalID string `json:"external_id"` } if err := decodeJSON(r, &req); err != nil { respond(w, http.StatusBadRequest, errResp("invalid request body")) @@ -173,6 +147,10 @@ func handleCreateTeam(db *sql.DB) http.HandlerFunc { } caller, isUser := userFromContext(r.Context()) + if req.ExternalID != "" && !isInstanceServiceAccount(r.Context()) { + respond(w, http.StatusForbidden, errResp("external_id is for instance-scoped service accounts")) + return + } if !isUser && !isInstanceServiceAccount(r.Context()) { // A team-scoped service account authenticates as owner of exactly // one team already (see serveAsServiceAccount); letting it create @@ -183,37 +161,57 @@ func handleCreateTeam(db *sql.DB) http.HandlerFunc { tx, err := db.BeginTx(r.Context(), nil) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer tx.Rollback() //nolint:errcheck var team models.Team var created int64 + if req.ExternalID != "" { + err := tx.QueryRowContext(r.Context(), + "SELECT id, name, created_at FROM teams WHERE external_id = $1", req.ExternalID). + Scan(&team.ID, &team.Name, &created) + if err == nil { + team.ExternalID = &req.ExternalID + team.CreatedAt = time.Unix(created, 0).UTC() + respond(w, http.StatusOK, team) + return + } + if !errors.Is(err, sql.ErrNoRows) { + serverError(w, r, err) + return + } + } + var externalID *string + if req.ExternalID != "" { + externalID = &req.ExternalID + } if err := tx.QueryRowContext(r.Context(), - "INSERT INTO teams (name) VALUES ($1) RETURNING id, name, created_at", - req.Name).Scan(&team.ID, &team.Name, &created); err != nil { + "INSERT INTO teams (name, external_id) VALUES ($1, $2) RETURNING id, name, created_at", + req.Name, externalID).Scan(&team.ID, &team.Name, &created); err != nil { if isUniqueViolation(err) { respond(w, http.StatusConflict, errResp("a team with that name already exists")) return } - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if isUser { if _, err := tx.ExecContext(r.Context(), "INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3)", team.ID, caller.ID, models.RoleOwner); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } } if err := tx.Commit(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } team.CreatedAt = time.Unix(created, 0).UTC() + team.ExternalID = externalID if isUser { team.Role = models.RoleOwner } @@ -241,7 +239,7 @@ func handleDeleteTeam(db *sql.DB) http.HandlerFunc { if err := db.QueryRowContext(r.Context(), "SELECT COUNT(*) FROM incidents WHERE team_id = $1 AND resolved_at IS NULL", teamID). Scan(&open); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if open > 0 { @@ -251,7 +249,7 @@ func handleDeleteTeam(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "DELETE FROM teams WHERE id = $1", teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -322,7 +320,7 @@ func handleListTeamMembers(db *sql.DB) http.HandlerFunc { WHERE m.team_id = $1 ORDER BY u.username`, teamID, todayUTC()) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -334,7 +332,7 @@ func handleListTeamMembers(db *sql.DB) http.HandlerFunc { var hasTopic, disabled bool if err := rows.Scan(&m.TeamID, &m.UserID, &m.Username, &m.Role, &joined, &m.Source, &hasTopic, &disabled, &lastActive, &m.OnCall, &m.NextShift); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } m.JoinedAt = time.Unix(joined, 0).UTC() @@ -360,7 +358,7 @@ func handleListTeamMembers(db *sql.DB) http.HandlerFunc { members = append(members, m) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, members) @@ -396,7 +394,7 @@ func handleAddTeamMember(db *sql.DB) http.HandlerFunc { } if managed, err := isSSOManagedMember(r.Context(), db, teamID, req.UserID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } else if managed { respond(w, http.StatusConflict, errResp(ssoManagedMsg)) @@ -408,7 +406,7 @@ func handleAddTeamMember(db *sql.DB) http.HandlerFunc { if req.Role == models.RoleMember { last, err := isLastTeamOwner(r.Context(), db, teamID, req.UserID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if last { @@ -453,7 +451,7 @@ func handleRemoveTeamMember(db *sql.DB) http.HandlerFunc { } if managed, err := isSSOManagedMember(r.Context(), db, teamID, userID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } else if managed { respond(w, http.StatusConflict, errResp(ssoManagedMsg)) @@ -462,7 +460,7 @@ func handleRemoveTeamMember(db *sql.DB) http.HandlerFunc { last, err := isLastTeamOwner(r.Context(), db, teamID, userID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if last { @@ -473,7 +471,7 @@ func handleRemoveTeamMember(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "DELETE FROM team_members WHERE team_id = $1 AND user_id = $2", teamID, userID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -572,7 +570,7 @@ func handleListIntegrations(db *sql.DB) http.HandlerFunc { WHERE i.team_id = $1 ORDER BY i.id`, teamID, now.Add(-sourceQuietAfter).Unix()) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -584,7 +582,7 @@ func handleListIntegrations(db *sql.DB) http.HandlerFunc { var lastUsed, lastAlert *int64 if err := rows.Scan(&i.ID, &i.TeamID, &i.Kind, &i.Name, &created, &lastUsed, &lastAlert, &i.Alerts24h); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } i.CreatedAt = time.Unix(created, 0).UTC() @@ -601,7 +599,7 @@ func handleListIntegrations(db *sql.DB) http.HandlerFunc { integrations = append(integrations, i) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, integrations) @@ -644,7 +642,7 @@ func handleCreateIntegration(db *sql.DB, publicURL string) http.HandlerFunc { raw, hash, err := randomToken() if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } @@ -656,7 +654,11 @@ func handleCreateIntegration(db *sql.DB, publicURL string) http.HandlerFunc { RETURNING id, team_id, kind, name, created_at`, teamID, req.Kind, req.Name, hash). Scan(&i.ID, &i.TeamID, &i.Kind, &i.Name, &created); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + if isUniqueViolation(err) { + respond(w, http.StatusConflict, errResp("an integration with that name already exists in this team")) + return + } + serverError(w, r, err) return } i.CreatedAt = time.Unix(created, 0).UTC() @@ -703,7 +705,11 @@ func handleRenameIntegration(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "UPDATE integrations SET name = $1 WHERE id = $2 AND team_id = $3", req.Name, id, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + if isUniqueViolation(err) { + respond(w, http.StatusConflict, errResp("an integration with that name already exists in this team")) + return + } + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -732,7 +738,7 @@ func handleDeleteIntegration(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "DELETE FROM integrations WHERE id = $1 AND team_id = $2", id, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -790,16 +796,6 @@ func teamParam(w http.ResponseWriter, r *http.Request) (int64, bool) { return id, true } -// defaultTeamID is the oldest team, which on an upgraded install is the -// "Default" team every pre-teams row was moved into and on a fresh one is the -// team migration 003 creates. Bootstrap puts the first user in it, so somebody -// signing in to a new server lands somewhere rather than in no team at all. -func defaultTeamID(ctx context.Context, db *sql.DB) (int64, error) { - var id int64 - err := db.QueryRowContext(ctx, "SELECT id FROM teams ORDER BY id LIMIT 1").Scan(&id) - return id, err -} - // --------------------------------------------------------------------------- // A team's dead man's switches // --------------------------------------------------------------------------- @@ -833,12 +829,12 @@ func handleListTeamDeadman(db *sql.DB) http.HandlerFunc { set, err := deadmanSetForTeam(r.Context(), db, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } out, err := deadmanStatuses(r.Context(), db, teamID, set, time.Now()) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, out) @@ -901,7 +897,11 @@ func handleCreateTeamDeadman(db *sql.DB) http.HandlerFunc { INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity) VALUES ($1, $2, $3, $4, $5) RETURNING id`, teamID, req.Name, m.config(), req.TimeoutSeconds, req.Severity).Scan(&id); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + if isUniqueViolation(err) { + respond(w, http.StatusConflict, errResp("a switch with that name already exists in this team")) + return + } + serverError(w, r, err) return } @@ -975,7 +975,11 @@ func handleUpdateTeamDeadman(db *sql.DB) http.HandlerFunc { WHERE id = $5 AND team_id = $6`, req.Name, m.config(), req.TimeoutSeconds, req.Severity, switchID, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + if isUniqueViolation(err) { + respond(w, http.StatusConflict, errResp("a switch with that name already exists in this team")) + return + } + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -990,12 +994,12 @@ func handleUpdateTeamDeadman(db *sql.DB) http.HandlerFunc { // handleListTeamDeadman would give it. set, err := deadmanSetForTeam(r.Context(), db, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } statuses, err := deadmanStatuses(r.Context(), db, teamID, set, time.Now()) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } for _, s := range statuses { @@ -1004,7 +1008,7 @@ func handleUpdateTeamDeadman(db *sql.DB) http.HandlerFunc { return } } - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) } } @@ -1029,7 +1033,7 @@ func handleDeleteTeamDeadman(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "DELETE FROM deadman_switches WHERE id = $1 AND team_id = $2", switchID, teamID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { diff --git a/internal/api/teams_test.go b/internal/api/teams_test.go index 8cebe1a..85879f9 100644 --- a/internal/api/teams_test.go +++ b/internal/api/teams_test.go @@ -6,8 +6,6 @@ import ( "io" "net/http" "testing" - - "git.ryuvia.com/niklas/terdut-server/internal/models" ) // The whole point of #4: two teams sharing one server must not see each other's @@ -146,7 +144,6 @@ func TestTeams_IncidentsAreScopedToTheReceivingTeam(t *testing.T) { otherID := int64(blueIncidents[0]["id"].(float64)) for _, path := range []string{ "/api/incidents/" + id64(otherID), - "/api/incidents/" + id64(otherID) + "/alerts", "/api/incidents/" + id64(otherID) + "/timeline", } { resp := red.call(http.MethodGet, path, nil) @@ -461,70 +458,58 @@ func TestTeams_OutsiderSeesNothing(t *testing.T) { // GET /api/teams?name= (TEAM-LOOKUP.md) // --------------------------------------------------------------------------- -func TestListTeamsByName_FindsExactMatch(t *testing.T) { - s := newTS(t) - instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0) - teamID := createTeamAs(t, s, instanceKey, "platform") - - teams := list(t, s.reqAs(t, instanceKey, http.MethodGet, "/api/teams?name=platform", nil)) - if len(teams) != 1 { - t.Fatalf("expected exactly one match for ?name=platform, got %d: %v", len(teams), teams) - } - if int64(teams[0]["id"].(float64)) != teamID { - t.Errorf("id = %v, want %d", teams[0]["id"], teamID) - } - // No membership, so no role to report (models.Team's own doc comment: - // "empty when nobody in particular is asking"). - if _, has := teams[0]["role"]; has { - t.Errorf("expected no role on a name-lookup match, got %v", teams[0]["role"]) - } -} - -func TestListTeamsByName_NoMatchIsAnEmptyArrayNotAnError(t *testing.T) { - s := newTS(t) - instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0) - - resp := s.reqAs(t, instanceKey, http.MethodGet, "/api/teams?name=does-not-exist", nil) - if resp.StatusCode != http.StatusOK { - t.Fatalf("expected 200 on no match, got %d", resp.StatusCode) - } - teams := list(t, resp) - if len(teams) != 0 { - t.Errorf("expected an empty array, got %v", teams) - } -} - -// The actual motivating scenario (TEAM-LOOKUP.md): a service account that -// already created a team, interrupted before it could remember the id, -// recovers it via ?name= on the same name its own POST 409s on. -func TestListTeamsByName_RecoversAfterCreateConflict(t *testing.T) { - s := newTS(t) - instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0) - original := createTeamAs(t, s, instanceKey, "recovered") - - conflict := s.reqAs(t, instanceKey, http.MethodPost, "/api/teams", map[string]string{"name": "recovered"}) - if conflict.StatusCode != http.StatusConflict { - t.Fatalf("expected 409 recreating the same name, got %d", conflict.StatusCode) - } - conflict.Body.Close() - - teams := list(t, s.reqAs(t, instanceKey, http.MethodGet, "/api/teams?name=recovered", nil)) - if len(teams) != 1 || int64(teams[0]["id"].(float64)) != original { - t.Fatalf("expected to recover the original team %d via ?name=, got %v", original, teams) - } -} - -// Not gated by isInstanceServiceAccount or AdminOnly (TEAM-LOOKUP.md): any -// authenticated caller may ask whether a name is taken, the same low -// sensitivity GET /api/service-accounts?name= already accepts. -func TestListTeamsByName_OpenToAnyAuthenticatedCaller(t *testing.T) { +// Everybody signed in can list users to name them, but only an admin (or the +// row's owner) sees an email or an ntfy topic, which is a publish secret. +func TestListUsers_RedactsEmailAndTopicForOthers(t *testing.T) { s := newTS(t) red := newTeam(t, s, "red") - _ = createTeamAs(t, s, s.key, "blue-target") + _ = newTeam(t, s, "blue") + s.exec(t, "UPDATE users SET ntfy_topic = 'secret-topic'") - // red's own member, not a member of "blue-target", still gets a match. - teams := list(t, red.call(http.MethodGet, "/api/teams?name=blue-target", nil)) - if len(teams) != 1 || teams[0]["name"] != "blue-target" { - t.Errorf("expected a non-member caller to still find the team by name, got %v", teams) + var asAdmin []map[string]any + decode(t, s.req(t, http.MethodGet, "/api/users", nil), &asAdmin) + for _, u := range asAdmin { + if u["email"] == "" || u["ntfy_topic"] != "secret-topic" { + t.Errorf("admin should see everything, got %v", u) + } + } + + asMember := list(t, red.call(http.MethodGet, "/api/users", nil)) + if len(asMember) < 3 { + t.Fatalf("expected the whole user list, got %v", asMember) + } + for _, u := range asMember { + own := u["username"] == "red-user" + if own != (u["email"] != "") || own != (u["ntfy_topic"] != nil) { + t.Errorf("only red-user's own row should keep email and topic, got %v", u) + } + } +} + +// Names identify integrations and switches within a team. +func TestTeamNames_AreUniquePerTeam(t *testing.T) { + s := newTS(t) // creates one integration named "test" in the default team + + dup := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/integrations", map[string]string{"name": "test"}) + dup.Body.Close() + if dup.StatusCode != http.StatusConflict { + t.Errorf("duplicate integration name: expected 409, got %d", dup.StatusCode) + } + + body := map[string]any{"matcher": "alertname=Watchdog", "timeout_seconds": 60, "severity": "critical"} + first := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches", body) + first.Body.Close() + second := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches", body) + second.Body.Close() + if first.StatusCode != http.StatusCreated || second.StatusCode != http.StatusConflict { + t.Errorf("duplicate switch name: expected 201 then 409, got %d then %d", first.StatusCode, second.StatusCode) + } + + // The same name in another team is fine. + other := newTeam(t, s, "elsewhere") + ok := s.req(t, http.MethodPost, "/api/teams/"+id64(other.id)+"/integrations", map[string]string{"name": "test"}) + ok.Body.Close() + if ok.StatusCode != http.StatusCreated { + t.Errorf("same name in another team: expected 201, got %d", ok.StatusCode) } } diff --git a/internal/api/testdb_test.go b/internal/api/testdb_test.go index 33d6294..25852e3 100644 --- a/internal/api/testdb_test.go +++ b/internal/api/testdb_test.go @@ -13,9 +13,8 @@ import ( "git.ryuvia.com/niklas/terdut-server/internal/db" ) -// Tests run against a real Postgres, because the server does. SQLite's -// ":memory:" gave every test a private database for free; Postgres has no -// equivalent, so isolation is bought with a schema per test. +// Tests run against a real Postgres, because the server does. Isolation is +// bought with a schema per test. // // A schema rather than a database: CREATE DATABASE copies a template on disk and // costs a hundred milliseconds or so each time, while CREATE SCHEMA plus the one diff --git a/internal/api/users.go b/internal/api/users.go index aa457a2..ba3f7f7 100644 --- a/internal/api/users.go +++ b/internal/api/users.go @@ -15,6 +15,10 @@ import ( "github.com/go-chi/chi/v5" ) +// bootstrapLockKey is the transaction-scoped advisory lock handleBootstrap +// holds; distinct from the migration and notifier keys. +const bootstrapLockKey = 0x7465726475744254 + func handleBootstrap(db *sql.DB) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { var req struct { @@ -40,15 +44,30 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc { } h, err := hashPassword(req.Password) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } passwordHash = &h } + // Check-then-insert has to be one atomic step: two concurrent calls on + // an empty install would otherwise both see zero users and both create + // an admin. The transaction-scoped lock serialises them, and the loser + // sees the winner's row. + tx, err := db.BeginTx(r.Context(), nil) + if err != nil { + serverError(w, r, err) + return + } + defer tx.Rollback() //nolint:errcheck + if _, err := tx.ExecContext(r.Context(), "SELECT pg_advisory_xact_lock($1)", bootstrapLockKey); err != nil { + serverError(w, r, err) + return + } + var count int - if err := db.QueryRowContext(r.Context(), "SELECT COUNT(*) FROM users").Scan(&count); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + if err := tx.QueryRowContext(r.Context(), "SELECT COUNT(*) FROM users").Scan(&count); err != nil { + serverError(w, r, err) return } if count > 0 { @@ -57,34 +76,29 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc { } var userID int64 - if err := db.QueryRowContext(r.Context(), + if err := tx.QueryRowContext(r.Context(), "INSERT INTO users (username, email, password_hash, is_admin) VALUES ($1, $2, $3, true) RETURNING id", req.Username, req.Email, passwordHash).Scan(&userID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } raw, hash, err := randomToken() if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } var keyID int64 - if err := db.QueryRowContext(r.Context(), + if err := tx.QueryRowContext(r.Context(), "INSERT INTO api_keys (user_id, key_hash, name) VALUES ($1, $2, $3) RETURNING id", userID, hash, "bootstrap").Scan(&keyID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } - // The default team exists from migration 003, on a fresh install too. - // Without a membership the first user signs in to a working server with - // no queue, no schedule and nowhere for an integration to hang off. - if teamID, err := defaultTeamID(r.Context(), db); err == nil { - db.ExecContext(r.Context(), //nolint:errcheck - "INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3) "+ - "ON CONFLICT (team_id, user_id) DO NOTHING", - teamID, userID, models.RoleOwner) + if err := tx.Commit(); err != nil { + serverError(w, r, err) + return } user, _ := fetchUser(r.Context(), db, userID) @@ -93,12 +107,19 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc { } } +// handleListUsers is readable by anyone signed in, because the assignment +// control and the schedule need to name people. What it returns about other +// people is therefore only what naming them takes: email and ntfy_topic are +// blanked unless the caller is an admin or the row is their own. The topic in +// particular is a publish secret. func handleListUsers(db *sql.DB) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { + caller, _ := userFromContext(r.Context()) + seeAll := caller.IsAdmin rows, err := db.QueryContext(r.Context(), "SELECT id, username, email, created_at, ntfy_topic, is_admin, admin_source, disabled_at FROM users ORDER BY id") if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -109,15 +130,19 @@ func handleListUsers(db *sql.DB) http.HandlerFunc { var ts int64 var disabled *int64 if err := rows.Scan(&u.ID, &u.Username, &u.Email, &ts, &u.NtfyTopic, &u.IsAdmin, &u.AdminSource, &disabled); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } u.CreatedAt = time.Unix(ts, 0).UTC() u.DisabledAt = unixPtr(disabled) + if !seeAll && u.ID != caller.ID { + u.Email = "" + u.NtfyTopic = nil + } users = append(users, u) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, users) @@ -147,7 +172,7 @@ func handleCreateUser(db *sql.DB) http.HandlerFunc { respond(w, http.StatusConflict, errResp("username or email already exists")) return } - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } user, _ := fetchUser(r.Context(), db, id) @@ -185,7 +210,7 @@ func handleSetNotifyTarget(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "UPDATE users SET ntfy_topic = $1 WHERE id = $2", topic, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -195,7 +220,7 @@ func handleSetNotifyTarget(db *sql.DB) http.HandlerFunc { user, err := fetchUser(r.Context(), db, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, user) @@ -217,7 +242,7 @@ func handleDeleteUser(db *sql.DB) http.HandlerFunc { return } if last, err := isLastAdmin(r.Context(), db, id); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } else if last { respond(w, http.StatusConflict, errResp("cannot delete the last administrator")) @@ -226,7 +251,7 @@ func handleDeleteUser(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "DELETE FROM users WHERE id = $1", id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } n, _ := res.RowsAffected() @@ -283,7 +308,7 @@ func handleCreateAPIKey(db *sql.DB) http.HandlerFunc { raw, hash, err := randomToken() if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } var expiresAt *int64 @@ -298,7 +323,7 @@ func handleCreateAPIKey(db *sql.DB) http.HandlerFunc { if err := db.QueryRowContext(r.Context(), "INSERT INTO api_keys (user_id, key_hash, name, expires_at) VALUES ($1, $2, $3, $4) RETURNING id", userID, hash, req.Name, expiresAt).Scan(&keyID); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } key := models.APIKey{ @@ -329,7 +354,7 @@ func handleListAPIKeys(db *sql.DB) http.HandlerFunc { `SELECT id, name, created_at, last_used_at, expires_at FROM api_keys WHERE user_id = $1 ORDER BY created_at DESC`, userID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } defer rows.Close() @@ -340,7 +365,7 @@ func handleListAPIKeys(db *sql.DB) http.HandlerFunc { var created int64 var lastUsed, expires *int64 if err := rows.Scan(&k.ID, &k.Name, &created, &lastUsed, &expires); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } k.UserID = userID @@ -350,7 +375,7 @@ func handleListAPIKeys(db *sql.DB) http.HandlerFunc { keys = append(keys, k) } if err := rows.Err(); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, keys) @@ -376,7 +401,7 @@ func handleDeleteAPIKey(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "DELETE FROM api_keys WHERE id = $1 AND user_id = $2", keyID, userID) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } n, _ := res.RowsAffected() @@ -442,7 +467,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc { if err := db.QueryRowContext(r.Context(), "SELECT EXISTS (SELECT 1 FROM users WHERE id = $1 AND is_admin AND admin_source = 'oidc')", id).Scan(&managed); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if managed { @@ -456,7 +481,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc { return } if last, err := isLastAdmin(r.Context(), db, id); err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } else if last { respond(w, http.StatusConflict, errResp("cannot revoke the last administrator")) @@ -467,7 +492,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc { res, err := db.ExecContext(r.Context(), "UPDATE users SET is_admin = $1 WHERE id = $2", *req.IsAdmin, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } if n, _ := res.RowsAffected(); n == 0 { @@ -477,7 +502,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc { user, err := fetchUser(r.Context(), db, id) if err != nil { - respond(w, http.StatusInternalServerError, errResp("internal error")) + serverError(w, r, err) return } respond(w, http.StatusOK, user) diff --git a/internal/config/config.go b/internal/config/config.go index 565638c..581bbd4 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -5,17 +5,21 @@ import ( "fmt" "net/url" "os" + "strconv" "strings" "time" ) +// MinOperatorKeyLength is the shortest TERDUT_OPERATOR_KEY accepted: it is a +// bearer credential with instance reach, so a short one is refused outright. +const MinOperatorKeyLength = 32 + type Config struct { Addr string // DSN is the Postgres connection string, e.g. // postgres://terdut:secret@host:5432/terdut?sslmode=require. Required: - // unlike the SQLite path it replaced there is no sensible default, and a - // server that silently came up against the wrong database would be worse + // there is no sensible default, and a server that silently came up against the wrong database would be worse // than one that refuses to start. DSN string @@ -26,25 +30,6 @@ type Config struct { // repeat_interval (default 4h), which is what refreshes the alert. StaleAfter time.Duration - // DeadmanMatchers selects the alerts that are heartbeats rather than - // problems: receiving one opens no incident, and the absence of one does. - // - // ";" separates matchers, "," the label conditions within one, "=" is exact - // equality — `alertname=Watchdog,cluster=prod; alertname=Heartbeat`. Every - // matcher must name an alertname. See api.ParseDeadmanConfig. - DeadmanMatchers string - - // DeadmanTimeout is how long a heartbeat may go unheard before its switch is - // declared dead. It must be *shorter* than the Alertmanager repeat_interval - // of the route carrying the heartbeat — the opposite of StaleAfter, and the - // reason a dead man's switch usually wants a route of its own. Zero disables - // dead man's switch handling entirely. - DeadmanTimeout time.Duration - - // DeadmanSeverity is the severity a dead man's switch incident opens at. - // These incidents have no member alerts to derive one from. - DeadmanSeverity string - // NtfyURL is the ntfy server push notifications are published to. Empty // disables notifications entirely. NtfyURL string @@ -70,6 +55,20 @@ type Config struct { // Config, which is what a test or a new caller builds, keeps passwords working. DisablePasswordLogin bool + // OperatorKey, when set, is the credential of the instance-scoped service + // account "terdut-operator", created or re-keyed at every start. It is how + // terdut-operator gets in without a bootstrap handshake: the operator + // generates the key, hands it to the server here, and uses it as its bearer + // token. Empty means no such account is managed. + OperatorKey string + + // TrustedProxies is how many reverse proxies sit in front of the server and + // append to X-Forwarded-For. The per-address rate limits take the client + // address that many entries from the right, because everything further left + // is whatever the client chose to send. 0 ignores the header and uses the + // connection's own address. + TrustedProxies int + // OIDC configures single sign-on. The zero value, with no Issuer, is off. OIDC OIDC @@ -133,24 +132,12 @@ func Load() Config { if addr == "" { addr = ":8080" } - deadmanMatchers := os.Getenv("TERDUT_DEADMAN_MATCHERS") - if deadmanMatchers == "" { - deadmanMatchers = "alertname=Watchdog" - } - deadmanSeverity := os.Getenv("TERDUT_DEADMAN_SEVERITY") - if deadmanSeverity == "" { - deadmanSeverity = "critical" - } return Config{ Addr: addr, DSN: os.Getenv("TERDUT_DB_DSN"), ArchiveAfter: duration("TERDUT_ARCHIVE_AFTER", 7*24*time.Hour), StaleAfter: duration("TERDUT_STALE_AFTER", 6*time.Hour), - DeadmanMatchers: deadmanMatchers, - DeadmanTimeout: duration("TERDUT_DEADMAN_TIMEOUT", 15*time.Minute), - DeadmanSeverity: deadmanSeverity, - NtfyURL: os.Getenv("TERDUT_NTFY_URL"), NtfyToken: os.Getenv("TERDUT_NTFY_TOKEN"), NtfyFallbackTopic: os.Getenv("TERDUT_NTFY_FALLBACK_TOPIC"), @@ -158,7 +145,10 @@ func Load() Config { NotifyRepeat: duration("TERDUT_NOTIFY_REPEAT", 15*time.Minute), DisablePasswordLogin: !boolean("TERDUT_PASSWORD_LOGIN", true), - OIDC: loadOIDC(), + + OperatorKey: strings.TrimSpace(os.Getenv("TERDUT_OPERATOR_KEY")), + TrustedProxies: integer("TERDUT_TRUSTED_PROXIES", 1), + OIDC: loadOIDC(), OperatorMode: boolean("TERDUT_OPERATOR_MODE", false), } @@ -187,6 +177,9 @@ func loadOIDC() OIDC { // provider would come up and then fail every login, which is harder to notice // than not starting. func (c Config) Validate() error { + if c.OperatorKey != "" && len(c.OperatorKey) < MinOperatorKeyLength { + return fmt.Errorf("TERDUT_OPERATOR_KEY must be at least %d characters", MinOperatorKeyLength) + } o := c.OIDC if !o.Enabled() { if c.DisablePasswordLogin { @@ -226,6 +219,16 @@ func str(env, def string) string { return def } +// integer reads a non-negative int env var; anything else takes the default. +func integer(env string, def int) int { + if s := os.Getenv(env); s != "" { + if n, err := strconv.Atoi(strings.TrimSpace(s)); err == nil && n >= 0 { + return n + } + } + return def +} + // list reads a comma- or space-separated env var. func list(env, def string) []string { s := os.Getenv(env) diff --git a/internal/db/db.go b/internal/db/db.go index c43bce8..5b14708 100644 --- a/internal/db/db.go +++ b/internal/db/db.go @@ -35,8 +35,7 @@ const ( // // The pool is modest on purpose: this server's concurrency comes from a handful // of HTTP handlers plus two background loops, and a cloud-native-pg instance -// sized for it has a low max_connections. It is still a pool, unlike the single -// connection SQLite forced, so the notifier no longer blocks a webhook. +// sized for it has a low max_connections. func Open(dsn string) (*sql.DB, error) { if dsn == "" { return nil, fmt.Errorf("empty DSN: set TERDUT_DB_DSN") @@ -76,9 +75,8 @@ const migrationLockKey int64 = 7265_0003 // Migrate applies every embedded migration that has not been applied yet, in // filename order, recording each in schema_migrations. // -// Each file runs inside a transaction, which SQLite's version did not do: a -// migration that failed half way used to leave the schema in whatever state it -// had reached. Postgres has transactional DDL, so the rollback is real. +// Each file runs inside a transaction, so a migration that fails half way +// leaves the schema as it was: Postgres has transactional DDL. func Migrate(db *sql.DB) error { ctx := context.Background() conn, err := db.Conn(ctx) diff --git a/internal/db/migrations/001_baseline.sql b/internal/db/migrations/001_baseline.sql deleted file mode 100644 index 2cfb54e..0000000 --- a/internal/db/migrations/001_baseline.sql +++ /dev/null @@ -1,182 +0,0 @@ --- The Postgres baseline: the schema as it stood at the end of the SQLite line, --- in one file rather than ten. --- --- The ten SQLite migrations are in git history up to the commit that introduced --- this one, and they replay against nothing here: their shape was incremental --- (columns added, then dropped again in 008) and 008's backfill rewrote data --- that a Postgres install never had. An existing SQLite database is carried over --- by scripts/sqlite-to-postgres.go, which copies rows into this schema. --- --- Two conventions inherited deliberately: --- --- * Timestamps are BIGINT unix seconds, not timestamptz. Everything in Go --- already speaks epochs, and converting was a second change riding along --- with the port. Worth revisiting on its own. --- --- * Ids are GENERATED BY DEFAULT, not ALWAYS, so the migration script can --- insert rows with their original ids and keep every foreign key intact. --- setval at the end of the copy puts the sequences past them. - -CREATE TABLE users ( - id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY, - username TEXT NOT NULL UNIQUE, - email TEXT NOT NULL UNIQUE, - created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint, - -- Where this user's notifications go. NULL means they get none; incidents - -- assigned to them fall back to the configured fallback topic. - ntfy_topic TEXT, - -- NULL means the user has no password and can only use API keys. - password_hash TEXT -); - -CREATE TABLE api_keys ( - id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY, - user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE, - key_hash TEXT NOT NULL UNIQUE, - name TEXT NOT NULL, - created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint, - last_used_at BIGINT -); - --- A session is a browser's credential, the cookie counterpart of an API key: --- only the hash of the token is stored. expires_at slides forward while the --- session is in use, so an on-call phone stays signed in. -CREATE TABLE sessions ( - id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY, - token_hash TEXT NOT NULL UNIQUE, - user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE, - created_at BIGINT NOT NULL, - last_seen_at BIGINT NOT NULL, - expires_at BIGINT NOT NULL, - user_agent TEXT -); - -CREATE INDEX idx_sessions_user ON sessions(user_id); - --- The machine-owned signal record: what Alertmanager says is true right now. --- Workflow state lives on incidents, never here, because the webhook upsert owns --- these rows and would overwrite it. -CREATE TABLE alerts ( - id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY, - fingerprint TEXT NOT NULL UNIQUE, - name TEXT NOT NULL, - status TEXT NOT NULL CHECK (status IN ('firing', 'resolved')), - labels JSONB NOT NULL DEFAULT '{}'::jsonb, - annotations JSONB NOT NULL DEFAULT '{}'::jsonb, - starts_at BIGINT NOT NULL, - ends_at BIGINT, - generator_url TEXT NOT NULL DEFAULT '', - received_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint, - archived_at BIGINT, - -- Why the alert left the firing state: 'alertmanager' when a resolved - -- webhook set it, 'expiry' when the sweeper inferred it from staleness. - resolution_source TEXT -); - -CREATE INDEX alerts_status_idx ON alerts(status); -CREATE INDEX alerts_name_idx ON alerts(name); -CREATE INDEX alerts_received_at_idx ON alerts(received_at DESC); -CREATE INDEX alerts_archived_at_idx ON alerts(archived_at); - -CREATE TABLE schedule_entries ( - id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY, - user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE, - date TEXT NOT NULL UNIQUE, -- YYYY-MM-DD; one person per day - created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint -); - -CREATE INDEX schedule_entries_date_idx ON schedule_entries(date); - --- The human work item: what people acknowledge, assign, snooze, discuss and --- resolve. Correlation uses Alertmanager's own groupKey, so incidents follow the --- group_by routing tree the operator already tuned. -CREATE TABLE incidents ( - id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY, - group_key TEXT NOT NULL, -- Alertmanager groupKey, opaque - title TEXT NOT NULL, -- rendered from group_labels - group_labels JSONB NOT NULL DEFAULT '{}'::jsonb, - status TEXT NOT NULL CHECK (status IN ('triggered', 'acknowledged', 'resolved')), - severity TEXT, -- highest `severity` label across firing members - triggered_at BIGINT NOT NULL, - acknowledged_by BIGINT REFERENCES users(id) ON DELETE SET NULL, - acknowledged_at BIGINT, - assigned_to BIGINT REFERENCES users(id) ON DELETE SET NULL, - snoozed_until BIGINT, - resolved_at BIGINT, - resolution_source TEXT, -- 'alerts' | 'manual' - archived_at BIGINT -); - --- Load-bearing: at most one OPEN incident per group_key. This is what makes --- "resolved incident + a new alert occurrence = a new incident" work, and it is --- the constraint the webhook's find-or-open lookup relies on. -CREATE UNIQUE INDEX incidents_open_group_key_idx ON incidents(group_key) WHERE resolved_at IS NULL; -CREATE INDEX incidents_status_idx ON incidents(status); -CREATE INDEX incidents_triggered_at_idx ON incidents(triggered_at DESC); -CREATE INDEX incidents_archived_at_idx ON incidents(archived_at); - --- Membership is historical, not a pointer on alerts: one alert row (one --- fingerprint) resolves and re-fires over time and belongs to a different --- incident each occurrence. -CREATE TABLE incident_alerts ( - incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE, - alert_id BIGINT NOT NULL REFERENCES alerts(id) ON DELETE CASCADE, - added_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint, - PRIMARY KEY (incident_id, alert_id) -); - -CREATE INDEX incident_alerts_alert_id_idx ON incident_alerts(alert_id); - --- The timeline. Append-only, and the only history this server keeps: alert rows --- are mutated in place, so without this there is no record that anything --- happened. Notes are events too, so one query renders the whole story. -CREATE TABLE incident_events ( - id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY, - incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE, - -- triggered | alert_added | alert_resolved | acknowledged | unacknowledged - -- | assigned | snoozed | unsnoozed | resolved | note | notified | notify_failed - type TEXT NOT NULL, - user_id BIGINT REFERENCES users(id) ON DELETE SET NULL, -- NULL = the server acted - alert_id BIGINT REFERENCES alerts(id) ON DELETE SET NULL, - detail TEXT, - created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint -); - -CREATE INDEX incident_events_incident_idx ON incident_events(incident_id, created_at); - --- Delivery is an outbox rather than an inline HTTP call: a POST made while --- holding the webhook's transaction would hold a connection open across a --- network round trip. The webhook inserts a row; the notifier goroutine --- delivers it. -CREATE TABLE notifications ( - id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY, - incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE, - -- Nullable: a notification sent to the fallback topic belongs to nobody, - -- because nobody was on call when the incident opened. - user_id BIGINT REFERENCES users(id) ON DELETE SET NULL, - topic TEXT NOT NULL, -- resolved at enqueue: who was on call then - kind TEXT NOT NULL CHECK (kind IN ('triggered', 'reminder', 'resolved')), - created_at BIGINT NOT NULL, - send_after BIGINT NOT NULL, -- retry backoff watermark - attempts BIGINT NOT NULL DEFAULT 0, - sent_at BIGINT, - last_error TEXT -- kept after the last attempt, for debugging -); - --- The delivery loop's only query: what is due and still unsent. -CREATE INDEX notifications_pending_idx ON notifications(send_after) WHERE sent_at IS NULL; --- Reminders and resolved notices both look up an incident's newest row. -CREATE INDEX notifications_incident_idx ON notifications(incident_id, id DESC); - --- A notification body is stored on the ntfy server and cached on the device, so --- a real API key must never appear in one. Each delivery mints its own token --- instead: one incident, one action, one day. -CREATE TABLE incident_ack_tokens ( - token_hash TEXT PRIMARY KEY, -- SHA-256 of the raw token, as with api_keys - incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE, - user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE, - created_at BIGINT NOT NULL, - expires_at BIGINT NOT NULL -); - -CREATE INDEX incident_ack_tokens_expires_idx ON incident_ack_tokens(expires_at); diff --git a/internal/db/migrations/001_schema.sql b/internal/db/migrations/001_schema.sql new file mode 100644 index 0000000..a7cb016 --- /dev/null +++ b/internal/db/migrations/001_schema.sql @@ -0,0 +1,587 @@ +-- Terdut Server schema. One baseline: the project has not shipped, so the +-- migration history that led here (SQLite import, a Default team, per-team +-- deadman configs later replaced by switches) is not carried. Later changes are +-- new numbered files after this one. +-- +-- Timestamps are Unix epoch seconds in BIGINT columns throughout. + +CREATE TABLE alerts ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + fingerprint text NOT NULL, + name text NOT NULL, + status text NOT NULL, + labels jsonb DEFAULT '{}'::jsonb NOT NULL, + annotations jsonb DEFAULT '{}'::jsonb NOT NULL, + starts_at bigint NOT NULL, + ends_at bigint, + generator_url text DEFAULT ''::text NOT NULL, + received_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + archived_at bigint, + resolution_source text, + team_id bigint NOT NULL, + integration_id bigint, + CONSTRAINT alerts_status_check CHECK ((status = ANY (ARRAY['firing'::text, 'resolved'::text]))) +); + +CREATE TABLE api_keys ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + user_id bigint NOT NULL, + key_hash text NOT NULL, + name text NOT NULL, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + last_used_at bigint, + expires_at bigint +); + +CREATE TABLE deadman_switches ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + team_id bigint NOT NULL, + name text NOT NULL, + matcher text NOT NULL, + timeout_seconds bigint NOT NULL, + severity text DEFAULT 'critical'::text NOT NULL, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + CONSTRAINT deadman_switches_timeout_seconds_check CHECK ((timeout_seconds > 0)) +); + +CREATE TABLE device_logins ( + device_hash text NOT NULL, + user_code text NOT NULL, + status text DEFAULT 'pending'::text NOT NULL, + user_id bigint, + expires_at bigint NOT NULL, + last_polled_at bigint DEFAULT 0 NOT NULL, + CONSTRAINT device_logins_status_check CHECK ((status = ANY (ARRAY['pending'::text, 'approved'::text, 'denied'::text]))) +); + +CREATE TABLE escalation_levels ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + team_id bigint NOT NULL, + "position" bigint NOT NULL, + timeout_seconds bigint NOT NULL, + CONSTRAINT escalation_levels_timeout_seconds_check CHECK ((timeout_seconds > 0)) +); + +CREATE TABLE escalation_policies ( + team_id bigint NOT NULL, + repeat_count bigint DEFAULT 0 NOT NULL, + fallback_topic text DEFAULT ''::text NOT NULL, + updated_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + CONSTRAINT escalation_policies_repeat_count_check CHECK (((repeat_count >= 0) AND (repeat_count <= 10))) +); + +CREATE TABLE escalation_targets ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + level_id bigint NOT NULL, + kind text NOT NULL, + user_id bigint, + CONSTRAINT escalation_targets_check CHECK ((((kind = 'user'::text) AND (user_id IS NOT NULL)) OR ((kind = 'oncall'::text) AND (user_id IS NULL)))), + CONSTRAINT escalation_targets_kind_check CHECK ((kind = ANY (ARRAY['user'::text, 'oncall'::text]))) +); + +CREATE TABLE incident_ack_tokens ( + token_hash text NOT NULL, + incident_id bigint NOT NULL, + user_id bigint NOT NULL, + created_at bigint NOT NULL, + expires_at bigint NOT NULL +); + +CREATE TABLE incident_alerts ( + incident_id bigint NOT NULL, + alert_id bigint NOT NULL, + added_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL +); + +CREATE TABLE incident_events ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + incident_id bigint NOT NULL, + type text NOT NULL, + user_id bigint, + alert_id bigint, + detail text, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + service_account_id bigint, + actor_user_id bigint, + actor_service_account_id bigint, + CONSTRAINT incident_events_actor_xor_chk CHECK (((user_id IS NULL) OR (service_account_id IS NULL))), + CONSTRAINT incident_events_assign_actor_xor_chk CHECK (((actor_user_id IS NULL) OR (actor_service_account_id IS NULL))) +); + +CREATE TABLE incidents ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + group_key text NOT NULL, + title text NOT NULL, + group_labels jsonb DEFAULT '{}'::jsonb NOT NULL, + status text NOT NULL, + severity text, + triggered_at bigint NOT NULL, + acknowledged_by bigint, + acknowledged_at bigint, + assigned_to bigint, + snoozed_until bigint, + resolved_at bigint, + resolution_source text, + archived_at bigint, + team_id bigint NOT NULL, + escalation_level bigint DEFAULT 0 NOT NULL, + escalation_level_at bigint, + escalation_round bigint DEFAULT 0 NOT NULL, + signature text DEFAULT ''::text NOT NULL, + acknowledged_by_service_account_id bigint, + CONSTRAINT incidents_ack_actor_xor_chk CHECK (((acknowledged_by IS NULL) OR (acknowledged_by_service_account_id IS NULL))), + CONSTRAINT incidents_status_check CHECK ((status = ANY (ARRAY['triggered'::text, 'acknowledged'::text, 'resolved'::text]))) +); + +CREATE TABLE integrations ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + team_id bigint NOT NULL, + kind text NOT NULL, + name text NOT NULL, + key_hash text NOT NULL, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + last_used_at bigint, + CONSTRAINT integrations_kind_check CHECK ((kind = 'alertmanager'::text)) +); + +CREATE TABLE invites ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + token_hash text NOT NULL, + team_id bigint NOT NULL, + role text NOT NULL, + created_by bigint, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + expires_at bigint NOT NULL, + max_uses bigint DEFAULT 1 NOT NULL, + uses bigint DEFAULT 0 NOT NULL, + revoked_at bigint, + CONSTRAINT invites_max_uses_check CHECK (((max_uses > 0) AND (max_uses <= 100))), + CONSTRAINT invites_role_check CHECK ((role = ANY (ARRAY['owner'::text, 'member'::text]))) +); + +CREATE TABLE notifications ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + incident_id bigint NOT NULL, + user_id bigint, + topic text NOT NULL, + kind text NOT NULL, + created_at bigint NOT NULL, + send_after bigint NOT NULL, + attempts bigint DEFAULT 0 NOT NULL, + sent_at bigint, + last_error text, + CONSTRAINT notifications_kind_check CHECK ((kind = ANY (ARRAY['triggered'::text, 'reminder'::text, 'resolved'::text, 'escalated'::text]))) +); + +CREATE TABLE oidc_logins ( + state_hash text NOT NULL, + nonce text NOT NULL, + pkce_verifier text NOT NULL, + expires_at bigint NOT NULL, + next text DEFAULT '/'::text NOT NULL +); + +CREATE TABLE rate_limit_counters ( + key text NOT NULL, + window_start bigint NOT NULL, + count integer NOT NULL +); + +CREATE TABLE schedule_entries ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + user_id bigint NOT NULL, + date text NOT NULL, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + team_id bigint NOT NULL +); + +CREATE TABLE service_account_keys ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + service_account_id bigint NOT NULL, + key_hash text NOT NULL, + name text NOT NULL, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + last_used_at bigint +); + +CREATE TABLE service_accounts ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + name text NOT NULL, + scope text NOT NULL, + team_id bigint, + created_by bigint, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + CONSTRAINT service_accounts_scope_check CHECK ((scope = ANY (ARRAY['instance'::text, 'team'::text]))), + CONSTRAINT service_accounts_scope_team_id_chk CHECK ((((scope = 'team'::text) AND (team_id IS NOT NULL)) OR ((scope = 'instance'::text) AND (team_id IS NULL)))) +); + +CREATE TABLE sessions ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + token_hash text NOT NULL, + user_id bigint NOT NULL, + created_at bigint NOT NULL, + last_seen_at bigint NOT NULL, + expires_at bigint NOT NULL, + user_agent text, + max_expires_at bigint +); + +CREATE TABLE settings ( + key text NOT NULL, + value text NOT NULL, + updated_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL +); + +CREATE TABLE team_members ( + team_id bigint NOT NULL, + user_id bigint NOT NULL, + role text NOT NULL, + joined_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + source text DEFAULT 'manual'::text NOT NULL, + CONSTRAINT team_members_role_check CHECK ((role = ANY (ARRAY['owner'::text, 'member'::text]))), + CONSTRAINT team_members_source_check CHECK ((source = ANY (ARRAY['manual'::text, 'oidc'::text]))) +); + +CREATE TABLE teams ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + name text NOT NULL, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + oidc_member_group text, + oidc_owner_group text, + -- A stable identity for a team managed by automation (terdut-operator: + -- "/" of its TerdutTeam), so it can find or recreate its own + -- team without trusting a display name. NULL for a team a person made. + external_id text +); + +CREATE TABLE user_identities ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + user_id bigint NOT NULL, + issuer text NOT NULL, + subject text NOT NULL, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + last_login_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL +); + +CREATE TABLE users ( + id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL, + username text NOT NULL, + email text NOT NULL, + created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL, + ntfy_topic text, + password_hash text, + is_admin boolean DEFAULT false NOT NULL, + disabled_at bigint, + invited_via bigint, + onboarding_dismissed_at bigint, + admin_source text DEFAULT 'manual'::text NOT NULL, + CONSTRAINT users_admin_source_check CHECK ((admin_source = ANY (ARRAY['manual'::text, 'oidc'::text]))) +); + +ALTER TABLE alerts + ADD CONSTRAINT alerts_pkey PRIMARY KEY (id); + +ALTER TABLE api_keys + ADD CONSTRAINT api_keys_key_hash_key UNIQUE (key_hash); + +ALTER TABLE api_keys + ADD CONSTRAINT api_keys_pkey PRIMARY KEY (id); + +ALTER TABLE deadman_switches + ADD CONSTRAINT deadman_switches_pkey PRIMARY KEY (id); + +ALTER TABLE device_logins + ADD CONSTRAINT device_logins_pkey PRIMARY KEY (device_hash); + +ALTER TABLE device_logins + ADD CONSTRAINT device_logins_user_code_key UNIQUE (user_code); + +ALTER TABLE escalation_levels + ADD CONSTRAINT escalation_levels_pkey PRIMARY KEY (id); + +ALTER TABLE escalation_levels + ADD CONSTRAINT escalation_levels_team_id_position_key UNIQUE (team_id, "position"); + +ALTER TABLE escalation_policies + ADD CONSTRAINT escalation_policies_pkey PRIMARY KEY (team_id); + +ALTER TABLE escalation_targets + ADD CONSTRAINT escalation_targets_pkey PRIMARY KEY (id); + +ALTER TABLE incident_ack_tokens + ADD CONSTRAINT incident_ack_tokens_pkey PRIMARY KEY (token_hash); + +ALTER TABLE incident_alerts + ADD CONSTRAINT incident_alerts_pkey PRIMARY KEY (incident_id, alert_id); + +ALTER TABLE incident_events + ADD CONSTRAINT incident_events_pkey PRIMARY KEY (id); + +ALTER TABLE incidents + ADD CONSTRAINT incidents_pkey PRIMARY KEY (id); + +ALTER TABLE integrations + ADD CONSTRAINT integrations_key_hash_key UNIQUE (key_hash); + +ALTER TABLE integrations + ADD CONSTRAINT integrations_pkey PRIMARY KEY (id); + +ALTER TABLE invites + ADD CONSTRAINT invites_pkey PRIMARY KEY (id); + +ALTER TABLE invites + ADD CONSTRAINT invites_token_hash_key UNIQUE (token_hash); + +ALTER TABLE notifications + ADD CONSTRAINT notifications_pkey PRIMARY KEY (id); + +ALTER TABLE oidc_logins + ADD CONSTRAINT oidc_logins_pkey PRIMARY KEY (state_hash); + +ALTER TABLE rate_limit_counters + ADD CONSTRAINT rate_limit_counters_pkey PRIMARY KEY (key); + +ALTER TABLE schedule_entries + ADD CONSTRAINT schedule_entries_pkey PRIMARY KEY (id); + +ALTER TABLE service_account_keys + ADD CONSTRAINT service_account_keys_key_hash_key UNIQUE (key_hash); + +ALTER TABLE service_account_keys + ADD CONSTRAINT service_account_keys_pkey PRIMARY KEY (id); + +ALTER TABLE service_accounts + ADD CONSTRAINT service_accounts_name_key UNIQUE (name); + +ALTER TABLE service_accounts + ADD CONSTRAINT service_accounts_pkey PRIMARY KEY (id); + +ALTER TABLE sessions + ADD CONSTRAINT sessions_pkey PRIMARY KEY (id); + +ALTER TABLE sessions + ADD CONSTRAINT sessions_token_hash_key UNIQUE (token_hash); + +ALTER TABLE settings + ADD CONSTRAINT settings_pkey PRIMARY KEY (key); + +ALTER TABLE team_members + ADD CONSTRAINT team_members_pkey PRIMARY KEY (team_id, user_id); + +ALTER TABLE teams + ADD CONSTRAINT teams_name_key UNIQUE (name); + +ALTER TABLE teams + ADD CONSTRAINT teams_external_id_key UNIQUE (external_id); + +ALTER TABLE teams + ADD CONSTRAINT teams_pkey PRIMARY KEY (id); + +ALTER TABLE user_identities + ADD CONSTRAINT user_identities_issuer_subject_key UNIQUE (issuer, subject); + +ALTER TABLE user_identities + ADD CONSTRAINT user_identities_pkey PRIMARY KEY (id); + +ALTER TABLE users + ADD CONSTRAINT users_email_key UNIQUE (email); + +ALTER TABLE users + ADD CONSTRAINT users_pkey PRIMARY KEY (id); + +ALTER TABLE users + ADD CONSTRAINT users_username_key UNIQUE (username); + +CREATE INDEX alerts_archived_at_idx ON alerts USING btree (archived_at); + +CREATE INDEX alerts_integration_idx ON alerts USING btree (integration_id, received_at) WHERE (integration_id IS NOT NULL); + +CREATE INDEX alerts_name_idx ON alerts USING btree (name); + +CREATE INDEX alerts_received_at_idx ON alerts USING btree (received_at DESC); + +CREATE INDEX alerts_status_idx ON alerts USING btree (status); + +CREATE UNIQUE INDEX alerts_team_fingerprint_idx ON alerts USING btree (team_id, fingerprint); + +CREATE INDEX alerts_team_received_idx ON alerts USING btree (team_id, received_at DESC); + +CREATE INDEX deadman_switches_team_idx ON deadman_switches USING btree (team_id); + +CREATE INDEX device_logins_expires_idx ON device_logins USING btree (expires_at); + +CREATE INDEX escalation_targets_level_idx ON escalation_targets USING btree (level_id); + +CREATE INDEX idx_sessions_user ON sessions USING btree (user_id); + +CREATE INDEX incident_ack_tokens_expires_idx ON incident_ack_tokens USING btree (expires_at); + +CREATE INDEX incident_alerts_alert_id_idx ON incident_alerts USING btree (alert_id); + +CREATE INDEX incident_events_actor_service_account_id_idx ON incident_events USING btree (actor_service_account_id); + +CREATE INDEX incident_events_actor_user_id_idx ON incident_events USING btree (actor_user_id); + +CREATE INDEX incident_events_incident_idx ON incident_events USING btree (incident_id, created_at); + +CREATE INDEX incident_events_service_account_id_idx ON incident_events USING btree (service_account_id); + +CREATE INDEX incidents_acknowledged_by_service_account_id_idx ON incidents USING btree (acknowledged_by_service_account_id); + +CREATE INDEX incidents_archived_at_idx ON incidents USING btree (archived_at); + +CREATE INDEX incidents_escalation_idx ON incidents USING btree (escalation_level_at) WHERE ((resolved_at IS NULL) AND (status = 'triggered'::text)); + +CREATE UNIQUE INDEX incidents_open_group_key_idx ON incidents USING btree (team_id, group_key) WHERE (resolved_at IS NULL); + +CREATE INDEX incidents_signature_idx ON incidents USING btree (team_id, signature, triggered_at DESC); + +CREATE INDEX incidents_status_idx ON incidents USING btree (status); + +CREATE INDEX incidents_team_triggered_idx ON incidents USING btree (team_id, triggered_at DESC); + +CREATE INDEX incidents_triggered_at_idx ON incidents USING btree (triggered_at DESC); + +CREATE INDEX integrations_team_idx ON integrations USING btree (team_id); + +-- A name identifies an integration (and a switch) within its team, so a client +-- that manages them declaratively can look one up by name instead of listing +-- and matching. +CREATE UNIQUE INDEX integrations_team_name_key ON integrations (team_id, name); +CREATE UNIQUE INDEX deadman_switches_team_name_key ON deadman_switches (team_id, name); + +CREATE INDEX invites_team_idx ON invites USING btree (team_id); + +CREATE INDEX notifications_incident_idx ON notifications USING btree (incident_id, id DESC); + +CREATE INDEX notifications_pending_idx ON notifications USING btree (send_after) WHERE (sent_at IS NULL); + +CREATE INDEX oidc_logins_expires_idx ON oidc_logins USING btree (expires_at); + +CREATE INDEX schedule_entries_date_idx ON schedule_entries USING btree (date); + +CREATE UNIQUE INDEX schedule_entries_team_date_idx ON schedule_entries USING btree (team_id, date); + +CREATE INDEX service_account_keys_service_account_id_idx ON service_account_keys USING btree (service_account_id); + +CREATE INDEX service_accounts_team_id_idx ON service_accounts USING btree (team_id); + +CREATE INDEX team_members_user_idx ON team_members USING btree (user_id); + +CREATE INDEX user_identities_user_idx ON user_identities USING btree (user_id); + +CREATE INDEX users_is_admin_idx ON users USING btree (is_admin) WHERE is_admin; + +ALTER TABLE alerts + ADD CONSTRAINT alerts_integration_id_fkey FOREIGN KEY (integration_id) REFERENCES integrations(id) ON DELETE SET NULL; + +ALTER TABLE alerts + ADD CONSTRAINT alerts_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE api_keys + ADD CONSTRAINT api_keys_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE; + +ALTER TABLE deadman_switches + ADD CONSTRAINT deadman_switches_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE device_logins + ADD CONSTRAINT device_logins_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE; + +ALTER TABLE escalation_levels + ADD CONSTRAINT escalation_levels_team_id_fkey FOREIGN KEY (team_id) REFERENCES escalation_policies(team_id) ON DELETE CASCADE; + +ALTER TABLE escalation_policies + ADD CONSTRAINT escalation_policies_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE escalation_targets + ADD CONSTRAINT escalation_targets_level_id_fkey FOREIGN KEY (level_id) REFERENCES escalation_levels(id) ON DELETE CASCADE; + +ALTER TABLE escalation_targets + ADD CONSTRAINT escalation_targets_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE; + +ALTER TABLE incident_ack_tokens + ADD CONSTRAINT incident_ack_tokens_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE; + +ALTER TABLE incident_ack_tokens + ADD CONSTRAINT incident_ack_tokens_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE; + +ALTER TABLE incident_alerts + ADD CONSTRAINT incident_alerts_alert_id_fkey FOREIGN KEY (alert_id) REFERENCES alerts(id) ON DELETE CASCADE; + +ALTER TABLE incident_alerts + ADD CONSTRAINT incident_alerts_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE; + +ALTER TABLE incident_events + ADD CONSTRAINT incident_events_actor_service_account_id_fkey FOREIGN KEY (actor_service_account_id) REFERENCES service_accounts(id) ON DELETE SET NULL; + +ALTER TABLE incident_events + ADD CONSTRAINT incident_events_actor_user_id_fkey FOREIGN KEY (actor_user_id) REFERENCES users(id) ON DELETE SET NULL; + +ALTER TABLE incident_events + ADD CONSTRAINT incident_events_alert_id_fkey FOREIGN KEY (alert_id) REFERENCES alerts(id) ON DELETE SET NULL; + +ALTER TABLE incident_events + ADD CONSTRAINT incident_events_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE; + +ALTER TABLE incident_events + ADD CONSTRAINT incident_events_service_account_id_fkey FOREIGN KEY (service_account_id) REFERENCES service_accounts(id) ON DELETE SET NULL; + +ALTER TABLE incident_events + ADD CONSTRAINT incident_events_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE SET NULL; + +ALTER TABLE incidents + ADD CONSTRAINT incidents_acknowledged_by_fkey FOREIGN KEY (acknowledged_by) REFERENCES users(id) ON DELETE SET NULL; + +ALTER TABLE incidents + ADD CONSTRAINT incidents_acknowledged_by_service_account_id_fkey FOREIGN KEY (acknowledged_by_service_account_id) REFERENCES service_accounts(id) ON DELETE SET NULL; + +ALTER TABLE incidents + ADD CONSTRAINT incidents_assigned_to_fkey FOREIGN KEY (assigned_to) REFERENCES users(id) ON DELETE SET NULL; + +ALTER TABLE incidents + ADD CONSTRAINT incidents_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE integrations + ADD CONSTRAINT integrations_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE invites + ADD CONSTRAINT invites_created_by_fkey FOREIGN KEY (created_by) REFERENCES users(id) ON DELETE SET NULL; + +ALTER TABLE invites + ADD CONSTRAINT invites_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE notifications + ADD CONSTRAINT notifications_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE; + +ALTER TABLE notifications + ADD CONSTRAINT notifications_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE SET NULL; + +ALTER TABLE schedule_entries + ADD CONSTRAINT schedule_entries_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE schedule_entries + ADD CONSTRAINT schedule_entries_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE; + +ALTER TABLE service_account_keys + ADD CONSTRAINT service_account_keys_service_account_id_fkey FOREIGN KEY (service_account_id) REFERENCES service_accounts(id) ON DELETE CASCADE; + +ALTER TABLE service_accounts + ADD CONSTRAINT service_accounts_created_by_fkey FOREIGN KEY (created_by) REFERENCES users(id) ON DELETE SET NULL; + +ALTER TABLE service_accounts + ADD CONSTRAINT service_accounts_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE sessions + ADD CONSTRAINT sessions_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE; + +ALTER TABLE team_members + ADD CONSTRAINT team_members_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE; + +ALTER TABLE team_members + ADD CONSTRAINT team_members_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE; + +ALTER TABLE user_identities + ADD CONSTRAINT user_identities_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE; + +ALTER TABLE users + ADD CONSTRAINT users_invited_via_fkey FOREIGN KEY (invited_via) REFERENCES invites(id) ON DELETE SET NULL; diff --git a/internal/db/migrations/002_admin_role.sql b/internal/db/migrations/002_admin_role.sql deleted file mode 100644 index 99cc095..0000000 --- a/internal/db/migrations/002_admin_role.sql +++ /dev/null @@ -1,25 +0,0 @@ --- A system administrator role, and the first thing in this server that one user --- can do and another cannot. --- --- Until now every authenticated caller could create and delete users, set --- anybody's password and mint API keys for anybody — auth.go said so in a --- comment. That was defensible with one operator and a hand-made account; it is --- not once people sign themselves up (see #7). --- --- EVERY EXISTING USER BECOMES AN ADMIN. They already hold these powers, so --- this migration changes nobody's access: it names what is already true, and --- leaves demotion as a deliberate act somebody performs afterwards. The --- alternative — promoting only user 1 — would silently strip the others, and --- could leave an install whose only admin is an account nobody has a password --- for. --- --- New users are not admins: the column defaults to false, and the only ways to --- become one are this backfill, the bootstrap endpoint, or an existing admin --- granting it. -ALTER TABLE users ADD COLUMN is_admin BOOLEAN NOT NULL DEFAULT false; - -UPDATE users SET is_admin = true; - --- The queue's assignment dropdown and the on-call schedule read every user, and --- the admin screens in #5 will filter on this. -CREATE INDEX users_is_admin_idx ON users(is_admin) WHERE is_admin; diff --git a/internal/db/migrations/003_teams.sql b/internal/db/migrations/003_teams.sql deleted file mode 100644 index ea1358d..0000000 --- a/internal/db/migrations/003_teams.sql +++ /dev/null @@ -1,103 +0,0 @@ --- Teams: the unit of tenancy. Everything a person works on now belongs to one. --- --- Until this migration the install was one shared space — every user saw every --- alert and every incident, and the Alertmanager webhook was unauthenticated, so --- anything that could reach the port could open an incident for everybody. --- --- The shape, in one paragraph: a team owns its incidents, alerts, schedule and --- integrations. A user belongs to as many teams as they like, with a role in --- each: an `owner` configures the team, a `member` works its incidents. An --- integration key is what an alert arrives on, and the key is what says which --- team the alert belongs to. --- --- EVERYTHING EXISTING MOVES INTO ONE DEFAULT TEAM, and every existing user --- becomes an owner of it. That keeps an upgrade a no-op for the people using it: --- the same queue, the same schedule, the same incidents, with a name on them. - -CREATE TABLE teams ( - id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY, - name TEXT NOT NULL UNIQUE, - created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint -); - --- role is free text with a CHECK rather than an enum, so adding a third role --- later is a migration and not a type rewrite. -CREATE TABLE team_members ( - team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE, - user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE, - role TEXT NOT NULL CHECK (role IN ('owner', 'member')), - joined_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint, - PRIMARY KEY (team_id, user_id) -); - -CREATE INDEX team_members_user_idx ON team_members(user_id); - --- How alerts get in, and the only thing that says which team they belong to. --- The key is stored as a SHA-256 hash, like api_keys and the ack tokens: a --- leaked database gives nobody the ability to post alerts. -CREATE TABLE integrations ( - id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY, - team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE, - kind TEXT NOT NULL CHECK (kind IN ('alertmanager')), - name TEXT NOT NULL, - key_hash TEXT NOT NULL UNIQUE, - created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint, - last_used_at BIGINT -); - -CREATE INDEX integrations_team_idx ON integrations(team_id); - --- --------------------------------------------------------------------------- --- The default team, and everything that already exists moving into it. --- --- Created unconditionally, even on an empty install, so there is always a team --- for the bootstrap user to land in and for the first integration to hang off. --- --------------------------------------------------------------------------- - -INSERT INTO teams (name) VALUES ('Default'); - -INSERT INTO team_members (team_id, user_id, role) -SELECT (SELECT id FROM teams WHERE name = 'Default'), id, 'owner' FROM users; - --- --------------------------------------------------------------------------- --- team_id on everything a team owns. --- --- Added nullable, backfilled, then made NOT NULL: adding a NOT NULL column with --- no default to a table with rows is rejected, and a DEFAULT pointing at the --- default team would quietly keep working after the default team is gone. --- --------------------------------------------------------------------------- - -ALTER TABLE alerts ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE; -ALTER TABLE incidents ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE; -ALTER TABLE schedule_entries ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE; - -UPDATE alerts SET team_id = (SELECT id FROM teams WHERE name = 'Default'); -UPDATE incidents SET team_id = (SELECT id FROM teams WHERE name = 'Default'); -UPDATE schedule_entries SET team_id = (SELECT id FROM teams WHERE name = 'Default'); - -ALTER TABLE alerts ALTER COLUMN team_id SET NOT NULL; -ALTER TABLE incidents ALTER COLUMN team_id SET NOT NULL; -ALTER TABLE schedule_entries ALTER COLUMN team_id SET NOT NULL; - --- --------------------------------------------------------------------------- --- The uniqueness rules were all written for one tenant, and every one of them --- is wrong now: two teams monitoring two clusters legitimately see the same --- fingerprint, the same groupKey, and want somebody on call on the same day. --- --------------------------------------------------------------------------- - -ALTER TABLE alerts DROP CONSTRAINT alerts_fingerprint_key; -CREATE UNIQUE INDEX alerts_team_fingerprint_idx ON alerts(team_id, fingerprint); - -DROP INDEX incidents_open_group_key_idx; --- Still load-bearing, now per team: at most one OPEN incident per group_key --- within a team. This is what makes "resolved incident + a new alert occurrence --- = a new incident" work, and what the webhook's find-or-open lookup relies on. -CREATE UNIQUE INDEX incidents_open_group_key_idx - ON incidents(team_id, group_key) WHERE resolved_at IS NULL; - -ALTER TABLE schedule_entries DROP CONSTRAINT schedule_entries_date_key; -CREATE UNIQUE INDEX schedule_entries_team_date_idx ON schedule_entries(team_id, date); - --- The list views all filter by team first. -CREATE INDEX alerts_team_received_idx ON alerts(team_id, received_at DESC); -CREATE INDEX incidents_team_triggered_idx ON incidents(team_id, triggered_at DESC); diff --git a/internal/db/migrations/004_team_deadman.sql b/internal/db/migrations/004_team_deadman.sql deleted file mode 100644 index 3a066fd..0000000 --- a/internal/db/migrations/004_team_deadman.sql +++ /dev/null @@ -1,39 +0,0 @@ --- Dead man's switches become a team's own configuration. --- --- They were three environment variables — TERDUT_DEADMAN_MATCHERS, _TIMEOUT and --- _SEVERITY — which made them one setting for the whole install. That was the --- last piece of the alerting path a team could not control: a team could take --- its own alerts on its own key and still not say which of them were --- heartbeats, or how long a silence had to last before somebody was paged. --- --- One row per team rather than one row per switch. The unit of monitoring is --- still the fingerprint, as it always was — two clusters sending the same --- heartbeat alertname are two independent switches — and the matcher string --- keeps the format the environment variable used, so a value can be moved from --- one to the other unchanged. --- --- No rows are seeded here: a migration cannot read the environment. The server --- inserts a row per team at startup from its own configuration, and the same --- values therefore carry forward into the first team's row without anybody --- retyping them. See seedDeadmanConfigs. -CREATE TABLE deadman_configs ( - team_id BIGINT PRIMARY KEY REFERENCES teams(id) ON DELETE CASCADE, - - -- ";" separates matchers, "," the label conditions within one, "=" is exact - -- equality: `alertname=Watchdog,cluster=prod; alertname=EdgeHeartbeat`. - -- Every matcher must name an alertname. Empty watches nothing. - matchers TEXT NOT NULL DEFAULT '', - - -- Seconds rather than a Go duration string: the column is compared and - -- arithmetic is done on it, and a value that has to be parsed before it can - -- be believed is a value that can be stored unparseable. Zero disables the - -- team's switches entirely. - timeout_seconds BIGINT NOT NULL DEFAULT 0, - - -- The severity these incidents open at. They have no member alerts to - -- derive one from, and a heartbeat's own severity label is meaningless — - -- Watchdog ships as "none". - severity TEXT NOT NULL DEFAULT 'critical', - - updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint -); diff --git a/internal/db/migrations/005_settings.sql b/internal/db/migrations/005_settings.sql deleted file mode 100644 index c9e060c..0000000 --- a/internal/db/migrations/005_settings.sql +++ /dev/null @@ -1,35 +0,0 @@ --- Settings that an administrator can change without a redeploy, and the flag --- that takes an account out of use without deleting it. --- --- Three of the server's tunables were environment variables, which meant --- changing how long an incident waits before it is paged again required editing --- a chart, merging it, and waiting for a reconcile. They are behaviour, not --- infrastructure, and the difference is who needs to change them and how often. --- --- What stays in the environment: the ntfy URL and token, the database DSN, the --- listen address and the public URL. Those are where the server is plugged in --- rather than how it behaves, they are needed before the database is open, and --- two of them are credentials. --- --- Key/value rather than a column per setting. A settings table with one row and --- a column per knob needs a migration for every new knob, and #6 and #7 will --- both add some. The cost is that values are text and the accessor has to say --- what type it wanted; settings.go does that in one place. --- --- No rows are seeded here: a migration cannot read the environment. The server --- inserts each key from its own configuration at startup, once, so an install --- that upgrades keeps exactly the behaviour it had. See SeedSettings. -CREATE TABLE settings ( - key TEXT PRIMARY KEY, - value TEXT NOT NULL, - updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint -); - --- Disabling an account rather than deleting it: the person has left, or the --- credential is suspect, and their incidents, acknowledgements and timeline --- entries must stay exactly where they are. Deleting a user nulls their --- acknowledged_by and assigned_to, which quietly rewrites history. --- --- A disabled user cannot sign in and their API keys stop working, but they are --- still a name the timeline can show and still a member of their teams. -ALTER TABLE users ADD COLUMN disabled_at BIGINT; diff --git a/internal/db/migrations/006_escalation.sql b/internal/db/migrations/006_escalation.sql deleted file mode 100644 index 245567d..0000000 --- a/internal/db/migrations/006_escalation.sql +++ /dev/null @@ -1,95 +0,0 @@ --- Escalation: page somebody else when the first person does not answer. --- --- This is the gap the whole multi-tenancy line of work was opened to close. --- Until now an unacknowledged incident re-paged the same topic every --- notify_repeat forever, which is a louder version of the same silence: if the --- person on call is asleep, has no signal, or has left, nothing else happens. --- --- Shape: one policy per team, an ordered list of levels, each level with a --- timeout and a set of targets. When a level's timeout passes and the incident --- is still triggered, the next level is paged. When the last level passes, the --- chain repeats repeat_count times, and then the team's fallback topic is paged --- once as the end of the line. --- --- A team WITHOUT a policy keeps exactly today's behaviour: page the assignee, --- then remind on the same topic. Escalation is opt-in per team, and the two --- never both run for one incident -- see enqueueReminders. -CREATE TABLE escalation_policies ( - -- One per team for now, hence the team as the key rather than an id with a - -- unique index: routing different alerts to different chains needs the - -- alert to carry something to route ON, which is a separate question. - team_id BIGINT PRIMARY KEY REFERENCES teams(id) ON DELETE CASCADE, - - -- How many extra times to run the whole chain after it has been walked - -- once. 0 means walk it once and stop at the fallback. - repeat_count BIGINT NOT NULL DEFAULT 0 CHECK (repeat_count >= 0 AND repeat_count <= 10), - - -- Where the last page goes when every level has been tried. Per team now: - -- TERDUT_NTFY_FALLBACK_TOPIC was one topic for the whole install, which in - -- a multi-team server pages the wrong people. Empty means the chain simply - -- ends. - fallback_topic TEXT NOT NULL DEFAULT '', - - updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint -); - -CREATE TABLE escalation_levels ( - id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY, - team_id BIGINT NOT NULL REFERENCES escalation_policies(team_id) ON DELETE CASCADE, - -- 1-based, dense. The API rewrites the whole ladder on every edit rather - -- than patching one rung, so there is no way to leave a gap. - position BIGINT NOT NULL, - -- How long this level has to produce an acknowledgement before the next one - -- is paged. Seconds, like every other duration in this schema. - timeout_seconds BIGINT NOT NULL CHECK (timeout_seconds > 0), - - UNIQUE (team_id, position) -); - --- Who a level pages. Either a named person, or whoever the team's rota says is --- on call today -- which is the target that keeps working when the rota --- changes and nobody remembers to edit the policy. -CREATE TABLE escalation_targets ( - id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY, - level_id BIGINT NOT NULL REFERENCES escalation_levels(id) ON DELETE CASCADE, - kind TEXT NOT NULL CHECK (kind IN ('user', 'oncall')), - -- Set for kind='user', NULL for kind='oncall'. - user_id BIGINT REFERENCES users(id) ON DELETE CASCADE, - - CHECK ((kind = 'user' AND user_id IS NOT NULL) OR (kind = 'oncall' AND user_id IS NULL)) -); - -CREATE INDEX escalation_targets_level_idx ON escalation_targets(level_id); - --- --------------------------------------------------------------------------- --- Where an incident is in its chain. --- --- On the incident rather than in a side table: it is read on every notifier --- tick alongside the incident's status, and one row per incident is exactly --- what the state is. --- --------------------------------------------------------------------------- - --- 0 means no level has been paged yet, which is the state of every incident --- that existed before escalation and of every incident in a team with no --- policy. 1 is the first level. -ALTER TABLE incidents ADD COLUMN escalation_level BIGINT NOT NULL DEFAULT 0; - --- When the current level was entered, and therefore what its timeout is --- measured from. NULL while escalation_level is 0. -ALTER TABLE incidents ADD COLUMN escalation_level_at BIGINT; - --- How many times the chain has been walked in full. Compared against the --- policy's repeat_count. -ALTER TABLE incidents ADD COLUMN escalation_round BIGINT NOT NULL DEFAULT 0; - --- The notifier's escalation query: incidents still waiting, oldest level first. -CREATE INDEX incidents_escalation_idx - ON incidents(escalation_level_at) - WHERE resolved_at IS NULL AND status = 'triggered'; - --- 'escalated' joins the outbox kinds: a page that went out because nobody --- answered the last one, which is worth telling apart from the first page and --- from a reminder when reading the timeline or debugging a delivery. -ALTER TABLE notifications DROP CONSTRAINT notifications_kind_check; -ALTER TABLE notifications ADD CONSTRAINT notifications_kind_check - CHECK (kind IN ('triggered', 'reminder', 'resolved', 'escalated')); diff --git a/internal/db/migrations/007_signup_invites.sql b/internal/db/migrations/007_signup_invites.sql deleted file mode 100644 index c615353..0000000 --- a/internal/db/migrations/007_signup_invites.sql +++ /dev/null @@ -1,49 +0,0 @@ --- Self-service sign-up, and the invite links that make it useful. --- --- Until now the only way to get an account was for somebody who already had one --- to create it, and the login page told people to "ask an admin". That is a --- workable arrangement for one operator and an impossible one for a team. --- --- An invite is a link, not an email: this server has no SMTP and adding it to --- send one message would be a new subsystem to run, secure and monitor. The --- person inviting sends the link however they already talk to the person they --- are inviting. -CREATE TABLE invites ( - id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY, - - -- SHA-256 of the raw token, like api_keys, the integration keys and the - -- acknowledgement tokens. A leaked database hands nobody an account. - token_hash TEXT NOT NULL UNIQUE, - - -- Which team the invitee lands in, and as what. An invite always names a - -- team: an account in no team sees an empty queue and can be paged by - -- nobody, which is not a state to invite somebody into. - team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE, - role TEXT NOT NULL CHECK (role IN ('owner', 'member')), - - created_by BIGINT REFERENCES users(id) ON DELETE SET NULL, - created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint, - - -- Invites expire. A link that works forever is a credential nobody - -- remembers issuing, sitting in a chat log. - expires_at BIGINT NOT NULL, - - -- Single-use by default: max_uses 1. A team onboarding six people at once - -- can raise it rather than minting six links. - max_uses BIGINT NOT NULL DEFAULT 1 CHECK (max_uses > 0 AND max_uses <= 100), - uses BIGINT NOT NULL DEFAULT 0, - - -- Revoked by hand, separately from expiry, so "this link is no longer - -- wanted" and "this link timed out" stay distinguishable in the listing. - revoked_at BIGINT -); - -CREATE INDEX invites_team_idx ON invites(team_id); - --- Who redeemed which invite. Kept after the invite is gone — the answer to "how --- did this account get here" should outlive the link that made it. -ALTER TABLE users ADD COLUMN invited_via BIGINT REFERENCES invites(id) ON DELETE SET NULL; - --- Where a person is in the first-run checklist, so it can be resumed and --- dismissed rather than nagging forever. One row per user, created on demand. -ALTER TABLE users ADD COLUMN onboarding_dismissed_at BIGINT; diff --git a/internal/db/migrations/008_incident_signature.sql b/internal/db/migrations/008_incident_signature.sql deleted file mode 100644 index 3e63b31..0000000 --- a/internal/db/migrations/008_incident_signature.sql +++ /dev/null @@ -1,23 +0,0 @@ --- Similar incidents: a signature per incident, so "has this happened before" --- is an indexed equality instead of a search. --- --- The signature is the alert name plus the group labels that identify WHAT is --- broken, minus the ones that only say WHERE it happened to run this time --- (instance, pod, ...). Two incidents with the same signature in the same team --- are the same problem for a responder's purposes. --- --- Computed in Go for new incidents (incidentSignature in incident_store.go). --- The backfill below MUST produce the same string; keep the volatile list in --- both places in step. -ALTER TABLE incidents ADD COLUMN signature TEXT NOT NULL DEFAULT ''; - -UPDATE incidents SET signature = - COALESCE(NULLIF(group_labels->>'alertname', ''), title) || '|' || - COALESCE(( - SELECT string_agg(e.k || '=' || e.v, ',' ORDER BY e.k) - FROM jsonb_each_text(incidents.group_labels) AS e(k, v) - WHERE e.k <> 'alertname' - AND e.k NOT IN ('instance', 'pod', 'pod_name', 'pod_ip', 'container', 'container_name', 'endpoint') - ), ''); - -CREATE INDEX incidents_signature_idx ON incidents(team_id, signature, triggered_at DESC); diff --git a/internal/db/migrations/009_deadman_switches.sql b/internal/db/migrations/009_deadman_switches.sql deleted file mode 100644 index e693d46..0000000 --- a/internal/db/migrations/009_deadman_switches.sql +++ /dev/null @@ -1,54 +0,0 @@ --- Dead man's switches become rows of their own. --- --- 004 kept a team's switches in one string with one timeout and one severity, --- which was enough to configure them and not enough to show them: there was no --- thing to list, nothing to hang a status on, and every switch in a team had to --- share a deadline. A row per switch gives each its own name, matcher, timeout --- and severity, and gives the Team → Switches page something to be a list of. --- --- The matcher keeps the syntax the string used, one matcher per row: --- `alertname=Watchdog,cluster=prod`. The unit of monitoring is still the --- fingerprint, so a matcher that many clusters satisfy is still one switch row --- watching several independent heartbeats. -CREATE TABLE deadman_switches ( - id BIGSERIAL PRIMARY KEY, - team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE, - - -- What the owner calls it. Defaults to the matcher when they do not say. - name TEXT NOT NULL, - - -- "," separates the label conditions, "=" is exact equality, and alertname is - -- mandatory: it is what keeps the sweeper's candidate query on an index. - matcher TEXT NOT NULL, - - -- Seconds of silence before the switch is declared dead. Never zero: a switch - -- that cannot fire is deleted, not disabled. - timeout_seconds BIGINT NOT NULL CHECK (timeout_seconds > 0), - - -- The severity its incidents open at. See 004 for why they carry their own. - severity TEXT NOT NULL DEFAULT 'critical', - - created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint -); - -CREATE INDEX deadman_switches_team_idx ON deadman_switches (team_id); - --- Carry every team's configuration over, one row per matcher. A team whose --- timeout was zero had switches turned off, which is now "no rows". -INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity) -SELECT c.team_id, btrim(m), btrim(m), c.timeout_seconds, c.severity -FROM deadman_configs c, - LATERAL regexp_split_to_table(c.matchers, ';') AS m -WHERE c.timeout_seconds > 0 - AND btrim(m) <> '' -ORDER BY c.team_id; - --- The server seeds environment defaults into teams once, and remembers that it --- did. An install that had a row per team was already seeded; without this --- marker the first start after upgrading would seed teams that had switched --- theirs off. -INSERT INTO settings (key, value) -SELECT 'deadman_seeded', '1' -WHERE EXISTS (SELECT 1 FROM deadman_configs); - -DROP TABLE deadman_configs; diff --git a/internal/db/migrations/010_alert_source.sql b/internal/db/migrations/010_alert_source.sql deleted file mode 100644 index f397fad..0000000 --- a/internal/db/migrations/010_alert_source.sql +++ /dev/null @@ -1,21 +0,0 @@ --- Which alert source an alert last arrived on. --- --- Team -> Sources shows when each source last posted, which integrations --- already knew (last_used_at, stamped on every webhook). What it could not say --- was what a source delivered: an alert never recorded the key it came in on, so --- "prod alertmanager" and "staging alertmanager" were indistinguishable once --- inside. This column is that link, and lets the page show each source's last --- alert and how many alerts it has kept fresh over the past day. --- --- Last sender wins: every accepted payload restamps it, the way it advances --- received_at. Two sources posting the same fingerprint into one team is --- already one alert, and it is attributed to whichever spoke last. --- --- Nullable, and not backfilled. Alerts that arrived before this migration have --- no source, and NULL says so honestly rather than guessing. It heals by itself: --- Alertmanager re-sends every alert each repeat_interval, and each re-send is an --- accepted payload. Deleting a source keeps its alerts, unattributed. -ALTER TABLE alerts ADD COLUMN integration_id BIGINT REFERENCES integrations(id) ON DELETE SET NULL; - -CREATE INDEX alerts_integration_idx ON alerts (integration_id, received_at) - WHERE integration_id IS NOT NULL; diff --git a/internal/db/migrations/011_oidc.sql b/internal/db/migrations/011_oidc.sql deleted file mode 100644 index 557d72c..0000000 --- a/internal/db/migrations/011_oidc.sql +++ /dev/null @@ -1,60 +0,0 @@ --- Single sign-on through an OpenID Connect provider (Authentik, and anything --- else that speaks OIDC). --- --- Four things change, and none of them touches a password user: every new column --- has a default that says "this is how it has always worked". --- --- 1. user_identities says which provider account a user is. It is keyed on --- (issuer, subject), never on email or username: those are mutable at the --- provider, and a recycled address must not inherit somebody's account. A --- user can have several identities (a second provider later), and none at all --- (a local, password-only user), which is why this is a table and not two --- columns on users. --- --- 2. team_members.source and users.admin_source record who granted a role. 'oidc' --- rows are owned by the group sync: it adds them when a group grants access --- and removes them when it stops, and nothing else may edit them. 'manual' rows --- are everything that existed before this migration, and are never touched by --- the sync. Without the marker the sync could not tell a membership it created --- from one an owner added by hand, and would have to either leave stale access --- behind or delete people it had no business deleting. --- --- 3. sessions.max_expires_at is a hard ceiling on a session's life. Ordinary --- sessions slide for as long as they are used; a session made by an SSO login --- must not, because the login is the only moment the groups are re-read. --- Capping the session is what makes "removed from the group in the provider" --- take effect within a bounded time. NULL means no ceiling. --- --- 4. oidc_logins holds a login that has been started and not yet finished: the --- state, nonce and PKCE verifier the callback must see again. A row rather --- than a signed cookie, so it survives a restart and needs no signing key. --- Only the hash of the state is stored, like every other token here; the --- nonce and verifier are useless without the state that names the row. -CREATE TABLE user_identities ( - id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY, - user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE, - issuer TEXT NOT NULL, - subject TEXT NOT NULL, - created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint, - last_login_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint, - UNIQUE (issuer, subject) -); - -CREATE INDEX user_identities_user_idx ON user_identities (user_id); - -ALTER TABLE team_members - ADD COLUMN source TEXT NOT NULL DEFAULT 'manual' CHECK (source IN ('manual', 'oidc')); - -ALTER TABLE users - ADD COLUMN admin_source TEXT NOT NULL DEFAULT 'manual' CHECK (admin_source IN ('manual', 'oidc')); - -ALTER TABLE sessions ADD COLUMN max_expires_at BIGINT; - -CREATE TABLE oidc_logins ( - state_hash TEXT PRIMARY KEY, - nonce TEXT NOT NULL, - pkce_verifier TEXT NOT NULL, - expires_at BIGINT NOT NULL -); - -CREATE INDEX oidc_logins_expires_idx ON oidc_logins (expires_at); diff --git a/internal/db/migrations/012_device_login.sql b/internal/db/migrations/012_device_login.sql deleted file mode 100644 index 1764e62..0000000 --- a/internal/db/migrations/012_device_login.sql +++ /dev/null @@ -1,40 +0,0 @@ --- Signing in from a terminal, for clients that cannot open a browser on the --- machine they run on (the TUI over SSH is the reason). --- --- The flow is the OAuth device authorization grant, run by terdut itself rather --- than the identity provider, so the terminal never talks to the provider and --- the server issues its ordinary session at the end: --- --- 1. The terminal asks for a login and gets two secrets: a device code it --- keeps and polls with, and a short user code it shows the person. --- 2. The person opens the verification URL on any device, signs in by whatever --- means the server offers, sees the user code, and approves it. --- 3. The terminal's next poll finds the row approved and is given a session. --- --- Only the hash of the device code is stored, like every other token here: the --- device code is what earns a session, so a database read must not yield one. --- The user code is shown on screens and typed by people, so it is stored as is; --- on its own it can only be approved, never redeemed. --- --- user_id is the person who approved. It is empty until then, and the session --- is minted at redemption, not at approval: an approval nobody collects must not --- leave a live session lying about. --- --- last_polled_at lets the server refuse a client that polls faster than the --- interval it was told. -CREATE TABLE device_logins ( - device_hash TEXT PRIMARY KEY, - user_code TEXT NOT NULL UNIQUE, - status TEXT NOT NULL DEFAULT 'pending' CHECK (status IN ('pending', 'approved', 'denied')), - user_id BIGINT REFERENCES users(id) ON DELETE CASCADE, - expires_at BIGINT NOT NULL, - last_polled_at BIGINT NOT NULL DEFAULT 0 -); - -CREATE INDEX device_logins_expires_idx ON device_logins (expires_at); - --- Where to send the browser once a single sign-on login completes. A person who --- opens /device?code=... without a session has to sign in first and then come --- back to it, and the same is true of any other deep link. Validated when it is --- stored: only a path on this server is ever kept. -ALTER TABLE oidc_logins ADD COLUMN next TEXT NOT NULL DEFAULT '/'; diff --git a/internal/db/migrations/013_oidc_team_groups.sql b/internal/db/migrations/013_oidc_team_groups.sql deleted file mode 100644 index d269402..0000000 --- a/internal/db/migrations/013_oidc_team_groups.sql +++ /dev/null @@ -1,26 +0,0 @@ --- Per-team OIDC group configuration, replacing the global --- TERDUT_OIDC_GROUP_MAPPINGS env var. --- --- Group -> team -> role used to be one global list an operator set for the --- whole install, matched against a team by name, and the sync would create --- the team if no team by that name existed yet. That put the decision of --- which group controls a team in the server's environment rather than the --- team's own hands, meant changing it needed an env var edit and a restart, --- and let a typo in a team name silently create a stray team. --- --- Each team now names, itself, which group grants membership and which --- grants ownership. Nullable: most teams need neither. No uniqueness --- constraint on either column — two teams may legitimately watch the same --- provider group (a broad team and a narrower one both keyed off overlapping --- groups is a choice for their owners to make, not one the schema should --- refuse). --- --- BREAKING CHANGE, deliberately not auto-migrated: TERDUT_OIDC_GROUP_MAPPINGS --- stops being read as of this version, and the sync no longer creates a team --- by name. Every team's group binding must be set again through --- PUT /api/teams/{teamID}/oidc-groups. Until an owner does that, an --- OIDC-sourced membership in that team is dropped at that user's next SSO --- sign-in, the same way any other loss of group access is handled. See the --- README's OIDC section. -ALTER TABLE teams ADD COLUMN oidc_member_group TEXT; -ALTER TABLE teams ADD COLUMN oidc_owner_group TEXT; diff --git a/internal/db/migrations/014_service_accounts.sql b/internal/db/migrations/014_service_accounts.sql deleted file mode 100644 index bd93644..0000000 --- a/internal/db/migrations/014_service_accounts.sql +++ /dev/null @@ -1,43 +0,0 @@ --- Service accounts: a scoped, non-human credential for automation (e.g. --- terdut-operator) that needs to manage teams, escalation policies, dead --- man's switches, integrations and OIDC group bindings without impersonating --- a human user. See SERVICE-ACCOUNTS.md for the design this implements. --- --- Deliberately not a users row: no password_hash, no is_admin, no --- user_identities linkage, so a service account can never be pulled into --- OIDC group sync or password login, and is never mistaken for a human in an --- audit trail. --- --- scope is 'instance' (acts with the same reach system administration has --- over teams: create one, list them, mint a 'team'-scoped account against --- any of them) or 'team' (acts as that one team's owner, and nothing else). --- The CHECK ties team_id's presence to scope directly, rather than leaving it --- to application code to keep the two consistent. -CREATE TABLE service_accounts ( - id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY, - name TEXT NOT NULL UNIQUE, - scope TEXT NOT NULL CHECK (scope IN ('instance', 'team')), - team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE, - created_by BIGINT REFERENCES users(id) ON DELETE SET NULL, - created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint, - CONSTRAINT service_accounts_scope_team_id_chk CHECK ( - (scope = 'team' AND team_id IS NOT NULL) OR - (scope = 'instance' AND team_id IS NULL) - ) -); - -CREATE INDEX service_accounts_team_id_idx ON service_accounts(team_id); - --- One account, many keys: rotation is minting a new one and revoking the --- old, the same shape api_keys already has, so an account's identity and --- audit history survive a rotation instead of being recreated by it. -CREATE TABLE service_account_keys ( - id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY, - service_account_id BIGINT NOT NULL REFERENCES service_accounts(id) ON DELETE CASCADE, - key_hash TEXT NOT NULL UNIQUE, - name TEXT NOT NULL, - created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint, - last_used_at BIGINT -); - -CREATE INDEX service_account_keys_service_account_id_idx ON service_account_keys(service_account_id); diff --git a/internal/db/migrations/015_incident_service_account_actors.sql b/internal/db/migrations/015_incident_service_account_actors.sql deleted file mode 100644 index b225f34..0000000 --- a/internal/db/migrations/015_incident_service_account_actors.sql +++ /dev/null @@ -1,39 +0,0 @@ --- Service-account actors on incident mutations (terdut-server#25). A --- team-scoped service account acknowledging/resolving/snoozing/noting an --- incident is not a users row, so it cannot be written into --- acknowledged_by/incident_events.user_id — doing so either violates the --- users(id) FK (new rows) or, for incident_events.user_id, silently matches --- zero rows on delete. These columns are the service-account-shaped parallel --- to the existing human ones: nullable, mutually exclusive with their human --- counterpart, ON DELETE SET NULL so a deleted service account doesn't take --- the incident history with it. -ALTER TABLE incidents - ADD COLUMN acknowledged_by_service_account_id BIGINT - REFERENCES service_accounts(id) ON DELETE SET NULL; - -ALTER TABLE incident_events - ADD COLUMN service_account_id BIGINT - REFERENCES service_accounts(id) ON DELETE SET NULL; - --- At most one actor kind per row: both NULL ("the server acted") is valid, --- exactly one set is valid, both set is a bug this constraint refuses to --- store rather than silently accepting. -ALTER TABLE incidents - ADD CONSTRAINT incidents_ack_actor_xor_chk CHECK ( - acknowledged_by IS NULL OR acknowledged_by_service_account_id IS NULL - ); - -ALTER TABLE incident_events - ADD CONSTRAINT incident_events_actor_xor_chk CHECK ( - user_id IS NULL OR service_account_id IS NULL - ); - -CREATE INDEX incidents_acknowledged_by_service_account_id_idx - ON incidents(acknowledged_by_service_account_id); -CREATE INDEX incident_events_service_account_id_idx - ON incident_events(service_account_id); - --- assigned_to_service_account_id is deliberately not added here: it would sit --- unpopulated until handleIncidentAssign itself tracks an actor, which is a --- separate, pre-existing gap (it records the assignee today, never the --- actor, for humans either) tracked in its own follow-up issue. diff --git a/internal/db/migrations/016_rate_limit_counters.sql b/internal/db/migrations/016_rate_limit_counters.sql deleted file mode 100644 index 5553401..0000000 --- a/internal/db/migrations/016_rate_limit_counters.sql +++ /dev/null @@ -1,16 +0,0 @@ --- Backs the rate limiters (failed logins, sign-ups, OIDC/device start) with --- Postgres instead of an in-memory map, now that the server runs more than --- one replica in production (v0.37.0): a counter that only ever sees its own --- pod's traffic quietly let every one of these limits through multiplied by --- the replica count. --- --- window_start is the start of the current fixed window for key, in the same --- "unix seconds" shape every other timestamp in this schema uses. The window --- resets rather than slides, matching the in-memory limiter it replaces: --- once a key's window is older than the limiter's window length, the next --- failure starts a fresh one instead of extending the stale one. -CREATE TABLE rate_limit_counters ( - key TEXT PRIMARY KEY, - window_start BIGINT NOT NULL, - count INT NOT NULL -); diff --git a/internal/db/migrations/017_api_key_expiry.sql b/internal/db/migrations/017_api_key_expiry.sql deleted file mode 100644 index 17f8ab5..0000000 --- a/internal/db/migrations/017_api_key_expiry.sql +++ /dev/null @@ -1,7 +0,0 @@ --- Optional expiry on a user's own API keys. NULL (the existing default for --- every row already in this table) means "never expires" -- the same --- behavior these keys have always had, so no existing integration breaks. --- Service account keys are deliberately NOT touched: they are a different --- table, managed by automation, and already distinguished by their own --- "tdsa_" prefix. -ALTER TABLE api_keys ADD COLUMN expires_at BIGINT; diff --git a/internal/db/migrations/018_incident_event_actor.sql b/internal/db/migrations/018_incident_event_actor.sql deleted file mode 100644 index cfee52f..0000000 --- a/internal/db/migrations/018_incident_event_actor.sql +++ /dev/null @@ -1,21 +0,0 @@ --- Who performed an assignment (terdut-server#35). On an 'assigned' event --- incident_events.user_id is the assignee, so the actor needs columns of its --- own. Only populated for 'assigned' events; every other event type keeps --- using user_id/service_account_id for the actor. Older 'assigned' rows stay --- NULL (the actor was never recorded). Same shape as migration 015: nullable, --- mutually exclusive, ON DELETE SET NULL. --- --- assigned_to_service_account_id is still deliberately not added: making --- service accounts assignable is a separate change (request body, assignee --- picker, notifier, filters). -ALTER TABLE incident_events - ADD COLUMN actor_user_id BIGINT REFERENCES users(id) ON DELETE SET NULL, - ADD COLUMN actor_service_account_id BIGINT REFERENCES service_accounts(id) ON DELETE SET NULL; - -ALTER TABLE incident_events - ADD CONSTRAINT incident_events_assign_actor_xor_chk CHECK ( - actor_user_id IS NULL OR actor_service_account_id IS NULL - ); - -CREATE INDEX incident_events_actor_user_id_idx ON incident_events(actor_user_id); -CREATE INDEX incident_events_actor_service_account_id_idx ON incident_events(actor_service_account_id); diff --git a/internal/models/alert.go b/internal/models/alert.go index 01047e8..c6e0356 100644 --- a/internal/models/alert.go +++ b/internal/models/alert.go @@ -33,7 +33,7 @@ type Alert struct { // refreshed. The sweeper stale-dates against it (see expireStale), API // clients render it, and GET /api/alerts is ordered by it. Anything that // stops the webhook handler from advancing it on a re-send is a breaking - // change — see "received_at is a liveness heartbeat" in the README and + // change — see "received_at is a liveness heartbeat" in docs/api.md and // TestWebhook_ResendBumpsReceivedAt. ReceivedAt time.Time `json:"received_at"` @@ -52,7 +52,7 @@ type Alert struct { // inferred. Under "expiry" nothing ever reported an end, so EndsAt is only // an upper bound (see expireStale) and ReceivedAt is the more truthful // signal. Treat the value set as open — see "resolution_source says how much - // to trust ends_at" in the README, and TestWebhook_ResolvedSetsSource / + // to trust ends_at" in docs/api.md, and TestWebhook_ResolvedSetsSource / // TestExpiry_StaleFiringAlert. ResolutionSource *string `json:"resolution_source,omitempty"` diff --git a/internal/models/team.go b/internal/models/team.go index 8bedf25..0ed6365 100644 --- a/internal/models/team.go +++ b/internal/models/team.go @@ -9,6 +9,10 @@ type Team struct { Name string `json:"name"` CreatedAt time.Time `json:"created_at"` + // ExternalID identifies a team managed by automation; see handleCreateTeam. + // Shown to instance service accounts and admins only. + ExternalID *string `json:"external_id,omitempty"` + // Role is the caller's own role in this team, populated when a team is // listed for a particular person. Empty when nobody in particular is // asking, as in the admin listing. diff --git a/internal/oidc/grants.go b/internal/oidc/grants.go index 2a67a65..4de53a1 100644 --- a/internal/oidc/grants.go +++ b/internal/oidc/grants.go @@ -102,6 +102,3 @@ func rank(role string) int { } return 0 } - -// HigherRole reports whether role a outranks role b. -func HigherRole(a, b string) bool { return rank(a) > rank(b) } diff --git a/internal/web/static/js/format.js b/internal/web/static/js/format.js index 9bf994f..98be815 100644 --- a/internal/web/static/js/format.js +++ b/internal/web/static/js/format.js @@ -123,7 +123,7 @@ export function initial(name) { // convention. It comes from Prometheus's externalLabels, so it is on every // alert; an incident carries it only when it is in Alertmanager's group_by, // which is also what keeps two clusters' identical alerts from merging into one -// incident (see the README, "Several clusters, one team"). +// incident (see docs/incidents.md, "Several clusters, one team"). export const ORIGIN_LABEL = 'cluster'; export function originOf(labels) {