Compare commits
96 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 5c4e0bdd0e | |||
| 9e5b085d8b | |||
| 0050738ca0 | |||
| 42180948d1 | |||
| c83c7c2a8b | |||
| 43beda9a30 | |||
| 710521a73c | |||
| d9492913ed | |||
| 1770e5d945 | |||
| 497086cb51 | |||
| e3090d2779 | |||
| 91f03c21e8 | |||
| 4358e84b24 | |||
| f45dc2f925 | |||
| 774fdfcaa8 | |||
| fd26fef1ba | |||
| a9d788cc83 | |||
| 4b15079ac2 | |||
| fc9f47cc8d | |||
| bc9f793f1f | |||
| 871274a3a0 | |||
| 6f8499fa42 | |||
| ef731e85c5 | |||
| a4dd60f6b8 | |||
| b5573fbca2 | |||
| 0ee576f793 | |||
| b610b1817a | |||
| 949d6595ba | |||
| 33356ca978 | |||
| e5b4df7c03 | |||
| 5b4683febf | |||
| 97a4814c04 | |||
| a2dc9e3b03 | |||
| 155f27ca62 | |||
| a27ff49171 | |||
| c5be55dcbc | |||
| b2c3868619 | |||
| 36c00acf62 | |||
| 9d1df2b611 | |||
| 9bf4c92bfe | |||
| e616c82646 | |||
| 2b396d22d6 | |||
| 1f1faa437c | |||
| dc92f51cf8 | |||
| d675f8ec9b | |||
| e8d45f9d3d | |||
| f3918b863c | |||
| 3ee8583f6f | |||
| 591d5b8df0 | |||
| d2cdcc9776 | |||
| 8b2789b9b2 | |||
| 71d7e1853a | |||
| 60ebb75cd2 | |||
| 734cd9c5fd | |||
| 423ed9b3a3 | |||
| 43f004499b | |||
| 559be6de6e | |||
| e77f04b55e | |||
| e536fdd2c0 | |||
| 429d5fdda3 | |||
| 3cdd5aee1f | |||
| 6a03698f65 | |||
| 67d68ce058 | |||
| a6fa673e08 | |||
| ee22eb000c | |||
| 07914d5cdb | |||
| 7b9a337d25 | |||
| fc8b0c8d58 | |||
| 828cf87656 | |||
| ac9af8e4f5 | |||
| 8869ac864f | |||
| 0677e74cf8 | |||
| 56b8191a78 | |||
| 93761056eb | |||
| a92da7dcc0 | |||
| b39aac36b7 | |||
| 19f168ab7e | |||
| d827ceedff | |||
| 4e8c52c28c | |||
| fb927aa67b | |||
| d728af53b1 | |||
| 53e5e03f4e | |||
| 4d62c1130b | |||
| 3183e7e5c5 | |||
| 94d23a593c | |||
| b0a02c010b | |||
| 303e7a3365 | |||
| 7c87ae2af8 | |||
| 4c85e7646c | |||
| 05f82220a6 | |||
| 5227eb0d5f | |||
| 74359c72ab | |||
| 43f69272f0 | |||
| 2de5c8412d | |||
| a4fbd60441 | |||
| 1377d9005b |
@@ -16,6 +16,12 @@ RUN CGO_ENABLED=0 GOOS=${TARGETOS} GOARCH=${TARGETARCH} \
|
||||
go build -ldflags="-w -s -X main.version=${VERSION}" -o /terdut ./cmd/terdut
|
||||
|
||||
FROM scratch
|
||||
# scratch has no trust store, and a Go binary on it fails every HTTPS call with
|
||||
# "x509: certificate signed by unknown authority". Nothing needed one until single
|
||||
# sign-on: discovery and the token exchange are HTTPS calls to the identity provider.
|
||||
# The bundle is the builder's, copied by name so a missing file fails the build
|
||||
# rather than shipping an image that cannot sign anybody in.
|
||||
COPY --from=builder /etc/ssl/certs/ca-certificates.crt /etc/ssl/certs/ca-certificates.crt
|
||||
COPY --from=builder /terdut /terdut
|
||||
EXPOSE 8080
|
||||
ENTRYPOINT ["/terdut"]
|
||||
|
||||
@@ -47,12 +47,13 @@ curl -H "Authorization: Bearer $KEY" http://localhost:8080/api/users
|
||||
|
||||
The server serves a web UI at `/`: the incident queue, each incident's alerts
|
||||
and timeline with every action (acknowledge, assign, snooze, note, resolve,
|
||||
archive), who is on call, the alert feed, and changing your own password. It is
|
||||
built for a phone first. On a phone it has a bottom tab bar and a sticky action
|
||||
bar, it follows the system's dark mode, and it can be added to the home screen.
|
||||
From 900px wide it switches to a sidebar with the queue and the incident side by
|
||||
side. Schedule editing, statistics and user management remain in
|
||||
[terdut-tui](https://github.com/yeniklas/terdut-tui) for now.
|
||||
archive), who is on call, the alert feed, and an *Account* tab for your own
|
||||
password and the ntfy topic your pages go to. It is built for a phone first. On a phone
|
||||
it navigates through a hamburger menu and has a sticky action bar, it follows the
|
||||
system's dark mode, and it can be added to the home screen. From 900px wide it switches
|
||||
to a sidebar with the queue and the incident side by side. The Stats page shows
|
||||
incident counts, MTTA and MTTR, and alert frequency by name, hour and day over a
|
||||
chosen range.
|
||||
|
||||
You sign in with a username and password. Users have no password until one is
|
||||
set, and a user without one can only use API keys:
|
||||
@@ -85,6 +86,156 @@ How a browser stays signed in:
|
||||
With `TERDUT_PUBLIC_URL` set, tapping a push notification opens the incident in
|
||||
the web UI (`/incidents/{id}`).
|
||||
|
||||
A **Team** tab holds everything a team owns, in five sub-sections with a URL
|
||||
each and a strip across the top to move between them: the on-call rota
|
||||
(`/team/rota`), the membership (`/team/members`), the escalation ladder
|
||||
(`/team/escalation`), the alert sources with their keys (`/team/sources`) and
|
||||
the dead man's switches (`/team/deadman`). `/team` itself is an overview — who
|
||||
is on call today, how many members and owners, how many ladder levels, how many
|
||||
keys and how many switches — so a page fetches only what it shows. An owner
|
||||
edits it; a member sees the same pages read-only, because the server refuses
|
||||
their writes anyway. Somebody in more than one team picks between them above
|
||||
the strip, since the choice changes the subject of all five.
|
||||
|
||||
The rota is a month at a time, one coloured initial per day with a legend
|
||||
underneath, and it says how many days are left uncovered — the question a rota
|
||||
is read for is who holds which stretch, and a run of one colour answers it
|
||||
where a list of dates does not. An owner taps a day to hand it to somebody or
|
||||
empty it, and fills a whole shift from the range form folded in below.
|
||||
|
||||
The **Admin** tab appears only for a system administrator, and holds what
|
||||
belongs to the whole server rather than to one team. It has three sub-sections,
|
||||
each with a URL of its own and a strip across the top to move between them:
|
||||
every team (`/admin/teams`), every user (`/admin/users`), and the settings that
|
||||
used to be environment variables (`/admin/settings`). `/admin` itself is an
|
||||
overview — how many of each, and what each section is for. Adding somebody is
|
||||
minting them an invite link into a team, rather than creating a bare account:
|
||||
the person who accepts it picks their own password, so one never passes through
|
||||
an administrator, and the link carries the team, so they land somewhere with a
|
||||
queue in it. That happens on the team's own page, since an invite is a fact
|
||||
about a team; the user list points there rather than asking which team beside a
|
||||
form.
|
||||
|
||||
A name in the team list opens **that team's page**, at `/admin/teams/{id}`: when it
|
||||
was created, how many are in it and how much is open, a field to rename it, the
|
||||
members with their roles, the invites into it, and deletion. The member list is the
|
||||
one thing there that needed a new endpoint — `GET /api/teams/{id}/members` is
|
||||
member-only and answers `404` to an administrator who is not in the team, which is
|
||||
the rule and not an oversight, so the page reads `GET /api/admin/teams/{id}` instead.
|
||||
An administrator still sees none of that team's incidents, alerts or rota.
|
||||
|
||||
A name in the user list opens **that person's page**, at `/admin/users/{id}`: their
|
||||
email and when they joined, where their notifications go, whether they are an
|
||||
administrator, whether the account is disabled, the teams they are in with their
|
||||
role in each, a password field for a first or forgotten one, and deletion. It is
|
||||
the one place membership is edited from the person's side — the Team tab answers
|
||||
"who is in this team", and answering "which teams is this person in" there means
|
||||
visiting each team in turn.
|
||||
|
||||
### Single sign-on (OIDC)
|
||||
|
||||
terdut can sign people in through any OpenID Connect provider; the examples use
|
||||
[Authentik](https://goauthentik.io/). Groups at the provider decide who may sign
|
||||
in, which teams they belong to and whether they administer the install, much as
|
||||
Grafana's OAuth role and org mapping does. Password login keeps working alongside
|
||||
it unless you turn it off.
|
||||
|
||||
**At the provider**, create an OAuth2/OpenID provider and an application for it:
|
||||
a *confidential* client, redirect URI `<TERDUT_PUBLIC_URL>/api/oidc/callback`, and
|
||||
the `openid`, `profile` and `email` scopes. The issuer is the application's, e.g.
|
||||
`https://auth.example.com/application/o/terdut/`. Then set:
|
||||
|
||||
```sh
|
||||
TERDUT_PUBLIC_URL=https://terdut.example.com
|
||||
TERDUT_OIDC_ISSUER=https://auth.example.com/application/o/terdut/
|
||||
TERDUT_OIDC_CLIENT_ID=terdut
|
||||
TERDUT_OIDC_CLIENT_SECRET=...
|
||||
TERDUT_OIDC_ALLOWED_GROUPS=terdut-users,terdut-admins
|
||||
TERDUT_OIDC_ADMIN_GROUP=terdut-admins
|
||||
```
|
||||
|
||||
Which team a group grants is not server-wide config: each team names its own
|
||||
group(s), set by that team's own owner (or an administrator) from its Members
|
||||
tab, or `PUT /api/teams/{teamID}/oidc-groups {"member_group":"sre","owner_group":"sre-leads"}`.
|
||||
A team must already exist before a group can grant access to it — the sync
|
||||
never creates one.
|
||||
|
||||
The web UI's sign-in page shows a "Sign in with <name>" button (a plain link to
|
||||
`/api/oidc/login`) above the password form, or instead of it when
|
||||
`TERDUT_PASSWORD_LOGIN=false`; it asks `GET /api/auth/config` what the server offers
|
||||
(`password_login`, `oidc.enabled`, `oidc.name`). A refused sign-in comes back to that
|
||||
page with the reason spelled out. Access the groups grant is badged **SSO** on the
|
||||
Team, Admin and per-user pages, with its edit and remove controls disabled, and the
|
||||
Account page does not offer to set a password nobody could use.
|
||||
|
||||
**What a sign-in does**
|
||||
|
||||
1. *Who.* The provider's `(issuer, subject)` is the identity. The first time, a
|
||||
user is found by email — only when the provider marks it verified, or
|
||||
`TERDUT_OIDC_TRUST_EMAIL` is set — or created with no password. A username taken
|
||||
by somebody else gets a numeric suffix (`alice-2`). Username and email follow the
|
||||
provider at each sign-in. Authentik reports `email_verified` as false unless
|
||||
configured otherwise, so linking existing users usually needs
|
||||
`TERDUT_OIDC_TRUST_EMAIL=true`.
|
||||
2. *Whether.* With `TERDUT_OIDC_ALLOWED_GROUPS` set, somebody in none of them is
|
||||
refused and nothing is created.
|
||||
3. *What.* The administrator flag follows `TERDUT_OIDC_ADMIN_GROUP`. Team roles
|
||||
follow each team's own `oidc_member_group`/`oidc_owner_group`; where both of a
|
||||
team's groups match, the owner group wins.
|
||||
|
||||
**Managed access.** What the sync grants is marked as managed by single sign-on,
|
||||
and only that is ever changed by it. It is added at sign-in, and removed at the
|
||||
next sign-in after the group is gone, even if that leaves a team without an owner
|
||||
(an administrator can always repair a team) — the provider is the source of truth
|
||||
for what it grants, so the last-owner and last-administrator guards do not apply.
|
||||
Memberships and administrators added by hand are left alone; the exception is a
|
||||
hand-added member whose team's own group grants a *higher* role, who is raised and
|
||||
from then on managed. Editing managed access by hand (`POST` or `DELETE` on a
|
||||
team's members, revoking an SSO-granted administrator) is refused with `409`, since
|
||||
the next sign-in would undo it.
|
||||
|
||||
> **Upgrading past migration 013: reconfigure every team's groups.**
|
||||
> `TERDUT_OIDC_GROUP_MAPPINGS` is gone, and the sync no longer creates a team by
|
||||
> name. Group-to-team-role mapping is now each team's own setting — an owner sets
|
||||
> it from the Members tab, or `PUT /api/teams/{teamID}/oidc-groups`. Until a team's
|
||||
> owner does that, an OIDC-sourced membership in it is dropped at that user's next
|
||||
> SSO sign-in, the same as any other loss of group access. Set every team's groups
|
||||
> before affected users next sign in, to avoid a visible gap in access.
|
||||
|
||||
**How fast changes arrive.** Groups are read only at sign-in. A session made by an
|
||||
SSO sign-in has a hard ceiling (`TERDUT_OIDC_SESSION_MAX_AGE`, default 12h) that
|
||||
sliding never extends, so a change at the provider reaches terdut within that time.
|
||||
Password sessions are unaffected.
|
||||
|
||||
> **API keys are not revoked when somebody is removed at the provider.** terdut
|
||||
> holds no refresh token and never asks the provider again, so a person removed
|
||||
> from every allowed group loses their sessions within `TERDUT_OIDC_SESSION_MAX_AGE`
|
||||
> and cannot sign in again, but keeps any API key they made (the TUI and scripts use
|
||||
> them) until an administrator disables the user in terdut.
|
||||
|
||||
**Signing in from a terminal.** A client with no browser of its own, such as the
|
||||
TUI over SSH, signs in with a device code, run by terdut itself so the terminal
|
||||
never talks to the provider:
|
||||
|
||||
1. The terminal calls `POST /api/oidc/device` and shows the person a link
|
||||
(`<TERDUT_PUBLIC_URL>/device?code=XXXX-XXXX`) and the code.
|
||||
2. On any device the person opens the link, signs in (by the provider or by
|
||||
password, whatever the login page offers), sees the code and the account, and
|
||||
presses **Approve**. Only a browser session can approve; an API key cannot.
|
||||
3. The terminal polls `POST /api/oidc/device/token` every 5 seconds and is given the
|
||||
ordinary `terdut_session` cookie once. A person who signs in through the provider
|
||||
gets the same `TERDUT_OIDC_SESSION_MAX_AGE` ceiling on the terminal's session as
|
||||
on their browser's.
|
||||
|
||||
A login expires after 10 minutes. `GET /api/auth/config` reports `device_login`.
|
||||
|
||||
**If the provider is down**, terdut still starts (discovery is fetched on first
|
||||
use) and password login is the way in. With `TERDUT_PASSWORD_LOGIN=false` that way
|
||||
is closed: set it back to `true`. The first administrator comes from the bootstrap
|
||||
endpoint, and stays a manual administrator that no group can revoke; on an SSO-only
|
||||
install set `bootstrap.enabled: false` in the chart if you don't want that account,
|
||||
or keep it and never give it a password.
|
||||
|
||||
### Docker
|
||||
|
||||
```bash
|
||||
@@ -155,20 +306,44 @@ somewhere to exec. The sidecar, the PVC and the `backupSidecar` values are all g
|
||||
|
||||
## Configuration
|
||||
|
||||
Two kinds of setting, split by who changes them and how often.
|
||||
|
||||
**Where the server is plugged in** stays in the environment: the listen address,
|
||||
the database DSN, the ntfy URL and token, the public URL. They are needed before
|
||||
the database is open, and two of them are credentials.
|
||||
|
||||
**How the server behaves** lives in the database and is edited by an
|
||||
administrator in the web UI or through `PUT /api/admin/settings`, taking effect
|
||||
on the next sweep rather than at the next restart. The variables below marked
|
||||
**seed** are the value each of those starts from: written once, on first start,
|
||||
and never overwritten afterwards — a redeploy cannot put a chart's default back
|
||||
over an administrator's edit.
|
||||
|
||||
| Variable | Default | Description |
|
||||
|---|---|---|
|
||||
| `TERDUT_ADDR` | `:8080` | TCP address to listen on |
|
||||
| `TERDUT_DB_DSN` | — | **Required.** Postgres connection string, e.g. `postgres://terdut:secret@localhost:5432/terdut?sslmode=require` |
|
||||
| `TERDUT_ARCHIVE_AFTER` | `168h` (7d) | How long a resolved alert or incident stays in the default list before being auto-archived |
|
||||
| `TERDUT_STALE_AFTER` | `6h` | How long a firing alert may go without a refreshing webhook before it is treated as resolved — **must exceed your Alertmanager `repeat_interval`** |
|
||||
| `TERDUT_DEADMAN_MATCHERS` | `alertname=Watchdog` | Which alerts are [dead man's switches](#dead-mans-switch). `;` separates matchers, `,` the label conditions within one, `=` is exact equality. Every matcher must name an `alertname` |
|
||||
| `TERDUT_ARCHIVE_AFTER` | `168h` (7d) | **seed.** How long a resolved alert or incident stays in the default list before being auto-archived |
|
||||
| `TERDUT_STALE_AFTER` | `6h` | **seed.** How long a firing alert may go without a refreshing webhook before it is treated as resolved — **must exceed your Alertmanager `repeat_interval`** |
|
||||
| `TERDUT_DEADMAN_MATCHERS` | `alertname=Watchdog` | The **default** matchers a team starts with — switches are per team now, and this seeds teams that have no configuration of their own. `;` separates matchers, `,` the label conditions within one, `=` is exact equality. Every matcher must name an `alertname` |
|
||||
| `TERDUT_DEADMAN_TIMEOUT` | `15m` | How long a heartbeat may go unheard before its switch is declared dead — **must be shorter than the `repeat_interval` of the route carrying it**. `0` disables dead man's switch handling |
|
||||
| `TERDUT_DEADMAN_SEVERITY` | `critical` | Severity a dead man's switch incident opens at |
|
||||
| `TERDUT_NTFY_URL` | — | ntfy server to publish push notifications to. Empty disables notifications entirely |
|
||||
| `TERDUT_NTFY_TOKEN` | — | Bearer token for an access-controlled ntfy |
|
||||
| `TERDUT_NTFY_FALLBACK_TOPIC` | — | Topic used when nobody is on call |
|
||||
| `TERDUT_PUBLIC_URL` | — | Base URL a phone uses to reach this server: the notification's link into the web UI, its Acknowledge button, and whether the session cookie is `Secure` |
|
||||
| `TERDUT_NOTIFY_REPEAT` | `15m` | How long an incident may sit unacknowledged before it is paged again. `0` notifies once and never repeats |
|
||||
| `TERDUT_NOTIFY_REPEAT` | `15m` | **seed.** How long an incident may sit unacknowledged before it is paged again. `0` notifies once and never repeats |
|
||||
| `TERDUT_PASSWORD_LOGIN` | `true` | `false` refuses password login and password sign-up (`403`), leaving single sign-on the only way in. Refused at startup unless SSO is configured |
|
||||
| `TERDUT_OPERATOR_MODE` | `false` | Declares this install gitops-managed: a session's or a user's own API key's writes to teams, escalation policies, dead man's switches and integrations are refused (`403 reason:"operator_managed"`); a [service account](#service-accounts)'s are not. Team membership and the schedule stay editable regardless |
|
||||
| `TERDUT_OIDC_ISSUER` | — | Turns single sign-on on. The provider's issuer URL; discovery is read from `<issuer>/.well-known/openid-configuration`. See [Single sign-on](#single-sign-on-oidc) |
|
||||
| `TERDUT_OIDC_CLIENT_ID` / `TERDUT_OIDC_CLIENT_SECRET` | — | **Required with an issuer.** The confidential client registered at the provider. Keep the secret in a Secret, not in values |
|
||||
| `TERDUT_OIDC_NAME` | `SSO` | What the sign-in button calls the provider |
|
||||
| `TERDUT_OIDC_SCOPES` | `openid profile email` | Scopes requested, comma or space separated. Authentik puts `groups` behind `profile` |
|
||||
| `TERDUT_OIDC_USERNAME_CLAIM` / `_EMAIL_CLAIM` / `_GROUPS_CLAIM` | `preferred_username` / `email` / `groups` | ID token claims read for the username, email and groups |
|
||||
| `TERDUT_OIDC_TRUST_EMAIL` | `false` | Link a first sign-in to an existing local user by email even if the provider does not mark the address verified |
|
||||
| `TERDUT_OIDC_ALLOWED_GROUPS` | — | Comma-separated. Only people in one of these may sign in. Empty admits everybody the provider authenticates |
|
||||
| `TERDUT_OIDC_ADMIN_GROUP` | — | Members are system administrators |
|
||||
| `TERDUT_OIDC_SESSION_MAX_AGE` | `12h` | Hard ceiling on a session made by an SSO sign-in |
|
||||
|
||||
Durations use Go syntax (`30m`, `12h`, `168h`). An unparseable value falls back to the default.
|
||||
|
||||
@@ -176,25 +351,45 @@ Note that `TERDUT_STALE_AFTER` and `TERDUT_DEADMAN_TIMEOUT` point in opposite di
|
||||
is a generous grace period around a `repeat_interval` you do not control; a dead man's switch is a
|
||||
deadline you set deliberately, and the heartbeat's route is configured to beat faster than it.
|
||||
|
||||
In the Helm chart the two sweeper durations are set via `sweeper.staleAfter` and `sweeper.archiveAfter`, dead man's switches via the `deadman.*` values, and notifications via the `notify.*` values.
|
||||
In the Helm chart the two sweeper durations are set via `sweeper.staleAfter` and `sweeper.archiveAfter`, dead man's switches via the `deadman.*` values, notifications via the `notify.*` values, single sign-on via `oidc.*` and `passwordLogin`, and operator mode via `operatorMode`.
|
||||
|
||||
---
|
||||
|
||||
## Alertmanager configuration
|
||||
|
||||
Add terdut-server as a webhook receiver in your `alertmanager.yml`:
|
||||
Alerts arrive on a team's **integration key**, which says both that the sender
|
||||
may post and which team the alerts belong to. Mint one as an owner of the team:
|
||||
|
||||
```bash
|
||||
curl -X POST https://terdut.example.com/api/teams/1/integrations \
|
||||
-H "Authorization: Bearer $TERDUT_API_KEY" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"name":"prod alertmanager"}'
|
||||
```
|
||||
|
||||
The response carries the key and the full URL **once**; only a SHA-256 hash is
|
||||
stored. Put it in your `alertmanager.yml`:
|
||||
|
||||
```yaml
|
||||
receivers:
|
||||
- name: terdut
|
||||
webhook_configs:
|
||||
- url: http://terdut-server:8080/api/alertmanager/webhook
|
||||
- url: http://terdut-server:8080/api/integrations/<key>/alertmanager
|
||||
send_resolved: true
|
||||
|
||||
route:
|
||||
receiver: terdut
|
||||
```
|
||||
|
||||
The whole URL is a credential, so treat it like one. Alertmanager 0.26 and
|
||||
later can read it from a file with `url_file:` instead, which keeps it out of
|
||||
your configuration repository:
|
||||
|
||||
```yaml
|
||||
- url_file: /etc/alertmanager/secrets/terdut-webhook-url/url
|
||||
send_resolved: true
|
||||
```
|
||||
|
||||
The webhook endpoint requires no authentication.
|
||||
|
||||
If you use the [dead man's switch](#dead-mans-switch) — and the default configuration does — give
|
||||
@@ -296,10 +491,20 @@ exactly as it was rather than with a hole in it.
|
||||
### Push notifications
|
||||
|
||||
With `TERDUT_NTFY_URL` set, an incident that opens is pushed to the on-call
|
||||
person's phone through [ntfy](https://ntfy.sh). Set each user's topic with
|
||||
`PUT /api/users/{id}/notify`; a user with no topic falls back to
|
||||
`TERDUT_NTFY_FALLBACK_TOPIC`, as does an incident that opens with nobody on call.
|
||||
If neither yields a topic, nothing is queued.
|
||||
person's phone through [ntfy](https://ntfy.sh). Everybody sets their own topic
|
||||
under *Account* in the web UI, where a **Send a test push** button proves it
|
||||
before an incident has to; `PUT /api/users/{id}/notify` is the same thing over
|
||||
the API, and an administrator may set somebody else's. A user with no topic
|
||||
falls back to `TERDUT_NTFY_FALLBACK_TOPIC`, as does an incident that opens with
|
||||
nobody on call. If neither yields a topic, nothing is queued.
|
||||
|
||||
The **server** is the install's one ntfy, from `TERDUT_NTFY_URL`, and is not
|
||||
something a user picks. Only the topic is per-person.
|
||||
|
||||
A topic is a shared secret with the ntfy server: anyone who knows it can both
|
||||
read the pages and publish to it, so an unguessable one is worth the trouble.
|
||||
That is also why the topic never appears in an incident's timeline, which every
|
||||
API key can read.
|
||||
|
||||
Three things get pushed:
|
||||
|
||||
@@ -344,6 +549,50 @@ exhausts its retries. Written from the result rather than at enqueue, so the
|
||||
timeline says what actually happened — and a page that never landed is visible
|
||||
instead of looking the same as one that did.
|
||||
|
||||
### Escalation
|
||||
|
||||
Without a ladder, an unacknowledged incident re-pages the same topic every
|
||||
`notify_repeat` forever. That is a louder version of the same silence: if the
|
||||
person on call is asleep, out of signal, or has left, nothing else happens.
|
||||
|
||||
A team can configure an ordered ladder instead. Each level has a timeout and a
|
||||
set of targets, and a target is either a named person or **whoever the team's
|
||||
rota says is on call today** — the target that keeps working when the rota
|
||||
changes and nobody remembers to edit the policy.
|
||||
|
||||
```
|
||||
level 1 5m oncall the rota gets first refusal
|
||||
level 2 5m user:bob then a named second
|
||||
then repeat_count more rounds
|
||||
then the team's fallback topic, once
|
||||
```
|
||||
|
||||
When a level's timeout passes with the incident still `triggered`, the next
|
||||
level is paged. Off the end of the ladder the whole thing runs again
|
||||
`repeat_count` times, and after that the team's `fallback_topic` is paged once
|
||||
as the end of the line. The incident stays open throughout: running out of
|
||||
people to wake is not the same as somebody answering.
|
||||
|
||||
**Acknowledging or resolving stops it**, which is the point — continuing to wake
|
||||
people after somebody has said "I have this" is how a tool teaches people to
|
||||
mute it. **Snoozing pauses it**: a deliberate "not now" holds the ladder where
|
||||
it is, and it resumes when the snooze runs out.
|
||||
|
||||
Every step is on the incident's timeline with the level and the names it woke,
|
||||
so somebody reading it afterwards can tell why their phone rang at 04:00. A
|
||||
level whose targets are all unreachable — no ntfy topic, a disabled account, an
|
||||
empty rota — is recorded as `nobody reachable` and the ladder moves on rather
|
||||
than stalling on a rung that cannot ring.
|
||||
|
||||
**Reminders and escalation never both run.** A team with a ladder gets
|
||||
escalation; a team without keeps the reminder behaviour exactly as it was. Two
|
||||
pages for one silence is the surest way to get a tool muted.
|
||||
|
||||
The ladder's `fallback_topic` is per team, unlike `TERDUT_NTFY_FALLBACK_TOPIC`,
|
||||
which is the install-wide topic used when an incident opens with nobody on call.
|
||||
They answer different questions: one is "nobody was scheduled", the other is
|
||||
"everybody scheduled has been tried".
|
||||
|
||||
### Stale alert expiry
|
||||
|
||||
A resolved webhook is the only signal that an alert has stopped firing, so a
|
||||
@@ -378,11 +627,31 @@ kube-prometheus-stack already ships the alert for this. `Watchdog` is
|
||||
nothing unless something downstream notices it stop. That is what
|
||||
`TERDUT_DEADMAN_MATCHERS` defaults to.
|
||||
|
||||
**Switches belong to a team**, which decides which of its own alerts are
|
||||
heartbeats and how long a silence has to last. Each **switch** is a row of its
|
||||
own — a name, one matcher, a timeout and a severity — so switches in one team
|
||||
can have different deadlines. An owner adds and removes them on **Team →
|
||||
Switches**, which lists each with a status (**healthy**, **dead**, or
|
||||
**dormant** until its first heartbeat), when it was last heard from, and when it
|
||||
last opened an incident; a matcher that several clusters satisfy is broken down
|
||||
per cluster. The API is `POST`/`DELETE /api/teams/{teamID}/deadman/switches`. A
|
||||
missed heartbeat opens an incident in the team whose integration received it.
|
||||
Removing a switch stops the watching; an incident it already opened stays open
|
||||
until somebody resolves it.
|
||||
|
||||
The environment variables are the starting point, not the setting: the **first**
|
||||
time the server starts, every team is given a switch per default matcher from
|
||||
them, once. After that a team's switches are its own — an owner's edit or
|
||||
deletion is never put back by a redeploy. A team created later starts watching
|
||||
nothing until its owner says otherwise — inheriting an install-wide heartbeat
|
||||
would page a new team about a source it has never heard of.
|
||||
|
||||
A matcher is a set of exact label conditions, one of which must be the
|
||||
`alertname`:
|
||||
`alertname`, in the format the environment variable uses (one matcher per switch; the
|
||||
variable takes several, separated by `;`):
|
||||
|
||||
```
|
||||
TERDUT_DEADMAN_MATCHERS="alertname=Watchdog,cluster=prod; alertname=EdgeHeartbeat"
|
||||
alertname=Watchdog,cluster=prod; alertname=EdgeHeartbeat
|
||||
```
|
||||
|
||||
**The unit of monitoring is the fingerprint, not the alert name.** Two clusters
|
||||
@@ -435,8 +704,10 @@ of the last heartbeat, and the heartbeat's labels are on the incident's
|
||||
|
||||
### Authentication
|
||||
|
||||
All endpoints except `/api/bootstrap`, `/api/alertmanager/webhook`,
|
||||
`/api/notify/ack/{token}`, `/api/login` and `/api/logout` require either an API key:
|
||||
All endpoints except `/api/bootstrap`, `/api/integrations/{key}/alertmanager`,
|
||||
`/api/notify/ack/{token}`, `/api/login`, `/api/logout`, `/api/auth/config`,
|
||||
`/api/version`, `/api/oidc/login`, `/api/oidc/callback`, `/api/oidc/device` and
|
||||
`/api/oidc/device/token` require either an API key:
|
||||
|
||||
```
|
||||
Authorization: Bearer <api-key>
|
||||
@@ -445,9 +716,84 @@ Authorization: Bearer <api-key>
|
||||
or the web UI's session cookie. A request that carries an `Authorization` header
|
||||
is judged on that header alone.
|
||||
|
||||
Two kinds of user exist. An **administrator** manages accounts: creating and
|
||||
deleting users, setting anybody's password, minting keys for anybody, and
|
||||
granting the flag itself. Everybody else works incidents — acknowledging,
|
||||
assigning, snoozing, resolving, noting — and manages their own account and
|
||||
nobody else's. An API key carries exactly the rights of the user it belongs to.
|
||||
|
||||
A third principal, the **service account**, exists for automation (a
|
||||
Kubernetes operator, most likely) that needs to manage teams, escalation
|
||||
policies, dead man's switches and integrations without impersonating a human.
|
||||
It is not a user — it never signs in, never appears in a team's member list,
|
||||
and never holds the administrator flag — and its key is prefixed `tdsa_` so it
|
||||
reads as one at a glance in a log line. See [Service accounts](#service-accounts).
|
||||
|
||||
**Getting an account.** The first one comes from `/api/bootstrap`. After that
|
||||
it depends on `signup_mode`, an administrator setting:
|
||||
|
||||
- `invite_only` (the default) — a team owner mints a link with
|
||||
`POST /api/teams/{teamID}/invites`, and the person who opens it picks a
|
||||
username and password and lands in that team with the role the link carries.
|
||||
Links are single-use unless told otherwise, expire after seven days, and can
|
||||
be revoked before that.
|
||||
- `open` — anybody who can reach the server can create an account, and must
|
||||
name a team, which they then own.
|
||||
|
||||
Invites are **links, not email**: this server has no SMTP, and adding it to send
|
||||
one message would be a subsystem to run, secure and monitor. Send the link
|
||||
however you already talk to the person.
|
||||
|
||||
A domain-restricted third mode was considered and dropped: with no email there
|
||||
is nothing to verify an address against, so it would only check the domain of a
|
||||
string somebody typed.
|
||||
|
||||
The first user, from `/api/bootstrap`, is an administrator. Users created
|
||||
afterwards are not, until an administrator says so. An install always keeps at
|
||||
least one: the last administrator can be neither deleted nor demoted, and
|
||||
nobody can delete or demote themselves.
|
||||
|
||||
Endpoints that require the flag answer `403` with
|
||||
`{"error":"administrator access required"}`.
|
||||
|
||||
**Teams** are the unit of tenancy, and are a separate axis from the administrator
|
||||
flag. A team owns its incidents, alerts, schedule and integrations, and a user
|
||||
sees exactly the teams they belong to. Within a team an **owner** configures it
|
||||
(schedule, integrations, membership) and a **member** works its incidents.
|
||||
|
||||
An administrator crosses that line in one direction only. They **configure any
|
||||
team** without being in it — every owner-only endpoint accepts the flag, because
|
||||
otherwise a team whose last owner left could never be repaired. They do **not
|
||||
read any team**: the queue, the alerts and the incidents are filtered by real
|
||||
membership, so an administrator sees a team's work only by joining it, which is
|
||||
a membership change and shows up as one. Administration is about accounts and
|
||||
the shape of a team, not about reading other people's incidents.
|
||||
|
||||
Anything belonging to a team you are not in answers `404`, not `403`: whether an
|
||||
incident exists is itself something only its team should learn.
|
||||
|
||||
**Operator mode** (`TERDUT_OPERATOR_MODE`, see [Configuration](#configuration))
|
||||
declares this install gitops-managed. When it is on, a session or a user's own
|
||||
API key gets `403 {"error": "...", "reason": "operator_managed"}` on every
|
||||
write this README marks **owner**-gated under Teams below (creating, renaming
|
||||
or deleting a team; its OIDC group binding; its escalation ladder; its dead
|
||||
man's switches; its integrations) — a service account's writes are unaffected.
|
||||
Team membership and invites are deliberately excluded: they are never
|
||||
gitops-managed, in operator mode or out of it. `GET /api/auth/config` reports
|
||||
`operator_mode` so a client can grey those sections out before a write is ever
|
||||
attempted.
|
||||
|
||||
| Method | Path | Description |
|
||||
|---|---|---|
|
||||
| `POST` | `/api/login` | `{"username","password"}` → sets the session cookie, returns `{user, has_password}`. `429` after too many failures |
|
||||
| `GET` | `/api/auth/config` | How to sign in: `{"password_login", "oidc": {"enabled","name"}, "device_login", "operator_mode"}`. No session needed |
|
||||
| `GET` | `/api/version` | `{"version"}` — this build's version string. No session needed, the same as `/healthz` |
|
||||
| `POST` | `/api/login` | `{"username","password"}` → sets the session cookie, returns `{user, has_password}`. `429` after too many failures; `403` when `TERDUT_PASSWORD_LOGIN=false` |
|
||||
| `GET` | `/api/oidc/login` | Starts a single sign-on sign-in: redirects the browser to the provider. `?next=/path` is where to land afterwards; only a path on this server is honoured. Only exists when SSO is configured |
|
||||
| `POST` | `/api/oidc/device` | Starts a device login: returns `{device_code, user_code, verification_url, interval, expires_in}`. Only exists when SSO is configured |
|
||||
| `POST` | `/api/oidc/device/token` | `{"device_code"}` → `202 {"status":"pending"}`, then `200` with the session cookie once approved (once only). `410` with `{"error":"expired"}` or `{"error":"denied"}`; `429 {"error":"slow_down"}` if polled faster than `interval` |
|
||||
| `POST` | `/api/oidc/device/approve` | **session** — `{"user_code"}`. Approves a pending device login as the caller. `403` for an API key; `404` for an unknown, expired or already decided code |
|
||||
| `POST` | `/api/oidc/device/deny` | **session** — `{"user_code"}`. Refuses it |
|
||||
| `GET` | `/api/oidc/callback` | Where the provider sends the browser back. Sets the session cookie and redirects to `/`, or to `/?sso_error=<code>` — one of `denied`, `expired`, `failed`, `unavailable`, `not_allowed`, `no_email`, `email_conflict`, `disabled`, `not_bootstrapped` (no user exists on this install yet — sign in again once something has called `/api/bootstrap`) |
|
||||
| `POST` | `/api/logout` | Ends the session and clears the cookie |
|
||||
| `GET` | `/api/me` | The caller: `{user, has_password}` |
|
||||
|
||||
@@ -455,20 +801,120 @@ is judged on that header alone.
|
||||
|
||||
| Method | Path | Description |
|
||||
|---|---|---|
|
||||
| `POST` | `/api/bootstrap` | Create first user + API key `{"username","email","password"?}` (only works on empty DB) |
|
||||
| `GET` | `/api/users` | List users |
|
||||
| `POST` | `/api/users` | Create user `{"username","email"}` |
|
||||
| `DELETE` | `/api/users/{id}` | Delete user (cascades to keys) |
|
||||
| `PUT` | `/api/users/{id}/notify` | Set push notification target `{"ntfy_topic"}` — empty string clears it |
|
||||
| `PUT` | `/api/users/{id}/password` | Set web UI password `{"password","current_password"}`. `current_password` is required only when changing your own existing password. Ends the user's other sessions |
|
||||
| `POST` | `/api/users/{id}/api-keys` | Issue API key `{"name"}` — key shown once |
|
||||
| `DELETE` | `/api/users/{id}/api-keys/{keyID}` | Revoke API key |
|
||||
**admin** marks an endpoint that requires the administrator flag; **self or
|
||||
admin** marks one you may use on your own account and an administrator may use
|
||||
on anybody's.
|
||||
|
||||
| Method | Path | Who | Description |
|
||||
|---|---|---|---|
|
||||
| `GET` | `/api/signup` | — | Whether sign-up is open, and whether `?invite=` is usable. No session needed: the caller has no account yet |
|
||||
| `POST` | `/api/signup` | — | Create an account `{"username","email","password","invite"?,"team_name"?}` and sign in. `403` without a usable invite when the mode is invite-only |
|
||||
| `POST` | `/api/bootstrap` | — | Create first user + API key `{"username","email","password"?}` (only works on empty DB). The user is an administrator |
|
||||
| `GET` | `/api/users` | any | List users. Open to everybody: the queue's assignment control and the schedule both have to name people |
|
||||
| `GET` | `/api/users/{id}/teams` | self or admin | The teams that user is in, each with their role. `/api/teams` is always about the caller; this one answers it about somebody else, for the admin page's per-user view. `404` for a user who does not exist, so "no teams" and "no such person" are distinguishable |
|
||||
| `POST` | `/api/users` | **admin** | Create user `{"username","email"}`. Not an administrator |
|
||||
| `DELETE` | `/api/users/{id}` | **admin** | Delete user (cascades to keys). `409` for yourself or the last administrator |
|
||||
| `PUT` | `/api/users/{id}/admin` | **admin** | Grant or revoke the administrator flag `{"is_admin"}`. `409` for yourself, the last administrator, or an administrator granted by single sign-on |
|
||||
| `PUT` | `/api/users/{id}/disabled` | **admin** | Take an account out of use, or put it back `{"disabled"}`. `409` for yourself or the last administrator |
|
||||
| `PUT` | `/api/users/{id}/notify` | self or admin | Set push notification target `{"ntfy_topic"}` — empty string clears it |
|
||||
| `PUT` | `/api/users/{id}/password` | self or admin | Set web UI password `{"password","current_password"}`. `current_password` is required only when changing your own existing password. Ends the user's other sessions |
|
||||
| `POST` | `/api/users/{id}/api-keys` | self or admin | Issue API key `{"name"}` — key shown once |
|
||||
| `DELETE` | `/api/users/{id}/api-keys/{keyID}` | self or admin | Revoke API key |
|
||||
|
||||
### Administration
|
||||
|
||||
| Method | Path | Who | Description |
|
||||
|---|---|---|---|
|
||||
| `GET` | `/api/admin/teams` | **admin** | Every team on the server, with its member and open-incident counts. `/api/teams` answers "what am I in"; this answers "what is there" |
|
||||
| `GET` | `/api/admin/teams/{teamID}` | **admin** | One team and who is in it: `{"team", "members"}`. `404` for a team that does not exist. `GET /api/teams/{teamID}/members` is **member**-only and still `404`s an administrator from outside the team — reading a team's shape and reading its work are different questions, so they are different endpoints |
|
||||
| `GET` | `/api/admin/settings` | **admin** | The editable settings with their bounds, plus the environment-configured ones, read-only. Never credentials |
|
||||
| `PUT` | `/api/admin/settings` | **admin** | Change one or more `{"key": seconds}`, or `{"signup_mode": "open"\|"invite_only"}`. `400` for an unknown key or a value outside its bounds |
|
||||
|
||||
### Service accounts
|
||||
|
||||
A service account is a scoped, non-human credential for automation — not a
|
||||
`users` row, so it never signs in, is never a team member, and never carries
|
||||
the administrator flag. Two scopes:
|
||||
|
||||
- **instance** — the same reach system administration has over teams: create
|
||||
one, and mint a **team**-scoped account against any of them. There is no
|
||||
cap on how many instance-scoped accounts exist, but ordinarily there is one,
|
||||
belonging to whatever is provisioning this install end to end.
|
||||
- **team** — owner-equivalent for that one team, and nothing else: every
|
||||
**owner**-gated endpoint under [Teams](#teams), membership and invites
|
||||
included. Nothing narrower is enforced server-side; what actually keeps
|
||||
membership out of automation's hands is that no operator built against this
|
||||
scope should ever call those two endpoints — see
|
||||
[operator mode](#authentication) and `SERVICE-ACCOUNTS.md`'s note on this.
|
||||
|
||||
A key is shown once, at creation or rotation, and only its hash is stored —
|
||||
the same handling as a user's API key. Losing it means minting a new one;
|
||||
there is no way to recover a raw key from the server.
|
||||
|
||||
| Method | Path | Who | Description |
|
||||
|---|---|---|---|
|
||||
| `GET` | `/api/service-accounts` | **admin** | Every service account. Pass `?name=` instead to look one up by its exact name — open to **any** authenticated caller (human or service account), since it returns no key material and is how an account finds its own id |
|
||||
| `POST` | `/api/service-accounts` | owner\* | Create one and mint its first key `{"name","scope","team_id"?}` (`team_id` required for `scope:"team"`, absent for `scope:"instance"`). Returns `{"service_account", "key"}` — `key.key` shown once |
|
||||
| `POST` | `/api/service-accounts/{id}/keys` | owner\* | Mint an additional key `{"name"}` — rotation without recreating the account. Shown once |
|
||||
| `DELETE` | `/api/service-accounts/{id}/keys/{keyID}` | owner\* | Revoke one key |
|
||||
|
||||
\* For an **instance**-scoped account: a system administrator only. For a
|
||||
**team**-scoped account: a system administrator, that team's own human owner,
|
||||
an instance-scoped service account (minting a narrower credential for a team
|
||||
it just created), or — for the two key endpoints only — the account rotating
|
||||
or revoking its own key, which is not a privilege escalation, the same
|
||||
reasoning a user's own API keys rest on.
|
||||
|
||||
### Alert ingestion
|
||||
|
||||
Alerts arrive on a team's integration key. The key is both the credential and the
|
||||
routing: it says that the sender may post, and which team the alerts belong to.
|
||||
Create one with `POST /api/teams/{teamID}/integrations`, which returns the key
|
||||
and the full URL once and stores only a SHA-256 hash.
|
||||
|
||||
| Method | Path | Description |
|
||||
|---|---|---|
|
||||
| `POST` | `/api/alertmanager/webhook` | Alertmanager v4 webhook receiver (no auth) |
|
||||
| `POST` | `/api/integrations/{key}/alertmanager` | Alertmanager v4 webhook receiver for the key's team. `401` for an unknown key |
|
||||
|
||||
This is the only way in. The pre-teams `POST /api/alertmanager/webhook` took no
|
||||
credential at all — anything able to reach the port could open an incident —
|
||||
and was removed in v0.13.0 once senders had moved onto keys.
|
||||
|
||||
### Teams
|
||||
|
||||
**owner** below means an owner of that team, a system administrator (who
|
||||
passes every one of these without being a member), or that team's own
|
||||
team-scoped [service account](#service-accounts) — including membership and
|
||||
invites, technically, though no automation this scope was designed for
|
||||
(a Kubernetes operator's CRDs, see `SERVICE-ACCOUNTS.md`) ever models team
|
||||
membership or would call those two. See [Authentication](#authentication).
|
||||
**member** means membership and nothing else: an administrator who is not in
|
||||
the team gets the same `404` as anybody else.
|
||||
|
||||
| Method | Path | Who | Description |
|
||||
|---|---|---|---|
|
||||
| `GET` | `/api/teams` | any | The caller's own teams, each with their role |
|
||||
| `POST` | `/api/teams` | any | Create a team `{"name"}`; a human creator becomes its first owner. An instance-scoped [service account](#service-accounts) may also create one, and it gets no owner at all — expected for a team an operator is about to hand a team-scoped credential to, not an orphaned team a human made |
|
||||
| `PUT` | `/api/teams/{teamID}` | **owner** | Rename it `{"name"}`. `409` if the name is taken |
|
||||
| `DELETE` | `/api/teams/{teamID}` | **owner** | Delete a team and everything under it. `409` while it has open incidents |
|
||||
| `GET` | `/api/teams/{teamID}/members` | member | Who is in the team, with `status` (`oncall` if the rota has them today, `unpageable` when a page to them would go nowhere — even if they are on call — else `reachable`), `on_call`, `next_shift` (first rota day after today), `pageable` and `problem` (`has no ntfy topic` / `account is disabled`; never the topic itself) and `last_active_at` (their newest session or API-key use). Every member sees the same list |
|
||||
| `POST` | `/api/teams/{teamID}/members` | **owner** | Add a member, or change their role `{"user_id","role"}`. `409` when it would demote the last owner, or the membership is managed by single sign-on |
|
||||
| `DELETE` | `/api/teams/{teamID}/members/{userID}` | **owner** | Remove a member. `409` for the last owner, or a membership managed by single sign-on |
|
||||
| `GET` | `/api/teams/{teamID}/oidc-groups` | member | Which groups control this team's membership: `{"member_group","owner_group"}`. An empty string means no group grants that role here |
|
||||
| `PUT` | `/api/teams/{teamID}/oidc-groups` | **owner** | Set them. An empty string clears a binding |
|
||||
| `GET` | `/api/teams/{teamID}/integrations` | member | List integrations. Never returns keys. Each carries `status` (`active` if its key posted within 24h, `quiet` if it has but not lately, `never`), `last_used_at` (last webhook, usable or not), `last_alert_at` (when an alert last arrived on it) and `alerts_24h` (distinct alerts it refreshed in the last day). Alerts delivered before the source was recorded (migration 010) have none, so the last two fill in as Alertmanager re-sends them |
|
||||
| `PATCH` | `/api/teams/{teamID}/integrations/{integrationID}` | **owner** | Rename `{"name"}`. The key does not change |
|
||||
| `POST` | `/api/teams/{teamID}/integrations` | **owner** | Mint an integration `{"name","kind"}` — key and URL shown once |
|
||||
| `DELETE` | `/api/teams/{teamID}/integrations/{integrationID}` | **owner** | Revoke an integration. Alerts it delivered stay, unattributed |
|
||||
| `GET` | `/api/teams/{teamID}/invites` | **owner** | The team's invite links, with their uses and expiry. Never the tokens |
|
||||
| `POST` | `/api/teams/{teamID}/invites` | **owner** | Mint one `{"role","max_uses"}` — the full URL is returned once |
|
||||
| `DELETE` | `/api/teams/{teamID}/invites/{inviteID}` | **owner** | Revoke a link before it expires |
|
||||
| `GET` | `/api/teams/{teamID}/escalation` | member | The team's [escalation ladder](#escalation) `{repeat_count, fallback_topic, levels[], last_escalated_at?, last_escalated_incident_id?}`. Empty levels means the team has none. Each level also carries `status` (`ready`, `escalating` when an unanswered incident has climbed to it, `unreachable` when nobody on it could be woken), `waiting` (ids of the open incidents on it) and, per target, `username` (who it means today — the person on call, for a rota target), `reachable` and `problem`. The extra fields are output only; `PUT` takes the plain shape |
|
||||
| `PUT` | `/api/teams/{teamID}/escalation` | **owner** | Replace it wholesale. `400` for a level with no targets or no timeout — a rung that pages nobody is a silence with a number on it |
|
||||
| `GET` | `/api/teams/{teamID}/deadman/switches` | member | The team's [dead man's switches](#dead-mans-switch), each `{id, name, matcher, timeout_seconds, severity, status, last_heartbeat_at, last_triggered_at, open_incident_id, sources[]}`. `status` is `healthy`, `dead` or `dormant`; `sources` has one entry per heartbeat fingerprint. Empty when the team watches nothing |
|
||||
| `POST` | `/api/teams/{teamID}/deadman/switches` | **owner** | Add one: `{name?, matcher, timeout_seconds, severity?}`. `400` when the matcher names no `alertname` or holds several, or the timeout is not positive — a switch that silently watches nothing is the failure this feature exists to prevent |
|
||||
| `PUT` | `/api/teams/{teamID}/deadman/switches/{switchID}` | **owner** | Replace one in place, same body and validation as create. Its id is unchanged — for an automated caller reconciling a spec change, unlike delete-and-recreate |
|
||||
| `DELETE` | `/api/teams/{teamID}/deadman/switches/{switchID}` | **owner** | Stop watching. An incident it opened stays open. `404` for a switch of another team |
|
||||
|
||||
### Notifications
|
||||
|
||||
@@ -655,13 +1101,23 @@ unknown" rather than being rejected.
|
||||
|
||||
| Method | Path | Description |
|
||||
|---|---|---|
|
||||
| `POST` | `/api/schedule` | Assign user to dates `{"user_id", "dates":["YYYY-MM-DD",...], "replace"}` — all-or-nothing |
|
||||
| `GET` | `/api/schedule` | List entries. Filters: `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD` |
|
||||
| `GET` | `/api/schedule/current` | Today's on-call user (UTC), 404 if none |
|
||||
| `DELETE` | `/api/schedule/{id}` | Remove schedule entry |
|
||||
Each team keeps its own rota, so two teams can have two different people on call
|
||||
on the same day. The person taking a shift has to be in the team — paging
|
||||
somebody who cannot open the incident is worse than paging nobody.
|
||||
|
||||
| Method | Path | Who | Description |
|
||||
|---|---|---|---|
|
||||
| `POST` | `/api/teams/{teamID}/schedule` | **owner** | Assign user to dates `{"user_id", "dates":["YYYY-MM-DD",...], "replace"}` — all-or-nothing |
|
||||
| `GET` | `/api/teams/{teamID}/schedule` | member | List entries. Filters: `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD` |
|
||||
| `DELETE` | `/api/teams/{teamID}/schedule/{id}` | **owner** | Remove schedule entry |
|
||||
| `GET` | `/api/schedule/current` | any | Who is on call today (UTC) in **every** team the caller is in — one entry per team, `[]` when nobody anywhere |
|
||||
|
||||
### Statistics
|
||||
|
||||
Every figure counts the caller's own teams only: a report that counted other
|
||||
teams' incidents would leak their volume, and their alert names through the
|
||||
top-alerts list, and would not be a number about the reader's work anyway.
|
||||
|
||||
All stat endpoints accept optional `?from=YYYY-MM-DD` and `?to=YYYY-MM-DD`, and exclude archived rows to match the default list views. Alert stats filter on `received_at`; incident stats filter on `triggered_at`.
|
||||
|
||||
| Method | Path | Description |
|
||||
@@ -678,32 +1134,80 @@ averages over incidents that have actually been acknowledged or resolved, and ar
|
||||
|
||||
---
|
||||
|
||||
## Upgrading from SQLite
|
||||
## Upgrading to teams
|
||||
|
||||
Versions up to v0.10.2 stored everything in a SQLite file. From the Postgres release onwards
|
||||
the server needs `TERDUT_DB_DSN` and keeps nothing on disk.
|
||||
Everything that existed before teams moves into one team called **Default**, and
|
||||
every existing user becomes an owner of it. The upgrade is a no-op for the
|
||||
people using it: the same queue, the same schedule, the same incidents, with a
|
||||
name on them.
|
||||
|
||||
The cutover is ordered — the server must not be running while the copy happens:
|
||||
What changes, and will need attention:
|
||||
|
||||
- **Alert ingestion moved.** Mint a key with
|
||||
`POST /api/teams/{teamID}/integrations` and point Alertmanager at the URL it
|
||||
returns. In v0.12.0 the old `POST /api/alertmanager/webhook` still worked,
|
||||
deprecated, routing everything to the oldest team; **v0.13.0 removes it**, so
|
||||
upgrade straight from v0.11.x to v0.13.0 only after the senders are moved.
|
||||
- **The schedule endpoints moved** under `/api/teams/{teamID}/schedule`, and
|
||||
editing the rota is now an owner's job. `GET /api/schedule/current` stayed
|
||||
where it was but now returns an **array** — one entry per team with somebody
|
||||
on call — instead of a single object or a 404. This is a breaking API change
|
||||
for anything that reads it, terdut-tui included.
|
||||
- **Uniqueness is per team now.** Two teams can legitimately see the same alert
|
||||
fingerprint, the same Alertmanager groupKey, and put somebody on call on the
|
||||
same date.
|
||||
|
||||
**Dead man's switches moved too.** `TERDUT_DEADMAN_MATCHERS`, `_TIMEOUT` and
|
||||
`_SEVERITY` are no longer the setting; they are the default each existing team
|
||||
is seeded with at startup, after which an owner manages them per team through
|
||||
`/api/teams/{teamID}/deadman/switches` and a redeploy never overwrites that.
|
||||
|
||||
Nothing else about an incident changes, and incidents never move between teams:
|
||||
an alert belongs to whichever team's key it arrived on.
|
||||
|
||||
## Upgrading to roles
|
||||
|
||||
Before this release every authenticated caller could create and delete users,
|
||||
set anybody's password and mint anybody's API keys. That is now the
|
||||
administrator flag, and the migration **makes every existing user an
|
||||
administrator** — they already held those powers, so nobody's access changes on
|
||||
upgrade and demotion is a deliberate act afterwards. Promoting only the first
|
||||
user would have silently stripped the rest, and could leave an install whose
|
||||
only administrator is an account nobody has a password for.
|
||||
|
||||
Users created after the upgrade are not administrators. Hand the flag out with:
|
||||
|
||||
```bash
|
||||
# 1. Stop the old server, keeping its database file.
|
||||
# 2. Create an empty Postgres database, then let the new binary build the schema:
|
||||
TERDUT_DB_DSN='postgres://terdut:secret@localhost:5432/terdut?sslmode=disable' ./terdut &
|
||||
# ...watch for "listening on", then stop it again.
|
||||
# 3. Copy the data across:
|
||||
go run -tags migrate ./scripts/sqlite-to-postgres.go \
|
||||
-sqlite /data/terdut.db \
|
||||
-dsn 'postgres://terdut:secret@localhost:5432/terdut?sslmode=disable'
|
||||
# 4. Start the new server for good.
|
||||
curl -X PUT https://terdut.example.com/api/users/7/admin \
|
||||
-H "Authorization: Bearer $TERDUT_API_KEY" \
|
||||
-H 'Content-Type: application/json' \
|
||||
-d '{"is_admin": true}'
|
||||
```
|
||||
|
||||
Nothing in the API changed shape, so terdut-tui needs no new version — but a
|
||||
non-administrator now gets `403` where a `200` used to come back.
|
||||
|
||||
## Upgrading from SQLite
|
||||
|
||||
Versions up to v0.10.2 stored everything in a SQLite file. From v0.11.1 the server needs
|
||||
`TERDUT_DB_DSN` and keeps nothing on disk.
|
||||
|
||||
The copy was done by `scripts/sqlite-to-postgres.go`, which **was deleted in v0.13.0** along
|
||||
with the SQLite driver it was the last user of. It is still in the history — check out the
|
||||
`v0.12.0` tag to get it:
|
||||
|
||||
```bash
|
||||
git show v0.12.0:scripts/sqlite-to-postgres.go > sqlite-to-postgres.go
|
||||
```
|
||||
|
||||
The cutover is ordered, and the server must not be running while the copy happens: stop the
|
||||
old version, let the new binary build the schema against an empty Postgres, run the script
|
||||
with `-sqlite` and `-dsn`, then start the new version for good. On Kubernetes step three runs
|
||||
as a Job with the same image against the PVC before it is removed.
|
||||
|
||||
The copy preserves every id, so incidents keep their numbers and the timeline, alert
|
||||
membership, outbox and ack tokens all still point where they did. It refuses a target that
|
||||
already has rows, so a second run cannot double-insert. On Kubernetes, step 3 runs as a Job
|
||||
with the same image against the PVC before it is removed.
|
||||
|
||||
The script is deliberately temporary: it is the only thing left that needs the SQLite driver,
|
||||
and both should be deleted once the installs that need them have migrated.
|
||||
already has rows, so a second run cannot double-insert.
|
||||
|
||||
## Upgrading to incidents
|
||||
|
||||
|
||||
@@ -0,0 +1,263 @@
|
||||
# Service accounts: a scoped, non-human credential type
|
||||
|
||||
This is a design note for a feature, not an implementation plan — it exists to
|
||||
propose the shape before writing code. It's raised directly by `terdut-operator`
|
||||
(a separate repo, no shared code — see its `DESIGN.md` §6, §9, §13), which needs
|
||||
a credential for unattended, repeatable API access and currently has no good one
|
||||
available. Anything automating terdut-server long-term (this operator, CI, future
|
||||
integrations) hits the same gap, so this is written as a general primitive, not
|
||||
operator-specific.
|
||||
|
||||
## The problem
|
||||
|
||||
terdut-server has two credential types today, and neither fits "an unattended
|
||||
process that manages teams/schedules/policies on someone's behalf":
|
||||
|
||||
- **User API keys** (`api_keys`, `internal/api/users.go`) are always tied to a
|
||||
real `users` row and carry that user's full rights — every team they're a
|
||||
member of, their admin flag if set. There's no `kind`/`service` marker
|
||||
distinguishing "a human's personal automation key" from "a login session," and
|
||||
no way to mint one scoped to less than the full user.
|
||||
- **Integration keys** (`integrations`, `internal/api/*teams*.go`) are team-scoped,
|
||||
but narrowly: they authenticate exactly one inbound Alertmanager webhook call
|
||||
(`POST /api/integrations/{key}/alertmanager`) and nothing else. They're not a
|
||||
general management-API credential and shouldn't become one — overloading a
|
||||
narrow, one-way ingestion credential with broad read/write access would weaken
|
||||
the one property that makes it safe to embed in an Alertmanager config today.
|
||||
|
||||
The result: any automation that needs to create teams, set escalation policies,
|
||||
manage dead-man switches, or rotate integration keys has to hold a real human
|
||||
admin's or team owner's API key. That key is exactly as powerful as that person
|
||||
logging in — full team access, and full instance access if they're an admin.
|
||||
`terdut-operator`'s design ran directly into this (its DESIGN.md §6): its
|
||||
described bootstrap/rotation flow assumed a repeatable, identity-scoped way to
|
||||
get a credential, and `/api/bootstrap`'s actual behavior (single-shot per
|
||||
install, gated on `COUNT(*) FROM users`, confirmed via `internal/api/users.go`
|
||||
and `charts/terdut-server/templates/bootstrap-job.yaml`) doesn't provide one —
|
||||
it mints exactly one founding admin, once, ever.
|
||||
|
||||
## Goals
|
||||
|
||||
- A credential type that isn't a human: doesn't touch OIDC group sync, login,
|
||||
session, or the `is_admin`/account-management semantics that come with a real
|
||||
`users` row.
|
||||
- Two scopes matching the two shapes automation actually needs: instance-wide
|
||||
(create/list teams — what a server-owning controller needs) and team-scoped
|
||||
(manage one team's escalation policy, dead-man switches, integrations,
|
||||
schedule, OIDC group bindings — what a per-team controller or integration
|
||||
needs).
|
||||
- Repeatable issuance and rotation — unlike `/api/bootstrap`, callable more than
|
||||
once, by anything that already holds admin rights, without destroying and
|
||||
recreating state to get a fresh credential.
|
||||
- Visibly distinct from a human in every place identity shows up (audit trails,
|
||||
timeline entries, UI attribution) — a service account acting on a team should
|
||||
never be indistinguishable from a person.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- Not a general OAuth2/OIDC client-credentials flow — this is a bearer-token
|
||||
primitive matching the shape `api_keys` already uses (SHA-256 hash stored,
|
||||
raw key shown once at creation), not a new auth protocol.
|
||||
- Not replacing integration keys — those stay as the narrow, one-way webhook
|
||||
credential they are today.
|
||||
- Not modeling per-endpoint or per-verb permissions within a scope — `instance`
|
||||
and `team` are the only two scopes for now; finer-grained scoping is future
|
||||
work if a real need shows up.
|
||||
|
||||
## Proposed shape
|
||||
|
||||
### Schema
|
||||
|
||||
```sql
|
||||
CREATE TABLE service_accounts (
|
||||
id BIGSERIAL PRIMARY KEY,
|
||||
name TEXT NOT NULL UNIQUE, -- e.g. "terdut-operator"
|
||||
scope TEXT NOT NULL CHECK (scope IN ('instance', 'team')),
|
||||
team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE,
|
||||
-- team_id required iff scope = 'team'; NULL iff scope = 'instance'
|
||||
created_by BIGINT REFERENCES users(id),
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT now()
|
||||
);
|
||||
|
||||
CREATE TABLE service_account_keys (
|
||||
id BIGSERIAL PRIMARY KEY,
|
||||
service_account_id BIGINT NOT NULL REFERENCES service_accounts(id) ON DELETE CASCADE,
|
||||
key_hash TEXT NOT NULL UNIQUE,
|
||||
name TEXT NOT NULL, -- e.g. "initial", "2026-Q4-rotation"
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
||||
last_used_at TIMESTAMPTZ
|
||||
);
|
||||
```
|
||||
|
||||
Deliberately not a `users` row: no `password_hash`, no `is_admin`, no
|
||||
`user_identities` linkage, so it's structurally impossible for a service account
|
||||
to be pulled into OIDC group sync or password login. Multiple keys per account
|
||||
(mirroring `api_keys`' existing one-user-many-keys shape) so rotation is "mint a
|
||||
new key, revoke the old one," not "recreate the account."
|
||||
|
||||
### Endpoints
|
||||
|
||||
- `POST /api/service-accounts` — instance-scope/admin-only. Body:
|
||||
`{"name": ..., "scope": "instance"|"team", "teamID": ... }` (teamID required
|
||||
iff scope=team, and caller must be that team's owner or a system admin).
|
||||
Returns the account plus its first raw key (shown once, same pattern as
|
||||
`POST /api/users/{id}/api-keys`). Safe to call again with the same `name` —
|
||||
see "idempotent lookup" below — unlike `/api/bootstrap`, which is inherently
|
||||
one-shot by design (it's answering "does any user exist yet," a question with
|
||||
no analogue once one already does).
|
||||
- `POST /api/service-accounts/{id}/keys` — mint an additional key on an existing
|
||||
account (self-service-equivalent: instance admin for `instance` scope, team
|
||||
owner or system admin for `team` scope). Enables rotation without recreating
|
||||
the account or losing its identity/audit history.
|
||||
- `DELETE /api/service-accounts/{id}/keys/{keyID}` — revoke one key, mirroring
|
||||
`DELETE /api/users/{id}/api-keys/{keyID}`.
|
||||
- `GET /api/service-accounts?name=` — look up an existing account by name.
|
||||
This is what turns "I tried to create my account and got a conflict" into a
|
||||
normal flow instead of an error: a controller that expects to have already
|
||||
registered itself calls this first, and only falls through to `POST` if
|
||||
nothing comes back.
|
||||
|
||||
### Auth middleware
|
||||
|
||||
**Revised** (this section originally described an aspiration that didn't
|
||||
match what shipped — `TEAM-LOOKUP.md` already caught one instance of that,
|
||||
and a fuller audit found three more; this is the corrected, as-built
|
||||
description, not the original proposal).
|
||||
|
||||
`internal/api/middleware.go`'s dual resolution (`Authorization: Bearer` →
|
||||
`apiKeyUser()`, or session cookie → `sessionUser()`) and the service-account
|
||||
path (`serviceAccountFor()`) both resolve into one `Caller` type
|
||||
(`internal/api/caller.go`), not two parallel, un-unified context
|
||||
representations the way an earlier version of this server kept them. Every
|
||||
authorization predicate reads `Caller`'s methods:
|
||||
|
||||
- `Caller.IsAdmin()` — true **only** for a human system administrator, never
|
||||
for a service account of either scope, under any circumstance. `AdminOnly`
|
||||
and `requireSelfOrAdmin` key on this alone — user management
|
||||
(`POST /api/users`, `PUT /api/users/{id}/admin`, etc.) and
|
||||
`GET/PUT /api/admin/settings` stay human-only, forever. The original text
|
||||
here claimed an instance-scoped service account satisfies `AdminOnly` "for
|
||||
team-creation/listing purposes" — that was never true of the shipped code
|
||||
(`TEAM-LOOKUP.md` caught the listing half; the creation half was always a
|
||||
separate, bespoke check in `handleCreateTeam`, not `AdminOnly` itself) and
|
||||
is not being made true now. Don't widen `AdminOnly`: every time this has
|
||||
come up, the fix has been a narrower, purpose-built capability instead
|
||||
(`?name=` lookups for teams and service accounts; now
|
||||
`terdut-operator`'s own invite-minting feature for the one real gap this
|
||||
boundary left — how a human ever gets a first login on a no-OIDC,
|
||||
operator-managed install. See the bottom of "What this unblocks.")
|
||||
- `Caller.IsInstanceServiceAccount()` — true only for an instance-scoped
|
||||
service account, never for a human (including a human admin).
|
||||
`handleCreateTeam` uses exactly this: a human creates a team by being a
|
||||
human (and becomes its owner); an instance-scoped service account creates
|
||||
one with no human owner at all. The two paths are not interchangeable, so
|
||||
this predicate deliberately does not also admit a human admin.
|
||||
- `Caller.Role(teamID)`/`TeamIDs()` — a human's real `team_members` rows, or
|
||||
a team-scoped service account's single synthetic owner membership
|
||||
(`serveAsServiceAccount`). This is what makes `requireTeamMember`/
|
||||
`requireTeamOwner` treat a team-scoped service account as owner-equivalent
|
||||
for that one team, with no separate branch needed in either function.
|
||||
- `Caller.ServiceAccountID()` — used by `OperatorModeBlock` ("any service
|
||||
account passes") and by `callerMayManageServiceAccount`'s self-rotation
|
||||
check.
|
||||
- `Caller.AsHuman()` — the accessor every handler that needs a real
|
||||
`user_id` to act on behalf of must call and check, instead of reading a
|
||||
user off context unconditionally. Before the `Caller` type existed, four
|
||||
handlers did the latter and silently misbehaved for a service-account
|
||||
caller: `handleMe` and `handleTestNotification` 500'd (a zero-value user id
|
||||
that matches no row), `handleDismissOnboarding` silently no-op'd (`UPDATE
|
||||
... WHERE id = 0` affects nothing, still returns 204), and
|
||||
`handleCreateInvite` wrote that same zero value into `invites.created_by`
|
||||
— a real foreign-key violation, not just a wrong answer, since that column
|
||||
is nullable but was never passed as `nil`. All four now call `AsHuman()`
|
||||
and return an explicit 403 ("this endpoint is for human accounts only")
|
||||
or, for the invite case, leave `created_by` `NULL` the same way
|
||||
`handleCreateServiceAccount` already did for the analogous situation.
|
||||
|
||||
**Team scope is owner-equivalent for every `requireTeamOwner` endpoint,
|
||||
membership and invites included — by design, not by an unclosed gap.** An
|
||||
earlier version of this document flagged this as "acknowledged rather than
|
||||
closed," kept in check only by the social convention that nobody *builds*
|
||||
automation against those two routes. That convention is retired:
|
||||
`terdut-operator`'s `TerdutTeam` controller now mints and revokes its own
|
||||
team's invite link through exactly this capability (its existing
|
||||
team-scoped credential, `POST`/`DELETE /api/teams/{teamID}/invites`), which
|
||||
is the real fix for the human-onboarding gap below — not a narrower
|
||||
carve-out of this capability. `service_accounts_test.go`'s
|
||||
`TestServiceAccount_TeamScopeManagesItsOwnInvites` pins it.
|
||||
|
||||
**A team-scoped account can also mint another service account scoped to its
|
||||
own team** (`handleCreateServiceAccount`'s `callerOwnsTeam` branch, which a
|
||||
team-scoped caller already satisfies for its own team via the synthetic
|
||||
membership above). Kept, not restricted, for the same reason: a team-scoped
|
||||
credential is that team's owner's reach, full stop — carving this one
|
||||
capability out while leaving membership/invites alone would be an arbitrary
|
||||
asymmetry. Pinned by
|
||||
`TestServiceAccount_TeamScopeCanMintAnotherAccountForItsOwnTeam`.
|
||||
|
||||
**`callerMayManageServiceAccount` gained the one load-bearing fix this
|
||||
redesign exists for:** an instance-scoped service account may manage
|
||||
(mint/revoke a key on) *any* team-scoped account, not only one admin, that
|
||||
team's human owner, or the account itself. `handleCreateServiceAccount`
|
||||
already let an instance-scoped caller *create* a team-scoped account for
|
||||
any team; this closes the gap where adopting or rotating one it didn't just
|
||||
create in the same call — exactly `terdut-operator`'s documented
|
||||
adopt-on-409 crash-window recovery (its own `DESIGN.md` §5) — 403'd forever
|
||||
instead of succeeding (`terdut-operator#3`). Pinned by
|
||||
`TestServiceAccount_InstanceScopeAdoptsAnExistingTeamScopedAccountsKey`.
|
||||
|
||||
Anywhere identity is recorded for a human (incident timeline
|
||||
`acknowledged_by`/`assigned_to`, audit-relevant fields), a service-account
|
||||
caller is still coerced into a bare `user_id` of `0` today — `Caller`'s new
|
||||
`Identity()` accessor exists for exactly this follow-up, but wiring it in
|
||||
needs a schema migration (an actor-attribution column distinct from
|
||||
`user_id`) and is deliberately out of scope here. Tracked separately, not by
|
||||
this document.
|
||||
|
||||
## What this unblocks
|
||||
|
||||
Directly resolves `terdut-operator` DESIGN.md §6's two broken assumptions:
|
||||
1. **Bootstrap becomes single-purpose again.** `/api/bootstrap` mints exactly
|
||||
the founding human admin, once. The operator's actual first-reconcile flow:
|
||||
call `/api/bootstrap` only on a genuinely empty install; otherwise (or
|
||||
immediately after, if it won the bootstrap race) call
|
||||
`GET /api/service-accounts?name=terdut-operator`, and `POST` one if it
|
||||
doesn't exist yet. From then on the operator never touches `/api/bootstrap`
|
||||
again.
|
||||
2. **Rotation becomes real.** `POST /api/service-accounts/{id}/keys` + revoke the
|
||||
old one — no destructive DB-level workaround, no re-triggering a single-shot
|
||||
endpoint that can't fire twice.
|
||||
3. **Cross-namespace credential mirroring is no longer needed at all.**
|
||||
`terdut-operator`'s current design holds every credential — instance- and
|
||||
team-scoped alike — privately in the operator's own namespace, never in
|
||||
the namespace of the CR each one authenticates for; reconciliation happens
|
||||
entirely inside the operator's controller loop, so no CR owner ever needs
|
||||
read access to a terdut-server credential regardless of same- or
|
||||
cross-namespace `serverRef`. Team scoping is still what bounds the blast
|
||||
radius of any individual credential: a leaked team-scoped key exposes
|
||||
exactly one team's resources, never the whole server, which is what makes
|
||||
holding many credentials in one place (the operator's namespace) an
|
||||
acceptable trade rather than reintroducing the mirrored design's
|
||||
server-admin-equivalent-everywhere problem.
|
||||
4. **A human can get a first login on a no-OIDC, operator-managed install —
|
||||
without ever touching `AdminOnly` or `/api/admin/settings`.** This was
|
||||
filed as `terdut-server#23` ("no API path to create a human login after
|
||||
bootstrap") and diagnosed, at the time, as this server needing to let a
|
||||
service account through `AdminOnly`. It doesn't: the fix lives entirely
|
||||
in `terdut-operator`, because a team-scoped credential was *already*
|
||||
owner-equivalent for `POST /api/teams/{teamID}/invites`, and invite
|
||||
redemption (`POST /api/signup` with an `invite` token) bypasses
|
||||
`signup_mode` entirely — `terdut-operator` just never grew a feature to
|
||||
use either fact. Its `TerdutTeam` controller now mints and surfaces one
|
||||
via its own existing team-scoped credential (`spec.invite`,
|
||||
`status.inviteSecretRef`, see that repo's own docs), so a human joins a
|
||||
CRD-managed team by a real invite link, the same way anyone else would.
|
||||
`terdut-server#23` is closed with this note once that feature ships — its
|
||||
named routes stay human-only, correctly, not a gap.
|
||||
|
||||
## Suggested sequencing
|
||||
|
||||
Land this before `terdut-operator` implements any bootstrap/credential-handling
|
||||
code — that code would otherwise be written against the current one-shot,
|
||||
user-only credential model as a known-temporary workaround, which is wasted
|
||||
effort on a repo that currently has zero implementation to begin with.
|
||||
@@ -0,0 +1,73 @@
|
||||
# Team lookup for service accounts: closing terdut-operator's create-path crash window
|
||||
|
||||
This is a design note for a feature, not an implementation plan — same posture as
|
||||
`SERVICE-ACCOUNTS.md`, and raised for the same reason: `terdut-operator`'s `TerdutTeam`
|
||||
controller (ROADMAP.md Stage 2, a separate repo, no shared code) hit a gap this server has
|
||||
no answer for yet.
|
||||
|
||||
## The problem
|
||||
|
||||
`POST /api/teams` (`handleCreateTeam`, confirmed against `internal/api/teams.go`) lets an
|
||||
instance-scoped service account create a team — it has its own explicit
|
||||
`isInstanceServiceAccount(...)` branch alongside the human-user path, not gated by
|
||||
`AdminOnly`. If that call succeeds server-side but the caller (`TerdutTeam`'s controller)
|
||||
crashes before persisting the resulting team ID locally, a retry's `POST` 409s on the name's
|
||||
unique constraint (confirmed: the `isUniqueViolation` branch in the same handler).
|
||||
|
||||
Recovering from that 409 means looking the team up by name, and nothing today permits that
|
||||
for a service account:
|
||||
|
||||
- `GET /api/teams` (`handleListTeams`) answers "what teams does the *caller* belong to", via
|
||||
a `team_members` join keyed on `userFromContext`'s `caller.ID` — confirmed against source.
|
||||
A service account is never a member of anything, so this always returns empty for one,
|
||||
regardless of what exists.
|
||||
- `GET /api/admin/teams` (`handleAdminListTeams`) is gated by `AdminOnly`, and `AdminOnly`'s
|
||||
actual code (`internal/api/middleware.go`) checks only `userFromContext(...).IsAdmin` — no
|
||||
branch for a service account at all, confirmed against source. This contradicts
|
||||
`SERVICE-ACCOUNTS.md`'s own text, which claims "an instance-scoped [service account
|
||||
satisfies] `AdminOnly` for team-creation/listing purposes" — that claim doesn't match this
|
||||
endpoint's actual, shipped code. (Team *creation* is fine: `handleCreateTeam` isn't behind
|
||||
`AdminOnly` at all, it has its own check. Only the listing half of that sentence is wrong.)
|
||||
|
||||
This is exactly the shape of gap `SERVICE-ACCOUNTS.md`'s own `GET /api/service-accounts?name=`
|
||||
closed for service accounts themselves (confirmed: that endpoint's own comment —
|
||||
"the name lookup is open to any authenticated caller... what lets a service account find its
|
||||
own account on the 403 that follows a second POST"). Teams never got the equivalent, because
|
||||
nothing needed it until an operator started creating them unattended.
|
||||
|
||||
## Goals
|
||||
|
||||
- A service-account-accessible way to look up one team by exact name, mirroring
|
||||
`GET /api/service-accounts?name=` as closely as possible — same shape, same reasoning,
|
||||
same low sensitivity of what it discloses.
|
||||
- No change to today's behavior for an empty/no-name request.
|
||||
|
||||
## Proposed shape
|
||||
|
||||
Extend `GET /api/teams` itself, the same way `handleListServiceAccounts` already branches on
|
||||
a `?name=` query param, rather than adding a new route:
|
||||
|
||||
- `name` unset (today's behavior, unchanged): the caller's own teams, via `team_members`.
|
||||
- `name=<value>` set: look up that one team by exact name — a one-or-zero-length array, not
|
||||
an error on no match, mirroring `GET /api/service-accounts?name=`'s own response shape and
|
||||
status codes exactly. Deliberately **not** gated by `isInstanceServiceAccount` or
|
||||
`AdminOnly`: a human caller who's already a member sees this same information in their own
|
||||
team list regardless, and a non-member learning only that a name is taken — not who's in
|
||||
the team, not any of its data — is the same low-sensitivity disclosure
|
||||
`GET /api/service-accounts?name=` already accepts for service-account names.
|
||||
|
||||
## What this unblocks
|
||||
|
||||
Directly resolves the crash-window gap in `terdut-operator`'s `TerdutTeam` controller: on a
|
||||
409 from `POST /api/teams`, `GET /api/teams?name=<the same name>` — authenticated with the
|
||||
same instance-scoped credential that just got the 409 — finds the id, and the controller
|
||||
proceeds as if its own create had returned it directly. The same adopt-on-409 pattern already
|
||||
proven for service accounts (that repo's `DESIGN.md` §6 point 1, §5's general rule), not a
|
||||
new one.
|
||||
|
||||
## Suggested sequencing
|
||||
|
||||
Land this before `TerdutTeam`'s create path is implemented — the same reasoning
|
||||
`SERVICE-ACCOUNTS.md` gave for its own sequencing: writing that code against today's gap as a
|
||||
"known-temporary workaround" is wasted effort when the fix is this small and this
|
||||
well-precedented.
|
||||
@@ -15,5 +15,5 @@ type: application
|
||||
# appVersion and image.tag in values.yaml no longer agree, and that is not an oversight:
|
||||
# image.tag stays "latest", which is what a local install actually pulls. appVersion is
|
||||
# metadata and drives nothing.
|
||||
version: 0.11.1
|
||||
appVersion: "v0.11.1"
|
||||
version: 0.36.0
|
||||
appVersion: "v0.36.0"
|
||||
|
||||
@@ -10,9 +10,14 @@ spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
{{- include "terdut-server.selectorLabels" . | nindent 6 }}
|
||||
# Recreate, not RollingUpdate, even though the PVC that forced it is gone: the
|
||||
# sweeper and the notifier are unsynchronised singletons, and two replicas
|
||||
# overlapping during a rollout would both page for the same incident.
|
||||
# Recreate, not RollingUpdate, even though the PVC that forced it is gone:
|
||||
# the sweeper, notifier and migration runner take a Postgres advisory lock
|
||||
# each, and new-incident creation on the first webhook for a brand-new
|
||||
# groupKey resolves its own insert conflict — so two replicas overlapping
|
||||
# during a rollout no longer double-page, race a migration, or drop a
|
||||
# webhook payload. Nothing left here actually requires Recreate anymore;
|
||||
# it stays the default pending a deliberate decision to raise replicas
|
||||
# above 1 and move to RollingUpdate.
|
||||
strategy:
|
||||
type: Recreate
|
||||
template:
|
||||
@@ -21,6 +26,23 @@ spec:
|
||||
{{- include "terdut-server.selectorLabels" . | nindent 8 }}
|
||||
spec:
|
||||
enableServiceLinks: false
|
||||
{{- if .Values.database.waitForPostgres.enabled }}
|
||||
initContainers:
|
||||
- name: wait-for-postgres
|
||||
image: "{{ .Values.database.waitForPostgres.image.repository }}:{{ .Values.database.waitForPostgres.image.tag }}"
|
||||
imagePullPolicy: {{ .Values.database.waitForPostgres.image.pullPolicy }}
|
||||
env:
|
||||
- name: TERDUT_DB_DSN
|
||||
value: {{ required "database.dsn is required" .Values.database.dsn | quote }}
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
- |
|
||||
until pg_isready -d "$TERDUT_DB_DSN"; do
|
||||
echo "wait-for-postgres: not ready yet, retrying in 2s"
|
||||
sleep 2
|
||||
done
|
||||
{{- end }}
|
||||
containers:
|
||||
- name: terdut-server
|
||||
image: "{{ .Values.image.repository }}:{{ .Values.image.tag }}"
|
||||
@@ -61,8 +83,6 @@ spec:
|
||||
value: "{{ .Values.notify.fallbackTopic }}"
|
||||
- name: TERDUT_NOTIFY_REPEAT
|
||||
value: "{{ .Values.notify.repeatEvery }}"
|
||||
- name: TERDUT_PUBLIC_URL
|
||||
value: "{{ .Values.notify.publicUrl | default (printf "https://%s" .Values.networking.hostname) }}"
|
||||
{{- if .Values.notify.tokenSecret.name }}
|
||||
- name: TERDUT_NTFY_TOKEN
|
||||
valueFrom:
|
||||
@@ -71,6 +91,47 @@ spec:
|
||||
key: {{ .Values.notify.tokenSecret.key }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
# Set whether or not ntfy is: single sign-on builds its redirect URI
|
||||
# from it, and sessions use it to decide the cookie's Secure flag.
|
||||
- name: TERDUT_PUBLIC_URL
|
||||
value: "{{ .Values.notify.publicUrl | default (printf "https://%s" .Values.networking.hostname) }}"
|
||||
- name: TERDUT_PASSWORD_LOGIN
|
||||
value: {{ .Values.passwordLogin | quote }}
|
||||
- name: TERDUT_OPERATOR_MODE
|
||||
value: {{ .Values.operatorMode | quote }}
|
||||
{{- if .Values.oidc.enabled }}
|
||||
- name: TERDUT_OIDC_ISSUER
|
||||
value: {{ required "oidc.issuer is required when oidc.enabled" .Values.oidc.issuer | quote }}
|
||||
- name: TERDUT_OIDC_CLIENT_ID
|
||||
value: {{ required "oidc.clientId is required when oidc.enabled" .Values.oidc.clientId | quote }}
|
||||
- name: TERDUT_OIDC_CLIENT_SECRET
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: {{ required "oidc.clientSecret.name is required when oidc.enabled" .Values.oidc.clientSecret.name }}
|
||||
key: {{ .Values.oidc.clientSecret.key }}
|
||||
- name: TERDUT_OIDC_NAME
|
||||
value: {{ .Values.oidc.name | quote }}
|
||||
- name: TERDUT_OIDC_SCOPES
|
||||
value: {{ .Values.oidc.scopes | quote }}
|
||||
- name: TERDUT_OIDC_USERNAME_CLAIM
|
||||
value: {{ .Values.oidc.usernameClaim | quote }}
|
||||
- name: TERDUT_OIDC_EMAIL_CLAIM
|
||||
value: {{ .Values.oidc.emailClaim | quote }}
|
||||
- name: TERDUT_OIDC_GROUPS_CLAIM
|
||||
value: {{ .Values.oidc.groupsClaim | quote }}
|
||||
- name: TERDUT_OIDC_TRUST_EMAIL
|
||||
value: {{ .Values.oidc.trustEmail | quote }}
|
||||
- name: TERDUT_OIDC_SESSION_MAX_AGE
|
||||
value: {{ .Values.oidc.sessionMaxAge | quote }}
|
||||
{{- if .Values.oidc.allowedGroups }}
|
||||
- name: TERDUT_OIDC_ALLOWED_GROUPS
|
||||
value: {{ join "," .Values.oidc.allowedGroups | quote }}
|
||||
{{- end }}
|
||||
{{- if .Values.oidc.adminGroup }}
|
||||
- name: TERDUT_OIDC_ADMIN_GROUP
|
||||
value: {{ .Values.oidc.adminGroup | quote }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
path: /healthz
|
||||
|
||||
@@ -33,6 +33,26 @@ database:
|
||||
passwordSecret:
|
||||
name: ""
|
||||
key: password
|
||||
# Blocks the main container from starting until Postgres accepts
|
||||
# connections. Without this, a Deployment created before Postgres has
|
||||
# finished its very first boot -- initdb plus Patroni leader election, on a
|
||||
# from-scratch postgres-operator cluster -- crash-loops a few times: the
|
||||
# app's own ping-retry budget on startup (pingAttempts/pingRetryDelay in
|
||||
# internal/db/db.go) is sized for a much shorter, different race --
|
||||
# NetworkPolicy propagation, a few seconds -- not for genuine first-time
|
||||
# cluster creation, which routinely takes longer, so it exhausts and the
|
||||
# process exits before ever binding its HTTP port. A startupProbe cannot
|
||||
# help here: the crash happens before there is anything to probe.
|
||||
#
|
||||
# pg_isready needs no credentials -- it reports PQPING_OK on anything that
|
||||
# amounts to "a Postgres backend answered", including an auth challenge --
|
||||
# so no PGPASSWORD is wired into this container.
|
||||
waitForPostgres:
|
||||
enabled: true
|
||||
image:
|
||||
repository: postgres
|
||||
tag: "17-alpine"
|
||||
pullPolicy: IfNotPresent
|
||||
|
||||
service:
|
||||
type: ClusterIP
|
||||
@@ -107,6 +127,62 @@ notify:
|
||||
name: ""
|
||||
key: token
|
||||
|
||||
# Whether a user may sign in, or sign up, with a password. Turn it off once
|
||||
# single sign-on works, to make it the only way in; turn it back on (and
|
||||
# redeploy) if the identity provider is down and somebody has to get in.
|
||||
passwordLogin: true
|
||||
|
||||
# Declares this install gitops-managed: writes to teams, escalation policies,
|
||||
# dead man's switches and integrations from a session or a user's own API key
|
||||
# are refused, while a service account's (see SERVICE-ACCOUNTS.md) are not.
|
||||
# Off by default — turning it on is a statement that something like
|
||||
# terdut-operator, not a person in the web UI, owns this install's
|
||||
# configuration from here on.
|
||||
operatorMode: false
|
||||
|
||||
# Single sign-on through an OpenID Connect provider such as Authentik.
|
||||
#
|
||||
# At the provider, create an OAuth2/OpenID application whose redirect URI is
|
||||
# <notify.publicUrl>/api/oidc/callback
|
||||
# (publicUrl defaults to https://<networking.hostname>), a confidential client, and
|
||||
# put the client secret in an existing Secret named by clientSecret below.
|
||||
#
|
||||
# Groups from the provider decide what a person can do. Access it grants is
|
||||
# marked as managed by single sign-on and is re-read at every sign-in; anything
|
||||
# added by hand in terdut is left alone. Changes in the provider take effect at
|
||||
# the person's next sign-in, at most sessionMaxAge later. API keys are NOT
|
||||
# revoked when somebody is removed at the provider: disable the user in terdut too.
|
||||
oidc:
|
||||
enabled: false
|
||||
# Issuer URL. For Authentik: https://<authentik>/application/o/<app-slug>/
|
||||
issuer: ""
|
||||
clientId: ""
|
||||
clientSecret:
|
||||
name: ""
|
||||
key: client-secret
|
||||
# What the sign-in button calls the provider.
|
||||
name: SSO
|
||||
# Authentik puts the groups claim behind the profile scope.
|
||||
scopes: "openid profile email"
|
||||
usernameClaim: preferred_username
|
||||
emailClaim: email
|
||||
groupsClaim: groups
|
||||
# Link a first sign-in to an existing local user with the same email even when
|
||||
# the provider does not mark the address verified. Authentik reports
|
||||
# email_verified as false unless configured otherwise.
|
||||
trustEmail: false
|
||||
# Only people in one of these groups may sign in. Empty admits everybody the
|
||||
# provider authenticates, and access control is left to the provider.
|
||||
allowedGroups: []
|
||||
# Members of this group are system administrators.
|
||||
adminGroup: ""
|
||||
# Which group grants a team's membership and ownership is each team's own
|
||||
# setting now, not chart config: an owner sets it from the Members tab, or
|
||||
# PUT /api/teams/{teamID}/oidc-groups. A team must already exist for a group
|
||||
# to grant access to it.
|
||||
# Hard ceiling on a session made by a single sign-on login.
|
||||
sessionMaxAge: 12h
|
||||
|
||||
# Backups are no longer this chart's business. The SQLite database lived on a PVC
|
||||
# beside the app, so it needed a sidecar with a sqlite3 module for k8up to exec a
|
||||
# dump in; Postgres is backed up where it runs, through a k8up.io/backupcommand
|
||||
|
||||
+18
-2
@@ -17,6 +17,9 @@ var version = "dev"
|
||||
|
||||
func main() {
|
||||
cfg := config.Load()
|
||||
if err := cfg.Validate(); err != nil {
|
||||
log.Fatalf("config: %v", err)
|
||||
}
|
||||
|
||||
database, err := db.Open(cfg.DSN)
|
||||
if err != nil {
|
||||
@@ -36,9 +39,22 @@ func main() {
|
||||
RepeatEvery: cfg.NotifyRepeat,
|
||||
}
|
||||
|
||||
// Dead man's switches live per team now. The environment variables are the
|
||||
// defaults a team starts from: every team without a configuration of its
|
||||
// own gets one from them here, and an owner's later edit is never
|
||||
// overwritten by a redeploy.
|
||||
deadman := api.ParseDeadmanConfig(cfg.DeadmanMatchers, cfg.DeadmanTimeout, cfg.DeadmanSeverity)
|
||||
if err := api.SeedDeadmanConfigs(context.Background(), database, deadman); err != nil {
|
||||
log.Fatalf("seed dead man's switch defaults: %v", err)
|
||||
}
|
||||
|
||||
router := api.NewRouter(database, notify, deadman)
|
||||
// The behaviour knobs move into the database on first start, after which an
|
||||
// administrator owns them and a redeploy leaves them alone.
|
||||
if err := api.SeedSettings(context.Background(), database, cfg); err != nil {
|
||||
log.Fatalf("seed settings: %v", err)
|
||||
}
|
||||
|
||||
router := api.NewRouter(database, notify, cfg, version)
|
||||
|
||||
srv := &http.Server{
|
||||
Addr: cfg.Addr,
|
||||
@@ -51,7 +67,7 @@ func main() {
|
||||
ctx, stop := signal.NotifyContext(context.Background(), syscall.SIGINT, syscall.SIGTERM)
|
||||
defer stop()
|
||||
|
||||
go api.StartArchiver(ctx, database, cfg.ArchiveAfter, cfg.StaleAfter, deadman, notify)
|
||||
go api.StartArchiver(ctx, database, cfg.ArchiveAfter, cfg.StaleAfter, notify)
|
||||
go api.StartNotifier(ctx, database, notify)
|
||||
|
||||
go func() {
|
||||
|
||||
@@ -3,26 +3,19 @@ module git.ryuvia.com/niklas/terdut-server
|
||||
go 1.25.9
|
||||
|
||||
require (
|
||||
github.com/coreos/go-oidc/v3 v3.21.0
|
||||
github.com/go-chi/chi/v5 v5.2.5
|
||||
github.com/jackc/pgerrcode v0.0.0-20250907135507-afb5586c32a6
|
||||
github.com/jackc/pgx/v5 v5.11.0
|
||||
golang.org/x/crypto v0.55.0
|
||||
modernc.org/sqlite v1.50.1
|
||||
golang.org/x/oauth2 v0.36.0
|
||||
)
|
||||
|
||||
require (
|
||||
github.com/dustin/go-humanize v1.0.1 // indirect
|
||||
github.com/google/uuid v1.6.0 // indirect
|
||||
github.com/go-jose/go-jose/v4 v4.1.4 // indirect
|
||||
github.com/jackc/pgpassfile v1.0.0 // indirect
|
||||
github.com/jackc/pgservicefile v0.0.0-20240606120523-5a60cdf6a761 // indirect
|
||||
github.com/jackc/puddle/v2 v2.2.2 // indirect
|
||||
github.com/mattn/go-isatty v0.0.20 // indirect
|
||||
github.com/ncruces/go-strftime v1.0.0 // indirect
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
||||
golang.org/x/sync v0.22.0 // indirect
|
||||
golang.org/x/sys v0.47.0 // indirect
|
||||
golang.org/x/text v0.41.0 // indirect
|
||||
modernc.org/libc v1.72.3 // indirect
|
||||
modernc.org/mathutil v1.7.1 // indirect
|
||||
modernc.org/memory v1.11.0 // indirect
|
||||
)
|
||||
|
||||
@@ -1,16 +1,12 @@
|
||||
github.com/coreos/go-oidc/v3 v3.21.0 h1:wZo4Q9Pum8dYEj0eMUPrqR+kvuGkeUplbLpNCkBqoWM=
|
||||
github.com/coreos/go-oidc/v3 v3.21.0/go.mod h1:DYCf24+ncYi+XkIH97GY1+dqoRlbaSI26KVTCI9SrY4=
|
||||
github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||
github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c=
|
||||
github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||
github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY=
|
||||
github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto=
|
||||
github.com/go-chi/chi/v5 v5.2.5 h1:Eg4myHZBjyvJmAFjFvWgrqDTXFyOzjj7YIm3L3mu6Ug=
|
||||
github.com/go-chi/chi/v5 v5.2.5/go.mod h1:X7Gx4mteadT3eDOMTsXzmI4/rwUpOwBHLpAfupzFJP0=
|
||||
github.com/google/pprof v0.0.0-20250317173921-a4b03ec1a45e h1:ijClszYn+mADRFY17kjQEVQ1XRhq2/JR1M3sGqeJoxs=
|
||||
github.com/google/pprof v0.0.0-20250317173921-a4b03ec1a45e/go.mod h1:boTsfXsheKC2y+lKOCMpSfarhxDeIzfZG1jqGcPl3cA=
|
||||
github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0=
|
||||
github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo=
|
||||
github.com/hashicorp/golang-lru/v2 v2.0.7 h1:a+bsQ5rvGLjzHuww6tVxozPZFVghXaHOwFs4luLUK2k=
|
||||
github.com/hashicorp/golang-lru/v2 v2.0.7/go.mod h1:QeFd9opnmA6QUJc5vARoKUSoFhyfM2/ZepoAG6RGpeM=
|
||||
github.com/go-jose/go-jose/v4 v4.1.4 h1:moDMcTHmvE6Groj34emNPLs/qtYXRVcd6S7NHbHz3kA=
|
||||
github.com/go-jose/go-jose/v4 v4.1.4/go.mod h1:x4oUasVrzR7071A4TnHLGSPpNOm2a21K9Kf04k1rs08=
|
||||
github.com/jackc/pgerrcode v0.0.0-20250907135507-afb5586c32a6 h1:D/V0gu4zQ3cL2WKeVNVM4r2gLxGGf6McLwgXzRTo2RQ=
|
||||
github.com/jackc/pgerrcode v0.0.0-20250907135507-afb5586c32a6/go.mod h1:a/s9Lp5W7n/DD0VrVoyJ00FbP2ytTPDVOivvn2bMlds=
|
||||
github.com/jackc/pgpassfile v1.0.0 h1:/6Hmqy13Ss2zCq62VdNG8tM1wchn8zjSGOBJ6icpsIM=
|
||||
@@ -21,14 +17,8 @@ github.com/jackc/pgx/v5 v5.11.0 h1:IzBBtyK9AHqf98cctWFifYSci2hgQR/cd56wB4p+ogg=
|
||||
github.com/jackc/pgx/v5 v5.11.0/go.mod h1:mal1tBGAFfLHvZzaYh77YS/eC6IX9OWbRV1QIIM0Jn4=
|
||||
github.com/jackc/puddle/v2 v2.2.2 h1:PR8nw+E/1w0GLuRFSmiioY6UooMp6KJv0/61nB7icHo=
|
||||
github.com/jackc/puddle/v2 v2.2.2/go.mod h1:vriiEXHvEE654aYKXXjOvZM39qJ0q+azkZFrfEOc3H4=
|
||||
github.com/mattn/go-isatty v0.0.20 h1:xfD0iDuEKnDkl03q4limB+vH+GxLEtL/jb4xVJSWWEY=
|
||||
github.com/mattn/go-isatty v0.0.20/go.mod h1:W+V8PltTTMOvKvAeJH7IuucS94S2C6jfK/D7dTCTo3Y=
|
||||
github.com/ncruces/go-strftime v1.0.0 h1:HMFp8mLCTPp341M/ZnA4qaf7ZlsbTc+miZjCLOFAw7w=
|
||||
github.com/ncruces/go-strftime v1.0.0/go.mod h1:Fwc5htZGVVkseilnfgOVb9mKy6w1naJmn9CehxcKcls=
|
||||
github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZbAQM=
|
||||
github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4=
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec h1:W09IVJc94icq4NjY3clb7Lk8O1qJ8BdBEF8z0ibU0rE=
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec/go.mod h1:qqbHyh8v60DhA7CoWK5oRCqLrMHRGoxYCSS9EjAz6Eo=
|
||||
github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME=
|
||||
github.com/stretchr/testify v1.3.0/go.mod h1:M5WIy9Dh21IEIfnGCwXGc5bZfKNJtfHm1UVUgZn+9EI=
|
||||
github.com/stretchr/testify v1.7.0/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg=
|
||||
@@ -36,46 +26,13 @@ github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu
|
||||
github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U=
|
||||
golang.org/x/crypto v0.55.0 h1:+KWHjbgOaAQ66dh/YlkZKHlz9ZUlq61AFirAR9ntP8M=
|
||||
golang.org/x/crypto v0.55.0/go.mod h1:uq0V9dE/fzQuJtbnL+2EhWOE63vo164FY8xqEnV9xis=
|
||||
golang.org/x/mod v0.38.0 h1:MECBjubtXD7yj4HrhIUcywNaGeNVUdfVnxmPajOk4yk=
|
||||
golang.org/x/mod v0.38.0/go.mod h1:V6Xz0pq8TQ3dGqVQ1FVHuelZpAL0uNhSkk9ogYP3c40=
|
||||
golang.org/x/oauth2 v0.36.0 h1:peZ/1z27fi9hUOFCAZaHyrpWG5lwe0RJEEEeH0ThlIs=
|
||||
golang.org/x/oauth2 v0.36.0/go.mod h1:YDBUJMTkDnJS+A4BP4eZBjCqtokkg1hODuPjwiGPO7Q=
|
||||
golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek=
|
||||
golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
|
||||
golang.org/x/sys v0.6.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
|
||||
golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/text v0.41.0 h1:vz/seA0lnX87Othu2f/0L24RcgrXD9/YFTSuGjj3rH8=
|
||||
golang.org/x/text v0.41.0/go.mod h1:jvf1O8ajNzZqhSrQBPbutR/EB83Cc0CFrezNQIwbb5M=
|
||||
golang.org/x/tools v0.48.0 h1:3+hClM1aLL5mjMKm5ovokw9epgRXPuu2tILgismM6RE=
|
||||
golang.org/x/tools v0.48.0/go.mod h1:08xX0orndb/F7jJxGDicx061tyd5pcMto75YMAXr6lk=
|
||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
||||
gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
|
||||
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
|
||||
gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
|
||||
modernc.org/cc/v4 v4.28.2 h1:3tQ0lf2ADtoby2EtSP+J7IE2SHwEJdP8ioR59wx7XpY=
|
||||
modernc.org/cc/v4 v4.28.2/go.mod h1:OnovgIhbbMXMu1aISnJ0wvVD1KnW+cAUJkIrAWh+kVI=
|
||||
modernc.org/ccgo/v4 v4.34.0 h1:yRLPFZieg532OT4rp4JFNIVcquwalMX26G95WQDqwCQ=
|
||||
modernc.org/ccgo/v4 v4.34.0/go.mod h1:AS5WYMyBakQ+fhsHhtP8mWB82KTGPkNNJDGfGQCe0/A=
|
||||
modernc.org/fileutil v1.4.0 h1:j6ZzNTftVS054gi281TyLjHPp6CPHr2KCxEXjEbD6SM=
|
||||
modernc.org/fileutil v1.4.0/go.mod h1:EqdKFDxiByqxLk8ozOxObDSfcVOv/54xDs/DUHdvCUU=
|
||||
modernc.org/gc/v2 v2.6.5 h1:nyqdV8q46KvTpZlsw66kWqwXRHdjIlJOhG6kxiV/9xI=
|
||||
modernc.org/gc/v2 v2.6.5/go.mod h1:YgIahr1ypgfe7chRuJi2gD7DBQiKSLMPgBQe9oIiito=
|
||||
modernc.org/gc/v3 v3.1.2 h1:ZtDCnhonXSZexk/AYsegNRV1lJGgaNZJuKjJSWKyEqo=
|
||||
modernc.org/gc/v3 v3.1.2/go.mod h1:HFK/6AGESC7Ex+EZJhJ2Gni6cTaYpSMmU/cT9RmlfYY=
|
||||
modernc.org/goabi0 v0.2.0 h1:HvEowk7LxcPd0eq6mVOAEMai46V+i7Jrj13t4AzuNks=
|
||||
modernc.org/goabi0 v0.2.0/go.mod h1:CEFRnnJhKvWT1c1JTI3Avm+tgOWbkOu5oPA8eH8LnMI=
|
||||
modernc.org/libc v1.72.3 h1:ZnDF4tXn4NBXFutMMQC4vtbTFSXhhKzR73fv0beZEAU=
|
||||
modernc.org/libc v1.72.3/go.mod h1:dn0dZNnnn1clLyvRxLxYExxiKRZIRENOfqQ8XEeg4Qs=
|
||||
modernc.org/mathutil v1.7.1 h1:GCZVGXdaN8gTqB1Mf/usp1Y/hSqgI2vAGGP4jZMCxOU=
|
||||
modernc.org/mathutil v1.7.1/go.mod h1:4p5IwJITfppl0G4sUEDtCr4DthTaT47/N3aT6MhfgJg=
|
||||
modernc.org/memory v1.11.0 h1:o4QC8aMQzmcwCK3t3Ux/ZHmwFPzE6hf2Y5LbkRs+hbI=
|
||||
modernc.org/memory v1.11.0/go.mod h1:/JP4VbVC+K5sU2wZi9bHoq2MAkCnrt2r98UGeSK7Mjw=
|
||||
modernc.org/opt v0.2.0 h1:tGyef5ApycA7FSEOMraay9SaTk5zmbx7Tu+cJs4QKZg=
|
||||
modernc.org/opt v0.2.0/go.mod h1:03fq9lsNfvkYSfxrfUhZCWPk1lm4cq4N+Bh//bEtgns=
|
||||
modernc.org/sortutil v1.2.1 h1:+xyoGf15mM3NMlPDnFqrteY07klSFxLElE2PVuWIJ7w=
|
||||
modernc.org/sortutil v1.2.1/go.mod h1:7ZI3a3REbai7gzCLcotuw9AC4VZVpYMjDzETGsSMqJE=
|
||||
modernc.org/sqlite v1.50.1 h1:l+cQvn0sd0zJJtfygGHuQJ5AjlrwXmWPw4KP3ZMwr9w=
|
||||
modernc.org/sqlite v1.50.1/go.mod h1:tcNzv5p84E0skkmJn038y+hWJbLQXQqEnQfeh5r2JLM=
|
||||
modernc.org/strutil v1.2.1 h1:UneZBkQA+DX2Rp35KcM69cSsNES9ly8mQWD71HKlOA0=
|
||||
modernc.org/strutil v1.2.1/go.mod h1:EHkiggD70koQxjVdSBM3JKM7k6L0FbGE5eymy9i3B9A=
|
||||
modernc.org/token v1.1.0 h1:Xl7Ap9dKaEs5kLoOQeQmPWevfnk/DM5qcLcYlA8ys6Y=
|
||||
modernc.org/token v1.1.0/go.mod h1:UGzOrNV1mAFSEB63lOFHIpNRUVMvYTc6yu1SMY/XTDM=
|
||||
|
||||
@@ -0,0 +1,504 @@
|
||||
package api_test
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"io"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// The bootstrap user is an administrator; everybody it creates afterwards is
|
||||
// not. These tests are about the line between them.
|
||||
|
||||
// id64 spells an id into a path segment.
|
||||
func id64(n int64) string { return strconv.FormatInt(n, 10) }
|
||||
|
||||
// member creates an ordinary user and an API key for it, and returns a caller
|
||||
// that authenticates as them. Minting the key goes through the admin's own
|
||||
// credentials, which is how a real install hands one out.
|
||||
func member(t *testing.T, s *ts, username string) (id int64, call func(method, path string, body any) *http.Response) {
|
||||
t.Helper()
|
||||
|
||||
resp := s.req(t, http.MethodPost, "/api/users",
|
||||
map[string]string{"username": username, "email": username + "@test.com"})
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("create %s: %d", username, resp.StatusCode)
|
||||
}
|
||||
var user struct {
|
||||
ID int64 `json:"id"`
|
||||
IsAdmin bool `json:"is_admin"`
|
||||
}
|
||||
decode(t, resp, &user)
|
||||
if user.IsAdmin {
|
||||
t.Fatalf("a created user must not be an administrator")
|
||||
}
|
||||
|
||||
// Into the default team as a plain member: being in a team is what lets
|
||||
// somebody work its incidents, and is separate from administering accounts.
|
||||
resp = s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/members",
|
||||
map[string]any{"user_id": user.ID, "role": "member"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Fatalf("add %s to the team: %d", username, resp.StatusCode)
|
||||
}
|
||||
|
||||
resp = s.req(t, http.MethodPost, "/api/users/"+id64(user.ID)+"/api-keys",
|
||||
map[string]string{"name": "test"})
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("mint key for %s: %d", username, resp.StatusCode)
|
||||
}
|
||||
var key struct {
|
||||
Key string `json:"key"`
|
||||
}
|
||||
decode(t, resp, &key)
|
||||
|
||||
return user.ID, func(method, path string, body any) *http.Response {
|
||||
t.Helper()
|
||||
var r io.Reader
|
||||
if body != nil {
|
||||
data, _ := json.Marshal(body)
|
||||
r = bytes.NewReader(data)
|
||||
}
|
||||
req, _ := http.NewRequest(method, s.URL+path, r)
|
||||
req.Header.Set("Authorization", "Bearer "+key.Key)
|
||||
if body != nil {
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
}
|
||||
resp, err := http.DefaultClient.Do(req)
|
||||
if err != nil {
|
||||
t.Fatalf("%s %s: %v", method, path, err)
|
||||
}
|
||||
return resp
|
||||
}
|
||||
}
|
||||
|
||||
// The whole point of the release: a user who is not an administrator cannot
|
||||
// manage other people's accounts. Every one of these was open to any
|
||||
// authenticated caller before.
|
||||
func TestAdmin_MemberIsRefusedAdministration(t *testing.T) {
|
||||
s := newTS(t)
|
||||
memberID, call := member(t, s, "member")
|
||||
|
||||
cases := []struct {
|
||||
name string
|
||||
method string
|
||||
path string
|
||||
body any
|
||||
}{
|
||||
{"create a user", http.MethodPost, "/api/users",
|
||||
map[string]string{"username": "sneaky", "email": "sneaky@test.com"}},
|
||||
{"delete the admin", http.MethodDelete, "/api/users/1", nil},
|
||||
{"grant themselves admin", http.MethodPut, "/api/users/" + id64(memberID) + "/admin",
|
||||
map[string]bool{"is_admin": true}},
|
||||
{"set the admin's password", http.MethodPut, "/api/users/1/password",
|
||||
map[string]string{"password": "hunter2-hunter2"}},
|
||||
{"mint a key for the admin", http.MethodPost, "/api/users/1/api-keys",
|
||||
map[string]string{"name": "borrowed"}},
|
||||
{"retarget the admin's notifications", http.MethodPut, "/api/users/1/notify",
|
||||
map[string]string{"ntfy_topic": "attacker-topic"}},
|
||||
}
|
||||
|
||||
for _, c := range cases {
|
||||
resp := call(c.method, c.path, c.body)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("%s: expected 403, got %d", c.name, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Being refused other people's accounts must not cost a user their own.
|
||||
func TestAdmin_MemberKeepsTheirOwnAccount(t *testing.T) {
|
||||
s := newTS(t)
|
||||
memberID, call := member(t, s, "member")
|
||||
self := "/api/users/" + id64(memberID)
|
||||
|
||||
resp := call(http.MethodPut, self+"/notify", map[string]string{"ntfy_topic": "terdut-member"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Errorf("own notify target: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
resp = call(http.MethodPut, self+"/password", map[string]string{"password": "correct-horse-battery"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK && resp.StatusCode != http.StatusNoContent {
|
||||
t.Errorf("own password: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
// An API key carries exactly the rights of its owner, so minting your own
|
||||
// is no more than signing in again.
|
||||
resp = call(http.MethodPost, self+"/api-keys", map[string]string{"name": "laptop"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Errorf("own API key: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
// And the queue still has to be able to name people.
|
||||
resp = call(http.MethodGet, "/api/users", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Errorf("list users: %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// Incident work is everybody's job; none of it is administration.
|
||||
func TestAdmin_MemberCanWorkIncidents(t *testing.T) {
|
||||
s := newTS(t)
|
||||
_, call := member(t, s, "responder")
|
||||
postWebhook(t, s, []map[string]any{
|
||||
amAlert("fp-admin", "DiskFull", "firing", "2026-09-20T10:00:00Z", zeroTime, nil),
|
||||
})
|
||||
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
method string
|
||||
path string
|
||||
}{
|
||||
{"list", http.MethodGet, "/api/incidents"},
|
||||
{"acknowledge", http.MethodPost, "/api/incidents/1/acknowledge"},
|
||||
{"resolve", http.MethodPost, "/api/incidents/1/resolve"},
|
||||
} {
|
||||
resp := call(c.method, c.path, nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Errorf("%s: expected 200, got %d", c.name, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// An install must never be left with nobody who can administer it.
|
||||
func TestAdmin_LastAdministratorIsProtected(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
resp := s.req(t, http.MethodPut, "/api/users/1/admin", map[string]bool{"is_admin": false})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusConflict {
|
||||
t.Errorf("self-demotion: expected 409, got %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
resp = s.req(t, http.MethodDelete, "/api/users/1", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusConflict {
|
||||
t.Errorf("deleting yourself: expected 409, got %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
// With a second administrator the first may stand down, but not while they
|
||||
// are the only one — which is the same rule from the other side.
|
||||
otherID, _ := member(t, s, "second")
|
||||
resp = s.req(t, http.MethodPut, "/api/users/"+id64(otherID)+"/admin", map[string]bool{"is_admin": true})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("granting admin: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
resp = s.req(t, http.MethodDelete, "/api/users/"+id64(otherID), nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Errorf("deleting the second admin: expected 204, got %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// A promoted user gets the powers with the flag, and loses them with it.
|
||||
func TestAdmin_GrantAndRevokeChangeWhatIsAllowed(t *testing.T) {
|
||||
s := newTS(t)
|
||||
memberID, call := member(t, s, "promotee")
|
||||
admin := "/api/users/" + id64(memberID) + "/admin"
|
||||
|
||||
resp := call(http.MethodPost, "/api/users", map[string]string{"username": "a", "email": "a@test.com"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Fatalf("before the grant: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
resp = s.req(t, http.MethodPut, admin, map[string]bool{"is_admin": true})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("grant: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
resp = call(http.MethodPost, "/api/users", map[string]string{"username": "b", "email": "b@test.com"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Errorf("after the grant: expected 201, got %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
resp = s.req(t, http.MethodPut, admin, map[string]bool{"is_admin": false})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("revoke: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
resp = call(http.MethodPost, "/api/users", map[string]string{"username": "c", "email": "c@test.com"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("after the revoke: expected 403, got %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// An administrator passes every team-owner check without being in the team,
|
||||
// which is what lets them repair a team whose owner has left. It has been true
|
||||
// since teams landed and nothing pinned it, so a later reading of the epic's
|
||||
// "an admin is not implicitly in every team" could quietly take it away.
|
||||
//
|
||||
// The line it draws: configuring a team, yes; reading what the team owns, no.
|
||||
// The queue below is the half that stays shut.
|
||||
func TestAdmin_ConfiguresATeamTheyAreNotIn(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
// A team the admin is deliberately not a member of. It is created by
|
||||
// somebody else, so the admin's only claim on it is the flag.
|
||||
_, call := member(t, s, "founder")
|
||||
var team struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
decode(t, call(http.MethodPost, "/api/teams", map[string]string{"name": "theirs"}), &team)
|
||||
if team.ID == 0 {
|
||||
t.Fatal("no team was created")
|
||||
}
|
||||
|
||||
var mine []struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodGet, "/api/teams", nil), &mine)
|
||||
for _, m := range mine {
|
||||
if m.ID == team.ID {
|
||||
t.Fatalf("the admin should not be a member of team %d", team.ID)
|
||||
}
|
||||
}
|
||||
|
||||
path := "/api/teams/" + id64(team.ID)
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
method string
|
||||
path string
|
||||
body any
|
||||
want int
|
||||
}{
|
||||
{"rename it", http.MethodPut, path,
|
||||
map[string]string{"name": "theirs, renamed"}, http.StatusNoContent},
|
||||
{"mint an invite", http.MethodPost, path + "/invites",
|
||||
map[string]any{"role": "member", "max_uses": 1}, http.StatusCreated},
|
||||
{"add a member", http.MethodPost, path + "/members",
|
||||
map[string]any{"user_id": 1, "role": "member"}, http.StatusNoContent},
|
||||
{"remove a member", http.MethodDelete, path + "/members/1", nil, http.StatusNoContent},
|
||||
} {
|
||||
resp := s.req(t, c.method, c.path, c.body)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != c.want {
|
||||
t.Errorf("%s: expected %d, got %d", c.name, c.want, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// The other half of the rule. An incident in that team is not the admin's
|
||||
// to read, because administration is about accounts — and the last case
|
||||
// above has just taken the admin back out of the membership.
|
||||
var integration struct {
|
||||
Key string `json:"key"`
|
||||
}
|
||||
decode(t, call(http.MethodPost, path+"/integrations",
|
||||
map[string]string{"name": "theirs alertmanager"}), &integration)
|
||||
postToIntegration(t, s, integration.Key, "fp-theirs", "TheirDiskFull")
|
||||
|
||||
var incidents []struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodGet, "/api/incidents", nil), &incidents)
|
||||
if len(incidents) != 0 {
|
||||
t.Errorf("the admin should see none of that team's incidents, got %d", len(incidents))
|
||||
}
|
||||
}
|
||||
|
||||
// The team page at /admin/teams/{id} needs the one question the test above
|
||||
// leaves shut: who is in a team the administrator is not in.
|
||||
//
|
||||
// It is answered by a separate endpoint under AdminOnly rather than by letting
|
||||
// the admin flag through requireTeamMember, and the second half of this test is
|
||||
// the reason — /api/teams/{id}/members must keep answering 404, so that "member
|
||||
// means membership and nothing else" stays true of the endpoint it was said
|
||||
// about. Reading a team's shape and reading a team's work are different things.
|
||||
func TestAdminGetTeam_ReadsAnyTeamWithoutJoiningIt(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
founderID, call := member(t, s, "founder")
|
||||
var team struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
decode(t, call(http.MethodPost, "/api/teams", map[string]string{"name": "theirs"}), &team)
|
||||
if team.ID == 0 {
|
||||
t.Fatal("no team was created")
|
||||
}
|
||||
|
||||
// The admin reads it whole, without being in it.
|
||||
var got struct {
|
||||
Team struct {
|
||||
ID int64 `json:"id"`
|
||||
Name string `json:"name"`
|
||||
Members int64 `json:"members"`
|
||||
OpenIncidents int64 `json:"open_incidents"`
|
||||
} `json:"team"`
|
||||
Members []struct {
|
||||
UserID int64 `json:"user_id"`
|
||||
Username string `json:"username"`
|
||||
Role string `json:"role"`
|
||||
} `json:"members"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodGet, "/api/admin/teams/"+id64(team.ID), nil), &got)
|
||||
|
||||
if got.Team.ID != team.ID || got.Team.Name != "theirs" {
|
||||
t.Errorf("expected team %d named theirs, got %d named %q", team.ID, got.Team.ID, got.Team.Name)
|
||||
}
|
||||
if got.Team.Members != 1 {
|
||||
t.Errorf("expected a member count of 1, got %d", got.Team.Members)
|
||||
}
|
||||
if len(got.Members) != 1 {
|
||||
t.Fatalf("expected one member, got %d", len(got.Members))
|
||||
}
|
||||
if got.Members[0].UserID != founderID || got.Members[0].Username != "founder" {
|
||||
t.Errorf("expected founder (%d), got %q (%d)",
|
||||
founderID, got.Members[0].Username, got.Members[0].UserID)
|
||||
}
|
||||
// Whoever creates a team owns it, and the page's role toggle depends on
|
||||
// that being reported rather than assumed.
|
||||
if got.Members[0].Role != "owner" {
|
||||
t.Errorf("expected the creator to be owner, got %q", got.Members[0].Role)
|
||||
}
|
||||
|
||||
// The rule this endpoint exists in order not to break. Same admin, same
|
||||
// team, the member-only endpoint: still not found.
|
||||
resp := s.req(t, http.MethodGet, "/api/teams/"+id64(team.ID)+"/members", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNotFound {
|
||||
t.Errorf("an admin outside the team must still get 404 from the member-only list, got %d",
|
||||
resp.StatusCode)
|
||||
}
|
||||
|
||||
// And the new one is administration, not membership: being in the team is
|
||||
// not enough.
|
||||
resp = call(http.MethodGet, "/api/admin/teams/"+id64(team.ID), nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("a non-admin member must get 403, got %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
path string
|
||||
want int
|
||||
}{
|
||||
{"a team that does not exist", "/api/admin/teams/999999", http.StatusNotFound},
|
||||
{"a team id that is not a number", "/api/admin/teams/nonsense", http.StatusBadRequest},
|
||||
} {
|
||||
resp := s.req(t, http.MethodGet, c.path, nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != c.want {
|
||||
t.Errorf("%s: expected %d, got %d", c.name, c.want, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A team name is trimmed when it is created, and renaming had not been, so " "
|
||||
// was a legal name to rename to and an illegal one to start with.
|
||||
func TestRenameTeam_TrimsTheName(t *testing.T) {
|
||||
s := newTS(t)
|
||||
var team struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodPost, "/api/teams", map[string]string{"name": "trimmed"}), &team)
|
||||
|
||||
path := "/api/teams/" + id64(team.ID)
|
||||
resp := s.req(t, http.MethodPut, path, map[string]string{"name": " "})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusBadRequest {
|
||||
t.Errorf("a blank name must be refused, got %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
resp = s.req(t, http.MethodPut, path, map[string]string{"name": " padded "})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Fatalf("expected 204, got %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
var got struct {
|
||||
Team struct {
|
||||
Name string `json:"name"`
|
||||
} `json:"team"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodGet, "/api/admin/teams/"+id64(team.ID), nil), &got)
|
||||
if got.Team.Name != "padded" {
|
||||
t.Errorf("expected the name to be trimmed to %q, got %q", "padded", got.Team.Name)
|
||||
}
|
||||
}
|
||||
|
||||
// The admin page's per-user view asks what somebody is in. Self or admin, like
|
||||
// the rest of the per-user endpoints.
|
||||
func TestUserTeams_SelfOrAdmin(t *testing.T) {
|
||||
s := newTS(t)
|
||||
memberID, call := member(t, s, "joiner")
|
||||
path := "/api/users/" + id64(memberID) + "/teams"
|
||||
|
||||
// member() puts them in the default team, so both readings agree on one.
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
do func() *http.Response
|
||||
}{
|
||||
{"the admin reading somebody else's", func() *http.Response { return s.req(t, http.MethodGet, path, nil) }},
|
||||
{"the user reading their own", func() *http.Response { return call(http.MethodGet, path, nil) }},
|
||||
} {
|
||||
var teams []struct {
|
||||
ID int64 `json:"id"`
|
||||
Name string `json:"name"`
|
||||
Role string `json:"role"`
|
||||
}
|
||||
decode(t, c.do(), &teams)
|
||||
if len(teams) != 1 {
|
||||
t.Fatalf("%s: expected 1 team, got %d", c.name, len(teams))
|
||||
}
|
||||
if teams[0].Role != "member" {
|
||||
t.Errorf("%s: expected role member, got %q", c.name, teams[0].Role)
|
||||
}
|
||||
}
|
||||
|
||||
// Somebody else's is not theirs to read.
|
||||
otherID, _ := member(t, s, "nosy")
|
||||
resp := call(http.MethodGet, "/api/users/"+id64(otherID)+"/teams", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("reading another user's teams: expected 403, got %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
// A user who does not exist is a 404 rather than an empty list, which is
|
||||
// how the page tells "no teams" from "no such person".
|
||||
resp = s.req(t, http.MethodGet, "/api/users/9999/teams", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNotFound {
|
||||
t.Errorf("a missing user: expected 404, got %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// The flag has to reach the client, or the web UI cannot decide what to show.
|
||||
func TestAdmin_MeReportsTheFlag(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
var me struct {
|
||||
User struct {
|
||||
IsAdmin bool `json:"is_admin"`
|
||||
} `json:"user"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodGet, "/api/me", nil), &me)
|
||||
if !me.User.IsAdmin {
|
||||
t.Error("the bootstrap user should be an administrator")
|
||||
}
|
||||
|
||||
_, call := member(t, s, "plain")
|
||||
var theirs struct {
|
||||
User struct {
|
||||
IsAdmin bool `json:"is_admin"`
|
||||
} `json:"user"`
|
||||
}
|
||||
decode(t, call(http.MethodGet, "/api/me", nil), &theirs)
|
||||
if theirs.User.IsAdmin {
|
||||
t.Error("a created user should not be an administrator")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,122 @@
|
||||
package api
|
||||
|
||||
// This file is internal (package api, not api_test) because withAdvisoryLock is
|
||||
// unexported and these tests exercise its locking semantics directly rather than
|
||||
// through the full StartArchiver/StartNotifier loop, which would make the "does
|
||||
// not run while held" case timing-dependent instead of deterministic. It opens a
|
||||
// plain connection to TERDUT_TEST_DSN rather than reusing testdb_test.go's
|
||||
// newTestDB, since that helper lives in the separate, already-compiled
|
||||
// api_test package and a Postgres advisory lock needs no schema or migration
|
||||
// to exercise.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"os"
|
||||
"testing"
|
||||
|
||||
_ "github.com/jackc/pgx/v5/stdlib"
|
||||
)
|
||||
|
||||
// advisoryTestDB opens a plain, unmigrated connection to the test database. An
|
||||
// unset DSN fails rather than skips, matching testdb_test.go's rationale: a
|
||||
// suite that quietly tests nothing is worse than one that does not run.
|
||||
func advisoryTestDB(t *testing.T) *sql.DB {
|
||||
t.Helper()
|
||||
|
||||
dsn := os.Getenv("TERDUT_TEST_DSN")
|
||||
if dsn == "" {
|
||||
t.Fatalf("TERDUT_TEST_DSN is not set: these tests need Postgres.\n" +
|
||||
"Run `make test-db` for a local one, then\n" +
|
||||
" export TERDUT_TEST_DSN=postgres://terdut:terdut@localhost:5432/terdut_test?sslmode=disable")
|
||||
}
|
||||
|
||||
db, err := sql.Open("pgx", dsn)
|
||||
if err != nil {
|
||||
t.Fatalf("connect to TERDUT_TEST_DSN: %v", err)
|
||||
}
|
||||
t.Cleanup(func() { db.Close() })
|
||||
return db
|
||||
}
|
||||
|
||||
func TestWithAdvisoryLock_RunsWhenFree(t *testing.T) {
|
||||
db := advisoryTestDB(t)
|
||||
ctx := context.Background()
|
||||
|
||||
ran := false
|
||||
withAdvisoryLock(ctx, db, archiverLockKey, "test", func() { ran = true })
|
||||
|
||||
if !ran {
|
||||
t.Fatal("fn did not run although the lock was free")
|
||||
}
|
||||
}
|
||||
|
||||
func TestWithAdvisoryLock_SkipsWhileHeldElsewhere(t *testing.T) {
|
||||
db := advisoryTestDB(t)
|
||||
ctx := context.Background()
|
||||
|
||||
// Hold the lock on a connection of our own, standing in for another
|
||||
// replica mid-pass.
|
||||
holder, err := db.Conn(ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("acquire holder connection: %v", err)
|
||||
}
|
||||
defer holder.Close()
|
||||
if _, err := holder.ExecContext(ctx, "SELECT pg_advisory_lock($1)", archiverLockKey); err != nil {
|
||||
t.Fatalf("pre-acquire lock: %v", err)
|
||||
}
|
||||
|
||||
ran := false
|
||||
withAdvisoryLock(ctx, db, archiverLockKey, "test", func() { ran = true })
|
||||
if ran {
|
||||
t.Fatal("fn ran although another connection already held the lock")
|
||||
}
|
||||
|
||||
if _, err := holder.ExecContext(ctx, "SELECT pg_advisory_unlock($1)", archiverLockKey); err != nil {
|
||||
t.Fatalf("release held lock: %v", err)
|
||||
}
|
||||
|
||||
// Now that the holder released it, the next caller should get it.
|
||||
ran = false
|
||||
withAdvisoryLock(ctx, db, archiverLockKey, "test", func() { ran = true })
|
||||
if !ran {
|
||||
t.Fatal("fn did not run after the other connection released the lock")
|
||||
}
|
||||
}
|
||||
|
||||
func TestWithAdvisoryLock_ReleasesAfterFnReturns(t *testing.T) {
|
||||
db := advisoryTestDB(t)
|
||||
ctx := context.Background()
|
||||
|
||||
withAdvisoryLock(ctx, db, notifierLockKey, "test", func() {})
|
||||
|
||||
// If the first call had leaked the lock, this one would see it held and
|
||||
// skip, leaving ran false.
|
||||
ran := false
|
||||
withAdvisoryLock(ctx, db, notifierLockKey, "test", func() { ran = true })
|
||||
if !ran {
|
||||
t.Fatal("fn did not run on a later call: the earlier call leaked its lock")
|
||||
}
|
||||
}
|
||||
|
||||
func TestWithAdvisoryLock_KeysAreIndependent(t *testing.T) {
|
||||
db := advisoryTestDB(t)
|
||||
ctx := context.Background()
|
||||
|
||||
holder, err := db.Conn(ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("acquire holder connection: %v", err)
|
||||
}
|
||||
defer holder.Close()
|
||||
if _, err := holder.ExecContext(ctx, "SELECT pg_advisory_lock($1)", archiverLockKey); err != nil {
|
||||
t.Fatalf("pre-acquire archiver lock: %v", err)
|
||||
}
|
||||
defer holder.ExecContext(ctx, "SELECT pg_advisory_unlock($1)", archiverLockKey)
|
||||
|
||||
// Holding archiverLockKey must not block notifierLockKey.
|
||||
ran := false
|
||||
withAdvisoryLock(ctx, db, notifierLockKey, "test", func() { ran = true })
|
||||
if !ran {
|
||||
t.Fatal("fn did not run under a different key although only archiverLockKey was held")
|
||||
}
|
||||
}
|
||||
+115
-37
@@ -4,9 +4,12 @@ import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"log"
|
||||
"net/http"
|
||||
"time"
|
||||
|
||||
"github.com/go-chi/chi/v5"
|
||||
)
|
||||
|
||||
// Values for alerts.resolution_source, recording why an alert left the firing
|
||||
@@ -70,36 +73,65 @@ type ingested struct {
|
||||
deadman bool
|
||||
}
|
||||
|
||||
func handleAlertmanagerWebhook(db *sql.DB, notify NotifyConfig, deadman DeadmanConfig) http.HandlerFunc {
|
||||
// handleIntegrationWebhook receives alerts on a team's own integration key.
|
||||
// The key in the path is both the credential and the routing: it says who may
|
||||
// post, and which team the alerts belong to.
|
||||
func handleIntegrationWebhook(db *sql.DB, notify NotifyConfig) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
var payload amPayload
|
||||
if err := decodeJSON(r, &payload); err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid payload"))
|
||||
src, err := sourceForKey(r.Context(), db, chi.URLParam(r, "key"))
|
||||
if err != nil {
|
||||
if errors.Is(err, errUnknownIntegration) {
|
||||
// 401 and not 404: the path is real, the key is not, and a
|
||||
// sender misconfigured this way should say so in its own logs
|
||||
// rather than believe it is delivering.
|
||||
respond(w, http.StatusUnauthorized, errResp("unknown integration key"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
|
||||
// Alertmanager retries anything that is not 2xx, and a retry of a payload
|
||||
// we failed to store is more useful than an error it cannot act on — so
|
||||
// failures are logged, not surfaced.
|
||||
if err := ingest(r.Context(), db, notify, deadman, payload); err != nil {
|
||||
log.Printf("webhook ingest (group %q): %v", payload.GroupKey, err)
|
||||
}
|
||||
|
||||
w.WriteHeader(http.StatusOK)
|
||||
receiveWebhook(w, r, db, notify, src)
|
||||
}
|
||||
}
|
||||
|
||||
func receiveWebhook(w http.ResponseWriter, r *http.Request, db *sql.DB, notify NotifyConfig, src alertSource) {
|
||||
teamID := src.teamID
|
||||
var payload amPayload
|
||||
if err := decodeJSON(r, &payload); err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid payload"))
|
||||
return
|
||||
}
|
||||
|
||||
// Alertmanager retries anything that is not 2xx, and a retry of a payload
|
||||
// we failed to store is more useful than an error it cannot act on — so
|
||||
// failures are logged, not surfaced.
|
||||
if err := ingest(r.Context(), db, notify, src, payload); err != nil {
|
||||
log.Printf("webhook ingest (team %d, group %q): %v", teamID, payload.GroupKey, err)
|
||||
}
|
||||
|
||||
w.WriteHeader(http.StatusOK)
|
||||
}
|
||||
|
||||
// ingest stores a payload's alerts and reconciles the incident for its group.
|
||||
// The whole payload is one transaction: an incident that opened but whose alerts
|
||||
// failed to link would be a work item nobody could act on.
|
||||
func ingest(ctx context.Context, db *sql.DB, notify NotifyConfig, deadman DeadmanConfig, payload amPayload) error {
|
||||
func ingest(ctx context.Context, db *sql.DB, notify NotifyConfig, src alertSource, payload amPayload) error {
|
||||
teamID := src.teamID
|
||||
tx, err := db.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer tx.Rollback() //nolint:errcheck
|
||||
|
||||
accepted, err := upsertAlerts(ctx, tx, deadman, payload.Alerts)
|
||||
// Which arriving alerts are heartbeats is the team's own answer, read
|
||||
// inside the transaction so an owner editing it mid-payload cannot split
|
||||
// one webhook across two interpretations.
|
||||
deadman, err := deadmanSetForTeam(ctx, tx, teamID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
accepted, err := upsertAlerts(ctx, tx, deadman, src, payload.Alerts)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -108,7 +140,7 @@ func ingest(ctx context.Context, db *sql.DB, notify NotifyConfig, deadman Deadma
|
||||
// resolution cascade are recomputed once per incident at the end.
|
||||
touched := map[int64]bool{}
|
||||
|
||||
incidentID, err := incidentForGroup(ctx, tx, notify, payload, accepted)
|
||||
incidentID, err := incidentForGroup(ctx, tx, notify, teamID, payload, accepted)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -156,7 +188,8 @@ func ingest(ctx context.Context, db *sql.DB, notify NotifyConfig, deadman Deadma
|
||||
|
||||
// upsertAlerts stores each alert of a payload and reports what changed. Payloads
|
||||
// the ordering guard rejected are left out entirely.
|
||||
func upsertAlerts(ctx context.Context, tx *sql.Tx, deadman DeadmanConfig, alerts []amAlert) ([]ingested, error) {
|
||||
func upsertAlerts(ctx context.Context, tx *sql.Tx, deadman deadmanSet, src alertSource, alerts []amAlert) ([]ingested, error) {
|
||||
teamID := src.teamID
|
||||
now := time.Now().Unix()
|
||||
accepted := make([]ingested, 0, len(alerts))
|
||||
|
||||
@@ -171,7 +204,8 @@ func upsertAlerts(ctx context.Context, tx *sql.Tx, deadman DeadmanConfig, alerts
|
||||
var prevStartsAt int64
|
||||
existed := true
|
||||
switch err := tx.QueryRowContext(ctx,
|
||||
"SELECT status, starts_at FROM alerts WHERE fingerprint = $1", a.Fingerprint,
|
||||
"SELECT status, starts_at FROM alerts WHERE team_id = $1 AND fingerprint = $2",
|
||||
teamID, a.Fingerprint,
|
||||
).Scan(&prevStatus, &prevStartsAt); {
|
||||
case err == sql.ErrNoRows:
|
||||
existed = false
|
||||
@@ -210,10 +244,10 @@ func upsertAlerts(ctx context.Context, tx *sql.Tx, deadman DeadmanConfig, alerts
|
||||
// undone by a stale retry.
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
INSERT INTO alerts
|
||||
(fingerprint, name, status, labels, annotations, starts_at, ends_at,
|
||||
generator_url, received_at, resolution_source)
|
||||
VALUES ($1, $2, $3, $4::jsonb, $5::jsonb, $6, $7, $8, $9, $10)
|
||||
ON CONFLICT (fingerprint) DO UPDATE SET
|
||||
(team_id, fingerprint, name, status, labels, annotations, starts_at, ends_at,
|
||||
generator_url, received_at, resolution_source, integration_id)
|
||||
VALUES ($1, $2, $3, $4, $5::jsonb, $6::jsonb, $7, $8, $9, $10, $11, $12)
|
||||
ON CONFLICT (team_id, fingerprint) DO UPDATE SET
|
||||
status = excluded.status,
|
||||
labels = excluded.labels,
|
||||
annotations = excluded.annotations,
|
||||
@@ -226,6 +260,8 @@ func upsertAlerts(ctx context.Context, tx *sql.Tx, deadman DeadmanConfig, alerts
|
||||
-- is a breaking API change — see models.Alert.ReceivedAt.
|
||||
received_at = excluded.received_at,
|
||||
resolution_source = excluded.resolution_source,
|
||||
-- Last sender wins; see migration 010.
|
||||
integration_id = excluded.integration_id,
|
||||
-- A re-fire makes the alert current again, so it leaves the archive.
|
||||
archived_at = CASE WHEN excluded.status = 'firing'
|
||||
THEN NULL ELSE alerts.archived_at END
|
||||
@@ -233,10 +269,10 @@ func upsertAlerts(ctx context.Context, tx *sql.Tx, deadman DeadmanConfig, alerts
|
||||
OR (excluded.starts_at = alerts.starts_at
|
||||
AND (alerts.resolution_source = '`+resolutionDeadman+`'
|
||||
OR NOT (alerts.status = 'resolved' AND excluded.status = 'firing')))`,
|
||||
a.Fingerprint, name, a.Status,
|
||||
teamID, a.Fingerprint, name, a.Status,
|
||||
string(labelsJSON), string(annotationsJSON),
|
||||
a.StartsAt.Unix(), endsAtUnix,
|
||||
a.GeneratorURL, now, resolutionSource,
|
||||
a.GeneratorURL, now, resolutionSource, src.integrationID,
|
||||
); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -245,7 +281,8 @@ func upsertAlerts(ctx context.Context, tx *sql.Tx, deadman DeadmanConfig, alerts
|
||||
var curStatus string
|
||||
var curStartsAt int64
|
||||
if err := tx.QueryRowContext(ctx,
|
||||
"SELECT id, status, starts_at FROM alerts WHERE fingerprint = $1", a.Fingerprint,
|
||||
"SELECT id, status, starts_at FROM alerts WHERE team_id = $1 AND fingerprint = $2",
|
||||
teamID, a.Fingerprint,
|
||||
).Scan(&id, &curStatus, &curStartsAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -284,7 +321,7 @@ func upsertAlerts(ctx context.Context, tx *sql.Tx, deadman DeadmanConfig, alerts
|
||||
// Heartbeats do not count as anything here. A group of nothing but dead man's
|
||||
// switch alerts opens no incident at all, and a mixed group gets an incident for
|
||||
// its real alerts only.
|
||||
func incidentForGroup(ctx context.Context, tx *sql.Tx, notify NotifyConfig, payload amPayload, accepted []ingested) (int64, error) {
|
||||
func incidentForGroup(ctx context.Context, tx *sql.Tx, notify NotifyConfig, teamID int64, payload amPayload, accepted []ingested) (int64, error) {
|
||||
var firstName string
|
||||
anyFiring, anyNew := false, false
|
||||
for _, a := range accepted {
|
||||
@@ -315,7 +352,8 @@ func incidentForGroup(ctx context.Context, tx *sql.Tx, notify NotifyConfig, payl
|
||||
|
||||
var id int64
|
||||
switch err := tx.QueryRowContext(ctx,
|
||||
"SELECT id FROM incidents WHERE group_key = $1 AND resolved_at IS NULL", groupKey,
|
||||
"SELECT id FROM incidents WHERE team_id = $1 AND group_key = $2 AND resolved_at IS NULL",
|
||||
teamID, groupKey,
|
||||
).Scan(&id); {
|
||||
case err == nil:
|
||||
return id, nil
|
||||
@@ -326,7 +364,7 @@ func incidentForGroup(ctx context.Context, tx *sql.Tx, notify NotifyConfig, payl
|
||||
if !anyNew {
|
||||
return 0, nil
|
||||
}
|
||||
return openIncident(ctx, tx, notify, groupKey,
|
||||
return openIncident(ctx, tx, notify, teamID, groupKey,
|
||||
incidentTitle(payload.GroupLabels, firstName), payload.GroupLabels, nil)
|
||||
}
|
||||
|
||||
@@ -338,8 +376,19 @@ func incidentForGroup(ctx context.Context, tx *sql.Tx, notify NotifyConfig, payl
|
||||
// its own. Hence the querier rather than a *sql.Tx. A nil severity leaves the
|
||||
// column for refreshSeverity to fill from the member alerts; the sweeper passes
|
||||
// one because its incidents have no members to derive it from.
|
||||
func openIncident(ctx context.Context, q querier, notify NotifyConfig, groupKey, title string, groupLabels map[string]string, severity *string) (int64, error) {
|
||||
onCall, err := currentOnCall(ctx, q)
|
||||
//
|
||||
// Both callers get here only after their own SELECT found no open incident for
|
||||
// this group_key — but on more than one replica, two webhook deliveries for the
|
||||
// very first occurrence of a brand-new group_key can both pass that SELECT
|
||||
// before either INSERTs. ON CONFLICT DO NOTHING against
|
||||
// incidents_open_group_key_idx is what makes the loser's INSERT a no-op instead
|
||||
// of a unique-violation error that would otherwise roll back its entire
|
||||
// payload; existingOpenIncident then hands it the winner's row. Postgres
|
||||
// resolves that conflict only once the winner's transaction has committed (or
|
||||
// rolled back), so by the time this RETURNING comes back empty, the winner's
|
||||
// row is guaranteed visible to that follow-up SELECT.
|
||||
func openIncident(ctx context.Context, q querier, notify NotifyConfig, teamID int64, groupKey, title string, groupLabels map[string]string, severity *string) (int64, error) {
|
||||
onCall, err := currentOnCall(ctx, q, teamID)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
@@ -351,12 +400,20 @@ func openIncident(ctx context.Context, q querier, notify NotifyConfig, groupKey,
|
||||
|
||||
var id int64
|
||||
err = q.QueryRowContext(ctx, `
|
||||
INSERT INTO incidents (group_key, title, group_labels, status, severity, triggered_at, assigned_to)
|
||||
VALUES ($1, $2, $3::jsonb, 'triggered', $4, $5, $6)
|
||||
INSERT INTO incidents (team_id, group_key, title, group_labels, signature, status, severity, triggered_at, assigned_to)
|
||||
VALUES ($1, $2, $3, $4::jsonb, $5, 'triggered', $6, $7, $8)
|
||||
ON CONFLICT (team_id, group_key) WHERE resolved_at IS NULL DO NOTHING
|
||||
RETURNING id`,
|
||||
groupKey, title, string(labelsJSON), severity,
|
||||
teamID, groupKey, title, string(labelsJSON), incidentSignature(groupLabels, title), severity,
|
||||
time.Now().Unix(), onCall).Scan(&id)
|
||||
if err != nil {
|
||||
switch {
|
||||
case err == sql.ErrNoRows:
|
||||
// Lost the race: someone else's incident for this group_key exists now.
|
||||
// Everything below — the trigger event, assignment, page, escalation
|
||||
// clock — already happened for that row when it was created; attach to
|
||||
// it rather than fail this call (and the whole payload) outright.
|
||||
return existingOpenIncident(ctx, q, teamID, groupKey)
|
||||
case err != nil:
|
||||
return 0, err
|
||||
}
|
||||
|
||||
@@ -370,12 +427,33 @@ func openIncident(ctx context.Context, q querier, notify NotifyConfig, groupKey,
|
||||
}
|
||||
}
|
||||
|
||||
// Queue the page, but do not send it here: this runs inside a transaction on
|
||||
// a single-connection pool, so an HTTP call would hold up every other
|
||||
// request. The notifier picks the row up within a tick.
|
||||
// Queue the page, but do not send it here: this runs inside the webhook's
|
||||
// transaction, and an HTTP call would hold a connection open across a
|
||||
// network round trip. The notifier picks the row up within a tick.
|
||||
if err := enqueueOpened(ctx, q, notify, id, onCall); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
|
||||
// And start the escalation clock, if the team keeps one. In the same
|
||||
// transaction, so an incident is never briefly open with nobody counting.
|
||||
if err := startEscalation(ctx, q, id, teamID); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return id, nil
|
||||
}
|
||||
|
||||
// existingOpenIncident looks up the open incident openIncident's own INSERT just
|
||||
// lost a conflict against — the same lookup incidentForGroup does before ever
|
||||
// calling openIncident, repeated here for the caller that arrived second.
|
||||
func existingOpenIncident(ctx context.Context, q querier, teamID int64, groupKey string) (int64, error) {
|
||||
var id int64
|
||||
err := q.QueryRowContext(ctx,
|
||||
"SELECT id FROM incidents WHERE team_id = $1 AND group_key = $2 AND resolved_at IS NULL",
|
||||
teamID, groupKey,
|
||||
).Scan(&id)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return id, nil
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,112 @@
|
||||
package api_test
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"sync"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// TestWebhook_ConcurrentFirstOccurrenceOpensOneIncident reproduces two
|
||||
// replicas racing the very first webhook delivery for a brand-new group_key:
|
||||
// both see no open incident yet (incidentForGroup's own SELECT finds
|
||||
// nothing) and race openIncident's INSERT.
|
||||
//
|
||||
// The DB's own unique index already guarantees at most one incident either
|
||||
// way, with or without this fix — so "exactly one incident" alone cannot
|
||||
// tell a fixed run from a broken one. What ON CONFLICT handling actually
|
||||
// changes is what happens to the *loser*: before it, the loser's INSERT hit
|
||||
// incidents_open_group_key_idx's unique violation, which — since
|
||||
// upsertAlerts ran earlier in that same transaction — rolled back its whole
|
||||
// payload, alert insert included. ingest's error is only logged and
|
||||
// receiveWebhook answers 200 regardless, so nothing ever retried it: the
|
||||
// loser's alert silently never existed. That is the regression signal this
|
||||
// test checks — every caller's fingerprint must show up in /api/alerts, not
|
||||
// just the winner's.
|
||||
func TestWebhook_ConcurrentFirstOccurrenceOpensOneIncident(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
const callers = 8
|
||||
const groupKey = "race-group"
|
||||
|
||||
// Every caller needs its own fingerprint. A shared one would serialize all
|
||||
// of them at upsertAlerts' own ON CONFLICT (team_id, fingerprint) row lock,
|
||||
// long before any of them reached incidentForGroup — which would hide the
|
||||
// very race this test exists to force.
|
||||
bodies := make([][]byte, callers)
|
||||
for i := range callers {
|
||||
payload := map[string]any{
|
||||
"version": "4",
|
||||
"status": "firing",
|
||||
"groupKey": groupKey,
|
||||
"groupLabels": map[string]string{"alertname": "RaceAlert"},
|
||||
"alerts": []map[string]any{amAlert(fmt.Sprintf("fp-race-%d", i), "RaceAlert", "firing",
|
||||
"2026-05-20T10:00:00Z", "0001-01-01T00:00:00Z", nil)},
|
||||
}
|
||||
bodies[i], _ = json.Marshal(payload)
|
||||
}
|
||||
|
||||
// A start line, so every request is fired as close to simultaneously as
|
||||
// goroutine scheduling allows, rather than trickling out one dial at a
|
||||
// time — the race window is the gap between incidentForGroup's SELECT and
|
||||
// openIncident's INSERT, which a staggered start could easily miss.
|
||||
var ready sync.WaitGroup
|
||||
start := make(chan struct{})
|
||||
statuses := make([]int, callers)
|
||||
var wg sync.WaitGroup
|
||||
for i := range callers {
|
||||
ready.Add(1)
|
||||
wg.Add(1)
|
||||
go func(i int) {
|
||||
defer wg.Done()
|
||||
ready.Done()
|
||||
<-start
|
||||
resp, err := http.Post(s.URL+"/api/integrations/"+s.ingestKey+"/alertmanager",
|
||||
"application/json", bytes.NewReader(bodies[i]))
|
||||
if err != nil {
|
||||
t.Errorf("post webhook #%d: %v", i, err)
|
||||
return
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
statuses[i] = resp.StatusCode
|
||||
}(i)
|
||||
}
|
||||
ready.Wait()
|
||||
close(start)
|
||||
wg.Wait()
|
||||
|
||||
for i, code := range statuses {
|
||||
if code != http.StatusOK {
|
||||
t.Errorf("webhook #%d returned %d, want 200", i, code)
|
||||
}
|
||||
}
|
||||
|
||||
var matched []any
|
||||
for _, inc := range listIncidents(t, s, "") {
|
||||
if inc["group_key"] == groupKey {
|
||||
matched = append(matched, inc["id"])
|
||||
}
|
||||
}
|
||||
if len(matched) != 1 {
|
||||
t.Fatalf("expected exactly 1 incident for group_key %q after %d concurrent deliveries, got %d: %v",
|
||||
groupKey, callers, len(matched), matched)
|
||||
}
|
||||
|
||||
var alerts []map[string]any
|
||||
decode(t, s.req(t, http.MethodGet, "/api/alerts", nil), &alerts)
|
||||
seen := map[string]bool{}
|
||||
for _, a := range alerts {
|
||||
if fp, ok := a["fingerprint"].(string); ok {
|
||||
seen[fp] = true
|
||||
}
|
||||
}
|
||||
for i := range callers {
|
||||
fp := fmt.Sprintf("fp-race-%d", i)
|
||||
if !seen[fp] {
|
||||
t.Errorf("alert %q is missing: its delivery's whole payload was silently rolled back "+
|
||||
"when it lost the race for the incident", fp)
|
||||
}
|
||||
}
|
||||
}
|
||||
+15
-6
@@ -19,7 +19,7 @@ import (
|
||||
// incident_alerts rather than as a column here, because one alert row is reused
|
||||
// across occurrences and belongs to a different incident each time.
|
||||
const alertSelectFrom = `
|
||||
SELECT a.id, a.fingerprint, a.name, a.status,
|
||||
SELECT a.id, a.team_id, t.name, a.fingerprint, a.name, a.status,
|
||||
a.labels, a.annotations,
|
||||
a.starts_at, a.ends_at, a.generator_url, a.received_at,
|
||||
(SELECT ia.incident_id
|
||||
@@ -29,7 +29,8 @@ const alertSelectFrom = `
|
||||
ORDER BY i.triggered_at DESC, i.id DESC
|
||||
LIMIT 1),
|
||||
a.resolution_source, a.archived_at
|
||||
FROM alerts a`
|
||||
FROM alerts a
|
||||
JOIN teams t ON t.id = a.team_id`
|
||||
|
||||
func handleListAlerts(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
@@ -38,6 +39,13 @@ func handleListAlerts(db *sql.DB) http.HandlerFunc {
|
||||
where := []string{}
|
||||
args := &sqlArgs{}
|
||||
|
||||
where = append(where, "a.team_id = ANY("+args.add(callerTeamIDs(r.Context()))+")")
|
||||
if team := q.Get("team_id"); team != "" {
|
||||
if n, err := strconv.ParseInt(team, 10, 64); err == nil {
|
||||
where = append(where, "a.team_id = "+args.add(n))
|
||||
}
|
||||
}
|
||||
|
||||
if status := q.Get("status"); status != "" {
|
||||
where = append(where, "a.status = "+args.add(status))
|
||||
}
|
||||
@@ -107,7 +115,7 @@ func handleGetAlert(db *sql.DB) http.HandlerFunc {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid alert id"))
|
||||
return
|
||||
}
|
||||
a, err := fetchAlert(r.Context(), db, id)
|
||||
a, err := fetchAlert(r.Context(), db, id, callerTeamIDs(r.Context()))
|
||||
if err == sql.ErrNoRows {
|
||||
respond(w, http.StatusNotFound, errResp("alert not found"))
|
||||
return
|
||||
@@ -121,8 +129,9 @@ func handleGetAlert(db *sql.DB) http.HandlerFunc {
|
||||
}
|
||||
|
||||
// fetchAlert loads a single alert by ID using the shared query.
|
||||
func fetchAlert(ctx context.Context, db *sql.DB, id int64) (models.Alert, error) {
|
||||
return scanAlert(db.QueryRowContext(ctx, alertSelectFrom+" WHERE a.id = $1", id))
|
||||
func fetchAlert(ctx context.Context, db *sql.DB, id int64, teamIDs []int64) (models.Alert, error) {
|
||||
return scanAlert(db.QueryRowContext(ctx,
|
||||
alertSelectFrom+" WHERE a.id = $1 AND a.team_id = ANY($2)", id, teamIDs))
|
||||
}
|
||||
|
||||
// scanner is satisfied by both *sql.Row and *sql.Rows.
|
||||
@@ -137,7 +146,7 @@ func scanAlert(s scanner) (models.Alert, error) {
|
||||
var endsAtUnix, archivedAtUnix *int64
|
||||
|
||||
if err := s.Scan(
|
||||
&a.ID, &a.Fingerprint, &a.Name, &a.Status,
|
||||
&a.ID, &a.TeamID, &a.TeamName, &a.Fingerprint, &a.Name, &a.Status,
|
||||
&labelsJSON, &annotationsJSON,
|
||||
&startsAtUnix, &endsAtUnix,
|
||||
&a.GeneratorURL, &receivedAtUnix,
|
||||
|
||||
+94
-26
@@ -9,20 +9,27 @@ import (
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||
)
|
||||
|
||||
// ts wraps httptest.Server with a pre-bootstrapped API key. db is exposed so
|
||||
// tests can age rows directly — the sweeper's inputs are wall-clock timestamps.
|
||||
type ts struct {
|
||||
*httptest.Server
|
||||
key string
|
||||
db *sql.DB
|
||||
notify api.NotifyConfig
|
||||
deadman api.DeadmanConfig
|
||||
key string
|
||||
// ingestKey is an integration key for the default team: the only way in
|
||||
// since the unauthenticated webhook was removed, so the tests exercise the
|
||||
// same path production does.
|
||||
ingestKey string
|
||||
db *sql.DB
|
||||
notify api.NotifyConfig
|
||||
deadman api.DeadmanConfig
|
||||
}
|
||||
|
||||
// newTS builds a server over a fresh database. Notifications are off
|
||||
@@ -37,16 +44,23 @@ func newTS(t *testing.T, notify ...api.NotifyConfig) *ts {
|
||||
return newDeadmanTS(t, api.DeadmanConfig{}, cfg)
|
||||
}
|
||||
|
||||
// newDeadmanTS is newTS with dead man's switch handling configured.
|
||||
// newDeadmanTS is newTS with the default team's dead man's switches configured.
|
||||
func newDeadmanTS(t *testing.T, deadman api.DeadmanConfig, notify ...api.NotifyConfig) *ts {
|
||||
t.Helper()
|
||||
var cfg api.NotifyConfig
|
||||
if len(notify) > 0 {
|
||||
cfg = notify[0]
|
||||
}
|
||||
return newTSWith(t, deadman, cfg, testConfig())
|
||||
}
|
||||
|
||||
// newTSWith is newDeadmanTS with the server's own configuration supplied, for
|
||||
// tests of behaviour that config switches on, such as single sign-on.
|
||||
func newTSWith(t *testing.T, deadman api.DeadmanConfig, cfg api.NotifyConfig, conf config.Config) *ts {
|
||||
t.Helper()
|
||||
|
||||
database := newTestDB(t)
|
||||
srv := httptest.NewServer(api.NewRouter(database, cfg, deadman))
|
||||
srv := httptest.NewServer(api.NewRouter(database, cfg, conf, "test"))
|
||||
t.Cleanup(srv.Close)
|
||||
|
||||
body, _ := json.Marshal(map[string]string{"username": "admin", "email": "admin@test.com"})
|
||||
@@ -62,7 +76,46 @@ func newDeadmanTS(t *testing.T, deadman api.DeadmanConfig, notify ...api.NotifyC
|
||||
json.NewDecoder(resp.Body).Decode(&result)
|
||||
key := result["api_key"].(map[string]any)["key"].(string)
|
||||
|
||||
return &ts{Server: srv, key: key, db: database, notify: cfg, deadman: deadman}
|
||||
s := &ts{Server: srv, key: key, db: database, notify: cfg, deadman: deadman}
|
||||
|
||||
var integration struct {
|
||||
Key string `json:"key"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/integrations",
|
||||
map[string]string{"name": "test"}), &integration)
|
||||
if integration.Key == "" {
|
||||
t.Fatal("no integration key was returned")
|
||||
}
|
||||
s.ingestKey = integration.Key
|
||||
|
||||
// Dead man's switches belong to a team now, so a test that wants them
|
||||
// configures the default team the way an owner would.
|
||||
if deadman.Timeout > 0 {
|
||||
setTeamDeadman(t, s, deadman)
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// setTeamDeadman gives the default team one switch per configured matcher, over
|
||||
// the API, the way an owner would add them.
|
||||
func setTeamDeadman(t *testing.T, s *ts, cfg api.DeadmanConfig) {
|
||||
t.Helper()
|
||||
for _, m := range cfg.Matchers {
|
||||
parts := []string{"alertname=" + m.Name}
|
||||
for k, v := range m.Labels {
|
||||
parts = append(parts, k+"="+v)
|
||||
}
|
||||
sort.Strings(parts[1:])
|
||||
resp := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches", map[string]any{
|
||||
"matcher": strings.Join(parts, ","),
|
||||
"timeout_seconds": int64(cfg.Timeout.Seconds()),
|
||||
"severity": cfg.Severity,
|
||||
})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("add a dead man's switch: %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// exec runs a statement against the test database.
|
||||
@@ -201,7 +254,8 @@ func postWebhook(t *testing.T, s *ts, alerts []map[string]any, groupKey ...strin
|
||||
}
|
||||
}
|
||||
data, _ := json.Marshal(payload)
|
||||
resp, err := http.Post(s.URL+"/api/alertmanager/webhook", "application/json", bytes.NewReader(data))
|
||||
resp, err := http.Post(s.URL+"/api/integrations/"+s.ingestKey+"/alertmanager",
|
||||
"application/json", bytes.NewReader(data))
|
||||
if err != nil {
|
||||
t.Fatalf("post webhook: %v", err)
|
||||
}
|
||||
@@ -276,14 +330,14 @@ func TestAlertUpsert_DifferentFingerprintsStored(t *testing.T) {
|
||||
func TestSchedule_ConflictOnSameDate(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
first := s.req(t, http.MethodPost, "/api/schedule",
|
||||
first := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/schedule",
|
||||
map[string]any{"user_id": 1, "dates": []string{"2026-06-01"}})
|
||||
if first.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("first assignment returned %d", first.StatusCode)
|
||||
}
|
||||
first.Body.Close()
|
||||
|
||||
second := s.req(t, http.MethodPost, "/api/schedule",
|
||||
second := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/schedule",
|
||||
map[string]any{"user_id": 1, "dates": []string{"2026-06-01"}})
|
||||
if second.StatusCode != http.StatusConflict {
|
||||
t.Errorf("expected 409 on duplicate date, got %d", second.StatusCode)
|
||||
@@ -295,11 +349,11 @@ func TestSchedule_MultiDateRollbackOnConflict(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
// Claim 2026-06-10 first.
|
||||
s.req(t, http.MethodPost, "/api/schedule",
|
||||
s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/schedule",
|
||||
map[string]any{"user_id": 1, "dates": []string{"2026-06-10"}}).Body.Close()
|
||||
|
||||
// Try to assign two dates in one request where the second conflicts.
|
||||
resp := s.req(t, http.MethodPost, "/api/schedule",
|
||||
resp := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/schedule",
|
||||
map[string]any{"user_id": 1, "dates": []string{"2026-06-09", "2026-06-10"}})
|
||||
if resp.StatusCode != http.StatusConflict {
|
||||
t.Fatalf("expected 409, got %d", resp.StatusCode)
|
||||
@@ -307,7 +361,7 @@ func TestSchedule_MultiDateRollbackOnConflict(t *testing.T) {
|
||||
resp.Body.Close()
|
||||
|
||||
// 2026-06-09 must NOT have been committed (transaction rolled back).
|
||||
listResp := s.req(t, http.MethodGet, "/api/schedule?from=2026-06-09&to=2026-06-09", nil)
|
||||
listResp := s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/schedule?from=2026-06-09&to=2026-06-09", nil)
|
||||
var entries []any
|
||||
decode(t, listResp, &entries)
|
||||
if len(entries) != 0 {
|
||||
@@ -321,21 +375,35 @@ func TestSchedule_MultiDateRollbackOnConflict(t *testing.T) {
|
||||
|
||||
// addUser creates a second person to hand a shift to. The bootstrap user is
|
||||
// admin, id 1.
|
||||
// addUser creates a user and puts them in the default team, because a user who
|
||||
// is in no team can be paged by nobody and take no shift — which is the rule
|
||||
// these tests exercise around, not the one they are testing.
|
||||
func addUser(t *testing.T, s *ts, username string) {
|
||||
t.Helper()
|
||||
resp := s.req(t, http.MethodPost, "/api/users",
|
||||
map[string]any{"username": username, "email": username + "@test.com"})
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
resp.Body.Close()
|
||||
t.Fatalf("create user returned %d", resp.StatusCode)
|
||||
}
|
||||
var user struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
decode(t, resp, &user)
|
||||
|
||||
member := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/members",
|
||||
map[string]any{"user_id": user.ID, "role": "member"})
|
||||
defer member.Body.Close()
|
||||
if member.StatusCode != http.StatusNoContent {
|
||||
t.Fatalf("add %s to the team returned %d", username, member.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// scheduleHolder reports who is on call for one date, or "" for nobody.
|
||||
func scheduleHolder(t *testing.T, s *ts, date string) string {
|
||||
t.Helper()
|
||||
var entries []map[string]any
|
||||
decode(t, s.req(t, http.MethodGet, "/api/schedule?from="+date+"&to="+date, nil), &entries)
|
||||
decode(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/schedule?from="+date+"&to="+date, nil), &entries)
|
||||
if len(entries) == 0 {
|
||||
return ""
|
||||
}
|
||||
@@ -347,10 +415,10 @@ func TestSchedule_ReplaceTakesAnAssignedDate(t *testing.T) {
|
||||
s := newTS(t)
|
||||
addUser(t, s, "alex")
|
||||
|
||||
s.req(t, http.MethodPost, "/api/schedule",
|
||||
s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/schedule",
|
||||
map[string]any{"user_id": 1, "dates": []string{"2026-06-01"}}).Body.Close()
|
||||
|
||||
resp := s.req(t, http.MethodPost, "/api/schedule",
|
||||
resp := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/schedule",
|
||||
map[string]any{"user_id": 2, "dates": []string{"2026-06-01"}, "replace": true})
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("expected replace to succeed, got %d", resp.StatusCode)
|
||||
@@ -364,7 +432,7 @@ func TestSchedule_ReplaceTakesAnAssignedDate(t *testing.T) {
|
||||
// One row, not two: two entries for a date would mean two people believing
|
||||
// they are on call for it.
|
||||
var entries []map[string]any
|
||||
decode(t, s.req(t, http.MethodGet, "/api/schedule?from=2026-06-01&to=2026-06-01", nil), &entries)
|
||||
decode(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/schedule?from=2026-06-01&to=2026-06-01", nil), &entries)
|
||||
if len(entries) != 1 {
|
||||
t.Errorf("expected exactly one entry for the date, got %d", len(entries))
|
||||
}
|
||||
@@ -376,11 +444,11 @@ func TestSchedule_ReplaceMixedWeek(t *testing.T) {
|
||||
s := newTS(t)
|
||||
addUser(t, s, "alex")
|
||||
|
||||
s.req(t, http.MethodPost, "/api/schedule",
|
||||
s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/schedule",
|
||||
map[string]any{"user_id": 1, "dates": []string{"2026-06-02", "2026-06-04"}}).Body.Close()
|
||||
|
||||
week := []string{"2026-06-01", "2026-06-02", "2026-06-03", "2026-06-04", "2026-06-05"}
|
||||
resp := s.req(t, http.MethodPost, "/api/schedule",
|
||||
resp := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/schedule",
|
||||
map[string]any{"user_id": 2, "dates": week, "replace": true})
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("expected the mixed week to succeed, got %d", resp.StatusCode)
|
||||
@@ -399,10 +467,10 @@ func TestSchedule_ReplaceDefaultsOff(t *testing.T) {
|
||||
s := newTS(t)
|
||||
addUser(t, s, "alex")
|
||||
|
||||
s.req(t, http.MethodPost, "/api/schedule",
|
||||
s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/schedule",
|
||||
map[string]any{"user_id": 1, "dates": []string{"2026-06-01"}}).Body.Close()
|
||||
|
||||
resp := s.req(t, http.MethodPost, "/api/schedule",
|
||||
resp := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/schedule",
|
||||
map[string]any{"user_id": 2, "dates": []string{"2026-06-01"}})
|
||||
if resp.StatusCode != http.StatusConflict {
|
||||
t.Fatalf("expected 409 without replace, got %d", resp.StatusCode)
|
||||
@@ -420,7 +488,7 @@ func TestSchedule_ReplaceDefaultsOff(t *testing.T) {
|
||||
func TestSchedule_ReplaceCollapsesRepeatedDates(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
resp := s.req(t, http.MethodPost, "/api/schedule",
|
||||
resp := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/schedule",
|
||||
map[string]any{"user_id": 1, "dates": []string{"2026-06-01", "2026-06-01"}, "replace": true})
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("expected a repeated date to be accepted under replace, got %d", resp.StatusCode)
|
||||
@@ -428,7 +496,7 @@ func TestSchedule_ReplaceCollapsesRepeatedDates(t *testing.T) {
|
||||
resp.Body.Close()
|
||||
|
||||
var entries []map[string]any
|
||||
decode(t, s.req(t, http.MethodGet, "/api/schedule?from=2026-06-01&to=2026-06-01", nil), &entries)
|
||||
decode(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/schedule?from=2026-06-01&to=2026-06-01", nil), &entries)
|
||||
if len(entries) != 1 {
|
||||
t.Errorf("expected one entry for the repeated date, got %d", len(entries))
|
||||
}
|
||||
@@ -498,7 +566,7 @@ func TestArchive_AlertListFilter(t *testing.T) {
|
||||
}
|
||||
|
||||
// 2. Let the sweeper archive it: ends_at is already well past archiveAfter.
|
||||
api.Sweep(context.Background(), s.db, time.Hour, 6*time.Hour, s.deadman, s.notify)
|
||||
api.Sweep(context.Background(), s.db, time.Hour, 6*time.Hour, s.notify)
|
||||
|
||||
// 3. Default list excludes it.
|
||||
decode(t, s.req(t, http.MethodGet, "/api/alerts", nil), &alerts)
|
||||
@@ -539,7 +607,7 @@ func postAlert(t *testing.T, s *ts, fingerprint, status, startsAt, endsAt string
|
||||
|
||||
func sweep(t *testing.T, s *ts, staleAfter time.Duration) {
|
||||
t.Helper()
|
||||
api.Sweep(context.Background(), s.db, noArchive, staleAfter, s.deadman, s.notify)
|
||||
api.Sweep(context.Background(), s.db, noArchive, staleAfter, s.notify)
|
||||
}
|
||||
|
||||
// A firing alert Alertmanager stopped refreshing is resolved via the
|
||||
|
||||
@@ -15,19 +15,39 @@ const (
|
||||
// expiryGrace absorbs clock skew and notification latency before an alert
|
||||
// whose ends_at watermark has passed is treated as stale.
|
||||
expiryGrace = 5 * time.Minute
|
||||
|
||||
// archiverLockKey is the Postgres advisory lock the sweeper takes for the
|
||||
// duration of each pass, so that running more than one replica does not run
|
||||
// the sweep concurrently on all of them. Its value has no meaning beyond
|
||||
// being distinct from notifierLockKey.
|
||||
archiverLockKey int64 = 7265_0001
|
||||
)
|
||||
|
||||
// StartArchiver runs the alert sweeper until ctx is cancelled, starting with an
|
||||
// immediate pass so a restart reconciles state right away.
|
||||
func StartArchiver(ctx context.Context, db *sql.DB, archiveAfter, staleAfter time.Duration, deadman DeadmanConfig, notify NotifyConfig) {
|
||||
// archiveAfter and staleAfter are the values the server started with. They are
|
||||
// the fallback, not the setting: each pass reads the current value from the
|
||||
// settings table, so an administrator's change takes effect on the next tick
|
||||
// instead of at the next restart.
|
||||
//
|
||||
// Each pass runs under archiverLockKey (see withAdvisoryLock), so that on more
|
||||
// than one replica only whichever instance's tick takes the lock first actually
|
||||
// sweeps; the rest skip that tick rather than racing the same pass.
|
||||
func StartArchiver(ctx context.Context, db *sql.DB, archiveAfter, staleAfter time.Duration, notify NotifyConfig) {
|
||||
ticker := time.NewTicker(sweepInterval)
|
||||
defer ticker.Stop()
|
||||
|
||||
Sweep(ctx, db, archiveAfter, staleAfter, deadman, notify)
|
||||
sweep := func() {
|
||||
withAdvisoryLock(ctx, db, archiverLockKey, "sweeper", func() {
|
||||
Sweep(ctx, db, archiveAfter, staleAfter, notify)
|
||||
})
|
||||
}
|
||||
|
||||
sweep()
|
||||
for {
|
||||
select {
|
||||
case <-ticker.C:
|
||||
Sweep(ctx, db, archiveAfter, staleAfter, deadman, notify)
|
||||
sweep()
|
||||
case <-ctx.Done():
|
||||
return
|
||||
}
|
||||
@@ -44,8 +64,12 @@ func StartArchiver(ctx context.Context, db *sql.DB, archiveAfter, staleAfter tim
|
||||
// touch: a heartbeat answers to its own, much tighter, timeout, and the generic
|
||||
// staleness rules would otherwise resolve it as 'expiry' long before that.
|
||||
// Exported so tests can drive a pass without waiting on the ticker.
|
||||
func Sweep(ctx context.Context, db *sql.DB, archiveAfter, staleAfter time.Duration, deadman DeadmanConfig, notify NotifyConfig) {
|
||||
heartbeats := sweepDeadman(ctx, db, deadman, notify)
|
||||
func Sweep(ctx context.Context, db *sql.DB, archiveAfter, staleAfter time.Duration, notify NotifyConfig) {
|
||||
settings := NewSettings(db)
|
||||
staleAfter = settings.Duration(ctx, SettingStaleAfter, staleAfter)
|
||||
archiveAfter = settings.Duration(ctx, SettingArchiveAfter, archiveAfter)
|
||||
|
||||
heartbeats := sweepDeadman(ctx, db, notify)
|
||||
expireStale(ctx, db, staleAfter, heartbeats)
|
||||
resolveSettledIncidents(ctx, db)
|
||||
archiveResolved(ctx, db, archiveAfter)
|
||||
|
||||
+69
-25
@@ -139,6 +139,50 @@ func hashPassword(pw string) (string, error) {
|
||||
return string(h), err
|
||||
}
|
||||
|
||||
// startSession mints a session and sets the cookie. Shared by login and
|
||||
// sign-up: somebody who has just chosen a password is signed in, rather than
|
||||
// being sent to a form to type the same credential again.
|
||||
func startSession(w http.ResponseWriter, r *http.Request, db *sql.DB, userID int64, publicURL string) error {
|
||||
return startSessionCapped(w, r, db, userID, publicURL, 0)
|
||||
}
|
||||
|
||||
// startSessionCapped is startSession with a hard ceiling on the session's life,
|
||||
// which sliding never extends. maxAge zero means no ceiling. A single sign-on
|
||||
// login uses it: the login is the only moment the provider's groups are read, so
|
||||
// a session that could outlive it indefinitely would keep access the provider
|
||||
// has since taken away.
|
||||
func startSessionCapped(w http.ResponseWriter, r *http.Request, db *sql.DB, userID int64, publicURL string, maxAge time.Duration) error {
|
||||
raw, tokenHash, err := randomToken()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
now := time.Now()
|
||||
life := sessionTTL
|
||||
var ceiling *int64
|
||||
if maxAge > 0 {
|
||||
c := now.Add(maxAge).Unix()
|
||||
ceiling = &c
|
||||
life = min(life, maxAge)
|
||||
}
|
||||
if _, err := db.ExecContext(r.Context(), `
|
||||
INSERT INTO sessions (token_hash, user_id, created_at, last_seen_at, expires_at, max_expires_at, user_agent)
|
||||
VALUES ($1, $2, $3, $4, $5, $6, $7)`,
|
||||
tokenHash, userID, now.Unix(), now.Unix(), now.Add(life).Unix(), ceiling, r.UserAgent()); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
http.SetCookie(w, &http.Cookie{
|
||||
Name: sessionCookie,
|
||||
Value: raw,
|
||||
Path: "/",
|
||||
MaxAge: int(life.Seconds()),
|
||||
HttpOnly: true,
|
||||
Secure: cookieSecure(publicURL, r),
|
||||
SameSite: http.SameSiteLaxMode,
|
||||
})
|
||||
return nil
|
||||
}
|
||||
|
||||
// handleLogin exchanges a username and password for a session cookie.
|
||||
func handleLogin(db *sql.DB, limiter *loginLimiter, publicURL string) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
@@ -182,29 +226,10 @@ func handleLogin(db *sql.DB, limiter *loginLimiter, publicURL string) http.Handl
|
||||
}
|
||||
limiter.clear(userKey)
|
||||
|
||||
raw, tokenHash, err := randomToken()
|
||||
if err != nil {
|
||||
if err := startSession(w, r, db, userID, publicURL); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
now := time.Now()
|
||||
if _, err := db.ExecContext(r.Context(), `
|
||||
INSERT INTO sessions (token_hash, user_id, created_at, last_seen_at, expires_at, user_agent)
|
||||
VALUES ($1, $2, $3, $4, $5, $6)`,
|
||||
tokenHash, userID, now.Unix(), now.Unix(), now.Add(sessionTTL).Unix(), r.UserAgent()); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
|
||||
http.SetCookie(w, &http.Cookie{
|
||||
Name: sessionCookie,
|
||||
Value: raw,
|
||||
Path: "/",
|
||||
MaxAge: int(sessionTTL.Seconds()),
|
||||
HttpOnly: true,
|
||||
Secure: cookieSecure(publicURL, r),
|
||||
SameSite: http.SameSiteLaxMode,
|
||||
})
|
||||
|
||||
user, err := fetchUser(r.Context(), db, userID)
|
||||
if err != nil {
|
||||
@@ -243,22 +268,37 @@ func handleLogout(db *sql.DB, publicURL string) http.HandlerFunc {
|
||||
type meResponse struct {
|
||||
User any `json:"user"`
|
||||
HasPassword bool `json:"has_password"`
|
||||
|
||||
// OnboardingDismissed is whether this person has put the first-run
|
||||
// checklist away. Per user rather than per browser: somebody who finishes
|
||||
// setting up on a laptop should not be nagged again on their phone.
|
||||
OnboardingDismissed bool `json:"onboarding_dismissed"`
|
||||
}
|
||||
|
||||
// handleMe says who the caller is. The web UI calls it on load to decide
|
||||
// between the login form and the app, since it cannot read its own cookie.
|
||||
func handleMe(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
caller, _ := userFromContext(r.Context())
|
||||
caller, ok := userFromContext(r.Context())
|
||||
if !ok {
|
||||
respond(w, http.StatusForbidden, errResp("this endpoint is for human accounts only"))
|
||||
return
|
||||
}
|
||||
user, err := fetchUser(r.Context(), db, caller.ID)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
var hash sql.NullString
|
||||
var dismissed *int64
|
||||
db.QueryRowContext(r.Context(),
|
||||
"SELECT password_hash FROM users WHERE id = $1", caller.ID).Scan(&hash)
|
||||
respond(w, http.StatusOK, meResponse{User: user, HasPassword: hash.Valid})
|
||||
"SELECT password_hash, onboarding_dismissed_at FROM users WHERE id = $1",
|
||||
caller.ID).Scan(&hash, &dismissed)
|
||||
respond(w, http.StatusOK, meResponse{
|
||||
User: user,
|
||||
HasPassword: hash.Valid,
|
||||
OnboardingDismissed: dismissed != nil,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -266,8 +306,9 @@ func handleMe(db *sql.DB) http.HandlerFunc {
|
||||
//
|
||||
// Changing your own password takes the current one, when there is one, so an
|
||||
// unattended signed-in browser cannot be used to take the account over. Setting
|
||||
// somebody else's is how an admin gives a user their first password, and like
|
||||
// the other user endpoints it is open to any authenticated caller.
|
||||
// somebody else's is how an admin gives a user their first password, and is
|
||||
// restricted to administrators: it hands over an account outright, without
|
||||
// knowing the password it replaces.
|
||||
//
|
||||
// Every other session of the target is ended: a password change is what you
|
||||
// do when you think someone else is signed in.
|
||||
@@ -278,6 +319,9 @@ func handleSetPassword(db *sql.DB) http.HandlerFunc {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid user id"))
|
||||
return
|
||||
}
|
||||
if !requireSelfOrAdmin(w, r, id) {
|
||||
return
|
||||
}
|
||||
var req struct {
|
||||
Password string `json:"password"`
|
||||
CurrentPassword string `json:"current_password"`
|
||||
|
||||
@@ -248,7 +248,7 @@ func TestSession_ExpiredIsRejected(t *testing.T) {
|
||||
if code := status(t, b.do(t, http.MethodGet, "/api/me", nil)); code != http.StatusUnauthorized {
|
||||
t.Errorf("expired session: %d", code)
|
||||
}
|
||||
api.Sweep(t.Context(), s.db, 0, 0, api.DeadmanConfig{}, api.NotifyConfig{})
|
||||
api.Sweep(t.Context(), s.db, 0, 0, api.NotifyConfig{})
|
||||
var n int
|
||||
s.db.QueryRow("SELECT COUNT(*) FROM sessions").Scan(&n)
|
||||
if n != 0 {
|
||||
@@ -301,7 +301,7 @@ func TestSetPassword_EndsOtherSessionsButNotThisOne(t *testing.T) {
|
||||
|
||||
func TestBootstrap_WithPassword(t *testing.T) {
|
||||
database := newTestDB(t)
|
||||
srv := httptest.NewServer(api.NewRouter(database, api.NotifyConfig{}, api.DeadmanConfig{}))
|
||||
srv := httptest.NewServer(api.NewRouter(database, api.NotifyConfig{}, testConfig(), "test"))
|
||||
t.Cleanup(srv.Close)
|
||||
|
||||
body := `{"username":"admin","email":"a@test.com","password":"` + adminPassword + `"}`
|
||||
|
||||
@@ -0,0 +1,123 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||
)
|
||||
|
||||
// Caller is the one principal type every authorization predicate in this
|
||||
// package reads from. Before this, a human (ctxUser + ctxTeams) and a
|
||||
// service account (ctxServiceAccount + a synthetic ctxTeams entry) were two
|
||||
// parallel, un-unified context representations — every predicate had to
|
||||
// remember which one(s) it needed to check, and the ones that forgot either
|
||||
// 403'd a service account that should have been let through (terdut-server#23,
|
||||
// terdut-operator#3), crashed on an unchecked zero-value user ID (handleMe,
|
||||
// handleTestNotification), or silently no-op'd (handleDismissOnboarding).
|
||||
// serveAs and serveAsServiceAccount now both build exactly one Caller and
|
||||
// store it under one context key; everything else in this file is a read
|
||||
// of one of its methods.
|
||||
type Caller struct {
|
||||
// user is set for a human caller (session cookie or a user's own API
|
||||
// key), nil for a service account of either scope.
|
||||
user *models.User
|
||||
|
||||
// sa is set for a service-account caller, nil for a human.
|
||||
sa *serviceAccountPrincipal
|
||||
|
||||
// memberships is the caller's real team_members rows for a human, or —
|
||||
// for a team-scoped service account — the single synthetic owner
|
||||
// membership serveAsServiceAccount injects (see its own comment for
|
||||
// why). Always nil for an instance-scoped service account: it acts on
|
||||
// teams by id, not by belonging to one.
|
||||
memberships []membership
|
||||
}
|
||||
|
||||
// AsHuman returns the real user behind this caller, or false for a service
|
||||
// account of either scope. Every handler that needs a real user_id to act
|
||||
// on behalf of — not just "is this caller sufficiently privileged" — calls
|
||||
// this and handles the false case explicitly, replacing the unchecked
|
||||
// userFromContext(ctx) zero-value reads that used to silently misbehave for
|
||||
// a service-account caller.
|
||||
func (c Caller) AsHuman() (models.User, bool) {
|
||||
if c.user == nil {
|
||||
return models.User{}, false
|
||||
}
|
||||
return *c.user, true
|
||||
}
|
||||
|
||||
// IsAdmin is true only for a human system administrator — never for a
|
||||
// service account, of either scope, under any circumstance. AdminOnly and
|
||||
// requireSelfOrAdmin key on this and nothing else: user management and
|
||||
// /api/admin/settings stay human-only forever, by design (SERVICE-ACCOUNTS.md).
|
||||
func (c Caller) IsAdmin() bool {
|
||||
return c.user != nil && c.user.IsAdmin
|
||||
}
|
||||
|
||||
// IsInstanceServiceAccount reports whether this caller is specifically an
|
||||
// instance-scoped service account — never true for a human, including a
|
||||
// human admin. handleCreateTeam needs exactly this: a human creates a team
|
||||
// by being a human (and becomes its owner as a side effect), an
|
||||
// instance-scoped service account creates one with no human owner at all;
|
||||
// the two paths are not interchangeable, so this predicate must not also
|
||||
// admit a human admin the way MayActAsInstanceAdmin deliberately does.
|
||||
func (c Caller) IsInstanceServiceAccount() bool {
|
||||
return c.sa != nil && c.sa.scope == models.ServiceAccountScopeInstance
|
||||
}
|
||||
|
||||
// Role reports the caller's role in teamID, and whether they belong to it
|
||||
// at all.
|
||||
func (c Caller) Role(teamID int64) (string, bool) {
|
||||
for _, m := range c.memberships {
|
||||
if m.teamID == teamID {
|
||||
return m.role, true
|
||||
}
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// TeamIDs lists every team this caller belongs to: a human's real
|
||||
// memberships, or a team-scoped service account's own single team. Always
|
||||
// empty for an instance-scoped service account.
|
||||
func (c Caller) TeamIDs() []int64 {
|
||||
ids := make([]int64, 0, len(c.memberships))
|
||||
for _, m := range c.memberships {
|
||||
ids = append(ids, m.teamID)
|
||||
}
|
||||
return ids
|
||||
}
|
||||
|
||||
// ServiceAccountID reports this caller's own service-account id, for the
|
||||
// "may manage/rotate its own credential" self-check in
|
||||
// callerMayManageServiceAccount, and for OperatorModeBlock's "any service
|
||||
// account passes" rule.
|
||||
func (c Caller) ServiceAccountID() (int64, bool) {
|
||||
if c.sa == nil {
|
||||
return 0, false
|
||||
}
|
||||
return c.sa.id, true
|
||||
}
|
||||
|
||||
// Identity is a stable, log/audit-facing string distinguishing a human
|
||||
// caller from a service account — "user:42" or "service-account:7". Not
|
||||
// wired into any database column today (incidents.go's acknowledged_by/
|
||||
// assigned_to/user_id are explicitly out of scope for this change — that
|
||||
// needs its own schema migration, tracked separately), but this is the one
|
||||
// place in the request path that already knows which kind of caller this
|
||||
// is, and that follow-up will want exactly this accessor.
|
||||
func (c Caller) Identity() string {
|
||||
switch {
|
||||
case c.user != nil:
|
||||
return fmt.Sprintf("user:%d", c.user.ID)
|
||||
case c.sa != nil:
|
||||
return fmt.Sprintf("service-account:%d", c.sa.id)
|
||||
default:
|
||||
return "unknown"
|
||||
}
|
||||
}
|
||||
|
||||
func callerFromContext(ctx context.Context) (Caller, bool) {
|
||||
c, ok := ctx.Value(ctxCaller).(Caller)
|
||||
return c, ok
|
||||
}
|
||||
+397
-95
@@ -4,6 +4,8 @@ import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log"
|
||||
"sort"
|
||||
"strings"
|
||||
@@ -39,6 +41,17 @@ func (m DeadmanMatcher) String() string {
|
||||
return m.Name + " (" + strings.Join(parts, ", ") + ")"
|
||||
}
|
||||
|
||||
// config renders the matcher in the form parseDeadmanMatcher reads, which is
|
||||
// what a switch row stores: `alertname=Watchdog,cluster=prod`.
|
||||
func (m DeadmanMatcher) config() string {
|
||||
parts := make([]string, 0, len(m.Labels))
|
||||
for k, v := range m.Labels {
|
||||
parts = append(parts, k+"="+v)
|
||||
}
|
||||
sort.Strings(parts)
|
||||
return strings.Join(append([]string{"alertname=" + m.Name}, parts...), ",")
|
||||
}
|
||||
|
||||
// matches reports whether an alert's labels satisfy every condition.
|
||||
func (m DeadmanMatcher) matches(labels map[string]string) bool {
|
||||
if labels["alertname"] != m.Name {
|
||||
@@ -52,12 +65,10 @@ func (m DeadmanMatcher) matches(labels map[string]string) bool {
|
||||
return true
|
||||
}
|
||||
|
||||
// DeadmanConfig inverts the handling of the alerts it matches: receiving one
|
||||
// opens nothing, and the absence of one opens an incident.
|
||||
//
|
||||
// The unit of monitoring is the fingerprint, not the matcher — two clusters
|
||||
// sending the same heartbeat alertname are two independent switches, so one
|
||||
// healthy cluster cannot mask a dead one.
|
||||
// DeadmanConfig is the server-wide default a team's switches are seeded from:
|
||||
// the environment's matchers, timeout and severity. Switches themselves are rows
|
||||
// of a team's own — see DeadmanSwitch — and this is only how a fresh install
|
||||
// starts out.
|
||||
type DeadmanConfig struct {
|
||||
Matchers []DeadmanMatcher
|
||||
|
||||
@@ -76,41 +87,81 @@ type DeadmanConfig struct {
|
||||
// enabled reports whether there is anything to watch.
|
||||
func (c DeadmanConfig) enabled() bool { return c.Timeout > 0 && len(c.Matchers) > 0 }
|
||||
|
||||
// match returns the first matcher an alert satisfies.
|
||||
func (c DeadmanConfig) match(labels map[string]string) (DeadmanMatcher, bool) {
|
||||
if !c.enabled() {
|
||||
return DeadmanMatcher{}, false
|
||||
}
|
||||
for _, m := range c.Matchers {
|
||||
if m.matches(labels) {
|
||||
return m, true
|
||||
}
|
||||
}
|
||||
return DeadmanMatcher{}, false
|
||||
// DeadmanSwitch inverts the handling of the alerts it matches: receiving one
|
||||
// opens nothing, and the absence of one opens an incident.
|
||||
//
|
||||
// The unit of monitoring is the fingerprint, not the switch — two clusters
|
||||
// sending the same heartbeat alertname are two independent heartbeats under one
|
||||
// switch, so one healthy cluster cannot mask a dead one.
|
||||
type DeadmanSwitch struct {
|
||||
ID int64
|
||||
Name string
|
||||
Matcher DeadmanMatcher
|
||||
|
||||
// Timeout is how long a heartbeat may go unheard before it is declared dead.
|
||||
Timeout time.Duration
|
||||
|
||||
// Severity is what the incident opens at.
|
||||
Severity string
|
||||
}
|
||||
|
||||
// isDeadman is match without the matcher, for the ingest path.
|
||||
func (c DeadmanConfig) isDeadman(labels map[string]string) bool {
|
||||
_, ok := c.match(labels)
|
||||
// deadmanSet is one team's switches.
|
||||
type deadmanSet []DeadmanSwitch
|
||||
|
||||
// match returns the first switch an alert satisfies.
|
||||
func (d deadmanSet) match(labels map[string]string) (DeadmanSwitch, bool) {
|
||||
for _, sw := range d {
|
||||
if sw.Matcher.matches(labels) {
|
||||
return sw, true
|
||||
}
|
||||
}
|
||||
return DeadmanSwitch{}, false
|
||||
}
|
||||
|
||||
// isDeadman is match without the switch, for the ingest path.
|
||||
func (d deadmanSet) isDeadman(labels map[string]string) bool {
|
||||
_, ok := d.match(labels)
|
||||
return ok
|
||||
}
|
||||
|
||||
// names lists the distinct alertnames worth loading from the database.
|
||||
func (c DeadmanConfig) names() []string {
|
||||
func (d deadmanSet) names() []string {
|
||||
seen := map[string]bool{}
|
||||
out := make([]string, 0, len(c.Matchers))
|
||||
for _, m := range c.Matchers {
|
||||
if !seen[m.Name] {
|
||||
seen[m.Name] = true
|
||||
out = append(out, m.Name)
|
||||
out := make([]string, 0, len(d))
|
||||
for _, sw := range d {
|
||||
if !seen[sw.Matcher.Name] {
|
||||
seen[sw.Matcher.Name] = true
|
||||
out = append(out, sw.Matcher.Name)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// parseDeadmanMatcher reads one matcher from its configured form: "," separates
|
||||
// the conditions and "=" is exact label equality — `alertname=Watchdog,cluster=prod`.
|
||||
// The error says what is wrong with it, in words a form can show.
|
||||
func parseDeadmanMatcher(entry string) (DeadmanMatcher, error) {
|
||||
m := DeadmanMatcher{Labels: map[string]string{}}
|
||||
for _, cond := range strings.Split(strings.TrimSpace(entry), ",") {
|
||||
k, v, ok := strings.Cut(cond, "=")
|
||||
k, v = strings.TrimSpace(k), strings.TrimSpace(v)
|
||||
if !ok || k == "" || v == "" {
|
||||
return DeadmanMatcher{}, fmt.Errorf("%q is not label=value", strings.TrimSpace(cond))
|
||||
}
|
||||
if k == "alertname" {
|
||||
m.Name = v
|
||||
continue
|
||||
}
|
||||
m.Labels[k] = v
|
||||
}
|
||||
if m.Name == "" {
|
||||
return DeadmanMatcher{}, errors.New("no alertname condition")
|
||||
}
|
||||
return m, nil
|
||||
}
|
||||
|
||||
// ParseDeadmanConfig reads the matcher list from its configured form:
|
||||
// ";" separates matchers, "," separates the conditions within one, and "=" is
|
||||
// exact label equality — `alertname=Watchdog,cluster=prod; alertname=Heartbeat`.
|
||||
// ";" separates matchers, and each is parsed as parseDeadmanMatcher does.
|
||||
//
|
||||
// A malformed or alertname-less entry is dropped rather than fatal, following
|
||||
// config.duration's rule that one bad tuning knob should not take the server
|
||||
@@ -125,28 +176,9 @@ func ParseDeadmanConfig(matchers string, timeout time.Duration, severity string)
|
||||
if entry == "" {
|
||||
continue
|
||||
}
|
||||
|
||||
m := DeadmanMatcher{Labels: map[string]string{}}
|
||||
malformed := false
|
||||
for _, cond := range strings.Split(entry, ",") {
|
||||
k, v, ok := strings.Cut(cond, "=")
|
||||
k, v = strings.TrimSpace(k), strings.TrimSpace(v)
|
||||
if !ok || k == "" || v == "" {
|
||||
log.Printf("deadman: ignoring matcher %q: %q is not label=value", entry, strings.TrimSpace(cond))
|
||||
malformed = true
|
||||
break
|
||||
}
|
||||
if k == "alertname" {
|
||||
m.Name = v
|
||||
continue
|
||||
}
|
||||
m.Labels[k] = v
|
||||
}
|
||||
if malformed {
|
||||
continue
|
||||
}
|
||||
if m.Name == "" {
|
||||
log.Printf("deadman: ignoring matcher %q: no alertname condition", entry)
|
||||
m, err := parseDeadmanMatcher(entry)
|
||||
if err != nil {
|
||||
log.Printf("deadman: ignoring matcher %q: %v", entry, err)
|
||||
continue
|
||||
}
|
||||
cfg.Matchers = append(cfg.Matchers, m)
|
||||
@@ -162,22 +194,35 @@ func ParseDeadmanConfig(matchers string, timeout time.Duration, severity string)
|
||||
for _, m := range cfg.Matchers {
|
||||
rendered = append(rendered, m.String())
|
||||
}
|
||||
log.Printf("deadman: watching %s, timeout %s, severity %s",
|
||||
log.Printf("deadman: default for new teams: %s, timeout %s, severity %s",
|
||||
strings.Join(rendered, "; "), timeout, severity)
|
||||
}
|
||||
return cfg
|
||||
}
|
||||
|
||||
// deadmanAlert is one switch: the alert row carrying its last heartbeat.
|
||||
// deadmanAlert is one heartbeat: the alert row carrying its last sighting, and
|
||||
// the switch that claimed it.
|
||||
type deadmanAlert struct {
|
||||
id int64
|
||||
teamID int64
|
||||
fingerprint string
|
||||
labels map[string]string
|
||||
matcher DeadmanMatcher
|
||||
sw DeadmanSwitch
|
||||
resolved bool
|
||||
receivedAt int64
|
||||
}
|
||||
|
||||
// dead is the one rule for a silent heartbeat, shared by the sweeper that pages
|
||||
// on it and the status the Switches page shows, so the page cannot disagree
|
||||
// with the pager.
|
||||
//
|
||||
// An explicit resolved from Alertmanager is a stronger death signal than mere
|
||||
// absence: the sender is telling us the heartbeat stopped, so there is nothing
|
||||
// left to wait out.
|
||||
func (a deadmanAlert) dead(now time.Time) bool {
|
||||
return a.resolved || a.receivedAt < now.Add(-a.sw.Timeout).Unix()
|
||||
}
|
||||
|
||||
// groupKey is the switch's identity as an incident. Per fingerprint, so each
|
||||
// source is tracked on its own.
|
||||
func (a deadmanAlert) groupKey() string { return deadmanGroupPrefix + a.fingerprint }
|
||||
@@ -188,46 +233,49 @@ func (a deadmanAlert) groupKey() string { return deadmanGroupPrefix + a.fingerpr
|
||||
// It returns the ids of the alerts it owns, because the generic staleness
|
||||
// expiry must leave them alone — staleAfter and ends_at would otherwise resolve
|
||||
// a heartbeat long before its own, much tighter, timeout ever fired.
|
||||
func sweepDeadman(ctx context.Context, db *sql.DB, cfg DeadmanConfig, notify NotifyConfig) map[int64]bool {
|
||||
// Each team is swept against its own switches, each with its own matcher,
|
||||
// timeout and severity. A team watching nothing is skipped entirely, which is
|
||||
// most of them.
|
||||
func sweepDeadman(ctx context.Context, db *sql.DB, notify NotifyConfig) map[int64]bool {
|
||||
owned := map[int64]bool{}
|
||||
if !cfg.enabled() {
|
||||
return owned
|
||||
}
|
||||
|
||||
switches, err := deadmanAlerts(ctx, db, cfg)
|
||||
configs, err := deadmanSets(ctx, db)
|
||||
if err != nil {
|
||||
log.Printf("deadman: load switches: %v", err)
|
||||
log.Printf("deadman: load configs: %v", err)
|
||||
return owned
|
||||
}
|
||||
|
||||
now := time.Now()
|
||||
cutoff := now.Add(-cfg.Timeout).Unix()
|
||||
|
||||
for _, sw := range switches {
|
||||
owned[sw.id] = true
|
||||
|
||||
// An explicit resolved from Alertmanager is a stronger death signal than
|
||||
// mere absence: the sender is telling us the heartbeat stopped, so there
|
||||
// is nothing left to wait out.
|
||||
if sw.resolved || sw.receivedAt < cutoff {
|
||||
if err := deadmanDied(ctx, db, cfg, notify, sw, now); err != nil {
|
||||
log.Printf("deadman: open incident for %s: %v", sw.matcher.Name, err)
|
||||
}
|
||||
for teamID, cfg := range configs {
|
||||
heartbeats, err := deadmanAlerts(ctx, db, teamID, cfg)
|
||||
if err != nil {
|
||||
log.Printf("deadman: load heartbeats for team %d: %v", teamID, err)
|
||||
continue
|
||||
}
|
||||
if err := deadmanRecovered(ctx, db, sw); err != nil {
|
||||
log.Printf("deadman: resolve incident for %s: %v", sw.matcher.Name, err)
|
||||
|
||||
for _, hb := range heartbeats {
|
||||
owned[hb.id] = true
|
||||
|
||||
if hb.dead(now) {
|
||||
if err := deadmanDied(ctx, db, notify, hb, now); err != nil {
|
||||
log.Printf("deadman: open incident for %s: %v", hb.sw.Matcher.Name, err)
|
||||
}
|
||||
continue
|
||||
}
|
||||
if err := deadmanRecovered(ctx, db, hb); err != nil {
|
||||
log.Printf("deadman: resolve incident for %s: %v", hb.sw.Matcher.Name, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
return owned
|
||||
}
|
||||
|
||||
// deadmanAlerts loads every alert row that a matcher claims. The candidate query
|
||||
// deadmanAlerts loads every alert row that one of a team's switches claims. The candidate query
|
||||
// is narrowed by alertname so it rides alerts_name_idx; the rest of the matching
|
||||
// happens in Go, which keeps one implementation of the rules. The rows are read
|
||||
// in full before the caller writes, so the writes do not run against an open
|
||||
// cursor over the same table.
|
||||
func deadmanAlerts(ctx context.Context, db *sql.DB, cfg DeadmanConfig) ([]deadmanAlert, error) {
|
||||
func deadmanAlerts(ctx context.Context, db *sql.DB, teamID int64, cfg deadmanSet) ([]deadmanAlert, error) {
|
||||
names := cfg.names()
|
||||
args := &sqlArgs{}
|
||||
nameList := make([]any, len(names))
|
||||
@@ -236,9 +284,10 @@ func deadmanAlerts(ctx context.Context, db *sql.DB, cfg DeadmanConfig) ([]deadma
|
||||
}
|
||||
|
||||
rows, err := db.QueryContext(ctx, `
|
||||
SELECT id, fingerprint, labels, status, received_at
|
||||
SELECT id, team_id, fingerprint, labels, status, received_at
|
||||
FROM alerts
|
||||
WHERE name IN (`+args.addList(nameList)+`)
|
||||
WHERE team_id = `+args.add(teamID)+`
|
||||
AND name IN (`+args.addList(nameList)+`)
|
||||
AND archived_at IS NULL`, args.all()...)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
@@ -249,16 +298,16 @@ func deadmanAlerts(ctx context.Context, db *sql.DB, cfg DeadmanConfig) ([]deadma
|
||||
for rows.Next() {
|
||||
var a deadmanAlert
|
||||
var labelsJSON, status string
|
||||
if err := rows.Scan(&a.id, &a.fingerprint, &labelsJSON, &status, &a.receivedAt); err != nil {
|
||||
if err := rows.Scan(&a.id, &a.teamID, &a.fingerprint, &labelsJSON, &status, &a.receivedAt); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
json.Unmarshal([]byte(labelsJSON), &a.labels) //nolint:errcheck
|
||||
|
||||
m, ok := cfg.match(a.labels)
|
||||
sw, ok := cfg.match(a.labels)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
a.matcher = m
|
||||
a.sw = sw
|
||||
a.resolved = status == "resolved"
|
||||
out = append(out, a)
|
||||
}
|
||||
@@ -275,16 +324,16 @@ func deadmanAlerts(ctx context.Context, db *sql.DB, cfg DeadmanConfig) ([]deadma
|
||||
// incidentForGroup), and a source that is gone for good is a one-time page
|
||||
// rather than a nag. Only a heartbeat that comes back and dies again earns a new
|
||||
// incident.
|
||||
func deadmanDied(ctx context.Context, db *sql.DB, cfg DeadmanConfig, notify NotifyConfig, sw deadmanAlert, now time.Time) error {
|
||||
func deadmanDied(ctx context.Context, db *sql.DB, notify NotifyConfig, hb deadmanAlert, now time.Time) error {
|
||||
var lastTriggered, open int64
|
||||
if err := db.QueryRowContext(ctx, `
|
||||
SELECT COALESCE(MAX(triggered_at), 0),
|
||||
COUNT(*) FILTER (WHERE resolved_at IS NULL)
|
||||
FROM incidents WHERE group_key = $1`,
|
||||
sw.groupKey()).Scan(&lastTriggered, &open); err != nil {
|
||||
FROM incidents WHERE team_id = $1 AND group_key = $2`,
|
||||
hb.teamID, hb.groupKey()).Scan(&lastTriggered, &open); err != nil {
|
||||
return err
|
||||
}
|
||||
if open > 0 || sw.receivedAt <= lastTriggered {
|
||||
if open > 0 || hb.receivedAt <= lastTriggered {
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -297,31 +346,33 @@ func deadmanDied(ctx context.Context, db *sql.DB, cfg DeadmanConfig, notify Noti
|
||||
// A heartbeat nobody has heard from is not firing, and saying otherwise in
|
||||
// the alert list would be a lie. An Alertmanager-sourced resolution keeps its
|
||||
// own source: it told us the truth first.
|
||||
if !sw.resolved {
|
||||
if !hb.resolved {
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE alerts
|
||||
SET status = 'resolved',
|
||||
resolution_source = $1,
|
||||
ends_at = COALESCE(ends_at, `+nowEpoch+`)
|
||||
WHERE id = $2 AND status = 'firing'`, resolutionDeadman, sw.id); err != nil {
|
||||
WHERE id = $2 AND status = 'firing'`, resolutionDeadman, hb.id); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
severity := cfg.Severity
|
||||
severity := hb.sw.Severity
|
||||
var sev *string
|
||||
if severity != "" {
|
||||
sev = &severity
|
||||
}
|
||||
|
||||
incidentID, err := openIncident(ctx, tx, notify, sw.groupKey(),
|
||||
"No heartbeat from "+sw.matcher.String(), sw.labels, sev)
|
||||
// The incident opens in the team whose integration received the heartbeat:
|
||||
// the switch belongs to whoever is watching that source, not to the install.
|
||||
incidentID, err := openIncident(ctx, tx, notify, hb.teamID, hb.groupKey(),
|
||||
"No heartbeat from "+hb.sw.Matcher.String(), hb.labels, sev)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
alertID := sw.id
|
||||
detail := "last heartbeat " + humanDuration(now.Sub(time.Unix(sw.receivedAt, 0))) + " ago"
|
||||
alertID := hb.id
|
||||
detail := "last heartbeat " + humanDuration(now.Sub(time.Unix(hb.receivedAt, 0))) + " ago"
|
||||
if err := logEvent(ctx, tx, incidentID, evDeadmanSilent, nil, &alertID, &detail); err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -329,7 +380,7 @@ func deadmanDied(ctx context.Context, db *sql.DB, cfg DeadmanConfig, notify Noti
|
||||
if err := tx.Commit(); err != nil {
|
||||
return err
|
||||
}
|
||||
log.Printf("deadman: %s went silent, opened incident %d", sw.matcher.String(), incidentID)
|
||||
log.Printf("deadman: %s went silent, opened incident %d", hb.sw.Matcher.String(), incidentID)
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -339,11 +390,12 @@ func deadmanDied(ctx context.Context, db *sql.DB, cfg DeadmanConfig, notify Noti
|
||||
// member alerts (linking the heartbeat would have the settled-incident cascade
|
||||
// close it on the very same sweep that opened it), so the alert-driven cascade
|
||||
// ignores it entirely and recovery is the only automatic way out.
|
||||
func deadmanRecovered(ctx context.Context, db *sql.DB, sw deadmanAlert) error {
|
||||
func deadmanRecovered(ctx context.Context, db *sql.DB, hb deadmanAlert) error {
|
||||
var incidentID int64
|
||||
switch err := db.QueryRowContext(ctx, `
|
||||
SELECT id FROM incidents
|
||||
WHERE group_key = $1 AND resolved_at IS NULL`, sw.groupKey()).Scan(&incidentID); {
|
||||
WHERE team_id = $1 AND group_key = $2 AND resolved_at IS NULL`,
|
||||
hb.teamID, hb.groupKey()).Scan(&incidentID); {
|
||||
case err == sql.ErrNoRows:
|
||||
return nil
|
||||
case err != nil:
|
||||
@@ -375,6 +427,256 @@ func deadmanRecovered(ctx context.Context, db *sql.DB, sw deadmanAlert) error {
|
||||
if err := tx.Commit(); err != nil {
|
||||
return err
|
||||
}
|
||||
log.Printf("deadman: %s is back, resolved incident %d", sw.matcher.String(), incidentID)
|
||||
log.Printf("deadman: %s is back, resolved incident %d", hb.sw.Matcher.String(), incidentID)
|
||||
return nil
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// A team's switches
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const deadmanSwitchColumns = "id, team_id, name, matcher, timeout_seconds, severity"
|
||||
|
||||
// scanDeadmanSwitches reads switch rows into per-team sets. A row whose matcher
|
||||
// no longer parses is skipped rather than fatal: the API refuses to store one,
|
||||
// so it can only mean a hand edit, and one bad row must not stop the others
|
||||
// from being watched.
|
||||
func scanDeadmanSwitches(rows *sql.Rows) (map[int64]deadmanSet, error) {
|
||||
defer rows.Close()
|
||||
out := map[int64]deadmanSet{}
|
||||
for rows.Next() {
|
||||
var sw DeadmanSwitch
|
||||
var teamID, timeout int64
|
||||
var matcher string
|
||||
if err := rows.Scan(&sw.ID, &teamID, &sw.Name, &matcher, &timeout, &sw.Severity); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
m, err := parseDeadmanMatcher(matcher)
|
||||
if err != nil {
|
||||
log.Printf("deadman: switch %d has an unusable matcher %q: %v", sw.ID, matcher, err)
|
||||
continue
|
||||
}
|
||||
sw.Matcher = m
|
||||
sw.Timeout = time.Duration(timeout) * time.Second
|
||||
out[teamID] = append(out[teamID], sw)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
// deadmanSetForTeam reads one team's switches. A team with none gets an empty
|
||||
// set — which is the right answer rather than an error: most teams watch no
|
||||
// heartbeat at all.
|
||||
func deadmanSetForTeam(ctx context.Context, q querier, teamID int64) (deadmanSet, error) {
|
||||
rows, err := q.QueryContext(ctx,
|
||||
"SELECT "+deadmanSwitchColumns+" FROM deadman_switches WHERE team_id = $1 ORDER BY id", teamID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
sets, err := scanDeadmanSwitches(rows)
|
||||
return sets[teamID], err
|
||||
}
|
||||
|
||||
// deadmanSets reads every team's switches in one query, for the sweeper.
|
||||
func deadmanSets(ctx context.Context, db *sql.DB) (map[int64]deadmanSet, error) {
|
||||
rows, err := db.QueryContext(ctx,
|
||||
"SELECT "+deadmanSwitchColumns+" FROM deadman_switches ORDER BY id")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return scanDeadmanSwitches(rows)
|
||||
}
|
||||
|
||||
// deadmanSeededKey is the settings row that records the environment defaults
|
||||
// were handed out. Without it, a team that deleted its last switch would get
|
||||
// the default back on the next restart.
|
||||
const deadmanSeededKey = "deadman_seeded"
|
||||
|
||||
// SeedDeadmanConfigs gives every team the server's environment defaults as
|
||||
// switches, exactly once per install, so a fresh install watches Watchdog
|
||||
// without anybody setting it up.
|
||||
//
|
||||
// Once seeded it never runs again: a team's switches are its own, and a redeploy
|
||||
// must not quietly put the environment's value back over an owner's edit or
|
||||
// deletion. Installs that upgraded from per-team configuration were already
|
||||
// seeded, which migration 009 records.
|
||||
//
|
||||
// A team created after that gets none and watches nothing until its owner says
|
||||
// otherwise. That is deliberate: inheriting an install-wide heartbeat would page
|
||||
// a new team about a source it has never heard of, and a switch nobody chose is
|
||||
// the kind that gets muted rather than fixed.
|
||||
func SeedDeadmanConfigs(ctx context.Context, db *sql.DB, cfg DeadmanConfig) error {
|
||||
if !cfg.enabled() {
|
||||
return nil
|
||||
}
|
||||
|
||||
tx, err := db.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer tx.Rollback() //nolint:errcheck
|
||||
|
||||
res, err := tx.ExecContext(ctx,
|
||||
"INSERT INTO settings (key, value) VALUES ($1, '1') ON CONFLICT (key) DO NOTHING",
|
||||
deadmanSeededKey)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if n, _ := res.RowsAffected(); n == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
for _, m := range cfg.Matchers {
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity)
|
||||
SELECT id, $1, $1, $2, $3 FROM teams`,
|
||||
m.config(), int64(cfg.Timeout.Seconds()), cfg.Severity); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Status
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const (
|
||||
switchHealthy = "healthy"
|
||||
switchDead = "dead"
|
||||
switchDormant = "dormant"
|
||||
)
|
||||
|
||||
// deadmanSource is one heartbeat under a switch: a fingerprint that matched.
|
||||
type deadmanSource struct {
|
||||
Fingerprint string `json:"fingerprint"`
|
||||
Labels map[string]string `json:"labels"`
|
||||
Status string `json:"status"`
|
||||
LastHeartbeatAt time.Time `json:"last_heartbeat_at"`
|
||||
LastTriggeredAt *time.Time `json:"last_triggered_at"`
|
||||
IncidentID *int64 `json:"incident_id"`
|
||||
}
|
||||
|
||||
// deadmanSwitchStatus is a switch as the Switches page shows it.
|
||||
type deadmanSwitchStatus struct {
|
||||
ID int64 `json:"id"`
|
||||
Name string `json:"name"`
|
||||
Matcher string `json:"matcher"`
|
||||
TimeoutSeconds int64 `json:"timeout_seconds"`
|
||||
Severity string `json:"severity"`
|
||||
|
||||
// Status is dead when any source is, dormant when none has ever been heard
|
||||
// from, healthy otherwise — a live cluster must not hide a dead one.
|
||||
Status string `json:"status"`
|
||||
LastHeartbeatAt *time.Time `json:"last_heartbeat_at"`
|
||||
LastTriggeredAt *time.Time `json:"last_triggered_at"`
|
||||
OpenIncidentID *int64 `json:"open_incident_id"`
|
||||
Sources []deadmanSource `json:"sources"`
|
||||
}
|
||||
|
||||
// deadmanStatuses reports every switch of a team with what its heartbeats are
|
||||
// doing. The liveness verdict is deadmanAlert.dead, the sweeper's own.
|
||||
func deadmanStatuses(ctx context.Context, db *sql.DB, teamID int64, set deadmanSet, now time.Time) ([]deadmanSwitchStatus, error) {
|
||||
out := make([]deadmanSwitchStatus, 0, len(set))
|
||||
if len(set) == 0 {
|
||||
return out, nil
|
||||
}
|
||||
|
||||
heartbeats, err := deadmanAlerts(ctx, db, teamID, set)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
// One query for every switch's incident history, keyed the way the sweeper
|
||||
// keys it.
|
||||
type history struct {
|
||||
triggeredAt int64
|
||||
openID int64
|
||||
}
|
||||
incidents := map[string]history{}
|
||||
rows, err := db.QueryContext(ctx, `
|
||||
SELECT group_key, MAX(triggered_at), COALESCE(MAX(id) FILTER (WHERE resolved_at IS NULL), 0)
|
||||
FROM incidents
|
||||
WHERE team_id = $1 AND group_key LIKE $2
|
||||
GROUP BY group_key`, teamID, deadmanGroupPrefix+"%")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
for rows.Next() {
|
||||
var key string
|
||||
var h history
|
||||
if err := rows.Scan(&key, &h.triggeredAt, &h.openID); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
incidents[key] = h
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
bySwitch := map[int64][]deadmanAlert{}
|
||||
for _, hb := range heartbeats {
|
||||
bySwitch[hb.sw.ID] = append(bySwitch[hb.sw.ID], hb)
|
||||
}
|
||||
|
||||
later := func(cur *time.Time, unix int64) *time.Time {
|
||||
t := time.Unix(unix, 0).UTC()
|
||||
if cur == nil || t.After(*cur) {
|
||||
return &t
|
||||
}
|
||||
return cur
|
||||
}
|
||||
|
||||
for _, sw := range set {
|
||||
st := deadmanSwitchStatus{
|
||||
ID: sw.ID, Name: sw.Name, Matcher: sw.Matcher.config(),
|
||||
TimeoutSeconds: int64(sw.Timeout.Seconds()), Severity: sw.Severity,
|
||||
Status: switchDormant, Sources: []deadmanSource{},
|
||||
}
|
||||
|
||||
for _, hb := range bySwitch[sw.ID] {
|
||||
src := deadmanSource{
|
||||
Fingerprint: hb.fingerprint,
|
||||
Labels: hb.labels,
|
||||
Status: switchHealthy,
|
||||
LastHeartbeatAt: time.Unix(hb.receivedAt, 0).UTC(),
|
||||
}
|
||||
if hb.dead(now) {
|
||||
src.Status = switchDead
|
||||
}
|
||||
if h, ok := incidents[hb.groupKey()]; ok {
|
||||
t := time.Unix(h.triggeredAt, 0).UTC()
|
||||
src.LastTriggeredAt = &t
|
||||
st.LastTriggeredAt = later(st.LastTriggeredAt, h.triggeredAt)
|
||||
if h.openID != 0 {
|
||||
id := h.openID
|
||||
src.IncidentID = &id
|
||||
if st.OpenIncidentID == nil || id > *st.OpenIncidentID {
|
||||
st.OpenIncidentID = &id
|
||||
}
|
||||
}
|
||||
}
|
||||
st.LastHeartbeatAt = later(st.LastHeartbeatAt, hb.receivedAt)
|
||||
st.Sources = append(st.Sources, src)
|
||||
|
||||
switch {
|
||||
case src.Status == switchDead:
|
||||
st.Status = switchDead
|
||||
case st.Status == switchDormant:
|
||||
st.Status = switchHealthy
|
||||
}
|
||||
}
|
||||
|
||||
// Dead ones first, then by fingerprint: what needs attention leads, and
|
||||
// the order does not shuffle between refreshes.
|
||||
sort.Slice(st.Sources, func(i, j int) bool {
|
||||
a, b := st.Sources[i], st.Sources[j]
|
||||
if (a.Status == switchDead) != (b.Status == switchDead) {
|
||||
return a.Status == switchDead
|
||||
}
|
||||
return a.Fingerprint < b.Fingerprint
|
||||
})
|
||||
out = append(out, st)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
@@ -1,7 +1,9 @@
|
||||
package api_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/http"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -469,3 +471,272 @@ func TestDeadman_DisabledConfigIsInert(t *testing.T) {
|
||||
t.Errorf("expected the generic sweeper to own the alert, got %v", source)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Per-team configuration
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// Each team decides for itself what a heartbeat is. The same alert is a
|
||||
// heartbeat in one team and an ordinary problem in another.
|
||||
func TestDeadman_ConfigurationIsPerTeam(t *testing.T) {
|
||||
s, _ := deadmanTS(t, deadmanCfg())
|
||||
watched := newTeam(t, s, "watched")
|
||||
unwatched := newTeam(t, s, "unwatched")
|
||||
|
||||
// Only the first team calls Watchdog a heartbeat.
|
||||
resp := s.req(t, http.MethodPost, "/api/teams/"+id64(watched.id)+"/deadman/switches", map[string]any{
|
||||
"matcher": "alertname=Watchdog",
|
||||
"timeout_seconds": 3600,
|
||||
"severity": "critical",
|
||||
})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("configure the watched team: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
postToIntegration(t, s, watched.key, "fp-watched", "Watchdog")
|
||||
postToIntegration(t, s, unwatched.key, "fp-unwatched", "Watchdog")
|
||||
|
||||
// A heartbeat opens nothing where it is one; an ordinary alert opens an
|
||||
// incident where it is not.
|
||||
if got := len(list(t, watched.call(http.MethodGet, "/api/incidents", nil))); got != 0 {
|
||||
t.Errorf("the watched team's heartbeat opened %d incident(s), want 0", got)
|
||||
}
|
||||
if got := len(list(t, unwatched.call(http.MethodGet, "/api/incidents", nil))); got != 1 {
|
||||
t.Errorf("the unwatched team's Watchdog opened %d incident(s), want 1", got)
|
||||
}
|
||||
|
||||
// Silence pages only the team that is watching.
|
||||
s.exec(t, "UPDATE alerts SET received_at = $1 WHERE fingerprint = $2",
|
||||
time.Now().Add(-2*time.Hour).Unix(), "fp-watched")
|
||||
s.exec(t, "UPDATE alerts SET received_at = $1 WHERE fingerprint = $2",
|
||||
time.Now().Add(-2*time.Hour).Unix(), "fp-unwatched")
|
||||
sweep(t, s, noArchive)
|
||||
|
||||
watchedIncidents := list(t, watched.call(http.MethodGet, "/api/incidents", nil))
|
||||
if len(watchedIncidents) != 1 {
|
||||
t.Fatalf("silence opened %d incident(s) for the watching team, want 1", len(watchedIncidents))
|
||||
}
|
||||
if title := watchedIncidents[0]["title"].(string); title != "No heartbeat from Watchdog" {
|
||||
t.Errorf("unexpected incident title %q", title)
|
||||
}
|
||||
if teamID := int64(watchedIncidents[0]["team_id"].(float64)); teamID != watched.id {
|
||||
t.Errorf("the incident opened in team %d, want %d", teamID, watched.id)
|
||||
}
|
||||
|
||||
// The unwatched team's alert went stale the ordinary way, so it has the one
|
||||
// incident it always had — not a second, dead man's switch one.
|
||||
if got := len(list(t, unwatched.call(http.MethodGet, "/api/incidents", nil))); got != 1 {
|
||||
t.Errorf("the unwatched team ended with %d incident(s), want 1", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Configuration is an owner's to change and a member's to read, like the rest of
|
||||
// a team's settings.
|
||||
func TestDeadman_ConfigurationIsOwnerOnly(t *testing.T) {
|
||||
s, _ := deadmanTS(t, deadmanCfg())
|
||||
team := newTeam(t, s, "red")
|
||||
|
||||
// A plain member of that team.
|
||||
var user struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodPost, "/api/users",
|
||||
map[string]string{"username": "plain", "email": "plain@test.com"}), &user)
|
||||
s.req(t, http.MethodPost, "/api/teams/"+id64(team.id)+"/members",
|
||||
map[string]any{"user_id": user.ID, "role": "member"}).Body.Close()
|
||||
var key struct {
|
||||
Key string `json:"key"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodPost, "/api/users/"+id64(user.ID)+"/api-keys",
|
||||
map[string]string{"name": "test"}), &key)
|
||||
|
||||
req, _ := http.NewRequest(http.MethodPost,
|
||||
s.URL+"/api/teams/"+id64(team.id)+"/deadman/switches",
|
||||
strings.NewReader(`{"matcher":"alertname=Watchdog","timeout_seconds":60}`))
|
||||
req.Header.Set("Authorization", "Bearer "+key.Key)
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
resp, err := http.DefaultClient.Do(req)
|
||||
if err != nil {
|
||||
t.Fatalf("put: %v", err)
|
||||
}
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("a member editing the switches: expected 403, got %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
read, _ := http.NewRequest(http.MethodGet, s.URL+"/api/teams/"+id64(team.id)+"/deadman/switches", nil)
|
||||
read.Header.Set("Authorization", "Bearer "+key.Key)
|
||||
got, err := http.DefaultClient.Do(read)
|
||||
if err != nil {
|
||||
t.Fatalf("get: %v", err)
|
||||
}
|
||||
got.Body.Close()
|
||||
if got.StatusCode != http.StatusOK {
|
||||
t.Errorf("a member reading the switches: expected 200, got %d", got.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// A matcher with no alertname watches nothing, silently, which is the failure
|
||||
// this feature exists to prevent — so it is refused at the door, along with the
|
||||
// other things that would make a switch unable to fire.
|
||||
func TestDeadman_UnusableSwitchesAreRejected(t *testing.T) {
|
||||
s, _ := deadmanTS(t, deadmanCfg())
|
||||
|
||||
for name, body := range map[string]map[string]any{
|
||||
"no alertname": {"matcher": "cluster=prod", "timeout_seconds": 900},
|
||||
"malformed": {"matcher": "alertname=Watchdog,garbage", "timeout_seconds": 900},
|
||||
"several": {"matcher": "alertname=A; alertname=B", "timeout_seconds": 900},
|
||||
"zero timeout": {"matcher": "alertname=Watchdog", "timeout_seconds": 0},
|
||||
"bad severity": {"matcher": "alertname=Watchdog", "timeout_seconds": 900, "severity": "loud"},
|
||||
"empty matcher": {"matcher": "", "timeout_seconds": 900},
|
||||
} {
|
||||
resp := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches", body)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusBadRequest {
|
||||
t.Errorf("%s: expected 400, got %d", name, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// The switch list
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// listSwitches reads the default team's switches as the Switches page does.
|
||||
func listSwitches(t *testing.T, s *ts) []map[string]any {
|
||||
t.Helper()
|
||||
return list(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/deadman/switches", nil))
|
||||
}
|
||||
|
||||
// A switch is healthy while its heartbeat is fresh, dead once it is silent, and
|
||||
// dormant until the first one arrives.
|
||||
func TestDeadman_ListReportsStatus(t *testing.T) {
|
||||
s, _ := deadmanTS(t, api.ParseDeadmanConfig("alertname=Watchdog; alertname=NeverSent", time.Hour, "critical"))
|
||||
|
||||
got := listSwitches(t, s)
|
||||
if len(got) != 2 {
|
||||
t.Fatalf("expected 2 switches, got %d", len(got))
|
||||
}
|
||||
for _, sw := range got {
|
||||
if sw["status"] != "dormant" || sw["last_heartbeat_at"] != nil || sw["last_triggered_at"] != nil {
|
||||
t.Errorf("a switch nobody has heard from should be dormant and blank, got %v", sw)
|
||||
}
|
||||
}
|
||||
|
||||
heartbeat(t, s, "fp-watchdog", nil)
|
||||
got = listSwitches(t, s)
|
||||
if got[0]["status"] != "healthy" || got[0]["last_heartbeat_at"] == nil {
|
||||
t.Errorf("a fresh heartbeat should be healthy with a timestamp, got %v", got[0])
|
||||
}
|
||||
if got[1]["status"] != "dormant" {
|
||||
t.Errorf("the other switch is still dormant, got %v", got[1]["status"])
|
||||
}
|
||||
|
||||
silence(t, s, "fp-watchdog", 2*time.Hour)
|
||||
sweep(t, s, noArchive)
|
||||
got = listSwitches(t, s)
|
||||
if got[0]["status"] != "dead" {
|
||||
t.Fatalf("a silent heartbeat should be dead, got %v", got[0]["status"])
|
||||
}
|
||||
if got[0]["last_triggered_at"] == nil || got[0]["open_incident_id"] == nil {
|
||||
t.Errorf("a dead switch should show when it triggered and its open incident, got %v", got[0])
|
||||
}
|
||||
}
|
||||
|
||||
// One matcher, several clusters: the switch is as bad as its worst heartbeat and
|
||||
// each heartbeat is listed on its own.
|
||||
func TestDeadman_ListBreaksDownByFingerprint(t *testing.T) {
|
||||
s, _ := deadmanTS(t, deadmanCfg())
|
||||
|
||||
heartbeat(t, s, "fp-a", map[string]string{"cluster": "a"})
|
||||
heartbeat(t, s, "fp-b", map[string]string{"cluster": "b"})
|
||||
silence(t, s, "fp-b", 2*time.Hour)
|
||||
|
||||
sw := listSwitches(t, s)[0]
|
||||
if sw["status"] != "dead" {
|
||||
t.Errorf("one dead cluster makes the switch dead, got %v", sw["status"])
|
||||
}
|
||||
sources := sw["sources"].([]any)
|
||||
if len(sources) != 2 {
|
||||
t.Fatalf("expected 2 sources, got %d", len(sources))
|
||||
}
|
||||
first, second := sources[0].(map[string]any), sources[1].(map[string]any)
|
||||
if first["fingerprint"] != "fp-b" || first["status"] != "dead" || second["status"] != "healthy" {
|
||||
t.Errorf("the dead source should lead, got %v then %v", first, second)
|
||||
}
|
||||
}
|
||||
|
||||
// Every switch keeps its own deadline.
|
||||
func TestDeadman_TimeoutsArePerSwitch(t *testing.T) {
|
||||
s, _ := deadmanTS(t, deadmanCfg())
|
||||
resp := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches", map[string]any{
|
||||
"matcher": "alertname=Edge", "timeout_seconds": 300,
|
||||
})
|
||||
resp.Body.Close()
|
||||
|
||||
heartbeat(t, s, "fp-watchdog", nil)
|
||||
postWebhook(t, s, []map[string]any{
|
||||
amAlert("fp-edge", "Edge", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||
}, `{}:{alertname="Edge"}`)
|
||||
|
||||
// Ten minutes of silence: past the Edge switch's five, inside Watchdog's hour.
|
||||
silence(t, s, "fp-watchdog", 10*time.Minute)
|
||||
silence(t, s, "fp-edge", 10*time.Minute)
|
||||
|
||||
got := listSwitches(t, s)
|
||||
if got[0]["status"] != "healthy" || got[1]["status"] != "dead" {
|
||||
t.Errorf("want Watchdog healthy and Edge dead, got %v and %v", got[0]["status"], got[1]["status"])
|
||||
}
|
||||
}
|
||||
|
||||
// Deleting is an owner's, is scoped to the team, and leaves what the switch
|
||||
// already opened alone.
|
||||
func TestDeadman_DeleteIsScopedToTheTeam(t *testing.T) {
|
||||
s, _ := deadmanTS(t, deadmanCfg())
|
||||
other := newTeam(t, s, "other")
|
||||
|
||||
id := int64(listSwitches(t, s)[0]["id"].(float64))
|
||||
|
||||
// Another team's owner cannot reach it.
|
||||
resp := other.call(http.MethodDelete, "/api/teams/"+id64(other.id)+"/deadman/switches/"+id64(id), nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNotFound {
|
||||
t.Errorf("deleting another team's switch: expected 404, got %d", resp.StatusCode)
|
||||
}
|
||||
if got := len(listSwitches(t, s)); got != 1 {
|
||||
t.Fatalf("the switch should have survived, %d left", got)
|
||||
}
|
||||
|
||||
resp = s.req(t, http.MethodDelete, "/api/teams/"+defaultTeam+"/deadman/switches/"+id64(id), nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Fatalf("deleting: expected 204, got %d", resp.StatusCode)
|
||||
}
|
||||
if got := len(listSwitches(t, s)); got != 0 {
|
||||
t.Errorf("expected no switches, got %d", got)
|
||||
}
|
||||
}
|
||||
|
||||
// The environment's defaults are handed out once and then belong to the teams.
|
||||
func TestDeadman_SeedRunsOnce(t *testing.T) {
|
||||
s := newTS(t)
|
||||
cfg := api.ParseDeadmanConfig("alertname=Watchdog", time.Hour, "critical")
|
||||
|
||||
if err := api.SeedDeadmanConfigs(context.Background(), s.db, cfg); err != nil {
|
||||
t.Fatalf("seed: %v", err)
|
||||
}
|
||||
if got := len(listSwitches(t, s)); got != 1 {
|
||||
t.Fatalf("the first seed should add the default, got %d switches", got)
|
||||
}
|
||||
|
||||
// The owner deletes it; a restart must not put it back.
|
||||
id := int64(listSwitches(t, s)[0]["id"].(float64))
|
||||
s.req(t, http.MethodDelete, "/api/teams/"+defaultTeam+"/deadman/switches/"+id64(id), nil).Body.Close()
|
||||
if err := api.SeedDeadmanConfigs(context.Background(), s.db, cfg); err != nil {
|
||||
t.Fatalf("seed again: %v", err)
|
||||
}
|
||||
if got := len(listSwitches(t, s)); got != 0 {
|
||||
t.Errorf("a second seed resurrected %d switch(es)", got)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,282 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"crypto/rand"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"log"
|
||||
"math/big"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// The device login flow lets a client that cannot open a browser sign in: it
|
||||
// shows a code, the person approves it in a browser they are signed in to, and
|
||||
// the client is handed an ordinary session. See migration 012.
|
||||
|
||||
const (
|
||||
// deviceTTL is how long a person has to get from the terminal's prompt to an
|
||||
// approval.
|
||||
deviceTTL = 10 * time.Minute
|
||||
|
||||
// deviceInterval is how often the client is told to poll. The server holds it
|
||||
// to that, with a second of slack for clocks and scheduling.
|
||||
deviceInterval = 5 * time.Second
|
||||
|
||||
// deviceStartMaxPerAddr bounds unauthenticated device logins started per
|
||||
// address, since each writes a row.
|
||||
deviceStartMaxPerAddr = 30
|
||||
|
||||
// userCodeAlphabet has no vowels, so a code cannot spell a word, and none of
|
||||
// the characters that read alike (0/O, 1/I/L).
|
||||
userCodeAlphabet = "BCDFGHJKMNPQRSTVWXZ23456789"
|
||||
userCodeLen = 8
|
||||
)
|
||||
|
||||
// newUserCode returns a code for a person to read, as XXXX-XXXX.
|
||||
func newUserCode() (string, error) {
|
||||
max := big.NewInt(int64(len(userCodeAlphabet)))
|
||||
b := make([]byte, userCodeLen)
|
||||
for i := range b {
|
||||
n, err := rand.Int(rand.Reader, max)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
b[i] = userCodeAlphabet[n.Int64()]
|
||||
}
|
||||
return string(b[:4]) + "-" + string(b[4:]), nil
|
||||
}
|
||||
|
||||
// normalizeUserCode reduces whatever a person typed or pasted to the stored
|
||||
// form, so "bcdf ghjk" and "BCDF-GHJK" name the same login. It returns "" for
|
||||
// anything that cannot be a code.
|
||||
func normalizeUserCode(s string) string {
|
||||
var b strings.Builder
|
||||
for _, r := range strings.ToUpper(s) {
|
||||
if strings.ContainsRune(userCodeAlphabet, r) {
|
||||
b.WriteRune(r)
|
||||
}
|
||||
}
|
||||
code := b.String()
|
||||
if len(code) != userCodeLen {
|
||||
return ""
|
||||
}
|
||||
return code[:4] + "-" + code[4:]
|
||||
}
|
||||
|
||||
// handleDeviceStart begins a device login: it returns the device code the
|
||||
// client polls with, and the user code and URL the person is shown.
|
||||
func handleDeviceStart(db *sql.DB, limiter *loginLimiter, publicURL string) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
addrKey := "device:" + clientAddr(r)
|
||||
if limiter.blocked(addrKey, deviceStartMaxPerAddr) {
|
||||
w.Header().Set("Retry-After", strconv.Itoa(int(loginWindow.Seconds())))
|
||||
respond(w, http.StatusTooManyRequests, errResp("too many sign-in attempts, try again later"))
|
||||
return
|
||||
}
|
||||
limiter.fail(addrKey)
|
||||
|
||||
deviceCode, deviceHash, err := randomToken()
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
|
||||
now := time.Now()
|
||||
db.ExecContext(r.Context(), "DELETE FROM device_logins WHERE expires_at < $1", now.Unix())
|
||||
|
||||
// A collision on the user code is one in 27^8; retrying a few times makes
|
||||
// it a non-event rather than a 500.
|
||||
var userCode string
|
||||
for range 5 {
|
||||
userCode, err = newUserCode()
|
||||
if err != nil {
|
||||
break
|
||||
}
|
||||
_, err = db.ExecContext(r.Context(), `
|
||||
INSERT INTO device_logins (device_hash, user_code, expires_at) VALUES ($1, $2, $3)`,
|
||||
deviceHash, userCode, now.Add(deviceTTL).Unix())
|
||||
if err == nil || !isUniqueViolation(err) {
|
||||
break
|
||||
}
|
||||
}
|
||||
if err != nil {
|
||||
log.Printf("device login: start: %v", err)
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
|
||||
respond(w, http.StatusOK, map[string]any{
|
||||
"device_code": deviceCode,
|
||||
"user_code": userCode,
|
||||
// The code is in the URL so nobody has to type it; it is shown anyway,
|
||||
// for the person to check against the terminal before approving.
|
||||
"verification_url": strings.TrimRight(publicURL, "/") + "/device?code=" + url.QueryEscape(userCode),
|
||||
"interval": int(deviceInterval.Seconds()),
|
||||
"expires_in": int(deviceTTL.Seconds()),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// handleDeviceDecision approves or denies a pending device login on behalf of
|
||||
// the signed-in caller.
|
||||
//
|
||||
// It takes a session, not an API key. Approving hands a terminal the caller's
|
||||
// identity, and the approval must come from a browser the person is looking at:
|
||||
// the page shows the code and asks. A script with a key has no business
|
||||
// approving one, and the check keeps it from being a way to mint sessions out of
|
||||
// keys.
|
||||
func handleDeviceDecision(db *sql.DB, approve bool) http.HandlerFunc {
|
||||
status := "denied"
|
||||
if approve {
|
||||
status = "approved"
|
||||
}
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
if _, viaSession := sessionFromContext(r.Context()); !viaSession {
|
||||
respond(w, http.StatusForbidden, errResp("sign in with the web UI to approve a device"))
|
||||
return
|
||||
}
|
||||
var req struct {
|
||||
UserCode string `json:"user_code"`
|
||||
}
|
||||
if err := decodeJSON(r, &req); err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||
return
|
||||
}
|
||||
code := normalizeUserCode(req.UserCode)
|
||||
if code == "" {
|
||||
respond(w, http.StatusBadRequest, errResp("that is not a sign-in code"))
|
||||
return
|
||||
}
|
||||
|
||||
caller, _ := userFromContext(r.Context())
|
||||
// Only a pending login can be decided, and only once: an approval cannot
|
||||
// be overwritten, so a second browser cannot take a login over.
|
||||
res, err := db.ExecContext(r.Context(), `
|
||||
UPDATE device_logins SET status = $1, user_id = $2
|
||||
WHERE user_code = $3 AND status = 'pending' AND expires_at > $4`,
|
||||
status, caller.ID, code, time.Now().Unix())
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if n, _ := res.RowsAffected(); n == 0 {
|
||||
respond(w, http.StatusNotFound, errResp("that sign-in code is unknown, expired or already used"))
|
||||
return
|
||||
}
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}
|
||||
}
|
||||
|
||||
// handleDeviceToken is what the client polls. Pending answers 202; an approval
|
||||
// answers 200 with the session cookie, once; anything else is 410.
|
||||
func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) http.HandlerFunc {
|
||||
gone := func(w http.ResponseWriter, why string) {
|
||||
respond(w, http.StatusGone, map[string]string{"error": why})
|
||||
}
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
var req struct {
|
||||
DeviceCode string `json:"device_code"`
|
||||
}
|
||||
if err := decodeJSON(r, &req); err != nil || req.DeviceCode == "" {
|
||||
respond(w, http.StatusBadRequest, errResp("device_code is required"))
|
||||
return
|
||||
}
|
||||
hash := hashToken(req.DeviceCode)
|
||||
now := time.Now()
|
||||
|
||||
tx, err := db.BeginTx(r.Context(), nil)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
defer tx.Rollback() //nolint:errcheck
|
||||
|
||||
var status string
|
||||
var userID sql.NullInt64
|
||||
var expires, lastPolled int64
|
||||
err = tx.QueryRowContext(r.Context(), `
|
||||
SELECT status, user_id, expires_at, last_polled_at FROM device_logins
|
||||
WHERE device_hash = $1 FOR UPDATE`, hash).Scan(&status, &userID, &expires, &lastPolled)
|
||||
if errors.Is(err, sql.ErrNoRows) || (err == nil && expires <= now.Unix()) {
|
||||
gone(w, "expired")
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
|
||||
switch status {
|
||||
case "denied":
|
||||
tx.ExecContext(r.Context(), "DELETE FROM device_logins WHERE device_hash = $1", hash)
|
||||
tx.Commit() //nolint:errcheck
|
||||
gone(w, "denied")
|
||||
return
|
||||
|
||||
case "pending":
|
||||
// Held to the interval it was given, less a second of slack.
|
||||
if now.Unix()-lastPolled < int64(deviceInterval.Seconds())-1 {
|
||||
w.Header().Set("Retry-After", strconv.Itoa(int(deviceInterval.Seconds())))
|
||||
respond(w, http.StatusTooManyRequests, map[string]string{"error": "slow_down"})
|
||||
return
|
||||
}
|
||||
if _, err := tx.ExecContext(r.Context(),
|
||||
"UPDATE device_logins SET last_polled_at = $1 WHERE device_hash = $2", now.Unix(), hash); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if err := tx.Commit(); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusAccepted, map[string]string{"status": "pending"})
|
||||
return
|
||||
}
|
||||
|
||||
// Approved. Single use: the row goes before the session is made, so two
|
||||
// racing polls cannot both be given one.
|
||||
if _, err := tx.ExecContext(r.Context(), "DELETE FROM device_logins WHERE device_hash = $1", hash); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
var disabled, sso bool
|
||||
if err := tx.QueryRowContext(r.Context(), `
|
||||
SELECT disabled_at IS NOT NULL,
|
||||
EXISTS (SELECT 1 FROM user_identities WHERE user_id = $1)
|
||||
FROM users WHERE id = $1`, userID.Int64).Scan(&disabled, &sso); err != nil {
|
||||
gone(w, "denied")
|
||||
return
|
||||
}
|
||||
if err := tx.Commit(); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if disabled {
|
||||
gone(w, "denied")
|
||||
return
|
||||
}
|
||||
|
||||
// A session for somebody who signs in through the provider carries the
|
||||
// same ceiling as their browser's would, so the terminal is not a way
|
||||
// round it. Password users have none.
|
||||
var maxAge time.Duration
|
||||
if sso {
|
||||
maxAge = ssoMaxAge
|
||||
}
|
||||
if err := startSessionCapped(w, r, db, userID.Int64, publicURL, maxAge); err != nil {
|
||||
log.Printf("device login: start session: %v", err)
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
user, err := fetchUser(r.Context(), db, userID.Int64)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusOK, meResponse{User: user, HasPassword: false})
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,347 @@
|
||||
package api_test
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
type deviceStart struct {
|
||||
DeviceCode string `json:"device_code"`
|
||||
UserCode string `json:"user_code"`
|
||||
VerificationURL string `json:"verification_url"`
|
||||
Interval int `json:"interval"`
|
||||
ExpiresIn int `json:"expires_in"`
|
||||
}
|
||||
|
||||
// startDevice is the terminal asking for a login.
|
||||
func startDevice(t *testing.T, s *ts) deviceStart {
|
||||
t.Helper()
|
||||
resp := newBrowser(t, s.URL).do(t, http.MethodPost, "/api/oidc/device", nil)
|
||||
var d deviceStart
|
||||
decode(t, resp, &d)
|
||||
if d.DeviceCode == "" || d.UserCode == "" {
|
||||
t.Fatalf("device start returned %+v", d)
|
||||
}
|
||||
return d
|
||||
}
|
||||
|
||||
// pollDevice is the terminal polling. It returns the status, and the session
|
||||
// cookie the response set, if any.
|
||||
func pollDevice(t *testing.T, s *ts, code string) (int, *http.Cookie, string) {
|
||||
t.Helper()
|
||||
resp := newBrowser(t, s.URL).do(t, http.MethodPost, "/api/oidc/device/token", map[string]string{"device_code": code})
|
||||
defer resp.Body.Close()
|
||||
var body map[string]any
|
||||
json.NewDecoder(resp.Body).Decode(&body)
|
||||
var cookie *http.Cookie
|
||||
for _, c := range resp.Cookies() {
|
||||
if c.Name == "terdut_session" {
|
||||
cookie = c
|
||||
}
|
||||
}
|
||||
msg, _ := body["error"].(string)
|
||||
if msg == "" {
|
||||
msg, _ = body["status"].(string)
|
||||
}
|
||||
return resp.StatusCode, cookie, msg
|
||||
}
|
||||
|
||||
// readyToPoll lets the next poll through: the server holds a client to the
|
||||
// interval it was given, which a test has no wish to wait out.
|
||||
func (s *ts) readyToPoll(t *testing.T) {
|
||||
t.Helper()
|
||||
s.exec(t, "UPDATE device_logins SET last_polled_at = 0")
|
||||
}
|
||||
|
||||
func decide(t *testing.T, b *browser, what, code string) int {
|
||||
t.Helper()
|
||||
resp := b.do(t, http.MethodPost, "/api/oidc/device/"+what, map[string]string{"user_code": code})
|
||||
resp.Body.Close()
|
||||
return resp.StatusCode
|
||||
}
|
||||
|
||||
func TestDevice_FullFlow(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
|
||||
d := startDevice(t, s)
|
||||
if !strings.HasPrefix(d.VerificationURL, "http://terdut.test/device?code=") ||
|
||||
!strings.Contains(d.VerificationURL, url.QueryEscape(d.UserCode)) {
|
||||
t.Errorf("verification url %q", d.VerificationURL)
|
||||
}
|
||||
if len(d.UserCode) != 9 || d.UserCode[4] != '-' || d.Interval != 5 || d.ExpiresIn != 600 {
|
||||
t.Errorf("start: %+v", d)
|
||||
}
|
||||
|
||||
if status, cookie, msg := pollDevice(t, s, d.DeviceCode); status != http.StatusAccepted || cookie != nil || msg != "pending" {
|
||||
t.Fatalf("first poll: %d %v %q, want 202 pending and no cookie", status, cookie, msg)
|
||||
}
|
||||
|
||||
// The person signs in through the provider in some browser and approves.
|
||||
person := ssoBrowser(t, s)
|
||||
signInSSO(t, idp, person, alice)
|
||||
if got := decide(t, person, "approve", d.UserCode); got != http.StatusNoContent {
|
||||
t.Fatalf("approve: %d", got)
|
||||
}
|
||||
|
||||
s.readyToPoll(t)
|
||||
status, cookie, _ := pollDevice(t, s, d.DeviceCode)
|
||||
if status != http.StatusOK || cookie == nil {
|
||||
t.Fatalf("poll after approval: %d, cookie %v", status, cookie)
|
||||
}
|
||||
// The cookie is a working session for the person who approved.
|
||||
term := newBrowser(t, s.URL)
|
||||
req, _ := http.NewRequest(http.MethodGet, s.URL+"/api/me", nil)
|
||||
req.AddCookie(cookie)
|
||||
resp, err := term.Do(req)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var me struct {
|
||||
User struct {
|
||||
Username string `json:"username"`
|
||||
} `json:"user"`
|
||||
}
|
||||
decode(t, resp, &me)
|
||||
if me.User.Username != "alice" {
|
||||
t.Errorf("session belongs to %q, want alice", me.User.Username)
|
||||
}
|
||||
|
||||
// Single use.
|
||||
if status, cookie, msg := pollDevice(t, s, d.DeviceCode); status != http.StatusGone || cookie != nil || msg != "expired" {
|
||||
t.Errorf("second redemption: %d %v %q, want 410 expired", status, cookie, msg)
|
||||
}
|
||||
// The session was made for an SSO user, so it carries the ceiling.
|
||||
var ceiling *int64
|
||||
s.db.QueryRow("SELECT max_expires_at FROM sessions ORDER BY id DESC LIMIT 1").Scan(&ceiling)
|
||||
if ceiling == nil {
|
||||
t.Error("a device session for an SSO user must carry the SSO session ceiling")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDevice_PasswordUserGetsNoCeiling(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
admin := signedIn(t, s) // password sign-in as the bootstrap admin
|
||||
|
||||
d := startDevice(t, s)
|
||||
if got := decide(t, admin, "approve", d.UserCode); got != http.StatusNoContent {
|
||||
t.Fatalf("approve: %d", got)
|
||||
}
|
||||
s.readyToPoll(t)
|
||||
if status, cookie, _ := pollDevice(t, s, d.DeviceCode); status != http.StatusOK || cookie == nil {
|
||||
t.Fatalf("poll: %d %v", status, cookie)
|
||||
}
|
||||
var ceiling *int64
|
||||
s.db.QueryRow("SELECT max_expires_at FROM sessions ORDER BY id DESC LIMIT 1").Scan(&ceiling)
|
||||
if ceiling != nil {
|
||||
t.Errorf("a password user's device session has a ceiling %d, want none", *ceiling)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDevice_Denied(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
person := ssoBrowser(t, s)
|
||||
signInSSO(t, idp, person, alice)
|
||||
|
||||
d := startDevice(t, s)
|
||||
if got := decide(t, person, "deny", d.UserCode); got != http.StatusNoContent {
|
||||
t.Fatalf("deny: %d", got)
|
||||
}
|
||||
s.readyToPoll(t)
|
||||
if status, cookie, msg := pollDevice(t, s, d.DeviceCode); status != http.StatusGone || cookie != nil || msg != "denied" {
|
||||
t.Errorf("poll: %d %v %q, want 410 denied", status, cookie, msg)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDevice_DecisionNeedsABrowserSession(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
d := startDevice(t, s)
|
||||
|
||||
// Nobody signed in.
|
||||
if got := decide(t, newBrowser(t, s.URL), "approve", d.UserCode); got != http.StatusUnauthorized {
|
||||
t.Errorf("anonymous approve: %d, want 401", got)
|
||||
}
|
||||
// An API key is a credential for scripts, not for approving a terminal.
|
||||
resp := s.req(t, http.MethodPost, "/api/oidc/device/approve", map[string]string{"user_code": d.UserCode})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("approve with an API key: %d, want 403", resp.StatusCode)
|
||||
}
|
||||
if status, _, msg := pollDevice(t, s, d.DeviceCode); status != http.StatusAccepted || msg != "pending" {
|
||||
t.Errorf("the login must still be pending: %d %q", status, msg)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDevice_ApprovalIsFinal(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
first, second := ssoBrowser(t, s), ssoBrowser(t, s)
|
||||
signInSSO(t, idp, first, alice)
|
||||
signInSSO(t, idp, second, idpUser{sub: "sub-mallory", username: "mallory", email: "mallory@example.com", groups: []string{"terdut-users"}})
|
||||
|
||||
d := startDevice(t, s)
|
||||
if got := decide(t, first, "approve", d.UserCode); got != http.StatusNoContent {
|
||||
t.Fatalf("approve: %d", got)
|
||||
}
|
||||
// A second browser cannot take the login over, nor refuse it.
|
||||
for _, what := range []string{"approve", "deny"} {
|
||||
if got := decide(t, second, what, d.UserCode); got != http.StatusNotFound {
|
||||
t.Errorf("%s after approval: %d, want 404", what, got)
|
||||
}
|
||||
}
|
||||
s.readyToPoll(t)
|
||||
_, cookie, _ := pollDevice(t, s, d.DeviceCode)
|
||||
if cookie == nil {
|
||||
t.Fatal("no session")
|
||||
}
|
||||
var name string
|
||||
s.db.QueryRow("SELECT u.username FROM sessions ss JOIN users u ON u.id = ss.user_id ORDER BY ss.id DESC LIMIT 1").Scan(&name)
|
||||
if name != "alice" {
|
||||
t.Errorf("session for %q, want alice", name)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDevice_CodeIsForgivingAboutHowItWasTyped(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
person := ssoBrowser(t, s)
|
||||
signInSSO(t, idp, person, alice)
|
||||
|
||||
d := startDevice(t, s)
|
||||
typed := strings.ToLower(strings.ReplaceAll(d.UserCode, "-", " "))
|
||||
if got := decide(t, person, "approve", typed); got != http.StatusNoContent {
|
||||
t.Errorf("approve %q: %d, want 204", typed, got)
|
||||
}
|
||||
if got := decide(t, person, "approve", "nonsense"); got != http.StatusBadRequest {
|
||||
t.Errorf("approve nonsense: %d, want 400", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDevice_ExpiredAndUnknown(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
person := ssoBrowser(t, s)
|
||||
signInSSO(t, idp, person, alice)
|
||||
|
||||
d := startDevice(t, s)
|
||||
s.exec(t, "UPDATE device_logins SET expires_at = 1")
|
||||
if got := decide(t, person, "approve", d.UserCode); got != http.StatusNotFound {
|
||||
t.Errorf("approve expired: %d, want 404", got)
|
||||
}
|
||||
if status, _, msg := pollDevice(t, s, d.DeviceCode); status != http.StatusGone || msg != "expired" {
|
||||
t.Errorf("poll expired: %d %q, want 410 expired", status, msg)
|
||||
}
|
||||
if status, _, msg := pollDevice(t, s, "not-a-device-code"); status != http.StatusGone || msg != "expired" {
|
||||
t.Errorf("poll unknown: %d %q, want 410 expired", status, msg)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDevice_PollingTooFastIsRefused(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
d := startDevice(t, s)
|
||||
if status, _, _ := pollDevice(t, s, d.DeviceCode); status != http.StatusAccepted {
|
||||
t.Fatalf("first poll: %d", status)
|
||||
}
|
||||
if status, _, msg := pollDevice(t, s, d.DeviceCode); status != http.StatusTooManyRequests || msg != "slow_down" {
|
||||
t.Errorf("immediate second poll: %d %q, want 429 slow_down", status, msg)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDevice_DisabledUserGetsNoSession(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
person := ssoBrowser(t, s)
|
||||
signInSSO(t, idp, person, alice)
|
||||
|
||||
d := startDevice(t, s)
|
||||
decide(t, person, "approve", d.UserCode)
|
||||
s.exec(t, "UPDATE users SET disabled_at = 1 WHERE username = 'alice'")
|
||||
s.readyToPoll(t)
|
||||
var before int
|
||||
s.db.QueryRow("SELECT COUNT(*) FROM sessions").Scan(&before)
|
||||
if status, cookie, _ := pollDevice(t, s, d.DeviceCode); status != http.StatusGone || cookie != nil {
|
||||
t.Errorf("poll: %d %v, want 410 and no cookie", status, cookie)
|
||||
}
|
||||
var after int
|
||||
s.db.QueryRow("SELECT COUNT(*) FROM sessions").Scan(&after)
|
||||
if after != before {
|
||||
t.Error("a session was created for a disabled user")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDevice_OnlyExistsWithSSOConfigured(t *testing.T) {
|
||||
s := newTS(t) // no SSO
|
||||
resp := newBrowser(t, s.URL).do(t, http.MethodPost, "/api/oidc/device", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNotFound {
|
||||
t.Errorf("device start with SSO off: %d, want 404", resp.StatusCode)
|
||||
}
|
||||
|
||||
idp := newFakeIdP(t)
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
s *ts
|
||||
want bool
|
||||
}{{"off", s, false}, {"on", newSSOTS(t, idp), true}} {
|
||||
var cfg struct {
|
||||
DeviceLogin bool `json:"device_login"`
|
||||
}
|
||||
decode(t, newBrowser(t, c.s.URL).do(t, http.MethodGet, "/api/auth/config", nil), &cfg)
|
||||
if cfg.DeviceLogin != c.want {
|
||||
t.Errorf("auth config device_login with SSO %s: %v, want %v", c.name, cfg.DeviceLogin, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestDevice_StartIsRateLimited(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
b := newBrowser(t, s.URL)
|
||||
var last int
|
||||
for range 32 {
|
||||
resp := b.do(t, http.MethodPost, "/api/oidc/device", nil)
|
||||
resp.Body.Close()
|
||||
last = resp.StatusCode
|
||||
}
|
||||
if last != http.StatusTooManyRequests {
|
||||
t.Errorf("32nd start: %d, want 429", last)
|
||||
}
|
||||
}
|
||||
|
||||
// After signing in the browser is sent on to where the person was going, which
|
||||
// is how somebody without a session gets from /device?code=... through the
|
||||
// provider and back to it. Only paths on this server are honoured.
|
||||
func TestSSO_NextIsHonouredOnlyForPathsOnThisServer(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
|
||||
for _, c := range []struct{ next, want string }{
|
||||
{"/device?code=BCDF-GHJK", "/device?code=BCDF-GHJK"},
|
||||
{"/team/members", "/team/members"},
|
||||
{"", "/"},
|
||||
{"//evil.example/x", "/"},
|
||||
{"/\\evil.example", "/"},
|
||||
{"https://evil.example/", "/"},
|
||||
{"evil.example", "/"},
|
||||
{"/api/users", "/"},
|
||||
{"/ok\r\nSet-Cookie: x=y", "/"},
|
||||
{"/" + strings.Repeat("a", 600), "/"},
|
||||
} {
|
||||
b := ssoBrowser(t, s)
|
||||
resp := b.do(t, http.MethodGet, "/api/oidc/login?next="+url.QueryEscape(c.next), nil)
|
||||
resp.Body.Close()
|
||||
loc, _ := url.Parse(resp.Header.Get("Location"))
|
||||
q := loc.Query()
|
||||
got := callback(t, b, idp.issueCode(alice, q.Get("nonce"), q.Get("code_challenge")), q.Get("state"))
|
||||
if got != c.want {
|
||||
t.Errorf("next %q: redirected to %q, want %q", c.next, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,690 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"log"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// evEscalated records a rung of the ladder on the incident's timeline: which
|
||||
// level, and who it woke.
|
||||
const evEscalated = "escalated"
|
||||
|
||||
// escalationPolicy is a team's ladder, loaded whole. It is small — a handful of
|
||||
// levels with a few targets each — and every use needs all of it, so there is
|
||||
// no point reading it a level at a time.
|
||||
type escalationPolicy struct {
|
||||
teamID int64
|
||||
repeatCount int64
|
||||
fallbackTopic string
|
||||
levels []escalationLevel
|
||||
}
|
||||
|
||||
type escalationLevel struct {
|
||||
id int64
|
||||
position int64
|
||||
timeout time.Duration
|
||||
targets []escalationTarget
|
||||
}
|
||||
|
||||
type escalationTarget struct {
|
||||
kind string // "user" or "oncall"
|
||||
userID *int64
|
||||
}
|
||||
|
||||
// configured reports whether this team has anything to escalate through. A
|
||||
// policy row with no levels is the same as no policy: the team gets the
|
||||
// pre-escalation behaviour, which is reminders on the assignee's topic.
|
||||
func (p *escalationPolicy) configured() bool { return p != nil && len(p.levels) > 0 }
|
||||
|
||||
// level returns the level at a 1-based position.
|
||||
func (p *escalationPolicy) level(pos int64) (escalationLevel, bool) {
|
||||
for _, l := range p.levels {
|
||||
if l.position == pos {
|
||||
return l, true
|
||||
}
|
||||
}
|
||||
return escalationLevel{}, false
|
||||
}
|
||||
|
||||
// loadEscalationPolicy reads one team's ladder. A team with no policy row
|
||||
// returns nil, which every caller treats as "not configured" rather than as an
|
||||
// error: most teams will never set one up.
|
||||
func loadEscalationPolicy(ctx context.Context, q querier, teamID int64) (*escalationPolicy, error) {
|
||||
p := &escalationPolicy{teamID: teamID}
|
||||
err := q.QueryRowContext(ctx,
|
||||
"SELECT repeat_count, fallback_topic FROM escalation_policies WHERE team_id = $1",
|
||||
teamID).Scan(&p.repeatCount, &p.fallbackTopic)
|
||||
if err == sql.ErrNoRows {
|
||||
return nil, nil
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
rows, err := q.QueryContext(ctx, `
|
||||
SELECT l.id, l.position, l.timeout_seconds, t.kind, t.user_id
|
||||
FROM escalation_levels l
|
||||
LEFT JOIN escalation_targets t ON t.level_id = l.id
|
||||
WHERE l.team_id = $1
|
||||
ORDER BY l.position, t.id`, teamID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
byPosition := map[int64]int{} // position -> index in p.levels
|
||||
for rows.Next() {
|
||||
var id, position, timeout int64
|
||||
var kind *string
|
||||
var userID *int64
|
||||
if err := rows.Scan(&id, &position, &timeout, &kind, &userID); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
idx, seen := byPosition[position]
|
||||
if !seen {
|
||||
p.levels = append(p.levels, escalationLevel{
|
||||
id: id,
|
||||
position: position,
|
||||
timeout: time.Duration(timeout) * time.Second,
|
||||
})
|
||||
idx = len(p.levels) - 1
|
||||
byPosition[position] = idx
|
||||
}
|
||||
// LEFT JOIN: a level with no targets yet still produces a row, with a
|
||||
// NULL kind. It is a rung that pages nobody, which the API refuses to
|
||||
// store but an older row could still hold.
|
||||
if kind != nil {
|
||||
p.levels[idx].targets = append(p.levels[idx].targets,
|
||||
escalationTarget{kind: *kind, userID: userID})
|
||||
}
|
||||
}
|
||||
return p, rows.Err()
|
||||
}
|
||||
|
||||
// escalate advances every incident whose current level has run out of time.
|
||||
//
|
||||
// Runs on the notifier's tick, beside the reminder pass, because it is the same
|
||||
// question asked differently: reminders ask "has this been ignored long
|
||||
// enough to say it again", escalation asks "long enough to say it to somebody
|
||||
// else". Sharing the tick means one query cadence and one outbox.
|
||||
func escalate(ctx context.Context, db *sql.DB, cfg NotifyConfig) {
|
||||
rows, err := db.QueryContext(ctx, `
|
||||
SELECT i.id, i.team_id, i.escalation_level, i.escalation_level_at, i.escalation_round
|
||||
FROM incidents i
|
||||
JOIN escalation_policies p ON p.team_id = i.team_id
|
||||
WHERE i.resolved_at IS NULL
|
||||
AND i.archived_at IS NULL
|
||||
AND i.status = 'triggered'
|
||||
AND (i.snoozed_until IS NULL OR i.snoozed_until <= $1)
|
||||
AND i.escalation_level > 0`, time.Now().Unix())
|
||||
if err != nil {
|
||||
log.Printf("escalation: find due: %v", err)
|
||||
return
|
||||
}
|
||||
|
||||
type pending struct {
|
||||
incidentID, teamID, level, round int64
|
||||
levelAt int64
|
||||
}
|
||||
var due []pending
|
||||
for rows.Next() {
|
||||
var p pending
|
||||
var levelAt *int64
|
||||
if err := rows.Scan(&p.incidentID, &p.teamID, &p.level, &levelAt, &p.round); err != nil {
|
||||
rows.Close()
|
||||
log.Printf("escalation: scan: %v", err)
|
||||
return
|
||||
}
|
||||
if levelAt == nil {
|
||||
continue
|
||||
}
|
||||
p.levelAt = *levelAt
|
||||
due = append(due, p)
|
||||
}
|
||||
rows.Close()
|
||||
if err := rows.Err(); err != nil {
|
||||
log.Printf("escalation: iterate: %v", err)
|
||||
return
|
||||
}
|
||||
|
||||
now := time.Now()
|
||||
for _, d := range due {
|
||||
policy, err := loadEscalationPolicy(ctx, db, d.teamID)
|
||||
if err != nil {
|
||||
log.Printf("escalation: load policy for team %d: %v", d.teamID, err)
|
||||
continue
|
||||
}
|
||||
if !policy.configured() {
|
||||
continue
|
||||
}
|
||||
current, ok := policy.level(d.level)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
if now.Sub(time.Unix(d.levelAt, 0)) < current.timeout {
|
||||
continue
|
||||
}
|
||||
if err := advanceEscalation(ctx, db, cfg, policy, d.incidentID, d.level, d.round, now); err != nil {
|
||||
log.Printf("escalation: advance incident %d: %v", d.incidentID, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// advanceEscalation moves one incident to its next rung, or off the end of the
|
||||
// ladder.
|
||||
//
|
||||
// The whole move is one transaction: the level, the page and the timeline entry
|
||||
// are one event, and an incident recorded as being at level 3 that nobody at
|
||||
// level 3 was told about is the worst of the possible half-states.
|
||||
func advanceEscalation(ctx context.Context, db *sql.DB, cfg NotifyConfig, policy *escalationPolicy, incidentID, level, round int64, now time.Time) error {
|
||||
tx, err := db.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer tx.Rollback() //nolint:errcheck
|
||||
|
||||
next := level + 1
|
||||
nextRound := round
|
||||
if _, ok := policy.level(next); !ok {
|
||||
// Off the end. Either start the chain again, or make the last call.
|
||||
if round < policy.repeatCount {
|
||||
next, nextRound = 1, round+1
|
||||
} else {
|
||||
if err := escalationExhausted(ctx, tx, policy, incidentID, now); err != nil {
|
||||
return err
|
||||
}
|
||||
return tx.Commit()
|
||||
}
|
||||
}
|
||||
|
||||
target, ok := policy.level(next)
|
||||
if !ok {
|
||||
return nil
|
||||
}
|
||||
paged, err := pageLevel(ctx, tx, cfg, policy, incidentID, target)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE incidents
|
||||
SET escalation_level = $1, escalation_level_at = $2, escalation_round = $3
|
||||
WHERE id = $4`, next, now.Unix(), nextRound, incidentID); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
detail := "level " + strconv.FormatInt(next, 10)
|
||||
if nextRound > round {
|
||||
detail += " (round " + strconv.FormatInt(nextRound+1, 10) + ")"
|
||||
}
|
||||
if len(paged) > 0 {
|
||||
detail += ": " + strings.Join(paged, ", ")
|
||||
} else {
|
||||
// Worth recording loudly: the rung exists, its turn came, and it woke
|
||||
// nobody. That is a policy that looks configured and is not.
|
||||
detail += ": nobody reachable"
|
||||
}
|
||||
if err := logEvent(ctx, tx, incidentID, evEscalated, nil, nil, &detail); err != nil {
|
||||
return err
|
||||
}
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
// escalationExhausted is the end of the line: the fallback topic, once, and a
|
||||
// timeline entry saying the ladder is finished. The incident stays triggered —
|
||||
// escalation running out is not the same as somebody answering.
|
||||
func escalationExhausted(ctx context.Context, tx *sql.Tx, policy *escalationPolicy, incidentID int64, now time.Time) error {
|
||||
detail := "escalation exhausted"
|
||||
if policy.fallbackTopic != "" {
|
||||
if err := enqueueNotification(ctx, tx, incidentID, nil, policy.fallbackTopic, notifyEscalated); err != nil {
|
||||
return err
|
||||
}
|
||||
detail += ": paged " + policy.fallbackTopic
|
||||
} else {
|
||||
detail += ": no fallback topic configured"
|
||||
}
|
||||
|
||||
// Level 0 again, so the sweep stops considering it. The round counter is
|
||||
// left where it is, as the record of how far it got.
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
"UPDATE incidents SET escalation_level = 0, escalation_level_at = NULL WHERE id = $1",
|
||||
incidentID); err != nil {
|
||||
return err
|
||||
}
|
||||
return logEvent(ctx, tx, incidentID, evEscalated, nil, nil, &detail)
|
||||
}
|
||||
|
||||
// pageLevel notifies every target of one level and reports who was woken.
|
||||
//
|
||||
// Each target gets its own outbox row, so each gets its own Acknowledge token:
|
||||
// the button in a notification must acknowledge as the person holding the
|
||||
// phone, not as whoever was paged first.
|
||||
func pageLevel(ctx context.Context, tx *sql.Tx, cfg NotifyConfig, policy *escalationPolicy, incidentID int64, level escalationLevel) ([]string, error) {
|
||||
var paged []string
|
||||
seen := map[int64]bool{}
|
||||
|
||||
for _, t := range level.targets {
|
||||
userID := t.userID
|
||||
if t.kind == "oncall" {
|
||||
onCall, err := currentOnCall(ctx, tx, policy.teamID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if onCall == nil {
|
||||
continue
|
||||
}
|
||||
userID = onCall
|
||||
}
|
||||
if userID == nil || seen[*userID] {
|
||||
continue
|
||||
}
|
||||
seen[*userID] = true
|
||||
|
||||
var topic *string
|
||||
var username string
|
||||
if err := tx.QueryRowContext(ctx,
|
||||
"SELECT ntfy_topic, username FROM users WHERE id = $1 AND disabled_at IS NULL",
|
||||
*userID).Scan(&topic, &username); err != nil {
|
||||
// A disabled or deleted account is not an error in the middle of an
|
||||
// escalation: it is a target that cannot be woken, and the next
|
||||
// level is the answer to that.
|
||||
continue
|
||||
}
|
||||
if topic == nil || *topic == "" {
|
||||
continue
|
||||
}
|
||||
if err := enqueueNotification(ctx, tx, incidentID, userID, *topic, notifyEscalated); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
paged = append(paged, username)
|
||||
}
|
||||
return paged, nil
|
||||
}
|
||||
|
||||
// startEscalation puts a newly opened incident on the first rung, when its team
|
||||
// has a ladder. Called from openIncident, inside the same transaction, so an
|
||||
// incident is never briefly open with no escalation clock running.
|
||||
func startEscalation(ctx context.Context, q querier, incidentID, teamID int64) error {
|
||||
policy, err := loadEscalationPolicy(ctx, q, teamID)
|
||||
if err != nil || !policy.configured() {
|
||||
return err
|
||||
}
|
||||
_, err = q.ExecContext(ctx,
|
||||
"UPDATE incidents SET escalation_level = 1, escalation_level_at = $1 WHERE id = $2",
|
||||
time.Now().Unix(), incidentID)
|
||||
return err
|
||||
}
|
||||
|
||||
// stopEscalation takes an incident off the ladder. Acknowledging or resolving
|
||||
// is somebody saying "I have this", and continuing to wake people after that is
|
||||
// the behaviour that teaches people to ignore the tool.
|
||||
func stopEscalation(ctx context.Context, q querier, incidentID int64) error {
|
||||
_, err := q.ExecContext(ctx,
|
||||
"UPDATE incidents SET escalation_level = 0, escalation_level_at = NULL WHERE id = $1",
|
||||
incidentID)
|
||||
return err
|
||||
}
|
||||
|
||||
// handleGetEscalation returns a team's ladder.
|
||||
func handleGetEscalation(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
teamID, ok := teamParam(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
if !requireTeamMember(w, r, teamID) {
|
||||
return
|
||||
}
|
||||
|
||||
policy, err := loadEscalationPolicy(r.Context(), db, teamID)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
view, err := escalationStatus(r.Context(), db, teamID, policy)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusOK, view)
|
||||
}
|
||||
}
|
||||
|
||||
// Level statuses, as the Escalation page colours them.
|
||||
const (
|
||||
levelReady = "ready"
|
||||
levelEscalating = "escalating"
|
||||
levelUnreachable = "unreachable"
|
||||
)
|
||||
|
||||
// escalationTargetView is a target with who it means today and whether that
|
||||
// person can actually be woken. The extra fields are output only: the PUT body
|
||||
// is the plain escalationTargetJSON, and anything else in it is ignored.
|
||||
type escalationTargetView struct {
|
||||
escalationTargetJSON
|
||||
|
||||
// Username is who the target resolves to right now: the named person, or
|
||||
// whoever the rota says is on call today. Empty when nobody is.
|
||||
Username string `json:"username,omitempty"`
|
||||
|
||||
// Reachable is whether a page to this target would go anywhere, and Problem
|
||||
// says why not when it would not — the same conditions pageLevel skips on.
|
||||
Reachable bool `json:"reachable"`
|
||||
Problem string `json:"problem,omitempty"`
|
||||
}
|
||||
|
||||
type escalationLevelView struct {
|
||||
Position int64 `json:"position"`
|
||||
TimeoutSeconds int64 `json:"timeout_seconds"`
|
||||
Targets []escalationTargetView `json:"targets"`
|
||||
|
||||
// Status is unreachable when no target of the level could be woken — a rung
|
||||
// that looks configured and pages nobody, which is worth seeing before an
|
||||
// incident finds it — escalating when an unanswered incident has climbed to
|
||||
// it, and ready otherwise.
|
||||
Status string `json:"status"`
|
||||
|
||||
// Waiting lists the open, unacknowledged incidents currently on this level.
|
||||
Waiting []int64 `json:"waiting"`
|
||||
}
|
||||
|
||||
type escalationView struct {
|
||||
TeamID int64 `json:"team_id"`
|
||||
RepeatCount int64 `json:"repeat_count"`
|
||||
FallbackTopic string `json:"fallback_topic"`
|
||||
Levels []escalationLevelView `json:"levels"`
|
||||
|
||||
// LastEscalatedAt is when an incident of this team last moved up the ladder,
|
||||
// or ran off the end of it, and LastEscalatedIncidentID which one. Absent
|
||||
// when nothing ever has: a ladder nobody has needed yet.
|
||||
LastEscalatedAt *time.Time `json:"last_escalated_at,omitempty"`
|
||||
LastEscalatedIncidentID *int64 `json:"last_escalated_incident_id,omitempty"`
|
||||
}
|
||||
|
||||
// escalationStatus is a team's ladder together with what it would do right now
|
||||
// and what it has been doing. The resolution follows pageLevel's rules, so the
|
||||
// page cannot promise a page that the notifier would skip.
|
||||
func escalationStatus(ctx context.Context, db *sql.DB, teamID int64, policy *escalationPolicy) (escalationView, error) {
|
||||
base := escalationResponse(policy, teamID)
|
||||
out := escalationView{
|
||||
TeamID: teamID, RepeatCount: base.RepeatCount, FallbackTopic: base.FallbackTopic,
|
||||
Levels: []escalationLevelView{},
|
||||
}
|
||||
if !policy.configured() {
|
||||
return out, nil
|
||||
}
|
||||
|
||||
onCall, err := currentOnCall(ctx, db, teamID)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
|
||||
type account struct {
|
||||
username string
|
||||
topic bool
|
||||
disabled bool
|
||||
}
|
||||
accounts := map[int64]account{}
|
||||
lookup := func(id int64) (account, error) {
|
||||
if a, ok := accounts[id]; ok {
|
||||
return a, nil
|
||||
}
|
||||
var a account
|
||||
var topic *string
|
||||
var disabledAt *int64
|
||||
if err := db.QueryRowContext(ctx,
|
||||
"SELECT username, ntfy_topic, disabled_at FROM users WHERE id = $1", id).
|
||||
Scan(&a.username, &topic, &disabledAt); err != nil {
|
||||
return a, err
|
||||
}
|
||||
a.topic = topic != nil && *topic != ""
|
||||
a.disabled = disabledAt != nil
|
||||
accounts[id] = a
|
||||
return a, nil
|
||||
}
|
||||
|
||||
waiting := map[int64][]int64{}
|
||||
rows, err := db.QueryContext(ctx, `
|
||||
SELECT id, escalation_level FROM incidents
|
||||
WHERE team_id = $1 AND resolved_at IS NULL AND archived_at IS NULL
|
||||
AND status = 'triggered' AND escalation_level > 0
|
||||
ORDER BY id`, teamID)
|
||||
if err != nil {
|
||||
return out, err
|
||||
}
|
||||
for rows.Next() {
|
||||
var id, level int64
|
||||
if err := rows.Scan(&id, &level); err != nil {
|
||||
rows.Close()
|
||||
return out, err
|
||||
}
|
||||
waiting[level] = append(waiting[level], id)
|
||||
}
|
||||
rows.Close()
|
||||
if err := rows.Err(); err != nil {
|
||||
return out, err
|
||||
}
|
||||
|
||||
for _, l := range base.Levels {
|
||||
level := escalationLevelView{
|
||||
Position: l.Position, TimeoutSeconds: l.TimeoutSeconds,
|
||||
Targets: []escalationTargetView{}, Waiting: []int64{},
|
||||
}
|
||||
if w := waiting[l.Position]; w != nil {
|
||||
level.Waiting = w
|
||||
}
|
||||
|
||||
anyReachable := false
|
||||
for _, t := range l.Targets {
|
||||
view := escalationTargetView{escalationTargetJSON: t}
|
||||
userID := t.UserID
|
||||
if t.Kind == "oncall" {
|
||||
userID = onCall
|
||||
}
|
||||
switch {
|
||||
case userID == nil:
|
||||
view.Problem = "nobody is on call today"
|
||||
default:
|
||||
a, err := lookup(*userID)
|
||||
switch {
|
||||
case err != nil:
|
||||
view.Problem = "account not found"
|
||||
case a.disabled:
|
||||
view.Username, view.Problem = a.username, "account is disabled"
|
||||
case !a.topic:
|
||||
view.Username, view.Problem = a.username, "has no ntfy topic"
|
||||
default:
|
||||
view.Username, view.Reachable = a.username, true
|
||||
}
|
||||
}
|
||||
anyReachable = anyReachable || view.Reachable
|
||||
level.Targets = append(level.Targets, view)
|
||||
}
|
||||
|
||||
switch {
|
||||
case !anyReachable:
|
||||
level.Status = levelUnreachable
|
||||
case l.Position >= 2 && len(level.Waiting) > 0:
|
||||
level.Status = levelEscalating
|
||||
default:
|
||||
level.Status = levelReady
|
||||
}
|
||||
out.Levels = append(out.Levels, level)
|
||||
}
|
||||
|
||||
var incidentID, at int64
|
||||
switch err := db.QueryRowContext(ctx, `
|
||||
SELECT e.incident_id, e.created_at
|
||||
FROM incident_events e JOIN incidents i ON i.id = e.incident_id
|
||||
WHERE i.team_id = $1 AND e.type = $2
|
||||
ORDER BY e.created_at DESC, e.id DESC LIMIT 1`, teamID, evEscalated).
|
||||
Scan(&incidentID, &at); {
|
||||
case err == sql.ErrNoRows:
|
||||
case err != nil:
|
||||
return out, err
|
||||
default:
|
||||
t := time.Unix(at, 0).UTC()
|
||||
out.LastEscalatedAt, out.LastEscalatedIncidentID = &t, &incidentID
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
type escalationLevelJSON struct {
|
||||
Position int64 `json:"position"`
|
||||
TimeoutSeconds int64 `json:"timeout_seconds"`
|
||||
Targets []escalationTargetJSON `json:"targets"`
|
||||
}
|
||||
|
||||
type escalationTargetJSON struct {
|
||||
Kind string `json:"kind"`
|
||||
UserID *int64 `json:"user_id,omitempty"`
|
||||
}
|
||||
|
||||
type escalationJSON struct {
|
||||
TeamID int64 `json:"team_id"`
|
||||
RepeatCount int64 `json:"repeat_count"`
|
||||
FallbackTopic string `json:"fallback_topic"`
|
||||
Levels []escalationLevelJSON `json:"levels"`
|
||||
}
|
||||
|
||||
func escalationResponse(p *escalationPolicy, teamID int64) escalationJSON {
|
||||
out := escalationJSON{TeamID: teamID, Levels: []escalationLevelJSON{}}
|
||||
if p == nil {
|
||||
return out
|
||||
}
|
||||
out.RepeatCount = p.repeatCount
|
||||
out.FallbackTopic = p.fallbackTopic
|
||||
for _, l := range p.levels {
|
||||
level := escalationLevelJSON{
|
||||
Position: l.position,
|
||||
TimeoutSeconds: int64(l.timeout.Seconds()),
|
||||
Targets: []escalationTargetJSON{},
|
||||
}
|
||||
for _, t := range l.targets {
|
||||
level.Targets = append(level.Targets, escalationTargetJSON{Kind: t.kind, UserID: t.userID})
|
||||
}
|
||||
out.Levels = append(out.Levels, level)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// handleSetEscalation replaces a team's ladder wholesale.
|
||||
//
|
||||
// Replace rather than patch: the levels are an order, and an API that edits one
|
||||
// rung has to answer what happens to the numbering of the others. Sending the
|
||||
// whole ladder makes the order the client's to decide and the server's to
|
||||
// store, and makes an edit atomic — there is no moment where level 2 exists
|
||||
// twice.
|
||||
func handleSetEscalation(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
teamID, ok := teamParam(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
if !requireTeamOwner(w, r, teamID) {
|
||||
return
|
||||
}
|
||||
|
||||
var req escalationJSON
|
||||
if err := decodeJSON(r, &req); err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||
return
|
||||
}
|
||||
if req.RepeatCount < 0 || req.RepeatCount > 10 {
|
||||
respond(w, http.StatusBadRequest, errResp("repeat_count must be between 0 and 10"))
|
||||
return
|
||||
}
|
||||
for i, l := range req.Levels {
|
||||
if l.TimeoutSeconds <= 0 {
|
||||
respond(w, http.StatusBadRequest, errResp("every level needs a timeout"))
|
||||
return
|
||||
}
|
||||
if len(l.Targets) == 0 {
|
||||
// A rung that pages nobody is not a delay, it is a silence with
|
||||
// a number on it.
|
||||
respond(w, http.StatusBadRequest,
|
||||
errResp("level "+strconv.FormatInt(int64(i+1), 10)+" has no targets"))
|
||||
return
|
||||
}
|
||||
for _, t := range l.Targets {
|
||||
switch t.Kind {
|
||||
case "oncall":
|
||||
if t.UserID != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("an oncall target takes no user_id"))
|
||||
return
|
||||
}
|
||||
case "user":
|
||||
if t.UserID == nil {
|
||||
respond(w, http.StatusBadRequest, errResp("a user target needs a user_id"))
|
||||
return
|
||||
}
|
||||
default:
|
||||
respond(w, http.StatusBadRequest, errResp("target kind must be user or oncall"))
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
tx, err := db.BeginTx(r.Context(), nil)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
defer tx.Rollback() //nolint:errcheck
|
||||
|
||||
if _, err := tx.ExecContext(r.Context(), `
|
||||
INSERT INTO escalation_policies (team_id, repeat_count, fallback_topic, updated_at)
|
||||
VALUES ($1, $2, $3, `+nowEpoch+`)
|
||||
ON CONFLICT (team_id) DO UPDATE SET
|
||||
repeat_count = excluded.repeat_count,
|
||||
fallback_topic = excluded.fallback_topic,
|
||||
updated_at = excluded.updated_at`,
|
||||
teamID, req.RepeatCount, req.FallbackTopic); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
// The levels are replaced, not merged; the cascade takes the targets.
|
||||
if _, err := tx.ExecContext(r.Context(),
|
||||
"DELETE FROM escalation_levels WHERE team_id = $1", teamID); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
|
||||
for i, l := range req.Levels {
|
||||
var levelID int64
|
||||
if err := tx.QueryRowContext(r.Context(), `
|
||||
INSERT INTO escalation_levels (team_id, position, timeout_seconds)
|
||||
VALUES ($1, $2, $3) RETURNING id`,
|
||||
teamID, int64(i+1), l.TimeoutSeconds).Scan(&levelID); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
for _, t := range l.Targets {
|
||||
if _, err := tx.ExecContext(r.Context(), `
|
||||
INSERT INTO escalation_targets (level_id, kind, user_id)
|
||||
VALUES ($1, $2, $3)`, levelID, t.Kind, t.UserID); err != nil {
|
||||
// The only foreign key here is the user.
|
||||
respond(w, http.StatusBadRequest, errResp("unknown user in targets"))
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if err := tx.Commit(); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
|
||||
policy, err := loadEscalationPolicy(r.Context(), db, teamID)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusOK, escalationResponse(policy, teamID))
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,504 @@
|
||||
package api_test
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||
)
|
||||
|
||||
// teamUser creates a user in the default team with an ntfy topic, so they can
|
||||
// actually be paged.
|
||||
func teamUser(t *testing.T, s *ts, username, topic string) int64 {
|
||||
t.Helper()
|
||||
var user struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodPost, "/api/users",
|
||||
map[string]string{"username": username, "email": username + "@test.com"}), &user)
|
||||
resp := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/members",
|
||||
map[string]any{"user_id": user.ID, "role": "member"})
|
||||
resp.Body.Close()
|
||||
setTopic(t, s, int(user.ID), topic)
|
||||
return user.ID
|
||||
}
|
||||
|
||||
// Escalation is all timeouts, and there is no fake clock in this package. The
|
||||
// tests back-date escalation_level_at instead, which is the same trick the dead
|
||||
// man's switch tests use on received_at: the sweeper reads a stored timestamp,
|
||||
// so moving the timestamp is moving the clock.
|
||||
|
||||
// ladder configures the default team with two levels: the rota first, then a
|
||||
// named person, then the fallback topic.
|
||||
func ladder(t *testing.T, s *ts, secondUserID int64, repeat int64, fallback string) {
|
||||
t.Helper()
|
||||
resp := s.req(t, http.MethodPut, "/api/teams/"+defaultTeam+"/escalation", map[string]any{
|
||||
"repeat_count": repeat,
|
||||
"fallback_topic": fallback,
|
||||
"levels": []map[string]any{
|
||||
{"timeout_seconds": 300, "targets": []map[string]any{{"kind": "oncall"}}},
|
||||
{"timeout_seconds": 300, "targets": []map[string]any{{"kind": "user", "user_id": secondUserID}}},
|
||||
},
|
||||
})
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("configure the ladder: %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// overdue back-dates an incident's current level so its timeout has passed.
|
||||
func overdue(t *testing.T, s *ts, incidentID int64) {
|
||||
t.Helper()
|
||||
s.exec(t, "UPDATE incidents SET escalation_level_at = $1 WHERE id = $2",
|
||||
time.Now().Add(-time.Hour).Unix(), incidentID)
|
||||
}
|
||||
|
||||
func escalationLevel(t *testing.T, s *ts, incidentID int64) (level, round int64) {
|
||||
t.Helper()
|
||||
if err := s.db.QueryRow(
|
||||
"SELECT escalation_level, escalation_round FROM incidents WHERE id = $1",
|
||||
incidentID).Scan(&level, &round); err != nil {
|
||||
t.Fatalf("read escalation state: %v", err)
|
||||
}
|
||||
return level, round
|
||||
}
|
||||
|
||||
// The whole point: nobody answers, so somebody else is woken.
|
||||
func TestEscalation_PagesTheNextLevel(t *testing.T) {
|
||||
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||
second := teamUser(t, s, "second", "terdut-second")
|
||||
ladder(t, s, second, 0, "terdut-fallback")
|
||||
|
||||
postWebhook(t, s, []map[string]any{
|
||||
amAlert("fp-esc", "DiskFull", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||
})
|
||||
s.sweepNotify(t)
|
||||
|
||||
// Level 1 is the rota, so the first page went to the admin.
|
||||
if level, _ := escalationLevel(t, s, 1); level != 1 {
|
||||
t.Fatalf("a new incident should start at level 1, got %d", level)
|
||||
}
|
||||
if got := f.topicsSince(t); len(got) == 0 || got[0] != "terdut-admin" {
|
||||
t.Fatalf("the first page should go to the on-call user, went to %v", got)
|
||||
}
|
||||
|
||||
// Time passes with no acknowledgement.
|
||||
f.forget()
|
||||
overdue(t, s, 1)
|
||||
s.sweepNotify(t)
|
||||
|
||||
if level, _ := escalationLevel(t, s, 1); level != 2 {
|
||||
t.Errorf("expected level 2, got %d", level)
|
||||
}
|
||||
if got := f.topicsSince(t); len(got) != 1 || got[0] != "terdut-second" {
|
||||
t.Errorf("level 2 should page the named user, paged %v", got)
|
||||
}
|
||||
|
||||
// And the timeline says so, which is what somebody reads afterwards to
|
||||
// understand why their phone rang at 04:00.
|
||||
timeline := list(t, s.req(t, http.MethodGet, "/api/incidents/1/timeline", nil))
|
||||
found := ""
|
||||
for _, e := range timeline {
|
||||
if e["type"] == "escalated" {
|
||||
found, _ = e["detail"].(string)
|
||||
}
|
||||
}
|
||||
if found == "" {
|
||||
t.Error("the timeline should record the escalation")
|
||||
} else if !strings.HasPrefix(found, "level 2") || !strings.Contains(found, "second") {
|
||||
t.Errorf("the escalation entry should say which level and who: %q", found)
|
||||
}
|
||||
}
|
||||
|
||||
// Acknowledging is somebody saying "I have this". Nobody else should be woken.
|
||||
func TestEscalation_AcknowledgementStopsIt(t *testing.T) {
|
||||
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||
second := teamUser(t, s, "second", "terdut-second")
|
||||
ladder(t, s, second, 0, "terdut-fallback")
|
||||
|
||||
postWebhook(t, s, []map[string]any{
|
||||
amAlert("fp-ack", "DiskFull", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||
})
|
||||
s.sweepNotify(t)
|
||||
|
||||
s.req(t, http.MethodPost, "/api/incidents/1/acknowledge", nil).Body.Close()
|
||||
if level, _ := escalationLevel(t, s, 1); level != 0 {
|
||||
t.Errorf("acknowledging should take the incident off the ladder, level is %d", level)
|
||||
}
|
||||
|
||||
f.forget()
|
||||
overdue(t, s, 1) // no-op: level is 0, so there is nothing due
|
||||
s.sweepNotify(t)
|
||||
if got := f.topicsSince(t); len(got) != 0 {
|
||||
t.Errorf("an acknowledged incident should page nobody, paged %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Resolving stops it too, and by the same mechanism.
|
||||
func TestEscalation_ResolutionStopsIt(t *testing.T) {
|
||||
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||
second := teamUser(t, s, "second", "terdut-second")
|
||||
ladder(t, s, second, 0, "terdut-fallback")
|
||||
|
||||
postWebhook(t, s, []map[string]any{
|
||||
amAlert("fp-res", "DiskFull", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||
})
|
||||
s.sweepNotify(t)
|
||||
s.req(t, http.MethodPost, "/api/incidents/1/resolve", nil).Body.Close()
|
||||
|
||||
f.forget()
|
||||
overdue(t, s, 1)
|
||||
s.sweepNotify(t)
|
||||
if level, _ := escalationLevel(t, s, 1); level != 0 {
|
||||
t.Errorf("a resolved incident should be off the ladder, level is %d", level)
|
||||
}
|
||||
}
|
||||
|
||||
// Snoozing is a deliberate "not now", so the ladder waits rather than carrying
|
||||
// on without the person who asked for quiet.
|
||||
func TestEscalation_SnoozePausesIt(t *testing.T) {
|
||||
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||
second := teamUser(t, s, "second", "terdut-second")
|
||||
ladder(t, s, second, 0, "terdut-fallback")
|
||||
|
||||
postWebhook(t, s, []map[string]any{
|
||||
amAlert("fp-snooze", "DiskFull", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||
})
|
||||
s.sweepNotify(t)
|
||||
|
||||
resp := s.req(t, http.MethodPost, "/api/incidents/1/snooze", map[string]any{"duration": "1h"})
|
||||
resp.Body.Close()
|
||||
|
||||
f.forget()
|
||||
overdue(t, s, 1)
|
||||
s.sweepNotify(t)
|
||||
|
||||
if level, _ := escalationLevel(t, s, 1); level != 1 {
|
||||
t.Errorf("a snoozed incident should stay where it is, level is %d", level)
|
||||
}
|
||||
if got := f.topicsSince(t); len(got) != 0 {
|
||||
t.Errorf("a snoozed incident should page nobody, paged %v", got)
|
||||
}
|
||||
|
||||
// When the snooze ends, the ladder picks up where it left off.
|
||||
s.exec(t, "UPDATE incidents SET snoozed_until = $1 WHERE id = 1", time.Now().Add(-time.Minute).Unix())
|
||||
s.sweepNotify(t)
|
||||
if level, _ := escalationLevel(t, s, 1); level != 2 {
|
||||
t.Errorf("after the snooze the ladder should resume, level is %d", level)
|
||||
}
|
||||
}
|
||||
|
||||
// Running out of ladder pages the team's fallback topic once, and says so.
|
||||
func TestEscalation_ExhaustionPagesTheFallback(t *testing.T) {
|
||||
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||
second := teamUser(t, s, "second", "terdut-second")
|
||||
ladder(t, s, second, 0, "terdut-fallback")
|
||||
|
||||
postWebhook(t, s, []map[string]any{
|
||||
amAlert("fp-end", "DiskFull", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||
})
|
||||
s.sweepNotify(t)
|
||||
|
||||
overdue(t, s, 1)
|
||||
s.sweepNotify(t) // level 2
|
||||
f.forget()
|
||||
overdue(t, s, 1)
|
||||
s.sweepNotify(t) // off the end
|
||||
|
||||
if got := f.topicsSince(t); len(got) != 1 || got[0] != "terdut-fallback" {
|
||||
t.Errorf("exhaustion should page the fallback topic once, paged %v", got)
|
||||
}
|
||||
level, _ := escalationLevel(t, s, 1)
|
||||
if level != 0 {
|
||||
t.Errorf("an exhausted ladder should stop asking, level is %d", level)
|
||||
}
|
||||
|
||||
// The incident is still open: running out of people is not an answer.
|
||||
var status string
|
||||
if err := s.db.QueryRow("SELECT status FROM incidents WHERE id = 1").Scan(&status); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if status != "triggered" {
|
||||
t.Errorf("exhaustion must not resolve the incident, status is %q", status)
|
||||
}
|
||||
}
|
||||
|
||||
// repeat_count walks the whole ladder again before giving up.
|
||||
func TestEscalation_RepeatsTheChain(t *testing.T) {
|
||||
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||
second := teamUser(t, s, "second", "terdut-second")
|
||||
ladder(t, s, second, 1, "terdut-fallback") // one extra round
|
||||
|
||||
postWebhook(t, s, []map[string]any{
|
||||
amAlert("fp-repeat", "DiskFull", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||
})
|
||||
s.sweepNotify(t)
|
||||
|
||||
overdue(t, s, 1)
|
||||
s.sweepNotify(t) // level 2
|
||||
f.forget()
|
||||
overdue(t, s, 1)
|
||||
s.sweepNotify(t) // back to level 1, round 2
|
||||
|
||||
level, round := escalationLevel(t, s, 1)
|
||||
if level != 1 || round != 1 {
|
||||
t.Errorf("expected level 1 round 1, got level %d round %d", level, round)
|
||||
}
|
||||
if got := f.topicsSince(t); len(got) != 1 || got[0] != "terdut-admin" {
|
||||
t.Errorf("the second round should start at the top again, paged %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// A team without a ladder keeps exactly the behaviour it had, and never gets
|
||||
// both a reminder and an escalation for the same silence.
|
||||
func TestEscalation_WithoutAPolicyRemindersStillRun(t *testing.T) {
|
||||
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||
postWebhook(t, s, []map[string]any{
|
||||
amAlert("fp-noesc", "DiskFull", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||
})
|
||||
s.sweepNotify(t)
|
||||
|
||||
// Age the first notification past the repeat interval.
|
||||
f.forget()
|
||||
s.exec(t, "UPDATE notifications SET created_at = $1, sent_at = $1",
|
||||
time.Now().Add(-time.Hour).Unix())
|
||||
s.sweepNotify(t)
|
||||
|
||||
if got := f.topicsSince(t); len(got) != 1 || got[0] != "terdut-admin" {
|
||||
t.Errorf("without a ladder the reminder should still fire, paged %v", got)
|
||||
}
|
||||
if level, _ := escalationLevel(t, s, 1); level != 0 {
|
||||
t.Errorf("an incident in a team with no ladder should not be on one, level is %d", level)
|
||||
}
|
||||
}
|
||||
|
||||
// With a ladder, reminders stop: two pages for one silence is how people learn
|
||||
// to mute the tool.
|
||||
func TestEscalation_WithAPolicyRemindersDoNotAlsoFire(t *testing.T) {
|
||||
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||
second := teamUser(t, s, "second", "terdut-second")
|
||||
ladder(t, s, second, 0, "terdut-fallback")
|
||||
|
||||
postWebhook(t, s, []map[string]any{
|
||||
amAlert("fp-both", "DiskFull", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||
})
|
||||
s.sweepNotify(t)
|
||||
|
||||
f.forget()
|
||||
// Old enough for a reminder, but not yet due for escalation.
|
||||
s.exec(t, "UPDATE notifications SET created_at = $1, sent_at = $1",
|
||||
time.Now().Add(-time.Hour).Unix())
|
||||
s.sweepNotify(t)
|
||||
|
||||
if got := f.topicsSince(t); len(got) != 0 {
|
||||
t.Errorf("a team with a ladder should not also get reminders, paged %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// The API refuses a ladder that cannot page anybody.
|
||||
func TestEscalation_RejectsAnUnusablePolicy(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
body map[string]any
|
||||
}{
|
||||
{"a level with no targets", map[string]any{
|
||||
"levels": []map[string]any{{"timeout_seconds": 300, "targets": []map[string]any{}}},
|
||||
}},
|
||||
{"a level with no timeout", map[string]any{
|
||||
"levels": []map[string]any{{"timeout_seconds": 0, "targets": []map[string]any{{"kind": "oncall"}}}},
|
||||
}},
|
||||
{"a user target with no user", map[string]any{
|
||||
"levels": []map[string]any{{"timeout_seconds": 300, "targets": []map[string]any{{"kind": "user"}}}},
|
||||
}},
|
||||
{"an unknown target kind", map[string]any{
|
||||
"levels": []map[string]any{{"timeout_seconds": 300, "targets": []map[string]any{{"kind": "everybody"}}}},
|
||||
}},
|
||||
{"an absurd repeat count", map[string]any{
|
||||
"repeat_count": 99,
|
||||
"levels": []map[string]any{{"timeout_seconds": 300, "targets": []map[string]any{{"kind": "oncall"}}}},
|
||||
}},
|
||||
} {
|
||||
resp := s.req(t, http.MethodPut, "/api/teams/"+defaultTeam+"/escalation", c.body)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusBadRequest {
|
||||
t.Errorf("%s: expected 400, got %d", c.name, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Editing the ladder is an owner's job; reading it is any member's.
|
||||
func TestEscalation_OwnerOnlyToEdit(t *testing.T) {
|
||||
s := newTS(t)
|
||||
_, call := member(t, s, "plain")
|
||||
|
||||
resp := call(http.MethodPut, "/api/teams/"+defaultTeam+"/escalation", map[string]any{
|
||||
"levels": []map[string]any{{"timeout_seconds": 300, "targets": []map[string]any{{"kind": "oncall"}}}},
|
||||
})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("a member editing the ladder: expected 403, got %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
resp = call(http.MethodGet, "/api/teams/"+defaultTeam+"/escalation", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Errorf("a member reading the ladder: expected 200, got %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// A target who cannot be woken is not a reason to stop: the next level is the
|
||||
// answer to an unreachable one.
|
||||
func TestEscalation_SkipsUnreachableTargets(t *testing.T) {
|
||||
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||
// Second user has no ntfy topic at all.
|
||||
var user struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodPost, "/api/users",
|
||||
map[string]string{"username": "silent", "email": "silent@test.com"}), &user)
|
||||
s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/members",
|
||||
map[string]any{"user_id": user.ID, "role": "member"}).Body.Close()
|
||||
|
||||
ladder(t, s, user.ID, 0, "terdut-fallback")
|
||||
|
||||
postWebhook(t, s, []map[string]any{
|
||||
amAlert("fp-silent", "DiskFull", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||
})
|
||||
s.sweepNotify(t)
|
||||
|
||||
f.forget()
|
||||
overdue(t, s, 1)
|
||||
s.sweepNotify(t)
|
||||
|
||||
// Level 2 was entered even though it woke nobody, so the ladder keeps
|
||||
// moving toward the fallback rather than stalling on a silent rung.
|
||||
if level, _ := escalationLevel(t, s, 1); level != 2 {
|
||||
t.Errorf("expected the ladder to advance past an unreachable target, level is %d", level)
|
||||
}
|
||||
if got := f.topicsSince(t); len(got) != 0 {
|
||||
t.Errorf("a target with no topic should page nothing, paged %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// The ladder as the Escalation page reads it
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
type ladderLevel struct {
|
||||
Status string `json:"status"`
|
||||
Waiting []int64 `json:"waiting"`
|
||||
Targets []struct {
|
||||
Kind string `json:"kind"`
|
||||
Username string `json:"username"`
|
||||
Reachable bool `json:"reachable"`
|
||||
Problem string `json:"problem"`
|
||||
} `json:"targets"`
|
||||
}
|
||||
|
||||
type ladderView struct {
|
||||
Levels []ladderLevel `json:"levels"`
|
||||
LastEscalatedAt *string `json:"last_escalated_at"`
|
||||
LastEscalatedIncidentID *int64 `json:"last_escalated_incident_id"`
|
||||
}
|
||||
|
||||
func readLadder(t *testing.T, s *ts) ladderView {
|
||||
t.Helper()
|
||||
var v ladderView
|
||||
decode(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/escalation", nil), &v)
|
||||
return v
|
||||
}
|
||||
|
||||
// Targets say who they mean today, so "whoever is on call" is a name and not a
|
||||
// promise.
|
||||
func TestEscalation_StatusResolvesTargets(t *testing.T) {
|
||||
s, _ := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||
second := teamUser(t, s, "second", "terdut-second")
|
||||
ladder(t, s, second, 0, "terdut-fallback")
|
||||
|
||||
v := readLadder(t, s)
|
||||
if len(v.Levels) != 2 {
|
||||
t.Fatalf("expected 2 levels, got %d", len(v.Levels))
|
||||
}
|
||||
if got := v.Levels[0].Targets[0]; got.Kind != "oncall" || got.Username != "admin" || !got.Reachable {
|
||||
t.Errorf("the rota target should resolve to the person on call, got %+v", got)
|
||||
}
|
||||
if got := v.Levels[1].Targets[0]; got.Username != "second" || !got.Reachable {
|
||||
t.Errorf("the named target should be reachable, got %+v", got)
|
||||
}
|
||||
if v.Levels[0].Status != "ready" || v.Levels[1].Status != "ready" || v.LastEscalatedAt != nil {
|
||||
t.Errorf("an idle, healthy ladder is ready and has never escalated, got %+v", v)
|
||||
}
|
||||
}
|
||||
|
||||
// A rung that would page nobody is called out before an incident finds it.
|
||||
func TestEscalation_StatusFlagsUnreachableLevels(t *testing.T) {
|
||||
s, _ := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||
silent := teamUser(t, s, "silent", "terdut-silent")
|
||||
ladder(t, s, silent, 0, "terdut-fallback")
|
||||
|
||||
// Nobody on call today, and the named person loses their topic.
|
||||
s.exec(t, "DELETE FROM schedule_entries")
|
||||
s.exec(t, "UPDATE users SET ntfy_topic = NULL WHERE id = $1", silent)
|
||||
|
||||
v := readLadder(t, s)
|
||||
if v.Levels[0].Status != "unreachable" || v.Levels[0].Targets[0].Problem != "nobody is on call today" {
|
||||
t.Errorf("an empty rota should make level 1 unreachable, got %+v", v.Levels[0])
|
||||
}
|
||||
if v.Levels[1].Status != "unreachable" || v.Levels[1].Targets[0].Problem != "has no ntfy topic" {
|
||||
t.Errorf("a person with no topic should make level 2 unreachable, got %+v", v.Levels[1])
|
||||
}
|
||||
|
||||
s.exec(t, "UPDATE users SET disabled_at = 1 WHERE id = $1", silent)
|
||||
if p := readLadder(t, s).Levels[1].Targets[0].Problem; p != "account is disabled" {
|
||||
t.Errorf("a disabled account should say so, got %q", p)
|
||||
}
|
||||
}
|
||||
|
||||
// Where unanswered incidents are right now, and when the ladder last did its
|
||||
// job.
|
||||
func TestEscalation_StatusShowsWhoIsWaitingAndLastEscalation(t *testing.T) {
|
||||
s, _ := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||
second := teamUser(t, s, "second", "terdut-second")
|
||||
ladder(t, s, second, 0, "terdut-fallback")
|
||||
|
||||
postWebhook(t, s, []map[string]any{
|
||||
amAlert("fp-wait", "DiskFull", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||
})
|
||||
s.sweepNotify(t)
|
||||
|
||||
// On level 1 it is waiting, which is normal and not yet an escalation.
|
||||
v := readLadder(t, s)
|
||||
if len(v.Levels[0].Waiting) != 1 || v.Levels[0].Status != "ready" || v.LastEscalatedAt != nil {
|
||||
t.Fatalf("a fresh incident waits on level 1 quietly, got %+v", v)
|
||||
}
|
||||
|
||||
overdue(t, s, 1)
|
||||
s.sweepNotify(t)
|
||||
v = readLadder(t, s)
|
||||
if v.Levels[1].Status != "escalating" || len(v.Levels[1].Waiting) != 1 || v.Levels[1].Waiting[0] != 1 {
|
||||
t.Errorf("level 2 should be escalating with the incident on it, got %+v", v.Levels[1])
|
||||
}
|
||||
if v.LastEscalatedAt == nil || v.LastEscalatedIncidentID == nil || *v.LastEscalatedIncidentID != 1 {
|
||||
t.Errorf("the escalation should be recorded, got %+v", v)
|
||||
}
|
||||
|
||||
// Somebody answers: nothing is waiting, but the history stays.
|
||||
s.req(t, http.MethodPost, "/api/incidents/1/acknowledge", nil).Body.Close()
|
||||
v = readLadder(t, s)
|
||||
if v.Levels[1].Status != "ready" || len(v.Levels[1].Waiting) != 0 || v.LastEscalatedAt == nil {
|
||||
t.Errorf("an acknowledged incident stops waiting but stays in the history, got %+v", v)
|
||||
}
|
||||
}
|
||||
|
||||
// No ladder is a real answer, not an error.
|
||||
func TestEscalation_StatusWithoutALadder(t *testing.T) {
|
||||
s, _ := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||
v := readLadder(t, s)
|
||||
if len(v.Levels) != 0 || v.LastEscalatedAt != nil {
|
||||
t.Errorf("a team with no ladder should read as empty, got %+v", v)
|
||||
}
|
||||
}
|
||||
@@ -1,8 +1,11 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"log"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"strings"
|
||||
@@ -77,3 +80,39 @@ func decodeJSON(r *http.Request, v any) error {
|
||||
func errResp(msg string) map[string]string {
|
||||
return map[string]string{"error": msg}
|
||||
}
|
||||
|
||||
// withAdvisoryLock runs fn only if it can take the named Postgres advisory lock on a
|
||||
// dedicated connection, and skips fn otherwise. This is what keeps the archiver and
|
||||
// notifier safe to run on more than one replica: whichever instance's tick gets there
|
||||
// first does the work; the rest see the lock held and simply wait for their next tick
|
||||
// instead of running the same pass concurrently.
|
||||
//
|
||||
// pg_try_advisory_lock is session-scoped, so taking and releasing it must happen on the
|
||||
// same connection, reserved via db.Conn rather than borrowed from the pool's shared
|
||||
// connections fn itself may use — and released (unlocked, then closed) before returning,
|
||||
// since a session lock otherwise outlives this call and leaks onto whatever reuses the
|
||||
// pooled connection next.
|
||||
func withAdvisoryLock(ctx context.Context, db *sql.DB, key int64, name string, fn func()) {
|
||||
conn, err := db.Conn(ctx)
|
||||
if err != nil {
|
||||
log.Printf("%s: advisory lock: acquire connection: %v", name, err)
|
||||
return
|
||||
}
|
||||
defer conn.Close()
|
||||
|
||||
var locked bool
|
||||
if err := conn.QueryRowContext(ctx, "SELECT pg_try_advisory_lock($1)", key).Scan(&locked); err != nil {
|
||||
log.Printf("%s: advisory lock: %v", name, err)
|
||||
return
|
||||
}
|
||||
if !locked {
|
||||
return // another replica is already running this pass
|
||||
}
|
||||
defer func() {
|
||||
if _, err := conn.ExecContext(ctx, "SELECT pg_advisory_unlock($1)", key); err != nil {
|
||||
log.Printf("%s: advisory unlock: %v", name, err)
|
||||
}
|
||||
}()
|
||||
|
||||
fn()
|
||||
}
|
||||
|
||||
@@ -36,6 +36,10 @@ const (
|
||||
evUnsnoozed = "unsnoozed"
|
||||
evResolved = "resolved"
|
||||
evNote = "note"
|
||||
// evResolutionNote is the note worth finding again: what fixed it. The
|
||||
// similar-incidents lookup and the page lead with these; plain notes are
|
||||
// the working chatter and stay one click away.
|
||||
evResolutionNote = "resolution_note"
|
||||
evDeadmanSilent = "deadman_silent"
|
||||
)
|
||||
|
||||
@@ -51,12 +55,20 @@ type querier interface {
|
||||
}
|
||||
|
||||
const incidentSelectFrom = `
|
||||
SELECT i.id, i.group_key, i.title, i.group_labels, i.status, i.severity,
|
||||
SELECT i.id, i.team_id, t.name, i.group_key, i.title, i.group_labels, i.status, i.severity,
|
||||
i.escalation_level,
|
||||
-- When this level runs out. Computed here rather than in Go because
|
||||
-- the timeout lives beside the level in the policy, and one join is
|
||||
-- cheaper than a second query per incident in a list.
|
||||
(SELECT i.escalation_level_at + el.timeout_seconds
|
||||
FROM escalation_levels el
|
||||
WHERE el.team_id = i.team_id AND el.position = i.escalation_level),
|
||||
i.triggered_at,
|
||||
i.acknowledged_by, i.acknowledged_at, ack.username,
|
||||
i.assigned_to, asg.username, i.snoozed_until,
|
||||
i.resolved_at, i.resolution_source, i.archived_at
|
||||
FROM incidents i
|
||||
JOIN teams t ON t.id = i.team_id
|
||||
LEFT JOIN users ack ON ack.id = i.acknowledged_by
|
||||
LEFT JOIN users asg ON asg.id = i.assigned_to`
|
||||
|
||||
@@ -64,10 +76,11 @@ func scanIncident(s scanner) (models.Incident, error) {
|
||||
var i models.Incident
|
||||
var groupLabelsJSON string
|
||||
var triggeredAt int64
|
||||
var ackAt, snoozedUntil, resolvedAt, archivedAt *int64
|
||||
var ackAt, snoozedUntil, resolvedAt, archivedAt, escalationDue *int64
|
||||
|
||||
if err := s.Scan(
|
||||
&i.ID, &i.GroupKey, &i.Title, &groupLabelsJSON, &i.Status, &i.Severity,
|
||||
&i.ID, &i.TeamID, &i.TeamName, &i.GroupKey, &i.Title, &groupLabelsJSON, &i.Status, &i.Severity,
|
||||
&i.EscalationLevel, &escalationDue,
|
||||
&triggeredAt,
|
||||
&i.AcknowledgedByID, &ackAt, &i.AcknowledgedByUser,
|
||||
&i.AssignedToID, &i.AssignedToUser, &snoozedUntil,
|
||||
@@ -82,6 +95,7 @@ func scanIncident(s scanner) (models.Incident, error) {
|
||||
i.SnoozedUntil = unixPtr(snoozedUntil)
|
||||
i.ResolvedAt = unixPtr(resolvedAt)
|
||||
i.ArchivedAt = unixPtr(archivedAt)
|
||||
i.EscalationDueAt = unixPtr(escalationDue)
|
||||
return i, nil
|
||||
}
|
||||
|
||||
@@ -113,12 +127,17 @@ func todayUTC() string {
|
||||
return time.Now().UTC().Format("2006-01-02")
|
||||
}
|
||||
|
||||
// currentOnCall returns today's on-call user, or nil when nobody is scheduled.
|
||||
// A missing schedule entry is not an error — incidents just open unassigned.
|
||||
func currentOnCall(ctx context.Context, q querier) (*int64, error) {
|
||||
// currentOnCall returns a team's on-call user for today, or nil when nobody is
|
||||
// scheduled. A missing schedule entry is not an error — incidents just open
|
||||
// unassigned.
|
||||
//
|
||||
// Per team: each team keeps its own rota, so two teams can have two different
|
||||
// people on call on the same day, which was the point of scoping the schedule.
|
||||
func currentOnCall(ctx context.Context, q querier, teamID int64) (*int64, error) {
|
||||
var userID int64
|
||||
err := q.QueryRowContext(ctx,
|
||||
"SELECT user_id FROM schedule_entries WHERE date = $1", todayUTC()).Scan(&userID)
|
||||
"SELECT user_id FROM schedule_entries WHERE team_id = $1 AND date = $2",
|
||||
teamID, todayUTC()).Scan(&userID)
|
||||
if err == sql.ErrNoRows {
|
||||
return nil, nil
|
||||
}
|
||||
@@ -229,6 +248,9 @@ func resolveIfSettled(ctx context.Context, q querier, incidentID int64) (bool, e
|
||||
if n == 0 {
|
||||
return false, nil
|
||||
}
|
||||
if err := stopEscalation(ctx, q, incidentID); err != nil {
|
||||
return false, err
|
||||
}
|
||||
if err := logEvent(ctx, q, incidentID, evResolved, nil, nil, nil); err != nil {
|
||||
return false, err
|
||||
}
|
||||
@@ -239,14 +261,17 @@ func resolveIfSettled(ctx context.Context, q querier, incidentID int64) (bool, e
|
||||
}
|
||||
|
||||
// acknowledgeIncident records that userID has picked an incident up, and reports
|
||||
// whether it changed anything — an already-resolved incident is left alone.
|
||||
// Shared by the authenticated handler and the Acknowledge button in a push
|
||||
// notification, so both write the same state and the same timeline entry.
|
||||
// whether it changed anything — an already-resolved or already-acknowledged
|
||||
// incident is left alone, so a second acknowledge (a retried request, or a
|
||||
// stale push notification tapped after the web UI already acked it) is a
|
||||
// no-op rather than a second "acknowledged" timeline entry. Shared by the
|
||||
// authenticated handler and the Acknowledge button in a push notification,
|
||||
// so both write the same state and the same timeline entry.
|
||||
func acknowledgeIncident(ctx context.Context, q querier, incidentID, userID int64) (bool, error) {
|
||||
res, err := q.ExecContext(ctx, `
|
||||
UPDATE incidents
|
||||
SET status = 'acknowledged', acknowledged_by = $1, acknowledged_at = $2
|
||||
WHERE id = $3 AND resolved_at IS NULL`,
|
||||
WHERE id = $3 AND status = 'triggered'`,
|
||||
userID, time.Now().Unix(), incidentID)
|
||||
if err != nil {
|
||||
return false, err
|
||||
@@ -254,6 +279,10 @@ func acknowledgeIncident(ctx context.Context, q querier, incidentID, userID int6
|
||||
if n, _ := res.RowsAffected(); n == 0 {
|
||||
return false, nil
|
||||
}
|
||||
// Somebody has it: stop waking anybody else.
|
||||
if err := stopEscalation(ctx, q, incidentID); err != nil {
|
||||
return false, err
|
||||
}
|
||||
return true, logEvent(ctx, q, incidentID, evAcknowledged, &userID, nil, nil)
|
||||
}
|
||||
|
||||
@@ -273,6 +302,34 @@ func openIncidentForAlert(ctx context.Context, q querier, alertID int64) (int64,
|
||||
return id, err
|
||||
}
|
||||
|
||||
// volatileLabels say where a problem ran this time, not what the problem is, so
|
||||
// they stay out of the signature. Migration 008's backfill lists the same set.
|
||||
var volatileLabels = map[string]bool{
|
||||
"instance": true, "pod": true, "pod_name": true, "pod_ip": true,
|
||||
"container": true, "container_name": true, "endpoint": true,
|
||||
}
|
||||
|
||||
// incidentSignature identifies "the same problem" across incidents: the alert
|
||||
// name plus the stable group labels, sorted. Incidents in one team with equal
|
||||
// signatures are what the similar-incidents lookup returns. title stands in for
|
||||
// the name when the payload carried no alertname (groupless and dead man's
|
||||
// switch incidents).
|
||||
func incidentSignature(groupLabels map[string]string, title string) string {
|
||||
name := groupLabels["alertname"]
|
||||
if name == "" {
|
||||
name = title
|
||||
}
|
||||
rest := make([]string, 0, len(groupLabels))
|
||||
for k, v := range groupLabels {
|
||||
if k == "alertname" || volatileLabels[k] {
|
||||
continue
|
||||
}
|
||||
rest = append(rest, k+"="+v)
|
||||
}
|
||||
sort.Strings(rest)
|
||||
return name + "|" + strings.Join(rest, ",")
|
||||
}
|
||||
|
||||
// incidentTitle renders a human-readable title from Alertmanager's groupLabels,
|
||||
// leading with the alert name and appending whatever else the operator grouped
|
||||
// by. Falls back to the alert's own name when the payload carried no groupLabels.
|
||||
|
||||
+80
-21
@@ -3,6 +3,7 @@ package api
|
||||
import (
|
||||
"database/sql"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"strings"
|
||||
@@ -19,6 +20,15 @@ func handleListIncidents(db *sql.DB) http.HandlerFunc {
|
||||
where := []string{}
|
||||
args := &sqlArgs{}
|
||||
|
||||
// The combined queue: every team the caller belongs to, in one list. A
|
||||
// caller in no team sees an empty queue rather than everybody's.
|
||||
where = append(where, "i.team_id = ANY("+args.add(callerTeamIDs(r.Context()))+")")
|
||||
if team := q.Get("team_id"); team != "" {
|
||||
if n, err := strconv.ParseInt(team, 10, 64); err == nil {
|
||||
where = append(where, "i.team_id = "+args.add(n))
|
||||
}
|
||||
}
|
||||
|
||||
// Without an explicit status the queue shows open work, which is what an
|
||||
// on-call person opens the tool to see.
|
||||
if status := q.Get("status"); status != "" {
|
||||
@@ -96,7 +106,7 @@ func handleListIncidents(db *sql.DB) http.HandlerFunc {
|
||||
|
||||
func handleGetIncident(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, ok := incidentIDParam(w, r)
|
||||
id, ok := incidentIDParam(w, r, db)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
@@ -119,7 +129,7 @@ func handleGetIncident(db *sql.DB) http.HandlerFunc {
|
||||
|
||||
func handleIncidentAlerts(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, ok := incidentIDParam(w, r)
|
||||
id, ok := incidentIDParam(w, r, db)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
@@ -137,7 +147,7 @@ func handleIncidentAlerts(db *sql.DB) http.HandlerFunc {
|
||||
|
||||
func handleIncidentTimeline(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, ok := incidentIDParam(w, r)
|
||||
id, ok := incidentIDParam(w, r, db)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
@@ -176,7 +186,7 @@ func handleIncidentTimeline(db *sql.DB) http.HandlerFunc {
|
||||
|
||||
func handleIncidentAcknowledge(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, ok := incidentIDParam(w, r)
|
||||
id, ok := incidentIDParam(w, r, db)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
@@ -187,10 +197,20 @@ func handleIncidentAcknowledge(db *sql.DB) http.HandlerFunc {
|
||||
return
|
||||
}
|
||||
if !acked {
|
||||
if !incidentExists(w, r, db, id) {
|
||||
// incidentIDParam above already confirmed the incident exists, so this
|
||||
// is either resolved, or already acknowledged — the latter is now a
|
||||
// no-op rather than an error, since the caller's desired state
|
||||
// (acknowledged) already holds.
|
||||
inc, err := fetchIncident(r.Context(), db, id)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusConflict, errResp("incident is resolved"))
|
||||
if inc.Status == "resolved" {
|
||||
respond(w, http.StatusConflict, errResp("incident is resolved"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusOK, inc)
|
||||
return
|
||||
}
|
||||
respondIncident(w, r, db, id)
|
||||
@@ -199,7 +219,7 @@ func handleIncidentAcknowledge(db *sql.DB) http.HandlerFunc {
|
||||
|
||||
func handleIncidentUnacknowledge(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, ok := incidentIDParam(w, r)
|
||||
id, ok := incidentIDParam(w, r, db)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
@@ -223,28 +243,48 @@ func handleIncidentUnacknowledge(db *sql.DB) http.HandlerFunc {
|
||||
// re-send of an alert that never stopped firing. Use snooze for "not now".
|
||||
func handleIncidentResolve(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, ok := incidentIDParam(w, r)
|
||||
id, ok := incidentIDParam(w, r, db)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
user, _ := userFromContext(r.Context())
|
||||
// The body is optional: clients that predate resolution notes send none.
|
||||
var req struct {
|
||||
Resolution string `json:"resolution"`
|
||||
}
|
||||
if err := decodeJSON(r, &req); err != nil && err != io.EOF {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||
return
|
||||
}
|
||||
req.Resolution = strings.TrimSpace(req.Resolution)
|
||||
if !updateOpenIncident(w, r, db, id,
|
||||
`UPDATE incidents SET status = 'resolved', resolved_at = $1, resolution_source = $2
|
||||
WHERE id = $3 AND resolved_at IS NULL`,
|
||||
time.Now().Unix(), incidentResolutionManual, id) {
|
||||
return
|
||||
}
|
||||
// A person closing an incident is the clearest possible "I have this".
|
||||
if err := stopEscalation(r.Context(), db, id); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if err := logEvent(r.Context(), db, id, evResolved, &user.ID, nil, nil); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if req.Resolution != "" {
|
||||
if err := logEvent(r.Context(), db, id, evResolutionNote, &user.ID, nil, &req.Resolution); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
}
|
||||
respondIncident(w, r, db, id)
|
||||
}
|
||||
}
|
||||
|
||||
func handleIncidentAssign(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, ok := incidentIDParam(w, r)
|
||||
id, ok := incidentIDParam(w, r, db)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
@@ -285,7 +325,7 @@ func handleIncidentAssign(db *sql.DB) http.HandlerFunc {
|
||||
// {"duration": "2h"}.
|
||||
func handleIncidentSnooze(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, ok := incidentIDParam(w, r)
|
||||
id, ok := incidentIDParam(w, r, db)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
@@ -340,7 +380,7 @@ func handleIncidentSnooze(db *sql.DB) http.HandlerFunc {
|
||||
|
||||
func handleIncidentUnsnooze(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, ok := incidentIDParam(w, r)
|
||||
id, ok := incidentIDParam(w, r, db)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
@@ -359,7 +399,7 @@ func handleIncidentUnsnooze(db *sql.DB) http.HandlerFunc {
|
||||
|
||||
func handleIncidentArchive(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, ok := incidentIDParam(w, r)
|
||||
id, ok := incidentIDParam(w, r, db)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
@@ -379,7 +419,7 @@ func handleIncidentArchive(db *sql.DB) http.HandlerFunc {
|
||||
|
||||
func handleIncidentUnarchive(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, ok := incidentIDParam(w, r)
|
||||
id, ok := incidentIDParam(w, r, db)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
@@ -401,17 +441,23 @@ func handleIncidentUnarchive(db *sql.DB) http.HandlerFunc {
|
||||
// single query renders the whole story of an incident in order.
|
||||
func handleCreateNote(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, ok := incidentIDParam(w, r)
|
||||
id, ok := incidentIDParam(w, r, db)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
var req struct {
|
||||
Content string `json:"content"`
|
||||
// Pinned files the note as the resolution note: what fixed it.
|
||||
Pinned bool `json:"pinned"`
|
||||
}
|
||||
if err := decodeJSON(r, &req); err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||
return
|
||||
}
|
||||
noteType := evNote
|
||||
if req.Pinned {
|
||||
noteType = evResolutionNote
|
||||
}
|
||||
if req.Content == "" {
|
||||
respond(w, http.StatusBadRequest, errResp("content is required"))
|
||||
return
|
||||
@@ -426,7 +472,7 @@ func handleCreateNote(db *sql.DB) http.HandlerFunc {
|
||||
err := db.QueryRowContext(r.Context(), `
|
||||
INSERT INTO incident_events (incident_id, type, user_id, detail, created_at)
|
||||
VALUES ($1, $2, $3, $4, $5)
|
||||
RETURNING id`, id, evNote, user.ID, req.Content, now.Unix()).Scan(&eventID)
|
||||
RETURNING id`, id, noteType, user.ID, req.Content, now.Unix()).Scan(&eventID)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
@@ -435,7 +481,7 @@ func handleCreateNote(db *sql.DB) http.HandlerFunc {
|
||||
respond(w, http.StatusCreated, models.IncidentEvent{
|
||||
ID: eventID,
|
||||
IncidentID: id,
|
||||
Type: evNote,
|
||||
Type: noteType,
|
||||
UserID: &user.ID,
|
||||
Username: &user.Username,
|
||||
Detail: &req.Content,
|
||||
@@ -448,7 +494,7 @@ func handleCreateNote(db *sql.DB) http.HandlerFunc {
|
||||
// rest of the timeline is what actually happened, and is not editable.
|
||||
func handleDeleteNote(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, ok := incidentIDParam(w, r)
|
||||
id, ok := incidentIDParam(w, r, db)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
@@ -461,8 +507,8 @@ func handleDeleteNote(db *sql.DB) http.HandlerFunc {
|
||||
user, _ := userFromContext(r.Context())
|
||||
res, err := db.ExecContext(r.Context(), `
|
||||
DELETE FROM incident_events
|
||||
WHERE id = $1 AND incident_id = $2 AND type = $3 AND user_id = $4`,
|
||||
eventID, id, evNote, user.ID)
|
||||
WHERE id = $1 AND incident_id = $2 AND type IN ($3, $4) AND user_id = $5`,
|
||||
eventID, id, evNote, evResolutionNote, user.ID)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
@@ -479,19 +525,32 @@ func handleDeleteNote(db *sql.DB) http.HandlerFunc {
|
||||
// Shared handler plumbing
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func incidentIDParam(w http.ResponseWriter, r *http.Request) (int64, bool) {
|
||||
// incidentIDParam reads {id} from the path AND confirms the incident belongs to
|
||||
// a team the caller is in. Both in one place, deliberately: every incident route
|
||||
// goes through here, so scoping cannot be forgotten by writing a new handler
|
||||
// that only remembers the first half.
|
||||
//
|
||||
// An incident in somebody else's team is reported as not found rather than
|
||||
// forbidden, because "there is an incident 41 you may not see" is itself
|
||||
// something only that team should know.
|
||||
func incidentIDParam(w http.ResponseWriter, r *http.Request, db *sql.DB) (int64, bool) {
|
||||
id, err := strconv.ParseInt(chi.URLParam(r, "id"), 10, 64)
|
||||
if err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid incident id"))
|
||||
return 0, false
|
||||
}
|
||||
if !incidentExists(w, r, db, id) {
|
||||
return 0, false
|
||||
}
|
||||
return id, true
|
||||
}
|
||||
|
||||
// incidentExists reports whether the incident is one the caller may see at all.
|
||||
func incidentExists(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64) bool {
|
||||
var exists int
|
||||
if err := db.QueryRowContext(r.Context(),
|
||||
"SELECT 1 FROM incidents WHERE id = $1", id).Scan(&exists); err != nil {
|
||||
"SELECT 1 FROM incidents WHERE id = $1 AND team_id = ANY($2)",
|
||||
id, callerTeamIDs(r.Context())).Scan(&exists); err != nil {
|
||||
respond(w, http.StatusNotFound, errResp("incident not found"))
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -338,6 +338,48 @@ func TestIncident_Acknowledge(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// A second acknowledge — a retried request, or a stale push notification
|
||||
// tapped after the web UI already acked it — must be a no-op: same state,
|
||||
// no second "acknowledged" timeline entry. Regression test for the bug
|
||||
// described in issue #26 ("two acknowledged entries look like a bug").
|
||||
func TestIncident_AcknowledgeTwiceIsIdempotent(t *testing.T) {
|
||||
s := newTS(t)
|
||||
postWebhook(t, s, []map[string]any{
|
||||
amAlert("fp-ack2", "Y", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||
})
|
||||
|
||||
resp := s.req(t, http.MethodPost, "/api/incidents/1/acknowledge", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("first acknowledge returned %d", resp.StatusCode)
|
||||
}
|
||||
var first map[string]any
|
||||
decode(t, resp, &first)
|
||||
|
||||
resp = s.req(t, http.MethodPost, "/api/incidents/1/acknowledge", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("second acknowledge returned %d, want 200 (idempotent)", resp.StatusCode)
|
||||
}
|
||||
var second map[string]any
|
||||
decode(t, resp, &second)
|
||||
if second["status"] != "acknowledged" {
|
||||
t.Errorf("expected status still acknowledged, got %v", second["status"])
|
||||
}
|
||||
if second["acknowledged_by"] != first["acknowledged_by"] {
|
||||
t.Errorf("expected the same acknowledged_by, got %v then %v", first["acknowledged_by"], second["acknowledged_by"])
|
||||
}
|
||||
|
||||
types := eventTypes(timeline(t, s, 1))
|
||||
n := 0
|
||||
for _, ty := range types {
|
||||
if ty == "acknowledged" {
|
||||
n++
|
||||
}
|
||||
}
|
||||
if n != 1 {
|
||||
t.Errorf("expected exactly one acknowledged event, got %d in %v", n, types)
|
||||
}
|
||||
}
|
||||
|
||||
func TestIncident_ManualResolveIsTerminal(t *testing.T) {
|
||||
s := newTS(t)
|
||||
postWebhook(t, s, []map[string]any{
|
||||
@@ -425,7 +467,7 @@ func TestIncident_AutoAssignedToCurrentOnCall(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
today := time.Now().UTC().Format("2006-01-02")
|
||||
resp := s.req(t, http.MethodPost, "/api/schedule",
|
||||
resp := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/schedule",
|
||||
map[string]any{"user_id": 1, "dates": []string{today}})
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("schedule assignment returned %d", resp.StatusCode)
|
||||
@@ -609,7 +651,7 @@ func TestSweeper_ArchivesResolvedIncidents(t *testing.T) {
|
||||
|
||||
s.exec(t, "UPDATE incidents SET resolved_at = $1 WHERE id = 1",
|
||||
time.Now().Add(-30*24*time.Hour).Unix())
|
||||
api.Sweep(context.Background(), s.db, 7*24*time.Hour, 6*time.Hour, s.deadman, s.notify)
|
||||
api.Sweep(context.Background(), s.db, 7*24*time.Hour, 6*time.Hour, s.notify)
|
||||
|
||||
if inc := getIncident(t, s, 1); inc["archived_at"] == nil {
|
||||
t.Error("expected the sweeper to archive a long-resolved incident")
|
||||
|
||||
@@ -0,0 +1,154 @@
|
||||
package api_test
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||
)
|
||||
|
||||
func testNotify() api.NotifyConfig {
|
||||
return api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute}
|
||||
}
|
||||
|
||||
type memberView struct {
|
||||
Username string `json:"username"`
|
||||
Role string `json:"role"`
|
||||
Status string `json:"status"`
|
||||
OnCall bool `json:"on_call"`
|
||||
NextShift *string `json:"next_shift"`
|
||||
Pageable bool `json:"pageable"`
|
||||
Problem string `json:"problem"`
|
||||
LastActiveAt *string `json:"last_active_at"`
|
||||
}
|
||||
|
||||
func readMembers(t *testing.T, s *ts) map[string]memberView {
|
||||
t.Helper()
|
||||
var list []memberView
|
||||
decode(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/members", nil), &list)
|
||||
out := map[string]memberView{}
|
||||
for _, m := range list {
|
||||
out[m.Username] = m
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// The list says who is on call, who could not be woken, and who is merely
|
||||
// there — and an on-call person who cannot be paged is the red one.
|
||||
func TestMembers_StatusReflectsRotaAndPageability(t *testing.T) {
|
||||
s, _ := notifyTS(t, testNotify()) // admin is on call today, with a topic
|
||||
teamUser(t, s, "reachable", "terdut-reachable")
|
||||
silent := teamUser(t, s, "silent", "terdut-silent")
|
||||
s.exec(t, "UPDATE users SET ntfy_topic = NULL WHERE id = $1", silent)
|
||||
|
||||
got := readMembers(t, s)
|
||||
if m := got["admin"]; m.Status != "oncall" || !m.OnCall || !m.Pageable {
|
||||
t.Errorf("the person on call should read on call, got %+v", m)
|
||||
}
|
||||
if m := got["reachable"]; m.Status != "reachable" || m.OnCall {
|
||||
t.Errorf("a member with a topic who is off the rota is reachable, got %+v", m)
|
||||
}
|
||||
if m := got["silent"]; m.Status != "unpageable" || m.Problem != "has no ntfy topic" {
|
||||
t.Errorf("no topic means they cannot be paged, got %+v", m)
|
||||
}
|
||||
|
||||
// Being on call does not rescue an account that cannot be woken.
|
||||
s.exec(t, "UPDATE users SET ntfy_topic = NULL WHERE username = 'admin'")
|
||||
if m := readMembers(t, s)["admin"]; m.Status != "unpageable" || !m.OnCall {
|
||||
t.Errorf("an on-call person with no topic is the red case, got %+v", m)
|
||||
}
|
||||
|
||||
s.exec(t, "UPDATE users SET disabled_at = 1 WHERE id = $1", silent)
|
||||
if m := readMembers(t, s)["silent"]; m.Problem != "account is disabled" {
|
||||
t.Errorf("a disabled account should say so, got %+v", m)
|
||||
}
|
||||
}
|
||||
|
||||
// The next shift is the next day after today, not today itself.
|
||||
func TestMembers_NextShiftIsAfterToday(t *testing.T) {
|
||||
s, _ := notifyTS(t, testNotify())
|
||||
tomorrow := time.Now().UTC().AddDate(0, 0, 3).Format("2006-01-02")
|
||||
resp := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/schedule",
|
||||
map[string]any{"user_id": 1, "dates": []string{tomorrow}})
|
||||
resp.Body.Close()
|
||||
|
||||
m := readMembers(t, s)["admin"]
|
||||
if !m.OnCall || m.NextShift == nil || *m.NextShift != tomorrow {
|
||||
t.Errorf("want on call today with the next shift on %s, got %+v", tomorrow, m)
|
||||
}
|
||||
teamUser(t, s, "idle", "terdut-idle")
|
||||
if m := readMembers(t, s)["idle"]; m.NextShift != nil {
|
||||
t.Errorf("somebody not on the rota has no next shift, got %v", *m.NextShift)
|
||||
}
|
||||
}
|
||||
|
||||
// Last active is the newer of a session and an API key, and absent when neither
|
||||
// has ever been used.
|
||||
func TestMembers_LastActive(t *testing.T) {
|
||||
s, _ := notifyTS(t, testNotify())
|
||||
idle := teamUser(t, s, "idle", "terdut-idle")
|
||||
|
||||
if m := readMembers(t, s)["idle"]; m.LastActiveAt != nil {
|
||||
t.Errorf("nobody has used idle's account, got %v", *m.LastActiveAt)
|
||||
}
|
||||
|
||||
old := time.Now().Add(-48 * time.Hour).Unix()
|
||||
s.exec(t, `INSERT INTO api_keys (user_id, key_hash, name, last_used_at) VALUES ($1, 'h1', 'k', $2)`, idle, old)
|
||||
s.exec(t, `INSERT INTO sessions (token_hash, user_id, created_at, last_seen_at, expires_at)
|
||||
VALUES ('h2', $1, $2, $3, $4)`, idle, old, old+3600, time.Now().Add(time.Hour).Unix())
|
||||
|
||||
m := readMembers(t, s)["idle"]
|
||||
if m.LastActiveAt == nil {
|
||||
t.Fatal("expected a last active time")
|
||||
}
|
||||
got, _ := time.Parse(time.RFC3339, *m.LastActiveAt)
|
||||
if got.Unix() != old+3600 {
|
||||
t.Errorf("last active should be the newer session (%d), got %d", old+3600, got.Unix())
|
||||
}
|
||||
}
|
||||
|
||||
// The last owner can be neither removed nor demoted; with another owner in
|
||||
// place, both are fine.
|
||||
func TestMembers_LastOwnerIsProtected(t *testing.T) {
|
||||
s, _ := notifyTS(t, testNotify())
|
||||
tm := newTeam(t, s, "red")
|
||||
base := "/api/teams/" + id64(tm.id) + "/members"
|
||||
|
||||
// Creating a team makes the creator an owner too; step the admin out so
|
||||
// "red-user" is the only one left.
|
||||
resp := s.req(t, http.MethodDelete, base+"/1", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Fatalf("removing the creator: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
var members []map[string]any
|
||||
decode(t, tm.call(http.MethodGet, base, nil), &members)
|
||||
var owner int64
|
||||
for _, m := range members {
|
||||
if m["username"] == "red-user" {
|
||||
owner = int64(m["user_id"].(float64))
|
||||
}
|
||||
}
|
||||
|
||||
resp = tm.call(http.MethodPost, base, map[string]any{"user_id": owner, "role": "member"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusConflict {
|
||||
t.Errorf("demoting the last owner: expected 409, got %d", resp.StatusCode)
|
||||
}
|
||||
resp = tm.call(http.MethodDelete, base+"/"+id64(owner), nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusConflict {
|
||||
t.Errorf("removing the last owner: expected 409, got %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
// A second owner frees the first to step down.
|
||||
resp = s.req(t, http.MethodPost, base, map[string]any{"user_id": 1, "role": "owner"})
|
||||
resp.Body.Close()
|
||||
resp = tm.call(http.MethodPost, base, map[string]any{"user_id": owner, "role": "member"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Errorf("demoting one of two owners: expected 204, got %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
+257
-12
@@ -9,13 +9,19 @@ import (
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||
)
|
||||
|
||||
type contextKey string
|
||||
|
||||
const (
|
||||
ctxUser contextKey = "user"
|
||||
// ctxCaller holds the one Caller (see caller.go) every authorization
|
||||
// predicate in this package reads from — a human and a service account
|
||||
// used to be two parallel, un-unified context keys (ctxUser/ctxTeams vs.
|
||||
// ctxServiceAccount); this is why that was a mistake, not a smaller
|
||||
// version of the same idea.
|
||||
ctxCaller contextKey = "caller"
|
||||
ctxSession contextKey = "session"
|
||||
)
|
||||
|
||||
@@ -38,12 +44,19 @@ func AuthMiddleware(db *sql.DB) func(http.Handler) http.Handler {
|
||||
respond(w, http.StatusUnauthorized, errResp("unauthorized"))
|
||||
return
|
||||
}
|
||||
userID, ok := apiKeyUser(r.Context(), db, token)
|
||||
if !ok {
|
||||
respond(w, http.StatusUnauthorized, errResp("unauthorized"))
|
||||
if userID, ok := apiKeyUser(r.Context(), db, token); ok {
|
||||
serveAs(w, r, next, db, userID, 0)
|
||||
return
|
||||
}
|
||||
serveAs(w, r, next, db, userID, 0)
|
||||
// Tried second, not first: a user API key is the common case,
|
||||
// and a service-account key is visibly prefixed (tdsa_) so this
|
||||
// second lookup is rarely reached on a request that was going
|
||||
// to fail anyway.
|
||||
if sa, ok := serviceAccountFor(r.Context(), db, token); ok {
|
||||
serveAsServiceAccount(w, r, next, sa)
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusUnauthorized, errResp("unauthorized"))
|
||||
return
|
||||
}
|
||||
|
||||
@@ -66,6 +79,39 @@ func AuthMiddleware(db *sql.DB) func(http.Handler) http.Handler {
|
||||
}
|
||||
}
|
||||
|
||||
// AdminOnly rejects a caller who is not a system administrator. It runs inside
|
||||
// AuthMiddleware's group, so by the time it sees a request the caller is known.
|
||||
//
|
||||
// 403 and not 404: the route exists and the caller is authenticated, they are
|
||||
// simply not allowed. Hiding the endpoint would buy nothing — every one of them
|
||||
// is in the README.
|
||||
func AdminOnly(next http.Handler) http.Handler {
|
||||
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
caller, ok := userFromContext(r.Context())
|
||||
if !ok || !caller.IsAdmin {
|
||||
respond(w, http.StatusForbidden, errResp("administrator access required"))
|
||||
return
|
||||
}
|
||||
next.ServeHTTP(w, r)
|
||||
})
|
||||
}
|
||||
|
||||
// requireSelfOrAdmin guards the endpoints that are self-service for your own
|
||||
// account and administration for anybody else's: your password, your ntfy
|
||||
// topic, your API keys. Reports whether the request may proceed, and answers it
|
||||
// if not.
|
||||
//
|
||||
// An API key is not an escalation: it carries exactly the rights of the user it
|
||||
// belongs to, so minting your own is no more than signing in again.
|
||||
func requireSelfOrAdmin(w http.ResponseWriter, r *http.Request, targetID int64) bool {
|
||||
caller, ok := userFromContext(r.Context())
|
||||
if !ok || (caller.ID != targetID && !caller.IsAdmin) {
|
||||
respond(w, http.StatusForbidden, errResp("administrator access required"))
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// apiKeyUser resolves an API key to its user and stamps its last use.
|
||||
func apiKeyUser(ctx context.Context, db *sql.DB, token string) (int64, bool) {
|
||||
var keyID, userID int64
|
||||
@@ -98,8 +144,13 @@ func sessionUser(ctx context.Context, db *sql.DB, token string) (sessionID, user
|
||||
}
|
||||
|
||||
if now.Sub(time.Unix(lastSeen, 0)) > sessionTouchEvery {
|
||||
db.ExecContext(ctx,
|
||||
"UPDATE sessions SET last_seen_at = $1, expires_at = $2 WHERE id = $3",
|
||||
// LEAST keeps a capped session (a single sign-on login) from sliding
|
||||
// past its ceiling; with no ceiling COALESCE makes it the plain slide.
|
||||
db.ExecContext(ctx, `
|
||||
UPDATE sessions
|
||||
SET last_seen_at = $1,
|
||||
expires_at = LEAST($2::bigint, COALESCE(max_expires_at, $2::bigint))
|
||||
WHERE id = $3`,
|
||||
now.Unix(), now.Add(sessionTTL).Unix(), sessionID)
|
||||
}
|
||||
return sessionID, userID, true
|
||||
@@ -110,15 +161,28 @@ func sessionUser(ctx context.Context, db *sql.DB, token string) (sessionID, user
|
||||
func serveAs(w http.ResponseWriter, r *http.Request, next http.Handler, db *sql.DB, userID, sessionID int64) {
|
||||
var u models.User
|
||||
var createdUnix int64
|
||||
// disabled_at IS NULL is part of the lookup rather than a check afterwards:
|
||||
// a disabled account is one that cannot authenticate, by either credential,
|
||||
// and the way to be sure of that is for there to be no path where the row
|
||||
// is loaded and the flag is then forgotten.
|
||||
if err := db.QueryRowContext(r.Context(),
|
||||
"SELECT id, username, email, created_at FROM users WHERE id = $1", userID,
|
||||
).Scan(&u.ID, &u.Username, &u.Email, &createdUnix); err != nil {
|
||||
"SELECT id, username, email, created_at, is_admin FROM users WHERE id = $1 AND disabled_at IS NULL", userID,
|
||||
).Scan(&u.ID, &u.Username, &u.Email, &createdUnix, &u.IsAdmin); err != nil {
|
||||
respond(w, http.StatusUnauthorized, errResp("unauthorized"))
|
||||
return
|
||||
}
|
||||
u.CreatedAt = time.Unix(createdUnix, 0).UTC()
|
||||
|
||||
ctx := context.WithValue(r.Context(), ctxUser, u)
|
||||
// Every scoped query needs the caller's teams, so they are loaded once here
|
||||
// rather than per handler. One extra round trip per request, against a
|
||||
// table with one row per membership.
|
||||
teams, err := callerMemberships(r.Context(), db, userID)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
|
||||
ctx := context.WithValue(r.Context(), ctxCaller, Caller{user: &u, memberships: teams})
|
||||
if sessionID != 0 {
|
||||
ctx = context.WithValue(ctx, ctxSession, sessionID)
|
||||
}
|
||||
@@ -130,9 +194,190 @@ func hashToken(token string) string {
|
||||
return hex.EncodeToString(h[:])
|
||||
}
|
||||
|
||||
// userFromContext is a thin compatibility wrapper over Caller.AsHuman(), so
|
||||
// every call site written before the Caller abstraction (alerts.go,
|
||||
// incidents.go, schedule.go, stats.go, and more) needs no change and keeps
|
||||
// its exact existing behavior.
|
||||
func userFromContext(ctx context.Context) (models.User, bool) {
|
||||
u, ok := ctx.Value(ctxUser).(models.User)
|
||||
return u, ok
|
||||
c, _ := callerFromContext(ctx)
|
||||
return c.AsHuman()
|
||||
}
|
||||
|
||||
// serviceAccountPrincipal is a service account as resolved from its key:
|
||||
// enough to authorize requests, never the key itself.
|
||||
type serviceAccountPrincipal struct {
|
||||
id int64
|
||||
name string
|
||||
scope string
|
||||
teamID int64 // meaningless (zero) for instance scope
|
||||
}
|
||||
|
||||
// serviceAccountFor resolves a service-account key to its account and stamps
|
||||
// its last use, the same shape apiKeyUser has for a user's own key.
|
||||
func serviceAccountFor(ctx context.Context, db *sql.DB, token string) (serviceAccountPrincipal, bool) {
|
||||
var sa serviceAccountPrincipal
|
||||
var keyID int64
|
||||
var teamID sql.NullInt64
|
||||
err := db.QueryRowContext(ctx, `
|
||||
SELECT k.id, a.id, a.name, a.scope, a.team_id
|
||||
FROM service_account_keys k
|
||||
JOIN service_accounts a ON a.id = k.service_account_id
|
||||
WHERE k.key_hash = $1`, hashToken(token),
|
||||
).Scan(&keyID, &sa.id, &sa.name, &sa.scope, &teamID)
|
||||
if err != nil {
|
||||
return serviceAccountPrincipal{}, false
|
||||
}
|
||||
if teamID.Valid {
|
||||
sa.teamID = teamID.Int64
|
||||
}
|
||||
|
||||
// best-effort; don't fail the request if this update fails
|
||||
db.ExecContext(ctx,
|
||||
"UPDATE service_account_keys SET last_used_at = $1 WHERE id = $2",
|
||||
time.Now().Unix(), keyID)
|
||||
return sa, true
|
||||
}
|
||||
|
||||
// serveAsServiceAccount hands the request on with a service account's
|
||||
// identity in context. A team-scoped account gets a single synthetic
|
||||
// membership — owner of its own team, nothing else — which is what makes it
|
||||
// satisfy requireTeamMember/requireTeamOwner exactly as a real owner would,
|
||||
// without teaching either function about a second kind of caller. An
|
||||
// instance-scoped account gets no memberships at all: it acts on teams by id,
|
||||
// not by belonging to one.
|
||||
//
|
||||
// No CSRF check, for the same reason an API key needs none: a service-account
|
||||
// key is only ever set by the client that holds it, never attached by a
|
||||
// browser to a request another site makes.
|
||||
func serveAsServiceAccount(w http.ResponseWriter, r *http.Request, next http.Handler, sa serviceAccountPrincipal) {
|
||||
var memberships []membership
|
||||
if sa.scope == models.ServiceAccountScopeTeam {
|
||||
memberships = []membership{{teamID: sa.teamID, role: models.RoleOwner}}
|
||||
}
|
||||
ctx := context.WithValue(r.Context(), ctxCaller, Caller{sa: &sa, memberships: memberships})
|
||||
next.ServeHTTP(w, r.WithContext(ctx))
|
||||
}
|
||||
|
||||
// isInstanceServiceAccount is a thin compatibility wrapper over
|
||||
// Caller.IsInstanceServiceAccount(), for call sites outside this package's
|
||||
// core predicates (handleCreateTeam, handleCreateServiceAccount) that
|
||||
// needed this exact, narrow check before the Caller abstraction existed.
|
||||
func isInstanceServiceAccount(ctx context.Context) bool {
|
||||
c, _ := callerFromContext(ctx)
|
||||
return c.IsInstanceServiceAccount()
|
||||
}
|
||||
|
||||
// operatorReason marks a write that operator mode refused as such, distinct
|
||||
// from every other 403 this server returns, so a client — the web UI or
|
||||
// terdut-tui — can tell "you may not" from "this is managed elsewhere" and
|
||||
// show the right message instead of a bare "forbidden".
|
||||
const operatorReason = "operator_managed"
|
||||
|
||||
// OperatorModeBlock refuses a human write (session or a user's own API key)
|
||||
// on a route it wraps, while letting a service account through. That is the
|
||||
// whole point of operator mode: automation holding a service-account key
|
||||
// (terdut-operator, most likely) keeps reconciling these resources, and a
|
||||
// person in the web UI or terdut-tui gets a clear "edit this through your
|
||||
// GitOps source instead" rather than a write that the next resync would only
|
||||
// undo.
|
||||
//
|
||||
// Checked after AuthMiddleware, the same way AdminOnly is: by the time a
|
||||
// request reaches here the caller is already known to be a service account
|
||||
// or not. A router that never enables operator mode pays nothing for this —
|
||||
// it hands back next unchanged rather than wrapping it in a check that would
|
||||
// always pass.
|
||||
func OperatorModeBlock(cfg config.Config) func(http.Handler) http.Handler {
|
||||
return func(next http.Handler) http.Handler {
|
||||
if !cfg.OperatorMode {
|
||||
return next
|
||||
}
|
||||
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
caller, _ := callerFromContext(r.Context())
|
||||
if _, ok := caller.ServiceAccountID(); ok {
|
||||
next.ServeHTTP(w, r)
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusForbidden, map[string]string{
|
||||
"error": "this server is in operator mode; edit this through your GitOps source instead of the web UI or API",
|
||||
"reason": operatorReason,
|
||||
})
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// membership is the caller's role in one team.
|
||||
type membership struct {
|
||||
teamID int64
|
||||
role string
|
||||
}
|
||||
|
||||
func callerMemberships(ctx context.Context, db *sql.DB, userID int64) ([]membership, error) {
|
||||
rows, err := db.QueryContext(ctx,
|
||||
"SELECT team_id, role FROM team_members WHERE user_id = $1 ORDER BY team_id", userID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
var out []membership
|
||||
for rows.Next() {
|
||||
var m membership
|
||||
if err := rows.Scan(&m.teamID, &m.role); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, m)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
// callerTeamIDs lists the teams the caller belongs to, for the `team_id = ANY`
|
||||
// filter every list query carries. An admin is NOT implicitly in every team:
|
||||
// administration is about accounts, not about reading other people's incidents,
|
||||
// and an admin who needs to see a team's queue can add themselves to it.
|
||||
func callerTeamIDs(ctx context.Context) []int64 {
|
||||
c, _ := callerFromContext(ctx)
|
||||
return c.TeamIDs()
|
||||
}
|
||||
|
||||
// callerRole reports the caller's role in one team, and whether they are in it
|
||||
// at all.
|
||||
func callerRole(ctx context.Context, teamID int64) (string, bool) {
|
||||
c, _ := callerFromContext(ctx)
|
||||
return c.Role(teamID)
|
||||
}
|
||||
|
||||
// requireTeamMember answers the request and reports false unless the caller
|
||||
// belongs to teamID.
|
||||
//
|
||||
// 404, not 403: whether a team exists is itself something only its members
|
||||
// should learn, and the same reasoning applies to every incident and alert
|
||||
// under it.
|
||||
func requireTeamMember(w http.ResponseWriter, r *http.Request, teamID int64) bool {
|
||||
if _, ok := callerRole(r.Context(), teamID); !ok {
|
||||
respond(w, http.StatusNotFound, errResp("not found"))
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// requireTeamOwner is requireTeamMember for the things only an owner may change:
|
||||
// the schedule, the integrations and who is in the team. A system administrator
|
||||
// passes without being a member, because somebody has to be able to repair a
|
||||
// team whose owner has left.
|
||||
func requireTeamOwner(w http.ResponseWriter, r *http.Request, teamID int64) bool {
|
||||
role, ok := callerRole(r.Context(), teamID)
|
||||
if ok && role == models.RoleOwner {
|
||||
return true
|
||||
}
|
||||
if caller, _ := userFromContext(r.Context()); caller.IsAdmin {
|
||||
return true
|
||||
}
|
||||
if !ok {
|
||||
respond(w, http.StatusNotFound, errResp("not found"))
|
||||
return false
|
||||
}
|
||||
respond(w, http.StatusForbidden, errResp("team owner access required"))
|
||||
return false
|
||||
}
|
||||
|
||||
// sessionFromContext returns the id of the session a request was authenticated
|
||||
|
||||
@@ -38,11 +38,22 @@ const (
|
||||
ackTokenTTL = 24 * time.Hour
|
||||
)
|
||||
|
||||
// notifierLockKey is the Postgres advisory lock the notifier takes for the
|
||||
// duration of each pass, so that running more than one replica does not
|
||||
// deliver (or double-deliver) the same notification from more than one of
|
||||
// them at once. Its value has no meaning beyond being distinct from
|
||||
// archiverLockKey.
|
||||
const notifierLockKey int64 = 7265_0002
|
||||
|
||||
// Notification kinds, recording why a push was sent.
|
||||
const (
|
||||
notifyTriggered = "triggered"
|
||||
notifyReminder = "reminder"
|
||||
notifyResolved = "resolved"
|
||||
|
||||
// notifyEscalated is a page that went out because nobody answered the last
|
||||
// one. Told apart from a reminder because it goes to somebody else.
|
||||
notifyEscalated = "escalated"
|
||||
)
|
||||
|
||||
// Timeline event types the notifier writes, so an incident's history says who
|
||||
@@ -93,6 +104,10 @@ var notifyClient = &http.Client{Timeout: 10 * time.Second}
|
||||
|
||||
// StartNotifier delivers queued notifications until ctx is cancelled, starting
|
||||
// with an immediate pass so a restart flushes whatever the last one left behind.
|
||||
//
|
||||
// Each pass runs under notifierLockKey (see withAdvisoryLock), so that on more
|
||||
// than one replica only whichever instance's tick takes the lock first actually
|
||||
// delivers; the rest skip that tick rather than racing the same pass.
|
||||
func StartNotifier(ctx context.Context, db *sql.DB, cfg NotifyConfig) {
|
||||
if !cfg.enabled() {
|
||||
log.Print("notifier: disabled (no ntfy URL configured)")
|
||||
@@ -103,11 +118,17 @@ func StartNotifier(ctx context.Context, db *sql.DB, cfg NotifyConfig) {
|
||||
ticker := time.NewTicker(notifyInterval)
|
||||
defer ticker.Stop()
|
||||
|
||||
NotifySweep(ctx, db, cfg)
|
||||
sweep := func() {
|
||||
withAdvisoryLock(ctx, db, notifierLockKey, "notifier", func() {
|
||||
NotifySweep(ctx, db, cfg)
|
||||
})
|
||||
}
|
||||
|
||||
sweep()
|
||||
for {
|
||||
select {
|
||||
case <-ticker.C:
|
||||
NotifySweep(ctx, db, cfg)
|
||||
sweep()
|
||||
case <-ctx.Done():
|
||||
return
|
||||
}
|
||||
@@ -120,6 +141,9 @@ func StartNotifier(ctx context.Context, db *sql.DB, cfg NotifyConfig) {
|
||||
// Exported so tests can drive a pass without waiting on the ticker.
|
||||
func NotifySweep(ctx context.Context, db *sql.DB, cfg NotifyConfig) {
|
||||
enqueueReminders(ctx, db, cfg)
|
||||
// Escalation before delivery, so a level that comes due on this tick is
|
||||
// paged on this tick rather than waiting for the next one.
|
||||
escalate(ctx, db, cfg)
|
||||
deliverPending(ctx, db, cfg)
|
||||
}
|
||||
|
||||
@@ -134,7 +158,11 @@ func NotifySweep(ctx context.Context, db *sql.DB, cfg NotifyConfig) {
|
||||
// queued, so an ntfy outage produces a retry backlog rather than a reminder
|
||||
// backlog that all lands at once when it comes back.
|
||||
func enqueueReminders(ctx context.Context, db *sql.DB, cfg NotifyConfig) {
|
||||
if cfg.RepeatEvery <= 0 {
|
||||
// cfg.RepeatEvery is what the server started with; the settings table is
|
||||
// what it runs on. Read per tick, so an administrator lengthening the
|
||||
// interval at 02:00 is obeyed at 02:00 and not at the next restart.
|
||||
repeat := NewSettings(db).Duration(ctx, SettingNotifyRepeat, cfg.RepeatEvery)
|
||||
if repeat <= 0 {
|
||||
return
|
||||
}
|
||||
now := time.Now()
|
||||
@@ -155,8 +183,13 @@ func enqueueReminders(ctx context.Context, db *sql.DB, cfg NotifyConfig) {
|
||||
AND i.resolved_at IS NULL
|
||||
AND i.archived_at IS NULL
|
||||
AND i.status = 'triggered'
|
||||
AND (i.snoozed_until IS NULL OR i.snoozed_until <= $2)`,
|
||||
now.Add(-cfg.RepeatEvery).Unix(), now.Unix())
|
||||
AND (i.snoozed_until IS NULL OR i.snoozed_until <= $2)
|
||||
-- A team with an escalation ladder gets escalation instead. Both
|
||||
-- would mean two pages for one silence, which is how people learn to
|
||||
-- mute a tool.
|
||||
AND NOT EXISTS (
|
||||
SELECT 1 FROM escalation_levels el WHERE el.team_id = i.team_id)`,
|
||||
now.Add(-repeat).Unix(), now.Unix())
|
||||
if err != nil {
|
||||
log.Printf("notifier: find reminders: %v", err)
|
||||
return
|
||||
@@ -314,6 +347,16 @@ func deliver(ctx context.Context, db *sql.DB, cfg NotifyConfig, n outboxRow) err
|
||||
|
||||
msg := renderNotification(inc, n, firing, cfg)
|
||||
|
||||
// The page that opens an incident carries what fixed it last time, so the
|
||||
// person woken up starts from that. Best effort: a failed lookup must not
|
||||
// hold back the page itself.
|
||||
if n.kind == notifyTriggered {
|
||||
if sim, err := similarIncidents(ctx, db, n.incidentID, 1); err == nil && len(sim) > 0 && len(sim[0].ResolutionNotes) > 0 {
|
||||
notes := sim[0].ResolutionNotes
|
||||
msg.Message += "\nLast time: " + shorten(derefString(notes[len(notes)-1].Detail), 160)
|
||||
}
|
||||
}
|
||||
|
||||
// An Acknowledge button needs both a user to attribute the acknowledgement
|
||||
// to and a URL the phone can reach. Minted per delivery, so every push
|
||||
// carries its own short-lived token rather than reusing one.
|
||||
@@ -557,6 +600,17 @@ func plural(n int) string {
|
||||
return "s"
|
||||
}
|
||||
|
||||
// shorten cuts s to at most n runes, marking the cut, and flattens newlines so
|
||||
// a multi-line note stays one line in a push.
|
||||
func shorten(s string, n int) string {
|
||||
s = strings.Join(strings.Fields(s), " ")
|
||||
r := []rune(s)
|
||||
if len(r) <= n {
|
||||
return s
|
||||
}
|
||||
return string(r[:n-1]) + "…"
|
||||
}
|
||||
|
||||
// derefString reads a nullable text column as a plain string.
|
||||
func derefString(s *string) string {
|
||||
if s == nil {
|
||||
|
||||
@@ -66,12 +66,19 @@ func handleNotifyAck(db *sql.DB) http.HandlerFunc {
|
||||
return
|
||||
}
|
||||
if !acked {
|
||||
// The incident closed between the page and the tap. Nothing to do,
|
||||
// and nothing the responder did wrong — report the state, not an error,
|
||||
// so ntfy shows a success toast rather than a failure.
|
||||
// Either the incident closed between the page and the tap, or it was
|
||||
// already acknowledged (e.g. from the web UI, or an earlier tap of
|
||||
// the same button) — either way nothing the responder did wrong, so
|
||||
// report the actual state rather than assuming "resolved", and let
|
||||
// ntfy show a success toast rather than a failure.
|
||||
inc, err := fetchIncident(r.Context(), db, incidentID)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusOK, map[string]any{
|
||||
"incident_id": incidentID,
|
||||
"status": "resolved",
|
||||
"status": inc.Status,
|
||||
})
|
||||
return
|
||||
}
|
||||
|
||||
@@ -69,6 +69,27 @@ func (f *fakeNtfy) messages() []pushed {
|
||||
return append([]pushed(nil), f.got...)
|
||||
}
|
||||
|
||||
// topicsSince lists the topics published to since the last forget, which is how
|
||||
// the escalation tests ask "who did this tick wake".
|
||||
func (f *fakeNtfy) topicsSince(t *testing.T) []string {
|
||||
t.Helper()
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
out := make([]string, 0, len(f.got))
|
||||
for _, m := range f.got {
|
||||
out = append(out, m.Topic)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// forget drops what has been published so far, so the next assertion is about
|
||||
// this tick rather than the whole test.
|
||||
func (f *fakeNtfy) forget() {
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
f.got = nil
|
||||
}
|
||||
|
||||
func (f *fakeNtfy) failWith(status int) {
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
@@ -95,7 +116,7 @@ func notifyTS(t *testing.T, cfg api.NotifyConfig) (*ts, *fakeNtfy) {
|
||||
func putOnCall(t *testing.T, s *ts, userID int) {
|
||||
t.Helper()
|
||||
today := time.Now().UTC().Format("2006-01-02")
|
||||
resp := s.req(t, http.MethodPost, "/api/schedule",
|
||||
resp := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/schedule",
|
||||
map[string]any{"user_id": userID, "dates": []string{today}})
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
@@ -387,6 +408,47 @@ func TestNotify_AckButtonAcknowledgesIncident(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// The ack token isn't single-use (it stays valid for a day, in case the
|
||||
// first tap never reaches the server), so tapping the same notification's
|
||||
// Acknowledge button twice is a real scenario, not just a retried request.
|
||||
// It must report the incident's actual state, not assume "resolved" —
|
||||
// see handleNotifyAck's !acked branch — and must not log a second
|
||||
// "acknowledged" event.
|
||||
func TestNotify_AckButtonTwiceIsIdempotent(t *testing.T) {
|
||||
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com"})
|
||||
|
||||
fireCritical(t, s)
|
||||
s.sweepNotify(t)
|
||||
|
||||
ackURL := f.messages()[0].Actions[0].URL
|
||||
path := ackURL[strings.Index(ackURL, "/api/notify/ack/"):]
|
||||
|
||||
for i := range 2 {
|
||||
resp, err := http.Post(s.URL+path, "application/json", nil)
|
||||
if err != nil {
|
||||
t.Fatalf("ack %d: %v", i+1, err)
|
||||
}
|
||||
var body map[string]any
|
||||
decode(t, resp, &body)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("ack %d returned %d", i+1, resp.StatusCode)
|
||||
}
|
||||
if body["status"] != "acknowledged" {
|
||||
t.Errorf("ack %d: expected status acknowledged, got %v", i+1, body["status"])
|
||||
}
|
||||
}
|
||||
|
||||
n := 0
|
||||
for _, ty := range eventTypes(timeline(t, s, 1)) {
|
||||
if ty == "acknowledged" {
|
||||
n++
|
||||
}
|
||||
}
|
||||
if n != 1 {
|
||||
t.Errorf("expected exactly one acknowledged event after two taps, got %d", n)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNotify_AckRejectsUnknownToken(t *testing.T) {
|
||||
s, _ := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com"})
|
||||
fireCritical(t, s)
|
||||
@@ -438,7 +500,7 @@ func TestNotify_SweepPurgesExpiredAckTokens(t *testing.T) {
|
||||
s.sweepNotify(t)
|
||||
s.exec(t, "UPDATE incident_ack_tokens SET expires_at = $1", time.Now().Add(-time.Minute).Unix())
|
||||
|
||||
api.Sweep(context.Background(), s.db, 168*time.Hour, 6*time.Hour, s.deadman, s.notify)
|
||||
api.Sweep(context.Background(), s.db, 168*time.Hour, 6*time.Hour, s.notify)
|
||||
|
||||
var n int
|
||||
if err := s.db.QueryRow("SELECT COUNT(*) FROM incident_ack_tokens").Scan(&n); err != nil {
|
||||
|
||||
@@ -0,0 +1,518 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"log"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/oidc"
|
||||
)
|
||||
|
||||
const (
|
||||
// oidcStateCookie ties an in-flight login to the browser that started it.
|
||||
// Without it anybody could start a login, and send the callback URL that
|
||||
// results to somebody else, who would be signed in as the attacker.
|
||||
oidcStateCookie = "terdut_oidc_state"
|
||||
|
||||
// oidcLoginTTL is how long a login may take between the redirect to the
|
||||
// provider and the callback, which includes the person typing a password
|
||||
// and a second factor.
|
||||
oidcLoginTTL = 10 * time.Minute
|
||||
|
||||
// oidcStartMaxPerAddr bounds unauthenticated logins started per address.
|
||||
// Each writes a row, so an unbounded endpoint is a way to grow the table.
|
||||
oidcStartMaxPerAddr = 30
|
||||
)
|
||||
|
||||
// ssoError is a sign-in refusal the person can be told about. Its value is the
|
||||
// code the web UI is sent back with, as ?sso_error=<code>; the detail stays in
|
||||
// the server log, since it can name accounts.
|
||||
type ssoError string
|
||||
|
||||
func (e ssoError) Error() string { return "sso: " + string(e) }
|
||||
|
||||
const (
|
||||
ssoDenied ssoError = "denied" // the provider reported an error, or the person declined
|
||||
ssoExpired ssoError = "expired" // unknown, used or expired state; start again
|
||||
ssoFailed ssoError = "failed" // the token exchange or its verification failed
|
||||
ssoUnavailable ssoError = "unavailable" // the provider could not be reached
|
||||
ssoNotAllowed ssoError = "not_allowed" // authenticated, but in none of the allowed groups
|
||||
ssoNoEmail ssoError = "no_email" // the provider sent no email address
|
||||
ssoEmailConflict ssoError = "email_conflict" // a local account has this email and cannot be linked
|
||||
ssoDisabled ssoError = "disabled" // the linked account is disabled
|
||||
// ssoNotBootstrapped: this identity has no existing account, and no user
|
||||
// exists on this install yet either -- creating one here would race
|
||||
// POST /api/bootstrap for the one gitops-managed installs expect to win
|
||||
// it (terdut-operator's own DESIGN.md §1, §6), which has no way to
|
||||
// recover if it loses. The person sees this for at most as long as it
|
||||
// takes whatever is bootstrapping this install to finish; signing in
|
||||
// again afterward hits the ordinary first-sign-in path. Found by
|
||||
// terdut-operator#1: nothing stopped an otherwise-ordinary OIDC sign-in
|
||||
// from quietly winning this race against an operator that assumed it
|
||||
// was the only caller.
|
||||
ssoNotBootstrapped ssoError = "not_bootstrapped"
|
||||
)
|
||||
|
||||
// handleAuthConfig says how this server can be signed in to, so the login form
|
||||
// and the TUI can offer the right choices before anybody types anything. It is
|
||||
// unauthenticated by necessity, and reveals nothing beyond what the login page
|
||||
// shows anyway.
|
||||
func handleAuthConfig(cfg config.Config) http.HandlerFunc {
|
||||
type oidcInfo struct {
|
||||
Enabled bool `json:"enabled"`
|
||||
Name string `json:"name,omitempty"`
|
||||
}
|
||||
type response struct {
|
||||
PasswordLogin bool `json:"password_login"`
|
||||
OIDC oidcInfo `json:"oidc"`
|
||||
|
||||
// DeviceLogin is whether a client that cannot open a browser (the TUI)
|
||||
// can sign in by showing a code, through /api/oidc/device.
|
||||
DeviceLogin bool `json:"device_login"`
|
||||
|
||||
// OperatorMode is whether this install is gitops-managed: writes to
|
||||
// teams, escalation policies, dead man's switches and integrations
|
||||
// from a session or a user's own API key are refused (OperatorModeBlock),
|
||||
// though a service account's are not. The web UI reads this before
|
||||
// anybody signs in, the same way it reads PasswordLogin/OIDC, so it can
|
||||
// show those sections read-only from the start rather than only after
|
||||
// a write fails.
|
||||
OperatorMode bool `json:"operator_mode"`
|
||||
}
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
resp := response{PasswordLogin: !cfg.DisablePasswordLogin, OperatorMode: cfg.OperatorMode}
|
||||
if cfg.OIDC.Enabled() {
|
||||
resp.OIDC = oidcInfo{Enabled: true, Name: cfg.OIDC.Name}
|
||||
resp.DeviceLogin = true
|
||||
}
|
||||
respond(w, http.StatusOK, resp)
|
||||
}
|
||||
}
|
||||
|
||||
// passwordLoginOnly refuses a route when password login is switched off.
|
||||
func passwordLoginOnly(enabled bool) func(http.Handler) http.Handler {
|
||||
return func(next http.Handler) http.Handler {
|
||||
if enabled {
|
||||
return next
|
||||
}
|
||||
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
respond(w, http.StatusForbidden, errResp("password login is disabled on this server"))
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// ssoRedirect sends the browser back to the web UI with the reason a sign-in
|
||||
// failed. It is a redirect and not a JSON error because the browser arrived
|
||||
// here by navigating from the provider: there is no page script to read one.
|
||||
func ssoRedirect(w http.ResponseWriter, r *http.Request, code ssoError) {
|
||||
http.Redirect(w, r, "/?sso_error="+url.QueryEscape(string(code)), http.StatusFound)
|
||||
}
|
||||
|
||||
// handleOIDCLogin starts a sign-in: it records the state, nonce and PKCE
|
||||
// verifier the callback will need and sends the browser to the provider.
|
||||
func handleOIDCLogin(db *sql.DB, prov *oidc.Provider, limiter *loginLimiter, publicURL string) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
addrKey := "oidc:" + clientAddr(r)
|
||||
if limiter.blocked(addrKey, oidcStartMaxPerAddr) {
|
||||
w.Header().Set("Retry-After", strconv.Itoa(int(loginWindow.Seconds())))
|
||||
respond(w, http.StatusTooManyRequests, errResp("too many sign-in attempts, try again later"))
|
||||
return
|
||||
}
|
||||
limiter.fail(addrKey)
|
||||
|
||||
state, stateHash, err := randomToken()
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
nonce, _, err := randomToken()
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
verifier := oidc.NewVerifier()
|
||||
next := safeNext(r.URL.Query().Get("next"))
|
||||
|
||||
// Abandoned logins are swept here rather than by the sweeper: this is
|
||||
// the only place they are made, so the table cannot outgrow its writers.
|
||||
now := time.Now()
|
||||
db.ExecContext(r.Context(), "DELETE FROM oidc_logins WHERE expires_at < $1", now.Unix())
|
||||
if _, err := db.ExecContext(r.Context(), `
|
||||
INSERT INTO oidc_logins (state_hash, nonce, pkce_verifier, next, expires_at)
|
||||
VALUES ($1, $2, $3, $4, $5)`,
|
||||
stateHash, nonce, verifier, next, now.Add(oidcLoginTTL).Unix()); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
|
||||
authURL, err := prov.AuthURL(r.Context(), state, nonce, verifier)
|
||||
if err != nil {
|
||||
log.Printf("oidc: start login: %v", err)
|
||||
ssoRedirect(w, r, ssoUnavailable)
|
||||
return
|
||||
}
|
||||
|
||||
http.SetCookie(w, &http.Cookie{
|
||||
Name: oidcStateCookie,
|
||||
Value: state,
|
||||
Path: "/api/oidc",
|
||||
MaxAge: int(oidcLoginTTL.Seconds()),
|
||||
HttpOnly: true,
|
||||
Secure: cookieSecure(publicURL, r),
|
||||
// Lax, not Strict: the callback is a top-level navigation from the
|
||||
// provider's site, which Strict would not send the cookie on.
|
||||
SameSite: http.SameSiteLaxMode,
|
||||
})
|
||||
http.Redirect(w, r, authURL, http.StatusFound)
|
||||
}
|
||||
}
|
||||
|
||||
// handleOIDCCallback finishes a sign-in: it verifies the provider's answer,
|
||||
// finds or creates the user, applies their groups and starts a session.
|
||||
func handleOIDCCallback(db *sql.DB, prov *oidc.Provider, publicURL string) http.HandlerFunc {
|
||||
cfg := prov.Config()
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
// The state cookie has done its job once the callback arrives, whatever
|
||||
// the outcome.
|
||||
http.SetCookie(w, &http.Cookie{
|
||||
Name: oidcStateCookie, Value: "", Path: "/api/oidc", MaxAge: -1,
|
||||
HttpOnly: true, Secure: cookieSecure(publicURL, r), SameSite: http.SameSiteLaxMode,
|
||||
})
|
||||
|
||||
q := r.URL.Query()
|
||||
if e := q.Get("error"); e != "" {
|
||||
log.Printf("oidc: provider returned error %q: %s", e, q.Get("error_description"))
|
||||
ssoRedirect(w, r, ssoDenied)
|
||||
return
|
||||
}
|
||||
state := q.Get("state")
|
||||
cookie, err := r.Cookie(oidcStateCookie)
|
||||
if state == "" || q.Get("code") == "" || err != nil || cookie.Value != state {
|
||||
ssoRedirect(w, r, ssoExpired)
|
||||
return
|
||||
}
|
||||
|
||||
// DELETE ... RETURNING makes the state single-use: a replayed callback
|
||||
// finds nothing.
|
||||
var nonce, verifier, next string
|
||||
err = db.QueryRowContext(r.Context(), `
|
||||
DELETE FROM oidc_logins WHERE state_hash = $1 AND expires_at > $2
|
||||
RETURNING nonce, pkce_verifier, next`,
|
||||
hashToken(state), time.Now().Unix()).Scan(&nonce, &verifier, &next)
|
||||
if errors.Is(err, sql.ErrNoRows) {
|
||||
ssoRedirect(w, r, ssoExpired)
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
log.Printf("oidc: load login state: %v", err)
|
||||
ssoRedirect(w, r, ssoFailed)
|
||||
return
|
||||
}
|
||||
|
||||
identity, err := prov.Exchange(r.Context(), q.Get("code"), verifier, nonce)
|
||||
if err != nil {
|
||||
log.Printf("oidc: %v", err)
|
||||
ssoRedirect(w, r, ssoFailed)
|
||||
return
|
||||
}
|
||||
|
||||
grants := oidc.ComputeGrants(cfg, identity.Groups)
|
||||
if !grants.Admitted {
|
||||
log.Printf("oidc: %q (%s) is in none of the allowed groups", identity.Username, identity.Subject)
|
||||
ssoRedirect(w, r, ssoNotAllowed)
|
||||
return
|
||||
}
|
||||
|
||||
teamGroups, err := loadTeamGroups(r.Context(), db)
|
||||
if err != nil {
|
||||
log.Printf("oidc: load team groups: %v", err)
|
||||
ssoRedirect(w, r, ssoFailed)
|
||||
return
|
||||
}
|
||||
teamGrants := oidc.ComputeTeamGrants(teamGroups, identity.Groups)
|
||||
|
||||
userID, err := signInSSO(r.Context(), db, cfg, identity, grants, teamGrants)
|
||||
if err != nil {
|
||||
var se ssoError
|
||||
if errors.As(err, &se) {
|
||||
log.Printf("oidc: refused %q (%s): %v", identity.Username, identity.Subject, se)
|
||||
ssoRedirect(w, r, se)
|
||||
return
|
||||
}
|
||||
log.Printf("oidc: sign in %q: %v", identity.Username, err)
|
||||
ssoRedirect(w, r, ssoFailed)
|
||||
return
|
||||
}
|
||||
|
||||
if err := startSessionCapped(w, r, db, userID, publicURL, cfg.SessionMaxAge); err != nil {
|
||||
log.Printf("oidc: start session: %v", err)
|
||||
ssoRedirect(w, r, ssoFailed)
|
||||
return
|
||||
}
|
||||
http.Redirect(w, r, safeNext(next), http.StatusFound)
|
||||
}
|
||||
}
|
||||
|
||||
// safeNext returns where to send the browser after a sign-in: the path asked
|
||||
// for, if it is one on this server, and the front page otherwise. It is the
|
||||
// only thing standing between a login link and an open redirect, so it accepts
|
||||
// a single leading slash and nothing that a browser could read as another host
|
||||
// ("//evil.example", "/\evil.example"), and never an API path, which would
|
||||
// land somebody on raw JSON.
|
||||
func safeNext(next string) string {
|
||||
switch {
|
||||
case next == "", len(next) > 512,
|
||||
!strings.HasPrefix(next, "/"),
|
||||
strings.HasPrefix(next, "//"),
|
||||
strings.HasPrefix(next, "/api/"),
|
||||
strings.ContainsAny(next, "\\\r\n"):
|
||||
return "/"
|
||||
}
|
||||
return next
|
||||
}
|
||||
|
||||
// signInSSO resolves the identity to a user and applies its grants, in one
|
||||
// transaction: a login that fails half way must not leave memberships changed.
|
||||
func signInSSO(ctx context.Context, db *sql.DB, cfg config.OIDC, id *oidc.Identity, g oidc.Grants, teamRoles map[int64]string) (int64, error) {
|
||||
tx, err := db.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
defer tx.Rollback() //nolint:errcheck
|
||||
|
||||
userID, err := resolveSSOUser(ctx, tx, cfg, id)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
var disabled bool
|
||||
if err := tx.QueryRowContext(ctx,
|
||||
"SELECT disabled_at IS NOT NULL FROM users WHERE id = $1", userID).Scan(&disabled); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
if disabled {
|
||||
return 0, ssoDisabled
|
||||
}
|
||||
if err := syncGrants(ctx, tx, userID, g, teamRoles); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return userID, tx.Commit()
|
||||
}
|
||||
|
||||
// loadTeamGroups reads every team's own OIDC group binding, for the sync to
|
||||
// evaluate against one user's groups at a time. Teams are few, so this reads
|
||||
// the whole table rather than filtering it.
|
||||
func loadTeamGroups(ctx context.Context, db *sql.DB) ([]oidc.TeamGroup, error) {
|
||||
rows, err := db.QueryContext(ctx,
|
||||
"SELECT id, COALESCE(oidc_member_group, ''), COALESCE(oidc_owner_group, '') FROM teams")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
var out []oidc.TeamGroup
|
||||
for rows.Next() {
|
||||
var tg oidc.TeamGroup
|
||||
if err := rows.Scan(&tg.TeamID, &tg.MemberGroup, &tg.OwnerGroup); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, tg)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
// resolveSSOUser finds the user an identity belongs to, linking or creating one
|
||||
// when this is its first sign-in.
|
||||
//
|
||||
// The order matters. The (issuer, subject) pair is the identity; email is only
|
||||
// a way to recognise an existing local account the first time. Once linked, a
|
||||
// changed email at the provider must not move the account to somebody else.
|
||||
func resolveSSOUser(ctx context.Context, tx *sql.Tx, cfg config.OIDC, id *oidc.Identity) (int64, error) {
|
||||
now := time.Now().Unix()
|
||||
|
||||
var userID int64
|
||||
err := tx.QueryRowContext(ctx,
|
||||
"SELECT user_id FROM user_identities WHERE issuer = $1 AND subject = $2",
|
||||
id.Issuer, id.Subject).Scan(&userID)
|
||||
if err == nil {
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
"UPDATE user_identities SET last_login_at = $1 WHERE issuer = $2 AND subject = $3",
|
||||
now, id.Issuer, id.Subject); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return userID, refreshProfile(ctx, tx, userID, id)
|
||||
}
|
||||
if !errors.Is(err, sql.ErrNoRows) {
|
||||
return 0, err
|
||||
}
|
||||
|
||||
// First sign-in with this identity.
|
||||
if id.Email == "" {
|
||||
return 0, ssoNoEmail
|
||||
}
|
||||
err = tx.QueryRowContext(ctx,
|
||||
"SELECT id FROM users WHERE lower(email) = lower($1)", id.Email).Scan(&userID)
|
||||
switch {
|
||||
case err == nil:
|
||||
if !id.EmailVerified && !cfg.TrustEmail {
|
||||
return 0, ssoEmailConflict
|
||||
}
|
||||
// A local account that already has an identity from this issuer is a
|
||||
// different person at the provider using a recycled address. Linking
|
||||
// them would hand one person's account to another.
|
||||
var linked bool
|
||||
if err := tx.QueryRowContext(ctx,
|
||||
"SELECT EXISTS (SELECT 1 FROM user_identities WHERE user_id = $1 AND issuer = $2)",
|
||||
userID, id.Issuer).Scan(&linked); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
if linked {
|
||||
return 0, ssoEmailConflict
|
||||
}
|
||||
case errors.Is(err, sql.ErrNoRows):
|
||||
// Creating the very first user is /api/bootstrap's own job (same
|
||||
// gate, same table: SELECT COUNT(*) FROM users in handleBootstrap).
|
||||
// An identity nobody has linked yet, on an install with no users at
|
||||
// all, is exactly the race terdut-operator#1 found: whoever gets
|
||||
// here first wins a slot the other side has no way to recover from
|
||||
// losing. Refusing it here costs an otherwise-ordinary sign-in
|
||||
// nothing but a retry once bootstrap has actually run.
|
||||
var userCount int
|
||||
if err := tx.QueryRowContext(ctx, "SELECT COUNT(*) FROM users").Scan(&userCount); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
if userCount == 0 {
|
||||
return 0, ssoNotBootstrapped
|
||||
}
|
||||
userID, err = createSSOUser(ctx, tx, id)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
default:
|
||||
return 0, err
|
||||
}
|
||||
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
"INSERT INTO user_identities (user_id, issuer, subject) VALUES ($1, $2, $3)",
|
||||
userID, id.Issuer, id.Subject); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return userID, nil
|
||||
}
|
||||
|
||||
// createSSOUser inserts a user with no password. The username is the provider's,
|
||||
// made unique with a numeric suffix when somebody local already has it.
|
||||
func createSSOUser(ctx context.Context, tx *sql.Tx, id *oidc.Identity) (int64, error) {
|
||||
base := strings.TrimSpace(id.Username)
|
||||
if base == "" {
|
||||
base, _, _ = strings.Cut(id.Email, "@")
|
||||
}
|
||||
if base == "" {
|
||||
base = "user"
|
||||
}
|
||||
for n := 1; n <= 100; n++ {
|
||||
name := base
|
||||
if n > 1 {
|
||||
name = base + "-" + strconv.Itoa(n)
|
||||
}
|
||||
var userID int64
|
||||
err := tx.QueryRowContext(ctx, `
|
||||
INSERT INTO users (username, email) VALUES ($1, $2)
|
||||
ON CONFLICT (username) DO NOTHING RETURNING id`,
|
||||
name, id.Email).Scan(&userID)
|
||||
if errors.Is(err, sql.ErrNoRows) {
|
||||
continue // taken; try the next suffix
|
||||
}
|
||||
return userID, err
|
||||
}
|
||||
return 0, errors.New("no free username for " + base)
|
||||
}
|
||||
|
||||
// refreshProfile brings a linked user's username and email in line with the
|
||||
// provider. Each update is skipped, not failed, when another user already holds
|
||||
// the value: both columns are unique, and a sign-in must not break over a name.
|
||||
func refreshProfile(ctx context.Context, tx *sql.Tx, userID int64, id *oidc.Identity) error {
|
||||
if id.Username != "" {
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE users SET username = $1
|
||||
WHERE id = $2 AND username <> $1
|
||||
AND NOT EXISTS (SELECT 1 FROM users WHERE username = $1)`,
|
||||
id.Username, userID); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
if id.Email != "" {
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
UPDATE users SET email = $1
|
||||
WHERE id = $2 AND email <> $1
|
||||
AND NOT EXISTS (SELECT 1 FROM users WHERE lower(email) = lower($1))`,
|
||||
id.Email, userID); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// syncGrants makes the user's OIDC-sourced access match what their groups grant
|
||||
// now, and touches nothing else.
|
||||
//
|
||||
// Rows the sync owns are marked source 'oidc'. It adds them, changes their role
|
||||
// and removes them. The last-owner and last-administrator guards do not apply:
|
||||
// they exist to stop a person's mistake, and the provider is the source of truth
|
||||
// for the access it grants, so a team or an install can be left without an
|
||||
// SSO-granted owner. Administrators can always repair a team, and the bootstrap
|
||||
// administrator is a manual one. Rows added by hand are 'manual', and the sync
|
||||
// only ever raises them (turning them into 'oidc' rows), never lowers or removes
|
||||
// them.
|
||||
//
|
||||
// teamRoles is keyed by team ID, not name: a team must already exist, with its
|
||||
// own oidc_member_group/oidc_owner_group set by its owner, before a group can
|
||||
// grant access to it. The sync never creates a team.
|
||||
func syncGrants(ctx context.Context, tx *sql.Tx, userID int64, g oidc.Grants, teamRoles map[int64]string) error {
|
||||
// Administrator. A manual administrator stays one whatever the groups say.
|
||||
if g.Admin {
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
"UPDATE users SET is_admin = true, admin_source = 'oidc' WHERE id = $1 AND NOT is_admin",
|
||||
userID); err != nil {
|
||||
return err
|
||||
}
|
||||
} else if _, err := tx.ExecContext(ctx,
|
||||
"UPDATE users SET is_admin = false, admin_source = 'manual' WHERE id = $1 AND admin_source = 'oidc'",
|
||||
userID); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
// Teams. The result of the loop is the set of teams the groups grant.
|
||||
granted := make([]int64, 0, len(teamRoles))
|
||||
for teamID, role := range teamRoles {
|
||||
granted = append(granted, teamID)
|
||||
|
||||
// A row the sync owns follows the groups in both directions. One added by
|
||||
// hand is only raised: a member the owner made an owner by hand is not
|
||||
// demoted because the group says member.
|
||||
if _, err := tx.ExecContext(ctx, `
|
||||
INSERT INTO team_members (team_id, user_id, role, source)
|
||||
VALUES ($1, $2, $3, 'oidc')
|
||||
ON CONFLICT (team_id, user_id) DO UPDATE
|
||||
SET role = excluded.role, source = 'oidc'
|
||||
WHERE team_members.source = 'oidc'
|
||||
OR (excluded.role = $4 AND team_members.role = $5)`,
|
||||
teamID, userID, role, models.RoleOwner, models.RoleMember); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
// Access the groups no longer grant. granted is never nil, or the ALL
|
||||
// comparison would be against NULL and delete nothing.
|
||||
_, err := tx.ExecContext(ctx,
|
||||
"DELETE FROM team_members WHERE user_id = $1 AND source = 'oidc' AND team_id <> ALL($2)",
|
||||
userID, granted)
|
||||
return err
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"database/sql"
|
||||
"net/http"
|
||||
)
|
||||
|
||||
// teamOIDCGroups is one team's own OIDC binding: which group, if any, grants
|
||||
// member access and which grants owner access. The same shape answers GET and
|
||||
// is accepted by PUT. An empty string means no group grants that role here.
|
||||
type teamOIDCGroups struct {
|
||||
MemberGroup string `json:"member_group"`
|
||||
OwnerGroup string `json:"owner_group"`
|
||||
}
|
||||
|
||||
// handleGetTeamOIDCGroups answers which groups control a team's membership.
|
||||
// Member-gated like the member list itself: this is part of "who is in the
|
||||
// team and why", not a setting only an owner should be able to see.
|
||||
func handleGetTeamOIDCGroups(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
teamID, ok := teamParam(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
if !requireTeamMember(w, r, teamID) {
|
||||
return
|
||||
}
|
||||
|
||||
var g teamOIDCGroups
|
||||
err := db.QueryRowContext(r.Context(),
|
||||
"SELECT COALESCE(oidc_member_group, ''), COALESCE(oidc_owner_group, '') FROM teams WHERE id = $1",
|
||||
teamID).Scan(&g.MemberGroup, &g.OwnerGroup)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusOK, g)
|
||||
}
|
||||
}
|
||||
|
||||
// handleSetTeamOIDCGroups sets which groups control a team's membership.
|
||||
//
|
||||
// Owner-gated, the same as the schedule, the integrations and the escalation
|
||||
// ladder: this decides who can end up in the team, which is exactly the kind
|
||||
// of thing only the team's own owner (or an administrator repairing it) should
|
||||
// be able to change. An empty string clears a binding.
|
||||
func handleSetTeamOIDCGroups(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
teamID, ok := teamParam(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
if !requireTeamOwner(w, r, teamID) {
|
||||
return
|
||||
}
|
||||
|
||||
var req teamOIDCGroups
|
||||
if err := decodeJSON(r, &req); err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||
return
|
||||
}
|
||||
|
||||
if _, err := db.ExecContext(r.Context(), `
|
||||
UPDATE teams
|
||||
SET oidc_member_group = NULLIF($1, ''),
|
||||
oidc_owner_group = NULLIF($2, '')
|
||||
WHERE id = $3`,
|
||||
req.MemberGroup, req.OwnerGroup, teamID); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,854 @@
|
||||
package api_test
|
||||
|
||||
import (
|
||||
"crypto"
|
||||
"crypto/rand"
|
||||
"crypto/rsa"
|
||||
"crypto/sha256"
|
||||
"encoding/base64"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"math/big"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"net/url"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||
)
|
||||
|
||||
// fakeIdP is just enough of an OpenID Connect provider for terdut to sign
|
||||
// somebody in against: discovery, a key set and a token endpoint that checks the
|
||||
// PKCE verifier. There is no authorize endpoint; the tests read the URL terdut
|
||||
// redirects to and play the part of the browser and the person themselves.
|
||||
type fakeIdP struct {
|
||||
*httptest.Server
|
||||
key *rsa.PrivateKey
|
||||
|
||||
mu sync.Mutex
|
||||
codes map[string]pendingCode
|
||||
}
|
||||
|
||||
type pendingCode struct {
|
||||
claims map[string]any
|
||||
challenge string
|
||||
}
|
||||
|
||||
const (
|
||||
idpClientID = "terdut"
|
||||
idpClientSecret = "s3cret"
|
||||
)
|
||||
|
||||
func newFakeIdP(t *testing.T) *fakeIdP {
|
||||
t.Helper()
|
||||
key, err := rsa.GenerateKey(rand.Reader, 2048)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
f := &fakeIdP{key: key, codes: map[string]pendingCode{}}
|
||||
|
||||
mux := http.NewServeMux()
|
||||
mux.HandleFunc("/.well-known/openid-configuration", func(w http.ResponseWriter, r *http.Request) {
|
||||
json.NewEncoder(w).Encode(map[string]any{
|
||||
"issuer": f.URL,
|
||||
"authorization_endpoint": f.URL + "/authorize",
|
||||
"token_endpoint": f.URL + "/token",
|
||||
"jwks_uri": f.URL + "/jwks",
|
||||
"id_token_signing_alg_values_supported": []string{"RS256"},
|
||||
"response_types_supported": []string{"code"},
|
||||
"subject_types_supported": []string{"public"},
|
||||
})
|
||||
})
|
||||
mux.HandleFunc("/jwks", func(w http.ResponseWriter, r *http.Request) {
|
||||
b64 := base64.RawURLEncoding.EncodeToString
|
||||
json.NewEncoder(w).Encode(map[string]any{"keys": []map[string]string{{
|
||||
"kty": "RSA", "kid": "k1", "use": "sig", "alg": "RS256",
|
||||
"n": b64(key.N.Bytes()),
|
||||
"e": b64(big.NewInt(int64(key.E)).Bytes()),
|
||||
}}})
|
||||
})
|
||||
mux.HandleFunc("/token", func(w http.ResponseWriter, r *http.Request) {
|
||||
r.ParseForm()
|
||||
user, pass, basic := r.BasicAuth()
|
||||
if !basic {
|
||||
user, pass = r.PostForm.Get("client_id"), r.PostForm.Get("client_secret")
|
||||
}
|
||||
if user != idpClientID || pass != idpClientSecret {
|
||||
http.Error(w, `{"error":"invalid_client"}`, http.StatusUnauthorized)
|
||||
return
|
||||
}
|
||||
f.mu.Lock()
|
||||
p, ok := f.codes[r.PostForm.Get("code")]
|
||||
delete(f.codes, r.PostForm.Get("code")) // single use, like a real provider
|
||||
f.mu.Unlock()
|
||||
sum := sha256.Sum256([]byte(r.PostForm.Get("code_verifier")))
|
||||
if !ok || base64.RawURLEncoding.EncodeToString(sum[:]) != p.challenge {
|
||||
http.Error(w, `{"error":"invalid_grant"}`, http.StatusBadRequest)
|
||||
return
|
||||
}
|
||||
// oauth2 picks the parser from the content type; without this it reads
|
||||
// the body as a form, finds no token and retries, spending the code.
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
json.NewEncoder(w).Encode(map[string]any{
|
||||
"access_token": "unused", "token_type": "Bearer", "expires_in": 300,
|
||||
"id_token": f.sign(t, p.claims),
|
||||
})
|
||||
})
|
||||
f.Server = httptest.NewServer(mux)
|
||||
t.Cleanup(f.Close)
|
||||
return f
|
||||
}
|
||||
|
||||
// sign returns claims as an RS256 JWT.
|
||||
func (f *fakeIdP) sign(t *testing.T, claims map[string]any) string {
|
||||
t.Helper()
|
||||
enc := func(v any) string {
|
||||
b, _ := json.Marshal(v)
|
||||
return base64.RawURLEncoding.EncodeToString(b)
|
||||
}
|
||||
signing := enc(map[string]string{"alg": "RS256", "kid": "k1", "typ": "JWT"}) + "." + enc(claims)
|
||||
sum := sha256.Sum256([]byte(signing))
|
||||
sig, err := rsa.SignPKCS1v15(rand.Reader, f.key, crypto.SHA256, sum[:])
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return signing + "." + base64.RawURLEncoding.EncodeToString(sig)
|
||||
}
|
||||
|
||||
// idpUser is who signs in, as the provider describes them.
|
||||
type idpUser struct {
|
||||
sub, username, email string
|
||||
unverified bool
|
||||
groups []string
|
||||
badNonce bool
|
||||
}
|
||||
|
||||
// ssoConfig is a terdut configuration wired to idp: terdut-users may sign in,
|
||||
// terdut-admins administer. Which groups grant which team is not config
|
||||
// anymore — it is each team's own oidc_member_group/oidc_owner_group, so a
|
||||
// test that needs one seeds it with seedTeam.
|
||||
func ssoConfig(idp *fakeIdP) config.Config {
|
||||
c := testConfig()
|
||||
c.OIDC = config.OIDC{
|
||||
Issuer: idp.URL,
|
||||
ClientID: idpClientID,
|
||||
ClientSecret: idpClientSecret,
|
||||
Name: "Authentik",
|
||||
Scopes: []string{"openid", "profile", "email"},
|
||||
UsernameClaim: "preferred_username",
|
||||
EmailClaim: "email",
|
||||
GroupsClaim: "groups",
|
||||
AllowedGroups: []string{"terdut-users"},
|
||||
AdminGroup: "terdut-admins",
|
||||
SessionMaxAge: 12 * time.Hour,
|
||||
}
|
||||
return c
|
||||
}
|
||||
|
||||
// seedTeam creates a team with an OIDC group binding, the way an owner would
|
||||
// set one from the Members tab. Teams are no longer created by the sync
|
||||
// itself, so a test whose groups should grant something needs the team to
|
||||
// already exist. An empty group means that role is not granted by one.
|
||||
func (s *ts) seedTeam(t *testing.T, name, memberGroup, ownerGroup string) int64 {
|
||||
t.Helper()
|
||||
var id int64
|
||||
err := s.db.QueryRow(`
|
||||
INSERT INTO teams (name, oidc_member_group, oidc_owner_group)
|
||||
VALUES ($1, NULLIF($2, ''), NULLIF($3, '')) RETURNING id`,
|
||||
name, memberGroup, ownerGroup).Scan(&id)
|
||||
if err != nil {
|
||||
t.Fatalf("seed team %q: %v", name, err)
|
||||
}
|
||||
return id
|
||||
}
|
||||
|
||||
func newSSOTS(t *testing.T, idp *fakeIdP, tweak ...func(*config.Config)) *ts {
|
||||
t.Helper()
|
||||
c := ssoConfig(idp)
|
||||
for _, f := range tweak {
|
||||
f(&c)
|
||||
}
|
||||
return newTSWith(t, api.DeadmanConfig{}, api.NotifyConfig{PublicURL: "http://terdut.test"}, c)
|
||||
}
|
||||
|
||||
// ssoBrowser is a browser that does not follow redirects, so a test can read
|
||||
// where each step sends it.
|
||||
func ssoBrowser(t *testing.T, s *ts) *browser {
|
||||
t.Helper()
|
||||
b := newBrowser(t, s.URL)
|
||||
b.CheckRedirect = func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse }
|
||||
return b
|
||||
}
|
||||
|
||||
// startLogin visits /api/oidc/login and returns what terdut asked the provider
|
||||
// for: the state, nonce and PKCE challenge.
|
||||
func startLogin(t *testing.T, idp *fakeIdP, b *browser) (state, nonce, challenge string) {
|
||||
t.Helper()
|
||||
resp := b.do(t, http.MethodGet, "/api/oidc/login", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusFound {
|
||||
t.Fatalf("login start: %d", resp.StatusCode)
|
||||
}
|
||||
loc, err := url.Parse(resp.Header.Get("Location"))
|
||||
if err != nil || !strings.HasPrefix(loc.String(), idp.URL+"/authorize") {
|
||||
t.Fatalf("login redirected to %q, want the provider", resp.Header.Get("Location"))
|
||||
}
|
||||
q := loc.Query()
|
||||
if q.Get("code_challenge_method") != "S256" || q.Get("client_id") != idpClientID ||
|
||||
q.Get("redirect_uri") != "http://terdut.test/api/oidc/callback" || q.Get("response_type") != "code" {
|
||||
t.Fatalf("unexpected authorization request: %v", q)
|
||||
}
|
||||
return q.Get("state"), q.Get("nonce"), q.Get("code_challenge")
|
||||
}
|
||||
|
||||
// issueCode has the provider authenticate u and hand back an authorization code.
|
||||
func (f *fakeIdP) issueCode(u idpUser, nonce, challenge string) string {
|
||||
if u.badNonce {
|
||||
nonce = "not-the-nonce"
|
||||
}
|
||||
claims := map[string]any{
|
||||
"iss": f.URL, "sub": u.sub, "aud": idpClientID,
|
||||
"iat": time.Now().Unix(), "exp": time.Now().Add(5 * time.Minute).Unix(),
|
||||
"nonce": nonce,
|
||||
"preferred_username": u.username,
|
||||
"email": u.email,
|
||||
"email_verified": !u.unverified,
|
||||
"groups": u.groups,
|
||||
}
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
code := fmt.Sprintf("code-%d", len(f.codes)+int(time.Now().UnixNano()%1e6))
|
||||
f.codes[code] = pendingCode{claims: claims, challenge: challenge}
|
||||
return code
|
||||
}
|
||||
|
||||
// callback delivers the provider's answer to terdut and returns where terdut
|
||||
// sends the browser next.
|
||||
func callback(t *testing.T, b *browser, code, state string) string {
|
||||
t.Helper()
|
||||
resp := b.do(t, http.MethodGet, "/api/oidc/callback?code="+url.QueryEscape(code)+"&state="+url.QueryEscape(state), nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusFound {
|
||||
t.Fatalf("callback: %d", resp.StatusCode)
|
||||
}
|
||||
return resp.Header.Get("Location")
|
||||
}
|
||||
|
||||
// signInSSO runs a whole sign-in and returns the Location the callback ended on.
|
||||
func signInSSO(t *testing.T, idp *fakeIdP, b *browser, u idpUser) string {
|
||||
t.Helper()
|
||||
state, nonce, challenge := startLogin(t, idp, b)
|
||||
return callback(t, b, idp.issueCode(u, nonce, challenge), state)
|
||||
}
|
||||
|
||||
var alice = idpUser{sub: "sub-alice", username: "alice", email: "alice@example.com", groups: []string{"terdut-users", "sre"}}
|
||||
|
||||
func withGroups(u idpUser, groups ...string) idpUser {
|
||||
u.groups = groups
|
||||
return u
|
||||
}
|
||||
|
||||
// meOf reads /api/me over the browser's session.
|
||||
func meOf(t *testing.T, b *browser) (status int, username string, isAdmin, hasPassword bool) {
|
||||
t.Helper()
|
||||
resp := b.do(t, http.MethodGet, "/api/me", nil)
|
||||
defer resp.Body.Close()
|
||||
var me struct {
|
||||
User struct {
|
||||
Username string `json:"username"`
|
||||
IsAdmin bool `json:"is_admin"`
|
||||
} `json:"user"`
|
||||
HasPassword bool `json:"has_password"`
|
||||
}
|
||||
json.NewDecoder(resp.Body).Decode(&me)
|
||||
return resp.StatusCode, me.User.Username, me.User.IsAdmin, me.HasPassword
|
||||
}
|
||||
|
||||
// memberships lists a user's teams as name -> "role/source".
|
||||
func (s *ts) memberships(t *testing.T, username string) map[string]string {
|
||||
t.Helper()
|
||||
rows, err := s.db.Query(`
|
||||
SELECT t.name, m.role, m.source FROM team_members m
|
||||
JOIN teams t ON t.id = m.team_id JOIN users u ON u.id = m.user_id
|
||||
WHERE u.username = $1`, username)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer rows.Close()
|
||||
out := map[string]string{}
|
||||
for rows.Next() {
|
||||
var name, role, source string
|
||||
rows.Scan(&name, &role, &source)
|
||||
out[name] = role + "/" + source
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func sameMap(a, b map[string]string) bool {
|
||||
if len(a) != len(b) {
|
||||
return false
|
||||
}
|
||||
for k, v := range a {
|
||||
if b[k] != v {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
func TestSSO_FirstSignInCreatesUserAndGrantsTeams(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
s.seedTeam(t, "SRE", "sre", "sre-leads")
|
||||
s.seedTeam(t, "Platform", "platform", "")
|
||||
b := ssoBrowser(t, s)
|
||||
|
||||
if loc := signInSSO(t, idp, b, withGroups(alice, "terdut-users", "sre", "platform")); loc != "/" {
|
||||
t.Fatalf("signed in and was sent to %q, want /", loc)
|
||||
}
|
||||
status, name, isAdmin, hasPassword := meOf(t, b)
|
||||
if status != http.StatusOK || name != "alice" || isAdmin || hasPassword {
|
||||
t.Fatalf("me: status %d user %q admin %v has_password %v", status, name, isAdmin, hasPassword)
|
||||
}
|
||||
want := map[string]string{"SRE": "member/oidc", "Platform": "member/oidc"}
|
||||
if got := s.memberships(t, "alice"); !sameMap(got, want) {
|
||||
t.Errorf("memberships %v, want %v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// A group matching no team's own binding grants nothing and creates nothing:
|
||||
// unlike the old global mapping, the sync never creates a team by name.
|
||||
func TestSSO_NoAutoCreateTeam(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
|
||||
var before int
|
||||
s.db.QueryRow("SELECT COUNT(*) FROM teams").Scan(&before)
|
||||
|
||||
signInSSO(t, idp, ssoBrowser(t, s), alice) // groups include "sre"; no team names it
|
||||
if got := s.memberships(t, "alice"); len(got) != 0 {
|
||||
t.Errorf("memberships %v, want none: no team's oidc_member_group/oidc_owner_group is set", got)
|
||||
}
|
||||
|
||||
var after int
|
||||
s.db.QueryRow("SELECT COUNT(*) FROM teams").Scan(&after)
|
||||
if after != before {
|
||||
t.Errorf("team count %d -> %d, want no team created", before, after)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_RefusedOutsideAllowedGroups(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
b := ssoBrowser(t, s)
|
||||
|
||||
loc := signInSSO(t, idp, b, withGroups(alice, "sre", "terdut-admins"))
|
||||
if loc != "/?sso_error=not_allowed" {
|
||||
t.Fatalf("sent to %q, want the not_allowed error", loc)
|
||||
}
|
||||
if status, _, _, _ := meOf(t, b); status != http.StatusUnauthorized {
|
||||
t.Errorf("a refused sign-in must not leave a session: /api/me %d", status)
|
||||
}
|
||||
var n int
|
||||
s.db.QueryRow("SELECT COUNT(*) FROM users WHERE username = 'alice'").Scan(&n)
|
||||
if n != 0 {
|
||||
t.Error("a refused sign-in must not create the user")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_AdminFollowsTheAdminGroup(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
|
||||
signInSSO(t, idp, ssoBrowser(t, s), withGroups(alice, "terdut-users", "terdut-admins"))
|
||||
var isAdmin bool
|
||||
var source string
|
||||
read := func() {
|
||||
s.db.QueryRow("SELECT is_admin, admin_source FROM users WHERE username = 'alice'").Scan(&isAdmin, &source)
|
||||
}
|
||||
if read(); !isAdmin || source != "oidc" {
|
||||
t.Fatalf("after admin sign-in: admin %v source %q", isAdmin, source)
|
||||
}
|
||||
|
||||
signInSSO(t, idp, ssoBrowser(t, s), withGroups(alice, "terdut-users"))
|
||||
if read(); isAdmin || source != "manual" {
|
||||
t.Errorf("after losing the group: admin %v source %q, want revoked and manual", isAdmin, source)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_ManualAdminIsNeverRevoked(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp, func(c *config.Config) { c.OIDC.TrustEmail = true })
|
||||
|
||||
// The bootstrap administrator is a manual one. Signing in through the
|
||||
// provider without the admin group must not take that away.
|
||||
signInSSO(t, idp, ssoBrowser(t, s), idpUser{sub: "sub-admin", username: "admin", email: "admin@test.com", groups: []string{"terdut-users"}})
|
||||
var isAdmin bool
|
||||
var source string
|
||||
s.db.QueryRow("SELECT is_admin, admin_source FROM users WHERE username = 'admin'").Scan(&isAdmin, &source)
|
||||
if !isAdmin || source != "manual" {
|
||||
t.Errorf("admin %v source %q, want still a manual admin", isAdmin, source)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_LosingAGroupRemovesOnlyManagedAccess(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
s.seedTeam(t, "SRE", "sre", "sre-leads")
|
||||
|
||||
signInSSO(t, idp, ssoBrowser(t, s), alice)
|
||||
// Somebody adds alice to another team by hand.
|
||||
s.exec(t, "INSERT INTO teams (name) VALUES ('Hand')")
|
||||
s.exec(t, `INSERT INTO team_members (team_id, user_id, role)
|
||||
SELECT (SELECT id FROM teams WHERE name = 'Hand'), id, 'member' FROM users WHERE username = 'alice'`)
|
||||
|
||||
signInSSO(t, idp, ssoBrowser(t, s), withGroups(alice, "terdut-users"))
|
||||
want := map[string]string{"Hand": "member/manual"}
|
||||
if got := s.memberships(t, "alice"); !sameMap(got, want) {
|
||||
t.Errorf("memberships %v, want %v: the SRE row is the sync's to remove, Hand is not", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_HighestRoleWinsAndRoleChangesFollow(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
s.seedTeam(t, "SRE", "sre", "sre-leads")
|
||||
|
||||
signInSSO(t, idp, ssoBrowser(t, s), withGroups(alice, "terdut-users", "sre", "sre-leads"))
|
||||
if got := s.memberships(t, "alice"); !sameMap(got, map[string]string{"SRE": "owner/oidc"}) {
|
||||
t.Errorf("both groups: %v, want owner", got)
|
||||
}
|
||||
signInSSO(t, idp, ssoBrowser(t, s), withGroups(alice, "terdut-users", "sre"))
|
||||
if got := s.memberships(t, "alice"); !sameMap(got, map[string]string{"SRE": "member/oidc"}) {
|
||||
t.Errorf("lead group dropped: %v, want member", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_ManualMemberIsRaisedNeverLowered(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
|
||||
// alice exists locally, is a manual owner of SRE, and is linked by email.
|
||||
s.exec(t, "INSERT INTO users (username, email) VALUES ('alice', 'alice@example.com')")
|
||||
s.seedTeam(t, "SRE", "sre", "")
|
||||
s.exec(t, `INSERT INTO team_members (team_id, user_id, role)
|
||||
VALUES ((SELECT id FROM teams WHERE name = 'SRE'), (SELECT id FROM users WHERE username = 'alice'), 'owner')`)
|
||||
|
||||
signInSSO(t, idp, ssoBrowser(t, s), alice) // the group only grants member
|
||||
if got := s.memberships(t, "alice"); !sameMap(got, map[string]string{"SRE": "owner/manual"}) {
|
||||
t.Errorf("%v: a hand-made owner must not be lowered by a member mapping", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_LinksExistingUserByVerifiedEmail(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
s.exec(t, "INSERT INTO users (username, email) VALUES ('alice-local', 'Alice@Example.com')")
|
||||
|
||||
b := ssoBrowser(t, s)
|
||||
signInSSO(t, idp, b, alice)
|
||||
if _, name, _, _ := meOf(t, b); name != "alice-local" {
|
||||
t.Errorf("signed in as %q, want the existing local user", name)
|
||||
}
|
||||
var users, identities int
|
||||
s.db.QueryRow("SELECT COUNT(*) FROM users").Scan(&users)
|
||||
s.db.QueryRow("SELECT COUNT(*) FROM user_identities").Scan(&identities)
|
||||
if users != 2 || identities != 1 { // admin + alice-local
|
||||
t.Errorf("%d users, %d identities: linking must not create a second user", users, identities)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_UnverifiedEmailIsNotLinkedUnlessTrusted(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
unverified := alice
|
||||
unverified.unverified = true
|
||||
|
||||
s := newSSOTS(t, idp)
|
||||
s.exec(t, "INSERT INTO users (username, email) VALUES ('alice-local', 'alice@example.com')")
|
||||
if loc := signInSSO(t, idp, ssoBrowser(t, s), unverified); loc != "/?sso_error=email_conflict" {
|
||||
t.Errorf("unverified email: sent to %q, want email_conflict", loc)
|
||||
}
|
||||
|
||||
trusting := newSSOTS(t, idp, func(c *config.Config) { c.OIDC.TrustEmail = true })
|
||||
trusting.exec(t, "INSERT INTO users (username, email) VALUES ('alice-local', 'alice@example.com')")
|
||||
b := ssoBrowser(t, trusting)
|
||||
if loc := signInSSO(t, idp, b, unverified); loc != "/" {
|
||||
t.Fatalf("trusted email: sent to %q, want /", loc)
|
||||
}
|
||||
if _, name, _, _ := meOf(t, b); name != "alice-local" {
|
||||
t.Errorf("signed in as %q, want the existing local user", name)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_RecycledEmailDoesNotTakeOverALinkedAccount(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
signInSSO(t, idp, ssoBrowser(t, s), alice)
|
||||
|
||||
// A different person at the provider, same address.
|
||||
other := alice
|
||||
other.sub = "sub-someone-else"
|
||||
if loc := signInSSO(t, idp, ssoBrowser(t, s), other); loc != "/?sso_error=email_conflict" {
|
||||
t.Errorf("sent to %q, want email_conflict", loc)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_UsernameCollisionGetsASuffix(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
s.exec(t, "INSERT INTO users (username, email) VALUES ('alice', 'someone-else@example.com')")
|
||||
|
||||
b := ssoBrowser(t, s)
|
||||
signInSSO(t, idp, b, alice)
|
||||
if _, name, _, _ := meOf(t, b); name != "alice-2" {
|
||||
t.Errorf("username %q, want alice-2", name)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_ProfileFollowsTheProvider(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
signInSSO(t, idp, ssoBrowser(t, s), alice)
|
||||
|
||||
renamed := alice
|
||||
renamed.username, renamed.email = "alice.smith", "alice.smith@example.com"
|
||||
b := ssoBrowser(t, s)
|
||||
signInSSO(t, idp, b, renamed)
|
||||
if _, name, _, _ := meOf(t, b); name != "alice.smith" {
|
||||
t.Errorf("username %q, want the provider's new one", name)
|
||||
}
|
||||
var email string
|
||||
s.db.QueryRow("SELECT email FROM users WHERE username = 'alice.smith'").Scan(&email)
|
||||
if email != "alice.smith@example.com" {
|
||||
t.Errorf("email %q", email)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_DisabledUserIsRefused(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
signInSSO(t, idp, ssoBrowser(t, s), alice)
|
||||
s.exec(t, "UPDATE users SET disabled_at = 1 WHERE username = 'alice'")
|
||||
|
||||
b := ssoBrowser(t, s)
|
||||
if loc := signInSSO(t, idp, b, alice); loc != "/?sso_error=disabled" {
|
||||
t.Errorf("sent to %q, want disabled", loc)
|
||||
}
|
||||
if status, _, _, _ := meOf(t, b); status != http.StatusUnauthorized {
|
||||
t.Errorf("/api/me %d, want 401", status)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSSO_FirstUserIsRefusedUntilBootstrap is terdut-operator#1: an
|
||||
// otherwise-ordinary OIDC sign-in against a brand-new, not-yet-bootstrapped
|
||||
// install must not be allowed to create the first user and win the race
|
||||
// POST /api/bootstrap expects to win uncontested. Built directly over
|
||||
// api.NewRouter rather than newSSOTS/newTS, both of which bootstrap before
|
||||
// a test body ever runs -- exactly the state this test needs to not have yet.
|
||||
func TestSSO_FirstUserIsRefusedUntilBootstrap(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
database := newTestDB(t)
|
||||
srv := httptest.NewServer(api.NewRouter(database, api.NotifyConfig{PublicURL: "http://terdut.test"}, ssoConfig(idp), "test"))
|
||||
t.Cleanup(srv.Close)
|
||||
|
||||
first := newBrowser(t, srv.URL)
|
||||
first.CheckRedirect = func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse }
|
||||
if loc := signInSSO(t, idp, first, alice); loc != "/?sso_error=not_bootstrapped" {
|
||||
t.Fatalf("sent to %q, want not_bootstrapped", loc)
|
||||
}
|
||||
|
||||
// Bootstrap the install for real, the way terdut-operator's own
|
||||
// reconcileBootstrap does.
|
||||
resp, err := http.Post(srv.URL+"/api/bootstrap", "application/json",
|
||||
strings.NewReader(`{"username":"admin","email":"admin@test.com"}`))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("bootstrap: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
// The same identity, signing in again, is this install's ordinary first
|
||||
// SSO user now -- no longer refused.
|
||||
second := newBrowser(t, srv.URL)
|
||||
second.CheckRedirect = func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse }
|
||||
if loc := signInSSO(t, idp, second, alice); loc != "/" {
|
||||
t.Errorf("sent to %q after bootstrap, want success", loc)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_NoEmailIsRefused(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
noEmail := alice
|
||||
noEmail.email = ""
|
||||
if loc := signInSSO(t, idp, ssoBrowser(t, s), noEmail); loc != "/?sso_error=no_email" {
|
||||
t.Errorf("sent to %q, want no_email", loc)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_SessionIsCappedAndDoesNotSlidePastTheCap(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
b := ssoBrowser(t, s)
|
||||
signInSSO(t, idp, b, alice)
|
||||
|
||||
var expires, ceiling int64
|
||||
s.db.QueryRow(`SELECT expires_at, max_expires_at FROM sessions ORDER BY id DESC LIMIT 1`).Scan(&expires, &ceiling)
|
||||
inTwelveHours := time.Now().Add(12 * time.Hour).Unix()
|
||||
if ceiling < inTwelveHours-60 || ceiling > inTwelveHours+60 || expires != ceiling {
|
||||
t.Fatalf("expires %d ceiling %d, want both about %d", expires, ceiling, inTwelveHours)
|
||||
}
|
||||
|
||||
// Age the session so the next request would slide it, with a ceiling well
|
||||
// inside the ordinary 30 days.
|
||||
s.exec(t, "UPDATE sessions SET last_seen_at = last_seen_at - 7200")
|
||||
if status, _, _, _ := meOf(t, b); status != http.StatusOK {
|
||||
t.Fatalf("/api/me %d", status)
|
||||
}
|
||||
var after int64
|
||||
s.db.QueryRow(`SELECT expires_at FROM sessions ORDER BY id DESC LIMIT 1`).Scan(&after)
|
||||
if after > ceiling {
|
||||
t.Errorf("expiry slid to %d, past the ceiling %d", after, ceiling)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_PasswordSessionsStillSlideWithoutACeiling(t *testing.T) {
|
||||
s := newTS(t)
|
||||
b := signedIn(t, s)
|
||||
var ceiling *int64
|
||||
s.db.QueryRow(`SELECT max_expires_at FROM sessions ORDER BY id DESC LIMIT 1`).Scan(&ceiling)
|
||||
if ceiling != nil {
|
||||
t.Errorf("a password session has a ceiling %d, want none", *ceiling)
|
||||
}
|
||||
s.exec(t, "UPDATE sessions SET last_seen_at = last_seen_at - 7200, expires_at = expires_at - 7200")
|
||||
var before, after int64
|
||||
s.db.QueryRow(`SELECT expires_at FROM sessions ORDER BY id DESC LIMIT 1`).Scan(&before)
|
||||
meOf(t, b)
|
||||
s.db.QueryRow(`SELECT expires_at FROM sessions ORDER BY id DESC LIMIT 1`).Scan(&after)
|
||||
if after <= before {
|
||||
t.Errorf("expiry %d -> %d, want it to slide forward", before, after)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_StateIsSingleUseAndBoundToTheBrowser(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
|
||||
// Replaying a callback finds no state.
|
||||
b := ssoBrowser(t, s)
|
||||
state, nonce, challenge := startLogin(t, idp, b)
|
||||
code := idp.issueCode(alice, nonce, challenge)
|
||||
if loc := callback(t, b, code, state); loc != "/" {
|
||||
t.Fatalf("first callback sent to %q", loc)
|
||||
}
|
||||
if loc := callback(t, b, idp.issueCode(alice, nonce, challenge), state); loc != "/?sso_error=expired" {
|
||||
t.Errorf("replayed state: sent to %q, want expired", loc)
|
||||
}
|
||||
|
||||
// A callback from a browser that did not start the login is refused, which
|
||||
// is what stops a login being planted on somebody else.
|
||||
victim := ssoBrowser(t, s)
|
||||
state, nonce, challenge = startLogin(t, idp, ssoBrowser(t, s)) // the attacker's
|
||||
if loc := callback(t, victim, idp.issueCode(alice, nonce, challenge), state); loc != "/?sso_error=expired" {
|
||||
t.Errorf("foreign browser: sent to %q, want expired", loc)
|
||||
}
|
||||
if status, _, _, _ := meOf(t, victim); status != http.StatusUnauthorized {
|
||||
t.Errorf("the victim has a session: /api/me %d", status)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_WrongNonceIsRefused(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
bad := alice
|
||||
bad.badNonce = true
|
||||
b := ssoBrowser(t, s)
|
||||
if loc := signInSSO(t, idp, b, bad); loc != "/?sso_error=failed" {
|
||||
t.Errorf("sent to %q, want failed", loc)
|
||||
}
|
||||
if status, _, _, _ := meOf(t, b); status != http.StatusUnauthorized {
|
||||
t.Errorf("/api/me %d, want 401", status)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_ProviderErrorGoesBackToTheUI(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
b := ssoBrowser(t, s)
|
||||
resp := b.do(t, http.MethodGet, "/api/oidc/callback?error=access_denied", nil)
|
||||
resp.Body.Close()
|
||||
if loc := resp.Header.Get("Location"); resp.StatusCode != http.StatusFound || loc != "/?sso_error=denied" {
|
||||
t.Errorf("%d to %q, want a redirect to denied", resp.StatusCode, loc)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_ManagedAccessCannotBeEditedByHand(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
s.seedTeam(t, "SRE", "sre", "")
|
||||
signInSSO(t, idp, ssoBrowser(t, s), withGroups(alice, "terdut-users", "sre", "terdut-admins"))
|
||||
|
||||
var aliceID, sreID int64
|
||||
s.db.QueryRow("SELECT id FROM users WHERE username = 'alice'").Scan(&aliceID)
|
||||
s.db.QueryRow("SELECT id FROM teams WHERE name = 'SRE'").Scan(&sreID)
|
||||
teamPath := fmt.Sprintf("/api/teams/%d/members", sreID)
|
||||
|
||||
// The bootstrap admin is a system administrator, so may manage SRE.
|
||||
for _, c := range []struct {
|
||||
name, method, path string
|
||||
body any
|
||||
}{
|
||||
{"role change", http.MethodPost, teamPath, map[string]any{"user_id": aliceID, "role": "owner"}},
|
||||
{"removal", http.MethodDelete, fmt.Sprintf("%s/%d", teamPath, aliceID), nil},
|
||||
{"admin revoke", http.MethodPut, fmt.Sprintf("/api/users/%d/admin", aliceID), map[string]any{"is_admin": false}},
|
||||
} {
|
||||
resp := s.req(t, c.method, c.path, c.body)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusConflict {
|
||||
t.Errorf("%s: %d, want 409", c.name, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
if got := s.memberships(t, "alice"); !sameMap(got, map[string]string{"SRE": "member/oidc"}) {
|
||||
t.Errorf("memberships changed by a refused edit: %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_PasswordLoginCanBeSwitchedOff(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp, func(c *config.Config) { c.DisablePasswordLogin = true })
|
||||
b := newBrowser(t, s.URL)
|
||||
|
||||
resp := b.login(t, "admin", "whatever-password")
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("login: %d, want 403", resp.StatusCode)
|
||||
}
|
||||
resp = b.do(t, http.MethodPost, "/api/signup", map[string]string{"username": "x", "email": "x@example.com", "password": "correct horse battery"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("signup: %d, want 403", resp.StatusCode)
|
||||
}
|
||||
|
||||
var cfg struct {
|
||||
PasswordLogin bool `json:"password_login"`
|
||||
OIDC struct {
|
||||
Enabled bool `json:"enabled"`
|
||||
Name string `json:"name"`
|
||||
} `json:"oidc"`
|
||||
}
|
||||
resp = b.do(t, http.MethodGet, "/api/auth/config", nil)
|
||||
defer resp.Body.Close()
|
||||
json.NewDecoder(resp.Body).Decode(&cfg)
|
||||
if cfg.PasswordLogin || !cfg.OIDC.Enabled || cfg.OIDC.Name != "Authentik" {
|
||||
t.Errorf("auth config: %+v", cfg)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAuthConfig_DefaultsToPasswordOnly(t *testing.T) {
|
||||
s := newTS(t)
|
||||
var cfg struct {
|
||||
PasswordLogin bool `json:"password_login"`
|
||||
OIDC struct {
|
||||
Enabled bool `json:"enabled"`
|
||||
} `json:"oidc"`
|
||||
}
|
||||
resp := newBrowser(t, s.URL).do(t, http.MethodGet, "/api/auth/config", nil)
|
||||
defer resp.Body.Close()
|
||||
json.NewDecoder(resp.Body).Decode(&cfg)
|
||||
if !cfg.PasswordLogin || cfg.OIDC.Enabled {
|
||||
t.Errorf("auth config: %+v", cfg)
|
||||
}
|
||||
|
||||
// With SSO off the routes do not exist, rather than answering with an error
|
||||
// page a person could land on.
|
||||
resp = newBrowser(t, s.URL).do(t, http.MethodGet, "/api/oidc/login", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNotFound {
|
||||
t.Errorf("/api/oidc/login with SSO off: %d, want 404", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_UnreachableProviderRedirectsWithAnError(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
idp.Close() // the provider goes down after terdut has started
|
||||
|
||||
b := ssoBrowser(t, s)
|
||||
resp := b.do(t, http.MethodGet, "/api/oidc/login", nil)
|
||||
resp.Body.Close()
|
||||
if loc := resp.Header.Get("Location"); resp.StatusCode != http.StatusFound || loc != "/?sso_error=unavailable" {
|
||||
t.Errorf("%d to %q, want a redirect to unavailable", resp.StatusCode, loc)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSSO_APIShowsWhereAccessCameFrom(t *testing.T) {
|
||||
idp := newFakeIdP(t)
|
||||
s := newSSOTS(t, idp)
|
||||
s.seedTeam(t, "SRE", "sre", "")
|
||||
b := ssoBrowser(t, s)
|
||||
signInSSO(t, idp, b, withGroups(alice, "terdut-users", "sre", "terdut-admins"))
|
||||
|
||||
var aliceID, sreID int64
|
||||
s.db.QueryRow("SELECT id FROM users WHERE username = 'alice'").Scan(&aliceID)
|
||||
s.db.QueryRow("SELECT id FROM teams WHERE name = 'SRE'").Scan(&sreID)
|
||||
|
||||
// Users: alice's administrator flag is the groups', the bootstrap admin's is not.
|
||||
var users []struct {
|
||||
Username string `json:"username"`
|
||||
AdminSource string `json:"admin_source"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodGet, "/api/users", nil), &users)
|
||||
got := map[string]string{}
|
||||
for _, u := range users {
|
||||
got[u.Username] = u.AdminSource
|
||||
}
|
||||
if got["alice"] != "oidc" || got["admin"] != "manual" {
|
||||
t.Errorf("admin_source by user: %v", got)
|
||||
}
|
||||
|
||||
// The team's own member list, as a member sees it.
|
||||
var members []struct {
|
||||
Username string `json:"username"`
|
||||
Source string `json:"source"`
|
||||
}
|
||||
resp := b.do(t, http.MethodGet, fmt.Sprintf("/api/teams/%d/members", sreID), nil)
|
||||
decode(t, resp, &members)
|
||||
if len(members) != 1 || members[0].Username != "alice" || members[0].Source != "oidc" {
|
||||
t.Errorf("team members: %+v", members)
|
||||
}
|
||||
|
||||
// The administrator's view of the same team, and of alice's teams.
|
||||
var adminTeam struct {
|
||||
Members []struct {
|
||||
Username string `json:"username"`
|
||||
Source string `json:"source"`
|
||||
} `json:"members"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodGet, fmt.Sprintf("/api/admin/teams/%d", sreID), nil), &adminTeam)
|
||||
if len(adminTeam.Members) != 1 || adminTeam.Members[0].Source != "oidc" {
|
||||
t.Errorf("admin team members: %+v", adminTeam.Members)
|
||||
}
|
||||
var teams []struct {
|
||||
Name string `json:"name"`
|
||||
Source string `json:"source"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodGet, fmt.Sprintf("/api/users/%d/teams", aliceID), nil), &teams)
|
||||
if len(teams) != 1 || teams[0].Name != "SRE" || teams[0].Source != "oidc" {
|
||||
t.Errorf("user teams: %+v", teams)
|
||||
}
|
||||
|
||||
// The bootstrap admin's own membership is manual.
|
||||
var mine []struct {
|
||||
Source string `json:"source"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodGet, "/api/users/1/teams", nil), &mine)
|
||||
if len(mine) == 0 || mine[0].Source != "manual" {
|
||||
t.Errorf("bootstrap admin's teams: %+v", mine)
|
||||
}
|
||||
}
|
||||
+159
-13
@@ -4,16 +4,29 @@ import (
|
||||
"database/sql"
|
||||
"net/http"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/oidc"
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/web"
|
||||
"github.com/go-chi/chi/v5"
|
||||
"github.com/go-chi/chi/v5/middleware"
|
||||
)
|
||||
|
||||
// NewRouter builds the HTTP surface. notify and deadman are passed through to
|
||||
// the webhook, the only handler that has to decide where a new incident's page
|
||||
// goes and which arriving alerts are heartbeats rather than problems. A zero
|
||||
// notify disables notifications; a zero deadman disables dead man's switches.
|
||||
func NewRouter(db *sql.DB, notify NotifyConfig, deadman DeadmanConfig) http.Handler {
|
||||
// NewRouter builds the HTTP surface. notify is passed through to the webhook,
|
||||
// the only handler that has to decide where a new incident's page goes; a zero
|
||||
// notify disables notifications. Dead man's switches are per team and read from
|
||||
// the database, so nothing about them is wired in here. version is reported
|
||||
// verbatim by GET /api/version, unauthenticated like /healthz: a client
|
||||
// deciding whether it can talk to this server — terdut-tui, terdut-operator —
|
||||
// needs to ask before it holds a credential for it, and the version is not a
|
||||
// secret.
|
||||
func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version string) http.Handler {
|
||||
// One limiter each, both process-wide for the life of the router: login
|
||||
// counts failed passwords, sign-up counts account creation, and mixing the
|
||||
// two would let a burst of sign-ups lock somebody out of logging in.
|
||||
loginLimit := newLoginLimiter()
|
||||
signupLimiter := newLoginLimiter()
|
||||
oidcLimit := newLoginLimiter()
|
||||
|
||||
r := chi.NewRouter()
|
||||
r.Use(middleware.Logger)
|
||||
r.Use(middleware.Recoverer)
|
||||
@@ -21,33 +34,109 @@ func NewRouter(db *sql.DB, notify NotifyConfig, deadman DeadmanConfig) http.Hand
|
||||
r.Get("/healthz", func(w http.ResponseWriter, r *http.Request) {
|
||||
respond(w, http.StatusOK, map[string]string{"status": "ok"})
|
||||
})
|
||||
r.Get("/api/version", func(w http.ResponseWriter, r *http.Request) {
|
||||
respond(w, http.StatusOK, map[string]string{"version": version})
|
||||
})
|
||||
|
||||
// Unauthenticated: bootstrap, the Alertmanager webhook receiver, and the
|
||||
// Acknowledge button in a push notification. The last one is authorised by
|
||||
// the scoped token in its path rather than an API key, and has to stay
|
||||
// reachable from outside the cluster for the button to work.
|
||||
r.Post("/api/bootstrap", handleBootstrap(db))
|
||||
r.Post("/api/alertmanager/webhook", handleAlertmanagerWebhook(db, notify, deadman))
|
||||
r.Post("/api/notify/ack/{token}", handleNotifyAck(db))
|
||||
|
||||
// Alert ingestion. The key in the path says both that the sender may post
|
||||
// and which team the alerts belong to, which is why it needs no session.
|
||||
//
|
||||
// This is the only way in. The pre-teams /api/alertmanager/webhook, which
|
||||
// took no credential at all, was removed in v0.13.0 once the cluster's
|
||||
// Alertmanager had moved onto a key; a sender still posting there gets the
|
||||
// JSON 404 every unknown /api path gets.
|
||||
r.Post("/api/integrations/{key}/alertmanager", handleIntegrationWebhook(db, notify))
|
||||
|
||||
// Signing up. Both are unauthenticated by necessity: the caller has no
|
||||
// account yet. The info endpoint says whether the door is open and whether
|
||||
// an invite link is good, so the form can say so before somebody picks a
|
||||
// password.
|
||||
r.Get("/api/signup", handleSignupInfo(db))
|
||||
r.With(passwordLoginOnly(!cfg.DisablePasswordLogin)).
|
||||
Post("/api/signup", handleSignup(db, signupLimiter, notify.PublicURL))
|
||||
|
||||
// How to sign in: what the login form and the TUI offer before anybody types.
|
||||
r.Get("/api/auth/config", handleAuthConfig(cfg))
|
||||
|
||||
// Signing in to the web UI. Login trades a password for a session cookie,
|
||||
// which AuthMiddleware accepts in place of an API key.
|
||||
r.Post("/api/login", handleLogin(db, newLoginLimiter(), notify.PublicURL))
|
||||
r.With(passwordLoginOnly(!cfg.DisablePasswordLogin)).
|
||||
Post("/api/login", handleLogin(db, loginLimit, notify.PublicURL))
|
||||
r.Post("/api/logout", handleLogout(db, notify.PublicURL))
|
||||
|
||||
// Single sign-on. Both routes are navigations the browser makes, to and from
|
||||
// the provider, so they answer with redirects rather than JSON.
|
||||
if cfg.OIDC.Enabled() {
|
||||
prov := oidc.New(cfg.OIDC, notify.PublicURL)
|
||||
r.Get("/api/oidc/login", handleOIDCLogin(db, prov, oidcLimit, notify.PublicURL))
|
||||
r.Get("/api/oidc/callback", handleOIDCCallback(db, prov, notify.PublicURL))
|
||||
|
||||
// Device login, for a client with no browser of its own. Both are
|
||||
// unauthenticated: the device code in the body is the credential.
|
||||
r.Post("/api/oidc/device", handleDeviceStart(db, oidcLimit, notify.PublicURL))
|
||||
r.Post("/api/oidc/device/token", handleDeviceToken(db, cfg.OIDC.SessionMaxAge, notify.PublicURL))
|
||||
}
|
||||
|
||||
// All other /api routes require a valid API key.
|
||||
r.Group(func(r chi.Router) {
|
||||
r.Use(AuthMiddleware(db))
|
||||
|
||||
r.Get("/api/me", handleMe(db))
|
||||
|
||||
// Approving or refusing a device login is done by somebody signed in
|
||||
// to a browser, and needs the same SSO configuration the flow does.
|
||||
if cfg.OIDC.Enabled() {
|
||||
r.Post("/api/oidc/device/approve", handleDeviceDecision(db, true))
|
||||
r.Post("/api/oidc/device/deny", handleDeviceDecision(db, false))
|
||||
}
|
||||
r.Put("/api/me/onboarding", handleDismissOnboarding(db))
|
||||
// Proves the topic works, which is the only part of "notifications are
|
||||
// set up" that the person holding the phone can confirm.
|
||||
r.Post("/api/me/notify/test", handleTestNotification(notify, db))
|
||||
|
||||
// Readable by anyone signed in: the queue's assignment control and the
|
||||
// on-call schedule both need to name people.
|
||||
r.Get("/api/users", handleListUsers(db))
|
||||
r.Post("/api/users", handleCreateUser(db))
|
||||
r.Delete("/api/users/{id}", handleDeleteUser(db))
|
||||
|
||||
// Your own account, or anybody's if you are an admin. The handlers call
|
||||
// requireSelfOrAdmin rather than sitting behind AdminOnly, because
|
||||
// which rule applies depends on the {id} in the path.
|
||||
r.Get("/api/users/{id}/teams", handleUserTeams(db))
|
||||
r.Put("/api/users/{id}/notify", handleSetNotifyTarget(db))
|
||||
r.Put("/api/users/{id}/password", handleSetPassword(db))
|
||||
r.Post("/api/users/{id}/api-keys", handleCreateAPIKey(db))
|
||||
r.Delete("/api/users/{id}/api-keys/{keyID}", handleDeleteAPIKey(db))
|
||||
|
||||
// Administration: who exists, and who is an administrator. Until #3
|
||||
// these were open to any authenticated caller, which meant every user
|
||||
// could delete every other one.
|
||||
r.Group(func(r chi.Router) {
|
||||
r.Use(AdminOnly)
|
||||
|
||||
r.Post("/api/users", handleCreateUser(db))
|
||||
r.Delete("/api/users/{id}", handleDeleteUser(db))
|
||||
r.Put("/api/users/{id}/admin", handleSetAdmin(db))
|
||||
r.Put("/api/users/{id}/disabled", handleSetUserDisabled(db))
|
||||
|
||||
// What exists on this server, and how it behaves. /api/teams
|
||||
// answers "what am I in"; this one answers "what is there".
|
||||
r.Get("/api/admin/teams", handleAdminListTeams(db))
|
||||
// One team and who is in it. The member list under
|
||||
// /api/teams/{id}/members stays member-only and still 404s
|
||||
// an administrator from outside; this is a different
|
||||
// question, so it is a different endpoint.
|
||||
r.Get("/api/admin/teams/{teamID}", handleAdminGetTeam(db))
|
||||
r.Get("/api/admin/settings", handleGetSettings(db, cfg))
|
||||
r.Put("/api/admin/settings", handleSetSettings(db))
|
||||
})
|
||||
|
||||
// Alerts are read-only: they are Alertmanager's record, not a work
|
||||
// queue. Everything a person does happens on the incident instead.
|
||||
r.Get("/api/alerts", handleListAlerts(db))
|
||||
@@ -57,6 +146,7 @@ func NewRouter(db *sql.DB, notify NotifyConfig, deadman DeadmanConfig) http.Hand
|
||||
r.Get("/api/incidents/{id}", handleGetIncident(db))
|
||||
r.Get("/api/incidents/{id}/alerts", handleIncidentAlerts(db))
|
||||
r.Get("/api/incidents/{id}/timeline", handleIncidentTimeline(db))
|
||||
r.Get("/api/incidents/{id}/similar", handleIncidentSimilar(db))
|
||||
r.Post("/api/incidents/{id}/acknowledge", handleIncidentAcknowledge(db))
|
||||
r.Delete("/api/incidents/{id}/acknowledge", handleIncidentUnacknowledge(db))
|
||||
r.Post("/api/incidents/{id}/resolve", handleIncidentResolve(db))
|
||||
@@ -68,10 +158,66 @@ func NewRouter(db *sql.DB, notify NotifyConfig, deadman DeadmanConfig) http.Hand
|
||||
r.Post("/api/incidents/{id}/notes", handleCreateNote(db))
|
||||
r.Delete("/api/incidents/{id}/notes/{eventID}", handleDeleteNote(db))
|
||||
|
||||
r.Post("/api/schedule", handleCreateSchedule(db))
|
||||
r.Get("/api/schedule/current", handleCurrentSchedule(db)) // must be before /{id}
|
||||
r.Get("/api/schedule", handleListSchedule(db))
|
||||
r.Delete("/api/schedule/{id}", handleDeleteSchedule(db))
|
||||
// Service accounts: a scoped, non-human credential for automation
|
||||
// (terdut-operator, most likely) that needs to manage the resources
|
||||
// below without impersonating a human user. See SERVICE-ACCOUNTS.md.
|
||||
r.Get("/api/service-accounts", handleListServiceAccounts(db))
|
||||
r.Post("/api/service-accounts", handleCreateServiceAccount(db))
|
||||
r.Post("/api/service-accounts/{id}/keys", handleCreateServiceAccountKey(db))
|
||||
r.Delete("/api/service-accounts/{id}/keys/{keyID}", handleDeleteServiceAccountKey(db))
|
||||
|
||||
// Operator mode (TERDUT_OPERATOR_MODE) makes every write below refuse a
|
||||
// human caller (a session or a user's own API key) while still letting
|
||||
// a service account through — see OperatorModeBlock. opMode is a no-op
|
||||
// wrapper when the flag is off, so this costs nothing on a server that
|
||||
// never sets it.
|
||||
opMode := OperatorModeBlock(cfg)
|
||||
|
||||
// Teams. A user sees the teams they belong to; an owner configures one.
|
||||
r.Get("/api/teams", handleListTeams(db))
|
||||
r.With(opMode).Post("/api/teams", handleCreateTeam(db))
|
||||
r.With(opMode).Put("/api/teams/{teamID}", handleRenameTeam(db))
|
||||
r.With(opMode).Delete("/api/teams/{teamID}", handleDeleteTeam(db))
|
||||
r.Get("/api/teams/{teamID}/members", handleListTeamMembers(db))
|
||||
r.Post("/api/teams/{teamID}/members", handleAddTeamMember(db))
|
||||
r.Delete("/api/teams/{teamID}/members/{userID}", handleRemoveTeamMember(db))
|
||||
|
||||
// A team's own OIDC group binding: which provider groups grant member
|
||||
// and owner access to it.
|
||||
r.Get("/api/teams/{teamID}/oidc-groups", handleGetTeamOIDCGroups(db))
|
||||
r.With(opMode).Put("/api/teams/{teamID}/oidc-groups", handleSetTeamOIDCGroups(db))
|
||||
|
||||
// Invite links into this team. Not operator-mode-gated: membership is
|
||||
// deliberately never gitops-managed (see terdut-operator's DESIGN.md
|
||||
// §4.2), so it stays editable regardless of this flag.
|
||||
r.Get("/api/teams/{teamID}/invites", handleListInvites(db))
|
||||
r.Post("/api/teams/{teamID}/invites", handleCreateInvite(db, notify.PublicURL))
|
||||
r.Delete("/api/teams/{teamID}/invites/{inviteID}", handleRevokeInvite(db))
|
||||
|
||||
// A team's escalation ladder: who is paged when nobody answers.
|
||||
r.Get("/api/teams/{teamID}/escalation", handleGetEscalation(db))
|
||||
r.With(opMode).Put("/api/teams/{teamID}/escalation", handleSetEscalation(db))
|
||||
|
||||
// A team's own dead man's switches: which of its alerts are heartbeats,
|
||||
// and how long a silence has to last before somebody is paged.
|
||||
r.Get("/api/teams/{teamID}/deadman/switches", handleListTeamDeadman(db))
|
||||
r.With(opMode).Post("/api/teams/{teamID}/deadman/switches", handleCreateTeamDeadman(db))
|
||||
r.With(opMode).Put("/api/teams/{teamID}/deadman/switches/{switchID}", handleUpdateTeamDeadman(db))
|
||||
r.With(opMode).Delete("/api/teams/{teamID}/deadman/switches/{switchID}", handleDeleteTeamDeadman(db))
|
||||
|
||||
// Integrations: where a team's alerts come in, and the key that says so.
|
||||
r.Get("/api/teams/{teamID}/integrations", handleListIntegrations(db))
|
||||
r.With(opMode).Post("/api/teams/{teamID}/integrations", handleCreateIntegration(db, notify.PublicURL))
|
||||
r.With(opMode).Patch("/api/teams/{teamID}/integrations/{integrationID}", handleRenameIntegration(db))
|
||||
r.With(opMode).Delete("/api/teams/{teamID}/integrations/{integrationID}", handleDeleteIntegration(db))
|
||||
|
||||
// The rota is per team. /api/schedule/current is the exception: it
|
||||
// answers across every team the caller is in, which is what somebody on
|
||||
// two rotas wants to see.
|
||||
r.Get("/api/schedule/current", handleCurrentSchedule(db))
|
||||
r.Post("/api/teams/{teamID}/schedule", handleCreateSchedule(db))
|
||||
r.Get("/api/teams/{teamID}/schedule", handleListSchedule(db))
|
||||
r.Delete("/api/teams/{teamID}/schedule/{id}", handleDeleteSchedule(db))
|
||||
|
||||
r.Get("/api/stats/incidents", handleStatsIncidents(db))
|
||||
r.Get("/api/stats/alerts", handleStatsAlerts(db))
|
||||
|
||||
+70
-27
@@ -12,8 +12,18 @@ import (
|
||||
"github.com/go-chi/chi/v5"
|
||||
)
|
||||
|
||||
// The schedule is per team: each team keeps its own rota, so two teams can have
|
||||
// two different people on call on the same day. Editing it is an owner's job,
|
||||
// like the rest of a team's configuration; reading it is any member's.
|
||||
func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
teamID, ok := teamParam(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
if !requireTeamOwner(w, r, teamID) {
|
||||
return
|
||||
}
|
||||
var req struct {
|
||||
UserID int64 `json:"user_id"`
|
||||
Dates []string `json:"dates"`
|
||||
@@ -43,10 +53,13 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
|
||||
}
|
||||
}
|
||||
|
||||
// Verify the user exists.
|
||||
// The person taking the shift has to be in the team: paging somebody
|
||||
// who cannot open the incident is worse than paging nobody.
|
||||
var exists int
|
||||
if err := db.QueryRowContext(r.Context(), "SELECT 1 FROM users WHERE id = $1", req.UserID).Scan(&exists); err != nil {
|
||||
respond(w, http.StatusNotFound, errResp("user not found"))
|
||||
if err := db.QueryRowContext(r.Context(),
|
||||
"SELECT 1 FROM team_members WHERE team_id = $1 AND user_id = $2",
|
||||
teamID, req.UserID).Scan(&exists); err != nil {
|
||||
respond(w, http.StatusNotFound, errResp("user is not a member of this team"))
|
||||
return
|
||||
}
|
||||
|
||||
@@ -64,13 +77,15 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
|
||||
for _, d := range req.Dates {
|
||||
if req.Replace {
|
||||
if _, err := tx.ExecContext(r.Context(),
|
||||
"DELETE FROM schedule_entries WHERE date = $1", d); err != nil {
|
||||
"DELETE FROM schedule_entries WHERE team_id = $1 AND date = $2",
|
||||
teamID, d); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
}
|
||||
if _, err := tx.ExecContext(r.Context(),
|
||||
"INSERT INTO schedule_entries (user_id, date) VALUES ($1, $2)", req.UserID, d); err != nil {
|
||||
"INSERT INTO schedule_entries (team_id, user_id, date) VALUES ($1, $2, $3)",
|
||||
teamID, req.UserID, d); err != nil {
|
||||
if isUniqueViolation(err) {
|
||||
respond(w, http.StatusConflict,
|
||||
errResp("date already assigned: "+d+" (pass replace to take it)"))
|
||||
@@ -90,7 +105,7 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
|
||||
for _, d := range req.Dates {
|
||||
dateSet[d] = true
|
||||
}
|
||||
all, err := scheduleRange(r.Context(), db, req.Dates[0], req.Dates[len(req.Dates)-1])
|
||||
all, err := scheduleRange(r.Context(), db, teamID, req.Dates[0], req.Dates[len(req.Dates)-1])
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
@@ -107,6 +122,13 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
|
||||
|
||||
func handleListSchedule(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
teamID, ok := teamParam(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
if !requireTeamMember(w, r, teamID) {
|
||||
return
|
||||
}
|
||||
q := r.URL.Query()
|
||||
from, to := q.Get("from"), q.Get("to")
|
||||
|
||||
@@ -123,7 +145,7 @@ func handleListSchedule(db *sql.DB) http.HandlerFunc {
|
||||
}
|
||||
}
|
||||
|
||||
entries, err := scheduleRange(r.Context(), db, from, to)
|
||||
entries, err := scheduleRange(r.Context(), db, teamID, from, to)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
@@ -134,12 +156,20 @@ func handleListSchedule(db *sql.DB) http.HandlerFunc {
|
||||
|
||||
func handleDeleteSchedule(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
teamID, ok := teamParam(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
if !requireTeamOwner(w, r, teamID) {
|
||||
return
|
||||
}
|
||||
id, err := strconv.ParseInt(chi.URLParam(r, "id"), 10, 64)
|
||||
if err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid schedule id"))
|
||||
return
|
||||
}
|
||||
res, err := db.ExecContext(r.Context(), "DELETE FROM schedule_entries WHERE id = $1", id)
|
||||
res, err := db.ExecContext(r.Context(),
|
||||
"DELETE FROM schedule_entries WHERE id = $1 AND team_id = $2", id, teamID)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
@@ -152,35 +182,50 @@ func handleDeleteSchedule(db *sql.DB) http.HandlerFunc {
|
||||
}
|
||||
}
|
||||
|
||||
// handleCurrentSchedule answers "who is on call right now" for every team the
|
||||
// caller belongs to — one entry per team, so somebody on two rotas sees both.
|
||||
// A team with nobody scheduled today simply does not appear.
|
||||
func handleCurrentSchedule(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
today := time.Now().UTC().Format("2006-01-02")
|
||||
|
||||
var e models.ScheduleEntry
|
||||
var ts int64
|
||||
err := db.QueryRowContext(r.Context(), `
|
||||
SELECT s.id, s.user_id, u.username, s.date, s.created_at
|
||||
rows, err := db.QueryContext(r.Context(), `
|
||||
SELECT s.id, s.team_id, t.name, s.user_id, u.username, s.date, s.created_at
|
||||
FROM schedule_entries s
|
||||
JOIN users u ON u.id = s.user_id
|
||||
WHERE s.date = $1`, today).Scan(&e.ID, &e.UserID, &e.Username, &e.Date, &ts)
|
||||
if err == sql.ErrNoRows {
|
||||
respond(w, http.StatusNotFound, errResp("no one is on call today"))
|
||||
return
|
||||
}
|
||||
JOIN teams t ON t.id = s.team_id
|
||||
WHERE s.date = $1 AND s.team_id = ANY($2)
|
||||
ORDER BY t.name`, today, callerTeamIDs(r.Context()))
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
e.CreatedAt = time.Unix(ts, 0).UTC()
|
||||
respond(w, http.StatusOK, e)
|
||||
defer rows.Close()
|
||||
|
||||
entries := []models.ScheduleEntry{}
|
||||
for rows.Next() {
|
||||
var e models.ScheduleEntry
|
||||
var ts int64
|
||||
if err := rows.Scan(&e.ID, &e.TeamID, &e.TeamName, &e.UserID, &e.Username, &e.Date, &ts); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
e.CreatedAt = time.Unix(ts, 0).UTC()
|
||||
entries = append(entries, e)
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusOK, entries)
|
||||
}
|
||||
}
|
||||
|
||||
// scheduleRange returns schedule entries ordered by date.
|
||||
// from and to are YYYY-MM-DD strings; an empty string means unbounded on that side.
|
||||
func scheduleRange(ctx context.Context, db *sql.DB, from, to string) ([]models.ScheduleEntry, error) {
|
||||
where := []string{}
|
||||
func scheduleRange(ctx context.Context, db *sql.DB, teamID int64, from, to string) ([]models.ScheduleEntry, error) {
|
||||
args := &sqlArgs{}
|
||||
where := []string{"s.team_id = " + args.add(teamID)}
|
||||
if from != "" {
|
||||
where = append(where, "s.date >= "+args.add(from))
|
||||
}
|
||||
@@ -188,15 +233,13 @@ func scheduleRange(ctx context.Context, db *sql.DB, from, to string) ([]models.S
|
||||
where = append(where, "s.date <= "+args.add(to))
|
||||
}
|
||||
|
||||
clause := "1=1"
|
||||
if len(where) > 0 {
|
||||
clause = strings.Join(where, " AND ")
|
||||
}
|
||||
clause := strings.Join(where, " AND ")
|
||||
|
||||
rows, err := db.QueryContext(ctx, `
|
||||
SELECT s.id, s.user_id, u.username, s.date, s.created_at
|
||||
SELECT s.id, s.team_id, t.name, s.user_id, u.username, s.date, s.created_at
|
||||
FROM schedule_entries s
|
||||
JOIN users u ON u.id = s.user_id
|
||||
JOIN teams t ON t.id = s.team_id
|
||||
WHERE `+clause+`
|
||||
ORDER BY s.date ASC`, args.all()...)
|
||||
if err != nil {
|
||||
@@ -208,7 +251,7 @@ func scheduleRange(ctx context.Context, db *sql.DB, from, to string) ([]models.S
|
||||
for rows.Next() {
|
||||
var e models.ScheduleEntry
|
||||
var ts int64
|
||||
if err := rows.Scan(&e.ID, &e.UserID, &e.Username, &e.Date, &ts); err != nil {
|
||||
if err := rows.Scan(&e.ID, &e.TeamID, &e.TeamName, &e.UserID, &e.Username, &e.Date, &ts); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
e.CreatedAt = time.Unix(ts, 0).UTC()
|
||||
|
||||
@@ -0,0 +1,350 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||
"github.com/go-chi/chi/v5"
|
||||
)
|
||||
|
||||
// serviceAccountKeyPrefix marks a service-account key visibly, in logs and at
|
||||
// a glance, distinct from a user's own personal API key. It carries no
|
||||
// meaning to the server itself — the hash is looked up the same way either
|
||||
// kind of key is — it exists entirely for whoever is reading a log line or an
|
||||
// audit trail.
|
||||
const serviceAccountKeyPrefix = "tdsa_"
|
||||
|
||||
// randomServiceAccountToken is randomToken with serviceAccountKeyPrefix on the
|
||||
// raw value, hashed as a whole: the prefix is not a fixed header stripped
|
||||
// before hashing, it is part of the secret, the same as if it had been
|
||||
// generated that long to begin with.
|
||||
func randomServiceAccountToken() (raw, hash string, err error) {
|
||||
body, _, err := randomToken()
|
||||
if err != nil {
|
||||
return "", "", err
|
||||
}
|
||||
raw = serviceAccountKeyPrefix + body
|
||||
return raw, hashToken(raw), nil
|
||||
}
|
||||
|
||||
// callerIsAdmin reports whether the caller is a signed-in human system
|
||||
// administrator. A service account never is, by design (SERVICE-ACCOUNTS.md):
|
||||
// account and user management stays human-only, service accounts included.
|
||||
func callerIsAdmin(ctx context.Context) bool {
|
||||
u, ok := userFromContext(ctx)
|
||||
return ok && u.IsAdmin
|
||||
}
|
||||
|
||||
// callerOwnsTeam reports whether the caller is owner-equivalent for teamID:
|
||||
// a human owner, or that team's own team-scoped service account (its single
|
||||
// synthetic membership, serveAsServiceAccount — ratified in
|
||||
// SERVICE-ACCOUNTS.md as intentional, not an accident: a team-scoped
|
||||
// credential is that team's owner's reach, full stop, membership and
|
||||
// invites included). Built on callerRole like requireTeamOwner, but without
|
||||
// writing a response: callers here need to combine it with other ways of
|
||||
// being allowed, not stop at the first no.
|
||||
func callerOwnsTeam(ctx context.Context, teamID int64) bool {
|
||||
role, ok := callerRole(ctx, teamID)
|
||||
return ok && role == models.RoleOwner
|
||||
}
|
||||
|
||||
// handleCreateServiceAccount creates a service account and mints its first
|
||||
// key. Who may do this depends on scope: an instance-scoped account (which
|
||||
// can in turn create a team and a team-scoped account for it) is system
|
||||
// administration's own reach extended to automation, so only a human admin
|
||||
// grants one. A team-scoped account is that team's owner's reach, so a human
|
||||
// admin, the target team's own human owner, or an existing instance-scoped
|
||||
// service account (minting itself a narrower credential for a team it just
|
||||
// created) may create one.
|
||||
func handleCreateServiceAccount(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
var req struct {
|
||||
Name string `json:"name"`
|
||||
Scope string `json:"scope"`
|
||||
TeamID int64 `json:"team_id"`
|
||||
}
|
||||
if err := decodeJSON(r, &req); err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||
return
|
||||
}
|
||||
req.Name = strings.TrimSpace(req.Name)
|
||||
if req.Name == "" {
|
||||
respond(w, http.StatusBadRequest, errResp("name is required"))
|
||||
return
|
||||
}
|
||||
if req.Scope != models.ServiceAccountScopeInstance && req.Scope != models.ServiceAccountScopeTeam {
|
||||
respond(w, http.StatusBadRequest, errResp("scope must be instance or team"))
|
||||
return
|
||||
}
|
||||
if req.Scope == models.ServiceAccountScopeTeam && req.TeamID == 0 {
|
||||
respond(w, http.StatusBadRequest, errResp("team_id is required for a team-scoped account"))
|
||||
return
|
||||
}
|
||||
if req.Scope == models.ServiceAccountScopeInstance && req.TeamID != 0 {
|
||||
respond(w, http.StatusBadRequest, errResp("team_id must not be set for an instance-scoped account"))
|
||||
return
|
||||
}
|
||||
|
||||
allowed := callerIsAdmin(r.Context())
|
||||
if !allowed && req.Scope == models.ServiceAccountScopeTeam {
|
||||
allowed = callerOwnsTeam(r.Context(), req.TeamID) || isInstanceServiceAccount(r.Context())
|
||||
}
|
||||
if !allowed {
|
||||
respond(w, http.StatusForbidden, errResp("team owner, system administrator, or instance-scoped service account access required"))
|
||||
return
|
||||
}
|
||||
|
||||
var callerUserID *int64
|
||||
if u, ok := userFromContext(r.Context()); ok {
|
||||
id := u.ID
|
||||
callerUserID = &id
|
||||
}
|
||||
var teamID *int64
|
||||
if req.Scope == models.ServiceAccountScopeTeam {
|
||||
teamID = &req.TeamID
|
||||
}
|
||||
|
||||
var sa models.ServiceAccount
|
||||
var created int64
|
||||
if err := db.QueryRowContext(r.Context(), `
|
||||
INSERT INTO service_accounts (name, scope, team_id, created_by)
|
||||
VALUES ($1, $2, $3, $4)
|
||||
RETURNING id, name, scope, team_id, created_by, created_at`,
|
||||
req.Name, req.Scope, teamID, callerUserID,
|
||||
).Scan(&sa.ID, &sa.Name, &sa.Scope, &sa.TeamID, &sa.CreatedBy, &created); err != nil {
|
||||
if isUniqueViolation(err) {
|
||||
respond(w, http.StatusConflict, errResp("a service account with that name already exists"))
|
||||
return
|
||||
}
|
||||
// The only foreign key that can fail here is team_id: an
|
||||
// instance-scoped caller is not otherwise checked against it
|
||||
// (callerOwnsTeam already proved it exists for a human owner).
|
||||
respond(w, http.StatusBadRequest, errResp("unknown team_id"))
|
||||
return
|
||||
}
|
||||
sa.CreatedAt = time.Unix(created, 0).UTC()
|
||||
|
||||
key, err := mintServiceAccountKey(r.Context(), db, sa.ID, "initial")
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusCreated, map[string]any{"service_account": sa, "key": key})
|
||||
}
|
||||
}
|
||||
|
||||
// mintServiceAccountKey inserts one key for an existing account and returns
|
||||
// it with its raw value populated — the one moment that value exists outside
|
||||
// the request that generated it.
|
||||
func mintServiceAccountKey(ctx context.Context, db *sql.DB, serviceAccountID int64, name string) (models.ServiceAccountKey, error) {
|
||||
raw, hash, err := randomServiceAccountToken()
|
||||
if err != nil {
|
||||
return models.ServiceAccountKey{}, err
|
||||
}
|
||||
var key models.ServiceAccountKey
|
||||
var created int64
|
||||
if err := db.QueryRowContext(ctx, `
|
||||
INSERT INTO service_account_keys (service_account_id, key_hash, name)
|
||||
VALUES ($1, $2, $3)
|
||||
RETURNING id, service_account_id, name, created_at`,
|
||||
serviceAccountID, hash, name,
|
||||
).Scan(&key.ID, &key.ServiceAccountID, &key.Name, &created); err != nil {
|
||||
return models.ServiceAccountKey{}, err
|
||||
}
|
||||
key.CreatedAt = time.Unix(created, 0).UTC()
|
||||
key.Key = raw
|
||||
return key, nil
|
||||
}
|
||||
|
||||
func fetchServiceAccount(ctx context.Context, db *sql.DB, id int64) (models.ServiceAccount, error) {
|
||||
var sa models.ServiceAccount
|
||||
var created int64
|
||||
err := db.QueryRowContext(ctx,
|
||||
"SELECT id, name, scope, team_id, created_by, created_at FROM service_accounts WHERE id = $1", id,
|
||||
).Scan(&sa.ID, &sa.Name, &sa.Scope, &sa.TeamID, &sa.CreatedBy, &created)
|
||||
if err != nil {
|
||||
return sa, err
|
||||
}
|
||||
sa.CreatedAt = time.Unix(created, 0).UTC()
|
||||
return sa, nil
|
||||
}
|
||||
|
||||
// callerMayManageServiceAccount reports whether the caller may mint or revoke
|
||||
// a key on sa: a system administrator, that team-scoped account's own human
|
||||
// owner, the account rotating its own credential (not a privilege
|
||||
// escalation, the same reasoning requireSelfOrAdmin already rests on for a
|
||||
// user's own API keys) — or, new, an instance-scoped service account
|
||||
// managing any team-scoped account.
|
||||
//
|
||||
// That last branch closes terdut-operator#3: handleCreateServiceAccount
|
||||
// already lets an instance-scoped caller *create* a team-scoped account for
|
||||
// any team (the branch below it, isInstanceServiceAccount(ctx)) — this
|
||||
// account didn't have an equivalent reach to *adopt or rotate* one it
|
||||
// didn't just create in the same call, which is exactly the recovery path
|
||||
// terdut-operator's own documented crash-window handling depends on
|
||||
// (DESIGN.md §5's general adopt-on-conflict rule): a reconcile that creates
|
||||
// the account successfully but crashes before persisting its credential
|
||||
// locally retries into a 409, and without this branch the only available
|
||||
// recovery — minting a fresh key on the now-existing account — 403'd
|
||||
// forever, with no way out. Granting it here is not a new power: it
|
||||
// mirrors the create-time reach this scope already has, just extended to
|
||||
// the retry path DESIGN.md's own crash-window reasoning requires.
|
||||
func callerMayManageServiceAccount(ctx context.Context, sa models.ServiceAccount) bool {
|
||||
if callerIsAdmin(ctx) {
|
||||
return true
|
||||
}
|
||||
if sa.TeamID != nil && callerOwnsTeam(ctx, *sa.TeamID) {
|
||||
return true
|
||||
}
|
||||
caller, _ := callerFromContext(ctx)
|
||||
if id, ok := caller.ServiceAccountID(); ok && id == sa.ID {
|
||||
return true
|
||||
}
|
||||
if sa.TeamID != nil && caller.IsInstanceServiceAccount() {
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func serviceAccountParam(w http.ResponseWriter, r *http.Request) (int64, bool) {
|
||||
id, err := strconv.ParseInt(chi.URLParam(r, "id"), 10, 64)
|
||||
if err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid service account id"))
|
||||
return 0, false
|
||||
}
|
||||
return id, true
|
||||
}
|
||||
|
||||
func handleCreateServiceAccountKey(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, ok := serviceAccountParam(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
sa, err := fetchServiceAccount(r.Context(), db, id)
|
||||
if errors.Is(err, sql.ErrNoRows) {
|
||||
respond(w, http.StatusNotFound, errResp("service account not found"))
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if !callerMayManageServiceAccount(r.Context(), sa) {
|
||||
respond(w, http.StatusForbidden, errResp("team owner, system administrator, or the account itself may rotate its key"))
|
||||
return
|
||||
}
|
||||
|
||||
var req struct {
|
||||
Name string `json:"name"`
|
||||
}
|
||||
if err := decodeJSON(r, &req); err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||
return
|
||||
}
|
||||
if req.Name == "" {
|
||||
respond(w, http.StatusBadRequest, errResp("name is required"))
|
||||
return
|
||||
}
|
||||
|
||||
key, err := mintServiceAccountKey(r.Context(), db, sa.ID, req.Name)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusCreated, key)
|
||||
}
|
||||
}
|
||||
|
||||
func handleDeleteServiceAccountKey(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, ok := serviceAccountParam(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
sa, err := fetchServiceAccount(r.Context(), db, id)
|
||||
if errors.Is(err, sql.ErrNoRows) {
|
||||
respond(w, http.StatusNotFound, errResp("service account not found"))
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if !callerMayManageServiceAccount(r.Context(), sa) {
|
||||
respond(w, http.StatusForbidden, errResp("team owner, system administrator, or the account itself may revoke its key"))
|
||||
return
|
||||
}
|
||||
keyID, err := strconv.ParseInt(chi.URLParam(r, "keyID"), 10, 64)
|
||||
if err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid key id"))
|
||||
return
|
||||
}
|
||||
|
||||
res, err := db.ExecContext(r.Context(),
|
||||
"DELETE FROM service_account_keys WHERE id = $1 AND service_account_id = $2", keyID, sa.ID)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if n, _ := res.RowsAffected(); n == 0 {
|
||||
respond(w, http.StatusNotFound, errResp("key not found"))
|
||||
return
|
||||
}
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}
|
||||
}
|
||||
|
||||
// handleListServiceAccounts lists every service account, or looks one up by
|
||||
// its exact name with ?name=. The name lookup is open to any authenticated
|
||||
// caller, human or service account: it returns no key material, and it is
|
||||
// what lets a service account find its own account on the 403 that follows a
|
||||
// second POST — the self-registration pattern SERVICE-ACCOUNTS.md describes.
|
||||
// Listing everything, with no filter, stays administrator-only.
|
||||
func handleListServiceAccounts(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
name := strings.TrimSpace(r.URL.Query().Get("name"))
|
||||
if name == "" && !callerIsAdmin(r.Context()) {
|
||||
respond(w, http.StatusForbidden, errResp("administrator access required to list every service account; pass ?name= to look up one by name"))
|
||||
return
|
||||
}
|
||||
|
||||
query := "SELECT id, name, scope, team_id, created_by, created_at FROM service_accounts"
|
||||
var args []any
|
||||
if name != "" {
|
||||
query += " WHERE name = $1"
|
||||
args = append(args, name)
|
||||
}
|
||||
query += " ORDER BY id"
|
||||
|
||||
rows, err := db.QueryContext(r.Context(), query, args...)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
accounts := []models.ServiceAccount{}
|
||||
for rows.Next() {
|
||||
var sa models.ServiceAccount
|
||||
var created int64
|
||||
if err := rows.Scan(&sa.ID, &sa.Name, &sa.Scope, &sa.TeamID, &sa.CreatedBy, &created); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
sa.CreatedAt = time.Unix(created, 0).UTC()
|
||||
accounts = append(accounts, sa)
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusOK, accounts)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,492 @@
|
||||
package api_test
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||
)
|
||||
|
||||
// reqAs is s.req with an arbitrary bearer credential in place of the admin's
|
||||
// own key, for exercising a service account's or another user's key.
|
||||
func (s *ts) reqAs(t *testing.T, key, method, path string, body any) *http.Response {
|
||||
t.Helper()
|
||||
var r io.Reader
|
||||
if body != nil {
|
||||
data, _ := json.Marshal(body)
|
||||
r = bytes.NewReader(data)
|
||||
}
|
||||
req, _ := http.NewRequest(method, s.URL+path, r)
|
||||
req.Header.Set("Authorization", "Bearer "+key)
|
||||
if body != nil {
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
}
|
||||
resp, err := http.DefaultClient.Do(req)
|
||||
if err != nil {
|
||||
t.Fatalf("%s %s: %v", method, path, err)
|
||||
}
|
||||
return resp
|
||||
}
|
||||
|
||||
// createServiceAccount creates a service account as callerKey and returns its
|
||||
// freshly minted raw key.
|
||||
func createServiceAccount(t *testing.T, s *ts, callerKey, name, scope string, teamID int64) string {
|
||||
t.Helper()
|
||||
body := map[string]any{"name": name, "scope": scope}
|
||||
if teamID != 0 {
|
||||
body["team_id"] = teamID
|
||||
}
|
||||
resp := s.reqAs(t, callerKey, http.MethodPost, "/api/service-accounts", body)
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
resp.Body.Close()
|
||||
t.Fatalf("create service account %s: %d", name, resp.StatusCode)
|
||||
}
|
||||
var result struct {
|
||||
Key struct {
|
||||
Key string `json:"key"`
|
||||
} `json:"key"`
|
||||
}
|
||||
decode(t, resp, &result)
|
||||
if result.Key.Key == "" {
|
||||
t.Fatalf("create service account %s: no key returned", name)
|
||||
}
|
||||
return result.Key.Key
|
||||
}
|
||||
|
||||
// createTeamAs creates a team as callerKey and returns its id.
|
||||
func createTeamAs(t *testing.T, s *ts, callerKey, name string) int64 {
|
||||
t.Helper()
|
||||
resp := s.reqAs(t, callerKey, http.MethodPost, "/api/teams", map[string]string{"name": name})
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
resp.Body.Close()
|
||||
t.Fatalf("create team %s: %d", name, resp.StatusCode)
|
||||
}
|
||||
var team struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
decode(t, resp, &team)
|
||||
return team.ID
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Instance scope
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func TestServiceAccount_InstanceScopeCreatesTeamWithNoHumanOwner(t *testing.T) {
|
||||
s := newTS(t)
|
||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||
|
||||
if !strings.HasPrefix(instanceKey, "tdsa_") {
|
||||
t.Errorf("expected a service-account key to carry the tdsa_ prefix, got %q", instanceKey)
|
||||
}
|
||||
|
||||
resp := s.reqAs(t, instanceKey, http.MethodPost, "/api/teams", map[string]string{"name": "provisioned"})
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("instance-scoped account creating a team: %d", resp.StatusCode)
|
||||
}
|
||||
var team struct {
|
||||
ID int64 `json:"id"`
|
||||
Role string `json:"role"`
|
||||
}
|
||||
decode(t, resp, &team)
|
||||
if team.Role != "" {
|
||||
t.Errorf("expected no role on a team a service account created (no human owner), got %q", team.Role)
|
||||
}
|
||||
|
||||
// It still exists, visible to an administrator, even with no member.
|
||||
var admin []map[string]any
|
||||
decode(t, s.req(t, http.MethodGet, "/api/admin/teams", nil), &admin)
|
||||
found := false
|
||||
for _, tm := range admin {
|
||||
if int64(tm["id"].(float64)) == team.ID {
|
||||
found = true
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Errorf("expected the service-account-created team to appear in /api/admin/teams")
|
||||
}
|
||||
}
|
||||
|
||||
func TestServiceAccount_TeamScopeCannotCreateTeam(t *testing.T) {
|
||||
s := newTS(t)
|
||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||
keyA := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||
|
||||
resp := s.reqAs(t, keyA, http.MethodPost, "/api/teams", map[string]string{"name": "should-fail"})
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("expected 403, a team-scoped account creating a team, got %d", resp.StatusCode)
|
||||
}
|
||||
resp.Body.Close()
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Team scope
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// The whole point of team scope: bound to its own team, refused everywhere
|
||||
// else, the same as an instance-scoped account minting a key per TerdutTeam
|
||||
// rather than sharing one server-admin-equivalent credential would need.
|
||||
func TestServiceAccount_TeamScopeIsBoundToItsOwnTeam(t *testing.T) {
|
||||
s := newTS(t)
|
||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||
|
||||
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||
teamB := createTeamAs(t, s, instanceKey, "team-b")
|
||||
keyA := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||
|
||||
policy := map[string]any{"repeat_count": 0, "fallback_topic": "", "levels": []any{}}
|
||||
|
||||
resp := s.reqAs(t, keyA, http.MethodPut, "/api/teams/"+id64(teamA)+"/escalation", policy)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("team-a's own key setting its escalation: %d", resp.StatusCode)
|
||||
}
|
||||
resp.Body.Close()
|
||||
|
||||
// 404, not 403: the same "does this exist" refusal a human non-member
|
||||
// gets from requireTeamMember, not a distinguishable "you may not".
|
||||
resp2 := s.reqAs(t, keyA, http.MethodPut, "/api/teams/"+id64(teamB)+"/escalation", policy)
|
||||
if resp2.StatusCode != http.StatusNotFound {
|
||||
t.Errorf("expected 404 reaching into another team, got %d", resp2.StatusCode)
|
||||
}
|
||||
resp2.Body.Close()
|
||||
}
|
||||
|
||||
// Team scope is owner-equivalent broadly (SERVICE-ACCOUNTS.md), not limited to
|
||||
// one endpoint: escalation, dead man's switches and integrations all work.
|
||||
func TestServiceAccount_TeamScopeManagesItsResources(t *testing.T) {
|
||||
s := newTS(t)
|
||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||
keyA := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||
|
||||
resp := s.reqAs(t, keyA, http.MethodPost, "/api/teams/"+id64(teamA)+"/deadman/switches",
|
||||
map[string]any{"matcher": "alertname=Watchdog", "timeout_seconds": 900, "severity": "critical"})
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Errorf("team-scoped account creating a dead man's switch: %d", resp.StatusCode)
|
||||
}
|
||||
resp.Body.Close()
|
||||
|
||||
resp2 := s.reqAs(t, keyA, http.MethodPost, "/api/teams/"+id64(teamA)+"/integrations",
|
||||
map[string]string{"name": "prod"})
|
||||
if resp2.StatusCode != http.StatusCreated {
|
||||
t.Errorf("team-scoped account creating an integration: %d", resp2.StatusCode)
|
||||
}
|
||||
resp2.Body.Close()
|
||||
}
|
||||
|
||||
// Documents the capability already granted at create time (handleCreateServiceAccount's
|
||||
// own callerOwnsTeam branch) also applies here: a team-scoped account is that
|
||||
// team's owner's reach, membership and further accounts included, not just
|
||||
// the handful of endpoints exercised above. Kept, not restricted, for
|
||||
// symmetry with the now-ratified membership/invite capability below.
|
||||
func TestServiceAccount_TeamScopeCanMintAnotherAccountForItsOwnTeam(t *testing.T) {
|
||||
s := newTS(t)
|
||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||
keyA := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||
|
||||
resp := s.reqAs(t, keyA, http.MethodPost, "/api/service-accounts",
|
||||
map[string]any{"name": "team-a-sa-2", "scope": models.ServiceAccountScopeTeam, "team_id": teamA})
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Errorf("team-scoped account minting another account for its own team: %d", resp.StatusCode)
|
||||
}
|
||||
resp.Body.Close()
|
||||
}
|
||||
|
||||
// SERVICE-ACCOUNTS.md ratifies this explicitly: a team-scoped account is
|
||||
// owner-equivalent for every requireTeamOwner endpoint, membership and
|
||||
// invites included — terdut-operator's own invite-minting feature depends on
|
||||
// exactly this. No test exercised handleCreateInvite from a service account
|
||||
// before this change, and it would have 500'd (created_by written as a bare
|
||||
// zero value against a NOT-validated-but-FK'd column) rather than succeeded;
|
||||
// see the signup_test.go addition for that half.
|
||||
func TestServiceAccount_TeamScopeManagesItsOwnInvites(t *testing.T) {
|
||||
s := newTS(t)
|
||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||
keyA := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||
|
||||
var invite struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
resp := s.reqAs(t, keyA, http.MethodPost, "/api/teams/"+id64(teamA)+"/invites", map[string]any{})
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("team-scoped account creating an invite: %d", resp.StatusCode)
|
||||
}
|
||||
decode(t, resp, &invite)
|
||||
|
||||
var list []map[string]any
|
||||
decode(t, s.reqAs(t, keyA, http.MethodGet, "/api/teams/"+id64(teamA)+"/invites", nil), &list)
|
||||
if len(list) != 1 {
|
||||
t.Errorf("expected the invite to list back, got %d", len(list))
|
||||
}
|
||||
|
||||
if resp := s.reqAs(t, keyA, http.MethodDelete,
|
||||
"/api/teams/"+id64(teamA)+"/invites/"+id64(invite.ID), nil); resp.StatusCode != http.StatusNoContent {
|
||||
t.Errorf("team-scoped account revoking its own invite: %d", resp.StatusCode)
|
||||
} else {
|
||||
resp.Body.Close()
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Key rotation
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// terdut-operator#3: an instance-scoped account is already trusted to CREATE
|
||||
// a team-scoped account for any team (handleCreateServiceAccount's own
|
||||
// isInstanceServiceAccount branch) — this pins that it is equally trusted to
|
||||
// manage/rotate a key on one that already exists and that it did not just
|
||||
// create in this call, which is the exact shape of terdut-operator's own
|
||||
// crash-window recovery (mint succeeds, a later step is interrupted before
|
||||
// persisting the credential locally, and the next reconcile retries into a
|
||||
// 409 then needs to mint a fresh key on the now-existing account). Before
|
||||
// this fix, the second POST .../keys below 403'd forever.
|
||||
func TestServiceAccount_InstanceScopeAdoptsAnExistingTeamScopedAccountsKey(t *testing.T) {
|
||||
s := newTS(t)
|
||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||
|
||||
resp := s.reqAs(t, instanceKey, http.MethodPost, "/api/service-accounts",
|
||||
map[string]any{"name": "team-a-sa", "scope": models.ServiceAccountScopeTeam, "team_id": teamA})
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("create team-scoped account: %d", resp.StatusCode)
|
||||
}
|
||||
var created struct {
|
||||
ServiceAccount struct {
|
||||
ID int64 `json:"id"`
|
||||
} `json:"service_account"`
|
||||
}
|
||||
decode(t, resp, &created)
|
||||
|
||||
// Simulates the adopt-on-409 recovery path: this instance-scoped caller
|
||||
// did not just create this account in this call (a fresh *tdclient.Client
|
||||
// request, same as a second, independent reconcile would issue), yet
|
||||
// still needs to mint it a fresh key.
|
||||
rotateResp := s.reqAs(t, instanceKey, http.MethodPost,
|
||||
"/api/service-accounts/"+id64(created.ServiceAccount.ID)+"/keys", map[string]string{"name": "adopted"})
|
||||
if rotateResp.StatusCode != http.StatusCreated {
|
||||
t.Errorf("instance-scoped account adopting a team-scoped account's key: %d", rotateResp.StatusCode)
|
||||
}
|
||||
rotateResp.Body.Close()
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// AdminOnly / requireSelfOrAdmin — unchanged after the Caller refactor
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// The Caller abstraction must not have widened AdminOnly/requireSelfOrAdmin:
|
||||
// user management and /api/admin/settings stay human-only, for every scope
|
||||
// of service account, exactly as before.
|
||||
func TestAdminOnly_RefusesEveryServiceAccountScope(t *testing.T) {
|
||||
s := newTS(t)
|
||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||
teamKey := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||
|
||||
for _, key := range []string{instanceKey, teamKey} {
|
||||
if resp := s.reqAs(t, key, http.MethodGet, "/api/admin/settings", nil); resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("expected 403 for a service account reading /api/admin/settings, got %d", resp.StatusCode)
|
||||
} else {
|
||||
resp.Body.Close()
|
||||
}
|
||||
if resp := s.reqAs(t, key, http.MethodPost, "/api/users",
|
||||
map[string]string{"username": "nope", "email": "nope@example.com"}); resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("expected 403 for a service account creating a user, got %d", resp.StatusCode)
|
||||
} else {
|
||||
resp.Body.Close()
|
||||
}
|
||||
if resp := s.reqAs(t, key, http.MethodGet, "/api/me", nil); resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("expected 403 for a service account calling /api/me, got %d", resp.StatusCode)
|
||||
} else {
|
||||
resp.Body.Close()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestServiceAccount_SelfRotatesItsOwnKey(t *testing.T) {
|
||||
s := newTS(t)
|
||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||
|
||||
// Self-lookup by name, the pattern that turns /api/bootstrap's 403 into a
|
||||
// normal flow instead of an unhandled error.
|
||||
var accounts []map[string]any
|
||||
decode(t, s.reqAs(t, instanceKey, http.MethodGet, "/api/service-accounts?name=terdut-operator", nil), &accounts)
|
||||
if len(accounts) != 1 {
|
||||
t.Fatalf("expected exactly one match for ?name=terdut-operator, got %d", len(accounts))
|
||||
}
|
||||
id := int64(accounts[0]["id"].(float64))
|
||||
|
||||
resp := s.reqAs(t, instanceKey, http.MethodPost, "/api/service-accounts/"+id64(id)+"/keys",
|
||||
map[string]string{"name": "rotated"})
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("self-rotation: %d", resp.StatusCode)
|
||||
}
|
||||
var newKey struct {
|
||||
Key string `json:"key"`
|
||||
}
|
||||
decode(t, resp, &newKey)
|
||||
|
||||
if resp := s.reqAs(t, newKey.Key, http.MethodPost, "/api/teams", map[string]string{"name": "after-rotation"}); resp.StatusCode != http.StatusCreated {
|
||||
t.Errorf("expected the newly rotated key to work, got %d", resp.StatusCode)
|
||||
} else {
|
||||
resp.Body.Close()
|
||||
}
|
||||
|
||||
// Rotation adds a key, it does not itself revoke the old one.
|
||||
if resp := s.reqAs(t, instanceKey, http.MethodGet, "/api/service-accounts?name=terdut-operator", nil); resp.StatusCode != http.StatusOK {
|
||||
t.Errorf("expected the original key to still work until explicitly revoked, got %d", resp.StatusCode)
|
||||
} else {
|
||||
resp.Body.Close()
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Operator mode
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// Operator mode is exercised against a second router over an
|
||||
// already-configured database, rather than turning it on for newTSWith's own
|
||||
// setup: that setup creates the default integration with the admin's (human)
|
||||
// key, which is precisely the write operator mode exists to refuse, and in
|
||||
// the real deployment this flag targets that setup was never done by a human
|
||||
// to begin with — the operator itself would have provisioned it.
|
||||
func TestOperatorMode_BlocksHumanWritesButAllowsServiceAccounts(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
conf := testConfig()
|
||||
conf.OperatorMode = true
|
||||
opSrv := httptest.NewServer(api.NewRouter(s.db, s.notify, conf, "test"))
|
||||
t.Cleanup(opSrv.Close)
|
||||
do := func(key, method, path string, body any) *http.Response {
|
||||
t.Helper()
|
||||
var r io.Reader
|
||||
if body != nil {
|
||||
data, _ := json.Marshal(body)
|
||||
r = bytes.NewReader(data)
|
||||
}
|
||||
req, _ := http.NewRequest(method, opSrv.URL+path, r)
|
||||
req.Header.Set("Authorization", "Bearer "+key)
|
||||
if body != nil {
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
}
|
||||
resp, err := http.DefaultClient.Do(req)
|
||||
if err != nil {
|
||||
t.Fatalf("%s %s: %v", method, path, err)
|
||||
}
|
||||
return resp
|
||||
}
|
||||
|
||||
// The bootstrap admin's own key is a human credential: refused.
|
||||
resp := do(s.key, http.MethodPost, "/api/teams", map[string]string{"name": "human-team"})
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Fatalf("expected 403 for a human write under operator mode, got %d", resp.StatusCode)
|
||||
}
|
||||
var refusal map[string]string
|
||||
decode(t, resp, &refusal)
|
||||
if refusal["reason"] != "operator_managed" {
|
||||
t.Errorf("expected reason=operator_managed, got %q", refusal["reason"])
|
||||
}
|
||||
|
||||
// Creating the service account itself is not gated by operator mode —
|
||||
// it is how an operator identifies itself, not one of the resources it
|
||||
// manages.
|
||||
resp2 := do(s.key, http.MethodPost, "/api/service-accounts",
|
||||
map[string]any{"name": "terdut-operator", "scope": models.ServiceAccountScopeInstance})
|
||||
if resp2.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("create service account under operator mode: %d", resp2.StatusCode)
|
||||
}
|
||||
var result struct {
|
||||
Key struct {
|
||||
Key string `json:"key"`
|
||||
} `json:"key"`
|
||||
}
|
||||
decode(t, resp2, &result)
|
||||
|
||||
resp3 := do(result.Key.Key, http.MethodPost, "/api/teams", map[string]string{"name": "operator-team"})
|
||||
if resp3.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("expected 201 for a service-account write under operator mode, got %d", resp3.StatusCode)
|
||||
}
|
||||
resp3.Body.Close()
|
||||
|
||||
// Reads are unaffected regardless of caller.
|
||||
if resp := do(s.key, http.MethodGet, "/api/teams", nil); resp.StatusCode != http.StatusOK {
|
||||
t.Errorf("expected reads to stay open under operator mode, got %d", resp.StatusCode)
|
||||
} else {
|
||||
resp.Body.Close()
|
||||
}
|
||||
}
|
||||
|
||||
func TestOperatorMode_OffLeavesHumanWritesAlone(t *testing.T) {
|
||||
s := newTS(t) // testConfig(): OperatorMode false
|
||||
resp := s.req(t, http.MethodPost, "/api/teams", map[string]string{"name": "still-fine"})
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Errorf("expected a human write to succeed with operator mode off, got %d", resp.StatusCode)
|
||||
}
|
||||
resp.Body.Close()
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Version
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func TestVersion(t *testing.T) {
|
||||
s := newTS(t)
|
||||
resp, err := http.Get(s.URL + "/api/version")
|
||||
if err != nil {
|
||||
t.Fatalf("get version: %v", err)
|
||||
}
|
||||
var v struct {
|
||||
Version string `json:"version"`
|
||||
}
|
||||
decode(t, resp, &v)
|
||||
if v.Version != "test" {
|
||||
t.Errorf("expected version %q, got %q", "test", v.Version)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Dead man's switch update-in-place
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func TestDeadman_UpdateInPlacePreservesID(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
var created struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches",
|
||||
map[string]any{"matcher": "alertname=Watchdog", "timeout_seconds": 900, "severity": "critical"}), &created)
|
||||
|
||||
resp := s.req(t, http.MethodPut, "/api/teams/"+defaultTeam+"/deadman/switches/"+id64(created.ID),
|
||||
map[string]any{"name": "renamed", "matcher": "alertname=Watchdog", "timeout_seconds": 1200, "severity": "warning"})
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("update switch: %d", resp.StatusCode)
|
||||
}
|
||||
var updated struct {
|
||||
ID int64 `json:"id"`
|
||||
Name string `json:"name"`
|
||||
TimeoutSeconds int64 `json:"timeout_seconds"`
|
||||
Severity string `json:"severity"`
|
||||
}
|
||||
decode(t, resp, &updated)
|
||||
if updated.ID != created.ID {
|
||||
t.Errorf("expected id to stay %d, got %d", created.ID, updated.ID)
|
||||
}
|
||||
if updated.Name != "renamed" || updated.TimeoutSeconds != 1200 || updated.Severity != "warning" {
|
||||
t.Errorf("expected the update to apply, got %+v", updated)
|
||||
}
|
||||
|
||||
var list []map[string]any
|
||||
decode(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/deadman/switches", nil), &list)
|
||||
if len(list) != 1 {
|
||||
t.Errorf("expected the update to replace in place, not add a row, got %d switches", len(list))
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,483 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||
"github.com/go-chi/chi/v5"
|
||||
)
|
||||
|
||||
// The settings an administrator can change at runtime. Each is behaviour rather
|
||||
// than infrastructure: what the server does, not where it is plugged in.
|
||||
//
|
||||
// The values are seconds, stored as text. A duration string would be friendlier
|
||||
// to read in psql and worse everywhere else — it can be stored unparseable, and
|
||||
// then the question is what a background loop should do at 02:00 with a
|
||||
// tuning knob it cannot understand.
|
||||
const (
|
||||
SettingNotifyRepeat = "notify_repeat_seconds"
|
||||
SettingStaleAfter = "stale_after_seconds"
|
||||
SettingArchiveAfter = "archive_after_seconds"
|
||||
)
|
||||
|
||||
// settingBounds keeps an edit from producing a server that cannot work. The
|
||||
// ceilings are loose — they exist to catch a slipped decimal point, not to have
|
||||
// an opinion about anybody's rota.
|
||||
var settingBounds = map[string]struct {
|
||||
min, max time.Duration
|
||||
label string
|
||||
}{
|
||||
SettingNotifyRepeat: {0, 24 * time.Hour, "how long an incident may sit unacknowledged before it is paged again; 0 disables reminders"},
|
||||
SettingStaleAfter: {5 * time.Minute, 30 * 24 * time.Hour, "how long a firing alert may go without a refreshing webhook before the sweeper resolves it"},
|
||||
SettingArchiveAfter: {time.Minute, 365 * 24 * time.Hour, "how long a resolved alert or incident stays in the default list"},
|
||||
}
|
||||
|
||||
// Settings reads the runtime configuration. It holds no cache: the readers are
|
||||
// two background loops that tick every 30 seconds and 15 minutes, and handlers
|
||||
// that run once per request, so a query each time costs nothing measurable and
|
||||
// means an administrator's change takes effect on the next tick rather than at
|
||||
// the next restart.
|
||||
type Settings struct{ db *sql.DB }
|
||||
|
||||
// NewSettings returns a reader over db.
|
||||
func NewSettings(db *sql.DB) *Settings { return &Settings{db: db} }
|
||||
|
||||
// Duration reads one setting, falling back to def when the row is missing or
|
||||
// unreadable. A tuning knob is never worth failing a sweep over: the fallback
|
||||
// is the value the server started with.
|
||||
func (s *Settings) Duration(ctx context.Context, key string, def time.Duration) time.Duration {
|
||||
var raw string
|
||||
err := s.db.QueryRowContext(ctx, "SELECT value FROM settings WHERE key = $1", key).Scan(&raw)
|
||||
if err != nil {
|
||||
return def
|
||||
}
|
||||
secs, err := strconv.ParseInt(raw, 10, 64)
|
||||
if err != nil {
|
||||
return def
|
||||
}
|
||||
return time.Duration(secs) * time.Second
|
||||
}
|
||||
|
||||
// SeedSettings writes each key from the server's environment configuration,
|
||||
// once. Never overwrites: after the first start the database owns these, and a
|
||||
// redeploy must not put a chart's default back over an administrator's edit —
|
||||
// the same rule as the per-team dead man's switches.
|
||||
func SeedSettings(ctx context.Context, db *sql.DB, cfg config.Config) error {
|
||||
seeds := map[string]time.Duration{
|
||||
SettingNotifyRepeat: cfg.NotifyRepeat,
|
||||
SettingStaleAfter: cfg.StaleAfter,
|
||||
SettingArchiveAfter: cfg.ArchiveAfter,
|
||||
}
|
||||
for key, d := range seeds {
|
||||
if _, err := db.ExecContext(ctx, `
|
||||
INSERT INTO settings (key, value) VALUES ($1, $2)
|
||||
ON CONFLICT (key) DO NOTHING`,
|
||||
key, strconv.FormatInt(int64(d.Seconds()), 10)); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// settingsResponse is what the admin page renders. The environment half is
|
||||
// included and marked read-only, so somebody looking for the ntfy URL finds out
|
||||
// where it lives rather than concluding the server does not have one.
|
||||
type settingsResponse struct {
|
||||
Editable map[string]settingValue `json:"editable"`
|
||||
FromEnv map[string]string `json:"from_env"`
|
||||
|
||||
// Choices are settings that are a word from a fixed list rather than a
|
||||
// duration. One so far: who may create an account.
|
||||
Choices map[string]choiceValue `json:"choices"`
|
||||
}
|
||||
|
||||
type choiceValue struct {
|
||||
Value string `json:"value"`
|
||||
Options []string `json:"options"`
|
||||
Description string `json:"description"`
|
||||
}
|
||||
|
||||
type settingValue struct {
|
||||
Seconds int64 `json:"seconds"`
|
||||
Description string `json:"description"`
|
||||
MinSeconds int64 `json:"min_seconds"`
|
||||
MaxSeconds int64 `json:"max_seconds"`
|
||||
}
|
||||
|
||||
func handleGetSettings(db *sql.DB, cfg config.Config) http.HandlerFunc {
|
||||
settings := NewSettings(db)
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
out := settingsResponse{
|
||||
Editable: map[string]settingValue{},
|
||||
Choices: map[string]choiceValue{
|
||||
SettingSignupMode: {
|
||||
Value: signupMode(r.Context(), db),
|
||||
Options: []string{SignupInviteOnly, SignupOpen},
|
||||
Description: "who may create an account: invite_only means a link from a team owner, " +
|
||||
"open means anybody who can reach this server",
|
||||
},
|
||||
},
|
||||
FromEnv: map[string]string{
|
||||
// Never the ntfy token or the DSN: both are credentials, and an
|
||||
// admin page that renders them turns a browser tab into a place
|
||||
// they leak from.
|
||||
"ntfy_url": cfg.NtfyURL,
|
||||
"ntfy_configured": strconv.FormatBool(cfg.NtfyURL != ""),
|
||||
"ntfy_token_set": strconv.FormatBool(cfg.NtfyToken != ""),
|
||||
"public_url": cfg.PublicURL,
|
||||
"listen_address": cfg.Addr,
|
||||
},
|
||||
}
|
||||
for key, b := range settingBounds {
|
||||
def := map[string]time.Duration{
|
||||
SettingNotifyRepeat: cfg.NotifyRepeat,
|
||||
SettingStaleAfter: cfg.StaleAfter,
|
||||
SettingArchiveAfter: cfg.ArchiveAfter,
|
||||
}[key]
|
||||
out.Editable[key] = settingValue{
|
||||
Seconds: int64(settings.Duration(r.Context(), key, def).Seconds()),
|
||||
Description: b.label,
|
||||
MinSeconds: int64(b.min.Seconds()),
|
||||
MaxSeconds: int64(b.max.Seconds()),
|
||||
}
|
||||
}
|
||||
respond(w, http.StatusOK, out)
|
||||
}
|
||||
}
|
||||
|
||||
// handleSetSettings changes one or more settings. Unknown keys are refused
|
||||
// rather than stored: a typo that writes notify_repeat_second would otherwise
|
||||
// sit in the table looking like configuration and doing nothing.
|
||||
func handleSetSettings(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
var req map[string]any
|
||||
if err := decodeJSON(r, &req); err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||
return
|
||||
}
|
||||
if len(req) == 0 {
|
||||
respond(w, http.StatusBadRequest, errResp("no settings given"))
|
||||
return
|
||||
}
|
||||
|
||||
// Validate everything before writing anything: a request that sets two
|
||||
// settings and gets one wrong should change neither.
|
||||
values := map[string]string{}
|
||||
for key, raw := range req {
|
||||
switch key {
|
||||
case SettingSignupMode:
|
||||
mode, _ := raw.(string)
|
||||
if mode != SignupOpen && mode != SignupInviteOnly {
|
||||
respond(w, http.StatusBadRequest,
|
||||
errResp("signup_mode must be "+SignupInviteOnly+" or "+SignupOpen))
|
||||
return
|
||||
}
|
||||
values[key] = mode
|
||||
default:
|
||||
b, known := settingBounds[key]
|
||||
if !known {
|
||||
respond(w, http.StatusBadRequest, errResp("unknown setting: "+key))
|
||||
return
|
||||
}
|
||||
secs, ok := raw.(float64) // JSON numbers decode as float64
|
||||
if !ok {
|
||||
respond(w, http.StatusBadRequest, errResp(key+" must be a number of seconds"))
|
||||
return
|
||||
}
|
||||
d := time.Duration(int64(secs)) * time.Second
|
||||
if d < b.min || d > b.max {
|
||||
respond(w, http.StatusBadRequest, errResp(
|
||||
key+" must be between "+b.min.String()+" and "+b.max.String()))
|
||||
return
|
||||
}
|
||||
values[key] = strconv.FormatInt(int64(secs), 10)
|
||||
}
|
||||
}
|
||||
|
||||
tx, err := db.BeginTx(r.Context(), nil)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
defer tx.Rollback() //nolint:errcheck
|
||||
|
||||
for key, value := range values {
|
||||
if _, err := tx.ExecContext(r.Context(), `
|
||||
INSERT INTO settings (key, value, updated_at)
|
||||
VALUES ($1, $2, `+nowEpoch+`)
|
||||
ON CONFLICT (key) DO UPDATE SET
|
||||
value = excluded.value, updated_at = excluded.updated_at`,
|
||||
key, value); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
}
|
||||
if err := tx.Commit(); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}
|
||||
}
|
||||
|
||||
// adminTeam is a team as an administrator sees it: what it is, plus how big it
|
||||
// is and how much is on fire in it. One definition, so a team in the list and a
|
||||
// team on its own page cannot describe themselves differently.
|
||||
type adminTeam struct {
|
||||
ID int64 `json:"id"`
|
||||
Name string `json:"name"`
|
||||
CreatedAt time.Time `json:"created_at"`
|
||||
Members int64 `json:"members"`
|
||||
OpenIncidents int64 `json:"open_incidents"`
|
||||
|
||||
// OIDCMemberGroup and OIDCOwnerGroup are the team's own group binding,
|
||||
// read-only here: an administrator can see why a team's OIDC-sourced
|
||||
// membership looks the way it does without being able to change it out
|
||||
// from under the team's owner. Setting it is PUT
|
||||
// /api/teams/{teamID}/oidc-groups, owner-only.
|
||||
OIDCMemberGroup string `json:"oidc_member_group,omitempty"`
|
||||
OIDCOwnerGroup string `json:"oidc_owner_group,omitempty"`
|
||||
}
|
||||
|
||||
// handleAdminListTeams lists every team on the server, with its size. The
|
||||
// ordinary /api/teams answers "what am I in"; this one answers "what exists",
|
||||
// which only an administrator may ask.
|
||||
func handleAdminListTeams(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
rows, err := db.QueryContext(r.Context(), `
|
||||
SELECT t.id, t.name, t.created_at,
|
||||
(SELECT COUNT(*) FROM team_members m WHERE m.team_id = t.id),
|
||||
(SELECT COUNT(*) FROM incidents i
|
||||
WHERE i.team_id = t.id AND i.resolved_at IS NULL),
|
||||
COALESCE(t.oidc_member_group, ''), COALESCE(t.oidc_owner_group, '')
|
||||
FROM teams t
|
||||
ORDER BY t.name`)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
teams := []adminTeam{}
|
||||
for rows.Next() {
|
||||
var t adminTeam
|
||||
var created int64
|
||||
if err := rows.Scan(&t.ID, &t.Name, &created, &t.Members, &t.OpenIncidents,
|
||||
&t.OIDCMemberGroup, &t.OIDCOwnerGroup); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
t.CreatedAt = time.Unix(created, 0).UTC()
|
||||
teams = append(teams, t)
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusOK, teams)
|
||||
}
|
||||
}
|
||||
|
||||
// handleAdminGetTeam answers "what is this team, and who is in it" for any team
|
||||
// on the server, which is the one question an administrator could not ask.
|
||||
//
|
||||
// GET /api/teams/{id}/members is requireTeamMember and answers 404 to somebody
|
||||
// outside the team, administrator or not, and that stays exactly as it is:
|
||||
// member means membership and nothing else. Reading a team's shape is a
|
||||
// different thing from reading its work, so it gets an endpoint of its own
|
||||
// under AdminOnly rather than an exception carved into that rule. An
|
||||
// administrator still sees none of the team's incidents, alerts or rota.
|
||||
func handleAdminGetTeam(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
teamID, ok := teamParam(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
|
||||
var t adminTeam
|
||||
var created int64
|
||||
err := db.QueryRowContext(r.Context(), `
|
||||
SELECT t.id, t.name, t.created_at,
|
||||
(SELECT COUNT(*) FROM team_members m WHERE m.team_id = t.id),
|
||||
(SELECT COUNT(*) FROM incidents i
|
||||
WHERE i.team_id = t.id AND i.resolved_at IS NULL),
|
||||
COALESCE(t.oidc_member_group, ''), COALESCE(t.oidc_owner_group, '')
|
||||
FROM teams t
|
||||
WHERE t.id = $1`, teamID).
|
||||
Scan(&t.ID, &t.Name, &created, &t.Members, &t.OpenIncidents,
|
||||
&t.OIDCMemberGroup, &t.OIDCOwnerGroup)
|
||||
if errors.Is(err, sql.ErrNoRows) {
|
||||
respond(w, http.StatusNotFound, errResp("not found"))
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
t.CreatedAt = time.Unix(created, 0).UTC()
|
||||
|
||||
// Same query and same ordering as handleListTeamMembers, so the two
|
||||
// answers to "who is in this team" cannot disagree about the answer.
|
||||
rows, err := db.QueryContext(r.Context(), `
|
||||
SELECT m.team_id, m.user_id, u.username, m.role, m.joined_at, m.source
|
||||
FROM team_members m
|
||||
JOIN users u ON u.id = m.user_id
|
||||
WHERE m.team_id = $1
|
||||
ORDER BY u.username`, teamID)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
members := []models.TeamMember{}
|
||||
for rows.Next() {
|
||||
var m models.TeamMember
|
||||
var joined int64
|
||||
if err := rows.Scan(&m.TeamID, &m.UserID, &m.Username, &m.Role, &joined, &m.Source); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
m.JoinedAt = time.Unix(joined, 0).UTC()
|
||||
members = append(members, m)
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
|
||||
// A wrapper rather than a team with the members hung off it: "members"
|
||||
// already means a count on the list endpoint, and one name must not be
|
||||
// a number in one answer and an array in the next.
|
||||
respond(w, http.StatusOK, map[string]any{"team": t, "members": members})
|
||||
}
|
||||
}
|
||||
|
||||
// handleRenameTeam renames a team. An owner's job, and an administrator's when
|
||||
// a team has nobody left to do it.
|
||||
func handleRenameTeam(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
teamID, ok := teamParam(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
if !requireTeamOwner(w, r, teamID) {
|
||||
return
|
||||
}
|
||||
|
||||
var req struct {
|
||||
Name string `json:"name"`
|
||||
}
|
||||
// Trimmed, as handleCreateTeam trims: without it " " is a team name
|
||||
// here but not at creation, which is one rule stated twice and only
|
||||
// half applied.
|
||||
if err := decodeJSON(r, &req); err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("name is required"))
|
||||
return
|
||||
}
|
||||
req.Name = strings.TrimSpace(req.Name)
|
||||
if req.Name == "" {
|
||||
respond(w, http.StatusBadRequest, errResp("name is required"))
|
||||
return
|
||||
}
|
||||
|
||||
res, err := db.ExecContext(r.Context(),
|
||||
"UPDATE teams SET name = $1 WHERE id = $2", req.Name, teamID)
|
||||
if err != nil {
|
||||
if isUniqueViolation(err) {
|
||||
respond(w, http.StatusConflict, errResp("a team with that name already exists"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if n, _ := res.RowsAffected(); n == 0 {
|
||||
respond(w, http.StatusNotFound, errResp("not found"))
|
||||
return
|
||||
}
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}
|
||||
}
|
||||
|
||||
// handleSetUserDisabled takes an account out of use, or puts it back.
|
||||
//
|
||||
// Not a delete: the person's acknowledgements, assignments and timeline entries
|
||||
// stay attached to them. Deleting a user nulls those columns, which rewrites
|
||||
// what happened during an incident months after the fact.
|
||||
func handleSetUserDisabled(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, err := strconv.ParseInt(chi.URLParam(r, "id"), 10, 64)
|
||||
if err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid user id"))
|
||||
return
|
||||
}
|
||||
var req struct {
|
||||
Disabled *bool `json:"disabled"`
|
||||
}
|
||||
if err := decodeJSON(r, &req); err != nil || req.Disabled == nil {
|
||||
respond(w, http.StatusBadRequest, errResp("disabled is required"))
|
||||
return
|
||||
}
|
||||
|
||||
if *req.Disabled {
|
||||
caller, _ := userFromContext(r.Context())
|
||||
if caller.ID == id {
|
||||
respond(w, http.StatusConflict, errResp("cannot disable your own account"))
|
||||
return
|
||||
}
|
||||
last, err := isLastAdmin(r.Context(), db, id)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if last {
|
||||
respond(w, http.StatusConflict, errResp("cannot disable the last administrator"))
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
var res sql.Result
|
||||
if *req.Disabled {
|
||||
res, err = db.ExecContext(r.Context(),
|
||||
"UPDATE users SET disabled_at = "+nowEpoch+" WHERE id = $1 AND disabled_at IS NULL", id)
|
||||
} else {
|
||||
res, err = db.ExecContext(r.Context(),
|
||||
"UPDATE users SET disabled_at = NULL WHERE id = $1", id)
|
||||
}
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if n, _ := res.RowsAffected(); n == 0 {
|
||||
// Either no such user, or already in the state asked for. The
|
||||
// second is not a failure, so check which before answering.
|
||||
var exists int
|
||||
if err := db.QueryRowContext(r.Context(),
|
||||
"SELECT 1 FROM users WHERE id = $1", id).Scan(&exists); errors.Is(err, sql.ErrNoRows) {
|
||||
respond(w, http.StatusNotFound, errResp("user not found"))
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
// Signing back in is the only way to use a re-enabled account, and a
|
||||
// disabled one must not keep a live session.
|
||||
if *req.Disabled {
|
||||
db.ExecContext(r.Context(), "DELETE FROM sessions WHERE user_id = $1", id) //nolint:errcheck
|
||||
}
|
||||
|
||||
user, err := fetchUser(r.Context(), db, id)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusOK, user)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,275 @@
|
||||
package api_test
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||
)
|
||||
|
||||
// The settings an administrator can change, and the ones they cannot.
|
||||
func TestSettings_EditableAndReadOnly(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
var got struct {
|
||||
Editable map[string]struct {
|
||||
Seconds int64 `json:"seconds"`
|
||||
Description string `json:"description"`
|
||||
MinSeconds int64 `json:"min_seconds"`
|
||||
MaxSeconds int64 `json:"max_seconds"`
|
||||
} `json:"editable"`
|
||||
FromEnv map[string]string `json:"from_env"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodGet, "/api/admin/settings", nil), &got)
|
||||
|
||||
// Seeded from the environment the server started with, not from zero.
|
||||
if v := got.Editable["notify_repeat_seconds"].Seconds; v != 900 {
|
||||
t.Errorf("notify_repeat_seconds seeded as %d, want 900", v)
|
||||
}
|
||||
if v := got.Editable["stale_after_seconds"].Seconds; v != 21600 {
|
||||
t.Errorf("stale_after_seconds seeded as %d, want 21600", v)
|
||||
}
|
||||
if got.Editable["archive_after_seconds"].Description == "" {
|
||||
t.Error("a setting without a description is a number nobody can act on")
|
||||
}
|
||||
|
||||
// The environment half is visible so somebody can see where it lives, but
|
||||
// never the credentials themselves.
|
||||
if _, ok := got.FromEnv["public_url"]; !ok {
|
||||
t.Error("public_url should be reported as environment-configured")
|
||||
}
|
||||
for _, leak := range []string{"ntfy_token", "dsn", "database_dsn", "password"} {
|
||||
if v, ok := got.FromEnv[leak]; ok {
|
||||
t.Errorf("%s must not be in the settings response (got %q)", leak, v)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Changing a setting takes effect on the next tick, without a restart. This is
|
||||
// the whole point of moving them out of the environment.
|
||||
func TestSettings_ChangeTakesEffectOnTheNextSweep(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
// An alert whose last webhook was two hours ago. Under the seeded
|
||||
// stale_after of six hours the sweeper leaves it alone.
|
||||
postWebhook(t, s, []map[string]any{
|
||||
amAlert("fp-settings", "Stale", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||
})
|
||||
s.exec(t, "UPDATE alerts SET received_at = $1 WHERE fingerprint = $2",
|
||||
time.Now().Add(-2*time.Hour).Unix(), "fp-settings")
|
||||
|
||||
sweep(t, s, noArchive)
|
||||
if status, _, _ := s.alertRow(t, "fp-settings"); status != "firing" {
|
||||
t.Fatalf("before the change the alert should still be firing, got %q", status)
|
||||
}
|
||||
|
||||
// Shorten it to an hour. Nothing restarts.
|
||||
resp := s.req(t, http.MethodPut, "/api/admin/settings",
|
||||
map[string]int64{"stale_after_seconds": 3600})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Fatalf("change setting: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
sweep(t, s, noArchive)
|
||||
status, source, _ := s.alertRow(t, "fp-settings")
|
||||
if status != "resolved" {
|
||||
t.Errorf("after the change the alert should have expired, got %q", status)
|
||||
}
|
||||
if source == nil || *source != "expiry" {
|
||||
t.Errorf("expected resolution_source expiry, got %v", source)
|
||||
}
|
||||
}
|
||||
|
||||
// A typo must not look like configuration, and a slipped decimal point must not
|
||||
// produce a server that sweeps every second.
|
||||
func TestSettings_RejectsUnknownKeysAndSillyValues(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
body map[string]int64
|
||||
}{
|
||||
{"unknown key", map[string]int64{"notify_repeat_second": 60}},
|
||||
{"below the floor", map[string]int64{"stale_after_seconds": 30}},
|
||||
{"above the ceiling", map[string]int64{"archive_after_seconds": 400 * 24 * 3600}},
|
||||
{"nothing at all", map[string]int64{}},
|
||||
} {
|
||||
resp := s.req(t, http.MethodPut, "/api/admin/settings", c.body)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusBadRequest {
|
||||
t.Errorf("%s: expected 400, got %d", c.name, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Settings are the server's behaviour, so only an administrator may change
|
||||
// them — or see where the rest of the configuration comes from.
|
||||
func TestSettings_AreAdminOnly(t *testing.T) {
|
||||
s := newTS(t)
|
||||
_, call := member(t, s, "member")
|
||||
|
||||
for _, c := range []struct {
|
||||
method string
|
||||
body any
|
||||
}{
|
||||
{http.MethodGet, nil},
|
||||
{http.MethodPut, map[string]int64{"notify_repeat_seconds": 60}},
|
||||
} {
|
||||
resp := call(c.method, "/api/admin/settings", c.body)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("%s /api/admin/settings: expected 403, got %d", c.method, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
resp := call(http.MethodGet, "/api/admin/teams", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("GET /api/admin/teams: expected 403, got %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// An administrator sees every team, including ones they are not in — which is
|
||||
// exactly what /api/teams must not show them.
|
||||
func TestSettings_AdminSeesEveryTeam(t *testing.T) {
|
||||
s := newTS(t)
|
||||
newTeam(t, s, "red")
|
||||
newTeam(t, s, "blue")
|
||||
|
||||
all := list(t, s.req(t, http.MethodGet, "/api/admin/teams", nil))
|
||||
if len(all) != 3 { // Default, red, blue
|
||||
t.Fatalf("admin should see all 3 teams, saw %d", len(all))
|
||||
}
|
||||
for _, team := range all {
|
||||
if _, ok := team["members"]; !ok {
|
||||
t.Error("the admin listing should say how big each team is")
|
||||
}
|
||||
}
|
||||
|
||||
// The admin created them, so they own them — but they are not a member of
|
||||
// a team somebody else makes, and /api/teams still answers "what am I in".
|
||||
mine := list(t, s.req(t, http.MethodGet, "/api/teams", nil))
|
||||
if len(mine) != 3 {
|
||||
t.Errorf("the creator is an owner of what they created, saw %d", len(mine))
|
||||
}
|
||||
}
|
||||
|
||||
// Disabling is not deleting: the account stops working and the history stays.
|
||||
func TestSettings_DisablingAnAccountKeepsItsHistory(t *testing.T) {
|
||||
s := newTS(t)
|
||||
memberID, call := member(t, s, "leaver")
|
||||
|
||||
// They acknowledge an incident, so there is history to preserve.
|
||||
postWebhook(t, s, []map[string]any{
|
||||
amAlert("fp-leaver", "DiskFull", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||
})
|
||||
resp := call(http.MethodPost, "/api/incidents/1/acknowledge", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("acknowledge: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
resp = s.req(t, http.MethodPut, "/api/users/"+id64(memberID)+"/disabled",
|
||||
map[string]bool{"disabled": true})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("disable: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
// Their API key stops working.
|
||||
resp = call(http.MethodGet, "/api/incidents", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusUnauthorized {
|
||||
t.Errorf("a disabled user's key: expected 401, got %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
// The acknowledgement still names them.
|
||||
var incident map[string]any
|
||||
decode(t, s.req(t, http.MethodGet, "/api/incidents/1", nil), &incident)
|
||||
if incident["acknowledged_by"] != "leaver" {
|
||||
t.Errorf("the acknowledgement should still name leaver, got %v", incident["acknowledged_by"])
|
||||
}
|
||||
if incident["status"] != "acknowledged" {
|
||||
t.Errorf("the incident should still be acknowledged, got %v", incident["status"])
|
||||
}
|
||||
|
||||
// And re-enabling gives the account back.
|
||||
resp = s.req(t, http.MethodPut, "/api/users/"+id64(memberID)+"/disabled",
|
||||
map[string]bool{"disabled": false})
|
||||
resp.Body.Close()
|
||||
resp = call(http.MethodGet, "/api/incidents", nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Errorf("after re-enabling: expected 200, got %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// The same two guards as deleting and demoting: an install must keep somebody
|
||||
// who can administer it.
|
||||
func TestSettings_CannotDisableYourselfOrTheLastAdmin(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
resp := s.req(t, http.MethodPut, "/api/users/1/disabled", map[string]bool{"disabled": true})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusConflict {
|
||||
t.Errorf("disabling yourself: expected 409, got %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// Renaming a team is an owner's job, and the name stays unique.
|
||||
func TestSettings_TeamRename(t *testing.T) {
|
||||
s := newTS(t)
|
||||
team := newTeam(t, s, "red")
|
||||
|
||||
resp := s.req(t, http.MethodPut, "/api/teams/"+id64(team.id), map[string]string{"name": "Platform"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Fatalf("rename: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
teams := list(t, s.req(t, http.MethodGet, "/api/admin/teams", nil))
|
||||
found := false
|
||||
for _, x := range teams {
|
||||
if x["name"] == "Platform" {
|
||||
found = true
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Error("the renamed team should be listed under its new name")
|
||||
}
|
||||
|
||||
// Taking a name that exists is a conflict, not a silent second team with
|
||||
// the same label.
|
||||
resp = s.req(t, http.MethodPut, "/api/teams/"+id64(team.id), map[string]string{"name": "Default"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusConflict {
|
||||
t.Errorf("renaming onto an existing name: expected 409, got %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// The seed runs once. A redeploy must not put the chart's default back over an
|
||||
// administrator's edit — the rule the dead man's switches already follow.
|
||||
func TestSettings_SeedDoesNotOverwrite(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
resp := s.req(t, http.MethodPut, "/api/admin/settings",
|
||||
map[string]int64{"notify_repeat_seconds": 60})
|
||||
resp.Body.Close()
|
||||
|
||||
// A second start, with the environment still saying 15 minutes.
|
||||
if err := api.SeedSettings(t.Context(), s.db, testConfig()); err != nil {
|
||||
t.Fatalf("re-seed: %v", err)
|
||||
}
|
||||
|
||||
var got struct {
|
||||
Editable map[string]struct {
|
||||
Seconds int64 `json:"seconds"`
|
||||
} `json:"editable"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodGet, "/api/admin/settings", nil), &got)
|
||||
if v := got.Editable["notify_repeat_seconds"].Seconds; v != 60 {
|
||||
t.Errorf("the edit should survive a restart, got %d", v)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,509 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||
"github.com/go-chi/chi/v5"
|
||||
)
|
||||
|
||||
// SettingSignupMode says who may create an account. It lives in the settings
|
||||
// table with the other behaviour settings, so an administrator changes it in
|
||||
// the admin page rather than in a chart.
|
||||
//
|
||||
// Two modes, not three. A domain-restricted mode was considered and dropped:
|
||||
// with no email in this server there is nothing to verify an address against,
|
||||
// so it would check the domain of a string somebody typed — a speed bump
|
||||
// dressed as a control.
|
||||
const (
|
||||
SettingSignupMode = "signup_mode"
|
||||
|
||||
SignupInviteOnly = "invite_only"
|
||||
SignupOpen = "open"
|
||||
)
|
||||
|
||||
// defaultSignupMode is invite-only. An install that gets a public hostname
|
||||
// before anybody has thought about sign-up should not be collecting accounts
|
||||
// from the internet by default.
|
||||
const defaultSignupMode = SignupInviteOnly
|
||||
|
||||
// inviteTTL is how long a new invite link lives. Long enough to send it and be
|
||||
// read tomorrow, short enough that a link in an old chat log stops working.
|
||||
const inviteTTL = 7 * 24 * time.Hour
|
||||
|
||||
// signupMode reads the current mode, falling back to invite-only for a missing
|
||||
// or unrecognised value: the failure mode of a typo in this setting should be
|
||||
// the closed door, not the open one.
|
||||
func signupMode(ctx context.Context, db *sql.DB) string {
|
||||
var raw string
|
||||
if err := db.QueryRowContext(ctx,
|
||||
"SELECT value FROM settings WHERE key = $1", SettingSignupMode).Scan(&raw); err != nil {
|
||||
return defaultSignupMode
|
||||
}
|
||||
if raw != SignupOpen && raw != SignupInviteOnly {
|
||||
return defaultSignupMode
|
||||
}
|
||||
return raw
|
||||
}
|
||||
|
||||
// handleSignupInfo tells the sign-up page what it may offer, without requiring
|
||||
// a session: whether open sign-up is on, and whether the invite in the URL is
|
||||
// any good. A bad invite is better reported before somebody picks a password.
|
||||
func handleSignupInfo(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
out := map[string]any{"mode": signupMode(r.Context(), db)}
|
||||
|
||||
if token := r.URL.Query().Get("invite"); token != "" {
|
||||
inv, err := loadInvite(r.Context(), db, token)
|
||||
switch {
|
||||
case err == nil:
|
||||
out["invite_valid"] = true
|
||||
out["invite_team"] = inv.teamName
|
||||
default:
|
||||
// Deliberately one answer for expired, revoked, used up and
|
||||
// never existed. Telling a stranger which it was tells them
|
||||
// something about links they do not hold.
|
||||
out["invite_valid"] = false
|
||||
}
|
||||
}
|
||||
respond(w, http.StatusOK, out)
|
||||
}
|
||||
}
|
||||
|
||||
type invite struct {
|
||||
id int64
|
||||
teamID int64
|
||||
teamName string
|
||||
role string
|
||||
}
|
||||
|
||||
// loadInvite resolves a raw token to a usable invite, or an error. Usable means
|
||||
// it exists, has not been revoked, has not expired and has uses left.
|
||||
func loadInvite(ctx context.Context, q querier, token string) (invite, error) {
|
||||
var inv invite
|
||||
err := q.QueryRowContext(ctx, `
|
||||
SELECT i.id, i.team_id, t.name, i.role
|
||||
FROM invites i
|
||||
JOIN teams t ON t.id = i.team_id
|
||||
WHERE i.token_hash = $1
|
||||
AND i.revoked_at IS NULL
|
||||
AND i.expires_at > `+nowEpoch+`
|
||||
AND i.uses < i.max_uses`, hashToken(token)).
|
||||
Scan(&inv.id, &inv.teamID, &inv.teamName, &inv.role)
|
||||
if errors.Is(err, sql.ErrNoRows) {
|
||||
return invite{}, errInviteUnusable
|
||||
}
|
||||
return inv, err
|
||||
}
|
||||
|
||||
var errInviteUnusable = errors.New("invite is not usable")
|
||||
|
||||
// handleSignup creates an account, and puts it somewhere.
|
||||
//
|
||||
// Rate-limited on the same limiter as login, by address: sign-up is the other
|
||||
// unauthenticated endpoint that writes, and an open install without this is a
|
||||
// way to fill somebody's user table.
|
||||
func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
addr := clientAddr(r)
|
||||
if limiter.blocked("signup:"+addr, maxSignupsPerAddr) {
|
||||
respond(w, http.StatusTooManyRequests, errResp("too many sign-ups from this address"))
|
||||
return
|
||||
}
|
||||
|
||||
var req struct {
|
||||
Username string `json:"username"`
|
||||
Email string `json:"email"`
|
||||
Password string `json:"password"`
|
||||
Invite string `json:"invite"`
|
||||
TeamName string `json:"team_name"`
|
||||
}
|
||||
if err := decodeJSON(r, &req); err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||
return
|
||||
}
|
||||
req.Username = strings.TrimSpace(req.Username)
|
||||
req.Email = strings.TrimSpace(req.Email)
|
||||
req.TeamName = strings.TrimSpace(req.TeamName)
|
||||
|
||||
if req.Username == "" || req.Email == "" {
|
||||
respond(w, http.StatusBadRequest, errResp("username and email are required"))
|
||||
return
|
||||
}
|
||||
if msg := validatePassword(req.Password); msg != "" {
|
||||
respond(w, http.StatusBadRequest, errResp(msg))
|
||||
return
|
||||
}
|
||||
|
||||
mode := signupMode(r.Context(), db)
|
||||
var inv invite
|
||||
hasInvite := false
|
||||
if req.Invite != "" {
|
||||
var err error
|
||||
inv, err = loadInvite(r.Context(), db, req.Invite)
|
||||
if err != nil {
|
||||
limiter.fail("signup:" + addr)
|
||||
respond(w, http.StatusForbidden, errResp("this invite link is not usable"))
|
||||
return
|
||||
}
|
||||
hasInvite = true
|
||||
}
|
||||
if !hasInvite && mode != SignupOpen {
|
||||
// No invite and the door is shut. Not 404: the endpoint exists and
|
||||
// saying so is how somebody knows to ask for a link.
|
||||
respond(w, http.StatusForbidden,
|
||||
errResp("sign-up is invite-only on this server"))
|
||||
return
|
||||
}
|
||||
if !hasInvite && req.TeamName == "" {
|
||||
// Open sign-up with no team would create an account that sees an
|
||||
// empty queue and can be paged by nobody.
|
||||
respond(w, http.StatusBadRequest, errResp("team_name is required"))
|
||||
return
|
||||
}
|
||||
|
||||
hash, err := hashPassword(req.Password)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
|
||||
tx, err := db.BeginTx(r.Context(), nil)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
defer tx.Rollback() //nolint:errcheck
|
||||
|
||||
var userID int64
|
||||
var invitedVia *int64
|
||||
if hasInvite {
|
||||
invitedVia = &inv.id
|
||||
}
|
||||
if err := tx.QueryRowContext(r.Context(), `
|
||||
INSERT INTO users (username, email, password_hash, invited_via)
|
||||
VALUES ($1, $2, $3, $4) RETURNING id`,
|
||||
req.Username, req.Email, hash, invitedVia).Scan(&userID); err != nil {
|
||||
if isUniqueViolation(err) {
|
||||
respond(w, http.StatusConflict, errResp("username or email already exists"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
|
||||
teamID, role := inv.teamID, inv.role
|
||||
if !hasInvite {
|
||||
// Open sign-up makes a team, and its creator owns it.
|
||||
if err := tx.QueryRowContext(r.Context(),
|
||||
"INSERT INTO teams (name) VALUES ($1) RETURNING id", req.TeamName).Scan(&teamID); err != nil {
|
||||
if isUniqueViolation(err) {
|
||||
respond(w, http.StatusConflict, errResp("a team with that name already exists"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
role = models.RoleOwner
|
||||
}
|
||||
|
||||
if _, err := tx.ExecContext(r.Context(),
|
||||
"INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3)",
|
||||
teamID, userID, role); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
|
||||
if hasInvite {
|
||||
// Counted inside the transaction, so two people redeeming the last
|
||||
// use of a link at once cannot both get in.
|
||||
res, err := tx.ExecContext(r.Context(),
|
||||
"UPDATE invites SET uses = uses + 1 WHERE id = $1 AND uses < max_uses", inv.id)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if n, _ := res.RowsAffected(); n == 0 {
|
||||
respond(w, http.StatusForbidden, errResp("this invite link is not usable"))
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
if err := tx.Commit(); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
|
||||
// Signed in immediately: the alternative is a form that says "now go
|
||||
// and log in", which is the same credential typed twice.
|
||||
if err := startSession(w, r, db, userID, publicURL); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
user, _ := fetchUser(r.Context(), db, userID)
|
||||
respond(w, http.StatusCreated, meResponse{User: user, HasPassword: true})
|
||||
}
|
||||
}
|
||||
|
||||
// maxSignupsPerAddr is looser than the login limit: several people joining from
|
||||
// one office share an address, and the thing being limited is account creation
|
||||
// rather than password guessing.
|
||||
const maxSignupsPerAddr = 10
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Invites
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
type inviteJSON struct {
|
||||
ID int64 `json:"id"`
|
||||
TeamID int64 `json:"team_id"`
|
||||
Role string `json:"role"`
|
||||
CreatedAt time.Time `json:"created_at"`
|
||||
ExpiresAt time.Time `json:"expires_at"`
|
||||
MaxUses int64 `json:"max_uses"`
|
||||
Uses int64 `json:"uses"`
|
||||
Revoked bool `json:"revoked"`
|
||||
|
||||
// URL is the whole link, returned once when the invite is created. Like an
|
||||
// integration key, only its hash is stored.
|
||||
URL string `json:"url,omitempty"`
|
||||
}
|
||||
|
||||
func handleListInvites(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
teamID, ok := teamParam(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
if !requireTeamOwner(w, r, teamID) {
|
||||
return
|
||||
}
|
||||
|
||||
rows, err := db.QueryContext(r.Context(), `
|
||||
SELECT id, team_id, role, created_at, expires_at, max_uses, uses, revoked_at
|
||||
FROM invites
|
||||
WHERE team_id = $1
|
||||
ORDER BY id DESC`, teamID)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
out := []inviteJSON{}
|
||||
for rows.Next() {
|
||||
var i inviteJSON
|
||||
var created, expires int64
|
||||
var revoked *int64
|
||||
if err := rows.Scan(&i.ID, &i.TeamID, &i.Role, &created, &expires,
|
||||
&i.MaxUses, &i.Uses, &revoked); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
i.CreatedAt = time.Unix(created, 0).UTC()
|
||||
i.ExpiresAt = time.Unix(expires, 0).UTC()
|
||||
i.Revoked = revoked != nil
|
||||
out = append(out, i)
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusOK, out)
|
||||
}
|
||||
}
|
||||
|
||||
// handleCreateInvite mints a link into this team. Owner-only, like the rest of
|
||||
// a team's configuration: deciding who joins is configuring the team.
|
||||
func handleCreateInvite(db *sql.DB, publicURL string) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
teamID, ok := teamParam(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
if !requireTeamOwner(w, r, teamID) {
|
||||
return
|
||||
}
|
||||
|
||||
var req struct {
|
||||
Role string `json:"role"`
|
||||
MaxUses int64 `json:"max_uses"`
|
||||
}
|
||||
if err := decodeJSON(r, &req); err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||
return
|
||||
}
|
||||
if req.Role == "" {
|
||||
req.Role = models.RoleMember
|
||||
}
|
||||
if req.Role != models.RoleOwner && req.Role != models.RoleMember {
|
||||
respond(w, http.StatusBadRequest, errResp("role must be owner or member"))
|
||||
return
|
||||
}
|
||||
if req.MaxUses == 0 {
|
||||
req.MaxUses = 1
|
||||
}
|
||||
if req.MaxUses < 1 || req.MaxUses > 100 {
|
||||
respond(w, http.StatusBadRequest, errResp("max_uses must be between 1 and 100"))
|
||||
return
|
||||
}
|
||||
|
||||
raw, hash, err := randomToken()
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
// created_by is nullable (ON DELETE SET NULL) for exactly this
|
||||
// reason: the caller minting an invite is not always a human with a
|
||||
// real users row. A team-scoped service account is owner-equivalent
|
||||
// here (requireTeamOwner above already let it through), and this
|
||||
// must leave created_by NULL for one the same way
|
||||
// handleCreateServiceAccount already does for the analogous case —
|
||||
// an unchecked zero value would violate the users(id) foreign key
|
||||
// instead of recording "nobody" cleanly.
|
||||
var createdBy *int64
|
||||
if u, ok := userFromContext(r.Context()); ok {
|
||||
id := u.ID
|
||||
createdBy = &id
|
||||
}
|
||||
expires := time.Now().Add(inviteTTL)
|
||||
|
||||
var out inviteJSON
|
||||
var created, expiresAt int64
|
||||
if err := db.QueryRowContext(r.Context(), `
|
||||
INSERT INTO invites (token_hash, team_id, role, created_by, expires_at, max_uses)
|
||||
VALUES ($1, $2, $3, $4, $5, $6)
|
||||
RETURNING id, team_id, role, created_at, expires_at, max_uses, uses`,
|
||||
hash, teamID, req.Role, createdBy, expires.Unix(), req.MaxUses).
|
||||
Scan(&out.ID, &out.TeamID, &out.Role, &created, &expiresAt, &out.MaxUses, &out.Uses); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
out.CreatedAt = time.Unix(created, 0).UTC()
|
||||
out.ExpiresAt = time.Unix(expiresAt, 0).UTC()
|
||||
out.URL = strings.TrimSuffix(publicURL, "/") + "/signup?invite=" + raw
|
||||
respond(w, http.StatusCreated, out)
|
||||
}
|
||||
}
|
||||
|
||||
// handleRevokeInvite stops a link working without waiting for it to expire.
|
||||
func handleRevokeInvite(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
teamID, ok := teamParam(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
if !requireTeamOwner(w, r, teamID) {
|
||||
return
|
||||
}
|
||||
id, err := strconv.ParseInt(chi.URLParam(r, "inviteID"), 10, 64)
|
||||
if err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid invite id"))
|
||||
return
|
||||
}
|
||||
|
||||
res, err := db.ExecContext(r.Context(),
|
||||
"UPDATE invites SET revoked_at = "+nowEpoch+
|
||||
" WHERE id = $1 AND team_id = $2 AND revoked_at IS NULL", id, teamID)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if n, _ := res.RowsAffected(); n == 0 {
|
||||
respond(w, http.StatusNotFound, errResp("not found"))
|
||||
return
|
||||
}
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Onboarding
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// handleTestNotification publishes one push to the caller's own topic.
|
||||
//
|
||||
// The point of the first-run checklist's notification step is not that a topic
|
||||
// string has been typed but that a phone buzzes, and only the person holding it
|
||||
// can tell whether it did. Published directly rather than through the outbox:
|
||||
// the outbox row requires an incident, and this deliberately belongs to no
|
||||
// incident.
|
||||
func handleTestNotification(cfg NotifyConfig, db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
if cfg.BaseURL == "" {
|
||||
respond(w, http.StatusServiceUnavailable,
|
||||
errResp("this server has no ntfy configured, so it can send nothing"))
|
||||
return
|
||||
}
|
||||
caller, ok := userFromContext(r.Context())
|
||||
if !ok {
|
||||
respond(w, http.StatusForbidden, errResp("this endpoint is for human accounts only"))
|
||||
return
|
||||
}
|
||||
|
||||
var topic *string
|
||||
if err := db.QueryRowContext(r.Context(),
|
||||
"SELECT ntfy_topic FROM users WHERE id = $1", caller.ID).Scan(&topic); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if topic == nil || *topic == "" {
|
||||
respond(w, http.StatusBadRequest, errResp("set a notification topic first"))
|
||||
return
|
||||
}
|
||||
|
||||
if err := publish(r.Context(), cfg, ntfyMessage{
|
||||
Topic: *topic,
|
||||
Title: "terdut test",
|
||||
Message: "If this arrived, your notifications work.",
|
||||
Tags: []string{"white_check_mark"},
|
||||
}); err != nil {
|
||||
// The failure is the useful part here: a wrong topic, a token the
|
||||
// ntfy server rejects, or an ntfy that is down all look the same
|
||||
// from the phone, which is silence.
|
||||
respond(w, http.StatusBadGateway, errResp("ntfy rejected the test: "+err.Error()))
|
||||
return
|
||||
}
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}
|
||||
}
|
||||
|
||||
// handleDismissOnboarding hides the first-run checklist, or brings it back.
|
||||
// Stored per user rather than in the browser: somebody who finishes setting up
|
||||
// on a laptop should not be nagged again on their phone.
|
||||
func handleDismissOnboarding(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
var req struct {
|
||||
Dismissed *bool `json:"dismissed"`
|
||||
}
|
||||
if err := decodeJSON(r, &req); err != nil || req.Dismissed == nil {
|
||||
respond(w, http.StatusBadRequest, errResp("dismissed is required"))
|
||||
return
|
||||
}
|
||||
caller, ok := userFromContext(r.Context())
|
||||
if !ok {
|
||||
respond(w, http.StatusForbidden, errResp("this endpoint is for human accounts only"))
|
||||
return
|
||||
}
|
||||
|
||||
var err error
|
||||
if *req.Dismissed {
|
||||
_, err = db.ExecContext(r.Context(),
|
||||
"UPDATE users SET onboarding_dismissed_at = "+nowEpoch+" WHERE id = $1", caller.ID)
|
||||
} else {
|
||||
_, err = db.ExecContext(r.Context(),
|
||||
"UPDATE users SET onboarding_dismissed_at = NULL WHERE id = $1", caller.ID)
|
||||
}
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,326 @@
|
||||
package api_test
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"database/sql"
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/http/cookiejar"
|
||||
"testing"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||
)
|
||||
|
||||
// signup posts to the unauthenticated sign-up endpoint, the way the form does,
|
||||
// and returns the response and a client holding whatever cookie came back.
|
||||
func signup(t *testing.T, s *ts, body map[string]any) (*http.Response, *http.Client) {
|
||||
t.Helper()
|
||||
data, _ := json.Marshal(body)
|
||||
jar, _ := cookiejar.New(nil)
|
||||
client := &http.Client{Jar: jar}
|
||||
req, _ := http.NewRequest(http.MethodPost, s.URL+"/api/signup", bytes.NewReader(data))
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
t.Fatalf("signup: %v", err)
|
||||
}
|
||||
return resp, client
|
||||
}
|
||||
|
||||
// invite mints a link into the default team and returns its raw token.
|
||||
func invite(t *testing.T, s *ts, role string, maxUses int64) string {
|
||||
t.Helper()
|
||||
var out struct {
|
||||
URL string `json:"url"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/invites",
|
||||
map[string]any{"role": role, "max_uses": maxUses}), &out)
|
||||
if out.URL == "" {
|
||||
t.Fatal("no invite URL returned")
|
||||
}
|
||||
// ...?invite=<token>
|
||||
i := len(out.URL) - 1
|
||||
for ; i >= 0 && out.URL[i] != '='; i-- {
|
||||
}
|
||||
return out.URL[i+1:]
|
||||
}
|
||||
|
||||
func setSignupMode(t *testing.T, s *ts, mode string) {
|
||||
t.Helper()
|
||||
resp := s.req(t, http.MethodPut, "/api/admin/settings", map[string]any{"signup_mode": mode})
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Fatalf("set signup mode: %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// The default is the closed door. An install that gets a public hostname before
|
||||
// anybody has thought about sign-up should not be collecting accounts.
|
||||
func TestSignup_InviteOnlyByDefault(t *testing.T) {
|
||||
s := newTS(t)
|
||||
|
||||
var info map[string]any
|
||||
decode(t, s.req(t, http.MethodGet, "/api/signup", nil), &info)
|
||||
if info["mode"] != "invite_only" {
|
||||
t.Errorf("default sign-up mode is %v, want invite_only", info["mode"])
|
||||
}
|
||||
|
||||
resp, _ := signup(t, s, map[string]any{
|
||||
"username": "stranger", "email": "s@test.com", "password": "correct-horse-battery",
|
||||
})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("sign-up without an invite: expected 403, got %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// An invite carries the team and the role, so redeeming one lands somewhere
|
||||
// usable rather than in an account that sees an empty queue.
|
||||
func TestSignup_InviteCreatesAMemberOfThatTeam(t *testing.T) {
|
||||
s := newTS(t)
|
||||
token := invite(t, s, "member", 1)
|
||||
|
||||
// The form checks the link before asking for a password.
|
||||
var info map[string]any
|
||||
decode(t, s.req(t, http.MethodGet, "/api/signup?invite="+token, nil), &info)
|
||||
if info["invite_valid"] != true {
|
||||
t.Fatalf("a fresh invite should be valid: %v", info)
|
||||
}
|
||||
if info["invite_team"] != "Default" {
|
||||
t.Errorf("the form should name the team: %v", info["invite_team"])
|
||||
}
|
||||
|
||||
resp, client := signup(t, s, map[string]any{
|
||||
"username": "newcomer", "email": "n@test.com",
|
||||
"password": "correct-horse-battery", "invite": token,
|
||||
})
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("redeeming an invite: %d", resp.StatusCode)
|
||||
}
|
||||
var me struct {
|
||||
User struct {
|
||||
ID int64 `json:"id"`
|
||||
IsAdmin bool `json:"is_admin"`
|
||||
} `json:"user"`
|
||||
}
|
||||
decode(t, resp, &me)
|
||||
if me.User.IsAdmin {
|
||||
t.Error("somebody who signs up must not be an administrator")
|
||||
}
|
||||
|
||||
// Signed in already: the cookie came back with the response.
|
||||
got, err := client.Get(s.URL + "/api/teams")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
teams := list(t, got)
|
||||
if len(teams) != 1 || teams[0]["name"] != "Default" || teams[0]["role"] != "member" {
|
||||
t.Errorf("expected membership of Default as member, got %v", teams)
|
||||
}
|
||||
}
|
||||
|
||||
// A single-use link is single-use, and the check is inside the transaction so
|
||||
// two people redeeming the last use at once cannot both get in.
|
||||
func TestSignup_InviteCannotBeUsedTwice(t *testing.T) {
|
||||
s := newTS(t)
|
||||
token := invite(t, s, "member", 1)
|
||||
|
||||
first, _ := signup(t, s, map[string]any{
|
||||
"username": "first", "email": "f@test.com",
|
||||
"password": "correct-horse-battery", "invite": token,
|
||||
})
|
||||
first.Body.Close()
|
||||
if first.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("first redemption: %d", first.StatusCode)
|
||||
}
|
||||
|
||||
second, _ := signup(t, s, map[string]any{
|
||||
"username": "second", "email": "s@test.com",
|
||||
"password": "correct-horse-battery", "invite": token,
|
||||
})
|
||||
second.Body.Close()
|
||||
if second.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("second redemption: expected 403, got %d", second.StatusCode)
|
||||
}
|
||||
|
||||
// And the link reports itself unusable before anybody types a password.
|
||||
var info map[string]any
|
||||
decode(t, s.req(t, http.MethodGet, "/api/signup?invite="+token, nil), &info)
|
||||
if info["invite_valid"] != false {
|
||||
t.Error("a used-up invite should report itself invalid")
|
||||
}
|
||||
}
|
||||
|
||||
// Revoking stops a link without waiting for it to expire.
|
||||
func TestSignup_RevokedInviteStopsWorking(t *testing.T) {
|
||||
s := newTS(t)
|
||||
token := invite(t, s, "member", 5)
|
||||
|
||||
invites := list(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/invites", nil))
|
||||
if len(invites) != 1 {
|
||||
t.Fatalf("expected one invite, got %d", len(invites))
|
||||
}
|
||||
id := int64(invites[0]["id"].(float64))
|
||||
|
||||
resp := s.req(t, http.MethodDelete, "/api/teams/"+defaultTeam+"/invites/"+id64(id), nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Fatalf("revoke: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
used, _ := signup(t, s, map[string]any{
|
||||
"username": "late", "email": "l@test.com",
|
||||
"password": "correct-horse-battery", "invite": token,
|
||||
})
|
||||
used.Body.Close()
|
||||
if used.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("a revoked invite: expected 403, got %d", used.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// Open sign-up makes a team, because an account in no team sees an empty queue
|
||||
// and can be paged by nobody.
|
||||
func TestSignup_OpenModeMakesATeam(t *testing.T) {
|
||||
s := newTS(t)
|
||||
setSignupMode(t, s, "open")
|
||||
|
||||
missing, _ := signup(t, s, map[string]any{
|
||||
"username": "solo", "email": "s@test.com", "password": "correct-horse-battery",
|
||||
})
|
||||
missing.Body.Close()
|
||||
if missing.StatusCode != http.StatusBadRequest {
|
||||
t.Errorf("open sign-up with no team name: expected 400, got %d", missing.StatusCode)
|
||||
}
|
||||
|
||||
resp, client := signup(t, s, map[string]any{
|
||||
"username": "solo", "email": "s@test.com",
|
||||
"password": "correct-horse-battery", "team_name": "Solo",
|
||||
})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("open sign-up: %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
got, err := client.Get(s.URL + "/api/teams")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
teams := list(t, got)
|
||||
if len(teams) != 1 || teams[0]["name"] != "Solo" || teams[0]["role"] != "owner" {
|
||||
t.Errorf("the creator should own their new team, got %v", teams)
|
||||
}
|
||||
}
|
||||
|
||||
// Switching the mode is an administrator's decision, and it takes effect at
|
||||
// once rather than at the next restart.
|
||||
func TestSignup_ModeIsAnAdminSetting(t *testing.T) {
|
||||
s := newTS(t)
|
||||
_, call := member(t, s, "plain")
|
||||
|
||||
resp := call(http.MethodPut, "/api/admin/settings", map[string]any{"signup_mode": "open"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("a member changing the mode: expected 403, got %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
bad := s.req(t, http.MethodPut, "/api/admin/settings", map[string]any{"signup_mode": "everybody"})
|
||||
bad.Body.Close()
|
||||
if bad.StatusCode != http.StatusBadRequest {
|
||||
t.Errorf("an unknown mode: expected 400, got %d", bad.StatusCode)
|
||||
}
|
||||
|
||||
setSignupMode(t, s, "open")
|
||||
var info map[string]any
|
||||
decode(t, s.req(t, http.MethodGet, "/api/signup", nil), &info)
|
||||
if info["mode"] != "open" {
|
||||
t.Errorf("the change should be visible at once, got %v", info["mode"])
|
||||
}
|
||||
}
|
||||
|
||||
// Minting a link is configuring the team, so it is an owner's job.
|
||||
func TestSignup_InvitesAreOwnerOnly(t *testing.T) {
|
||||
s := newTS(t)
|
||||
_, call := member(t, s, "plain")
|
||||
|
||||
resp := call(http.MethodPost, "/api/teams/"+defaultTeam+"/invites", map[string]any{"role": "member"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("a member minting an invite: expected 403, got %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// A password still has to be a password, and a taken username is still taken.
|
||||
func TestSignup_ValidatesLikeTheRestOfTheServer(t *testing.T) {
|
||||
s := newTS(t)
|
||||
token := invite(t, s, "member", 5)
|
||||
|
||||
short, _ := signup(t, s, map[string]any{
|
||||
"username": "shorty", "email": "sh@test.com", "password": "abc", "invite": token,
|
||||
})
|
||||
short.Body.Close()
|
||||
if short.StatusCode != http.StatusBadRequest {
|
||||
t.Errorf("a short password: expected 400, got %d", short.StatusCode)
|
||||
}
|
||||
|
||||
taken, _ := signup(t, s, map[string]any{
|
||||
"username": "admin", "email": "other@test.com",
|
||||
"password": "correct-horse-battery", "invite": token,
|
||||
})
|
||||
taken.Body.Close()
|
||||
if taken.StatusCode != http.StatusConflict {
|
||||
t.Errorf("an existing username: expected 409, got %d", taken.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// A team-scoped service account has no users row to attribute created_by to.
|
||||
// Before this fix, handleCreateInvite wrote its zero-value caller.ID straight
|
||||
// into that (nullable, ON DELETE SET NULL) foreign key instead of leaving it
|
||||
// NULL the way handleCreateServiceAccount already does for the same
|
||||
// situation — a 500, not the 201 TestServiceAccount_TeamScopeManagesItsOwnInvites
|
||||
// now confirms. This pins the column itself ends up NULL, not just "some
|
||||
// response came back".
|
||||
func TestSignup_InviteCreatedByAServiceAccountLeavesCreatedByNull(t *testing.T) {
|
||||
s := newTS(t)
|
||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||
keyA := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||
|
||||
var created struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
decode(t, s.reqAs(t, keyA, http.MethodPost, "/api/teams/"+id64(teamA)+"/invites", map[string]any{}), &created)
|
||||
|
||||
var createdBy sql.NullInt64
|
||||
if err := s.db.QueryRow("SELECT created_by FROM invites WHERE id = $1", created.ID).Scan(&createdBy); err != nil {
|
||||
t.Fatalf("read back invites.created_by: %v", err)
|
||||
}
|
||||
if createdBy.Valid {
|
||||
t.Errorf("expected created_by to be NULL for a service-account-minted invite, got %d", createdBy.Int64)
|
||||
}
|
||||
}
|
||||
|
||||
// Neither of these has a real user_id to act on behalf of; both must 403 a
|
||||
// service account explicitly rather than 500 (handleTestNotification, which
|
||||
// used to query ntfy_topic for user id 0) or silently no-op (handleDismissOnboarding,
|
||||
// which used to UPDATE ... WHERE id = 0, affecting nothing and still
|
||||
// returning 204).
|
||||
func TestServiceAccount_HumanOnlyEndpointsRefuseExplicitly(t *testing.T) {
|
||||
// BaseURL set (even to a fake, unreachable address) so handleTestNotification
|
||||
// reaches its AsHuman() check instead of short-circuiting on "ntfy not
|
||||
// configured" first — this test is about the human-only check, not ntfy.
|
||||
s := newTS(t, api.NotifyConfig{BaseURL: "http://ntfy.invalid"})
|
||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||
|
||||
if resp := s.reqAs(t, instanceKey, http.MethodPost, "/api/me/notify/test", nil); resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("expected 403 for a service account testing notifications, got %d", resp.StatusCode)
|
||||
} else {
|
||||
resp.Body.Close()
|
||||
}
|
||||
if resp := s.reqAs(t, instanceKey, http.MethodPut, "/api/me/onboarding",
|
||||
map[string]bool{"dismissed": true}); resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("expected 403 for a service account dismissing onboarding, got %d", resp.StatusCode)
|
||||
} else {
|
||||
resp.Body.Close()
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,113 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||
)
|
||||
|
||||
const (
|
||||
similarDefaultLimit = 5
|
||||
similarMaxLimit = 20
|
||||
)
|
||||
|
||||
// handleIncidentSimilar lists earlier, resolved incidents in the same team with
|
||||
// the same signature that someone left notes on, incidents with a resolution
|
||||
// note first. This is the "have we seen this before" answer for a responder
|
||||
// looking at a fresh incident; the plain notes are one timeline fetch away.
|
||||
func handleIncidentSimilar(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, ok := incidentIDParam(w, r, db)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
limit := similarDefaultLimit
|
||||
if v := r.URL.Query().Get("limit"); v != "" {
|
||||
n, err := strconv.Atoi(v)
|
||||
if err != nil || n < 1 {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid limit"))
|
||||
return
|
||||
}
|
||||
limit = min(n, similarMaxLimit)
|
||||
}
|
||||
|
||||
out, err := similarIncidents(r.Context(), db, id, limit)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusOK, out)
|
||||
}
|
||||
}
|
||||
|
||||
func similarIncidents(ctx context.Context, q querier, id int64, limit int) ([]models.SimilarIncident, error) {
|
||||
rows, err := q.QueryContext(ctx, `
|
||||
SELECT o.id, o.title, o.triggered_at, o.resolved_at,
|
||||
(SELECT COUNT(*) FROM incident_events e
|
||||
WHERE e.incident_id = o.id AND e.type = $3)
|
||||
FROM incidents i
|
||||
JOIN incidents o ON o.team_id = i.team_id AND o.signature = i.signature
|
||||
WHERE i.id = $1 AND o.id <> i.id AND o.resolved_at IS NOT NULL
|
||||
AND EXISTS (SELECT 1 FROM incident_events e
|
||||
WHERE e.incident_id = o.id AND e.type IN ($3, $4))
|
||||
ORDER BY EXISTS (SELECT 1 FROM incident_events e
|
||||
WHERE e.incident_id = o.id AND e.type = $4) DESC,
|
||||
o.triggered_at DESC
|
||||
LIMIT $2`, id, limit, evNote, evResolutionNote)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
out := []models.SimilarIncident{}
|
||||
ids := []int64{}
|
||||
for rows.Next() {
|
||||
var s models.SimilarIncident
|
||||
var triggered, resolved int64
|
||||
if err := rows.Scan(&s.ID, &s.Title, &triggered, &resolved, &s.NoteCount); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
s.TriggeredAt = time.Unix(triggered, 0).UTC()
|
||||
s.ResolvedAt = time.Unix(resolved, 0).UTC()
|
||||
s.ResolutionNotes = []models.IncidentEvent{}
|
||||
out = append(out, s)
|
||||
ids = append(ids, s.ID)
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if len(out) == 0 {
|
||||
return out, nil
|
||||
}
|
||||
|
||||
nrows, err := q.QueryContext(ctx, `
|
||||
SELECT e.id, e.incident_id, e.type, e.user_id, u.username, e.detail, e.created_at
|
||||
FROM incident_events e
|
||||
LEFT JOIN users u ON u.id = e.user_id
|
||||
WHERE e.incident_id = ANY($1) AND e.type = $2
|
||||
ORDER BY e.created_at, e.id`, ids, evResolutionNote)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer nrows.Close()
|
||||
|
||||
byID := make(map[int64]*models.SimilarIncident, len(out))
|
||||
for i := range out {
|
||||
byID[out[i].ID] = &out[i]
|
||||
}
|
||||
for nrows.Next() {
|
||||
var e models.IncidentEvent
|
||||
var ts int64
|
||||
if err := nrows.Scan(&e.ID, &e.IncidentID, &e.Type, &e.UserID, &e.Username, &e.Detail, &ts); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
e.CreatedAt = time.Unix(ts, 0).UTC()
|
||||
s := byID[e.IncidentID]
|
||||
s.ResolutionNotes = append(s.ResolutionNotes, e)
|
||||
}
|
||||
return out, nrows.Err()
|
||||
}
|
||||
@@ -0,0 +1,99 @@
|
||||
package api_test
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// postGrouped posts a firing webhook whose group labels are exactly the given
|
||||
// map, unlike postWebhook, which only ever groups by alertname.
|
||||
func postGrouped(t *testing.T, s *ts, fingerprint, startsAt string, groupLabels map[string]string) {
|
||||
t.Helper()
|
||||
labels := map[string]string{}
|
||||
for k, v := range groupLabels {
|
||||
labels[k] = v
|
||||
}
|
||||
payload := map[string]any{
|
||||
"version": "4", "status": "firing",
|
||||
"groupKey": fingerprint,
|
||||
"groupLabels": groupLabels,
|
||||
"alerts": []map[string]any{
|
||||
amAlert(fingerprint, groupLabels["alertname"], "firing", startsAt, zeroTime, labels),
|
||||
},
|
||||
}
|
||||
data, _ := json.Marshal(payload)
|
||||
resp, err := http.Post(s.URL+"/api/integrations/"+s.ingestKey+"/alertmanager",
|
||||
"application/json", bytes.NewReader(data))
|
||||
if err != nil {
|
||||
t.Fatalf("post webhook: %v", err)
|
||||
}
|
||||
resp.Body.Close()
|
||||
}
|
||||
|
||||
func similar(t *testing.T, s *ts, id int) []map[string]any {
|
||||
t.Helper()
|
||||
var out []map[string]any
|
||||
decode(t, s.req(t, http.MethodGet, "/api/incidents/"+strconv.Itoa(id)+"/similar", nil), &out)
|
||||
return out
|
||||
}
|
||||
|
||||
// Same alert on another instance is the same problem; a resolution note left on
|
||||
// the first one is what the second one should be shown.
|
||||
func TestSimilar_IgnoresVolatileLabelsAndLeadsWithResolutionNote(t *testing.T) {
|
||||
s := newTS(t)
|
||||
postGrouped(t, s, "fp-a", "2026-05-20T10:00:00Z",
|
||||
map[string]string{"alertname": "DiskFull", "instance": "web-1", "job": "node"})
|
||||
s.req(t, http.MethodPost, "/api/incidents/1/resolve",
|
||||
map[string]string{"resolution": "rotated the logs"}).Body.Close()
|
||||
|
||||
postGrouped(t, s, "fp-b", "2026-05-21T10:00:00Z",
|
||||
map[string]string{"alertname": "DiskFull", "instance": "web-2", "job": "node"})
|
||||
|
||||
got := similar(t, s, 2)
|
||||
if len(got) != 1 || int(got[0]["id"].(float64)) != 1 {
|
||||
t.Fatalf("expected incident 1 as the only similar one, got %v", got)
|
||||
}
|
||||
notes := got[0]["resolution_notes"].([]any)
|
||||
if len(notes) != 1 || notes[0].(map[string]any)["detail"] != "rotated the logs" {
|
||||
t.Fatalf("expected the resolution note, got %v", notes)
|
||||
}
|
||||
}
|
||||
|
||||
// A different stable label (job) is a different problem, and an incident nobody
|
||||
// wrote a note on has nothing to show.
|
||||
func TestSimilar_DifferentSignatureOrNoNotesIsExcluded(t *testing.T) {
|
||||
s := newTS(t)
|
||||
postGrouped(t, s, "fp-1", "2026-05-20T10:00:00Z",
|
||||
map[string]string{"alertname": "DiskFull", "job": "node"})
|
||||
s.req(t, http.MethodPost, "/api/incidents/1/notes", map[string]string{"content": "checked"}).Body.Close()
|
||||
s.req(t, http.MethodPost, "/api/incidents/1/resolve", nil).Body.Close()
|
||||
|
||||
postGrouped(t, s, "fp-2", "2026-05-20T11:00:00Z",
|
||||
map[string]string{"alertname": "DiskFull", "job": "db"})
|
||||
s.req(t, http.MethodPost, "/api/incidents/2/resolve", nil).Body.Close()
|
||||
|
||||
postGrouped(t, s, "fp-3", "2026-05-21T10:00:00Z",
|
||||
map[string]string{"alertname": "DiskFull", "job": "db"})
|
||||
|
||||
// Incident 3 matches 2 by signature, but 2 has no notes.
|
||||
if got := similar(t, s, 3); len(got) != 0 {
|
||||
t.Fatalf("expected nothing similar to incident 3, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
// An open incident is not "earlier experience" yet, and the incident itself is
|
||||
// never its own match.
|
||||
func TestSimilar_OpenIncidentsAreNotListed(t *testing.T) {
|
||||
s := newTS(t)
|
||||
postGrouped(t, s, "fp-o1", "2026-05-20T10:00:00Z", map[string]string{"alertname": "Flap"})
|
||||
s.req(t, http.MethodPost, "/api/incidents/1/notes",
|
||||
map[string]any{"content": "still open", "pinned": true}).Body.Close()
|
||||
postGrouped(t, s, "fp-o2", "2026-05-21T10:00:00Z", map[string]string{"alertname": "Flap"})
|
||||
|
||||
if got := similar(t, s, 2); len(got) != 0 {
|
||||
t.Fatalf("expected an open incident not to be listed, got %v", got)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,171 @@
|
||||
package api_test
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"net/http"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// listSources reads a team's alert sources as the Sources page does.
|
||||
func listSources(t *testing.T, tm teamFixture) []map[string]any {
|
||||
t.Helper()
|
||||
return list(t, tm.call(http.MethodGet, "/api/teams/"+id64(tm.id)+"/integrations", nil))
|
||||
}
|
||||
|
||||
// addSource mints a second source in a team and returns its key.
|
||||
func addSource(t *testing.T, tm teamFixture, name string) string {
|
||||
t.Helper()
|
||||
var out struct {
|
||||
Key string `json:"key"`
|
||||
}
|
||||
decode(t, tm.call(http.MethodPost, "/api/teams/"+id64(tm.id)+"/integrations",
|
||||
map[string]string{"name": name}), &out)
|
||||
return out.Key
|
||||
}
|
||||
|
||||
// A source that has never posted is "never", with nothing to say about alerts.
|
||||
func TestSources_NeverUsedIsBlank(t *testing.T) {
|
||||
s := newTS(t)
|
||||
tm := newTeam(t, s, "red")
|
||||
|
||||
got := listSources(t, tm)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("expected 1 source, got %d", len(got))
|
||||
}
|
||||
src := got[0]
|
||||
if src["status"] != "never" || src["last_used_at"] != nil || src["last_alert_at"] != nil {
|
||||
t.Errorf("a source nobody has posted on should be blank, got %v", src)
|
||||
}
|
||||
if src["alerts_24h"].(float64) != 0 {
|
||||
t.Errorf("alerts_24h = %v, want 0", src["alerts_24h"])
|
||||
}
|
||||
}
|
||||
|
||||
// Each source is credited with what arrived on its own key, and only that.
|
||||
func TestSources_AlertsAreAttributedToTheirSource(t *testing.T) {
|
||||
s := newTS(t)
|
||||
tm := newTeam(t, s, "red")
|
||||
second := addSource(t, tm, "staging")
|
||||
|
||||
postToIntegration(t, s, tm.key, "fp-1", "DiskFull")
|
||||
postToIntegration(t, s, tm.key, "fp-2", "CPUHot")
|
||||
|
||||
got := listSources(t, tm)
|
||||
first, other := got[0], got[1]
|
||||
if first["status"] != "active" || first["last_used_at"] == nil || first["last_alert_at"] == nil {
|
||||
t.Errorf("the source that posted should be active with timestamps, got %v", first)
|
||||
}
|
||||
if first["alerts_24h"].(float64) != 2 {
|
||||
t.Errorf("alerts_24h = %v, want 2", first["alerts_24h"])
|
||||
}
|
||||
if other["status"] != "never" || other["alerts_24h"].(float64) != 0 {
|
||||
t.Errorf("the other source should be untouched, got %v", other)
|
||||
}
|
||||
|
||||
// Re-sending the same alert on the other key moves it: last sender wins.
|
||||
postToIntegration(t, s, second, "fp-1", "DiskFull")
|
||||
got = listSources(t, tm)
|
||||
if got[0]["alerts_24h"].(float64) != 1 || got[1]["alerts_24h"].(float64) != 1 {
|
||||
t.Errorf("fp-1 should have moved to the second source, got %v and %v",
|
||||
got[0]["alerts_24h"], got[1]["alerts_24h"])
|
||||
}
|
||||
}
|
||||
|
||||
// A payload with no alerts in it is a webhook, not an alert: the source was
|
||||
// heard from, and nothing arrived.
|
||||
func TestSources_EmptyPayloadStampsUseButNotAlert(t *testing.T) {
|
||||
s := newTS(t)
|
||||
tm := newTeam(t, s, "red")
|
||||
|
||||
resp, err := http.Post(s.URL+"/api/integrations/"+tm.key+"/alertmanager",
|
||||
"application/json", bytes.NewReader([]byte(`{"version":"4","status":"firing","alerts":[]}`)))
|
||||
if err != nil {
|
||||
t.Fatalf("post: %v", err)
|
||||
}
|
||||
resp.Body.Close()
|
||||
|
||||
src := listSources(t, tm)[0]
|
||||
if src["status"] != "active" || src["last_alert_at"] != nil {
|
||||
t.Errorf("want active with no alert yet, got %v", src)
|
||||
}
|
||||
}
|
||||
|
||||
// Quiet is "has posted, not lately"; the alert counter forgets after a day but
|
||||
// the last alert's timestamp is kept.
|
||||
func TestSources_QuietAfterADay(t *testing.T) {
|
||||
s := newTS(t)
|
||||
tm := newTeam(t, s, "red")
|
||||
postToIntegration(t, s, tm.key, "fp-1", "DiskFull")
|
||||
|
||||
old := time.Now().Add(-48 * time.Hour).Unix()
|
||||
s.exec(t, "UPDATE integrations SET last_used_at = $1", old)
|
||||
s.exec(t, "UPDATE alerts SET received_at = $1 WHERE fingerprint = 'fp-1'", old)
|
||||
|
||||
src := listSources(t, tm)[0]
|
||||
if src["status"] != "quiet" {
|
||||
t.Errorf("status = %v, want quiet", src["status"])
|
||||
}
|
||||
if src["alerts_24h"].(float64) != 0 {
|
||||
t.Errorf("alerts_24h = %v, want 0", src["alerts_24h"])
|
||||
}
|
||||
if src["last_alert_at"] == nil {
|
||||
t.Error("last_alert_at should survive the day")
|
||||
}
|
||||
}
|
||||
|
||||
// Revoking a source does not take its alerts with it.
|
||||
func TestSources_RevokeKeepsTheAlerts(t *testing.T) {
|
||||
s := newTS(t)
|
||||
tm := newTeam(t, s, "red")
|
||||
postToIntegration(t, s, tm.key, "fp-1", "DiskFull")
|
||||
|
||||
id := int64(listSources(t, tm)[0]["id"].(float64))
|
||||
resp := tm.call(http.MethodDelete, "/api/teams/"+id64(tm.id)+"/integrations/"+id64(id), nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Fatalf("revoke: %d", resp.StatusCode)
|
||||
}
|
||||
if got := len(list(t, tm.call(http.MethodGet, "/api/alerts", nil))); got != 1 {
|
||||
t.Errorf("the alert should outlive its source, got %d alerts", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Renaming is an owner's, scoped to the team, and does not touch the key.
|
||||
func TestSources_Rename(t *testing.T) {
|
||||
s := newTS(t)
|
||||
tm := newTeam(t, s, "red")
|
||||
other := newTeam(t, s, "blue")
|
||||
id := int64(listSources(t, tm)[0]["id"].(float64))
|
||||
path := "/api/teams/" + id64(tm.id) + "/integrations/" + id64(id)
|
||||
|
||||
resp := tm.call(http.MethodPatch, path, map[string]string{"name": " prod "})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Fatalf("rename: %d", resp.StatusCode)
|
||||
}
|
||||
if name := listSources(t, tm)[0]["name"]; name != "prod" {
|
||||
t.Errorf("name = %q, want it trimmed to prod", name)
|
||||
}
|
||||
postToIntegration(t, s, tm.key, "fp-1", "DiskFull") // the old key still works
|
||||
|
||||
for name, body := range map[string]map[string]string{
|
||||
"empty": {"name": " "},
|
||||
"too long": {"name": strings.Repeat("x", 101)},
|
||||
} {
|
||||
resp := tm.call(http.MethodPatch, path, body)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusBadRequest {
|
||||
t.Errorf("%s name: expected 400, got %d", name, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// Another team's owner cannot reach it.
|
||||
resp = other.call(http.MethodPatch, "/api/teams/"+id64(other.id)+"/integrations/"+id64(id),
|
||||
map[string]string{"name": "mine now"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNotFound {
|
||||
t.Errorf("renaming another team's source: expected 404, got %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
+11
-7
@@ -11,7 +11,7 @@ import (
|
||||
|
||||
func handleStatsAlerts(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
where, args := statsFilter(r.URL.Query(), "received_at")
|
||||
where, args := statsFilter(r.URL.Query(), "received_at", callerTeamIDs(r.Context()))
|
||||
|
||||
// COALESCE because SUM over zero rows is NULL, not 0, and a count of
|
||||
// nothing is 0 — without it an empty window is a 500 rather than a
|
||||
@@ -37,7 +37,7 @@ func handleStatsAlerts(db *sql.DB) http.HandlerFunc {
|
||||
|
||||
func handleStatsTop(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
where, args := statsFilter(r.URL.Query(), "received_at")
|
||||
where, args := statsFilter(r.URL.Query(), "received_at", callerTeamIDs(r.Context()))
|
||||
|
||||
limit := 10
|
||||
if l := r.URL.Query().Get("limit"); l != "" {
|
||||
@@ -79,7 +79,7 @@ func handleStatsTop(db *sql.DB) http.HandlerFunc {
|
||||
|
||||
func handleStatsByHour(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
where, args := statsFilter(r.URL.Query(), "received_at")
|
||||
where, args := statsFilter(r.URL.Query(), "received_at", callerTeamIDs(r.Context()))
|
||||
|
||||
rows, err := db.QueryContext(r.Context(), fmt.Sprintf(`
|
||||
SELECT EXTRACT(HOUR FROM to_timestamp(received_at) AT TIME ZONE 'UTC')::int AS hr,
|
||||
@@ -119,7 +119,7 @@ func handleStatsByHour(db *sql.DB) http.HandlerFunc {
|
||||
|
||||
func handleStatsByDay(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
where, args := statsFilter(r.URL.Query(), "received_at")
|
||||
where, args := statsFilter(r.URL.Query(), "received_at", callerTeamIDs(r.Context()))
|
||||
|
||||
// Postgres EXTRACT(DOW …) → 0=Sunday … 6=Saturday, the same numbering
|
||||
// SQLite's strftime('%w') returned, so the frontend needs no change.
|
||||
@@ -167,7 +167,7 @@ func handleStatsByDay(db *sql.DB) http.HandlerFunc {
|
||||
// mutated in place and carry no acknowledgement or closure time.
|
||||
func handleStatsIncidents(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
where, args := statsFilter(r.URL.Query(), "triggered_at")
|
||||
where, args := statsFilter(r.URL.Query(), "triggered_at", callerTeamIDs(r.Context()))
|
||||
|
||||
// The counts are COALESCEd because SUM over zero rows is NULL, not 0.
|
||||
// The averages are not: mtta and mttr stay null on purpose, since zero
|
||||
@@ -206,9 +206,13 @@ func handleStatsIncidents(db *sql.DB) http.HandlerFunc {
|
||||
// statsFilter builds a WHERE clause and args from optional ?from and ?to query
|
||||
// params, filtering on timeCol. Archived rows are always excluded, matching the
|
||||
// default list views.
|
||||
func statsFilter(q url.Values, timeCol string) (where string, args *sqlArgs) {
|
||||
//
|
||||
// teamIDs scopes every figure to the caller's own teams: a report that counted
|
||||
// other teams' incidents would leak their volume and their names through the
|
||||
// top-alerts list, and would not be a number about the reader's work anyway.
|
||||
func statsFilter(q url.Values, timeCol string, teamIDs []int64) (where string, args *sqlArgs) {
|
||||
args = &sqlArgs{}
|
||||
clauses := []string{"archived_at IS NULL"}
|
||||
clauses := []string{"archived_at IS NULL", "team_id = ANY(" + args.add(teamIDs) + ")"}
|
||||
if from := q.Get("from"); from != "" {
|
||||
if t, err := time.Parse("2006-01-02", from); err == nil {
|
||||
clauses = append(clauses, timeCol+" >= "+args.add(t.UTC().Unix()))
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,530 @@
|
||||
package api_test
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"io"
|
||||
"net/http"
|
||||
"testing"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||
)
|
||||
|
||||
// The whole point of #4: two teams sharing one server must not see each other's
|
||||
// work. These tests build two of them and check the boundary from both sides.
|
||||
|
||||
type teamFixture struct {
|
||||
id int64
|
||||
key string // integration key: how alerts get in
|
||||
call func(method, path string, body any) *http.Response
|
||||
}
|
||||
|
||||
// newTeam creates a team with its own member, integration key and API key. The
|
||||
// admin does the creating, as an install's first user would.
|
||||
func newTeam(t *testing.T, s *ts, name string) teamFixture {
|
||||
t.Helper()
|
||||
|
||||
var team struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodPost, "/api/teams", map[string]string{"name": name}), &team)
|
||||
|
||||
var integration struct {
|
||||
Key string `json:"key"`
|
||||
URL string `json:"url"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodPost, "/api/teams/"+id64(team.ID)+"/integrations",
|
||||
map[string]string{"name": name + " alertmanager"}), &integration)
|
||||
if integration.Key == "" {
|
||||
t.Fatalf("%s: integration key was not returned", name)
|
||||
}
|
||||
|
||||
// A member of this team and no other.
|
||||
var user struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodPost, "/api/users",
|
||||
map[string]string{"username": name + "-user", "email": name + "@test.com"}), &user)
|
||||
|
||||
resp := s.req(t, http.MethodPost, "/api/teams/"+id64(team.ID)+"/members",
|
||||
map[string]any{"user_id": user.ID, "role": "owner"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Fatalf("%s: add member: %d", name, resp.StatusCode)
|
||||
}
|
||||
|
||||
var key struct {
|
||||
Key string `json:"key"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodPost, "/api/users/"+id64(user.ID)+"/api-keys",
|
||||
map[string]string{"name": "test"}), &key)
|
||||
|
||||
return teamFixture{
|
||||
id: team.ID,
|
||||
key: integration.Key,
|
||||
call: func(method, path string, body any) *http.Response {
|
||||
t.Helper()
|
||||
var r io.Reader
|
||||
if body != nil {
|
||||
data, _ := json.Marshal(body)
|
||||
r = bytes.NewReader(data)
|
||||
}
|
||||
req, _ := http.NewRequest(method, s.URL+path, r)
|
||||
req.Header.Set("Authorization", "Bearer "+key.Key)
|
||||
if body != nil {
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
}
|
||||
resp, err := http.DefaultClient.Do(req)
|
||||
if err != nil {
|
||||
t.Fatalf("%s %s: %v", method, path, err)
|
||||
}
|
||||
return resp
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// postToIntegration sends one firing alert on a team's integration key, the way
|
||||
// a real Alertmanager receiver would.
|
||||
func postToIntegration(t *testing.T, s *ts, key, fingerprint, name string) {
|
||||
t.Helper()
|
||||
payload := map[string]any{
|
||||
"version": "4",
|
||||
"status": "firing",
|
||||
"groupKey": "{}:{alertname=\"" + name + "\"}",
|
||||
"groupLabels": map[string]string{"alertname": name},
|
||||
"alerts": []map[string]any{
|
||||
amAlert(fingerprint, name, "firing", "2026-09-20T10:00:00Z", zeroTime, nil),
|
||||
},
|
||||
}
|
||||
data, _ := json.Marshal(payload)
|
||||
resp, err := http.Post(s.URL+"/api/integrations/"+key+"/alertmanager",
|
||||
"application/json", bytes.NewReader(data))
|
||||
if err != nil {
|
||||
t.Fatalf("post alert: %v", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("post alert: %d", resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
func list(t *testing.T, resp *http.Response) []map[string]any {
|
||||
t.Helper()
|
||||
var out []map[string]any
|
||||
decode(t, resp, &out)
|
||||
return out
|
||||
}
|
||||
|
||||
// An alert posted on one team's key opens an incident in that team and nowhere
|
||||
// else, and neither team can read the other's queue.
|
||||
func TestTeams_IncidentsAreScopedToTheReceivingTeam(t *testing.T) {
|
||||
s := newTS(t)
|
||||
red := newTeam(t, s, "red")
|
||||
blue := newTeam(t, s, "blue")
|
||||
|
||||
postToIntegration(t, s, red.key, "fp-red", "RedDiskFull")
|
||||
postToIntegration(t, s, blue.key, "fp-blue", "BlueDiskFull")
|
||||
|
||||
redIncidents := list(t, red.call(http.MethodGet, "/api/incidents", nil))
|
||||
if len(redIncidents) != 1 {
|
||||
t.Fatalf("red should see exactly its own incident, saw %d", len(redIncidents))
|
||||
}
|
||||
if title := redIncidents[0]["title"]; title != "RedDiskFull" {
|
||||
t.Errorf("red saw %v", title)
|
||||
}
|
||||
if teamID := int64(redIncidents[0]["team_id"].(float64)); teamID != red.id {
|
||||
t.Errorf("red's incident belongs to team %d, want %d", teamID, red.id)
|
||||
}
|
||||
|
||||
blueIncidents := list(t, blue.call(http.MethodGet, "/api/incidents", nil))
|
||||
if len(blueIncidents) != 1 || blueIncidents[0]["title"] != "BlueDiskFull" {
|
||||
t.Fatalf("blue should see exactly its own incident, saw %v", blueIncidents)
|
||||
}
|
||||
|
||||
// Reading the other team's incident by id is not found rather than
|
||||
// forbidden: its existence is the other team's business.
|
||||
otherID := int64(blueIncidents[0]["id"].(float64))
|
||||
for _, path := range []string{
|
||||
"/api/incidents/" + id64(otherID),
|
||||
"/api/incidents/" + id64(otherID) + "/alerts",
|
||||
"/api/incidents/" + id64(otherID) + "/timeline",
|
||||
} {
|
||||
resp := red.call(http.MethodGet, path, nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNotFound {
|
||||
t.Errorf("red reading %s: expected 404, got %d", path, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// And cannot act on it either.
|
||||
for _, path := range []string{"/acknowledge", "/resolve", "/archive"} {
|
||||
resp := red.call(http.MethodPost, "/api/incidents/"+id64(otherID)+path, nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNotFound {
|
||||
t.Errorf("red posting %s: expected 404, got %d", path, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Alerts, the raw signal record, are scoped the same way.
|
||||
func TestTeams_AlertsAndStatsAreScoped(t *testing.T) {
|
||||
s := newTS(t)
|
||||
red := newTeam(t, s, "red")
|
||||
blue := newTeam(t, s, "blue")
|
||||
|
||||
postToIntegration(t, s, red.key, "fp-red", "RedDiskFull")
|
||||
postToIntegration(t, s, blue.key, "fp-blue-1", "BlueDiskFull")
|
||||
postToIntegration(t, s, blue.key, "fp-blue-2", "BlueMemory")
|
||||
|
||||
if alerts := list(t, red.call(http.MethodGet, "/api/alerts", nil)); len(alerts) != 1 {
|
||||
t.Errorf("red should see 1 alert, saw %d", len(alerts))
|
||||
}
|
||||
if alerts := list(t, blue.call(http.MethodGet, "/api/alerts", nil)); len(alerts) != 2 {
|
||||
t.Errorf("blue should see 2 alerts, saw %d", len(alerts))
|
||||
}
|
||||
|
||||
// Statistics count your own work only — otherwise a team's volume, and the
|
||||
// names of its alerts, leak through the totals.
|
||||
var stats map[string]any
|
||||
decode(t, red.call(http.MethodGet, "/api/stats/alerts", nil), &stats)
|
||||
if total := stats["total"].(float64); total != 1 {
|
||||
t.Errorf("red's alert stats counted %v alerts, want 1", total)
|
||||
}
|
||||
|
||||
top := list(t, red.call(http.MethodGet, "/api/stats/alerts/top", nil))
|
||||
for _, row := range top {
|
||||
if name := row["name"].(string); name != "RedDiskFull" {
|
||||
t.Errorf("red's top alerts named %q, which is not theirs", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The same fingerprint, the same groupKey and the same date are all legitimate
|
||||
// in two teams at once: two clusters running the same rules, two rotas.
|
||||
func TestTeams_SameFingerprintInTwoTeams(t *testing.T) {
|
||||
s := newTS(t)
|
||||
red := newTeam(t, s, "red")
|
||||
blue := newTeam(t, s, "blue")
|
||||
|
||||
postToIntegration(t, s, red.key, "fp-shared", "DiskFull")
|
||||
postToIntegration(t, s, blue.key, "fp-shared", "DiskFull")
|
||||
|
||||
for _, team := range []struct {
|
||||
name string
|
||||
f teamFixture
|
||||
}{{"red", red}, {"blue", blue}} {
|
||||
incidents := list(t, team.f.call(http.MethodGet, "/api/incidents", nil))
|
||||
if len(incidents) != 1 {
|
||||
t.Errorf("%s: expected its own incident for the shared fingerprint, saw %d",
|
||||
team.name, len(incidents))
|
||||
}
|
||||
}
|
||||
|
||||
// And both rotas can name somebody for the same day.
|
||||
for _, team := range []struct {
|
||||
name string
|
||||
f teamFixture
|
||||
}{{"red", red}, {"blue", blue}} {
|
||||
var members []map[string]any
|
||||
decode(t, team.f.call(http.MethodGet, "/api/teams/"+id64(team.f.id)+"/members", nil), &members)
|
||||
userID := int64(members[0]["user_id"].(float64))
|
||||
|
||||
resp := team.f.call(http.MethodPost, "/api/teams/"+id64(team.f.id)+"/schedule",
|
||||
map[string]any{"user_id": userID, "dates": []string{"2026-10-01"}})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Errorf("%s: taking 2026-10-01 returned %d", team.name, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// An unknown key delivers nothing, and says so rather than accepting silently.
|
||||
func TestTeams_UnknownIntegrationKeyIsRejected(t *testing.T) {
|
||||
s := newTS(t)
|
||||
team := newTeam(t, s, "red")
|
||||
|
||||
resp, err := http.Post(s.URL+"/api/integrations/not-a-real-key/alertmanager",
|
||||
"application/json", bytes.NewReader([]byte(`{"version":"4","status":"firing","alerts":[]}`)))
|
||||
if err != nil {
|
||||
t.Fatalf("post: %v", err)
|
||||
}
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusUnauthorized {
|
||||
t.Errorf("expected 401 for an unknown key, got %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
if incidents := list(t, team.call(http.MethodGet, "/api/incidents", nil)); len(incidents) != 0 {
|
||||
t.Errorf("a rejected payload opened %d incident(s)", len(incidents))
|
||||
}
|
||||
}
|
||||
|
||||
// Team configuration is an owner's job; working incidents is a member's.
|
||||
func TestTeams_MemberCannotConfigureTheTeam(t *testing.T) {
|
||||
s := newTS(t)
|
||||
team := newTeam(t, s, "red")
|
||||
|
||||
// A plain member of the same team.
|
||||
var user struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodPost, "/api/users",
|
||||
map[string]string{"username": "plain", "email": "plain@test.com"}), &user)
|
||||
resp := s.req(t, http.MethodPost, "/api/teams/"+id64(team.id)+"/members",
|
||||
map[string]any{"user_id": user.ID, "role": "member"})
|
||||
resp.Body.Close()
|
||||
var key struct {
|
||||
Key string `json:"key"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodPost, "/api/users/"+id64(user.ID)+"/api-keys",
|
||||
map[string]string{"name": "test"}), &key)
|
||||
|
||||
call := func(method, path string, body any) *http.Response {
|
||||
t.Helper()
|
||||
var r io.Reader
|
||||
if body != nil {
|
||||
data, _ := json.Marshal(body)
|
||||
r = bytes.NewReader(data)
|
||||
}
|
||||
req, _ := http.NewRequest(method, s.URL+path, r)
|
||||
req.Header.Set("Authorization", "Bearer "+key.Key)
|
||||
if body != nil {
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
}
|
||||
resp, err := http.DefaultClient.Do(req)
|
||||
if err != nil {
|
||||
t.Fatalf("%s %s: %v", method, path, err)
|
||||
}
|
||||
return resp
|
||||
}
|
||||
|
||||
base := "/api/teams/" + id64(team.id)
|
||||
for _, c := range []struct {
|
||||
name string
|
||||
method string
|
||||
path string
|
||||
body any
|
||||
}{
|
||||
{"mint an integration key", http.MethodPost, base + "/integrations",
|
||||
map[string]string{"name": "mine"}},
|
||||
{"take a shift", http.MethodPost, base + "/schedule",
|
||||
map[string]any{"user_id": user.ID, "dates": []string{"2026-11-01"}}},
|
||||
{"add a member", http.MethodPost, base + "/members",
|
||||
map[string]any{"user_id": 1}},
|
||||
{"delete the team", http.MethodDelete, base, nil},
|
||||
} {
|
||||
resp := call(c.method, c.path, c.body)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("%s: expected 403, got %d", c.name, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// But they can read what the team is doing.
|
||||
for _, path := range []string{base + "/members", base + "/integrations", base + "/schedule"} {
|
||||
resp := call(http.MethodGet, path, nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Errorf("reading %s: expected 200, got %d", path, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A team's own OIDC group binding follows the same rule as its schedule and
|
||||
// its integrations: an owner sets it, a member may only read it, an outsider
|
||||
// learns nothing, and an administrator can still reach it to repair a team
|
||||
// whose owner has left.
|
||||
func TestTeamOIDCGroups_OwnerOnlyToEdit(t *testing.T) {
|
||||
s := newTS(t)
|
||||
team := newTeam(t, s, "sre") // team.call authenticates as its owner
|
||||
|
||||
// A plain member of the same team.
|
||||
var plain struct {
|
||||
ID int64 `json:"id"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodPost, "/api/users",
|
||||
map[string]string{"username": "plain", "email": "plain@test.com"}), &plain)
|
||||
resp := s.req(t, http.MethodPost, "/api/teams/"+id64(team.id)+"/members",
|
||||
map[string]any{"user_id": plain.ID, "role": "member"})
|
||||
resp.Body.Close()
|
||||
var key struct {
|
||||
Key string `json:"key"`
|
||||
}
|
||||
decode(t, s.req(t, http.MethodPost, "/api/users/"+id64(plain.ID)+"/api-keys",
|
||||
map[string]string{"name": "test"}), &key)
|
||||
memberCall := func(method, path string, body any) *http.Response {
|
||||
t.Helper()
|
||||
var r io.Reader
|
||||
if body != nil {
|
||||
data, _ := json.Marshal(body)
|
||||
r = bytes.NewReader(data)
|
||||
}
|
||||
req, _ := http.NewRequest(method, s.URL+path, r)
|
||||
req.Header.Set("Authorization", "Bearer "+key.Key)
|
||||
if body != nil {
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
}
|
||||
resp, err := http.DefaultClient.Do(req)
|
||||
if err != nil {
|
||||
t.Fatalf("%s %s: %v", method, path, err)
|
||||
}
|
||||
return resp
|
||||
}
|
||||
|
||||
// A member of a different team altogether.
|
||||
_, outsiderCall := member(t, s, "outsider")
|
||||
|
||||
path := "/api/teams/" + id64(team.id) + "/oidc-groups"
|
||||
|
||||
resp = team.call(http.MethodPut, path, map[string]string{"member_group": "sre", "owner_group": "sre-leads"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Errorf("owner PUT: %d, want 204", resp.StatusCode)
|
||||
}
|
||||
var got struct {
|
||||
MemberGroup string `json:"member_group"`
|
||||
OwnerGroup string `json:"owner_group"`
|
||||
}
|
||||
decode(t, team.call(http.MethodGet, path, nil), &got)
|
||||
if got.MemberGroup != "sre" || got.OwnerGroup != "sre-leads" {
|
||||
t.Errorf("owner GET after PUT: %+v", got)
|
||||
}
|
||||
|
||||
resp = memberCall(http.MethodGet, path, nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Errorf("member GET: %d, want 200", resp.StatusCode)
|
||||
}
|
||||
resp = memberCall(http.MethodPut, path, map[string]string{"member_group": "anything"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusForbidden {
|
||||
t.Errorf("member PUT: %d, want 403", resp.StatusCode)
|
||||
}
|
||||
|
||||
// 404, not 403: whether the team exists is itself something only its
|
||||
// members should learn.
|
||||
resp = outsiderCall(http.MethodGet, path, nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNotFound {
|
||||
t.Errorf("outsider GET: %d, want 404", resp.StatusCode)
|
||||
}
|
||||
resp = outsiderCall(http.MethodPut, path, map[string]string{"member_group": "anything"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNotFound {
|
||||
t.Errorf("outsider PUT: %d, want 404", resp.StatusCode)
|
||||
}
|
||||
|
||||
// An administrator who is not a member may still set it, the same bypass
|
||||
// that lets one repair a team whose owner has left.
|
||||
resp = s.req(t, http.MethodPut, path, map[string]string{"member_group": "sre2"})
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
t.Errorf("admin PUT: %d, want 204", resp.StatusCode)
|
||||
}
|
||||
|
||||
// An empty string clears a binding, stored as NULL rather than the literal
|
||||
// empty string, so an empty group claim can never accidentally match it.
|
||||
resp = team.call(http.MethodPut, path, map[string]string{"member_group": "", "owner_group": ""})
|
||||
resp.Body.Close()
|
||||
var cleared struct {
|
||||
MemberGroup string `json:"member_group"`
|
||||
OwnerGroup string `json:"owner_group"`
|
||||
}
|
||||
decode(t, team.call(http.MethodGet, path, nil), &cleared)
|
||||
if cleared.MemberGroup != "" || cleared.OwnerGroup != "" {
|
||||
t.Errorf("cleared: %+v", cleared)
|
||||
}
|
||||
}
|
||||
|
||||
// A team is not somewhere an outsider can look, whatever they know about it.
|
||||
func TestTeams_OutsiderSeesNothing(t *testing.T) {
|
||||
s := newTS(t)
|
||||
red := newTeam(t, s, "red")
|
||||
blue := newTeam(t, s, "blue")
|
||||
|
||||
base := "/api/teams/" + id64(red.id)
|
||||
for _, path := range []string{base + "/members", base + "/integrations", base + "/schedule"} {
|
||||
resp := blue.call(http.MethodGet, path, nil)
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusNotFound {
|
||||
t.Errorf("blue reading %s: expected 404, got %d", path, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
// /api/teams lists your own, never the install's.
|
||||
teams := list(t, blue.call(http.MethodGet, "/api/teams", nil))
|
||||
if len(teams) != 1 || teams[0]["name"] != "blue" {
|
||||
t.Errorf("blue's team list: %v", teams)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// GET /api/teams?name= (TEAM-LOOKUP.md)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
func TestListTeamsByName_FindsExactMatch(t *testing.T) {
|
||||
s := newTS(t)
|
||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||
teamID := createTeamAs(t, s, instanceKey, "platform")
|
||||
|
||||
teams := list(t, s.reqAs(t, instanceKey, http.MethodGet, "/api/teams?name=platform", nil))
|
||||
if len(teams) != 1 {
|
||||
t.Fatalf("expected exactly one match for ?name=platform, got %d: %v", len(teams), teams)
|
||||
}
|
||||
if int64(teams[0]["id"].(float64)) != teamID {
|
||||
t.Errorf("id = %v, want %d", teams[0]["id"], teamID)
|
||||
}
|
||||
// No membership, so no role to report (models.Team's own doc comment:
|
||||
// "empty when nobody in particular is asking").
|
||||
if _, has := teams[0]["role"]; has {
|
||||
t.Errorf("expected no role on a name-lookup match, got %v", teams[0]["role"])
|
||||
}
|
||||
}
|
||||
|
||||
func TestListTeamsByName_NoMatchIsAnEmptyArrayNotAnError(t *testing.T) {
|
||||
s := newTS(t)
|
||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||
|
||||
resp := s.reqAs(t, instanceKey, http.MethodGet, "/api/teams?name=does-not-exist", nil)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("expected 200 on no match, got %d", resp.StatusCode)
|
||||
}
|
||||
teams := list(t, resp)
|
||||
if len(teams) != 0 {
|
||||
t.Errorf("expected an empty array, got %v", teams)
|
||||
}
|
||||
}
|
||||
|
||||
// The actual motivating scenario (TEAM-LOOKUP.md): a service account that
|
||||
// already created a team, interrupted before it could remember the id,
|
||||
// recovers it via ?name= on the same name its own POST 409s on.
|
||||
func TestListTeamsByName_RecoversAfterCreateConflict(t *testing.T) {
|
||||
s := newTS(t)
|
||||
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||
original := createTeamAs(t, s, instanceKey, "recovered")
|
||||
|
||||
conflict := s.reqAs(t, instanceKey, http.MethodPost, "/api/teams", map[string]string{"name": "recovered"})
|
||||
if conflict.StatusCode != http.StatusConflict {
|
||||
t.Fatalf("expected 409 recreating the same name, got %d", conflict.StatusCode)
|
||||
}
|
||||
conflict.Body.Close()
|
||||
|
||||
teams := list(t, s.reqAs(t, instanceKey, http.MethodGet, "/api/teams?name=recovered", nil))
|
||||
if len(teams) != 1 || int64(teams[0]["id"].(float64)) != original {
|
||||
t.Fatalf("expected to recover the original team %d via ?name=, got %v", original, teams)
|
||||
}
|
||||
}
|
||||
|
||||
// Not gated by isInstanceServiceAccount or AdminOnly (TEAM-LOOKUP.md): any
|
||||
// authenticated caller may ask whether a name is taken, the same low
|
||||
// sensitivity GET /api/service-accounts?name= already accepts.
|
||||
func TestListTeamsByName_OpenToAnyAuthenticatedCaller(t *testing.T) {
|
||||
s := newTS(t)
|
||||
red := newTeam(t, s, "red")
|
||||
_ = createTeamAs(t, s, s.key, "blue-target")
|
||||
|
||||
// red's own member, not a member of "blue-target", still gets a match.
|
||||
teams := list(t, red.call(http.MethodGet, "/api/teams?name=blue-target", nil))
|
||||
if len(teams) != 1 || teams[0]["name"] != "blue-target" {
|
||||
t.Errorf("expected a non-member caller to still find the team by name, got %v", teams)
|
||||
}
|
||||
}
|
||||
@@ -7,7 +7,9 @@ import (
|
||||
"os"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/db"
|
||||
)
|
||||
|
||||
@@ -30,6 +32,23 @@ import (
|
||||
// tests nothing is worse than one that does not run.
|
||||
const testDSNEnv = "TERDUT_TEST_DSN"
|
||||
|
||||
// testConfig is the environment half of the server's configuration, which the
|
||||
// admin settings page renders read-only and SeedSettings seeds the editable
|
||||
// half from. The durations match the defaults config.Load would produce, so a
|
||||
// test that never touches the settings table behaves as a fresh install does.
|
||||
func testConfig() config.Config {
|
||||
return config.Config{
|
||||
Addr: ":8080",
|
||||
ArchiveAfter: 7 * 24 * time.Hour,
|
||||
StaleAfter: 6 * time.Hour,
|
||||
NotifyRepeat: 15 * time.Minute,
|
||||
}
|
||||
}
|
||||
|
||||
// defaultTeam is the team migration 003 creates and the bootstrap user owns, as
|
||||
// a path segment. Every test that does not say otherwise works inside it.
|
||||
const defaultTeam = "1"
|
||||
|
||||
var schemaSeq int
|
||||
|
||||
// newTestDB returns a migrated database private to this test, and drops it
|
||||
|
||||
+121
-5
@@ -58,7 +58,7 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc {
|
||||
|
||||
var userID int64
|
||||
if err := db.QueryRowContext(r.Context(),
|
||||
"INSERT INTO users (username, email, password_hash) VALUES ($1, $2, $3) RETURNING id",
|
||||
"INSERT INTO users (username, email, password_hash, is_admin) VALUES ($1, $2, $3, true) RETURNING id",
|
||||
req.Username, req.Email, passwordHash).Scan(&userID); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
@@ -77,6 +77,16 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc {
|
||||
return
|
||||
}
|
||||
|
||||
// The default team exists from migration 003, on a fresh install too.
|
||||
// Without a membership the first user signs in to a working server with
|
||||
// no queue, no schedule and nowhere for an integration to hang off.
|
||||
if teamID, err := defaultTeamID(r.Context(), db); err == nil {
|
||||
db.ExecContext(r.Context(), //nolint:errcheck
|
||||
"INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3) "+
|
||||
"ON CONFLICT (team_id, user_id) DO NOTHING",
|
||||
teamID, userID, models.RoleOwner)
|
||||
}
|
||||
|
||||
user, _ := fetchUser(r.Context(), db, userID)
|
||||
key := models.APIKey{ID: keyID, UserID: userID, Name: "bootstrap", Key: raw, CreatedAt: user.CreatedAt}
|
||||
respond(w, http.StatusCreated, map[string]any{"user": user, "api_key": key})
|
||||
@@ -86,7 +96,7 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc {
|
||||
func handleListUsers(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
rows, err := db.QueryContext(r.Context(),
|
||||
"SELECT id, username, email, created_at, ntfy_topic FROM users ORDER BY id")
|
||||
"SELECT id, username, email, created_at, ntfy_topic, is_admin, admin_source, disabled_at FROM users ORDER BY id")
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
@@ -97,11 +107,13 @@ func handleListUsers(db *sql.DB) http.HandlerFunc {
|
||||
for rows.Next() {
|
||||
var u models.User
|
||||
var ts int64
|
||||
if err := rows.Scan(&u.ID, &u.Username, &u.Email, &ts, &u.NtfyTopic); err != nil {
|
||||
var disabled *int64
|
||||
if err := rows.Scan(&u.ID, &u.Username, &u.Email, &ts, &u.NtfyTopic, &u.IsAdmin, &u.AdminSource, &disabled); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
u.CreatedAt = time.Unix(ts, 0).UTC()
|
||||
u.DisabledAt = unixPtr(disabled)
|
||||
users = append(users, u)
|
||||
}
|
||||
respond(w, http.StatusOK, users)
|
||||
@@ -150,6 +162,9 @@ func handleSetNotifyTarget(db *sql.DB) http.HandlerFunc {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid user id"))
|
||||
return
|
||||
}
|
||||
if !requireSelfOrAdmin(w, r, id) {
|
||||
return
|
||||
}
|
||||
var req struct {
|
||||
NtfyTopic string `json:"ntfy_topic"`
|
||||
}
|
||||
@@ -190,6 +205,21 @@ func handleDeleteUser(db *sql.DB) http.HandlerFunc {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid user id"))
|
||||
return
|
||||
}
|
||||
// Deleting yourself is how an install ends up with no administrator at
|
||||
// all, and it is never what somebody meant to do.
|
||||
caller, _ := userFromContext(r.Context())
|
||||
if caller.ID == id {
|
||||
respond(w, http.StatusConflict, errResp("cannot delete your own account"))
|
||||
return
|
||||
}
|
||||
if last, err := isLastAdmin(r.Context(), db, id); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
} else if last {
|
||||
respond(w, http.StatusConflict, errResp("cannot delete the last administrator"))
|
||||
return
|
||||
}
|
||||
|
||||
res, err := db.ExecContext(r.Context(), "DELETE FROM users WHERE id = $1", id)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
@@ -211,6 +241,9 @@ func handleCreateAPIKey(db *sql.DB) http.HandlerFunc {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid user id"))
|
||||
return
|
||||
}
|
||||
if !requireSelfOrAdmin(w, r, userID) {
|
||||
return
|
||||
}
|
||||
|
||||
var req struct {
|
||||
Name string `json:"name"`
|
||||
@@ -254,6 +287,9 @@ func handleDeleteAPIKey(db *sql.DB) http.HandlerFunc {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid user id"))
|
||||
return
|
||||
}
|
||||
if !requireSelfOrAdmin(w, r, userID) {
|
||||
return
|
||||
}
|
||||
keyID, err := strconv.ParseInt(chi.URLParam(r, "keyID"), 10, 64)
|
||||
if err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid key id"))
|
||||
@@ -291,12 +327,92 @@ func randomToken() (raw, hash string, err error) {
|
||||
func fetchUser(ctx context.Context, db *sql.DB, id int64) (models.User, error) {
|
||||
var u models.User
|
||||
var ts int64
|
||||
var disabled *int64
|
||||
err := db.QueryRowContext(ctx,
|
||||
"SELECT id, username, email, created_at, ntfy_topic FROM users WHERE id = $1", id).
|
||||
Scan(&u.ID, &u.Username, &u.Email, &ts, &u.NtfyTopic)
|
||||
"SELECT id, username, email, created_at, ntfy_topic, is_admin, admin_source, disabled_at FROM users WHERE id = $1", id).
|
||||
Scan(&u.ID, &u.Username, &u.Email, &ts, &u.NtfyTopic, &u.IsAdmin, &u.AdminSource, &disabled)
|
||||
if err != nil {
|
||||
return u, err
|
||||
}
|
||||
u.CreatedAt = time.Unix(ts, 0).UTC()
|
||||
u.DisabledAt = unixPtr(disabled)
|
||||
return u, nil
|
||||
}
|
||||
|
||||
// handleSetAdmin grants or revokes the system administrator flag.
|
||||
//
|
||||
// Revoking is guarded twice: an install must keep at least one administrator,
|
||||
// and you cannot demote yourself. The first stops the flag being lost
|
||||
// altogether; the second stops the likelier accident, where the only admin
|
||||
// clears their own flag while tidying up and locks the door behind them.
|
||||
func handleSetAdmin(db *sql.DB) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
id, err := strconv.ParseInt(chi.URLParam(r, "id"), 10, 64)
|
||||
if err != nil {
|
||||
respond(w, http.StatusBadRequest, errResp("invalid user id"))
|
||||
return
|
||||
}
|
||||
var req struct {
|
||||
IsAdmin *bool `json:"is_admin"`
|
||||
}
|
||||
if err := decodeJSON(r, &req); err != nil || req.IsAdmin == nil {
|
||||
respond(w, http.StatusBadRequest, errResp("is_admin is required"))
|
||||
return
|
||||
}
|
||||
|
||||
if !*req.IsAdmin {
|
||||
var managed bool
|
||||
if err := db.QueryRowContext(r.Context(),
|
||||
"SELECT EXISTS (SELECT 1 FROM users WHERE id = $1 AND is_admin AND admin_source = 'oidc')",
|
||||
id).Scan(&managed); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if managed {
|
||||
respond(w, http.StatusConflict, errResp("administrator access is managed by single sign-on; change the user's groups in the identity provider"))
|
||||
return
|
||||
}
|
||||
|
||||
caller, _ := userFromContext(r.Context())
|
||||
if caller.ID == id {
|
||||
respond(w, http.StatusConflict, errResp("cannot revoke your own administrator access"))
|
||||
return
|
||||
}
|
||||
if last, err := isLastAdmin(r.Context(), db, id); err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
} else if last {
|
||||
respond(w, http.StatusConflict, errResp("cannot revoke the last administrator"))
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
res, err := db.ExecContext(r.Context(),
|
||||
"UPDATE users SET is_admin = $1 WHERE id = $2", *req.IsAdmin, id)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
if n, _ := res.RowsAffected(); n == 0 {
|
||||
respond(w, http.StatusNotFound, errResp("user not found"))
|
||||
return
|
||||
}
|
||||
|
||||
user, err := fetchUser(r.Context(), db, id)
|
||||
if err != nil {
|
||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||
return
|
||||
}
|
||||
respond(w, http.StatusOK, user)
|
||||
}
|
||||
}
|
||||
|
||||
// isLastAdmin reports whether id is an administrator and no other user is one.
|
||||
// A non-admin id is never the last one, so removing them is always allowed.
|
||||
func isLastAdmin(ctx context.Context, db *sql.DB, id int64) (bool, error) {
|
||||
var last bool
|
||||
err := db.QueryRowContext(ctx, `
|
||||
SELECT EXISTS (SELECT 1 FROM users WHERE id = $1 AND is_admin)
|
||||
AND NOT EXISTS (SELECT 1 FROM users WHERE id <> $1 AND is_admin)`, id).Scan(&last)
|
||||
return last, err
|
||||
}
|
||||
|
||||
@@ -1,7 +1,11 @@
|
||||
package config
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/url"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
@@ -59,8 +63,71 @@ type Config struct {
|
||||
// NotifyRepeat is how long an incident may sit unacknowledged before it is
|
||||
// notified again. Zero disables reminders.
|
||||
NotifyRepeat time.Duration
|
||||
|
||||
// DisablePasswordLogin refuses signing in, or signing up, with a password.
|
||||
// It is how an install moves to SSO only, and turning it back off is the way
|
||||
// in when the identity provider is down. Stated negatively so that the zero
|
||||
// Config, which is what a test or a new caller builds, keeps passwords working.
|
||||
DisablePasswordLogin bool
|
||||
|
||||
// OIDC configures single sign-on. The zero value, with no Issuer, is off.
|
||||
OIDC OIDC
|
||||
|
||||
// OperatorMode declares this install gitops-managed: writes to teams,
|
||||
// escalation policies, dead man's switches and integrations from a human
|
||||
// (a session or a user's own API key) are refused, while a service
|
||||
// account's are not. Deploy-time and restart-required, like the rest of
|
||||
// "where this server is plugged in" — it is a statement about who owns
|
||||
// this install's configuration, not a per-request toggle.
|
||||
OperatorMode bool
|
||||
}
|
||||
|
||||
// OIDC is the single sign-on configuration. Groups from the provider decide
|
||||
// who may sign in, which teams they belong to, and whether they administer the
|
||||
// install, in the manner of Grafana's org and role mapping.
|
||||
type OIDC struct {
|
||||
// Issuer is the provider's issuer URL. Discovery is fetched from
|
||||
// <Issuer>/.well-known/openid-configuration. For Authentik this is the
|
||||
// application's issuer, e.g. https://auth.example.com/application/o/terdut/.
|
||||
// Empty turns single sign-on off.
|
||||
Issuer string
|
||||
ClientID string
|
||||
ClientSecret string
|
||||
|
||||
// Name is what the sign-in button calls the provider.
|
||||
Name string
|
||||
|
||||
// Scopes to request. The groups claim normally needs "profile" on Authentik.
|
||||
Scopes []string
|
||||
|
||||
// UsernameClaim, EmailClaim and GroupsClaim name the ID token claims read.
|
||||
UsernameClaim string
|
||||
EmailClaim string
|
||||
GroupsClaim string
|
||||
|
||||
// TrustEmail links a sign-in to an existing local user by email even when the
|
||||
// provider does not vouch that the address is verified. Authentik reports
|
||||
// email_verified false unless told otherwise, and an install that runs its
|
||||
// own provider has already decided that its addresses can be trusted.
|
||||
TrustEmail bool
|
||||
|
||||
// AllowedGroups gates sign-in: somebody in none of them is refused, however
|
||||
// well the provider authenticated them. Empty admits everybody the provider
|
||||
// authenticates, and access control is left to the provider.
|
||||
AllowedGroups []string
|
||||
|
||||
// AdminGroup grants the system administrator flag while the user is in it.
|
||||
AdminGroup string
|
||||
|
||||
// SessionMaxAge is the hard ceiling on a session made by an SSO login. The
|
||||
// login is the only moment groups are re-read, so this is how long a change
|
||||
// in the provider may take to reach terdut.
|
||||
SessionMaxAge time.Duration
|
||||
}
|
||||
|
||||
// Enabled reports whether single sign-on is configured.
|
||||
func (o OIDC) Enabled() bool { return o.Issuer != "" }
|
||||
|
||||
func Load() Config {
|
||||
addr := os.Getenv("TERDUT_ADDR")
|
||||
if addr == "" {
|
||||
@@ -89,9 +156,98 @@ func Load() Config {
|
||||
NtfyFallbackTopic: os.Getenv("TERDUT_NTFY_FALLBACK_TOPIC"),
|
||||
PublicURL: os.Getenv("TERDUT_PUBLIC_URL"),
|
||||
NotifyRepeat: duration("TERDUT_NOTIFY_REPEAT", 15*time.Minute),
|
||||
|
||||
DisablePasswordLogin: !boolean("TERDUT_PASSWORD_LOGIN", true),
|
||||
OIDC: loadOIDC(),
|
||||
|
||||
OperatorMode: boolean("TERDUT_OPERATOR_MODE", false),
|
||||
}
|
||||
}
|
||||
|
||||
func loadOIDC() OIDC {
|
||||
o := OIDC{
|
||||
Issuer: strings.TrimSpace(os.Getenv("TERDUT_OIDC_ISSUER")),
|
||||
ClientID: os.Getenv("TERDUT_OIDC_CLIENT_ID"),
|
||||
ClientSecret: os.Getenv("TERDUT_OIDC_CLIENT_SECRET"),
|
||||
Name: str("TERDUT_OIDC_NAME", "SSO"),
|
||||
Scopes: list("TERDUT_OIDC_SCOPES", "openid profile email"),
|
||||
UsernameClaim: str("TERDUT_OIDC_USERNAME_CLAIM", "preferred_username"),
|
||||
EmailClaim: str("TERDUT_OIDC_EMAIL_CLAIM", "email"),
|
||||
GroupsClaim: str("TERDUT_OIDC_GROUPS_CLAIM", "groups"),
|
||||
TrustEmail: boolean("TERDUT_OIDC_TRUST_EMAIL", false),
|
||||
AllowedGroups: list("TERDUT_OIDC_ALLOWED_GROUPS", ""),
|
||||
AdminGroup: os.Getenv("TERDUT_OIDC_ADMIN_GROUP"),
|
||||
SessionMaxAge: duration("TERDUT_OIDC_SESSION_MAX_AGE", 12*time.Hour),
|
||||
}
|
||||
return o
|
||||
}
|
||||
|
||||
// Validate reports a configuration the server should refuse to start with.
|
||||
// Single sign-on is the only part that can be inconsistent: a half-configured
|
||||
// provider would come up and then fail every login, which is harder to notice
|
||||
// than not starting.
|
||||
func (c Config) Validate() error {
|
||||
o := c.OIDC
|
||||
if !o.Enabled() {
|
||||
if c.DisablePasswordLogin {
|
||||
return errors.New("TERDUT_PASSWORD_LOGIN=false without TERDUT_OIDC_ISSUER leaves no way to sign in")
|
||||
}
|
||||
if o.AdminGroup != "" || len(o.AllowedGroups) > 0 {
|
||||
return errors.New("TERDUT_OIDC_* group settings are set but TERDUT_OIDC_ISSUER is not")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
if u, err := url.Parse(o.Issuer); err != nil || u.Scheme == "" || u.Host == "" {
|
||||
return fmt.Errorf("TERDUT_OIDC_ISSUER %q is not a URL", o.Issuer)
|
||||
}
|
||||
if o.ClientID == "" || o.ClientSecret == "" {
|
||||
return errors.New("TERDUT_OIDC_CLIENT_ID and TERDUT_OIDC_CLIENT_SECRET are required with TERDUT_OIDC_ISSUER")
|
||||
}
|
||||
if c.PublicURL == "" {
|
||||
return errors.New("TERDUT_PUBLIC_URL is required with TERDUT_OIDC_ISSUER: it is the base of the redirect URI")
|
||||
}
|
||||
if o.SessionMaxAge <= 0 {
|
||||
return errors.New("TERDUT_OIDC_SESSION_MAX_AGE must be positive")
|
||||
}
|
||||
// Team grants are no longer visible here: they live on each team's own
|
||||
// oidc_member_group/oidc_owner_group columns, set by that team's owner, not
|
||||
// in config Validate can see at startup. The one thing left to guard against
|
||||
// is an install nobody can administer at all.
|
||||
if c.DisablePasswordLogin && o.AdminGroup == "" {
|
||||
return errors.New("TERDUT_PASSWORD_LOGIN=false with no TERDUT_OIDC_ADMIN_GROUP leaves nobody able to administer the install")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func str(env, def string) string {
|
||||
if s := strings.TrimSpace(os.Getenv(env)); s != "" {
|
||||
return s
|
||||
}
|
||||
return def
|
||||
}
|
||||
|
||||
// list reads a comma- or space-separated env var.
|
||||
func list(env, def string) []string {
|
||||
s := os.Getenv(env)
|
||||
if strings.TrimSpace(s) == "" {
|
||||
s = def
|
||||
}
|
||||
return strings.FieldsFunc(s, func(r rune) bool { return r == ',' || r == ' ' })
|
||||
}
|
||||
|
||||
// boolean reads a true/false env var. An unrecognised value takes the default,
|
||||
// so the two flags read this way (password login on, trusting email off) both
|
||||
// fail towards the cautious setting.
|
||||
func boolean(env string, def bool) bool {
|
||||
switch strings.ToLower(strings.TrimSpace(os.Getenv(env))) {
|
||||
case "true", "1", "yes":
|
||||
return true
|
||||
case "false", "0", "no":
|
||||
return false
|
||||
}
|
||||
return def
|
||||
}
|
||||
|
||||
// duration reads a time.ParseDuration-formatted env var. An unset or
|
||||
// unparseable value falls back to def rather than failing startup: a typo in one
|
||||
// tuning knob should not take the server down.
|
||||
|
||||
@@ -0,0 +1,79 @@
|
||||
package config
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestValidate(t *testing.T) {
|
||||
base := func() map[string]string {
|
||||
return map[string]string{
|
||||
"TERDUT_PUBLIC_URL": "https://terdut.example.com",
|
||||
"TERDUT_OIDC_ISSUER": "https://auth.example.com/application/o/terdut/",
|
||||
"TERDUT_OIDC_CLIENT_ID": "id",
|
||||
"TERDUT_OIDC_CLIENT_SECRET": "secret",
|
||||
}
|
||||
}
|
||||
tests := []struct {
|
||||
name string
|
||||
env func(map[string]string)
|
||||
wantErr string // substring; empty means valid
|
||||
}{
|
||||
{"off by default", func(m map[string]string) { clear(m) }, ""},
|
||||
{"minimal sso", func(m map[string]string) {}, ""},
|
||||
{"groups without issuer", func(m map[string]string) {
|
||||
clear(m)
|
||||
m["TERDUT_OIDC_ADMIN_GROUP"] = "admins"
|
||||
}, "ISSUER is not"},
|
||||
{"missing secret", func(m map[string]string) { delete(m, "TERDUT_OIDC_CLIENT_SECRET") }, "CLIENT_SECRET"},
|
||||
{"missing public url", func(m map[string]string) { delete(m, "TERDUT_PUBLIC_URL") }, "PUBLIC_URL"},
|
||||
{"bad issuer", func(m map[string]string) { m["TERDUT_OIDC_ISSUER"] = "not a url" }, "not a URL"},
|
||||
{"password off without sso", func(m map[string]string) {
|
||||
clear(m)
|
||||
m["TERDUT_PASSWORD_LOGIN"] = "false"
|
||||
}, "no way to sign in"},
|
||||
{"password off with sso but no grants", func(m map[string]string) {
|
||||
m["TERDUT_PASSWORD_LOGIN"] = "false"
|
||||
}, "nobody able"},
|
||||
{"password off with admin group", func(m map[string]string) {
|
||||
m["TERDUT_PASSWORD_LOGIN"] = "false"
|
||||
m["TERDUT_OIDC_ADMIN_GROUP"] = "admins"
|
||||
}, ""},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
env := base()
|
||||
tt.env(env)
|
||||
for _, k := range []string{
|
||||
"TERDUT_PUBLIC_URL", "TERDUT_PASSWORD_LOGIN", "TERDUT_OIDC_ISSUER", "TERDUT_OIDC_CLIENT_ID",
|
||||
"TERDUT_OIDC_CLIENT_SECRET", "TERDUT_OIDC_ADMIN_GROUP",
|
||||
} {
|
||||
t.Setenv(k, env[k])
|
||||
}
|
||||
err := Load().Validate()
|
||||
switch {
|
||||
case tt.wantErr == "" && err != nil:
|
||||
t.Errorf("unexpected error: %v", err)
|
||||
case tt.wantErr != "" && (err == nil || !strings.Contains(err.Error(), tt.wantErr)):
|
||||
t.Errorf("error %v, want one containing %q", err, tt.wantErr)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestLoad_OIDCDefaults(t *testing.T) {
|
||||
t.Setenv("TERDUT_OIDC_ISSUER", "https://auth.example.com/")
|
||||
o := Load().OIDC
|
||||
if o.UsernameClaim != "preferred_username" || o.EmailClaim != "email" || o.GroupsClaim != "groups" {
|
||||
t.Errorf("claim defaults: %+v", o)
|
||||
}
|
||||
if strings.Join(o.Scopes, " ") != "openid profile email" {
|
||||
t.Errorf("scopes: %v", o.Scopes)
|
||||
}
|
||||
if o.SessionMaxAge.Hours() != 12 {
|
||||
t.Errorf("max age: %v", o.SessionMaxAge)
|
||||
}
|
||||
if Load().DisablePasswordLogin {
|
||||
t.Error("password login should be on by default")
|
||||
}
|
||||
}
|
||||
+53
-4
@@ -1,10 +1,12 @@
|
||||
package db
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"embed"
|
||||
"fmt"
|
||||
"io/fs"
|
||||
"log"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
@@ -15,6 +17,19 @@ import (
|
||||
//go:embed migrations
|
||||
var migrationsFS embed.FS
|
||||
|
||||
// pingAttempts and pingRetryDelay bound the retry on the first connection.
|
||||
// This pod's own IP can reach the Postgres pod's node before that node's
|
||||
// NetworkPolicy enforcement (kube-router, reacting to the pod's creation
|
||||
// event) has added it to the allowed-source set, which fails the ping with
|
||||
// "connection refused" rather than a timeout. That race resolves within
|
||||
// several seconds in practice; five attempts two seconds apart give it
|
||||
// comfortable room without turning a genuinely absent database into a long
|
||||
// hang.
|
||||
const (
|
||||
pingAttempts = 5
|
||||
pingRetryDelay = 2 * time.Second
|
||||
)
|
||||
|
||||
// Open connects to Postgres. dsn is a libpq connection string or URL, e.g.
|
||||
// postgres://terdut:secret@localhost:5432/terdut?sslmode=disable.
|
||||
//
|
||||
@@ -33,13 +48,31 @@ func Open(dsn string) (*sql.DB, error) {
|
||||
db.SetMaxOpenConns(10)
|
||||
db.SetMaxIdleConns(5)
|
||||
db.SetConnMaxLifetime(time.Hour)
|
||||
if err := db.Ping(); err != nil {
|
||||
db.Close()
|
||||
return nil, fmt.Errorf("ping: %w", err)
|
||||
|
||||
for attempt := 1; ; attempt++ {
|
||||
err = db.Ping()
|
||||
if err == nil {
|
||||
return db, nil
|
||||
}
|
||||
if attempt == pingAttempts {
|
||||
db.Close()
|
||||
return nil, fmt.Errorf("ping: %w", err)
|
||||
}
|
||||
log.Printf("open db: ping attempt %d/%d failed, retrying in %s: %v", attempt, pingAttempts, pingRetryDelay, err)
|
||||
time.Sleep(pingRetryDelay)
|
||||
}
|
||||
return db, nil
|
||||
}
|
||||
|
||||
// migrationLockKey is the Postgres advisory lock Migrate holds for its whole
|
||||
// run. Two replicas starting at once would otherwise race the check-then-apply
|
||||
// loop below against schema_migrations: the loser could crash on a
|
||||
// duplicate-key insert, or contend with the winner's uncommitted DDL. Blocking
|
||||
// (pg_advisory_lock, not pg_try_advisory_lock as the archiver and notifier
|
||||
// use): on boot there is no later tick to defer to, so the right behaviour is
|
||||
// to wait for the other replica to finish migrating, not to skip ahead and
|
||||
// start serving against an unmigrated schema.
|
||||
const migrationLockKey int64 = 7265_0003
|
||||
|
||||
// Migrate applies every embedded migration that has not been applied yet, in
|
||||
// filename order, recording each in schema_migrations.
|
||||
//
|
||||
@@ -47,6 +80,22 @@ func Open(dsn string) (*sql.DB, error) {
|
||||
// migration that failed half way used to leave the schema in whatever state it
|
||||
// had reached. Postgres has transactional DDL, so the rollback is real.
|
||||
func Migrate(db *sql.DB) error {
|
||||
ctx := context.Background()
|
||||
conn, err := db.Conn(ctx)
|
||||
if err != nil {
|
||||
return fmt.Errorf("migrate: acquire connection: %w", err)
|
||||
}
|
||||
defer conn.Close()
|
||||
|
||||
if _, err := conn.ExecContext(ctx, "SELECT pg_advisory_lock($1)", migrationLockKey); err != nil {
|
||||
return fmt.Errorf("migrate: acquire advisory lock: %w", err)
|
||||
}
|
||||
defer func() {
|
||||
if _, err := conn.ExecContext(ctx, "SELECT pg_advisory_unlock($1)", migrationLockKey); err != nil {
|
||||
log.Printf("migrate: release advisory lock: %v", err)
|
||||
}
|
||||
}()
|
||||
|
||||
if _, err := db.Exec(`CREATE TABLE IF NOT EXISTS schema_migrations (
|
||||
version TEXT PRIMARY KEY,
|
||||
applied_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
package db_test
|
||||
|
||||
import (
|
||||
"database/sql"
|
||||
"fmt"
|
||||
"net/url"
|
||||
"os"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/db"
|
||||
|
||||
_ "github.com/jackc/pgx/v5/stdlib"
|
||||
)
|
||||
|
||||
// TERDUT_TEST_DSN must point at a database the test role may create schemas
|
||||
// in; see internal/api/testdb_test.go for the fuller rationale this mirrors.
|
||||
// An unset DSN fails rather than skips, deliberately.
|
||||
const testDSNEnv = "TERDUT_TEST_DSN"
|
||||
|
||||
// TestMigrate_ConcurrentCallersDoNotRace reproduces two replicas starting at
|
||||
// once against a brand-new, unmigrated schema: both call db.Migrate at the
|
||||
// same time. Before migrationLockKey, the loser could crash on a
|
||||
// duplicate-key insert into schema_migrations, or contend with the winner's
|
||||
// uncommitted DDL; with the advisory lock, one blocks until the other
|
||||
// finishes and both return cleanly.
|
||||
func TestMigrate_ConcurrentCallersDoNotRace(t *testing.T) {
|
||||
dsn := os.Getenv(testDSNEnv)
|
||||
if dsn == "" {
|
||||
t.Fatalf("%s is not set: these tests need Postgres.\n"+
|
||||
"Run `make test-db` for a local one, then\n"+
|
||||
" export %s=postgres://terdut:terdut@localhost:5432/terdut_test?sslmode=disable",
|
||||
testDSNEnv, testDSNEnv)
|
||||
}
|
||||
|
||||
schema := fmt.Sprintf("migrate_race_%d", os.Getpid())
|
||||
admin, err := sql.Open("pgx", dsn)
|
||||
if err != nil {
|
||||
t.Fatalf("connect to %s: %v", testDSNEnv, err)
|
||||
}
|
||||
defer admin.Close()
|
||||
if _, err := admin.Exec("CREATE SCHEMA " + schema); err != nil {
|
||||
t.Fatalf("create schema %s: %v", schema, err)
|
||||
}
|
||||
t.Cleanup(func() {
|
||||
cleanup, err := sql.Open("pgx", dsn)
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
defer cleanup.Close()
|
||||
if _, err := cleanup.Exec("DROP SCHEMA " + schema + " CASCADE"); err != nil {
|
||||
t.Logf("drop schema %s: %v", schema, err)
|
||||
}
|
||||
})
|
||||
|
||||
scoped := withSearchPath(dsn, schema)
|
||||
|
||||
const callers = 2
|
||||
errs := make([]error, callers)
|
||||
var wg sync.WaitGroup
|
||||
for i := range callers {
|
||||
wg.Add(1)
|
||||
go func(i int) {
|
||||
defer wg.Done()
|
||||
database, err := db.Open(scoped)
|
||||
if err != nil {
|
||||
errs[i] = fmt.Errorf("open: %w", err)
|
||||
return
|
||||
}
|
||||
defer database.Close()
|
||||
errs[i] = db.Migrate(database)
|
||||
}(i)
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
for i, err := range errs {
|
||||
if err != nil {
|
||||
t.Fatalf("Migrate #%d: %v", i, err)
|
||||
}
|
||||
}
|
||||
|
||||
entries, err := os.ReadDir("migrations")
|
||||
if err != nil {
|
||||
t.Fatalf("read migrations dir: %v", err)
|
||||
}
|
||||
var want int
|
||||
for _, e := range entries {
|
||||
if !e.IsDir() && strings.HasSuffix(e.Name(), ".sql") {
|
||||
want++
|
||||
}
|
||||
}
|
||||
|
||||
check, err := sql.Open("pgx", scoped)
|
||||
if err != nil {
|
||||
t.Fatalf("connect for verification: %v", err)
|
||||
}
|
||||
defer check.Close()
|
||||
|
||||
var got int
|
||||
if err := check.QueryRow("SELECT COUNT(*) FROM schema_migrations").Scan(&got); err != nil {
|
||||
t.Fatalf("count schema_migrations: %v", err)
|
||||
}
|
||||
if got != want {
|
||||
t.Fatalf("schema_migrations has %d row(s) after two concurrent Migrate calls, want %d (one per migration file, no duplicates)", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// withSearchPath pins a DSN to one schema. Copied from
|
||||
// internal/api/testdb_test.go rather than shared: that helper lives in the
|
||||
// api_test package, a separate compiled package this one cannot import.
|
||||
func withSearchPath(dsn, schema string) string {
|
||||
opt := "-csearch_path=" + schema
|
||||
|
||||
if strings.HasPrefix(dsn, "postgres://") || strings.HasPrefix(dsn, "postgresql://") {
|
||||
u, err := url.Parse(dsn)
|
||||
if err == nil {
|
||||
q := u.Query()
|
||||
q.Set("options", opt)
|
||||
u.RawQuery = q.Encode()
|
||||
return u.String()
|
||||
}
|
||||
}
|
||||
return dsn + " options='" + opt + "'"
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
-- A system administrator role, and the first thing in this server that one user
|
||||
-- can do and another cannot.
|
||||
--
|
||||
-- Until now every authenticated caller could create and delete users, set
|
||||
-- anybody's password and mint API keys for anybody — auth.go said so in a
|
||||
-- comment. That was defensible with one operator and a hand-made account; it is
|
||||
-- not once people sign themselves up (see #7).
|
||||
--
|
||||
-- EVERY EXISTING USER BECOMES AN ADMIN. They already hold these powers, so
|
||||
-- this migration changes nobody's access: it names what is already true, and
|
||||
-- leaves demotion as a deliberate act somebody performs afterwards. The
|
||||
-- alternative — promoting only user 1 — would silently strip the others, and
|
||||
-- could leave an install whose only admin is an account nobody has a password
|
||||
-- for.
|
||||
--
|
||||
-- New users are not admins: the column defaults to false, and the only ways to
|
||||
-- become one are this backfill, the bootstrap endpoint, or an existing admin
|
||||
-- granting it.
|
||||
ALTER TABLE users ADD COLUMN is_admin BOOLEAN NOT NULL DEFAULT false;
|
||||
|
||||
UPDATE users SET is_admin = true;
|
||||
|
||||
-- The queue's assignment dropdown and the on-call schedule read every user, and
|
||||
-- the admin screens in #5 will filter on this.
|
||||
CREATE INDEX users_is_admin_idx ON users(is_admin) WHERE is_admin;
|
||||
@@ -0,0 +1,103 @@
|
||||
-- Teams: the unit of tenancy. Everything a person works on now belongs to one.
|
||||
--
|
||||
-- Until this migration the install was one shared space — every user saw every
|
||||
-- alert and every incident, and the Alertmanager webhook was unauthenticated, so
|
||||
-- anything that could reach the port could open an incident for everybody.
|
||||
--
|
||||
-- The shape, in one paragraph: a team owns its incidents, alerts, schedule and
|
||||
-- integrations. A user belongs to as many teams as they like, with a role in
|
||||
-- each: an `owner` configures the team, a `member` works its incidents. An
|
||||
-- integration key is what an alert arrives on, and the key is what says which
|
||||
-- team the alert belongs to.
|
||||
--
|
||||
-- EVERYTHING EXISTING MOVES INTO ONE DEFAULT TEAM, and every existing user
|
||||
-- becomes an owner of it. That keeps an upgrade a no-op for the people using it:
|
||||
-- the same queue, the same schedule, the same incidents, with a name on them.
|
||||
|
||||
CREATE TABLE teams (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
name TEXT NOT NULL UNIQUE,
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
||||
);
|
||||
|
||||
-- role is free text with a CHECK rather than an enum, so adding a third role
|
||||
-- later is a migration and not a type rewrite.
|
||||
CREATE TABLE team_members (
|
||||
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
|
||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
||||
role TEXT NOT NULL CHECK (role IN ('owner', 'member')),
|
||||
joined_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
PRIMARY KEY (team_id, user_id)
|
||||
);
|
||||
|
||||
CREATE INDEX team_members_user_idx ON team_members(user_id);
|
||||
|
||||
-- How alerts get in, and the only thing that says which team they belong to.
|
||||
-- The key is stored as a SHA-256 hash, like api_keys and the ack tokens: a
|
||||
-- leaked database gives nobody the ability to post alerts.
|
||||
CREATE TABLE integrations (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
|
||||
kind TEXT NOT NULL CHECK (kind IN ('alertmanager')),
|
||||
name TEXT NOT NULL,
|
||||
key_hash TEXT NOT NULL UNIQUE,
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
last_used_at BIGINT
|
||||
);
|
||||
|
||||
CREATE INDEX integrations_team_idx ON integrations(team_id);
|
||||
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- The default team, and everything that already exists moving into it.
|
||||
--
|
||||
-- Created unconditionally, even on an empty install, so there is always a team
|
||||
-- for the bootstrap user to land in and for the first integration to hang off.
|
||||
-- ---------------------------------------------------------------------------
|
||||
|
||||
INSERT INTO teams (name) VALUES ('Default');
|
||||
|
||||
INSERT INTO team_members (team_id, user_id, role)
|
||||
SELECT (SELECT id FROM teams WHERE name = 'Default'), id, 'owner' FROM users;
|
||||
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- team_id on everything a team owns.
|
||||
--
|
||||
-- Added nullable, backfilled, then made NOT NULL: adding a NOT NULL column with
|
||||
-- no default to a table with rows is rejected, and a DEFAULT pointing at the
|
||||
-- default team would quietly keep working after the default team is gone.
|
||||
-- ---------------------------------------------------------------------------
|
||||
|
||||
ALTER TABLE alerts ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
|
||||
ALTER TABLE incidents ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
|
||||
ALTER TABLE schedule_entries ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
|
||||
|
||||
UPDATE alerts SET team_id = (SELECT id FROM teams WHERE name = 'Default');
|
||||
UPDATE incidents SET team_id = (SELECT id FROM teams WHERE name = 'Default');
|
||||
UPDATE schedule_entries SET team_id = (SELECT id FROM teams WHERE name = 'Default');
|
||||
|
||||
ALTER TABLE alerts ALTER COLUMN team_id SET NOT NULL;
|
||||
ALTER TABLE incidents ALTER COLUMN team_id SET NOT NULL;
|
||||
ALTER TABLE schedule_entries ALTER COLUMN team_id SET NOT NULL;
|
||||
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- The uniqueness rules were all written for one tenant, and every one of them
|
||||
-- is wrong now: two teams monitoring two clusters legitimately see the same
|
||||
-- fingerprint, the same groupKey, and want somebody on call on the same day.
|
||||
-- ---------------------------------------------------------------------------
|
||||
|
||||
ALTER TABLE alerts DROP CONSTRAINT alerts_fingerprint_key;
|
||||
CREATE UNIQUE INDEX alerts_team_fingerprint_idx ON alerts(team_id, fingerprint);
|
||||
|
||||
DROP INDEX incidents_open_group_key_idx;
|
||||
-- Still load-bearing, now per team: at most one OPEN incident per group_key
|
||||
-- within a team. This is what makes "resolved incident + a new alert occurrence
|
||||
-- = a new incident" work, and what the webhook's find-or-open lookup relies on.
|
||||
CREATE UNIQUE INDEX incidents_open_group_key_idx
|
||||
ON incidents(team_id, group_key) WHERE resolved_at IS NULL;
|
||||
|
||||
ALTER TABLE schedule_entries DROP CONSTRAINT schedule_entries_date_key;
|
||||
CREATE UNIQUE INDEX schedule_entries_team_date_idx ON schedule_entries(team_id, date);
|
||||
|
||||
-- The list views all filter by team first.
|
||||
CREATE INDEX alerts_team_received_idx ON alerts(team_id, received_at DESC);
|
||||
CREATE INDEX incidents_team_triggered_idx ON incidents(team_id, triggered_at DESC);
|
||||
@@ -0,0 +1,39 @@
|
||||
-- Dead man's switches become a team's own configuration.
|
||||
--
|
||||
-- They were three environment variables — TERDUT_DEADMAN_MATCHERS, _TIMEOUT and
|
||||
-- _SEVERITY — which made them one setting for the whole install. That was the
|
||||
-- last piece of the alerting path a team could not control: a team could take
|
||||
-- its own alerts on its own key and still not say which of them were
|
||||
-- heartbeats, or how long a silence had to last before somebody was paged.
|
||||
--
|
||||
-- One row per team rather than one row per switch. The unit of monitoring is
|
||||
-- still the fingerprint, as it always was — two clusters sending the same
|
||||
-- heartbeat alertname are two independent switches — and the matcher string
|
||||
-- keeps the format the environment variable used, so a value can be moved from
|
||||
-- one to the other unchanged.
|
||||
--
|
||||
-- No rows are seeded here: a migration cannot read the environment. The server
|
||||
-- inserts a row per team at startup from its own configuration, and the same
|
||||
-- values therefore carry forward into the first team's row without anybody
|
||||
-- retyping them. See seedDeadmanConfigs.
|
||||
CREATE TABLE deadman_configs (
|
||||
team_id BIGINT PRIMARY KEY REFERENCES teams(id) ON DELETE CASCADE,
|
||||
|
||||
-- ";" separates matchers, "," the label conditions within one, "=" is exact
|
||||
-- equality: `alertname=Watchdog,cluster=prod; alertname=EdgeHeartbeat`.
|
||||
-- Every matcher must name an alertname. Empty watches nothing.
|
||||
matchers TEXT NOT NULL DEFAULT '',
|
||||
|
||||
-- Seconds rather than a Go duration string: the column is compared and
|
||||
-- arithmetic is done on it, and a value that has to be parsed before it can
|
||||
-- be believed is a value that can be stored unparseable. Zero disables the
|
||||
-- team's switches entirely.
|
||||
timeout_seconds BIGINT NOT NULL DEFAULT 0,
|
||||
|
||||
-- The severity these incidents open at. They have no member alerts to
|
||||
-- derive one from, and a heartbeat's own severity label is meaningless —
|
||||
-- Watchdog ships as "none".
|
||||
severity TEXT NOT NULL DEFAULT 'critical',
|
||||
|
||||
updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
||||
);
|
||||
@@ -0,0 +1,35 @@
|
||||
-- Settings that an administrator can change without a redeploy, and the flag
|
||||
-- that takes an account out of use without deleting it.
|
||||
--
|
||||
-- Three of the server's tunables were environment variables, which meant
|
||||
-- changing how long an incident waits before it is paged again required editing
|
||||
-- a chart, merging it, and waiting for a reconcile. They are behaviour, not
|
||||
-- infrastructure, and the difference is who needs to change them and how often.
|
||||
--
|
||||
-- What stays in the environment: the ntfy URL and token, the database DSN, the
|
||||
-- listen address and the public URL. Those are where the server is plugged in
|
||||
-- rather than how it behaves, they are needed before the database is open, and
|
||||
-- two of them are credentials.
|
||||
--
|
||||
-- Key/value rather than a column per setting. A settings table with one row and
|
||||
-- a column per knob needs a migration for every new knob, and #6 and #7 will
|
||||
-- both add some. The cost is that values are text and the accessor has to say
|
||||
-- what type it wanted; settings.go does that in one place.
|
||||
--
|
||||
-- No rows are seeded here: a migration cannot read the environment. The server
|
||||
-- inserts each key from its own configuration at startup, once, so an install
|
||||
-- that upgrades keeps exactly the behaviour it had. See SeedSettings.
|
||||
CREATE TABLE settings (
|
||||
key TEXT PRIMARY KEY,
|
||||
value TEXT NOT NULL,
|
||||
updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
||||
);
|
||||
|
||||
-- Disabling an account rather than deleting it: the person has left, or the
|
||||
-- credential is suspect, and their incidents, acknowledgements and timeline
|
||||
-- entries must stay exactly where they are. Deleting a user nulls their
|
||||
-- acknowledged_by and assigned_to, which quietly rewrites history.
|
||||
--
|
||||
-- A disabled user cannot sign in and their API keys stop working, but they are
|
||||
-- still a name the timeline can show and still a member of their teams.
|
||||
ALTER TABLE users ADD COLUMN disabled_at BIGINT;
|
||||
@@ -0,0 +1,95 @@
|
||||
-- Escalation: page somebody else when the first person does not answer.
|
||||
--
|
||||
-- This is the gap the whole multi-tenancy line of work was opened to close.
|
||||
-- Until now an unacknowledged incident re-paged the same topic every
|
||||
-- notify_repeat forever, which is a louder version of the same silence: if the
|
||||
-- person on call is asleep, has no signal, or has left, nothing else happens.
|
||||
--
|
||||
-- Shape: one policy per team, an ordered list of levels, each level with a
|
||||
-- timeout and a set of targets. When a level's timeout passes and the incident
|
||||
-- is still triggered, the next level is paged. When the last level passes, the
|
||||
-- chain repeats repeat_count times, and then the team's fallback topic is paged
|
||||
-- once as the end of the line.
|
||||
--
|
||||
-- A team WITHOUT a policy keeps exactly today's behaviour: page the assignee,
|
||||
-- then remind on the same topic. Escalation is opt-in per team, and the two
|
||||
-- never both run for one incident -- see enqueueReminders.
|
||||
CREATE TABLE escalation_policies (
|
||||
-- One per team for now, hence the team as the key rather than an id with a
|
||||
-- unique index: routing different alerts to different chains needs the
|
||||
-- alert to carry something to route ON, which is a separate question.
|
||||
team_id BIGINT PRIMARY KEY REFERENCES teams(id) ON DELETE CASCADE,
|
||||
|
||||
-- How many extra times to run the whole chain after it has been walked
|
||||
-- once. 0 means walk it once and stop at the fallback.
|
||||
repeat_count BIGINT NOT NULL DEFAULT 0 CHECK (repeat_count >= 0 AND repeat_count <= 10),
|
||||
|
||||
-- Where the last page goes when every level has been tried. Per team now:
|
||||
-- TERDUT_NTFY_FALLBACK_TOPIC was one topic for the whole install, which in
|
||||
-- a multi-team server pages the wrong people. Empty means the chain simply
|
||||
-- ends.
|
||||
fallback_topic TEXT NOT NULL DEFAULT '',
|
||||
|
||||
updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
||||
);
|
||||
|
||||
CREATE TABLE escalation_levels (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
team_id BIGINT NOT NULL REFERENCES escalation_policies(team_id) ON DELETE CASCADE,
|
||||
-- 1-based, dense. The API rewrites the whole ladder on every edit rather
|
||||
-- than patching one rung, so there is no way to leave a gap.
|
||||
position BIGINT NOT NULL,
|
||||
-- How long this level has to produce an acknowledgement before the next one
|
||||
-- is paged. Seconds, like every other duration in this schema.
|
||||
timeout_seconds BIGINT NOT NULL CHECK (timeout_seconds > 0),
|
||||
|
||||
UNIQUE (team_id, position)
|
||||
);
|
||||
|
||||
-- Who a level pages. Either a named person, or whoever the team's rota says is
|
||||
-- on call today -- which is the target that keeps working when the rota
|
||||
-- changes and nobody remembers to edit the policy.
|
||||
CREATE TABLE escalation_targets (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
level_id BIGINT NOT NULL REFERENCES escalation_levels(id) ON DELETE CASCADE,
|
||||
kind TEXT NOT NULL CHECK (kind IN ('user', 'oncall')),
|
||||
-- Set for kind='user', NULL for kind='oncall'.
|
||||
user_id BIGINT REFERENCES users(id) ON DELETE CASCADE,
|
||||
|
||||
CHECK ((kind = 'user' AND user_id IS NOT NULL) OR (kind = 'oncall' AND user_id IS NULL))
|
||||
);
|
||||
|
||||
CREATE INDEX escalation_targets_level_idx ON escalation_targets(level_id);
|
||||
|
||||
-- ---------------------------------------------------------------------------
|
||||
-- Where an incident is in its chain.
|
||||
--
|
||||
-- On the incident rather than in a side table: it is read on every notifier
|
||||
-- tick alongside the incident's status, and one row per incident is exactly
|
||||
-- what the state is.
|
||||
-- ---------------------------------------------------------------------------
|
||||
|
||||
-- 0 means no level has been paged yet, which is the state of every incident
|
||||
-- that existed before escalation and of every incident in a team with no
|
||||
-- policy. 1 is the first level.
|
||||
ALTER TABLE incidents ADD COLUMN escalation_level BIGINT NOT NULL DEFAULT 0;
|
||||
|
||||
-- When the current level was entered, and therefore what its timeout is
|
||||
-- measured from. NULL while escalation_level is 0.
|
||||
ALTER TABLE incidents ADD COLUMN escalation_level_at BIGINT;
|
||||
|
||||
-- How many times the chain has been walked in full. Compared against the
|
||||
-- policy's repeat_count.
|
||||
ALTER TABLE incidents ADD COLUMN escalation_round BIGINT NOT NULL DEFAULT 0;
|
||||
|
||||
-- The notifier's escalation query: incidents still waiting, oldest level first.
|
||||
CREATE INDEX incidents_escalation_idx
|
||||
ON incidents(escalation_level_at)
|
||||
WHERE resolved_at IS NULL AND status = 'triggered';
|
||||
|
||||
-- 'escalated' joins the outbox kinds: a page that went out because nobody
|
||||
-- answered the last one, which is worth telling apart from the first page and
|
||||
-- from a reminder when reading the timeline or debugging a delivery.
|
||||
ALTER TABLE notifications DROP CONSTRAINT notifications_kind_check;
|
||||
ALTER TABLE notifications ADD CONSTRAINT notifications_kind_check
|
||||
CHECK (kind IN ('triggered', 'reminder', 'resolved', 'escalated'));
|
||||
@@ -0,0 +1,49 @@
|
||||
-- Self-service sign-up, and the invite links that make it useful.
|
||||
--
|
||||
-- Until now the only way to get an account was for somebody who already had one
|
||||
-- to create it, and the login page told people to "ask an admin". That is a
|
||||
-- workable arrangement for one operator and an impossible one for a team.
|
||||
--
|
||||
-- An invite is a link, not an email: this server has no SMTP and adding it to
|
||||
-- send one message would be a new subsystem to run, secure and monitor. The
|
||||
-- person inviting sends the link however they already talk to the person they
|
||||
-- are inviting.
|
||||
CREATE TABLE invites (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
|
||||
-- SHA-256 of the raw token, like api_keys, the integration keys and the
|
||||
-- acknowledgement tokens. A leaked database hands nobody an account.
|
||||
token_hash TEXT NOT NULL UNIQUE,
|
||||
|
||||
-- Which team the invitee lands in, and as what. An invite always names a
|
||||
-- team: an account in no team sees an empty queue and can be paged by
|
||||
-- nobody, which is not a state to invite somebody into.
|
||||
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
|
||||
role TEXT NOT NULL CHECK (role IN ('owner', 'member')),
|
||||
|
||||
created_by BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
|
||||
-- Invites expire. A link that works forever is a credential nobody
|
||||
-- remembers issuing, sitting in a chat log.
|
||||
expires_at BIGINT NOT NULL,
|
||||
|
||||
-- Single-use by default: max_uses 1. A team onboarding six people at once
|
||||
-- can raise it rather than minting six links.
|
||||
max_uses BIGINT NOT NULL DEFAULT 1 CHECK (max_uses > 0 AND max_uses <= 100),
|
||||
uses BIGINT NOT NULL DEFAULT 0,
|
||||
|
||||
-- Revoked by hand, separately from expiry, so "this link is no longer
|
||||
-- wanted" and "this link timed out" stay distinguishable in the listing.
|
||||
revoked_at BIGINT
|
||||
);
|
||||
|
||||
CREATE INDEX invites_team_idx ON invites(team_id);
|
||||
|
||||
-- Who redeemed which invite. Kept after the invite is gone — the answer to "how
|
||||
-- did this account get here" should outlive the link that made it.
|
||||
ALTER TABLE users ADD COLUMN invited_via BIGINT REFERENCES invites(id) ON DELETE SET NULL;
|
||||
|
||||
-- Where a person is in the first-run checklist, so it can be resumed and
|
||||
-- dismissed rather than nagging forever. One row per user, created on demand.
|
||||
ALTER TABLE users ADD COLUMN onboarding_dismissed_at BIGINT;
|
||||
@@ -0,0 +1,23 @@
|
||||
-- Similar incidents: a signature per incident, so "has this happened before"
|
||||
-- is an indexed equality instead of a search.
|
||||
--
|
||||
-- The signature is the alert name plus the group labels that identify WHAT is
|
||||
-- broken, minus the ones that only say WHERE it happened to run this time
|
||||
-- (instance, pod, ...). Two incidents with the same signature in the same team
|
||||
-- are the same problem for a responder's purposes.
|
||||
--
|
||||
-- Computed in Go for new incidents (incidentSignature in incident_store.go).
|
||||
-- The backfill below MUST produce the same string; keep the volatile list in
|
||||
-- both places in step.
|
||||
ALTER TABLE incidents ADD COLUMN signature TEXT NOT NULL DEFAULT '';
|
||||
|
||||
UPDATE incidents SET signature =
|
||||
COALESCE(NULLIF(group_labels->>'alertname', ''), title) || '|' ||
|
||||
COALESCE((
|
||||
SELECT string_agg(e.k || '=' || e.v, ',' ORDER BY e.k)
|
||||
FROM jsonb_each_text(incidents.group_labels) AS e(k, v)
|
||||
WHERE e.k <> 'alertname'
|
||||
AND e.k NOT IN ('instance', 'pod', 'pod_name', 'pod_ip', 'container', 'container_name', 'endpoint')
|
||||
), '');
|
||||
|
||||
CREATE INDEX incidents_signature_idx ON incidents(team_id, signature, triggered_at DESC);
|
||||
@@ -0,0 +1,54 @@
|
||||
-- Dead man's switches become rows of their own.
|
||||
--
|
||||
-- 004 kept a team's switches in one string with one timeout and one severity,
|
||||
-- which was enough to configure them and not enough to show them: there was no
|
||||
-- thing to list, nothing to hang a status on, and every switch in a team had to
|
||||
-- share a deadline. A row per switch gives each its own name, matcher, timeout
|
||||
-- and severity, and gives the Team → Switches page something to be a list of.
|
||||
--
|
||||
-- The matcher keeps the syntax the string used, one matcher per row:
|
||||
-- `alertname=Watchdog,cluster=prod`. The unit of monitoring is still the
|
||||
-- fingerprint, so a matcher that many clusters satisfy is still one switch row
|
||||
-- watching several independent heartbeats.
|
||||
CREATE TABLE deadman_switches (
|
||||
id BIGSERIAL PRIMARY KEY,
|
||||
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
|
||||
|
||||
-- What the owner calls it. Defaults to the matcher when they do not say.
|
||||
name TEXT NOT NULL,
|
||||
|
||||
-- "," separates the label conditions, "=" is exact equality, and alertname is
|
||||
-- mandatory: it is what keeps the sweeper's candidate query on an index.
|
||||
matcher TEXT NOT NULL,
|
||||
|
||||
-- Seconds of silence before the switch is declared dead. Never zero: a switch
|
||||
-- that cannot fire is deleted, not disabled.
|
||||
timeout_seconds BIGINT NOT NULL CHECK (timeout_seconds > 0),
|
||||
|
||||
-- The severity its incidents open at. See 004 for why they carry their own.
|
||||
severity TEXT NOT NULL DEFAULT 'critical',
|
||||
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
||||
);
|
||||
|
||||
CREATE INDEX deadman_switches_team_idx ON deadman_switches (team_id);
|
||||
|
||||
-- Carry every team's configuration over, one row per matcher. A team whose
|
||||
-- timeout was zero had switches turned off, which is now "no rows".
|
||||
INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity)
|
||||
SELECT c.team_id, btrim(m), btrim(m), c.timeout_seconds, c.severity
|
||||
FROM deadman_configs c,
|
||||
LATERAL regexp_split_to_table(c.matchers, ';') AS m
|
||||
WHERE c.timeout_seconds > 0
|
||||
AND btrim(m) <> ''
|
||||
ORDER BY c.team_id;
|
||||
|
||||
-- The server seeds environment defaults into teams once, and remembers that it
|
||||
-- did. An install that had a row per team was already seeded; without this
|
||||
-- marker the first start after upgrading would seed teams that had switched
|
||||
-- theirs off.
|
||||
INSERT INTO settings (key, value)
|
||||
SELECT 'deadman_seeded', '1'
|
||||
WHERE EXISTS (SELECT 1 FROM deadman_configs);
|
||||
|
||||
DROP TABLE deadman_configs;
|
||||
@@ -0,0 +1,21 @@
|
||||
-- Which alert source an alert last arrived on.
|
||||
--
|
||||
-- Team -> Sources shows when each source last posted, which integrations
|
||||
-- already knew (last_used_at, stamped on every webhook). What it could not say
|
||||
-- was what a source delivered: an alert never recorded the key it came in on, so
|
||||
-- "prod alertmanager" and "staging alertmanager" were indistinguishable once
|
||||
-- inside. This column is that link, and lets the page show each source's last
|
||||
-- alert and how many alerts it has kept fresh over the past day.
|
||||
--
|
||||
-- Last sender wins: every accepted payload restamps it, the way it advances
|
||||
-- received_at. Two sources posting the same fingerprint into one team is
|
||||
-- already one alert, and it is attributed to whichever spoke last.
|
||||
--
|
||||
-- Nullable, and not backfilled. Alerts that arrived before this migration have
|
||||
-- no source, and NULL says so honestly rather than guessing. It heals by itself:
|
||||
-- Alertmanager re-sends every alert each repeat_interval, and each re-send is an
|
||||
-- accepted payload. Deleting a source keeps its alerts, unattributed.
|
||||
ALTER TABLE alerts ADD COLUMN integration_id BIGINT REFERENCES integrations(id) ON DELETE SET NULL;
|
||||
|
||||
CREATE INDEX alerts_integration_idx ON alerts (integration_id, received_at)
|
||||
WHERE integration_id IS NOT NULL;
|
||||
@@ -0,0 +1,60 @@
|
||||
-- Single sign-on through an OpenID Connect provider (Authentik, and anything
|
||||
-- else that speaks OIDC).
|
||||
--
|
||||
-- Four things change, and none of them touches a password user: every new column
|
||||
-- has a default that says "this is how it has always worked".
|
||||
--
|
||||
-- 1. user_identities says which provider account a user is. It is keyed on
|
||||
-- (issuer, subject), never on email or username: those are mutable at the
|
||||
-- provider, and a recycled address must not inherit somebody's account. A
|
||||
-- user can have several identities (a second provider later), and none at all
|
||||
-- (a local, password-only user), which is why this is a table and not two
|
||||
-- columns on users.
|
||||
--
|
||||
-- 2. team_members.source and users.admin_source record who granted a role. 'oidc'
|
||||
-- rows are owned by the group sync: it adds them when a group grants access
|
||||
-- and removes them when it stops, and nothing else may edit them. 'manual' rows
|
||||
-- are everything that existed before this migration, and are never touched by
|
||||
-- the sync. Without the marker the sync could not tell a membership it created
|
||||
-- from one an owner added by hand, and would have to either leave stale access
|
||||
-- behind or delete people it had no business deleting.
|
||||
--
|
||||
-- 3. sessions.max_expires_at is a hard ceiling on a session's life. Ordinary
|
||||
-- sessions slide for as long as they are used; a session made by an SSO login
|
||||
-- must not, because the login is the only moment the groups are re-read.
|
||||
-- Capping the session is what makes "removed from the group in the provider"
|
||||
-- take effect within a bounded time. NULL means no ceiling.
|
||||
--
|
||||
-- 4. oidc_logins holds a login that has been started and not yet finished: the
|
||||
-- state, nonce and PKCE verifier the callback must see again. A row rather
|
||||
-- than a signed cookie, so it survives a restart and needs no signing key.
|
||||
-- Only the hash of the state is stored, like every other token here; the
|
||||
-- nonce and verifier are useless without the state that names the row.
|
||||
CREATE TABLE user_identities (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
||||
issuer TEXT NOT NULL,
|
||||
subject TEXT NOT NULL,
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
last_login_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
UNIQUE (issuer, subject)
|
||||
);
|
||||
|
||||
CREATE INDEX user_identities_user_idx ON user_identities (user_id);
|
||||
|
||||
ALTER TABLE team_members
|
||||
ADD COLUMN source TEXT NOT NULL DEFAULT 'manual' CHECK (source IN ('manual', 'oidc'));
|
||||
|
||||
ALTER TABLE users
|
||||
ADD COLUMN admin_source TEXT NOT NULL DEFAULT 'manual' CHECK (admin_source IN ('manual', 'oidc'));
|
||||
|
||||
ALTER TABLE sessions ADD COLUMN max_expires_at BIGINT;
|
||||
|
||||
CREATE TABLE oidc_logins (
|
||||
state_hash TEXT PRIMARY KEY,
|
||||
nonce TEXT NOT NULL,
|
||||
pkce_verifier TEXT NOT NULL,
|
||||
expires_at BIGINT NOT NULL
|
||||
);
|
||||
|
||||
CREATE INDEX oidc_logins_expires_idx ON oidc_logins (expires_at);
|
||||
@@ -0,0 +1,40 @@
|
||||
-- Signing in from a terminal, for clients that cannot open a browser on the
|
||||
-- machine they run on (the TUI over SSH is the reason).
|
||||
--
|
||||
-- The flow is the OAuth device authorization grant, run by terdut itself rather
|
||||
-- than the identity provider, so the terminal never talks to the provider and
|
||||
-- the server issues its ordinary session at the end:
|
||||
--
|
||||
-- 1. The terminal asks for a login and gets two secrets: a device code it
|
||||
-- keeps and polls with, and a short user code it shows the person.
|
||||
-- 2. The person opens the verification URL on any device, signs in by whatever
|
||||
-- means the server offers, sees the user code, and approves it.
|
||||
-- 3. The terminal's next poll finds the row approved and is given a session.
|
||||
--
|
||||
-- Only the hash of the device code is stored, like every other token here: the
|
||||
-- device code is what earns a session, so a database read must not yield one.
|
||||
-- The user code is shown on screens and typed by people, so it is stored as is;
|
||||
-- on its own it can only be approved, never redeemed.
|
||||
--
|
||||
-- user_id is the person who approved. It is empty until then, and the session
|
||||
-- is minted at redemption, not at approval: an approval nobody collects must not
|
||||
-- leave a live session lying about.
|
||||
--
|
||||
-- last_polled_at lets the server refuse a client that polls faster than the
|
||||
-- interval it was told.
|
||||
CREATE TABLE device_logins (
|
||||
device_hash TEXT PRIMARY KEY,
|
||||
user_code TEXT NOT NULL UNIQUE,
|
||||
status TEXT NOT NULL DEFAULT 'pending' CHECK (status IN ('pending', 'approved', 'denied')),
|
||||
user_id BIGINT REFERENCES users(id) ON DELETE CASCADE,
|
||||
expires_at BIGINT NOT NULL,
|
||||
last_polled_at BIGINT NOT NULL DEFAULT 0
|
||||
);
|
||||
|
||||
CREATE INDEX device_logins_expires_idx ON device_logins (expires_at);
|
||||
|
||||
-- Where to send the browser once a single sign-on login completes. A person who
|
||||
-- opens /device?code=... without a session has to sign in first and then come
|
||||
-- back to it, and the same is true of any other deep link. Validated when it is
|
||||
-- stored: only a path on this server is ever kept.
|
||||
ALTER TABLE oidc_logins ADD COLUMN next TEXT NOT NULL DEFAULT '/';
|
||||
@@ -0,0 +1,26 @@
|
||||
-- Per-team OIDC group configuration, replacing the global
|
||||
-- TERDUT_OIDC_GROUP_MAPPINGS env var.
|
||||
--
|
||||
-- Group -> team -> role used to be one global list an operator set for the
|
||||
-- whole install, matched against a team by name, and the sync would create
|
||||
-- the team if no team by that name existed yet. That put the decision of
|
||||
-- which group controls a team in the server's environment rather than the
|
||||
-- team's own hands, meant changing it needed an env var edit and a restart,
|
||||
-- and let a typo in a team name silently create a stray team.
|
||||
--
|
||||
-- Each team now names, itself, which group grants membership and which
|
||||
-- grants ownership. Nullable: most teams need neither. No uniqueness
|
||||
-- constraint on either column — two teams may legitimately watch the same
|
||||
-- provider group (a broad team and a narrower one both keyed off overlapping
|
||||
-- groups is a choice for their owners to make, not one the schema should
|
||||
-- refuse).
|
||||
--
|
||||
-- BREAKING CHANGE, deliberately not auto-migrated: TERDUT_OIDC_GROUP_MAPPINGS
|
||||
-- stops being read as of this version, and the sync no longer creates a team
|
||||
-- by name. Every team's group binding must be set again through
|
||||
-- PUT /api/teams/{teamID}/oidc-groups. Until an owner does that, an
|
||||
-- OIDC-sourced membership in that team is dropped at that user's next SSO
|
||||
-- sign-in, the same way any other loss of group access is handled. See the
|
||||
-- README's OIDC section.
|
||||
ALTER TABLE teams ADD COLUMN oidc_member_group TEXT;
|
||||
ALTER TABLE teams ADD COLUMN oidc_owner_group TEXT;
|
||||
@@ -0,0 +1,43 @@
|
||||
-- Service accounts: a scoped, non-human credential for automation (e.g.
|
||||
-- terdut-operator) that needs to manage teams, escalation policies, dead
|
||||
-- man's switches, integrations and OIDC group bindings without impersonating
|
||||
-- a human user. See SERVICE-ACCOUNTS.md for the design this implements.
|
||||
--
|
||||
-- Deliberately not a users row: no password_hash, no is_admin, no
|
||||
-- user_identities linkage, so a service account can never be pulled into
|
||||
-- OIDC group sync or password login, and is never mistaken for a human in an
|
||||
-- audit trail.
|
||||
--
|
||||
-- scope is 'instance' (acts with the same reach system administration has
|
||||
-- over teams: create one, list them, mint a 'team'-scoped account against
|
||||
-- any of them) or 'team' (acts as that one team's owner, and nothing else).
|
||||
-- The CHECK ties team_id's presence to scope directly, rather than leaving it
|
||||
-- to application code to keep the two consistent.
|
||||
CREATE TABLE service_accounts (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
name TEXT NOT NULL UNIQUE,
|
||||
scope TEXT NOT NULL CHECK (scope IN ('instance', 'team')),
|
||||
team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE,
|
||||
created_by BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
CONSTRAINT service_accounts_scope_team_id_chk CHECK (
|
||||
(scope = 'team' AND team_id IS NOT NULL) OR
|
||||
(scope = 'instance' AND team_id IS NULL)
|
||||
)
|
||||
);
|
||||
|
||||
CREATE INDEX service_accounts_team_id_idx ON service_accounts(team_id);
|
||||
|
||||
-- One account, many keys: rotation is minting a new one and revoking the
|
||||
-- old, the same shape api_keys already has, so an account's identity and
|
||||
-- audit history survive a rotation instead of being recreated by it.
|
||||
CREATE TABLE service_account_keys (
|
||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
||||
service_account_id BIGINT NOT NULL REFERENCES service_accounts(id) ON DELETE CASCADE,
|
||||
key_hash TEXT NOT NULL UNIQUE,
|
||||
name TEXT NOT NULL,
|
||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
||||
last_used_at BIGINT
|
||||
);
|
||||
|
||||
CREATE INDEX service_account_keys_service_account_id_idx ON service_account_keys(service_account_id);
|
||||
@@ -7,7 +7,13 @@ import "time"
|
||||
// to it — acknowledgement, assignment, notes and closure all live on the
|
||||
// Incident an alert belongs to.
|
||||
type Alert struct {
|
||||
ID int64 `json:"id"`
|
||||
ID int64 `json:"id"`
|
||||
|
||||
// TeamID is the team whose integration received this alert, and TeamName
|
||||
// rides along so a combined list can label a row without a second request.
|
||||
TeamID int64 `json:"team_id"`
|
||||
TeamName string `json:"team_name,omitempty"`
|
||||
|
||||
Fingerprint string `json:"fingerprint"`
|
||||
Name string `json:"name"`
|
||||
Status string `json:"status"` // "firing" or "resolved"
|
||||
|
||||
@@ -11,6 +11,19 @@ import "time"
|
||||
// the webhook and the sweeper may flip to "resolved" once every member alert has
|
||||
// stopped firing.
|
||||
type Incident struct {
|
||||
// EscalationLevel is which rung of its team's ladder this incident is on,
|
||||
// 0 for none — either the team has no ladder, or somebody has answered.
|
||||
// EscalationDueAt is when the current level runs out, so a client can say
|
||||
// how long is left rather than only what already happened.
|
||||
EscalationLevel int64 `json:"escalation_level"`
|
||||
EscalationDueAt *time.Time `json:"escalation_due_at,omitempty"`
|
||||
|
||||
// TeamID is the team that owns this incident, fixed when it opens: an
|
||||
// incident never moves between teams. TeamName rides along so the combined
|
||||
// queue can badge each row without a second request.
|
||||
TeamID int64 `json:"team_id"`
|
||||
TeamName string `json:"team_name,omitempty"`
|
||||
|
||||
ID int64 `json:"id"`
|
||||
GroupKey string `json:"group_key"`
|
||||
Title string `json:"title"`
|
||||
@@ -66,3 +79,15 @@ type IncidentEvent struct {
|
||||
Detail *string `json:"detail,omitempty"`
|
||||
CreatedAt time.Time `json:"created_at"`
|
||||
}
|
||||
|
||||
// SimilarIncident is an earlier, resolved incident with the same signature as
|
||||
// the one being looked at. ResolutionNotes are the "what fixed it" notes;
|
||||
// NoteCount counts the plain working notes, which live on the timeline.
|
||||
type SimilarIncident struct {
|
||||
ID int64 `json:"id"`
|
||||
Title string `json:"title"`
|
||||
TriggeredAt time.Time `json:"triggered_at"`
|
||||
ResolvedAt time.Time `json:"resolved_at"`
|
||||
NoteCount int `json:"note_count"`
|
||||
ResolutionNotes []IncidentEvent `json:"resolution_notes"`
|
||||
}
|
||||
|
||||
@@ -3,7 +3,14 @@ package models
|
||||
import "time"
|
||||
|
||||
type ScheduleEntry struct {
|
||||
ID int64 `json:"id"`
|
||||
ID int64 `json:"id"`
|
||||
|
||||
// TeamID is whose rota this shift belongs to; TeamName rides along so the
|
||||
// combined "who is on call" view can label each entry without a second
|
||||
// request.
|
||||
TeamID int64 `json:"team_id"`
|
||||
TeamName string `json:"team_name,omitempty"`
|
||||
|
||||
UserID int64 `json:"user_id"`
|
||||
Username string `json:"username"`
|
||||
Date string `json:"date"` // YYYY-MM-DD
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
package models
|
||||
|
||||
import "time"
|
||||
|
||||
// Service account scopes. Instance acts with the same reach system
|
||||
// administration has over teams: create one, list them, mint a team-scoped
|
||||
// account against any of them. Team acts as that one team's owner, and
|
||||
// nothing else.
|
||||
const (
|
||||
ServiceAccountScopeInstance = "instance"
|
||||
ServiceAccountScopeTeam = "team"
|
||||
)
|
||||
|
||||
// ServiceAccount is a non-human credential: not a users row, so it never
|
||||
// touches OIDC group sync, login, or the is_admin flag, and is never mistaken
|
||||
// for a human in an audit trail (see api_keys' user_id, which every service
|
||||
// account key deliberately does not have).
|
||||
type ServiceAccount struct {
|
||||
ID int64 `json:"id"`
|
||||
Name string `json:"name"`
|
||||
Scope string `json:"scope"`
|
||||
TeamID *int64 `json:"team_id,omitempty"`
|
||||
CreatedBy *int64 `json:"created_by,omitempty"`
|
||||
CreatedAt time.Time `json:"created_at"`
|
||||
}
|
||||
|
||||
// ServiceAccountKey is one bearer credential on a ServiceAccount. Multiple
|
||||
// keys per account, the same shape as APIKey, are what let rotation mint a
|
||||
// new one and revoke the old without recreating the account.
|
||||
type ServiceAccountKey struct {
|
||||
ID int64 `json:"id"`
|
||||
ServiceAccountID int64 `json:"service_account_id"`
|
||||
Name string `json:"name"`
|
||||
CreatedAt time.Time `json:"created_at"`
|
||||
LastUsedAt *time.Time `json:"last_used_at,omitempty"`
|
||||
Key string `json:"key,omitempty"` // populated only on creation, never stored
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
package models
|
||||
|
||||
import "time"
|
||||
|
||||
// Team is the unit of tenancy: it owns its incidents, alerts, schedule and
|
||||
// integrations, and a user sees exactly the teams they belong to.
|
||||
type Team struct {
|
||||
ID int64 `json:"id"`
|
||||
Name string `json:"name"`
|
||||
CreatedAt time.Time `json:"created_at"`
|
||||
|
||||
// Role is the caller's own role in this team, populated when a team is
|
||||
// listed for a particular person. Empty when nobody in particular is
|
||||
// asking, as in the admin listing.
|
||||
Role string `json:"role,omitempty"`
|
||||
|
||||
// Source says who granted Role, on the endpoint that lists one user's teams:
|
||||
// "manual", or "oidc" when the identity provider's groups did.
|
||||
Source string `json:"source,omitempty"`
|
||||
}
|
||||
|
||||
// Team roles. An owner configures the team — its schedule, its integrations and
|
||||
// who is in it. A member works its incidents.
|
||||
const (
|
||||
RoleOwner = "owner"
|
||||
RoleMember = "member"
|
||||
)
|
||||
|
||||
// TeamMember is one person's membership of one team.
|
||||
type TeamMember struct {
|
||||
TeamID int64 `json:"team_id"`
|
||||
UserID int64 `json:"user_id"`
|
||||
Username string `json:"username"`
|
||||
Role string `json:"role"`
|
||||
JoinedAt time.Time `json:"joined_at"`
|
||||
|
||||
// Source is who granted the membership: "manual", or "oidc" when the
|
||||
// identity provider's groups did and only they can change it.
|
||||
Source string `json:"source"`
|
||||
}
|
||||
|
||||
// Integration is how alerts get in, and the only thing that says which team an
|
||||
// arriving alert belongs to.
|
||||
type Integration struct {
|
||||
ID int64 `json:"id"`
|
||||
TeamID int64 `json:"team_id"`
|
||||
Kind string `json:"kind"`
|
||||
Name string `json:"name"`
|
||||
CreatedAt time.Time `json:"created_at"`
|
||||
LastUsedAt *time.Time `json:"last_used_at,omitempty"`
|
||||
|
||||
// Key is the raw integration key, shown once when the integration is
|
||||
// created and never stored. URL is the address to point the sender at,
|
||||
// likewise only complete at creation time.
|
||||
Key string `json:"key,omitempty"`
|
||||
URL string `json:"url,omitempty"`
|
||||
}
|
||||
|
||||
// Integration kinds.
|
||||
const (
|
||||
IntegrationAlertmanager = "alertmanager"
|
||||
)
|
||||
@@ -12,6 +12,23 @@ type User struct {
|
||||
// none of their own; incidents assigned to them fall back to the configured
|
||||
// fallback topic instead.
|
||||
NtfyTopic *string `json:"ntfy_topic,omitempty"`
|
||||
|
||||
// DisabledAt is when the account was taken out of use, or nil. A disabled
|
||||
// user cannot authenticate by either credential, and keeps their name on
|
||||
// every acknowledgement and timeline entry they made.
|
||||
DisabledAt *time.Time `json:"disabled_at,omitempty"`
|
||||
|
||||
// IsAdmin is the system administrator flag: managing users and API keys.
|
||||
// Not omitempty — a client has to be able to tell "false" from "this server
|
||||
// is too old to have the field", and the web UI decides what to show from
|
||||
// it.
|
||||
IsAdmin bool `json:"is_admin"`
|
||||
|
||||
// AdminSource is who granted the flag: "manual" or "oidc". An "oidc"
|
||||
// administrator follows the identity provider's groups, so the UI shows it as
|
||||
// managed there and the API refuses to revoke it by hand. Only set on the
|
||||
// user endpoints that show it.
|
||||
AdminSource string `json:"admin_source,omitempty"`
|
||||
}
|
||||
|
||||
type APIKey struct {
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
// Package oidc signs users in through an OpenID Connect provider and turns the
|
||||
// groups it reports into the access terdut grants.
|
||||
//
|
||||
// The package knows nothing about the database or HTTP handlers: Grants is a
|
||||
// pure function of configuration and groups, and Provider is the protocol. The
|
||||
// api package joins them to users, teams and sessions.
|
||||
package oidc
|
||||
|
||||
import (
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||
)
|
||||
|
||||
// Role names match models.RoleOwner and RoleMember. They are restated here so
|
||||
// the package stays free of the models import; config.Validate has already
|
||||
// refused anything else.
|
||||
const (
|
||||
roleOwner = "owner"
|
||||
roleMember = "member"
|
||||
)
|
||||
|
||||
// Grants is the account-wide access a set of groups confers. Team access is a
|
||||
// separate question — see TeamGroup and ComputeTeamGrants — because it is
|
||||
// configured per team in the database, not in this package's cfg.
|
||||
type Grants struct {
|
||||
// Admitted is false when AllowedGroups is set and the user is in none of
|
||||
// them. Nothing else in the struct means anything then.
|
||||
Admitted bool
|
||||
|
||||
// Admin is whether the user is in the admin group.
|
||||
Admin bool
|
||||
}
|
||||
|
||||
// ComputeGrants evaluates the account-wide configuration against groups.
|
||||
func ComputeGrants(cfg config.OIDC, groups []string) Grants {
|
||||
in := make(map[string]bool, len(groups))
|
||||
for _, g := range groups {
|
||||
in[g] = true
|
||||
}
|
||||
|
||||
var g Grants
|
||||
|
||||
g.Admitted = len(cfg.AllowedGroups) == 0
|
||||
for _, allowed := range cfg.AllowedGroups {
|
||||
if in[allowed] {
|
||||
g.Admitted = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !g.Admitted {
|
||||
return g
|
||||
}
|
||||
|
||||
g.Admin = cfg.AdminGroup != "" && in[cfg.AdminGroup]
|
||||
return g
|
||||
}
|
||||
|
||||
// TeamGroup is one team's own OIDC binding: which group, if any, grants
|
||||
// member access to it and which grants owner access, as read from
|
||||
// teams.oidc_member_group / teams.oidc_owner_group.
|
||||
type TeamGroup struct {
|
||||
TeamID int64
|
||||
MemberGroup string // "" means no group grants member access here.
|
||||
OwnerGroup string // "" means no group grants owner access here.
|
||||
}
|
||||
|
||||
// ComputeTeamGrants evaluates every team's own group binding against groups,
|
||||
// and returns the role each team grants, keyed by team ID. A team absent from
|
||||
// the result is not granted at all. Where a team's member and owner groups
|
||||
// both match, the owner group wins — the same "highest role wins" rule that
|
||||
// applied across the old global mapping list applies here across one team's
|
||||
// two fields, so belonging to both groups makes somebody an owner rather than
|
||||
// whichever field happened to be checked last.
|
||||
func ComputeTeamGrants(teamGroups []TeamGroup, groups []string) map[int64]string {
|
||||
in := make(map[string]bool, len(groups))
|
||||
for _, g := range groups {
|
||||
in[g] = true
|
||||
}
|
||||
|
||||
out := map[int64]string{}
|
||||
for _, tg := range teamGroups {
|
||||
role := ""
|
||||
if tg.MemberGroup != "" && in[tg.MemberGroup] {
|
||||
role = roleMember
|
||||
}
|
||||
if tg.OwnerGroup != "" && in[tg.OwnerGroup] && rank(roleOwner) > rank(role) {
|
||||
role = roleOwner
|
||||
}
|
||||
if role != "" {
|
||||
out[tg.TeamID] = role
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// rank orders roles; an unknown or absent role ranks lowest.
|
||||
func rank(role string) int {
|
||||
switch role {
|
||||
case roleOwner:
|
||||
return 2
|
||||
case roleMember:
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// HigherRole reports whether role a outranks role b.
|
||||
func HigherRole(a, b string) bool { return rank(a) > rank(b) }
|
||||
@@ -0,0 +1,140 @@
|
||||
package oidc
|
||||
|
||||
import (
|
||||
"reflect"
|
||||
"testing"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||
)
|
||||
|
||||
func testCfg() config.OIDC {
|
||||
return config.OIDC{
|
||||
AllowedGroups: []string{"terdut-users"},
|
||||
AdminGroup: "terdut-admins",
|
||||
}
|
||||
}
|
||||
|
||||
func TestComputeGrants(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
groups []string
|
||||
want Grants
|
||||
}{
|
||||
{
|
||||
name: "not in an allowed group is refused",
|
||||
groups: []string{"sre", "terdut-admins"},
|
||||
want: Grants{Admitted: false},
|
||||
},
|
||||
{
|
||||
name: "allowed but no grants",
|
||||
groups: []string{"terdut-users"},
|
||||
want: Grants{Admitted: true},
|
||||
},
|
||||
{
|
||||
name: "admin group grants admin",
|
||||
groups: []string{"terdut-users", "terdut-admins"},
|
||||
want: Grants{Admitted: true, Admin: true},
|
||||
},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := ComputeGrants(testCfg(), tt.groups)
|
||||
if !reflect.DeepEqual(got, tt.want) {
|
||||
t.Errorf("got %+v, want %+v", got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestComputeGrants_NoAllowedGroupsAdmitsEveryone(t *testing.T) {
|
||||
cfg := testCfg()
|
||||
cfg.AllowedGroups = nil
|
||||
if g := ComputeGrants(cfg, nil); !g.Admitted {
|
||||
t.Error("with no allowed groups configured, everybody the provider authenticates is admitted")
|
||||
}
|
||||
}
|
||||
|
||||
// testTeamGroups is one SRE team keyed off two groups (a member group and a
|
||||
// higher owner group) and one Platform team keyed off a member group only —
|
||||
// the same shape the old global TERDUT_OIDC_GROUP_MAPPINGS example used.
|
||||
func testTeamGroups() []TeamGroup {
|
||||
return []TeamGroup{
|
||||
{TeamID: 1, MemberGroup: "sre", OwnerGroup: "sre-leads"},
|
||||
{TeamID: 2, MemberGroup: "platform"},
|
||||
}
|
||||
}
|
||||
|
||||
func TestComputeTeamGrants(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
teamGroups []TeamGroup
|
||||
groups []string
|
||||
want map[int64]string
|
||||
}{
|
||||
{
|
||||
name: "no matching group grants nothing",
|
||||
teamGroups: testTeamGroups(),
|
||||
groups: []string{"terdut-users"},
|
||||
want: map[int64]string{},
|
||||
},
|
||||
{
|
||||
name: "member group grants member",
|
||||
teamGroups: testTeamGroups(),
|
||||
groups: []string{"sre"},
|
||||
want: map[int64]string{1: roleMember},
|
||||
},
|
||||
{
|
||||
name: "owner group grants owner",
|
||||
teamGroups: testTeamGroups(),
|
||||
groups: []string{"sre-leads"},
|
||||
want: map[int64]string{1: roleOwner},
|
||||
},
|
||||
{
|
||||
name: "in both of a team's groups, owner wins",
|
||||
teamGroups: testTeamGroups(),
|
||||
groups: []string{"sre", "sre-leads"},
|
||||
want: map[int64]string{1: roleOwner},
|
||||
},
|
||||
{
|
||||
name: "several teams from several groups",
|
||||
teamGroups: testTeamGroups(),
|
||||
groups: []string{"sre", "platform"},
|
||||
want: map[int64]string{1: roleMember, 2: roleMember},
|
||||
},
|
||||
{
|
||||
name: "two teams may share a group",
|
||||
teamGroups: []TeamGroup{
|
||||
{TeamID: 1, MemberGroup: "sre"},
|
||||
{TeamID: 2, MemberGroup: "sre"},
|
||||
},
|
||||
groups: []string{"sre"},
|
||||
want: map[int64]string{1: roleMember, 2: roleMember},
|
||||
},
|
||||
{
|
||||
name: "a team with neither field set is never granted",
|
||||
teamGroups: []TeamGroup{{TeamID: 1}},
|
||||
groups: []string{"sre", "sre-leads", "platform"},
|
||||
want: map[int64]string{},
|
||||
},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := ComputeTeamGrants(tt.teamGroups, tt.groups)
|
||||
if !reflect.DeepEqual(got, tt.want) {
|
||||
t.Errorf("got %+v, want %+v", got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestStringList(t *testing.T) {
|
||||
if got := stringList([]any{"a", "", 3, "b"}); !reflect.DeepEqual(got, []string{"a", "b"}) {
|
||||
t.Errorf("list: %v", got)
|
||||
}
|
||||
if got := stringList("solo"); !reflect.DeepEqual(got, []string{"solo"}) {
|
||||
t.Errorf("single string: %v", got)
|
||||
}
|
||||
if got := stringList(nil); got != nil {
|
||||
t.Errorf("nil: %v", got)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,173 @@
|
||||
package oidc
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
gooidc "github.com/coreos/go-oidc/v3/oidc"
|
||||
"golang.org/x/oauth2"
|
||||
|
||||
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||
)
|
||||
|
||||
// CallbackPath is where the provider sends the browser back to. Register
|
||||
// <TERDUT_PUBLIC_URL>/api/oidc/callback as the redirect URI at the provider.
|
||||
const CallbackPath = "/api/oidc/callback"
|
||||
|
||||
// Identity is what the provider says about somebody who has just signed in.
|
||||
type Identity struct {
|
||||
Issuer string
|
||||
Subject string
|
||||
Username string
|
||||
Email string
|
||||
|
||||
// EmailVerified is the provider's own claim. Whether to believe it is
|
||||
// config.OIDC.TrustEmail's business, not this package's.
|
||||
EmailVerified bool
|
||||
|
||||
Groups []string
|
||||
}
|
||||
|
||||
// Provider runs the authorization-code flow with PKCE against one issuer.
|
||||
type Provider struct {
|
||||
cfg config.OIDC
|
||||
redirectURL string
|
||||
http *http.Client
|
||||
|
||||
// Discovery is fetched on first use, not at startup. A provider that is
|
||||
// down when terdut starts must not stop terdut starting: password login is
|
||||
// the way in while it is down, and it can only be that if the server is up.
|
||||
mu sync.Mutex
|
||||
provider *gooidc.Provider
|
||||
}
|
||||
|
||||
// New returns a Provider for cfg. publicURL is the base of the redirect URI.
|
||||
func New(cfg config.OIDC, publicURL string) *Provider {
|
||||
return &Provider{
|
||||
cfg: cfg,
|
||||
redirectURL: trimSlash(publicURL) + CallbackPath,
|
||||
http: &http.Client{Timeout: 10 * time.Second},
|
||||
}
|
||||
}
|
||||
|
||||
func trimSlash(s string) string {
|
||||
for len(s) > 0 && s[len(s)-1] == '/' {
|
||||
s = s[:len(s)-1]
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// Name is what the sign-in button calls the provider.
|
||||
func (p *Provider) Name() string { return p.cfg.Name }
|
||||
|
||||
// Config is the configuration this provider was built from.
|
||||
func (p *Provider) Config() config.OIDC { return p.cfg }
|
||||
|
||||
// discover returns the provider's metadata, fetching it if need be. A failure is
|
||||
// not cached, so the next login tries again.
|
||||
func (p *Provider) discover(ctx context.Context) (*gooidc.Provider, error) {
|
||||
p.mu.Lock()
|
||||
defer p.mu.Unlock()
|
||||
if p.provider != nil {
|
||||
return p.provider, nil
|
||||
}
|
||||
ctx = gooidc.ClientContext(ctx, p.http)
|
||||
prov, err := gooidc.NewProvider(ctx, p.cfg.Issuer)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("oidc discovery: %w", err)
|
||||
}
|
||||
p.provider = prov
|
||||
return prov, nil
|
||||
}
|
||||
|
||||
func (p *Provider) oauth(prov *gooidc.Provider) *oauth2.Config {
|
||||
return &oauth2.Config{
|
||||
ClientID: p.cfg.ClientID,
|
||||
ClientSecret: p.cfg.ClientSecret,
|
||||
Endpoint: prov.Endpoint(),
|
||||
RedirectURL: p.redirectURL,
|
||||
Scopes: p.cfg.Scopes,
|
||||
}
|
||||
}
|
||||
|
||||
// NewVerifier returns a fresh PKCE code verifier.
|
||||
func NewVerifier() string { return oauth2.GenerateVerifier() }
|
||||
|
||||
// AuthURL is where to send the browser to sign in.
|
||||
func (p *Provider) AuthURL(ctx context.Context, state, nonce, verifier string) (string, error) {
|
||||
prov, err := p.discover(ctx)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return p.oauth(prov).AuthCodeURL(state,
|
||||
oauth2.S256ChallengeOption(verifier),
|
||||
gooidc.Nonce(nonce),
|
||||
), nil
|
||||
}
|
||||
|
||||
// Exchange trades the authorization code for tokens, verifies the ID token
|
||||
// (signature, issuer, audience, expiry and nonce) and returns who it names.
|
||||
func (p *Provider) Exchange(ctx context.Context, code, verifier, nonce string) (*Identity, error) {
|
||||
prov, err := p.discover(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
ctx = gooidc.ClientContext(ctx, p.http)
|
||||
|
||||
tok, err := p.oauth(prov).Exchange(ctx, code, oauth2.VerifierOption(verifier))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("oidc token exchange: %w", err)
|
||||
}
|
||||
raw, _ := tok.Extra("id_token").(string)
|
||||
if raw == "" {
|
||||
return nil, errors.New("oidc: token response has no id_token")
|
||||
}
|
||||
idToken, err := prov.Verifier(&gooidc.Config{ClientID: p.cfg.ClientID}).Verify(ctx, raw)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("oidc: verify id_token: %w", err)
|
||||
}
|
||||
if idToken.Nonce != nonce {
|
||||
return nil, errors.New("oidc: id_token nonce mismatch")
|
||||
}
|
||||
|
||||
var claims map[string]any
|
||||
if err := idToken.Claims(&claims); err != nil {
|
||||
return nil, fmt.Errorf("oidc: read claims: %w", err)
|
||||
}
|
||||
return p.identity(idToken.Issuer, idToken.Subject, claims), nil
|
||||
}
|
||||
|
||||
// identity maps raw claims onto an Identity using the configured claim names.
|
||||
func (p *Provider) identity(issuer, subject string, claims map[string]any) *Identity {
|
||||
id := &Identity{Issuer: issuer, Subject: subject}
|
||||
id.Username, _ = claims[p.cfg.UsernameClaim].(string)
|
||||
id.Email, _ = claims[p.cfg.EmailClaim].(string)
|
||||
id.EmailVerified, _ = claims["email_verified"].(bool)
|
||||
id.Groups = stringList(claims[p.cfg.GroupsClaim])
|
||||
return id
|
||||
}
|
||||
|
||||
// stringList reads a claim that is a list of strings, or a single string, which
|
||||
// some providers send for a one-element list.
|
||||
func stringList(v any) []string {
|
||||
switch t := v.(type) {
|
||||
case string:
|
||||
if t == "" {
|
||||
return nil
|
||||
}
|
||||
return []string{t}
|
||||
case []any:
|
||||
out := make([]string, 0, len(t))
|
||||
for _, e := range t {
|
||||
if s, ok := e.(string); ok && s != "" {
|
||||
out = append(out, s)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
package web
|
||||
|
||||
import (
|
||||
"io/fs"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestCopyIncidentIsEmbedded(t *testing.T) {
|
||||
sub, err := fs.Sub(files, "static")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for file, want := range map[string]string{
|
||||
"js/incident.js": "copyIncident",
|
||||
"js/ui.js": "copy:",
|
||||
} {
|
||||
b, err := fs.ReadFile(sub, file)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !strings.Contains(string(b), want) {
|
||||
t.Errorf("%s lacks %s", file, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,86 @@
|
||||
package web
|
||||
|
||||
import (
|
||||
"io/fs"
|
||||
"regexp"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func read(t *testing.T, name string) string {
|
||||
t.Helper()
|
||||
sub, err := fs.Sub(files, "static")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
b, err := fs.ReadFile(sub, name)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
|
||||
// The sign-in button has to be a link the browser navigates, not script: the
|
||||
// CSP's connect-src is 'self', so a fetch to the identity provider is blocked,
|
||||
// and it is a redirect to the provider that the server answers.
|
||||
func TestLoginPageOffersSSOAsAPlainLink(t *testing.T) {
|
||||
html := read(t, "index.html")
|
||||
if !regexp.MustCompile(`<a[^>]*id="sso-link"[^>]*href="/api/oidc/login"|<a[^>]*href="/api/oidc/login"[^>]*id="sso-link"`).MatchString(html) {
|
||||
t.Error("index.html has no <a id=sso-link href=/api/oidc/login>")
|
||||
}
|
||||
if !strings.Contains(html, `id="password-login"`) {
|
||||
t.Error("the password fields must sit in #password-login so a server can hide them")
|
||||
}
|
||||
}
|
||||
|
||||
// Every code the server can put in ?sso_error= must have a message, or a
|
||||
// refused person sees a generic failure and cannot tell what to ask for.
|
||||
func TestLoginExplainsEverySSOError(t *testing.T) {
|
||||
js := read(t, "js/app.js")
|
||||
for _, code := range []string{
|
||||
"denied", "expired", "failed", "unavailable",
|
||||
"not_allowed", "no_email", "email_conflict", "disabled",
|
||||
} {
|
||||
if !regexp.MustCompile(`\b` + code + `:`).MatchString(js) {
|
||||
t.Errorf("app.js has no message for sso_error=%s", code)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The page a terminal's prompt links to has to be reachable as a route, or the
|
||||
// link 404s into the queue and the code is never seen.
|
||||
func TestDevicePageIsRoutedAndCallsTheApprovalAPI(t *testing.T) {
|
||||
if !strings.Contains(read(t, "index.html"), `id="view-device"`) {
|
||||
t.Error("index.html has no #view-device section")
|
||||
}
|
||||
app := read(t, "js/app.js")
|
||||
if !strings.Contains(app, "name === 'device'") || !strings.Contains(app, "device: {") {
|
||||
t.Error("app.js does not route /device")
|
||||
}
|
||||
// The SSO button must carry the page asked for through the provider.
|
||||
if !strings.Contains(app, "/api/oidc/login?next=") {
|
||||
t.Error("the SSO link does not carry next=")
|
||||
}
|
||||
dev := read(t, "js/device.js")
|
||||
for _, want := range []string{"approveDevice", "denyDevice"} {
|
||||
if !strings.Contains(dev, want) {
|
||||
t.Errorf("device.js never calls %s", want)
|
||||
}
|
||||
}
|
||||
api := read(t, "js/api.js")
|
||||
for _, want := range []string{"/oidc/device/approve", "/oidc/device/deny"} {
|
||||
if !strings.Contains(api, want) {
|
||||
t.Errorf("api.js has no call to %s", want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// SSO-managed access must be marked in every view that edits it.
|
||||
func TestManagedAccessIsMarkedWhereItIsEdited(t *testing.T) {
|
||||
for _, file := range []string{"js/team.js", "js/adminteam.js", "js/adminuser.js", "js/admin.js"} {
|
||||
js := read(t, file)
|
||||
if !strings.Contains(js, "ssoBadge") || !strings.Contains(js, "SSO_MANAGED") {
|
||||
t.Errorf("%s does not mark or explain SSO-managed access", file)
|
||||
}
|
||||
}
|
||||
}
|
||||
+471
-15
@@ -31,6 +31,14 @@
|
||||
--snooze: #6b5bd2;
|
||||
--snooze-soft: #efedfb;
|
||||
|
||||
/* Two hues that mean nothing on their own. The rota needs six colours to
|
||||
tell six people apart and the palette above only has four that are not
|
||||
already an alarm. */
|
||||
--teal: #0f7d8c;
|
||||
--teal-soft: #e3f4f6;
|
||||
--pink: #b3427e;
|
||||
--pink-soft: #fbe8f2;
|
||||
|
||||
--radius: 10px;
|
||||
--radius-sm: 6px;
|
||||
--shadow: 0 1px 2px rgb(16 24 40 / 6%), 0 1px 3px rgb(16 24 40 / 8%);
|
||||
@@ -39,7 +47,26 @@
|
||||
--font: system-ui, -apple-system, "Segoe UI", Roboto, "Helvetica Neue", Arial, sans-serif;
|
||||
--mono: ui-monospace, SFMono-Regular, Menlo, Consolas, monospace;
|
||||
|
||||
/* A ~4-size type scale, per issue #26's "use one type scale" ask. This
|
||||
file still has a dozen one-off font-size values below; migrating the
|
||||
low-risk, purely cosmetic ones (standalone titles with no dimensional
|
||||
or functional constraint) onto these tokens is a start, not the whole
|
||||
job — the rest (sizes tied to a fixed shape like the avatar circle, to
|
||||
a deliberately prominent display like a stat tile or the on-call name,
|
||||
or to a non-negotiable constraint like the 16px that stops iOS zooming
|
||||
into an input) stay as either their own pixel value or a documented
|
||||
exception, since guessing at those without seeing them render risks
|
||||
trading one inconsistency for a worse one. */
|
||||
--fs-xs: 12px;
|
||||
--fs-sm: 13px;
|
||||
--fs-base: 14px;
|
||||
--fs-lg: 18px;
|
||||
--fs-xl: 21px;
|
||||
|
||||
--topbar-h: 52px;
|
||||
/* The phone-width bottom tab bar's height. Unused above 900px: the
|
||||
desktop block overrides .nav/.view/.toast directly rather than reading
|
||||
this back down to 0. */
|
||||
--tabbar-h: 58px;
|
||||
--safe-top: env(safe-area-inset-top, 0px);
|
||||
--safe-bottom: env(safe-area-inset-bottom, 0px);
|
||||
@@ -72,6 +99,11 @@
|
||||
--snooze: #a89bff;
|
||||
--snooze-soft: #262245;
|
||||
|
||||
--teal: #4fc2d4;
|
||||
--teal-soft: #0f2e33;
|
||||
--pink: #f07fb8;
|
||||
--pink-soft: #3a1c2d;
|
||||
|
||||
--shadow: 0 1px 2px rgb(0 0 0 / 40%);
|
||||
--shadow-lg: 0 16px 40px rgb(0 0 0 / 55%);
|
||||
}
|
||||
@@ -120,6 +152,13 @@ h1, h2, h3 { margin: 0; line-height: 1.25; }
|
||||
.login-brand { display: flex; align-items: center; gap: 10px; margin-bottom: 8px; }
|
||||
.login-brand h1 { font-size: 24px; letter-spacing: -0.01em; }
|
||||
.login-hint { color: var(--faint); font-size: 13px; margin: 4px 0 0; }
|
||||
.login-password { display: grid; gap: 14px; }
|
||||
/* "or" between the single sign-on button and the password form. */
|
||||
.login-divider {
|
||||
display: flex; align-items: center; gap: 10px; margin: 0;
|
||||
color: var(--faint); font-size: 13px;
|
||||
}
|
||||
.login-divider::before, .login-divider::after { content: ''; flex: 1; height: 1px; background: var(--border); }
|
||||
|
||||
label { display: grid; gap: 6px; }
|
||||
label > span { font-size: 13px; font-weight: 600; color: var(--muted); }
|
||||
@@ -156,7 +195,7 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
||||
transition: background 0.12s, border-color 0.12s, opacity 0.12s;
|
||||
}
|
||||
.btn:hover { background: var(--surface-hover); }
|
||||
.btn:disabled { opacity: 0.55; cursor: default; }
|
||||
.btn:disabled, .btn-sm:disabled { opacity: 0.55; cursor: default; }
|
||||
.btn-primary { background: var(--accent); border-color: var(--accent); color: var(--accent-text); }
|
||||
.btn-primary:hover { background: var(--accent); filter: brightness(1.06); }
|
||||
.btn-danger { background: var(--crit); border-color: var(--crit); color: #fff; }
|
||||
@@ -183,8 +222,11 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
||||
-webkit-backdrop-filter: saturate(1.4) blur(12px);
|
||||
border-bottom: 1px solid var(--border);
|
||||
}
|
||||
.topbar-title { font-size: 18px; font-weight: 700; letter-spacing: -0.01em; }
|
||||
|
||||
.topbar-left { display: flex; align-items: center; gap: 8px; min-width: 0; }
|
||||
.topbar-title {
|
||||
font-size: 18px; font-weight: 700; letter-spacing: -0.01em;
|
||||
overflow: hidden; text-overflow: ellipsis; white-space: nowrap; min-width: 0;
|
||||
}
|
||||
.open-pill {
|
||||
display: inline-flex; align-items: center; gap: 6px;
|
||||
padding: 3px 10px; border-radius: 999px;
|
||||
@@ -196,10 +238,12 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
||||
.open-pill.has-triggered::before { background: var(--crit); }
|
||||
.open-pill.all-acked::before { background: var(--warn); }
|
||||
|
||||
/* Bottom tab bar on phones. */
|
||||
/* Bottom tab bar on phones (Queue, On-call, Alerts, Team, More); becomes the
|
||||
left sidebar from 900px, where the desktop block below redeclares display
|
||||
and shows every section flat, .nav-link-secondary included. */
|
||||
.nav {
|
||||
display: flex;
|
||||
position: fixed; left: 0; right: 0; bottom: 0; z-index: 20;
|
||||
display: grid; grid-template-columns: repeat(4, 1fr);
|
||||
height: calc(var(--tabbar-h) + var(--safe-bottom));
|
||||
padding-bottom: var(--safe-bottom);
|
||||
background: color-mix(in srgb, var(--surface) 92%, transparent);
|
||||
@@ -208,12 +252,29 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
||||
border-top: 1px solid var(--border);
|
||||
}
|
||||
.nav-brand { display: none; }
|
||||
/* Hidden here (shown from 900px below): on the phone bar the team switcher
|
||||
lives in the topbar instead, as #team-selector-mobile. */
|
||||
.nav-team-selector { display: none; }
|
||||
.nav-link {
|
||||
position: relative;
|
||||
/* flex: 1 spreads the tabs evenly across the bar's width; the desktop
|
||||
block below cancels it back to a natural-width row item. */
|
||||
flex: 1 1 0;
|
||||
display: flex; flex-direction: column; align-items: center; justify-content: center; gap: 2px;
|
||||
color: var(--faint); font-size: 11px; font-weight: 600;
|
||||
/* min-width lets a column shrink below its label's natural width, which is
|
||||
what stops six tabs widening the bar past the screen. */
|
||||
min-width: 0; padding: 0 2px;
|
||||
}
|
||||
.nav-link svg { width: 24px; height: 24px; fill: none; stroke: currentColor; stroke-width: 1.8; stroke-linecap: round; stroke-linejoin: round; }
|
||||
/* Must come after .nav-link above: same specificity (one class each), so
|
||||
whichever is later in the file wins for an element wearing both classes,
|
||||
and this needs to beat .nav-link's display:flex here on the phone bar. */
|
||||
.nav-link-secondary { display: none; }
|
||||
.nav-label {
|
||||
max-width: 100%; overflow: hidden; text-overflow: ellipsis; white-space: nowrap;
|
||||
}
|
||||
.nav-link svg { width: 24px; height: 24px; flex: none; fill: none; stroke: currentColor; stroke-width: 1.8; stroke-linecap: round; stroke-linejoin: round; }
|
||||
|
||||
.nav-link[aria-current="page"] { color: var(--accent); }
|
||||
.nav-badge {
|
||||
position: absolute; top: 6px; left: calc(50% + 6px);
|
||||
@@ -222,6 +283,34 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
||||
font-size: 11px; font-weight: 700; line-height: 18px; text-align: center;
|
||||
}
|
||||
|
||||
/* ---------- team selector ---------- */
|
||||
/* The global control for which team the app is scoped to. Hidden (via the
|
||||
`hidden` attribute, set from teamselector.js) for anybody in fewer than two
|
||||
teams, the same rule every other team-aware control in this file follows. */
|
||||
|
||||
.nav-team-selector,
|
||||
.team-selector-mobile {
|
||||
display: inline-flex; align-items: center; gap: 8px;
|
||||
border: 1px solid var(--border-strong); border-radius: 999px;
|
||||
background: var(--surface); color: var(--text);
|
||||
font-size: 13px; font-weight: 600; cursor: pointer;
|
||||
padding: 4px 12px; max-width: 100%;
|
||||
}
|
||||
.team-selector-mobile { padding: 4px 10px; font-size: 12px; max-width: 120px; }
|
||||
.team-selector-label { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
|
||||
.team-selector-chevron { width: 14px; height: 14px; flex: none; color: var(--faint); margin-left: -2px; }
|
||||
|
||||
/* A team's identity colour — not a status, so never the severity palette. Six
|
||||
colours, then they repeat; teamColorClass() in format.js picks one by the
|
||||
team's id, the same rcN convention the rota's per-person chips use. */
|
||||
.team-dot { flex: none; width: 8px; height: 8px; border-radius: 50%; background: var(--border-strong); }
|
||||
.team-dot.rc1 { background: var(--accent); }
|
||||
.team-dot.rc2 { background: var(--ok); }
|
||||
.team-dot.rc3 { background: var(--snooze); }
|
||||
.team-dot.rc4 { background: var(--warn); }
|
||||
.team-dot.rc5 { background: var(--teal); }
|
||||
.team-dot.rc6 { background: var(--pink); }
|
||||
|
||||
.view { padding-bottom: calc(var(--tabbar-h) + var(--safe-bottom)); }
|
||||
.view-page { padding-left: 16px; padding-right: 16px; }
|
||||
.view-page > * { max-width: 760px; margin-left: auto; margin-right: auto; }
|
||||
@@ -237,12 +326,14 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
||||
/* ---------- chips ---------- */
|
||||
|
||||
.chips {
|
||||
position: relative;
|
||||
display: flex; gap: 6px;
|
||||
padding: 12px 16px 8px;
|
||||
overflow-x: auto; scrollbar-width: none;
|
||||
}
|
||||
.chips::-webkit-scrollbar { display: none; }
|
||||
.chip {
|
||||
display: inline-flex; align-items: center; gap: 6px;
|
||||
flex: none;
|
||||
min-height: 34px; padding: 0 12px;
|
||||
border: 1px solid var(--border-strong); border-radius: 999px;
|
||||
@@ -251,6 +342,17 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
||||
}
|
||||
.chip[aria-selected="true"] { background: var(--text); border-color: var(--text); color: var(--bg); }
|
||||
.chip .count { margin-left: 4px; opacity: 0.7; }
|
||||
/* An overlay, not a flex item: absolute against .chips' own (non-scrolling)
|
||||
box stays flush with its real right edge regardless of scroll position,
|
||||
which turned out not to be true of position:sticky here — as a flex
|
||||
item, its sticky offset interacted with the row's gap and its own
|
||||
negative margin, landing short of the edge by about one gap's width. */
|
||||
.chips-fade {
|
||||
position: absolute; top: 0; right: 0; bottom: 0;
|
||||
width: 24px;
|
||||
background: linear-gradient(to right, transparent, var(--bg));
|
||||
pointer-events: none;
|
||||
}
|
||||
|
||||
/* ---------- lists ---------- */
|
||||
|
||||
@@ -285,6 +387,26 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
||||
min-width: 0;
|
||||
}
|
||||
.row-meta .labels { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; min-width: 0; max-width: 100%; color: var(--faint); }
|
||||
/* Which team's queue a row came from. Only rendered for somebody in more than
|
||||
one team, so it never repeats the same word down the whole list. */
|
||||
/* Separates the status chips from the team chips in the queue's filter row. */
|
||||
.chip-sep { width: 1px; align-self: stretch; background: var(--border); margin: 0 2px; }
|
||||
|
||||
.row-team {
|
||||
padding: 1px 6px; border-radius: 4px;
|
||||
background: var(--surface-2); border: 1px solid var(--border);
|
||||
color: var(--muted); font-size: 12px; white-space: nowrap;
|
||||
}
|
||||
/* The page a terminal's sign-in prompt links to. */
|
||||
.device-card { max-width: 420px; margin: 24px auto; padding: 20px; display: grid; gap: 14px; }
|
||||
.device-code {
|
||||
margin: 0; padding: 14px; text-align: center;
|
||||
font: 700 30px/1 ui-monospace, SFMono-Regular, Menlo, Consolas, monospace;
|
||||
letter-spacing: 0.12em;
|
||||
background: var(--surface-2); border: 1px solid var(--border); border-radius: var(--radius-sm);
|
||||
}
|
||||
/* Access the identity provider's groups grant. The tint says "not yours to edit here". */
|
||||
.row-team.sso { margin-left: 6px; background: var(--accent-soft); border-color: transparent; color: var(--accent); }
|
||||
.row.resolved .row-title { color: var(--muted); }
|
||||
|
||||
.sev-critical { --sev: var(--crit); }
|
||||
@@ -294,7 +416,7 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
||||
.empty {
|
||||
padding: 48px 16px; text-align: center; color: var(--muted);
|
||||
}
|
||||
.empty strong { display: block; color: var(--text); font-size: 16px; margin-bottom: 4px; }
|
||||
.empty strong { display: block; color: var(--text); font-size: var(--fs-lg); margin-bottom: 4px; }
|
||||
.empty .icon { width: 36px; height: 36px; color: var(--ok); margin-bottom: 8px; }
|
||||
|
||||
.load-error {
|
||||
@@ -317,7 +439,16 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
||||
.badge.st-triggered, .badge.st-firing { background: var(--crit-soft); color: var(--crit); }
|
||||
.badge.st-acknowledged { background: var(--warn-soft); color: var(--warn); }
|
||||
.badge.st-snoozed { background: var(--snooze-soft); color: var(--snooze); }
|
||||
.badge.st-resolved { background: var(--ok-soft); color: var(--ok); }
|
||||
.badge.st-resolved, .badge.st-healthy { background: var(--ok-soft); color: var(--ok); }
|
||||
.badge.st-dead { background: var(--crit-soft); color: var(--crit); }
|
||||
/* Dormant is the plain badge on purpose: nothing has gone wrong and nothing has
|
||||
gone right, which is what the muted default already says. */
|
||||
.badge.st-dormant, .badge.st-never { background: var(--surface-2); color: var(--muted); }
|
||||
.badge.st-active { background: var(--ok-soft); color: var(--ok); }
|
||||
.badge.st-quiet, .badge.st-escalating { background: var(--warn-soft); color: var(--warn); }
|
||||
.badge.st-ready, .badge.st-oncall { background: var(--ok-soft); color: var(--ok); }
|
||||
.badge.st-reachable { background: var(--surface-2); color: var(--muted); }
|
||||
.badge.st-unreachable, .badge.st-unpageable { background: var(--crit-soft); color: var(--crit); }
|
||||
.badge.sev-critical { background: var(--crit-soft); color: var(--crit); }
|
||||
.badge.sev-warning { background: var(--warn-soft); color: var(--warn); }
|
||||
.badge.sev-info { background: var(--info-soft); color: var(--info); }
|
||||
@@ -335,9 +466,15 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
||||
-webkit-backdrop-filter: saturate(1.4) blur(12px);
|
||||
border-bottom: 1px solid var(--border);
|
||||
}
|
||||
.detail-head .copy { margin-left: auto; }
|
||||
.clip-buffer { position: fixed; top: 0; left: 0; opacity: 0; pointer-events: none; }
|
||||
.detail-head .crumb { font-weight: 600; color: var(--muted); font-size: 14px; }
|
||||
.detail-title { font-size: 21px; font-weight: 750; letter-spacing: -0.01em; margin: 16px 0 8px; overflow-wrap: anywhere; }
|
||||
.detail-title { font-size: var(--fs-xl); font-weight: 750; letter-spacing: -0.01em; margin: 16px 0 8px; overflow-wrap: anywhere; }
|
||||
.detail-badges { display: flex; flex-wrap: wrap; gap: 6px; margin-bottom: 14px; }
|
||||
/* A copy of the sticky actionbar's primary button, right under the status
|
||||
it responds to — see quickActions() in incident.js. */
|
||||
.detail-quick-actions { margin-bottom: 14px; }
|
||||
.detail-quick-actions .btn-primary { font-size: 16px; min-height: 44px; }
|
||||
|
||||
.card {
|
||||
background: var(--surface);
|
||||
@@ -352,6 +489,9 @@ input:focus, textarea:focus { outline: none; border-color: var(--accent); box-sh
|
||||
.facts dt { color: var(--muted); }
|
||||
.facts dd { margin: 0; overflow-wrap: anywhere; }
|
||||
.facts .sub { color: var(--faint); }
|
||||
.facts dd.fact-summary { display: flex; flex-wrap: wrap; align-items: center; gap: 8px; }
|
||||
.fact-chip { display: inline-flex; align-items: center; gap: 4px; }
|
||||
.fact-icon { width: 15px; height: 15px; color: var(--faint); }
|
||||
|
||||
.section { margin-top: 22px; }
|
||||
.section-title {
|
||||
@@ -383,6 +523,17 @@ details > summary::before { content: "▸ "; }
|
||||
details[open] > summary::before { content: "▾ "; }
|
||||
details[open] > summary { margin-bottom: 8px; }
|
||||
|
||||
/* One .tl-phase per status the incident has been through (see
|
||||
timelinePhases() in incident.js) — each with its own .timeline <ol>, so
|
||||
the existing :first-child/:last-child rail-capping below gives each phase
|
||||
its own self-contained connecting line rather than one running through
|
||||
the headings. */
|
||||
.tl-phase + .tl-phase { border-top: 1px solid var(--border); }
|
||||
.tl-phase-title {
|
||||
padding: 10px 14px 0;
|
||||
font-size: 11px; font-weight: 700; text-transform: uppercase; letter-spacing: 0.06em;
|
||||
color: var(--faint);
|
||||
}
|
||||
.timeline { list-style: none; margin: 0; padding: 4px 0; }
|
||||
.tl-item {
|
||||
position: relative;
|
||||
@@ -408,11 +559,16 @@ details[open] > summary { margin-bottom: 8px; }
|
||||
.tl-text { overflow-wrap: anywhere; }
|
||||
.tl-text .who { font-weight: 650; }
|
||||
.tl-time { color: var(--faint); font-size: 12px; }
|
||||
.tl-note .note {
|
||||
.note {
|
||||
margin-top: 6px; padding: 10px 12px;
|
||||
background: var(--surface-2); border-radius: var(--radius-sm);
|
||||
white-space: pre-wrap; overflow-wrap: anywhere;
|
||||
}
|
||||
.note-fix { background: var(--ok-soft); border-left: 3px solid var(--ok); }
|
||||
.similar { list-style: none; margin: 0; padding: 0; }
|
||||
.similar-item { padding: 10px 14px; }
|
||||
.similar-item + .similar-item { border-top: 1px solid var(--border, var(--surface-2)); }
|
||||
.check { display: flex; align-items: center; gap: 8px; font-size: 14px; color: var(--muted); }
|
||||
.note-actions { display: flex; justify-content: flex-end; }
|
||||
.note-actions .btn { color: var(--muted); }
|
||||
|
||||
@@ -449,7 +605,7 @@ details[open] > summary { margin-bottom: 8px; }
|
||||
@keyframes sheet-up { from { transform: translateY(24px); opacity: 0.6; } }
|
||||
.sheet-inner { padding: 8px 16px calc(16px + var(--safe-bottom)); }
|
||||
.sheet-grab { width: 40px; height: 4px; margin: 0 auto 12px; border-radius: 2px; background: var(--border-strong); }
|
||||
.sheet-title { font-size: 17px; font-weight: 700; margin: 0 0 4px; }
|
||||
.sheet-title { font-size: var(--fs-lg); font-weight: 700; margin: 0 0 4px; }
|
||||
.sheet-text { color: var(--muted); margin: 0 0 14px; font-size: 14px; }
|
||||
.sheet-form { display: grid; gap: 12px; }
|
||||
.sheet-actions { display: flex; gap: 8px; margin-top: 16px; }
|
||||
@@ -501,9 +657,18 @@ details[open] > summary { margin-bottom: 8px; }
|
||||
.now-label { color: var(--muted); font-size: 13px; font-weight: 600; }
|
||||
.now-name { font-size: 20px; font-weight: 750; }
|
||||
.you { color: var(--accent); font-weight: 650; font-size: 13px; margin-left: 6px; }
|
||||
/* Oncall's own "you" indicator only (see you() in oncall.js) — a pill badge
|
||||
is easier to spot there than this plain accent-coloured text. */
|
||||
.you-badge { margin-left: 6px; }
|
||||
|
||||
.week-nav { display: flex; align-items: center; gap: 4px; }
|
||||
.week-nav .label { font-size: 14px; font-weight: 650; min-width: 9em; text-align: center; }
|
||||
/* The week-nav button that shows the date range, "28 Sep – 4 Oct", with the
|
||||
ISO week number as secondary text inside it — a separate class from
|
||||
.label above (team.js's month-nav uses that one) so its <small> isn't
|
||||
caught by the unrelated .label > span styling meant for label chips. */
|
||||
.week-nav .week-label { font-size: 14px; font-weight: 650; min-width: 11.5em; text-align: center; white-space: nowrap; }
|
||||
.week-label small { color: var(--faint); font-weight: 600; font-size: 11px; margin-left: 2px; }
|
||||
.days { list-style: none; margin: 0; padding: 0; }
|
||||
.day { display: grid; grid-template-columns: 3.2em 4.2em 1fr; align-items: center; gap: 8px; min-height: 50px; padding: 0 14px; }
|
||||
.day + .day { border-top: 1px solid var(--border); }
|
||||
@@ -511,11 +676,18 @@ details[open] > summary { margin-bottom: 8px; }
|
||||
.day-date { color: var(--faint); font-size: 13px; }
|
||||
.day-who { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
|
||||
.day-who.nobody { color: var(--faint); font-style: italic; }
|
||||
/* A run of several days held by the same person (or left empty), replacing
|
||||
what used to be one identical row per day — see weekRuns() in oncall.js. */
|
||||
.day.range { grid-template-columns: 1fr auto; }
|
||||
.day-range { font-weight: 650; }
|
||||
.day.today { background: var(--accent-soft); }
|
||||
.day.today:first-child { border-radius: var(--radius) var(--radius) 0 0; }
|
||||
.day.today:last-child { border-radius: 0 0 var(--radius) var(--radius); }
|
||||
.day.today .day-name { color: var(--accent); }
|
||||
.day.today .day-name, .day.today .day-range { color: var(--accent); }
|
||||
.day.past { opacity: 0.6; }
|
||||
/* Highlights whichever row is yours, same soft tint as .today — they already
|
||||
read fine layered (today's own row is almost always one of yours too). */
|
||||
.day.mine { background: var(--accent-soft); }
|
||||
|
||||
.shift-list { list-style: none; margin: 0; padding: 0; }
|
||||
.shift-list li { display: flex; justify-content: space-between; padding: 12px 14px; }
|
||||
@@ -536,6 +708,8 @@ details[open] > summary { margin-bottom: 8px; }
|
||||
.account-name { font-size: 18px; font-weight: 750; }
|
||||
.account-email { color: var(--muted); font-size: 14px; overflow-wrap: anywhere; }
|
||||
.pw-form { display: grid; gap: 12px; padding: 16px; }
|
||||
.account-fold { border-top: 1px solid var(--border); padding-top: 12px; }
|
||||
.account-fold .stacked-form { margin-top: 8px; }
|
||||
.form-ok {
|
||||
margin: 0; padding: 10px 12px;
|
||||
background: var(--ok-soft); color: var(--ok);
|
||||
@@ -556,8 +730,6 @@ kbd {
|
||||
/* ---------- desktop ---------- */
|
||||
|
||||
@media (min-width: 900px) {
|
||||
:root { --tabbar-h: 0px; }
|
||||
|
||||
.app { display: grid; grid-template-columns: 220px 1fr; height: 100dvh; }
|
||||
|
||||
.nav {
|
||||
@@ -572,8 +744,11 @@ kbd {
|
||||
display: flex; align-items: center; gap: 10px;
|
||||
padding: 4px 10px 18px; font-size: 18px; font-weight: 750; letter-spacing: -0.01em;
|
||||
}
|
||||
.nav-team-selector { display: inline-flex; margin: -8px 10px 14px; width: calc(100% - 20px); }
|
||||
.nav-link-secondary { display: flex; }
|
||||
.nav-more-btn { display: none; }
|
||||
.nav-link {
|
||||
flex-direction: row; justify-content: flex-start; gap: 12px;
|
||||
flex: none; flex-direction: row; justify-content: flex-start; gap: 12px;
|
||||
min-height: 40px; padding: 0 10px; border-radius: var(--radius-sm);
|
||||
color: var(--muted); font-size: 14px;
|
||||
}
|
||||
@@ -593,6 +768,11 @@ kbd {
|
||||
.view-queue .pane { overflow: auto; height: 100dvh; }
|
||||
.pane-list { border-right: 1px solid var(--border); }
|
||||
.pane-list .chips { position: sticky; top: 0; z-index: 2; background: var(--bg); padding-top: 16px; }
|
||||
/* The pane is 340-420px wide and a mouse cannot scroll a row whose scrollbar
|
||||
is hidden, so the chips wrap here instead: Archived stays reachable. */
|
||||
.pane-list .chips { flex-wrap: wrap; overflow-x: visible; }
|
||||
.pane-list .chip-sep { display: none; }
|
||||
.chips-fade { display: none; }
|
||||
.view-queue:not(.has-detail) .pane-detail { display: block; }
|
||||
|
||||
/* On desktop the list stays visible next to the detail. */
|
||||
@@ -620,3 +800,279 @@ kbd {
|
||||
.toast, .app.detail-open ~ .toast { bottom: 24px; }
|
||||
.only-desktop { display: block; }
|
||||
}
|
||||
|
||||
/* --- admin ---------------------------------------------------------------
|
||||
The admin page is three tables of things you act on, so it needs table
|
||||
styling the rest of the app never did: the queue is a list of links and the
|
||||
account page is a form. */
|
||||
.admin-table { width: 100%; border-collapse: collapse; font-size: 14px; }
|
||||
.admin-table th {
|
||||
text-align: left; font-weight: 600; color: var(--muted); font-size: 12px;
|
||||
text-transform: uppercase; letter-spacing: 0.04em;
|
||||
padding: 4px 8px 4px 0; border-bottom: 1px solid var(--border);
|
||||
}
|
||||
.admin-table td { padding: 8px 8px 8px 0; border-bottom: 1px solid var(--border); vertical-align: middle; }
|
||||
.admin-table tr:last-child td { border-bottom: none; }
|
||||
.admin-table .num { text-align: right; font-variant-numeric: tabular-nums; }
|
||||
.admin-table td .btn-sm + .btn-sm { margin-left: 6px; }
|
||||
/* A disabled account stays readable — it is still the name on old
|
||||
acknowledgements — but should not look like a working one. */
|
||||
.disabled-row td { opacity: 0.55; }
|
||||
.btn-sm.danger { color: var(--crit); border-color: var(--crit-soft); }
|
||||
|
||||
/* --- status lists: alert sources and dead man's switches -----------------
|
||||
Six columns do not fit a phone, so the table scrolls inside its card rather
|
||||
than the page. A heartbeat under a switch with several is indented, the way
|
||||
the escalation ladder indents its levels. */
|
||||
.card-head { display: flex; align-items: center; justify-content: space-between; gap: 12px; flex-wrap: wrap; }
|
||||
.table-scroll { overflow-x: auto; margin-top: 12px; }
|
||||
.status-table th, .status-table td { white-space: nowrap; }
|
||||
.status-table td.wrap { white-space: normal; min-width: 12em; }
|
||||
.status-table .source-row td { border-bottom-style: dashed; }
|
||||
.status-table .source-row td:first-child { padding-left: 16px; }
|
||||
.source-labels { display: flex; flex-wrap: wrap; gap: 4px; align-items: center; }
|
||||
|
||||
.inline-form { display: flex; gap: 8px; margin-top: 12px; }
|
||||
.inline-form input { flex: 1; min-width: 0; }
|
||||
|
||||
.admin-settings .setting-value { width: 5.5em; margin-right: 6px; }
|
||||
.admin-settings .setting-unit { max-width: 8em; }
|
||||
.admin-settings button[type="submit"] { margin-top: 12px; }
|
||||
.small { font-size: 13px; }
|
||||
|
||||
/* A name in an admin table is the way to that row's own page -- a person's or
|
||||
a team's. */
|
||||
.row-link { color: var(--text); font-weight: 650; text-decoration: none; }
|
||||
.row-link:hover { color: var(--accent); text-decoration: underline; }
|
||||
|
||||
.invite-block { margin-top: 20px; border-top: 1px solid var(--border); padding-top: 12px; }
|
||||
.invite-block h3 { margin: 0 0 4px; font-size: 14px; }
|
||||
/* The link is shown once and never stored, so it has to be selectable and
|
||||
wrap rather than scroll off the side of a phone. */
|
||||
.invite-out { margin-top: 12px; font-size: 13px; }
|
||||
.invite-link {
|
||||
display: block; margin-top: 6px; padding: 8px; border-radius: var(--radius-sm);
|
||||
background: var(--surface-2); font-family: var(--mono); font-size: 12px;
|
||||
word-break: break-all; user-select: all;
|
||||
}
|
||||
|
||||
/* --- one user, one team --------------------------------------------------
|
||||
Both subject pages share this: .user-head and .user-facts are generic
|
||||
despite the names, and a team fills them with its own facts. */
|
||||
.back-link {
|
||||
display: inline-flex; align-items: center; gap: 2px; margin-bottom: 12px;
|
||||
color: var(--muted); font-size: 14px; text-decoration: none;
|
||||
}
|
||||
.back-link:hover { color: var(--text); }
|
||||
.back-link svg { width: 18px; height: 18px; }
|
||||
|
||||
.user-head { display: flex; align-items: center; flex-wrap: wrap; gap: 8px; }
|
||||
.user-head h2 { margin: 0; }
|
||||
|
||||
.user-facts {
|
||||
display: grid; grid-template-columns: max-content 1fr; gap: 4px 16px;
|
||||
margin: 12px 0 0; font-size: 14px;
|
||||
}
|
||||
.user-facts dt { color: var(--muted); }
|
||||
.user-facts dd { margin: 0; overflow-wrap: anywhere; }
|
||||
|
||||
.row-actions { display: flex; flex-wrap: wrap; gap: 8px; margin-top: 16px; }
|
||||
.admin-table .row-actions { margin-top: 0; gap: 6px; }
|
||||
|
||||
/* --- team settings -------------------------------------------------------
|
||||
Forms with a label above each control, rather than the queue's rows of
|
||||
links. The escalation ladder is the only nested structure in the app, so it
|
||||
gets a little indentation to make the levels read as an order. */
|
||||
.stacked-form { display: flex; flex-direction: column; gap: 10px; margin-top: 12px; align-items: flex-start; }
|
||||
.stacked-form label { display: flex; align-items: center; gap: 6px; flex-wrap: wrap; font-size: 14px; }
|
||||
.stacked-form label.checkbox { gap: 8px; }
|
||||
.stacked-form input.wide { min-width: min(420px, 100%); }
|
||||
|
||||
/* The rota, a month at a time. A name is too wide to print thirty times and
|
||||
too alike down a column to read, so a day carries an initial in that
|
||||
person's colour and the legend underneath says whose. A shift is then a run
|
||||
of one colour, which is the shape the question actually has. */
|
||||
.rota-grid { display: grid; grid-template-columns: 2.4em repeat(7, 1fr); gap: 2px; padding: 10px; }
|
||||
.rota-wd {
|
||||
padding-bottom: 4px; text-align: center;
|
||||
color: var(--muted); font-size: 11px; font-weight: 700;
|
||||
text-transform: uppercase; letter-spacing: 0.04em;
|
||||
}
|
||||
.rota-day {
|
||||
display: flex; flex-direction: column; align-items: center; gap: 4px;
|
||||
min-height: 52px; padding: 6px 0 8px;
|
||||
border: 0; border-radius: var(--radius-sm); background: none;
|
||||
font: inherit; color: inherit;
|
||||
}
|
||||
button.rota-day { cursor: pointer; }
|
||||
button.rota-day:hover { background: var(--surface-2); }
|
||||
/* The week number starts each row. Quiet by default, because it is a label
|
||||
first; an owner's tap on it is the second thing it does. */
|
||||
.rota-week {
|
||||
display: grid; place-items: center;
|
||||
border: 0; border-radius: var(--radius-sm); background: none;
|
||||
font: inherit; font-size: 12px; font-variant-numeric: tabular-nums;
|
||||
color: var(--faint);
|
||||
}
|
||||
button.rota-week { cursor: pointer; }
|
||||
button.rota-week:hover { background: var(--surface-2); color: var(--text); }
|
||||
.rota-week.current { color: var(--accent); font-weight: 700; }
|
||||
/* The sheet's row of who holds each day of the week. */
|
||||
.week-holders { display: flex; justify-content: space-between; gap: 4px; margin: 4px 0 12px; }
|
||||
.week-holder { display: flex; flex-direction: column; align-items: center; gap: 4px; flex: 1; }
|
||||
.week-holder.past { opacity: 0.55; }
|
||||
.rota-num { color: var(--muted); font-size: 12px; font-variant-numeric: tabular-nums; }
|
||||
.rota-day.today { background: var(--accent-soft); }
|
||||
.rota-day.today .rota-num { color: var(--accent); font-weight: 700; }
|
||||
.rota-day.past { opacity: 0.55; }
|
||||
/* The days either side of the month are real days and are drawn, but they
|
||||
belong to the month you are not looking at. */
|
||||
.rota-day.outside { opacity: 0.35; }
|
||||
|
||||
.rota-chip {
|
||||
display: grid; place-items: center;
|
||||
width: 26px; height: 26px; border-radius: 50%;
|
||||
font-size: 12px; font-weight: 750; text-transform: uppercase;
|
||||
}
|
||||
/* An empty day is a dot rather than a hole, and keeps the chip's box so the
|
||||
rows stay on one baseline. */
|
||||
.rota-chip.none { width: 8px; height: 8px; margin: 9px; background: var(--border-strong); }
|
||||
|
||||
/* Six colours, then they repeat; the initial inside still tells two people
|
||||
apart. Deliberately not the severity palette — nothing here is critical. */
|
||||
.rc1 { background: var(--accent-soft); color: var(--accent); }
|
||||
.rc2 { background: var(--ok-soft); color: var(--ok); }
|
||||
.rc3 { background: var(--snooze-soft); color: var(--snooze); }
|
||||
.rc4 { background: var(--warn-soft); color: var(--warn); }
|
||||
.rc5 { background: var(--teal-soft); color: var(--teal); }
|
||||
.rc6 { background: var(--pink-soft); color: var(--pink); }
|
||||
|
||||
.rota-foot { padding: 12px 14px; border-top: 1px solid var(--border); }
|
||||
.rota-legend { display: flex; flex-wrap: wrap; align-items: center; gap: 6px 14px; font-size: 14px; }
|
||||
.rota-key { display: inline-flex; align-items: center; gap: 6px; }
|
||||
.rota-key .rota-chip { width: 22px; height: 22px; font-size: 11px; }
|
||||
.rota-note { margin: 10px 0 0; color: var(--muted); font-size: 13px; }
|
||||
.rota-note:first-child { margin-top: 0; }
|
||||
.rota-bulk { padding: 12px 14px; border-top: 1px solid var(--border); }
|
||||
.rota-bulk .stacked-form { margin-top: 4px; }
|
||||
.sheet-pick { display: flex; align-items: center; gap: 8px; font-size: 14px; }
|
||||
|
||||
.ladder-level {
|
||||
border-left: 3px solid var(--border-strong);
|
||||
padding: 8px 0 8px 12px; margin: 12px 0;
|
||||
}
|
||||
.ladder-head { display: flex; align-items: center; gap: 10px; margin-bottom: 6px; }
|
||||
.ladder-targets { display: flex; flex-direction: column; gap: 6px; margin-top: 8px; }
|
||||
.target-row { display: flex; gap: 6px; align-items: center; flex-wrap: wrap; }
|
||||
.ladder-editor { display: flex; flex-direction: column; gap: 10px; align-items: flex-start; margin-top: 12px; }
|
||||
/* A target that would not wake anybody says why, in place: it is the reason a
|
||||
level is red, and the thing to go and fix. */
|
||||
.target-line { display: flex; gap: 8px; align-items: baseline; flex-wrap: wrap; }
|
||||
.target-problem { color: var(--crit); font-size: 12px; font-weight: 600; }
|
||||
|
||||
/* An integration key is shown exactly once, so it should look like something
|
||||
to act on rather than another row of text. */
|
||||
.key-panel {
|
||||
margin-top: 12px; padding: 12px;
|
||||
border: 1px solid var(--accent); border-radius: 8px; background: var(--accent-soft);
|
||||
}
|
||||
.key-panel pre {
|
||||
overflow-x: auto; background: var(--surface); border: 1px solid var(--border);
|
||||
border-radius: 6px; padding: 8px; font-size: 12px;
|
||||
}
|
||||
.key-url code { word-break: break-all; }
|
||||
|
||||
/* --- onboarding checklist ------------------------------------------------
|
||||
Sits above the queue until it is finished or hidden. Deliberately plain:
|
||||
it is a list of things to do, not a celebration. */
|
||||
.onboarding { border-left: 3px solid var(--accent); }
|
||||
.onboarding-head { display: flex; align-items: center; gap: 10px; }
|
||||
.onboarding-head h2 { flex: 1; margin: 0; }
|
||||
.checklist { list-style: none; margin: 12px 0 0; padding: 0; display: flex; flex-direction: column; gap: 12px; }
|
||||
.checklist .step { display: flex; gap: 10px; align-items: flex-start; }
|
||||
.checklist .step p { margin: 2px 0 0; }
|
||||
.step-mark {
|
||||
flex: none; width: 20px; height: 20px; border-radius: 50%;
|
||||
border: 1px solid var(--border-strong); color: var(--accent);
|
||||
display: flex; align-items: center; justify-content: center; font-size: 13px;
|
||||
}
|
||||
.step.done .step-mark { border-color: var(--accent); }
|
||||
.step.done > div > strong { color: var(--muted); text-decoration: line-through; }
|
||||
.step-actions { display: flex; gap: 6px; margin-top: 6px; flex-wrap: wrap; }
|
||||
|
||||
.signup-intro { margin: 0 0 4px; font-size: 14px; color: var(--muted); }
|
||||
|
||||
/* --- sub-navigation -------------------------------------------------------
|
||||
A strip of links across the top of every Admin and every Team page, one per
|
||||
sub-section. Deliberately not .chip: chips filter what a page already shows,
|
||||
here and in the queue, and these go somewhere. Same aria-current convention
|
||||
as the tab bar, so the state lives on the attribute rather than in a class.
|
||||
|
||||
Six entries do not fit a phone's width, which is what the horizontal scroll
|
||||
below is for -- the Team tab's strip is the one that needs it. */
|
||||
.subnav {
|
||||
display: flex; gap: 2px;
|
||||
margin: 12px auto 0;
|
||||
border-bottom: 1px solid var(--border);
|
||||
overflow-x: auto; scrollbar-width: none;
|
||||
}
|
||||
.subnav::-webkit-scrollbar { display: none; }
|
||||
.subnav-link {
|
||||
flex: none;
|
||||
padding: 8px 12px; margin-bottom: -1px;
|
||||
border-bottom: 2px solid transparent;
|
||||
color: var(--muted); font-size: 14px; font-weight: 600; white-space: nowrap;
|
||||
}
|
||||
.subnav-link:hover { color: var(--text); }
|
||||
.subnav-link[aria-current="page"] { color: var(--accent); border-bottom-color: var(--accent); }
|
||||
|
||||
/* The overview a tab opens on, at /admin and at /team. The strip above already
|
||||
links to the sections, so these carry the counts, which is the part a menu
|
||||
cannot say. Not named for either tab: both use it, and the one that renamed
|
||||
.user-link to .row-link is the same rename for the same reason. */
|
||||
.overview-menu { display: grid; gap: 10px; margin-top: 16px; }
|
||||
/* The grid's gap is the spacing here, so .card + .card must not add its own. */
|
||||
.overview-menu .card + .card { margin-top: 0; }
|
||||
.overview-item { display: block; padding: 14px; }
|
||||
.overview-item:hover { background: var(--surface-hover); }
|
||||
.overview-head { display: flex; align-items: baseline; gap: 8px; }
|
||||
.overview-count { margin-left: auto; color: var(--muted); font-size: 18px; font-weight: 700; }
|
||||
.overview-item p { margin: 4px 0 0; }
|
||||
.overview-note-icon { width: 13px; height: 13px; vertical-align: -2px; color: var(--ok); }
|
||||
|
||||
/* ---------- stats page ---------- */
|
||||
|
||||
#view-stats .chips { padding-left: 0; padding-right: 0; }
|
||||
.stats { display: grid; gap: 14px; padding-bottom: 16px; }
|
||||
.stat-tiles { display: grid; grid-template-columns: repeat(2, 1fr); gap: 8px; }
|
||||
.stat-tile {
|
||||
--sev: var(--border-strong);
|
||||
background: var(--surface); border: 1px solid var(--border);
|
||||
border-left: 3px solid var(--sev); border-radius: var(--radius);
|
||||
box-shadow: var(--shadow); padding: 12px;
|
||||
}
|
||||
.stat-tile.st-triggered { --sev: var(--crit); }
|
||||
.stat-tile.st-acknowledged { --sev: var(--warn); }
|
||||
.stat-tile.st-resolved { --sev: var(--ok); }
|
||||
.stat-value { font-size: 24px; font-weight: 700; font-variant-numeric: tabular-nums; }
|
||||
.stat-label { margin-top: 2px; font-size: 12px; color: var(--muted); }
|
||||
.chart-card + .chart-card { margin-top: 0; }
|
||||
.chart-title {
|
||||
margin-bottom: 10px; font-size: 13px; font-weight: 700;
|
||||
text-transform: uppercase; letter-spacing: 0.06em; color: var(--muted);
|
||||
}
|
||||
.hbars { list-style: none; display: grid; gap: 6px; }
|
||||
.hbar { display: grid; grid-template-columns: minmax(80px, 34%) 1fr auto; align-items: center; gap: 8px; font-size: 13px; }
|
||||
.hbar-name { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
|
||||
.hbar-track { height: 10px; border-radius: 5px; background: var(--surface-2); overflow: hidden; }
|
||||
.hbar-fill { display: block; height: 100%; border-radius: 5px; background: var(--ok); }
|
||||
.hbar-count { min-width: 2ch; text-align: right; color: var(--muted); font-variant-numeric: tabular-nums; }
|
||||
.columns { display: block; width: 100%; height: auto; }
|
||||
.columns .axis { stroke: var(--border-strong); stroke-width: 1; }
|
||||
.columns .col-hit { fill: transparent; }
|
||||
.columns .col-bar { fill: var(--accent); }
|
||||
.columns .col:hover .col-bar { opacity: 0.75; }
|
||||
.columns .col-label { fill: var(--faint); font-size: 10px; font-family: var(--font); }
|
||||
@media (min-width: 900px) {
|
||||
.stat-tiles { grid-template-columns: repeat(3, 1fr); }
|
||||
}
|
||||
|
||||
@@ -25,18 +25,58 @@
|
||||
<img src="/icon.svg" alt="" width="40" height="40">
|
||||
<h1>terdut</h1>
|
||||
</div>
|
||||
<!-- Why a sign-in failed, when the identity provider sent the browser back
|
||||
here with ?sso_error=. Kept apart from the password form's own error. -->
|
||||
<p class="form-error" id="sso-error" role="alert" hidden></p>
|
||||
<!-- A plain link, not a fetch: the browser has to navigate to the provider,
|
||||
and the page's CSP allows no connection to anywhere else. -->
|
||||
<a class="btn btn-primary btn-block" id="sso-link" href="/api/oidc/login" hidden>Sign in with SSO</a>
|
||||
<p class="login-divider" id="login-or" hidden><span>or</span></p>
|
||||
<div class="login-password" id="password-login">
|
||||
<label>
|
||||
<span>Username</span>
|
||||
<input name="username" autocomplete="username" autocapitalize="none" spellcheck="false" required>
|
||||
</label>
|
||||
<label>
|
||||
<span>Password</span>
|
||||
<input name="password" type="password" autocomplete="current-password" required>
|
||||
</label>
|
||||
<p class="form-error" role="alert" hidden></p>
|
||||
<button class="btn btn-primary btn-block" type="submit">Sign in</button>
|
||||
<p class="login-hint">No password yet? Ask an admin to set one, or run
|
||||
<code>PUT /api/users/{id}/password</code> with your API key.</p>
|
||||
<p class="login-hint" id="signup-link" hidden>
|
||||
No account? <a href="/signup">Create one</a>.</p>
|
||||
</div>
|
||||
</form>
|
||||
|
||||
<!-- Sign-up. Shown instead of the login card at /signup, and only offers
|
||||
what the server allows: an invite link, or open sign-up. -->
|
||||
<form id="signup-form" class="login-card" autocomplete="on" hidden>
|
||||
<div class="login-brand">
|
||||
<img src="/icon.svg" alt="" width="40" height="40">
|
||||
<h1>terdut</h1>
|
||||
</div>
|
||||
<p class="signup-intro" id="signup-intro"></p>
|
||||
<label>
|
||||
<span>Username</span>
|
||||
<input name="username" autocomplete="username" autocapitalize="none" spellcheck="false" required>
|
||||
</label>
|
||||
<label>
|
||||
<span>Email</span>
|
||||
<input name="email" type="email" autocomplete="email" required>
|
||||
</label>
|
||||
<label>
|
||||
<span>Password</span>
|
||||
<input name="password" type="password" autocomplete="current-password" required>
|
||||
<input name="password" type="password" autocomplete="new-password" minlength="10" required>
|
||||
</label>
|
||||
<label id="signup-team-label" hidden>
|
||||
<span>Team name</span>
|
||||
<input name="team_name" autocomplete="off">
|
||||
</label>
|
||||
<p class="form-error" role="alert" hidden></p>
|
||||
<button class="btn btn-primary btn-block" type="submit">Sign in</button>
|
||||
<p class="login-hint">No password yet? Ask an admin to set one, or run
|
||||
<code>PUT /api/users/{id}/password</code> with your API key.</p>
|
||||
<button class="btn btn-primary btn-block" type="submit">Create account</button>
|
||||
<p class="login-hint">Already have one? <a href="/">Sign in</a>.</p>
|
||||
</form>
|
||||
</main>
|
||||
|
||||
@@ -46,27 +86,59 @@
|
||||
<img src="/icon.svg" alt="" width="28" height="28">
|
||||
<span>terdut</span>
|
||||
</a>
|
||||
<a class="nav-link" href="/" data-section="queue">
|
||||
<!-- Which team the app is scoped to. Hidden unless the signed-in user is
|
||||
in more than one; teamselector.js fills it in and wires the click. -->
|
||||
<button class="nav-team-selector" id="team-selector" type="button" hidden></button>
|
||||
<a class="nav-link" href="/" data-section="queue" aria-label="Queue">
|
||||
<svg viewBox="0 0 24 24" aria-hidden="true"><path d="M4 6h16M4 12h16M4 18h10"/></svg>
|
||||
<span class="nav-label">Queue</span>
|
||||
<span class="nav-badge" data-badge hidden></span>
|
||||
</a>
|
||||
<a class="nav-link" href="/oncall" data-section="oncall">
|
||||
<a class="nav-link" href="/oncall" data-section="oncall" aria-label="On-call">
|
||||
<svg viewBox="0 0 24 24" aria-hidden="true"><rect x="3.5" y="5" width="17" height="15" rx="2"/><path d="M3.5 10h17M8 3v4M16 3v4"/></svg>
|
||||
<span class="nav-label">On-call</span>
|
||||
</a>
|
||||
<a class="nav-link" href="/alerts" data-section="alerts">
|
||||
<a class="nav-link" href="/alerts" data-section="alerts" aria-label="Alerts">
|
||||
<svg viewBox="0 0 24 24" aria-hidden="true"><path d="M6 16V11a6 6 0 0 1 12 0v5l1.5 2h-15z"/><path d="M10 20.5a2 2 0 0 0 4 0"/></svg>
|
||||
<span class="nav-label">Alerts</span>
|
||||
</a>
|
||||
<a class="nav-link" href="/more" data-section="more">
|
||||
<!-- Secondary: full-width in the desktop sidebar, folded into the
|
||||
"More" tab's sheet on the phone-width bottom bar instead (see
|
||||
.nav-link-secondary in app.css and openNavMenu in app.js). -->
|
||||
<a class="nav-link nav-link-secondary" href="/stats" data-section="stats" aria-label="Stats">
|
||||
<svg viewBox="0 0 24 24" aria-hidden="true"><path d="M4 20h16M7 20v-7M12 20V6M17 20v-10"/></svg>
|
||||
<span class="nav-label">Stats</span>
|
||||
</a>
|
||||
<a class="nav-link" href="/team" data-section="team" aria-label="Team">
|
||||
<svg viewBox="0 0 24 24" aria-hidden="true"><circle cx="9" cy="8" r="3"/><circle cx="17" cy="9" r="2.5"/><path d="M3 19a6 6 0 0 1 12 0M15 19a5 5 0 0 1 6-4"/></svg>
|
||||
<span class="nav-label">Team</span>
|
||||
</a>
|
||||
<!-- Hidden unless the signed-in user is a system administrator; app.js
|
||||
unhides it once /api/me says so. The server refuses every admin
|
||||
endpoint regardless, so this is a courtesy and not a gate. -->
|
||||
<a class="nav-link nav-link-secondary" href="/admin" data-section="admin" aria-label="Admin" id="nav-admin" hidden>
|
||||
<svg viewBox="0 0 24 24" aria-hidden="true"><path d="M12 3l7 3v6c0 4-3 7-7 9-4-2-7-5-7-9V6z"/></svg>
|
||||
<span class="nav-label">Admin</span>
|
||||
</a>
|
||||
<a class="nav-link nav-link-secondary" href="/more" data-section="more" aria-label="Account">
|
||||
<svg viewBox="0 0 24 24" aria-hidden="true"><circle cx="12" cy="8" r="3.5"/><path d="M5 20a7 7 0 0 1 14 0"/></svg>
|
||||
<span class="nav-label">Account</span>
|
||||
</a>
|
||||
<!-- Phone-width only (see .nav-more-btn in app.css): opens the same
|
||||
sheet the old hamburger button did, for the sections the bottom
|
||||
bar has no room for. Not shown on the desktop sidebar, which lists
|
||||
every section already. -->
|
||||
<button class="nav-link nav-more-btn" id="nav-more-btn" type="button" aria-label="More sections" aria-haspopup="menu">
|
||||
<svg viewBox="0 0 24 24" aria-hidden="true"><path d="M5 12h.01M12 12h.01M19 12h.01"/></svg>
|
||||
<span class="nav-label">More</span>
|
||||
</button>
|
||||
</nav>
|
||||
|
||||
<header class="topbar">
|
||||
<h1 class="topbar-title" id="topbar-title">Queue</h1>
|
||||
<div class="topbar-left">
|
||||
<button class="team-selector-mobile" id="team-selector-mobile" type="button" hidden></button>
|
||||
<h1 class="topbar-title" id="topbar-title">Queue</h1>
|
||||
</div>
|
||||
<span class="open-pill" id="open-pill" hidden></span>
|
||||
</header>
|
||||
|
||||
@@ -80,7 +152,18 @@
|
||||
|
||||
<section id="view-oncall" class="view view-page" data-view="oncall" hidden></section>
|
||||
<section id="view-alerts" class="view view-page" data-view="alerts" hidden></section>
|
||||
<section id="view-stats" class="view view-page" data-view="stats" hidden></section>
|
||||
<section id="view-team" class="view view-page" data-view="team" hidden></section>
|
||||
<section id="view-admin" class="view view-page" data-view="admin" hidden></section>
|
||||
<!-- One person, at /admin/users/{id}: reached from the Admin tab's user
|
||||
list, and a section of its own so a deep link survives a reload. -->
|
||||
<section id="view-adminuser" class="view view-page" data-view="adminuser" hidden></section>
|
||||
<!-- One team, at /admin/teams/{id}: who is in it and the invites into it,
|
||||
which the Team tab cannot show for a team you are not a member of. -->
|
||||
<section id="view-adminteam" class="view view-page" data-view="adminteam" hidden></section>
|
||||
<section id="view-more" class="view view-page" data-view="more" hidden></section>
|
||||
<!-- Approve a terminal's sign-in, at /device?code=...: the page its prompt links to. -->
|
||||
<section id="view-device" class="view view-page" data-view="device" hidden></section>
|
||||
</div>
|
||||
|
||||
<dialog id="sheet" class="sheet"></dialog>
|
||||
|
||||
@@ -23,8 +23,10 @@ function render() {
|
||||
h('div', { class: 'account-name', text: user.username }),
|
||||
h('div', { class: 'account-email', text: user.email }))),
|
||||
|
||||
h('div', { class: 'page-head' }, h('h2', { text: hasPassword ? 'Change password' : 'Set a password' })),
|
||||
passwordForm(user, hasPassword),
|
||||
h('div', { class: 'page-head' }, h('h2', { text: 'Notifications' })),
|
||||
notifyForm(user),
|
||||
|
||||
...passwordSection(user, hasPassword),
|
||||
|
||||
h('div', { class: 'only-desktop' },
|
||||
h('div', { class: 'page-head' }, h('h2', { text: 'Keyboard' })),
|
||||
@@ -32,13 +34,116 @@ function render() {
|
||||
|
||||
h('div', { class: 'page-head' }),
|
||||
h('button', { class: 'btn btn-block', type: 'button', onclick: signOut }, icon('logout'), 'Sign out'),
|
||||
h('p', { class: 'foot-note', text: 'Schedule editing, statistics and user management are in terdut-tui for now.' }),
|
||||
);
|
||||
}
|
||||
|
||||
// Where this user's pages go. The onboarding checklist's first step sends
|
||||
// people here for it, and until now there was nothing here to send them to:
|
||||
// the topic could only be set with curl or by an administrator.
|
||||
//
|
||||
// The topic is the whole address — the server it is published to is the
|
||||
// install's one ntfy, set in the deployment and not something a user picks.
|
||||
function notifyStatus(topic) {
|
||||
return topic ? `Topic: ${topic}` : 'No topic set — pages go to the team’s fallback topic.';
|
||||
}
|
||||
|
||||
function notifyForm(user) {
|
||||
const err = h('p', { class: 'form-error', role: 'alert', hidden: true });
|
||||
const topic = h('input', {
|
||||
name: 'ntfy_topic', type: 'text', autocomplete: 'off',
|
||||
autocapitalize: 'none', spellcheck: false,
|
||||
value: user.ntfy_topic || '',
|
||||
placeholder: 'terdut-a7f3c91e',
|
||||
});
|
||||
const submit = h('button', { class: 'btn btn-primary', type: 'submit', text: 'Save topic' });
|
||||
|
||||
const form = h('form', { class: 'stacked-form' },
|
||||
h('label', {},
|
||||
h('span', { text: 'ntfy topic' }),
|
||||
topic),
|
||||
h('p', { class: 'muted small' },
|
||||
'Subscribe to this topic in the ntfy app and incidents assigned to you ',
|
||||
'reach your phone. Leave it empty and they page the team’s fallback ',
|
||||
'topic instead.'),
|
||||
// Worth saying plainly: people reach for their own name, and the topic is
|
||||
// the only thing standing between a stranger and their pages.
|
||||
h('p', { class: 'muted small' },
|
||||
'Anyone who knows the topic can read your pages and publish to it, so ',
|
||||
'pick something unguessable rather than your name.'),
|
||||
err,
|
||||
submit,
|
||||
);
|
||||
|
||||
const status = h('p', { class: 'muted', text: notifyStatus(user.ntfy_topic) });
|
||||
const summary = h('summary', { text: user.ntfy_topic ? 'Change topic' : 'Set a topic' });
|
||||
const details = h('details', { class: 'account-fold' }, summary, form);
|
||||
|
||||
// Only offered once a topic is saved: the test publishes to whatever the
|
||||
// server has stored, not to whatever is half-typed in the field.
|
||||
const test = h('button', {
|
||||
class: 'btn', type: 'button', text: 'Send a test push',
|
||||
hidden: !user.ntfy_topic,
|
||||
onclick: async () => {
|
||||
test.disabled = true;
|
||||
try {
|
||||
await api.testNotification();
|
||||
toast('Sent. If nothing arrives, the topic is wrong or ntfy is not reachable.');
|
||||
} catch (ex) {
|
||||
toast(ex.message, 'error');
|
||||
} finally {
|
||||
test.disabled = false;
|
||||
}
|
||||
},
|
||||
});
|
||||
|
||||
form.addEventListener('submit', async (e) => {
|
||||
e.preventDefault();
|
||||
err.hidden = true;
|
||||
submit.disabled = true;
|
||||
try {
|
||||
const updated = await api.setNotifyTarget(user.id, topic.value.trim());
|
||||
// Keep the cached user in step, so the onboarding checklist stops
|
||||
// asking for this and the test button appears without a reload.
|
||||
state.me.user = updated;
|
||||
status.textContent = notifyStatus(updated.ntfy_topic);
|
||||
summary.textContent = updated.ntfy_topic ? 'Change topic' : 'Set a topic';
|
||||
test.hidden = !updated.ntfy_topic;
|
||||
details.open = false;
|
||||
toast(updated.ntfy_topic ? 'Topic saved' : 'Topic cleared. Your pages go to the team’s fallback topic.');
|
||||
} catch (ex) {
|
||||
err.textContent = ex.message;
|
||||
err.hidden = false;
|
||||
} finally {
|
||||
submit.disabled = false;
|
||||
}
|
||||
});
|
||||
|
||||
return h('div', { class: 'card pw-form' },
|
||||
status,
|
||||
h('div', { class: 'row-actions' }, test),
|
||||
details);
|
||||
}
|
||||
|
||||
// With password login switched off a password opens nothing, so somebody who
|
||||
// has none is not asked to make one. Somebody who does keeps the form: it is
|
||||
// how they change or get rid of a credential the server still remembers.
|
||||
function passwordSection(user, hasPassword) {
|
||||
if (!hasPassword && state.auth.password_login === false) {
|
||||
const name = state.auth.oidc?.name || 'single sign-on';
|
||||
return [
|
||||
h('div', { class: 'page-head' }, h('h2', { text: 'Password' })),
|
||||
h('div', { class: 'card' },
|
||||
h('p', { class: 'muted', text: `You sign in with ${name}, and this server has turned password login off.` })),
|
||||
];
|
||||
}
|
||||
return [
|
||||
h('div', { class: 'page-head' }, h('h2', { text: 'Password' })),
|
||||
passwordForm(user, hasPassword),
|
||||
];
|
||||
}
|
||||
|
||||
function passwordForm(user, hasPassword) {
|
||||
const err = h('p', { class: 'form-error', role: 'alert', hidden: true });
|
||||
const ok = h('p', { class: 'form-ok', role: 'status', hidden: true });
|
||||
const current = hasPassword
|
||||
? h('input', { name: 'current', type: 'password', autocomplete: 'current-password', required: true })
|
||||
: null;
|
||||
@@ -48,18 +153,24 @@ function passwordForm(user, hasPassword) {
|
||||
|
||||
// A hidden username field lets password managers file the new password
|
||||
// under the right account.
|
||||
const form = h('form', { class: 'card pw-form', autocomplete: 'on' },
|
||||
const form = h('form', { class: 'stacked-form', autocomplete: 'on' },
|
||||
h('input', { type: 'text', name: 'username', autocomplete: 'username', value: user.username, hidden: true, readonly: true }),
|
||||
current && h('label', {}, h('span', { text: 'Current password' }), current),
|
||||
h('label', {}, h('span', { text: 'New password' }), next),
|
||||
h('label', {}, h('span', { text: 'Repeat new password' }), again),
|
||||
err, ok, submit,
|
||||
err, submit,
|
||||
);
|
||||
|
||||
const status = h('p', {
|
||||
class: 'muted',
|
||||
text: hasPassword ? 'Password set.' : 'No password set — sign-in needs one of the other methods.',
|
||||
});
|
||||
const summary = h('summary', { text: hasPassword ? 'Change password' : 'Set a password' });
|
||||
const details = h('details', { class: 'account-fold' }, summary, form);
|
||||
|
||||
form.addEventListener('submit', async (e) => {
|
||||
e.preventDefault();
|
||||
err.hidden = true;
|
||||
ok.hidden = true;
|
||||
if (next.value !== again.value) {
|
||||
err.textContent = 'The new passwords do not match.';
|
||||
err.hidden = false;
|
||||
@@ -71,13 +182,14 @@ function passwordForm(user, hasPassword) {
|
||||
state.me.has_password = true;
|
||||
form.reset();
|
||||
if (!current) {
|
||||
// From now on the form needs the current-password field.
|
||||
// From now on the form needs the current-password field, and a fresh
|
||||
// render already comes up with the fold closed.
|
||||
render();
|
||||
toast('Password saved');
|
||||
return;
|
||||
}
|
||||
ok.textContent = 'Password saved. Other devices have been signed out.';
|
||||
ok.hidden = false;
|
||||
details.open = false;
|
||||
toast('Password saved. Other devices have been signed out.');
|
||||
} catch (ex) {
|
||||
err.textContent = ex.message;
|
||||
err.hidden = false;
|
||||
@@ -85,7 +197,7 @@ function passwordForm(user, hasPassword) {
|
||||
submit.disabled = false;
|
||||
}
|
||||
});
|
||||
return form;
|
||||
return h('div', { class: 'card pw-form' }, status, details);
|
||||
}
|
||||
|
||||
function shortcuts() {
|
||||
|
||||
@@ -0,0 +1,375 @@
|
||||
// Administration: the teams on this server, the people who can sign in, and
|
||||
// the settings that change how the server behaves.
|
||||
//
|
||||
// Each of those three is a route of its own, reached from a strip across the
|
||||
// top, with /admin itself an overview. They used to be three cards stacked on
|
||||
// one page, which meant no way to link to the settings, no way back to the top
|
||||
// of the user list but scrolling, and a poll that refetched all three endpoints
|
||||
// however little of the page you were looking at.
|
||||
//
|
||||
// Only rendered for a system administrator. The server enforces that on every
|
||||
// endpoint regardless — hiding a section is a courtesy to the reader, not a
|
||||
// permission — so this view simply says so rather than pretending to be a
|
||||
// gate.
|
||||
|
||||
import * as api from './api.js';
|
||||
import { h, clear, spinner, confirm, menuCard, ssoBadge, SSO_MANAGED } from './ui.js';
|
||||
import { state, myID } from './state.js';
|
||||
|
||||
const view = () => document.getElementById('view-admin');
|
||||
|
||||
// The sub-sections, in the order the strip shows them. The overview is /admin
|
||||
// itself, so it has no tab of its own. This table is the only place the four
|
||||
// routes are written down: app.js parses against it and the strip is built
|
||||
// from it, so adding a fifth is one line here.
|
||||
export const TABS = [
|
||||
{ tab: null, path: '/admin', label: 'Overview' },
|
||||
{ tab: 'teams', path: '/admin/teams', label: 'Teams' },
|
||||
{ tab: 'users', path: '/admin/users', label: 'Users' },
|
||||
{ tab: 'settings', path: '/admin/settings', label: 'Settings' },
|
||||
];
|
||||
|
||||
// Which sub-section is open. Remembered rather than passed, because the poll
|
||||
// loop calls refresh() with no route — the same reason adminuser.js keeps its
|
||||
// user ID in the module.
|
||||
let tab = null;
|
||||
let data = null; // whatever the current tab needs; the shape varies by tab
|
||||
let error = null;
|
||||
let busy = false;
|
||||
|
||||
export function show(route) {
|
||||
const next = route?.tab ?? null;
|
||||
// A different sub-section wants different data, so the old answer goes
|
||||
// rather than being shown under the new heading until the fetch lands.
|
||||
if (next !== tab) {
|
||||
tab = next;
|
||||
data = null;
|
||||
}
|
||||
if (!data) clear(view(), subnav(), spinner());
|
||||
refresh();
|
||||
}
|
||||
|
||||
export async function refresh() {
|
||||
if (!state.me?.user?.is_admin) {
|
||||
data = null;
|
||||
render();
|
||||
return;
|
||||
}
|
||||
try {
|
||||
data = await load();
|
||||
error = null;
|
||||
} catch (err) {
|
||||
error = err.message;
|
||||
}
|
||||
render();
|
||||
}
|
||||
|
||||
// Only what the open sub-section shows. Users is the one that needs two: it
|
||||
// only points at Teams for an invite if there is a team to point at, and the
|
||||
// overview counts both.
|
||||
async function load() {
|
||||
if (tab === 'teams') return { teams: await api.adminTeams() };
|
||||
if (tab === 'settings') return { settings: await api.adminSettings() };
|
||||
const [teams, users] = await Promise.all([api.adminTeams(), api.users()]);
|
||||
return { teams, users };
|
||||
}
|
||||
|
||||
function render() {
|
||||
if (!state.me?.user?.is_admin) {
|
||||
clear(view(), h('div', { class: 'card' },
|
||||
h('p', { class: 'muted', text: 'Administration is for system administrators. Ask one for access.' })));
|
||||
return;
|
||||
}
|
||||
if (!data) {
|
||||
clear(view(), subnav(), error ? h('div', { class: 'load-error', text: error }) : spinner());
|
||||
return;
|
||||
}
|
||||
clear(view(),
|
||||
subnav(),
|
||||
error && h('div', { class: 'load-error', text: `Showing older data: ${error}` }),
|
||||
section(),
|
||||
);
|
||||
}
|
||||
|
||||
function section() {
|
||||
if (tab === 'teams') return teamsCard();
|
||||
if (tab === 'users') return usersCard();
|
||||
if (tab === 'settings') return settingsCard();
|
||||
return overview();
|
||||
}
|
||||
|
||||
// The strip across the top of every admin page. Ordinary links rather than
|
||||
// buttons, because these are four URLs: app.js intercepts the click, the
|
||||
// browser's Back walks them, and a reload lands where you were.
|
||||
function subnav() {
|
||||
return h('nav', { class: 'subnav', 'aria-label': 'Administration' },
|
||||
TABS.map((t) => h('a', {
|
||||
class: 'subnav-link',
|
||||
href: t.path,
|
||||
text: t.label,
|
||||
'aria-current': t.tab === tab ? 'page' : null,
|
||||
})));
|
||||
}
|
||||
|
||||
// --- overview --------------------------------------------------------------
|
||||
|
||||
// /admin itself. The strip already links to the three, so this earns its place
|
||||
// by saying how much of each there is — the one thing a menu cannot.
|
||||
function overview() {
|
||||
const admins = data.users.filter((u) => u.is_admin).length;
|
||||
const disabled = data.users.filter((u) => u.disabled_at).length;
|
||||
const open = data.teams.reduce((n, t) => n + t.open_incidents, 0);
|
||||
|
||||
const people = [`${admins} ${admins === 1 ? 'administrator' : 'administrators'}`];
|
||||
if (disabled > 0) people.push(`${disabled} disabled`);
|
||||
|
||||
return h('div', { class: 'overview-menu' },
|
||||
menuCard('/admin/teams', 'Teams', data.teams.length,
|
||||
open > 0
|
||||
? `${open} open ${open === 1 ? 'incident' : 'incidents'} between them.`
|
||||
: 'Nothing open anywhere.'),
|
||||
menuCard('/admin/users', 'Users', data.users.length, `${people.join(', ')}.`),
|
||||
menuCard('/admin/settings', 'Settings', null,
|
||||
'How the server behaves, and where it is plugged in.'),
|
||||
);
|
||||
}
|
||||
|
||||
// --- teams -----------------------------------------------------------------
|
||||
|
||||
function teamsCard() {
|
||||
const rows = data.teams.map((t) =>
|
||||
h('tr', {},
|
||||
// The name is the way in: everything about one team lives on its own
|
||||
// page, and this table stays a list rather than becoming a form.
|
||||
h('td', {}, h('a', { class: 'row-link', href: `/admin/teams/${t.id}`, text: t.name })),
|
||||
h('td', { class: 'num', text: String(t.members) }),
|
||||
h('td', { class: 'num', text: String(t.open_incidents) }),
|
||||
));
|
||||
|
||||
return h('div', { class: 'card' },
|
||||
h('h2', { text: 'Teams' }),
|
||||
h('p', { class: 'muted small' },
|
||||
'Open a team for who is in it, the invites into it, and renaming or ',
|
||||
'deleting it. Deleting takes its alerts, incidents, schedule and ',
|
||||
'integrations with it, and is refused while anything is still open.'),
|
||||
h('table', { class: 'admin-table' },
|
||||
h('thead', {}, h('tr', {},
|
||||
h('th', { text: 'Name' }),
|
||||
h('th', { class: 'num', text: 'Members' }),
|
||||
h('th', { class: 'num', text: 'Open' }))),
|
||||
h('tbody', {}, rows)),
|
||||
newTeamForm(),
|
||||
);
|
||||
}
|
||||
|
||||
function newTeamForm() {
|
||||
const name = h('input', { name: 'name', type: 'text', placeholder: 'New team name', required: true });
|
||||
const form = h('form', { class: 'inline-form' }, name,
|
||||
h('button', { class: 'btn', type: 'submit', text: 'Create' }));
|
||||
form.addEventListener('submit', async (e) => {
|
||||
e.preventDefault();
|
||||
if (busy) return;
|
||||
busy = true;
|
||||
try {
|
||||
await api.createTeam(name.value.trim());
|
||||
name.value = '';
|
||||
await refresh();
|
||||
} catch (err) {
|
||||
error = err.message;
|
||||
render();
|
||||
} finally {
|
||||
busy = false;
|
||||
}
|
||||
});
|
||||
return form;
|
||||
}
|
||||
|
||||
// --- users -----------------------------------------------------------------
|
||||
|
||||
function usersCard() {
|
||||
const rows = data.users.map((u) => {
|
||||
const self = u.id === myID();
|
||||
return h('tr', { class: u.disabled_at ? 'disabled-row' : '' },
|
||||
h('td', {},
|
||||
// The name is the way in: everything about one person lives on their
|
||||
// own page, and this table stays a list rather than becoming a form.
|
||||
h('a', { class: 'row-link', href: `/admin/users/${u.id}`, text: u.username }),
|
||||
u.disabled_at && h('span', { class: 'row-team', text: 'disabled' }),
|
||||
self && h('span', { class: 'you', text: 'you' })),
|
||||
h('td', { class: 'muted', text: u.email }),
|
||||
h('td', {},
|
||||
u.is_admin ? h('span', { class: 'row-team', text: 'admin' }) : null,
|
||||
u.is_admin && u.admin_source === 'oidc' ? ssoBadge() : null),
|
||||
h('td', {},
|
||||
// Neither action is offered for your own account: the server refuses
|
||||
// both, and an enabled-looking button that always fails is worse than
|
||||
// no button.
|
||||
!self && h('button', {
|
||||
class: 'btn-sm',
|
||||
type: 'button',
|
||||
text: u.is_admin ? 'Revoke admin' : 'Make admin',
|
||||
// The server refuses to revoke what the groups grant.
|
||||
disabled: ssoAdmin(u),
|
||||
title: ssoAdmin(u) ? SSO_MANAGED : null,
|
||||
onclick: () => setAdmin(u, !u.is_admin),
|
||||
}),
|
||||
!self && h('button', {
|
||||
class: 'btn-sm danger',
|
||||
type: 'button',
|
||||
text: u.disabled_at ? 'Enable' : 'Disable',
|
||||
onclick: () => setDisabled(u, !u.disabled_at),
|
||||
}),
|
||||
),
|
||||
);
|
||||
});
|
||||
|
||||
return h('div', { class: 'card' },
|
||||
h('h2', { text: 'Users' }),
|
||||
h('p', { class: 'muted small' },
|
||||
'Disabling an account stops it signing in and stops its API keys, and keeps ',
|
||||
'its acknowledgements and timeline entries. Deleting a user erases those. ',
|
||||
'Open a name for their teams, their password and the rest.'),
|
||||
h('table', { class: 'admin-table' },
|
||||
h('thead', {}, h('tr', {},
|
||||
h('th', { text: 'User' }),
|
||||
h('th', { text: 'Email' }),
|
||||
h('th', { text: '' }),
|
||||
h('th', { text: '' }))),
|
||||
h('tbody', {}, rows)),
|
||||
invitePointer(),
|
||||
);
|
||||
}
|
||||
|
||||
// Adding a person is minting them an invite into a team, not creating a row:
|
||||
// whoever accepts it picks their own password, so one never passes through an
|
||||
// administrator, and the link carries the team, so they do not land on an empty
|
||||
// queue.
|
||||
//
|
||||
// The form for it lives on the team's own page. It always needed a team beside
|
||||
// it, and a picker here was the admission that an invite is a fact about a team
|
||||
// rather than about the server.
|
||||
function invitePointer() {
|
||||
return h('div', { class: 'invite-block' },
|
||||
h('h3', { text: 'Add someone' }),
|
||||
data.teams.length > 0
|
||||
? h('p', { class: 'muted small' },
|
||||
'Open the team you want them in, under ',
|
||||
h('a', { class: 'row-link', href: '/admin/teams', text: 'Teams' }),
|
||||
', and mint an invite there.')
|
||||
: h('p', { class: 'muted small', text: 'Create a team first — an invite has to lead somewhere.' }),
|
||||
);
|
||||
}
|
||||
|
||||
const ssoAdmin = (u) => u.is_admin && u.admin_source === 'oidc';
|
||||
|
||||
async function setAdmin(user, next) {
|
||||
if (next && !(await confirm({
|
||||
title: `Make ${user.username} an administrator?`,
|
||||
text: 'They will be able to create and delete users, and grant this to others.',
|
||||
confirmLabel: 'Make admin',
|
||||
}))) return;
|
||||
try {
|
||||
await api.setUserAdmin(user.id, next);
|
||||
} catch (err) {
|
||||
error = err.message;
|
||||
}
|
||||
refresh();
|
||||
}
|
||||
|
||||
async function setDisabled(user, next) {
|
||||
if (next && !(await confirm({
|
||||
title: `Disable ${user.username}?`,
|
||||
text: 'They cannot sign in and their API keys stop working. Their history stays.',
|
||||
confirmLabel: 'Disable',
|
||||
danger: true,
|
||||
}))) return;
|
||||
try {
|
||||
await api.setUserDisabled(user.id, next);
|
||||
} catch (err) {
|
||||
error = err.message;
|
||||
}
|
||||
refresh();
|
||||
}
|
||||
|
||||
// --- settings --------------------------------------------------------------
|
||||
|
||||
// Seconds are what the API speaks; people think in minutes and hours. The two
|
||||
// are converted here rather than in the server, which should keep exactly one
|
||||
// unit.
|
||||
const UNITS = [
|
||||
{ label: 'minutes', seconds: 60 },
|
||||
{ label: 'hours', seconds: 3600 },
|
||||
{ label: 'days', seconds: 86400 },
|
||||
];
|
||||
|
||||
function bestUnit(seconds) {
|
||||
for (const u of [...UNITS].reverse()) {
|
||||
if (seconds > 0 && seconds % u.seconds === 0) return u;
|
||||
}
|
||||
return UNITS[0];
|
||||
}
|
||||
|
||||
function settingsCard() {
|
||||
const editable = data.settings.editable || {};
|
||||
const inputs = new Map();
|
||||
|
||||
const rows = Object.entries(editable).map(([key, s]) => {
|
||||
const unit = bestUnit(s.seconds);
|
||||
const value = h('input', {
|
||||
type: 'number',
|
||||
min: '0',
|
||||
value: String(Math.round(s.seconds / unit.seconds)),
|
||||
class: 'setting-value',
|
||||
});
|
||||
const select = h('select', { class: 'setting-unit' },
|
||||
...UNITS.map((u) => h('option', {
|
||||
value: String(u.seconds),
|
||||
text: u.label,
|
||||
selected: u.seconds === unit.seconds,
|
||||
})));
|
||||
inputs.set(key, () => Number(value.value) * Number(select.value));
|
||||
|
||||
return h('tr', {},
|
||||
h('td', {}, h('strong', { text: key.replace(/_seconds$/, '').replace(/_/g, ' ') })),
|
||||
h('td', { class: 'muted small', text: s.description }),
|
||||
h('td', {}, value, select),
|
||||
);
|
||||
});
|
||||
|
||||
const form = h('form', { class: 'admin-settings' },
|
||||
h('table', { class: 'admin-table' }, h('tbody', {}, rows)),
|
||||
h('button', { class: 'btn', type: 'submit', text: 'Save settings' }));
|
||||
|
||||
form.addEventListener('submit', async (e) => {
|
||||
e.preventDefault();
|
||||
if (busy) return;
|
||||
busy = true;
|
||||
const body = {};
|
||||
for (const [key, read] of inputs) body[key] = read();
|
||||
try {
|
||||
await api.setAdminSettings(body);
|
||||
await refresh();
|
||||
} catch (err) {
|
||||
error = err.message;
|
||||
render();
|
||||
} finally {
|
||||
busy = false;
|
||||
}
|
||||
});
|
||||
|
||||
const env = Object.entries(data.settings.from_env || {}).map(([k, v]) =>
|
||||
h('tr', {},
|
||||
h('td', {}, h('code', { text: k })),
|
||||
h('td', { class: 'muted', text: v === '' ? '(unset)' : v })));
|
||||
|
||||
return h('div', { class: 'card' },
|
||||
h('h2', { text: 'Settings' }),
|
||||
h('p', { class: 'muted small', text: 'Saved changes take effect on the next sweep — no restart.' }),
|
||||
form,
|
||||
h('h3', { text: 'From the environment' }),
|
||||
h('p', { class: 'muted small' },
|
||||
'Where the server is plugged in, rather than how it behaves. These are set ',
|
||||
'in the deployment and are read-only here. Credentials are never shown.'),
|
||||
h('table', { class: 'admin-table' }, h('tbody', {}, env)),
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,353 @@
|
||||
// One team, at /admin/teams/{id}: what it is, who is in it, the invites into
|
||||
// it, and the two destructive things an administrator can do to it.
|
||||
//
|
||||
// The mirror of adminuser.js. That page answers "which teams is this person
|
||||
// in"; this one answers "who is in this team" for a team the administrator
|
||||
// need not be a member of — which the Team tab cannot do, because it only
|
||||
// offers teams the viewer is in.
|
||||
//
|
||||
// Only rendered for a system administrator. The server enforces that on every
|
||||
// endpoint regardless, so this view says so rather than pretending to be a
|
||||
// gate.
|
||||
|
||||
import * as api from './api.js';
|
||||
import { h, clear, spinner, confirm, toast, icon, ssoBadge, SSO_MANAGED } from './ui.js';
|
||||
import { state } from './state.js';
|
||||
import { navigate } from './app.js';
|
||||
import { when } from './format.js';
|
||||
|
||||
const view = () => document.getElementById('view-adminteam');
|
||||
|
||||
let teamID = null;
|
||||
let data = null; // { team, members, users, invites }
|
||||
let error = null;
|
||||
let busy = false;
|
||||
// An invite link is shown once and never stored, so it lives here until the
|
||||
// page is left rather than being toasted away after three seconds.
|
||||
let freshInvite = null;
|
||||
|
||||
export function show(route) {
|
||||
const next = route && route.team != null ? route.team : null;
|
||||
if (next !== teamID) {
|
||||
teamID = next;
|
||||
data = null;
|
||||
error = null;
|
||||
freshInvite = null;
|
||||
}
|
||||
if (!data) clear(view(), spinner());
|
||||
refresh();
|
||||
}
|
||||
|
||||
export async function refresh() {
|
||||
if (teamID == null || !state.me?.user?.is_admin) {
|
||||
render();
|
||||
return;
|
||||
}
|
||||
try {
|
||||
// The team and its members come from the admin endpoint in one answer:
|
||||
// /teams/{id}/members is member-only and 404s an administrator from
|
||||
// outside the team, deliberately. users() is the add-a-member picker.
|
||||
const [team, users, invites] = await Promise.all([
|
||||
api.adminTeam(teamID),
|
||||
api.users(),
|
||||
api.invites(teamID),
|
||||
]);
|
||||
data = { team: team.team, members: team.members, users, invites };
|
||||
error = null;
|
||||
} catch (err) {
|
||||
// A team that is gone answers 404, where a missing user is simply absent
|
||||
// from a list adminuser.js already has. So the "no such team" state has to
|
||||
// be recognised here; left to the error banner it would read as a fetch
|
||||
// that failed, which is a different thing and invites a retry.
|
||||
if (err.status === 404) {
|
||||
data = { team: null, members: [], users: [], invites: [] };
|
||||
error = null;
|
||||
} else {
|
||||
error = err.message;
|
||||
}
|
||||
}
|
||||
render();
|
||||
}
|
||||
|
||||
function render() {
|
||||
const el = view();
|
||||
if (!state.me?.user?.is_admin) {
|
||||
clear(el, backLink(), h('div', { class: 'card' },
|
||||
h('p', { class: 'muted', text: 'Administration is for system administrators. Ask one for access.' })));
|
||||
return;
|
||||
}
|
||||
if (!data) {
|
||||
clear(el, backLink(), error ? h('div', { class: 'load-error', text: error }) : spinner());
|
||||
return;
|
||||
}
|
||||
if (!data.team) {
|
||||
clear(el, backLink(), h('div', { class: 'card' },
|
||||
h('p', { class: 'muted', text: 'No such team. It may have just been deleted.' })));
|
||||
return;
|
||||
}
|
||||
clear(el,
|
||||
backLink(),
|
||||
error && h('div', { class: 'load-error', text: `Showing older data: ${error}` }),
|
||||
identityCard(),
|
||||
membersCard(),
|
||||
invitesCard(),
|
||||
dangerCard(),
|
||||
);
|
||||
}
|
||||
|
||||
function backLink() {
|
||||
return h('a', { class: 'back-link', href: '/admin/teams' }, icon('chevronLeft'), h('span', { text: 'Teams' }));
|
||||
}
|
||||
|
||||
// --- identity --------------------------------------------------------------
|
||||
|
||||
function identityCard() {
|
||||
const t = data.team;
|
||||
const err = h('p', { class: 'form-error', role: 'alert', hidden: true });
|
||||
const ok = h('p', { class: 'form-ok', role: 'status', hidden: true });
|
||||
const name = h('input', {
|
||||
name: 'name', type: 'text', value: t.name, required: true,
|
||||
autocomplete: 'off', spellcheck: false,
|
||||
});
|
||||
const submit = h('button', { class: 'btn btn-primary', type: 'submit', text: 'Save name' });
|
||||
|
||||
// A field rather than the window.prompt this used to be. The server answers
|
||||
// 409 for a name already taken, and a dialog is the wrong place to read that.
|
||||
const form = h('form', { class: 'inline-form' }, name, submit);
|
||||
form.addEventListener('submit', async (e) => {
|
||||
e.preventDefault();
|
||||
if (busy) return;
|
||||
err.hidden = true;
|
||||
ok.hidden = true;
|
||||
const next = name.value.trim();
|
||||
if (!next || next === t.name) return;
|
||||
busy = true;
|
||||
submit.disabled = true;
|
||||
try {
|
||||
await api.renameTeam(teamID, next);
|
||||
ok.textContent = 'Name saved.';
|
||||
ok.hidden = false;
|
||||
error = null;
|
||||
} catch (ex) {
|
||||
err.textContent = ex.message;
|
||||
err.hidden = false;
|
||||
busy = false;
|
||||
submit.disabled = false;
|
||||
return;
|
||||
}
|
||||
busy = false;
|
||||
submit.disabled = false;
|
||||
await refresh();
|
||||
});
|
||||
|
||||
return h('div', { class: 'card' },
|
||||
h('div', { class: 'user-head' }, h('h2', { text: t.name })),
|
||||
h('dl', { class: 'user-facts' },
|
||||
fact('Created', when(t.created_at)),
|
||||
fact('Members', String(t.members)),
|
||||
fact('Open incidents', String(t.open_incidents)),
|
||||
// Read-only here: an administrator can see why a team's OIDC-sourced
|
||||
// membership looks the way it does, but setting it is the team's own
|
||||
// owner's call, from the Team tab.
|
||||
...(state.auth?.oidc?.enabled ? [
|
||||
fact('OIDC member group', t.oidc_member_group || '—'),
|
||||
fact('OIDC owner group', t.oidc_owner_group || '—'),
|
||||
] : []),
|
||||
),
|
||||
form, err, ok,
|
||||
);
|
||||
}
|
||||
|
||||
function fact(label, value) {
|
||||
return [h('dt', { text: label }), h('dd', { text: value })];
|
||||
}
|
||||
|
||||
// --- members ---------------------------------------------------------------
|
||||
|
||||
// An administrator passes every team-owner check without being in the team,
|
||||
// which is what lets them repair a team whose owner has left. So this card
|
||||
// edits rather than reporting what somebody else would have to do.
|
||||
function membersCard() {
|
||||
const rows = data.members.map((m) =>
|
||||
h('tr', {},
|
||||
// Unlike the Team tab's own member list, the name is a link: that
|
||||
// person's page is where the rest of them lives.
|
||||
h('td', {}, h('a', { class: 'row-link', href: `/admin/users/${m.user_id}`, text: m.username })),
|
||||
h('td', { class: 'muted small' }, m.role, m.source === 'oidc' && ssoBadge()),
|
||||
h('td', { class: 'row-actions' },
|
||||
h('button', {
|
||||
class: 'btn-sm', type: 'button',
|
||||
text: m.role === 'owner' ? 'Make member' : 'Make owner',
|
||||
// The server refuses to edit a membership the groups grant.
|
||||
disabled: m.source === 'oidc',
|
||||
title: m.source === 'oidc' ? SSO_MANAGED : null,
|
||||
// The same endpoint both ways: adding is an upsert on the role.
|
||||
onclick: () => act(() =>
|
||||
api.addTeamMember(teamID, m.user_id, m.role === 'owner' ? 'member' : 'owner')),
|
||||
}),
|
||||
h('button', {
|
||||
class: 'btn-sm danger', type: 'button', text: 'Remove',
|
||||
disabled: m.source === 'oidc',
|
||||
title: m.source === 'oidc' ? SSO_MANAGED : null,
|
||||
// The server refuses the last owner with a 409, which act() shows.
|
||||
onclick: () => act(() => api.removeTeamMember(teamID, m.user_id)),
|
||||
}),
|
||||
),
|
||||
));
|
||||
|
||||
const inTeam = new Set(data.members.map((m) => m.user_id));
|
||||
// A disabled account cannot sign in, so putting one on a rota would be
|
||||
// staffing the team with somebody who cannot answer.
|
||||
const candidates = data.users.filter((u) => !inTeam.has(u.id) && !u.disabled_at);
|
||||
const pick = h('select', {},
|
||||
...candidates.map((u) => h('option', { value: String(u.id), text: u.username })));
|
||||
const role = h('select', {},
|
||||
h('option', { value: 'member', text: 'member' }),
|
||||
h('option', { value: 'owner', text: 'owner' }));
|
||||
const form = h('form', { class: 'inline-form' }, pick, role,
|
||||
h('button', { class: 'btn', type: 'submit', text: 'Add' }));
|
||||
form.addEventListener('submit', (e) => {
|
||||
e.preventDefault();
|
||||
act(() => api.addTeamMember(teamID, Number(pick.value), role.value));
|
||||
});
|
||||
|
||||
return h('div', { class: 'card' },
|
||||
h('h2', { text: 'Members' }),
|
||||
data.members.length === 0 && h('p', { class: 'muted small' },
|
||||
'Nobody is in this team. Its queue has no one to work it and its ',
|
||||
'escalation has no one to reach — add somebody, or delete it.'),
|
||||
data.members.length > 0 && h('table', { class: 'admin-table' }, h('tbody', {}, rows)),
|
||||
candidates.length > 0 && form,
|
||||
);
|
||||
}
|
||||
|
||||
// --- invites ---------------------------------------------------------------
|
||||
|
||||
// Adding a person to the server is minting them an invite into a team, not
|
||||
// creating a row: whoever accepts it picks their own password, so one never
|
||||
// passes through an administrator, and the link carries the team, so they do
|
||||
// not land on an empty queue.
|
||||
//
|
||||
// This lives on the team rather than on the Users page, where it used to be
|
||||
// with a team picker beside it. The picker was the admission that an invite is
|
||||
// a fact about a team.
|
||||
function invitesCard() {
|
||||
const role = h('select', {},
|
||||
h('option', { value: 'member', text: 'member' }),
|
||||
h('option', { value: 'owner', text: 'owner' }));
|
||||
const form = h('form', { class: 'inline-form' }, role,
|
||||
h('button', { class: 'btn', type: 'submit', text: 'Create invite' }));
|
||||
form.addEventListener('submit', async (e) => {
|
||||
e.preventDefault();
|
||||
if (busy) return;
|
||||
busy = true;
|
||||
try {
|
||||
const inv = await api.createInvite(teamID, role.value, 1);
|
||||
freshInvite = inv.url;
|
||||
error = null;
|
||||
} catch (err) {
|
||||
error = err.message;
|
||||
} finally {
|
||||
busy = false;
|
||||
}
|
||||
await refresh();
|
||||
});
|
||||
|
||||
// The server lists spent and revoked invites too, and they are worth seeing:
|
||||
// "who was invited here" is part of the answer to "who is in this team".
|
||||
// Only a live one can be revoked, so only a live one offers the button.
|
||||
const rows = (data.invites || []).map((inv) => {
|
||||
const state = inviteState(inv);
|
||||
return h('tr', { class: state === 'live' ? '' : 'disabled-row' },
|
||||
h('td', {}, h('strong', { text: inv.role })),
|
||||
h('td', { class: 'muted small', text: `${inv.uses}/${inv.max_uses} used` }),
|
||||
h('td', { class: 'muted small', text: state === 'live' ? `expires ${when(inv.expires_at)}` : state }),
|
||||
h('td', { class: 'row-actions' },
|
||||
state === 'live' && h('button', {
|
||||
class: 'btn-sm danger', type: 'button', text: 'Revoke',
|
||||
onclick: () => act(() => api.revokeInvite(teamID, inv.id)),
|
||||
})),
|
||||
);
|
||||
});
|
||||
|
||||
return h('div', { class: 'card' },
|
||||
h('h2', { text: 'Invites' }),
|
||||
h('p', { class: 'muted small' },
|
||||
'An invite link puts somebody in this team and lets them choose their ',
|
||||
'own password. It lasts a week and can be used once.'),
|
||||
rows.length > 0 && h('table', { class: 'admin-table' }, h('tbody', {}, rows)),
|
||||
form,
|
||||
// Shown once and never stored, so it goes on the page to be copied.
|
||||
freshInvite && h('p', { class: 'invite-out' },
|
||||
h('strong', { text: 'Send them this link. It is shown once.' }),
|
||||
h('code', { class: 'invite-link', text: freshInvite })),
|
||||
);
|
||||
}
|
||||
|
||||
// Why a link no longer works, in the server's own order of precedence: revoked
|
||||
// beats spent beats expired. Only 'live' is still usable.
|
||||
function inviteState(inv) {
|
||||
if (inv.revoked) return 'revoked';
|
||||
if (inv.uses >= inv.max_uses) return 'used up';
|
||||
if (new Date(inv.expires_at).getTime() <= Date.now()) return 'expired';
|
||||
return 'live';
|
||||
}
|
||||
|
||||
// --- delete ----------------------------------------------------------------
|
||||
|
||||
function dangerCard() {
|
||||
const t = data.team;
|
||||
const blocked = t.open_incidents > 0;
|
||||
return h('div', { class: 'card' },
|
||||
h('h2', { text: 'Delete' }),
|
||||
h('p', { class: 'muted small' },
|
||||
'Its alerts, incidents, schedule and integrations go with it. This ',
|
||||
'cannot be undone. Everybody in it keeps their account and stays in ',
|
||||
'whatever other teams they are in.'),
|
||||
h('button', {
|
||||
class: 'btn btn-danger', type: 'button', text: `Delete ${t.name}`,
|
||||
// Saying so before the click is kinder than a 409 afterwards.
|
||||
disabled: blocked,
|
||||
title: blocked ? 'Resolve its open incidents first' : '',
|
||||
onclick: deleteTeam,
|
||||
}),
|
||||
);
|
||||
}
|
||||
|
||||
async function deleteTeam() {
|
||||
if (!(await confirm({
|
||||
title: `Delete ${data.team.name}?`,
|
||||
text: 'Its alerts, incidents, schedule and integrations go with it. This cannot be undone.',
|
||||
confirmLabel: 'Delete',
|
||||
danger: true,
|
||||
}))) return;
|
||||
try {
|
||||
await api.deleteTeam(teamID);
|
||||
} catch (err) {
|
||||
error = err.message;
|
||||
render();
|
||||
return;
|
||||
}
|
||||
toast('Team deleted.');
|
||||
// Not act(): there is no longer a page here to refresh.
|
||||
navigate('/admin/teams');
|
||||
}
|
||||
|
||||
// --- plumbing --------------------------------------------------------------
|
||||
|
||||
// act runs a write and reloads. Errors are shown rather than thrown away: the
|
||||
// 409 from the last-owner guard, and the one for a duplicate name, are the
|
||||
// server explaining itself, and the reader needs to see it.
|
||||
async function act(fn) {
|
||||
if (busy) return;
|
||||
busy = true;
|
||||
try {
|
||||
await fn();
|
||||
error = null;
|
||||
} catch (err) {
|
||||
error = err.message;
|
||||
} finally {
|
||||
busy = false;
|
||||
}
|
||||
await refresh();
|
||||
}
|
||||
@@ -0,0 +1,303 @@
|
||||
// One person, at /admin/users/{id}: what they are, what they are in, and the
|
||||
// levers an administrator has over the account.
|
||||
//
|
||||
// A section of its own rather than an expanding row in the Admin tab's table,
|
||||
// because memberships and the account actions together are more than a row can
|
||||
// hold and still be read on a phone.
|
||||
//
|
||||
// Like the Admin tab, this hides nothing the server would allow and shows
|
||||
// nothing it would refuse: every write here is an endpoint that answers 403
|
||||
// without the flag, so the view is a description of the rules rather than an
|
||||
// enforcement of them.
|
||||
|
||||
import * as api from './api.js';
|
||||
import { h, clear, spinner, confirm, toast, icon, ssoBadge, SSO_MANAGED } from './ui.js';
|
||||
import { state, myID } from './state.js';
|
||||
import { navigate } from './app.js';
|
||||
import { when } from './format.js';
|
||||
|
||||
const view = () => document.getElementById('view-adminuser');
|
||||
|
||||
let userID = null;
|
||||
let data = null; // { user, teams, allTeams }
|
||||
let error = null;
|
||||
let busy = false;
|
||||
|
||||
export function show(route) {
|
||||
const next = route && route.user != null ? route.user : null;
|
||||
if (next !== userID) {
|
||||
userID = next;
|
||||
data = null;
|
||||
error = null;
|
||||
}
|
||||
if (!data) clear(view(), spinner());
|
||||
refresh();
|
||||
}
|
||||
|
||||
export async function refresh() {
|
||||
if (userID == null || !state.me?.user?.is_admin) {
|
||||
render();
|
||||
return;
|
||||
}
|
||||
try {
|
||||
// The user comes from the list rather than a show endpoint: there is no
|
||||
// GET /api/users/{id}, and adding one for a row the list already carries
|
||||
// would be a second way to say the same thing.
|
||||
const [users, teams, allTeams] = await Promise.all([
|
||||
api.users(),
|
||||
api.userTeams(userID),
|
||||
api.adminTeams(),
|
||||
]);
|
||||
const user = users.find((u) => u.id === userID) || null;
|
||||
data = { user, teams, allTeams };
|
||||
error = null;
|
||||
} catch (err) {
|
||||
error = err.message;
|
||||
}
|
||||
render();
|
||||
}
|
||||
|
||||
function render() {
|
||||
const el = view();
|
||||
if (!state.me?.user?.is_admin) {
|
||||
clear(el, backLink(), h('div', { class: 'card' },
|
||||
h('p', { class: 'muted', text: 'Administration is for system administrators. Ask one for access.' })));
|
||||
return;
|
||||
}
|
||||
if (!data) {
|
||||
clear(el, backLink(), error ? h('div', { class: 'load-error', text: error }) : spinner());
|
||||
return;
|
||||
}
|
||||
if (!data.user) {
|
||||
clear(el, backLink(), h('div', { class: 'card' },
|
||||
h('p', { class: 'muted', text: 'No such user. They may have just been deleted.' })));
|
||||
return;
|
||||
}
|
||||
clear(el,
|
||||
backLink(),
|
||||
error && h('div', { class: 'load-error', text: `Showing older data: ${error}` }),
|
||||
identityCard(),
|
||||
teamsCard(),
|
||||
accountCard(),
|
||||
);
|
||||
}
|
||||
|
||||
function backLink() {
|
||||
return h('a', { class: 'back-link', href: '/admin/users' }, icon('chevronLeft'), h('span', { text: 'Users' }));
|
||||
}
|
||||
|
||||
// --- identity --------------------------------------------------------------
|
||||
|
||||
function identityCard() {
|
||||
const u = data.user;
|
||||
const self = u.id === myID();
|
||||
|
||||
return h('div', { class: 'card' },
|
||||
h('div', { class: 'user-head' },
|
||||
h('h2', { text: u.username }),
|
||||
u.is_admin && h('span', { class: 'row-team', text: 'admin' }),
|
||||
u.is_admin && u.admin_source === 'oidc' && ssoBadge(),
|
||||
u.disabled_at && h('span', { class: 'row-team', text: 'disabled' }),
|
||||
self && h('span', { class: 'you', text: 'you' })),
|
||||
h('dl', { class: 'user-facts' },
|
||||
fact('Email', u.email),
|
||||
fact('Joined', when(u.created_at)),
|
||||
fact('Notifications', u.ntfy_topic ? `ntfy: ${u.ntfy_topic}` : 'None of their own'),
|
||||
u.disabled_at && fact('Disabled', when(u.disabled_at)),
|
||||
),
|
||||
// Both of these refuse your own account, and the last administrator's. An
|
||||
// enabled button that always fails is worse than no button.
|
||||
h('div', { class: 'row-actions' },
|
||||
!self && h('button', {
|
||||
class: 'btn', type: 'button',
|
||||
text: u.is_admin ? 'Revoke admin' : 'Make admin',
|
||||
// The server refuses to revoke what the groups grant.
|
||||
disabled: u.is_admin && u.admin_source === 'oidc',
|
||||
title: u.is_admin && u.admin_source === 'oidc' ? SSO_MANAGED : null,
|
||||
onclick: () => setAdmin(!u.is_admin),
|
||||
}),
|
||||
!self && h('button', {
|
||||
class: 'btn', type: 'button',
|
||||
text: u.disabled_at ? 'Enable account' : 'Disable account',
|
||||
onclick: () => setDisabled(!u.disabled_at),
|
||||
}),
|
||||
),
|
||||
self && h('p', { class: 'muted small' },
|
||||
'You cannot change your own administrator flag or disable yourself — ',
|
||||
'that is how an install ends up with nobody who can administer it.'),
|
||||
);
|
||||
}
|
||||
|
||||
function fact(label, value) {
|
||||
return [h('dt', { text: label }), h('dd', { text: value })];
|
||||
}
|
||||
|
||||
async function setAdmin(next) {
|
||||
if (next && !(await confirm({
|
||||
title: `Make ${data.user.username} an administrator?`,
|
||||
text: 'They will be able to manage every account, configure any team, and grant this to others.',
|
||||
confirmLabel: 'Make admin',
|
||||
}))) return;
|
||||
await act(() => api.setUserAdmin(userID, next));
|
||||
}
|
||||
|
||||
async function setDisabled(next) {
|
||||
if (next && !(await confirm({
|
||||
title: `Disable ${data.user.username}?`,
|
||||
text: 'They cannot sign in and their API keys stop working. Their acknowledgements and timeline entries stay.',
|
||||
confirmLabel: 'Disable',
|
||||
danger: true,
|
||||
}))) return;
|
||||
await act(() => api.setUserDisabled(userID, next));
|
||||
}
|
||||
|
||||
// --- teams -----------------------------------------------------------------
|
||||
|
||||
// An administrator passes every team-owner check without being in the team,
|
||||
// which is what lets them repair a team whose owner has left. So this card
|
||||
// edits, rather than reporting what somebody else would have to do.
|
||||
//
|
||||
// It is the one place membership can be changed from the person's side: the
|
||||
// Team tab asks "who is in this team", and answering "which teams is this
|
||||
// person in" there means visiting each team in turn.
|
||||
function teamsCard() {
|
||||
const rows = data.teams.map((t) =>
|
||||
h('tr', {},
|
||||
// Not a link: the Team tab always shows the viewer's own team, so
|
||||
// sending them there from somebody else's membership would be a lie.
|
||||
h('td', {}, h('strong', { text: t.name })),
|
||||
h('td', { class: 'muted small' }, t.role, t.source === 'oidc' && ssoBadge()),
|
||||
h('td', { class: 'row-actions' },
|
||||
h('button', {
|
||||
class: 'btn-sm', type: 'button',
|
||||
text: t.role === 'owner' ? 'Make member' : 'Make owner',
|
||||
// The server refuses to edit a membership the groups grant.
|
||||
disabled: t.source === 'oidc',
|
||||
title: t.source === 'oidc' ? SSO_MANAGED : null,
|
||||
onclick: () => act(() =>
|
||||
api.addTeamMember(t.id, userID, t.role === 'owner' ? 'member' : 'owner')),
|
||||
}),
|
||||
h('button', {
|
||||
class: 'btn-sm danger', type: 'button', text: 'Remove',
|
||||
disabled: t.source === 'oidc',
|
||||
title: t.source === 'oidc' ? SSO_MANAGED : null,
|
||||
// The server refuses the last owner with a 409, which act() shows.
|
||||
onclick: () => act(() => api.removeTeamMember(t.id, userID)),
|
||||
}),
|
||||
),
|
||||
));
|
||||
|
||||
const inTeam = new Set(data.teams.map((t) => t.id));
|
||||
const candidates = (data.allTeams || []).filter((t) => !inTeam.has(t.id));
|
||||
const pick = h('select', {},
|
||||
...candidates.map((t) => h('option', { value: String(t.id), text: t.name })));
|
||||
const role = h('select', {},
|
||||
h('option', { value: 'member', text: 'member' }),
|
||||
h('option', { value: 'owner', text: 'owner' }));
|
||||
const form = h('form', { class: 'inline-form' }, pick, role,
|
||||
h('button', { class: 'btn', type: 'submit', text: 'Add' }));
|
||||
form.addEventListener('submit', (e) => {
|
||||
e.preventDefault();
|
||||
act(() => api.addTeamMember(Number(pick.value), userID, role.value));
|
||||
});
|
||||
|
||||
return h('div', { class: 'card' },
|
||||
h('h2', { text: 'Teams' }),
|
||||
data.teams.length === 0 && h('p', { class: 'muted small' },
|
||||
'In no team. They can sign in, but there is no queue for them to work ',
|
||||
'and nothing to page them about.'),
|
||||
data.teams.length > 0 && h('table', { class: 'admin-table' }, h('tbody', {}, rows)),
|
||||
candidates.length > 0 && form,
|
||||
);
|
||||
}
|
||||
|
||||
// --- account ---------------------------------------------------------------
|
||||
|
||||
function accountCard() {
|
||||
const u = data.user;
|
||||
const self = u.id === myID();
|
||||
|
||||
const pw = h('input', {
|
||||
type: 'password', name: 'password', autocomplete: 'new-password',
|
||||
minlength: '10', required: true, placeholder: 'At least 10 characters',
|
||||
});
|
||||
const form = h('form', { class: 'inline-form' }, pw,
|
||||
h('button', { class: 'btn', type: 'submit', text: 'Set password' }));
|
||||
form.addEventListener('submit', async (e) => {
|
||||
e.preventDefault();
|
||||
if (busy) return;
|
||||
busy = true;
|
||||
try {
|
||||
// No current password: that check is for changing your own, and an
|
||||
// administrator setting somebody else's does not know it by design.
|
||||
await api.setPassword(userID, pw.value);
|
||||
pw.value = '';
|
||||
toast(`Password set for ${u.username}. Their other sessions are signed out.`);
|
||||
error = null;
|
||||
} catch (err) {
|
||||
error = err.message;
|
||||
} finally {
|
||||
busy = false;
|
||||
}
|
||||
await refresh();
|
||||
});
|
||||
|
||||
return h('div', { class: 'card' },
|
||||
h('h2', { text: 'Account' }),
|
||||
h('p', { class: 'muted small' },
|
||||
'Setting a password here is how somebody gets their first one, or a new ',
|
||||
'one after forgetting it. It signs them out everywhere else. They change ',
|
||||
'it themselves under Account afterwards.'),
|
||||
self ? h('p', { class: 'muted small' },
|
||||
'Change your own password under Account, where the current one is asked for.')
|
||||
: form,
|
||||
h('h3', { text: 'Delete' }),
|
||||
h('p', { class: 'muted small' },
|
||||
'Deleting erases their acknowledgements and timeline entries — incidents ',
|
||||
'they handled stop saying who did. Disabling keeps the history and is ',
|
||||
'almost always what is meant.'),
|
||||
h('button', {
|
||||
class: 'btn btn-danger', type: 'button', text: `Delete ${u.username}`,
|
||||
disabled: self,
|
||||
title: self ? 'You cannot delete your own account' : '',
|
||||
onclick: deleteUser,
|
||||
}),
|
||||
);
|
||||
}
|
||||
|
||||
async function deleteUser() {
|
||||
if (!(await confirm({
|
||||
title: `Delete ${data.user.username}?`,
|
||||
text: 'Their API keys go with them, and their name comes off every incident they acknowledged. This cannot be undone.',
|
||||
confirmLabel: 'Delete',
|
||||
danger: true,
|
||||
}))) return;
|
||||
try {
|
||||
await api.deleteUser(userID);
|
||||
} catch (err) {
|
||||
error = err.message;
|
||||
render();
|
||||
return;
|
||||
}
|
||||
toast('User deleted.');
|
||||
navigate('/admin/users');
|
||||
}
|
||||
|
||||
// --- plumbing --------------------------------------------------------------
|
||||
|
||||
// act runs a write and reloads. Errors are shown rather than thrown away: the
|
||||
// 409 from the last-owner or last-administrator guard is the server explaining
|
||||
// itself, and the reader needs to see it.
|
||||
async function act(fn) {
|
||||
if (busy) return;
|
||||
busy = true;
|
||||
try {
|
||||
await fn();
|
||||
error = null;
|
||||
} catch (err) {
|
||||
error = err.message;
|
||||
} finally {
|
||||
busy = false;
|
||||
}
|
||||
await refresh();
|
||||
}
|
||||
@@ -60,7 +60,11 @@ function render() {
|
||||
let body;
|
||||
if (error && !items) body = h('div', { class: 'load-error', text: error });
|
||||
else if (!items) body = spinner();
|
||||
else if (!items.length) body = emptyState(filter === 'firing' ? 'Nothing firing' : 'No alerts', '', filter === 'firing' ? 'checkCircle' : null);
|
||||
else if (!items.length) {
|
||||
body = filter === 'firing'
|
||||
? emptyState('Nothing firing', 'No alerts are currently firing.', 'checkCircle')
|
||||
: emptyState('No alerts', 'None match this filter.', 'bell');
|
||||
}
|
||||
else {
|
||||
body = h('div', { class: 'list' },
|
||||
error && h('div', { class: 'load-error', text: `Showing older data: ${error}` }),
|
||||
|
||||
@@ -59,14 +59,28 @@ async function call(method, path, { query, body, signal } = {}) {
|
||||
|
||||
// session
|
||||
export const me = () => call('GET', '/me');
|
||||
// How this server can be signed in to: { password_login, oidc: { enabled, name } }.
|
||||
export const authConfig = () => call('GET', '/auth/config');
|
||||
export const login = (username, password) => call('POST', '/login', { body: { username, password } });
|
||||
export const logout = () => call('POST', '/logout');
|
||||
// Approve or refuse a sign-in a terminal started; code is what it is showing.
|
||||
export const approveDevice = (code) => call('POST', '/oidc/device/approve', { body: { user_code: code } });
|
||||
export const denyDevice = (code) => call('POST', '/oidc/device/deny', { body: { user_code: code } });
|
||||
export const setPassword = (userID, password, currentPassword) =>
|
||||
call('PUT', `/users/${userID}/password`, { body: { password, current_password: currentPassword } });
|
||||
|
||||
// users
|
||||
export const users = () => call('GET', '/users');
|
||||
|
||||
// What one person is in. /teams answers "what am I in" and cannot be asked
|
||||
// about anybody else, which is what the admin page's per-user view needs.
|
||||
export const userTeams = (id) => call('GET', `/users/${id}/teams`);
|
||||
|
||||
// Where this user's pages go. An empty topic clears it, which the server
|
||||
// treats as "no topic of their own" rather than an error.
|
||||
export const setNotifyTarget = (id, ntfyTopic) =>
|
||||
call('PUT', `/users/${id}/notify`, { body: { ntfy_topic: ntfyTopic } });
|
||||
|
||||
// incidents
|
||||
export const incidents = (query, opts) => call('GET', '/incidents', { query, ...opts });
|
||||
export const incident = (id) => call('GET', `/incidents/${id}`);
|
||||
@@ -74,25 +88,96 @@ export const timeline = (id) => call('GET', `/incidents/${id}/timeline`);
|
||||
|
||||
export const acknowledge = (id) => call('POST', `/incidents/${id}/acknowledge`);
|
||||
export const unacknowledge = (id) => call('DELETE', `/incidents/${id}/acknowledge`);
|
||||
export const resolve = (id) => call('POST', `/incidents/${id}/resolve`);
|
||||
export const resolve = (id, resolution) => call('POST', `/incidents/${id}/resolve`, resolution ? { body: { resolution } } : {});
|
||||
export const assign = (id, userID) => call('POST', `/incidents/${id}/assign`, { body: { user_id: userID } });
|
||||
export const snooze = (id, spec) => call('POST', `/incidents/${id}/snooze`, { body: spec });
|
||||
export const unsnooze = (id) => call('DELETE', `/incidents/${id}/snooze`);
|
||||
export const archive = (id) => call('POST', `/incidents/${id}/archive`);
|
||||
export const unarchive = (id) => call('DELETE', `/incidents/${id}/archive`);
|
||||
export const addNote = (id, content) => call('POST', `/incidents/${id}/notes`, { body: { content } });
|
||||
export const addNote = (id, content, pinned = false) => call('POST', `/incidents/${id}/notes`, { body: { content, pinned } });
|
||||
export const similar = (id) => call('GET', `/incidents/${id}/similar`);
|
||||
export const deleteNote = (id, eventID) => call('DELETE', `/incidents/${id}/notes/${eventID}`);
|
||||
|
||||
// stats
|
||||
export const statsIncidents = (query) => call('GET', '/stats/incidents', { query });
|
||||
export const statsTop = (query) => call('GET', '/stats/alerts/top', { query });
|
||||
export const statsByHour = (query) => call('GET', '/stats/alerts/by-hour', { query });
|
||||
export const statsByDay = (query) => call('GET', '/stats/alerts/by-day', { query });
|
||||
|
||||
// alerts
|
||||
export const alerts = (query, opts) => call('GET', '/alerts', { query, ...opts });
|
||||
|
||||
// schedule
|
||||
export const schedule = (from, to) => call('GET', '/schedule', { query: { from, to } });
|
||||
export async function onCallNow() {
|
||||
try {
|
||||
return await call('GET', '/schedule/current');
|
||||
} catch (err) {
|
||||
if (err instanceof ApiError && err.status === 404) return null;
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
// Sign-up, both halves unauthenticated: the caller has no account yet.
|
||||
export const signupInfo = (invite) =>
|
||||
call('GET', '/signup', { query: invite ? { invite } : {} });
|
||||
export const signup = (body) => call('POST', '/signup', { body });
|
||||
|
||||
export const invites = (id) => call('GET', `/teams/${id}/invites`);
|
||||
export const createInvite = (id, role, maxUses) =>
|
||||
call('POST', `/teams/${id}/invites`, { body: { role, max_uses: maxUses } });
|
||||
export const revokeInvite = (id, inviteID) => call('DELETE', `/teams/${id}/invites/${inviteID}`);
|
||||
|
||||
export const testNotification = () => call('POST', '/me/notify/test');
|
||||
export const dismissOnboarding = (dismissed) =>
|
||||
call('PUT', '/me/onboarding', { body: { dismissed } });
|
||||
|
||||
export const teams = () => call('GET', '/teams');
|
||||
export const createTeam = (name) => call('POST', '/teams', { body: { name } });
|
||||
export const renameTeam = (id, name) => call('PUT', `/teams/${id}`, { body: { name } });
|
||||
export const deleteTeam = (id) => call('DELETE', `/teams/${id}`);
|
||||
|
||||
// A team's own settings. Every write is owner-only and every read is
|
||||
// member-only; the server answers 403 and 404 respectively, so the UI shows
|
||||
// what the role allows rather than guarding it.
|
||||
export const teamMembers = (id) => call('GET', `/teams/${id}/members`);
|
||||
export const addTeamMember = (id, userID, role) =>
|
||||
call('POST', `/teams/${id}/members`, { body: { user_id: userID, role } });
|
||||
export const removeTeamMember = (id, userID) => call('DELETE', `/teams/${id}/members/${userID}`);
|
||||
|
||||
// Which OIDC groups grant member and owner access to this team.
|
||||
export const oidcGroups = (id) => call('GET', `/teams/${id}/oidc-groups`);
|
||||
export const setOidcGroups = (id, body) => call('PUT', `/teams/${id}/oidc-groups`, { body });
|
||||
|
||||
export const integrations = (id) => call('GET', `/teams/${id}/integrations`);
|
||||
export const createIntegration = (id, name) =>
|
||||
call('POST', `/teams/${id}/integrations`, { body: { name } });
|
||||
export const renameIntegration = (id, integrationID, name) =>
|
||||
call('PATCH', `/teams/${id}/integrations/${integrationID}`, { body: { name } });
|
||||
export const deleteIntegration = (id, integrationID) =>
|
||||
call('DELETE', `/teams/${id}/integrations/${integrationID}`);
|
||||
|
||||
export const deadmanSwitches = (id) => call('GET', `/teams/${id}/deadman/switches`);
|
||||
export const createDeadmanSwitch = (id, body) =>
|
||||
call('POST', `/teams/${id}/deadman/switches`, { body });
|
||||
export const deleteDeadmanSwitch = (id, switchID) =>
|
||||
call('DELETE', `/teams/${id}/deadman/switches/${switchID}`);
|
||||
|
||||
export const escalation = (id) => call('GET', `/teams/${id}/escalation`);
|
||||
export const setEscalation = (id, body) => call('PUT', `/teams/${id}/escalation`, { body });
|
||||
|
||||
export const assignSchedule = (id, userID, dates, replace = false) =>
|
||||
call('POST', `/teams/${id}/schedule`, { body: { user_id: userID, dates, replace } });
|
||||
export const unassignSchedule = (id, entryID) => call('DELETE', `/teams/${id}/schedule/${entryID}`);
|
||||
|
||||
// Administration. Every one of these is refused with 403 for anybody without
|
||||
// the flag, so the UI hides the section rather than guarding it.
|
||||
export const adminTeams = () => call('GET', '/admin/teams');
|
||||
// One team and who is in it: { team, members }. /teams/{id}/members is
|
||||
// member-only and answers 404 to an administrator from outside the team, which
|
||||
// is the rule rather than an oversight -- this asks the other question.
|
||||
export const adminTeam = (id) => call('GET', `/admin/teams/${id}`);
|
||||
export const adminSettings = () => call('GET', '/admin/settings');
|
||||
export const setAdminSettings = (body) => call('PUT', '/admin/settings', { body });
|
||||
export const setUserAdmin = (id, isAdmin) =>
|
||||
call('PUT', `/users/${id}/admin`, { body: { is_admin: isAdmin } });
|
||||
export const setUserDisabled = (id, disabled) =>
|
||||
call('PUT', `/users/${id}/disabled`, { body: { disabled } });
|
||||
export const deleteUser = (id) => call('DELETE', `/users/${id}`);
|
||||
export const schedule = (teamID, from, to) =>
|
||||
call('GET', `/teams/${teamID}/schedule`, { query: { from, to } });
|
||||
|
||||
// One entry per team the viewer belongs to, for the teams that have somebody
|
||||
// scheduled today. An empty array means nobody anywhere, which is a real answer
|
||||
// rather than an error — unlike the pre-teams endpoint, which 404ed.
|
||||
export const onCallNow = () => call('GET', '/schedule/current');
|
||||
|
||||
+288
-15
@@ -3,31 +3,84 @@
|
||||
import * as api from './api.js';
|
||||
import * as ui from './ui.js';
|
||||
import * as poll from './poll.js';
|
||||
import { state, reset } from './state.js';
|
||||
import { state, reset, loadTeams } from './state.js';
|
||||
import * as queue from './queue.js';
|
||||
import * as incident from './incident.js';
|
||||
import * as oncall from './oncall.js';
|
||||
import * as alerts from './alerts.js';
|
||||
import * as stats from './stats.js';
|
||||
import * as account from './account.js';
|
||||
import * as team from './team.js';
|
||||
import * as teamselector from './teamselector.js';
|
||||
import * as admin from './admin.js';
|
||||
import * as adminuser from './adminuser.js';
|
||||
import * as adminteam from './adminteam.js';
|
||||
import * as device from './device.js';
|
||||
|
||||
const $ = (id) => document.getElementById(id);
|
||||
|
||||
// One route per section; /incidents/{id} is the queue with a detail open.
|
||||
// One route per section; /incidents/{id} is the queue with a detail open, and
|
||||
// /admin/users/{id} is a section of its own rather than a mode of the Admin
|
||||
// tab, because it replaces the page rather than opening beside it.
|
||||
const SECTIONS = {
|
||||
queue: { title: 'Queue', view: queue },
|
||||
oncall: { title: 'On-call', view: oncall },
|
||||
alerts: { title: 'Alerts', view: alerts },
|
||||
stats: { title: 'Stats', view: stats },
|
||||
team: { title: 'Team', view: team },
|
||||
admin: { title: 'Admin', view: admin },
|
||||
adminuser: { title: 'User', view: adminuser, nav: 'admin' },
|
||||
adminteam: { title: 'Team', view: adminteam, nav: 'admin' },
|
||||
more: { title: 'Account', view: account },
|
||||
// Reached by link from a terminal's sign-in prompt, not from the nav.
|
||||
device: { title: 'Sign in a terminal', view: device },
|
||||
};
|
||||
|
||||
// The desktop sidebar's .nav-link list in index.html, in the same order.
|
||||
// `secondary` marks the ones that fold into the phone bottom bar's "More"
|
||||
// sheet (openNavMenu below) instead of getting a tab of their own there.
|
||||
const NAV_ITEMS = [
|
||||
{ path: '/', section: 'queue', label: 'Queue', icon: 'queueList' },
|
||||
{ path: '/oncall', section: 'oncall', label: 'On-call', icon: 'calendar' },
|
||||
{ path: '/alerts', section: 'alerts', label: 'Alerts', icon: 'bell' },
|
||||
{ path: '/stats', section: 'stats', label: 'Stats', icon: 'chart', secondary: true },
|
||||
{ path: '/team', section: 'team', label: 'Team', icon: 'team' },
|
||||
{ path: '/admin', section: 'admin', label: 'Admin', icon: 'shield', adminOnly: true, secondary: true },
|
||||
{ path: '/more', section: 'more', label: 'Account', icon: 'user', secondary: true },
|
||||
];
|
||||
|
||||
function parseRoute(pathname) {
|
||||
const m = pathname.match(/^\/incidents\/(\d+)\/?$/);
|
||||
if (m) return { section: 'queue', incident: Number(m[1]) };
|
||||
const u = pathname.match(/^\/admin\/users\/(\d+)\/?$/);
|
||||
if (u) return { section: 'adminuser', user: Number(u[1]) };
|
||||
// Before the TABS lookup below, which matches a path exactly and would let
|
||||
// /admin/teams/7 fall through to the queue.
|
||||
const g = pathname.match(/^\/admin\/teams\/(\d+)\/?$/);
|
||||
if (g) return { section: 'adminteam', team: Number(g[1]) };
|
||||
const name = pathname.replace(/^\/|\/$/g, '');
|
||||
if (name === 'oncall' || name === 'alerts' || name === 'more') return { section: name };
|
||||
// The Admin and Team tabs' sub-sections are routes of their own. Each view
|
||||
// owns the table of its own, since each also builds the strip that links to
|
||||
// them; /team is in team.TABS as the overview, so it is matched here too.
|
||||
const t = admin.TABS.find((x) => x.path === `/${name}`);
|
||||
if (t) return { section: 'admin', tab: t.tab };
|
||||
const tt = team.TABS.find((x) => x.path === `/${name}`);
|
||||
if (tt) return { section: 'team', tab: tt.tab };
|
||||
if (name === 'oncall' || name === 'alerts' || name === 'stats' || name === 'more' || name === 'device') return { section: name };
|
||||
return { section: 'queue', incident: null };
|
||||
}
|
||||
|
||||
// What the top bar and the document title call this route. The sub-sections of
|
||||
// Admin and Team are pages in their own right, so they say which one rather
|
||||
// than the tab's name four or six times; either overview keeps the tab's own
|
||||
// name. A tab may carry a `title` where its strip label is too short to name a
|
||||
// page on its own.
|
||||
function title(r) {
|
||||
const tabs = r.section === 'admin' ? admin.TABS : r.section === 'team' ? team.TABS : null;
|
||||
const t = tabs && r.tab ? tabs.find((x) => x.tab === r.tab) : null;
|
||||
return t ? (t.title || t.label) : SECTIONS[r.section].title;
|
||||
}
|
||||
|
||||
let route = parseRoute(location.pathname);
|
||||
// How many in-app navigations deep we are, so Back can use the browser's
|
||||
// history when there is somewhere to go back to, and the queue otherwise.
|
||||
@@ -60,13 +113,16 @@ function render() {
|
||||
route = parseRoute(location.pathname);
|
||||
const app = $('app');
|
||||
|
||||
for (const [name, s] of Object.entries(SECTIONS)) {
|
||||
for (const name of Object.keys(SECTIONS)) {
|
||||
const el = $(`view-${name}`);
|
||||
el.hidden = name !== route.section;
|
||||
if (name === route.section) $('topbar-title').textContent = s.title;
|
||||
if (name === route.section) $('topbar-title').textContent = title(route);
|
||||
}
|
||||
// A section may light up somebody else's tab: /admin/users/{id} is still the
|
||||
// Admin tab as far as the nav is concerned, since there is no tab of its own.
|
||||
const current = SECTIONS[route.section].nav || route.section;
|
||||
for (const link of document.querySelectorAll('.nav-link')) {
|
||||
if (link.dataset.section === route.section) link.setAttribute('aria-current', 'page');
|
||||
if (link.dataset.section === current) link.setAttribute('aria-current', 'page');
|
||||
else link.removeAttribute('aria-current');
|
||||
}
|
||||
|
||||
@@ -81,16 +137,49 @@ function render() {
|
||||
incident.show(route.incident);
|
||||
} else {
|
||||
incident.show(null);
|
||||
SECTIONS[route.section].view.show();
|
||||
SECTIONS[route.section].view.show(route);
|
||||
}
|
||||
|
||||
if (detailOpen && !wasOpen) window.scrollTo(0, 0);
|
||||
else if (!detailOpen && wasOpen) requestAnimationFrame(() => window.scrollTo(0, listScroll));
|
||||
else if (prev.section !== route.section) window.scrollTo(0, 0);
|
||||
// A changed tab counts as a changed page: stepping from a long user list to
|
||||
// the settings should not land you halfway down them. So does a changed
|
||||
// subject — one team to the next is two pages, not one scrolled page.
|
||||
else if (prev.section !== route.section || prev.tab !== route.tab
|
||||
|| prev.user !== route.user || prev.team !== route.team) window.scrollTo(0, 0);
|
||||
|
||||
updateTitle();
|
||||
}
|
||||
|
||||
// ---------- nav menu ----------
|
||||
|
||||
// The phone bottom bar's "More" sheet: same shape as the sheet-based action
|
||||
// menus in incident.js (openSheet + a <ul class="menu"> of menu-item
|
||||
// buttons), one item per secondary NAV_ITEMS entry — the ones the bar itself
|
||||
// has no room for, since Queue/On-call/Alerts/Team already have their own
|
||||
// tab and don't need to be reachable here too.
|
||||
function openNavMenu() {
|
||||
const current = SECTIONS[route.section].nav || route.section;
|
||||
const items = NAV_ITEMS.filter((n) => n.secondary && (!n.adminOnly || state.me?.user?.is_admin));
|
||||
ui.openSheet(() => [
|
||||
ui.h('h2', { class: 'sheet-title', text: 'More' }),
|
||||
ui.h('ul', { class: 'menu', role: 'menu' }, items.map((n) =>
|
||||
ui.h('li', {}, ui.h('button', {
|
||||
class: 'menu-item',
|
||||
type: 'button',
|
||||
role: 'menuitemradio',
|
||||
'aria-checked': String(n.section === current),
|
||||
onclick: () => ui.closeSheet(n.path),
|
||||
},
|
||||
ui.icon(n.icon),
|
||||
n.label,
|
||||
))),
|
||||
),
|
||||
]).then((path) => {
|
||||
if (path) navigate(path);
|
||||
});
|
||||
}
|
||||
|
||||
// ---------- refresh + badges ----------
|
||||
|
||||
async function refresh() {
|
||||
@@ -117,15 +206,18 @@ function updateBadges() {
|
||||
pill.classList.toggle('has-triggered', triggered > 0);
|
||||
pill.classList.toggle('all-acked', open > 0 && triggered === 0);
|
||||
|
||||
const badge = document.querySelector('[data-badge]');
|
||||
badge.hidden = triggered === 0;
|
||||
badge.textContent = String(triggered);
|
||||
// Just the Queue tab's own badge now — phone bottom bar and desktop
|
||||
// sidebar both read the same [data-badge] span on that one nav-link.
|
||||
for (const badge of document.querySelectorAll('[data-badge]')) {
|
||||
badge.hidden = triggered === 0;
|
||||
badge.textContent = String(triggered);
|
||||
}
|
||||
updateTitle();
|
||||
}
|
||||
|
||||
function updateTitle() {
|
||||
const triggered = state.open.filter((i) => i.status === 'triggered').length;
|
||||
const section = SECTIONS[route.section].title;
|
||||
const section = title(route);
|
||||
const base = route.section === 'queue' && route.incident == null ? 'terdut' : `${section} · terdut`;
|
||||
document.title = triggered ? `(${triggered}) ${base}` : base;
|
||||
}
|
||||
@@ -138,9 +230,28 @@ async function boot() {
|
||||
document.addEventListener('click', interceptLinks);
|
||||
document.addEventListener('keydown', onKey);
|
||||
$('login-form').addEventListener('submit', onLogin);
|
||||
$('signup-form').addEventListener('submit', onSignup);
|
||||
$('nav-more-btn').addEventListener('click', openNavMenu);
|
||||
teamselector.init();
|
||||
ssoErrorCode = takeSSOError();
|
||||
|
||||
// /signup is the one route that works without a session.
|
||||
if (location.pathname.replace(/\/$/, '') === '/signup') {
|
||||
$('boot').hidden = true;
|
||||
await showSignup();
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
// Before anything renders: the account view decides from it whether a
|
||||
// password is worth offering to set.
|
||||
await loadAuthConfig();
|
||||
state.me = await api.me();
|
||||
await loadTeams();
|
||||
teamselector.render();
|
||||
// The Admin tab exists only for an administrator. Somebody who types /admin
|
||||
// anyway gets the view's own "ask an administrator" card, not a blank page.
|
||||
$('nav-admin').hidden = !state.me?.user?.is_admin;
|
||||
showApp();
|
||||
} catch (err) {
|
||||
if (err.status === 401) showLogin();
|
||||
@@ -153,15 +264,172 @@ function showBootError(err) {
|
||||
$('boot').append(ui.h('button', { class: 'btn', onclick: () => location.reload(), text: 'Retry' }));
|
||||
}
|
||||
|
||||
function showLogin() {
|
||||
// The sign-up screen. Reached at /signup, with an optional ?invite= that the
|
||||
// server has already judged — the form says whether the link is good before
|
||||
// somebody picks a password, rather than after.
|
||||
async function showSignup() {
|
||||
poll.stop();
|
||||
ui.closeSheet(null);
|
||||
reset();
|
||||
$('boot').hidden = true;
|
||||
$('app').hidden = true;
|
||||
$('login').hidden = false;
|
||||
const form = $('login-form');
|
||||
$('login-form').hidden = true;
|
||||
$('signup-form').hidden = false;
|
||||
|
||||
const invite = new URLSearchParams(location.search).get('invite');
|
||||
const intro = $('signup-intro');
|
||||
const form = $('signup-form');
|
||||
const teamLabel = $('signup-team-label');
|
||||
form.querySelector('.form-error').hidden = true;
|
||||
|
||||
let info;
|
||||
try {
|
||||
info = await api.signupInfo(invite);
|
||||
} catch (err) {
|
||||
intro.textContent = err.message;
|
||||
return;
|
||||
}
|
||||
|
||||
if (invite && info.invite_valid) {
|
||||
intro.textContent = `You have been invited to ${info.invite_team}.`;
|
||||
teamLabel.hidden = true;
|
||||
form.team_name.required = false;
|
||||
} else if (invite) {
|
||||
// One answer for expired, revoked, used up and never existed, matching the
|
||||
// server: which it was is not a stranger's business.
|
||||
intro.textContent = 'That invite link is not usable. Ask whoever sent it for a new one.';
|
||||
form.querySelector('button[type=submit]').disabled = true;
|
||||
} else if (info.mode === 'open') {
|
||||
intro.textContent = 'Create an account and a team to put your alerts in.';
|
||||
teamLabel.hidden = false;
|
||||
form.team_name.required = true;
|
||||
} else {
|
||||
intro.textContent = 'Sign-up on this server is invite-only. Ask a team owner for a link.';
|
||||
form.querySelector('button[type=submit]').disabled = true;
|
||||
}
|
||||
form.username.focus();
|
||||
}
|
||||
|
||||
async function onSignup(e) {
|
||||
e.preventDefault();
|
||||
const form = e.currentTarget;
|
||||
const err = form.querySelector('.form-error');
|
||||
const btn = form.querySelector('button[type=submit]');
|
||||
err.hidden = true;
|
||||
btn.disabled = true;
|
||||
try {
|
||||
state.me = await api.signup({
|
||||
username: form.username.value.trim(),
|
||||
email: form.email.value.trim(),
|
||||
password: form.password.value,
|
||||
invite: new URLSearchParams(location.search).get('invite') || undefined,
|
||||
team_name: form.team_name.value.trim() || undefined,
|
||||
});
|
||||
form.password.value = '';
|
||||
// Signing up signs you in, so go straight to the queue rather than to a
|
||||
// login form asking for the credential just chosen.
|
||||
history.replaceState({ depth: 0 }, '', '/');
|
||||
route = parseRoute('/');
|
||||
await loadTeams();
|
||||
teamselector.render();
|
||||
$('nav-admin').hidden = !state.me?.user?.is_admin;
|
||||
showApp();
|
||||
} catch (ex) {
|
||||
err.textContent = ex.message;
|
||||
err.hidden = false;
|
||||
} finally {
|
||||
btn.disabled = false;
|
||||
}
|
||||
}
|
||||
|
||||
// Why single sign-on sent the browser back, by the code the server puts in
|
||||
// ?sso_error=. The provider's name is the one the administrator configured.
|
||||
function ssoErrorText(code, name) {
|
||||
const sso = name || 'single sign-on';
|
||||
return {
|
||||
denied: `Signing in with ${sso} was cancelled or refused.`,
|
||||
expired: 'That sign-in expired or was already used. Try again.',
|
||||
failed: `Signing in with ${sso} failed. Try again, and tell an administrator if it keeps happening.`,
|
||||
unavailable: `${sso} could not be reached. Try again in a moment.`,
|
||||
not_allowed: 'Your account is not allowed to use terdut. Ask an administrator to add you to the right group.',
|
||||
no_email: `${sso} did not send an email address for you, which terdut needs.`,
|
||||
email_conflict: 'An account with your email address already exists and could not be linked to this sign-in. Ask an administrator.',
|
||||
disabled: 'Your account is disabled. Ask an administrator.',
|
||||
not_bootstrapped: 'This install is still setting up. Try again in a moment.',
|
||||
}[code] || `Signing in with ${sso} failed.`;
|
||||
}
|
||||
|
||||
// Reads how the server can be signed in to. An older server has no such
|
||||
// endpoint, and one that cannot be asked is treated as offering passwords only:
|
||||
// the form that always existed is better than a blank page.
|
||||
async function loadAuthConfig() {
|
||||
try {
|
||||
state.auth = await api.authConfig();
|
||||
} catch {
|
||||
/* keep the last answer, or the defaults */
|
||||
}
|
||||
return state.auth;
|
||||
}
|
||||
|
||||
// The reason the last single sign-on attempt failed, read once at boot. It is
|
||||
// held here rather than re-read from the address because showLogin runs more
|
||||
// than once on the way to the form (the 401 from /api/me reaches it through the
|
||||
// API layer and again through boot's own catch), and only the first would see it.
|
||||
let ssoErrorCode = null;
|
||||
|
||||
// Takes ?sso_error= off the address, so a reload does not repeat the message.
|
||||
function takeSSOError() {
|
||||
const params = new URLSearchParams(location.search);
|
||||
const code = params.get('sso_error');
|
||||
if (code === null) return null;
|
||||
params.delete('sso_error');
|
||||
const query = params.toString();
|
||||
history.replaceState(null, '', location.pathname + (query ? `?${query}` : '') + location.hash);
|
||||
return code;
|
||||
}
|
||||
|
||||
async function showLogin() {
|
||||
poll.stop();
|
||||
ui.closeSheet(null);
|
||||
reset();
|
||||
$('app').hidden = true;
|
||||
$('signup-form').hidden = true;
|
||||
|
||||
// Ask before showing anything, so the form does not flash the password
|
||||
// fields at somebody whose server has turned them off.
|
||||
const auth = await loadAuthConfig();
|
||||
const sso = auth.oidc?.enabled ? auth.oidc : null;
|
||||
const passwords = auth.password_login !== false;
|
||||
|
||||
const link = $('sso-link');
|
||||
link.hidden = !sso;
|
||||
if (sso) {
|
||||
link.textContent = `Sign in with ${sso.name || 'SSO'}`;
|
||||
// Come back to the page that was asked for: a link to /device?code=... has
|
||||
// to survive the trip through the provider. The server only honours paths
|
||||
// on this server, and ignores the front page.
|
||||
const here = location.pathname + location.search;
|
||||
link.href = here === '/' ? '/api/oidc/login' : `/api/oidc/login?next=${encodeURIComponent(here)}`;
|
||||
}
|
||||
$('login-or').hidden = !(sso && passwords);
|
||||
$('password-login').hidden = !passwords;
|
||||
|
||||
const ssoErr = $('sso-error');
|
||||
ssoErr.hidden = ssoErrorCode === null;
|
||||
if (ssoErrorCode !== null) ssoErr.textContent = ssoErrorText(ssoErrorCode, sso?.name);
|
||||
|
||||
$('boot').hidden = true;
|
||||
$('login').hidden = false;
|
||||
$('login-form').hidden = false;
|
||||
const form = $('login-form');
|
||||
form.querySelector('#password-login .form-error').hidden = true;
|
||||
if (!passwords) return;
|
||||
// Only offer the door that is open. Somebody without an invite on an
|
||||
// invite-only server should be told, not sent to a form that refuses them.
|
||||
api.signupInfo().then((info) => {
|
||||
$('signup-link').hidden = info.mode !== 'open';
|
||||
}).catch(() => {});
|
||||
form.password.value = '';
|
||||
(form.username.value ? form.password : form.username).focus();
|
||||
}
|
||||
@@ -169,9 +437,13 @@ function showLogin() {
|
||||
async function onLogin(e) {
|
||||
e.preventDefault();
|
||||
const form = e.currentTarget;
|
||||
const err = form.querySelector('.form-error');
|
||||
// Not the first .form-error: that one is the single sign-on message above.
|
||||
const err = form.querySelector('#password-login .form-error');
|
||||
const btn = form.querySelector('button[type=submit]');
|
||||
err.hidden = true;
|
||||
// Whatever single sign-on said is about the last attempt, not this one.
|
||||
ssoErrorCode = null;
|
||||
$('sso-error').hidden = true;
|
||||
btn.disabled = true;
|
||||
try {
|
||||
state.me = await api.login(form.username.value.trim(), form.password.value);
|
||||
@@ -195,6 +467,7 @@ export async function signOut() {
|
||||
}
|
||||
|
||||
function showApp() {
|
||||
ssoErrorCode = null;
|
||||
$('boot').hidden = true;
|
||||
$('login').hidden = true;
|
||||
$('app').hidden = false;
|
||||
|
||||
@@ -0,0 +1,88 @@
|
||||
// Approving a sign-in that a terminal started, at /device?code=XXXX-XXXX.
|
||||
//
|
||||
// The terminal (the TUI) shows a code and a link to this page. Whoever opens it
|
||||
// is already signed in — by the provider or by password, whichever the login
|
||||
// page offered — and is asked to approve. Approving hands that terminal a
|
||||
// session for *this* account, so the page names the account and the code, and
|
||||
// tells anybody who did not start this to refuse.
|
||||
|
||||
import * as api from './api.js';
|
||||
import { h, clear } from './ui.js';
|
||||
import { state } from './state.js';
|
||||
import { navigate } from './app.js';
|
||||
|
||||
const view = () => document.getElementById('view-device');
|
||||
|
||||
// What has been decided for the code on screen, so a re-render does not offer
|
||||
// to approve it a second time.
|
||||
let outcome = null; // { code, approved }
|
||||
|
||||
export function show() {
|
||||
render();
|
||||
}
|
||||
|
||||
function render() {
|
||||
const code = new URLSearchParams(location.search).get('code') || '';
|
||||
if (!code) return clear(view(), enterCode());
|
||||
if (outcome && outcome.code === code) return clear(view(), decided(outcome.approved));
|
||||
return clear(view(), confirmCard(code));
|
||||
}
|
||||
|
||||
// Reached without a code, for somebody who typed the address by hand.
|
||||
function enterCode() {
|
||||
const input = h('input', {
|
||||
name: 'code', autocomplete: 'off', autocapitalize: 'characters', spellcheck: 'false',
|
||||
placeholder: 'XXXX-XXXX', required: true,
|
||||
});
|
||||
const form = h('form', { class: 'card device-card' },
|
||||
h('h2', { text: 'Sign in a terminal' }),
|
||||
h('p', { class: 'muted', text: 'Enter the code the terminal is showing.' }),
|
||||
h('label', {}, h('span', { text: 'Code' }), input),
|
||||
h('button', { class: 'btn btn-primary', type: 'submit', text: 'Continue' }));
|
||||
form.addEventListener('submit', (e) => {
|
||||
e.preventDefault();
|
||||
navigate(`/device?code=${encodeURIComponent(input.value.trim())}`);
|
||||
});
|
||||
return form;
|
||||
}
|
||||
|
||||
function confirmCard(code) {
|
||||
const err = h('p', { class: 'form-error', role: 'alert', hidden: true });
|
||||
const approve = h('button', { class: 'btn btn-primary', type: 'button', text: 'Approve' });
|
||||
const refuse = h('button', { class: 'btn', type: 'button', text: 'Refuse' });
|
||||
|
||||
const decide = (approved) => async () => {
|
||||
err.hidden = true;
|
||||
approve.disabled = refuse.disabled = true;
|
||||
try {
|
||||
await (approved ? api.approveDevice(code) : api.denyDevice(code));
|
||||
outcome = { code, approved };
|
||||
render();
|
||||
} catch (ex) {
|
||||
err.textContent = ex.message;
|
||||
err.hidden = false;
|
||||
approve.disabled = refuse.disabled = false;
|
||||
}
|
||||
};
|
||||
approve.addEventListener('click', decide(true));
|
||||
refuse.addEventListener('click', decide(false));
|
||||
|
||||
return h('div', { class: 'card device-card' },
|
||||
h('h2', { text: 'Sign in a terminal?' }),
|
||||
h('p', {}, 'A terminal is asking to sign in as ', h('strong', { text: state.me.user.username }),
|
||||
'. Check that this code matches the one it is showing:'),
|
||||
h('p', { class: 'device-code', text: code }),
|
||||
h('p', { class: 'muted small',
|
||||
text: 'Only approve a sign-in you started yourself. Whoever is approved here acts as you.' }),
|
||||
err,
|
||||
h('div', { class: 'row-actions' }, approve, refuse));
|
||||
}
|
||||
|
||||
function decided(approved) {
|
||||
return h('div', { class: 'card device-card' },
|
||||
h('h2', { text: approved ? 'Approved' : 'Refused' }),
|
||||
h('p', { class: 'muted', text: approved
|
||||
? 'You can go back to your terminal. It signs in within a few seconds.'
|
||||
: 'That terminal will not be signed in.' }),
|
||||
h('a', { class: 'btn', href: '/', text: 'Go to the queue' }));
|
||||
}
|
||||
@@ -66,21 +66,22 @@ export function mondayOf(d) {
|
||||
return r;
|
||||
}
|
||||
|
||||
// ISO 8601 week number: weeks start on Monday and week 1 is the one holding the
|
||||
// year's first Thursday, which is what a rota that runs Monday to Sunday means
|
||||
// by "week 40". Taken from the Thursday of d's week, whose year is the week's.
|
||||
export function isoWeek(d) {
|
||||
const thu = new Date(d.getFullYear(), d.getMonth(), d.getDate());
|
||||
thu.setDate(thu.getDate() + 3 - ((thu.getDay() + 6) % 7));
|
||||
const jan4 = new Date(thu.getFullYear(), 0, 4);
|
||||
return 1 + Math.round(((thu - jan4) / 86400000 - 3 + ((jan4.getDay() + 6) % 7)) / 7);
|
||||
}
|
||||
|
||||
export function addDays(d, n) {
|
||||
const r = new Date(d);
|
||||
r.setDate(r.getDate() + n);
|
||||
return r;
|
||||
}
|
||||
|
||||
// ISO 8601 week number.
|
||||
export function isoWeek(d) {
|
||||
const t = new Date(Date.UTC(d.getFullYear(), d.getMonth(), d.getDate()));
|
||||
const day = t.getUTCDay() || 7;
|
||||
t.setUTCDate(t.getUTCDate() + 4 - day);
|
||||
const yearStart = new Date(Date.UTC(t.getUTCFullYear(), 0, 1));
|
||||
return Math.ceil(((t - yearStart) / DAY + 1) / 7);
|
||||
}
|
||||
|
||||
export const STATUS_LABEL = {
|
||||
triggered: 'Triggered',
|
||||
acknowledged: 'Acknowledged',
|
||||
@@ -97,6 +98,14 @@ export function severityClass(sev) {
|
||||
return '';
|
||||
}
|
||||
|
||||
// A stable identity colour for a team, so the same team always reads the same
|
||||
// colour without the server needing to store one. Teams have no colour field;
|
||||
// this hashes the id into the six-colour rcN palette app.css already has for
|
||||
// the rota's per-person chips (a team is not a status, so never severity).
|
||||
export function teamColorClass(teamID) {
|
||||
return `rc${(((teamID % 6) + 6) % 6) + 1}`;
|
||||
}
|
||||
|
||||
// A one-line summary of the group labels, without the one the title already shows.
|
||||
export function labelSummary(labels, skip = 'alertname') {
|
||||
return Object.entries(labels || {})
|
||||
|
||||
@@ -7,7 +7,7 @@ import {
|
||||
h, clear, icon, badge, labelChip, openSheet, closeSheet, confirm, toast, spinner, emptyState,
|
||||
} from './ui.js';
|
||||
import {
|
||||
ago, when, until, isFuture, severityClass, STATUS_LABEL,
|
||||
ago, when, until, duration, isFuture, severityClass, STATUS_LABEL,
|
||||
} from './format.js';
|
||||
import { myID, users } from './state.js';
|
||||
import { back } from './app.js';
|
||||
@@ -17,6 +17,7 @@ const pane = () => document.getElementById('detail');
|
||||
let currentID = null;
|
||||
let inc = null;
|
||||
let events = [];
|
||||
let similarList = [];
|
||||
let error = null;
|
||||
let busy = false;
|
||||
|
||||
@@ -38,10 +39,15 @@ export async function refresh() {
|
||||
const id = currentID;
|
||||
if (id == null) return;
|
||||
try {
|
||||
const [i, t] = await Promise.all([api.incident(id), api.timeline(id)]);
|
||||
// Similar incidents are a courtesy: an older server answers 404 and a
|
||||
// failure here must not hide the incident itself.
|
||||
const [i, t, sim] = await Promise.all([
|
||||
api.incident(id), api.timeline(id), api.similar(id).catch(() => []),
|
||||
]);
|
||||
if (id !== currentID) return;
|
||||
inc = i;
|
||||
events = t;
|
||||
similarList = sim;
|
||||
error = null;
|
||||
} catch (err) {
|
||||
if (id !== currentID) return;
|
||||
@@ -60,6 +66,8 @@ function render() {
|
||||
h('button', { class: 'btn btn-ghost btn-icon back', type: 'button', 'aria-label': 'Back to queue', onclick: back },
|
||||
icon('back')),
|
||||
h('span', { class: 'crumb', text: currentID != null ? `Incident #${currentID}` : '' }),
|
||||
inc && h('button', { class: 'btn btn-ghost btn-icon copy', type: 'button', 'aria-label': 'Copy incident', title: 'Copy incident (y)', onclick: copyIncident },
|
||||
icon('copy')),
|
||||
);
|
||||
|
||||
if (!inc) {
|
||||
@@ -78,9 +86,11 @@ function render() {
|
||||
error && h('div', { class: 'load-error', text: `Showing older data: ${error}` }),
|
||||
h('h1', { class: 'detail-title', text: inc.title }),
|
||||
h('div', { class: 'detail-badges' }, statusBadges()),
|
||||
quickActions(),
|
||||
facts(),
|
||||
groupLabels(),
|
||||
alertsSection(),
|
||||
similarSection(),
|
||||
timelineSection(),
|
||||
),
|
||||
actionBar(),
|
||||
@@ -95,6 +105,15 @@ function statusBadges() {
|
||||
if (inc.status !== 'resolved' && isFuture(inc.snoozed_until)) {
|
||||
out.push(badge(`Snoozed · ${until(inc.snoozed_until)} left`, 'st-snoozed'));
|
||||
}
|
||||
// Where it is on the ladder, while it is still climbing. The queue shows
|
||||
// what happened; this says what happens next, which is the question somebody
|
||||
// looking at an unacknowledged incident actually has.
|
||||
if (inc.escalation_level > 0) {
|
||||
const left = inc.escalation_due_at && isFuture(inc.escalation_due_at)
|
||||
? ` · next in ${until(inc.escalation_due_at)}`
|
||||
: ' · next page due';
|
||||
out.push(badge(`Escalating · level ${inc.escalation_level}${left}`, 'st-triggered'));
|
||||
}
|
||||
if (inc.archived_at) out.push(badge('Archived', 'plain'));
|
||||
return out;
|
||||
}
|
||||
@@ -107,6 +126,20 @@ function who(id, name) {
|
||||
function facts() {
|
||||
const rows = [];
|
||||
const add = (k, ...v) => rows.push(h('dt', { text: k }), h('dd', {}, ...v));
|
||||
|
||||
// Duration, severity and who's on it, in one scannable row up top — the
|
||||
// rest of this card has each of those too, but spread across rows that
|
||||
// take reading top to bottom to piece together.
|
||||
const elapsedTo = inc.resolved_at ? Date.parse(inc.resolved_at) : Date.now();
|
||||
const responsible = inc.assigned_to_id != null ? who(inc.assigned_to_id, inc.assigned_to)
|
||||
: inc.acknowledged_by_id != null ? who(inc.acknowledged_by_id, inc.acknowledged_by)
|
||||
: 'Unassigned';
|
||||
rows.push(h('dt', { text: 'At a glance' }), h('dd', { class: 'fact-summary' },
|
||||
h('span', { class: 'fact-chip' }, icon('clock', 'icon fact-icon'), duration(elapsedTo - Date.parse(inc.triggered_at))),
|
||||
inc.severity && badge(inc.severity, `plain ${severityClass(inc.severity)}`),
|
||||
h('span', { class: 'fact-chip' }, icon('user', 'icon fact-icon'), responsible),
|
||||
));
|
||||
|
||||
add('Triggered', when(inc.triggered_at), h('span', { class: 'sub', text: ` · ${ago(inc.triggered_at)}` }));
|
||||
if (inc.acknowledged_at) {
|
||||
add('Acknowledged', `${who(inc.acknowledged_by_id, inc.acknowledged_by)} · ${when(inc.acknowledged_at)}`);
|
||||
@@ -115,6 +148,10 @@ function facts() {
|
||||
if (inc.status !== 'resolved' && isFuture(inc.snoozed_until)) {
|
||||
add('Snoozed until', when(inc.snoozed_until));
|
||||
}
|
||||
if (inc.escalation_level > 0 && inc.escalation_due_at) {
|
||||
add('Escalates next', when(inc.escalation_due_at),
|
||||
h('span', { class: 'sub', text: ` · level ${inc.escalation_level}` }));
|
||||
}
|
||||
if (inc.resolved_at) {
|
||||
const how = inc.resolution_source === 'manual' ? 'by hand' : 'alerts stopped firing';
|
||||
add('Resolved', when(inc.resolved_at), h('span', { class: 'sub', text: ` · ${how}` }));
|
||||
@@ -126,9 +163,12 @@ function facts() {
|
||||
function groupLabels() {
|
||||
const entries = Object.entries(inc.group_labels || {});
|
||||
if (!entries.length) return null;
|
||||
// Collapsed by default, the same disclosure alertItem() below uses for an
|
||||
// alert's own labels — this is background, not something to scan past.
|
||||
return h('section', { class: 'section' },
|
||||
h('h2', { class: 'section-title', text: 'Grouped by' }),
|
||||
h('div', { class: 'labels-wrap' }, entries.map(([k, v]) => labelChip(k, v))),
|
||||
h('details', {},
|
||||
h('summary', { text: `Grouped by (${entries.length})` }),
|
||||
h('div', { class: 'labels-wrap' }, entries.map(([k, v]) => labelChip(k, v)))),
|
||||
);
|
||||
}
|
||||
|
||||
@@ -166,8 +206,9 @@ function alertItem(a) {
|
||||
|
||||
// ---------- timeline ----------
|
||||
|
||||
function eventText(ev) {
|
||||
const person = ev.user_id != null ? who(ev.user_id, ev.username) : null;
|
||||
// named spells users out instead of "you", for text that leaves this page.
|
||||
function eventText(ev, named = false) {
|
||||
const person = ev.user_id != null ? (named ? ev.username || 'someone' : who(ev.user_id, ev.username)) : null;
|
||||
const strong = (t) => h('span', { class: 'who', text: t || 'someone' });
|
||||
const alertName = () => {
|
||||
const a = (inc.alerts || []).find((x) => x.id === ev.alert_id);
|
||||
@@ -183,11 +224,20 @@ function eventText(ev) {
|
||||
case 'snoozed': return [strong(person), ` snoozed until ${ev.detail ? when(ev.detail) : '…'}`];
|
||||
case 'unsnoozed': return [strong(person), ' ended the snooze'];
|
||||
case 'resolved': return person ? [strong(person), ' resolved the incident'] : ['Resolved: every alert stopped firing'];
|
||||
// Falls through to the generic `${ev.type}: ${ev.detail}` below otherwise
|
||||
// — this just capitalises it and drops the redundant "escalated:" prefix
|
||||
// from detail (already "level 2: alice, bob" or "escalation exhausted: …").
|
||||
case 'escalated': return [`Escalated — ${ev.detail}`];
|
||||
case 'note': return [strong(person), ' added a note'];
|
||||
case 'resolution_note': return [strong(person), ' noted what fixed it'];
|
||||
case 'notified': {
|
||||
const to = person ? strong(person) : 'the fallback topic';
|
||||
if (ev.detail === 'reminder') return ['Reminder sent to ', to];
|
||||
if (ev.detail === 'resolved') return ['Resolution sent to ', to];
|
||||
// 'escalated' is a second, later page — the next level firing, not the
|
||||
// same page landing twice — so it reads as a bug unless told apart
|
||||
// from the initial 'triggered' page below.
|
||||
if (ev.detail === 'escalated') return ['Escalation paged ', to];
|
||||
return ['Paged ', to];
|
||||
}
|
||||
case 'notify_failed': return ['Notification failed', ev.detail ? `: ${ev.detail}` : ''];
|
||||
@@ -196,8 +246,50 @@ function eventText(ev) {
|
||||
}
|
||||
}
|
||||
|
||||
// Earlier incidents with the same signature that someone left notes on, the
|
||||
// ones that recorded what fixed it first. Plain notes are on that incident's
|
||||
// own page.
|
||||
function similarSection() {
|
||||
if (!similarList.length) return null;
|
||||
return h('section', { class: 'section' },
|
||||
h('h2', { class: 'section-title' }, h('span', { text: 'Seen before' })),
|
||||
h('div', { class: 'card' },
|
||||
h('ul', { class: 'similar' }, similarList.map((s) => h('li', { class: 'similar-item' },
|
||||
h('a', { href: `/incidents/${s.id}`, text: `#${s.id} ${s.title}` }),
|
||||
h('div', { class: 'sub', text: `${when(s.resolved_at)} · ${ago(s.resolved_at)}${s.note_count ? ` · ${s.note_count} note${s.note_count === 1 ? '' : 's'}` : ''}` }),
|
||||
...s.resolution_notes.map((n) => h('div', { class: 'note note-fix', text: n.detail || '' })),
|
||||
)))),
|
||||
);
|
||||
}
|
||||
|
||||
// Splits the already-sorted timeline on the status transitions that matter —
|
||||
// first acknowledged, then resolved — so a long incident reads as "before
|
||||
// anyone had it" / "while someone did" / "after it closed" instead of one
|
||||
// undifferentiated list. A later re-acknowledge (after an unacknowledge)
|
||||
// doesn't open a second "Acknowledged" phase; it's still the same spell of
|
||||
// somebody owning it.
|
||||
function timelinePhases(sorted) {
|
||||
const phases = [{ label: 'Triggered', events: [] }];
|
||||
let acked = false;
|
||||
for (const ev of sorted) {
|
||||
if (ev.type === 'acknowledged' && !acked) {
|
||||
phases.push({ label: 'Acknowledged', events: [] });
|
||||
acked = true;
|
||||
} else if (ev.type === 'resolved') {
|
||||
phases.push({ label: 'Resolved', events: [] });
|
||||
}
|
||||
phases[phases.length - 1].events.push(ev);
|
||||
}
|
||||
return phases.filter((p) => p.events.length);
|
||||
}
|
||||
|
||||
function timelineSection() {
|
||||
const sorted = [...events].sort((a, b) => Date.parse(a.created_at) - Date.parse(b.created_at) || a.id - b.id);
|
||||
const phases = timelinePhases(sorted);
|
||||
// A single phase (the common case: most incidents are acked once and
|
||||
// resolved) names nothing extra — only a split timeline needs the
|
||||
// headings to make sense of.
|
||||
const named = phases.length > 1;
|
||||
return h('section', { class: 'section' },
|
||||
h('h2', { class: 'section-title' },
|
||||
h('span', { text: 'Timeline' }),
|
||||
@@ -205,51 +297,165 @@ function timelineSection() {
|
||||
icon('note'), 'Add note')),
|
||||
h('div', { class: 'card' },
|
||||
sorted.length
|
||||
? h('ol', { class: 'timeline' }, sorted.map(timelineItem))
|
||||
: emptyState('No events yet', '')),
|
||||
? phases.map((p) => h('div', { class: 'tl-phase' },
|
||||
named && h('div', { class: 'tl-phase-title', text: p.label }),
|
||||
h('ol', { class: 'timeline' }, p.events.map(timelineItem))))
|
||||
: emptyState('No events yet', 'Nothing has happened on this incident yet.', 'clock')),
|
||||
);
|
||||
}
|
||||
|
||||
const isNote = (ev) => ev.type === 'note' || ev.type === 'resolution_note';
|
||||
|
||||
function timelineItem(ev) {
|
||||
const mine = ev.type === 'note' && ev.user_id === myID();
|
||||
const mine = isNote(ev) && ev.user_id === myID();
|
||||
return h('li', { class: `tl-item tl-${ev.type}` },
|
||||
h('span', { class: 'tl-dot' }),
|
||||
h('div', { class: 'tl-body' },
|
||||
h('div', { class: 'tl-text' }, eventText(ev)),
|
||||
h('div', { class: 'tl-time', title: ev.created_at, text: `${when(ev.created_at)} · ${ago(ev.created_at)}` }),
|
||||
ev.type === 'note' && h('div', { class: 'note', text: ev.detail || '' }),
|
||||
isNote(ev) && h('div', { class: ev.type === 'resolution_note' ? 'note note-fix' : 'note', text: ev.detail || '' }),
|
||||
mine && h('div', { class: 'note-actions' },
|
||||
h('button', { class: 'btn btn-ghost btn-sm', type: 'button', onclick: () => deleteNote(ev) }, icon('trash'), 'Delete')),
|
||||
),
|
||||
);
|
||||
}
|
||||
|
||||
// ---------- copy ----------
|
||||
|
||||
const fence = (rows) => ['```', ...rows, '```'];
|
||||
const pairs = (obj) => Object.entries(obj || {}).sort(([a], [b]) => a.localeCompare(b)).map(([k, v]) => `${k}=${v}`);
|
||||
|
||||
// incidentMarkdown is everything on this page as text that reads well in a chat
|
||||
// or an agent prompt. Times are ISO 8601, since "3 min ago" means nothing once
|
||||
// it has been pasted somewhere else.
|
||||
function incidentMarkdown() {
|
||||
const out = [`# Incident #${inc.id}: ${inc.title}`, ''];
|
||||
const add = (k, v) => { if (v != null && v !== '') out.push(`- ${k}: ${v}`); };
|
||||
add('Status', inc.status);
|
||||
add('Severity', inc.severity);
|
||||
add('Team', inc.team_name);
|
||||
add('Assigned to', inc.assigned_to_id != null ? inc.assigned_to || 'someone' : 'unassigned');
|
||||
add('Triggered', inc.triggered_at);
|
||||
if (inc.acknowledged_at) add('Acknowledged', `${inc.acknowledged_at} by ${inc.acknowledged_by || 'someone'}`);
|
||||
if (isOpen() && isFuture(inc.snoozed_until)) add('Snoozed until', inc.snoozed_until);
|
||||
if (inc.escalation_level > 0) add('Escalation level', inc.escalation_level);
|
||||
if (inc.resolved_at) add('Resolved', `${inc.resolved_at} (${inc.resolution_source === 'manual' ? 'manually' : 'all alerts stopped firing'})`);
|
||||
if (inc.archived_at) add('Archived', inc.archived_at);
|
||||
const group = pairs(inc.group_labels);
|
||||
if (group.length) out.push('- Grouped by:', ...group.map((g) => ` - ${g}`));
|
||||
|
||||
const alerts = inc.alerts || [];
|
||||
out.push('', `## Alerts (${alerts.length})`);
|
||||
for (const a of alerts) {
|
||||
out.push('', `### ${a.name} (${a.status})`);
|
||||
out.push(`- Started: ${a.starts_at}`);
|
||||
if (a.status === 'resolved' && a.ends_at) out.push(`- Ended: ${a.ends_at}`);
|
||||
if (a.generator_url) out.push(`- Source: ${a.generator_url}`);
|
||||
const labels = pairs(a.labels);
|
||||
if (labels.length) out.push('', 'Labels:', ...fence(labels));
|
||||
const annotations = Object.entries(a.annotations || {}).sort(([x], [y]) => x.localeCompare(y));
|
||||
if (annotations.length) out.push('', 'Annotations:', ...fence(annotations.map(([k, v]) => `${k}: ${v}`)));
|
||||
}
|
||||
|
||||
const sorted = [...events].sort((a, b) => Date.parse(a.created_at) - Date.parse(b.created_at) || a.id - b.id);
|
||||
if (sorted.length) {
|
||||
out.push('', '## Timeline', '');
|
||||
for (const ev of sorted) {
|
||||
const text = eventText(ev, true).map((f) => (f instanceof Node ? f.textContent : f)).join('');
|
||||
out.push(`- ${ev.created_at} ${text}`);
|
||||
if (isNote(ev) && ev.detail) {
|
||||
const label = ev.type === 'resolution_note' ? ' (what fixed it)' : '';
|
||||
out.push(...(label ? [label] : []), ...ev.detail.split('\n').map((l) => ` > ${l}`));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (similarList.length) {
|
||||
out.push('', '## Seen before', '', 'Earlier incidents with the same signature:');
|
||||
for (const s of similarList) {
|
||||
out.push(`- #${s.id} ${s.title} (resolved ${s.resolved_at})`);
|
||||
for (const n of s.resolution_notes || []) {
|
||||
out.push(' - What fixed it:', ...(n.detail || '').split('\n').map((l) => ` > ${l}`));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
out.push('', `_Copied from Terminal Duty at ${new Date().toISOString()}_`, '');
|
||||
return out.join('\n');
|
||||
}
|
||||
|
||||
// writeClipboard falls back to execCommand: the async API needs a secure
|
||||
// context, and this server is often reached over plain HTTP.
|
||||
async function writeClipboard(text) {
|
||||
try {
|
||||
await navigator.clipboard.writeText(text);
|
||||
return;
|
||||
} catch {
|
||||
// fall through
|
||||
}
|
||||
const ta = h('textarea', { readonly: true, 'aria-hidden': 'true', class: 'clip-buffer' });
|
||||
ta.value = text;
|
||||
document.body.append(ta);
|
||||
ta.select();
|
||||
try {
|
||||
if (!document.execCommand('copy')) throw new Error('copy refused');
|
||||
} finally {
|
||||
ta.remove();
|
||||
}
|
||||
}
|
||||
|
||||
async function copyIncident() {
|
||||
if (!inc) return;
|
||||
try {
|
||||
await writeClipboard(incidentMarkdown());
|
||||
toast('Copied incident');
|
||||
} catch {
|
||||
toast('Could not copy', 'error');
|
||||
}
|
||||
}
|
||||
|
||||
// ---------- actions ----------
|
||||
|
||||
const isOpen = () => inc.status !== 'resolved';
|
||||
const isSnoozed = () => isOpen() && isFuture(inc.snoozed_until);
|
||||
|
||||
function actionBar() {
|
||||
let primary;
|
||||
let secondary;
|
||||
// primaryAction and secondaryAction are factories, not shared nodes — a
|
||||
// button can only live in one place, and quickActions() below needs its own
|
||||
// copy of the primary one rather than the actionbar's.
|
||||
function primaryAction() {
|
||||
if (inc.status === 'triggered') {
|
||||
primary = h('button', { class: 'btn btn-primary', type: 'button', onclick: acknowledge }, icon('check'), 'Acknowledge');
|
||||
} else if (inc.status === 'acknowledged') {
|
||||
primary = h('button', { class: 'btn btn-primary', type: 'button', onclick: resolve }, icon('checkCircle'), 'Resolve');
|
||||
} else {
|
||||
primary = inc.archived_at
|
||||
? h('button', { class: 'btn btn-primary', type: 'button', onclick: unarchive }, icon('undo'), 'Unarchive')
|
||||
: h('button', { class: 'btn btn-primary', type: 'button', onclick: archive }, icon('archive'), 'Archive');
|
||||
return h('button', { class: 'btn btn-primary', type: 'button', onclick: acknowledge }, icon('check'), 'Acknowledge');
|
||||
}
|
||||
if (inc.status === 'acknowledged') {
|
||||
return h('button', { class: 'btn btn-primary', type: 'button', onclick: resolve }, icon('checkCircle'), 'Resolve');
|
||||
}
|
||||
return inc.archived_at
|
||||
? h('button', { class: 'btn btn-primary', type: 'button', onclick: unarchive }, icon('undo'), 'Unarchive')
|
||||
: h('button', { class: 'btn btn-primary', type: 'button', onclick: archive }, icon('archive'), 'Archive');
|
||||
}
|
||||
|
||||
function secondaryAction() {
|
||||
if (isOpen()) {
|
||||
secondary = isSnoozed()
|
||||
return isSnoozed()
|
||||
? h('button', { class: 'btn', type: 'button', onclick: unsnooze }, icon('bell'), 'Unsnooze')
|
||||
: h('button', { class: 'btn', type: 'button', onclick: snooze }, icon('clock'), 'Snooze');
|
||||
} else {
|
||||
secondary = h('button', { class: 'btn', type: 'button', onclick: addNote }, icon('note'), 'Note');
|
||||
}
|
||||
const more = h('button', { class: 'btn btn-icon', type: 'button', 'aria-label': 'More actions', onclick: moreMenu }, icon('more'));
|
||||
const bar = h('div', { class: 'actionbar' }, primary, secondary, more);
|
||||
return h('button', { class: 'btn', type: 'button', onclick: addNote }, icon('note'), 'Note');
|
||||
}
|
||||
|
||||
// A copy of the primary action (Acknowledge/Resolve/…) up where it's seen
|
||||
// right away, next to the status it responds to. The sticky actionbar below
|
||||
// keeps carrying every action, primary included, for whenever the page has
|
||||
// been scrolled past it.
|
||||
function quickActions() {
|
||||
const div = h('div', { class: 'detail-quick-actions' }, primaryAction());
|
||||
if (busy) for (const b of div.querySelectorAll('button')) b.disabled = true;
|
||||
return div;
|
||||
}
|
||||
|
||||
function actionBar() {
|
||||
const more = h('button', { class: 'btn', type: 'button', 'aria-label': 'More actions', onclick: moreMenu }, icon('more'), 'More');
|
||||
const bar = h('div', { class: 'actionbar' }, primaryAction(), secondaryAction(), more);
|
||||
if (busy) for (const b of bar.querySelectorAll('button')) b.disabled = true;
|
||||
return bar;
|
||||
}
|
||||
@@ -283,15 +489,35 @@ function unacknowledge() {
|
||||
|
||||
async function resolve() {
|
||||
const id = inc.id;
|
||||
const ok = await confirm({
|
||||
title: 'Resolve this incident?',
|
||||
text: 'Resolving is final. If these alerts fire again they open a new incident, '
|
||||
+ 'and if any are still firing this one stays closed regardless. '
|
||||
+ 'Use snooze if you only need it out of the way.',
|
||||
confirmLabel: 'Resolve',
|
||||
danger: true,
|
||||
const res = await openSheet(() => {
|
||||
const textarea = h('textarea', {
|
||||
name: 'resolution', autofocus: true, maxlength: '10000',
|
||||
placeholder: 'What fixed it? Optional, shown on the next similar incident.',
|
||||
});
|
||||
const form = h('form', {
|
||||
class: 'sheet-form',
|
||||
onsubmit: (e) => {
|
||||
e.preventDefault();
|
||||
closeSheet({ resolution: textarea.value.trim() });
|
||||
},
|
||||
},
|
||||
h('h2', { class: 'sheet-title', text: 'Resolve this incident?' }),
|
||||
h('p', {
|
||||
text: 'Resolving is final. If these alerts fire again they open a new incident, '
|
||||
+ 'and if any are still firing this one stays closed regardless. '
|
||||
+ 'Use snooze if you only need it out of the way.',
|
||||
}),
|
||||
textarea,
|
||||
h('div', { class: 'sheet-actions' },
|
||||
h('button', { class: 'btn', type: 'button', onclick: () => closeSheet(null), text: 'Cancel' }),
|
||||
h('button', { class: 'btn btn-danger', type: 'submit', text: 'Resolve' })),
|
||||
);
|
||||
textarea.addEventListener('keydown', (e) => {
|
||||
if (e.key === 'Enter' && (e.ctrlKey || e.metaKey)) form.requestSubmit();
|
||||
});
|
||||
return form;
|
||||
});
|
||||
if (ok) await run(() => api.resolve(id), 'Resolved');
|
||||
if (res) await run(() => api.resolve(id, res.resolution), 'Resolved');
|
||||
}
|
||||
|
||||
function archive() {
|
||||
@@ -369,16 +595,18 @@ async function addNote() {
|
||||
const textarea = h('textarea', {
|
||||
name: 'content', required: true, autofocus: true, placeholder: 'What did you find? What did you do?', maxlength: '10000',
|
||||
});
|
||||
const fix = h('input', { type: 'checkbox', name: 'fix' });
|
||||
const form = h('form', {
|
||||
class: 'sheet-form',
|
||||
onsubmit: (e) => {
|
||||
e.preventDefault();
|
||||
const v = textarea.value.trim();
|
||||
if (v) closeSheet(v);
|
||||
if (v) closeSheet({ content: v, pinned: fix.checked });
|
||||
},
|
||||
},
|
||||
h('h2', { class: 'sheet-title', text: 'Add note' }),
|
||||
textarea,
|
||||
h('label', { class: 'check' }, fix, ' This is what fixed it (shown on similar incidents)'),
|
||||
h('div', { class: 'sheet-actions' },
|
||||
h('button', { class: 'btn', type: 'button', onclick: () => closeSheet(null), text: 'Cancel' }),
|
||||
h('button', { class: 'btn btn-primary', type: 'submit', text: 'Save note' })),
|
||||
@@ -389,7 +617,7 @@ async function addNote() {
|
||||
});
|
||||
return form;
|
||||
});
|
||||
if (content) await run(() => api.addNote(id, content), 'Note added');
|
||||
if (content) await run(() => api.addNote(id, content.content, content.pinned), 'Note added');
|
||||
}
|
||||
|
||||
async function deleteNote(ev) {
|
||||
@@ -409,10 +637,12 @@ async function moreMenu() {
|
||||
items.push(item('user', 'Assign…', assign));
|
||||
items.push(isSnoozed() ? item('bell', 'End snooze', unsnooze) : item('clock', 'Snooze…', snooze));
|
||||
items.push(item('note', 'Add note…', addNote));
|
||||
items.push(item('copy', 'Copy incident', copyIncident));
|
||||
items.push(h('li', { class: 'menu-sep', role: 'separator' }));
|
||||
items.push(item('checkCircle', 'Resolve…', resolve, 'danger'));
|
||||
} else {
|
||||
items.push(item('note', 'Add note…', addNote));
|
||||
items.push(item('copy', 'Copy incident', copyIncident));
|
||||
items.push(inc.archived_at ? item('undo', 'Unarchive', unarchive) : item('archive', 'Archive', archive));
|
||||
}
|
||||
|
||||
@@ -441,6 +671,7 @@ export function key(e) {
|
||||
case 'z': if (isOpen() && !isSnoozed()) snooze(); return true;
|
||||
case 'Z': if (isSnoozed()) unsnooze(); return true;
|
||||
case 'c': addNote(); return true;
|
||||
case 'y': copyIncident(); return true;
|
||||
case 'x': if (!isOpen()) (inc.archived_at ? unarchive() : archive()); return true;
|
||||
default: return false;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,147 @@
|
||||
// The first-run checklist: the four things a new install or a new person has
|
||||
// to do before an alert reaches a phone.
|
||||
//
|
||||
// It is computed from what the server already knows rather than from stored
|
||||
// progress — a topic is set or it is not, an integration exists or it does not
|
||||
// — so it cannot claim a step is done when it is not, and it comes back by
|
||||
// itself if somebody deletes their integration a month later.
|
||||
//
|
||||
// Dismissal is the one piece of state, kept per user so finishing on a laptop
|
||||
// does not leave the phone nagging.
|
||||
|
||||
import * as api from './api.js';
|
||||
import { h, clear, spinner } from './ui.js';
|
||||
import { state, currentTeam } from './state.js';
|
||||
import { navigate } from './app.js';
|
||||
import { isoDate } from './format.js';
|
||||
|
||||
let steps = null;
|
||||
let error = null;
|
||||
let busy = false;
|
||||
let testResult = null;
|
||||
|
||||
// done() is deliberately a question about the world, not a flag: each step asks
|
||||
// the data whether it happened.
|
||||
export async function load() {
|
||||
const team = currentTeam();
|
||||
if (!team) {
|
||||
steps = null;
|
||||
return;
|
||||
}
|
||||
try {
|
||||
const [schedule, integrations, alerts] = await Promise.all([
|
||||
api.schedule(team.id, isoDate(new Date()), isoDate(new Date())),
|
||||
api.integrations(team.id),
|
||||
api.alerts({ limit: 1 }),
|
||||
]);
|
||||
steps = [
|
||||
{
|
||||
id: 'topic',
|
||||
title: 'Set where your pages go',
|
||||
text: 'An ntfy topic on your account. Without one, incidents assigned to you page the team’s fallback topic instead of your phone.',
|
||||
done: Boolean(state.me?.user?.ntfy_topic),
|
||||
action: { label: 'Account', go: '/more' },
|
||||
},
|
||||
{
|
||||
id: 'rota',
|
||||
title: 'Put somebody on call',
|
||||
text: 'An incident opens assigned to whoever the rota says is on call today. With an empty rota it opens unassigned.',
|
||||
done: (schedule || []).length > 0,
|
||||
action: { label: 'Team', go: '/team' },
|
||||
},
|
||||
{
|
||||
id: 'integration',
|
||||
title: 'Create an alert source',
|
||||
text: 'Alerts arrive on an integration key, which says which team they belong to. Nothing can reach this team without one.',
|
||||
done: (integrations || []).length > 0,
|
||||
action: { label: 'Team', go: '/team' },
|
||||
},
|
||||
{
|
||||
id: 'alert',
|
||||
title: 'Send a test alert',
|
||||
text: 'Post to the integration URL and watch it appear in the queue. Until one arrives, none of the above is proven.',
|
||||
done: (alerts || []).length > 0,
|
||||
action: { label: 'How', go: '/team' },
|
||||
},
|
||||
];
|
||||
error = null;
|
||||
} catch (err) {
|
||||
error = err.message;
|
||||
}
|
||||
}
|
||||
|
||||
// visible reports whether there is anything worth showing: something undone,
|
||||
// and not dismissed.
|
||||
export function visible() {
|
||||
if (!steps || state.me?.onboarding_dismissed) return false;
|
||||
return steps.some((s) => !s.done);
|
||||
}
|
||||
|
||||
export function card() {
|
||||
if (!visible()) return null;
|
||||
const remaining = steps.filter((s) => !s.done).length;
|
||||
|
||||
return h('div', { class: 'card onboarding' },
|
||||
h('div', { class: 'onboarding-head' },
|
||||
h('h2', { text: 'Finish setting up' }),
|
||||
h('span', { class: 'muted small', text: `${remaining} left` }),
|
||||
h('button', {
|
||||
class: 'btn-sm', type: 'button', text: 'Hide',
|
||||
title: 'Hide this checklist for good',
|
||||
onclick: async () => {
|
||||
try {
|
||||
await api.dismissOnboarding(true);
|
||||
if (state.me) state.me.onboarding_dismissed = true;
|
||||
} catch (err) {
|
||||
error = err.message;
|
||||
}
|
||||
rerender();
|
||||
},
|
||||
})),
|
||||
error && h('p', { class: 'load-error', text: error }),
|
||||
h('ol', { class: 'checklist' }, ...steps.map(stepRow)),
|
||||
testResult && h('p', { class: testResult.ok ? 'muted small' : 'load-error', text: testResult.text }),
|
||||
);
|
||||
}
|
||||
|
||||
function stepRow(step) {
|
||||
return h('li', { class: step.done ? 'step done' : 'step' },
|
||||
h('span', { class: 'step-mark', text: step.done ? '✓' : '' }),
|
||||
h('div', {},
|
||||
h('strong', { text: step.title }),
|
||||
h('p', { class: 'muted small', text: step.text }),
|
||||
!step.done && h('div', { class: 'step-actions' },
|
||||
h('button', {
|
||||
class: 'btn-sm', type: 'button', text: step.action.label,
|
||||
onclick: () => navigate(step.action.go),
|
||||
}),
|
||||
// The topic step is the only one this page can finish by itself, and
|
||||
// the only proof that matters is a phone buzzing.
|
||||
step.id === 'topic' && state.me?.user?.ntfy_topic && h('button', {
|
||||
class: 'btn-sm', type: 'button', text: 'Send a test push',
|
||||
disabled: busy,
|
||||
onclick: sendTest,
|
||||
}),
|
||||
)),
|
||||
);
|
||||
}
|
||||
|
||||
async function sendTest() {
|
||||
busy = true;
|
||||
try {
|
||||
await api.testNotification();
|
||||
testResult = { ok: true, text: 'Sent. If nothing arrives, the topic is wrong or ntfy is not reachable.' };
|
||||
} catch (err) {
|
||||
testResult = { ok: false, text: err.message };
|
||||
} finally {
|
||||
busy = false;
|
||||
}
|
||||
rerender();
|
||||
}
|
||||
|
||||
// The queue owns the card's place on the page, so ask it to redraw rather than
|
||||
// reaching into its list.
|
||||
let rerender = () => {};
|
||||
export function onRerender(fn) {
|
||||
rerender = fn;
|
||||
}
|
||||
@@ -1,10 +1,14 @@
|
||||
// On-call: who is on duty now, the week around it, and your own next shifts.
|
||||
// Read-only for now; the TUI edits the schedule.
|
||||
//
|
||||
// One team's rota at a time — the viewer's first team, since a viewer in one
|
||||
// team has nothing to choose between. "On call now" is the exception and shows
|
||||
// every team the viewer is in, because somebody on two rotas wants both.
|
||||
|
||||
import * as api from './api.js';
|
||||
import { h, clear, icon, spinner } from './ui.js';
|
||||
import { isoDate, mondayOf, addDays, isoWeek, initial } from './format.js';
|
||||
import { myID } from './state.js';
|
||||
import { h, clear, badge, icon, spinner } from './ui.js';
|
||||
import { isoDate, mondayOf, addDays, isoWeek, initial, duration } from './format.js';
|
||||
import { myID, currentTeam } from './state.js';
|
||||
|
||||
const view = () => document.getElementById('view-oncall');
|
||||
|
||||
@@ -24,10 +28,17 @@ export async function refresh() {
|
||||
const start = weekStart;
|
||||
const today = new Date();
|
||||
try {
|
||||
const team = currentTeam();
|
||||
if (!team) {
|
||||
data = { now: [], week: [], upcoming: [] };
|
||||
error = null;
|
||||
render();
|
||||
return;
|
||||
}
|
||||
const [now, week, upcoming] = await Promise.all([
|
||||
api.onCallNow(),
|
||||
api.schedule(isoDate(start), isoDate(addDays(start, 6))),
|
||||
api.schedule(isoDate(today), isoDate(addDays(today, 60))),
|
||||
api.schedule(team.id, isoDate(start), isoDate(addDays(start, 6))),
|
||||
api.schedule(team.id, isoDate(today), isoDate(addDays(today, 60))),
|
||||
]);
|
||||
if (start !== weekStart) return;
|
||||
data = { now, week, upcoming };
|
||||
@@ -57,34 +68,79 @@ function render() {
|
||||
}
|
||||
|
||||
function you(userID) {
|
||||
return userID === myID() ? h('span', { class: 'you', text: 'you' }) : null;
|
||||
return userID === myID() ? badge('you', 'plain st-oncall you-badge') : null;
|
||||
}
|
||||
|
||||
// One card per team with somebody on call, and a single empty card when there
|
||||
// is nobody anywhere. The team's name is shown only when the viewer is in more
|
||||
// than one, so the common case reads exactly as it did before teams existed.
|
||||
function nowCard() {
|
||||
const n = data.now;
|
||||
return h('div', { class: 'card now-card' },
|
||||
h('div', { class: `avatar ${n ? '' : 'none'}`, text: n ? initial(n.username) : '–' }),
|
||||
h('div', {},
|
||||
h('div', { class: 'now-label', text: 'On call now' }),
|
||||
h('div', { class: 'now-name' }, n ? n.username : 'Nobody', n && you(n.user_id)),
|
||||
),
|
||||
);
|
||||
const entries = data.now || [];
|
||||
const showTeam = entries.length > 1;
|
||||
if (entries.length === 0) {
|
||||
return h('div', { class: 'card now-card' },
|
||||
h('div', { class: 'avatar none', text: '–' }),
|
||||
h('div', {},
|
||||
h('div', { class: 'now-label', text: 'On call now' }),
|
||||
h('div', { class: 'now-name', text: 'Nobody' }),
|
||||
),
|
||||
);
|
||||
}
|
||||
return h('div', {}, ...entries.map((n) =>
|
||||
h('div', { class: 'card now-card' },
|
||||
h('div', { class: 'avatar', text: initial(n.username) }),
|
||||
h('div', {},
|
||||
h('div', {
|
||||
class: 'now-label',
|
||||
text: showTeam ? `On call now · ${n.team_name}` : 'On call now',
|
||||
}),
|
||||
h('div', { class: 'now-name' }, n.username, you(n.user_id)),
|
||||
),
|
||||
)));
|
||||
}
|
||||
|
||||
// Groups the week's 7 days into runs held by the same person (or the same
|
||||
// empty slot) — the week's own version of the consecutive-day grouping
|
||||
// myShifts does for a single person's own dates, below. Seven identical rows
|
||||
// for one person all week collapses to the one bar this way.
|
||||
function weekRuns(byDate) {
|
||||
const runs = [];
|
||||
for (let i = 0; i < 7; i++) {
|
||||
const date = isoDate(addDays(weekStart, i));
|
||||
const e = byDate.get(date) || null;
|
||||
const uid = e ? e.user_id : null;
|
||||
const last = runs[runs.length - 1];
|
||||
if (last && last.uid === uid) last.to = date;
|
||||
else runs.push({ uid, entry: e, from: date, to: date });
|
||||
}
|
||||
return runs;
|
||||
}
|
||||
|
||||
function weekCard() {
|
||||
const byDate = new Map(data.week.map((e) => [e.date, e]));
|
||||
const today = isoDate(new Date());
|
||||
const days = [];
|
||||
for (let i = 0; i < 7; i++) {
|
||||
const d = addDays(weekStart, i);
|
||||
const key = isoDate(d);
|
||||
const e = byDate.get(key);
|
||||
days.push(h('li', { class: `day ${key === today ? 'today' : ''} ${key < today ? 'past' : ''}` },
|
||||
h('span', { class: 'day-name', text: dayName.format(d) }),
|
||||
h('span', { class: 'day-date', text: dayDate.format(d) }),
|
||||
h('span', { class: `day-who ${e ? '' : 'nobody'}` }, e ? e.username : 'nobody', e && you(e.user_id)),
|
||||
));
|
||||
}
|
||||
const mine = myID();
|
||||
const days = weekRuns(byDate).map((r) => {
|
||||
const single = r.from === r.to;
|
||||
const cls = [
|
||||
'day',
|
||||
single ? '' : 'range',
|
||||
r.from <= today && today <= r.to ? 'today' : '',
|
||||
r.to < today ? 'past' : '',
|
||||
r.uid === mine ? 'mine' : '',
|
||||
].filter(Boolean).join(' ');
|
||||
const label = single
|
||||
? [h('span', { class: 'day-name', text: dayName.format(parse(r.from)) }),
|
||||
h('span', { class: 'day-date', text: dayDate.format(parse(r.from)) })]
|
||||
: [h('span', {
|
||||
class: 'day-range',
|
||||
text: `${dayName.format(parse(r.from))} ${dayDate.format(parse(r.from))} – ${dayName.format(parse(r.to))} ${dayDate.format(parse(r.to))}`,
|
||||
})];
|
||||
return h('li', { class: cls },
|
||||
...label,
|
||||
h('span', { class: `day-who ${r.entry ? '' : 'nobody'}` }, r.entry ? r.entry.username : 'nobody', r.entry && you(r.entry.user_id)),
|
||||
);
|
||||
});
|
||||
const thisWeek = isoDate(weekStart) === isoDate(mondayOf(new Date()));
|
||||
return [
|
||||
h('div', { class: 'page-head' },
|
||||
@@ -93,12 +149,12 @@ function weekCard() {
|
||||
h('button', { class: 'btn btn-ghost btn-icon', type: 'button', 'aria-label': 'Previous week', onclick: () => shiftWeek(-1) },
|
||||
icon('chevronLeft')),
|
||||
h('button', {
|
||||
class: 'btn btn-ghost label',
|
||||
class: 'btn btn-ghost week-label',
|
||||
type: 'button',
|
||||
title: 'Back to this week',
|
||||
onclick: () => { weekStart = mondayOf(new Date()); refresh(); },
|
||||
text: `Week ${isoWeek(weekStart)}`,
|
||||
}),
|
||||
text: `${dayDate.format(weekStart)} – ${dayDate.format(addDays(weekStart, 6))}`,
|
||||
}, h('small', { text: ` Week ${isoWeek(weekStart)}` })),
|
||||
h('button', { class: 'btn btn-ghost btn-icon', type: 'button', 'aria-label': 'Next week', onclick: () => shiftWeek(1) },
|
||||
icon('chevronRight')),
|
||||
),
|
||||
@@ -107,7 +163,9 @@ function weekCard() {
|
||||
];
|
||||
}
|
||||
|
||||
// myShifts groups your upcoming dates into runs of consecutive days.
|
||||
// myShifts groups your upcoming dates into runs of consecutive days, then
|
||||
// splits off the one you're already in — listing it again under "Next
|
||||
// shifts" told people they hadn't started a shift they were already on.
|
||||
function myShifts() {
|
||||
const mine = data.upcoming.filter((e) => e.user_id === myID()).map((e) => e.date).sort();
|
||||
const runs = [];
|
||||
@@ -117,16 +175,34 @@ function myShifts() {
|
||||
else runs.push({ from: date, to: date });
|
||||
}
|
||||
const fmt = (s) => `${dayName.format(parse(s))} ${dayDate.format(parse(s))}`;
|
||||
return [
|
||||
h('div', { class: 'page-head' }, h('h2', { text: 'Your next shifts' })),
|
||||
const label = (r) => (r.from === r.to ? fmt(r.from) : `${fmt(r.from)} – ${fmt(r.to)}`);
|
||||
|
||||
const today = isoDate(new Date());
|
||||
const current = runs[0] && runs[0].from <= today ? runs[0] : null;
|
||||
const next = current ? runs.slice(1) : runs;
|
||||
|
||||
const currentCard = current ? [
|
||||
h('div', { class: 'page-head' }, h('h2', { text: 'Current shift' })),
|
||||
h('div', { class: 'card' },
|
||||
runs.length
|
||||
? h('ul', { class: 'shift-list' }, runs.slice(0, 8).map((r) =>
|
||||
h('ul', { class: 'shift-list' }, h('li', {},
|
||||
h('span', { text: label(current) }),
|
||||
h('span', { class: 'muted', text: `ends in ${duration(addDays(parse(current.to), 1) - Date.now())}` })))),
|
||||
] : [];
|
||||
|
||||
return [
|
||||
...currentCard,
|
||||
h('div', { class: 'page-head' }, h('h2', { text: current ? 'Next shifts' : 'Your next shifts' })),
|
||||
h('div', { class: 'card' },
|
||||
next.length
|
||||
? h('ul', { class: 'shift-list' }, next.slice(0, 8).map((r) =>
|
||||
h('li', {},
|
||||
h('span', { text: r.from === r.to ? fmt(r.from) : `${fmt(r.from)} – ${fmt(r.to)}` }),
|
||||
h('span', { text: label(r) }),
|
||||
h('span', { class: 'muted', text: days(r) })),
|
||||
))
|
||||
: h('div', { class: 'empty', text: 'Nothing scheduled in the next 60 days.' })),
|
||||
: h('div', {
|
||||
class: 'empty',
|
||||
text: current ? 'Nothing else scheduled in the next 60 days.' : 'Nothing scheduled in the next 60 days.',
|
||||
})),
|
||||
];
|
||||
}
|
||||
|
||||
|
||||
@@ -2,8 +2,9 @@
|
||||
|
||||
import * as api from './api.js';
|
||||
import { h, clear, badge, emptyState, spinner } from './ui.js';
|
||||
import { age, until, isFuture, severityClass, labelSummary } from './format.js';
|
||||
import { state, myID } from './state.js';
|
||||
import { ago, until, isFuture, severityClass, labelSummary, teamColorClass } from './format.js';
|
||||
import { state, myID, setSelectedTeam, onTeamChange } from './state.js';
|
||||
import * as onboarding from './onboarding.js';
|
||||
import { navigate } from './app.js';
|
||||
|
||||
// The same filters as the TUI's `f` cycle, plus archived ones to get back to.
|
||||
@@ -17,14 +18,23 @@ const FILTERS = [
|
||||
];
|
||||
|
||||
const EMPTY = {
|
||||
open: ['All clear', 'Nothing open right now.'],
|
||||
triggered: ['Nothing triggered', 'Every open incident has been acknowledged.'],
|
||||
acknowledged: ['Nothing acknowledged', 'No one is working an incident right now.'],
|
||||
snoozed: ['Nothing snoozed', 'Snoozed incidents show up here until the snooze runs out.'],
|
||||
resolved: ['Nothing resolved', 'Resolved incidents are archived after a while.'],
|
||||
archived: ['Nothing archived', ''],
|
||||
open: ['All clear', 'Nothing open right now.', 'checkCircle'],
|
||||
triggered: ['Nothing triggered', 'Every open incident has been acknowledged.', 'checkCircle'],
|
||||
acknowledged: ['Nothing acknowledged', 'No one is working an incident right now.', 'checkCircle'],
|
||||
snoozed: ['Nothing snoozed', 'Snoozed incidents show up here until the snooze runs out.', 'clock'],
|
||||
resolved: ['Nothing resolved', 'Resolved incidents are archived after a while.', null],
|
||||
archived: ['Nothing archived', 'Resolved incidents land here once archived.', 'archive'],
|
||||
};
|
||||
|
||||
onboarding.onRerender(() => renderList());
|
||||
// The queue used to keep its own team filter (a per-tab sessionStorage value,
|
||||
// out of step with team.js's own picker); both now defer to the global
|
||||
// selector's shared state, so re-render whenever it changes.
|
||||
onTeamChange(() => {
|
||||
renderChips();
|
||||
refresh({ fresh: true });
|
||||
});
|
||||
|
||||
let filter = loadFilter();
|
||||
let items = null; // null while loading
|
||||
let error = null;
|
||||
@@ -64,7 +74,12 @@ export async function refresh({ fresh = false } = {}) {
|
||||
const requested = filter;
|
||||
try {
|
||||
// The open list is already fetched for the badges; no need to ask twice.
|
||||
const result = filter === 'open' && !fresh ? state.open : await api.incidents(f.query);
|
||||
// The cached open queue covers every team, so it can only be reused when
|
||||
// no team filter is applied.
|
||||
const query = state.selectedTeamID != null ? { ...f.query, team_id: state.selectedTeamID } : f.query;
|
||||
const cached = filter === 'open' && !fresh && state.selectedTeamID == null;
|
||||
const result = cached ? state.open : await api.incidents(query);
|
||||
await onboarding.load();
|
||||
if (requested !== filter) return;
|
||||
items = result;
|
||||
error = null;
|
||||
@@ -72,6 +87,10 @@ export async function refresh({ fresh = false } = {}) {
|
||||
if (requested !== filter) return;
|
||||
error = err.message;
|
||||
}
|
||||
// state.open (what the chip counts read) has just been refreshed too, by
|
||||
// whichever caller updated it before calling here — app.js's poll, or the
|
||||
// `cached` branch above.
|
||||
renderChips();
|
||||
renderList();
|
||||
}
|
||||
|
||||
@@ -86,36 +105,86 @@ function setFilter(id) {
|
||||
refresh({ fresh: true });
|
||||
}
|
||||
|
||||
// Counts for the three chips derivable from the open list already fetched
|
||||
// for the badges — Snoozed/Resolved/Archived would need a request of their
|
||||
// own, so those chips stay count-less for now.
|
||||
function chipCount(id) {
|
||||
const open = state.selectedTeamID == null
|
||||
? state.open
|
||||
: state.open.filter((i) => i.team_id === state.selectedTeamID);
|
||||
if (id === 'open') return open.length;
|
||||
if (id === 'triggered') return open.filter((i) => i.status === 'triggered').length;
|
||||
if (id === 'acknowledged') return open.filter((i) => i.status === 'acknowledged').length;
|
||||
return null;
|
||||
}
|
||||
|
||||
function renderChips() {
|
||||
const el = document.getElementById('queue-filters');
|
||||
clear(el, FILTERS.map((f) =>
|
||||
h('button', {
|
||||
const chips = FILTERS.map((f) => {
|
||||
const count = chipCount(f.id);
|
||||
return h('button', {
|
||||
class: 'chip',
|
||||
type: 'button',
|
||||
role: 'tab',
|
||||
'aria-selected': String(f.id === filter),
|
||||
onclick: () => setFilter(f.id),
|
||||
text: f.label,
|
||||
}),
|
||||
));
|
||||
}, count != null && h('span', { class: 'count', text: String(count) }));
|
||||
});
|
||||
|
||||
// Somebody in one team has nothing to choose between, so the row of team
|
||||
// chips appears only when there is more than one. The default is all of
|
||||
// them: the combined queue is the point.
|
||||
if (state.teams.length > 1) {
|
||||
chips.push(h('span', { class: 'chip-sep' }));
|
||||
chips.push(h('button', {
|
||||
class: 'chip',
|
||||
type: 'button',
|
||||
role: 'tab',
|
||||
'aria-selected': String(state.selectedTeamID == null),
|
||||
onclick: () => setSelectedTeam(null),
|
||||
}, h('span', { class: 'team-dot' }), ' All teams'));
|
||||
for (const team of state.teams) {
|
||||
chips.push(h('button', {
|
||||
class: 'chip',
|
||||
type: 'button',
|
||||
role: 'tab',
|
||||
'aria-selected': String(team.id === state.selectedTeamID),
|
||||
onclick: () => setSelectedTeam(team.id),
|
||||
},
|
||||
h('span', { class: `team-dot ${teamColorClass(team.id)}` }),
|
||||
' ' + team.name,
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
// A scroll hint for the phone-width row, where the chips can run off the
|
||||
// right edge with nothing to suggest there's more; the desktop sidebar
|
||||
// wraps instead of scrolling (see .pane-list .chips), so this fades out
|
||||
// there via CSS rather than being left out here.
|
||||
chips.push(h('span', { class: 'chips-fade', 'aria-hidden': 'true' }));
|
||||
|
||||
clear(el, chips);
|
||||
}
|
||||
|
||||
function renderList() {
|
||||
const el = document.getElementById('queue-list');
|
||||
const checklist = onboarding.card();
|
||||
if (error && !items) {
|
||||
clear(el, h('div', { class: 'load-error', text: error }));
|
||||
clear(el, checklist, h('div', { class: 'load-error', text: error }));
|
||||
return;
|
||||
}
|
||||
if (!items) {
|
||||
clear(el, spinner());
|
||||
clear(el, checklist, spinner());
|
||||
return;
|
||||
}
|
||||
if (!items.length) {
|
||||
const [title, text] = EMPTY[filter];
|
||||
clear(el, emptyState(title, text, filter === 'open' ? 'checkCircle' : null));
|
||||
const [title, text, iconName] = EMPTY[filter];
|
||||
clear(el, checklist, emptyState(title, text, iconName));
|
||||
return;
|
||||
}
|
||||
clear(el,
|
||||
checklist,
|
||||
error && h('div', { class: 'load-error', text: `Showing older data: ${error}` }),
|
||||
items.map((inc, i) => row(inc, i)),
|
||||
);
|
||||
@@ -142,6 +211,13 @@ function row(inc, index) {
|
||||
// The server already puts the group labels in the title; show only the rest.
|
||||
const labels = labelSummary(Object.fromEntries(
|
||||
Object.entries(inc.group_labels || {}).filter(([k, v]) => !inc.title.includes(`${k}=${v}`))));
|
||||
// The team is shown only to somebody who is in more than one. For everybody
|
||||
// else it is the same word on every row, which is noise rather than
|
||||
// information.
|
||||
const team = state.teams.length > 1 && inc.team_name
|
||||
? h('span', { class: 'row-team', text: inc.team_name })
|
||||
: null;
|
||||
|
||||
return h('a', {
|
||||
class: `row ${severityClass(inc.severity)} ${resolved ? 'resolved' : ''} ${index === cursor ? 'kbd-focus' : ''}`,
|
||||
href: `/incidents/${inc.id}`,
|
||||
@@ -149,10 +225,14 @@ function row(inc, index) {
|
||||
dataset: { index: String(index) },
|
||||
},
|
||||
h('div', { class: 'row-title', text: inc.title }),
|
||||
h('div', { class: 'row-age', title: inc.triggered_at, text: age(inc.triggered_at) }),
|
||||
h('div', { class: 'row-age', title: inc.triggered_at, text: `Triggered ${ago(inc.triggered_at)}` }),
|
||||
h('div', { class: 'row-meta' },
|
||||
status,
|
||||
// The left-border colour alone doesn't say what it means; spell it out
|
||||
// too, same badge the incident detail page uses for severity.
|
||||
inc.severity && badge(inc.severity, `plain ${severityClass(inc.severity)}`),
|
||||
assignee,
|
||||
team,
|
||||
labels && h('span', { class: 'labels', text: labels }),
|
||||
),
|
||||
);
|
||||
|
||||
@@ -5,9 +5,60 @@ import * as api from './api.js';
|
||||
|
||||
export const state = {
|
||||
me: null, // { user, has_password }
|
||||
// How the server can be signed in to. The defaults are an older server's
|
||||
// answer: passwords, no single sign-on.
|
||||
auth: { password_login: true, oidc: { enabled: false, name: '' } },
|
||||
open: [], // the default queue: open, not snoozed
|
||||
teams: [], // the teams the viewer belongs to, each with their role
|
||||
// Which team the whole app is scoped to right now; null means "All teams".
|
||||
// Set only through setSelectedTeam below, never assigned directly, so every
|
||||
// view stays in sync and the choice is remembered across reloads.
|
||||
selectedTeamID: loadSelectedTeam(),
|
||||
};
|
||||
|
||||
const SELECTED_TEAM_KEY = 'terdut.selectedTeam';
|
||||
|
||||
function loadSelectedTeam() {
|
||||
try {
|
||||
const raw = localStorage.getItem(SELECTED_TEAM_KEY);
|
||||
return raw ? Number(raw) : null;
|
||||
} catch {
|
||||
return null; // storage unavailable, or nothing saved yet
|
||||
}
|
||||
}
|
||||
|
||||
// Callbacks to run whenever the selected team changes, so every view that
|
||||
// cares — the queue's filter, the Team settings page, the selector's own
|
||||
// trigger — stays in sync without a general event bus, following the one
|
||||
// precedent for this in the codebase: onboarding.js's onRerender.
|
||||
const teamListeners = [];
|
||||
export function onTeamChange(cb) {
|
||||
teamListeners.push(cb);
|
||||
}
|
||||
|
||||
// setSelectedTeam changes which team the app is scoped to (id, or null for
|
||||
// "All teams"), persists it — a durable preference, unlike the per-tab
|
||||
// sessionStorage filter this replaces — and tells every registered listener.
|
||||
export function setSelectedTeam(id) {
|
||||
state.selectedTeamID = id;
|
||||
try {
|
||||
if (id == null) localStorage.removeItem(SELECTED_TEAM_KEY);
|
||||
else localStorage.setItem(SELECTED_TEAM_KEY, String(id));
|
||||
} catch {
|
||||
/* storage unavailable */
|
||||
}
|
||||
for (const cb of teamListeners) cb();
|
||||
}
|
||||
|
||||
// The team whose schedule and settings the views act on: the selected team,
|
||||
// falling back to the first one the viewer belongs to — which is everybody's
|
||||
// only team until somebody makes a second, or the stored selection naming a
|
||||
// team the account has since left.
|
||||
export function currentTeam() {
|
||||
const teams = state.teams || [];
|
||||
return teams.find((t) => t.id === state.selectedTeamID) || teams[0] || null;
|
||||
}
|
||||
|
||||
export function myID() {
|
||||
return state.me ? state.me.user.id : null;
|
||||
}
|
||||
@@ -24,8 +75,20 @@ export async function users() {
|
||||
return usersCache;
|
||||
}
|
||||
|
||||
export async function loadTeams() {
|
||||
state.teams = await api.teams();
|
||||
// A stored id that no longer names one of the account's teams — left it, or
|
||||
// this is simply a different account signed in on the same browser — is as
|
||||
// good as unset.
|
||||
if (state.selectedTeamID != null && !state.teams.some((t) => t.id === state.selectedTeamID)) {
|
||||
state.selectedTeamID = null;
|
||||
}
|
||||
return state.teams;
|
||||
}
|
||||
|
||||
export function reset() {
|
||||
state.me = null;
|
||||
state.open = [];
|
||||
state.teams = [];
|
||||
usersCache = null;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,168 @@
|
||||
// Statistics: how many incidents, how fast they are answered, and when and
|
||||
// what the alerts are. The same figures the TUI's Stats tab shows, over a
|
||||
// range picked with the chips. The server scopes them to the caller's teams.
|
||||
|
||||
import * as api from './api.js';
|
||||
import { h, clear, emptyState, spinner } from './ui.js';
|
||||
import { duration } from './format.js';
|
||||
|
||||
const DAY_MS = 24 * 60 * 60 * 1000;
|
||||
|
||||
// `days` counts back from today, inclusive; the server reads from/to as UTC
|
||||
// dates, so these are too.
|
||||
const RANGES = [
|
||||
{ id: 'today', label: 'Today', days: 1 },
|
||||
{ id: '7d', label: '7d', days: 7 },
|
||||
{ id: '30d', label: '30d', days: 30 },
|
||||
{ id: '90d', label: '90d', days: 90 },
|
||||
{ id: 'all', label: 'All', days: null },
|
||||
];
|
||||
|
||||
const view = () => document.getElementById('view-stats');
|
||||
|
||||
let range = '30d';
|
||||
let data = null;
|
||||
let error = null;
|
||||
|
||||
const utcDate = (ms) => new Date(ms).toISOString().slice(0, 10);
|
||||
|
||||
function query(r) {
|
||||
if (!r.days) return {};
|
||||
const now = Date.now();
|
||||
return { from: utcDate(now - (r.days - 1) * DAY_MS), to: utcDate(now) };
|
||||
}
|
||||
|
||||
export function show() {
|
||||
render();
|
||||
refresh();
|
||||
}
|
||||
|
||||
export async function refresh() {
|
||||
const requested = range;
|
||||
const q = query(RANGES.find((x) => x.id === range));
|
||||
try {
|
||||
const [incidents, top, byHour, byDay] = await Promise.all([
|
||||
api.statsIncidents(q),
|
||||
api.statsTop({ ...q, limit: 10 }),
|
||||
api.statsByHour(q),
|
||||
api.statsByDay(q),
|
||||
]);
|
||||
if (requested !== range) return;
|
||||
data = { incidents, top, byHour, byDay };
|
||||
error = null;
|
||||
} catch (err) {
|
||||
if (requested !== range) return;
|
||||
error = err.message;
|
||||
}
|
||||
render();
|
||||
}
|
||||
|
||||
function setRange(id) {
|
||||
if (id === range) return;
|
||||
range = id;
|
||||
data = null;
|
||||
render();
|
||||
refresh();
|
||||
}
|
||||
|
||||
function render() {
|
||||
const chips = h('div', { class: 'chips', role: 'tablist', 'aria-label': 'Time range' },
|
||||
RANGES.map((r) => h('button', {
|
||||
class: 'chip',
|
||||
type: 'button',
|
||||
role: 'tab',
|
||||
'aria-selected': String(r.id === range),
|
||||
onclick: () => setRange(r.id),
|
||||
text: r.label,
|
||||
})));
|
||||
|
||||
let body;
|
||||
if (error && !data) body = h('div', { class: 'load-error', text: error });
|
||||
else if (!data) body = spinner();
|
||||
else if (!data.incidents.total && !data.byHour.some((x) => x.count)) {
|
||||
body = emptyState('No data in this range', 'Nothing happened in this window.', 'chart');
|
||||
} else {
|
||||
body = h('div', { class: 'stats' },
|
||||
error && h('div', { class: 'load-error', text: `Showing older data: ${error}` }),
|
||||
tiles(data.incidents),
|
||||
data.top.length > 0 && card('Top alerts', topAlerts(data.top)),
|
||||
card('Alerts by hour (UTC)', columns(
|
||||
data.byHour.map((x) => ({ label: String(x.hour), value: x.count, tick: x.hour % 6 === 0 })),
|
||||
'Alerts by hour of day')),
|
||||
card('Alerts by day', columns(
|
||||
data.byDay.map((x) => ({ label: x.day_name.slice(0, 3), value: x.count, tick: true })),
|
||||
'Alerts by day of week')));
|
||||
}
|
||||
clear(view(), h('div', {}, chips, body));
|
||||
}
|
||||
|
||||
// A missing mean means nothing has been acknowledged or resolved yet.
|
||||
const mean = (s) => (s == null ? '—' : duration(s * 1000));
|
||||
|
||||
function tiles(s) {
|
||||
const tile = (label, value, cls = '') => h('div', { class: `stat-tile ${cls}` },
|
||||
h('div', { class: 'stat-value', text: String(value) }),
|
||||
h('div', { class: 'stat-label', text: label }));
|
||||
return h('div', { class: 'stat-tiles' },
|
||||
tile('Incidents', s.total),
|
||||
tile('Triggered', s.triggered, 'st-triggered'),
|
||||
tile('Acknowledged', s.acknowledged, 'st-acknowledged'),
|
||||
tile('Resolved', s.resolved, 'st-resolved'),
|
||||
tile('Mean time to acknowledge', mean(s.mtta_seconds)),
|
||||
tile('Mean time to resolve', mean(s.mttr_seconds)));
|
||||
}
|
||||
|
||||
function card(title, content) {
|
||||
return h('section', { class: 'chart-card card card-pad' },
|
||||
h('h3', { class: 'chart-title', text: title }), content);
|
||||
}
|
||||
|
||||
// Ranked names with a bar scaled to the busiest one.
|
||||
function topAlerts(items) {
|
||||
const max = Math.max(...items.map((x) => x.count), 1);
|
||||
return h('ol', { class: 'hbars' }, items.map((x) => {
|
||||
const fill = h('span', { class: 'hbar-fill' });
|
||||
fill.style.width = `${Math.max(2, (x.count / max) * 100)}%`;
|
||||
return h('li', { class: 'hbar' },
|
||||
h('span', { class: 'hbar-name', title: x.name, text: x.name }),
|
||||
h('span', { class: 'hbar-track' }, fill),
|
||||
h('span', { class: 'hbar-count', text: String(x.count) }));
|
||||
}));
|
||||
}
|
||||
|
||||
const SVG_NS = 'http://www.w3.org/2000/svg';
|
||||
|
||||
function svg(tag, attrs = {}, text) {
|
||||
const el = document.createElementNS(SVG_NS, tag);
|
||||
for (const [k, v] of Object.entries(attrs)) el.setAttribute(k, String(v));
|
||||
if (text != null) el.textContent = text;
|
||||
return el;
|
||||
}
|
||||
|
||||
// A column chart: one bar per item, the value in a tooltip, and a label under
|
||||
// the items marked `tick`.
|
||||
function columns(items, label) {
|
||||
const W = 480;
|
||||
const H = 140;
|
||||
const base = H - 18;
|
||||
const step = W / items.length;
|
||||
const max = Math.max(...items.map((x) => x.value), 1);
|
||||
const root = svg('svg', {
|
||||
class: 'columns', viewBox: `0 0 ${W} ${H}`, role: 'img', 'aria-label': label,
|
||||
});
|
||||
root.appendChild(svg('line', { class: 'axis', x1: 0, x2: W, y1: base, y2: base }));
|
||||
items.forEach((it, i) => {
|
||||
const bh = it.value ? Math.max(2, (it.value / max) * (base - 6)) : 0;
|
||||
const x = i * step + step * 0.15;
|
||||
const g = svg('g', { class: 'col' });
|
||||
g.appendChild(svg('title', {}, `${it.label}: ${it.value}`));
|
||||
// A full-height transparent hit area, so a tiny bar is still hoverable.
|
||||
g.appendChild(svg('rect', { class: 'col-hit', x: i * step, y: 0, width: step, height: base }));
|
||||
if (bh) g.appendChild(svg('rect', { class: 'col-bar', x, y: base - bh, width: step * 0.7, height: bh, rx: 2 }));
|
||||
root.appendChild(g);
|
||||
if (it.tick) {
|
||||
root.appendChild(svg('text', { class: 'col-label', x: i * step + step / 2, y: H - 4, 'text-anchor': 'middle' }, it.label));
|
||||
}
|
||||
});
|
||||
return root;
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user