Compare commits
121 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 01922291f6 | |||
| eb63e5e138 | |||
| 9029d48584 | |||
| 44b2eb2cc3 | |||
| 1cb09525e3 | |||
| c4833067f0 | |||
| 2bc8f336a0 | |||
| a955356821 | |||
| db474ca909 | |||
| a23e88c16d | |||
| def0f68d00 | |||
| fb86a18988 | |||
| b848143471 | |||
| 942517c7a8 | |||
| 7665e5e52f | |||
| bcf1a3e99b | |||
| 3a96b20cbe | |||
| 7b9d309d13 | |||
| 7412456c5a | |||
| 065b557860 | |||
| a8dc89e23d | |||
| ead5df1574 | |||
| 3ced069134 | |||
| 0f88574a41 | |||
| 3613fd5732 | |||
| dc62278788 | |||
| 0aaea8efb5 | |||
| 584d3441fc | |||
| 926aa2d3ec | |||
| a2ca9c25d0 | |||
| 92959cac38 | |||
| f15db0e20a | |||
| b82c10acf4 | |||
| 7cd6fbf571 | |||
| df83adfe47 | |||
| 9da913080f | |||
| 2b2609e98f | |||
| fa6d82d6e5 | |||
| 7efd1bbba7 | |||
| 3bf94a5d7f | |||
| 5c4e0bdd0e | |||
| 9e5b085d8b | |||
| 0050738ca0 | |||
| 42180948d1 | |||
| c83c7c2a8b | |||
| 43beda9a30 | |||
| 710521a73c | |||
| d9492913ed | |||
| 1770e5d945 | |||
| 497086cb51 | |||
| e3090d2779 | |||
| 91f03c21e8 | |||
| 4358e84b24 | |||
| f45dc2f925 | |||
| 774fdfcaa8 | |||
| fd26fef1ba | |||
| a9d788cc83 | |||
| 4b15079ac2 | |||
| fc9f47cc8d | |||
| bc9f793f1f | |||
| 871274a3a0 | |||
| 6f8499fa42 | |||
| ef731e85c5 | |||
| a4dd60f6b8 | |||
| b5573fbca2 | |||
| 0ee576f793 | |||
| b610b1817a | |||
| 949d6595ba | |||
| 33356ca978 | |||
| e5b4df7c03 | |||
| 5b4683febf | |||
| 97a4814c04 | |||
| a2dc9e3b03 | |||
| 155f27ca62 | |||
| a27ff49171 | |||
| c5be55dcbc | |||
| b2c3868619 | |||
| 36c00acf62 | |||
| 9d1df2b611 | |||
| 9bf4c92bfe | |||
| e616c82646 | |||
| 2b396d22d6 | |||
| 1f1faa437c | |||
| dc92f51cf8 | |||
| d675f8ec9b | |||
| e8d45f9d3d | |||
| f3918b863c | |||
| 3ee8583f6f | |||
| 591d5b8df0 | |||
| d2cdcc9776 | |||
| 8b2789b9b2 | |||
| 71d7e1853a | |||
| 60ebb75cd2 | |||
| 734cd9c5fd | |||
| 423ed9b3a3 | |||
| 43f004499b | |||
| 559be6de6e | |||
| e77f04b55e | |||
| e536fdd2c0 | |||
| 429d5fdda3 | |||
| 3cdd5aee1f | |||
| 6a03698f65 | |||
| 67d68ce058 | |||
| a6fa673e08 | |||
| ee22eb000c | |||
| 07914d5cdb | |||
| 7b9a337d25 | |||
| fc8b0c8d58 | |||
| 828cf87656 | |||
| ac9af8e4f5 | |||
| 8869ac864f | |||
| 0677e74cf8 | |||
| 56b8191a78 | |||
| 93761056eb | |||
| a92da7dcc0 | |||
| b39aac36b7 | |||
| 19f168ab7e | |||
| d827ceedff | |||
| 4e8c52c28c | |||
| fb927aa67b | |||
| d728af53b1 |
@@ -35,7 +35,7 @@ jobs:
|
|||||||
# Runs inside the toolchain image rather than installing Go per job. Note this puts
|
# Runs inside the toolchain image rather than installing Go per job. Note this puts
|
||||||
# the job on the dind bridge, which cannot reach github.com or get.helm.sh --
|
# the job on the dind bridge, which cannot reach github.com or get.helm.sh --
|
||||||
# proxy.golang.org and git.ryuvia.com are reachable, which is all this job needs.
|
# proxy.golang.org and git.ryuvia.com are reachable, which is all this job needs.
|
||||||
image: golang:1.26.6-bookworm
|
image: golang:1.26.9-bookworm
|
||||||
# act_runner destroys a job's own volumes when it finishes, so without these every
|
# act_runner destroys a job's own volumes when it finishes, so without these every
|
||||||
# run re-downloads the whole module graph. The names must appear in the runner's
|
# run re-downloads the whole module graph. The names must appear in the runner's
|
||||||
# container.valid_volumes allowlist (charts/act-runner in the k8s repo); unlisted
|
# container.valid_volumes allowlist (charts/act-runner in the k8s repo); unlisted
|
||||||
@@ -46,8 +46,8 @@ jobs:
|
|||||||
- go-build-cache:/root/.cache/go-build
|
- go-build-cache:/root/.cache/go-build
|
||||||
- gobin-cache:/go/bin
|
- gobin-cache:/go/bin
|
||||||
|
|
||||||
# The suite needs a real Postgres -- there is no in-memory Postgres the way there was
|
# The suite needs a real Postgres -- there is no in-memory Postgres,
|
||||||
# an in-memory SQLite, so each test gets its own schema on a shared server instead.
|
# so each test gets its own schema on a shared server instead.
|
||||||
# The job and the service share the dind bridge, so the service is reachable by its
|
# The job and the service share the dind bridge, so the service is reachable by its
|
||||||
# name rather than on localhost.
|
# name rather than on localhost.
|
||||||
services:
|
services:
|
||||||
@@ -101,7 +101,7 @@ jobs:
|
|||||||
security:
|
security:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
container:
|
container:
|
||||||
image: golang:1.26.6-bookworm
|
image: golang:1.26.9-bookworm
|
||||||
volumes:
|
volumes:
|
||||||
- go-mod-cache:/go/pkg/mod
|
- go-mod-cache:/go/pkg/mod
|
||||||
- go-build-cache:/root/.cache/go-build
|
- go-build-cache:/root/.cache/go-build
|
||||||
@@ -125,6 +125,9 @@ jobs:
|
|||||||
- name: Secret scan (gitleaks)
|
- name: Secret scan (gitleaks)
|
||||||
run: make security-secrets
|
run: make security-secrets
|
||||||
|
|
||||||
|
- name: Code security scan (gosec)
|
||||||
|
run: make security-code
|
||||||
|
|
||||||
# Host mode, no `container:`: helm is baked into the runner image, and a container job
|
# Host mode, no `container:`: helm is baked into the runner image, and a container job
|
||||||
# could not install it -- get.helm.sh is unreachable from the dind bridge. Same reason
|
# could not install it -- get.helm.sh is unreachable from the dind bridge. Same reason
|
||||||
# release.yaml's chart job runs on the host.
|
# release.yaml's chart job runs on the host.
|
||||||
|
|||||||
@@ -30,7 +30,7 @@ jobs:
|
|||||||
test:
|
test:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
container:
|
container:
|
||||||
image: golang:1.26.6-bookworm
|
image: golang:1.26.9-bookworm
|
||||||
volumes:
|
volumes:
|
||||||
- go-mod-cache:/go/pkg/mod
|
- go-mod-cache:/go/pkg/mod
|
||||||
- go-build-cache:/root/.cache/go-build
|
- go-build-cache:/root/.cache/go-build
|
||||||
@@ -71,7 +71,7 @@ jobs:
|
|||||||
needs: test
|
needs: test
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
container:
|
container:
|
||||||
image: golang:1.26.6-bookworm
|
image: golang:1.26.9-bookworm
|
||||||
volumes:
|
volumes:
|
||||||
- go-mod-cache:/go/pkg/mod
|
- go-mod-cache:/go/pkg/mod
|
||||||
- go-build-cache:/root/.cache/go-build
|
- go-build-cache:/root/.cache/go-build
|
||||||
|
|||||||
@@ -7,8 +7,6 @@
|
|||||||
# one (which has an unreachable entry) cannot abort a release
|
# one (which has an unreachable entry) cannot abort a release
|
||||||
/.helm-repos.yaml
|
/.helm-repos.yaml
|
||||||
|
|
||||||
# SQLite database files
|
|
||||||
*.db
|
|
||||||
*.db-shm
|
*.db-shm
|
||||||
*.db-wal
|
*.db-wal
|
||||||
|
|
||||||
|
|||||||
@@ -12,9 +12,9 @@ Preconditions and the plan, without side effects:
|
|||||||
```
|
```
|
||||||
|
|
||||||
Config is `.release.conf` here plus `make release-vars`. The process itself lives in
|
Config is `.release.conf` here plus `make release-vars`. The process itself lives in
|
||||||
`~/.claude/skills/release/`; why it is shaped this way is in README.md §Releasing.
|
`~/.claude/skills/release/`; why it is shaped this way is in docs/development.md (Releasing).
|
||||||
|
|
||||||
Two things about this repo specifically:
|
Three things about this repo specifically:
|
||||||
|
|
||||||
- **The image is scanned after it is published, not before.** `scan-image` runs trivy
|
- **The image is scanned after it is published, not before.** `scan-image` runs trivy
|
||||||
against the pushed image, because trivy cannot read a locally built one on this runner.
|
against the pushed image, because trivy cannot read a locally built one on this runner.
|
||||||
@@ -25,6 +25,16 @@ Two things about this repo specifically:
|
|||||||
so `chart-bump` needs `--image "$IMAGE"` to know which one moves. That sidecar backs up
|
so `chart-bump` needs `--image "$IMAGE"` to know which one moves. That sidecar backs up
|
||||||
SQLite; the Postgres move (#2) retires it in favour of a `postgresql` CR with a k8up
|
SQLite; the Postgres move (#2) retires it in favour of a `postgresql` CR with a k8up
|
||||||
`pg_dump` annotation, after which only the app image's tag is left.
|
`pg_dump` annotation, after which only the app image's tag is left.
|
||||||
|
- **Two demos pin this image, and `chart-bump` moves neither.** `terdut-demo` in
|
||||||
|
`Ryuvia/charts` is a `TerdutServer` CR that terdut-operator reconciles, and its
|
||||||
|
`values.yaml` `image.tag` is meant to match production's pin (same digest). The kind demo
|
||||||
|
in terdut-operator (`examples/demo/01-server.yaml`) pins a tag too. A release only bumps
|
||||||
|
the `terdut-server` wrapper, so both drift silently: `terdut-demo` sat at v0.37.0 through
|
||||||
|
v0.41.0-v0.43.0 until it was synced on 2026-10-08. After a release, bump `terdut-demo`'s
|
||||||
|
tag to the same `image-digest` and its `Chart.yaml` `version:` (Flux reconciles on
|
||||||
|
ChartVersion), as its own PR, and say in the release report whether you did. Neither
|
||||||
|
demo has anything but the pin to change, but read the version range's migrations first:
|
||||||
|
the demo's Postgres migrates forward at startup.
|
||||||
|
|
||||||
## Checks
|
## Checks
|
||||||
|
|
||||||
|
|||||||
@@ -16,6 +16,18 @@ RUN CGO_ENABLED=0 GOOS=${TARGETOS} GOARCH=${TARGETARCH} \
|
|||||||
go build -ldflags="-w -s -X main.version=${VERSION}" -o /terdut ./cmd/terdut
|
go build -ldflags="-w -s -X main.version=${VERSION}" -o /terdut ./cmd/terdut
|
||||||
|
|
||||||
FROM scratch
|
FROM scratch
|
||||||
|
# scratch has no trust store, and a Go binary on it fails every HTTPS call with
|
||||||
|
# "x509: certificate signed by unknown authority". Nothing needed one until single
|
||||||
|
# sign-on: discovery and the token exchange are HTTPS calls to the identity provider.
|
||||||
|
# The bundle is the builder's, copied by name so a missing file fails the build
|
||||||
|
# rather than shipping an image that cannot sign anybody in.
|
||||||
|
COPY --from=builder /etc/ssl/certs/ca-certificates.crt /etc/ssl/certs/ca-certificates.crt
|
||||||
COPY --from=builder /terdut /terdut
|
COPY --from=builder /terdut /terdut
|
||||||
EXPOSE 8080
|
EXPOSE 8080
|
||||||
|
# Numeric, not a name: scratch has no /etc/passwd for one to resolve against,
|
||||||
|
# and Docker's USER accepts a bare UID:GID without it. 65532 is the common
|
||||||
|
# "nonroot" convention (distroless's own uid), chosen so the chart's pod
|
||||||
|
# securityContext (runAsNonRoot, runAsUser: 65532) matches what the image
|
||||||
|
# already runs as rather than fighting it.
|
||||||
|
USER 65532:65532
|
||||||
ENTRYPOINT ["/terdut"]
|
ENTRYPOINT ["/terdut"]
|
||||||
|
|||||||
@@ -28,7 +28,7 @@ help: ## Show this help
|
|||||||
# -race below. Both need a Postgres to test against; see test-db.
|
# -race below. Both need a Postgres to test against; see test-db.
|
||||||
|
|
||||||
# The suite needs a Postgres, because the server does: there is no in-memory
|
# The suite needs a Postgres, because the server does: there is no in-memory
|
||||||
# Postgres the way there was an in-memory SQLite. TERDUT_TEST_DSN says where, and
|
# Postgres. TERDUT_TEST_DSN says where, and
|
||||||
# the tests fail rather than skip without it — a suite that quietly tests nothing
|
# the tests fail rather than skip without it — a suite that quietly tests nothing
|
||||||
# is worse than one that does not run. `make test-db` starts a local one;
|
# is worse than one that does not run. `make test-db` starts a local one;
|
||||||
# ci.yaml runs the same thing as a service container.
|
# ci.yaml runs the same thing as a service container.
|
||||||
@@ -143,6 +143,7 @@ BUILDX_BUILDER ?= terdut
|
|||||||
TRIVY_VERSION := 0.73.0
|
TRIVY_VERSION := 0.73.0
|
||||||
GOVULNCHECK_VERSION := v1.1.4
|
GOVULNCHECK_VERSION := v1.1.4
|
||||||
GITLEAKS_VERSION := v8.30.0
|
GITLEAKS_VERSION := v8.30.0
|
||||||
|
GOSEC_VERSION := v2.29.0
|
||||||
|
|
||||||
# --pull, not --no-cache: refresh the base image without discarding the layer cache.
|
# --pull, not --no-cache: refresh the base image without discarding the layer cache.
|
||||||
DOCKER_BUILD_FLAGS ?= --pull
|
DOCKER_BUILD_FLAGS ?= --pull
|
||||||
@@ -222,6 +223,26 @@ release: push helm-package helm-push ## Publish image + chart (the workflow's on
|
|||||||
security-go: ## Scan Go deps for known CVEs (govulncheck)
|
security-go: ## Scan Go deps for known CVEs (govulncheck)
|
||||||
go run golang.org/x/vuln/cmd/govulncheck@$(GOVULNCHECK_VERSION) ./...
|
go run golang.org/x/vuln/cmd/govulncheck@$(GOVULNCHECK_VERSION) ./...
|
||||||
|
|
||||||
|
# Code-level, not dependency- or secret-level: gosec reads this repo's own source for
|
||||||
|
# known-dangerous patterns (weak crypto, SQL/command injection shapes, insecure file
|
||||||
|
# permissions, …) rather than its module graph or working tree for leaked credentials,
|
||||||
|
# which is what security-go and security-secrets above already cover.
|
||||||
|
#
|
||||||
|
# G104 (unchecked error) is excluded. Every hit it found here on first run was this
|
||||||
|
# codebase's existing, deliberate idiom for a best-effort write or an already-reviewed
|
||||||
|
# json.Unmarshal of this server's own JSONB (see the "best-effort" comments in
|
||||||
|
# middleware.go and the //nolint:errcheck lines in alerts.go/deadman.go) -- a style that
|
||||||
|
# predates gosec and that G104 cannot distinguish from a mistake. Reaching the same
|
||||||
|
# green result by adding a dozens of individual #nosec comments would not add
|
||||||
|
# information; it would just make a future *real* G104 regression one more suppressed
|
||||||
|
# line instead of a visible one. Same reasoning as the chi-advisories note on
|
||||||
|
# security-go above: what gosec reports here (nothing, beyond G104) is the useful
|
||||||
|
# property, not a loophole. -exclude-generated skips web.go's embedded, build-time-only
|
||||||
|
# assets.
|
||||||
|
.PHONY: security-code
|
||||||
|
security-code: ## Scan this repo's own source for risky patterns (gosec)
|
||||||
|
go run github.com/securego/gosec/v2/cmd/gosec@$(GOSEC_VERSION) -exclude-generated -exclude=G104 ./...
|
||||||
|
|
||||||
# --no-git scans the working tree rather than the history, so this catches a secret on the
|
# --no-git scans the working tree rather than the history, so this catches a secret on the
|
||||||
# way in. It is not a history audit and finding nothing here says nothing about what is
|
# way in. It is not a history audit and finding nothing here says nothing about what is
|
||||||
# already committed. --redact because the finding is printed into a CI log.
|
# already committed. --redact because the finding is printed into a CI log.
|
||||||
|
|||||||
@@ -0,0 +1,60 @@
|
|||||||
|
# Service accounts
|
||||||
|
|
||||||
|
A non-human credential for automation (terdut-operator, CI, scripts). It is not a
|
||||||
|
`users` row: no password, no `is_admin`, no OIDC identity, so it can never be
|
||||||
|
pulled into login or group sync, and it is never mistaken for a person in an
|
||||||
|
audit trail. The bearer token has the same shape as an API key (SHA-256 hash
|
||||||
|
stored, raw value shown once), prefixed `tdsa_`.
|
||||||
|
|
||||||
|
## Scopes
|
||||||
|
|
||||||
|
- **instance** — acts as owner of every team's *configuration* (rename, OIDC
|
||||||
|
groups, escalation, dead man's switches, integrations, members, delete) and may
|
||||||
|
create teams. It is not a member of any team, so it reads no incidents or
|
||||||
|
queue. It is never an administrator: user management and
|
||||||
|
`/api/admin/settings` stay human-only.
|
||||||
|
- **team** — acts as owner of exactly one team, through a single synthetic
|
||||||
|
membership. It may also mint another service account for its own team.
|
||||||
|
|
||||||
|
An account has many keys, so rotating is "mint a new key, revoke the old one"
|
||||||
|
without losing the account's identity or history.
|
||||||
|
|
||||||
|
## Endpoints
|
||||||
|
|
||||||
|
- `POST /api/service-accounts` `{name, scope, team_id}` — returns the account and
|
||||||
|
its first key. An instance-scoped account is granted by a human administrator;
|
||||||
|
a team-scoped one by an administrator, that team's owner, or an instance-scoped
|
||||||
|
account.
|
||||||
|
- `GET /api/service-accounts?name=` — look one up by name.
|
||||||
|
- `POST /api/service-accounts/{id}/keys`, `DELETE .../keys/{keyID}` — mint or
|
||||||
|
revoke a key. An instance-scoped account may manage any team-scoped account's
|
||||||
|
keys, and any account may manage its own.
|
||||||
|
|
||||||
|
## Seeding the operator's account
|
||||||
|
|
||||||
|
`TERDUT_OPERATOR_KEY` (at least 32 characters) creates the instance-scoped account
|
||||||
|
`terdut-operator` if missing and replaces its `seed` key with this value at every
|
||||||
|
start (`internal/api/operator_key.go`). The deployer generates the key and
|
||||||
|
nothing has to call `/api/bootstrap` for it; rotating is a restart with a new
|
||||||
|
value. With `TERDUT_OPERATOR_MODE` on, configuration writes by humans are refused
|
||||||
|
and a service account of either scope passes.
|
||||||
|
|
||||||
|
## How it is enforced
|
||||||
|
|
||||||
|
Every request resolves to one `Caller` (`internal/api/caller.go`): a human
|
||||||
|
(session or API key) or a service account.
|
||||||
|
|
||||||
|
- `Caller.IsAdmin()` is true only for a human administrator. `AdminOnly` and
|
||||||
|
`requireSelfOrAdmin` key on it alone; do not widen them — each time a gap came up
|
||||||
|
the fix was a narrower purpose-built capability instead.
|
||||||
|
- `Caller.IsInstanceServiceAccount()` is true only for an instance-scoped account,
|
||||||
|
never for a human. `requireTeamOwner` and `callerOwnsTeam` admit it for any team.
|
||||||
|
- `Caller.Role(teamID)`/`TeamIDs()` are a human's memberships or a team-scoped
|
||||||
|
account's single owner membership; instance scope has none.
|
||||||
|
- `Caller.AsHuman()` is what a handler must call when it needs a real `user_id`;
|
||||||
|
handlers meant for people answer 403 to a service account instead of writing a
|
||||||
|
zero id.
|
||||||
|
|
||||||
|
Where a service account acts on an incident (acknowledge, resolve), the
|
||||||
|
timeline and `acknowledged_by` record it through parallel `*_service_account_id`
|
||||||
|
columns, never as a user.
|
||||||
@@ -15,5 +15,5 @@ type: application
|
|||||||
# appVersion and image.tag in values.yaml no longer agree, and that is not an oversight:
|
# appVersion and image.tag in values.yaml no longer agree, and that is not an oversight:
|
||||||
# image.tag stays "latest", which is what a local install actually pulls. appVersion is
|
# image.tag stays "latest", which is what a local install actually pulls. appVersion is
|
||||||
# metadata and drives nothing.
|
# metadata and drives nothing.
|
||||||
version: 0.13.0
|
version: 0.43.0
|
||||||
appVersion: "v0.13.0"
|
appVersion: "v0.43.0"
|
||||||
|
|||||||
@@ -6,25 +6,72 @@ metadata:
|
|||||||
labels:
|
labels:
|
||||||
{{- include "terdut-server.labels" . | nindent 4 }}
|
{{- include "terdut-server.labels" . | nindent 4 }}
|
||||||
spec:
|
spec:
|
||||||
replicas: 1
|
replicas: {{ .Values.replicaCount }}
|
||||||
selector:
|
selector:
|
||||||
matchLabels:
|
matchLabels:
|
||||||
{{- include "terdut-server.selectorLabels" . | nindent 6 }}
|
{{- include "terdut-server.selectorLabels" . | nindent 6 }}
|
||||||
# Recreate, not RollingUpdate, even though the PVC that forced it is gone: the
|
# RollingUpdate, not Recreate: the sweeper, notifier and migration runner
|
||||||
# sweeper and the notifier are unsynchronised singletons, and two replicas
|
# each take a Postgres advisory lock around their own pass, and new-incident
|
||||||
# overlapping during a rollout would both page for the same incident.
|
# creation on the first webhook for a brand-new groupKey resolves its own
|
||||||
|
# insert conflict -- so two replicas overlapping during a rollout no longer
|
||||||
|
# double-page, race a migration, or drop a webhook payload (v0.36.0). No
|
||||||
|
# explicit maxUnavailable/maxSurge: the 25%/25% default rounds to 0/1 at
|
||||||
|
# replicaCount: 2, which is zero-downtime already.
|
||||||
strategy:
|
strategy:
|
||||||
type: Recreate
|
type: RollingUpdate
|
||||||
template:
|
template:
|
||||||
metadata:
|
metadata:
|
||||||
labels:
|
labels:
|
||||||
{{- include "terdut-server.selectorLabels" . | nindent 8 }}
|
{{- include "terdut-server.selectorLabels" . | nindent 8 }}
|
||||||
spec:
|
spec:
|
||||||
enableServiceLinks: false
|
enableServiceLinks: false
|
||||||
|
# Pod-wide default; both containers below run as this UID regardless of
|
||||||
|
# what their own image would otherwise pick (postgres:17-alpine's
|
||||||
|
# pg_isready needs no particular user, and 65532 is what the app image
|
||||||
|
# itself runs as now — see the Dockerfile's USER). seccompProfile here
|
||||||
|
# rather than per-container: there is no reason it would ever differ
|
||||||
|
# between them.
|
||||||
|
securityContext:
|
||||||
|
runAsNonRoot: true
|
||||||
|
runAsUser: 65532
|
||||||
|
runAsGroup: 65532
|
||||||
|
seccompProfile:
|
||||||
|
type: RuntimeDefault
|
||||||
|
{{- if .Values.database.waitForPostgres.enabled }}
|
||||||
|
initContainers:
|
||||||
|
- name: wait-for-postgres
|
||||||
|
image: "{{ .Values.database.waitForPostgres.image.repository }}:{{ .Values.database.waitForPostgres.image.tag }}"
|
||||||
|
imagePullPolicy: {{ .Values.database.waitForPostgres.image.pullPolicy }}
|
||||||
|
# No capability this loop needs, and nothing in it writes to disk:
|
||||||
|
# sh, pg_isready, echo and sleep all run read-only.
|
||||||
|
securityContext:
|
||||||
|
allowPrivilegeEscalation: false
|
||||||
|
readOnlyRootFilesystem: true
|
||||||
|
capabilities:
|
||||||
|
drop: ["ALL"]
|
||||||
|
env:
|
||||||
|
- name: TERDUT_DB_DSN
|
||||||
|
value: {{ required "database.dsn is required" .Values.database.dsn | quote }}
|
||||||
|
command:
|
||||||
|
- sh
|
||||||
|
- -c
|
||||||
|
- |
|
||||||
|
until pg_isready -d "$TERDUT_DB_DSN"; do
|
||||||
|
echo "wait-for-postgres: not ready yet, retrying in 2s"
|
||||||
|
sleep 2
|
||||||
|
done
|
||||||
|
{{- end }}
|
||||||
containers:
|
containers:
|
||||||
- name: terdut-server
|
- name: terdut-server
|
||||||
image: "{{ .Values.image.repository }}:{{ .Values.image.tag }}"
|
image: "{{ .Values.image.repository }}:{{ .Values.image.tag }}"
|
||||||
imagePullPolicy: {{ .Values.image.pullPolicy }}
|
imagePullPolicy: {{ .Values.image.pullPolicy }}
|
||||||
|
# scratch, nothing to write: the binary keeps no local state and
|
||||||
|
# writes nothing to disk, so the root filesystem can stay read-only.
|
||||||
|
securityContext:
|
||||||
|
allowPrivilegeEscalation: false
|
||||||
|
readOnlyRootFilesystem: true
|
||||||
|
capabilities:
|
||||||
|
drop: ["ALL"]
|
||||||
ports:
|
ports:
|
||||||
- name: http
|
- name: http
|
||||||
containerPort: {{ .Values.service.port }}
|
containerPort: {{ .Values.service.port }}
|
||||||
@@ -48,12 +95,6 @@ spec:
|
|||||||
value: "{{ .Values.sweeper.staleAfter }}"
|
value: "{{ .Values.sweeper.staleAfter }}"
|
||||||
- name: TERDUT_ARCHIVE_AFTER
|
- name: TERDUT_ARCHIVE_AFTER
|
||||||
value: "{{ .Values.sweeper.archiveAfter }}"
|
value: "{{ .Values.sweeper.archiveAfter }}"
|
||||||
- name: TERDUT_DEADMAN_MATCHERS
|
|
||||||
value: "{{ .Values.deadman.matchers }}"
|
|
||||||
- name: TERDUT_DEADMAN_TIMEOUT
|
|
||||||
value: "{{ .Values.deadman.timeout }}"
|
|
||||||
- name: TERDUT_DEADMAN_SEVERITY
|
|
||||||
value: "{{ .Values.deadman.severity }}"
|
|
||||||
{{- if .Values.notify.ntfyUrl }}
|
{{- if .Values.notify.ntfyUrl }}
|
||||||
- name: TERDUT_NTFY_URL
|
- name: TERDUT_NTFY_URL
|
||||||
value: "{{ .Values.notify.ntfyUrl }}"
|
value: "{{ .Values.notify.ntfyUrl }}"
|
||||||
@@ -61,8 +102,6 @@ spec:
|
|||||||
value: "{{ .Values.notify.fallbackTopic }}"
|
value: "{{ .Values.notify.fallbackTopic }}"
|
||||||
- name: TERDUT_NOTIFY_REPEAT
|
- name: TERDUT_NOTIFY_REPEAT
|
||||||
value: "{{ .Values.notify.repeatEvery }}"
|
value: "{{ .Values.notify.repeatEvery }}"
|
||||||
- name: TERDUT_PUBLIC_URL
|
|
||||||
value: "{{ .Values.notify.publicUrl | default (printf "https://%s" .Values.networking.hostname) }}"
|
|
||||||
{{- if .Values.notify.tokenSecret.name }}
|
{{- if .Values.notify.tokenSecret.name }}
|
||||||
- name: TERDUT_NTFY_TOKEN
|
- name: TERDUT_NTFY_TOKEN
|
||||||
valueFrom:
|
valueFrom:
|
||||||
@@ -71,6 +110,49 @@ spec:
|
|||||||
key: {{ .Values.notify.tokenSecret.key }}
|
key: {{ .Values.notify.tokenSecret.key }}
|
||||||
{{- end }}
|
{{- end }}
|
||||||
{{- end }}
|
{{- end }}
|
||||||
|
# Set whether or not ntfy is: single sign-on builds its redirect URI
|
||||||
|
# from it, and sessions use it to decide the cookie's Secure flag.
|
||||||
|
- name: TERDUT_PUBLIC_URL
|
||||||
|
value: "{{ .Values.notify.publicUrl | default (printf "https://%s" .Values.networking.hostname) }}"
|
||||||
|
- name: TERDUT_PASSWORD_LOGIN
|
||||||
|
value: {{ .Values.passwordLogin | quote }}
|
||||||
|
- name: TERDUT_TRUSTED_PROXIES
|
||||||
|
value: {{ .Values.trustedProxies | quote }}
|
||||||
|
- name: TERDUT_OPERATOR_MODE
|
||||||
|
value: {{ .Values.operatorMode | quote }}
|
||||||
|
{{- if .Values.oidc.enabled }}
|
||||||
|
- name: TERDUT_OIDC_ISSUER
|
||||||
|
value: {{ required "oidc.issuer is required when oidc.enabled" .Values.oidc.issuer | quote }}
|
||||||
|
- name: TERDUT_OIDC_CLIENT_ID
|
||||||
|
value: {{ required "oidc.clientId is required when oidc.enabled" .Values.oidc.clientId | quote }}
|
||||||
|
- name: TERDUT_OIDC_CLIENT_SECRET
|
||||||
|
valueFrom:
|
||||||
|
secretKeyRef:
|
||||||
|
name: {{ required "oidc.clientSecret.name is required when oidc.enabled" .Values.oidc.clientSecret.name }}
|
||||||
|
key: {{ .Values.oidc.clientSecret.key }}
|
||||||
|
- name: TERDUT_OIDC_NAME
|
||||||
|
value: {{ .Values.oidc.name | quote }}
|
||||||
|
- name: TERDUT_OIDC_SCOPES
|
||||||
|
value: {{ .Values.oidc.scopes | quote }}
|
||||||
|
- name: TERDUT_OIDC_USERNAME_CLAIM
|
||||||
|
value: {{ .Values.oidc.usernameClaim | quote }}
|
||||||
|
- name: TERDUT_OIDC_EMAIL_CLAIM
|
||||||
|
value: {{ .Values.oidc.emailClaim | quote }}
|
||||||
|
- name: TERDUT_OIDC_GROUPS_CLAIM
|
||||||
|
value: {{ .Values.oidc.groupsClaim | quote }}
|
||||||
|
- name: TERDUT_OIDC_TRUST_EMAIL
|
||||||
|
value: {{ .Values.oidc.trustEmail | quote }}
|
||||||
|
- name: TERDUT_OIDC_SESSION_MAX_AGE
|
||||||
|
value: {{ .Values.oidc.sessionMaxAge | quote }}
|
||||||
|
{{- if .Values.oidc.allowedGroups }}
|
||||||
|
- name: TERDUT_OIDC_ALLOWED_GROUPS
|
||||||
|
value: {{ join "," .Values.oidc.allowedGroups | quote }}
|
||||||
|
{{- end }}
|
||||||
|
{{- if .Values.oidc.adminGroup }}
|
||||||
|
- name: TERDUT_OIDC_ADMIN_GROUP
|
||||||
|
value: {{ .Values.oidc.adminGroup | quote }}
|
||||||
|
{{- end }}
|
||||||
|
{{- end }}
|
||||||
livenessProbe:
|
livenessProbe:
|
||||||
httpGet:
|
httpGet:
|
||||||
path: /healthz
|
path: /healthz
|
||||||
|
|||||||
@@ -1,3 +1,10 @@
|
|||||||
|
# Safe above 1 since v0.36.0: the sweeper, notifier and migration runner each
|
||||||
|
# take a Postgres advisory lock around their own pass, and a webhook that
|
||||||
|
# loses the race to open a brand-new incident attaches to the winner's row
|
||||||
|
# instead of dropping its payload. An image older than v0.36.0 does not have
|
||||||
|
# these guards -- do not raise this against one.
|
||||||
|
replicaCount: 2
|
||||||
|
|
||||||
networking:
|
networking:
|
||||||
hostname: "terdut.example.com"
|
hostname: "terdut.example.com"
|
||||||
servicePort: 8080
|
servicePort: 8080
|
||||||
@@ -33,6 +40,26 @@ database:
|
|||||||
passwordSecret:
|
passwordSecret:
|
||||||
name: ""
|
name: ""
|
||||||
key: password
|
key: password
|
||||||
|
# Blocks the main container from starting until Postgres accepts
|
||||||
|
# connections. Without this, a Deployment created before Postgres has
|
||||||
|
# finished its very first boot -- initdb plus Patroni leader election, on a
|
||||||
|
# from-scratch postgres-operator cluster -- crash-loops a few times: the
|
||||||
|
# app's own ping-retry budget on startup (pingAttempts/pingRetryDelay in
|
||||||
|
# internal/db/db.go) is sized for a much shorter, different race --
|
||||||
|
# NetworkPolicy propagation, a few seconds -- not for genuine first-time
|
||||||
|
# cluster creation, which routinely takes longer, so it exhausts and the
|
||||||
|
# process exits before ever binding its HTTP port. A startupProbe cannot
|
||||||
|
# help here: the crash happens before there is anything to probe.
|
||||||
|
#
|
||||||
|
# pg_isready needs no credentials -- it reports PQPING_OK on anything that
|
||||||
|
# amounts to "a Postgres backend answered", including an auth challenge --
|
||||||
|
# so no PGPASSWORD is wired into this container.
|
||||||
|
waitForPostgres:
|
||||||
|
enabled: true
|
||||||
|
image:
|
||||||
|
repository: postgres
|
||||||
|
tag: "17-alpine"
|
||||||
|
pullPolicy: IfNotPresent
|
||||||
|
|
||||||
service:
|
service:
|
||||||
type: ClusterIP
|
type: ClusterIP
|
||||||
@@ -45,43 +72,10 @@ sweeper:
|
|||||||
# How long a resolved alert stays in the default list before auto-archiving.
|
# How long a resolved alert stays in the default list before auto-archiving.
|
||||||
archiveAfter: 168h
|
archiveAfter: 168h
|
||||||
|
|
||||||
# Alerts treated as dead man's switches: receiving one opens no incident, and
|
# How many reverse proxies in front of the server append to X-Forwarded-For.
|
||||||
# the absence of one does. The Watchdog alert kube-prometheus-stack ships is
|
# The per-address login/sign-up rate limits take the client address that many
|
||||||
# exactly this — an always-firing alert whose only value is something noticing
|
# entries from the right. 0 ignores the header.
|
||||||
# when it stops.
|
trustedProxies: 1
|
||||||
deadman:
|
|
||||||
# Which alerts to treat as heartbeats. ";" separates matchers, "," separates
|
|
||||||
# the label conditions within one, "=" is exact equality. Every matcher must
|
|
||||||
# name an alertname:
|
|
||||||
# alertname=Watchdog,cluster=prod; alertname=EdgeHeartbeat
|
|
||||||
# Each distinct label set is watched independently, so two clusters sending
|
|
||||||
# the same alertname are two switches and a live one cannot mask a dead one.
|
|
||||||
matchers: "alertname=Watchdog"
|
|
||||||
# How long a heartbeat may go unheard before its switch is declared dead.
|
|
||||||
#
|
|
||||||
# This must be SHORTER than the Alertmanager repeat_interval of the route
|
|
||||||
# carrying the heartbeat — the opposite of sweeper.staleAfter. The default
|
|
||||||
# repeat_interval of 4h (12h in many setups) makes for a useless dead man's
|
|
||||||
# switch, so give the heartbeat a route of its own:
|
|
||||||
#
|
|
||||||
# - matchers: [ 'alertname = "Watchdog"' ]
|
|
||||||
# receiver: terdut
|
|
||||||
# group_wait: 0s
|
|
||||||
# group_interval: 1m
|
|
||||||
# repeat_interval: 1m
|
|
||||||
#
|
|
||||||
# That delivers every 2m rather than every 1m: a group is only reconsidered
|
|
||||||
# each group_interval, and at exactly one elapsed interval repeat_interval has
|
|
||||||
# not quite passed, so equal values give 2x. Fine against 15m; use
|
|
||||||
# group_interval: 30s if you want a true 1m.
|
|
||||||
#
|
|
||||||
# Set to 0 to disable dead man's switch handling entirely.
|
|
||||||
timeout: 15m
|
|
||||||
# Severity a dead man's switch incident opens at. These incidents have no
|
|
||||||
# member alerts to derive one from, and the heartbeat's own severity label is
|
|
||||||
# meaningless — Watchdog ships as "none". Only "critical" maps to the ntfy
|
|
||||||
# priority that overrides a phone's quiet hours.
|
|
||||||
severity: critical
|
|
||||||
|
|
||||||
notify:
|
notify:
|
||||||
# ntfy server that push notifications are published to, e.g.
|
# ntfy server that push notifications are published to, e.g.
|
||||||
@@ -107,10 +101,64 @@ notify:
|
|||||||
name: ""
|
name: ""
|
||||||
key: token
|
key: token
|
||||||
|
|
||||||
# Backups are no longer this chart's business. The SQLite database lived on a PVC
|
# Whether a user may sign in, or sign up, with a password. Turn it off once
|
||||||
# beside the app, so it needed a sidecar with a sqlite3 module for k8up to exec a
|
# single sign-on works, to make it the only way in; turn it back on (and
|
||||||
# dump in; Postgres is backed up where it runs, through a k8up.io/backupcommand
|
# redeploy) if the identity provider is down and somebody has to get in.
|
||||||
# pg_dump annotation on the database pod itself.
|
passwordLogin: true
|
||||||
|
|
||||||
|
# Declares this install gitops-managed: writes to teams, escalation policies,
|
||||||
|
# dead man's switches and integrations from a session or a user's own API key
|
||||||
|
# are refused, while a service account's (see SERVICE-ACCOUNTS.md) are not.
|
||||||
|
# Off by default — turning it on is a statement that something like
|
||||||
|
# terdut-operator, not a person in the web UI, owns this install's
|
||||||
|
# configuration from here on.
|
||||||
|
operatorMode: false
|
||||||
|
|
||||||
|
# Single sign-on through an OpenID Connect provider such as Authentik.
|
||||||
|
#
|
||||||
|
# At the provider, create an OAuth2/OpenID application whose redirect URI is
|
||||||
|
# <notify.publicUrl>/api/oidc/callback
|
||||||
|
# (publicUrl defaults to https://<networking.hostname>), a confidential client, and
|
||||||
|
# put the client secret in an existing Secret named by clientSecret below.
|
||||||
|
#
|
||||||
|
# Groups from the provider decide what a person can do. Access it grants is
|
||||||
|
# marked as managed by single sign-on and is re-read at every sign-in; anything
|
||||||
|
# added by hand in terdut is left alone. Changes in the provider take effect at
|
||||||
|
# the person's next sign-in, at most sessionMaxAge later. API keys are NOT
|
||||||
|
# revoked when somebody is removed at the provider: disable the user in terdut too.
|
||||||
|
oidc:
|
||||||
|
enabled: false
|
||||||
|
# Issuer URL. For Authentik: https://<authentik>/application/o/<app-slug>/
|
||||||
|
issuer: ""
|
||||||
|
clientId: ""
|
||||||
|
clientSecret:
|
||||||
|
name: ""
|
||||||
|
key: client-secret
|
||||||
|
# What the sign-in button calls the provider.
|
||||||
|
name: SSO
|
||||||
|
# Authentik puts the groups claim behind the profile scope.
|
||||||
|
scopes: "openid profile email"
|
||||||
|
usernameClaim: preferred_username
|
||||||
|
emailClaim: email
|
||||||
|
groupsClaim: groups
|
||||||
|
# Link a first sign-in to an existing local user with the same email even when
|
||||||
|
# the provider does not mark the address verified. Authentik reports
|
||||||
|
# email_verified as false unless configured otherwise.
|
||||||
|
trustEmail: false
|
||||||
|
# Only people in one of these groups may sign in. Empty admits everybody the
|
||||||
|
# provider authenticates, and access control is left to the provider.
|
||||||
|
allowedGroups: []
|
||||||
|
# Members of this group are system administrators.
|
||||||
|
adminGroup: ""
|
||||||
|
# Which group grants a team's membership and ownership is each team's own
|
||||||
|
# setting now, not chart config: an owner sets it from the Members tab, or
|
||||||
|
# PUT /api/teams/{teamID}/oidc-groups. A team must already exist for a group
|
||||||
|
# to grant access to it.
|
||||||
|
# Hard ceiling on a session made by a single sign-on login.
|
||||||
|
sessionMaxAge: 12h
|
||||||
|
|
||||||
|
# Backups are not this chart's business: Postgres is backed up where it runs,
|
||||||
|
# through a k8up.io/backupcommand pg_dump annotation on the database pod itself.
|
||||||
|
|
||||||
bootstrap:
|
bootstrap:
|
||||||
enabled: true
|
enabled: true
|
||||||
|
|||||||
@@ -17,6 +17,9 @@ var version = "dev"
|
|||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
cfg := config.Load()
|
cfg := config.Load()
|
||||||
|
if err := cfg.Validate(); err != nil {
|
||||||
|
log.Fatalf("config: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
database, err := db.Open(cfg.DSN)
|
database, err := db.Open(cfg.DSN)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -36,22 +39,17 @@ func main() {
|
|||||||
RepeatEvery: cfg.NotifyRepeat,
|
RepeatEvery: cfg.NotifyRepeat,
|
||||||
}
|
}
|
||||||
|
|
||||||
// Dead man's switches live per team now. The environment variables are the
|
|
||||||
// defaults a team starts from: every team without a configuration of its
|
|
||||||
// own gets one from them here, and an owner's later edit is never
|
|
||||||
// overwritten by a redeploy.
|
|
||||||
deadman := api.ParseDeadmanConfig(cfg.DeadmanMatchers, cfg.DeadmanTimeout, cfg.DeadmanSeverity)
|
|
||||||
if err := api.SeedDeadmanConfigs(context.Background(), database, deadman); err != nil {
|
|
||||||
log.Fatalf("seed dead man's switch defaults: %v", err)
|
|
||||||
}
|
|
||||||
|
|
||||||
// The behaviour knobs move into the database on first start, after which an
|
// The behaviour knobs move into the database on first start, after which an
|
||||||
// administrator owns them and a redeploy leaves them alone.
|
// administrator owns them and a redeploy leaves them alone.
|
||||||
if err := api.SeedSettings(context.Background(), database, cfg); err != nil {
|
if err := api.SeedSettings(context.Background(), database, cfg); err != nil {
|
||||||
log.Fatalf("seed settings: %v", err)
|
log.Fatalf("seed settings: %v", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
router := api.NewRouter(database, notify, cfg)
|
if err := api.SeedOperatorKey(context.Background(), database, cfg.OperatorKey); err != nil {
|
||||||
|
log.Fatalf("%v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
router := api.NewRouter(database, notify, cfg, version)
|
||||||
|
|
||||||
srv := &http.Server{
|
srv := &http.Server{
|
||||||
Addr: cfg.Addr,
|
Addr: cfg.Addr,
|
||||||
|
|||||||
@@ -0,0 +1,21 @@
|
|||||||
|
# Terminal Duty documentation
|
||||||
|
|
||||||
|
The [README](../README.md) is the short tour. These pages hold the detail.
|
||||||
|
|
||||||
|
**Running it**
|
||||||
|
- [Deployment](./deployment.md): Docker, the Helm chart, the database, backups, and the operator.
|
||||||
|
- [Configuration](./configuration.md): environment variables and settings.
|
||||||
|
- [Single sign-on](./single-sign-on.md): OIDC, group mapping, the terminal device flow.
|
||||||
|
|
||||||
|
**Using it**
|
||||||
|
- [The web UI](./web-ui.md): sessions, the Team and Admin tabs.
|
||||||
|
- [Alertmanager configuration](./alertmanager.md): routes, integration keys and webhooks.
|
||||||
|
- [Alerts and incidents](./incidents.md): correlation, lifecycle, on-call assignment, stale-alert expiry.
|
||||||
|
- [Push notifications](./notifications.md): ntfy pages and acknowledging from them.
|
||||||
|
- [Escalation](./escalation.md): ladders, repeats and the fallback topic.
|
||||||
|
- [Dead man's switches](./dead-mans-switch.md): noticing that alerts stopped arriving.
|
||||||
|
|
||||||
|
**Integrating and contributing**
|
||||||
|
- [API reference](./api.md): every endpoint, authentication and error shape.
|
||||||
|
- [Service accounts](../SERVICE-ACCOUNTS.md): non-human credentials for automation.
|
||||||
|
- [Development and releasing](./development.md): tests, the CI gate, the release pipeline.
|
||||||
@@ -0,0 +1,62 @@
|
|||||||
|
# Alertmanager configuration
|
||||||
|
|
||||||
|
_Pointing Alertmanager at the server._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
Alerts arrive on a team's **integration key**, which says both that the sender
|
||||||
|
may post and which team the alerts belong to. Mint one as an owner of the team:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl -X POST https://terdut.example.com/api/teams/1/integrations \
|
||||||
|
-H "Authorization: Bearer $TERDUT_API_KEY" \
|
||||||
|
-H 'Content-Type: application/json' \
|
||||||
|
-d '{"name":"prod alertmanager"}'
|
||||||
|
```
|
||||||
|
|
||||||
|
The response carries the key and the full URL **once**; only a SHA-256 hash is
|
||||||
|
stored. Put it in your `alertmanager.yml`:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
receivers:
|
||||||
|
- name: terdut
|
||||||
|
webhook_configs:
|
||||||
|
- url: http://terdut-server:8080/api/integrations/<key>/alertmanager
|
||||||
|
send_resolved: true
|
||||||
|
|
||||||
|
route:
|
||||||
|
receiver: terdut
|
||||||
|
```
|
||||||
|
|
||||||
|
The whole URL is a credential, so treat it like one. Alertmanager 0.26 and
|
||||||
|
later can read it from a file with `url_file:` instead, which keeps it out of
|
||||||
|
your configuration repository:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
- url_file: /etc/alertmanager/secrets/terdut-webhook-url/url
|
||||||
|
send_resolved: true
|
||||||
|
```
|
||||||
|
|
||||||
|
The webhook endpoint requires no authentication.
|
||||||
|
|
||||||
|
If you use the [dead man's switch](./dead-mans-switch.md) — and the default configuration does — give
|
||||||
|
the heartbeat a route of its own, because the deadline is only as tight as the interval feeding it:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
route:
|
||||||
|
receiver: terdut
|
||||||
|
repeat_interval: 4h
|
||||||
|
routes:
|
||||||
|
- matchers: [ 'alertname = "Watchdog"' ]
|
||||||
|
receiver: terdut
|
||||||
|
group_wait: 0s
|
||||||
|
group_interval: 1m
|
||||||
|
repeat_interval: 1m
|
||||||
|
```
|
||||||
|
|
||||||
|
That delivers a heartbeat every **2 minutes**, not every minute. Alertmanager only reconsiders a
|
||||||
|
group every `group_interval`, and at exactly one elapsed interval `repeat_interval` has not *quite*
|
||||||
|
passed, so the send slips to the next tick — equal values give 2×. Two minutes against the 15 minute
|
||||||
|
default is seven heartbeats per window, which is the point; use `group_interval: 30s` if you want
|
||||||
|
the numbers to mean what they say.
|
||||||
|
|
||||||
|
kube-prometheus-stack users get the `Watchdog` alert (`expr: vector(1)`) for free; it just needs
|
||||||
|
routing to terdut rather than to `null`.
|
||||||
@@ -0,0 +1,436 @@
|
|||||||
|
# API reference
|
||||||
|
|
||||||
|
_The REST API._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
## Authentication
|
||||||
|
|
||||||
|
All endpoints except `/api/bootstrap`, `/api/integrations/{key}/alertmanager`,
|
||||||
|
`/api/notify/ack/{token}`, `/api/login`, `/api/logout`, `/api/auth/config`,
|
||||||
|
`/api/version`, `/api/oidc/login`, `/api/oidc/callback`, `/api/oidc/device` and
|
||||||
|
`/api/oidc/device/token` require either an API key:
|
||||||
|
|
||||||
|
```
|
||||||
|
Authorization: Bearer <api-key>
|
||||||
|
```
|
||||||
|
|
||||||
|
or the web UI's session cookie. A request that carries an `Authorization` header
|
||||||
|
is judged on that header alone.
|
||||||
|
|
||||||
|
Two kinds of user exist. An **administrator** manages accounts: creating and
|
||||||
|
deleting users, setting anybody's password, minting keys for anybody, and
|
||||||
|
granting the flag itself. Everybody else works incidents — acknowledging,
|
||||||
|
assigning, snoozing, resolving, noting — and manages their own account and
|
||||||
|
nobody else's. An API key carries exactly the rights of the user it belongs to.
|
||||||
|
|
||||||
|
A third principal, the **service account**, exists for automation (a
|
||||||
|
Kubernetes operator, most likely) that needs to manage teams, escalation
|
||||||
|
policies, dead man's switches and integrations without impersonating a human.
|
||||||
|
It is not a user — it never signs in, never appears in a team's member list,
|
||||||
|
and never holds the administrator flag — and its key is prefixed `tdsa_` so it
|
||||||
|
reads as one at a glance in a log line. See [Service accounts](#service-accounts).
|
||||||
|
|
||||||
|
**Getting an account.** The first one comes from `/api/bootstrap`. After that
|
||||||
|
it depends on `signup_mode`, an administrator setting:
|
||||||
|
|
||||||
|
- `invite_only` (the default) — a team owner mints a link with
|
||||||
|
`POST /api/teams/{teamID}/invites`, and the person who opens it picks a
|
||||||
|
username and password and lands in that team with the role the link carries.
|
||||||
|
Links are single-use unless told otherwise, expire after seven days, and can
|
||||||
|
be revoked before that.
|
||||||
|
- `open` — anybody who can reach the server can create an account, and must
|
||||||
|
name a team, which they then own.
|
||||||
|
|
||||||
|
Invites are **links, not email**: this server has no SMTP, and adding it to send
|
||||||
|
one message would be a subsystem to run, secure and monitor. Send the link
|
||||||
|
however you already talk to the person.
|
||||||
|
|
||||||
|
A domain-restricted third mode was considered and dropped: with no email there
|
||||||
|
is nothing to verify an address against, so it would only check the domain of a
|
||||||
|
string somebody typed.
|
||||||
|
|
||||||
|
The first user, from `/api/bootstrap`, is an administrator. Users created
|
||||||
|
afterwards are not, until an administrator says so. An install always keeps at
|
||||||
|
least one: the last administrator can be neither deleted nor demoted, and
|
||||||
|
nobody can delete or demote themselves.
|
||||||
|
|
||||||
|
Endpoints that require the flag answer `403` with
|
||||||
|
`{"error":"administrator access required"}`.
|
||||||
|
|
||||||
|
**Teams** are the unit of tenancy, and are a separate axis from the administrator
|
||||||
|
flag. A team owns its incidents, alerts, schedule and integrations, and a user
|
||||||
|
sees exactly the teams they belong to. Within a team an **owner** configures it
|
||||||
|
(schedule, integrations, membership) and a **member** works its incidents.
|
||||||
|
|
||||||
|
An administrator crosses that line in one direction only. They **configure any
|
||||||
|
team** without being in it — every owner-only endpoint accepts the flag, because
|
||||||
|
otherwise a team whose last owner left could never be repaired. They do **not
|
||||||
|
read any team**: the queue, the alerts and the incidents are filtered by real
|
||||||
|
membership, so an administrator sees a team's work only by joining it, which is
|
||||||
|
a membership change and shows up as one. Administration is about accounts and
|
||||||
|
the shape of a team, not about reading other people's incidents.
|
||||||
|
|
||||||
|
Anything belonging to a team you are not in answers `404`, not `403`: whether an
|
||||||
|
incident exists is itself something only its team should learn.
|
||||||
|
|
||||||
|
**Operator mode** (`TERDUT_OPERATOR_MODE`, see [Configuration](./configuration.md#configuration))
|
||||||
|
declares this install gitops-managed. When it is on, a session or a user's own
|
||||||
|
API key gets `403 {"error": "...", "reason": "operator_managed"}` on every
|
||||||
|
write this page marks **owner**-gated under Teams below (creating, renaming
|
||||||
|
or deleting a team; its OIDC group binding; its escalation ladder; its dead
|
||||||
|
man's switches; its integrations) — a service account's writes are unaffected.
|
||||||
|
Team membership and invites are deliberately excluded: they are never
|
||||||
|
gitops-managed, in operator mode or out of it. `GET /api/auth/config` reports
|
||||||
|
`operator_mode` so a client can grey those sections out before a write is ever
|
||||||
|
attempted.
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `GET` | `/api/auth/config` | How to sign in: `{"password_login", "oidc": {"enabled","name"}, "device_login", "operator_mode"}`. No session needed |
|
||||||
|
| `GET` | `/api/version` | `{"version"}` — this build's version string. No session needed, the same as `/healthz` |
|
||||||
|
| `POST` | `/api/login` | `{"username","password"}` → sets the session cookie, returns `{user, has_password}`. `429` after too many failures; `403` when `TERDUT_PASSWORD_LOGIN=false` |
|
||||||
|
| `GET` | `/api/oidc/login` | Starts a single sign-on sign-in: redirects the browser to the provider. `?next=/path` is where to land afterwards; only a path on this server is honoured. Only exists when SSO is configured |
|
||||||
|
| `POST` | `/api/oidc/device` | Starts a device login: returns `{device_code, user_code, verification_url, interval, expires_in}`. Only exists when SSO is configured |
|
||||||
|
| `POST` | `/api/oidc/device/token` | `{"device_code"}` → `202 {"status":"pending"}`, then `200` with the session cookie once approved (once only). `410` with `{"error":"expired"}` or `{"error":"denied"}`; `429 {"error":"slow_down"}` if polled faster than `interval` |
|
||||||
|
| `POST` | `/api/oidc/device/approve` | **session** — `{"user_code"}`. Approves a pending device login as the caller. `403` for an API key; `404` for an unknown, expired or already decided code |
|
||||||
|
| `POST` | `/api/oidc/device/deny` | **session** — `{"user_code"}`. Refuses it |
|
||||||
|
| `GET` | `/api/oidc/callback` | Where the provider sends the browser back. Sets the session cookie and redirects to `/`, or to `/?sso_error=<code>` — one of `denied`, `expired`, `failed`, `unavailable`, `not_allowed`, `no_email`, `email_conflict`, `disabled`, `not_bootstrapped` (no user exists on this install yet — sign in again once something has called `/api/bootstrap`) |
|
||||||
|
| `POST` | `/api/logout` | Ends the session and clears the cookie |
|
||||||
|
| `GET` | `/api/me` | The caller: `{user, has_password}` |
|
||||||
|
|
||||||
|
## Users
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
|---|---|---|
|
||||||
|
**admin** marks an endpoint that requires the administrator flag; **self or
|
||||||
|
admin** marks one you may use on your own account and an administrator may use
|
||||||
|
on anybody's.
|
||||||
|
|
||||||
|
| Method | Path | Who | Description |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `GET` | `/api/signup` | — | Whether sign-up is open, and whether `?invite=` is usable. No session needed: the caller has no account yet |
|
||||||
|
| `POST` | `/api/signup` | — | Create an account `{"username","email","password","invite"?,"team_name"?}` and sign in. `403` without a usable invite when the mode is invite-only |
|
||||||
|
| `POST` | `/api/bootstrap` | — | Create first user + API key `{"username","email","password"?}` (only works on empty DB). The user is an administrator |
|
||||||
|
| `GET` | `/api/users` | any | List users. Open to everybody: the queue's assignment control and the schedule both have to name people |
|
||||||
|
| `GET` | `/api/users/{id}/teams` | self or admin | The teams that user is in, each with their role. `/api/teams` is always about the caller; this one answers it about somebody else, for the admin page's per-user view. `404` for a user who does not exist, so "no teams" and "no such person" are distinguishable |
|
||||||
|
| `POST` | `/api/users` | **admin** | Create user `{"username","email"}`. Not an administrator |
|
||||||
|
| `DELETE` | `/api/users/{id}` | **admin** | Delete user (cascades to keys). `409` for yourself or the last administrator |
|
||||||
|
| `PUT` | `/api/users/{id}/admin` | **admin** | Grant or revoke the administrator flag `{"is_admin"}`. `409` for yourself, the last administrator, or an administrator granted by single sign-on |
|
||||||
|
| `PUT` | `/api/users/{id}/disabled` | **admin** | Take an account out of use, or put it back `{"disabled"}`. `409` for yourself or the last administrator |
|
||||||
|
| `PUT` | `/api/users/{id}/notify` | self or admin | Set push notification target `{"ntfy_topic"}` — empty string clears it |
|
||||||
|
| `PUT` | `/api/users/{id}/password` | self or admin | Set web UI password `{"password","current_password"}`. `current_password` is required only when changing your own existing password. Ends the user's other sessions |
|
||||||
|
| `POST` | `/api/users/{id}/api-keys` | self or admin | Issue API key `{"name"}` — key shown once |
|
||||||
|
| `DELETE` | `/api/users/{id}/api-keys/{keyID}` | self or admin | Revoke API key |
|
||||||
|
|
||||||
|
## Administration
|
||||||
|
|
||||||
|
| Method | Path | Who | Description |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `GET` | `/api/admin/teams` | **admin** | Every team on the server, with its member and open-incident counts. `/api/teams` answers "what am I in"; this answers "what is there" |
|
||||||
|
| `GET` | `/api/admin/teams/{teamID}` | **admin** | One team and who is in it: `{"team", "members"}`. `404` for a team that does not exist. `GET /api/teams/{teamID}/members` is **member**-only and still `404`s an administrator from outside the team — reading a team's shape and reading its work are different questions, so they are different endpoints |
|
||||||
|
| `GET` | `/api/admin/settings` | **admin** | The editable settings with their bounds, plus the environment-configured ones, read-only. Never credentials |
|
||||||
|
| `PUT` | `/api/admin/settings` | **admin** | Change one or more `{"key": seconds}`, or `{"signup_mode": "open"\|"invite_only"}`. `400` for an unknown key or a value outside its bounds |
|
||||||
|
|
||||||
|
## Service accounts
|
||||||
|
|
||||||
|
A service account is a scoped, non-human credential for automation — not a
|
||||||
|
`users` row, so it never signs in, is never a team member, and never carries
|
||||||
|
the administrator flag. Two scopes:
|
||||||
|
|
||||||
|
- **instance** — the same reach system administration has over teams: create
|
||||||
|
one, and mint a **team**-scoped account against any of them. There is no
|
||||||
|
cap on how many instance-scoped accounts exist, but ordinarily there is one,
|
||||||
|
belonging to whatever is provisioning this install end to end.
|
||||||
|
- **team** — owner-equivalent for that one team, and nothing else: every
|
||||||
|
**owner**-gated endpoint under [Teams](#teams), membership and invites
|
||||||
|
included. Nothing narrower is enforced server-side; what actually keeps
|
||||||
|
membership out of automation's hands is that no operator built against this
|
||||||
|
scope should ever call those two endpoints — see
|
||||||
|
[operator mode](#authentication) and [`SERVICE-ACCOUNTS.md`](../SERVICE-ACCOUNTS.md)'s note on this.
|
||||||
|
|
||||||
|
A key is shown once, at creation or rotation, and only its hash is stored —
|
||||||
|
the same handling as a user's API key. Losing it means minting a new one;
|
||||||
|
there is no way to recover a raw key from the server.
|
||||||
|
|
||||||
|
| Method | Path | Who | Description |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `GET` | `/api/service-accounts` | **admin** | Every service account. Pass `?name=` instead to look one up by its exact name — open to **any** authenticated caller (human or service account), since it returns no key material and is how an account finds its own id |
|
||||||
|
| `POST` | `/api/service-accounts` | owner\* | Create one and mint its first key `{"name","scope","team_id"?}` (`team_id` required for `scope:"team"`, absent for `scope:"instance"`). Returns `{"service_account", "key"}` — `key.key` shown once |
|
||||||
|
| `POST` | `/api/service-accounts/{id}/keys` | owner\* | Mint an additional key `{"name"}` — rotation without recreating the account. Shown once |
|
||||||
|
| `DELETE` | `/api/service-accounts/{id}/keys/{keyID}` | owner\* | Revoke one key |
|
||||||
|
|
||||||
|
\* For an **instance**-scoped account: a system administrator only. For a
|
||||||
|
**team**-scoped account: a system administrator, that team's own human owner,
|
||||||
|
an instance-scoped service account (minting a narrower credential for a team
|
||||||
|
it just created), or — for the two key endpoints only — the account rotating
|
||||||
|
or revoking its own key, which is not a privilege escalation, the same
|
||||||
|
reasoning a user's own API keys rest on.
|
||||||
|
|
||||||
|
## Alert ingestion
|
||||||
|
|
||||||
|
Alerts arrive on a team's integration key. The key is both the credential and the
|
||||||
|
routing: it says that the sender may post, and which team the alerts belong to.
|
||||||
|
Create one with `POST /api/teams/{teamID}/integrations`, which returns the key
|
||||||
|
and the full URL once and stores only a SHA-256 hash.
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `POST` | `/api/integrations/{key}/alertmanager` | Alertmanager v4 webhook receiver for the key's team. `401` for an unknown key |
|
||||||
|
|
||||||
|
This is the only way in. The pre-teams `POST /api/alertmanager/webhook` took no
|
||||||
|
credential at all — anything able to reach the port could open an incident —
|
||||||
|
and was removed in v0.13.0 once senders had moved onto keys.
|
||||||
|
|
||||||
|
## Teams
|
||||||
|
|
||||||
|
**owner** below means an owner of that team, a system administrator (who
|
||||||
|
passes every one of these without being a member), or that team's own
|
||||||
|
team-scoped [service account](#service-accounts) — including membership and
|
||||||
|
invites, technically, though no automation this scope was designed for
|
||||||
|
(a Kubernetes operator's CRDs, see [`SERVICE-ACCOUNTS.md`](../SERVICE-ACCOUNTS.md)) ever models team
|
||||||
|
membership or would call those two. See [Authentication](#authentication).
|
||||||
|
**member** means membership and nothing else: an administrator who is not in
|
||||||
|
the team gets the same `404` as anybody else.
|
||||||
|
|
||||||
|
| Method | Path | Who | Description |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `GET` | `/api/teams` | any | The caller's own teams, each with their role |
|
||||||
|
| `POST` | `/api/teams` | any | Create a team `{"name"}`; a human creator becomes its first owner. An instance-scoped [service account](#service-accounts) may also create one, and it gets no owner at all — expected for a team an operator is about to hand a team-scoped credential to, not an orphaned team a human made |
|
||||||
|
| `PUT` | `/api/teams/{teamID}` | **owner** | Rename it `{"name"}`. `409` if the name is taken |
|
||||||
|
| `DELETE` | `/api/teams/{teamID}` | **owner** | Delete a team and everything under it. `409` while it has open incidents |
|
||||||
|
| `GET` | `/api/teams/{teamID}/members` | member | Who is in the team, with `status` (`oncall` if the rota has them today, `unpageable` when a page to them would go nowhere — even if they are on call — else `reachable`), `on_call`, `next_shift` (first rota day after today), `pageable` and `problem` (`has no ntfy topic` / `account is disabled`; never the topic itself) and `last_active_at` (their newest session or API-key use). Every member sees the same list |
|
||||||
|
| `POST` | `/api/teams/{teamID}/members` | **owner** | Add a member, or change their role `{"user_id","role"}`. `409` when it would demote the last owner, or the membership is managed by single sign-on |
|
||||||
|
| `DELETE` | `/api/teams/{teamID}/members/{userID}` | **owner** | Remove a member. `409` for the last owner, or a membership managed by single sign-on |
|
||||||
|
| `GET` | `/api/teams/{teamID}/oidc-groups` | member | Which groups control this team's membership: `{"member_group","owner_group"}`. An empty string means no group grants that role here |
|
||||||
|
| `PUT` | `/api/teams/{teamID}/oidc-groups` | **owner** | Set them. An empty string clears a binding |
|
||||||
|
| `GET` | `/api/teams/{teamID}/integrations` | member | List integrations. Never returns keys. Each carries `status` (`active` if its key posted within 24h, `quiet` if it has but not lately, `never`), `last_used_at` (last webhook, usable or not), `last_alert_at` (when an alert last arrived on it) and `alerts_24h` (distinct alerts it refreshed in the last day). Alerts delivered before the source was recorded (migration 010) have none, so the last two fill in as Alertmanager re-sends them |
|
||||||
|
| `PATCH` | `/api/teams/{teamID}/integrations/{integrationID}` | **owner** | Rename `{"name"}`. The key does not change |
|
||||||
|
| `POST` | `/api/teams/{teamID}/integrations` | **owner** | Mint an integration `{"name","kind"}` — key and URL shown once |
|
||||||
|
| `DELETE` | `/api/teams/{teamID}/integrations/{integrationID}` | **owner** | Revoke an integration. Alerts it delivered stay, unattributed |
|
||||||
|
| `GET` | `/api/teams/{teamID}/invites` | **owner** | The team's invite links, with their uses and expiry. Never the tokens |
|
||||||
|
| `POST` | `/api/teams/{teamID}/invites` | **owner** | Mint one `{"role","max_uses"}` — the full URL is returned once |
|
||||||
|
| `DELETE` | `/api/teams/{teamID}/invites/{inviteID}` | **owner** | Revoke a link before it expires |
|
||||||
|
| `GET` | `/api/teams/{teamID}/escalation` | member | The team's [escalation ladder](./escalation.md#escalation) `{repeat_count, fallback_topic, levels[], last_escalated_at?, last_escalated_incident_id?}`. Empty levels means the team has none. Each level also carries `status` (`ready`, `escalating` when an unanswered incident has climbed to it, `unreachable` when nobody on it could be woken), `waiting` (ids of the open incidents on it) and, per target, `username` (who it means today — the person on call, for a rota target), `reachable` and `problem`. The extra fields are output only; `PUT` takes the plain shape |
|
||||||
|
| `PUT` | `/api/teams/{teamID}/escalation` | **owner** | Replace it wholesale. `400` for a level with no targets or no timeout — a rung that pages nobody is a silence with a number on it |
|
||||||
|
| `GET` | `/api/teams/{teamID}/deadman/switches` | member | The team's [dead man's switches](./dead-mans-switch.md), each `{id, name, matcher, timeout_seconds, severity, status, last_heartbeat_at, last_triggered_at, open_incident_id, sources[]}`. `status` is `healthy`, `dead` or `dormant`; `sources` has one entry per heartbeat fingerprint. Empty when the team watches nothing |
|
||||||
|
| `POST` | `/api/teams/{teamID}/deadman/switches` | **owner** | Add one: `{name?, matcher, timeout_seconds, severity?}`. `400` when the matcher names no `alertname` or holds several, or the timeout is not positive — a switch that silently watches nothing is the failure this feature exists to prevent |
|
||||||
|
| `PUT` | `/api/teams/{teamID}/deadman/switches/{switchID}` | **owner** | Replace one in place, same body and validation as create. Its id is unchanged — for an automated caller reconciling a spec change, unlike delete-and-recreate |
|
||||||
|
| `DELETE` | `/api/teams/{teamID}/deadman/switches/{switchID}` | **owner** | Stop watching. An incident it opened stays open. `404` for a switch of another team |
|
||||||
|
|
||||||
|
## Notifications
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `POST` | `/api/notify/ack/{token}` | Acknowledge an incident from a push notification's Acknowledge button. No auth: the token in the path is the credential — one incident, one action, 24 hours, idempotent. Must stay publicly reachable |
|
||||||
|
|
||||||
|
## Incidents
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `GET` | `/api/incidents` | List incidents. Filters: `?status=triggered\|acknowledged\|resolved`, `?severity=`, `?assigned_to=<user id>`, `?archived=true`, `?snoozed=true`, `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD`, `?sort=severity`, `?cluster=<value of the cluster group label>`, `?limit=` (default 50, max 500) |
|
||||||
|
| `GET` | `/api/incidents/clusters` | The distinct `cluster` values on the caller's incidents from the last 90 days, sorted (`?team_id=` narrows it). An empty array when nothing carries the label |
|
||||||
|
| `GET` | `/api/incidents/{id}` | Get single incident, with its alerts inline |
|
||||||
|
| `GET` | `/api/incidents/{id}/alerts` | Alerts under this incident |
|
||||||
|
| `GET` | `/api/incidents/{id}/timeline` | Full event history, chronological |
|
||||||
|
| `POST` | `/api/incidents/{id}/acknowledge` | Acknowledge (stamps authed user + time) |
|
||||||
|
| `DELETE` | `/api/incidents/{id}/acknowledge` | Clear acknowledgement, back to `triggered` |
|
||||||
|
| `POST` | `/api/incidents/{id}/resolve` | Close by hand — **terminal**, see above |
|
||||||
|
| `POST` | `/api/incidents/{id}/assign` | Reassign `{"user_id"}` |
|
||||||
|
| `POST` | `/api/incidents/{id}/snooze` | Hide until `{"until": RFC3339}` or `{"duration": "2h"}` |
|
||||||
|
| `DELETE` | `/api/incidents/{id}/snooze` | Un-snooze |
|
||||||
|
| `POST` | `/api/incidents/{id}/archive` | Archive (hides from the default list) |
|
||||||
|
| `DELETE` | `/api/incidents/{id}/archive` | Un-archive |
|
||||||
|
| `POST` | `/api/incidents/{id}/notes` | Add a note `{"content"}` |
|
||||||
|
| `DELETE` | `/api/incidents/{id}/notes/{eventID}` | Delete own note |
|
||||||
|
|
||||||
|
With no `?status=` filter, `GET /api/incidents` returns **open** incidents only —
|
||||||
|
the queue an on-call person wants. Currently snoozed and archived incidents are
|
||||||
|
excluded unless asked for. Actions that only make sense on an open incident
|
||||||
|
return `409` once it is resolved.
|
||||||
|
|
||||||
|
Notes are ordinary timeline events of type `note`; only they are deletable, and
|
||||||
|
only by their author. The rest of the timeline is a record of what happened.
|
||||||
|
|
||||||
|
### The incident object
|
||||||
|
|
||||||
|
| Field | Type | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| `id` | integer | Server-assigned |
|
||||||
|
| `group_key` | string | Alertmanager's `groupKey` — opaque, treat as an identifier |
|
||||||
|
| `title` | string | Rendered from `groupLabels` |
|
||||||
|
| `group_labels` | object | String→string, as sent by Alertmanager |
|
||||||
|
| `status` | string | `"triggered"`, `"acknowledged"` or `"resolved"` |
|
||||||
|
| `severity` | string | *optional* — high-water mark across the incident's alerts; never lowered |
|
||||||
|
| `triggered_at` | timestamp | When the incident opened |
|
||||||
|
| `acknowledged_by_id` / `acknowledged_by` / `acknowledged_at` | | *optional* — user id, username, time |
|
||||||
|
| `assigned_to_id` / `assigned_to` | | *optional* — user id, username |
|
||||||
|
| `snoozed_until` | timestamp | *optional* — a value in the past reads as not snoozed |
|
||||||
|
| `resolved_at` | timestamp | *optional* |
|
||||||
|
| `resolution_source` | string | *optional* — `"alerts"`, `"manual"` or `"recovered"` |
|
||||||
|
| `archived_at` | timestamp | *optional* |
|
||||||
|
| `alerts` | array | Only on `GET /api/incidents/{id}` |
|
||||||
|
|
||||||
|
Treat `resolution_source` as an open set, as with the alert field of the same
|
||||||
|
name: degrade unknown values to "resolved, reason unknown".
|
||||||
|
|
||||||
|
### The timeline event object
|
||||||
|
|
||||||
|
| Field | Type | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| `id` | integer | |
|
||||||
|
| `incident_id` | integer | |
|
||||||
|
| `type` | string | See below — treat as an open set |
|
||||||
|
| `user_id` / `username` | | *optional* — absent when the server acted rather than a person |
|
||||||
|
| `alert_id` | integer | *optional* — the alert an `alert_added` / `alert_resolved` event refers to |
|
||||||
|
| `detail` | string | *optional* — the note body, the snooze deadline, etc. |
|
||||||
|
| `created_at` | timestamp | |
|
||||||
|
|
||||||
|
Types written today: `triggered`, `alert_added`, `alert_resolved`,
|
||||||
|
`acknowledged`, `unacknowledged`, `assigned`, `archived`, `unarchived`, `snoozed`,
|
||||||
|
`unsnoozed`, `resolved`, `note`, `notified`, `notify_failed`, `deadman_silent`. On an
|
||||||
|
`assigned` event `user_id` is the **assignee**, not the actor; the actor is in
|
||||||
|
`actor_user_id`/`actor_username` or `actor_service_account_id`/`actor_service_account_name`
|
||||||
|
(absent on assignments made before they were recorded). New types may be added; render
|
||||||
|
unknown ones generically rather than dropping them.
|
||||||
|
|
||||||
|
On `notified` and `notify_failed`, `detail` carries the notification kind
|
||||||
|
(`triggered` | `reminder` | `resolved`), and on a failure the reason after it.
|
||||||
|
`user_id` is who was paged — absent means the page went to the shared fallback
|
||||||
|
topic and so belongs to nobody. The topic itself is never written to the
|
||||||
|
timeline: it is a shared secret with the ntfy server, and every API key can read
|
||||||
|
this.
|
||||||
|
|
||||||
|
## Alerts
|
||||||
|
|
||||||
|
Alerts are read-only. Everything a person does happens on the incident.
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `GET` | `/api/alerts` | List alerts. Filters: `?status=firing\|resolved`, `?name=`, `?incident_id=`, `?archived=true`, `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD`, `?limit=` (default 50, max 500) |
|
||||||
|
| `GET` | `/api/alerts/{id}` | Get single alert |
|
||||||
|
|
||||||
|
Archived alerts are hidden from `GET /api/alerts` unless `?archived=true` is
|
||||||
|
passed; alert archiving is automatic housekeeping by the sweeper, not a user
|
||||||
|
action. Resolved alerts carry `resolution_source`: `"alertmanager"` for a real
|
||||||
|
resolved webhook, `"expiry"` when the sweeper inferred it (see
|
||||||
|
[Stale alert expiry](./incidents.md#stale-alert-expiry)), `"deadman"` for a heartbeat declared
|
||||||
|
dead (see [Dead man's switch](./dead-mans-switch.md)).
|
||||||
|
|
||||||
|
### The alert object
|
||||||
|
|
||||||
|
Returned by `GET /api/alerts` (as an array) and `GET /api/alerts/{id}`.
|
||||||
|
Timestamps are RFC 3339 in UTC. Fields marked *optional* are omitted entirely
|
||||||
|
when unset, so clients must treat them as nullable.
|
||||||
|
|
||||||
|
| Field | Type | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| `id` | integer | Server-assigned; stable for the life of the row |
|
||||||
|
| `fingerprint` | string | Alertmanager's fingerprint — the upsert key |
|
||||||
|
| `name` | string | From the `alertname` label |
|
||||||
|
| `status` | string | `"firing"` or `"resolved"` |
|
||||||
|
| `labels` | object | String→string, as sent by Alertmanager |
|
||||||
|
| `annotations` | object | String→string, as sent by Alertmanager |
|
||||||
|
| `starts_at` | timestamp | When the alert instance began, **per Prometheus** |
|
||||||
|
| `ends_at` | timestamp | *optional* — absent while no end is known |
|
||||||
|
| `generator_url` | string | Link back to the originating Prometheus |
|
||||||
|
| `received_at` | timestamp | When the server last accepted a webhook for this alert — see below |
|
||||||
|
| `incident_id` | integer | *optional* — the most recent incident this alert belongs to |
|
||||||
|
| `resolution_source` | string | *optional* — `"alertmanager"`, `"expiry"` or `"deadman"` |
|
||||||
|
| `archived_at` | timestamp | *optional* — set while archived |
|
||||||
|
|
||||||
|
#### `received_at` is a liveness heartbeat
|
||||||
|
|
||||||
|
`starts_at` comes from Prometheus and **never changes** for the lifetime of an
|
||||||
|
alert instance. It says when the problem began, not whether it is still
|
||||||
|
happening — an alert that started twelve days ago looks identical whether
|
||||||
|
Alertmanager refreshed it a minute ago or went silent a week ago.
|
||||||
|
|
||||||
|
`received_at` is the field that answers "is this still live". It is set to the
|
||||||
|
server's clock on **every accepted webhook** for that fingerprint, including the
|
||||||
|
unchanged firing notifications Alertmanager re-sends every `repeat_interval`.
|
||||||
|
Clients may rely on this:
|
||||||
|
|
||||||
|
- **A firing alert whose `received_at` is advancing is still being refreshed.**
|
||||||
|
Stale-dating it against `repeat_interval` is a valid liveness check, and it is
|
||||||
|
what the built-in sweeper does (see
|
||||||
|
[Stale alert expiry](./incidents.md#stale-alert-expiry)).
|
||||||
|
- **`received_at` tracks accepted payloads, not delivery attempts.** A retry
|
||||||
|
that describes an older instance than the stored one is discarded, and a
|
||||||
|
discarded payload does not move `received_at`.
|
||||||
|
- **It stops advancing once the alert resolves,** because Alertmanager stops
|
||||||
|
re-sending. On an alert resolved by the sweeper
|
||||||
|
(`"resolution_source": "expiry"`) it therefore marks the last time
|
||||||
|
Alertmanager was actually heard from, which is earlier than `ends_at`.
|
||||||
|
|
||||||
|
`GET /api/alerts` is ordered by `received_at` descending — most recently
|
||||||
|
refreshed first — and the `?from=` / `?to=` filters on both the alert and stats
|
||||||
|
endpoints select on `received_at`, not `starts_at`.
|
||||||
|
|
||||||
|
#### `resolution_source` says how much to trust `ends_at`
|
||||||
|
|
||||||
|
An alert can leave the firing state two ways, and `resolution_source` records
|
||||||
|
which happened. Clients may rely on this:
|
||||||
|
|
||||||
|
- **Absent while firing.** It is set only on resolve, and a re-fire under the
|
||||||
|
same fingerprint clears it again, so its presence always agrees with
|
||||||
|
`"status": "resolved"`.
|
||||||
|
- **`"alertmanager"` — a real resolved webhook arrived.** `ends_at` is the end
|
||||||
|
time Alertmanager reported. It is an observed value and can be displayed as
|
||||||
|
fact.
|
||||||
|
- **`"expiry"` — the sweeper inferred the resolve** because Alertmanager stopped
|
||||||
|
refreshing the alert (see [Stale alert expiry](./incidents.md#stale-alert-expiry)). Nothing
|
||||||
|
ever reported an end, so **`ends_at` is approximate**: it is either the stale
|
||||||
|
`endsAt` watermark from the last notification, or — when that notification
|
||||||
|
carried none — the time the sweep ran, which lags the last real contact by up
|
||||||
|
to `TERDUT_STALE_AFTER` plus a sweep interval. Treat it as "no later than",
|
||||||
|
not as when the problem stopped.
|
||||||
|
|
||||||
|
On these alerts `received_at` is the more truthful signal: it marks the last
|
||||||
|
time Alertmanager was actually heard from. Surfacing the distinction is
|
||||||
|
worthwhile, since `"expiry"` can also mean the alert is still firing and the
|
||||||
|
notification path broke.
|
||||||
|
|
||||||
|
- **`"deadman"` — a heartbeat was declared dead** (see
|
||||||
|
[Dead man's switch](./dead-mans-switch.md)). Like `"expiry"`, an inference from
|
||||||
|
silence rather than an observed end, so `ends_at` is approximate — but a much
|
||||||
|
tighter one, bounded by the switch's timeout. It is also the one resolution
|
||||||
|
a re-fire under the same `starts_at` can undo, since the switch coming back is
|
||||||
|
exactly the evidence that the inference was wrong.
|
||||||
|
|
||||||
|
Treat the value as an open set and tolerate ones you do not recognise — new
|
||||||
|
sources may be added, and unknown values should degrade to "resolved, reason
|
||||||
|
unknown" rather than being rejected.
|
||||||
|
|
||||||
|
## On-call schedule
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
|---|---|---|
|
||||||
|
Each team keeps its own rota, so two teams can have two different people on call
|
||||||
|
on the same day. The person taking a shift has to be in the team — paging
|
||||||
|
somebody who cannot open the incident is worse than paging nobody.
|
||||||
|
|
||||||
|
| Method | Path | Who | Description |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `POST` | `/api/teams/{teamID}/schedule` | **owner** | Assign user to dates `{"user_id", "dates":["YYYY-MM-DD",...], "replace"}` — all-or-nothing |
|
||||||
|
| `GET` | `/api/teams/{teamID}/schedule` | member | List entries. Filters: `?from=YYYY-MM-DD`, `?to=YYYY-MM-DD` |
|
||||||
|
| `DELETE` | `/api/teams/{teamID}/schedule/{id}` | **owner** | Remove schedule entry |
|
||||||
|
| `GET` | `/api/schedule/current` | any | Who is on call today (UTC) in **every** team the caller is in — one entry per team, `[]` when nobody anywhere |
|
||||||
|
|
||||||
|
## Statistics
|
||||||
|
|
||||||
|
Every figure counts the caller's own teams only: a report that counted other
|
||||||
|
teams' incidents would leak their volume, and their alert names through the
|
||||||
|
top-alerts list, and would not be a number about the reader's work anyway.
|
||||||
|
|
||||||
|
All stat endpoints accept optional `?from=YYYY-MM-DD` and `?to=YYYY-MM-DD`, and exclude archived rows to match the default list views. Alert stats filter on `received_at`; incident stats filter on `triggered_at`.
|
||||||
|
|
||||||
|
| Method | Path | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `GET` | `/api/stats/incidents` | `{total, triggered, acknowledged, resolved, mtta_seconds, mttr_seconds}` |
|
||||||
|
| `GET` | `/api/stats/alerts` | `{total, firing, resolved}` counts |
|
||||||
|
| `GET` | `/api/stats/alerts/top` | Most frequent alert names. `?limit=` (default 10, max 100) |
|
||||||
|
| `GET` | `/api/stats/alerts/by-hour` | Count per hour-of-day (UTC), all 24 slots returned |
|
||||||
|
| `GET` | `/api/stats/alerts/by-day` | Count per day-of-week, all 7 slots with names returned |
|
||||||
|
|
||||||
|
`mtta_seconds` (time to acknowledge) and `mttr_seconds` (time to resolve) are
|
||||||
|
averages over incidents that have actually been acknowledged or resolved, and are
|
||||||
|
**null** until there are any — null means "no data", not zero.
|
||||||
@@ -0,0 +1,49 @@
|
|||||||
|
# Configuration
|
||||||
|
|
||||||
|
_Environment variables and settings._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
Two kinds of setting, split by who changes them and how often.
|
||||||
|
|
||||||
|
**Where the server is plugged in** stays in the environment: the listen address,
|
||||||
|
the database DSN, the ntfy URL and token, the public URL. They are needed before
|
||||||
|
the database is open, and two of them are credentials.
|
||||||
|
|
||||||
|
**How the server behaves** lives in the database and is edited by an
|
||||||
|
administrator in the web UI or through `PUT /api/admin/settings`, taking effect
|
||||||
|
on the next sweep rather than at the next restart. The variables below marked
|
||||||
|
**seed** are the value each of those starts from: written once, on first start,
|
||||||
|
and never overwritten afterwards — a redeploy cannot put a chart's default back
|
||||||
|
over an administrator's edit.
|
||||||
|
|
||||||
|
| Variable | Default | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `TERDUT_ADDR` | `:8080` | TCP address to listen on |
|
||||||
|
| `TERDUT_DB_DSN` | — | **Required.** Postgres connection string, e.g. `postgres://terdut:secret@localhost:5432/terdut?sslmode=require` |
|
||||||
|
| `TERDUT_ARCHIVE_AFTER` | `168h` (7d) | **seed.** How long a resolved alert or incident stays in the default list before being auto-archived |
|
||||||
|
| `TERDUT_STALE_AFTER` | `6h` | **seed.** How long a firing alert may go without a refreshing webhook before it is treated as resolved — **must exceed your Alertmanager `repeat_interval`** |
|
||||||
|
| `TERDUT_NTFY_URL` | — | ntfy server to publish push notifications to. Empty disables notifications entirely |
|
||||||
|
| `TERDUT_NTFY_TOKEN` | — | Bearer token for an access-controlled ntfy |
|
||||||
|
| `TERDUT_NTFY_FALLBACK_TOPIC` | — | Topic used when nobody is on call |
|
||||||
|
| `TERDUT_PUBLIC_URL` | — | Base URL a phone uses to reach this server: the notification's link into the web UI, its Acknowledge button, and whether the session cookie is `Secure` |
|
||||||
|
| `TERDUT_NOTIFY_REPEAT` | `15m` | **seed.** How long an incident may sit unacknowledged before it is paged again. `0` notifies once and never repeats |
|
||||||
|
| `TERDUT_PASSWORD_LOGIN` | `true` | `false` refuses password login and password sign-up (`403`), leaving single sign-on the only way in. Refused at startup unless SSO is configured |
|
||||||
|
| `TERDUT_TRUSTED_PROXIES` | `1` | How many reverse proxies in front of the server append to `X-Forwarded-For`; the per-address rate limits use the entry that many hops from the right. `0` ignores the header |
|
||||||
|
| `TERDUT_OPERATOR_KEY` | — | At least 32 characters. When set, the instance-scoped service account `terdut-operator` is created if missing and its `seed` key replaced with this value at every start — how terdut-operator authenticates without a bootstrap handshake. An instance-scoped account acts as owner of every team (team configuration) but is not a member of any, so it reads no incidents |
|
||||||
|
| `TERDUT_OPERATOR_MODE` | `false` | Declares this install gitops-managed: a session's or a user's own API key's writes to teams, escalation policies, dead man's switches and integrations are refused (`403 reason:"operator_managed"`); a [service account](./api.md#service-accounts)'s are not. Team membership and the schedule stay editable regardless |
|
||||||
|
| `TERDUT_OIDC_ISSUER` | — | Turns single sign-on on. The provider's issuer URL; discovery is read from `<issuer>/.well-known/openid-configuration`. See [Single sign-on](./single-sign-on.md#single-sign-on-oidc) |
|
||||||
|
| `TERDUT_OIDC_CLIENT_ID` / `TERDUT_OIDC_CLIENT_SECRET` | — | **Required with an issuer.** The confidential client registered at the provider. Keep the secret in a Secret, not in values |
|
||||||
|
| `TERDUT_OIDC_NAME` | `SSO` | What the sign-in button calls the provider |
|
||||||
|
| `TERDUT_OIDC_SCOPES` | `openid profile email` | Scopes requested, comma or space separated. Authentik puts `groups` behind `profile` |
|
||||||
|
| `TERDUT_OIDC_USERNAME_CLAIM` / `_EMAIL_CLAIM` / `_GROUPS_CLAIM` | `preferred_username` / `email` / `groups` | ID token claims read for the username, email and groups |
|
||||||
|
| `TERDUT_OIDC_TRUST_EMAIL` | `false` | Link a first sign-in to an existing local user by email even if the provider does not mark the address verified |
|
||||||
|
| `TERDUT_OIDC_ALLOWED_GROUPS` | — | Comma-separated. Only people in one of these may sign in. Empty admits everybody the provider authenticates |
|
||||||
|
| `TERDUT_OIDC_ADMIN_GROUP` | — | Members are system administrators |
|
||||||
|
| `TERDUT_OIDC_SESSION_MAX_AGE` | `12h` | Hard ceiling on a session made by an SSO sign-in |
|
||||||
|
|
||||||
|
Durations use Go syntax (`30m`, `12h`, `168h`). An unparseable value falls back to the default.
|
||||||
|
|
||||||
|
Note that `TERDUT_STALE_AFTER` and a dead man's switch timeout point in opposite directions. Staleness
|
||||||
|
is a generous grace period around a `repeat_interval` you do not control; a dead man's switch is a
|
||||||
|
deadline you set deliberately, and the heartbeat's route is configured to beat faster than it.
|
||||||
|
|
||||||
|
In the Helm chart the two sweeper durations are set via `sweeper.staleAfter` and `sweeper.archiveAfter`, notifications via the `notify.*` values, single sign-on via `oidc.*` and `passwordLogin`, and operator mode via `operatorMode`.
|
||||||
@@ -0,0 +1,81 @@
|
|||||||
|
# Dead man's switches
|
||||||
|
|
||||||
|
_Detecting that alerts have stopped arriving._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
Everything above assumes alerts arrive. If Prometheus stops evaluating, or
|
||||||
|
Alertmanager cannot reach this server, nothing arrives — and silence looks
|
||||||
|
exactly like everything being fine. A dead man's switch inverts the handling for
|
||||||
|
one designated alert so that silence is the signal:
|
||||||
|
|
||||||
|
- **receiving** it opens no incident, and
|
||||||
|
- the **absence** of it does.
|
||||||
|
|
||||||
|
kube-prometheus-stack already ships the alert for this. `Watchdog` is
|
||||||
|
`expr: vector(1)`, so it fires permanently and is re-sent forever; it is worth
|
||||||
|
nothing unless something downstream notices it stop. It is the usual first switch.
|
||||||
|
|
||||||
|
**Switches belong to a team**, which decides which of its own alerts are
|
||||||
|
heartbeats and how long a silence has to last. Each **switch** is a row of its
|
||||||
|
own — a name, one matcher, a timeout and a severity — so switches in one team
|
||||||
|
can have different deadlines. An owner adds and removes them on **Team →
|
||||||
|
Switches**, which lists each with a status (**healthy**, **dead**, or
|
||||||
|
**dormant** until its first heartbeat), when it was last heard from, and when it
|
||||||
|
last opened an incident; a matcher that several clusters satisfy is broken down
|
||||||
|
per cluster. The API is `POST`/`DELETE /api/teams/{teamID}/deadman/switches`. A
|
||||||
|
missed heartbeat opens an incident in the team whose integration received it.
|
||||||
|
Removing a switch stops the watching; an incident it already opened stays open
|
||||||
|
until somebody resolves it.
|
||||||
|
|
||||||
|
A new team watches nothing until its owner (or terdut-operator, from a
|
||||||
|
`TerdutTeam`) adds a switch: inheriting an install-wide heartbeat would page a
|
||||||
|
new team about a source it has never heard of.
|
||||||
|
|
||||||
|
A matcher is a set of exact label conditions, one of which must be the
|
||||||
|
`alertname`, , one matcher per switch, `,` between the label conditions:
|
||||||
|
|
||||||
|
```
|
||||||
|
alertname=Watchdog,cluster=prod
|
||||||
|
```
|
||||||
|
|
||||||
|
**The unit of monitoring is the fingerprint, not the alert name.** Two clusters
|
||||||
|
sending the same `Watchdog` are two independent switches, so a healthy one can
|
||||||
|
never mask a dead one.
|
||||||
|
|
||||||
|
## The lifecycle
|
||||||
|
|
||||||
|
A switch is **dormant** until its first heartbeat arrives. A configured matcher
|
||||||
|
that has never been heard from opens nothing, so a fresh deploy or a restored
|
||||||
|
database does not page. It also means a matcher that never matches anything is
|
||||||
|
silently inert.
|
||||||
|
|
||||||
|
Once armed, the sweeper declares it **dead** when either the heartbeat has not
|
||||||
|
been refreshed within the switch's `timeout_seconds`, or Alertmanager explicitly
|
||||||
|
resolved it — the sender saying the heartbeat stopped needs no further waiting.
|
||||||
|
That opens an incident at the switch's `severity`, assigned and paged like any
|
||||||
|
other, and marks the heartbeat alert `"resolution_source": "deadman"` so the
|
||||||
|
alert list stops claiming a dead switch is firing.
|
||||||
|
|
||||||
|
It **recovers** when the heartbeat starts arriving again: the incident resolves
|
||||||
|
with `"resolution_source": "recovered"` and the all-clear goes to whoever was
|
||||||
|
paged.
|
||||||
|
|
||||||
|
Resolving the incident by hand sticks, the same way it does for an alert-backed
|
||||||
|
one. While the switch stays silent nothing new opens — so a decommissioned
|
||||||
|
source is a one-time page rather than a nag. The switch **re-arms** on the next
|
||||||
|
heartbeat: come back and die again, and that is a new incident.
|
||||||
|
|
||||||
|
## Two things to know
|
||||||
|
|
||||||
|
A switch's timeout must be **shorter** than the `repeat_interval` of the
|
||||||
|
route carrying the heartbeat, which is the exact opposite of
|
||||||
|
`TERDUT_STALE_AFTER`. Inheriting a default `repeat_interval` of 4h gives you a
|
||||||
|
switch that takes four hours to notice anything, so give the heartbeat
|
||||||
|
[its own route](./alertmanager.md#alertmanager-configuration). Matched alerts are exempt from
|
||||||
|
stale-alert expiry — a heartbeat answers to its own timeout and nothing else.
|
||||||
|
|
||||||
|
A dead man's switch incident has **no member alerts**:
|
||||||
|
`GET /api/incidents/{id}/alerts` returns an empty list. There is no alert
|
||||||
|
describing the problem, because the problem is that no alert arrived. What
|
||||||
|
happened is on the timeline instead, as a `deadman_silent` event carrying the age
|
||||||
|
of the last heartbeat, and the heartbeat's labels are on the incident's
|
||||||
|
`group_labels`.
|
||||||
@@ -0,0 +1,74 @@
|
|||||||
|
# Deployment
|
||||||
|
|
||||||
|
_Running the server in a container and on Kubernetes with the Helm chart._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
## Docker
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker build -t terdut-server .
|
||||||
|
docker run -p 8080:8080 \
|
||||||
|
-e TERDUT_DB_DSN='postgres://terdut:secret@host.docker.internal:5432/terdut?sslmode=disable' \
|
||||||
|
terdut-server
|
||||||
|
```
|
||||||
|
|
||||||
|
The server creates its own schema on startup and needs a reachable Postgres; it stores nothing on
|
||||||
|
disk, so there is no volume to mount.
|
||||||
|
|
||||||
|
## Kubernetes
|
||||||
|
|
||||||
|
A Helm chart is published from this repository as an OCI artifact, versioned in lockstep
|
||||||
|
with the app — chart `x.y.z` is always app `vx.y.z`:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
helm upgrade --install terdut-server oci://git.ryuvia.com/niklas/terdut-server \
|
||||||
|
--version 0.9.2 \
|
||||||
|
--namespace terdut-server --create-namespace \
|
||||||
|
--set networking.hostname=terdut.example.com
|
||||||
|
```
|
||||||
|
|
||||||
|
The chart expects a [Gateway API](https://gateway-api.sigs.k8s.io/) Gateway named `envoy-main` in
|
||||||
|
the `envoy-gateway-system` namespace to already exist — it renders an `HTTPRoute` against it rather
|
||||||
|
than an `Ingress`. TLS is terminated at the gateway, so the server itself never sees a certificate.
|
||||||
|
|
||||||
|
| Value | Default | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `networking.hostname` | `terdut.example.com` | Hostname the `HTTPRoute` serves |
|
||||||
|
| `networking.listener` | `""` | Gateway listener (`sectionName`) to bind to. Empty attaches to every matching listener, **including plaintext HTTP** — set it to the HTTPS listener's name to serve TLS only |
|
||||||
|
| `networking.servicePort` | `8080` | Port the route forwards to; keep in sync with `service.port` |
|
||||||
|
| `bootstrap.enabled` | `true` | Runs a post-install hook that creates the first user and stores its API key in the `<release>-admin-key` Secret. Already-bootstrapped servers are left alone |
|
||||||
|
| `database.dsn` | `""` | **Required.** Postgres DSN, with no password in it. The chart provisions no database |
|
||||||
|
| `database.passwordSecret.name` | `""` | Secret supplying `PGPASSWORD`. With the Zalando postgres operator, the Secret it generates for the role |
|
||||||
|
| `database.passwordSecret.key` | `password` | Key within that Secret |
|
||||||
|
|
||||||
|
The API key travels in an `Authorization: Bearer` header, so set `networking.listener` whenever the
|
||||||
|
hostname is reachable outside a trusted network.
|
||||||
|
|
||||||
|
### The database
|
||||||
|
|
||||||
|
The chart provisions no database: it takes a DSN and expects a Postgres that already exists. In this
|
||||||
|
cluster the wrapper chart declares an `acid.zalan.do/v1 postgresql` CR; anywhere else, any reachable
|
||||||
|
Postgres 14+ will do.
|
||||||
|
|
||||||
|
The DSN carries no password. pgx falls back to libpq's environment variables for whatever the DSN
|
||||||
|
leaves out, so the password arrives as `PGPASSWORD` from a Secret and never appears in values, in
|
||||||
|
the rendered manifest or in `kubectl describe pod`. With the postgres operator that Secret is the
|
||||||
|
one it generates for the role, so a rebuild mints a new password with nothing to keep in sync —
|
||||||
|
the same wiring miniflux uses.
|
||||||
|
|
||||||
|
The server migrates its own schema on startup, so a new database only has to exist and be writable.
|
||||||
|
|
||||||
|
### Backups
|
||||||
|
|
||||||
|
Postgres is backed up where it runs, not from here. The database pod carries a
|
||||||
|
[k8up](https://k8up.io/) `k8up.io/backupcommand` annotation that streams a `pg_dump`, the same way
|
||||||
|
gitea and immich do in this cluster.
|
||||||
|
|
||||||
|
## On Kubernetes with the operator
|
||||||
|
|
||||||
|
[terdut-operator](https://git.ryuvia.com/niklas/terdut-operator) runs a server for you from a
|
||||||
|
`TerdutServer` object and manages its teams, escalation ladders, dead man's switches and alert
|
||||||
|
sources as Kubernetes objects. It hands the server a generated key through `TERDUT_OPERATOR_KEY`
|
||||||
|
(see [Configuration](./configuration.md) and [`SERVICE-ACCOUNTS.md`](../SERVICE-ACCOUNTS.md)), and
|
||||||
|
the server then treats configuration as operator-managed (`TERDUT_OPERATOR_MODE`), refusing edits
|
||||||
|
made by hand in the web UI. Use the Helm chart above for a plain install, the operator when you
|
||||||
|
want that configuration in gitops.
|
||||||
@@ -0,0 +1,93 @@
|
|||||||
|
# Development and releasing
|
||||||
|
|
||||||
|
_Building, testing and releasing the server._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
## Upgrading
|
||||||
|
|
||||||
|
The schema is a single baseline (`internal/db/migrations/001_schema.sql`) and no
|
||||||
|
release has shipped yet, so there is no upgrade path from earlier development
|
||||||
|
databases: start from an empty one. Changes after the first release arrive as
|
||||||
|
new numbered migrations.
|
||||||
|
|
||||||
|
## Development
|
||||||
|
|
||||||
|
```bash
|
||||||
|
make test-db # start a local Postgres for the tests (podman or docker)
|
||||||
|
make test # run all tests
|
||||||
|
go build ./... # compile all packages
|
||||||
|
go run ./cmd/terdut # run locally (needs TERDUT_DB_DSN)
|
||||||
|
```
|
||||||
|
|
||||||
|
The tests need a real Postgres, because the server does — there is no in-memory Postgres.
|
||||||
|
`TERDUT_TEST_DSN` says where it is, `make test-db` starts
|
||||||
|
one on port 5433 and prints the DSN, and `make test-db-stop` removes it. Each test gets its
|
||||||
|
own schema on that server, so tests cannot see each other's rows. An unset `TERDUT_TEST_DSN`
|
||||||
|
fails the suite rather than skipping it: a run that quietly tests nothing is worse than one
|
||||||
|
that does not run.
|
||||||
|
|
||||||
|
`make fmt lint test helm-lint` is the gate. It mirrors `.gitea/workflows/ci.yaml` step for
|
||||||
|
step, so a green run here means a green pipeline — with one deliberate exception: `make test`
|
||||||
|
adds `-race`, which CI does not. The sweeper, the notifier goroutine and the dead man's switch
|
||||||
|
sweep all run concurrently against the same database, and a race between them would surface as
|
||||||
|
a flaky incident in production rather than as a red build.
|
||||||
|
|
||||||
|
The web UI lives in `internal/web/static/` as plain HTML, CSS and ES modules,
|
||||||
|
embedded into the binary with `go:embed`. It has no build step and no npm, so
|
||||||
|
editing a file and restarting the server is the whole loop.
|
||||||
|
|
||||||
|
## Releasing
|
||||||
|
|
||||||
|
```
|
||||||
|
push or PR → ci.yaml gofmt, go vet, go test -race
|
||||||
|
govulncheck, gitleaks
|
||||||
|
helm lint + render
|
||||||
|
push tag vX.Y.Z → release.yaml the same gate, then publish:
|
||||||
|
git.ryuvia.com/niklas/terdut-server:vX.Y.Z
|
||||||
|
oci://git.ryuvia.com/niklas/terdut-server X.Y.Z
|
||||||
|
then trivy-scan the pushed image
|
||||||
|
PR to Ryuvia/charts → bump the wrapper chart to X.Y.Z; on merge
|
||||||
|
Flux reconciles and the release rolls out
|
||||||
|
```
|
||||||
|
|
||||||
|
Both artifacts go to the **personal** Gitea namespace rather than `ryuvia`, because Gitea
|
||||||
|
scopes package visibility to the owner with no per-package override — so `ryuvia/*` is private
|
||||||
|
because the org is. Publishing to `niklas` keeps them anonymously pullable, which is why no
|
||||||
|
pull secret is needed in the cluster. Same reasoning, and the same choice, as riksdata and
|
||||||
|
rd-web.
|
||||||
|
|
||||||
|
Saying **"Release"** runs all three rows: the `release` skill commits, pushes, tags, waits for
|
||||||
|
the pipeline, and opens the `Ryuvia/charts` PR, stopping before the merge. See
|
||||||
|
`~/.claude/skills/release/`, or `.release.conf` here for this repo's part of it.
|
||||||
|
|
||||||
|
The chart is published **only** from the tag, by the `chart` job. There used to be a second
|
||||||
|
publisher on every `charts/**` push to main, and the two raced for the same chart version with
|
||||||
|
different answers — chart 0.9.0 went out reading `appVersion: "latest"` that way. One
|
||||||
|
publisher, triggered by the tag (`766f439`). The cost is that a chart-only change has no
|
||||||
|
version of its own and rides the next app tag.
|
||||||
|
|
||||||
|
Both workflows are thin drivers over the Makefile: `ci.yaml` runs `make fmt lint test` and
|
||||||
|
`make helm-lint`, `release.yaml` adds `make binaries`, `make push`, `make helm-package` and
|
||||||
|
`make helm-push`. That is deliberate — it is what makes a green local gate and a green
|
||||||
|
pipeline the same code rather than two descriptions of it, and it is how riksdata and rd-web
|
||||||
|
have always worked.
|
||||||
|
|
||||||
|
`make push` builds and pushes in one step, unlike those two, because the image is
|
||||||
|
`linux/amd64,linux/arm64` and buildx cannot load a multi-platform result into the local image
|
||||||
|
store. `make build` stays single-platform and local-only. Both refuse `VERSION=dev`:
|
||||||
|
publishing is one command, so it is also one command to run by accident. Publishing happens
|
||||||
|
by pushing a tag.
|
||||||
|
|
||||||
|
Two things the release process needs to know about this repo:
|
||||||
|
|
||||||
|
- **The image scan runs after publishing**, like riksdata's and rd-web's: trivy cannot read
|
||||||
|
a locally built image on this runner, so it pulls the pushed one. A red `scan-image` means
|
||||||
|
do not bump the wrapper chart to that version — it cannot unpublish anything. The image is
|
||||||
|
`FROM scratch`, so trivy sees exactly one target, the Go binary and its module graph.
|
||||||
|
- **The wrapper chart's `values.yaml` has two `tag:` lines** — the app image and the python
|
||||||
|
backup sidecar — so `chart-bump` is given `--image` to say which one moves. Once the wrapper
|
||||||
|
chart drops the sidecar and declares a `postgresql` CR instead, there is one `tag:` line
|
||||||
|
again, and `--image` becomes belt and braces.
|
||||||
|
|
||||||
|
The wrapper chart must have **its own `version:` bumped in the same commit**. Flux reconciles
|
||||||
|
with `reconcileStrategy: ChartVersion`, so a chart whose version did not change produces no
|
||||||
|
new artifact and the change is never deployed — with no error anywhere.
|
||||||
@@ -0,0 +1,45 @@
|
|||||||
|
# Escalation
|
||||||
|
|
||||||
|
_Escalation ladders: who is paged next when nobody acknowledges._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
Without a ladder, an unacknowledged incident re-pages the same topic every
|
||||||
|
`notify_repeat` forever. That is a louder version of the same silence: if the
|
||||||
|
person on call is asleep, out of signal, or has left, nothing else happens.
|
||||||
|
|
||||||
|
A team can configure an ordered ladder instead. Each level has a timeout and a
|
||||||
|
set of targets, and a target is either a named person or **whoever the team's
|
||||||
|
rota says is on call today** — the target that keeps working when the rota
|
||||||
|
changes and nobody remembers to edit the policy.
|
||||||
|
|
||||||
|
```
|
||||||
|
level 1 5m oncall the rota gets first refusal
|
||||||
|
level 2 5m user:bob then a named second
|
||||||
|
then repeat_count more rounds
|
||||||
|
then the team's fallback topic, once
|
||||||
|
```
|
||||||
|
|
||||||
|
When a level's timeout passes with the incident still `triggered`, the next
|
||||||
|
level is paged. Off the end of the ladder the whole thing runs again
|
||||||
|
`repeat_count` times, and after that the team's `fallback_topic` is paged once
|
||||||
|
as the end of the line. The incident stays open throughout: running out of
|
||||||
|
people to wake is not the same as somebody answering.
|
||||||
|
|
||||||
|
**Acknowledging or resolving stops it**, which is the point — continuing to wake
|
||||||
|
people after somebody has said "I have this" is how a tool teaches people to
|
||||||
|
mute it. **Snoozing pauses it**: a deliberate "not now" holds the ladder where
|
||||||
|
it is, and it resumes when the snooze runs out.
|
||||||
|
|
||||||
|
Every step is on the incident's timeline with the level and the names it woke,
|
||||||
|
so somebody reading it afterwards can tell why their phone rang at 04:00. A
|
||||||
|
level whose targets are all unreachable — no ntfy topic, a disabled account, an
|
||||||
|
empty rota — is recorded as `nobody reachable` and the ladder moves on rather
|
||||||
|
than stalling on a rung that cannot ring.
|
||||||
|
|
||||||
|
**Reminders and escalation never both run.** A team with a ladder gets
|
||||||
|
escalation; a team without keeps the reminder behaviour exactly as it was. Two
|
||||||
|
pages for one silence is the surest way to get a tool muted.
|
||||||
|
|
||||||
|
The ladder's `fallback_topic` is per team, unlike `TERDUT_NTFY_FALLBACK_TOPIC`,
|
||||||
|
which is the install-wide topic used when an incident opens with nobody on call.
|
||||||
|
They answer different questions: one is "nobody was scheduled", the other is
|
||||||
|
"everybody scheduled has been tried".
|
||||||
|
After Width: | Height: | Size: 23 KiB |
|
After Width: | Height: | Size: 55 KiB |
|
After Width: | Height: | Size: 88 KiB |
|
After Width: | Height: | Size: 52 KiB |
|
After Width: | Height: | Size: 53 KiB |
|
After Width: | Height: | Size: 100 KiB |
|
After Width: | Height: | Size: 93 KiB |
|
After Width: | Height: | Size: 94 KiB |
|
After Width: | Height: | Size: 42 KiB |
|
After Width: | Height: | Size: 35 KiB |
|
After Width: | Height: | Size: 44 KiB |
@@ -0,0 +1,113 @@
|
|||||||
|
# Alerts and incidents
|
||||||
|
|
||||||
|
_How alerts become incidents and how incidents are worked, notified, escalated and expired._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
There are two objects, and the difference between them is the whole design.
|
||||||
|
|
||||||
|
**An alert is Alertmanager's record.** It has two states, `firing` and
|
||||||
|
`resolved`, one row per fingerprint, and no human ever writes to it. The API
|
||||||
|
exposes alerts read-only.
|
||||||
|
|
||||||
|
**An incident is the work item.** It goes `triggered → acknowledged → resolved`,
|
||||||
|
carries an assignee, a snooze, notes and a timeline, and is the only thing people
|
||||||
|
act on. Many alerts belong to one incident.
|
||||||
|
|
||||||
|
## Correlation uses Alertmanager's `groupKey`
|
||||||
|
|
||||||
|
Alertmanager has already grouped alerts according to the `group_by` routing tree
|
||||||
|
you configured, and it sends the resulting `groupKey` and `groupLabels` on every
|
||||||
|
webhook. Incidents adopt that answer rather than re-grouping alerts a second
|
||||||
|
time — if you want different correlation, change `group_by` in
|
||||||
|
`alertmanager.yml` and terdut follows.
|
||||||
|
|
||||||
|
At most one incident is open per `groupKey` at a time. Alerts firing in a group
|
||||||
|
that already has an open incident join it. The incident's `severity` is a
|
||||||
|
high-water mark — the highest `severity` label any of its alerts has carried — so
|
||||||
|
an incident that hit `critical` still reads as critical after the critical alert
|
||||||
|
clears.
|
||||||
|
|
||||||
|
## Several clusters, one team
|
||||||
|
|
||||||
|
A team with one Alertmanager per Kubernetes cluster, each posting to its own
|
||||||
|
source, needs two settings or the clusters run together.
|
||||||
|
|
||||||
|
1. Give every alert a `cluster` label at the source. In Prometheus that is
|
||||||
|
`externalLabels: {cluster: prod-eu}` (kube-prometheus-stack:
|
||||||
|
`prometheus.prometheusSpec.externalLabels`).
|
||||||
|
2. Add `cluster` to `group_by` in `alertmanager.yml`.
|
||||||
|
|
||||||
|
The second one is the one that matters. Incidents are matched on the team and
|
||||||
|
Alertmanager's `groupKey`, and the `groupKey` does not include external labels:
|
||||||
|
without `cluster` in `group_by`, the same alert in two clusters has the same
|
||||||
|
key and joins one incident. With it, each cluster gets its own, `cluster` is in
|
||||||
|
the incident's `group_labels`, and the web UI shows it as a coloured chip on the
|
||||||
|
queue, the incident and the alert list, instead of leaving it in the title.
|
||||||
|
An alert that is not grouped by `cluster` still shows the chip on the alert
|
||||||
|
list, which reads the label from the alert itself.
|
||||||
|
|
||||||
|
The queue has a cluster dropdown once there are two or more values to choose
|
||||||
|
between. It filters on the incident's `cluster` group label
|
||||||
|
(`GET /api/incidents?cluster=...`), so it only sees incidents grouped by it.
|
||||||
|
|
||||||
|
## An incident opens only on a new occurrence
|
||||||
|
|
||||||
|
An incident opens when an alert **transitions into firing**: a fingerprint that
|
||||||
|
was never seen, an alert with a newer `startsAt`, or a resolved alert that
|
||||||
|
started again. The unchanged firing notifications Alertmanager re-sends every
|
||||||
|
`repeat_interval` are none of those, and open nothing.
|
||||||
|
|
||||||
|
This is what makes closing an incident by hand mean something. Without the rule,
|
||||||
|
`POST /api/incidents/{id}/resolve` would be undone by the next re-send of an
|
||||||
|
alert that never stopped firing.
|
||||||
|
|
||||||
|
## Leaving the open state
|
||||||
|
|
||||||
|
- **Automatically**, once every alert under the incident has stopped firing —
|
||||||
|
whether by a resolved webhook or by the sweeper's
|
||||||
|
[stale-alert expiry](#stale-alert-expiry). The incident gets
|
||||||
|
`"resolution_source": "alerts"`.
|
||||||
|
- **By hand**, via `POST /api/incidents/{id}/resolve`
|
||||||
|
(`"resolution_source": "manual"`). This is **terminal**: a later occurrence in
|
||||||
|
that group opens a *new* incident rather than reopening this one. If the alert
|
||||||
|
underneath never stops firing, the incident stays closed — that is what
|
||||||
|
resolving by hand asserts.
|
||||||
|
- **On recovery**, for a [dead man's switch](./dead-mans-switch.md) incident whose
|
||||||
|
heartbeat started arriving again (`"resolution_source": "recovered"`). These
|
||||||
|
incidents have no member alerts, so the automatic cascade above cannot reach
|
||||||
|
them.
|
||||||
|
|
||||||
|
To quieten an incident you expect to come back, snooze it instead
|
||||||
|
(`POST /api/incidents/{id}/snooze`). A snooze hides the incident from the default
|
||||||
|
list without closing it, and expires by simply falling into the past.
|
||||||
|
|
||||||
|
## On-call assignment
|
||||||
|
|
||||||
|
A new incident is assigned to whoever holds today's schedule entry at the moment
|
||||||
|
it opens (`GET /api/schedule/current`). If nobody is scheduled it opens
|
||||||
|
unassigned. Reassign with `POST /api/incidents/{id}/assign`.
|
||||||
|
|
||||||
|
One person holds a given day, so `POST /api/schedule` refuses a date somebody
|
||||||
|
already has: taking a shift off the person expecting to be paged for it should
|
||||||
|
not be something a plain call does by accident. Pass `"replace": true` to take
|
||||||
|
them anyway. Either way the whole request is one transaction — a week where some
|
||||||
|
days are free and some are taken moves as a unit, and a failure leaves the rota
|
||||||
|
exactly as it was rather than with a hole in it.
|
||||||
|
|
||||||
|
## Stale alert expiry
|
||||||
|
|
||||||
|
A resolved webhook is the only signal that an alert has stopped firing, so a
|
||||||
|
notification that is dropped, silenced, or lost to a restart would otherwise pin
|
||||||
|
that alert as firing forever. A background sweeper resolves firing alerts that
|
||||||
|
Alertmanager has stopped refreshing, using either signal:
|
||||||
|
|
||||||
|
- the `endsAt` watermark on the last notification has passed, or
|
||||||
|
- no webhook has refreshed the alert within `TERDUT_STALE_AFTER`.
|
||||||
|
|
||||||
|
Alertmanager re-sends firing notifications every `repeat_interval`, which is what
|
||||||
|
keeps a live alert fresh — so `TERDUT_STALE_AFTER` must be comfortably larger
|
||||||
|
than your `repeat_interval` (default 4h), or live alerts will be resolved
|
||||||
|
prematurely. Alerts resolved this way are marked `"resolution_source": "expiry"`
|
||||||
|
to distinguish them from a real Alertmanager resolve (`"alertmanager"`).
|
||||||
|
|
||||||
|
An expiry cascades: once it leaves an incident with nothing firing under it, the
|
||||||
|
incident resolves too, in the same sweep.
|
||||||
@@ -0,0 +1,62 @@
|
|||||||
|
# Push notifications
|
||||||
|
|
||||||
|
_Pages through ntfy, who gets them and how to acknowledge from the notification._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
With `TERDUT_NTFY_URL` set, an incident that opens is pushed to the on-call
|
||||||
|
person's phone through [ntfy](https://ntfy.sh). Everybody sets their own topic
|
||||||
|
under *Account* in the web UI, where a **Send a test push** button proves it
|
||||||
|
before an incident has to; `PUT /api/users/{id}/notify` is the same thing over
|
||||||
|
the API, and an administrator may set somebody else's. A user with no topic
|
||||||
|
falls back to `TERDUT_NTFY_FALLBACK_TOPIC`, as does an incident that opens with
|
||||||
|
nobody on call. If neither yields a topic, nothing is queued.
|
||||||
|
|
||||||
|
The **server** is the install's one ntfy, from `TERDUT_NTFY_URL`, and is not
|
||||||
|
something a user picks. Only the topic is per-person.
|
||||||
|
|
||||||
|
A topic is a shared secret with the ntfy server: anyone who knows it can both
|
||||||
|
read the pages and publish to it, so an unguessable one is worth the trouble.
|
||||||
|
That is also why the topic never appears in an incident's timeline, which every
|
||||||
|
API key can read.
|
||||||
|
|
||||||
|
Three things get pushed:
|
||||||
|
|
||||||
|
- **triggered** — an incident opened. Priority follows severity (`critical` maps
|
||||||
|
to ntfy's max priority, the one that overrides the phone's quiet settings).
|
||||||
|
- **reminder** — the incident is still `triggered` after `TERDUT_NOTIFY_REPEAT`.
|
||||||
|
Repeats until somebody acts. Acknowledging, snoozing, resolving or archiving
|
||||||
|
all stop it — snooze is the mute button.
|
||||||
|
- **resolved** — every alert under the incident stopped firing. Only sent to
|
||||||
|
whoever was paged in the first place, and only for the automatic cascade:
|
||||||
|
resolving by hand pushes nothing, since the person who did it already knows.
|
||||||
|
|
||||||
|
Notifications carry an **Acknowledge** button that acknowledges the incident
|
||||||
|
without opening anything. It POSTs to `/api/notify/ack/{token}`, an
|
||||||
|
unauthenticated route authorised by the 256-bit token in its path — minted fresh
|
||||||
|
per notification, scoped to one incident and one action, and valid for 24 hours.
|
||||||
|
A real API key is never put in a notification, because the message is stored on
|
||||||
|
the ntfy server and cached on the device.
|
||||||
|
|
||||||
|
The token is **not** consumed by use. Acknowledging is idempotent, so a token
|
||||||
|
stays valid for its full 24 hours and a second tap is a no-op that reports the
|
||||||
|
incident's current state rather than an error — which is what you want when a
|
||||||
|
tap is retried on a flaky mobile connection. What bounds it is scope, not a use
|
||||||
|
count: one incident, one action, one day. Expired tokens are purged by the
|
||||||
|
sweeper.
|
||||||
|
|
||||||
|
Two consequences worth planning for:
|
||||||
|
|
||||||
|
- `/api/notify/ack/{token}` **must stay publicly reachable**, or the button will
|
||||||
|
not work when the responder is off your network.
|
||||||
|
- Notifications sent to the fallback topic carry **no** Acknowledge button. The
|
||||||
|
topic is shared, and a button on it would let any subscriber acknowledge as
|
||||||
|
somebody else.
|
||||||
|
|
||||||
|
Delivery is a queue, not an inline call: the webhook writes a row and a
|
||||||
|
background notifier sends it within 30 seconds, retrying with exponential
|
||||||
|
backoff up to 8 attempts. Nothing about ingestion blocks on ntfy being reachable.
|
||||||
|
|
||||||
|
Every delivery is recorded on the incident's timeline: a `notified` event once
|
||||||
|
ntfy accepts the publish, and a `notify_failed` event when a notification
|
||||||
|
exhausts its retries. Written from the result rather than at enqueue, so the
|
||||||
|
timeline says what actually happened — and a page that never landed is visible
|
||||||
|
instead of looking the same as one that did.
|
||||||
@@ -0,0 +1,105 @@
|
|||||||
|
# Single sign-on (OIDC)
|
||||||
|
|
||||||
|
_Signing in through an OpenID Connect provider, and mapping its groups to teams and administrators._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
terdut can sign people in through any OpenID Connect provider; the examples use
|
||||||
|
[Authentik](https://goauthentik.io/). Groups at the provider decide who may sign
|
||||||
|
in, which teams they belong to and whether they administer the install, much as
|
||||||
|
Grafana's OAuth role and org mapping does. Password login keeps working alongside
|
||||||
|
it unless you turn it off.
|
||||||
|
|
||||||
|
**At the provider**, create an OAuth2/OpenID provider and an application for it:
|
||||||
|
a *confidential* client, redirect URI `<TERDUT_PUBLIC_URL>/api/oidc/callback`, and
|
||||||
|
the `openid`, `profile` and `email` scopes. The issuer is the application's, e.g.
|
||||||
|
`https://auth.example.com/application/o/terdut/`. Then set:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
TERDUT_PUBLIC_URL=https://terdut.example.com
|
||||||
|
TERDUT_OIDC_ISSUER=https://auth.example.com/application/o/terdut/
|
||||||
|
TERDUT_OIDC_CLIENT_ID=terdut
|
||||||
|
TERDUT_OIDC_CLIENT_SECRET=...
|
||||||
|
TERDUT_OIDC_ALLOWED_GROUPS=terdut-users,terdut-admins
|
||||||
|
TERDUT_OIDC_ADMIN_GROUP=terdut-admins
|
||||||
|
```
|
||||||
|
|
||||||
|
Which team a group grants is not server-wide config: each team names its own
|
||||||
|
group(s), set by that team's own owner (or an administrator) from its Members
|
||||||
|
tab, or `PUT /api/teams/{teamID}/oidc-groups {"member_group":"sre","owner_group":"sre-leads"}`.
|
||||||
|
A team must already exist before a group can grant access to it — the sync
|
||||||
|
never creates one.
|
||||||
|
|
||||||
|
The web UI's sign-in page shows a "Sign in with <name>" button (a plain link to
|
||||||
|
`/api/oidc/login`) above the password form, or instead of it when
|
||||||
|
`TERDUT_PASSWORD_LOGIN=false`; it asks `GET /api/auth/config` what the server offers
|
||||||
|
(`password_login`, `oidc.enabled`, `oidc.name`). A refused sign-in comes back to that
|
||||||
|
page with the reason spelled out. Access the groups grant is badged **SSO** on the
|
||||||
|
Team, Admin and per-user pages, with its edit and remove controls disabled, and the
|
||||||
|
Account page does not offer to set a password nobody could use.
|
||||||
|
|
||||||
|
**What a sign-in does**
|
||||||
|
|
||||||
|
1. *Who.* The provider's `(issuer, subject)` is the identity. The first time, a
|
||||||
|
user is found by email — only when the provider marks it verified, or
|
||||||
|
`TERDUT_OIDC_TRUST_EMAIL` is set — or created with no password. A username taken
|
||||||
|
by somebody else gets a numeric suffix (`alice-2`). Username and email follow the
|
||||||
|
provider at each sign-in. Authentik reports `email_verified` as false unless
|
||||||
|
configured otherwise, so linking existing users usually needs
|
||||||
|
`TERDUT_OIDC_TRUST_EMAIL=true`.
|
||||||
|
2. *Whether.* With `TERDUT_OIDC_ALLOWED_GROUPS` set, somebody in none of them is
|
||||||
|
refused and nothing is created.
|
||||||
|
3. *What.* The administrator flag follows `TERDUT_OIDC_ADMIN_GROUP`. Team roles
|
||||||
|
follow each team's own `oidc_member_group`/`oidc_owner_group`; where both of a
|
||||||
|
team's groups match, the owner group wins.
|
||||||
|
|
||||||
|
**Managed access.** What the sync grants is marked as managed by single sign-on,
|
||||||
|
and only that is ever changed by it. It is added at sign-in, and removed at the
|
||||||
|
next sign-in after the group is gone, even if that leaves a team without an owner
|
||||||
|
(an administrator can always repair a team) — the provider is the source of truth
|
||||||
|
for what it grants, so the last-owner and last-administrator guards do not apply.
|
||||||
|
Memberships and administrators added by hand are left alone; the exception is a
|
||||||
|
hand-added member whose team's own group grants a *higher* role, who is raised and
|
||||||
|
from then on managed. Editing managed access by hand (`POST` or `DELETE` on a
|
||||||
|
team's members, revoking an SSO-granted administrator) is refused with `409`, since
|
||||||
|
the next sign-in would undo it.
|
||||||
|
|
||||||
|
> **Upgrading past migration 013: reconfigure every team's groups.**
|
||||||
|
> `TERDUT_OIDC_GROUP_MAPPINGS` is gone, and the sync no longer creates a team by
|
||||||
|
> name. Group-to-team-role mapping is now each team's own setting — an owner sets
|
||||||
|
> it from the Members tab, or `PUT /api/teams/{teamID}/oidc-groups`. Until a team's
|
||||||
|
> owner does that, an OIDC-sourced membership in it is dropped at that user's next
|
||||||
|
> SSO sign-in, the same as any other loss of group access. Set every team's groups
|
||||||
|
> before affected users next sign in, to avoid a visible gap in access.
|
||||||
|
|
||||||
|
**How fast changes arrive.** Groups are read only at sign-in. A session made by an
|
||||||
|
SSO sign-in has a hard ceiling (`TERDUT_OIDC_SESSION_MAX_AGE`, default 12h) that
|
||||||
|
sliding never extends, so a change at the provider reaches terdut within that time.
|
||||||
|
Password sessions are unaffected.
|
||||||
|
|
||||||
|
> **API keys are not revoked when somebody is removed at the provider.** terdut
|
||||||
|
> holds no refresh token and never asks the provider again, so a person removed
|
||||||
|
> from every allowed group loses their sessions within `TERDUT_OIDC_SESSION_MAX_AGE`
|
||||||
|
> and cannot sign in again, but keeps any API key they made (the TUI and scripts use
|
||||||
|
> them) until an administrator disables the user in terdut.
|
||||||
|
|
||||||
|
**Signing in from a terminal.** A client with no browser of its own, such as the
|
||||||
|
TUI over SSH, signs in with a device code, run by terdut itself so the terminal
|
||||||
|
never talks to the provider:
|
||||||
|
|
||||||
|
1. The terminal calls `POST /api/oidc/device` and shows the person a link
|
||||||
|
(`<TERDUT_PUBLIC_URL>/device?code=XXXX-XXXX`) and the code.
|
||||||
|
2. On any device the person opens the link, signs in (by the provider or by
|
||||||
|
password, whatever the login page offers), sees the code and the account, and
|
||||||
|
presses **Approve**. Only a browser session can approve; an API key cannot.
|
||||||
|
3. The terminal polls `POST /api/oidc/device/token` every 5 seconds and is given the
|
||||||
|
ordinary `terdut_session` cookie once. A person who signs in through the provider
|
||||||
|
gets the same `TERDUT_OIDC_SESSION_MAX_AGE` ceiling on the terminal's session as
|
||||||
|
on their browser's.
|
||||||
|
|
||||||
|
A login expires after 10 minutes. `GET /api/auth/config` reports `device_login`.
|
||||||
|
|
||||||
|
**If the provider is down**, terdut still starts (discovery is fetched on first
|
||||||
|
use) and password login is the way in. With `TERDUT_PASSWORD_LOGIN=false` that way
|
||||||
|
is closed: set it back to `true`. The first administrator comes from the bootstrap
|
||||||
|
endpoint, and stays a manual administrator that no group can revoke; on an SSO-only
|
||||||
|
install set `bootstrap.enabled: false` in the chart if you don't want that account,
|
||||||
|
or keep it and never give it a password.
|
||||||
@@ -0,0 +1,90 @@
|
|||||||
|
# The web UI
|
||||||
|
|
||||||
|
_What the web UI offers, how sign-in and sessions work, and the Team and Admin tabs._ Back to the [README](../README.md) and the [documentation index](./README.md).
|
||||||
|
|
||||||
|
The server serves a web UI at `/`: the incident queue, each incident's alerts
|
||||||
|
and timeline with every action (acknowledge, assign, snooze, note, resolve,
|
||||||
|
archive), who is on call, the alert feed, and an *Account* tab for your own
|
||||||
|
password and the ntfy topic your pages go to. It is built for a phone first. On a phone
|
||||||
|
it navigates through a hamburger menu and has a sticky action bar, it follows the
|
||||||
|
system's dark mode, and it can be added to the home screen. From 900px wide it switches
|
||||||
|
to a sidebar with the queue and the incident side by side. The Stats page shows
|
||||||
|
incident counts, MTTA and MTTR, and alert frequency by name, hour and day over a
|
||||||
|
chosen range.
|
||||||
|
|
||||||
|
You sign in with a username and password. Users have no password until one is
|
||||||
|
set, and a user without one can only use API keys:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# an admin sets someone's first password with their API key
|
||||||
|
curl -X PUT http://localhost:8080/api/users/2/password \
|
||||||
|
-H "Authorization: Bearer $KEY" -H "Content-Type: application/json" \
|
||||||
|
-d '{"password": "<at least 10 characters>"}'
|
||||||
|
```
|
||||||
|
|
||||||
|
After that, users change it themselves under *Account*. Changing your own
|
||||||
|
password requires the current one.
|
||||||
|
|
||||||
|
How a browser stays signed in:
|
||||||
|
|
||||||
|
- A successful login sets an `HttpOnly`, `SameSite=Lax` session cookie. It lasts
|
||||||
|
30 days and slides forward while it is used, so an on-call phone stays signed
|
||||||
|
in.
|
||||||
|
- The cookie is marked `Secure` when `TERDUT_PUBLIC_URL` starts with `https://`,
|
||||||
|
so set it to the HTTPS address. TLS terminates at the gateway and the server
|
||||||
|
itself only ever sees plain HTTP.
|
||||||
|
- Requests authenticated by the cookie are checked for cross-origin use (Go's
|
||||||
|
`http.CrossOriginProtection`). That is the CSRF guard. Bearer-key clients are
|
||||||
|
not affected.
|
||||||
|
- Setting a password signs that user out everywhere else.
|
||||||
|
- Ten failed logins for one username within 15 minutes lock that username for
|
||||||
|
the rest of the window.
|
||||||
|
|
||||||
|
With `TERDUT_PUBLIC_URL` set, tapping a push notification opens the incident in
|
||||||
|
the web UI (`/incidents/{id}`).
|
||||||
|
|
||||||
|
A **Team** tab holds everything a team owns, in five sub-sections with a URL
|
||||||
|
each and a strip across the top to move between them: the on-call rota
|
||||||
|
(`/team/rota`), the membership (`/team/members`), the escalation ladder
|
||||||
|
(`/team/escalation`), the alert sources with their keys (`/team/sources`) and
|
||||||
|
the dead man's switches (`/team/deadman`). `/team` itself is an overview — who
|
||||||
|
is on call today, how many members and owners, how many ladder levels, how many
|
||||||
|
keys and how many switches — so a page fetches only what it shows. An owner
|
||||||
|
edits it; a member sees the same pages read-only, because the server refuses
|
||||||
|
their writes anyway. Somebody in more than one team picks between them above
|
||||||
|
the strip, since the choice changes the subject of all five.
|
||||||
|
|
||||||
|
The rota is a month at a time, one coloured initial per day with a legend
|
||||||
|
underneath, and it says how many days are left uncovered — the question a rota
|
||||||
|
is read for is who holds which stretch, and a run of one colour answers it
|
||||||
|
where a list of dates does not. An owner taps a day to hand it to somebody or
|
||||||
|
empty it, and fills a whole shift from the range form folded in below.
|
||||||
|
|
||||||
|
The **Admin** tab appears only for a system administrator, and holds what
|
||||||
|
belongs to the whole server rather than to one team. It has three sub-sections,
|
||||||
|
each with a URL of its own and a strip across the top to move between them:
|
||||||
|
every team (`/admin/teams`), every user (`/admin/users`), and the settings that
|
||||||
|
used to be environment variables (`/admin/settings`). `/admin` itself is an
|
||||||
|
overview — how many of each, and what each section is for. Adding somebody is
|
||||||
|
minting them an invite link into a team, rather than creating a bare account:
|
||||||
|
the person who accepts it picks their own password, so one never passes through
|
||||||
|
an administrator, and the link carries the team, so they land somewhere with a
|
||||||
|
queue in it. That happens on the team's own page, since an invite is a fact
|
||||||
|
about a team; the user list points there rather than asking which team beside a
|
||||||
|
form.
|
||||||
|
|
||||||
|
A name in the team list opens **that team's page**, at `/admin/teams/{id}`: when it
|
||||||
|
was created, how many are in it and how much is open, a field to rename it, the
|
||||||
|
members with their roles, the invites into it, and deletion. The member list is the
|
||||||
|
one thing there that needed a new endpoint — `GET /api/teams/{id}/members` is
|
||||||
|
member-only and answers `404` to an administrator who is not in the team, which is
|
||||||
|
the rule and not an oversight, so the page reads `GET /api/admin/teams/{id}` instead.
|
||||||
|
An administrator still sees none of that team's incidents, alerts or rota.
|
||||||
|
|
||||||
|
A name in the user list opens **that person's page**, at `/admin/users/{id}`: their
|
||||||
|
email and when they joined, where their notifications go, whether they are an
|
||||||
|
administrator, whether the account is disabled, the teams they are in with their
|
||||||
|
role in each, a password field for a first or forgotten one, and deletion. It is
|
||||||
|
the one place membership is edited from the person's side — the Team tab answers
|
||||||
|
"who is in this team", and answering "which teams is this person in" there means
|
||||||
|
visiting each team in turn.
|
||||||
@@ -3,13 +3,16 @@ module git.ryuvia.com/niklas/terdut-server
|
|||||||
go 1.25.9
|
go 1.25.9
|
||||||
|
|
||||||
require (
|
require (
|
||||||
|
github.com/coreos/go-oidc/v3 v3.21.0
|
||||||
github.com/go-chi/chi/v5 v5.2.5
|
github.com/go-chi/chi/v5 v5.2.5
|
||||||
github.com/jackc/pgerrcode v0.0.0-20250907135507-afb5586c32a6
|
github.com/jackc/pgerrcode v0.0.0-20250907135507-afb5586c32a6
|
||||||
github.com/jackc/pgx/v5 v5.11.0
|
github.com/jackc/pgx/v5 v5.11.0
|
||||||
golang.org/x/crypto v0.55.0
|
golang.org/x/crypto v0.55.0
|
||||||
|
golang.org/x/oauth2 v0.36.0
|
||||||
)
|
)
|
||||||
|
|
||||||
require (
|
require (
|
||||||
|
github.com/go-jose/go-jose/v4 v4.1.4 // indirect
|
||||||
github.com/jackc/pgpassfile v1.0.0 // indirect
|
github.com/jackc/pgpassfile v1.0.0 // indirect
|
||||||
github.com/jackc/pgservicefile v0.0.0-20240606120523-5a60cdf6a761 // indirect
|
github.com/jackc/pgservicefile v0.0.0-20240606120523-5a60cdf6a761 // indirect
|
||||||
github.com/jackc/puddle/v2 v2.2.2 // indirect
|
github.com/jackc/puddle/v2 v2.2.2 // indirect
|
||||||
|
|||||||
@@ -1,8 +1,12 @@
|
|||||||
|
github.com/coreos/go-oidc/v3 v3.21.0 h1:wZo4Q9Pum8dYEj0eMUPrqR+kvuGkeUplbLpNCkBqoWM=
|
||||||
|
github.com/coreos/go-oidc/v3 v3.21.0/go.mod h1:DYCf24+ncYi+XkIH97GY1+dqoRlbaSI26KVTCI9SrY4=
|
||||||
github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||||
github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c=
|
github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c=
|
||||||
github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38=
|
||||||
github.com/go-chi/chi/v5 v5.2.5 h1:Eg4myHZBjyvJmAFjFvWgrqDTXFyOzjj7YIm3L3mu6Ug=
|
github.com/go-chi/chi/v5 v5.2.5 h1:Eg4myHZBjyvJmAFjFvWgrqDTXFyOzjj7YIm3L3mu6Ug=
|
||||||
github.com/go-chi/chi/v5 v5.2.5/go.mod h1:X7Gx4mteadT3eDOMTsXzmI4/rwUpOwBHLpAfupzFJP0=
|
github.com/go-chi/chi/v5 v5.2.5/go.mod h1:X7Gx4mteadT3eDOMTsXzmI4/rwUpOwBHLpAfupzFJP0=
|
||||||
|
github.com/go-jose/go-jose/v4 v4.1.4 h1:moDMcTHmvE6Groj34emNPLs/qtYXRVcd6S7NHbHz3kA=
|
||||||
|
github.com/go-jose/go-jose/v4 v4.1.4/go.mod h1:x4oUasVrzR7071A4TnHLGSPpNOm2a21K9Kf04k1rs08=
|
||||||
github.com/jackc/pgerrcode v0.0.0-20250907135507-afb5586c32a6 h1:D/V0gu4zQ3cL2WKeVNVM4r2gLxGGf6McLwgXzRTo2RQ=
|
github.com/jackc/pgerrcode v0.0.0-20250907135507-afb5586c32a6 h1:D/V0gu4zQ3cL2WKeVNVM4r2gLxGGf6McLwgXzRTo2RQ=
|
||||||
github.com/jackc/pgerrcode v0.0.0-20250907135507-afb5586c32a6/go.mod h1:a/s9Lp5W7n/DD0VrVoyJ00FbP2ytTPDVOivvn2bMlds=
|
github.com/jackc/pgerrcode v0.0.0-20250907135507-afb5586c32a6/go.mod h1:a/s9Lp5W7n/DD0VrVoyJ00FbP2ytTPDVOivvn2bMlds=
|
||||||
github.com/jackc/pgpassfile v1.0.0 h1:/6Hmqy13Ss2zCq62VdNG8tM1wchn8zjSGOBJ6icpsIM=
|
github.com/jackc/pgpassfile v1.0.0 h1:/6Hmqy13Ss2zCq62VdNG8tM1wchn8zjSGOBJ6icpsIM=
|
||||||
@@ -22,6 +26,8 @@ github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu
|
|||||||
github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U=
|
github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U=
|
||||||
golang.org/x/crypto v0.55.0 h1:+KWHjbgOaAQ66dh/YlkZKHlz9ZUlq61AFirAR9ntP8M=
|
golang.org/x/crypto v0.55.0 h1:+KWHjbgOaAQ66dh/YlkZKHlz9ZUlq61AFirAR9ntP8M=
|
||||||
golang.org/x/crypto v0.55.0/go.mod h1:uq0V9dE/fzQuJtbnL+2EhWOE63vo164FY8xqEnV9xis=
|
golang.org/x/crypto v0.55.0/go.mod h1:uq0V9dE/fzQuJtbnL+2EhWOE63vo164FY8xqEnV9xis=
|
||||||
|
golang.org/x/oauth2 v0.36.0 h1:peZ/1z27fi9hUOFCAZaHyrpWG5lwe0RJEEEeH0ThlIs=
|
||||||
|
golang.org/x/oauth2 v0.36.0/go.mod h1:YDBUJMTkDnJS+A4BP4eZBjCqtokkg1hODuPjwiGPO7Q=
|
||||||
golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek=
|
golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek=
|
||||||
golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
|
golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
|
||||||
golang.org/x/text v0.41.0 h1:vz/seA0lnX87Othu2f/0L24RcgrXD9/YFTSuGjj3rH8=
|
golang.org/x/text v0.41.0 h1:vz/seA0lnX87Othu2f/0L24RcgrXD9/YFTSuGjj3rH8=
|
||||||
|
|||||||
@@ -237,6 +237,246 @@ func TestAdmin_GrantAndRevokeChangeWhatIsAllowed(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// An administrator passes every team-owner check without being in the team,
|
||||||
|
// which is what lets them repair a team whose owner has left. It has been true
|
||||||
|
// since teams landed and nothing pinned it, so a later reading of the epic's
|
||||||
|
// "an admin is not implicitly in every team" could quietly take it away.
|
||||||
|
//
|
||||||
|
// The line it draws: configuring a team, yes; reading what the team owns, no.
|
||||||
|
// The queue below is the half that stays shut.
|
||||||
|
func TestAdmin_ConfiguresATeamTheyAreNotIn(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
|
||||||
|
// A team the admin is deliberately not a member of. It is created by
|
||||||
|
// somebody else, so the admin's only claim on it is the flag.
|
||||||
|
_, call := member(t, s, "founder")
|
||||||
|
var team struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
}
|
||||||
|
decode(t, call(http.MethodPost, "/api/teams", map[string]string{"name": "theirs"}), &team)
|
||||||
|
if team.ID == 0 {
|
||||||
|
t.Fatal("no team was created")
|
||||||
|
}
|
||||||
|
|
||||||
|
var mine []struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/teams", nil), &mine)
|
||||||
|
for _, m := range mine {
|
||||||
|
if m.ID == team.ID {
|
||||||
|
t.Fatalf("the admin should not be a member of team %d", team.ID)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
path := "/api/teams/" + id64(team.ID)
|
||||||
|
for _, c := range []struct {
|
||||||
|
name string
|
||||||
|
method string
|
||||||
|
path string
|
||||||
|
body any
|
||||||
|
want int
|
||||||
|
}{
|
||||||
|
{"rename it", http.MethodPut, path,
|
||||||
|
map[string]string{"name": "theirs, renamed"}, http.StatusNoContent},
|
||||||
|
{"mint an invite", http.MethodPost, path + "/invites",
|
||||||
|
map[string]any{"role": "member", "max_uses": 1}, http.StatusCreated},
|
||||||
|
{"add a member", http.MethodPost, path + "/members",
|
||||||
|
map[string]any{"user_id": 1, "role": "member"}, http.StatusNoContent},
|
||||||
|
{"remove a member", http.MethodDelete, path + "/members/1", nil, http.StatusNoContent},
|
||||||
|
} {
|
||||||
|
resp := s.req(t, c.method, c.path, c.body)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != c.want {
|
||||||
|
t.Errorf("%s: expected %d, got %d", c.name, c.want, resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The other half of the rule. An incident in that team is not the admin's
|
||||||
|
// to read, because administration is about accounts — and the last case
|
||||||
|
// above has just taken the admin back out of the membership.
|
||||||
|
var integration struct {
|
||||||
|
Key string `json:"key"`
|
||||||
|
}
|
||||||
|
decode(t, call(http.MethodPost, path+"/integrations",
|
||||||
|
map[string]string{"name": "theirs alertmanager"}), &integration)
|
||||||
|
postToIntegration(t, s, integration.Key, "fp-theirs", "TheirDiskFull")
|
||||||
|
|
||||||
|
var incidents []struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/incidents", nil), &incidents)
|
||||||
|
if len(incidents) != 0 {
|
||||||
|
t.Errorf("the admin should see none of that team's incidents, got %d", len(incidents))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The team page at /admin/teams/{id} needs the one question the test above
|
||||||
|
// leaves shut: who is in a team the administrator is not in.
|
||||||
|
//
|
||||||
|
// It is answered by a separate endpoint under AdminOnly rather than by letting
|
||||||
|
// the admin flag through requireTeamMember, and the second half of this test is
|
||||||
|
// the reason — /api/teams/{id}/members must keep answering 404, so that "member
|
||||||
|
// means membership and nothing else" stays true of the endpoint it was said
|
||||||
|
// about. Reading a team's shape and reading a team's work are different things.
|
||||||
|
func TestAdminGetTeam_ReadsAnyTeamWithoutJoiningIt(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
|
||||||
|
founderID, call := member(t, s, "founder")
|
||||||
|
var team struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
}
|
||||||
|
decode(t, call(http.MethodPost, "/api/teams", map[string]string{"name": "theirs"}), &team)
|
||||||
|
if team.ID == 0 {
|
||||||
|
t.Fatal("no team was created")
|
||||||
|
}
|
||||||
|
|
||||||
|
// The admin reads it whole, without being in it.
|
||||||
|
var got struct {
|
||||||
|
Team struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
Name string `json:"name"`
|
||||||
|
Members int64 `json:"members"`
|
||||||
|
OpenIncidents int64 `json:"open_incidents"`
|
||||||
|
} `json:"team"`
|
||||||
|
Members []struct {
|
||||||
|
UserID int64 `json:"user_id"`
|
||||||
|
Username string `json:"username"`
|
||||||
|
Role string `json:"role"`
|
||||||
|
} `json:"members"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/admin/teams/"+id64(team.ID), nil), &got)
|
||||||
|
|
||||||
|
if got.Team.ID != team.ID || got.Team.Name != "theirs" {
|
||||||
|
t.Errorf("expected team %d named theirs, got %d named %q", team.ID, got.Team.ID, got.Team.Name)
|
||||||
|
}
|
||||||
|
if got.Team.Members != 1 {
|
||||||
|
t.Errorf("expected a member count of 1, got %d", got.Team.Members)
|
||||||
|
}
|
||||||
|
if len(got.Members) != 1 {
|
||||||
|
t.Fatalf("expected one member, got %d", len(got.Members))
|
||||||
|
}
|
||||||
|
if got.Members[0].UserID != founderID || got.Members[0].Username != "founder" {
|
||||||
|
t.Errorf("expected founder (%d), got %q (%d)",
|
||||||
|
founderID, got.Members[0].Username, got.Members[0].UserID)
|
||||||
|
}
|
||||||
|
// Whoever creates a team owns it, and the page's role toggle depends on
|
||||||
|
// that being reported rather than assumed.
|
||||||
|
if got.Members[0].Role != "owner" {
|
||||||
|
t.Errorf("expected the creator to be owner, got %q", got.Members[0].Role)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The rule this endpoint exists in order not to break. Same admin, same
|
||||||
|
// team, the member-only endpoint: still not found.
|
||||||
|
resp := s.req(t, http.MethodGet, "/api/teams/"+id64(team.ID)+"/members", nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNotFound {
|
||||||
|
t.Errorf("an admin outside the team must still get 404 from the member-only list, got %d",
|
||||||
|
resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// And the new one is administration, not membership: being in the team is
|
||||||
|
// not enough.
|
||||||
|
resp = call(http.MethodGet, "/api/admin/teams/"+id64(team.ID), nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("a non-admin member must get 403, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, c := range []struct {
|
||||||
|
name string
|
||||||
|
path string
|
||||||
|
want int
|
||||||
|
}{
|
||||||
|
{"a team that does not exist", "/api/admin/teams/999999", http.StatusNotFound},
|
||||||
|
{"a team id that is not a number", "/api/admin/teams/nonsense", http.StatusBadRequest},
|
||||||
|
} {
|
||||||
|
resp := s.req(t, http.MethodGet, c.path, nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != c.want {
|
||||||
|
t.Errorf("%s: expected %d, got %d", c.name, c.want, resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A team name is trimmed when it is created, and renaming had not been, so " "
|
||||||
|
// was a legal name to rename to and an illegal one to start with.
|
||||||
|
func TestRenameTeam_TrimsTheName(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
var team struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodPost, "/api/teams", map[string]string{"name": "trimmed"}), &team)
|
||||||
|
|
||||||
|
path := "/api/teams/" + id64(team.ID)
|
||||||
|
resp := s.req(t, http.MethodPut, path, map[string]string{"name": " "})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusBadRequest {
|
||||||
|
t.Errorf("a blank name must be refused, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
resp = s.req(t, http.MethodPut, path, map[string]string{"name": " padded "})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNoContent {
|
||||||
|
t.Fatalf("expected 204, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
var got struct {
|
||||||
|
Team struct {
|
||||||
|
Name string `json:"name"`
|
||||||
|
} `json:"team"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/admin/teams/"+id64(team.ID), nil), &got)
|
||||||
|
if got.Team.Name != "padded" {
|
||||||
|
t.Errorf("expected the name to be trimmed to %q, got %q", "padded", got.Team.Name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The admin page's per-user view asks what somebody is in. Self or admin, like
|
||||||
|
// the rest of the per-user endpoints.
|
||||||
|
func TestUserTeams_SelfOrAdmin(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
memberID, call := member(t, s, "joiner")
|
||||||
|
path := "/api/users/" + id64(memberID) + "/teams"
|
||||||
|
|
||||||
|
// member() puts them in the default team, so both readings agree on one.
|
||||||
|
for _, c := range []struct {
|
||||||
|
name string
|
||||||
|
do func() *http.Response
|
||||||
|
}{
|
||||||
|
{"the admin reading somebody else's", func() *http.Response { return s.req(t, http.MethodGet, path, nil) }},
|
||||||
|
{"the user reading their own", func() *http.Response { return call(http.MethodGet, path, nil) }},
|
||||||
|
} {
|
||||||
|
var teams []struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
Name string `json:"name"`
|
||||||
|
Role string `json:"role"`
|
||||||
|
}
|
||||||
|
decode(t, c.do(), &teams)
|
||||||
|
if len(teams) != 1 {
|
||||||
|
t.Fatalf("%s: expected 1 team, got %d", c.name, len(teams))
|
||||||
|
}
|
||||||
|
if teams[0].Role != "member" {
|
||||||
|
t.Errorf("%s: expected role member, got %q", c.name, teams[0].Role)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Somebody else's is not theirs to read.
|
||||||
|
otherID, _ := member(t, s, "nosy")
|
||||||
|
resp := call(http.MethodGet, "/api/users/"+id64(otherID)+"/teams", nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("reading another user's teams: expected 403, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A user who does not exist is a 404 rather than an empty list, which is
|
||||||
|
// how the page tells "no teams" from "no such person".
|
||||||
|
resp = s.req(t, http.MethodGet, "/api/users/9999/teams", nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNotFound {
|
||||||
|
t.Errorf("a missing user: expected 404, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// The flag has to reach the client, or the web UI cannot decide what to show.
|
// The flag has to reach the client, or the web UI cannot decide what to show.
|
||||||
func TestAdmin_MeReportsTheFlag(t *testing.T) {
|
func TestAdmin_MeReportsTheFlag(t *testing.T) {
|
||||||
s := newTS(t)
|
s := newTS(t)
|
||||||
|
|||||||
@@ -0,0 +1,122 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
// This file is internal (package api, not api_test) because withAdvisoryLock is
|
||||||
|
// unexported and these tests exercise its locking semantics directly rather than
|
||||||
|
// through the full StartArchiver/StartNotifier loop, which would make the "does
|
||||||
|
// not run while held" case timing-dependent instead of deterministic. It opens a
|
||||||
|
// plain connection to TERDUT_TEST_DSN rather than reusing testdb_test.go's
|
||||||
|
// newTestDB, since that helper lives in the separate, already-compiled
|
||||||
|
// api_test package and a Postgres advisory lock needs no schema or migration
|
||||||
|
// to exercise.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"database/sql"
|
||||||
|
"os"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
_ "github.com/jackc/pgx/v5/stdlib"
|
||||||
|
)
|
||||||
|
|
||||||
|
// advisoryTestDB opens a plain, unmigrated connection to the test database. An
|
||||||
|
// unset DSN fails rather than skips, matching testdb_test.go's rationale: a
|
||||||
|
// suite that quietly tests nothing is worse than one that does not run.
|
||||||
|
func advisoryTestDB(t *testing.T) *sql.DB {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
dsn := os.Getenv("TERDUT_TEST_DSN")
|
||||||
|
if dsn == "" {
|
||||||
|
t.Fatalf("TERDUT_TEST_DSN is not set: these tests need Postgres.\n" +
|
||||||
|
"Run `make test-db` for a local one, then\n" +
|
||||||
|
" export TERDUT_TEST_DSN=postgres://terdut:terdut@localhost:5432/terdut_test?sslmode=disable")
|
||||||
|
}
|
||||||
|
|
||||||
|
db, err := sql.Open("pgx", dsn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("connect to TERDUT_TEST_DSN: %v", err)
|
||||||
|
}
|
||||||
|
t.Cleanup(func() { db.Close() })
|
||||||
|
return db
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestWithAdvisoryLock_RunsWhenFree(t *testing.T) {
|
||||||
|
db := advisoryTestDB(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
|
||||||
|
ran := false
|
||||||
|
withAdvisoryLock(ctx, db, archiverLockKey, "test", func() { ran = true })
|
||||||
|
|
||||||
|
if !ran {
|
||||||
|
t.Fatal("fn did not run although the lock was free")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestWithAdvisoryLock_SkipsWhileHeldElsewhere(t *testing.T) {
|
||||||
|
db := advisoryTestDB(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
|
||||||
|
// Hold the lock on a connection of our own, standing in for another
|
||||||
|
// replica mid-pass.
|
||||||
|
holder, err := db.Conn(ctx)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("acquire holder connection: %v", err)
|
||||||
|
}
|
||||||
|
defer holder.Close()
|
||||||
|
if _, err := holder.ExecContext(ctx, "SELECT pg_advisory_lock($1)", archiverLockKey); err != nil {
|
||||||
|
t.Fatalf("pre-acquire lock: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
ran := false
|
||||||
|
withAdvisoryLock(ctx, db, archiverLockKey, "test", func() { ran = true })
|
||||||
|
if ran {
|
||||||
|
t.Fatal("fn ran although another connection already held the lock")
|
||||||
|
}
|
||||||
|
|
||||||
|
if _, err := holder.ExecContext(ctx, "SELECT pg_advisory_unlock($1)", archiverLockKey); err != nil {
|
||||||
|
t.Fatalf("release held lock: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Now that the holder released it, the next caller should get it.
|
||||||
|
ran = false
|
||||||
|
withAdvisoryLock(ctx, db, archiverLockKey, "test", func() { ran = true })
|
||||||
|
if !ran {
|
||||||
|
t.Fatal("fn did not run after the other connection released the lock")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestWithAdvisoryLock_ReleasesAfterFnReturns(t *testing.T) {
|
||||||
|
db := advisoryTestDB(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
|
||||||
|
withAdvisoryLock(ctx, db, notifierLockKey, "test", func() {})
|
||||||
|
|
||||||
|
// If the first call had leaked the lock, this one would see it held and
|
||||||
|
// skip, leaving ran false.
|
||||||
|
ran := false
|
||||||
|
withAdvisoryLock(ctx, db, notifierLockKey, "test", func() { ran = true })
|
||||||
|
if !ran {
|
||||||
|
t.Fatal("fn did not run on a later call: the earlier call leaked its lock")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestWithAdvisoryLock_KeysAreIndependent(t *testing.T) {
|
||||||
|
db := advisoryTestDB(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
|
||||||
|
holder, err := db.Conn(ctx)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("acquire holder connection: %v", err)
|
||||||
|
}
|
||||||
|
defer holder.Close()
|
||||||
|
if _, err := holder.ExecContext(ctx, "SELECT pg_advisory_lock($1)", archiverLockKey); err != nil {
|
||||||
|
t.Fatalf("pre-acquire archiver lock: %v", err)
|
||||||
|
}
|
||||||
|
defer holder.ExecContext(ctx, "SELECT pg_advisory_unlock($1)", archiverLockKey)
|
||||||
|
|
||||||
|
// Holding archiverLockKey must not block notifierLockKey.
|
||||||
|
ran := false
|
||||||
|
withAdvisoryLock(ctx, db, notifierLockKey, "test", func() { ran = true })
|
||||||
|
if !ran {
|
||||||
|
t.Fatal("fn did not run under a different key although only archiverLockKey was held")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -78,7 +78,7 @@ type ingested struct {
|
|||||||
// post, and which team the alerts belong to.
|
// post, and which team the alerts belong to.
|
||||||
func handleIntegrationWebhook(db *sql.DB, notify NotifyConfig) http.HandlerFunc {
|
func handleIntegrationWebhook(db *sql.DB, notify NotifyConfig) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
teamID, err := teamIDForKey(r.Context(), db, chi.URLParam(r, "key"))
|
src, err := sourceForKey(r.Context(), db, chi.URLParam(r, "key"))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
if errors.Is(err, errUnknownIntegration) {
|
if errors.Is(err, errUnknownIntegration) {
|
||||||
// 401 and not 404: the path is real, the key is not, and a
|
// 401 and not 404: the path is real, the key is not, and a
|
||||||
@@ -87,16 +87,23 @@ func handleIntegrationWebhook(db *sql.DB, notify NotifyConfig) http.HandlerFunc
|
|||||||
respond(w, http.StatusUnauthorized, errResp("unknown integration key"))
|
respond(w, http.StatusUnauthorized, errResp("unknown integration key"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
receiveWebhook(w, r, db, notify, teamID)
|
receiveWebhook(w, r, db, notify, src)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func receiveWebhook(w http.ResponseWriter, r *http.Request, db *sql.DB, notify NotifyConfig, teamID int64) {
|
// maxWebhookBodyBytes is larger than maxBodyBytes: a real Alertmanager batch
|
||||||
|
// can carry many alerts, each with several labels and annotations, and the
|
||||||
|
// sender is a trusted piece of infrastructure rather than an arbitrary
|
||||||
|
// caller.
|
||||||
|
const maxWebhookBodyBytes = 8 << 20
|
||||||
|
|
||||||
|
func receiveWebhook(w http.ResponseWriter, r *http.Request, db *sql.DB, notify NotifyConfig, src alertSource) {
|
||||||
|
teamID := src.teamID
|
||||||
var payload amPayload
|
var payload amPayload
|
||||||
if err := decodeJSON(r, &payload); err != nil {
|
if err := decodeJSONLimit(r, &payload, maxWebhookBodyBytes); err != nil {
|
||||||
respond(w, http.StatusBadRequest, errResp("invalid payload"))
|
respond(w, http.StatusBadRequest, errResp("invalid payload"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
@@ -104,7 +111,7 @@ func receiveWebhook(w http.ResponseWriter, r *http.Request, db *sql.DB, notify N
|
|||||||
// Alertmanager retries anything that is not 2xx, and a retry of a payload
|
// Alertmanager retries anything that is not 2xx, and a retry of a payload
|
||||||
// we failed to store is more useful than an error it cannot act on — so
|
// we failed to store is more useful than an error it cannot act on — so
|
||||||
// failures are logged, not surfaced.
|
// failures are logged, not surfaced.
|
||||||
if err := ingest(r.Context(), db, notify, teamID, payload); err != nil {
|
if err := ingest(r.Context(), db, notify, src, payload); err != nil {
|
||||||
log.Printf("webhook ingest (team %d, group %q): %v", teamID, payload.GroupKey, err)
|
log.Printf("webhook ingest (team %d, group %q): %v", teamID, payload.GroupKey, err)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -114,7 +121,8 @@ func receiveWebhook(w http.ResponseWriter, r *http.Request, db *sql.DB, notify N
|
|||||||
// ingest stores a payload's alerts and reconciles the incident for its group.
|
// ingest stores a payload's alerts and reconciles the incident for its group.
|
||||||
// The whole payload is one transaction: an incident that opened but whose alerts
|
// The whole payload is one transaction: an incident that opened but whose alerts
|
||||||
// failed to link would be a work item nobody could act on.
|
// failed to link would be a work item nobody could act on.
|
||||||
func ingest(ctx context.Context, db *sql.DB, notify NotifyConfig, teamID int64, payload amPayload) error {
|
func ingest(ctx context.Context, db *sql.DB, notify NotifyConfig, src alertSource, payload amPayload) error {
|
||||||
|
teamID := src.teamID
|
||||||
tx, err := db.BeginTx(ctx, nil)
|
tx, err := db.BeginTx(ctx, nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
@@ -124,12 +132,12 @@ func ingest(ctx context.Context, db *sql.DB, notify NotifyConfig, teamID int64,
|
|||||||
// Which arriving alerts are heartbeats is the team's own answer, read
|
// Which arriving alerts are heartbeats is the team's own answer, read
|
||||||
// inside the transaction so an owner editing it mid-payload cannot split
|
// inside the transaction so an owner editing it mid-payload cannot split
|
||||||
// one webhook across two interpretations.
|
// one webhook across two interpretations.
|
||||||
deadman, err := deadmanConfigForTeam(ctx, tx, teamID)
|
deadman, err := deadmanSetForTeam(ctx, tx, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
accepted, err := upsertAlerts(ctx, tx, deadman, teamID, payload.Alerts)
|
accepted, err := upsertAlerts(ctx, tx, deadman, src, payload.Alerts)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
@@ -167,7 +175,7 @@ func ingest(ctx context.Context, db *sql.DB, notify NotifyConfig, teamID int64,
|
|||||||
}
|
}
|
||||||
touched[id] = true
|
touched[id] = true
|
||||||
alertID := a.id
|
alertID := a.id
|
||||||
if err := logEvent(ctx, tx, id, evAlertResolved, nil, &alertID, nil); err != nil {
|
if err := logEvent(ctx, tx, id, evAlertResolved, nil, nil, &alertID, nil); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -186,7 +194,8 @@ func ingest(ctx context.Context, db *sql.DB, notify NotifyConfig, teamID int64,
|
|||||||
|
|
||||||
// upsertAlerts stores each alert of a payload and reports what changed. Payloads
|
// upsertAlerts stores each alert of a payload and reports what changed. Payloads
|
||||||
// the ordering guard rejected are left out entirely.
|
// the ordering guard rejected are left out entirely.
|
||||||
func upsertAlerts(ctx context.Context, tx *sql.Tx, deadman DeadmanConfig, teamID int64, alerts []amAlert) ([]ingested, error) {
|
func upsertAlerts(ctx context.Context, tx *sql.Tx, deadman deadmanSet, src alertSource, alerts []amAlert) ([]ingested, error) {
|
||||||
|
teamID := src.teamID
|
||||||
now := time.Now().Unix()
|
now := time.Now().Unix()
|
||||||
accepted := make([]ingested, 0, len(alerts))
|
accepted := make([]ingested, 0, len(alerts))
|
||||||
|
|
||||||
@@ -242,8 +251,8 @@ func upsertAlerts(ctx context.Context, tx *sql.Tx, deadman DeadmanConfig, teamID
|
|||||||
if _, err := tx.ExecContext(ctx, `
|
if _, err := tx.ExecContext(ctx, `
|
||||||
INSERT INTO alerts
|
INSERT INTO alerts
|
||||||
(team_id, fingerprint, name, status, labels, annotations, starts_at, ends_at,
|
(team_id, fingerprint, name, status, labels, annotations, starts_at, ends_at,
|
||||||
generator_url, received_at, resolution_source)
|
generator_url, received_at, resolution_source, integration_id)
|
||||||
VALUES ($1, $2, $3, $4, $5::jsonb, $6::jsonb, $7, $8, $9, $10, $11)
|
VALUES ($1, $2, $3, $4, $5::jsonb, $6::jsonb, $7, $8, $9, $10, $11, $12)
|
||||||
ON CONFLICT (team_id, fingerprint) DO UPDATE SET
|
ON CONFLICT (team_id, fingerprint) DO UPDATE SET
|
||||||
status = excluded.status,
|
status = excluded.status,
|
||||||
labels = excluded.labels,
|
labels = excluded.labels,
|
||||||
@@ -257,6 +266,8 @@ func upsertAlerts(ctx context.Context, tx *sql.Tx, deadman DeadmanConfig, teamID
|
|||||||
-- is a breaking API change — see models.Alert.ReceivedAt.
|
-- is a breaking API change — see models.Alert.ReceivedAt.
|
||||||
received_at = excluded.received_at,
|
received_at = excluded.received_at,
|
||||||
resolution_source = excluded.resolution_source,
|
resolution_source = excluded.resolution_source,
|
||||||
|
-- Last sender wins; see migration 010.
|
||||||
|
integration_id = excluded.integration_id,
|
||||||
-- A re-fire makes the alert current again, so it leaves the archive.
|
-- A re-fire makes the alert current again, so it leaves the archive.
|
||||||
archived_at = CASE WHEN excluded.status = 'firing'
|
archived_at = CASE WHEN excluded.status = 'firing'
|
||||||
THEN NULL ELSE alerts.archived_at END
|
THEN NULL ELSE alerts.archived_at END
|
||||||
@@ -267,7 +278,7 @@ func upsertAlerts(ctx context.Context, tx *sql.Tx, deadman DeadmanConfig, teamID
|
|||||||
teamID, a.Fingerprint, name, a.Status,
|
teamID, a.Fingerprint, name, a.Status,
|
||||||
string(labelsJSON), string(annotationsJSON),
|
string(labelsJSON), string(annotationsJSON),
|
||||||
a.StartsAt.Unix(), endsAtUnix,
|
a.StartsAt.Unix(), endsAtUnix,
|
||||||
a.GeneratorURL, now, resolutionSource,
|
a.GeneratorURL, now, resolutionSource, src.integrationID,
|
||||||
); err != nil {
|
); err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
@@ -371,6 +382,17 @@ func incidentForGroup(ctx context.Context, tx *sql.Tx, notify NotifyConfig, team
|
|||||||
// its own. Hence the querier rather than a *sql.Tx. A nil severity leaves the
|
// its own. Hence the querier rather than a *sql.Tx. A nil severity leaves the
|
||||||
// column for refreshSeverity to fill from the member alerts; the sweeper passes
|
// column for refreshSeverity to fill from the member alerts; the sweeper passes
|
||||||
// one because its incidents have no members to derive it from.
|
// one because its incidents have no members to derive it from.
|
||||||
|
//
|
||||||
|
// Both callers get here only after their own SELECT found no open incident for
|
||||||
|
// this group_key — but on more than one replica, two webhook deliveries for the
|
||||||
|
// very first occurrence of a brand-new group_key can both pass that SELECT
|
||||||
|
// before either INSERTs. ON CONFLICT DO NOTHING against
|
||||||
|
// incidents_open_group_key_idx is what makes the loser's INSERT a no-op instead
|
||||||
|
// of a unique-violation error that would otherwise roll back its entire
|
||||||
|
// payload; existingOpenIncident then hands it the winner's row. Postgres
|
||||||
|
// resolves that conflict only once the winner's transaction has committed (or
|
||||||
|
// rolled back), so by the time this RETURNING comes back empty, the winner's
|
||||||
|
// row is guaranteed visible to that follow-up SELECT.
|
||||||
func openIncident(ctx context.Context, q querier, notify NotifyConfig, teamID int64, groupKey, title string, groupLabels map[string]string, severity *string) (int64, error) {
|
func openIncident(ctx context.Context, q querier, notify NotifyConfig, teamID int64, groupKey, title string, groupLabels map[string]string, severity *string) (int64, error) {
|
||||||
onCall, err := currentOnCall(ctx, q, teamID)
|
onCall, err := currentOnCall(ctx, q, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -384,21 +406,29 @@ func openIncident(ctx context.Context, q querier, notify NotifyConfig, teamID in
|
|||||||
|
|
||||||
var id int64
|
var id int64
|
||||||
err = q.QueryRowContext(ctx, `
|
err = q.QueryRowContext(ctx, `
|
||||||
INSERT INTO incidents (team_id, group_key, title, group_labels, status, severity, triggered_at, assigned_to)
|
INSERT INTO incidents (team_id, group_key, title, group_labels, signature, status, severity, triggered_at, assigned_to)
|
||||||
VALUES ($1, $2, $3, $4::jsonb, 'triggered', $5, $6, $7)
|
VALUES ($1, $2, $3, $4::jsonb, $5, 'triggered', $6, $7, $8)
|
||||||
|
ON CONFLICT (team_id, group_key) WHERE resolved_at IS NULL DO NOTHING
|
||||||
RETURNING id`,
|
RETURNING id`,
|
||||||
teamID, groupKey, title, string(labelsJSON), severity,
|
teamID, groupKey, title, string(labelsJSON), incidentSignature(groupLabels, title), severity,
|
||||||
time.Now().Unix(), onCall).Scan(&id)
|
time.Now().Unix(), onCall).Scan(&id)
|
||||||
if err != nil {
|
switch {
|
||||||
|
case err == sql.ErrNoRows:
|
||||||
|
// Lost the race: someone else's incident for this group_key exists now.
|
||||||
|
// Everything below — the trigger event, assignment, page, escalation
|
||||||
|
// clock — already happened for that row when it was created; attach to
|
||||||
|
// it rather than fail this call (and the whole payload) outright.
|
||||||
|
return existingOpenIncident(ctx, q, teamID, groupKey)
|
||||||
|
case err != nil:
|
||||||
return 0, err
|
return 0, err
|
||||||
}
|
}
|
||||||
|
|
||||||
if err := logEvent(ctx, q, id, evTriggered, nil, nil, nil); err != nil {
|
if err := logEvent(ctx, q, id, evTriggered, nil, nil, nil, nil); err != nil {
|
||||||
return 0, err
|
return 0, err
|
||||||
}
|
}
|
||||||
if onCall != nil {
|
if onCall != nil {
|
||||||
// On an "assigned" event user_id is the assignee, not the actor.
|
// On an "assigned" event user_id is the assignee, not the actor.
|
||||||
if err := logEvent(ctx, q, id, evAssigned, onCall, nil, nil); err != nil {
|
if err := logEvent(ctx, q, id, evAssigned, onCall, nil, nil, nil); err != nil {
|
||||||
return 0, err
|
return 0, err
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -418,6 +448,21 @@ func openIncident(ctx context.Context, q querier, notify NotifyConfig, teamID in
|
|||||||
return id, nil
|
return id, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// existingOpenIncident looks up the open incident openIncident's own INSERT just
|
||||||
|
// lost a conflict against — the same lookup incidentForGroup does before ever
|
||||||
|
// calling openIncident, repeated here for the caller that arrived second.
|
||||||
|
func existingOpenIncident(ctx context.Context, q querier, teamID int64, groupKey string) (int64, error) {
|
||||||
|
var id int64
|
||||||
|
err := q.QueryRowContext(ctx,
|
||||||
|
"SELECT id FROM incidents WHERE team_id = $1 AND group_key = $2 AND resolved_at IS NULL",
|
||||||
|
teamID, groupKey,
|
||||||
|
).Scan(&id)
|
||||||
|
if err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
return id, nil
|
||||||
|
}
|
||||||
|
|
||||||
// linkAlert adds an alert to an incident, emitting a timeline entry only the
|
// linkAlert adds an alert to an incident, emitting a timeline entry only the
|
||||||
// first time. Re-sends of an already-linked alert are silent.
|
// first time. Re-sends of an already-linked alert are silent.
|
||||||
func linkAlert(ctx context.Context, tx *sql.Tx, incidentID, alertID int64) error {
|
func linkAlert(ctx context.Context, tx *sql.Tx, incidentID, alertID int64) error {
|
||||||
@@ -431,5 +476,5 @@ func linkAlert(ctx context.Context, tx *sql.Tx, incidentID, alertID int64) error
|
|||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
return logEvent(ctx, tx, incidentID, evAlertAdded, nil, &alertID, nil)
|
return logEvent(ctx, tx, incidentID, evAlertAdded, nil, nil, &alertID, nil)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,112 @@
|
|||||||
|
package api_test
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"net/http"
|
||||||
|
"sync"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestWebhook_ConcurrentFirstOccurrenceOpensOneIncident reproduces two
|
||||||
|
// replicas racing the very first webhook delivery for a brand-new group_key:
|
||||||
|
// both see no open incident yet (incidentForGroup's own SELECT finds
|
||||||
|
// nothing) and race openIncident's INSERT.
|
||||||
|
//
|
||||||
|
// The DB's own unique index already guarantees at most one incident either
|
||||||
|
// way, with or without this fix — so "exactly one incident" alone cannot
|
||||||
|
// tell a fixed run from a broken one. What ON CONFLICT handling actually
|
||||||
|
// changes is what happens to the *loser*: before it, the loser's INSERT hit
|
||||||
|
// incidents_open_group_key_idx's unique violation, which — since
|
||||||
|
// upsertAlerts ran earlier in that same transaction — rolled back its whole
|
||||||
|
// payload, alert insert included. ingest's error is only logged and
|
||||||
|
// receiveWebhook answers 200 regardless, so nothing ever retried it: the
|
||||||
|
// loser's alert silently never existed. That is the regression signal this
|
||||||
|
// test checks — every caller's fingerprint must show up in /api/alerts, not
|
||||||
|
// just the winner's.
|
||||||
|
func TestWebhook_ConcurrentFirstOccurrenceOpensOneIncident(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
|
||||||
|
const callers = 8
|
||||||
|
const groupKey = "race-group"
|
||||||
|
|
||||||
|
// Every caller needs its own fingerprint. A shared one would serialize all
|
||||||
|
// of them at upsertAlerts' own ON CONFLICT (team_id, fingerprint) row lock,
|
||||||
|
// long before any of them reached incidentForGroup — which would hide the
|
||||||
|
// very race this test exists to force.
|
||||||
|
bodies := make([][]byte, callers)
|
||||||
|
for i := range callers {
|
||||||
|
payload := map[string]any{
|
||||||
|
"version": "4",
|
||||||
|
"status": "firing",
|
||||||
|
"groupKey": groupKey,
|
||||||
|
"groupLabels": map[string]string{"alertname": "RaceAlert"},
|
||||||
|
"alerts": []map[string]any{amAlert(fmt.Sprintf("fp-race-%d", i), "RaceAlert", "firing",
|
||||||
|
"2026-05-20T10:00:00Z", "0001-01-01T00:00:00Z", nil)},
|
||||||
|
}
|
||||||
|
bodies[i], _ = json.Marshal(payload)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A start line, so every request is fired as close to simultaneously as
|
||||||
|
// goroutine scheduling allows, rather than trickling out one dial at a
|
||||||
|
// time — the race window is the gap between incidentForGroup's SELECT and
|
||||||
|
// openIncident's INSERT, which a staggered start could easily miss.
|
||||||
|
var ready sync.WaitGroup
|
||||||
|
start := make(chan struct{})
|
||||||
|
statuses := make([]int, callers)
|
||||||
|
var wg sync.WaitGroup
|
||||||
|
for i := range callers {
|
||||||
|
ready.Add(1)
|
||||||
|
wg.Add(1)
|
||||||
|
go func(i int) {
|
||||||
|
defer wg.Done()
|
||||||
|
ready.Done()
|
||||||
|
<-start
|
||||||
|
resp, err := http.Post(s.URL+"/api/integrations/"+s.ingestKey+"/alertmanager",
|
||||||
|
"application/json", bytes.NewReader(bodies[i]))
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("post webhook #%d: %v", i, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer resp.Body.Close()
|
||||||
|
statuses[i] = resp.StatusCode
|
||||||
|
}(i)
|
||||||
|
}
|
||||||
|
ready.Wait()
|
||||||
|
close(start)
|
||||||
|
wg.Wait()
|
||||||
|
|
||||||
|
for i, code := range statuses {
|
||||||
|
if code != http.StatusOK {
|
||||||
|
t.Errorf("webhook #%d returned %d, want 200", i, code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
var matched []any
|
||||||
|
for _, inc := range listIncidents(t, s, "") {
|
||||||
|
if inc["group_key"] == groupKey {
|
||||||
|
matched = append(matched, inc["id"])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(matched) != 1 {
|
||||||
|
t.Fatalf("expected exactly 1 incident for group_key %q after %d concurrent deliveries, got %d: %v",
|
||||||
|
groupKey, callers, len(matched), matched)
|
||||||
|
}
|
||||||
|
|
||||||
|
var alerts []map[string]any
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/alerts", nil), &alerts)
|
||||||
|
seen := map[string]bool{}
|
||||||
|
for _, a := range alerts {
|
||||||
|
if fp, ok := a["fingerprint"].(string); ok {
|
||||||
|
seen[fp] = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for i := range callers {
|
||||||
|
fp := fmt.Sprintf("fp-race-%d", i)
|
||||||
|
if !seen[fp] {
|
||||||
|
t.Errorf("alert %q is missing: its delivery's whole payload was silently rolled back "+
|
||||||
|
"when it lost the race for the incident", fp)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -90,7 +90,7 @@ func handleListAlerts(db *sql.DB) http.HandlerFunc {
|
|||||||
fmt.Sprintf("%s WHERE %s ORDER BY a.received_at DESC LIMIT %s", alertSelectFrom, clause, args.add(limit)),
|
fmt.Sprintf("%s WHERE %s ORDER BY a.received_at DESC LIMIT %s", alertSelectFrom, clause, args.add(limit)),
|
||||||
args.all()...)
|
args.all()...)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -99,7 +99,7 @@ func handleListAlerts(db *sql.DB) http.HandlerFunc {
|
|||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
a, err := scanAlert(rows)
|
a, err := scanAlert(rows)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
alerts = append(alerts, a)
|
alerts = append(alerts, a)
|
||||||
@@ -121,7 +121,7 @@ func handleGetAlert(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, a)
|
respond(w, http.StatusOK, a)
|
||||||
|
|||||||
@@ -0,0 +1,135 @@
|
|||||||
|
package api_test
|
||||||
|
|
||||||
|
import (
|
||||||
|
"net/http"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestAPIKey_DefaultsToNeverExpiring(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
|
||||||
|
var key struct {
|
||||||
|
Key string `json:"key"`
|
||||||
|
ExpiresAt *string `json:"expires_at"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodPost, "/api/users/1/api-keys",
|
||||||
|
map[string]string{"name": "no-expiry"}), &key)
|
||||||
|
|
||||||
|
if key.ExpiresAt != nil {
|
||||||
|
t.Errorf("expires_at = %v, want nil (unset expires_in_days means never expires)", *key.ExpiresAt)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAPIKey_ExpiresInDaysSetsExpiresAt(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
|
||||||
|
var key struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
ExpiresAt *string `json:"expires_at"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodPost, "/api/users/1/api-keys",
|
||||||
|
map[string]any{"name": "rotates", "expires_in_days": 30}), &key)
|
||||||
|
|
||||||
|
if key.ExpiresAt == nil {
|
||||||
|
t.Fatal("expires_at = nil, want a timestamp roughly 30 days out")
|
||||||
|
}
|
||||||
|
got, err := time.Parse(time.RFC3339, *key.ExpiresAt)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse expires_at: %v", err)
|
||||||
|
}
|
||||||
|
want := time.Now().AddDate(0, 0, 30)
|
||||||
|
if diff := want.Sub(got).Abs(); diff > time.Hour {
|
||||||
|
t.Errorf("expires_at = %v, want close to %v (30 days out)", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAPIKey_ExpiresInDaysRejectsOutOfRange(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
|
||||||
|
for _, days := range []int{-1, 3651} {
|
||||||
|
resp := s.req(t, http.MethodPost, "/api/users/1/api-keys",
|
||||||
|
map[string]any{"name": "bad", "expires_in_days": days})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusBadRequest {
|
||||||
|
t.Errorf("expires_in_days=%d: status = %d, want %d", days, resp.StatusCode, http.StatusBadRequest)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAPIKey_AnExpiredKeyCannotAuthenticate(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
|
||||||
|
var key struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
Key string `json:"key"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodPost, "/api/users/1/api-keys",
|
||||||
|
map[string]any{"name": "soon-expired", "expires_in_days": 1}), &key)
|
||||||
|
|
||||||
|
// A fresh key works...
|
||||||
|
req, _ := http.NewRequest(http.MethodGet, s.URL+"/api/me", nil)
|
||||||
|
req.Header.Set("Authorization", "Bearer "+key.Key)
|
||||||
|
resp, err := http.DefaultClient.Do(req)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GET /api/me: %v", err)
|
||||||
|
}
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Fatalf("fresh key: status = %d, want %d", resp.StatusCode, http.StatusOK)
|
||||||
|
}
|
||||||
|
|
||||||
|
// ...and stops working once its expiry has passed.
|
||||||
|
s.exec(t, "UPDATE api_keys SET expires_at = $1 WHERE id = $2", time.Now().Add(-time.Hour).Unix(), key.ID)
|
||||||
|
|
||||||
|
req2, _ := http.NewRequest(http.MethodGet, s.URL+"/api/me", nil)
|
||||||
|
req2.Header.Set("Authorization", "Bearer "+key.Key)
|
||||||
|
resp2, err := http.DefaultClient.Do(req2)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("GET /api/me: %v", err)
|
||||||
|
}
|
||||||
|
defer resp2.Body.Close()
|
||||||
|
if resp2.StatusCode != http.StatusUnauthorized {
|
||||||
|
t.Errorf("expired key: status = %d, want %d", resp2.StatusCode, http.StatusUnauthorized)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAPIKey_ListNeverReturnsTheRawKey(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
|
||||||
|
decode(t, s.req(t, http.MethodPost, "/api/users/1/api-keys",
|
||||||
|
map[string]string{"name": "listed"}), new(struct {
|
||||||
|
Key string `json:"key"`
|
||||||
|
}))
|
||||||
|
|
||||||
|
var keys []struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
Name string `json:"name"`
|
||||||
|
Key string `json:"key"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/users/1/api-keys", nil), &keys)
|
||||||
|
|
||||||
|
found := false
|
||||||
|
for _, k := range keys {
|
||||||
|
if k.Name == "listed" {
|
||||||
|
found = true
|
||||||
|
}
|
||||||
|
if k.Key != "" {
|
||||||
|
t.Errorf("key %d (%s): raw key present in listing", k.ID, k.Name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !found {
|
||||||
|
t.Error("the key just created does not appear in the listing")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAPIKey_ListIsSelfOrAdmin(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
a := newTeam(t, s, "apikeys-a")
|
||||||
|
|
||||||
|
resp := a.call(http.MethodGet, "/api/users/1/api-keys", nil)
|
||||||
|
defer resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("status = %d, want %d (not self, not an admin)", resp.StatusCode, http.StatusForbidden)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -15,6 +15,7 @@ import (
|
|||||||
"time"
|
"time"
|
||||||
|
|
||||||
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||||
)
|
)
|
||||||
|
|
||||||
// ts wraps httptest.Server with a pre-bootstrapped API key. db is exposed so
|
// ts wraps httptest.Server with a pre-bootstrapped API key. db is exposed so
|
||||||
@@ -50,9 +51,16 @@ func newDeadmanTS(t *testing.T, deadman api.DeadmanConfig, notify ...api.NotifyC
|
|||||||
if len(notify) > 0 {
|
if len(notify) > 0 {
|
||||||
cfg = notify[0]
|
cfg = notify[0]
|
||||||
}
|
}
|
||||||
|
return newTSWith(t, deadman, cfg, testConfig())
|
||||||
|
}
|
||||||
|
|
||||||
|
// newTSWith is newDeadmanTS with the server's own configuration supplied, for
|
||||||
|
// tests of behaviour that config switches on, such as single sign-on.
|
||||||
|
func newTSWith(t *testing.T, deadman api.DeadmanConfig, cfg api.NotifyConfig, conf config.Config) *ts {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
database := newTestDB(t)
|
database := newTestDB(t)
|
||||||
srv := httptest.NewServer(api.NewRouter(database, cfg, testConfig()))
|
srv := httptest.NewServer(api.NewRouter(database, cfg, conf, "test"))
|
||||||
t.Cleanup(srv.Close)
|
t.Cleanup(srv.Close)
|
||||||
|
|
||||||
body, _ := json.Marshal(map[string]string{"username": "admin", "email": "admin@test.com"})
|
body, _ := json.Marshal(map[string]string{"username": "admin", "email": "admin@test.com"})
|
||||||
@@ -70,6 +78,10 @@ func newDeadmanTS(t *testing.T, deadman api.DeadmanConfig, notify ...api.NotifyC
|
|||||||
|
|
||||||
s := &ts{Server: srv, key: key, db: database, notify: cfg, deadman: deadman}
|
s := &ts{Server: srv, key: key, db: database, notify: cfg, deadman: deadman}
|
||||||
|
|
||||||
|
// A fresh install has no team, so the tests that want "the" team make it
|
||||||
|
// here: it is id 1, owned by the admin, which is what defaultTeam names.
|
||||||
|
decode(t, s.req(t, http.MethodPost, "/api/teams", map[string]string{"name": "Default"}), &struct{}{})
|
||||||
|
|
||||||
var integration struct {
|
var integration struct {
|
||||||
Key string `json:"key"`
|
Key string `json:"key"`
|
||||||
}
|
}
|
||||||
@@ -88,27 +100,25 @@ func newDeadmanTS(t *testing.T, deadman api.DeadmanConfig, notify ...api.NotifyC
|
|||||||
return s
|
return s
|
||||||
}
|
}
|
||||||
|
|
||||||
// setTeamDeadman configures the default team's switches over the API, rendering
|
// setTeamDeadman gives the default team one switch per configured matcher, over
|
||||||
// the matchers back into the string form the endpoint takes.
|
// the API, the way an owner would add them.
|
||||||
func setTeamDeadman(t *testing.T, s *ts, cfg api.DeadmanConfig) {
|
func setTeamDeadman(t *testing.T, s *ts, cfg api.DeadmanConfig) {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
matchers := make([]string, 0, len(cfg.Matchers))
|
|
||||||
for _, m := range cfg.Matchers {
|
for _, m := range cfg.Matchers {
|
||||||
parts := []string{"alertname=" + m.Name}
|
parts := []string{"alertname=" + m.Name}
|
||||||
for k, v := range m.Labels {
|
for k, v := range m.Labels {
|
||||||
parts = append(parts, k+"="+v)
|
parts = append(parts, k+"="+v)
|
||||||
}
|
}
|
||||||
sort.Strings(parts[1:])
|
sort.Strings(parts[1:])
|
||||||
matchers = append(matchers, strings.Join(parts, ","))
|
resp := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches", map[string]any{
|
||||||
}
|
"matcher": strings.Join(parts, ","),
|
||||||
resp := s.req(t, http.MethodPut, "/api/teams/"+defaultTeam+"/deadman", map[string]any{
|
"timeout_seconds": int64(cfg.Timeout.Seconds()),
|
||||||
"matchers": strings.Join(matchers, "; "),
|
"severity": cfg.Severity,
|
||||||
"timeout_seconds": int64(cfg.Timeout.Seconds()),
|
})
|
||||||
"severity": cfg.Severity,
|
resp.Body.Close()
|
||||||
})
|
if resp.StatusCode != http.StatusCreated {
|
||||||
defer resp.Body.Close()
|
t.Fatalf("add a dead man's switch: %d", resp.StatusCode)
|
||||||
if resp.StatusCode != http.StatusOK {
|
}
|
||||||
t.Fatalf("configure the team's dead man's switches: %d", resp.StatusCode)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -762,7 +772,7 @@ func TestWebhook_IgnoresOutOfOrderRetry(t *testing.T) {
|
|||||||
// An expiry resolve writes ends_at as an upper bound, not an observed end: an
|
// An expiry resolve writes ends_at as an upper bound, not an observed end: an
|
||||||
// Alertmanager watermark already on the row is preserved, and a row that never
|
// Alertmanager watermark already on the row is preserved, and a row that never
|
||||||
// carried one is stamped at sweep time. Clients are told to read it that way —
|
// carried one is stamped at sweep time. Clients are told to read it that way —
|
||||||
// see "resolution_source says how much to trust ends_at" in the README.
|
// see "resolution_source says how much to trust ends_at" in docs/api.md.
|
||||||
func TestExpiry_EndsAtIsUpperBound(t *testing.T) {
|
func TestExpiry_EndsAtIsUpperBound(t *testing.T) {
|
||||||
s := newTS(t)
|
s := newTS(t)
|
||||||
|
|
||||||
@@ -807,7 +817,7 @@ func TestExpiry_EndsAtIsUpperBound(t *testing.T) {
|
|||||||
//
|
//
|
||||||
// received_at is documented as a public liveness signal, so these lock the
|
// received_at is documented as a public liveness signal, so these lock the
|
||||||
// behaviour clients are told they may rely on. See "received_at is a liveness
|
// behaviour clients are told they may rely on. See "received_at is a liveness
|
||||||
// heartbeat" in the README and the comment on models.Alert.ReceivedAt.
|
// heartbeat" in docs/api.md and the comment on models.Alert.ReceivedAt.
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
// The heartbeat itself: an unchanged firing notification — what Alertmanager
|
// The heartbeat itself: an unchanged firing notification — what Alertmanager
|
||||||
@@ -869,3 +879,38 @@ func TestStats_ByDayReturnsSevenSlots(t *testing.T) {
|
|||||||
t.Errorf("expected 7 day slots, got %d", len(slots))
|
t.Errorf("expected 7 day slots, got %d", len(slots))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Two simultaneous bootstraps on an empty install must not both win.
|
||||||
|
func TestBootstrap_ConcurrentCallsCreateOneAdmin(t *testing.T) {
|
||||||
|
database := newTestDB(t)
|
||||||
|
srv := httptest.NewServer(api.NewRouter(database, api.NotifyConfig{}, testConfig(), "test"))
|
||||||
|
t.Cleanup(srv.Close)
|
||||||
|
|
||||||
|
const n = 8
|
||||||
|
codes := make(chan int, n)
|
||||||
|
for i := 0; i < n; i++ {
|
||||||
|
go func(i int) {
|
||||||
|
body, _ := json.Marshal(map[string]string{"username": fmt.Sprintf("u%d", i), "email": fmt.Sprintf("u%d@x.com", i)})
|
||||||
|
resp, err := http.Post(srv.URL+"/api/bootstrap", "application/json", bytes.NewReader(body))
|
||||||
|
if err != nil {
|
||||||
|
codes <- 0
|
||||||
|
return
|
||||||
|
}
|
||||||
|
resp.Body.Close()
|
||||||
|
codes <- resp.StatusCode
|
||||||
|
}(i)
|
||||||
|
}
|
||||||
|
created := 0
|
||||||
|
for i := 0; i < n; i++ {
|
||||||
|
if <-codes == http.StatusCreated {
|
||||||
|
created++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
var users int
|
||||||
|
if err := database.QueryRow("SELECT COUNT(*) FROM users").Scan(&users); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if created != 1 || users != 1 {
|
||||||
|
t.Errorf("expected exactly one bootstrap to win, got %d created and %d users", created, users)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -15,6 +15,12 @@ const (
|
|||||||
// expiryGrace absorbs clock skew and notification latency before an alert
|
// expiryGrace absorbs clock skew and notification latency before an alert
|
||||||
// whose ends_at watermark has passed is treated as stale.
|
// whose ends_at watermark has passed is treated as stale.
|
||||||
expiryGrace = 5 * time.Minute
|
expiryGrace = 5 * time.Minute
|
||||||
|
|
||||||
|
// archiverLockKey is the Postgres advisory lock the sweeper takes for the
|
||||||
|
// duration of each pass, so that running more than one replica does not run
|
||||||
|
// the sweep concurrently on all of them. Its value has no meaning beyond
|
||||||
|
// being distinct from notifierLockKey.
|
||||||
|
archiverLockKey int64 = 7265_0001
|
||||||
)
|
)
|
||||||
|
|
||||||
// StartArchiver runs the alert sweeper until ctx is cancelled, starting with an
|
// StartArchiver runs the alert sweeper until ctx is cancelled, starting with an
|
||||||
@@ -23,15 +29,25 @@ const (
|
|||||||
// the fallback, not the setting: each pass reads the current value from the
|
// the fallback, not the setting: each pass reads the current value from the
|
||||||
// settings table, so an administrator's change takes effect on the next tick
|
// settings table, so an administrator's change takes effect on the next tick
|
||||||
// instead of at the next restart.
|
// instead of at the next restart.
|
||||||
|
//
|
||||||
|
// Each pass runs under archiverLockKey (see withAdvisoryLock), so that on more
|
||||||
|
// than one replica only whichever instance's tick takes the lock first actually
|
||||||
|
// sweeps; the rest skip that tick rather than racing the same pass.
|
||||||
func StartArchiver(ctx context.Context, db *sql.DB, archiveAfter, staleAfter time.Duration, notify NotifyConfig) {
|
func StartArchiver(ctx context.Context, db *sql.DB, archiveAfter, staleAfter time.Duration, notify NotifyConfig) {
|
||||||
ticker := time.NewTicker(sweepInterval)
|
ticker := time.NewTicker(sweepInterval)
|
||||||
defer ticker.Stop()
|
defer ticker.Stop()
|
||||||
|
|
||||||
Sweep(ctx, db, archiveAfter, staleAfter, notify)
|
sweep := func() {
|
||||||
|
withAdvisoryLock(ctx, db, archiverLockKey, "sweeper", func() {
|
||||||
|
Sweep(ctx, db, archiveAfter, staleAfter, notify)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
sweep()
|
||||||
for {
|
for {
|
||||||
select {
|
select {
|
||||||
case <-ticker.C:
|
case <-ticker.C:
|
||||||
Sweep(ctx, db, archiveAfter, staleAfter, notify)
|
sweep()
|
||||||
case <-ctx.Done():
|
case <-ctx.Done():
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
@@ -60,6 +76,7 @@ func Sweep(ctx context.Context, db *sql.DB, archiveAfter, staleAfter time.Durati
|
|||||||
archiveResolvedIncidents(ctx, db, archiveAfter)
|
archiveResolvedIncidents(ctx, db, archiveAfter)
|
||||||
purgeAckTokens(ctx, db)
|
purgeAckTokens(ctx, db)
|
||||||
purgeSessions(ctx, db)
|
purgeSessions(ctx, db)
|
||||||
|
purgeRateLimits(ctx, db)
|
||||||
}
|
}
|
||||||
|
|
||||||
// expireStale resolves firing alerts that Alertmanager has stopped refreshing.
|
// expireStale resolves firing alerts that Alertmanager has stopped refreshing.
|
||||||
@@ -105,6 +122,10 @@ func expireStale(ctx context.Context, db *sql.DB, staleAfter time.Duration, skip
|
|||||||
for i, id := range ids {
|
for i, id := range ids {
|
||||||
idList[i] = id
|
idList[i] = id
|
||||||
}
|
}
|
||||||
|
// #nosec G202 -- sqlArgs.add/addList only ever splice in the "$N"
|
||||||
|
// placeholder they hand back, never a value; every value travels through
|
||||||
|
// args.all() as a bound parameter. See the sqlArgs doc comment in
|
||||||
|
// helpers.go.
|
||||||
if _, err := db.ExecContext(ctx, `
|
if _, err := db.ExecContext(ctx, `
|
||||||
UPDATE alerts
|
UPDATE alerts
|
||||||
SET status = 'resolved',
|
SET status = 'resolved',
|
||||||
@@ -126,16 +147,14 @@ func expireStale(ctx context.Context, db *sql.DB, staleAfter time.Duration, skip
|
|||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
alertID := id
|
alertID := id
|
||||||
if err := logEvent(ctx, db, incidentID, evAlertResolved, nil, &alertID, nil); err != nil {
|
if err := logEvent(ctx, db, incidentID, evAlertResolved, nil, nil, &alertID, nil); err != nil {
|
||||||
log.Printf("sweeper: log expiry event: %v", err)
|
log.Printf("sweeper: log expiry event: %v", err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// staleAlertIDs reads the ids in one go and closes the cursor before the caller
|
// staleAlertIDs reads the ids in one go and closes the cursor before the caller
|
||||||
// writes. Under SQLite's single connection an open read would have blocked the
|
// writes, which keeps the write off a cursor the same transaction is walking.
|
||||||
// update outright; with a pool it is no longer a deadlock, but reading the set
|
|
||||||
// first still keeps the write off a cursor the same transaction is walking.
|
|
||||||
func staleAlertIDs(ctx context.Context, db *sql.DB, now time.Time, staleAfter time.Duration) ([]int64, error) {
|
func staleAlertIDs(ctx context.Context, db *sql.DB, now time.Time, staleAfter time.Duration) ([]int64, error) {
|
||||||
rows, err := db.QueryContext(ctx, `
|
rows, err := db.QueryContext(ctx, `
|
||||||
SELECT id FROM alerts
|
SELECT id FROM alerts
|
||||||
|
|||||||
@@ -10,6 +10,7 @@ import (
|
|||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
"sync"
|
"sync"
|
||||||
|
"sync/atomic"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
"github.com/go-chi/chi/v5"
|
"github.com/go-chi/chi/v5"
|
||||||
@@ -28,6 +29,9 @@ const (
|
|||||||
// sessionTouchEvery bounds how often a request may slide the expiry.
|
// sessionTouchEvery bounds how often a request may slide the expiry.
|
||||||
sessionTouchEvery = time.Hour
|
sessionTouchEvery = time.Hour
|
||||||
|
|
||||||
|
// keyTouchEvery is the same bound for an API key's last_used_at.
|
||||||
|
keyTouchEvery = 5 * time.Minute
|
||||||
|
|
||||||
minPasswordLen = 10
|
minPasswordLen = 10
|
||||||
// maxPasswordLen is bcrypt's limit; it rejects longer input outright.
|
// maxPasswordLen is bcrypt's limit; it rejects longer input outright.
|
||||||
maxPasswordLen = 72
|
maxPasswordLen = 72
|
||||||
@@ -45,67 +49,113 @@ var dummyHash = sync.OnceValue(func() []byte {
|
|||||||
return h
|
return h
|
||||||
})
|
})
|
||||||
|
|
||||||
// loginLimiter counts failed logins in a fixed window, per username and per
|
// loginLimiter counts failed logins (and other unauthenticated attempts:
|
||||||
// client address. The username limit is what stops guessing one account; the
|
// sign-up, OIDC/device start) in a fixed window, per key — a username, a
|
||||||
// address limit is looser because every user behind the same gateway or NAT
|
// client address, or both, depending on the caller.
|
||||||
// shares it.
|
//
|
||||||
|
// Backed by Postgres rather than an in-memory map: this server runs more
|
||||||
|
// than one replica in production (v0.37.0), and a counter that only ever
|
||||||
|
// sees its own pod's traffic would quietly let every limit through
|
||||||
|
// multiplied by the replica count — two loginLimiter values pointed at the
|
||||||
|
// same db, standing in for two replicas, now share exactly one count per
|
||||||
|
// key instead of each keeping their own.
|
||||||
|
//
|
||||||
|
// The window resets rather than slides, the same behavior the in-memory
|
||||||
|
// version it replaces had: once a key's window is older than loginWindow,
|
||||||
|
// the next fail() starts a fresh one instead of extending the stale one.
|
||||||
type loginLimiter struct {
|
type loginLimiter struct {
|
||||||
mu sync.Mutex
|
db *sql.DB
|
||||||
failures map[string]*loginWindowCount
|
|
||||||
}
|
}
|
||||||
|
|
||||||
type loginWindowCount struct {
|
func newLoginLimiter(db *sql.DB) *loginLimiter {
|
||||||
start time.Time
|
return &loginLimiter{db: db}
|
||||||
n int
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func newLoginLimiter() *loginLimiter {
|
func (l *loginLimiter) blocked(ctx context.Context, key string, max int) bool {
|
||||||
return &loginLimiter{failures: map[string]*loginWindowCount{}}
|
cutoff := time.Now().Unix() - int64(loginWindow.Seconds())
|
||||||
}
|
var count int
|
||||||
|
err := l.db.QueryRowContext(ctx, `
|
||||||
func (l *loginLimiter) blocked(key string, max int) bool {
|
SELECT count FROM rate_limit_counters
|
||||||
l.mu.Lock()
|
WHERE key = $1 AND window_start > $2`,
|
||||||
defer l.mu.Unlock()
|
key, cutoff,
|
||||||
c, ok := l.failures[key]
|
).Scan(&count)
|
||||||
if !ok || time.Since(c.start) > loginWindow {
|
if err != nil {
|
||||||
|
// No row (never failed, or its window already expired): not blocked.
|
||||||
|
// A real query error fails the same way — a rate limiter that locks
|
||||||
|
// everyone out during a brief database hiccup is worse than one that
|
||||||
|
// is briefly too generous.
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
return c.n >= max
|
return count >= max
|
||||||
}
|
}
|
||||||
|
|
||||||
func (l *loginLimiter) fail(keys ...string) {
|
func (l *loginLimiter) fail(ctx context.Context, keys ...string) {
|
||||||
l.mu.Lock()
|
now := time.Now().Unix()
|
||||||
defer l.mu.Unlock()
|
windowSecs := int64(loginWindow.Seconds())
|
||||||
now := time.Now()
|
|
||||||
for k, c := range l.failures {
|
|
||||||
if now.Sub(c.start) > loginWindow {
|
|
||||||
delete(l.failures, k)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for _, key := range keys {
|
for _, key := range keys {
|
||||||
c, ok := l.failures[key]
|
if _, err := l.db.ExecContext(ctx, `
|
||||||
if !ok {
|
INSERT INTO rate_limit_counters (key, window_start, count)
|
||||||
c = &loginWindowCount{start: now}
|
VALUES ($1, $2, 1)
|
||||||
l.failures[key] = c
|
ON CONFLICT (key) DO UPDATE SET
|
||||||
|
window_start = CASE WHEN rate_limit_counters.window_start <= $2 - $3
|
||||||
|
THEN $2 ELSE rate_limit_counters.window_start END,
|
||||||
|
count = CASE WHEN rate_limit_counters.window_start <= $2 - $3
|
||||||
|
THEN 1 ELSE rate_limit_counters.count + 1 END`,
|
||||||
|
key, now, windowSecs,
|
||||||
|
); err != nil {
|
||||||
|
log.Printf("rate limiter: record failure for %q: %v", key, err) // #nosec G706 -- %q
|
||||||
}
|
}
|
||||||
c.n++
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func (l *loginLimiter) clear(key string) {
|
func (l *loginLimiter) clear(ctx context.Context, key string) {
|
||||||
l.mu.Lock()
|
if _, err := l.db.ExecContext(ctx, "DELETE FROM rate_limit_counters WHERE key = $1", key); err != nil {
|
||||||
defer l.mu.Unlock()
|
log.Printf("rate limiter: clear %q: %v", key, err)
|
||||||
delete(l.failures, key)
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// purgeRateLimits deletes rate-limit windows that have expired, from the
|
||||||
|
// sweeper — otherwise every distinct username and address this server has
|
||||||
|
// ever seen a failed attempt from would stay a row forever.
|
||||||
|
func purgeRateLimits(ctx context.Context, db *sql.DB) {
|
||||||
|
cutoff := time.Now().Unix() - int64(loginWindow.Seconds())
|
||||||
|
res, err := db.ExecContext(ctx,
|
||||||
|
"DELETE FROM rate_limit_counters WHERE window_start <= $1", cutoff)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("sweeper: purge rate limit counters: %v", err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if n, _ := res.RowsAffected(); n > 0 {
|
||||||
|
log.Printf("sweeper: purged %d expired rate limit counter(s)", n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// trustedProxies is how many X-Forwarded-For hops clientAddr trusts. Set once
|
||||||
|
// by NewRouter from config.
|
||||||
|
var trustedProxies atomic.Int64
|
||||||
|
|
||||||
// clientAddr is the address a login is counted against. Behind the gateway
|
// clientAddr is the address a login is counted against. Behind the gateway
|
||||||
// RemoteAddr is the gateway itself, so the first X-Forwarded-For hop is used
|
// RemoteAddr is the gateway itself, so the client address is read from
|
||||||
// when present. It can be forged, but only to dodge the address limit; the
|
// X-Forwarded-For, counting trustedProxies entries from the right: each trusted
|
||||||
// per-username limit does not depend on it.
|
// proxy appends the address it saw, so the entries to the left of those are
|
||||||
|
// client-supplied and could be forged to dodge the limit.
|
||||||
func clientAddr(r *http.Request) string {
|
func clientAddr(r *http.Request) string {
|
||||||
if xff := r.Header.Get("X-Forwarded-For"); xff != "" {
|
if n := int(trustedProxies.Load()); n > 0 {
|
||||||
first, _, _ := strings.Cut(xff, ",")
|
var hops []string
|
||||||
return strings.TrimSpace(first)
|
for _, v := range r.Header.Values("X-Forwarded-For") {
|
||||||
|
for _, h := range strings.Split(v, ",") {
|
||||||
|
if h = strings.TrimSpace(h); h != "" {
|
||||||
|
hops = append(hops, h)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(hops) > 0 {
|
||||||
|
i := len(hops) - n
|
||||||
|
if i < 0 {
|
||||||
|
i = 0
|
||||||
|
}
|
||||||
|
return hops[i]
|
||||||
|
}
|
||||||
}
|
}
|
||||||
host, _, err := net.SplitHostPort(r.RemoteAddr)
|
host, _, err := net.SplitHostPort(r.RemoteAddr)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -139,6 +189,53 @@ func hashPassword(pw string) (string, error) {
|
|||||||
return string(h), err
|
return string(h), err
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// startSession mints a session and sets the cookie. Shared by login and
|
||||||
|
// sign-up: somebody who has just chosen a password is signed in, rather than
|
||||||
|
// being sent to a form to type the same credential again.
|
||||||
|
func startSession(w http.ResponseWriter, r *http.Request, db *sql.DB, userID int64, publicURL string) error {
|
||||||
|
return startSessionCapped(w, r, db, userID, publicURL, 0)
|
||||||
|
}
|
||||||
|
|
||||||
|
// startSessionCapped is startSession with a hard ceiling on the session's life,
|
||||||
|
// which sliding never extends. maxAge zero means no ceiling. A single sign-on
|
||||||
|
// login uses it: the login is the only moment the provider's groups are read, so
|
||||||
|
// a session that could outlive it indefinitely would keep access the provider
|
||||||
|
// has since taken away.
|
||||||
|
func startSessionCapped(w http.ResponseWriter, r *http.Request, db *sql.DB, userID int64, publicURL string, maxAge time.Duration) error {
|
||||||
|
raw, tokenHash, err := randomToken()
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
now := time.Now()
|
||||||
|
life := sessionTTL
|
||||||
|
var ceiling *int64
|
||||||
|
if maxAge > 0 {
|
||||||
|
c := now.Add(maxAge).Unix()
|
||||||
|
ceiling = &c
|
||||||
|
life = min(life, maxAge)
|
||||||
|
}
|
||||||
|
if _, err := db.ExecContext(r.Context(), `
|
||||||
|
INSERT INTO sessions (token_hash, user_id, created_at, last_seen_at, expires_at, max_expires_at, user_agent)
|
||||||
|
VALUES ($1, $2, $3, $4, $5, $6, $7)`,
|
||||||
|
tokenHash, userID, now.Unix(), now.Unix(), now.Add(life).Unix(), ceiling, r.UserAgent()); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
// #nosec G124 -- HttpOnly/SameSite are literal below; Secure is
|
||||||
|
// cookieSecure(publicURL, r), not a literal true, which is what trips
|
||||||
|
// this rule. See cookieSecure's own doc comment above.
|
||||||
|
http.SetCookie(w, &http.Cookie{
|
||||||
|
Name: sessionCookie,
|
||||||
|
Value: raw,
|
||||||
|
Path: "/",
|
||||||
|
MaxAge: int(life.Seconds()),
|
||||||
|
HttpOnly: true,
|
||||||
|
Secure: cookieSecure(publicURL, r),
|
||||||
|
SameSite: http.SameSiteLaxMode,
|
||||||
|
})
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
// handleLogin exchanges a username and password for a session cookie.
|
// handleLogin exchanges a username and password for a session cookie.
|
||||||
func handleLogin(db *sql.DB, limiter *loginLimiter, publicURL string) http.HandlerFunc {
|
func handleLogin(db *sql.DB, limiter *loginLimiter, publicURL string) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
@@ -154,7 +251,7 @@ func handleLogin(db *sql.DB, limiter *loginLimiter, publicURL string) http.Handl
|
|||||||
userKey := "user:" + strings.ToLower(username)
|
userKey := "user:" + strings.ToLower(username)
|
||||||
addrKey := "addr:" + clientAddr(r)
|
addrKey := "addr:" + clientAddr(r)
|
||||||
|
|
||||||
if limiter.blocked(userKey, loginMaxPerUser) || limiter.blocked(addrKey, loginMaxPerAddr) {
|
if limiter.blocked(r.Context(), userKey, loginMaxPerUser) || limiter.blocked(r.Context(), addrKey, loginMaxPerAddr) {
|
||||||
w.Header().Set("Retry-After", strconv.Itoa(int(loginWindow.Seconds())))
|
w.Header().Set("Retry-After", strconv.Itoa(int(loginWindow.Seconds())))
|
||||||
respond(w, http.StatusTooManyRequests, errResp("too many failed attempts, try again later"))
|
respond(w, http.StatusTooManyRequests, errResp("too many failed attempts, try again later"))
|
||||||
return
|
return
|
||||||
@@ -166,7 +263,7 @@ func handleLogin(db *sql.DB, limiter *loginLimiter, publicURL string) http.Handl
|
|||||||
"SELECT id, password_hash FROM users WHERE username = $1", username,
|
"SELECT id, password_hash FROM users WHERE username = $1", username,
|
||||||
).Scan(&userID, &hash)
|
).Scan(&userID, &hash)
|
||||||
if err != nil && !errors.Is(err, sql.ErrNoRows) {
|
if err != nil && !errors.Is(err, sql.ErrNoRows) {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -176,39 +273,20 @@ func handleLogin(db *sql.DB, limiter *loginLimiter, publicURL string) http.Handl
|
|||||||
}
|
}
|
||||||
match := bcrypt.CompareHashAndPassword(stored, []byte(req.Password)) == nil
|
match := bcrypt.CompareHashAndPassword(stored, []byte(req.Password)) == nil
|
||||||
if !match || !hash.Valid {
|
if !match || !hash.Valid {
|
||||||
limiter.fail(userKey, addrKey)
|
limiter.fail(r.Context(), userKey, addrKey)
|
||||||
respond(w, http.StatusUnauthorized, errResp("invalid username or password"))
|
respond(w, http.StatusUnauthorized, errResp("invalid username or password"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
limiter.clear(userKey)
|
limiter.clear(r.Context(), userKey)
|
||||||
|
|
||||||
raw, tokenHash, err := randomToken()
|
if err := startSession(w, r, db, userID, publicURL); err != nil {
|
||||||
if err != nil {
|
serverError(w, r, err)
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
now := time.Now()
|
|
||||||
if _, err := db.ExecContext(r.Context(), `
|
|
||||||
INSERT INTO sessions (token_hash, user_id, created_at, last_seen_at, expires_at, user_agent)
|
|
||||||
VALUES ($1, $2, $3, $4, $5, $6)`,
|
|
||||||
tokenHash, userID, now.Unix(), now.Unix(), now.Add(sessionTTL).Unix(), r.UserAgent()); err != nil {
|
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
|
||||||
return
|
|
||||||
}
|
|
||||||
|
|
||||||
http.SetCookie(w, &http.Cookie{
|
|
||||||
Name: sessionCookie,
|
|
||||||
Value: raw,
|
|
||||||
Path: "/",
|
|
||||||
MaxAge: int(sessionTTL.Seconds()),
|
|
||||||
HttpOnly: true,
|
|
||||||
Secure: cookieSecure(publicURL, r),
|
|
||||||
SameSite: http.SameSiteLaxMode,
|
|
||||||
})
|
|
||||||
|
|
||||||
user, err := fetchUser(r.Context(), db, userID)
|
user, err := fetchUser(r.Context(), db, userID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, meResponse{User: user, HasPassword: true})
|
respond(w, http.StatusOK, meResponse{User: user, HasPassword: true})
|
||||||
@@ -227,6 +305,9 @@ func handleLogout(db *sql.DB, publicURL string) http.HandlerFunc {
|
|||||||
if c, err := r.Cookie(sessionCookie); err == nil && c.Value != "" {
|
if c, err := r.Cookie(sessionCookie); err == nil && c.Value != "" {
|
||||||
db.ExecContext(r.Context(), "DELETE FROM sessions WHERE token_hash = $1", hashToken(c.Value))
|
db.ExecContext(r.Context(), "DELETE FROM sessions WHERE token_hash = $1", hashToken(c.Value))
|
||||||
}
|
}
|
||||||
|
// #nosec G124 -- HttpOnly/SameSite are literal below; Secure is
|
||||||
|
// cookieSecure(publicURL, r), not a literal true, which is what
|
||||||
|
// trips this rule. See cookieSecure's own doc comment above.
|
||||||
http.SetCookie(w, &http.Cookie{
|
http.SetCookie(w, &http.Cookie{
|
||||||
Name: sessionCookie,
|
Name: sessionCookie,
|
||||||
Value: "",
|
Value: "",
|
||||||
@@ -243,22 +324,44 @@ func handleLogout(db *sql.DB, publicURL string) http.HandlerFunc {
|
|||||||
type meResponse struct {
|
type meResponse struct {
|
||||||
User any `json:"user"`
|
User any `json:"user"`
|
||||||
HasPassword bool `json:"has_password"`
|
HasPassword bool `json:"has_password"`
|
||||||
|
|
||||||
|
// OnboardingDismissed is whether this person has put the first-run
|
||||||
|
// checklist away. Per user rather than per browser: somebody who finishes
|
||||||
|
// setting up on a laptop should not be nagged again on their phone.
|
||||||
|
OnboardingDismissed bool `json:"onboarding_dismissed"`
|
||||||
}
|
}
|
||||||
|
|
||||||
// handleMe says who the caller is. The web UI calls it on load to decide
|
// handleMe says who the caller is. The web UI calls it on load to decide
|
||||||
// between the login form and the app, since it cannot read its own cookie.
|
// between the login form and the app, since it cannot read its own cookie.
|
||||||
func handleMe(db *sql.DB) http.HandlerFunc {
|
func handleMe(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
caller, _ := userFromContext(r.Context())
|
caller, ok := userFromContext(r.Context())
|
||||||
|
if !ok {
|
||||||
|
respond(w, http.StatusForbidden, errResp("this endpoint is for human accounts only"))
|
||||||
|
return
|
||||||
|
}
|
||||||
user, err := fetchUser(r.Context(), db, caller.ID)
|
user, err := fetchUser(r.Context(), db, caller.ID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
var hash sql.NullString
|
var hash sql.NullString
|
||||||
db.QueryRowContext(r.Context(),
|
var dismissed *int64
|
||||||
"SELECT password_hash FROM users WHERE id = $1", caller.ID).Scan(&hash)
|
if err := db.QueryRowContext(r.Context(),
|
||||||
respond(w, http.StatusOK, meResponse{User: user, HasPassword: hash.Valid})
|
"SELECT password_hash, onboarding_dismissed_at FROM users WHERE id = $1",
|
||||||
|
caller.ID).Scan(&hash, &dismissed); err != nil {
|
||||||
|
// fetchUser above already found this row, so an error here is a
|
||||||
|
// transient database problem, not a missing user — worth a 500
|
||||||
|
// rather than silently answering "no password, not dismissed",
|
||||||
|
// which a client would otherwise take at face value.
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusOK, meResponse{
|
||||||
|
User: user,
|
||||||
|
HasPassword: hash.Valid,
|
||||||
|
OnboardingDismissed: dismissed != nil,
|
||||||
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -303,7 +406,7 @@ func handleSetPassword(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -316,30 +419,30 @@ func handleSetPassword(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
hash, err := hashPassword(req.Password)
|
hash, err := hashPassword(req.Password)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
tx, err := db.BeginTx(r.Context(), nil)
|
tx, err := db.BeginTx(r.Context(), nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer tx.Rollback()
|
defer tx.Rollback()
|
||||||
|
|
||||||
if _, err := tx.ExecContext(r.Context(),
|
if _, err := tx.ExecContext(r.Context(),
|
||||||
"UPDATE users SET password_hash = $1 WHERE id = $2", hash, id); err != nil {
|
"UPDATE users SET password_hash = $1 WHERE id = $2", hash, id); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
keep, _ := sessionFromContext(r.Context()) // zero when changed with an API key
|
keep, _ := sessionFromContext(r.Context()) // zero when changed with an API key
|
||||||
if _, err := tx.ExecContext(r.Context(),
|
if _, err := tx.ExecContext(r.Context(),
|
||||||
"DELETE FROM sessions WHERE user_id = $1 AND id != $2", id, keep); err != nil {
|
"DELETE FROM sessions WHERE user_id = $1 AND id != $2", id, keep); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err := tx.Commit(); err != nil {
|
if err := tx.Commit(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
w.WriteHeader(http.StatusNoContent)
|
w.WriteHeader(http.StatusNoContent)
|
||||||
|
|||||||
@@ -301,7 +301,7 @@ func TestSetPassword_EndsOtherSessionsButNotThisOne(t *testing.T) {
|
|||||||
|
|
||||||
func TestBootstrap_WithPassword(t *testing.T) {
|
func TestBootstrap_WithPassword(t *testing.T) {
|
||||||
database := newTestDB(t)
|
database := newTestDB(t)
|
||||||
srv := httptest.NewServer(api.NewRouter(database, api.NotifyConfig{}, testConfig()))
|
srv := httptest.NewServer(api.NewRouter(database, api.NotifyConfig{}, testConfig(), "test"))
|
||||||
t.Cleanup(srv.Close)
|
t.Cleanup(srv.Close)
|
||||||
|
|
||||||
body := `{"username":"admin","email":"a@test.com","password":"` + adminPassword + `"}`
|
body := `{"username":"admin","email":"a@test.com","password":"` + adminPassword + `"}`
|
||||||
|
|||||||
@@ -0,0 +1,182 @@
|
|||||||
|
package api_test
|
||||||
|
|
||||||
|
import (
|
||||||
|
"net/http"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// This file is the regression test for the pattern documented throughout
|
||||||
|
// middleware.go: every team-scoped handler calls requireTeamMember or
|
||||||
|
// requireTeamOwner before touching data, every self-or-admin handler calls
|
||||||
|
// requireSelfOrAdmin, and every admin-only route sits behind AdminOnly. That
|
||||||
|
// pattern is enforced by convention, not by the type system — a new handler
|
||||||
|
// that forgets the call would compile and pass review on a quick read just
|
||||||
|
// as easily as one that remembers it. These tests exercise every route that
|
||||||
|
// carries one of those guards as a caller who should be refused, so a future
|
||||||
|
// handler missing its guard fails CI instead of becoming a silent IDOR.
|
||||||
|
|
||||||
|
// TestAuthzScope_TeamScopedRoutesRefuseANonMember builds two teams and, for
|
||||||
|
// every team-scoped route, calls it as team A's owner against team B's
|
||||||
|
// resources. requireTeamMember and requireTeamOwner both answer a non-member
|
||||||
|
// with 404 (team.go's own reasoning: whether a team exists is itself
|
||||||
|
// something only its members should learn), so every one of these must come
|
||||||
|
// back 404 regardless of which of the two guards its handler uses.
|
||||||
|
func TestAuthzScope_TeamScopedRoutesRefuseANonMember(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
a := newTeam(t, s, "authz-a")
|
||||||
|
b := newTeam(t, s, "authz-b")
|
||||||
|
|
||||||
|
// An incident in B, to cover the ID-based routes under /api/incidents —
|
||||||
|
// scoped by the incident's own team_id rather than a {teamID} path
|
||||||
|
// segment, but through the same single chokepoint (incidentIDParam).
|
||||||
|
postToIntegration(t, s, b.key, "fp-authz-scope", "AuthzScopeAlert")
|
||||||
|
var incidents []struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
}
|
||||||
|
decode(t, b.call(http.MethodGet, "/api/incidents", nil), &incidents)
|
||||||
|
if len(incidents) == 0 {
|
||||||
|
t.Fatal("setup: no incident in team B to test against")
|
||||||
|
}
|
||||||
|
incidentPath := "/api/incidents/" + id64(incidents[0].ID)
|
||||||
|
|
||||||
|
bPath := "/api/teams/" + id64(b.id)
|
||||||
|
tests := []struct {
|
||||||
|
method, path string
|
||||||
|
}{
|
||||||
|
// Team membership/ownership itself.
|
||||||
|
{http.MethodPut, bPath},
|
||||||
|
{http.MethodDelete, bPath},
|
||||||
|
{http.MethodGet, bPath + "/members"},
|
||||||
|
{http.MethodPost, bPath + "/members"},
|
||||||
|
{http.MethodDelete, bPath + "/members/1"},
|
||||||
|
|
||||||
|
// OIDC group binding.
|
||||||
|
{http.MethodGet, bPath + "/oidc-groups"},
|
||||||
|
{http.MethodPut, bPath + "/oidc-groups"},
|
||||||
|
|
||||||
|
// Invites.
|
||||||
|
{http.MethodGet, bPath + "/invites"},
|
||||||
|
{http.MethodPost, bPath + "/invites"},
|
||||||
|
{http.MethodDelete, bPath + "/invites/1"},
|
||||||
|
|
||||||
|
// Escalation.
|
||||||
|
{http.MethodGet, bPath + "/escalation"},
|
||||||
|
{http.MethodPut, bPath + "/escalation"},
|
||||||
|
|
||||||
|
// Dead man's switches.
|
||||||
|
{http.MethodGet, bPath + "/deadman/switches"},
|
||||||
|
{http.MethodPost, bPath + "/deadman/switches"},
|
||||||
|
{http.MethodPut, bPath + "/deadman/switches/1"},
|
||||||
|
{http.MethodDelete, bPath + "/deadman/switches/1"},
|
||||||
|
|
||||||
|
// Integrations.
|
||||||
|
{http.MethodGet, bPath + "/integrations"},
|
||||||
|
{http.MethodPost, bPath + "/integrations"},
|
||||||
|
{http.MethodPatch, bPath + "/integrations/1"},
|
||||||
|
{http.MethodDelete, bPath + "/integrations/1"},
|
||||||
|
|
||||||
|
// Schedule.
|
||||||
|
{http.MethodGet, bPath + "/schedule"},
|
||||||
|
{http.MethodPost, bPath + "/schedule"},
|
||||||
|
{http.MethodDelete, bPath + "/schedule/1"},
|
||||||
|
|
||||||
|
// Incidents, scoped by the incident's own team rather than a
|
||||||
|
// {teamID} segment.
|
||||||
|
{http.MethodGet, incidentPath},
|
||||||
|
{http.MethodGet, incidentPath + "/alerts"},
|
||||||
|
{http.MethodGet, incidentPath + "/timeline"},
|
||||||
|
{http.MethodGet, incidentPath + "/similar"},
|
||||||
|
{http.MethodPost, incidentPath + "/acknowledge"},
|
||||||
|
{http.MethodDelete, incidentPath + "/acknowledge"},
|
||||||
|
{http.MethodPost, incidentPath + "/resolve"},
|
||||||
|
{http.MethodPost, incidentPath + "/assign"},
|
||||||
|
{http.MethodPost, incidentPath + "/snooze"},
|
||||||
|
{http.MethodDelete, incidentPath + "/snooze"},
|
||||||
|
{http.MethodPost, incidentPath + "/archive"},
|
||||||
|
{http.MethodDelete, incidentPath + "/archive"},
|
||||||
|
{http.MethodPost, incidentPath + "/notes"},
|
||||||
|
{http.MethodDelete, incidentPath + "/notes/1"},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tc := range tests {
|
||||||
|
t.Run(tc.method+" "+tc.path, func(t *testing.T) {
|
||||||
|
resp := a.call(tc.method, tc.path, nil)
|
||||||
|
defer resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNotFound {
|
||||||
|
t.Errorf("status = %d, want %d (A is not a member of B)", resp.StatusCode, http.StatusNotFound)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAuthzScope_AdminOnlyRoutesRefuseANonAdmin exercises AdminOnly's group
|
||||||
|
// in router.go directly: a signed-in, non-admin caller gets 403 from every
|
||||||
|
// route in it, before any handler body runs.
|
||||||
|
func TestAuthzScope_AdminOnlyRoutesRefuseANonAdmin(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
a := newTeam(t, s, "authz-admin")
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
method, path string
|
||||||
|
}{
|
||||||
|
{http.MethodPost, "/api/users"},
|
||||||
|
{http.MethodDelete, "/api/users/1"},
|
||||||
|
{http.MethodPut, "/api/users/1/admin"},
|
||||||
|
{http.MethodPut, "/api/users/1/disabled"},
|
||||||
|
{http.MethodGet, "/api/admin/teams"},
|
||||||
|
{http.MethodGet, "/api/admin/teams/" + id64(a.id)},
|
||||||
|
{http.MethodGet, "/api/admin/settings"},
|
||||||
|
{http.MethodPut, "/api/admin/settings"},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tc := range tests {
|
||||||
|
t.Run(tc.method+" "+tc.path, func(t *testing.T) {
|
||||||
|
resp := a.call(tc.method, tc.path, nil)
|
||||||
|
defer resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("status = %d, want %d (not an admin)", resp.StatusCode, http.StatusForbidden)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAuthzScope_SelfOrAdminRoutesRefuseAnotherNonAdminUser exercises
|
||||||
|
// requireSelfOrAdmin's call sites: a non-admin caller acting on a *different*
|
||||||
|
// user's account must be refused, the same as AdminOnly's routes, even
|
||||||
|
// though these sit in the general authenticated group rather than behind
|
||||||
|
// AdminOnly itself.
|
||||||
|
func TestAuthzScope_SelfOrAdminRoutesRefuseAnotherNonAdminUser(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
a := newTeam(t, s, "authz-self-a")
|
||||||
|
b := newTeam(t, s, "authz-self-b")
|
||||||
|
|
||||||
|
var members []struct {
|
||||||
|
UserID int64 `json:"user_id"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/teams/"+id64(b.id)+"/members", nil), &members)
|
||||||
|
if len(members) == 0 {
|
||||||
|
t.Fatal("setup: team B has no members")
|
||||||
|
}
|
||||||
|
bUserID := id64(members[0].UserID)
|
||||||
|
|
||||||
|
tests := []struct {
|
||||||
|
method, path string
|
||||||
|
}{
|
||||||
|
{http.MethodGet, "/api/users/" + bUserID + "/teams"},
|
||||||
|
{http.MethodPut, "/api/users/" + bUserID + "/notify"},
|
||||||
|
{http.MethodPut, "/api/users/" + bUserID + "/password"},
|
||||||
|
{http.MethodGet, "/api/users/" + bUserID + "/api-keys"},
|
||||||
|
{http.MethodPost, "/api/users/" + bUserID + "/api-keys"},
|
||||||
|
{http.MethodDelete, "/api/users/" + bUserID + "/api-keys/1"},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tc := range tests {
|
||||||
|
t.Run(tc.method+" "+tc.path, func(t *testing.T) {
|
||||||
|
resp := a.call(tc.method, tc.path, nil)
|
||||||
|
defer resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("status = %d, want %d (not self, not an admin)", resp.StatusCode, http.StatusForbidden)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,115 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Caller is the one principal type every authorization predicate in this
|
||||||
|
// package reads from. Before this, a human (ctxUser + ctxTeams) and a
|
||||||
|
// service account (ctxServiceAccount + a synthetic ctxTeams entry) were two
|
||||||
|
// parallel, un-unified context representations — every predicate had to
|
||||||
|
// remember which one(s) it needed to check, and the ones that forgot either
|
||||||
|
// 403'd a service account that should have been let through (terdut-server#23,
|
||||||
|
// terdut-operator#3), crashed on an unchecked zero-value user ID (handleMe,
|
||||||
|
// handleTestNotification), or silently no-op'd (handleDismissOnboarding).
|
||||||
|
// serveAs and serveAsServiceAccount now both build exactly one Caller and
|
||||||
|
// store it under one context key; everything else in this file is a read
|
||||||
|
// of one of its methods.
|
||||||
|
type Caller struct {
|
||||||
|
// user is set for a human caller (session cookie or a user's own API
|
||||||
|
// key), nil for a service account of either scope.
|
||||||
|
user *models.User
|
||||||
|
|
||||||
|
// sa is set for a service-account caller, nil for a human.
|
||||||
|
sa *serviceAccountPrincipal
|
||||||
|
|
||||||
|
// memberships is the caller's real team_members rows for a human, or —
|
||||||
|
// for a team-scoped service account — the single synthetic owner
|
||||||
|
// membership serveAsServiceAccount injects (see its own comment for
|
||||||
|
// why). Always nil for an instance-scoped service account: it acts on
|
||||||
|
// teams by id, not by belonging to one.
|
||||||
|
memberships []membership
|
||||||
|
}
|
||||||
|
|
||||||
|
// AsHuman returns the real user behind this caller, or false for a service
|
||||||
|
// account of either scope. Every handler that needs a real user_id to act
|
||||||
|
// on behalf of — not just "is this caller sufficiently privileged" — calls
|
||||||
|
// this and handles the false case explicitly, replacing the unchecked
|
||||||
|
// userFromContext(ctx) zero-value reads that used to silently misbehave for
|
||||||
|
// a service-account caller.
|
||||||
|
func (c Caller) AsHuman() (models.User, bool) {
|
||||||
|
if c.user == nil {
|
||||||
|
return models.User{}, false
|
||||||
|
}
|
||||||
|
return *c.user, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// IsAdmin is true only for a human system administrator — never for a
|
||||||
|
// service account, of either scope, under any circumstance. AdminOnly and
|
||||||
|
// requireSelfOrAdmin key on this and nothing else: user management and
|
||||||
|
// /api/admin/settings stay human-only forever, by design (SERVICE-ACCOUNTS.md).
|
||||||
|
func (c Caller) IsAdmin() bool {
|
||||||
|
return c.user != nil && c.user.IsAdmin
|
||||||
|
}
|
||||||
|
|
||||||
|
// IsInstanceServiceAccount reports whether this caller is specifically an
|
||||||
|
// instance-scoped service account — never true for a human, including a
|
||||||
|
// human admin. handleCreateTeam needs exactly this: a human creates a team
|
||||||
|
// by being a human (and becomes its owner as a side effect), an
|
||||||
|
// instance-scoped service account creates one with no human owner at all;
|
||||||
|
// the two paths are not interchangeable, so this predicate must not also
|
||||||
|
// admit a human admin the way an administrator check would.
|
||||||
|
func (c Caller) IsInstanceServiceAccount() bool {
|
||||||
|
return c.sa != nil && c.sa.scope == models.ServiceAccountScopeInstance
|
||||||
|
}
|
||||||
|
|
||||||
|
// Role reports the caller's role in teamID, and whether they belong to it
|
||||||
|
// at all.
|
||||||
|
func (c Caller) Role(teamID int64) (string, bool) {
|
||||||
|
for _, m := range c.memberships {
|
||||||
|
if m.teamID == teamID {
|
||||||
|
return m.role, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
|
||||||
|
// TeamIDs lists every team this caller belongs to: a human's real
|
||||||
|
// memberships, or a team-scoped service account's own single team. Always
|
||||||
|
// empty for an instance-scoped service account.
|
||||||
|
func (c Caller) TeamIDs() []int64 {
|
||||||
|
ids := make([]int64, 0, len(c.memberships))
|
||||||
|
for _, m := range c.memberships {
|
||||||
|
ids = append(ids, m.teamID)
|
||||||
|
}
|
||||||
|
return ids
|
||||||
|
}
|
||||||
|
|
||||||
|
// ServiceAccountID reports this caller's own service-account id, for the
|
||||||
|
// "may manage/rotate its own credential" self-check in
|
||||||
|
// callerMayManageServiceAccount, and for OperatorModeBlock's "any service
|
||||||
|
// account passes" rule.
|
||||||
|
func (c Caller) ServiceAccountID() (int64, bool) {
|
||||||
|
if c.sa == nil {
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
return c.sa.id, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// ServiceAccountName reports this caller's own service-account name, for a
|
||||||
|
// handler's synchronous response — the same credential it authenticated
|
||||||
|
// with, already resolved onto the Caller by serveAsServiceAccount, so no
|
||||||
|
// extra query is needed.
|
||||||
|
func (c Caller) ServiceAccountName() (string, bool) {
|
||||||
|
if c.sa == nil {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
return c.sa.name, true
|
||||||
|
}
|
||||||
|
|
||||||
|
func callerFromContext(ctx context.Context) (Caller, bool) {
|
||||||
|
c, ok := ctx.Value(ctxCaller).(Caller)
|
||||||
|
return c, ok
|
||||||
|
}
|
||||||
@@ -4,6 +4,8 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"database/sql"
|
"database/sql"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
"log"
|
"log"
|
||||||
"sort"
|
"sort"
|
||||||
"strings"
|
"strings"
|
||||||
@@ -39,6 +41,17 @@ func (m DeadmanMatcher) String() string {
|
|||||||
return m.Name + " (" + strings.Join(parts, ", ") + ")"
|
return m.Name + " (" + strings.Join(parts, ", ") + ")"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// config renders the matcher in the form parseDeadmanMatcher reads, which is
|
||||||
|
// what a switch row stores: `alertname=Watchdog,cluster=prod`.
|
||||||
|
func (m DeadmanMatcher) config() string {
|
||||||
|
parts := make([]string, 0, len(m.Labels))
|
||||||
|
for k, v := range m.Labels {
|
||||||
|
parts = append(parts, k+"="+v)
|
||||||
|
}
|
||||||
|
sort.Strings(parts)
|
||||||
|
return strings.Join(append([]string{"alertname=" + m.Name}, parts...), ",")
|
||||||
|
}
|
||||||
|
|
||||||
// matches reports whether an alert's labels satisfy every condition.
|
// matches reports whether an alert's labels satisfy every condition.
|
||||||
func (m DeadmanMatcher) matches(labels map[string]string) bool {
|
func (m DeadmanMatcher) matches(labels map[string]string) bool {
|
||||||
if labels["alertname"] != m.Name {
|
if labels["alertname"] != m.Name {
|
||||||
@@ -52,133 +65,102 @@ func (m DeadmanMatcher) matches(labels map[string]string) bool {
|
|||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
|
||||||
// DeadmanConfig inverts the handling of the alerts it matches: receiving one
|
// DeadmanSwitch inverts the handling of the alerts it matches: receiving one
|
||||||
// opens nothing, and the absence of one opens an incident.
|
// opens nothing, and the absence of one opens an incident.
|
||||||
//
|
//
|
||||||
// The unit of monitoring is the fingerprint, not the matcher — two clusters
|
// The unit of monitoring is the fingerprint, not the switch — two clusters
|
||||||
// sending the same heartbeat alertname are two independent switches, so one
|
// sending the same heartbeat alertname are two independent heartbeats under one
|
||||||
// healthy cluster cannot mask a dead one.
|
// switch, so one healthy cluster cannot mask a dead one.
|
||||||
type DeadmanConfig struct {
|
type DeadmanSwitch struct {
|
||||||
Matchers []DeadmanMatcher
|
ID int64
|
||||||
|
Name string
|
||||||
|
Matcher DeadmanMatcher
|
||||||
|
|
||||||
// Timeout is how long a matched alert may go without a refreshing webhook
|
// Timeout is how long a heartbeat may go unheard before it is declared dead.
|
||||||
// before it is declared dead. It must be shorter than Alertmanager's
|
|
||||||
// repeat_interval for the heartbeat's route, which is what refreshes it.
|
|
||||||
// Zero disables dead man's switch handling entirely.
|
|
||||||
Timeout time.Duration
|
Timeout time.Duration
|
||||||
|
|
||||||
// Severity is the severity every dead man's switch incident opens at. These
|
// Severity is what the incident opens at.
|
||||||
// incidents have no member alerts to derive one from, and the heartbeat's
|
|
||||||
// own severity label is meaningless — Watchdog ships as "none".
|
|
||||||
Severity string
|
Severity string
|
||||||
}
|
}
|
||||||
|
|
||||||
// enabled reports whether there is anything to watch.
|
// deadmanSet is one team's switches.
|
||||||
func (c DeadmanConfig) enabled() bool { return c.Timeout > 0 && len(c.Matchers) > 0 }
|
type deadmanSet []DeadmanSwitch
|
||||||
|
|
||||||
// match returns the first matcher an alert satisfies.
|
// match returns the first switch an alert satisfies.
|
||||||
func (c DeadmanConfig) match(labels map[string]string) (DeadmanMatcher, bool) {
|
func (d deadmanSet) match(labels map[string]string) (DeadmanSwitch, bool) {
|
||||||
if !c.enabled() {
|
for _, sw := range d {
|
||||||
return DeadmanMatcher{}, false
|
if sw.Matcher.matches(labels) {
|
||||||
}
|
return sw, true
|
||||||
for _, m := range c.Matchers {
|
|
||||||
if m.matches(labels) {
|
|
||||||
return m, true
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return DeadmanMatcher{}, false
|
return DeadmanSwitch{}, false
|
||||||
}
|
}
|
||||||
|
|
||||||
// isDeadman is match without the matcher, for the ingest path.
|
// isDeadman is match without the switch, for the ingest path.
|
||||||
func (c DeadmanConfig) isDeadman(labels map[string]string) bool {
|
func (d deadmanSet) isDeadman(labels map[string]string) bool {
|
||||||
_, ok := c.match(labels)
|
_, ok := d.match(labels)
|
||||||
return ok
|
return ok
|
||||||
}
|
}
|
||||||
|
|
||||||
// names lists the distinct alertnames worth loading from the database.
|
// names lists the distinct alertnames worth loading from the database.
|
||||||
func (c DeadmanConfig) names() []string {
|
func (d deadmanSet) names() []string {
|
||||||
seen := map[string]bool{}
|
seen := map[string]bool{}
|
||||||
out := make([]string, 0, len(c.Matchers))
|
out := make([]string, 0, len(d))
|
||||||
for _, m := range c.Matchers {
|
for _, sw := range d {
|
||||||
if !seen[m.Name] {
|
if !seen[sw.Matcher.Name] {
|
||||||
seen[m.Name] = true
|
seen[sw.Matcher.Name] = true
|
||||||
out = append(out, m.Name)
|
out = append(out, sw.Matcher.Name)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return out
|
return out
|
||||||
}
|
}
|
||||||
|
|
||||||
// ParseDeadmanConfig reads the matcher list from its configured form:
|
// parseDeadmanMatcher reads one matcher from its configured form: "," separates
|
||||||
// ";" separates matchers, "," separates the conditions within one, and "=" is
|
// the conditions and "=" is exact label equality — `alertname=Watchdog,cluster=prod`.
|
||||||
// exact label equality — `alertname=Watchdog,cluster=prod; alertname=Heartbeat`.
|
// The error says what is wrong with it, in words a form can show.
|
||||||
//
|
func parseDeadmanMatcher(entry string) (DeadmanMatcher, error) {
|
||||||
// A malformed or alertname-less entry is dropped rather than fatal, following
|
m := DeadmanMatcher{Labels: map[string]string{}}
|
||||||
// config.duration's rule that one bad tuning knob should not take the server
|
for _, cond := range strings.Split(strings.TrimSpace(entry), ",") {
|
||||||
// down. Silence would be worse here than elsewhere, though — a typo that
|
k, v, ok := strings.Cut(cond, "=")
|
||||||
// disarms the switch is exactly the failure this feature exists to catch — so
|
k, v = strings.TrimSpace(k), strings.TrimSpace(v)
|
||||||
// the matchers that survived are logged.
|
if !ok || k == "" || v == "" {
|
||||||
func ParseDeadmanConfig(matchers string, timeout time.Duration, severity string) DeadmanConfig {
|
return DeadmanMatcher{}, fmt.Errorf("%q is not label=value", strings.TrimSpace(cond))
|
||||||
cfg := DeadmanConfig{Timeout: timeout, Severity: severity}
|
}
|
||||||
|
if k == "alertname" {
|
||||||
for _, entry := range strings.Split(matchers, ";") {
|
m.Name = v
|
||||||
entry = strings.TrimSpace(entry)
|
|
||||||
if entry == "" {
|
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
m.Labels[k] = v
|
||||||
m := DeadmanMatcher{Labels: map[string]string{}}
|
|
||||||
malformed := false
|
|
||||||
for _, cond := range strings.Split(entry, ",") {
|
|
||||||
k, v, ok := strings.Cut(cond, "=")
|
|
||||||
k, v = strings.TrimSpace(k), strings.TrimSpace(v)
|
|
||||||
if !ok || k == "" || v == "" {
|
|
||||||
log.Printf("deadman: ignoring matcher %q: %q is not label=value", entry, strings.TrimSpace(cond))
|
|
||||||
malformed = true
|
|
||||||
break
|
|
||||||
}
|
|
||||||
if k == "alertname" {
|
|
||||||
m.Name = v
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
m.Labels[k] = v
|
|
||||||
}
|
|
||||||
if malformed {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
if m.Name == "" {
|
|
||||||
log.Printf("deadman: ignoring matcher %q: no alertname condition", entry)
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
cfg.Matchers = append(cfg.Matchers, m)
|
|
||||||
}
|
}
|
||||||
|
if m.Name == "" {
|
||||||
switch {
|
return DeadmanMatcher{}, errors.New("no alertname condition")
|
||||||
case timeout <= 0:
|
|
||||||
log.Print("deadman: disabled (timeout is zero)")
|
|
||||||
case len(cfg.Matchers) == 0:
|
|
||||||
log.Print("deadman: disabled (no usable matchers)")
|
|
||||||
default:
|
|
||||||
rendered := make([]string, 0, len(cfg.Matchers))
|
|
||||||
for _, m := range cfg.Matchers {
|
|
||||||
rendered = append(rendered, m.String())
|
|
||||||
}
|
|
||||||
log.Printf("deadman: watching %s, timeout %s, severity %s",
|
|
||||||
strings.Join(rendered, "; "), timeout, severity)
|
|
||||||
}
|
}
|
||||||
return cfg
|
return m, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// deadmanAlert is one switch: the alert row carrying its last heartbeat.
|
// deadmanAlert is one heartbeat: the alert row carrying its last sighting, and
|
||||||
|
// the switch that claimed it.
|
||||||
type deadmanAlert struct {
|
type deadmanAlert struct {
|
||||||
id int64
|
id int64
|
||||||
teamID int64
|
teamID int64
|
||||||
fingerprint string
|
fingerprint string
|
||||||
labels map[string]string
|
labels map[string]string
|
||||||
matcher DeadmanMatcher
|
sw DeadmanSwitch
|
||||||
resolved bool
|
resolved bool
|
||||||
receivedAt int64
|
receivedAt int64
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// dead is the one rule for a silent heartbeat, shared by the sweeper that pages
|
||||||
|
// on it and the status the Switches page shows, so the page cannot disagree
|
||||||
|
// with the pager.
|
||||||
|
//
|
||||||
|
// An explicit resolved from Alertmanager is a stronger death signal than mere
|
||||||
|
// absence: the sender is telling us the heartbeat stopped, so there is nothing
|
||||||
|
// left to wait out.
|
||||||
|
func (a deadmanAlert) dead(now time.Time) bool {
|
||||||
|
return a.resolved || a.receivedAt < now.Add(-a.sw.Timeout).Unix()
|
||||||
|
}
|
||||||
|
|
||||||
// groupKey is the switch's identity as an incident. Per fingerprint, so each
|
// groupKey is the switch's identity as an incident. Per fingerprint, so each
|
||||||
// source is tracked on its own.
|
// source is tracked on its own.
|
||||||
func (a deadmanAlert) groupKey() string { return deadmanGroupPrefix + a.fingerprint }
|
func (a deadmanAlert) groupKey() string { return deadmanGroupPrefix + a.fingerprint }
|
||||||
@@ -189,13 +171,13 @@ func (a deadmanAlert) groupKey() string { return deadmanGroupPrefix + a.fingerpr
|
|||||||
// It returns the ids of the alerts it owns, because the generic staleness
|
// It returns the ids of the alerts it owns, because the generic staleness
|
||||||
// expiry must leave them alone — staleAfter and ends_at would otherwise resolve
|
// expiry must leave them alone — staleAfter and ends_at would otherwise resolve
|
||||||
// a heartbeat long before its own, much tighter, timeout ever fired.
|
// a heartbeat long before its own, much tighter, timeout ever fired.
|
||||||
// Each team is swept against its own configuration: its own matchers, its own
|
// Each team is swept against its own switches, each with its own matcher,
|
||||||
// timeout, its own severity. A team watching nothing is skipped entirely, which
|
// timeout and severity. A team watching nothing is skipped entirely, which is
|
||||||
// is most of them.
|
// most of them.
|
||||||
func sweepDeadman(ctx context.Context, db *sql.DB, notify NotifyConfig) map[int64]bool {
|
func sweepDeadman(ctx context.Context, db *sql.DB, notify NotifyConfig) map[int64]bool {
|
||||||
owned := map[int64]bool{}
|
owned := map[int64]bool{}
|
||||||
|
|
||||||
configs, err := deadmanConfigs(ctx, db)
|
configs, err := deadmanSets(ctx, db)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
log.Printf("deadman: load configs: %v", err)
|
log.Printf("deadman: load configs: %v", err)
|
||||||
return owned
|
return owned
|
||||||
@@ -203,39 +185,35 @@ func sweepDeadman(ctx context.Context, db *sql.DB, notify NotifyConfig) map[int6
|
|||||||
|
|
||||||
now := time.Now()
|
now := time.Now()
|
||||||
for teamID, cfg := range configs {
|
for teamID, cfg := range configs {
|
||||||
switches, err := deadmanAlerts(ctx, db, teamID, cfg)
|
heartbeats, err := deadmanAlerts(ctx, db, teamID, cfg)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
log.Printf("deadman: load switches for team %d: %v", teamID, err)
|
log.Printf("deadman: load heartbeats for team %d: %v", teamID, err)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
cutoff := now.Add(-cfg.Timeout).Unix()
|
|
||||||
|
|
||||||
for _, sw := range switches {
|
for _, hb := range heartbeats {
|
||||||
owned[sw.id] = true
|
owned[hb.id] = true
|
||||||
|
|
||||||
// An explicit resolved from Alertmanager is a stronger death signal
|
if hb.dead(now) {
|
||||||
// than mere absence: the sender is telling us the heartbeat
|
if err := deadmanDied(ctx, db, notify, hb, now); err != nil {
|
||||||
// stopped, so there is nothing left to wait out.
|
log.Printf("deadman: open incident for %s: %v", hb.sw.Matcher.Name, err)
|
||||||
if sw.resolved || sw.receivedAt < cutoff {
|
|
||||||
if err := deadmanDied(ctx, db, cfg, notify, sw, now); err != nil {
|
|
||||||
log.Printf("deadman: open incident for %s: %v", sw.matcher.Name, err)
|
|
||||||
}
|
}
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
if err := deadmanRecovered(ctx, db, sw); err != nil {
|
if err := deadmanRecovered(ctx, db, hb); err != nil {
|
||||||
log.Printf("deadman: resolve incident for %s: %v", sw.matcher.Name, err)
|
log.Printf("deadman: resolve incident for %s: %v", hb.sw.Matcher.Name, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return owned
|
return owned
|
||||||
}
|
}
|
||||||
|
|
||||||
// deadmanAlerts loads every alert row that a matcher claims. The candidate query
|
// deadmanAlerts loads every alert row that one of a team's switches claims. The candidate query
|
||||||
// is narrowed by alertname so it rides alerts_name_idx; the rest of the matching
|
// is narrowed by alertname so it rides alerts_name_idx; the rest of the matching
|
||||||
// happens in Go, which keeps one implementation of the rules. The rows are read
|
// happens in Go, which keeps one implementation of the rules. The rows are read
|
||||||
// in full before the caller writes, so the writes do not run against an open
|
// in full before the caller writes, so the writes do not run against an open
|
||||||
// cursor over the same table.
|
// cursor over the same table.
|
||||||
func deadmanAlerts(ctx context.Context, db *sql.DB, teamID int64, cfg DeadmanConfig) ([]deadmanAlert, error) {
|
func deadmanAlerts(ctx context.Context, db *sql.DB, teamID int64, cfg deadmanSet) ([]deadmanAlert, error) {
|
||||||
names := cfg.names()
|
names := cfg.names()
|
||||||
args := &sqlArgs{}
|
args := &sqlArgs{}
|
||||||
nameList := make([]any, len(names))
|
nameList := make([]any, len(names))
|
||||||
@@ -243,6 +221,10 @@ func deadmanAlerts(ctx context.Context, db *sql.DB, teamID int64, cfg DeadmanCon
|
|||||||
nameList[i] = n
|
nameList[i] = n
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// #nosec G202 -- sqlArgs.add/addList only ever splice in the "$N"
|
||||||
|
// placeholder they hand back, never a value; every value travels through
|
||||||
|
// args.all() as a bound parameter. See the sqlArgs doc comment in
|
||||||
|
// helpers.go.
|
||||||
rows, err := db.QueryContext(ctx, `
|
rows, err := db.QueryContext(ctx, `
|
||||||
SELECT id, team_id, fingerprint, labels, status, received_at
|
SELECT id, team_id, fingerprint, labels, status, received_at
|
||||||
FROM alerts
|
FROM alerts
|
||||||
@@ -263,11 +245,11 @@ func deadmanAlerts(ctx context.Context, db *sql.DB, teamID int64, cfg DeadmanCon
|
|||||||
}
|
}
|
||||||
json.Unmarshal([]byte(labelsJSON), &a.labels) //nolint:errcheck
|
json.Unmarshal([]byte(labelsJSON), &a.labels) //nolint:errcheck
|
||||||
|
|
||||||
m, ok := cfg.match(a.labels)
|
sw, ok := cfg.match(a.labels)
|
||||||
if !ok {
|
if !ok {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
a.matcher = m
|
a.sw = sw
|
||||||
a.resolved = status == "resolved"
|
a.resolved = status == "resolved"
|
||||||
out = append(out, a)
|
out = append(out, a)
|
||||||
}
|
}
|
||||||
@@ -284,16 +266,16 @@ func deadmanAlerts(ctx context.Context, db *sql.DB, teamID int64, cfg DeadmanCon
|
|||||||
// incidentForGroup), and a source that is gone for good is a one-time page
|
// incidentForGroup), and a source that is gone for good is a one-time page
|
||||||
// rather than a nag. Only a heartbeat that comes back and dies again earns a new
|
// rather than a nag. Only a heartbeat that comes back and dies again earns a new
|
||||||
// incident.
|
// incident.
|
||||||
func deadmanDied(ctx context.Context, db *sql.DB, cfg DeadmanConfig, notify NotifyConfig, sw deadmanAlert, now time.Time) error {
|
func deadmanDied(ctx context.Context, db *sql.DB, notify NotifyConfig, hb deadmanAlert, now time.Time) error {
|
||||||
var lastTriggered, open int64
|
var lastTriggered, open int64
|
||||||
if err := db.QueryRowContext(ctx, `
|
if err := db.QueryRowContext(ctx, `
|
||||||
SELECT COALESCE(MAX(triggered_at), 0),
|
SELECT COALESCE(MAX(triggered_at), 0),
|
||||||
COUNT(*) FILTER (WHERE resolved_at IS NULL)
|
COUNT(*) FILTER (WHERE resolved_at IS NULL)
|
||||||
FROM incidents WHERE team_id = $1 AND group_key = $2`,
|
FROM incidents WHERE team_id = $1 AND group_key = $2`,
|
||||||
sw.teamID, sw.groupKey()).Scan(&lastTriggered, &open); err != nil {
|
hb.teamID, hb.groupKey()).Scan(&lastTriggered, &open); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
if open > 0 || sw.receivedAt <= lastTriggered {
|
if open > 0 || hb.receivedAt <= lastTriggered {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -306,18 +288,18 @@ func deadmanDied(ctx context.Context, db *sql.DB, cfg DeadmanConfig, notify Noti
|
|||||||
// A heartbeat nobody has heard from is not firing, and saying otherwise in
|
// A heartbeat nobody has heard from is not firing, and saying otherwise in
|
||||||
// the alert list would be a lie. An Alertmanager-sourced resolution keeps its
|
// the alert list would be a lie. An Alertmanager-sourced resolution keeps its
|
||||||
// own source: it told us the truth first.
|
// own source: it told us the truth first.
|
||||||
if !sw.resolved {
|
if !hb.resolved {
|
||||||
if _, err := tx.ExecContext(ctx, `
|
if _, err := tx.ExecContext(ctx, `
|
||||||
UPDATE alerts
|
UPDATE alerts
|
||||||
SET status = 'resolved',
|
SET status = 'resolved',
|
||||||
resolution_source = $1,
|
resolution_source = $1,
|
||||||
ends_at = COALESCE(ends_at, `+nowEpoch+`)
|
ends_at = COALESCE(ends_at, `+nowEpoch+`)
|
||||||
WHERE id = $2 AND status = 'firing'`, resolutionDeadman, sw.id); err != nil {
|
WHERE id = $2 AND status = 'firing'`, resolutionDeadman, hb.id); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
severity := cfg.Severity
|
severity := hb.sw.Severity
|
||||||
var sev *string
|
var sev *string
|
||||||
if severity != "" {
|
if severity != "" {
|
||||||
sev = &severity
|
sev = &severity
|
||||||
@@ -325,22 +307,22 @@ func deadmanDied(ctx context.Context, db *sql.DB, cfg DeadmanConfig, notify Noti
|
|||||||
|
|
||||||
// The incident opens in the team whose integration received the heartbeat:
|
// The incident opens in the team whose integration received the heartbeat:
|
||||||
// the switch belongs to whoever is watching that source, not to the install.
|
// the switch belongs to whoever is watching that source, not to the install.
|
||||||
incidentID, err := openIncident(ctx, tx, notify, sw.teamID, sw.groupKey(),
|
incidentID, err := openIncident(ctx, tx, notify, hb.teamID, hb.groupKey(),
|
||||||
"No heartbeat from "+sw.matcher.String(), sw.labels, sev)
|
"No heartbeat from "+hb.sw.Matcher.String(), hb.labels, sev)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
alertID := sw.id
|
alertID := hb.id
|
||||||
detail := "last heartbeat " + humanDuration(now.Sub(time.Unix(sw.receivedAt, 0))) + " ago"
|
detail := "last heartbeat " + humanDuration(now.Sub(time.Unix(hb.receivedAt, 0))) + " ago"
|
||||||
if err := logEvent(ctx, tx, incidentID, evDeadmanSilent, nil, &alertID, &detail); err != nil {
|
if err := logEvent(ctx, tx, incidentID, evDeadmanSilent, nil, nil, &alertID, &detail); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
if err := tx.Commit(); err != nil {
|
if err := tx.Commit(); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
log.Printf("deadman: %s went silent, opened incident %d", sw.matcher.String(), incidentID)
|
log.Printf("deadman: %s went silent, opened incident %d", hb.sw.Matcher.String(), incidentID)
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -350,12 +332,12 @@ func deadmanDied(ctx context.Context, db *sql.DB, cfg DeadmanConfig, notify Noti
|
|||||||
// member alerts (linking the heartbeat would have the settled-incident cascade
|
// member alerts (linking the heartbeat would have the settled-incident cascade
|
||||||
// close it on the very same sweep that opened it), so the alert-driven cascade
|
// close it on the very same sweep that opened it), so the alert-driven cascade
|
||||||
// ignores it entirely and recovery is the only automatic way out.
|
// ignores it entirely and recovery is the only automatic way out.
|
||||||
func deadmanRecovered(ctx context.Context, db *sql.DB, sw deadmanAlert) error {
|
func deadmanRecovered(ctx context.Context, db *sql.DB, hb deadmanAlert) error {
|
||||||
var incidentID int64
|
var incidentID int64
|
||||||
switch err := db.QueryRowContext(ctx, `
|
switch err := db.QueryRowContext(ctx, `
|
||||||
SELECT id FROM incidents
|
SELECT id FROM incidents
|
||||||
WHERE team_id = $1 AND group_key = $2 AND resolved_at IS NULL`,
|
WHERE team_id = $1 AND group_key = $2 AND resolved_at IS NULL`,
|
||||||
sw.teamID, sw.groupKey()).Scan(&incidentID); {
|
hb.teamID, hb.groupKey()).Scan(&incidentID); {
|
||||||
case err == sql.ErrNoRows:
|
case err == sql.ErrNoRows:
|
||||||
return nil
|
return nil
|
||||||
case err != nil:
|
case err != nil:
|
||||||
@@ -375,7 +357,7 @@ func deadmanRecovered(ctx context.Context, db *sql.DB, sw deadmanAlert) error {
|
|||||||
time.Now().Unix(), incidentResolutionRecovered, incidentID); err != nil {
|
time.Now().Unix(), incidentResolutionRecovered, incidentID); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
if err := logEvent(ctx, tx, incidentID, evResolved, nil, nil, nil); err != nil {
|
if err := logEvent(ctx, tx, incidentID, evResolved, nil, nil, nil, nil); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
// The all-clear goes to whoever was paged, which enqueueResolved works out
|
// The all-clear goes to whoever was paged, which enqueueResolved works out
|
||||||
@@ -387,116 +369,206 @@ func deadmanRecovered(ctx context.Context, db *sql.DB, sw deadmanAlert) error {
|
|||||||
if err := tx.Commit(); err != nil {
|
if err := tx.Commit(); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
log.Printf("deadman: %s is back, resolved incident %d", sw.matcher.String(), incidentID)
|
log.Printf("deadman: %s is back, resolved incident %d", hb.sw.Matcher.String(), incidentID)
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Per-team configuration
|
// A team's switches
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
// deadmanConfigForTeam reads one team's switches. A team with no row, or with
|
const deadmanSwitchColumns = "id, team_id, name, matcher, timeout_seconds, severity"
|
||||||
// nothing configured, gets a disabled config — which is the right answer rather
|
|
||||||
// than an error: most teams watch no heartbeat at all.
|
|
||||||
func deadmanConfigForTeam(ctx context.Context, q querier, teamID int64) (DeadmanConfig, error) {
|
|
||||||
var matchers, severity string
|
|
||||||
var timeout int64
|
|
||||||
err := q.QueryRowContext(ctx,
|
|
||||||
"SELECT matchers, timeout_seconds, severity FROM deadman_configs WHERE team_id = $1",
|
|
||||||
teamID).Scan(&matchers, &timeout, &severity)
|
|
||||||
if err == sql.ErrNoRows {
|
|
||||||
return DeadmanConfig{}, nil
|
|
||||||
}
|
|
||||||
if err != nil {
|
|
||||||
return DeadmanConfig{}, err
|
|
||||||
}
|
|
||||||
return parseDeadmanQuietly(matchers, time.Duration(timeout)*time.Second, severity), nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// deadmanConfigs reads every team's switches in one query, for the sweeper.
|
// scanDeadmanSwitches reads switch rows into per-team sets. A row whose matcher
|
||||||
func deadmanConfigs(ctx context.Context, db *sql.DB) (map[int64]DeadmanConfig, error) {
|
// no longer parses is skipped rather than fatal: the API refuses to store one,
|
||||||
rows, err := db.QueryContext(ctx,
|
// so it can only mean a hand edit, and one bad row must not stop the others
|
||||||
"SELECT team_id, matchers, timeout_seconds, severity FROM deadman_configs")
|
// from being watched.
|
||||||
if err != nil {
|
func scanDeadmanSwitches(rows *sql.Rows) (map[int64]deadmanSet, error) {
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
|
out := map[int64]deadmanSet{}
|
||||||
out := map[int64]DeadmanConfig{}
|
|
||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
|
var sw DeadmanSwitch
|
||||||
var teamID, timeout int64
|
var teamID, timeout int64
|
||||||
var matchers, severity string
|
var matcher string
|
||||||
if err := rows.Scan(&teamID, &matchers, &timeout, &severity); err != nil {
|
if err := rows.Scan(&sw.ID, &teamID, &sw.Name, &matcher, &timeout, &sw.Severity); err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
cfg := parseDeadmanQuietly(matchers, time.Duration(timeout)*time.Second, severity)
|
m, err := parseDeadmanMatcher(matcher)
|
||||||
if cfg.enabled() {
|
if err != nil {
|
||||||
out[teamID] = cfg
|
log.Printf("deadman: switch %d has an unusable matcher %q: %v", sw.ID, matcher, err)
|
||||||
|
continue
|
||||||
}
|
}
|
||||||
|
sw.Matcher = m
|
||||||
|
sw.Timeout = time.Duration(timeout) * time.Second
|
||||||
|
out[teamID] = append(out[teamID], sw)
|
||||||
}
|
}
|
||||||
return out, rows.Err()
|
return out, rows.Err()
|
||||||
}
|
}
|
||||||
|
|
||||||
// SeedDeadmanConfigs gives every team without a row the server's environment
|
// deadmanSetForTeam reads one team's switches. A team with none gets an empty
|
||||||
// configuration, so the install that upgrades into per-team switches keeps
|
// set — which is the right answer rather than an error: most teams watch no
|
||||||
// watching exactly what it was watching before.
|
// heartbeat at all.
|
||||||
//
|
func deadmanSetForTeam(ctx context.Context, q querier, teamID int64) (deadmanSet, error) {
|
||||||
// Idempotent, and never overwrites: once a team has a row it owns its own
|
rows, err := q.QueryContext(ctx,
|
||||||
// configuration, and a redeploy must not quietly put the environment's value
|
"SELECT "+deadmanSwitchColumns+" FROM deadman_switches WHERE team_id = $1 ORDER BY id", teamID)
|
||||||
// back over an owner's edit.
|
if err != nil {
|
||||||
//
|
return nil, err
|
||||||
// A team created after startup gets no row and therefore watches nothing until
|
|
||||||
// its owner says otherwise. That is deliberate: inheriting an install-wide
|
|
||||||
// heartbeat would page a new team about a source it has never heard of, and a
|
|
||||||
// switch nobody chose is the kind that gets muted rather than fixed.
|
|
||||||
func SeedDeadmanConfigs(ctx context.Context, db *sql.DB, cfg DeadmanConfig) error {
|
|
||||||
matchers := make([]string, 0, len(cfg.Matchers))
|
|
||||||
for _, m := range cfg.Matchers {
|
|
||||||
parts := []string{"alertname=" + m.Name}
|
|
||||||
for k, v := range m.Labels {
|
|
||||||
parts = append(parts, k+"="+v)
|
|
||||||
}
|
|
||||||
sort.Strings(parts[1:])
|
|
||||||
matchers = append(matchers, strings.Join(parts, ","))
|
|
||||||
}
|
}
|
||||||
|
sets, err := scanDeadmanSwitches(rows)
|
||||||
_, err := db.ExecContext(ctx, `
|
return sets[teamID], err
|
||||||
INSERT INTO deadman_configs (team_id, matchers, timeout_seconds, severity)
|
|
||||||
SELECT id, $1, $2, $3 FROM teams
|
|
||||||
ON CONFLICT (team_id) DO NOTHING`,
|
|
||||||
strings.Join(matchers, "; "), int64(cfg.Timeout.Seconds()), cfg.Severity)
|
|
||||||
return err
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// parseDeadmanQuietly is ParseDeadmanConfig without the startup logging: a
|
// deadmanSets reads every team's switches in one query, for the sweeper.
|
||||||
// team's configuration is read on every sweep and every webhook, and logging it
|
func deadmanSets(ctx context.Context, db *sql.DB) (map[int64]deadmanSet, error) {
|
||||||
// each time would bury everything else.
|
rows, err := db.QueryContext(ctx,
|
||||||
func parseDeadmanQuietly(matchers string, timeout time.Duration, severity string) DeadmanConfig {
|
"SELECT "+deadmanSwitchColumns+" FROM deadman_switches ORDER BY id")
|
||||||
cfg := DeadmanConfig{Timeout: timeout, Severity: severity}
|
if err != nil {
|
||||||
for _, entry := range strings.Split(matchers, ";") {
|
return nil, err
|
||||||
entry = strings.TrimSpace(entry)
|
|
||||||
if entry == "" {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
m := DeadmanMatcher{Labels: map[string]string{}}
|
|
||||||
malformed := false
|
|
||||||
for _, cond := range strings.Split(entry, ",") {
|
|
||||||
k, v, ok := strings.Cut(cond, "=")
|
|
||||||
k, v = strings.TrimSpace(k), strings.TrimSpace(v)
|
|
||||||
if !ok || k == "" || v == "" {
|
|
||||||
malformed = true
|
|
||||||
break
|
|
||||||
}
|
|
||||||
if k == "alertname" {
|
|
||||||
m.Name = v
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
m.Labels[k] = v
|
|
||||||
}
|
|
||||||
if malformed || m.Name == "" {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
cfg.Matchers = append(cfg.Matchers, m)
|
|
||||||
}
|
}
|
||||||
return cfg
|
return scanDeadmanSwitches(rows)
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Status
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
const (
|
||||||
|
switchHealthy = "healthy"
|
||||||
|
switchDead = "dead"
|
||||||
|
switchDormant = "dormant"
|
||||||
|
)
|
||||||
|
|
||||||
|
// deadmanSource is one heartbeat under a switch: a fingerprint that matched.
|
||||||
|
type deadmanSource struct {
|
||||||
|
Fingerprint string `json:"fingerprint"`
|
||||||
|
Labels map[string]string `json:"labels"`
|
||||||
|
Status string `json:"status"`
|
||||||
|
LastHeartbeatAt time.Time `json:"last_heartbeat_at"`
|
||||||
|
LastTriggeredAt *time.Time `json:"last_triggered_at"`
|
||||||
|
IncidentID *int64 `json:"incident_id"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// deadmanSwitchStatus is a switch as the Switches page shows it.
|
||||||
|
type deadmanSwitchStatus struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
Name string `json:"name"`
|
||||||
|
Matcher string `json:"matcher"`
|
||||||
|
TimeoutSeconds int64 `json:"timeout_seconds"`
|
||||||
|
Severity string `json:"severity"`
|
||||||
|
|
||||||
|
// Status is dead when any source is, dormant when none has ever been heard
|
||||||
|
// from, healthy otherwise — a live cluster must not hide a dead one.
|
||||||
|
Status string `json:"status"`
|
||||||
|
LastHeartbeatAt *time.Time `json:"last_heartbeat_at"`
|
||||||
|
LastTriggeredAt *time.Time `json:"last_triggered_at"`
|
||||||
|
OpenIncidentID *int64 `json:"open_incident_id"`
|
||||||
|
Sources []deadmanSource `json:"sources"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// deadmanStatuses reports every switch of a team with what its heartbeats are
|
||||||
|
// doing. The liveness verdict is deadmanAlert.dead, the sweeper's own.
|
||||||
|
func deadmanStatuses(ctx context.Context, db *sql.DB, teamID int64, set deadmanSet, now time.Time) ([]deadmanSwitchStatus, error) {
|
||||||
|
out := make([]deadmanSwitchStatus, 0, len(set))
|
||||||
|
if len(set) == 0 {
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
heartbeats, err := deadmanAlerts(ctx, db, teamID, set)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
// One query for every switch's incident history, keyed the way the sweeper
|
||||||
|
// keys it.
|
||||||
|
type history struct {
|
||||||
|
triggeredAt int64
|
||||||
|
openID int64
|
||||||
|
}
|
||||||
|
incidents := map[string]history{}
|
||||||
|
rows, err := db.QueryContext(ctx, `
|
||||||
|
SELECT group_key, MAX(triggered_at), COALESCE(MAX(id) FILTER (WHERE resolved_at IS NULL), 0)
|
||||||
|
FROM incidents
|
||||||
|
WHERE team_id = $1 AND group_key LIKE $2
|
||||||
|
GROUP BY group_key`, teamID, deadmanGroupPrefix+"%")
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
defer rows.Close()
|
||||||
|
for rows.Next() {
|
||||||
|
var key string
|
||||||
|
var h history
|
||||||
|
if err := rows.Scan(&key, &h.triggeredAt, &h.openID); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
incidents[key] = h
|
||||||
|
}
|
||||||
|
if err := rows.Err(); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
bySwitch := map[int64][]deadmanAlert{}
|
||||||
|
for _, hb := range heartbeats {
|
||||||
|
bySwitch[hb.sw.ID] = append(bySwitch[hb.sw.ID], hb)
|
||||||
|
}
|
||||||
|
|
||||||
|
later := func(cur *time.Time, unix int64) *time.Time {
|
||||||
|
t := time.Unix(unix, 0).UTC()
|
||||||
|
if cur == nil || t.After(*cur) {
|
||||||
|
return &t
|
||||||
|
}
|
||||||
|
return cur
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, sw := range set {
|
||||||
|
st := deadmanSwitchStatus{
|
||||||
|
ID: sw.ID, Name: sw.Name, Matcher: sw.Matcher.config(),
|
||||||
|
TimeoutSeconds: int64(sw.Timeout.Seconds()), Severity: sw.Severity,
|
||||||
|
Status: switchDormant, Sources: []deadmanSource{},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, hb := range bySwitch[sw.ID] {
|
||||||
|
src := deadmanSource{
|
||||||
|
Fingerprint: hb.fingerprint,
|
||||||
|
Labels: hb.labels,
|
||||||
|
Status: switchHealthy,
|
||||||
|
LastHeartbeatAt: time.Unix(hb.receivedAt, 0).UTC(),
|
||||||
|
}
|
||||||
|
if hb.dead(now) {
|
||||||
|
src.Status = switchDead
|
||||||
|
}
|
||||||
|
if h, ok := incidents[hb.groupKey()]; ok {
|
||||||
|
t := time.Unix(h.triggeredAt, 0).UTC()
|
||||||
|
src.LastTriggeredAt = &t
|
||||||
|
st.LastTriggeredAt = later(st.LastTriggeredAt, h.triggeredAt)
|
||||||
|
if h.openID != 0 {
|
||||||
|
id := h.openID
|
||||||
|
src.IncidentID = &id
|
||||||
|
if st.OpenIncidentID == nil || id > *st.OpenIncidentID {
|
||||||
|
st.OpenIncidentID = &id
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
st.LastHeartbeatAt = later(st.LastHeartbeatAt, hb.receivedAt)
|
||||||
|
st.Sources = append(st.Sources, src)
|
||||||
|
|
||||||
|
switch {
|
||||||
|
case src.Status == switchDead:
|
||||||
|
st.Status = switchDead
|
||||||
|
case st.Status == switchDormant:
|
||||||
|
st.Status = switchHealthy
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Dead ones first, then by fingerprint: what needs attention leads, and
|
||||||
|
// the order does not shuffle between refreshes.
|
||||||
|
sort.Slice(st.Sources, func(i, j int) bool {
|
||||||
|
a, b := st.Sources[i], st.Sources[j]
|
||||||
|
if (a.Status == switchDead) != (b.Status == switchDead) {
|
||||||
|
return a.Status == switchDead
|
||||||
|
}
|
||||||
|
return a.Fingerprint < b.Fingerprint
|
||||||
|
})
|
||||||
|
out = append(out, st)
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -121,8 +121,11 @@ func TestDeadman_MixedGroupExcludesHeartbeat(t *testing.T) {
|
|||||||
t.Fatalf("expected 1 incident for the real alert, got %d", got)
|
t.Fatalf("expected 1 incident for the real alert, got %d", got)
|
||||||
}
|
}
|
||||||
|
|
||||||
var alerts []map[string]any
|
var incident struct {
|
||||||
decode(t, s.req(t, http.MethodGet, "/api/incidents/1/alerts", nil), &alerts)
|
Alerts []map[string]any `json:"alerts"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/incidents/1", nil), &incident)
|
||||||
|
alerts := incident.Alerts
|
||||||
if len(alerts) != 1 {
|
if len(alerts) != 1 {
|
||||||
t.Fatalf("expected 1 member alert, got %d", len(alerts))
|
t.Fatalf("expected 1 member alert, got %d", len(alerts))
|
||||||
}
|
}
|
||||||
@@ -483,13 +486,13 @@ func TestDeadman_ConfigurationIsPerTeam(t *testing.T) {
|
|||||||
unwatched := newTeam(t, s, "unwatched")
|
unwatched := newTeam(t, s, "unwatched")
|
||||||
|
|
||||||
// Only the first team calls Watchdog a heartbeat.
|
// Only the first team calls Watchdog a heartbeat.
|
||||||
resp := s.req(t, http.MethodPut, "/api/teams/"+id64(watched.id)+"/deadman", map[string]any{
|
resp := s.req(t, http.MethodPost, "/api/teams/"+id64(watched.id)+"/deadman/switches", map[string]any{
|
||||||
"matchers": "alertname=Watchdog",
|
"matcher": "alertname=Watchdog",
|
||||||
"timeout_seconds": 3600,
|
"timeout_seconds": 3600,
|
||||||
"severity": "critical",
|
"severity": "critical",
|
||||||
})
|
})
|
||||||
resp.Body.Close()
|
resp.Body.Close()
|
||||||
if resp.StatusCode != http.StatusOK {
|
if resp.StatusCode != http.StatusCreated {
|
||||||
t.Fatalf("configure the watched team: %d", resp.StatusCode)
|
t.Fatalf("configure the watched team: %d", resp.StatusCode)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -550,9 +553,9 @@ func TestDeadman_ConfigurationIsOwnerOnly(t *testing.T) {
|
|||||||
decode(t, s.req(t, http.MethodPost, "/api/users/"+id64(user.ID)+"/api-keys",
|
decode(t, s.req(t, http.MethodPost, "/api/users/"+id64(user.ID)+"/api-keys",
|
||||||
map[string]string{"name": "test"}), &key)
|
map[string]string{"name": "test"}), &key)
|
||||||
|
|
||||||
req, _ := http.NewRequest(http.MethodPut,
|
req, _ := http.NewRequest(http.MethodPost,
|
||||||
s.URL+"/api/teams/"+id64(team.id)+"/deadman",
|
s.URL+"/api/teams/"+id64(team.id)+"/deadman/switches",
|
||||||
strings.NewReader(`{"matchers":"alertname=Watchdog","timeout_seconds":60}`))
|
strings.NewReader(`{"matcher":"alertname=Watchdog","timeout_seconds":60}`))
|
||||||
req.Header.Set("Authorization", "Bearer "+key.Key)
|
req.Header.Set("Authorization", "Bearer "+key.Key)
|
||||||
req.Header.Set("Content-Type", "application/json")
|
req.Header.Set("Content-Type", "application/json")
|
||||||
resp, err := http.DefaultClient.Do(req)
|
resp, err := http.DefaultClient.Do(req)
|
||||||
@@ -564,7 +567,7 @@ func TestDeadman_ConfigurationIsOwnerOnly(t *testing.T) {
|
|||||||
t.Errorf("a member editing the switches: expected 403, got %d", resp.StatusCode)
|
t.Errorf("a member editing the switches: expected 403, got %d", resp.StatusCode)
|
||||||
}
|
}
|
||||||
|
|
||||||
read, _ := http.NewRequest(http.MethodGet, s.URL+"/api/teams/"+id64(team.id)+"/deadman", nil)
|
read, _ := http.NewRequest(http.MethodGet, s.URL+"/api/teams/"+id64(team.id)+"/deadman/switches", nil)
|
||||||
read.Header.Set("Authorization", "Bearer "+key.Key)
|
read.Header.Set("Authorization", "Bearer "+key.Key)
|
||||||
got, err := http.DefaultClient.Do(read)
|
got, err := http.DefaultClient.Do(read)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -577,16 +580,142 @@ func TestDeadman_ConfigurationIsOwnerOnly(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// A matcher with no alertname watches nothing, silently, which is the failure
|
// A matcher with no alertname watches nothing, silently, which is the failure
|
||||||
// this feature exists to prevent — so it is refused at the door.
|
// this feature exists to prevent — so it is refused at the door, along with the
|
||||||
func TestDeadman_UnusableMatchersAreRejected(t *testing.T) {
|
// other things that would make a switch unable to fire.
|
||||||
|
func TestDeadman_UnusableSwitchesAreRejected(t *testing.T) {
|
||||||
s, _ := deadmanTS(t, deadmanCfg())
|
s, _ := deadmanTS(t, deadmanCfg())
|
||||||
|
|
||||||
resp := s.req(t, http.MethodPut, "/api/teams/"+defaultTeam+"/deadman", map[string]any{
|
for name, body := range map[string]map[string]any{
|
||||||
"matchers": "cluster=prod",
|
"no alertname": {"matcher": "cluster=prod", "timeout_seconds": 900},
|
||||||
"timeout_seconds": 900,
|
"malformed": {"matcher": "alertname=Watchdog,garbage", "timeout_seconds": 900},
|
||||||
})
|
"several": {"matcher": "alertname=A; alertname=B", "timeout_seconds": 900},
|
||||||
resp.Body.Close()
|
"zero timeout": {"matcher": "alertname=Watchdog", "timeout_seconds": 0},
|
||||||
if resp.StatusCode != http.StatusBadRequest {
|
"bad severity": {"matcher": "alertname=Watchdog", "timeout_seconds": 900, "severity": "loud"},
|
||||||
t.Errorf("expected 400 for a matcher with no alertname, got %d", resp.StatusCode)
|
"empty matcher": {"matcher": "", "timeout_seconds": 900},
|
||||||
|
} {
|
||||||
|
resp := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches", body)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusBadRequest {
|
||||||
|
t.Errorf("%s: expected 400, got %d", name, resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// The switch list
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
// listSwitches reads the default team's switches as the Switches page does.
|
||||||
|
func listSwitches(t *testing.T, s *ts) []map[string]any {
|
||||||
|
t.Helper()
|
||||||
|
return list(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/deadman/switches", nil))
|
||||||
|
}
|
||||||
|
|
||||||
|
// A switch is healthy while its heartbeat is fresh, dead once it is silent, and
|
||||||
|
// dormant until the first one arrives.
|
||||||
|
func TestDeadman_ListReportsStatus(t *testing.T) {
|
||||||
|
s, _ := deadmanTS(t, api.ParseDeadmanConfig("alertname=Watchdog; alertname=NeverSent", time.Hour, "critical"))
|
||||||
|
|
||||||
|
got := listSwitches(t, s)
|
||||||
|
if len(got) != 2 {
|
||||||
|
t.Fatalf("expected 2 switches, got %d", len(got))
|
||||||
|
}
|
||||||
|
for _, sw := range got {
|
||||||
|
if sw["status"] != "dormant" || sw["last_heartbeat_at"] != nil || sw["last_triggered_at"] != nil {
|
||||||
|
t.Errorf("a switch nobody has heard from should be dormant and blank, got %v", sw)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
heartbeat(t, s, "fp-watchdog", nil)
|
||||||
|
got = listSwitches(t, s)
|
||||||
|
if got[0]["status"] != "healthy" || got[0]["last_heartbeat_at"] == nil {
|
||||||
|
t.Errorf("a fresh heartbeat should be healthy with a timestamp, got %v", got[0])
|
||||||
|
}
|
||||||
|
if got[1]["status"] != "dormant" {
|
||||||
|
t.Errorf("the other switch is still dormant, got %v", got[1]["status"])
|
||||||
|
}
|
||||||
|
|
||||||
|
silence(t, s, "fp-watchdog", 2*time.Hour)
|
||||||
|
sweep(t, s, noArchive)
|
||||||
|
got = listSwitches(t, s)
|
||||||
|
if got[0]["status"] != "dead" {
|
||||||
|
t.Fatalf("a silent heartbeat should be dead, got %v", got[0]["status"])
|
||||||
|
}
|
||||||
|
if got[0]["last_triggered_at"] == nil || got[0]["open_incident_id"] == nil {
|
||||||
|
t.Errorf("a dead switch should show when it triggered and its open incident, got %v", got[0])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// One matcher, several clusters: the switch is as bad as its worst heartbeat and
|
||||||
|
// each heartbeat is listed on its own.
|
||||||
|
func TestDeadman_ListBreaksDownByFingerprint(t *testing.T) {
|
||||||
|
s, _ := deadmanTS(t, deadmanCfg())
|
||||||
|
|
||||||
|
heartbeat(t, s, "fp-a", map[string]string{"cluster": "a"})
|
||||||
|
heartbeat(t, s, "fp-b", map[string]string{"cluster": "b"})
|
||||||
|
silence(t, s, "fp-b", 2*time.Hour)
|
||||||
|
|
||||||
|
sw := listSwitches(t, s)[0]
|
||||||
|
if sw["status"] != "dead" {
|
||||||
|
t.Errorf("one dead cluster makes the switch dead, got %v", sw["status"])
|
||||||
|
}
|
||||||
|
sources := sw["sources"].([]any)
|
||||||
|
if len(sources) != 2 {
|
||||||
|
t.Fatalf("expected 2 sources, got %d", len(sources))
|
||||||
|
}
|
||||||
|
first, second := sources[0].(map[string]any), sources[1].(map[string]any)
|
||||||
|
if first["fingerprint"] != "fp-b" || first["status"] != "dead" || second["status"] != "healthy" {
|
||||||
|
t.Errorf("the dead source should lead, got %v then %v", first, second)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Every switch keeps its own deadline.
|
||||||
|
func TestDeadman_TimeoutsArePerSwitch(t *testing.T) {
|
||||||
|
s, _ := deadmanTS(t, deadmanCfg())
|
||||||
|
resp := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches", map[string]any{
|
||||||
|
"matcher": "alertname=Edge", "timeout_seconds": 300,
|
||||||
|
})
|
||||||
|
resp.Body.Close()
|
||||||
|
|
||||||
|
heartbeat(t, s, "fp-watchdog", nil)
|
||||||
|
postWebhook(t, s, []map[string]any{
|
||||||
|
amAlert("fp-edge", "Edge", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||||
|
}, `{}:{alertname="Edge"}`)
|
||||||
|
|
||||||
|
// Ten minutes of silence: past the Edge switch's five, inside Watchdog's hour.
|
||||||
|
silence(t, s, "fp-watchdog", 10*time.Minute)
|
||||||
|
silence(t, s, "fp-edge", 10*time.Minute)
|
||||||
|
|
||||||
|
got := listSwitches(t, s)
|
||||||
|
if got[0]["status"] != "healthy" || got[1]["status"] != "dead" {
|
||||||
|
t.Errorf("want Watchdog healthy and Edge dead, got %v and %v", got[0]["status"], got[1]["status"])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Deleting is an owner's, is scoped to the team, and leaves what the switch
|
||||||
|
// already opened alone.
|
||||||
|
func TestDeadman_DeleteIsScopedToTheTeam(t *testing.T) {
|
||||||
|
s, _ := deadmanTS(t, deadmanCfg())
|
||||||
|
other := newTeam(t, s, "other")
|
||||||
|
|
||||||
|
id := int64(listSwitches(t, s)[0]["id"].(float64))
|
||||||
|
|
||||||
|
// Another team's owner cannot reach it.
|
||||||
|
resp := other.call(http.MethodDelete, "/api/teams/"+id64(other.id)+"/deadman/switches/"+id64(id), nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNotFound {
|
||||||
|
t.Errorf("deleting another team's switch: expected 404, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
if got := len(listSwitches(t, s)); got != 1 {
|
||||||
|
t.Fatalf("the switch should have survived, %d left", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
resp = s.req(t, http.MethodDelete, "/api/teams/"+defaultTeam+"/deadman/switches/"+id64(id), nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNoContent {
|
||||||
|
t.Fatalf("deleting: expected 204, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
if got := len(listSwitches(t, s)); got != 0 {
|
||||||
|
t.Errorf("expected no switches, got %d", got)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,282 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"crypto/rand"
|
||||||
|
"database/sql"
|
||||||
|
"errors"
|
||||||
|
"log"
|
||||||
|
"math/big"
|
||||||
|
"net/http"
|
||||||
|
"net/url"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// The device login flow lets a client that cannot open a browser sign in: it
|
||||||
|
// shows a code, the person approves it in a browser they are signed in to, and
|
||||||
|
// the client is handed an ordinary session. See migration 012.
|
||||||
|
|
||||||
|
const (
|
||||||
|
// deviceTTL is how long a person has to get from the terminal's prompt to an
|
||||||
|
// approval.
|
||||||
|
deviceTTL = 10 * time.Minute
|
||||||
|
|
||||||
|
// deviceInterval is how often the client is told to poll. The server holds it
|
||||||
|
// to that, with a second of slack for clocks and scheduling.
|
||||||
|
deviceInterval = 5 * time.Second
|
||||||
|
|
||||||
|
// deviceStartMaxPerAddr bounds unauthenticated device logins started per
|
||||||
|
// address, since each writes a row.
|
||||||
|
deviceStartMaxPerAddr = 30
|
||||||
|
|
||||||
|
// userCodeAlphabet has no vowels, so a code cannot spell a word, and none of
|
||||||
|
// the characters that read alike (0/O, 1/I/L).
|
||||||
|
userCodeAlphabet = "BCDFGHJKMNPQRSTVWXZ23456789"
|
||||||
|
userCodeLen = 8
|
||||||
|
)
|
||||||
|
|
||||||
|
// newUserCode returns a code for a person to read, as XXXX-XXXX.
|
||||||
|
func newUserCode() (string, error) {
|
||||||
|
max := big.NewInt(int64(len(userCodeAlphabet)))
|
||||||
|
b := make([]byte, userCodeLen)
|
||||||
|
for i := range b {
|
||||||
|
n, err := rand.Int(rand.Reader, max)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
b[i] = userCodeAlphabet[n.Int64()]
|
||||||
|
}
|
||||||
|
return string(b[:4]) + "-" + string(b[4:]), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// normalizeUserCode reduces whatever a person typed or pasted to the stored
|
||||||
|
// form, so "bcdf ghjk" and "BCDF-GHJK" name the same login. It returns "" for
|
||||||
|
// anything that cannot be a code.
|
||||||
|
func normalizeUserCode(s string) string {
|
||||||
|
var b strings.Builder
|
||||||
|
for _, r := range strings.ToUpper(s) {
|
||||||
|
if strings.ContainsRune(userCodeAlphabet, r) {
|
||||||
|
b.WriteRune(r)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
code := b.String()
|
||||||
|
if len(code) != userCodeLen {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
return code[:4] + "-" + code[4:]
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleDeviceStart begins a device login: it returns the device code the
|
||||||
|
// client polls with, and the user code and URL the person is shown.
|
||||||
|
func handleDeviceStart(db *sql.DB, limiter *loginLimiter, publicURL string) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
addrKey := "device:" + clientAddr(r)
|
||||||
|
if limiter.blocked(r.Context(), addrKey, deviceStartMaxPerAddr) {
|
||||||
|
w.Header().Set("Retry-After", strconv.Itoa(int(loginWindow.Seconds())))
|
||||||
|
respond(w, http.StatusTooManyRequests, errResp("too many sign-in attempts, try again later"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
limiter.fail(r.Context(), addrKey)
|
||||||
|
|
||||||
|
deviceCode, deviceHash, err := randomToken()
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
now := time.Now()
|
||||||
|
db.ExecContext(r.Context(), "DELETE FROM device_logins WHERE expires_at < $1", now.Unix())
|
||||||
|
|
||||||
|
// A collision on the user code is one in 27^8; retrying a few times makes
|
||||||
|
// it a non-event rather than a 500.
|
||||||
|
var userCode string
|
||||||
|
for range 5 {
|
||||||
|
userCode, err = newUserCode()
|
||||||
|
if err != nil {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
_, err = db.ExecContext(r.Context(), `
|
||||||
|
INSERT INTO device_logins (device_hash, user_code, expires_at) VALUES ($1, $2, $3)`,
|
||||||
|
deviceHash, userCode, now.Add(deviceTTL).Unix())
|
||||||
|
if err == nil || !isUniqueViolation(err) {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("device login: start: %v", err)
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
respond(w, http.StatusOK, map[string]any{
|
||||||
|
"device_code": deviceCode,
|
||||||
|
"user_code": userCode,
|
||||||
|
// The code is in the URL so nobody has to type it; it is shown anyway,
|
||||||
|
// for the person to check against the terminal before approving.
|
||||||
|
"verification_url": strings.TrimRight(publicURL, "/") + "/device?code=" + url.QueryEscape(userCode),
|
||||||
|
"interval": int(deviceInterval.Seconds()),
|
||||||
|
"expires_in": int(deviceTTL.Seconds()),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleDeviceDecision approves or denies a pending device login on behalf of
|
||||||
|
// the signed-in caller.
|
||||||
|
//
|
||||||
|
// It takes a session, not an API key. Approving hands a terminal the caller's
|
||||||
|
// identity, and the approval must come from a browser the person is looking at:
|
||||||
|
// the page shows the code and asks. A script with a key has no business
|
||||||
|
// approving one, and the check keeps it from being a way to mint sessions out of
|
||||||
|
// keys.
|
||||||
|
func handleDeviceDecision(db *sql.DB, approve bool) http.HandlerFunc {
|
||||||
|
status := "denied"
|
||||||
|
if approve {
|
||||||
|
status = "approved"
|
||||||
|
}
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
if _, viaSession := sessionFromContext(r.Context()); !viaSession {
|
||||||
|
respond(w, http.StatusForbidden, errResp("sign in with the web UI to approve a device"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
var req struct {
|
||||||
|
UserCode string `json:"user_code"`
|
||||||
|
}
|
||||||
|
if err := decodeJSON(r, &req); err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
code := normalizeUserCode(req.UserCode)
|
||||||
|
if code == "" {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("that is not a sign-in code"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
caller, _ := userFromContext(r.Context())
|
||||||
|
// Only a pending login can be decided, and only once: an approval cannot
|
||||||
|
// be overwritten, so a second browser cannot take a login over.
|
||||||
|
res, err := db.ExecContext(r.Context(), `
|
||||||
|
UPDATE device_logins SET status = $1, user_id = $2
|
||||||
|
WHERE user_code = $3 AND status = 'pending' AND expires_at > $4`,
|
||||||
|
status, caller.ID, code, time.Now().Unix())
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
|
respond(w, http.StatusNotFound, errResp("that sign-in code is unknown, expired or already used"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
w.WriteHeader(http.StatusNoContent)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleDeviceToken is what the client polls. Pending answers 202; an approval
|
||||||
|
// answers 200 with the session cookie, once; anything else is 410.
|
||||||
|
func handleDeviceToken(db *sql.DB, ssoMaxAge time.Duration, publicURL string) http.HandlerFunc {
|
||||||
|
gone := func(w http.ResponseWriter, why string) {
|
||||||
|
respond(w, http.StatusGone, map[string]string{"error": why})
|
||||||
|
}
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
var req struct {
|
||||||
|
DeviceCode string `json:"device_code"`
|
||||||
|
}
|
||||||
|
if err := decodeJSON(r, &req); err != nil || req.DeviceCode == "" {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("device_code is required"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
hash := hashToken(req.DeviceCode)
|
||||||
|
now := time.Now()
|
||||||
|
|
||||||
|
tx, err := db.BeginTx(r.Context(), nil)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer tx.Rollback() //nolint:errcheck
|
||||||
|
|
||||||
|
var status string
|
||||||
|
var userID sql.NullInt64
|
||||||
|
var expires, lastPolled int64
|
||||||
|
err = tx.QueryRowContext(r.Context(), `
|
||||||
|
SELECT status, user_id, expires_at, last_polled_at FROM device_logins
|
||||||
|
WHERE device_hash = $1 FOR UPDATE`, hash).Scan(&status, &userID, &expires, &lastPolled)
|
||||||
|
if errors.Is(err, sql.ErrNoRows) || (err == nil && expires <= now.Unix()) {
|
||||||
|
gone(w, "expired")
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
switch status {
|
||||||
|
case "denied":
|
||||||
|
tx.ExecContext(r.Context(), "DELETE FROM device_logins WHERE device_hash = $1", hash)
|
||||||
|
tx.Commit() //nolint:errcheck
|
||||||
|
gone(w, "denied")
|
||||||
|
return
|
||||||
|
|
||||||
|
case "pending":
|
||||||
|
// Held to the interval it was given, less a second of slack.
|
||||||
|
if now.Unix()-lastPolled < int64(deviceInterval.Seconds())-1 {
|
||||||
|
w.Header().Set("Retry-After", strconv.Itoa(int(deviceInterval.Seconds())))
|
||||||
|
respond(w, http.StatusTooManyRequests, map[string]string{"error": "slow_down"})
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if _, err := tx.ExecContext(r.Context(),
|
||||||
|
"UPDATE device_logins SET last_polled_at = $1 WHERE device_hash = $2", now.Unix(), hash); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err := tx.Commit(); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusAccepted, map[string]string{"status": "pending"})
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// Approved. Single use: the row goes before the session is made, so two
|
||||||
|
// racing polls cannot both be given one.
|
||||||
|
if _, err := tx.ExecContext(r.Context(), "DELETE FROM device_logins WHERE device_hash = $1", hash); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
var disabled, sso bool
|
||||||
|
if err := tx.QueryRowContext(r.Context(), `
|
||||||
|
SELECT disabled_at IS NOT NULL,
|
||||||
|
EXISTS (SELECT 1 FROM user_identities WHERE user_id = $1)
|
||||||
|
FROM users WHERE id = $1`, userID.Int64).Scan(&disabled, &sso); err != nil {
|
||||||
|
gone(w, "denied")
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err := tx.Commit(); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if disabled {
|
||||||
|
gone(w, "denied")
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// A session for somebody who signs in through the provider carries the
|
||||||
|
// same ceiling as their browser's would, so the terminal is not a way
|
||||||
|
// round it. Password users have none.
|
||||||
|
var maxAge time.Duration
|
||||||
|
if sso {
|
||||||
|
maxAge = ssoMaxAge
|
||||||
|
}
|
||||||
|
if err := startSessionCapped(w, r, db, userID.Int64, publicURL, maxAge); err != nil {
|
||||||
|
log.Printf("device login: start session: %v", err)
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
user, err := fetchUser(r.Context(), db, userID.Int64)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusOK, meResponse{User: user, HasPassword: false})
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,347 @@
|
|||||||
|
package api_test
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/json"
|
||||||
|
"net/http"
|
||||||
|
"net/url"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
type deviceStart struct {
|
||||||
|
DeviceCode string `json:"device_code"`
|
||||||
|
UserCode string `json:"user_code"`
|
||||||
|
VerificationURL string `json:"verification_url"`
|
||||||
|
Interval int `json:"interval"`
|
||||||
|
ExpiresIn int `json:"expires_in"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// startDevice is the terminal asking for a login.
|
||||||
|
func startDevice(t *testing.T, s *ts) deviceStart {
|
||||||
|
t.Helper()
|
||||||
|
resp := newBrowser(t, s.URL).do(t, http.MethodPost, "/api/oidc/device", nil)
|
||||||
|
var d deviceStart
|
||||||
|
decode(t, resp, &d)
|
||||||
|
if d.DeviceCode == "" || d.UserCode == "" {
|
||||||
|
t.Fatalf("device start returned %+v", d)
|
||||||
|
}
|
||||||
|
return d
|
||||||
|
}
|
||||||
|
|
||||||
|
// pollDevice is the terminal polling. It returns the status, and the session
|
||||||
|
// cookie the response set, if any.
|
||||||
|
func pollDevice(t *testing.T, s *ts, code string) (int, *http.Cookie, string) {
|
||||||
|
t.Helper()
|
||||||
|
resp := newBrowser(t, s.URL).do(t, http.MethodPost, "/api/oidc/device/token", map[string]string{"device_code": code})
|
||||||
|
defer resp.Body.Close()
|
||||||
|
var body map[string]any
|
||||||
|
json.NewDecoder(resp.Body).Decode(&body)
|
||||||
|
var cookie *http.Cookie
|
||||||
|
for _, c := range resp.Cookies() {
|
||||||
|
if c.Name == "terdut_session" {
|
||||||
|
cookie = c
|
||||||
|
}
|
||||||
|
}
|
||||||
|
msg, _ := body["error"].(string)
|
||||||
|
if msg == "" {
|
||||||
|
msg, _ = body["status"].(string)
|
||||||
|
}
|
||||||
|
return resp.StatusCode, cookie, msg
|
||||||
|
}
|
||||||
|
|
||||||
|
// readyToPoll lets the next poll through: the server holds a client to the
|
||||||
|
// interval it was given, which a test has no wish to wait out.
|
||||||
|
func (s *ts) readyToPoll(t *testing.T) {
|
||||||
|
t.Helper()
|
||||||
|
s.exec(t, "UPDATE device_logins SET last_polled_at = 0")
|
||||||
|
}
|
||||||
|
|
||||||
|
func decide(t *testing.T, b *browser, what, code string) int {
|
||||||
|
t.Helper()
|
||||||
|
resp := b.do(t, http.MethodPost, "/api/oidc/device/"+what, map[string]string{"user_code": code})
|
||||||
|
resp.Body.Close()
|
||||||
|
return resp.StatusCode
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDevice_FullFlow(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
|
||||||
|
d := startDevice(t, s)
|
||||||
|
if !strings.HasPrefix(d.VerificationURL, "http://terdut.test/device?code=") ||
|
||||||
|
!strings.Contains(d.VerificationURL, url.QueryEscape(d.UserCode)) {
|
||||||
|
t.Errorf("verification url %q", d.VerificationURL)
|
||||||
|
}
|
||||||
|
if len(d.UserCode) != 9 || d.UserCode[4] != '-' || d.Interval != 5 || d.ExpiresIn != 600 {
|
||||||
|
t.Errorf("start: %+v", d)
|
||||||
|
}
|
||||||
|
|
||||||
|
if status, cookie, msg := pollDevice(t, s, d.DeviceCode); status != http.StatusAccepted || cookie != nil || msg != "pending" {
|
||||||
|
t.Fatalf("first poll: %d %v %q, want 202 pending and no cookie", status, cookie, msg)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The person signs in through the provider in some browser and approves.
|
||||||
|
person := ssoBrowser(t, s)
|
||||||
|
signInSSO(t, idp, person, alice)
|
||||||
|
if got := decide(t, person, "approve", d.UserCode); got != http.StatusNoContent {
|
||||||
|
t.Fatalf("approve: %d", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
s.readyToPoll(t)
|
||||||
|
status, cookie, _ := pollDevice(t, s, d.DeviceCode)
|
||||||
|
if status != http.StatusOK || cookie == nil {
|
||||||
|
t.Fatalf("poll after approval: %d, cookie %v", status, cookie)
|
||||||
|
}
|
||||||
|
// The cookie is a working session for the person who approved.
|
||||||
|
term := newBrowser(t, s.URL)
|
||||||
|
req, _ := http.NewRequest(http.MethodGet, s.URL+"/api/me", nil)
|
||||||
|
req.AddCookie(cookie)
|
||||||
|
resp, err := term.Do(req)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var me struct {
|
||||||
|
User struct {
|
||||||
|
Username string `json:"username"`
|
||||||
|
} `json:"user"`
|
||||||
|
}
|
||||||
|
decode(t, resp, &me)
|
||||||
|
if me.User.Username != "alice" {
|
||||||
|
t.Errorf("session belongs to %q, want alice", me.User.Username)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Single use.
|
||||||
|
if status, cookie, msg := pollDevice(t, s, d.DeviceCode); status != http.StatusGone || cookie != nil || msg != "expired" {
|
||||||
|
t.Errorf("second redemption: %d %v %q, want 410 expired", status, cookie, msg)
|
||||||
|
}
|
||||||
|
// The session was made for an SSO user, so it carries the ceiling.
|
||||||
|
var ceiling *int64
|
||||||
|
s.db.QueryRow("SELECT max_expires_at FROM sessions ORDER BY id DESC LIMIT 1").Scan(&ceiling)
|
||||||
|
if ceiling == nil {
|
||||||
|
t.Error("a device session for an SSO user must carry the SSO session ceiling")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDevice_PasswordUserGetsNoCeiling(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
admin := signedIn(t, s) // password sign-in as the bootstrap admin
|
||||||
|
|
||||||
|
d := startDevice(t, s)
|
||||||
|
if got := decide(t, admin, "approve", d.UserCode); got != http.StatusNoContent {
|
||||||
|
t.Fatalf("approve: %d", got)
|
||||||
|
}
|
||||||
|
s.readyToPoll(t)
|
||||||
|
if status, cookie, _ := pollDevice(t, s, d.DeviceCode); status != http.StatusOK || cookie == nil {
|
||||||
|
t.Fatalf("poll: %d %v", status, cookie)
|
||||||
|
}
|
||||||
|
var ceiling *int64
|
||||||
|
s.db.QueryRow("SELECT max_expires_at FROM sessions ORDER BY id DESC LIMIT 1").Scan(&ceiling)
|
||||||
|
if ceiling != nil {
|
||||||
|
t.Errorf("a password user's device session has a ceiling %d, want none", *ceiling)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDevice_Denied(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
person := ssoBrowser(t, s)
|
||||||
|
signInSSO(t, idp, person, alice)
|
||||||
|
|
||||||
|
d := startDevice(t, s)
|
||||||
|
if got := decide(t, person, "deny", d.UserCode); got != http.StatusNoContent {
|
||||||
|
t.Fatalf("deny: %d", got)
|
||||||
|
}
|
||||||
|
s.readyToPoll(t)
|
||||||
|
if status, cookie, msg := pollDevice(t, s, d.DeviceCode); status != http.StatusGone || cookie != nil || msg != "denied" {
|
||||||
|
t.Errorf("poll: %d %v %q, want 410 denied", status, cookie, msg)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDevice_DecisionNeedsABrowserSession(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
d := startDevice(t, s)
|
||||||
|
|
||||||
|
// Nobody signed in.
|
||||||
|
if got := decide(t, newBrowser(t, s.URL), "approve", d.UserCode); got != http.StatusUnauthorized {
|
||||||
|
t.Errorf("anonymous approve: %d, want 401", got)
|
||||||
|
}
|
||||||
|
// An API key is a credential for scripts, not for approving a terminal.
|
||||||
|
resp := s.req(t, http.MethodPost, "/api/oidc/device/approve", map[string]string{"user_code": d.UserCode})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("approve with an API key: %d, want 403", resp.StatusCode)
|
||||||
|
}
|
||||||
|
if status, _, msg := pollDevice(t, s, d.DeviceCode); status != http.StatusAccepted || msg != "pending" {
|
||||||
|
t.Errorf("the login must still be pending: %d %q", status, msg)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDevice_ApprovalIsFinal(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
first, second := ssoBrowser(t, s), ssoBrowser(t, s)
|
||||||
|
signInSSO(t, idp, first, alice)
|
||||||
|
signInSSO(t, idp, second, idpUser{sub: "sub-mallory", username: "mallory", email: "mallory@example.com", groups: []string{"terdut-users"}})
|
||||||
|
|
||||||
|
d := startDevice(t, s)
|
||||||
|
if got := decide(t, first, "approve", d.UserCode); got != http.StatusNoContent {
|
||||||
|
t.Fatalf("approve: %d", got)
|
||||||
|
}
|
||||||
|
// A second browser cannot take the login over, nor refuse it.
|
||||||
|
for _, what := range []string{"approve", "deny"} {
|
||||||
|
if got := decide(t, second, what, d.UserCode); got != http.StatusNotFound {
|
||||||
|
t.Errorf("%s after approval: %d, want 404", what, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
s.readyToPoll(t)
|
||||||
|
_, cookie, _ := pollDevice(t, s, d.DeviceCode)
|
||||||
|
if cookie == nil {
|
||||||
|
t.Fatal("no session")
|
||||||
|
}
|
||||||
|
var name string
|
||||||
|
s.db.QueryRow("SELECT u.username FROM sessions ss JOIN users u ON u.id = ss.user_id ORDER BY ss.id DESC LIMIT 1").Scan(&name)
|
||||||
|
if name != "alice" {
|
||||||
|
t.Errorf("session for %q, want alice", name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDevice_CodeIsForgivingAboutHowItWasTyped(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
person := ssoBrowser(t, s)
|
||||||
|
signInSSO(t, idp, person, alice)
|
||||||
|
|
||||||
|
d := startDevice(t, s)
|
||||||
|
typed := strings.ToLower(strings.ReplaceAll(d.UserCode, "-", " "))
|
||||||
|
if got := decide(t, person, "approve", typed); got != http.StatusNoContent {
|
||||||
|
t.Errorf("approve %q: %d, want 204", typed, got)
|
||||||
|
}
|
||||||
|
if got := decide(t, person, "approve", "nonsense"); got != http.StatusBadRequest {
|
||||||
|
t.Errorf("approve nonsense: %d, want 400", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDevice_ExpiredAndUnknown(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
person := ssoBrowser(t, s)
|
||||||
|
signInSSO(t, idp, person, alice)
|
||||||
|
|
||||||
|
d := startDevice(t, s)
|
||||||
|
s.exec(t, "UPDATE device_logins SET expires_at = 1")
|
||||||
|
if got := decide(t, person, "approve", d.UserCode); got != http.StatusNotFound {
|
||||||
|
t.Errorf("approve expired: %d, want 404", got)
|
||||||
|
}
|
||||||
|
if status, _, msg := pollDevice(t, s, d.DeviceCode); status != http.StatusGone || msg != "expired" {
|
||||||
|
t.Errorf("poll expired: %d %q, want 410 expired", status, msg)
|
||||||
|
}
|
||||||
|
if status, _, msg := pollDevice(t, s, "not-a-device-code"); status != http.StatusGone || msg != "expired" {
|
||||||
|
t.Errorf("poll unknown: %d %q, want 410 expired", status, msg)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDevice_PollingTooFastIsRefused(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
d := startDevice(t, s)
|
||||||
|
if status, _, _ := pollDevice(t, s, d.DeviceCode); status != http.StatusAccepted {
|
||||||
|
t.Fatalf("first poll: %d", status)
|
||||||
|
}
|
||||||
|
if status, _, msg := pollDevice(t, s, d.DeviceCode); status != http.StatusTooManyRequests || msg != "slow_down" {
|
||||||
|
t.Errorf("immediate second poll: %d %q, want 429 slow_down", status, msg)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDevice_DisabledUserGetsNoSession(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
person := ssoBrowser(t, s)
|
||||||
|
signInSSO(t, idp, person, alice)
|
||||||
|
|
||||||
|
d := startDevice(t, s)
|
||||||
|
decide(t, person, "approve", d.UserCode)
|
||||||
|
s.exec(t, "UPDATE users SET disabled_at = 1 WHERE username = 'alice'")
|
||||||
|
s.readyToPoll(t)
|
||||||
|
var before int
|
||||||
|
s.db.QueryRow("SELECT COUNT(*) FROM sessions").Scan(&before)
|
||||||
|
if status, cookie, _ := pollDevice(t, s, d.DeviceCode); status != http.StatusGone || cookie != nil {
|
||||||
|
t.Errorf("poll: %d %v, want 410 and no cookie", status, cookie)
|
||||||
|
}
|
||||||
|
var after int
|
||||||
|
s.db.QueryRow("SELECT COUNT(*) FROM sessions").Scan(&after)
|
||||||
|
if after != before {
|
||||||
|
t.Error("a session was created for a disabled user")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDevice_OnlyExistsWithSSOConfigured(t *testing.T) {
|
||||||
|
s := newTS(t) // no SSO
|
||||||
|
resp := newBrowser(t, s.URL).do(t, http.MethodPost, "/api/oidc/device", nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNotFound {
|
||||||
|
t.Errorf("device start with SSO off: %d, want 404", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
for _, c := range []struct {
|
||||||
|
name string
|
||||||
|
s *ts
|
||||||
|
want bool
|
||||||
|
}{{"off", s, false}, {"on", newSSOTS(t, idp), true}} {
|
||||||
|
var cfg struct {
|
||||||
|
DeviceLogin bool `json:"device_login"`
|
||||||
|
}
|
||||||
|
decode(t, newBrowser(t, c.s.URL).do(t, http.MethodGet, "/api/auth/config", nil), &cfg)
|
||||||
|
if cfg.DeviceLogin != c.want {
|
||||||
|
t.Errorf("auth config device_login with SSO %s: %v, want %v", c.name, cfg.DeviceLogin, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestDevice_StartIsRateLimited(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
b := newBrowser(t, s.URL)
|
||||||
|
var last int
|
||||||
|
for range 32 {
|
||||||
|
resp := b.do(t, http.MethodPost, "/api/oidc/device", nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
last = resp.StatusCode
|
||||||
|
}
|
||||||
|
if last != http.StatusTooManyRequests {
|
||||||
|
t.Errorf("32nd start: %d, want 429", last)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// After signing in the browser is sent on to where the person was going, which
|
||||||
|
// is how somebody without a session gets from /device?code=... through the
|
||||||
|
// provider and back to it. Only paths on this server are honoured.
|
||||||
|
func TestSSO_NextIsHonouredOnlyForPathsOnThisServer(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
|
||||||
|
for _, c := range []struct{ next, want string }{
|
||||||
|
{"/device?code=BCDF-GHJK", "/device?code=BCDF-GHJK"},
|
||||||
|
{"/team/members", "/team/members"},
|
||||||
|
{"", "/"},
|
||||||
|
{"//evil.example/x", "/"},
|
||||||
|
{"/\\evil.example", "/"},
|
||||||
|
{"https://evil.example/", "/"},
|
||||||
|
{"evil.example", "/"},
|
||||||
|
{"/api/users", "/"},
|
||||||
|
{"/ok\r\nSet-Cookie: x=y", "/"},
|
||||||
|
{"/" + strings.Repeat("a", 600), "/"},
|
||||||
|
} {
|
||||||
|
b := ssoBrowser(t, s)
|
||||||
|
resp := b.do(t, http.MethodGet, "/api/oidc/login?next="+url.QueryEscape(c.next), nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
loc, _ := url.Parse(resp.Header.Get("Location"))
|
||||||
|
q := loc.Query()
|
||||||
|
got := callback(t, b, idp.issueCode(alice, q.Get("nonce"), q.Get("code_challenge")), q.Get("state"))
|
||||||
|
if got != c.want {
|
||||||
|
t.Errorf("next %q: redirected to %q, want %q", c.next, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -3,6 +3,7 @@ package api
|
|||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
"database/sql"
|
"database/sql"
|
||||||
|
"errors"
|
||||||
"log"
|
"log"
|
||||||
"net/http"
|
"net/http"
|
||||||
"strconv"
|
"strconv"
|
||||||
@@ -229,7 +230,7 @@ func advanceEscalation(ctx context.Context, db *sql.DB, cfg NotifyConfig, policy
|
|||||||
// nobody. That is a policy that looks configured and is not.
|
// nobody. That is a policy that looks configured and is not.
|
||||||
detail += ": nobody reachable"
|
detail += ": nobody reachable"
|
||||||
}
|
}
|
||||||
if err := logEvent(ctx, tx, incidentID, evEscalated, nil, nil, &detail); err != nil {
|
if err := logEvent(ctx, tx, incidentID, evEscalated, nil, nil, nil, &detail); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
return tx.Commit()
|
return tx.Commit()
|
||||||
@@ -256,7 +257,7 @@ func escalationExhausted(ctx context.Context, tx *sql.Tx, policy *escalationPoli
|
|||||||
incidentID); err != nil {
|
incidentID); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
return logEvent(ctx, tx, incidentID, evEscalated, nil, nil, &detail)
|
return logEvent(ctx, tx, incidentID, evEscalated, nil, nil, nil, &detail)
|
||||||
}
|
}
|
||||||
|
|
||||||
// pageLevel notifies every target of one level and reports who was woken.
|
// pageLevel notifies every target of one level and reports who was woken.
|
||||||
@@ -343,13 +344,197 @@ func handleGetEscalation(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
policy, err := loadEscalationPolicy(r.Context(), db, teamID)
|
policy, err := loadEscalationPolicy(r.Context(), db, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, escalationResponse(policy, teamID))
|
view, err := escalationStatus(r.Context(), db, teamID, policy)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusOK, view)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Level statuses, as the Escalation page colours them.
|
||||||
|
const (
|
||||||
|
levelReady = "ready"
|
||||||
|
levelEscalating = "escalating"
|
||||||
|
levelUnreachable = "unreachable"
|
||||||
|
)
|
||||||
|
|
||||||
|
// escalationTargetView is a target with who it means today and whether that
|
||||||
|
// person can actually be woken. The extra fields are output only: the PUT body
|
||||||
|
// is the plain escalationTargetJSON, and anything else in it is ignored.
|
||||||
|
type escalationTargetView struct {
|
||||||
|
escalationTargetJSON
|
||||||
|
|
||||||
|
// Username is who the target resolves to right now: the named person, or
|
||||||
|
// whoever the rota says is on call today. Empty when nobody is.
|
||||||
|
Username string `json:"username,omitempty"`
|
||||||
|
|
||||||
|
// Reachable is whether a page to this target would go anywhere, and Problem
|
||||||
|
// says why not when it would not — the same conditions pageLevel skips on.
|
||||||
|
Reachable bool `json:"reachable"`
|
||||||
|
Problem string `json:"problem,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type escalationLevelView struct {
|
||||||
|
Position int64 `json:"position"`
|
||||||
|
TimeoutSeconds int64 `json:"timeout_seconds"`
|
||||||
|
Targets []escalationTargetView `json:"targets"`
|
||||||
|
|
||||||
|
// Status is unreachable when no target of the level could be woken — a rung
|
||||||
|
// that looks configured and pages nobody, which is worth seeing before an
|
||||||
|
// incident finds it — escalating when an unanswered incident has climbed to
|
||||||
|
// it, and ready otherwise.
|
||||||
|
Status string `json:"status"`
|
||||||
|
|
||||||
|
// Waiting lists the open, unacknowledged incidents currently on this level.
|
||||||
|
Waiting []int64 `json:"waiting"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type escalationView struct {
|
||||||
|
TeamID int64 `json:"team_id"`
|
||||||
|
RepeatCount int64 `json:"repeat_count"`
|
||||||
|
FallbackTopic string `json:"fallback_topic"`
|
||||||
|
Levels []escalationLevelView `json:"levels"`
|
||||||
|
|
||||||
|
// LastEscalatedAt is when an incident of this team last moved up the ladder,
|
||||||
|
// or ran off the end of it, and LastEscalatedIncidentID which one. Absent
|
||||||
|
// when nothing ever has: a ladder nobody has needed yet.
|
||||||
|
LastEscalatedAt *time.Time `json:"last_escalated_at,omitempty"`
|
||||||
|
LastEscalatedIncidentID *int64 `json:"last_escalated_incident_id,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// escalationStatus is a team's ladder together with what it would do right now
|
||||||
|
// and what it has been doing. The resolution follows pageLevel's rules, so the
|
||||||
|
// page cannot promise a page that the notifier would skip.
|
||||||
|
func escalationStatus(ctx context.Context, db *sql.DB, teamID int64, policy *escalationPolicy) (escalationView, error) {
|
||||||
|
base := escalationResponse(policy, teamID)
|
||||||
|
out := escalationView{
|
||||||
|
TeamID: teamID, RepeatCount: base.RepeatCount, FallbackTopic: base.FallbackTopic,
|
||||||
|
Levels: []escalationLevelView{},
|
||||||
|
}
|
||||||
|
if !policy.configured() {
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
onCall, err := currentOnCall(ctx, db, teamID)
|
||||||
|
if err != nil {
|
||||||
|
return out, err
|
||||||
|
}
|
||||||
|
|
||||||
|
type account struct {
|
||||||
|
username string
|
||||||
|
topic bool
|
||||||
|
disabled bool
|
||||||
|
}
|
||||||
|
accounts := map[int64]account{}
|
||||||
|
lookup := func(id int64) (account, error) {
|
||||||
|
if a, ok := accounts[id]; ok {
|
||||||
|
return a, nil
|
||||||
|
}
|
||||||
|
var a account
|
||||||
|
var topic *string
|
||||||
|
var disabledAt *int64
|
||||||
|
if err := db.QueryRowContext(ctx,
|
||||||
|
"SELECT username, ntfy_topic, disabled_at FROM users WHERE id = $1", id).
|
||||||
|
Scan(&a.username, &topic, &disabledAt); err != nil {
|
||||||
|
return a, err
|
||||||
|
}
|
||||||
|
a.topic = topic != nil && *topic != ""
|
||||||
|
a.disabled = disabledAt != nil
|
||||||
|
accounts[id] = a
|
||||||
|
return a, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
waiting := map[int64][]int64{}
|
||||||
|
rows, err := db.QueryContext(ctx, `
|
||||||
|
SELECT id, escalation_level FROM incidents
|
||||||
|
WHERE team_id = $1 AND resolved_at IS NULL AND archived_at IS NULL
|
||||||
|
AND status = 'triggered' AND escalation_level > 0
|
||||||
|
ORDER BY id`, teamID)
|
||||||
|
if err != nil {
|
||||||
|
return out, err
|
||||||
|
}
|
||||||
|
for rows.Next() {
|
||||||
|
var id, level int64
|
||||||
|
if err := rows.Scan(&id, &level); err != nil {
|
||||||
|
rows.Close()
|
||||||
|
return out, err
|
||||||
|
}
|
||||||
|
waiting[level] = append(waiting[level], id)
|
||||||
|
}
|
||||||
|
rows.Close()
|
||||||
|
if err := rows.Err(); err != nil {
|
||||||
|
return out, err
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, l := range base.Levels {
|
||||||
|
level := escalationLevelView{
|
||||||
|
Position: l.Position, TimeoutSeconds: l.TimeoutSeconds,
|
||||||
|
Targets: []escalationTargetView{}, Waiting: []int64{},
|
||||||
|
}
|
||||||
|
if w := waiting[l.Position]; w != nil {
|
||||||
|
level.Waiting = w
|
||||||
|
}
|
||||||
|
|
||||||
|
anyReachable := false
|
||||||
|
for _, t := range l.Targets {
|
||||||
|
view := escalationTargetView{escalationTargetJSON: t}
|
||||||
|
userID := t.UserID
|
||||||
|
if t.Kind == "oncall" {
|
||||||
|
userID = onCall
|
||||||
|
}
|
||||||
|
switch {
|
||||||
|
case userID == nil:
|
||||||
|
view.Problem = "nobody is on call today"
|
||||||
|
default:
|
||||||
|
a, err := lookup(*userID)
|
||||||
|
switch {
|
||||||
|
case err != nil:
|
||||||
|
view.Problem = "account not found"
|
||||||
|
case a.disabled:
|
||||||
|
view.Username, view.Problem = a.username, "account is disabled"
|
||||||
|
case !a.topic:
|
||||||
|
view.Username, view.Problem = a.username, "has no ntfy topic"
|
||||||
|
default:
|
||||||
|
view.Username, view.Reachable = a.username, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
anyReachable = anyReachable || view.Reachable
|
||||||
|
level.Targets = append(level.Targets, view)
|
||||||
|
}
|
||||||
|
|
||||||
|
switch {
|
||||||
|
case !anyReachable:
|
||||||
|
level.Status = levelUnreachable
|
||||||
|
case l.Position >= 2 && len(level.Waiting) > 0:
|
||||||
|
level.Status = levelEscalating
|
||||||
|
default:
|
||||||
|
level.Status = levelReady
|
||||||
|
}
|
||||||
|
out.Levels = append(out.Levels, level)
|
||||||
|
}
|
||||||
|
|
||||||
|
var incidentID, at int64
|
||||||
|
switch err := db.QueryRowContext(ctx, `
|
||||||
|
SELECT e.incident_id, e.created_at
|
||||||
|
FROM incident_events e JOIN incidents i ON i.id = e.incident_id
|
||||||
|
WHERE i.team_id = $1 AND e.type = $2
|
||||||
|
ORDER BY e.created_at DESC, e.id DESC LIMIT 1`, teamID, evEscalated).
|
||||||
|
Scan(&incidentID, &at); {
|
||||||
|
case err == sql.ErrNoRows:
|
||||||
|
case err != nil:
|
||||||
|
return out, err
|
||||||
|
default:
|
||||||
|
t := time.Unix(at, 0).UTC()
|
||||||
|
out.LastEscalatedAt, out.LastEscalatedIncidentID = &t, &incidentID
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
type escalationLevelJSON struct {
|
type escalationLevelJSON struct {
|
||||||
Position int64 `json:"position"`
|
Position int64 `json:"position"`
|
||||||
TimeoutSeconds int64 `json:"timeout_seconds"`
|
TimeoutSeconds int64 `json:"timeout_seconds"`
|
||||||
@@ -359,6 +544,10 @@ type escalationLevelJSON struct {
|
|||||||
type escalationTargetJSON struct {
|
type escalationTargetJSON struct {
|
||||||
Kind string `json:"kind"`
|
Kind string `json:"kind"`
|
||||||
UserID *int64 `json:"user_id,omitempty"`
|
UserID *int64 `json:"user_id,omitempty"`
|
||||||
|
// Username is accepted in place of user_id on a PUT, and resolved to the
|
||||||
|
// id before anything is stored. It is never returned: the stored form is
|
||||||
|
// the id, which survives a rename.
|
||||||
|
Username string `json:"username,omitempty"`
|
||||||
}
|
}
|
||||||
|
|
||||||
type escalationJSON struct {
|
type escalationJSON struct {
|
||||||
@@ -416,6 +605,27 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
for i, l := range req.Levels {
|
for i, l := range req.Levels {
|
||||||
|
for j, t := range l.Targets {
|
||||||
|
if t.Username == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if t.Kind != "user" || t.UserID != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("username belongs on a user target, instead of user_id"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
var id int64
|
||||||
|
err := db.QueryRowContext(r.Context(), "SELECT id FROM users WHERE username = $1", t.Username).Scan(&id)
|
||||||
|
if errors.Is(err, sql.ErrNoRows) {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("unknown user "+strconv.Quote(t.Username)))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
req.Levels[i].Targets[j].UserID = &id
|
||||||
|
req.Levels[i].Targets[j].Username = ""
|
||||||
|
}
|
||||||
if l.TimeoutSeconds <= 0 {
|
if l.TimeoutSeconds <= 0 {
|
||||||
respond(w, http.StatusBadRequest, errResp("every level needs a timeout"))
|
respond(w, http.StatusBadRequest, errResp("every level needs a timeout"))
|
||||||
return
|
return
|
||||||
@@ -448,7 +658,7 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
tx, err := db.BeginTx(r.Context(), nil)
|
tx, err := db.BeginTx(r.Context(), nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer tx.Rollback() //nolint:errcheck
|
defer tx.Rollback() //nolint:errcheck
|
||||||
@@ -461,13 +671,13 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc {
|
|||||||
fallback_topic = excluded.fallback_topic,
|
fallback_topic = excluded.fallback_topic,
|
||||||
updated_at = excluded.updated_at`,
|
updated_at = excluded.updated_at`,
|
||||||
teamID, req.RepeatCount, req.FallbackTopic); err != nil {
|
teamID, req.RepeatCount, req.FallbackTopic); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
// The levels are replaced, not merged; the cascade takes the targets.
|
// The levels are replaced, not merged; the cascade takes the targets.
|
||||||
if _, err := tx.ExecContext(r.Context(),
|
if _, err := tx.ExecContext(r.Context(),
|
||||||
"DELETE FROM escalation_levels WHERE team_id = $1", teamID); err != nil {
|
"DELETE FROM escalation_levels WHERE team_id = $1", teamID); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -477,7 +687,7 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc {
|
|||||||
INSERT INTO escalation_levels (team_id, position, timeout_seconds)
|
INSERT INTO escalation_levels (team_id, position, timeout_seconds)
|
||||||
VALUES ($1, $2, $3) RETURNING id`,
|
VALUES ($1, $2, $3) RETURNING id`,
|
||||||
teamID, int64(i+1), l.TimeoutSeconds).Scan(&levelID); err != nil {
|
teamID, int64(i+1), l.TimeoutSeconds).Scan(&levelID); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
for _, t := range l.Targets {
|
for _, t := range l.Targets {
|
||||||
@@ -492,13 +702,13 @@ func handleSetEscalation(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
|
|
||||||
if err := tx.Commit(); err != nil {
|
if err := tx.Commit(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
policy, err := loadEscalationPolicy(r.Context(), db, teamID)
|
policy, err := loadEscalationPolicy(r.Context(), db, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, escalationResponse(policy, teamID))
|
respond(w, http.StatusOK, escalationResponse(policy, teamID))
|
||||||
|
|||||||
@@ -383,3 +383,164 @@ func TestEscalation_SkipsUnreachableTargets(t *testing.T) {
|
|||||||
t.Errorf("a target with no topic should page nothing, paged %v", got)
|
t.Errorf("a target with no topic should page nothing, paged %v", got)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// The ladder as the Escalation page reads it
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
type ladderLevel struct {
|
||||||
|
Status string `json:"status"`
|
||||||
|
Waiting []int64 `json:"waiting"`
|
||||||
|
Targets []struct {
|
||||||
|
Kind string `json:"kind"`
|
||||||
|
Username string `json:"username"`
|
||||||
|
Reachable bool `json:"reachable"`
|
||||||
|
Problem string `json:"problem"`
|
||||||
|
} `json:"targets"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type ladderView struct {
|
||||||
|
Levels []ladderLevel `json:"levels"`
|
||||||
|
LastEscalatedAt *string `json:"last_escalated_at"`
|
||||||
|
LastEscalatedIncidentID *int64 `json:"last_escalated_incident_id"`
|
||||||
|
}
|
||||||
|
|
||||||
|
func readLadder(t *testing.T, s *ts) ladderView {
|
||||||
|
t.Helper()
|
||||||
|
var v ladderView
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/escalation", nil), &v)
|
||||||
|
return v
|
||||||
|
}
|
||||||
|
|
||||||
|
// Targets say who they mean today, so "whoever is on call" is a name and not a
|
||||||
|
// promise.
|
||||||
|
func TestEscalation_StatusResolvesTargets(t *testing.T) {
|
||||||
|
s, _ := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||||
|
second := teamUser(t, s, "second", "terdut-second")
|
||||||
|
ladder(t, s, second, 0, "terdut-fallback")
|
||||||
|
|
||||||
|
v := readLadder(t, s)
|
||||||
|
if len(v.Levels) != 2 {
|
||||||
|
t.Fatalf("expected 2 levels, got %d", len(v.Levels))
|
||||||
|
}
|
||||||
|
if got := v.Levels[0].Targets[0]; got.Kind != "oncall" || got.Username != "admin" || !got.Reachable {
|
||||||
|
t.Errorf("the rota target should resolve to the person on call, got %+v", got)
|
||||||
|
}
|
||||||
|
if got := v.Levels[1].Targets[0]; got.Username != "second" || !got.Reachable {
|
||||||
|
t.Errorf("the named target should be reachable, got %+v", got)
|
||||||
|
}
|
||||||
|
if v.Levels[0].Status != "ready" || v.Levels[1].Status != "ready" || v.LastEscalatedAt != nil {
|
||||||
|
t.Errorf("an idle, healthy ladder is ready and has never escalated, got %+v", v)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A rung that would page nobody is called out before an incident finds it.
|
||||||
|
func TestEscalation_StatusFlagsUnreachableLevels(t *testing.T) {
|
||||||
|
s, _ := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||||
|
silent := teamUser(t, s, "silent", "terdut-silent")
|
||||||
|
ladder(t, s, silent, 0, "terdut-fallback")
|
||||||
|
|
||||||
|
// Nobody on call today, and the named person loses their topic.
|
||||||
|
s.exec(t, "DELETE FROM schedule_entries")
|
||||||
|
s.exec(t, "UPDATE users SET ntfy_topic = NULL WHERE id = $1", silent)
|
||||||
|
|
||||||
|
v := readLadder(t, s)
|
||||||
|
if v.Levels[0].Status != "unreachable" || v.Levels[0].Targets[0].Problem != "nobody is on call today" {
|
||||||
|
t.Errorf("an empty rota should make level 1 unreachable, got %+v", v.Levels[0])
|
||||||
|
}
|
||||||
|
if v.Levels[1].Status != "unreachable" || v.Levels[1].Targets[0].Problem != "has no ntfy topic" {
|
||||||
|
t.Errorf("a person with no topic should make level 2 unreachable, got %+v", v.Levels[1])
|
||||||
|
}
|
||||||
|
|
||||||
|
s.exec(t, "UPDATE users SET disabled_at = 1 WHERE id = $1", silent)
|
||||||
|
if p := readLadder(t, s).Levels[1].Targets[0].Problem; p != "account is disabled" {
|
||||||
|
t.Errorf("a disabled account should say so, got %q", p)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Where unanswered incidents are right now, and when the ladder last did its
|
||||||
|
// job.
|
||||||
|
func TestEscalation_StatusShowsWhoIsWaitingAndLastEscalation(t *testing.T) {
|
||||||
|
s, _ := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||||
|
second := teamUser(t, s, "second", "terdut-second")
|
||||||
|
ladder(t, s, second, 0, "terdut-fallback")
|
||||||
|
|
||||||
|
postWebhook(t, s, []map[string]any{
|
||||||
|
amAlert("fp-wait", "DiskFull", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||||
|
})
|
||||||
|
s.sweepNotify(t)
|
||||||
|
|
||||||
|
// On level 1 it is waiting, which is normal and not yet an escalation.
|
||||||
|
v := readLadder(t, s)
|
||||||
|
if len(v.Levels[0].Waiting) != 1 || v.Levels[0].Status != "ready" || v.LastEscalatedAt != nil {
|
||||||
|
t.Fatalf("a fresh incident waits on level 1 quietly, got %+v", v)
|
||||||
|
}
|
||||||
|
|
||||||
|
overdue(t, s, 1)
|
||||||
|
s.sweepNotify(t)
|
||||||
|
v = readLadder(t, s)
|
||||||
|
if v.Levels[1].Status != "escalating" || len(v.Levels[1].Waiting) != 1 || v.Levels[1].Waiting[0] != 1 {
|
||||||
|
t.Errorf("level 2 should be escalating with the incident on it, got %+v", v.Levels[1])
|
||||||
|
}
|
||||||
|
if v.LastEscalatedAt == nil || v.LastEscalatedIncidentID == nil || *v.LastEscalatedIncidentID != 1 {
|
||||||
|
t.Errorf("the escalation should be recorded, got %+v", v)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Somebody answers: nothing is waiting, but the history stays.
|
||||||
|
s.req(t, http.MethodPost, "/api/incidents/1/acknowledge", nil).Body.Close()
|
||||||
|
v = readLadder(t, s)
|
||||||
|
if v.Levels[1].Status != "ready" || len(v.Levels[1].Waiting) != 0 || v.LastEscalatedAt == nil {
|
||||||
|
t.Errorf("an acknowledged incident stops waiting but stays in the history, got %+v", v)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// No ladder is a real answer, not an error.
|
||||||
|
func TestEscalation_StatusWithoutALadder(t *testing.T) {
|
||||||
|
s, _ := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute})
|
||||||
|
v := readLadder(t, s)
|
||||||
|
if len(v.Levels) != 0 || v.LastEscalatedAt != nil {
|
||||||
|
t.Errorf("a team with no ladder should read as empty, got %+v", v)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A user target may name the person instead of carrying an id; the server
|
||||||
|
// resolves it and stores the id.
|
||||||
|
func TestEscalation_UserTargetByUsername(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
id := teamUser(t, s, "alice", "")
|
||||||
|
|
||||||
|
put := func(username string) *http.Response {
|
||||||
|
return s.req(t, http.MethodPut, "/api/teams/"+defaultTeam+"/escalation", map[string]any{
|
||||||
|
"repeat_count": 0,
|
||||||
|
"levels": []map[string]any{{
|
||||||
|
"timeout_seconds": 300,
|
||||||
|
"targets": []map[string]any{{"kind": "user", "username": username}},
|
||||||
|
}},
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
resp := put("alice")
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode >= 300 {
|
||||||
|
t.Fatalf("PUT by username: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
var got struct {
|
||||||
|
Levels []struct {
|
||||||
|
Targets []struct {
|
||||||
|
UserID *int64 `json:"user_id"`
|
||||||
|
Username string `json:"username"`
|
||||||
|
} `json:"targets"`
|
||||||
|
} `json:"levels"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/escalation", nil), &got)
|
||||||
|
if len(got.Levels) != 1 || len(got.Levels[0].Targets) != 1 ||
|
||||||
|
got.Levels[0].Targets[0].UserID == nil || *got.Levels[0].Targets[0].UserID != id {
|
||||||
|
t.Errorf("expected the target stored as user %d, got %+v", id, got)
|
||||||
|
}
|
||||||
|
|
||||||
|
bad := put("nobody")
|
||||||
|
bad.Body.Close()
|
||||||
|
if bad.StatusCode != http.StatusBadRequest {
|
||||||
|
t.Errorf("unknown username should be a 400, got %d", bad.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,31 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// DeadmanConfig is a test fixture only: a set of matchers with one timeout and
|
||||||
|
// severity, turned into switches over the API by the test helpers. Production
|
||||||
|
// has no server-wide default any more -- switches belong to teams.
|
||||||
|
type DeadmanConfig struct {
|
||||||
|
Matchers []DeadmanMatcher
|
||||||
|
Timeout time.Duration
|
||||||
|
Severity string
|
||||||
|
}
|
||||||
|
|
||||||
|
// ParseDeadmanConfig reads a ";"-separated matcher list the way the removed
|
||||||
|
// environment variable did, dropping malformed entries.
|
||||||
|
func ParseDeadmanConfig(matchers string, timeout time.Duration, severity string) DeadmanConfig {
|
||||||
|
cfg := DeadmanConfig{Timeout: timeout, Severity: severity}
|
||||||
|
for _, entry := range strings.Split(matchers, ";") {
|
||||||
|
entry = strings.TrimSpace(entry)
|
||||||
|
if entry == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if m, err := parseDeadmanMatcher(entry); err == nil {
|
||||||
|
cfg.Matchers = append(cfg.Matchers, m)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return cfg
|
||||||
|
}
|
||||||
@@ -1,8 +1,12 @@
|
|||||||
package api
|
package api
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"context"
|
||||||
|
"database/sql"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
"errors"
|
"errors"
|
||||||
|
"github.com/go-chi/chi/v5"
|
||||||
|
"log"
|
||||||
"net/http"
|
"net/http"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
@@ -14,8 +18,7 @@ import (
|
|||||||
// sqlArgs accumulates query arguments and hands back the placeholder for each.
|
// sqlArgs accumulates query arguments and hands back the placeholder for each.
|
||||||
//
|
//
|
||||||
// Postgres numbers its placeholders, so a dynamically assembled WHERE clause has
|
// Postgres numbers its placeholders, so a dynamically assembled WHERE clause has
|
||||||
// to keep its $1, $2, … in step with the order of the values — which SQLite's
|
// to keep its $1, $2, … in step with the order of the values. Handing out the placeholder and storing the value
|
||||||
// positional `?` did for free. Handing out the placeholder and storing the value
|
|
||||||
// in one call is what keeps them in step: a filter can be added, removed or
|
// in one call is what keeps them in step: a filter can be added, removed or
|
||||||
// reordered without renumbering anything by hand.
|
// reordered without renumbering anything by hand.
|
||||||
type sqlArgs struct{ vals []any }
|
type sqlArgs struct{ vals []any }
|
||||||
@@ -28,8 +31,7 @@ func (a *sqlArgs) add(v any) string {
|
|||||||
|
|
||||||
// addList stores every value and returns their placeholders as "$1, $2, …",
|
// addList stores every value and returns their placeholders as "$1, $2, …",
|
||||||
// ready to drop into an IN (…) clause. Returns an empty string for no values,
|
// ready to drop into an IN (…) clause. Returns an empty string for no values,
|
||||||
// which no caller should reach: `IN ()` is a syntax error in Postgres as it was
|
// which no caller should reach: `IN ()` is a syntax error in Postgres, so callers check for an empty set before building the query.
|
||||||
// in SQLite, so callers check for an empty set before building the query.
|
|
||||||
func (a *sqlArgs) addList(vs []any) string {
|
func (a *sqlArgs) addList(vs []any) string {
|
||||||
parts := make([]string, len(vs))
|
parts := make([]string, len(vs))
|
||||||
for i, v := range vs {
|
for i, v := range vs {
|
||||||
@@ -42,7 +44,7 @@ func (a *sqlArgs) addList(vs []any) string {
|
|||||||
func (a *sqlArgs) all() []any { return a.vals }
|
func (a *sqlArgs) all() []any { return a.vals }
|
||||||
|
|
||||||
// nowEpoch is the SQL expression for "now, as unix seconds", matching how every
|
// nowEpoch is the SQL expression for "now, as unix seconds", matching how every
|
||||||
// timestamp in this schema is stored. SQLite spelled it unixepoch().
|
// timestamp in this schema is stored.
|
||||||
//
|
//
|
||||||
// FLOOR, not a bare cast: EXTRACT returns fractional seconds and casting to
|
// FLOOR, not a bare cast: EXTRACT returns fractional seconds and casting to
|
||||||
// bigint rounds half up, so a row written at .6 of a second would claim a
|
// bigint rounds half up, so a row written at .6 of a second would claim a
|
||||||
@@ -53,11 +55,9 @@ const nowEpoch = "FLOOR(EXTRACT(EPOCH FROM now()))::bigint"
|
|||||||
// isUniqueViolation reports whether err is a broken unique constraint, which
|
// isUniqueViolation reports whether err is a broken unique constraint, which
|
||||||
// callers turn into 409 Conflict rather than 500.
|
// callers turn into 409 Conflict rather than 500.
|
||||||
//
|
//
|
||||||
// Postgres reports it as SQLSTATE 23505 on a typed error; the SQLite driver this
|
// Postgres reports it as SQLSTATE 23505 on a typed error. Matching the code
|
||||||
// replaced only put "UNIQUE constraint failed" in the message, which is why the
|
// means a renamed constraint or a translated message cannot quietly turn a
|
||||||
// check used to be a substring match. Matching the code means a renamed
|
// conflict back into a 500.
|
||||||
// constraint or a translated message cannot quietly turn a conflict back into a
|
|
||||||
// 500.
|
|
||||||
func isUniqueViolation(err error) bool {
|
func isUniqueViolation(err error) bool {
|
||||||
var pgErr *pgconn.PgError
|
var pgErr *pgconn.PgError
|
||||||
return errors.As(err, &pgErr) && pgErr.Code == pgerrcode.UniqueViolation
|
return errors.As(err, &pgErr) && pgErr.Code == pgerrcode.UniqueViolation
|
||||||
@@ -69,11 +69,76 @@ func respond(w http.ResponseWriter, status int, v any) {
|
|||||||
json.NewEncoder(w).Encode(v)
|
json.NewEncoder(w).Encode(v)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// serverError answers 500 and logs why. The response stays opaque, so the log
|
||||||
|
// line is the only record of what failed.
|
||||||
|
func serverError(w http.ResponseWriter, r *http.Request, err error) {
|
||||||
|
// The route pattern, not the path: two routes carry a credential in it.
|
||||||
|
route := r.URL.Path
|
||||||
|
if rc := chi.RouteContext(r.Context()); rc != nil && rc.RoutePattern() != "" {
|
||||||
|
route = rc.RoutePattern()
|
||||||
|
}
|
||||||
|
log.Printf("%s %s: %v", strconv.Quote(r.Method), strconv.Quote(route), err)
|
||||||
|
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
||||||
|
}
|
||||||
|
|
||||||
|
// maxBodyBytes caps an ordinary JSON request body. 1 MiB is far more than any
|
||||||
|
// endpoint below needs — it exists so an unauthenticated caller (signup,
|
||||||
|
// login, bootstrap) can't make the server buffer an arbitrarily large body
|
||||||
|
// before the request is even validated.
|
||||||
|
const maxBodyBytes = 1 << 20
|
||||||
|
|
||||||
func decodeJSON(r *http.Request, v any) error {
|
func decodeJSON(r *http.Request, v any) error {
|
||||||
|
return decodeJSONLimit(r, v, maxBodyBytes)
|
||||||
|
}
|
||||||
|
|
||||||
|
// decodeJSONLimit is decodeJSON with an explicit cap, for the one endpoint
|
||||||
|
// (the Alertmanager webhook, see maxWebhookBodyBytes) whose real payloads can
|
||||||
|
// legitimately be larger than maxBodyBytes.
|
||||||
|
func decodeJSONLimit(r *http.Request, v any, limit int64) error {
|
||||||
defer r.Body.Close()
|
defer r.Body.Close()
|
||||||
|
// w is nil: there is no ResponseWriter here to disable keep-alive with,
|
||||||
|
// which net/http documents as fine — the limit is still enforced, the
|
||||||
|
// connection just isn't closed early on a request that blows past it.
|
||||||
|
r.Body = http.MaxBytesReader(nil, r.Body, limit)
|
||||||
return json.NewDecoder(r.Body).Decode(v)
|
return json.NewDecoder(r.Body).Decode(v)
|
||||||
}
|
}
|
||||||
|
|
||||||
func errResp(msg string) map[string]string {
|
func errResp(msg string) map[string]string {
|
||||||
return map[string]string{"error": msg}
|
return map[string]string{"error": msg}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// withAdvisoryLock runs fn only if it can take the named Postgres advisory lock on a
|
||||||
|
// dedicated connection, and skips fn otherwise. This is what keeps the archiver and
|
||||||
|
// notifier safe to run on more than one replica: whichever instance's tick gets there
|
||||||
|
// first does the work; the rest see the lock held and simply wait for their next tick
|
||||||
|
// instead of running the same pass concurrently.
|
||||||
|
//
|
||||||
|
// pg_try_advisory_lock is session-scoped, so taking and releasing it must happen on the
|
||||||
|
// same connection, reserved via db.Conn rather than borrowed from the pool's shared
|
||||||
|
// connections fn itself may use — and released (unlocked, then closed) before returning,
|
||||||
|
// since a session lock otherwise outlives this call and leaks onto whatever reuses the
|
||||||
|
// pooled connection next.
|
||||||
|
func withAdvisoryLock(ctx context.Context, db *sql.DB, key int64, name string, fn func()) {
|
||||||
|
conn, err := db.Conn(ctx)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("%s: advisory lock: acquire connection: %v", name, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer conn.Close()
|
||||||
|
|
||||||
|
var locked bool
|
||||||
|
if err := conn.QueryRowContext(ctx, "SELECT pg_try_advisory_lock($1)", key).Scan(&locked); err != nil {
|
||||||
|
log.Printf("%s: advisory lock: %v", name, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !locked {
|
||||||
|
return // another replica is already running this pass
|
||||||
|
}
|
||||||
|
defer func() {
|
||||||
|
if _, err := conn.ExecContext(ctx, "SELECT pg_advisory_unlock($1)", key); err != nil {
|
||||||
|
log.Printf("%s: advisory unlock: %v", name, err)
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
|
||||||
|
fn()
|
||||||
|
}
|
||||||
|
|||||||
@@ -32,10 +32,16 @@ const (
|
|||||||
evAcknowledged = "acknowledged"
|
evAcknowledged = "acknowledged"
|
||||||
evUnacknowledged = "unacknowledged"
|
evUnacknowledged = "unacknowledged"
|
||||||
evAssigned = "assigned"
|
evAssigned = "assigned"
|
||||||
|
evArchived = "archived"
|
||||||
|
evUnarchived = "unarchived"
|
||||||
evSnoozed = "snoozed"
|
evSnoozed = "snoozed"
|
||||||
evUnsnoozed = "unsnoozed"
|
evUnsnoozed = "unsnoozed"
|
||||||
evResolved = "resolved"
|
evResolved = "resolved"
|
||||||
evNote = "note"
|
evNote = "note"
|
||||||
|
// evResolutionNote is the note worth finding again: what fixed it. The
|
||||||
|
// similar-incidents lookup and the page lead with these; plain notes are
|
||||||
|
// the working chatter and stay one click away.
|
||||||
|
evResolutionNote = "resolution_note"
|
||||||
evDeadmanSilent = "deadman_silent"
|
evDeadmanSilent = "deadman_silent"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -52,25 +58,36 @@ type querier interface {
|
|||||||
|
|
||||||
const incidentSelectFrom = `
|
const incidentSelectFrom = `
|
||||||
SELECT i.id, i.team_id, t.name, i.group_key, i.title, i.group_labels, i.status, i.severity,
|
SELECT i.id, i.team_id, t.name, i.group_key, i.title, i.group_labels, i.status, i.severity,
|
||||||
|
i.escalation_level,
|
||||||
|
-- When this level runs out. Computed here rather than in Go because
|
||||||
|
-- the timeout lives beside the level in the policy, and one join is
|
||||||
|
-- cheaper than a second query per incident in a list.
|
||||||
|
(SELECT i.escalation_level_at + el.timeout_seconds
|
||||||
|
FROM escalation_levels el
|
||||||
|
WHERE el.team_id = i.team_id AND el.position = i.escalation_level),
|
||||||
i.triggered_at,
|
i.triggered_at,
|
||||||
i.acknowledged_by, i.acknowledged_at, ack.username,
|
i.acknowledged_by, i.acknowledged_at, ack.username,
|
||||||
|
i.acknowledged_by_service_account_id, acksa.name,
|
||||||
i.assigned_to, asg.username, i.snoozed_until,
|
i.assigned_to, asg.username, i.snoozed_until,
|
||||||
i.resolved_at, i.resolution_source, i.archived_at
|
i.resolved_at, i.resolution_source, i.archived_at
|
||||||
FROM incidents i
|
FROM incidents i
|
||||||
JOIN teams t ON t.id = i.team_id
|
JOIN teams t ON t.id = i.team_id
|
||||||
LEFT JOIN users ack ON ack.id = i.acknowledged_by
|
LEFT JOIN users ack ON ack.id = i.acknowledged_by
|
||||||
|
LEFT JOIN service_accounts acksa ON acksa.id = i.acknowledged_by_service_account_id
|
||||||
LEFT JOIN users asg ON asg.id = i.assigned_to`
|
LEFT JOIN users asg ON asg.id = i.assigned_to`
|
||||||
|
|
||||||
func scanIncident(s scanner) (models.Incident, error) {
|
func scanIncident(s scanner) (models.Incident, error) {
|
||||||
var i models.Incident
|
var i models.Incident
|
||||||
var groupLabelsJSON string
|
var groupLabelsJSON string
|
||||||
var triggeredAt int64
|
var triggeredAt int64
|
||||||
var ackAt, snoozedUntil, resolvedAt, archivedAt *int64
|
var ackAt, snoozedUntil, resolvedAt, archivedAt, escalationDue *int64
|
||||||
|
|
||||||
if err := s.Scan(
|
if err := s.Scan(
|
||||||
&i.ID, &i.TeamID, &i.TeamName, &i.GroupKey, &i.Title, &groupLabelsJSON, &i.Status, &i.Severity,
|
&i.ID, &i.TeamID, &i.TeamName, &i.GroupKey, &i.Title, &groupLabelsJSON, &i.Status, &i.Severity,
|
||||||
|
&i.EscalationLevel, &escalationDue,
|
||||||
&triggeredAt,
|
&triggeredAt,
|
||||||
&i.AcknowledgedByID, &ackAt, &i.AcknowledgedByUser,
|
&i.AcknowledgedByID, &ackAt, &i.AcknowledgedByUser,
|
||||||
|
&i.AcknowledgedByServiceAccountID, &i.AcknowledgedByServiceAccountName,
|
||||||
&i.AssignedToID, &i.AssignedToUser, &snoozedUntil,
|
&i.AssignedToID, &i.AssignedToUser, &snoozedUntil,
|
||||||
&resolvedAt, &i.ResolutionSource, &archivedAt,
|
&resolvedAt, &i.ResolutionSource, &archivedAt,
|
||||||
); err != nil {
|
); err != nil {
|
||||||
@@ -83,6 +100,7 @@ func scanIncident(s scanner) (models.Incident, error) {
|
|||||||
i.SnoozedUntil = unixPtr(snoozedUntil)
|
i.SnoozedUntil = unixPtr(snoozedUntil)
|
||||||
i.ResolvedAt = unixPtr(resolvedAt)
|
i.ResolvedAt = unixPtr(resolvedAt)
|
||||||
i.ArchivedAt = unixPtr(archivedAt)
|
i.ArchivedAt = unixPtr(archivedAt)
|
||||||
|
i.EscalationDueAt = unixPtr(escalationDue)
|
||||||
return i, nil
|
return i, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -99,13 +117,42 @@ func fetchIncident(ctx context.Context, q querier, id int64) (models.Incident, e
|
|||||||
return scanIncident(q.QueryRowContext(ctx, incidentSelectFrom+" WHERE i.id = $1", id))
|
return scanIncident(q.QueryRowContext(ctx, incidentSelectFrom+" WHERE i.id = $1", id))
|
||||||
}
|
}
|
||||||
|
|
||||||
// logEvent appends one entry to an incident's timeline. A nil userID means the
|
// callerActorIDs resolves the current request's caller into the pair of
|
||||||
// server acted rather than a person.
|
// nilable ids logEvent/acknowledgeIncidentAs expect: exactly one of userID/
|
||||||
func logEvent(ctx context.Context, q querier, incidentID int64, evType string, userID, alertID *int64, detail *string) error {
|
// serviceAccountID is set (never both), replacing the unchecked
|
||||||
|
// userFromContext(ctx) zero-value reads that used to write a human-only id
|
||||||
|
// of 0 for a service-account caller (terdut-server#25).
|
||||||
|
func callerActorIDs(ctx context.Context) (userID, serviceAccountID *int64) {
|
||||||
|
caller, _ := callerFromContext(ctx)
|
||||||
|
if u, ok := caller.AsHuman(); ok {
|
||||||
|
return &u.ID, nil
|
||||||
|
}
|
||||||
|
if id, ok := caller.ServiceAccountID(); ok {
|
||||||
|
return nil, &id
|
||||||
|
}
|
||||||
|
return nil, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// logEvent appends one entry to an incident's timeline. userID and
|
||||||
|
// serviceAccountID are mutually exclusive and both nilable; both nil means
|
||||||
|
// the server acted rather than any caller (see incident_events_actor_xor_chk,
|
||||||
|
// migration 015).
|
||||||
|
func logEvent(ctx context.Context, q querier, incidentID int64, evType string, userID, serviceAccountID, alertID *int64, detail *string) error {
|
||||||
_, err := q.ExecContext(ctx, `
|
_, err := q.ExecContext(ctx, `
|
||||||
INSERT INTO incident_events (incident_id, type, user_id, alert_id, detail, created_at)
|
INSERT INTO incident_events (incident_id, type, user_id, service_account_id, alert_id, detail, created_at)
|
||||||
|
VALUES ($1, $2, $3, $4, $5, $6, $7)`,
|
||||||
|
incidentID, evType, userID, serviceAccountID, alertID, detail, time.Now().Unix())
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
// logAssignedEvent records an assignment: user_id is the assignee, and the
|
||||||
|
// caller who performed it goes in the actor_* columns (migration 018), since
|
||||||
|
// user_id cannot hold both.
|
||||||
|
func logAssignedEvent(ctx context.Context, q querier, incidentID, assigneeID int64, actorUserID, actorServiceAccountID *int64) error {
|
||||||
|
_, err := q.ExecContext(ctx, `
|
||||||
|
INSERT INTO incident_events (incident_id, type, user_id, actor_user_id, actor_service_account_id, created_at)
|
||||||
VALUES ($1, $2, $3, $4, $5, $6)`,
|
VALUES ($1, $2, $3, $4, $5, $6)`,
|
||||||
incidentID, evType, userID, alertID, detail, time.Now().Unix())
|
incidentID, evAssigned, assigneeID, actorUserID, actorServiceAccountID, time.Now().Unix())
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -238,7 +285,7 @@ func resolveIfSettled(ctx context.Context, q querier, incidentID int64) (bool, e
|
|||||||
if err := stopEscalation(ctx, q, incidentID); err != nil {
|
if err := stopEscalation(ctx, q, incidentID); err != nil {
|
||||||
return false, err
|
return false, err
|
||||||
}
|
}
|
||||||
if err := logEvent(ctx, q, incidentID, evResolved, nil, nil, nil); err != nil {
|
if err := logEvent(ctx, q, incidentID, evResolved, nil, nil, nil, nil); err != nil {
|
||||||
return false, err
|
return false, err
|
||||||
}
|
}
|
||||||
// The all-clear goes only to whoever was paged in the first place, which
|
// The all-clear goes only to whoever was paged in the first place, which
|
||||||
@@ -247,16 +294,30 @@ func resolveIfSettled(ctx context.Context, q querier, incidentID int64) (bool, e
|
|||||||
return true, enqueueResolved(ctx, q, incidentID)
|
return true, enqueueResolved(ctx, q, incidentID)
|
||||||
}
|
}
|
||||||
|
|
||||||
// acknowledgeIncident records that userID has picked an incident up, and reports
|
// acknowledgeIncident records that userID — a human — has picked an incident
|
||||||
// whether it changed anything — an already-resolved incident is left alone.
|
// up, and reports whether it changed anything — an already-resolved or
|
||||||
// Shared by the authenticated handler and the Acknowledge button in a push
|
// already-acknowledged incident is left alone, so a second acknowledge (a
|
||||||
// notification, so both write the same state and the same timeline entry.
|
// retried request, or a stale push notification tapped after the web UI
|
||||||
|
// already acked it) is a no-op rather than a second "acknowledged" timeline
|
||||||
|
// entry. Used only by the Acknowledge button in a push notification
|
||||||
|
// (notify_ack.go), which always resolves a human from
|
||||||
|
// incident_ack_tokens.user_id — there is no service-account equivalent of
|
||||||
|
// that flow, so this keeps its human-only signature; the authenticated
|
||||||
|
// handler goes through acknowledgeIncidentAs below instead.
|
||||||
func acknowledgeIncident(ctx context.Context, q querier, incidentID, userID int64) (bool, error) {
|
func acknowledgeIncident(ctx context.Context, q querier, incidentID, userID int64) (bool, error) {
|
||||||
|
return acknowledgeIncidentAs(ctx, q, incidentID, &userID, nil)
|
||||||
|
}
|
||||||
|
|
||||||
|
// acknowledgeIncidentAs is acknowledgeIncident generalized to either actor
|
||||||
|
// kind. userID and serviceAccountID are mutually exclusive and nilable the
|
||||||
|
// same way logEvent's are (see incidents_ack_actor_xor_chk, migration 015).
|
||||||
|
func acknowledgeIncidentAs(ctx context.Context, q querier, incidentID int64, userID, serviceAccountID *int64) (bool, error) {
|
||||||
res, err := q.ExecContext(ctx, `
|
res, err := q.ExecContext(ctx, `
|
||||||
UPDATE incidents
|
UPDATE incidents
|
||||||
SET status = 'acknowledged', acknowledged_by = $1, acknowledged_at = $2
|
SET status = 'acknowledged', acknowledged_by = $1, acknowledged_by_service_account_id = $2,
|
||||||
WHERE id = $3 AND resolved_at IS NULL`,
|
acknowledged_at = $3
|
||||||
userID, time.Now().Unix(), incidentID)
|
WHERE id = $4 AND status = 'triggered'`,
|
||||||
|
userID, serviceAccountID, time.Now().Unix(), incidentID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return false, err
|
return false, err
|
||||||
}
|
}
|
||||||
@@ -267,7 +328,7 @@ func acknowledgeIncident(ctx context.Context, q querier, incidentID, userID int6
|
|||||||
if err := stopEscalation(ctx, q, incidentID); err != nil {
|
if err := stopEscalation(ctx, q, incidentID); err != nil {
|
||||||
return false, err
|
return false, err
|
||||||
}
|
}
|
||||||
return true, logEvent(ctx, q, incidentID, evAcknowledged, &userID, nil, nil)
|
return true, logEvent(ctx, q, incidentID, evAcknowledged, userID, serviceAccountID, nil, nil)
|
||||||
}
|
}
|
||||||
|
|
||||||
// openIncidentForAlert returns the open incident an alert currently belongs to,
|
// openIncidentForAlert returns the open incident an alert currently belongs to,
|
||||||
@@ -286,6 +347,34 @@ func openIncidentForAlert(ctx context.Context, q querier, alertID int64) (int64,
|
|||||||
return id, err
|
return id, err
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// volatileLabels say where a problem ran this time, not what the problem is, so
|
||||||
|
// they stay out of the signature. Migration 008's backfill lists the same set.
|
||||||
|
var volatileLabels = map[string]bool{
|
||||||
|
"instance": true, "pod": true, "pod_name": true, "pod_ip": true,
|
||||||
|
"container": true, "container_name": true, "endpoint": true,
|
||||||
|
}
|
||||||
|
|
||||||
|
// incidentSignature identifies "the same problem" across incidents: the alert
|
||||||
|
// name plus the stable group labels, sorted. Incidents in one team with equal
|
||||||
|
// signatures are what the similar-incidents lookup returns. title stands in for
|
||||||
|
// the name when the payload carried no alertname (groupless and dead man's
|
||||||
|
// switch incidents).
|
||||||
|
func incidentSignature(groupLabels map[string]string, title string) string {
|
||||||
|
name := groupLabels["alertname"]
|
||||||
|
if name == "" {
|
||||||
|
name = title
|
||||||
|
}
|
||||||
|
rest := make([]string, 0, len(groupLabels))
|
||||||
|
for k, v := range groupLabels {
|
||||||
|
if k == "alertname" || volatileLabels[k] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
rest = append(rest, k+"="+v)
|
||||||
|
}
|
||||||
|
sort.Strings(rest)
|
||||||
|
return name + "|" + strings.Join(rest, ",")
|
||||||
|
}
|
||||||
|
|
||||||
// incidentTitle renders a human-readable title from Alertmanager's groupLabels,
|
// incidentTitle renders a human-readable title from Alertmanager's groupLabels,
|
||||||
// leading with the alert name and appending whatever else the operator grouped
|
// leading with the alert name and appending whatever else the operator grouped
|
||||||
// by. Falls back to the alert's own name when the payload carried no groupLabels.
|
// by. Falls back to the alert's own name when the payload carried no groupLabels.
|
||||||
|
|||||||
@@ -3,6 +3,7 @@ package api
|
|||||||
import (
|
import (
|
||||||
"database/sql"
|
"database/sql"
|
||||||
"fmt"
|
"fmt"
|
||||||
|
"io"
|
||||||
"net/http"
|
"net/http"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
@@ -12,6 +13,52 @@ import (
|
|||||||
"github.com/go-chi/chi/v5"
|
"github.com/go-chi/chi/v5"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// handleListClusters answers GET /api/incidents/clusters: the distinct values of
|
||||||
|
// the origin label across the caller's incidents from the last 90 days, sorted,
|
||||||
|
// so the queue can offer them as a filter. Optional team_id narrows it to one
|
||||||
|
// team. Empty when nothing carries the label, which is how the UI knows to show
|
||||||
|
// no filter at all.
|
||||||
|
func handleListClusters(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
args := &sqlArgs{}
|
||||||
|
where := []string{
|
||||||
|
"team_id = ANY(" + args.add(callerTeamIDs(r.Context())) + ")",
|
||||||
|
"triggered_at >= " + args.add(time.Now().AddDate(0, 0, -90).Unix()),
|
||||||
|
}
|
||||||
|
if team := r.URL.Query().Get("team_id"); team != "" {
|
||||||
|
if n, err := strconv.ParseInt(team, 10, 64); err == nil {
|
||||||
|
where = append(where, "team_id = "+args.add(n))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
label := args.add(originLabel)
|
||||||
|
|
||||||
|
rows, err := db.QueryContext(r.Context(),
|
||||||
|
fmt.Sprintf("SELECT DISTINCT group_labels->>%[1]s AS v FROM incidents WHERE %[2]s AND group_labels->>%[1]s <> '' ORDER BY v LIMIT 200",
|
||||||
|
label, strings.Join(where, " AND ")),
|
||||||
|
args.all()...)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer rows.Close()
|
||||||
|
|
||||||
|
clusters := []string{}
|
||||||
|
for rows.Next() {
|
||||||
|
var v string
|
||||||
|
if err := rows.Scan(&v); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
clusters = append(clusters, v)
|
||||||
|
}
|
||||||
|
if err := rows.Err(); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusOK, clusters)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func handleListIncidents(db *sql.DB) http.HandlerFunc {
|
func handleListIncidents(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
q := r.URL.Query()
|
q := r.URL.Query()
|
||||||
@@ -52,6 +99,12 @@ func handleListIncidents(db *sql.DB) http.HandlerFunc {
|
|||||||
if severity := q.Get("severity"); severity != "" {
|
if severity := q.Get("severity"); severity != "" {
|
||||||
where = append(where, "i.severity = "+args.add(severity))
|
where = append(where, "i.severity = "+args.add(severity))
|
||||||
}
|
}
|
||||||
|
// Where it came from: the value of the origin label (originLabel, by
|
||||||
|
// convention "cluster") among the incident's group labels. An incident has
|
||||||
|
// it only when the label is in Alertmanager's group_by.
|
||||||
|
if cluster := q.Get("cluster"); cluster != "" {
|
||||||
|
where = append(where, "i.group_labels->>"+args.add(originLabel)+" = "+args.add(cluster))
|
||||||
|
}
|
||||||
if assignee := q.Get("assigned_to"); assignee != "" {
|
if assignee := q.Get("assigned_to"); assignee != "" {
|
||||||
if n, err := strconv.ParseInt(assignee, 10, 64); err == nil {
|
if n, err := strconv.ParseInt(assignee, 10, 64); err == nil {
|
||||||
where = append(where, "i.assigned_to = "+args.add(n))
|
where = append(where, "i.assigned_to = "+args.add(n))
|
||||||
@@ -85,7 +138,7 @@ func handleListIncidents(db *sql.DB) http.HandlerFunc {
|
|||||||
incidentSelectFrom, strings.Join(where, " AND "), order, args.add(limit)),
|
incidentSelectFrom, strings.Join(where, " AND "), order, args.add(limit)),
|
||||||
args.all()...)
|
args.all()...)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -94,11 +147,15 @@ func handleListIncidents(db *sql.DB) http.HandlerFunc {
|
|||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
i, err := scanIncident(rows)
|
i, err := scanIncident(rows)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
incidents = append(incidents, i)
|
incidents = append(incidents, i)
|
||||||
}
|
}
|
||||||
|
if err := rows.Err(); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
respond(w, http.StatusOK, incidents)
|
respond(w, http.StatusOK, incidents)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -115,35 +172,17 @@ func handleGetIncident(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if inc.Alerts, err = incidentAlerts(r, db, id); err != nil {
|
if inc.Alerts, err = incidentAlerts(r, db, id); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, inc)
|
respond(w, http.StatusOK, inc)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func handleIncidentAlerts(db *sql.DB) http.HandlerFunc {
|
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
|
||||||
id, ok := incidentIDParam(w, r, db)
|
|
||||||
if !ok {
|
|
||||||
return
|
|
||||||
}
|
|
||||||
if !incidentExists(w, r, db, id) {
|
|
||||||
return
|
|
||||||
}
|
|
||||||
alerts, err := incidentAlerts(r, db, id)
|
|
||||||
if err != nil {
|
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
|
||||||
return
|
|
||||||
}
|
|
||||||
respond(w, http.StatusOK, alerts)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func handleIncidentTimeline(db *sql.DB) http.HandlerFunc {
|
func handleIncidentTimeline(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
id, ok := incidentIDParam(w, r, db)
|
id, ok := incidentIDParam(w, r, db)
|
||||||
@@ -156,13 +195,19 @@ func handleIncidentTimeline(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
rows, err := db.QueryContext(r.Context(), `
|
rows, err := db.QueryContext(r.Context(), `
|
||||||
SELECT e.id, e.incident_id, e.type, e.user_id, u.username,
|
SELECT e.id, e.incident_id, e.type, e.user_id, u.username,
|
||||||
|
e.service_account_id, sa.name,
|
||||||
|
e.actor_user_id, au.username,
|
||||||
|
e.actor_service_account_id, asa.name,
|
||||||
e.alert_id, e.detail, e.created_at
|
e.alert_id, e.detail, e.created_at
|
||||||
FROM incident_events e
|
FROM incident_events e
|
||||||
LEFT JOIN users u ON u.id = e.user_id
|
LEFT JOIN users u ON u.id = e.user_id
|
||||||
|
LEFT JOIN service_accounts sa ON sa.id = e.service_account_id
|
||||||
|
LEFT JOIN users au ON au.id = e.actor_user_id
|
||||||
|
LEFT JOIN service_accounts asa ON asa.id = e.actor_service_account_id
|
||||||
WHERE e.incident_id = $1
|
WHERE e.incident_id = $1
|
||||||
ORDER BY e.created_at ASC, e.id ASC`, id)
|
ORDER BY e.created_at ASC, e.id ASC`, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -172,13 +217,20 @@ func handleIncidentTimeline(db *sql.DB) http.HandlerFunc {
|
|||||||
var e models.IncidentEvent
|
var e models.IncidentEvent
|
||||||
var ts int64
|
var ts int64
|
||||||
if err := rows.Scan(&e.ID, &e.IncidentID, &e.Type, &e.UserID, &e.Username,
|
if err := rows.Scan(&e.ID, &e.IncidentID, &e.Type, &e.UserID, &e.Username,
|
||||||
|
&e.ServiceAccountID, &e.ServiceAccountName,
|
||||||
|
&e.ActorUserID, &e.ActorUsername,
|
||||||
|
&e.ActorServiceAccountID, &e.ActorServiceAccountName,
|
||||||
&e.AlertID, &e.Detail, &ts); err != nil {
|
&e.AlertID, &e.Detail, &ts); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
e.CreatedAt = time.Unix(ts, 0).UTC()
|
e.CreatedAt = time.Unix(ts, 0).UTC()
|
||||||
events = append(events, e)
|
events = append(events, e)
|
||||||
}
|
}
|
||||||
|
if err := rows.Err(); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
respond(w, http.StatusOK, events)
|
respond(w, http.StatusOK, events)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -189,17 +241,27 @@ func handleIncidentAcknowledge(db *sql.DB) http.HandlerFunc {
|
|||||||
if !ok {
|
if !ok {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
user, _ := userFromContext(r.Context())
|
userID, saID := callerActorIDs(r.Context())
|
||||||
acked, err := acknowledgeIncident(r.Context(), db, id, user.ID)
|
acked, err := acknowledgeIncidentAs(r.Context(), db, id, userID, saID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if !acked {
|
if !acked {
|
||||||
if !incidentExists(w, r, db, id) {
|
// incidentIDParam above already confirmed the incident exists, so this
|
||||||
|
// is either resolved, or already acknowledged — the latter is now a
|
||||||
|
// no-op rather than an error, since the caller's desired state
|
||||||
|
// (acknowledged) already holds.
|
||||||
|
inc, err := fetchIncident(r.Context(), db, id)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusConflict, errResp("incident is resolved"))
|
if inc.Status == "resolved" {
|
||||||
|
respond(w, http.StatusConflict, errResp("incident is resolved"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusOK, inc)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respondIncident(w, r, db, id)
|
respondIncident(w, r, db, id)
|
||||||
@@ -212,14 +274,15 @@ func handleIncidentUnacknowledge(db *sql.DB) http.HandlerFunc {
|
|||||||
if !ok {
|
if !ok {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
user, _ := userFromContext(r.Context())
|
userID, saID := callerActorIDs(r.Context())
|
||||||
if !updateOpenIncident(w, r, db, id,
|
if !updateOpenIncident(w, r, db, id,
|
||||||
`UPDATE incidents SET status = 'triggered', acknowledged_by = NULL, acknowledged_at = NULL
|
`UPDATE incidents SET status = 'triggered', acknowledged_by = NULL,
|
||||||
|
acknowledged_by_service_account_id = NULL, acknowledged_at = NULL
|
||||||
WHERE id = $1 AND resolved_at IS NULL`, id) {
|
WHERE id = $1 AND resolved_at IS NULL`, id) {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err := logEvent(r.Context(), db, id, evUnacknowledged, &user.ID, nil, nil); err != nil {
|
if err := logEvent(r.Context(), db, id, evUnacknowledged, userID, saID, nil, nil); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
w.WriteHeader(http.StatusNoContent)
|
w.WriteHeader(http.StatusNoContent)
|
||||||
@@ -236,7 +299,16 @@ func handleIncidentResolve(db *sql.DB) http.HandlerFunc {
|
|||||||
if !ok {
|
if !ok {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
user, _ := userFromContext(r.Context())
|
userID, saID := callerActorIDs(r.Context())
|
||||||
|
// The body is optional: clients that predate resolution notes send none.
|
||||||
|
var req struct {
|
||||||
|
Resolution string `json:"resolution"`
|
||||||
|
}
|
||||||
|
if err := decodeJSON(r, &req); err != nil && err != io.EOF {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
req.Resolution = strings.TrimSpace(req.Resolution)
|
||||||
if !updateOpenIncident(w, r, db, id,
|
if !updateOpenIncident(w, r, db, id,
|
||||||
`UPDATE incidents SET status = 'resolved', resolved_at = $1, resolution_source = $2
|
`UPDATE incidents SET status = 'resolved', resolved_at = $1, resolution_source = $2
|
||||||
WHERE id = $3 AND resolved_at IS NULL`,
|
WHERE id = $3 AND resolved_at IS NULL`,
|
||||||
@@ -245,13 +317,19 @@ func handleIncidentResolve(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
// A person closing an incident is the clearest possible "I have this".
|
// A person closing an incident is the clearest possible "I have this".
|
||||||
if err := stopEscalation(r.Context(), db, id); err != nil {
|
if err := stopEscalation(r.Context(), db, id); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err := logEvent(r.Context(), db, id, evResolved, &user.ID, nil, nil); err != nil {
|
if err := logEvent(r.Context(), db, id, evResolved, userID, saID, nil, nil); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
if req.Resolution != "" {
|
||||||
|
if err := logEvent(r.Context(), db, id, evResolutionNote, userID, saID, nil, &req.Resolution); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
}
|
||||||
respondIncident(w, r, db, id)
|
respondIncident(w, r, db, id)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -285,9 +363,11 @@ func handleIncidentAssign(db *sql.DB) http.HandlerFunc {
|
|||||||
req.UserID, id) {
|
req.UserID, id) {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
// On an "assigned" event user_id is the assignee, not the actor.
|
// On an "assigned" event user_id is the assignee; the actor goes in
|
||||||
if err := logEvent(r.Context(), db, id, evAssigned, &req.UserID, nil, nil); err != nil {
|
// the actor_* columns.
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
actorUserID, actorSAID := callerActorIDs(r.Context())
|
||||||
|
if err := logAssignedEvent(r.Context(), db, id, req.UserID, actorUserID, actorSAID); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respondIncident(w, r, db, id)
|
respondIncident(w, r, db, id)
|
||||||
@@ -337,15 +417,15 @@ func handleIncidentSnooze(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
user, _ := userFromContext(r.Context())
|
userID, saID := callerActorIDs(r.Context())
|
||||||
if !updateOpenIncident(w, r, db, id,
|
if !updateOpenIncident(w, r, db, id,
|
||||||
"UPDATE incidents SET snoozed_until = $1 WHERE id = $2 AND resolved_at IS NULL",
|
"UPDATE incidents SET snoozed_until = $1 WHERE id = $2 AND resolved_at IS NULL",
|
||||||
until.Unix(), id) {
|
until.Unix(), id) {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
detail := until.UTC().Format(time.RFC3339)
|
detail := until.UTC().Format(time.RFC3339)
|
||||||
if err := logEvent(r.Context(), db, id, evSnoozed, &user.ID, nil, &detail); err != nil {
|
if err := logEvent(r.Context(), db, id, evSnoozed, userID, saID, nil, &detail); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respondIncident(w, r, db, id)
|
respondIncident(w, r, db, id)
|
||||||
@@ -358,13 +438,13 @@ func handleIncidentUnsnooze(db *sql.DB) http.HandlerFunc {
|
|||||||
if !ok {
|
if !ok {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
user, _ := userFromContext(r.Context())
|
userID, saID := callerActorIDs(r.Context())
|
||||||
if !updateOpenIncident(w, r, db, id,
|
if !updateOpenIncident(w, r, db, id,
|
||||||
"UPDATE incidents SET snoozed_until = NULL WHERE id = $1 AND resolved_at IS NULL", id) {
|
"UPDATE incidents SET snoozed_until = NULL WHERE id = $1 AND resolved_at IS NULL", id) {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if err := logEvent(r.Context(), db, id, evUnsnoozed, &user.ID, nil, nil); err != nil {
|
if err := logEvent(r.Context(), db, id, evUnsnoozed, userID, saID, nil, nil); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
w.WriteHeader(http.StatusNoContent)
|
w.WriteHeader(http.StatusNoContent)
|
||||||
@@ -380,13 +460,18 @@ func handleIncidentArchive(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"UPDATE incidents SET archived_at = "+nowEpoch+" WHERE id = $1", id)
|
"UPDATE incidents SET archived_at = "+nowEpoch+" WHERE id = $1", id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
respond(w, http.StatusNotFound, errResp("incident not found"))
|
respond(w, http.StatusNotFound, errResp("incident not found"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
userID, saID := callerActorIDs(r.Context())
|
||||||
|
if err := logEvent(r.Context(), db, id, evArchived, userID, saID, nil, nil); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
respondIncident(w, r, db, id)
|
respondIncident(w, r, db, id)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -400,13 +485,18 @@ func handleIncidentUnarchive(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"UPDATE incidents SET archived_at = NULL WHERE id = $1", id)
|
"UPDATE incidents SET archived_at = NULL WHERE id = $1", id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
respond(w, http.StatusNotFound, errResp("incident not found"))
|
respond(w, http.StatusNotFound, errResp("incident not found"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
userID, saID := callerActorIDs(r.Context())
|
||||||
|
if err := logEvent(r.Context(), db, id, evUnarchived, userID, saID, nil, nil); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
w.WriteHeader(http.StatusNoContent)
|
w.WriteHeader(http.StatusNoContent)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -421,11 +511,17 @@ func handleCreateNote(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
var req struct {
|
var req struct {
|
||||||
Content string `json:"content"`
|
Content string `json:"content"`
|
||||||
|
// Pinned files the note as the resolution note: what fixed it.
|
||||||
|
Pinned bool `json:"pinned"`
|
||||||
}
|
}
|
||||||
if err := decodeJSON(r, &req); err != nil {
|
if err := decodeJSON(r, &req); err != nil {
|
||||||
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
noteType := evNote
|
||||||
|
if req.Pinned {
|
||||||
|
noteType = evResolutionNote
|
||||||
|
}
|
||||||
if req.Content == "" {
|
if req.Content == "" {
|
||||||
respond(w, http.StatusBadRequest, errResp("content is required"))
|
respond(w, http.StatusBadRequest, errResp("content is required"))
|
||||||
return
|
return
|
||||||
@@ -434,27 +530,32 @@ func handleCreateNote(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
user, _ := userFromContext(r.Context())
|
caller, _ := callerFromContext(r.Context())
|
||||||
|
userID, saID := callerActorIDs(r.Context())
|
||||||
now := time.Now()
|
now := time.Now()
|
||||||
var eventID int64
|
var eventID int64
|
||||||
err := db.QueryRowContext(r.Context(), `
|
err := db.QueryRowContext(r.Context(), `
|
||||||
INSERT INTO incident_events (incident_id, type, user_id, detail, created_at)
|
INSERT INTO incident_events (incident_id, type, user_id, service_account_id, detail, created_at)
|
||||||
VALUES ($1, $2, $3, $4, $5)
|
VALUES ($1, $2, $3, $4, $5, $6)
|
||||||
RETURNING id`, id, evNote, user.ID, req.Content, now.Unix()).Scan(&eventID)
|
RETURNING id`, id, noteType, userID, saID, req.Content, now.Unix()).Scan(&eventID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
respond(w, http.StatusCreated, models.IncidentEvent{
|
resp := models.IncidentEvent{
|
||||||
ID: eventID,
|
ID: eventID,
|
||||||
IncidentID: id,
|
IncidentID: id,
|
||||||
Type: evNote,
|
Type: noteType,
|
||||||
UserID: &user.ID,
|
|
||||||
Username: &user.Username,
|
|
||||||
Detail: &req.Content,
|
Detail: &req.Content,
|
||||||
CreatedAt: now.UTC().Truncate(time.Second),
|
CreatedAt: now.UTC().Truncate(time.Second),
|
||||||
})
|
}
|
||||||
|
if u, ok := caller.AsHuman(); ok {
|
||||||
|
resp.UserID, resp.Username = &u.ID, &u.Username
|
||||||
|
} else if saName, ok := caller.ServiceAccountName(); ok {
|
||||||
|
resp.ServiceAccountID, resp.ServiceAccountName = saID, &saName
|
||||||
|
}
|
||||||
|
respond(w, http.StatusCreated, resp)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -472,13 +573,14 @@ func handleDeleteNote(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
user, _ := userFromContext(r.Context())
|
userID, saID := callerActorIDs(r.Context())
|
||||||
res, err := db.ExecContext(r.Context(), `
|
res, err := db.ExecContext(r.Context(), `
|
||||||
DELETE FROM incident_events
|
DELETE FROM incident_events
|
||||||
WHERE id = $1 AND incident_id = $2 AND type = $3 AND user_id = $4`,
|
WHERE id = $1 AND incident_id = $2 AND type IN ($3, $4)
|
||||||
eventID, id, evNote, user.ID)
|
AND (user_id = $5 OR service_account_id = $6)`,
|
||||||
|
eventID, id, evNote, evResolutionNote, userID, saID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -532,7 +634,7 @@ func incidentExists(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64
|
|||||||
func updateOpenIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64, query string, args ...any) bool {
|
func updateOpenIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64, query string, args ...any) bool {
|
||||||
res, err := db.ExecContext(r.Context(), query, args...)
|
res, err := db.ExecContext(r.Context(), query, args...)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n > 0 {
|
if n, _ := res.RowsAffected(); n > 0 {
|
||||||
@@ -548,7 +650,7 @@ func updateOpenIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id i
|
|||||||
func respondIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64) {
|
func respondIncident(w http.ResponseWriter, r *http.Request, db *sql.DB, id int64) {
|
||||||
inc, err := fetchIncident(r.Context(), db, id)
|
inc, err := fetchIncident(r.Context(), db, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, inc)
|
respond(w, http.StatusOK, inc)
|
||||||
|
|||||||
@@ -4,10 +4,12 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"fmt"
|
"fmt"
|
||||||
"net/http"
|
"net/http"
|
||||||
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||||
)
|
)
|
||||||
|
|
||||||
// amAlert builds one alert of a webhook payload.
|
// amAlert builds one alert of a webhook payload.
|
||||||
@@ -338,6 +340,48 @@ func TestIncident_Acknowledge(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A second acknowledge — a retried request, or a stale push notification
|
||||||
|
// tapped after the web UI already acked it — must be a no-op: same state,
|
||||||
|
// no second "acknowledged" timeline entry. Regression test for the bug
|
||||||
|
// described in issue #26 ("two acknowledged entries look like a bug").
|
||||||
|
func TestIncident_AcknowledgeTwiceIsIdempotent(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
postWebhook(t, s, []map[string]any{
|
||||||
|
amAlert("fp-ack2", "Y", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||||
|
})
|
||||||
|
|
||||||
|
resp := s.req(t, http.MethodPost, "/api/incidents/1/acknowledge", nil)
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Fatalf("first acknowledge returned %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
var first map[string]any
|
||||||
|
decode(t, resp, &first)
|
||||||
|
|
||||||
|
resp = s.req(t, http.MethodPost, "/api/incidents/1/acknowledge", nil)
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Fatalf("second acknowledge returned %d, want 200 (idempotent)", resp.StatusCode)
|
||||||
|
}
|
||||||
|
var second map[string]any
|
||||||
|
decode(t, resp, &second)
|
||||||
|
if second["status"] != "acknowledged" {
|
||||||
|
t.Errorf("expected status still acknowledged, got %v", second["status"])
|
||||||
|
}
|
||||||
|
if second["acknowledged_by"] != first["acknowledged_by"] {
|
||||||
|
t.Errorf("expected the same acknowledged_by, got %v then %v", first["acknowledged_by"], second["acknowledged_by"])
|
||||||
|
}
|
||||||
|
|
||||||
|
types := eventTypes(timeline(t, s, 1))
|
||||||
|
n := 0
|
||||||
|
for _, ty := range types {
|
||||||
|
if ty == "acknowledged" {
|
||||||
|
n++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if n != 1 {
|
||||||
|
t.Errorf("expected exactly one acknowledged event, got %d in %v", n, types)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestIncident_ManualResolveIsTerminal(t *testing.T) {
|
func TestIncident_ManualResolveIsTerminal(t *testing.T) {
|
||||||
s := newTS(t)
|
s := newTS(t)
|
||||||
postWebhook(t, s, []map[string]any{
|
postWebhook(t, s, []map[string]any{
|
||||||
@@ -600,6 +644,106 @@ func TestIncident_ArchiveRoundTrip(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Service accounts (terdut-server#25)
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
// TestServiceAccount_CanActOnItsTeamsIncidents is #25's regression test.
|
||||||
|
// Before the fix: acknowledge/resolve/snooze/create-note each 500'd (writing
|
||||||
|
// acknowledged_by/user_id = 0, violating the users(id) FK for a service
|
||||||
|
// account), and delete-note silently matched zero rows (WHERE user_id = 0)
|
||||||
|
// instead of deleting.
|
||||||
|
func TestServiceAccount_CanActOnItsTeamsIncidents(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
instanceKey := createServiceAccount(t, s, s.key, "operator", models.ServiceAccountScopeInstance, 0)
|
||||||
|
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||||
|
keyA := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||||
|
|
||||||
|
var integration struct {
|
||||||
|
Key string `json:"key"`
|
||||||
|
}
|
||||||
|
decode(t, s.reqAs(t, keyA, http.MethodPost, "/api/teams/"+id64(teamA)+"/integrations",
|
||||||
|
map[string]string{"name": "test"}), &integration)
|
||||||
|
postToIntegration(t, s, integration.Key, "fp-sa", "SAIncident") // incident 1
|
||||||
|
|
||||||
|
// Acknowledge.
|
||||||
|
resp := s.reqAs(t, keyA, http.MethodPost, "/api/incidents/1/acknowledge", nil)
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Fatalf("service account acknowledge: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
var inc map[string]any
|
||||||
|
decode(t, resp, &inc)
|
||||||
|
if inc["acknowledged_by_service_account_id"] == nil {
|
||||||
|
t.Error("expected acknowledged_by_service_account_id to be set")
|
||||||
|
}
|
||||||
|
if inc["acknowledged_by_id"] != nil {
|
||||||
|
t.Errorf("expected acknowledged_by_id to stay nil for a service-account actor, got %v", inc["acknowledged_by_id"])
|
||||||
|
}
|
||||||
|
|
||||||
|
// Unacknowledge.
|
||||||
|
resp = s.reqAs(t, keyA, http.MethodDelete, "/api/incidents/1/acknowledge", nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNoContent {
|
||||||
|
t.Errorf("service account unacknowledge: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Snooze, then unsnooze.
|
||||||
|
resp = s.reqAs(t, keyA, http.MethodPost, "/api/incidents/1/snooze",
|
||||||
|
map[string]string{"duration": "1h"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Errorf("service account snooze: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
resp = s.reqAs(t, keyA, http.MethodDelete, "/api/incidents/1/snooze", nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNoContent {
|
||||||
|
t.Errorf("service account unsnooze: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Create, then delete, a note.
|
||||||
|
var note map[string]any
|
||||||
|
decode(t, s.reqAs(t, keyA, http.MethodPost, "/api/incidents/1/notes",
|
||||||
|
map[string]string{"content": "looking into it"}), ¬e)
|
||||||
|
if note["service_account_id"] == nil {
|
||||||
|
t.Error("expected service_account_id on the note event")
|
||||||
|
}
|
||||||
|
if note["user_id"] != nil {
|
||||||
|
t.Errorf("expected no user_id on a service-account note, got %v", note["user_id"])
|
||||||
|
}
|
||||||
|
noteID := int(note["id"].(float64))
|
||||||
|
delResp := s.reqAs(t, keyA, http.MethodDelete, fmt.Sprintf("/api/incidents/1/notes/%d", noteID), nil)
|
||||||
|
delResp.Body.Close()
|
||||||
|
if delResp.StatusCode != http.StatusNoContent {
|
||||||
|
t.Errorf("service account deleting its own note: %d", delResp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Resolve.
|
||||||
|
resp = s.reqAs(t, keyA, http.MethodPost, "/api/incidents/1/resolve", nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Errorf("service account resolve: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Regression guard: a human actor must still write only the human columns,
|
||||||
|
// unaffected by the service-account branch added above.
|
||||||
|
func TestIncident_AcknowledgeStillWritesOnlyHumanColumn(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
postWebhook(t, s, []map[string]any{
|
||||||
|
amAlert("fp-human-ack", "Z", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||||
|
})
|
||||||
|
|
||||||
|
var inc map[string]any
|
||||||
|
decode(t, s.req(t, http.MethodPost, "/api/incidents/1/acknowledge", nil), &inc)
|
||||||
|
if inc["acknowledged_by_id"] == nil {
|
||||||
|
t.Error("expected acknowledged_by_id to be set for a human actor")
|
||||||
|
}
|
||||||
|
if inc["acknowledged_by_service_account_id"] != nil {
|
||||||
|
t.Errorf("expected acknowledged_by_service_account_id to stay nil for a human actor, got %v",
|
||||||
|
inc["acknowledged_by_service_account_id"])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestSweeper_ArchivesResolvedIncidents(t *testing.T) {
|
func TestSweeper_ArchivesResolvedIncidents(t *testing.T) {
|
||||||
s := newTS(t)
|
s := newTS(t)
|
||||||
postWebhook(t, s, []map[string]any{
|
postWebhook(t, s, []map[string]any{
|
||||||
@@ -650,8 +794,7 @@ func TestStats_Incidents(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// An empty window is a report of zero, not a failure. SUM over no rows is NULL
|
// An empty window is a report of zero, not a failure. SUM over no rows is NULL
|
||||||
// in Postgres as it was in SQLite, and that used to come back as a 500 the
|
// in Postgres, and that used to come back as a 500 the moment every incident was archived — the state a quiet installation settles
|
||||||
// moment every incident was archived — the state a quiet installation settles
|
|
||||||
// into.
|
// into.
|
||||||
func TestStats_IncidentsEmptyWindowIsZeroNotAnError(t *testing.T) {
|
func TestStats_IncidentsEmptyWindowIsZeroNotAnError(t *testing.T) {
|
||||||
s := newTS(t)
|
s := newTS(t)
|
||||||
@@ -727,3 +870,139 @@ func contains(haystack []string, needle string) bool {
|
|||||||
}
|
}
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Actor on assign / archive / unarchive (terdut-server#35)
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
// lastEvent returns the newest timeline event of the given type.
|
||||||
|
func lastEvent(t *testing.T, events []map[string]any, typ string) map[string]any {
|
||||||
|
t.Helper()
|
||||||
|
for i := len(events) - 1; i >= 0; i-- {
|
||||||
|
if events[i]["type"] == typ {
|
||||||
|
return events[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
t.Fatalf("no %q event in %v", typ, eventTypes(events))
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestIncident_AssignRecordsHumanActor(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
postWebhook(t, s, []map[string]any{
|
||||||
|
amAlert("fp-asg-actor", "Assignable", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||||
|
})
|
||||||
|
s.req(t, http.MethodPost, "/api/users",
|
||||||
|
map[string]string{"username": "alice", "email": "alice@test.com"}).Body.Close()
|
||||||
|
s.req(t, http.MethodPost, "/api/incidents/1/assign", map[string]any{"user_id": 2}).Body.Close()
|
||||||
|
|
||||||
|
ev := lastEvent(t, timeline(t, s, 1), "assigned")
|
||||||
|
if ev["username"] != "alice" {
|
||||||
|
t.Errorf("expected the assignee alice in username, got %v", ev["username"])
|
||||||
|
}
|
||||||
|
if ev["actor_user_id"] == nil || ev["actor_username"] == nil {
|
||||||
|
t.Errorf("expected the assigning human in actor_*, got %v", ev)
|
||||||
|
}
|
||||||
|
if ev["actor_service_account_id"] != nil {
|
||||||
|
t.Errorf("expected no service-account actor, got %v", ev["actor_service_account_id"])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestIncident_ArchiveUnarchiveRecordHumanActor(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
postWebhook(t, s, []map[string]any{
|
||||||
|
amAlert("fp-arc-actor", "Archivable", "firing", "2026-05-20T10:00:00Z", zeroTime, nil),
|
||||||
|
})
|
||||||
|
s.req(t, http.MethodPost, "/api/incidents/1/resolve", nil).Body.Close()
|
||||||
|
s.req(t, http.MethodPost, "/api/incidents/1/archive", nil).Body.Close()
|
||||||
|
s.req(t, http.MethodDelete, "/api/incidents/1/archive", nil).Body.Close()
|
||||||
|
|
||||||
|
events := timeline(t, s, 1)
|
||||||
|
for _, typ := range []string{"archived", "unarchived"} {
|
||||||
|
ev := lastEvent(t, events, typ)
|
||||||
|
if ev["user_id"] == nil || ev["service_account_id"] != nil {
|
||||||
|
t.Errorf("%s: expected only the human actor, got %v", typ, ev)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestServiceAccount_AssignArchiveUnarchiveRecordActor(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
instanceKey := createServiceAccount(t, s, s.key, "operator", models.ServiceAccountScopeInstance, 0)
|
||||||
|
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||||
|
keyA := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||||
|
var integration struct {
|
||||||
|
Key string `json:"key"`
|
||||||
|
}
|
||||||
|
decode(t, s.reqAs(t, keyA, http.MethodPost, "/api/teams/"+id64(teamA)+"/integrations",
|
||||||
|
map[string]string{"name": "test"}), &integration)
|
||||||
|
postToIntegration(t, s, integration.Key, "fp-sa-35", "SA35") // incident 1
|
||||||
|
|
||||||
|
resp := s.reqAs(t, keyA, http.MethodPost, "/api/incidents/1/assign", map[string]any{"user_id": 1})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Fatalf("service account assign: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
resp = s.reqAs(t, keyA, http.MethodPost, "/api/incidents/1/resolve", nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
resp = s.reqAs(t, keyA, http.MethodPost, "/api/incidents/1/archive", nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Fatalf("service account archive: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
resp = s.reqAs(t, keyA, http.MethodDelete, "/api/incidents/1/archive", nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNoContent {
|
||||||
|
t.Fatalf("service account unarchive: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
var events []map[string]any
|
||||||
|
decode(t, s.reqAs(t, keyA, http.MethodGet, "/api/incidents/1/timeline", nil), &events)
|
||||||
|
asg := lastEvent(t, events, "assigned")
|
||||||
|
if asg["actor_service_account_id"] == nil || asg["actor_user_id"] != nil {
|
||||||
|
t.Errorf("assigned: expected only the service-account actor, got %v", asg)
|
||||||
|
}
|
||||||
|
if asg["user_id"] == nil {
|
||||||
|
t.Errorf("assigned: user_id must stay the assignee, got %v", asg)
|
||||||
|
}
|
||||||
|
for _, typ := range []string{"archived", "unarchived"} {
|
||||||
|
ev := lastEvent(t, events, typ)
|
||||||
|
if ev["service_account_id"] == nil || ev["user_id"] != nil {
|
||||||
|
t.Errorf("%s: expected only the service-account actor, got %v", typ, ev)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The queue can be narrowed to one cluster, and the distinct clusters are
|
||||||
|
// offered so the filter has something to list.
|
||||||
|
func TestIncidents_FilterByCluster(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
|
||||||
|
for _, c := range []string{"prod-eu", "prod-us"} {
|
||||||
|
fireGroupedAs(t, s, "fp-"+c,
|
||||||
|
map[string]string{"alertname": "PodRestarting", "cluster": c, "namespace": "n"},
|
||||||
|
`{}:{alertname="PodRestarting",cluster="`+c+`",namespace="n"}`)
|
||||||
|
}
|
||||||
|
// A cluster-less incident exists too, and must never match a cluster filter.
|
||||||
|
fireGroupedAs(t, s, "fp-none", map[string]string{"alertname": "DiskFull"}, `{}:{alertname="DiskFull"}`)
|
||||||
|
|
||||||
|
if got := len(listIncidents(t, s, "")); got != 3 {
|
||||||
|
t.Fatalf("expected 3 open incidents, got %d", got)
|
||||||
|
}
|
||||||
|
eu := listIncidents(t, s, "?cluster=prod-eu")
|
||||||
|
if len(eu) != 1 {
|
||||||
|
t.Fatalf("expected 1 incident for prod-eu, got %d", len(eu))
|
||||||
|
}
|
||||||
|
if labels, _ := eu[0]["group_labels"].(map[string]any); labels["cluster"] != "prod-eu" {
|
||||||
|
t.Errorf("filtered to the wrong cluster: %v", eu[0]["group_labels"])
|
||||||
|
}
|
||||||
|
if got := len(listIncidents(t, s, "?cluster=nowhere")); got != 0 {
|
||||||
|
t.Errorf("an unknown cluster should match nothing, got %d", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
var clusters []string
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/incidents/clusters", nil), &clusters)
|
||||||
|
if want := "prod-eu,prod-us"; strings.Join(clusters, ",") != want {
|
||||||
|
t.Errorf("clusters = %v, want %s", clusters, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,154 @@
|
|||||||
|
package api_test
|
||||||
|
|
||||||
|
import (
|
||||||
|
"net/http"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||||
|
)
|
||||||
|
|
||||||
|
func testNotify() api.NotifyConfig {
|
||||||
|
return api.NotifyConfig{PublicURL: "https://terdut.example.com", RepeatEvery: 15 * time.Minute}
|
||||||
|
}
|
||||||
|
|
||||||
|
type memberView struct {
|
||||||
|
Username string `json:"username"`
|
||||||
|
Role string `json:"role"`
|
||||||
|
Status string `json:"status"`
|
||||||
|
OnCall bool `json:"on_call"`
|
||||||
|
NextShift *string `json:"next_shift"`
|
||||||
|
Pageable bool `json:"pageable"`
|
||||||
|
Problem string `json:"problem"`
|
||||||
|
LastActiveAt *string `json:"last_active_at"`
|
||||||
|
}
|
||||||
|
|
||||||
|
func readMembers(t *testing.T, s *ts) map[string]memberView {
|
||||||
|
t.Helper()
|
||||||
|
var list []memberView
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/members", nil), &list)
|
||||||
|
out := map[string]memberView{}
|
||||||
|
for _, m := range list {
|
||||||
|
out[m.Username] = m
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// The list says who is on call, who could not be woken, and who is merely
|
||||||
|
// there — and an on-call person who cannot be paged is the red one.
|
||||||
|
func TestMembers_StatusReflectsRotaAndPageability(t *testing.T) {
|
||||||
|
s, _ := notifyTS(t, testNotify()) // admin is on call today, with a topic
|
||||||
|
teamUser(t, s, "reachable", "terdut-reachable")
|
||||||
|
silent := teamUser(t, s, "silent", "terdut-silent")
|
||||||
|
s.exec(t, "UPDATE users SET ntfy_topic = NULL WHERE id = $1", silent)
|
||||||
|
|
||||||
|
got := readMembers(t, s)
|
||||||
|
if m := got["admin"]; m.Status != "oncall" || !m.OnCall || !m.Pageable {
|
||||||
|
t.Errorf("the person on call should read on call, got %+v", m)
|
||||||
|
}
|
||||||
|
if m := got["reachable"]; m.Status != "reachable" || m.OnCall {
|
||||||
|
t.Errorf("a member with a topic who is off the rota is reachable, got %+v", m)
|
||||||
|
}
|
||||||
|
if m := got["silent"]; m.Status != "unpageable" || m.Problem != "has no ntfy topic" {
|
||||||
|
t.Errorf("no topic means they cannot be paged, got %+v", m)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Being on call does not rescue an account that cannot be woken.
|
||||||
|
s.exec(t, "UPDATE users SET ntfy_topic = NULL WHERE username = 'admin'")
|
||||||
|
if m := readMembers(t, s)["admin"]; m.Status != "unpageable" || !m.OnCall {
|
||||||
|
t.Errorf("an on-call person with no topic is the red case, got %+v", m)
|
||||||
|
}
|
||||||
|
|
||||||
|
s.exec(t, "UPDATE users SET disabled_at = 1 WHERE id = $1", silent)
|
||||||
|
if m := readMembers(t, s)["silent"]; m.Problem != "account is disabled" {
|
||||||
|
t.Errorf("a disabled account should say so, got %+v", m)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The next shift is the next day after today, not today itself.
|
||||||
|
func TestMembers_NextShiftIsAfterToday(t *testing.T) {
|
||||||
|
s, _ := notifyTS(t, testNotify())
|
||||||
|
tomorrow := time.Now().UTC().AddDate(0, 0, 3).Format("2006-01-02")
|
||||||
|
resp := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/schedule",
|
||||||
|
map[string]any{"user_id": 1, "dates": []string{tomorrow}})
|
||||||
|
resp.Body.Close()
|
||||||
|
|
||||||
|
m := readMembers(t, s)["admin"]
|
||||||
|
if !m.OnCall || m.NextShift == nil || *m.NextShift != tomorrow {
|
||||||
|
t.Errorf("want on call today with the next shift on %s, got %+v", tomorrow, m)
|
||||||
|
}
|
||||||
|
teamUser(t, s, "idle", "terdut-idle")
|
||||||
|
if m := readMembers(t, s)["idle"]; m.NextShift != nil {
|
||||||
|
t.Errorf("somebody not on the rota has no next shift, got %v", *m.NextShift)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Last active is the newer of a session and an API key, and absent when neither
|
||||||
|
// has ever been used.
|
||||||
|
func TestMembers_LastActive(t *testing.T) {
|
||||||
|
s, _ := notifyTS(t, testNotify())
|
||||||
|
idle := teamUser(t, s, "idle", "terdut-idle")
|
||||||
|
|
||||||
|
if m := readMembers(t, s)["idle"]; m.LastActiveAt != nil {
|
||||||
|
t.Errorf("nobody has used idle's account, got %v", *m.LastActiveAt)
|
||||||
|
}
|
||||||
|
|
||||||
|
old := time.Now().Add(-48 * time.Hour).Unix()
|
||||||
|
s.exec(t, `INSERT INTO api_keys (user_id, key_hash, name, last_used_at) VALUES ($1, 'h1', 'k', $2)`, idle, old)
|
||||||
|
s.exec(t, `INSERT INTO sessions (token_hash, user_id, created_at, last_seen_at, expires_at)
|
||||||
|
VALUES ('h2', $1, $2, $3, $4)`, idle, old, old+3600, time.Now().Add(time.Hour).Unix())
|
||||||
|
|
||||||
|
m := readMembers(t, s)["idle"]
|
||||||
|
if m.LastActiveAt == nil {
|
||||||
|
t.Fatal("expected a last active time")
|
||||||
|
}
|
||||||
|
got, _ := time.Parse(time.RFC3339, *m.LastActiveAt)
|
||||||
|
if got.Unix() != old+3600 {
|
||||||
|
t.Errorf("last active should be the newer session (%d), got %d", old+3600, got.Unix())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The last owner can be neither removed nor demoted; with another owner in
|
||||||
|
// place, both are fine.
|
||||||
|
func TestMembers_LastOwnerIsProtected(t *testing.T) {
|
||||||
|
s, _ := notifyTS(t, testNotify())
|
||||||
|
tm := newTeam(t, s, "red")
|
||||||
|
base := "/api/teams/" + id64(tm.id) + "/members"
|
||||||
|
|
||||||
|
// Creating a team makes the creator an owner too; step the admin out so
|
||||||
|
// "red-user" is the only one left.
|
||||||
|
resp := s.req(t, http.MethodDelete, base+"/1", nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNoContent {
|
||||||
|
t.Fatalf("removing the creator: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
var members []map[string]any
|
||||||
|
decode(t, tm.call(http.MethodGet, base, nil), &members)
|
||||||
|
var owner int64
|
||||||
|
for _, m := range members {
|
||||||
|
if m["username"] == "red-user" {
|
||||||
|
owner = int64(m["user_id"].(float64))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
resp = tm.call(http.MethodPost, base, map[string]any{"user_id": owner, "role": "member"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusConflict {
|
||||||
|
t.Errorf("demoting the last owner: expected 409, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
resp = tm.call(http.MethodDelete, base+"/"+id64(owner), nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusConflict {
|
||||||
|
t.Errorf("removing the last owner: expected 409, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A second owner frees the first to step down.
|
||||||
|
resp = s.req(t, http.MethodPost, base, map[string]any{"user_id": 1, "role": "owner"})
|
||||||
|
resp.Body.Close()
|
||||||
|
resp = tm.call(http.MethodPost, base, map[string]any{"user_id": owner, "role": "member"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNoContent {
|
||||||
|
t.Errorf("demoting one of two owners: expected 204, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -5,19 +5,28 @@ import (
|
|||||||
"crypto/sha256"
|
"crypto/sha256"
|
||||||
"database/sql"
|
"database/sql"
|
||||||
"encoding/hex"
|
"encoding/hex"
|
||||||
|
"log"
|
||||||
"net/http"
|
"net/http"
|
||||||
"strings"
|
"strings"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
|
"github.com/go-chi/chi/v5"
|
||||||
|
"github.com/go-chi/chi/v5/middleware"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||||
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||||
)
|
)
|
||||||
|
|
||||||
type contextKey string
|
type contextKey string
|
||||||
|
|
||||||
const (
|
const (
|
||||||
ctxUser contextKey = "user"
|
// ctxCaller holds the one Caller (see caller.go) every authorization
|
||||||
|
// predicate in this package reads from — a human and a service account
|
||||||
|
// used to be two parallel, un-unified context keys (ctxUser/ctxTeams vs.
|
||||||
|
// ctxServiceAccount); this is why that was a mistake, not a smaller
|
||||||
|
// version of the same idea.
|
||||||
|
ctxCaller contextKey = "caller"
|
||||||
ctxSession contextKey = "session"
|
ctxSession contextKey = "session"
|
||||||
ctxTeams contextKey = "teams"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
// AuthMiddleware accepts either of the two credentials the server issues: an
|
// AuthMiddleware accepts either of the two credentials the server issues: an
|
||||||
@@ -39,12 +48,19 @@ func AuthMiddleware(db *sql.DB) func(http.Handler) http.Handler {
|
|||||||
respond(w, http.StatusUnauthorized, errResp("unauthorized"))
|
respond(w, http.StatusUnauthorized, errResp("unauthorized"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
userID, ok := apiKeyUser(r.Context(), db, token)
|
if userID, ok := apiKeyUser(r.Context(), db, token); ok {
|
||||||
if !ok {
|
serveAs(w, r, next, db, userID, 0)
|
||||||
respond(w, http.StatusUnauthorized, errResp("unauthorized"))
|
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
serveAs(w, r, next, db, userID, 0)
|
// Tried second, not first: a user API key is the common case,
|
||||||
|
// and a service-account key is visibly prefixed (tdsa_) so this
|
||||||
|
// second lookup is rarely reached on a request that was going
|
||||||
|
// to fail anyway.
|
||||||
|
if sa, ok := serviceAccountFor(r.Context(), db, token); ok {
|
||||||
|
serveAsServiceAccount(w, r, next, sa)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusUnauthorized, errResp("unauthorized"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -67,12 +83,34 @@ func AuthMiddleware(db *sql.DB) func(http.Handler) http.Handler {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// securityHeaders sets headers that cost nothing to send on every response,
|
||||||
|
// API or static site alike. nosniff is unconditional; HSTS only fires once
|
||||||
|
// cookieSecure's signal says the browser is actually looking at this server
|
||||||
|
// over HTTPS — TLS terminates at the gateway, which (as of this writing) sets
|
||||||
|
// neither header itself.
|
||||||
|
//
|
||||||
|
// max-age is 180 days rather than the usual year-plus: short enough that if
|
||||||
|
// HTTPS here ever broke for real, the header would age out of a browser's
|
||||||
|
// cache well within a release cycle instead of locking anyone out of a
|
||||||
|
// working server. Raise it once this has run clean for a while.
|
||||||
|
func securityHeaders(publicURL string) func(http.Handler) http.Handler {
|
||||||
|
return func(next http.Handler) http.Handler {
|
||||||
|
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
w.Header().Set("X-Content-Type-Options", "nosniff")
|
||||||
|
if cookieSecure(publicURL, r) {
|
||||||
|
w.Header().Set("Strict-Transport-Security", "max-age=15552000; includeSubDomains")
|
||||||
|
}
|
||||||
|
next.ServeHTTP(w, r)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// AdminOnly rejects a caller who is not a system administrator. It runs inside
|
// AdminOnly rejects a caller who is not a system administrator. It runs inside
|
||||||
// AuthMiddleware's group, so by the time it sees a request the caller is known.
|
// AuthMiddleware's group, so by the time it sees a request the caller is known.
|
||||||
//
|
//
|
||||||
// 403 and not 404: the route exists and the caller is authenticated, they are
|
// 403 and not 404: the route exists and the caller is authenticated, they are
|
||||||
// simply not allowed. Hiding the endpoint would buy nothing — every one of them
|
// simply not allowed. Hiding the endpoint would buy nothing — every one of them
|
||||||
// is in the README.
|
// is in docs/api.md.
|
||||||
func AdminOnly(next http.Handler) http.Handler {
|
func AdminOnly(next http.Handler) http.Handler {
|
||||||
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||||
caller, ok := userFromContext(r.Context())
|
caller, ok := userFromContext(r.Context())
|
||||||
@@ -101,19 +139,29 @@ func requireSelfOrAdmin(w http.ResponseWriter, r *http.Request, targetID int64)
|
|||||||
}
|
}
|
||||||
|
|
||||||
// apiKeyUser resolves an API key to its user and stamps its last use.
|
// apiKeyUser resolves an API key to its user and stamps its last use.
|
||||||
|
// expires_at IS NULL OR > now is part of the lookup itself, the same way
|
||||||
|
// serveAs's disabled_at check is: an expired key is one that cannot
|
||||||
|
// authenticate, by construction, rather than one that happens to still
|
||||||
|
// resolve and has to be caught afterwards.
|
||||||
func apiKeyUser(ctx context.Context, db *sql.DB, token string) (int64, bool) {
|
func apiKeyUser(ctx context.Context, db *sql.DB, token string) (int64, bool) {
|
||||||
var keyID, userID int64
|
var keyID, userID int64
|
||||||
|
var lastUsed sql.NullInt64
|
||||||
err := db.QueryRowContext(ctx,
|
err := db.QueryRowContext(ctx,
|
||||||
"SELECT id, user_id FROM api_keys WHERE key_hash = $1", hashToken(token),
|
`SELECT id, user_id, last_used_at FROM api_keys
|
||||||
).Scan(&keyID, &userID)
|
WHERE key_hash = $1 AND (expires_at IS NULL OR expires_at > $2)`,
|
||||||
|
hashToken(token), time.Now().Unix(),
|
||||||
|
).Scan(&keyID, &userID, &lastUsed)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return 0, false
|
return 0, false
|
||||||
}
|
}
|
||||||
|
|
||||||
// best-effort; don't fail the request if this update fails
|
// best-effort; don't fail the request if this update fails. Throttled like
|
||||||
db.ExecContext(ctx,
|
// the session expiry, so a polling client does not write a row per request.
|
||||||
"UPDATE api_keys SET last_used_at = $1 WHERE id = $2",
|
if now := time.Now(); !lastUsed.Valid || now.Sub(time.Unix(lastUsed.Int64, 0)) > keyTouchEvery {
|
||||||
time.Now().Unix(), keyID)
|
db.ExecContext(ctx,
|
||||||
|
"UPDATE api_keys SET last_used_at = $1 WHERE id = $2",
|
||||||
|
now.Unix(), keyID)
|
||||||
|
}
|
||||||
return userID, true
|
return userID, true
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -132,8 +180,13 @@ func sessionUser(ctx context.Context, db *sql.DB, token string) (sessionID, user
|
|||||||
}
|
}
|
||||||
|
|
||||||
if now.Sub(time.Unix(lastSeen, 0)) > sessionTouchEvery {
|
if now.Sub(time.Unix(lastSeen, 0)) > sessionTouchEvery {
|
||||||
db.ExecContext(ctx,
|
// LEAST keeps a capped session (a single sign-on login) from sliding
|
||||||
"UPDATE sessions SET last_seen_at = $1, expires_at = $2 WHERE id = $3",
|
// past its ceiling; with no ceiling COALESCE makes it the plain slide.
|
||||||
|
db.ExecContext(ctx, `
|
||||||
|
UPDATE sessions
|
||||||
|
SET last_seen_at = $1,
|
||||||
|
expires_at = LEAST($2::bigint, COALESCE(max_expires_at, $2::bigint))
|
||||||
|
WHERE id = $3`,
|
||||||
now.Unix(), now.Add(sessionTTL).Unix(), sessionID)
|
now.Unix(), now.Add(sessionTTL).Unix(), sessionID)
|
||||||
}
|
}
|
||||||
return sessionID, userID, true
|
return sessionID, userID, true
|
||||||
@@ -161,12 +214,11 @@ func serveAs(w http.ResponseWriter, r *http.Request, next http.Handler, db *sql.
|
|||||||
// table with one row per membership.
|
// table with one row per membership.
|
||||||
teams, err := callerMemberships(r.Context(), db, userID)
|
teams, err := callerMemberships(r.Context(), db, userID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
ctx := context.WithValue(r.Context(), ctxTeams, teams)
|
ctx := context.WithValue(r.Context(), ctxCaller, Caller{user: &u, memberships: teams})
|
||||||
ctx = context.WithValue(ctx, ctxUser, u)
|
|
||||||
if sessionID != 0 {
|
if sessionID != 0 {
|
||||||
ctx = context.WithValue(ctx, ctxSession, sessionID)
|
ctx = context.WithValue(ctx, ctxSession, sessionID)
|
||||||
}
|
}
|
||||||
@@ -178,9 +230,113 @@ func hashToken(token string) string {
|
|||||||
return hex.EncodeToString(h[:])
|
return hex.EncodeToString(h[:])
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// userFromContext returns the human behind the request, or false for a service
|
||||||
|
// account: Caller.AsHuman() on the request's Caller.
|
||||||
func userFromContext(ctx context.Context) (models.User, bool) {
|
func userFromContext(ctx context.Context) (models.User, bool) {
|
||||||
u, ok := ctx.Value(ctxUser).(models.User)
|
c, _ := callerFromContext(ctx)
|
||||||
return u, ok
|
return c.AsHuman()
|
||||||
|
}
|
||||||
|
|
||||||
|
// serviceAccountPrincipal is a service account as resolved from its key:
|
||||||
|
// enough to authorize requests, never the key itself.
|
||||||
|
type serviceAccountPrincipal struct {
|
||||||
|
id int64
|
||||||
|
name string
|
||||||
|
scope string
|
||||||
|
teamID int64 // meaningless (zero) for instance scope
|
||||||
|
}
|
||||||
|
|
||||||
|
// serviceAccountFor resolves a service-account key to its account and stamps
|
||||||
|
// its last use, the same shape apiKeyUser has for a user's own key.
|
||||||
|
func serviceAccountFor(ctx context.Context, db *sql.DB, token string) (serviceAccountPrincipal, bool) {
|
||||||
|
var sa serviceAccountPrincipal
|
||||||
|
var keyID int64
|
||||||
|
var teamID, lastUsed sql.NullInt64
|
||||||
|
err := db.QueryRowContext(ctx, `
|
||||||
|
SELECT k.id, a.id, a.name, a.scope, a.team_id, k.last_used_at
|
||||||
|
FROM service_account_keys k
|
||||||
|
JOIN service_accounts a ON a.id = k.service_account_id
|
||||||
|
WHERE k.key_hash = $1`, hashToken(token),
|
||||||
|
).Scan(&keyID, &sa.id, &sa.name, &sa.scope, &teamID, &lastUsed)
|
||||||
|
if err != nil {
|
||||||
|
return serviceAccountPrincipal{}, false
|
||||||
|
}
|
||||||
|
if teamID.Valid {
|
||||||
|
sa.teamID = teamID.Int64
|
||||||
|
}
|
||||||
|
|
||||||
|
// best-effort; don't fail the request if this update fails
|
||||||
|
if now := time.Now(); !lastUsed.Valid || now.Sub(time.Unix(lastUsed.Int64, 0)) > keyTouchEvery {
|
||||||
|
db.ExecContext(ctx,
|
||||||
|
"UPDATE service_account_keys SET last_used_at = $1 WHERE id = $2",
|
||||||
|
now.Unix(), keyID)
|
||||||
|
}
|
||||||
|
return sa, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// serveAsServiceAccount hands the request on with a service account's
|
||||||
|
// identity in context. A team-scoped account gets a single synthetic
|
||||||
|
// membership — owner of its own team, nothing else — which is what makes it
|
||||||
|
// satisfy requireTeamMember/requireTeamOwner exactly as a real owner would,
|
||||||
|
// without teaching either function about a second kind of caller. An
|
||||||
|
// instance-scoped account gets no memberships at all: it acts on teams by id,
|
||||||
|
// not by belonging to one.
|
||||||
|
//
|
||||||
|
// No CSRF check, for the same reason an API key needs none: a service-account
|
||||||
|
// key is only ever set by the client that holds it, never attached by a
|
||||||
|
// browser to a request another site makes.
|
||||||
|
func serveAsServiceAccount(w http.ResponseWriter, r *http.Request, next http.Handler, sa serviceAccountPrincipal) {
|
||||||
|
var memberships []membership
|
||||||
|
if sa.scope == models.ServiceAccountScopeTeam {
|
||||||
|
memberships = []membership{{teamID: sa.teamID, role: models.RoleOwner}}
|
||||||
|
}
|
||||||
|
ctx := context.WithValue(r.Context(), ctxCaller, Caller{sa: &sa, memberships: memberships})
|
||||||
|
next.ServeHTTP(w, r.WithContext(ctx))
|
||||||
|
}
|
||||||
|
|
||||||
|
// isInstanceServiceAccount is Caller.IsInstanceServiceAccount() on the
|
||||||
|
// request's Caller.
|
||||||
|
func isInstanceServiceAccount(ctx context.Context) bool {
|
||||||
|
c, _ := callerFromContext(ctx)
|
||||||
|
return c.IsInstanceServiceAccount()
|
||||||
|
}
|
||||||
|
|
||||||
|
// operatorReason marks a write that operator mode refused as such, distinct
|
||||||
|
// from every other 403 this server returns, so a client — the web UI or
|
||||||
|
// terdut-tui — can tell "you may not" from "this is managed elsewhere" and
|
||||||
|
// show the right message instead of a bare "forbidden".
|
||||||
|
const operatorReason = "operator_managed"
|
||||||
|
|
||||||
|
// OperatorModeBlock refuses a human write (session or a user's own API key)
|
||||||
|
// on a route it wraps, while letting a service account through. That is the
|
||||||
|
// whole point of operator mode: automation holding a service-account key
|
||||||
|
// (terdut-operator, most likely) keeps reconciling these resources, and a
|
||||||
|
// person in the web UI or terdut-tui gets a clear "edit this through your
|
||||||
|
// GitOps source instead" rather than a write that the next resync would only
|
||||||
|
// undo.
|
||||||
|
//
|
||||||
|
// Checked after AuthMiddleware, the same way AdminOnly is: by the time a
|
||||||
|
// request reaches here the caller is already known to be a service account
|
||||||
|
// or not. A router that never enables operator mode pays nothing for this —
|
||||||
|
// it hands back next unchanged rather than wrapping it in a check that would
|
||||||
|
// always pass.
|
||||||
|
func OperatorModeBlock(cfg config.Config) func(http.Handler) http.Handler {
|
||||||
|
return func(next http.Handler) http.Handler {
|
||||||
|
if !cfg.OperatorMode {
|
||||||
|
return next
|
||||||
|
}
|
||||||
|
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
caller, _ := callerFromContext(r.Context())
|
||||||
|
if _, ok := caller.ServiceAccountID(); ok {
|
||||||
|
next.ServeHTTP(w, r)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusForbidden, map[string]string{
|
||||||
|
"error": "this server is in operator mode; edit this through your GitOps source instead of the web UI or API",
|
||||||
|
"reason": operatorReason,
|
||||||
|
})
|
||||||
|
})
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// membership is the caller's role in one team.
|
// membership is the caller's role in one team.
|
||||||
@@ -213,24 +369,15 @@ func callerMemberships(ctx context.Context, db *sql.DB, userID int64) ([]members
|
|||||||
// administration is about accounts, not about reading other people's incidents,
|
// administration is about accounts, not about reading other people's incidents,
|
||||||
// and an admin who needs to see a team's queue can add themselves to it.
|
// and an admin who needs to see a team's queue can add themselves to it.
|
||||||
func callerTeamIDs(ctx context.Context) []int64 {
|
func callerTeamIDs(ctx context.Context) []int64 {
|
||||||
ms, _ := ctx.Value(ctxTeams).([]membership)
|
c, _ := callerFromContext(ctx)
|
||||||
ids := make([]int64, 0, len(ms))
|
return c.TeamIDs()
|
||||||
for _, m := range ms {
|
|
||||||
ids = append(ids, m.teamID)
|
|
||||||
}
|
|
||||||
return ids
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// callerRole reports the caller's role in one team, and whether they are in it
|
// callerRole reports the caller's role in one team, and whether they are in it
|
||||||
// at all.
|
// at all.
|
||||||
func callerRole(ctx context.Context, teamID int64) (string, bool) {
|
func callerRole(ctx context.Context, teamID int64) (string, bool) {
|
||||||
ms, _ := ctx.Value(ctxTeams).([]membership)
|
c, _ := callerFromContext(ctx)
|
||||||
for _, m := range ms {
|
return c.Role(teamID)
|
||||||
if m.teamID == teamID {
|
|
||||||
return m.role, true
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return "", false
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// requireTeamMember answers the request and reports false unless the caller
|
// requireTeamMember answers the request and reports false unless the caller
|
||||||
@@ -256,6 +403,13 @@ func requireTeamOwner(w http.ResponseWriter, r *http.Request, teamID int64) bool
|
|||||||
if ok && role == models.RoleOwner {
|
if ok && role == models.RoleOwner {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
|
// The operator's instance-scoped account manages every team's
|
||||||
|
// configuration, which is what lets it use one credential instead of
|
||||||
|
// minting one per team. This is owner reach only: it does not make the
|
||||||
|
// account a member, so it still reads no team's incidents.
|
||||||
|
if c, _ := callerFromContext(r.Context()); c.IsInstanceServiceAccount() {
|
||||||
|
return true
|
||||||
|
}
|
||||||
if caller, _ := userFromContext(r.Context()); caller.IsAdmin {
|
if caller, _ := userFromContext(r.Context()); caller.IsAdmin {
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
@@ -273,3 +427,26 @@ func sessionFromContext(ctx context.Context) (int64, bool) {
|
|||||||
id, ok := ctx.Value(ctxSession).(int64)
|
id, ok := ctx.Value(ctxSession).(int64)
|
||||||
return id, ok
|
return id, ok
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// requestLogger logs one line per request with the matched route pattern in
|
||||||
|
// place of the URL path. Two routes carry a credential in the path (the
|
||||||
|
// integration key and the ack token), and chi's stock logger would write it to
|
||||||
|
// the log verbatim.
|
||||||
|
func requestLogger(next http.Handler) http.Handler {
|
||||||
|
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
start := time.Now()
|
||||||
|
ww := middleware.NewWrapResponseWriter(w, r.ProtoMajor)
|
||||||
|
next.ServeHTTP(ww, r)
|
||||||
|
route := "unmatched"
|
||||||
|
if rc := chi.RouteContext(r.Context()); rc != nil {
|
||||||
|
if p := rc.RoutePattern(); p != "" {
|
||||||
|
route = p
|
||||||
|
}
|
||||||
|
}
|
||||||
|
status := ww.Status()
|
||||||
|
if status == 0 {
|
||||||
|
status = http.StatusOK
|
||||||
|
}
|
||||||
|
log.Printf("%q %q %d %dB %s", r.Method, route, status, ww.BytesWritten(), time.Since(start).Round(time.Millisecond)) // #nosec G706 -- method and route are %q-quoted, the route is a registered pattern, the rest are numbers
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|||||||
@@ -8,6 +8,7 @@ import (
|
|||||||
"fmt"
|
"fmt"
|
||||||
"log"
|
"log"
|
||||||
"net/http"
|
"net/http"
|
||||||
|
"regexp"
|
||||||
"strings"
|
"strings"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
@@ -38,6 +39,13 @@ const (
|
|||||||
ackTokenTTL = 24 * time.Hour
|
ackTokenTTL = 24 * time.Hour
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// notifierLockKey is the Postgres advisory lock the notifier takes for the
|
||||||
|
// duration of each pass, so that running more than one replica does not
|
||||||
|
// deliver (or double-deliver) the same notification from more than one of
|
||||||
|
// them at once. Its value has no meaning beyond being distinct from
|
||||||
|
// archiverLockKey.
|
||||||
|
const notifierLockKey int64 = 7265_0002
|
||||||
|
|
||||||
// Notification kinds, recording why a push was sent.
|
// Notification kinds, recording why a push was sent.
|
||||||
const (
|
const (
|
||||||
notifyTriggered = "triggered"
|
notifyTriggered = "triggered"
|
||||||
@@ -97,6 +105,10 @@ var notifyClient = &http.Client{Timeout: 10 * time.Second}
|
|||||||
|
|
||||||
// StartNotifier delivers queued notifications until ctx is cancelled, starting
|
// StartNotifier delivers queued notifications until ctx is cancelled, starting
|
||||||
// with an immediate pass so a restart flushes whatever the last one left behind.
|
// with an immediate pass so a restart flushes whatever the last one left behind.
|
||||||
|
//
|
||||||
|
// Each pass runs under notifierLockKey (see withAdvisoryLock), so that on more
|
||||||
|
// than one replica only whichever instance's tick takes the lock first actually
|
||||||
|
// delivers; the rest skip that tick rather than racing the same pass.
|
||||||
func StartNotifier(ctx context.Context, db *sql.DB, cfg NotifyConfig) {
|
func StartNotifier(ctx context.Context, db *sql.DB, cfg NotifyConfig) {
|
||||||
if !cfg.enabled() {
|
if !cfg.enabled() {
|
||||||
log.Print("notifier: disabled (no ntfy URL configured)")
|
log.Print("notifier: disabled (no ntfy URL configured)")
|
||||||
@@ -107,11 +119,17 @@ func StartNotifier(ctx context.Context, db *sql.DB, cfg NotifyConfig) {
|
|||||||
ticker := time.NewTicker(notifyInterval)
|
ticker := time.NewTicker(notifyInterval)
|
||||||
defer ticker.Stop()
|
defer ticker.Stop()
|
||||||
|
|
||||||
NotifySweep(ctx, db, cfg)
|
sweep := func() {
|
||||||
|
withAdvisoryLock(ctx, db, notifierLockKey, "notifier", func() {
|
||||||
|
NotifySweep(ctx, db, cfg)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
sweep()
|
||||||
for {
|
for {
|
||||||
select {
|
select {
|
||||||
case <-ticker.C:
|
case <-ticker.C:
|
||||||
NotifySweep(ctx, db, cfg)
|
sweep()
|
||||||
case <-ctx.Done():
|
case <-ctx.Done():
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
@@ -240,7 +258,7 @@ func deliverPending(ctx context.Context, db *sql.DB, cfg NotifyConfig) {
|
|||||||
}
|
}
|
||||||
// Logged, not returned: the page has already gone out, and treating a
|
// Logged, not returned: the page has already gone out, and treating a
|
||||||
// failed timeline write as a failed delivery would send it again.
|
// failed timeline write as a failed delivery would send it again.
|
||||||
if err := logEvent(ctx, db, n.incidentID, eventNotified, n.userID, nil, &n.kind); err != nil {
|
if err := logEvent(ctx, db, n.incidentID, eventNotified, n.userID, nil, nil, &n.kind); err != nil {
|
||||||
log.Printf("notifier: log delivery of %d: %v", n.id, err)
|
log.Printf("notifier: log delivery of %d: %v", n.id, err)
|
||||||
}
|
}
|
||||||
sent++
|
sent++
|
||||||
@@ -295,7 +313,7 @@ func markFailed(ctx context.Context, db *sql.DB, n outboxRow, cause error) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
detail := fmt.Sprintf("%s: %s", n.kind, cause)
|
detail := fmt.Sprintf("%s: %s", n.kind, cause)
|
||||||
if err := logEvent(ctx, db, n.incidentID, eventNotifyFailed, n.userID, nil, &detail); err != nil {
|
if err := logEvent(ctx, db, n.incidentID, eventNotifyFailed, n.userID, nil, nil, &detail); err != nil {
|
||||||
log.Printf("notifier: log failure of %d: %v", n.id, err)
|
log.Printf("notifier: log failure of %d: %v", n.id, err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -330,6 +348,16 @@ func deliver(ctx context.Context, db *sql.DB, cfg NotifyConfig, n outboxRow) err
|
|||||||
|
|
||||||
msg := renderNotification(inc, n, firing, cfg)
|
msg := renderNotification(inc, n, firing, cfg)
|
||||||
|
|
||||||
|
// The page that opens an incident carries what fixed it last time, so the
|
||||||
|
// person woken up starts from that. Best effort: a failed lookup must not
|
||||||
|
// hold back the page itself.
|
||||||
|
if n.kind == notifyTriggered {
|
||||||
|
if sim, err := similarIncidents(ctx, db, n.incidentID, 1); err == nil && len(sim) > 0 && len(sim[0].ResolutionNotes) > 0 {
|
||||||
|
notes := sim[0].ResolutionNotes
|
||||||
|
msg.Message += "\nLast time: " + shorten(derefString(notes[len(notes)-1].Detail), 160)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// An Acknowledge button needs both a user to attribute the acknowledgement
|
// An Acknowledge button needs both a user to attribute the acknowledgement
|
||||||
// to and a URL the phone can reach. Minted per delivery, so every push
|
// to and a URL the phone can reach. Minted per delivery, so every push
|
||||||
// carries its own short-lived token rather than reusing one.
|
// carries its own short-lived token rather than reusing one.
|
||||||
@@ -382,9 +410,11 @@ func renderNotification(inc models.Incident, n outboxRow, firing int, cfg Notify
|
|||||||
strings.TrimSuffix(cfg.PublicURL, "/"), inc.ID)
|
strings.TrimSuffix(cfg.PublicURL, "/"), inc.ID)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
title := pageTitle(inc)
|
||||||
|
|
||||||
switch n.kind {
|
switch n.kind {
|
||||||
case notifyResolved:
|
case notifyResolved:
|
||||||
msg.Title = "Resolved: " + inc.Title
|
msg.Title = "Resolved: " + title
|
||||||
msg.Message = "All alerts stopped firing after " +
|
msg.Message = "All alerts stopped firing after " +
|
||||||
humanDuration(time.Since(inc.TriggeredAt))
|
humanDuration(time.Since(inc.TriggeredAt))
|
||||||
msg.Priority = ntfyPriorityLow
|
msg.Priority = ntfyPriorityLow
|
||||||
@@ -392,9 +422,9 @@ func renderNotification(inc models.Incident, n outboxRow, firing int, cfg Notify
|
|||||||
return msg
|
return msg
|
||||||
|
|
||||||
case notifyReminder:
|
case notifyReminder:
|
||||||
msg.Title = "Still unacknowledged: " + inc.Title
|
msg.Title = "Still unacknowledged: " + title
|
||||||
default:
|
default:
|
||||||
msg.Title = inc.Title
|
msg.Title = title
|
||||||
}
|
}
|
||||||
|
|
||||||
severity := derefString(inc.Severity)
|
severity := derefString(inc.Severity)
|
||||||
@@ -416,6 +446,48 @@ func renderNotification(inc models.Incident, n outboxRow, firing int, cfg Notify
|
|||||||
return msg
|
return msg
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// originLabel is the label that says where an alert came from, for a team with
|
||||||
|
// several Kubernetes clusters behind it. It comes from Prometheus's
|
||||||
|
// externalLabels and reaches an incident through Alertmanager's group_by; the
|
||||||
|
// web UI reads the same label, and docs/incidents.md ("Several clusters, one team")
|
||||||
|
// explains how to set it up.
|
||||||
|
const originLabel = "cluster"
|
||||||
|
|
||||||
|
// pageTitle is the incident's title for a notification. A phone's lock screen
|
||||||
|
// cuts a long title off at the end, and the incident title puts the grouping
|
||||||
|
// labels there, so the cluster would be the first thing lost. When the incident
|
||||||
|
// has an origin it leads instead, "[prod-eu] PodRestarting (namespace=foo)", and
|
||||||
|
// is dropped from the parenthesis so it is not said twice. A title that is not
|
||||||
|
// in incidentTitle's "name (k=v, k=v)" shape keeps its text and gains the prefix.
|
||||||
|
func pageTitle(inc models.Incident) string {
|
||||||
|
origin := inc.GroupLabels[originLabel]
|
||||||
|
if origin == "" {
|
||||||
|
return inc.Title
|
||||||
|
}
|
||||||
|
return "[" + origin + "] " + titleWithoutLabel(inc.Title, originLabel, origin)
|
||||||
|
}
|
||||||
|
|
||||||
|
var titleShape = regexp.MustCompile(`(?s)^(.*?) \((.*)\)$`)
|
||||||
|
|
||||||
|
// titleWithoutLabel removes "key=value" from the parenthesised tail of a title
|
||||||
|
// built by incidentTitle, and the parentheses with it if nothing else is left.
|
||||||
|
func titleWithoutLabel(title, key, value string) string {
|
||||||
|
m := titleShape.FindStringSubmatch(title)
|
||||||
|
if m == nil {
|
||||||
|
return title
|
||||||
|
}
|
||||||
|
var rest []string
|
||||||
|
for _, part := range strings.Split(m[2], ", ") {
|
||||||
|
if part != key+"="+value {
|
||||||
|
rest = append(rest, part)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(rest) == 0 {
|
||||||
|
return m[1]
|
||||||
|
}
|
||||||
|
return m[1] + " (" + strings.Join(rest, ", ") + ")"
|
||||||
|
}
|
||||||
|
|
||||||
// ntfy's priority scale. Max is the one that overrides the phone's quiet
|
// ntfy's priority scale. Max is the one that overrides the phone's quiet
|
||||||
// settings, which is the whole point of paging on critical.
|
// settings, which is the whole point of paging on critical.
|
||||||
const (
|
const (
|
||||||
@@ -573,6 +645,17 @@ func plural(n int) string {
|
|||||||
return "s"
|
return "s"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// shorten cuts s to at most n runes, marking the cut, and flattens newlines so
|
||||||
|
// a multi-line note stays one line in a push.
|
||||||
|
func shorten(s string, n int) string {
|
||||||
|
s = strings.Join(strings.Fields(s), " ")
|
||||||
|
r := []rune(s)
|
||||||
|
if len(r) <= n {
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
return string(r[:n-1]) + "…"
|
||||||
|
}
|
||||||
|
|
||||||
// derefString reads a nullable text column as a plain string.
|
// derefString reads a nullable text column as a plain string.
|
||||||
func derefString(s *string) string {
|
func derefString(s *string) string {
|
||||||
if s == nil {
|
if s == nil {
|
||||||
|
|||||||
@@ -62,16 +62,23 @@ func handleNotifyAck(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
acked, err := acknowledgeIncident(r.Context(), db, incidentID, userID)
|
acked, err := acknowledgeIncident(r.Context(), db, incidentID, userID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if !acked {
|
if !acked {
|
||||||
// The incident closed between the page and the tap. Nothing to do,
|
// Either the incident closed between the page and the tap, or it was
|
||||||
// and nothing the responder did wrong — report the state, not an error,
|
// already acknowledged (e.g. from the web UI, or an earlier tap of
|
||||||
// so ntfy shows a success toast rather than a failure.
|
// the same button) — either way nothing the responder did wrong, so
|
||||||
|
// report the actual state rather than assuming "resolved", and let
|
||||||
|
// ntfy show a success toast rather than a failure.
|
||||||
|
inc, err := fetchIncident(r.Context(), db, incidentID)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
respond(w, http.StatusOK, map[string]any{
|
respond(w, http.StatusOK, map[string]any{
|
||||||
"incident_id": incidentID,
|
"incident_id": incidentID,
|
||||||
"status": "resolved",
|
"status": inc.Status,
|
||||||
})
|
})
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
package api_test
|
package api_test
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"bytes"
|
||||||
"context"
|
"context"
|
||||||
"encoding/json"
|
"encoding/json"
|
||||||
"fmt"
|
"fmt"
|
||||||
@@ -164,10 +165,91 @@ func fireCritical(t *testing.T, s *ts) {
|
|||||||
}, "{}:{alertname=\"DiskFull\"}")
|
}, "{}:{alertname=\"DiskFull\"}")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// fireGrouped posts one critical alert whose Alertmanager group carries the
|
||||||
|
// given labels, the way group_by puts them on the webhook.
|
||||||
|
func fireGrouped(t *testing.T, s *ts, groupLabels map[string]string, groupKey string) {
|
||||||
|
t.Helper()
|
||||||
|
fireGroupedAs(t, s, "fp-grouped", groupLabels, groupKey)
|
||||||
|
}
|
||||||
|
|
||||||
|
// fireGroupedAs is fireGrouped with its own alert fingerprint, for a test that
|
||||||
|
// needs several alerts open at once.
|
||||||
|
func fireGroupedAs(t *testing.T, s *ts, fingerprint string, groupLabels map[string]string, groupKey string) {
|
||||||
|
t.Helper()
|
||||||
|
payload := map[string]any{
|
||||||
|
"version": "4", "status": "firing", "groupKey": groupKey, "groupLabels": groupLabels,
|
||||||
|
"alerts": []map[string]any{amAlert(fingerprint, "PodRestarting", "firing",
|
||||||
|
"2026-05-20T10:00:00Z", zeroTime, map[string]string{"severity": "critical"})},
|
||||||
|
}
|
||||||
|
data, _ := json.Marshal(payload)
|
||||||
|
resp, err := http.Post(s.URL+"/api/integrations/"+s.ingestKey+"/alertmanager",
|
||||||
|
"application/json", bytes.NewReader(data))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("post webhook: %v", err)
|
||||||
|
}
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Fatalf("webhook returned %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Delivery
|
// Delivery
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
// A phone cuts a long title off at the end, and the incident title keeps the
|
||||||
|
// grouping labels there, so the cluster leads the page instead.
|
||||||
|
func TestNotify_ClusterLeadsTheTitle(t *testing.T) {
|
||||||
|
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com"})
|
||||||
|
|
||||||
|
fireGrouped(t, s, map[string]string{
|
||||||
|
"alertname": "PodRestarting", "cluster": "prod-eu", "namespace": "shop",
|
||||||
|
}, `{}:{alertname="PodRestarting",cluster="prod-eu",namespace="shop"}`)
|
||||||
|
s.sweepNotify(t)
|
||||||
|
|
||||||
|
msgs := f.messages()
|
||||||
|
if len(msgs) != 1 {
|
||||||
|
t.Fatalf("expected 1 push, got %d", len(msgs))
|
||||||
|
}
|
||||||
|
if want := "[prod-eu] PodRestarting (namespace=shop)"; msgs[0].Title != want {
|
||||||
|
t.Errorf("title = %q, want %q", msgs[0].Title, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Only the cluster is the whole grouping: no parenthesis is left behind.
|
||||||
|
func TestNotify_ClusterAloneLeavesNoParenthesis(t *testing.T) {
|
||||||
|
s, f := notifyTS(t, api.NotifyConfig{})
|
||||||
|
|
||||||
|
fireGrouped(t, s, map[string]string{"alertname": "PodRestarting", "cluster": "prod-eu"},
|
||||||
|
`{}:{alertname="PodRestarting",cluster="prod-eu"}`)
|
||||||
|
s.sweepNotify(t)
|
||||||
|
|
||||||
|
msgs := f.messages()
|
||||||
|
if len(msgs) != 1 {
|
||||||
|
t.Fatalf("expected 1 push, got %d", len(msgs))
|
||||||
|
}
|
||||||
|
if want := "[prod-eu] PodRestarting"; msgs[0].Title != want {
|
||||||
|
t.Errorf("title = %q, want %q", msgs[0].Title, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Nothing changes for a team whose alerts have no cluster label.
|
||||||
|
func TestNotify_NoClusterKeepsTheTitle(t *testing.T) {
|
||||||
|
s, f := notifyTS(t, api.NotifyConfig{})
|
||||||
|
|
||||||
|
fireGrouped(t, s, map[string]string{"alertname": "PodRestarting", "namespace": "shop"},
|
||||||
|
`{}:{alertname="PodRestarting",namespace="shop"}`)
|
||||||
|
s.sweepNotify(t)
|
||||||
|
|
||||||
|
msgs := f.messages()
|
||||||
|
if len(msgs) != 1 {
|
||||||
|
t.Fatalf("expected 1 push, got %d", len(msgs))
|
||||||
|
}
|
||||||
|
if want := "PodRestarting (namespace=shop)"; msgs[0].Title != want {
|
||||||
|
t.Errorf("title = %q, want %q", msgs[0].Title, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestNotify_TriggeredIncidentPagesOnCall(t *testing.T) {
|
func TestNotify_TriggeredIncidentPagesOnCall(t *testing.T) {
|
||||||
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com"})
|
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com"})
|
||||||
|
|
||||||
@@ -408,6 +490,47 @@ func TestNotify_AckButtonAcknowledgesIncident(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The ack token isn't single-use (it stays valid for a day, in case the
|
||||||
|
// first tap never reaches the server), so tapping the same notification's
|
||||||
|
// Acknowledge button twice is a real scenario, not just a retried request.
|
||||||
|
// It must report the incident's actual state, not assume "resolved" —
|
||||||
|
// see handleNotifyAck's !acked branch — and must not log a second
|
||||||
|
// "acknowledged" event.
|
||||||
|
func TestNotify_AckButtonTwiceIsIdempotent(t *testing.T) {
|
||||||
|
s, f := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com"})
|
||||||
|
|
||||||
|
fireCritical(t, s)
|
||||||
|
s.sweepNotify(t)
|
||||||
|
|
||||||
|
ackURL := f.messages()[0].Actions[0].URL
|
||||||
|
path := ackURL[strings.Index(ackURL, "/api/notify/ack/"):]
|
||||||
|
|
||||||
|
for i := range 2 {
|
||||||
|
resp, err := http.Post(s.URL+path, "application/json", nil)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ack %d: %v", i+1, err)
|
||||||
|
}
|
||||||
|
var body map[string]any
|
||||||
|
decode(t, resp, &body)
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Fatalf("ack %d returned %d", i+1, resp.StatusCode)
|
||||||
|
}
|
||||||
|
if body["status"] != "acknowledged" {
|
||||||
|
t.Errorf("ack %d: expected status acknowledged, got %v", i+1, body["status"])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
n := 0
|
||||||
|
for _, ty := range eventTypes(timeline(t, s, 1)) {
|
||||||
|
if ty == "acknowledged" {
|
||||||
|
n++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if n != 1 {
|
||||||
|
t.Errorf("expected exactly one acknowledged event after two taps, got %d", n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestNotify_AckRejectsUnknownToken(t *testing.T) {
|
func TestNotify_AckRejectsUnknownToken(t *testing.T) {
|
||||||
s, _ := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com"})
|
s, _ := notifyTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com"})
|
||||||
fireCritical(t, s)
|
fireCritical(t, s)
|
||||||
|
|||||||
@@ -0,0 +1,530 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"database/sql"
|
||||||
|
"errors"
|
||||||
|
"log"
|
||||||
|
"net/http"
|
||||||
|
"net/url"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/oidc"
|
||||||
|
)
|
||||||
|
|
||||||
|
const (
|
||||||
|
// oidcStateCookie ties an in-flight login to the browser that started it.
|
||||||
|
// Without it anybody could start a login, and send the callback URL that
|
||||||
|
// results to somebody else, who would be signed in as the attacker.
|
||||||
|
oidcStateCookie = "terdut_oidc_state"
|
||||||
|
|
||||||
|
// oidcLoginTTL is how long a login may take between the redirect to the
|
||||||
|
// provider and the callback, which includes the person typing a password
|
||||||
|
// and a second factor.
|
||||||
|
oidcLoginTTL = 10 * time.Minute
|
||||||
|
|
||||||
|
// oidcStartMaxPerAddr bounds unauthenticated logins started per address.
|
||||||
|
// Each writes a row, so an unbounded endpoint is a way to grow the table.
|
||||||
|
oidcStartMaxPerAddr = 30
|
||||||
|
)
|
||||||
|
|
||||||
|
// ssoError is a sign-in refusal the person can be told about. Its value is the
|
||||||
|
// code the web UI is sent back with, as ?sso_error=<code>; the detail stays in
|
||||||
|
// the server log, since it can name accounts.
|
||||||
|
type ssoError string
|
||||||
|
|
||||||
|
func (e ssoError) Error() string { return "sso: " + string(e) }
|
||||||
|
|
||||||
|
const (
|
||||||
|
ssoDenied ssoError = "denied" // the provider reported an error, or the person declined
|
||||||
|
ssoExpired ssoError = "expired" // unknown, used or expired state; start again
|
||||||
|
ssoFailed ssoError = "failed" // the token exchange or its verification failed
|
||||||
|
ssoUnavailable ssoError = "unavailable" // the provider could not be reached
|
||||||
|
ssoNotAllowed ssoError = "not_allowed" // authenticated, but in none of the allowed groups
|
||||||
|
ssoNoEmail ssoError = "no_email" // the provider sent no email address
|
||||||
|
ssoEmailConflict ssoError = "email_conflict" // a local account has this email and cannot be linked
|
||||||
|
ssoDisabled ssoError = "disabled" // the linked account is disabled
|
||||||
|
// ssoNotBootstrapped: this identity has no existing account, and no user
|
||||||
|
// exists on this install yet either -- creating one here would race
|
||||||
|
// POST /api/bootstrap for the one gitops-managed installs expect to win
|
||||||
|
// it (terdut-operator's own DESIGN.md §1, §6), which has no way to
|
||||||
|
// recover if it loses. The person sees this for at most as long as it
|
||||||
|
// takes whatever is bootstrapping this install to finish; signing in
|
||||||
|
// again afterward hits the ordinary first-sign-in path. Found by
|
||||||
|
// terdut-operator#1: nothing stopped an otherwise-ordinary OIDC sign-in
|
||||||
|
// from quietly winning this race against an operator that assumed it
|
||||||
|
// was the only caller.
|
||||||
|
ssoNotBootstrapped ssoError = "not_bootstrapped"
|
||||||
|
)
|
||||||
|
|
||||||
|
// handleAuthConfig says how this server can be signed in to, so the login form
|
||||||
|
// and the TUI can offer the right choices before anybody types anything. It is
|
||||||
|
// unauthenticated by necessity, and reveals nothing beyond what the login page
|
||||||
|
// shows anyway.
|
||||||
|
func handleAuthConfig(cfg config.Config) http.HandlerFunc {
|
||||||
|
type oidcInfo struct {
|
||||||
|
Enabled bool `json:"enabled"`
|
||||||
|
Name string `json:"name,omitempty"`
|
||||||
|
}
|
||||||
|
type response struct {
|
||||||
|
PasswordLogin bool `json:"password_login"`
|
||||||
|
OIDC oidcInfo `json:"oidc"`
|
||||||
|
|
||||||
|
// DeviceLogin is whether a client that cannot open a browser (the TUI)
|
||||||
|
// can sign in by showing a code, through /api/oidc/device.
|
||||||
|
DeviceLogin bool `json:"device_login"`
|
||||||
|
|
||||||
|
// OperatorMode is whether this install is gitops-managed: writes to
|
||||||
|
// teams, escalation policies, dead man's switches and integrations
|
||||||
|
// from a session or a user's own API key are refused (OperatorModeBlock),
|
||||||
|
// though a service account's are not. The web UI reads this before
|
||||||
|
// anybody signs in, the same way it reads PasswordLogin/OIDC, so it can
|
||||||
|
// show those sections read-only from the start rather than only after
|
||||||
|
// a write fails.
|
||||||
|
OperatorMode bool `json:"operator_mode"`
|
||||||
|
}
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
resp := response{PasswordLogin: !cfg.DisablePasswordLogin, OperatorMode: cfg.OperatorMode}
|
||||||
|
if cfg.OIDC.Enabled() {
|
||||||
|
resp.OIDC = oidcInfo{Enabled: true, Name: cfg.OIDC.Name}
|
||||||
|
resp.DeviceLogin = true
|
||||||
|
}
|
||||||
|
respond(w, http.StatusOK, resp)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// passwordLoginOnly refuses a route when password login is switched off.
|
||||||
|
func passwordLoginOnly(enabled bool) func(http.Handler) http.Handler {
|
||||||
|
return func(next http.Handler) http.Handler {
|
||||||
|
if enabled {
|
||||||
|
return next
|
||||||
|
}
|
||||||
|
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
respond(w, http.StatusForbidden, errResp("password login is disabled on this server"))
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ssoRedirect sends the browser back to the web UI with the reason a sign-in
|
||||||
|
// failed. It is a redirect and not a JSON error because the browser arrived
|
||||||
|
// here by navigating from the provider: there is no page script to read one.
|
||||||
|
func ssoRedirect(w http.ResponseWriter, r *http.Request, code ssoError) {
|
||||||
|
http.Redirect(w, r, "/?sso_error="+url.QueryEscape(string(code)), http.StatusFound)
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleOIDCLogin starts a sign-in: it records the state, nonce and PKCE
|
||||||
|
// verifier the callback will need and sends the browser to the provider.
|
||||||
|
func handleOIDCLogin(db *sql.DB, prov *oidc.Provider, limiter *loginLimiter, publicURL string) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
addrKey := "oidc:" + clientAddr(r)
|
||||||
|
if limiter.blocked(r.Context(), addrKey, oidcStartMaxPerAddr) {
|
||||||
|
w.Header().Set("Retry-After", strconv.Itoa(int(loginWindow.Seconds())))
|
||||||
|
respond(w, http.StatusTooManyRequests, errResp("too many sign-in attempts, try again later"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
limiter.fail(r.Context(), addrKey)
|
||||||
|
|
||||||
|
state, stateHash, err := randomToken()
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
nonce, _, err := randomToken()
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
verifier := oidc.NewVerifier()
|
||||||
|
next := safeNext(r.URL.Query().Get("next"))
|
||||||
|
|
||||||
|
// Abandoned logins are swept here rather than by the sweeper: this is
|
||||||
|
// the only place they are made, so the table cannot outgrow its writers.
|
||||||
|
now := time.Now()
|
||||||
|
db.ExecContext(r.Context(), "DELETE FROM oidc_logins WHERE expires_at < $1", now.Unix())
|
||||||
|
if _, err := db.ExecContext(r.Context(), `
|
||||||
|
INSERT INTO oidc_logins (state_hash, nonce, pkce_verifier, next, expires_at)
|
||||||
|
VALUES ($1, $2, $3, $4, $5)`,
|
||||||
|
stateHash, nonce, verifier, next, now.Add(oidcLoginTTL).Unix()); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
authURL, err := prov.AuthURL(r.Context(), state, nonce, verifier)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("oidc: start login: %v", err)
|
||||||
|
ssoRedirect(w, r, ssoUnavailable)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// #nosec G124 -- HttpOnly/SameSite are literal below; Secure is
|
||||||
|
// cookieSecure(publicURL, r), not a literal true, which is what
|
||||||
|
// trips this rule. See cookieSecure's own doc comment in auth.go.
|
||||||
|
http.SetCookie(w, &http.Cookie{
|
||||||
|
Name: oidcStateCookie,
|
||||||
|
Value: state,
|
||||||
|
Path: "/api/oidc",
|
||||||
|
MaxAge: int(oidcLoginTTL.Seconds()),
|
||||||
|
HttpOnly: true,
|
||||||
|
Secure: cookieSecure(publicURL, r),
|
||||||
|
// Lax, not Strict: the callback is a top-level navigation from the
|
||||||
|
// provider's site, which Strict would not send the cookie on.
|
||||||
|
SameSite: http.SameSiteLaxMode,
|
||||||
|
})
|
||||||
|
http.Redirect(w, r, authURL, http.StatusFound)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleOIDCCallback finishes a sign-in: it verifies the provider's answer,
|
||||||
|
// finds or creates the user, applies their groups and starts a session.
|
||||||
|
func handleOIDCCallback(db *sql.DB, prov *oidc.Provider, publicURL string) http.HandlerFunc {
|
||||||
|
cfg := prov.Config()
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
// The state cookie has done its job once the callback arrives, whatever
|
||||||
|
// the outcome.
|
||||||
|
// #nosec G124 -- HttpOnly/SameSite are literal below; Secure is
|
||||||
|
// cookieSecure(publicURL, r), not a literal true, which is what
|
||||||
|
// trips this rule. See cookieSecure's own doc comment in auth.go.
|
||||||
|
http.SetCookie(w, &http.Cookie{
|
||||||
|
Name: oidcStateCookie, Value: "", Path: "/api/oidc", MaxAge: -1,
|
||||||
|
HttpOnly: true, Secure: cookieSecure(publicURL, r), SameSite: http.SameSiteLaxMode,
|
||||||
|
})
|
||||||
|
|
||||||
|
q := r.URL.Query()
|
||||||
|
if e := q.Get("error"); e != "" {
|
||||||
|
// %q on both: this runs before state is checked against the
|
||||||
|
// cookie, so error and error_description are still whatever the
|
||||||
|
// request's query string says, not yet known to be the real
|
||||||
|
// provider's. %q keeps a crafted value (say, one holding a
|
||||||
|
// newline) from forging a second log line rather than just
|
||||||
|
// being a quoted string within this one.
|
||||||
|
log.Printf("oidc: provider returned error %q: %q", e, q.Get("error_description")) // #nosec G706 -- both %q
|
||||||
|
ssoRedirect(w, r, ssoDenied)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
state := q.Get("state")
|
||||||
|
cookie, err := r.Cookie(oidcStateCookie)
|
||||||
|
if state == "" || q.Get("code") == "" || err != nil || cookie.Value != state {
|
||||||
|
ssoRedirect(w, r, ssoExpired)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// DELETE ... RETURNING makes the state single-use: a replayed callback
|
||||||
|
// finds nothing.
|
||||||
|
var nonce, verifier, next string
|
||||||
|
err = db.QueryRowContext(r.Context(), `
|
||||||
|
DELETE FROM oidc_logins WHERE state_hash = $1 AND expires_at > $2
|
||||||
|
RETURNING nonce, pkce_verifier, next`,
|
||||||
|
hashToken(state), time.Now().Unix()).Scan(&nonce, &verifier, &next)
|
||||||
|
if errors.Is(err, sql.ErrNoRows) {
|
||||||
|
ssoRedirect(w, r, ssoExpired)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("oidc: load login state: %v", err)
|
||||||
|
ssoRedirect(w, r, ssoFailed)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
identity, err := prov.Exchange(r.Context(), q.Get("code"), verifier, nonce)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("oidc: %v", err)
|
||||||
|
ssoRedirect(w, r, ssoFailed)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
grants := oidc.ComputeGrants(cfg, identity.Groups)
|
||||||
|
if !grants.Admitted {
|
||||||
|
log.Printf("oidc: %q (%q) is in none of the allowed groups", identity.Username, identity.Subject) // #nosec G706 -- both %q
|
||||||
|
ssoRedirect(w, r, ssoNotAllowed)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
teamGroups, err := loadTeamGroups(r.Context(), db)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("oidc: load team groups: %v", err)
|
||||||
|
ssoRedirect(w, r, ssoFailed)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
teamGrants := oidc.ComputeTeamGrants(teamGroups, identity.Groups)
|
||||||
|
|
||||||
|
userID, err := signInSSO(r.Context(), db, cfg, identity, grants, teamGrants)
|
||||||
|
if err != nil {
|
||||||
|
var se ssoError
|
||||||
|
if errors.As(err, &se) {
|
||||||
|
log.Printf("oidc: refused %q (%q): %v", identity.Username, identity.Subject, se) // #nosec G706 -- both %q
|
||||||
|
ssoRedirect(w, r, se)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
log.Printf("oidc: sign in %q: %v", identity.Username, err) // #nosec G706 -- %q
|
||||||
|
ssoRedirect(w, r, ssoFailed)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := startSessionCapped(w, r, db, userID, publicURL, cfg.SessionMaxAge); err != nil {
|
||||||
|
log.Printf("oidc: start session: %v", err)
|
||||||
|
ssoRedirect(w, r, ssoFailed)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
http.Redirect(w, r, safeNext(next), http.StatusFound)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// safeNext returns where to send the browser after a sign-in: the path asked
|
||||||
|
// for, if it is one on this server, and the front page otherwise. It is the
|
||||||
|
// only thing standing between a login link and an open redirect, so it accepts
|
||||||
|
// a single leading slash and nothing that a browser could read as another host
|
||||||
|
// ("//evil.example", "/\evil.example"), and never an API path, which would
|
||||||
|
// land somebody on raw JSON.
|
||||||
|
func safeNext(next string) string {
|
||||||
|
switch {
|
||||||
|
case next == "", len(next) > 512,
|
||||||
|
!strings.HasPrefix(next, "/"),
|
||||||
|
strings.HasPrefix(next, "//"),
|
||||||
|
strings.HasPrefix(next, "/api/"),
|
||||||
|
strings.ContainsAny(next, "\\\r\n"):
|
||||||
|
return "/"
|
||||||
|
}
|
||||||
|
return next
|
||||||
|
}
|
||||||
|
|
||||||
|
// signInSSO resolves the identity to a user and applies its grants, in one
|
||||||
|
// transaction: a login that fails half way must not leave memberships changed.
|
||||||
|
func signInSSO(ctx context.Context, db *sql.DB, cfg config.OIDC, id *oidc.Identity, g oidc.Grants, teamRoles map[int64]string) (int64, error) {
|
||||||
|
tx, err := db.BeginTx(ctx, nil)
|
||||||
|
if err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
defer tx.Rollback() //nolint:errcheck
|
||||||
|
|
||||||
|
userID, err := resolveSSOUser(ctx, tx, cfg, id)
|
||||||
|
if err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
var disabled bool
|
||||||
|
if err := tx.QueryRowContext(ctx,
|
||||||
|
"SELECT disabled_at IS NOT NULL FROM users WHERE id = $1", userID).Scan(&disabled); err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
if disabled {
|
||||||
|
return 0, ssoDisabled
|
||||||
|
}
|
||||||
|
if err := syncGrants(ctx, tx, userID, g, teamRoles); err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
return userID, tx.Commit()
|
||||||
|
}
|
||||||
|
|
||||||
|
// loadTeamGroups reads every team's own OIDC group binding, for the sync to
|
||||||
|
// evaluate against one user's groups at a time. Teams are few, so this reads
|
||||||
|
// the whole table rather than filtering it.
|
||||||
|
func loadTeamGroups(ctx context.Context, db *sql.DB) ([]oidc.TeamGroup, error) {
|
||||||
|
rows, err := db.QueryContext(ctx,
|
||||||
|
"SELECT id, COALESCE(oidc_member_group, ''), COALESCE(oidc_owner_group, '') FROM teams")
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
defer rows.Close()
|
||||||
|
|
||||||
|
var out []oidc.TeamGroup
|
||||||
|
for rows.Next() {
|
||||||
|
var tg oidc.TeamGroup
|
||||||
|
if err := rows.Scan(&tg.TeamID, &tg.MemberGroup, &tg.OwnerGroup); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
out = append(out, tg)
|
||||||
|
}
|
||||||
|
return out, rows.Err()
|
||||||
|
}
|
||||||
|
|
||||||
|
// resolveSSOUser finds the user an identity belongs to, linking or creating one
|
||||||
|
// when this is its first sign-in.
|
||||||
|
//
|
||||||
|
// The order matters. The (issuer, subject) pair is the identity; email is only
|
||||||
|
// a way to recognise an existing local account the first time. Once linked, a
|
||||||
|
// changed email at the provider must not move the account to somebody else.
|
||||||
|
func resolveSSOUser(ctx context.Context, tx *sql.Tx, cfg config.OIDC, id *oidc.Identity) (int64, error) {
|
||||||
|
now := time.Now().Unix()
|
||||||
|
|
||||||
|
var userID int64
|
||||||
|
err := tx.QueryRowContext(ctx,
|
||||||
|
"SELECT user_id FROM user_identities WHERE issuer = $1 AND subject = $2",
|
||||||
|
id.Issuer, id.Subject).Scan(&userID)
|
||||||
|
if err == nil {
|
||||||
|
if _, err := tx.ExecContext(ctx,
|
||||||
|
"UPDATE user_identities SET last_login_at = $1 WHERE issuer = $2 AND subject = $3",
|
||||||
|
now, id.Issuer, id.Subject); err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
return userID, refreshProfile(ctx, tx, userID, id)
|
||||||
|
}
|
||||||
|
if !errors.Is(err, sql.ErrNoRows) {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
|
||||||
|
// First sign-in with this identity.
|
||||||
|
if id.Email == "" {
|
||||||
|
return 0, ssoNoEmail
|
||||||
|
}
|
||||||
|
err = tx.QueryRowContext(ctx,
|
||||||
|
"SELECT id FROM users WHERE lower(email) = lower($1)", id.Email).Scan(&userID)
|
||||||
|
switch {
|
||||||
|
case err == nil:
|
||||||
|
if !id.EmailVerified && !cfg.TrustEmail {
|
||||||
|
return 0, ssoEmailConflict
|
||||||
|
}
|
||||||
|
// A local account that already has an identity from this issuer is a
|
||||||
|
// different person at the provider using a recycled address. Linking
|
||||||
|
// them would hand one person's account to another.
|
||||||
|
var linked bool
|
||||||
|
if err := tx.QueryRowContext(ctx,
|
||||||
|
"SELECT EXISTS (SELECT 1 FROM user_identities WHERE user_id = $1 AND issuer = $2)",
|
||||||
|
userID, id.Issuer).Scan(&linked); err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
if linked {
|
||||||
|
return 0, ssoEmailConflict
|
||||||
|
}
|
||||||
|
case errors.Is(err, sql.ErrNoRows):
|
||||||
|
// Creating the very first user is /api/bootstrap's own job (same
|
||||||
|
// gate, same table: SELECT COUNT(*) FROM users in handleBootstrap).
|
||||||
|
// An identity nobody has linked yet, on an install with no users at
|
||||||
|
// all, is exactly the race terdut-operator#1 found: whoever gets
|
||||||
|
// here first wins a slot the other side has no way to recover from
|
||||||
|
// losing. Refusing it here costs an otherwise-ordinary sign-in
|
||||||
|
// nothing but a retry once bootstrap has actually run.
|
||||||
|
var userCount int
|
||||||
|
if err := tx.QueryRowContext(ctx, "SELECT COUNT(*) FROM users").Scan(&userCount); err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
if userCount == 0 {
|
||||||
|
return 0, ssoNotBootstrapped
|
||||||
|
}
|
||||||
|
userID, err = createSSOUser(ctx, tx, id)
|
||||||
|
if err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
default:
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
|
||||||
|
if _, err := tx.ExecContext(ctx,
|
||||||
|
"INSERT INTO user_identities (user_id, issuer, subject) VALUES ($1, $2, $3)",
|
||||||
|
userID, id.Issuer, id.Subject); err != nil {
|
||||||
|
return 0, err
|
||||||
|
}
|
||||||
|
return userID, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// createSSOUser inserts a user with no password. The username is the provider's,
|
||||||
|
// made unique with a numeric suffix when somebody local already has it.
|
||||||
|
func createSSOUser(ctx context.Context, tx *sql.Tx, id *oidc.Identity) (int64, error) {
|
||||||
|
base := strings.TrimSpace(id.Username)
|
||||||
|
if base == "" {
|
||||||
|
base, _, _ = strings.Cut(id.Email, "@")
|
||||||
|
}
|
||||||
|
if base == "" {
|
||||||
|
base = "user"
|
||||||
|
}
|
||||||
|
for n := 1; n <= 100; n++ {
|
||||||
|
name := base
|
||||||
|
if n > 1 {
|
||||||
|
name = base + "-" + strconv.Itoa(n)
|
||||||
|
}
|
||||||
|
var userID int64
|
||||||
|
err := tx.QueryRowContext(ctx, `
|
||||||
|
INSERT INTO users (username, email) VALUES ($1, $2)
|
||||||
|
ON CONFLICT (username) DO NOTHING RETURNING id`,
|
||||||
|
name, id.Email).Scan(&userID)
|
||||||
|
if errors.Is(err, sql.ErrNoRows) {
|
||||||
|
continue // taken; try the next suffix
|
||||||
|
}
|
||||||
|
return userID, err
|
||||||
|
}
|
||||||
|
return 0, errors.New("no free username for " + base)
|
||||||
|
}
|
||||||
|
|
||||||
|
// refreshProfile brings a linked user's username and email in line with the
|
||||||
|
// provider. Each update is skipped, not failed, when another user already holds
|
||||||
|
// the value: both columns are unique, and a sign-in must not break over a name.
|
||||||
|
func refreshProfile(ctx context.Context, tx *sql.Tx, userID int64, id *oidc.Identity) error {
|
||||||
|
if id.Username != "" {
|
||||||
|
if _, err := tx.ExecContext(ctx, `
|
||||||
|
UPDATE users SET username = $1
|
||||||
|
WHERE id = $2 AND username <> $1
|
||||||
|
AND NOT EXISTS (SELECT 1 FROM users WHERE username = $1)`,
|
||||||
|
id.Username, userID); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if id.Email != "" {
|
||||||
|
if _, err := tx.ExecContext(ctx, `
|
||||||
|
UPDATE users SET email = $1
|
||||||
|
WHERE id = $2 AND email <> $1
|
||||||
|
AND NOT EXISTS (SELECT 1 FROM users WHERE lower(email) = lower($1))`,
|
||||||
|
id.Email, userID); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// syncGrants makes the user's OIDC-sourced access match what their groups grant
|
||||||
|
// now, and touches nothing else.
|
||||||
|
//
|
||||||
|
// Rows the sync owns are marked source 'oidc'. It adds them, changes their role
|
||||||
|
// and removes them. The last-owner and last-administrator guards do not apply:
|
||||||
|
// they exist to stop a person's mistake, and the provider is the source of truth
|
||||||
|
// for the access it grants, so a team or an install can be left without an
|
||||||
|
// SSO-granted owner. Administrators can always repair a team, and the bootstrap
|
||||||
|
// administrator is a manual one. Rows added by hand are 'manual', and the sync
|
||||||
|
// only ever raises them (turning them into 'oidc' rows), never lowers or removes
|
||||||
|
// them.
|
||||||
|
//
|
||||||
|
// teamRoles is keyed by team ID, not name: a team must already exist, with its
|
||||||
|
// own oidc_member_group/oidc_owner_group set by its owner, before a group can
|
||||||
|
// grant access to it. The sync never creates a team.
|
||||||
|
func syncGrants(ctx context.Context, tx *sql.Tx, userID int64, g oidc.Grants, teamRoles map[int64]string) error {
|
||||||
|
// Administrator. A manual administrator stays one whatever the groups say.
|
||||||
|
if g.Admin {
|
||||||
|
if _, err := tx.ExecContext(ctx,
|
||||||
|
"UPDATE users SET is_admin = true, admin_source = 'oidc' WHERE id = $1 AND NOT is_admin",
|
||||||
|
userID); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
} else if _, err := tx.ExecContext(ctx,
|
||||||
|
"UPDATE users SET is_admin = false, admin_source = 'manual' WHERE id = $1 AND admin_source = 'oidc'",
|
||||||
|
userID); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
|
// Teams. The result of the loop is the set of teams the groups grant.
|
||||||
|
granted := make([]int64, 0, len(teamRoles))
|
||||||
|
for teamID, role := range teamRoles {
|
||||||
|
granted = append(granted, teamID)
|
||||||
|
|
||||||
|
// A row the sync owns follows the groups in both directions. One added by
|
||||||
|
// hand is only raised: a member the owner made an owner by hand is not
|
||||||
|
// demoted because the group says member.
|
||||||
|
if _, err := tx.ExecContext(ctx, `
|
||||||
|
INSERT INTO team_members (team_id, user_id, role, source)
|
||||||
|
VALUES ($1, $2, $3, 'oidc')
|
||||||
|
ON CONFLICT (team_id, user_id) DO UPDATE
|
||||||
|
SET role = excluded.role, source = 'oidc'
|
||||||
|
WHERE team_members.source = 'oidc'
|
||||||
|
OR (excluded.role = $4 AND team_members.role = $5)`,
|
||||||
|
teamID, userID, role, models.RoleOwner, models.RoleMember); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Access the groups no longer grant. granted is never nil, or the ALL
|
||||||
|
// comparison would be against NULL and delete nothing.
|
||||||
|
_, err := tx.ExecContext(ctx,
|
||||||
|
"DELETE FROM team_members WHERE user_id = $1 AND source = 'oidc' AND team_id <> ALL($2)",
|
||||||
|
userID, granted)
|
||||||
|
return err
|
||||||
|
}
|
||||||
@@ -0,0 +1,74 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"database/sql"
|
||||||
|
"net/http"
|
||||||
|
)
|
||||||
|
|
||||||
|
// teamOIDCGroups is one team's own OIDC binding: which group, if any, grants
|
||||||
|
// member access and which grants owner access. The same shape answers GET and
|
||||||
|
// is accepted by PUT. An empty string means no group grants that role here.
|
||||||
|
type teamOIDCGroups struct {
|
||||||
|
MemberGroup string `json:"member_group"`
|
||||||
|
OwnerGroup string `json:"owner_group"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleGetTeamOIDCGroups answers which groups control a team's membership.
|
||||||
|
// Member-gated like the member list itself: this is part of "who is in the
|
||||||
|
// team and why", not a setting only an owner should be able to see.
|
||||||
|
func handleGetTeamOIDCGroups(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
teamID, ok := teamParam(w, r)
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !requireTeamMember(w, r, teamID) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
var g teamOIDCGroups
|
||||||
|
err := db.QueryRowContext(r.Context(),
|
||||||
|
"SELECT COALESCE(oidc_member_group, ''), COALESCE(oidc_owner_group, '') FROM teams WHERE id = $1",
|
||||||
|
teamID).Scan(&g.MemberGroup, &g.OwnerGroup)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusOK, g)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleSetTeamOIDCGroups sets which groups control a team's membership.
|
||||||
|
//
|
||||||
|
// Owner-gated, the same as the schedule, the integrations and the escalation
|
||||||
|
// ladder: this decides who can end up in the team, which is exactly the kind
|
||||||
|
// of thing only the team's own owner (or an administrator repairing it) should
|
||||||
|
// be able to change. An empty string clears a binding.
|
||||||
|
func handleSetTeamOIDCGroups(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
teamID, ok := teamParam(w, r)
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !requireTeamOwner(w, r, teamID) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
var req teamOIDCGroups
|
||||||
|
if err := decodeJSON(r, &req); err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
if _, err := db.ExecContext(r.Context(), `
|
||||||
|
UPDATE teams
|
||||||
|
SET oidc_member_group = NULLIF($1, ''),
|
||||||
|
oidc_owner_group = NULLIF($2, '')
|
||||||
|
WHERE id = $3`,
|
||||||
|
req.MemberGroup, req.OwnerGroup, teamID); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
w.WriteHeader(http.StatusNoContent)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,854 @@
|
|||||||
|
package api_test
|
||||||
|
|
||||||
|
import (
|
||||||
|
"crypto"
|
||||||
|
"crypto/rand"
|
||||||
|
"crypto/rsa"
|
||||||
|
"crypto/sha256"
|
||||||
|
"encoding/base64"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"math/big"
|
||||||
|
"net/http"
|
||||||
|
"net/http/httptest"
|
||||||
|
"net/url"
|
||||||
|
"strings"
|
||||||
|
"sync"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||||
|
)
|
||||||
|
|
||||||
|
// fakeIdP is just enough of an OpenID Connect provider for terdut to sign
|
||||||
|
// somebody in against: discovery, a key set and a token endpoint that checks the
|
||||||
|
// PKCE verifier. There is no authorize endpoint; the tests read the URL terdut
|
||||||
|
// redirects to and play the part of the browser and the person themselves.
|
||||||
|
type fakeIdP struct {
|
||||||
|
*httptest.Server
|
||||||
|
key *rsa.PrivateKey
|
||||||
|
|
||||||
|
mu sync.Mutex
|
||||||
|
codes map[string]pendingCode
|
||||||
|
}
|
||||||
|
|
||||||
|
type pendingCode struct {
|
||||||
|
claims map[string]any
|
||||||
|
challenge string
|
||||||
|
}
|
||||||
|
|
||||||
|
const (
|
||||||
|
idpClientID = "terdut"
|
||||||
|
idpClientSecret = "s3cret"
|
||||||
|
)
|
||||||
|
|
||||||
|
func newFakeIdP(t *testing.T) *fakeIdP {
|
||||||
|
t.Helper()
|
||||||
|
key, err := rsa.GenerateKey(rand.Reader, 2048)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
f := &fakeIdP{key: key, codes: map[string]pendingCode{}}
|
||||||
|
|
||||||
|
mux := http.NewServeMux()
|
||||||
|
mux.HandleFunc("/.well-known/openid-configuration", func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
json.NewEncoder(w).Encode(map[string]any{
|
||||||
|
"issuer": f.URL,
|
||||||
|
"authorization_endpoint": f.URL + "/authorize",
|
||||||
|
"token_endpoint": f.URL + "/token",
|
||||||
|
"jwks_uri": f.URL + "/jwks",
|
||||||
|
"id_token_signing_alg_values_supported": []string{"RS256"},
|
||||||
|
"response_types_supported": []string{"code"},
|
||||||
|
"subject_types_supported": []string{"public"},
|
||||||
|
})
|
||||||
|
})
|
||||||
|
mux.HandleFunc("/jwks", func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
b64 := base64.RawURLEncoding.EncodeToString
|
||||||
|
json.NewEncoder(w).Encode(map[string]any{"keys": []map[string]string{{
|
||||||
|
"kty": "RSA", "kid": "k1", "use": "sig", "alg": "RS256",
|
||||||
|
"n": b64(key.N.Bytes()),
|
||||||
|
"e": b64(big.NewInt(int64(key.E)).Bytes()),
|
||||||
|
}}})
|
||||||
|
})
|
||||||
|
mux.HandleFunc("/token", func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
r.ParseForm()
|
||||||
|
user, pass, basic := r.BasicAuth()
|
||||||
|
if !basic {
|
||||||
|
user, pass = r.PostForm.Get("client_id"), r.PostForm.Get("client_secret")
|
||||||
|
}
|
||||||
|
if user != idpClientID || pass != idpClientSecret {
|
||||||
|
http.Error(w, `{"error":"invalid_client"}`, http.StatusUnauthorized)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
f.mu.Lock()
|
||||||
|
p, ok := f.codes[r.PostForm.Get("code")]
|
||||||
|
delete(f.codes, r.PostForm.Get("code")) // single use, like a real provider
|
||||||
|
f.mu.Unlock()
|
||||||
|
sum := sha256.Sum256([]byte(r.PostForm.Get("code_verifier")))
|
||||||
|
if !ok || base64.RawURLEncoding.EncodeToString(sum[:]) != p.challenge {
|
||||||
|
http.Error(w, `{"error":"invalid_grant"}`, http.StatusBadRequest)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
// oauth2 picks the parser from the content type; without this it reads
|
||||||
|
// the body as a form, finds no token and retries, spending the code.
|
||||||
|
w.Header().Set("Content-Type", "application/json")
|
||||||
|
json.NewEncoder(w).Encode(map[string]any{
|
||||||
|
"access_token": "unused", "token_type": "Bearer", "expires_in": 300,
|
||||||
|
"id_token": f.sign(t, p.claims),
|
||||||
|
})
|
||||||
|
})
|
||||||
|
f.Server = httptest.NewServer(mux)
|
||||||
|
t.Cleanup(f.Close)
|
||||||
|
return f
|
||||||
|
}
|
||||||
|
|
||||||
|
// sign returns claims as an RS256 JWT.
|
||||||
|
func (f *fakeIdP) sign(t *testing.T, claims map[string]any) string {
|
||||||
|
t.Helper()
|
||||||
|
enc := func(v any) string {
|
||||||
|
b, _ := json.Marshal(v)
|
||||||
|
return base64.RawURLEncoding.EncodeToString(b)
|
||||||
|
}
|
||||||
|
signing := enc(map[string]string{"alg": "RS256", "kid": "k1", "typ": "JWT"}) + "." + enc(claims)
|
||||||
|
sum := sha256.Sum256([]byte(signing))
|
||||||
|
sig, err := rsa.SignPKCS1v15(rand.Reader, f.key, crypto.SHA256, sum[:])
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
return signing + "." + base64.RawURLEncoding.EncodeToString(sig)
|
||||||
|
}
|
||||||
|
|
||||||
|
// idpUser is who signs in, as the provider describes them.
|
||||||
|
type idpUser struct {
|
||||||
|
sub, username, email string
|
||||||
|
unverified bool
|
||||||
|
groups []string
|
||||||
|
badNonce bool
|
||||||
|
}
|
||||||
|
|
||||||
|
// ssoConfig is a terdut configuration wired to idp: terdut-users may sign in,
|
||||||
|
// terdut-admins administer. Which groups grant which team is not config
|
||||||
|
// anymore — it is each team's own oidc_member_group/oidc_owner_group, so a
|
||||||
|
// test that needs one seeds it with seedTeam.
|
||||||
|
func ssoConfig(idp *fakeIdP) config.Config {
|
||||||
|
c := testConfig()
|
||||||
|
c.OIDC = config.OIDC{
|
||||||
|
Issuer: idp.URL,
|
||||||
|
ClientID: idpClientID,
|
||||||
|
ClientSecret: idpClientSecret,
|
||||||
|
Name: "Authentik",
|
||||||
|
Scopes: []string{"openid", "profile", "email"},
|
||||||
|
UsernameClaim: "preferred_username",
|
||||||
|
EmailClaim: "email",
|
||||||
|
GroupsClaim: "groups",
|
||||||
|
AllowedGroups: []string{"terdut-users"},
|
||||||
|
AdminGroup: "terdut-admins",
|
||||||
|
SessionMaxAge: 12 * time.Hour,
|
||||||
|
}
|
||||||
|
return c
|
||||||
|
}
|
||||||
|
|
||||||
|
// seedTeam creates a team with an OIDC group binding, the way an owner would
|
||||||
|
// set one from the Members tab. Teams are no longer created by the sync
|
||||||
|
// itself, so a test whose groups should grant something needs the team to
|
||||||
|
// already exist. An empty group means that role is not granted by one.
|
||||||
|
func (s *ts) seedTeam(t *testing.T, name, memberGroup, ownerGroup string) int64 {
|
||||||
|
t.Helper()
|
||||||
|
var id int64
|
||||||
|
err := s.db.QueryRow(`
|
||||||
|
INSERT INTO teams (name, oidc_member_group, oidc_owner_group)
|
||||||
|
VALUES ($1, NULLIF($2, ''), NULLIF($3, '')) RETURNING id`,
|
||||||
|
name, memberGroup, ownerGroup).Scan(&id)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("seed team %q: %v", name, err)
|
||||||
|
}
|
||||||
|
return id
|
||||||
|
}
|
||||||
|
|
||||||
|
func newSSOTS(t *testing.T, idp *fakeIdP, tweak ...func(*config.Config)) *ts {
|
||||||
|
t.Helper()
|
||||||
|
c := ssoConfig(idp)
|
||||||
|
for _, f := range tweak {
|
||||||
|
f(&c)
|
||||||
|
}
|
||||||
|
return newTSWith(t, api.DeadmanConfig{}, api.NotifyConfig{PublicURL: "http://terdut.test"}, c)
|
||||||
|
}
|
||||||
|
|
||||||
|
// ssoBrowser is a browser that does not follow redirects, so a test can read
|
||||||
|
// where each step sends it.
|
||||||
|
func ssoBrowser(t *testing.T, s *ts) *browser {
|
||||||
|
t.Helper()
|
||||||
|
b := newBrowser(t, s.URL)
|
||||||
|
b.CheckRedirect = func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse }
|
||||||
|
return b
|
||||||
|
}
|
||||||
|
|
||||||
|
// startLogin visits /api/oidc/login and returns what terdut asked the provider
|
||||||
|
// for: the state, nonce and PKCE challenge.
|
||||||
|
func startLogin(t *testing.T, idp *fakeIdP, b *browser) (state, nonce, challenge string) {
|
||||||
|
t.Helper()
|
||||||
|
resp := b.do(t, http.MethodGet, "/api/oidc/login", nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusFound {
|
||||||
|
t.Fatalf("login start: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
loc, err := url.Parse(resp.Header.Get("Location"))
|
||||||
|
if err != nil || !strings.HasPrefix(loc.String(), idp.URL+"/authorize") {
|
||||||
|
t.Fatalf("login redirected to %q, want the provider", resp.Header.Get("Location"))
|
||||||
|
}
|
||||||
|
q := loc.Query()
|
||||||
|
if q.Get("code_challenge_method") != "S256" || q.Get("client_id") != idpClientID ||
|
||||||
|
q.Get("redirect_uri") != "http://terdut.test/api/oidc/callback" || q.Get("response_type") != "code" {
|
||||||
|
t.Fatalf("unexpected authorization request: %v", q)
|
||||||
|
}
|
||||||
|
return q.Get("state"), q.Get("nonce"), q.Get("code_challenge")
|
||||||
|
}
|
||||||
|
|
||||||
|
// issueCode has the provider authenticate u and hand back an authorization code.
|
||||||
|
func (f *fakeIdP) issueCode(u idpUser, nonce, challenge string) string {
|
||||||
|
if u.badNonce {
|
||||||
|
nonce = "not-the-nonce"
|
||||||
|
}
|
||||||
|
claims := map[string]any{
|
||||||
|
"iss": f.URL, "sub": u.sub, "aud": idpClientID,
|
||||||
|
"iat": time.Now().Unix(), "exp": time.Now().Add(5 * time.Minute).Unix(),
|
||||||
|
"nonce": nonce,
|
||||||
|
"preferred_username": u.username,
|
||||||
|
"email": u.email,
|
||||||
|
"email_verified": !u.unverified,
|
||||||
|
"groups": u.groups,
|
||||||
|
}
|
||||||
|
f.mu.Lock()
|
||||||
|
defer f.mu.Unlock()
|
||||||
|
code := fmt.Sprintf("code-%d", len(f.codes)+int(time.Now().UnixNano()%1e6))
|
||||||
|
f.codes[code] = pendingCode{claims: claims, challenge: challenge}
|
||||||
|
return code
|
||||||
|
}
|
||||||
|
|
||||||
|
// callback delivers the provider's answer to terdut and returns where terdut
|
||||||
|
// sends the browser next.
|
||||||
|
func callback(t *testing.T, b *browser, code, state string) string {
|
||||||
|
t.Helper()
|
||||||
|
resp := b.do(t, http.MethodGet, "/api/oidc/callback?code="+url.QueryEscape(code)+"&state="+url.QueryEscape(state), nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusFound {
|
||||||
|
t.Fatalf("callback: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
return resp.Header.Get("Location")
|
||||||
|
}
|
||||||
|
|
||||||
|
// signInSSO runs a whole sign-in and returns the Location the callback ended on.
|
||||||
|
func signInSSO(t *testing.T, idp *fakeIdP, b *browser, u idpUser) string {
|
||||||
|
t.Helper()
|
||||||
|
state, nonce, challenge := startLogin(t, idp, b)
|
||||||
|
return callback(t, b, idp.issueCode(u, nonce, challenge), state)
|
||||||
|
}
|
||||||
|
|
||||||
|
var alice = idpUser{sub: "sub-alice", username: "alice", email: "alice@example.com", groups: []string{"terdut-users", "sre"}}
|
||||||
|
|
||||||
|
func withGroups(u idpUser, groups ...string) idpUser {
|
||||||
|
u.groups = groups
|
||||||
|
return u
|
||||||
|
}
|
||||||
|
|
||||||
|
// meOf reads /api/me over the browser's session.
|
||||||
|
func meOf(t *testing.T, b *browser) (status int, username string, isAdmin, hasPassword bool) {
|
||||||
|
t.Helper()
|
||||||
|
resp := b.do(t, http.MethodGet, "/api/me", nil)
|
||||||
|
defer resp.Body.Close()
|
||||||
|
var me struct {
|
||||||
|
User struct {
|
||||||
|
Username string `json:"username"`
|
||||||
|
IsAdmin bool `json:"is_admin"`
|
||||||
|
} `json:"user"`
|
||||||
|
HasPassword bool `json:"has_password"`
|
||||||
|
}
|
||||||
|
json.NewDecoder(resp.Body).Decode(&me)
|
||||||
|
return resp.StatusCode, me.User.Username, me.User.IsAdmin, me.HasPassword
|
||||||
|
}
|
||||||
|
|
||||||
|
// memberships lists a user's teams as name -> "role/source".
|
||||||
|
func (s *ts) memberships(t *testing.T, username string) map[string]string {
|
||||||
|
t.Helper()
|
||||||
|
rows, err := s.db.Query(`
|
||||||
|
SELECT t.name, m.role, m.source FROM team_members m
|
||||||
|
JOIN teams t ON t.id = m.team_id JOIN users u ON u.id = m.user_id
|
||||||
|
WHERE u.username = $1`, username)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
defer rows.Close()
|
||||||
|
out := map[string]string{}
|
||||||
|
for rows.Next() {
|
||||||
|
var name, role, source string
|
||||||
|
rows.Scan(&name, &role, &source)
|
||||||
|
out[name] = role + "/" + source
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
func sameMap(a, b map[string]string) bool {
|
||||||
|
if len(a) != len(b) {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
for k, v := range a {
|
||||||
|
if b[k] != v {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_FirstSignInCreatesUserAndGrantsTeams(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
s.seedTeam(t, "SRE", "sre", "sre-leads")
|
||||||
|
s.seedTeam(t, "Platform", "platform", "")
|
||||||
|
b := ssoBrowser(t, s)
|
||||||
|
|
||||||
|
if loc := signInSSO(t, idp, b, withGroups(alice, "terdut-users", "sre", "platform")); loc != "/" {
|
||||||
|
t.Fatalf("signed in and was sent to %q, want /", loc)
|
||||||
|
}
|
||||||
|
status, name, isAdmin, hasPassword := meOf(t, b)
|
||||||
|
if status != http.StatusOK || name != "alice" || isAdmin || hasPassword {
|
||||||
|
t.Fatalf("me: status %d user %q admin %v has_password %v", status, name, isAdmin, hasPassword)
|
||||||
|
}
|
||||||
|
want := map[string]string{"SRE": "member/oidc", "Platform": "member/oidc"}
|
||||||
|
if got := s.memberships(t, "alice"); !sameMap(got, want) {
|
||||||
|
t.Errorf("memberships %v, want %v", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A group matching no team's own binding grants nothing and creates nothing:
|
||||||
|
// unlike the old global mapping, the sync never creates a team by name.
|
||||||
|
func TestSSO_NoAutoCreateTeam(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
|
||||||
|
var before int
|
||||||
|
s.db.QueryRow("SELECT COUNT(*) FROM teams").Scan(&before)
|
||||||
|
|
||||||
|
signInSSO(t, idp, ssoBrowser(t, s), alice) // groups include "sre"; no team names it
|
||||||
|
if got := s.memberships(t, "alice"); len(got) != 0 {
|
||||||
|
t.Errorf("memberships %v, want none: no team's oidc_member_group/oidc_owner_group is set", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
var after int
|
||||||
|
s.db.QueryRow("SELECT COUNT(*) FROM teams").Scan(&after)
|
||||||
|
if after != before {
|
||||||
|
t.Errorf("team count %d -> %d, want no team created", before, after)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_RefusedOutsideAllowedGroups(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
b := ssoBrowser(t, s)
|
||||||
|
|
||||||
|
loc := signInSSO(t, idp, b, withGroups(alice, "sre", "terdut-admins"))
|
||||||
|
if loc != "/?sso_error=not_allowed" {
|
||||||
|
t.Fatalf("sent to %q, want the not_allowed error", loc)
|
||||||
|
}
|
||||||
|
if status, _, _, _ := meOf(t, b); status != http.StatusUnauthorized {
|
||||||
|
t.Errorf("a refused sign-in must not leave a session: /api/me %d", status)
|
||||||
|
}
|
||||||
|
var n int
|
||||||
|
s.db.QueryRow("SELECT COUNT(*) FROM users WHERE username = 'alice'").Scan(&n)
|
||||||
|
if n != 0 {
|
||||||
|
t.Error("a refused sign-in must not create the user")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_AdminFollowsTheAdminGroup(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
|
||||||
|
signInSSO(t, idp, ssoBrowser(t, s), withGroups(alice, "terdut-users", "terdut-admins"))
|
||||||
|
var isAdmin bool
|
||||||
|
var source string
|
||||||
|
read := func() {
|
||||||
|
s.db.QueryRow("SELECT is_admin, admin_source FROM users WHERE username = 'alice'").Scan(&isAdmin, &source)
|
||||||
|
}
|
||||||
|
if read(); !isAdmin || source != "oidc" {
|
||||||
|
t.Fatalf("after admin sign-in: admin %v source %q", isAdmin, source)
|
||||||
|
}
|
||||||
|
|
||||||
|
signInSSO(t, idp, ssoBrowser(t, s), withGroups(alice, "terdut-users"))
|
||||||
|
if read(); isAdmin || source != "manual" {
|
||||||
|
t.Errorf("after losing the group: admin %v source %q, want revoked and manual", isAdmin, source)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_ManualAdminIsNeverRevoked(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp, func(c *config.Config) { c.OIDC.TrustEmail = true })
|
||||||
|
|
||||||
|
// The bootstrap administrator is a manual one. Signing in through the
|
||||||
|
// provider without the admin group must not take that away.
|
||||||
|
signInSSO(t, idp, ssoBrowser(t, s), idpUser{sub: "sub-admin", username: "admin", email: "admin@test.com", groups: []string{"terdut-users"}})
|
||||||
|
var isAdmin bool
|
||||||
|
var source string
|
||||||
|
s.db.QueryRow("SELECT is_admin, admin_source FROM users WHERE username = 'admin'").Scan(&isAdmin, &source)
|
||||||
|
if !isAdmin || source != "manual" {
|
||||||
|
t.Errorf("admin %v source %q, want still a manual admin", isAdmin, source)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_LosingAGroupRemovesOnlyManagedAccess(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
s.seedTeam(t, "SRE", "sre", "sre-leads")
|
||||||
|
|
||||||
|
signInSSO(t, idp, ssoBrowser(t, s), alice)
|
||||||
|
// Somebody adds alice to another team by hand.
|
||||||
|
s.exec(t, "INSERT INTO teams (name) VALUES ('Hand')")
|
||||||
|
s.exec(t, `INSERT INTO team_members (team_id, user_id, role)
|
||||||
|
SELECT (SELECT id FROM teams WHERE name = 'Hand'), id, 'member' FROM users WHERE username = 'alice'`)
|
||||||
|
|
||||||
|
signInSSO(t, idp, ssoBrowser(t, s), withGroups(alice, "terdut-users"))
|
||||||
|
want := map[string]string{"Hand": "member/manual"}
|
||||||
|
if got := s.memberships(t, "alice"); !sameMap(got, want) {
|
||||||
|
t.Errorf("memberships %v, want %v: the SRE row is the sync's to remove, Hand is not", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_HighestRoleWinsAndRoleChangesFollow(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
s.seedTeam(t, "SRE", "sre", "sre-leads")
|
||||||
|
|
||||||
|
signInSSO(t, idp, ssoBrowser(t, s), withGroups(alice, "terdut-users", "sre", "sre-leads"))
|
||||||
|
if got := s.memberships(t, "alice"); !sameMap(got, map[string]string{"SRE": "owner/oidc"}) {
|
||||||
|
t.Errorf("both groups: %v, want owner", got)
|
||||||
|
}
|
||||||
|
signInSSO(t, idp, ssoBrowser(t, s), withGroups(alice, "terdut-users", "sre"))
|
||||||
|
if got := s.memberships(t, "alice"); !sameMap(got, map[string]string{"SRE": "member/oidc"}) {
|
||||||
|
t.Errorf("lead group dropped: %v, want member", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_ManualMemberIsRaisedNeverLowered(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
|
||||||
|
// alice exists locally, is a manual owner of SRE, and is linked by email.
|
||||||
|
s.exec(t, "INSERT INTO users (username, email) VALUES ('alice', 'alice@example.com')")
|
||||||
|
s.seedTeam(t, "SRE", "sre", "")
|
||||||
|
s.exec(t, `INSERT INTO team_members (team_id, user_id, role)
|
||||||
|
VALUES ((SELECT id FROM teams WHERE name = 'SRE'), (SELECT id FROM users WHERE username = 'alice'), 'owner')`)
|
||||||
|
|
||||||
|
signInSSO(t, idp, ssoBrowser(t, s), alice) // the group only grants member
|
||||||
|
if got := s.memberships(t, "alice"); !sameMap(got, map[string]string{"SRE": "owner/manual"}) {
|
||||||
|
t.Errorf("%v: a hand-made owner must not be lowered by a member mapping", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_LinksExistingUserByVerifiedEmail(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
s.exec(t, "INSERT INTO users (username, email) VALUES ('alice-local', 'Alice@Example.com')")
|
||||||
|
|
||||||
|
b := ssoBrowser(t, s)
|
||||||
|
signInSSO(t, idp, b, alice)
|
||||||
|
if _, name, _, _ := meOf(t, b); name != "alice-local" {
|
||||||
|
t.Errorf("signed in as %q, want the existing local user", name)
|
||||||
|
}
|
||||||
|
var users, identities int
|
||||||
|
s.db.QueryRow("SELECT COUNT(*) FROM users").Scan(&users)
|
||||||
|
s.db.QueryRow("SELECT COUNT(*) FROM user_identities").Scan(&identities)
|
||||||
|
if users != 2 || identities != 1 { // admin + alice-local
|
||||||
|
t.Errorf("%d users, %d identities: linking must not create a second user", users, identities)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_UnverifiedEmailIsNotLinkedUnlessTrusted(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
unverified := alice
|
||||||
|
unverified.unverified = true
|
||||||
|
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
s.exec(t, "INSERT INTO users (username, email) VALUES ('alice-local', 'alice@example.com')")
|
||||||
|
if loc := signInSSO(t, idp, ssoBrowser(t, s), unverified); loc != "/?sso_error=email_conflict" {
|
||||||
|
t.Errorf("unverified email: sent to %q, want email_conflict", loc)
|
||||||
|
}
|
||||||
|
|
||||||
|
trusting := newSSOTS(t, idp, func(c *config.Config) { c.OIDC.TrustEmail = true })
|
||||||
|
trusting.exec(t, "INSERT INTO users (username, email) VALUES ('alice-local', 'alice@example.com')")
|
||||||
|
b := ssoBrowser(t, trusting)
|
||||||
|
if loc := signInSSO(t, idp, b, unverified); loc != "/" {
|
||||||
|
t.Fatalf("trusted email: sent to %q, want /", loc)
|
||||||
|
}
|
||||||
|
if _, name, _, _ := meOf(t, b); name != "alice-local" {
|
||||||
|
t.Errorf("signed in as %q, want the existing local user", name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_RecycledEmailDoesNotTakeOverALinkedAccount(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
signInSSO(t, idp, ssoBrowser(t, s), alice)
|
||||||
|
|
||||||
|
// A different person at the provider, same address.
|
||||||
|
other := alice
|
||||||
|
other.sub = "sub-someone-else"
|
||||||
|
if loc := signInSSO(t, idp, ssoBrowser(t, s), other); loc != "/?sso_error=email_conflict" {
|
||||||
|
t.Errorf("sent to %q, want email_conflict", loc)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_UsernameCollisionGetsASuffix(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
s.exec(t, "INSERT INTO users (username, email) VALUES ('alice', 'someone-else@example.com')")
|
||||||
|
|
||||||
|
b := ssoBrowser(t, s)
|
||||||
|
signInSSO(t, idp, b, alice)
|
||||||
|
if _, name, _, _ := meOf(t, b); name != "alice-2" {
|
||||||
|
t.Errorf("username %q, want alice-2", name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_ProfileFollowsTheProvider(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
signInSSO(t, idp, ssoBrowser(t, s), alice)
|
||||||
|
|
||||||
|
renamed := alice
|
||||||
|
renamed.username, renamed.email = "alice.smith", "alice.smith@example.com"
|
||||||
|
b := ssoBrowser(t, s)
|
||||||
|
signInSSO(t, idp, b, renamed)
|
||||||
|
if _, name, _, _ := meOf(t, b); name != "alice.smith" {
|
||||||
|
t.Errorf("username %q, want the provider's new one", name)
|
||||||
|
}
|
||||||
|
var email string
|
||||||
|
s.db.QueryRow("SELECT email FROM users WHERE username = 'alice.smith'").Scan(&email)
|
||||||
|
if email != "alice.smith@example.com" {
|
||||||
|
t.Errorf("email %q", email)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_DisabledUserIsRefused(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
signInSSO(t, idp, ssoBrowser(t, s), alice)
|
||||||
|
s.exec(t, "UPDATE users SET disabled_at = 1 WHERE username = 'alice'")
|
||||||
|
|
||||||
|
b := ssoBrowser(t, s)
|
||||||
|
if loc := signInSSO(t, idp, b, alice); loc != "/?sso_error=disabled" {
|
||||||
|
t.Errorf("sent to %q, want disabled", loc)
|
||||||
|
}
|
||||||
|
if status, _, _, _ := meOf(t, b); status != http.StatusUnauthorized {
|
||||||
|
t.Errorf("/api/me %d, want 401", status)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSSO_FirstUserIsRefusedUntilBootstrap is terdut-operator#1: an
|
||||||
|
// otherwise-ordinary OIDC sign-in against a brand-new, not-yet-bootstrapped
|
||||||
|
// install must not be allowed to create the first user and win the race
|
||||||
|
// POST /api/bootstrap expects to win uncontested. Built directly over
|
||||||
|
// api.NewRouter rather than newSSOTS/newTS, both of which bootstrap before
|
||||||
|
// a test body ever runs -- exactly the state this test needs to not have yet.
|
||||||
|
func TestSSO_FirstUserIsRefusedUntilBootstrap(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
database := newTestDB(t)
|
||||||
|
srv := httptest.NewServer(api.NewRouter(database, api.NotifyConfig{PublicURL: "http://terdut.test"}, ssoConfig(idp), "test"))
|
||||||
|
t.Cleanup(srv.Close)
|
||||||
|
|
||||||
|
first := newBrowser(t, srv.URL)
|
||||||
|
first.CheckRedirect = func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse }
|
||||||
|
if loc := signInSSO(t, idp, first, alice); loc != "/?sso_error=not_bootstrapped" {
|
||||||
|
t.Fatalf("sent to %q, want not_bootstrapped", loc)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Bootstrap the install for real, the way terdut-operator's own
|
||||||
|
// reconcileBootstrap does.
|
||||||
|
resp, err := http.Post(srv.URL+"/api/bootstrap", "application/json",
|
||||||
|
strings.NewReader(`{"username":"admin","email":"admin@test.com"}`))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusCreated {
|
||||||
|
t.Fatalf("bootstrap: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The same identity, signing in again, is this install's ordinary first
|
||||||
|
// SSO user now -- no longer refused.
|
||||||
|
second := newBrowser(t, srv.URL)
|
||||||
|
second.CheckRedirect = func(*http.Request, []*http.Request) error { return http.ErrUseLastResponse }
|
||||||
|
if loc := signInSSO(t, idp, second, alice); loc != "/" {
|
||||||
|
t.Errorf("sent to %q after bootstrap, want success", loc)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_NoEmailIsRefused(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
noEmail := alice
|
||||||
|
noEmail.email = ""
|
||||||
|
if loc := signInSSO(t, idp, ssoBrowser(t, s), noEmail); loc != "/?sso_error=no_email" {
|
||||||
|
t.Errorf("sent to %q, want no_email", loc)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_SessionIsCappedAndDoesNotSlidePastTheCap(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
b := ssoBrowser(t, s)
|
||||||
|
signInSSO(t, idp, b, alice)
|
||||||
|
|
||||||
|
var expires, ceiling int64
|
||||||
|
s.db.QueryRow(`SELECT expires_at, max_expires_at FROM sessions ORDER BY id DESC LIMIT 1`).Scan(&expires, &ceiling)
|
||||||
|
inTwelveHours := time.Now().Add(12 * time.Hour).Unix()
|
||||||
|
if ceiling < inTwelveHours-60 || ceiling > inTwelveHours+60 || expires != ceiling {
|
||||||
|
t.Fatalf("expires %d ceiling %d, want both about %d", expires, ceiling, inTwelveHours)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Age the session so the next request would slide it, with a ceiling well
|
||||||
|
// inside the ordinary 30 days.
|
||||||
|
s.exec(t, "UPDATE sessions SET last_seen_at = last_seen_at - 7200")
|
||||||
|
if status, _, _, _ := meOf(t, b); status != http.StatusOK {
|
||||||
|
t.Fatalf("/api/me %d", status)
|
||||||
|
}
|
||||||
|
var after int64
|
||||||
|
s.db.QueryRow(`SELECT expires_at FROM sessions ORDER BY id DESC LIMIT 1`).Scan(&after)
|
||||||
|
if after > ceiling {
|
||||||
|
t.Errorf("expiry slid to %d, past the ceiling %d", after, ceiling)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_PasswordSessionsStillSlideWithoutACeiling(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
b := signedIn(t, s)
|
||||||
|
var ceiling *int64
|
||||||
|
s.db.QueryRow(`SELECT max_expires_at FROM sessions ORDER BY id DESC LIMIT 1`).Scan(&ceiling)
|
||||||
|
if ceiling != nil {
|
||||||
|
t.Errorf("a password session has a ceiling %d, want none", *ceiling)
|
||||||
|
}
|
||||||
|
s.exec(t, "UPDATE sessions SET last_seen_at = last_seen_at - 7200, expires_at = expires_at - 7200")
|
||||||
|
var before, after int64
|
||||||
|
s.db.QueryRow(`SELECT expires_at FROM sessions ORDER BY id DESC LIMIT 1`).Scan(&before)
|
||||||
|
meOf(t, b)
|
||||||
|
s.db.QueryRow(`SELECT expires_at FROM sessions ORDER BY id DESC LIMIT 1`).Scan(&after)
|
||||||
|
if after <= before {
|
||||||
|
t.Errorf("expiry %d -> %d, want it to slide forward", before, after)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_StateIsSingleUseAndBoundToTheBrowser(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
|
||||||
|
// Replaying a callback finds no state.
|
||||||
|
b := ssoBrowser(t, s)
|
||||||
|
state, nonce, challenge := startLogin(t, idp, b)
|
||||||
|
code := idp.issueCode(alice, nonce, challenge)
|
||||||
|
if loc := callback(t, b, code, state); loc != "/" {
|
||||||
|
t.Fatalf("first callback sent to %q", loc)
|
||||||
|
}
|
||||||
|
if loc := callback(t, b, idp.issueCode(alice, nonce, challenge), state); loc != "/?sso_error=expired" {
|
||||||
|
t.Errorf("replayed state: sent to %q, want expired", loc)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A callback from a browser that did not start the login is refused, which
|
||||||
|
// is what stops a login being planted on somebody else.
|
||||||
|
victim := ssoBrowser(t, s)
|
||||||
|
state, nonce, challenge = startLogin(t, idp, ssoBrowser(t, s)) // the attacker's
|
||||||
|
if loc := callback(t, victim, idp.issueCode(alice, nonce, challenge), state); loc != "/?sso_error=expired" {
|
||||||
|
t.Errorf("foreign browser: sent to %q, want expired", loc)
|
||||||
|
}
|
||||||
|
if status, _, _, _ := meOf(t, victim); status != http.StatusUnauthorized {
|
||||||
|
t.Errorf("the victim has a session: /api/me %d", status)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_WrongNonceIsRefused(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
bad := alice
|
||||||
|
bad.badNonce = true
|
||||||
|
b := ssoBrowser(t, s)
|
||||||
|
if loc := signInSSO(t, idp, b, bad); loc != "/?sso_error=failed" {
|
||||||
|
t.Errorf("sent to %q, want failed", loc)
|
||||||
|
}
|
||||||
|
if status, _, _, _ := meOf(t, b); status != http.StatusUnauthorized {
|
||||||
|
t.Errorf("/api/me %d, want 401", status)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_ProviderErrorGoesBackToTheUI(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
b := ssoBrowser(t, s)
|
||||||
|
resp := b.do(t, http.MethodGet, "/api/oidc/callback?error=access_denied", nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if loc := resp.Header.Get("Location"); resp.StatusCode != http.StatusFound || loc != "/?sso_error=denied" {
|
||||||
|
t.Errorf("%d to %q, want a redirect to denied", resp.StatusCode, loc)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_ManagedAccessCannotBeEditedByHand(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
s.seedTeam(t, "SRE", "sre", "")
|
||||||
|
signInSSO(t, idp, ssoBrowser(t, s), withGroups(alice, "terdut-users", "sre", "terdut-admins"))
|
||||||
|
|
||||||
|
var aliceID, sreID int64
|
||||||
|
s.db.QueryRow("SELECT id FROM users WHERE username = 'alice'").Scan(&aliceID)
|
||||||
|
s.db.QueryRow("SELECT id FROM teams WHERE name = 'SRE'").Scan(&sreID)
|
||||||
|
teamPath := fmt.Sprintf("/api/teams/%d/members", sreID)
|
||||||
|
|
||||||
|
// The bootstrap admin is a system administrator, so may manage SRE.
|
||||||
|
for _, c := range []struct {
|
||||||
|
name, method, path string
|
||||||
|
body any
|
||||||
|
}{
|
||||||
|
{"role change", http.MethodPost, teamPath, map[string]any{"user_id": aliceID, "role": "owner"}},
|
||||||
|
{"removal", http.MethodDelete, fmt.Sprintf("%s/%d", teamPath, aliceID), nil},
|
||||||
|
{"admin revoke", http.MethodPut, fmt.Sprintf("/api/users/%d/admin", aliceID), map[string]any{"is_admin": false}},
|
||||||
|
} {
|
||||||
|
resp := s.req(t, c.method, c.path, c.body)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusConflict {
|
||||||
|
t.Errorf("%s: %d, want 409", c.name, resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if got := s.memberships(t, "alice"); !sameMap(got, map[string]string{"SRE": "member/oidc"}) {
|
||||||
|
t.Errorf("memberships changed by a refused edit: %v", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_PasswordLoginCanBeSwitchedOff(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp, func(c *config.Config) { c.DisablePasswordLogin = true })
|
||||||
|
b := newBrowser(t, s.URL)
|
||||||
|
|
||||||
|
resp := b.login(t, "admin", "whatever-password")
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("login: %d, want 403", resp.StatusCode)
|
||||||
|
}
|
||||||
|
resp = b.do(t, http.MethodPost, "/api/signup", map[string]string{"username": "x", "email": "x@example.com", "password": "correct horse battery"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("signup: %d, want 403", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
var cfg struct {
|
||||||
|
PasswordLogin bool `json:"password_login"`
|
||||||
|
OIDC struct {
|
||||||
|
Enabled bool `json:"enabled"`
|
||||||
|
Name string `json:"name"`
|
||||||
|
} `json:"oidc"`
|
||||||
|
}
|
||||||
|
resp = b.do(t, http.MethodGet, "/api/auth/config", nil)
|
||||||
|
defer resp.Body.Close()
|
||||||
|
json.NewDecoder(resp.Body).Decode(&cfg)
|
||||||
|
if cfg.PasswordLogin || !cfg.OIDC.Enabled || cfg.OIDC.Name != "Authentik" {
|
||||||
|
t.Errorf("auth config: %+v", cfg)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAuthConfig_DefaultsToPasswordOnly(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
var cfg struct {
|
||||||
|
PasswordLogin bool `json:"password_login"`
|
||||||
|
OIDC struct {
|
||||||
|
Enabled bool `json:"enabled"`
|
||||||
|
} `json:"oidc"`
|
||||||
|
}
|
||||||
|
resp := newBrowser(t, s.URL).do(t, http.MethodGet, "/api/auth/config", nil)
|
||||||
|
defer resp.Body.Close()
|
||||||
|
json.NewDecoder(resp.Body).Decode(&cfg)
|
||||||
|
if !cfg.PasswordLogin || cfg.OIDC.Enabled {
|
||||||
|
t.Errorf("auth config: %+v", cfg)
|
||||||
|
}
|
||||||
|
|
||||||
|
// With SSO off the routes do not exist, rather than answering with an error
|
||||||
|
// page a person could land on.
|
||||||
|
resp = newBrowser(t, s.URL).do(t, http.MethodGet, "/api/oidc/login", nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNotFound {
|
||||||
|
t.Errorf("/api/oidc/login with SSO off: %d, want 404", resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_UnreachableProviderRedirectsWithAnError(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
idp.Close() // the provider goes down after terdut has started
|
||||||
|
|
||||||
|
b := ssoBrowser(t, s)
|
||||||
|
resp := b.do(t, http.MethodGet, "/api/oidc/login", nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if loc := resp.Header.Get("Location"); resp.StatusCode != http.StatusFound || loc != "/?sso_error=unavailable" {
|
||||||
|
t.Errorf("%d to %q, want a redirect to unavailable", resp.StatusCode, loc)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSSO_APIShowsWhereAccessCameFrom(t *testing.T) {
|
||||||
|
idp := newFakeIdP(t)
|
||||||
|
s := newSSOTS(t, idp)
|
||||||
|
s.seedTeam(t, "SRE", "sre", "")
|
||||||
|
b := ssoBrowser(t, s)
|
||||||
|
signInSSO(t, idp, b, withGroups(alice, "terdut-users", "sre", "terdut-admins"))
|
||||||
|
|
||||||
|
var aliceID, sreID int64
|
||||||
|
s.db.QueryRow("SELECT id FROM users WHERE username = 'alice'").Scan(&aliceID)
|
||||||
|
s.db.QueryRow("SELECT id FROM teams WHERE name = 'SRE'").Scan(&sreID)
|
||||||
|
|
||||||
|
// Users: alice's administrator flag is the groups', the bootstrap admin's is not.
|
||||||
|
var users []struct {
|
||||||
|
Username string `json:"username"`
|
||||||
|
AdminSource string `json:"admin_source"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/users", nil), &users)
|
||||||
|
got := map[string]string{}
|
||||||
|
for _, u := range users {
|
||||||
|
got[u.Username] = u.AdminSource
|
||||||
|
}
|
||||||
|
if got["alice"] != "oidc" || got["admin"] != "manual" {
|
||||||
|
t.Errorf("admin_source by user: %v", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The team's own member list, as a member sees it.
|
||||||
|
var members []struct {
|
||||||
|
Username string `json:"username"`
|
||||||
|
Source string `json:"source"`
|
||||||
|
}
|
||||||
|
resp := b.do(t, http.MethodGet, fmt.Sprintf("/api/teams/%d/members", sreID), nil)
|
||||||
|
decode(t, resp, &members)
|
||||||
|
if len(members) != 1 || members[0].Username != "alice" || members[0].Source != "oidc" {
|
||||||
|
t.Errorf("team members: %+v", members)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The administrator's view of the same team, and of alice's teams.
|
||||||
|
var adminTeam struct {
|
||||||
|
Members []struct {
|
||||||
|
Username string `json:"username"`
|
||||||
|
Source string `json:"source"`
|
||||||
|
} `json:"members"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodGet, fmt.Sprintf("/api/admin/teams/%d", sreID), nil), &adminTeam)
|
||||||
|
if len(adminTeam.Members) != 1 || adminTeam.Members[0].Source != "oidc" {
|
||||||
|
t.Errorf("admin team members: %+v", adminTeam.Members)
|
||||||
|
}
|
||||||
|
var teams []struct {
|
||||||
|
Name string `json:"name"`
|
||||||
|
Source string `json:"source"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodGet, fmt.Sprintf("/api/users/%d/teams", aliceID), nil), &teams)
|
||||||
|
if len(teams) != 1 || teams[0].Name != "SRE" || teams[0].Source != "oidc" {
|
||||||
|
t.Errorf("user teams: %+v", teams)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The bootstrap admin's own membership is manual.
|
||||||
|
var mine []struct {
|
||||||
|
Source string `json:"source"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/users/1/teams", nil), &mine)
|
||||||
|
if len(mine) == 0 || mine[0].Source != "manual" {
|
||||||
|
t.Errorf("bootstrap admin's teams: %+v", mine)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,72 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"database/sql"
|
||||||
|
"fmt"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||||
|
)
|
||||||
|
|
||||||
|
const (
|
||||||
|
// operatorAccountName is the instance-scoped service account the operator
|
||||||
|
// key belongs to.
|
||||||
|
operatorAccountName = "terdut-operator"
|
||||||
|
|
||||||
|
// operatorKeyName names the one key SeedOperatorKey manages on it, so a
|
||||||
|
// rotation replaces that key and leaves any others alone.
|
||||||
|
operatorKeyName = "seed"
|
||||||
|
)
|
||||||
|
|
||||||
|
// SeedOperatorKey makes key the operator account's credential: it creates the
|
||||||
|
// instance-scoped service account if needed and replaces its "seed" key with
|
||||||
|
// this one. Idempotent, so every replica can run it at every start, and a
|
||||||
|
// rotated key simply wins on the next restart.
|
||||||
|
//
|
||||||
|
// The key is hashed like any other, so only the caller that generated it ever
|
||||||
|
// holds the raw value. An empty key does nothing.
|
||||||
|
func SeedOperatorKey(ctx context.Context, db *sql.DB, key string) error {
|
||||||
|
if key == "" {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
tx, err := db.BeginTx(ctx, nil)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("seed operator key: %w", err)
|
||||||
|
}
|
||||||
|
defer tx.Rollback() //nolint:errcheck
|
||||||
|
|
||||||
|
// Serialise replicas starting together; transaction-scoped, so it needs no
|
||||||
|
// explicit release.
|
||||||
|
if _, err := tx.ExecContext(ctx, "SELECT pg_advisory_xact_lock($1)", operatorKeyLockKey); err != nil {
|
||||||
|
return fmt.Errorf("seed operator key: lock: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
var accountID int64
|
||||||
|
err = tx.QueryRowContext(ctx,
|
||||||
|
"SELECT id FROM service_accounts WHERE name = $1 AND scope = $2",
|
||||||
|
operatorAccountName, models.ServiceAccountScopeInstance).Scan(&accountID)
|
||||||
|
if err == sql.ErrNoRows {
|
||||||
|
err = tx.QueryRowContext(ctx,
|
||||||
|
"INSERT INTO service_accounts (name, scope) VALUES ($1, $2) RETURNING id",
|
||||||
|
operatorAccountName, models.ServiceAccountScopeInstance).Scan(&accountID)
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("seed operator key: account: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if _, err := tx.ExecContext(ctx,
|
||||||
|
"DELETE FROM service_account_keys WHERE service_account_id = $1 AND name = $2",
|
||||||
|
accountID, operatorKeyName); err != nil {
|
||||||
|
return fmt.Errorf("seed operator key: drop old key: %w", err)
|
||||||
|
}
|
||||||
|
if _, err := tx.ExecContext(ctx,
|
||||||
|
"INSERT INTO service_account_keys (service_account_id, key_hash, name) VALUES ($1, $2, $3)",
|
||||||
|
accountID, hashToken(key), operatorKeyName); err != nil {
|
||||||
|
return fmt.Errorf("seed operator key: store key: %w", err)
|
||||||
|
}
|
||||||
|
return tx.Commit()
|
||||||
|
}
|
||||||
|
|
||||||
|
// operatorKeyLockKey is the transaction-scoped advisory lock SeedOperatorKey
|
||||||
|
// holds; distinct from the other lock keys in this package.
|
||||||
|
const operatorKeyLockKey int64 = 7265_0010
|
||||||
@@ -0,0 +1,134 @@
|
|||||||
|
package api_test
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"net/http"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||||
|
)
|
||||||
|
|
||||||
|
const (
|
||||||
|
operatorKeyOne = "tdsa_operator-key-number-one-0123456789"
|
||||||
|
operatorKeyTwo = "tdsa_operator-key-number-two-0123456789"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestSeedOperatorKey_AuthenticatesAsInstanceAccount(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
if err := api.SeedOperatorKey(context.Background(), s.db, operatorKeyOne); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
resp := s.reqAs(t, operatorKeyOne, http.MethodPost, "/api/teams", map[string]string{"name": "seeded"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusCreated {
|
||||||
|
t.Fatalf("seeded key should create a team, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSeedOperatorKey_RotationReplacesAndIsIdempotent(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
for _, key := range []string{operatorKeyOne, operatorKeyOne, operatorKeyTwo} {
|
||||||
|
if err := api.SeedOperatorKey(ctx, s.db, key); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
old := s.reqAs(t, operatorKeyOne, http.MethodGet, "/api/teams", nil)
|
||||||
|
old.Body.Close()
|
||||||
|
if old.StatusCode != http.StatusUnauthorized {
|
||||||
|
t.Errorf("rotated-out key should be refused, got %d", old.StatusCode)
|
||||||
|
}
|
||||||
|
cur := s.reqAs(t, operatorKeyTwo, http.MethodGet, "/api/teams", nil)
|
||||||
|
cur.Body.Close()
|
||||||
|
if cur.StatusCode != http.StatusOK {
|
||||||
|
t.Errorf("current key should work, got %d", cur.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
var accounts, keys int
|
||||||
|
s.db.QueryRow("SELECT COUNT(*) FROM service_accounts WHERE name = 'terdut-operator'").Scan(&accounts)
|
||||||
|
s.db.QueryRow("SELECT COUNT(*) FROM service_account_keys").Scan(&keys)
|
||||||
|
if accounts != 1 || keys != 1 {
|
||||||
|
t.Errorf("expected one account and one key, got %d and %d", accounts, keys)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSeedOperatorKey_EmptyKeyDoesNothing(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
if err := api.SeedOperatorKey(context.Background(), s.db, ""); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
var n int
|
||||||
|
s.db.QueryRow("SELECT COUNT(*) FROM service_accounts").Scan(&n)
|
||||||
|
if n != 0 {
|
||||||
|
t.Errorf("expected no service account, got %d", n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The instance account configures a team it did not create, which is what lets
|
||||||
|
// the operator hold one credential instead of one per team, yet it is not a
|
||||||
|
// member and so reads none of the team's incidents.
|
||||||
|
func TestInstanceAccount_ActsAsOwnerOfAnyTeamButIsNoMember(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
other := newTeam(t, s, "other")
|
||||||
|
if err := api.SeedOperatorKey(context.Background(), s.db, operatorKeyOne); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
rename := s.reqAs(t, operatorKeyOne, http.MethodPut, "/api/teams/"+id64(other.id), map[string]string{"name": "renamed"})
|
||||||
|
rename.Body.Close()
|
||||||
|
if rename.StatusCode >= 300 {
|
||||||
|
t.Errorf("instance account should rename any team, got %d", rename.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Not a member: the team's queue is not visible to it.
|
||||||
|
var queue []map[string]any
|
||||||
|
decode(t, s.reqAs(t, operatorKeyOne, http.MethodGet, "/api/incidents", nil), &queue)
|
||||||
|
if len(queue) != 0 {
|
||||||
|
t.Errorf("instance account should see no incidents, got %v", queue)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// external_id lets automation find its own team again after a crash, without
|
||||||
|
// trusting a display name.
|
||||||
|
func TestCreateTeam_ExternalIDIsIdempotentAndInstanceOnly(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
if err := api.SeedOperatorKey(context.Background(), s.db, operatorKeyOne); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
create := func(name string) (int, map[string]any) {
|
||||||
|
resp := s.reqAs(t, operatorKeyOne, http.MethodPost, "/api/teams",
|
||||||
|
map[string]string{"name": name, "external_id": "ns/platform"})
|
||||||
|
var out map[string]any
|
||||||
|
_ = json.NewDecoder(resp.Body).Decode(&out)
|
||||||
|
resp.Body.Close()
|
||||||
|
return resp.StatusCode, out
|
||||||
|
}
|
||||||
|
|
||||||
|
code, first := create("Platform")
|
||||||
|
if code != http.StatusCreated {
|
||||||
|
t.Fatalf("first create: %d", code)
|
||||||
|
}
|
||||||
|
// Same identity, even under a new display name: the same team comes back.
|
||||||
|
code, again := create("Platform renamed")
|
||||||
|
if code != http.StatusOK || again["id"] != first["id"] {
|
||||||
|
t.Errorf("repeat with the same external_id: want 200 and team %v, got %d %v", first["id"], code, again)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A different identity cannot take the name.
|
||||||
|
resp := s.reqAs(t, operatorKeyOne, http.MethodPost, "/api/teams",
|
||||||
|
map[string]string{"name": "Platform", "external_id": "other/platform"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusConflict {
|
||||||
|
t.Errorf("taken name under another external_id: want 409, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A person cannot set one.
|
||||||
|
resp = s.req(t, http.MethodPost, "/api/teams", map[string]string{"name": "Mine", "external_id": "x/y"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("a user setting external_id: want 403, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,161 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
// This file is internal (package api, not api_test) because loginLimiter and
|
||||||
|
// its blocked/fail/clear methods are unexported, and TestLoginLimiter_SharedAcrossReplicas
|
||||||
|
// specifically needs to construct two separate loginLimiter values pointed at
|
||||||
|
// one database — standing in for two replicas — which only this package can
|
||||||
|
// do. It duplicates testdb_test.go's newTestDB/withSearchPath rather than
|
||||||
|
// importing them: those live in the separate api_test package, compiled from
|
||||||
|
// this directory's external test files, and are not visible here. Same
|
||||||
|
// reasoning as advisory_lock_test.go, which makes the same trade for the
|
||||||
|
// same reason.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"database/sql"
|
||||||
|
"fmt"
|
||||||
|
"net/url"
|
||||||
|
"os"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/db"
|
||||||
|
_ "github.com/jackc/pgx/v5/stdlib"
|
||||||
|
)
|
||||||
|
|
||||||
|
var rateLimiterSchemaSeq int
|
||||||
|
|
||||||
|
// rateLimiterTestDB returns a migrated database private to this test.
|
||||||
|
func rateLimiterTestDB(t *testing.T) *sql.DB {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
dsn := os.Getenv("TERDUT_TEST_DSN")
|
||||||
|
if dsn == "" {
|
||||||
|
t.Fatalf("TERDUT_TEST_DSN is not set: these tests need Postgres.\n" +
|
||||||
|
"Run `make test-db` for a local one, then\n" +
|
||||||
|
" export TERDUT_TEST_DSN=postgres://terdut:terdut@localhost:5432/terdut_test?sslmode=disable")
|
||||||
|
}
|
||||||
|
|
||||||
|
rateLimiterSchemaSeq++
|
||||||
|
schema := fmt.Sprintf("test_rl_%d_%d", os.Getpid(), rateLimiterSchemaSeq)
|
||||||
|
|
||||||
|
admin, err := sql.Open("pgx", dsn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("connect to TERDUT_TEST_DSN: %v", err)
|
||||||
|
}
|
||||||
|
defer admin.Close()
|
||||||
|
if _, err := admin.Exec("CREATE SCHEMA " + schema); err != nil {
|
||||||
|
t.Fatalf("create schema %s: %v", schema, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
database, err := db.Open(rateLimiterWithSearchPath(dsn, schema))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("open db: %v", err)
|
||||||
|
}
|
||||||
|
if err := db.Migrate(database); err != nil {
|
||||||
|
t.Fatalf("migrate: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
t.Cleanup(func() {
|
||||||
|
database.Close()
|
||||||
|
cleanup, err := sql.Open("pgx", dsn)
|
||||||
|
if err != nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer cleanup.Close()
|
||||||
|
if _, err := cleanup.Exec("DROP SCHEMA " + schema + " CASCADE"); err != nil {
|
||||||
|
t.Logf("drop schema %s: %v", schema, err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
return database
|
||||||
|
}
|
||||||
|
|
||||||
|
func rateLimiterWithSearchPath(dsn, schema string) string {
|
||||||
|
opt := "-csearch_path=" + schema
|
||||||
|
if strings.HasPrefix(dsn, "postgres://") || strings.HasPrefix(dsn, "postgresql://") {
|
||||||
|
u, err := url.Parse(dsn)
|
||||||
|
if err == nil {
|
||||||
|
q := u.Query()
|
||||||
|
q.Set("options", opt)
|
||||||
|
u.RawQuery = q.Encode()
|
||||||
|
return u.String()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return dsn + " options='" + opt + "'"
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLoginLimiter_BlocksAtMax(t *testing.T) {
|
||||||
|
database := rateLimiterTestDB(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
l := newLoginLimiter(database)
|
||||||
|
|
||||||
|
for range 3 {
|
||||||
|
if l.blocked(ctx, "k", 3) {
|
||||||
|
t.Fatal("blocked before reaching max")
|
||||||
|
}
|
||||||
|
l.fail(ctx, "k")
|
||||||
|
}
|
||||||
|
if !l.blocked(ctx, "k", 3) {
|
||||||
|
t.Fatal("not blocked after reaching max")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLoginLimiter_ClearResetsTheCount(t *testing.T) {
|
||||||
|
database := rateLimiterTestDB(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
l := newLoginLimiter(database)
|
||||||
|
|
||||||
|
l.fail(ctx, "k")
|
||||||
|
l.fail(ctx, "k")
|
||||||
|
l.clear(ctx, "k")
|
||||||
|
|
||||||
|
if l.blocked(ctx, "k", 1) {
|
||||||
|
t.Fatal("still blocked after clear")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLoginLimiter_KeysAreIndependent(t *testing.T) {
|
||||||
|
database := rateLimiterTestDB(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
l := newLoginLimiter(database)
|
||||||
|
|
||||||
|
l.fail(ctx, "a")
|
||||||
|
if l.blocked(ctx, "b", 1) {
|
||||||
|
t.Fatal("failing one key blocked an unrelated one")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestLoginLimiter_SharedAcrossReplicas is the regression test for the gap
|
||||||
|
// this migration closes: an in-memory limiter would let each replica count
|
||||||
|
// independently, so a caller hitting two different pods could rack up
|
||||||
|
// max*replicaCount failures before either one blocked. Two loginLimiter
|
||||||
|
// values sharing one database, standing in for two replicas behind the same
|
||||||
|
// load balancer, must instead see one combined count.
|
||||||
|
func TestLoginLimiter_SharedAcrossReplicas(t *testing.T) {
|
||||||
|
database := rateLimiterTestDB(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
replicaA := newLoginLimiter(database)
|
||||||
|
replicaB := newLoginLimiter(database)
|
||||||
|
|
||||||
|
const max = 4
|
||||||
|
// Alternate which "replica" records the failure, as a real deployment
|
||||||
|
// would split requests across pods.
|
||||||
|
for i := range max {
|
||||||
|
replica := replicaA
|
||||||
|
if i%2 == 1 {
|
||||||
|
replica = replicaB
|
||||||
|
}
|
||||||
|
if replicaA.blocked(ctx, "k", max) || replicaB.blocked(ctx, "k", max) {
|
||||||
|
t.Fatalf("blocked after only %d of %d failures", i, max)
|
||||||
|
}
|
||||||
|
replica.fail(ctx, "k")
|
||||||
|
}
|
||||||
|
|
||||||
|
if !replicaA.blocked(ctx, "k", max) {
|
||||||
|
t.Fatal("replica A does not see the combined count as blocked")
|
||||||
|
}
|
||||||
|
if !replicaB.blocked(ctx, "k", max) {
|
||||||
|
t.Fatal("replica B does not see the combined count as blocked")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -5,6 +5,7 @@ import (
|
|||||||
"net/http"
|
"net/http"
|
||||||
|
|
||||||
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/oidc"
|
||||||
"git.ryuvia.com/niklas/terdut-server/internal/web"
|
"git.ryuvia.com/niklas/terdut-server/internal/web"
|
||||||
"github.com/go-chi/chi/v5"
|
"github.com/go-chi/chi/v5"
|
||||||
"github.com/go-chi/chi/v5/middleware"
|
"github.com/go-chi/chi/v5/middleware"
|
||||||
@@ -13,15 +14,31 @@ import (
|
|||||||
// NewRouter builds the HTTP surface. notify is passed through to the webhook,
|
// NewRouter builds the HTTP surface. notify is passed through to the webhook,
|
||||||
// the only handler that has to decide where a new incident's page goes; a zero
|
// the only handler that has to decide where a new incident's page goes; a zero
|
||||||
// notify disables notifications. Dead man's switches are per team and read from
|
// notify disables notifications. Dead man's switches are per team and read from
|
||||||
// the database, so nothing about them is wired in here.
|
// the database, so nothing about them is wired in here. version is reported
|
||||||
func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config) http.Handler {
|
// verbatim by GET /api/version, unauthenticated like /healthz: a client
|
||||||
|
// deciding whether it can talk to this server — terdut-tui, terdut-operator —
|
||||||
|
// needs to ask before it holds a credential for it, and the version is not a
|
||||||
|
// secret.
|
||||||
|
func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config, version string) http.Handler {
|
||||||
|
// One limiter each, both process-wide for the life of the router: login
|
||||||
|
// counts failed passwords, sign-up counts account creation, and mixing the
|
||||||
|
// two would let a burst of sign-ups lock somebody out of logging in.
|
||||||
|
trustedProxies.Store(int64(cfg.TrustedProxies))
|
||||||
|
loginLimit := newLoginLimiter(db)
|
||||||
|
signupLimiter := newLoginLimiter(db)
|
||||||
|
oidcLimit := newLoginLimiter(db)
|
||||||
|
|
||||||
r := chi.NewRouter()
|
r := chi.NewRouter()
|
||||||
r.Use(middleware.Logger)
|
r.Use(requestLogger)
|
||||||
r.Use(middleware.Recoverer)
|
r.Use(middleware.Recoverer)
|
||||||
|
r.Use(securityHeaders(notify.PublicURL))
|
||||||
|
|
||||||
r.Get("/healthz", func(w http.ResponseWriter, r *http.Request) {
|
r.Get("/healthz", func(w http.ResponseWriter, r *http.Request) {
|
||||||
respond(w, http.StatusOK, map[string]string{"status": "ok"})
|
respond(w, http.StatusOK, map[string]string{"status": "ok"})
|
||||||
})
|
})
|
||||||
|
r.Get("/api/version", func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
respond(w, http.StatusOK, map[string]string{"version": version})
|
||||||
|
})
|
||||||
|
|
||||||
// Unauthenticated: bootstrap, the Alertmanager webhook receiver, and the
|
// Unauthenticated: bootstrap, the Alertmanager webhook receiver, and the
|
||||||
// Acknowledge button in a push notification. The last one is authorised by
|
// Acknowledge button in a push notification. The last one is authorised by
|
||||||
@@ -32,24 +49,56 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config) http.Handler
|
|||||||
|
|
||||||
// Alert ingestion. The key in the path says both that the sender may post
|
// Alert ingestion. The key in the path says both that the sender may post
|
||||||
// and which team the alerts belong to, which is why it needs no session.
|
// and which team the alerts belong to, which is why it needs no session.
|
||||||
//
|
// This is the only way in.
|
||||||
// This is the only way in. The pre-teams /api/alertmanager/webhook, which
|
|
||||||
// took no credential at all, was removed in v0.13.0 once the cluster's
|
|
||||||
// Alertmanager had moved onto a key; a sender still posting there gets the
|
|
||||||
// JSON 404 every unknown /api path gets.
|
|
||||||
r.Post("/api/integrations/{key}/alertmanager", handleIntegrationWebhook(db, notify))
|
r.Post("/api/integrations/{key}/alertmanager", handleIntegrationWebhook(db, notify))
|
||||||
|
|
||||||
|
// Signing up. Both are unauthenticated by necessity: the caller has no
|
||||||
|
// account yet. The info endpoint says whether the door is open and whether
|
||||||
|
// an invite link is good, so the form can say so before somebody picks a
|
||||||
|
// password.
|
||||||
|
r.Get("/api/signup", handleSignupInfo(db))
|
||||||
|
r.With(passwordLoginOnly(!cfg.DisablePasswordLogin)).
|
||||||
|
Post("/api/signup", handleSignup(db, signupLimiter, notify.PublicURL))
|
||||||
|
|
||||||
|
// How to sign in: what the login form and the TUI offer before anybody types.
|
||||||
|
r.Get("/api/auth/config", handleAuthConfig(cfg))
|
||||||
|
|
||||||
// Signing in to the web UI. Login trades a password for a session cookie,
|
// Signing in to the web UI. Login trades a password for a session cookie,
|
||||||
// which AuthMiddleware accepts in place of an API key.
|
// which AuthMiddleware accepts in place of an API key.
|
||||||
r.Post("/api/login", handleLogin(db, newLoginLimiter(), notify.PublicURL))
|
r.With(passwordLoginOnly(!cfg.DisablePasswordLogin)).
|
||||||
|
Post("/api/login", handleLogin(db, loginLimit, notify.PublicURL))
|
||||||
r.Post("/api/logout", handleLogout(db, notify.PublicURL))
|
r.Post("/api/logout", handleLogout(db, notify.PublicURL))
|
||||||
|
|
||||||
|
// Single sign-on. Both routes are navigations the browser makes, to and from
|
||||||
|
// the provider, so they answer with redirects rather than JSON.
|
||||||
|
if cfg.OIDC.Enabled() {
|
||||||
|
prov := oidc.New(cfg.OIDC, notify.PublicURL)
|
||||||
|
r.Get("/api/oidc/login", handleOIDCLogin(db, prov, oidcLimit, notify.PublicURL))
|
||||||
|
r.Get("/api/oidc/callback", handleOIDCCallback(db, prov, notify.PublicURL))
|
||||||
|
|
||||||
|
// Device login, for a client with no browser of its own. Both are
|
||||||
|
// unauthenticated: the device code in the body is the credential.
|
||||||
|
r.Post("/api/oidc/device", handleDeviceStart(db, oidcLimit, notify.PublicURL))
|
||||||
|
r.Post("/api/oidc/device/token", handleDeviceToken(db, cfg.OIDC.SessionMaxAge, notify.PublicURL))
|
||||||
|
}
|
||||||
|
|
||||||
// All other /api routes require a valid API key.
|
// All other /api routes require a valid API key.
|
||||||
r.Group(func(r chi.Router) {
|
r.Group(func(r chi.Router) {
|
||||||
r.Use(AuthMiddleware(db))
|
r.Use(AuthMiddleware(db))
|
||||||
|
|
||||||
r.Get("/api/me", handleMe(db))
|
r.Get("/api/me", handleMe(db))
|
||||||
|
|
||||||
|
// Approving or refusing a device login is done by somebody signed in
|
||||||
|
// to a browser, and needs the same SSO configuration the flow does.
|
||||||
|
if cfg.OIDC.Enabled() {
|
||||||
|
r.Post("/api/oidc/device/approve", handleDeviceDecision(db, true))
|
||||||
|
r.Post("/api/oidc/device/deny", handleDeviceDecision(db, false))
|
||||||
|
}
|
||||||
|
r.Put("/api/me/onboarding", handleDismissOnboarding(db))
|
||||||
|
// Proves the topic works, which is the only part of "notifications are
|
||||||
|
// set up" that the person holding the phone can confirm.
|
||||||
|
r.Post("/api/me/notify/test", handleTestNotification(notify, db))
|
||||||
|
|
||||||
// Readable by anyone signed in: the queue's assignment control and the
|
// Readable by anyone signed in: the queue's assignment control and the
|
||||||
// on-call schedule both need to name people.
|
// on-call schedule both need to name people.
|
||||||
r.Get("/api/users", handleListUsers(db))
|
r.Get("/api/users", handleListUsers(db))
|
||||||
@@ -57,14 +106,14 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config) http.Handler
|
|||||||
// Your own account, or anybody's if you are an admin. The handlers call
|
// Your own account, or anybody's if you are an admin. The handlers call
|
||||||
// requireSelfOrAdmin rather than sitting behind AdminOnly, because
|
// requireSelfOrAdmin rather than sitting behind AdminOnly, because
|
||||||
// which rule applies depends on the {id} in the path.
|
// which rule applies depends on the {id} in the path.
|
||||||
|
r.Get("/api/users/{id}/teams", handleUserTeams(db))
|
||||||
r.Put("/api/users/{id}/notify", handleSetNotifyTarget(db))
|
r.Put("/api/users/{id}/notify", handleSetNotifyTarget(db))
|
||||||
r.Put("/api/users/{id}/password", handleSetPassword(db))
|
r.Put("/api/users/{id}/password", handleSetPassword(db))
|
||||||
|
r.Get("/api/users/{id}/api-keys", handleListAPIKeys(db))
|
||||||
r.Post("/api/users/{id}/api-keys", handleCreateAPIKey(db))
|
r.Post("/api/users/{id}/api-keys", handleCreateAPIKey(db))
|
||||||
r.Delete("/api/users/{id}/api-keys/{keyID}", handleDeleteAPIKey(db))
|
r.Delete("/api/users/{id}/api-keys/{keyID}", handleDeleteAPIKey(db))
|
||||||
|
|
||||||
// Administration: who exists, and who is an administrator. Until #3
|
// Administration: who exists, and who is an administrator.
|
||||||
// these were open to any authenticated caller, which meant every user
|
|
||||||
// could delete every other one.
|
|
||||||
r.Group(func(r chi.Router) {
|
r.Group(func(r chi.Router) {
|
||||||
r.Use(AdminOnly)
|
r.Use(AdminOnly)
|
||||||
|
|
||||||
@@ -76,6 +125,11 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config) http.Handler
|
|||||||
// What exists on this server, and how it behaves. /api/teams
|
// What exists on this server, and how it behaves. /api/teams
|
||||||
// answers "what am I in"; this one answers "what is there".
|
// answers "what am I in"; this one answers "what is there".
|
||||||
r.Get("/api/admin/teams", handleAdminListTeams(db))
|
r.Get("/api/admin/teams", handleAdminListTeams(db))
|
||||||
|
// One team and who is in it. The member list under
|
||||||
|
// /api/teams/{id}/members stays member-only and still 404s
|
||||||
|
// an administrator from outside; this is a different
|
||||||
|
// question, so it is a different endpoint.
|
||||||
|
r.Get("/api/admin/teams/{teamID}", handleAdminGetTeam(db))
|
||||||
r.Get("/api/admin/settings", handleGetSettings(db, cfg))
|
r.Get("/api/admin/settings", handleGetSettings(db, cfg))
|
||||||
r.Put("/api/admin/settings", handleSetSettings(db))
|
r.Put("/api/admin/settings", handleSetSettings(db))
|
||||||
})
|
})
|
||||||
@@ -86,9 +140,10 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config) http.Handler
|
|||||||
r.Get("/api/alerts/{id}", handleGetAlert(db))
|
r.Get("/api/alerts/{id}", handleGetAlert(db))
|
||||||
|
|
||||||
r.Get("/api/incidents", handleListIncidents(db))
|
r.Get("/api/incidents", handleListIncidents(db))
|
||||||
|
r.Get("/api/incidents/clusters", handleListClusters(db))
|
||||||
r.Get("/api/incidents/{id}", handleGetIncident(db))
|
r.Get("/api/incidents/{id}", handleGetIncident(db))
|
||||||
r.Get("/api/incidents/{id}/alerts", handleIncidentAlerts(db))
|
|
||||||
r.Get("/api/incidents/{id}/timeline", handleIncidentTimeline(db))
|
r.Get("/api/incidents/{id}/timeline", handleIncidentTimeline(db))
|
||||||
|
r.Get("/api/incidents/{id}/similar", handleIncidentSimilar(db))
|
||||||
r.Post("/api/incidents/{id}/acknowledge", handleIncidentAcknowledge(db))
|
r.Post("/api/incidents/{id}/acknowledge", handleIncidentAcknowledge(db))
|
||||||
r.Delete("/api/incidents/{id}/acknowledge", handleIncidentUnacknowledge(db))
|
r.Delete("/api/incidents/{id}/acknowledge", handleIncidentUnacknowledge(db))
|
||||||
r.Post("/api/incidents/{id}/resolve", handleIncidentResolve(db))
|
r.Post("/api/incidents/{id}/resolve", handleIncidentResolve(db))
|
||||||
@@ -100,28 +155,58 @@ func NewRouter(db *sql.DB, notify NotifyConfig, cfg config.Config) http.Handler
|
|||||||
r.Post("/api/incidents/{id}/notes", handleCreateNote(db))
|
r.Post("/api/incidents/{id}/notes", handleCreateNote(db))
|
||||||
r.Delete("/api/incidents/{id}/notes/{eventID}", handleDeleteNote(db))
|
r.Delete("/api/incidents/{id}/notes/{eventID}", handleDeleteNote(db))
|
||||||
|
|
||||||
|
// Service accounts: a scoped, non-human credential for automation
|
||||||
|
// (terdut-operator, most likely) that needs to manage the resources
|
||||||
|
// below without impersonating a human user. See SERVICE-ACCOUNTS.md.
|
||||||
|
r.Get("/api/service-accounts", handleListServiceAccounts(db))
|
||||||
|
r.Post("/api/service-accounts", handleCreateServiceAccount(db))
|
||||||
|
r.Post("/api/service-accounts/{id}/keys", handleCreateServiceAccountKey(db))
|
||||||
|
r.Delete("/api/service-accounts/{id}/keys/{keyID}", handleDeleteServiceAccountKey(db))
|
||||||
|
|
||||||
|
// Operator mode (TERDUT_OPERATOR_MODE) makes every write below refuse a
|
||||||
|
// human caller (a session or a user's own API key) while still letting
|
||||||
|
// a service account through — see OperatorModeBlock. opMode is a no-op
|
||||||
|
// wrapper when the flag is off, so this costs nothing on a server that
|
||||||
|
// never sets it.
|
||||||
|
opMode := OperatorModeBlock(cfg)
|
||||||
|
|
||||||
// Teams. A user sees the teams they belong to; an owner configures one.
|
// Teams. A user sees the teams they belong to; an owner configures one.
|
||||||
r.Get("/api/teams", handleListTeams(db))
|
r.Get("/api/teams", handleListTeams(db))
|
||||||
r.Post("/api/teams", handleCreateTeam(db))
|
r.With(opMode).Post("/api/teams", handleCreateTeam(db))
|
||||||
r.Put("/api/teams/{teamID}", handleRenameTeam(db))
|
r.With(opMode).Put("/api/teams/{teamID}", handleRenameTeam(db))
|
||||||
r.Delete("/api/teams/{teamID}", handleDeleteTeam(db))
|
r.With(opMode).Delete("/api/teams/{teamID}", handleDeleteTeam(db))
|
||||||
r.Get("/api/teams/{teamID}/members", handleListTeamMembers(db))
|
r.Get("/api/teams/{teamID}/members", handleListTeamMembers(db))
|
||||||
r.Post("/api/teams/{teamID}/members", handleAddTeamMember(db))
|
r.Post("/api/teams/{teamID}/members", handleAddTeamMember(db))
|
||||||
r.Delete("/api/teams/{teamID}/members/{userID}", handleRemoveTeamMember(db))
|
r.Delete("/api/teams/{teamID}/members/{userID}", handleRemoveTeamMember(db))
|
||||||
|
|
||||||
|
// A team's own OIDC group binding: which provider groups grant member
|
||||||
|
// and owner access to it.
|
||||||
|
r.Get("/api/teams/{teamID}/oidc-groups", handleGetTeamOIDCGroups(db))
|
||||||
|
r.With(opMode).Put("/api/teams/{teamID}/oidc-groups", handleSetTeamOIDCGroups(db))
|
||||||
|
|
||||||
|
// Invite links into this team. Not operator-mode-gated: membership is
|
||||||
|
// deliberately never gitops-managed (see terdut-operator's DESIGN.md
|
||||||
|
// §4.2), so it stays editable regardless of this flag.
|
||||||
|
r.Get("/api/teams/{teamID}/invites", handleListInvites(db))
|
||||||
|
r.Post("/api/teams/{teamID}/invites", handleCreateInvite(db, notify.PublicURL))
|
||||||
|
r.Delete("/api/teams/{teamID}/invites/{inviteID}", handleRevokeInvite(db))
|
||||||
|
|
||||||
// A team's escalation ladder: who is paged when nobody answers.
|
// A team's escalation ladder: who is paged when nobody answers.
|
||||||
r.Get("/api/teams/{teamID}/escalation", handleGetEscalation(db))
|
r.Get("/api/teams/{teamID}/escalation", handleGetEscalation(db))
|
||||||
r.Put("/api/teams/{teamID}/escalation", handleSetEscalation(db))
|
r.With(opMode).Put("/api/teams/{teamID}/escalation", handleSetEscalation(db))
|
||||||
|
|
||||||
// A team's own dead man's switches: which of its alerts are heartbeats,
|
// A team's own dead man's switches: which of its alerts are heartbeats,
|
||||||
// and how long a silence has to last before somebody is paged.
|
// and how long a silence has to last before somebody is paged.
|
||||||
r.Get("/api/teams/{teamID}/deadman", handleGetTeamDeadman(db))
|
r.Get("/api/teams/{teamID}/deadman/switches", handleListTeamDeadman(db))
|
||||||
r.Put("/api/teams/{teamID}/deadman", handleSetTeamDeadman(db))
|
r.With(opMode).Post("/api/teams/{teamID}/deadman/switches", handleCreateTeamDeadman(db))
|
||||||
|
r.With(opMode).Put("/api/teams/{teamID}/deadman/switches/{switchID}", handleUpdateTeamDeadman(db))
|
||||||
|
r.With(opMode).Delete("/api/teams/{teamID}/deadman/switches/{switchID}", handleDeleteTeamDeadman(db))
|
||||||
|
|
||||||
// Integrations: where a team's alerts come in, and the key that says so.
|
// Integrations: where a team's alerts come in, and the key that says so.
|
||||||
r.Get("/api/teams/{teamID}/integrations", handleListIntegrations(db))
|
r.Get("/api/teams/{teamID}/integrations", handleListIntegrations(db))
|
||||||
r.Post("/api/teams/{teamID}/integrations", handleCreateIntegration(db, notify.PublicURL))
|
r.With(opMode).Post("/api/teams/{teamID}/integrations", handleCreateIntegration(db, notify.PublicURL))
|
||||||
r.Delete("/api/teams/{teamID}/integrations/{integrationID}", handleDeleteIntegration(db))
|
r.With(opMode).Patch("/api/teams/{teamID}/integrations/{integrationID}", handleRenameIntegration(db))
|
||||||
|
r.With(opMode).Delete("/api/teams/{teamID}/integrations/{integrationID}", handleDeleteIntegration(db))
|
||||||
|
|
||||||
// The rota is per team. /api/schedule/current is the exception: it
|
// The rota is per team. /api/schedule/current is the exception: it
|
||||||
// answers across every team the caller is in, which is what somebody on
|
// answers across every team the caller is in, which is what somebody on
|
||||||
|
|||||||
@@ -69,7 +69,7 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
|
|||||||
// be, so the delete and the insert share one transaction.
|
// be, so the delete and the insert share one transaction.
|
||||||
tx, err := db.BeginTx(r.Context(), nil)
|
tx, err := db.BeginTx(r.Context(), nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer tx.Rollback()
|
defer tx.Rollback()
|
||||||
@@ -79,7 +79,7 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
|
|||||||
if _, err := tx.ExecContext(r.Context(),
|
if _, err := tx.ExecContext(r.Context(),
|
||||||
"DELETE FROM schedule_entries WHERE team_id = $1 AND date = $2",
|
"DELETE FROM schedule_entries WHERE team_id = $1 AND date = $2",
|
||||||
teamID, d); err != nil {
|
teamID, d); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -91,12 +91,12 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
|
|||||||
errResp("date already assigned: "+d+" (pass replace to take it)"))
|
errResp("date already assigned: "+d+" (pass replace to take it)"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if err := tx.Commit(); err != nil {
|
if err := tx.Commit(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -107,7 +107,7 @@ func handleCreateSchedule(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
all, err := scheduleRange(r.Context(), db, teamID, req.Dates[0], req.Dates[len(req.Dates)-1])
|
all, err := scheduleRange(r.Context(), db, teamID, req.Dates[0], req.Dates[len(req.Dates)-1])
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
created := []models.ScheduleEntry{}
|
created := []models.ScheduleEntry{}
|
||||||
@@ -147,7 +147,7 @@ func handleListSchedule(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
entries, err := scheduleRange(r.Context(), db, teamID, from, to)
|
entries, err := scheduleRange(r.Context(), db, teamID, from, to)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, entries)
|
respond(w, http.StatusOK, entries)
|
||||||
@@ -171,7 +171,7 @@ func handleDeleteSchedule(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"DELETE FROM schedule_entries WHERE id = $1 AND team_id = $2", id, teamID)
|
"DELETE FROM schedule_entries WHERE id = $1 AND team_id = $2", id, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -197,7 +197,7 @@ func handleCurrentSchedule(db *sql.DB) http.HandlerFunc {
|
|||||||
WHERE s.date = $1 AND s.team_id = ANY($2)
|
WHERE s.date = $1 AND s.team_id = ANY($2)
|
||||||
ORDER BY t.name`, today, callerTeamIDs(r.Context()))
|
ORDER BY t.name`, today, callerTeamIDs(r.Context()))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -207,14 +207,14 @@ func handleCurrentSchedule(db *sql.DB) http.HandlerFunc {
|
|||||||
var e models.ScheduleEntry
|
var e models.ScheduleEntry
|
||||||
var ts int64
|
var ts int64
|
||||||
if err := rows.Scan(&e.ID, &e.TeamID, &e.TeamName, &e.UserID, &e.Username, &e.Date, &ts); err != nil {
|
if err := rows.Scan(&e.ID, &e.TeamID, &e.TeamName, &e.UserID, &e.Username, &e.Date, &ts); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
e.CreatedAt = time.Unix(ts, 0).UTC()
|
e.CreatedAt = time.Unix(ts, 0).UTC()
|
||||||
entries = append(entries, e)
|
entries = append(entries, e)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, entries)
|
respond(w, http.StatusOK, entries)
|
||||||
@@ -235,6 +235,9 @@ func scheduleRange(ctx context.Context, db *sql.DB, teamID int64, from, to strin
|
|||||||
|
|
||||||
clause := strings.Join(where, " AND ")
|
clause := strings.Join(where, " AND ")
|
||||||
|
|
||||||
|
// #nosec G202 -- clause is built from sqlArgs.add's "$N" placeholders
|
||||||
|
// only, never a value; every value travels through args.all() as a
|
||||||
|
// bound parameter. See the sqlArgs doc comment in helpers.go.
|
||||||
rows, err := db.QueryContext(ctx, `
|
rows, err := db.QueryContext(ctx, `
|
||||||
SELECT s.id, s.team_id, t.name, s.user_id, u.username, s.date, s.created_at
|
SELECT s.id, s.team_id, t.name, s.user_id, u.username, s.date, s.created_at
|
||||||
FROM schedule_entries s
|
FROM schedule_entries s
|
||||||
|
|||||||
@@ -0,0 +1,115 @@
|
|||||||
|
package api_test
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"net/http"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||||
|
)
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Security headers
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
func TestSecurityHeaders_NosniffAlwaysSet(t *testing.T) {
|
||||||
|
s := newTS(t) // no PublicURL: the HTTPS signal is off
|
||||||
|
resp := s.req(t, http.MethodGet, "/api/me", nil)
|
||||||
|
defer resp.Body.Close()
|
||||||
|
|
||||||
|
if got := resp.Header.Get("X-Content-Type-Options"); got != "nosniff" {
|
||||||
|
t.Errorf("X-Content-Type-Options = %q, want nosniff", got)
|
||||||
|
}
|
||||||
|
if got := resp.Header.Get("Strict-Transport-Security"); got != "" {
|
||||||
|
t.Errorf("Strict-Transport-Security = %q, want unset without an https PublicURL", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestSecurityHeaders_HSTSWhenPublicURLIsHTTPS(t *testing.T) {
|
||||||
|
s := newTS(t, api.NotifyConfig{PublicURL: "https://terdut.example.com"})
|
||||||
|
resp := s.req(t, http.MethodGet, "/api/me", nil)
|
||||||
|
defer resp.Body.Close()
|
||||||
|
|
||||||
|
got := resp.Header.Get("Strict-Transport-Security")
|
||||||
|
if !strings.HasPrefix(got, "max-age=") || !strings.Contains(got, "includeSubDomains") {
|
||||||
|
t.Errorf("Strict-Transport-Security = %q, want a max-age with includeSubDomains", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Request body size limits
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
// TestBodySizeLimit_OrdinaryEndpointRejectsOversizedBody confirms an
|
||||||
|
// unauthenticated endpoint can't be made to buffer an arbitrarily large body:
|
||||||
|
// past maxBodyBytes, decodeJSON fails exactly as it would on any other
|
||||||
|
// malformed body, rather than the server reading the whole thing first.
|
||||||
|
func TestBodySizeLimit_OrdinaryEndpointRejectsOversizedBody(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
|
||||||
|
huge := bytes.Repeat([]byte("a"), 2<<20) // 2 MiB, past the 1 MiB default
|
||||||
|
body := []byte(`{"username":"` + string(huge) + `","password":"x"}`)
|
||||||
|
|
||||||
|
resp, err := http.Post(s.URL+"/api/login", "application/json", bytes.NewReader(body))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("POST /api/login: %v", err)
|
||||||
|
}
|
||||||
|
defer resp.Body.Close()
|
||||||
|
|
||||||
|
if resp.StatusCode != http.StatusBadRequest {
|
||||||
|
t.Errorf("status = %d, want %d (oversized body treated as invalid)", resp.StatusCode, http.StatusBadRequest)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestBodySizeLimit_WebhookAllowsLargerBodyThanDefault confirms the
|
||||||
|
// Alertmanager webhook's separate, larger cap actually takes effect: a body
|
||||||
|
// bigger than the ordinary default but within maxWebhookBodyBytes is still
|
||||||
|
// accepted, not rejected by the smaller limit every other endpoint gets.
|
||||||
|
func TestBodySizeLimit_WebhookAllowsLargerBodyThanDefault(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
|
||||||
|
// Padding kept inside one alert's annotation, comfortably past the 1 MiB
|
||||||
|
// default and still well under the webhook's 8 MiB cap.
|
||||||
|
padding := strings.Repeat("a", 3<<20) // 3 MiB
|
||||||
|
payload := `{"version":"4","status":"firing","groupKey":"big-group",` +
|
||||||
|
`"groupLabels":{"alertname":"BigAlert"},"alerts":[{"status":"firing",` +
|
||||||
|
`"labels":{"alertname":"BigAlert"},"annotations":{"note":"` + padding + `"},` +
|
||||||
|
`"startsAt":"2026-05-20T10:00:00Z","endsAt":"0001-01-01T00:00:00Z",` +
|
||||||
|
`"fingerprint":"fp-big"}]}`
|
||||||
|
|
||||||
|
resp, err := http.Post(s.URL+"/api/integrations/"+s.ingestKey+"/alertmanager",
|
||||||
|
"application/json", strings.NewReader(payload))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("POST webhook: %v", err)
|
||||||
|
}
|
||||||
|
defer resp.Body.Close()
|
||||||
|
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Errorf("status = %d, want %d (body under the webhook's own cap)", resp.StatusCode, http.StatusOK)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestBodySizeLimit_WebhookRejectsPastItsOwnCap confirms the webhook's larger
|
||||||
|
// cap is still a cap, not an exemption from one.
|
||||||
|
func TestBodySizeLimit_WebhookRejectsPastItsOwnCap(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
|
||||||
|
huge := strings.Repeat("a", 9<<20) // 9 MiB, past the 8 MiB webhook cap
|
||||||
|
payload := `{"version":"4","status":"firing","groupKey":"huge-group",` +
|
||||||
|
`"groupLabels":{"alertname":"HugeAlert"},"alerts":[{"status":"firing",` +
|
||||||
|
`"labels":{"alertname":"HugeAlert"},"annotations":{"note":"` + huge + `"},` +
|
||||||
|
`"startsAt":"2026-05-20T10:00:00Z","endsAt":"0001-01-01T00:00:00Z",` +
|
||||||
|
`"fingerprint":"fp-huge"}]}`
|
||||||
|
|
||||||
|
resp, err := http.Post(s.URL+"/api/integrations/"+s.ingestKey+"/alertmanager",
|
||||||
|
"application/json", strings.NewReader(payload))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("POST webhook: %v", err)
|
||||||
|
}
|
||||||
|
defer resp.Body.Close()
|
||||||
|
|
||||||
|
if resp.StatusCode != http.StatusBadRequest {
|
||||||
|
t.Errorf("status = %d, want %d (body past the webhook's own cap)", resp.StatusCode, http.StatusBadRequest)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,353 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"database/sql"
|
||||||
|
"errors"
|
||||||
|
"net/http"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||||
|
"github.com/go-chi/chi/v5"
|
||||||
|
)
|
||||||
|
|
||||||
|
// serviceAccountKeyPrefix marks a service-account key visibly, in logs and at
|
||||||
|
// a glance, distinct from a user's own personal API key. It carries no
|
||||||
|
// meaning to the server itself — the hash is looked up the same way either
|
||||||
|
// kind of key is — it exists entirely for whoever is reading a log line or an
|
||||||
|
// audit trail.
|
||||||
|
const serviceAccountKeyPrefix = "tdsa_"
|
||||||
|
|
||||||
|
// randomServiceAccountToken is randomToken with serviceAccountKeyPrefix on the
|
||||||
|
// raw value, hashed as a whole: the prefix is not a fixed header stripped
|
||||||
|
// before hashing, it is part of the secret, the same as if it had been
|
||||||
|
// generated that long to begin with.
|
||||||
|
func randomServiceAccountToken() (raw, hash string, err error) {
|
||||||
|
body, _, err := randomToken()
|
||||||
|
if err != nil {
|
||||||
|
return "", "", err
|
||||||
|
}
|
||||||
|
raw = serviceAccountKeyPrefix + body
|
||||||
|
return raw, hashToken(raw), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// callerIsAdmin reports whether the caller is a signed-in human system
|
||||||
|
// administrator. A service account never is, by design (SERVICE-ACCOUNTS.md):
|
||||||
|
// account and user management stays human-only, service accounts included.
|
||||||
|
func callerIsAdmin(ctx context.Context) bool {
|
||||||
|
u, ok := userFromContext(ctx)
|
||||||
|
return ok && u.IsAdmin
|
||||||
|
}
|
||||||
|
|
||||||
|
// callerOwnsTeam reports whether the caller is owner-equivalent for teamID:
|
||||||
|
// a human owner, or that team's own team-scoped service account (its single
|
||||||
|
// synthetic membership, serveAsServiceAccount — ratified in
|
||||||
|
// SERVICE-ACCOUNTS.md as intentional, not an accident: a team-scoped
|
||||||
|
// credential is that team's owner's reach, full stop, membership and
|
||||||
|
// invites included). Built on callerRole like requireTeamOwner, but without
|
||||||
|
// writing a response: callers here need to combine it with other ways of
|
||||||
|
// being allowed, not stop at the first no.
|
||||||
|
func callerOwnsTeam(ctx context.Context, teamID int64) bool {
|
||||||
|
if c, _ := callerFromContext(ctx); c.IsInstanceServiceAccount() {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
role, ok := callerRole(ctx, teamID)
|
||||||
|
return ok && role == models.RoleOwner
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleCreateServiceAccount creates a service account and mints its first
|
||||||
|
// key. Who may do this depends on scope: an instance-scoped account (which
|
||||||
|
// can in turn create a team and a team-scoped account for it) is system
|
||||||
|
// administration's own reach extended to automation, so only a human admin
|
||||||
|
// grants one. A team-scoped account is that team's owner's reach, so a human
|
||||||
|
// admin, the target team's own human owner, or an existing instance-scoped
|
||||||
|
// service account (minting itself a narrower credential for a team it just
|
||||||
|
// created) may create one.
|
||||||
|
func handleCreateServiceAccount(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
var req struct {
|
||||||
|
Name string `json:"name"`
|
||||||
|
Scope string `json:"scope"`
|
||||||
|
TeamID int64 `json:"team_id"`
|
||||||
|
}
|
||||||
|
if err := decodeJSON(r, &req); err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
req.Name = strings.TrimSpace(req.Name)
|
||||||
|
if req.Name == "" {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("name is required"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if req.Scope != models.ServiceAccountScopeInstance && req.Scope != models.ServiceAccountScopeTeam {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("scope must be instance or team"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if req.Scope == models.ServiceAccountScopeTeam && req.TeamID == 0 {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("team_id is required for a team-scoped account"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if req.Scope == models.ServiceAccountScopeInstance && req.TeamID != 0 {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("team_id must not be set for an instance-scoped account"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
allowed := callerIsAdmin(r.Context())
|
||||||
|
if !allowed && req.Scope == models.ServiceAccountScopeTeam {
|
||||||
|
allowed = callerOwnsTeam(r.Context(), req.TeamID) || isInstanceServiceAccount(r.Context())
|
||||||
|
}
|
||||||
|
if !allowed {
|
||||||
|
respond(w, http.StatusForbidden, errResp("team owner, system administrator, or instance-scoped service account access required"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
var callerUserID *int64
|
||||||
|
if u, ok := userFromContext(r.Context()); ok {
|
||||||
|
id := u.ID
|
||||||
|
callerUserID = &id
|
||||||
|
}
|
||||||
|
var teamID *int64
|
||||||
|
if req.Scope == models.ServiceAccountScopeTeam {
|
||||||
|
teamID = &req.TeamID
|
||||||
|
}
|
||||||
|
|
||||||
|
var sa models.ServiceAccount
|
||||||
|
var created int64
|
||||||
|
if err := db.QueryRowContext(r.Context(), `
|
||||||
|
INSERT INTO service_accounts (name, scope, team_id, created_by)
|
||||||
|
VALUES ($1, $2, $3, $4)
|
||||||
|
RETURNING id, name, scope, team_id, created_by, created_at`,
|
||||||
|
req.Name, req.Scope, teamID, callerUserID,
|
||||||
|
).Scan(&sa.ID, &sa.Name, &sa.Scope, &sa.TeamID, &sa.CreatedBy, &created); err != nil {
|
||||||
|
if isUniqueViolation(err) {
|
||||||
|
respond(w, http.StatusConflict, errResp("a service account with that name already exists"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
// The only foreign key that can fail here is team_id: an
|
||||||
|
// instance-scoped caller is not otherwise checked against it
|
||||||
|
// (callerOwnsTeam already proved it exists for a human owner).
|
||||||
|
respond(w, http.StatusBadRequest, errResp("unknown team_id"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
sa.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
|
|
||||||
|
key, err := mintServiceAccountKey(r.Context(), db, sa.ID, "initial")
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusCreated, map[string]any{"service_account": sa, "key": key})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// mintServiceAccountKey inserts one key for an existing account and returns
|
||||||
|
// it with its raw value populated — the one moment that value exists outside
|
||||||
|
// the request that generated it.
|
||||||
|
func mintServiceAccountKey(ctx context.Context, db *sql.DB, serviceAccountID int64, name string) (models.ServiceAccountKey, error) {
|
||||||
|
raw, hash, err := randomServiceAccountToken()
|
||||||
|
if err != nil {
|
||||||
|
return models.ServiceAccountKey{}, err
|
||||||
|
}
|
||||||
|
var key models.ServiceAccountKey
|
||||||
|
var created int64
|
||||||
|
if err := db.QueryRowContext(ctx, `
|
||||||
|
INSERT INTO service_account_keys (service_account_id, key_hash, name)
|
||||||
|
VALUES ($1, $2, $3)
|
||||||
|
RETURNING id, service_account_id, name, created_at`,
|
||||||
|
serviceAccountID, hash, name,
|
||||||
|
).Scan(&key.ID, &key.ServiceAccountID, &key.Name, &created); err != nil {
|
||||||
|
return models.ServiceAccountKey{}, err
|
||||||
|
}
|
||||||
|
key.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
|
key.Key = raw
|
||||||
|
return key, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func fetchServiceAccount(ctx context.Context, db *sql.DB, id int64) (models.ServiceAccount, error) {
|
||||||
|
var sa models.ServiceAccount
|
||||||
|
var created int64
|
||||||
|
err := db.QueryRowContext(ctx,
|
||||||
|
"SELECT id, name, scope, team_id, created_by, created_at FROM service_accounts WHERE id = $1", id,
|
||||||
|
).Scan(&sa.ID, &sa.Name, &sa.Scope, &sa.TeamID, &sa.CreatedBy, &created)
|
||||||
|
if err != nil {
|
||||||
|
return sa, err
|
||||||
|
}
|
||||||
|
sa.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
|
return sa, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// callerMayManageServiceAccount reports whether the caller may mint or revoke
|
||||||
|
// a key on sa: a system administrator, that team-scoped account's own human
|
||||||
|
// owner, the account rotating its own credential (not a privilege
|
||||||
|
// escalation, the same reasoning requireSelfOrAdmin already rests on for a
|
||||||
|
// user's own API keys) — or, new, an instance-scoped service account
|
||||||
|
// managing any team-scoped account.
|
||||||
|
//
|
||||||
|
// That last branch closes terdut-operator#3: handleCreateServiceAccount
|
||||||
|
// already lets an instance-scoped caller *create* a team-scoped account for
|
||||||
|
// any team (the branch below it, isInstanceServiceAccount(ctx)) — this
|
||||||
|
// account didn't have an equivalent reach to *adopt or rotate* one it
|
||||||
|
// didn't just create in the same call, which is exactly the recovery path
|
||||||
|
// terdut-operator's own documented crash-window handling depends on
|
||||||
|
// (DESIGN.md §5's general adopt-on-conflict rule): a reconcile that creates
|
||||||
|
// the account successfully but crashes before persisting its credential
|
||||||
|
// locally retries into a 409, and without this branch the only available
|
||||||
|
// recovery — minting a fresh key on the now-existing account — 403'd
|
||||||
|
// forever, with no way out. Granting it here is not a new power: it
|
||||||
|
// mirrors the create-time reach this scope already has, just extended to
|
||||||
|
// the retry path DESIGN.md's own crash-window reasoning requires.
|
||||||
|
func callerMayManageServiceAccount(ctx context.Context, sa models.ServiceAccount) bool {
|
||||||
|
if callerIsAdmin(ctx) {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if sa.TeamID != nil && callerOwnsTeam(ctx, *sa.TeamID) {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
caller, _ := callerFromContext(ctx)
|
||||||
|
if id, ok := caller.ServiceAccountID(); ok && id == sa.ID {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if sa.TeamID != nil && caller.IsInstanceServiceAccount() {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
func serviceAccountParam(w http.ResponseWriter, r *http.Request) (int64, bool) {
|
||||||
|
id, err := strconv.ParseInt(chi.URLParam(r, "id"), 10, 64)
|
||||||
|
if err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid service account id"))
|
||||||
|
return 0, false
|
||||||
|
}
|
||||||
|
return id, true
|
||||||
|
}
|
||||||
|
|
||||||
|
func handleCreateServiceAccountKey(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
id, ok := serviceAccountParam(w, r)
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
sa, err := fetchServiceAccount(r.Context(), db, id)
|
||||||
|
if errors.Is(err, sql.ErrNoRows) {
|
||||||
|
respond(w, http.StatusNotFound, errResp("service account not found"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !callerMayManageServiceAccount(r.Context(), sa) {
|
||||||
|
respond(w, http.StatusForbidden, errResp("team owner, system administrator, or the account itself may rotate its key"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
var req struct {
|
||||||
|
Name string `json:"name"`
|
||||||
|
}
|
||||||
|
if err := decodeJSON(r, &req); err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if req.Name == "" {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("name is required"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
key, err := mintServiceAccountKey(r.Context(), db, sa.ID, req.Name)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusCreated, key)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func handleDeleteServiceAccountKey(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
id, ok := serviceAccountParam(w, r)
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
sa, err := fetchServiceAccount(r.Context(), db, id)
|
||||||
|
if errors.Is(err, sql.ErrNoRows) {
|
||||||
|
respond(w, http.StatusNotFound, errResp("service account not found"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !callerMayManageServiceAccount(r.Context(), sa) {
|
||||||
|
respond(w, http.StatusForbidden, errResp("team owner, system administrator, or the account itself may revoke its key"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
keyID, err := strconv.ParseInt(chi.URLParam(r, "keyID"), 10, 64)
|
||||||
|
if err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid key id"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
res, err := db.ExecContext(r.Context(),
|
||||||
|
"DELETE FROM service_account_keys WHERE id = $1 AND service_account_id = $2", keyID, sa.ID)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
|
respond(w, http.StatusNotFound, errResp("key not found"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
w.WriteHeader(http.StatusNoContent)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleListServiceAccounts lists every service account, or looks one up by
|
||||||
|
// its exact name with ?name=. The name lookup is open to any authenticated
|
||||||
|
// caller, human or service account: it returns no key material, and it is
|
||||||
|
// what lets a service account find its own account on the 403 that follows a
|
||||||
|
// second POST — the self-registration pattern SERVICE-ACCOUNTS.md describes.
|
||||||
|
// Listing everything, with no filter, stays administrator-only.
|
||||||
|
func handleListServiceAccounts(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
name := strings.TrimSpace(r.URL.Query().Get("name"))
|
||||||
|
if name == "" && !callerIsAdmin(r.Context()) {
|
||||||
|
respond(w, http.StatusForbidden, errResp("administrator access required to list every service account; pass ?name= to look up one by name"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
query := "SELECT id, name, scope, team_id, created_by, created_at FROM service_accounts"
|
||||||
|
var args []any
|
||||||
|
if name != "" {
|
||||||
|
query += " WHERE name = $1"
|
||||||
|
args = append(args, name)
|
||||||
|
}
|
||||||
|
query += " ORDER BY id"
|
||||||
|
|
||||||
|
rows, err := db.QueryContext(r.Context(), query, args...)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer rows.Close()
|
||||||
|
|
||||||
|
accounts := []models.ServiceAccount{}
|
||||||
|
for rows.Next() {
|
||||||
|
var sa models.ServiceAccount
|
||||||
|
var created int64
|
||||||
|
if err := rows.Scan(&sa.ID, &sa.Name, &sa.Scope, &sa.TeamID, &sa.CreatedBy, &created); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
sa.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
|
accounts = append(accounts, sa)
|
||||||
|
}
|
||||||
|
if err := rows.Err(); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusOK, accounts)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,492 @@
|
|||||||
|
package api_test
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/json"
|
||||||
|
"io"
|
||||||
|
"net/http"
|
||||||
|
"net/http/httptest"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||||
|
)
|
||||||
|
|
||||||
|
// reqAs is s.req with an arbitrary bearer credential in place of the admin's
|
||||||
|
// own key, for exercising a service account's or another user's key.
|
||||||
|
func (s *ts) reqAs(t *testing.T, key, method, path string, body any) *http.Response {
|
||||||
|
t.Helper()
|
||||||
|
var r io.Reader
|
||||||
|
if body != nil {
|
||||||
|
data, _ := json.Marshal(body)
|
||||||
|
r = bytes.NewReader(data)
|
||||||
|
}
|
||||||
|
req, _ := http.NewRequest(method, s.URL+path, r)
|
||||||
|
req.Header.Set("Authorization", "Bearer "+key)
|
||||||
|
if body != nil {
|
||||||
|
req.Header.Set("Content-Type", "application/json")
|
||||||
|
}
|
||||||
|
resp, err := http.DefaultClient.Do(req)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("%s %s: %v", method, path, err)
|
||||||
|
}
|
||||||
|
return resp
|
||||||
|
}
|
||||||
|
|
||||||
|
// createServiceAccount creates a service account as callerKey and returns its
|
||||||
|
// freshly minted raw key.
|
||||||
|
func createServiceAccount(t *testing.T, s *ts, callerKey, name, scope string, teamID int64) string {
|
||||||
|
t.Helper()
|
||||||
|
body := map[string]any{"name": name, "scope": scope}
|
||||||
|
if teamID != 0 {
|
||||||
|
body["team_id"] = teamID
|
||||||
|
}
|
||||||
|
resp := s.reqAs(t, callerKey, http.MethodPost, "/api/service-accounts", body)
|
||||||
|
if resp.StatusCode != http.StatusCreated {
|
||||||
|
resp.Body.Close()
|
||||||
|
t.Fatalf("create service account %s: %d", name, resp.StatusCode)
|
||||||
|
}
|
||||||
|
var result struct {
|
||||||
|
Key struct {
|
||||||
|
Key string `json:"key"`
|
||||||
|
} `json:"key"`
|
||||||
|
}
|
||||||
|
decode(t, resp, &result)
|
||||||
|
if result.Key.Key == "" {
|
||||||
|
t.Fatalf("create service account %s: no key returned", name)
|
||||||
|
}
|
||||||
|
return result.Key.Key
|
||||||
|
}
|
||||||
|
|
||||||
|
// createTeamAs creates a team as callerKey and returns its id.
|
||||||
|
func createTeamAs(t *testing.T, s *ts, callerKey, name string) int64 {
|
||||||
|
t.Helper()
|
||||||
|
resp := s.reqAs(t, callerKey, http.MethodPost, "/api/teams", map[string]string{"name": name})
|
||||||
|
if resp.StatusCode != http.StatusCreated {
|
||||||
|
resp.Body.Close()
|
||||||
|
t.Fatalf("create team %s: %d", name, resp.StatusCode)
|
||||||
|
}
|
||||||
|
var team struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
}
|
||||||
|
decode(t, resp, &team)
|
||||||
|
return team.ID
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Instance scope
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
func TestServiceAccount_InstanceScopeCreatesTeamWithNoHumanOwner(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||||
|
|
||||||
|
if !strings.HasPrefix(instanceKey, "tdsa_") {
|
||||||
|
t.Errorf("expected a service-account key to carry the tdsa_ prefix, got %q", instanceKey)
|
||||||
|
}
|
||||||
|
|
||||||
|
resp := s.reqAs(t, instanceKey, http.MethodPost, "/api/teams", map[string]string{"name": "provisioned"})
|
||||||
|
if resp.StatusCode != http.StatusCreated {
|
||||||
|
t.Fatalf("instance-scoped account creating a team: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
var team struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
Role string `json:"role"`
|
||||||
|
}
|
||||||
|
decode(t, resp, &team)
|
||||||
|
if team.Role != "" {
|
||||||
|
t.Errorf("expected no role on a team a service account created (no human owner), got %q", team.Role)
|
||||||
|
}
|
||||||
|
|
||||||
|
// It still exists, visible to an administrator, even with no member.
|
||||||
|
var admin []map[string]any
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/admin/teams", nil), &admin)
|
||||||
|
found := false
|
||||||
|
for _, tm := range admin {
|
||||||
|
if int64(tm["id"].(float64)) == team.ID {
|
||||||
|
found = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !found {
|
||||||
|
t.Errorf("expected the service-account-created team to appear in /api/admin/teams")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestServiceAccount_TeamScopeCannotCreateTeam(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||||
|
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||||
|
keyA := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||||
|
|
||||||
|
resp := s.reqAs(t, keyA, http.MethodPost, "/api/teams", map[string]string{"name": "should-fail"})
|
||||||
|
if resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("expected 403, a team-scoped account creating a team, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
resp.Body.Close()
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Team scope
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
// The whole point of team scope: bound to its own team, refused everywhere
|
||||||
|
// else, the same as an instance-scoped account minting a key per TerdutTeam
|
||||||
|
// rather than sharing one server-admin-equivalent credential would need.
|
||||||
|
func TestServiceAccount_TeamScopeIsBoundToItsOwnTeam(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||||
|
|
||||||
|
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||||
|
teamB := createTeamAs(t, s, instanceKey, "team-b")
|
||||||
|
keyA := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||||
|
|
||||||
|
policy := map[string]any{"repeat_count": 0, "fallback_topic": "", "levels": []any{}}
|
||||||
|
|
||||||
|
resp := s.reqAs(t, keyA, http.MethodPut, "/api/teams/"+id64(teamA)+"/escalation", policy)
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Fatalf("team-a's own key setting its escalation: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
resp.Body.Close()
|
||||||
|
|
||||||
|
// 404, not 403: the same "does this exist" refusal a human non-member
|
||||||
|
// gets from requireTeamMember, not a distinguishable "you may not".
|
||||||
|
resp2 := s.reqAs(t, keyA, http.MethodPut, "/api/teams/"+id64(teamB)+"/escalation", policy)
|
||||||
|
if resp2.StatusCode != http.StatusNotFound {
|
||||||
|
t.Errorf("expected 404 reaching into another team, got %d", resp2.StatusCode)
|
||||||
|
}
|
||||||
|
resp2.Body.Close()
|
||||||
|
}
|
||||||
|
|
||||||
|
// Team scope is owner-equivalent broadly (SERVICE-ACCOUNTS.md), not limited to
|
||||||
|
// one endpoint: escalation, dead man's switches and integrations all work.
|
||||||
|
func TestServiceAccount_TeamScopeManagesItsResources(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||||
|
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||||
|
keyA := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||||
|
|
||||||
|
resp := s.reqAs(t, keyA, http.MethodPost, "/api/teams/"+id64(teamA)+"/deadman/switches",
|
||||||
|
map[string]any{"matcher": "alertname=Watchdog", "timeout_seconds": 900, "severity": "critical"})
|
||||||
|
if resp.StatusCode != http.StatusCreated {
|
||||||
|
t.Errorf("team-scoped account creating a dead man's switch: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
resp.Body.Close()
|
||||||
|
|
||||||
|
resp2 := s.reqAs(t, keyA, http.MethodPost, "/api/teams/"+id64(teamA)+"/integrations",
|
||||||
|
map[string]string{"name": "prod"})
|
||||||
|
if resp2.StatusCode != http.StatusCreated {
|
||||||
|
t.Errorf("team-scoped account creating an integration: %d", resp2.StatusCode)
|
||||||
|
}
|
||||||
|
resp2.Body.Close()
|
||||||
|
}
|
||||||
|
|
||||||
|
// Documents the capability already granted at create time (handleCreateServiceAccount's
|
||||||
|
// own callerOwnsTeam branch) also applies here: a team-scoped account is that
|
||||||
|
// team's owner's reach, membership and further accounts included, not just
|
||||||
|
// the handful of endpoints exercised above. Kept, not restricted, for
|
||||||
|
// symmetry with the now-ratified membership/invite capability below.
|
||||||
|
func TestServiceAccount_TeamScopeCanMintAnotherAccountForItsOwnTeam(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||||
|
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||||
|
keyA := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||||
|
|
||||||
|
resp := s.reqAs(t, keyA, http.MethodPost, "/api/service-accounts",
|
||||||
|
map[string]any{"name": "team-a-sa-2", "scope": models.ServiceAccountScopeTeam, "team_id": teamA})
|
||||||
|
if resp.StatusCode != http.StatusCreated {
|
||||||
|
t.Errorf("team-scoped account minting another account for its own team: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
resp.Body.Close()
|
||||||
|
}
|
||||||
|
|
||||||
|
// SERVICE-ACCOUNTS.md ratifies this explicitly: a team-scoped account is
|
||||||
|
// owner-equivalent for every requireTeamOwner endpoint, membership and
|
||||||
|
// invites included — terdut-operator's own invite-minting feature depends on
|
||||||
|
// exactly this. No test exercised handleCreateInvite from a service account
|
||||||
|
// before this change, and it would have 500'd (created_by written as a bare
|
||||||
|
// zero value against a NOT-validated-but-FK'd column) rather than succeeded;
|
||||||
|
// see the signup_test.go addition for that half.
|
||||||
|
func TestServiceAccount_TeamScopeManagesItsOwnInvites(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||||
|
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||||
|
keyA := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||||
|
|
||||||
|
var invite struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
}
|
||||||
|
resp := s.reqAs(t, keyA, http.MethodPost, "/api/teams/"+id64(teamA)+"/invites", map[string]any{})
|
||||||
|
if resp.StatusCode != http.StatusCreated {
|
||||||
|
t.Fatalf("team-scoped account creating an invite: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
decode(t, resp, &invite)
|
||||||
|
|
||||||
|
var list []map[string]any
|
||||||
|
decode(t, s.reqAs(t, keyA, http.MethodGet, "/api/teams/"+id64(teamA)+"/invites", nil), &list)
|
||||||
|
if len(list) != 1 {
|
||||||
|
t.Errorf("expected the invite to list back, got %d", len(list))
|
||||||
|
}
|
||||||
|
|
||||||
|
if resp := s.reqAs(t, keyA, http.MethodDelete,
|
||||||
|
"/api/teams/"+id64(teamA)+"/invites/"+id64(invite.ID), nil); resp.StatusCode != http.StatusNoContent {
|
||||||
|
t.Errorf("team-scoped account revoking its own invite: %d", resp.StatusCode)
|
||||||
|
} else {
|
||||||
|
resp.Body.Close()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Key rotation
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
// terdut-operator#3: an instance-scoped account is already trusted to CREATE
|
||||||
|
// a team-scoped account for any team (handleCreateServiceAccount's own
|
||||||
|
// isInstanceServiceAccount branch) — this pins that it is equally trusted to
|
||||||
|
// manage/rotate a key on one that already exists and that it did not just
|
||||||
|
// create in this call, which is the exact shape of terdut-operator's own
|
||||||
|
// crash-window recovery (mint succeeds, a later step is interrupted before
|
||||||
|
// persisting the credential locally, and the next reconcile retries into a
|
||||||
|
// 409 then needs to mint a fresh key on the now-existing account). Before
|
||||||
|
// this fix, the second POST .../keys below 403'd forever.
|
||||||
|
func TestServiceAccount_InstanceScopeAdoptsAnExistingTeamScopedAccountsKey(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||||
|
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||||
|
|
||||||
|
resp := s.reqAs(t, instanceKey, http.MethodPost, "/api/service-accounts",
|
||||||
|
map[string]any{"name": "team-a-sa", "scope": models.ServiceAccountScopeTeam, "team_id": teamA})
|
||||||
|
if resp.StatusCode != http.StatusCreated {
|
||||||
|
t.Fatalf("create team-scoped account: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
var created struct {
|
||||||
|
ServiceAccount struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
} `json:"service_account"`
|
||||||
|
}
|
||||||
|
decode(t, resp, &created)
|
||||||
|
|
||||||
|
// Simulates the adopt-on-409 recovery path: this instance-scoped caller
|
||||||
|
// did not just create this account in this call (a fresh *tdclient.Client
|
||||||
|
// request, same as a second, independent reconcile would issue), yet
|
||||||
|
// still needs to mint it a fresh key.
|
||||||
|
rotateResp := s.reqAs(t, instanceKey, http.MethodPost,
|
||||||
|
"/api/service-accounts/"+id64(created.ServiceAccount.ID)+"/keys", map[string]string{"name": "adopted"})
|
||||||
|
if rotateResp.StatusCode != http.StatusCreated {
|
||||||
|
t.Errorf("instance-scoped account adopting a team-scoped account's key: %d", rotateResp.StatusCode)
|
||||||
|
}
|
||||||
|
rotateResp.Body.Close()
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// AdminOnly / requireSelfOrAdmin — unchanged after the Caller refactor
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
// The Caller abstraction must not have widened AdminOnly/requireSelfOrAdmin:
|
||||||
|
// user management and /api/admin/settings stay human-only, for every scope
|
||||||
|
// of service account, exactly as before.
|
||||||
|
func TestAdminOnly_RefusesEveryServiceAccountScope(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||||
|
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||||
|
teamKey := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||||
|
|
||||||
|
for _, key := range []string{instanceKey, teamKey} {
|
||||||
|
if resp := s.reqAs(t, key, http.MethodGet, "/api/admin/settings", nil); resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("expected 403 for a service account reading /api/admin/settings, got %d", resp.StatusCode)
|
||||||
|
} else {
|
||||||
|
resp.Body.Close()
|
||||||
|
}
|
||||||
|
if resp := s.reqAs(t, key, http.MethodPost, "/api/users",
|
||||||
|
map[string]string{"username": "nope", "email": "nope@example.com"}); resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("expected 403 for a service account creating a user, got %d", resp.StatusCode)
|
||||||
|
} else {
|
||||||
|
resp.Body.Close()
|
||||||
|
}
|
||||||
|
if resp := s.reqAs(t, key, http.MethodGet, "/api/me", nil); resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("expected 403 for a service account calling /api/me, got %d", resp.StatusCode)
|
||||||
|
} else {
|
||||||
|
resp.Body.Close()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestServiceAccount_SelfRotatesItsOwnKey(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||||
|
|
||||||
|
// Self-lookup by name, the pattern that turns /api/bootstrap's 403 into a
|
||||||
|
// normal flow instead of an unhandled error.
|
||||||
|
var accounts []map[string]any
|
||||||
|
decode(t, s.reqAs(t, instanceKey, http.MethodGet, "/api/service-accounts?name=terdut-operator", nil), &accounts)
|
||||||
|
if len(accounts) != 1 {
|
||||||
|
t.Fatalf("expected exactly one match for ?name=terdut-operator, got %d", len(accounts))
|
||||||
|
}
|
||||||
|
id := int64(accounts[0]["id"].(float64))
|
||||||
|
|
||||||
|
resp := s.reqAs(t, instanceKey, http.MethodPost, "/api/service-accounts/"+id64(id)+"/keys",
|
||||||
|
map[string]string{"name": "rotated"})
|
||||||
|
if resp.StatusCode != http.StatusCreated {
|
||||||
|
t.Fatalf("self-rotation: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
var newKey struct {
|
||||||
|
Key string `json:"key"`
|
||||||
|
}
|
||||||
|
decode(t, resp, &newKey)
|
||||||
|
|
||||||
|
if resp := s.reqAs(t, newKey.Key, http.MethodPost, "/api/teams", map[string]string{"name": "after-rotation"}); resp.StatusCode != http.StatusCreated {
|
||||||
|
t.Errorf("expected the newly rotated key to work, got %d", resp.StatusCode)
|
||||||
|
} else {
|
||||||
|
resp.Body.Close()
|
||||||
|
}
|
||||||
|
|
||||||
|
// Rotation adds a key, it does not itself revoke the old one.
|
||||||
|
if resp := s.reqAs(t, instanceKey, http.MethodGet, "/api/service-accounts?name=terdut-operator", nil); resp.StatusCode != http.StatusOK {
|
||||||
|
t.Errorf("expected the original key to still work until explicitly revoked, got %d", resp.StatusCode)
|
||||||
|
} else {
|
||||||
|
resp.Body.Close()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Operator mode
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
// Operator mode is exercised against a second router over an
|
||||||
|
// already-configured database, rather than turning it on for newTSWith's own
|
||||||
|
// setup: that setup creates the default integration with the admin's (human)
|
||||||
|
// key, which is precisely the write operator mode exists to refuse, and in
|
||||||
|
// the real deployment this flag targets that setup was never done by a human
|
||||||
|
// to begin with — the operator itself would have provisioned it.
|
||||||
|
func TestOperatorMode_BlocksHumanWritesButAllowsServiceAccounts(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
|
||||||
|
conf := testConfig()
|
||||||
|
conf.OperatorMode = true
|
||||||
|
opSrv := httptest.NewServer(api.NewRouter(s.db, s.notify, conf, "test"))
|
||||||
|
t.Cleanup(opSrv.Close)
|
||||||
|
do := func(key, method, path string, body any) *http.Response {
|
||||||
|
t.Helper()
|
||||||
|
var r io.Reader
|
||||||
|
if body != nil {
|
||||||
|
data, _ := json.Marshal(body)
|
||||||
|
r = bytes.NewReader(data)
|
||||||
|
}
|
||||||
|
req, _ := http.NewRequest(method, opSrv.URL+path, r)
|
||||||
|
req.Header.Set("Authorization", "Bearer "+key)
|
||||||
|
if body != nil {
|
||||||
|
req.Header.Set("Content-Type", "application/json")
|
||||||
|
}
|
||||||
|
resp, err := http.DefaultClient.Do(req)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("%s %s: %v", method, path, err)
|
||||||
|
}
|
||||||
|
return resp
|
||||||
|
}
|
||||||
|
|
||||||
|
// The bootstrap admin's own key is a human credential: refused.
|
||||||
|
resp := do(s.key, http.MethodPost, "/api/teams", map[string]string{"name": "human-team"})
|
||||||
|
if resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Fatalf("expected 403 for a human write under operator mode, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
var refusal map[string]string
|
||||||
|
decode(t, resp, &refusal)
|
||||||
|
if refusal["reason"] != "operator_managed" {
|
||||||
|
t.Errorf("expected reason=operator_managed, got %q", refusal["reason"])
|
||||||
|
}
|
||||||
|
|
||||||
|
// Creating the service account itself is not gated by operator mode —
|
||||||
|
// it is how an operator identifies itself, not one of the resources it
|
||||||
|
// manages.
|
||||||
|
resp2 := do(s.key, http.MethodPost, "/api/service-accounts",
|
||||||
|
map[string]any{"name": "terdut-operator", "scope": models.ServiceAccountScopeInstance})
|
||||||
|
if resp2.StatusCode != http.StatusCreated {
|
||||||
|
t.Fatalf("create service account under operator mode: %d", resp2.StatusCode)
|
||||||
|
}
|
||||||
|
var result struct {
|
||||||
|
Key struct {
|
||||||
|
Key string `json:"key"`
|
||||||
|
} `json:"key"`
|
||||||
|
}
|
||||||
|
decode(t, resp2, &result)
|
||||||
|
|
||||||
|
resp3 := do(result.Key.Key, http.MethodPost, "/api/teams", map[string]string{"name": "operator-team"})
|
||||||
|
if resp3.StatusCode != http.StatusCreated {
|
||||||
|
t.Fatalf("expected 201 for a service-account write under operator mode, got %d", resp3.StatusCode)
|
||||||
|
}
|
||||||
|
resp3.Body.Close()
|
||||||
|
|
||||||
|
// Reads are unaffected regardless of caller.
|
||||||
|
if resp := do(s.key, http.MethodGet, "/api/teams", nil); resp.StatusCode != http.StatusOK {
|
||||||
|
t.Errorf("expected reads to stay open under operator mode, got %d", resp.StatusCode)
|
||||||
|
} else {
|
||||||
|
resp.Body.Close()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestOperatorMode_OffLeavesHumanWritesAlone(t *testing.T) {
|
||||||
|
s := newTS(t) // testConfig(): OperatorMode false
|
||||||
|
resp := s.req(t, http.MethodPost, "/api/teams", map[string]string{"name": "still-fine"})
|
||||||
|
if resp.StatusCode != http.StatusCreated {
|
||||||
|
t.Errorf("expected a human write to succeed with operator mode off, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
resp.Body.Close()
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Version
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
func TestVersion(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
resp, err := http.Get(s.URL + "/api/version")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("get version: %v", err)
|
||||||
|
}
|
||||||
|
var v struct {
|
||||||
|
Version string `json:"version"`
|
||||||
|
}
|
||||||
|
decode(t, resp, &v)
|
||||||
|
if v.Version != "test" {
|
||||||
|
t.Errorf("expected version %q, got %q", "test", v.Version)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Dead man's switch update-in-place
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
func TestDeadman_UpdateInPlacePreservesID(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
|
||||||
|
var created struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches",
|
||||||
|
map[string]any{"matcher": "alertname=Watchdog", "timeout_seconds": 900, "severity": "critical"}), &created)
|
||||||
|
|
||||||
|
resp := s.req(t, http.MethodPut, "/api/teams/"+defaultTeam+"/deadman/switches/"+id64(created.ID),
|
||||||
|
map[string]any{"name": "renamed", "matcher": "alertname=Watchdog", "timeout_seconds": 1200, "severity": "warning"})
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Fatalf("update switch: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
var updated struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
Name string `json:"name"`
|
||||||
|
TimeoutSeconds int64 `json:"timeout_seconds"`
|
||||||
|
Severity string `json:"severity"`
|
||||||
|
}
|
||||||
|
decode(t, resp, &updated)
|
||||||
|
if updated.ID != created.ID {
|
||||||
|
t.Errorf("expected id to stay %d, got %d", created.ID, updated.ID)
|
||||||
|
}
|
||||||
|
if updated.Name != "renamed" || updated.TimeoutSeconds != 1200 || updated.Severity != "warning" {
|
||||||
|
t.Errorf("expected the update to apply, got %+v", updated)
|
||||||
|
}
|
||||||
|
|
||||||
|
var list []map[string]any
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/deadman/switches", nil), &list)
|
||||||
|
if len(list) != 1 {
|
||||||
|
t.Errorf("expected the update to replace in place, not add a row, got %d switches", len(list))
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -6,9 +6,11 @@ import (
|
|||||||
"errors"
|
"errors"
|
||||||
"net/http"
|
"net/http"
|
||||||
"strconv"
|
"strconv"
|
||||||
|
"strings"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
"git.ryuvia.com/niklas/terdut-server/internal/config"
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||||
"github.com/go-chi/chi/v5"
|
"github.com/go-chi/chi/v5"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -90,6 +92,16 @@ func SeedSettings(ctx context.Context, db *sql.DB, cfg config.Config) error {
|
|||||||
type settingsResponse struct {
|
type settingsResponse struct {
|
||||||
Editable map[string]settingValue `json:"editable"`
|
Editable map[string]settingValue `json:"editable"`
|
||||||
FromEnv map[string]string `json:"from_env"`
|
FromEnv map[string]string `json:"from_env"`
|
||||||
|
|
||||||
|
// Choices are settings that are a word from a fixed list rather than a
|
||||||
|
// duration. One so far: who may create an account.
|
||||||
|
Choices map[string]choiceValue `json:"choices"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type choiceValue struct {
|
||||||
|
Value string `json:"value"`
|
||||||
|
Options []string `json:"options"`
|
||||||
|
Description string `json:"description"`
|
||||||
}
|
}
|
||||||
|
|
||||||
type settingValue struct {
|
type settingValue struct {
|
||||||
@@ -104,6 +116,14 @@ func handleGetSettings(db *sql.DB, cfg config.Config) http.HandlerFunc {
|
|||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
out := settingsResponse{
|
out := settingsResponse{
|
||||||
Editable: map[string]settingValue{},
|
Editable: map[string]settingValue{},
|
||||||
|
Choices: map[string]choiceValue{
|
||||||
|
SettingSignupMode: {
|
||||||
|
Value: signupMode(r.Context(), db),
|
||||||
|
Options: []string{SignupInviteOnly, SignupOpen},
|
||||||
|
Description: "who may create an account: invite_only means a link from a team owner, " +
|
||||||
|
"open means anybody who can reach this server",
|
||||||
|
},
|
||||||
|
},
|
||||||
FromEnv: map[string]string{
|
FromEnv: map[string]string{
|
||||||
// Never the ntfy token or the DSN: both are credentials, and an
|
// Never the ntfy token or the DSN: both are credentials, and an
|
||||||
// admin page that renders them turns a browser tab into a place
|
// admin page that renders them turns a browser tab into a place
|
||||||
@@ -137,7 +157,7 @@ func handleGetSettings(db *sql.DB, cfg config.Config) http.HandlerFunc {
|
|||||||
// sit in the table looking like configuration and doing nothing.
|
// sit in the table looking like configuration and doing nothing.
|
||||||
func handleSetSettings(db *sql.DB) http.HandlerFunc {
|
func handleSetSettings(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
var req map[string]int64
|
var req map[string]any
|
||||||
if err := decodeJSON(r, &req); err != nil {
|
if err := decodeJSON(r, &req); err != nil {
|
||||||
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||||
return
|
return
|
||||||
@@ -147,40 +167,60 @@ func handleSetSettings(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
for key, secs := range req {
|
// Validate everything before writing anything: a request that sets two
|
||||||
b, known := settingBounds[key]
|
// settings and gets one wrong should change neither.
|
||||||
if !known {
|
values := map[string]string{}
|
||||||
respond(w, http.StatusBadRequest, errResp("unknown setting: "+key))
|
for key, raw := range req {
|
||||||
return
|
switch key {
|
||||||
}
|
case SettingSignupMode:
|
||||||
d := time.Duration(secs) * time.Second
|
mode, _ := raw.(string)
|
||||||
if d < b.min || d > b.max {
|
if mode != SignupOpen && mode != SignupInviteOnly {
|
||||||
respond(w, http.StatusBadRequest, errResp(
|
respond(w, http.StatusBadRequest,
|
||||||
key+" must be between "+b.min.String()+" and "+b.max.String()))
|
errResp("signup_mode must be "+SignupInviteOnly+" or "+SignupOpen))
|
||||||
return
|
return
|
||||||
|
}
|
||||||
|
values[key] = mode
|
||||||
|
default:
|
||||||
|
b, known := settingBounds[key]
|
||||||
|
if !known {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("unknown setting: "+key))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
secs, ok := raw.(float64) // JSON numbers decode as float64
|
||||||
|
if !ok {
|
||||||
|
respond(w, http.StatusBadRequest, errResp(key+" must be a number of seconds"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
d := time.Duration(int64(secs)) * time.Second
|
||||||
|
if d < b.min || d > b.max {
|
||||||
|
respond(w, http.StatusBadRequest, errResp(
|
||||||
|
key+" must be between "+b.min.String()+" and "+b.max.String()))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
values[key] = strconv.FormatInt(int64(secs), 10)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
tx, err := db.BeginTx(r.Context(), nil)
|
tx, err := db.BeginTx(r.Context(), nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer tx.Rollback() //nolint:errcheck
|
defer tx.Rollback() //nolint:errcheck
|
||||||
|
|
||||||
for key, secs := range req {
|
for key, value := range values {
|
||||||
if _, err := tx.ExecContext(r.Context(), `
|
if _, err := tx.ExecContext(r.Context(), `
|
||||||
INSERT INTO settings (key, value, updated_at)
|
INSERT INTO settings (key, value, updated_at)
|
||||||
VALUES ($1, $2, `+nowEpoch+`)
|
VALUES ($1, $2, `+nowEpoch+`)
|
||||||
ON CONFLICT (key) DO UPDATE SET
|
ON CONFLICT (key) DO UPDATE SET
|
||||||
value = excluded.value, updated_at = excluded.updated_at`,
|
value = excluded.value, updated_at = excluded.updated_at`,
|
||||||
key, strconv.FormatInt(secs, 10)); err != nil {
|
key, value); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if err := tx.Commit(); err != nil {
|
if err := tx.Commit(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -188,6 +228,25 @@ func handleSetSettings(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// adminTeam is a team as an administrator sees it: what it is, plus how big it
|
||||||
|
// is and how much is on fire in it. One definition, so a team in the list and a
|
||||||
|
// team on its own page cannot describe themselves differently.
|
||||||
|
type adminTeam struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
Name string `json:"name"`
|
||||||
|
CreatedAt time.Time `json:"created_at"`
|
||||||
|
Members int64 `json:"members"`
|
||||||
|
OpenIncidents int64 `json:"open_incidents"`
|
||||||
|
|
||||||
|
// OIDCMemberGroup and OIDCOwnerGroup are the team's own group binding,
|
||||||
|
// read-only here: an administrator can see why a team's OIDC-sourced
|
||||||
|
// membership looks the way it does without being able to change it out
|
||||||
|
// from under the team's owner. Setting it is PUT
|
||||||
|
// /api/teams/{teamID}/oidc-groups, owner-only.
|
||||||
|
OIDCMemberGroup string `json:"oidc_member_group,omitempty"`
|
||||||
|
OIDCOwnerGroup string `json:"oidc_owner_group,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
// handleAdminListTeams lists every team on the server, with its size. The
|
// handleAdminListTeams lists every team on the server, with its size. The
|
||||||
// ordinary /api/teams answers "what am I in"; this one answers "what exists",
|
// ordinary /api/teams answers "what am I in"; this one answers "what exists",
|
||||||
// which only an administrator may ask.
|
// which only an administrator may ask.
|
||||||
@@ -197,41 +256,111 @@ func handleAdminListTeams(db *sql.DB) http.HandlerFunc {
|
|||||||
SELECT t.id, t.name, t.created_at,
|
SELECT t.id, t.name, t.created_at,
|
||||||
(SELECT COUNT(*) FROM team_members m WHERE m.team_id = t.id),
|
(SELECT COUNT(*) FROM team_members m WHERE m.team_id = t.id),
|
||||||
(SELECT COUNT(*) FROM incidents i
|
(SELECT COUNT(*) FROM incidents i
|
||||||
WHERE i.team_id = t.id AND i.resolved_at IS NULL)
|
WHERE i.team_id = t.id AND i.resolved_at IS NULL),
|
||||||
|
COALESCE(t.oidc_member_group, ''), COALESCE(t.oidc_owner_group, '')
|
||||||
FROM teams t
|
FROM teams t
|
||||||
ORDER BY t.name`)
|
ORDER BY t.name`)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
|
|
||||||
type adminTeam struct {
|
|
||||||
ID int64 `json:"id"`
|
|
||||||
Name string `json:"name"`
|
|
||||||
CreatedAt time.Time `json:"created_at"`
|
|
||||||
Members int64 `json:"members"`
|
|
||||||
OpenIncidents int64 `json:"open_incidents"`
|
|
||||||
}
|
|
||||||
teams := []adminTeam{}
|
teams := []adminTeam{}
|
||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
var t adminTeam
|
var t adminTeam
|
||||||
var created int64
|
var created int64
|
||||||
if err := rows.Scan(&t.ID, &t.Name, &created, &t.Members, &t.OpenIncidents); err != nil {
|
if err := rows.Scan(&t.ID, &t.Name, &created, &t.Members, &t.OpenIncidents,
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
&t.OIDCMemberGroup, &t.OIDCOwnerGroup); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
t.CreatedAt = time.Unix(created, 0).UTC()
|
t.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
teams = append(teams, t)
|
teams = append(teams, t)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, teams)
|
respond(w, http.StatusOK, teams)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// handleAdminGetTeam answers "what is this team, and who is in it" for any team
|
||||||
|
// on the server, which is the one question an administrator could not ask.
|
||||||
|
//
|
||||||
|
// GET /api/teams/{id}/members is requireTeamMember and answers 404 to somebody
|
||||||
|
// outside the team, administrator or not, and that stays exactly as it is:
|
||||||
|
// member means membership and nothing else. Reading a team's shape is a
|
||||||
|
// different thing from reading its work, so it gets an endpoint of its own
|
||||||
|
// under AdminOnly rather than an exception carved into that rule. An
|
||||||
|
// administrator still sees none of the team's incidents, alerts or rota.
|
||||||
|
func handleAdminGetTeam(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
teamID, ok := teamParam(w, r)
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
var t adminTeam
|
||||||
|
var created int64
|
||||||
|
err := db.QueryRowContext(r.Context(), `
|
||||||
|
SELECT t.id, t.name, t.created_at,
|
||||||
|
(SELECT COUNT(*) FROM team_members m WHERE m.team_id = t.id),
|
||||||
|
(SELECT COUNT(*) FROM incidents i
|
||||||
|
WHERE i.team_id = t.id AND i.resolved_at IS NULL),
|
||||||
|
COALESCE(t.oidc_member_group, ''), COALESCE(t.oidc_owner_group, '')
|
||||||
|
FROM teams t
|
||||||
|
WHERE t.id = $1`, teamID).
|
||||||
|
Scan(&t.ID, &t.Name, &created, &t.Members, &t.OpenIncidents,
|
||||||
|
&t.OIDCMemberGroup, &t.OIDCOwnerGroup)
|
||||||
|
if errors.Is(err, sql.ErrNoRows) {
|
||||||
|
respond(w, http.StatusNotFound, errResp("not found"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
t.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
|
|
||||||
|
// Same query and same ordering as handleListTeamMembers, so the two
|
||||||
|
// answers to "who is in this team" cannot disagree about the answer.
|
||||||
|
rows, err := db.QueryContext(r.Context(), `
|
||||||
|
SELECT m.team_id, m.user_id, u.username, m.role, m.joined_at, m.source
|
||||||
|
FROM team_members m
|
||||||
|
JOIN users u ON u.id = m.user_id
|
||||||
|
WHERE m.team_id = $1
|
||||||
|
ORDER BY u.username`, teamID)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer rows.Close()
|
||||||
|
|
||||||
|
members := []models.TeamMember{}
|
||||||
|
for rows.Next() {
|
||||||
|
var m models.TeamMember
|
||||||
|
var joined int64
|
||||||
|
if err := rows.Scan(&m.TeamID, &m.UserID, &m.Username, &m.Role, &joined, &m.Source); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
m.JoinedAt = time.Unix(joined, 0).UTC()
|
||||||
|
members = append(members, m)
|
||||||
|
}
|
||||||
|
if err := rows.Err(); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// A wrapper rather than a team with the members hung off it: "members"
|
||||||
|
// already means a count on the list endpoint, and one name must not be
|
||||||
|
// a number in one answer and an array in the next.
|
||||||
|
respond(w, http.StatusOK, map[string]any{"team": t, "members": members})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// handleRenameTeam renames a team. An owner's job, and an administrator's when
|
// handleRenameTeam renames a team. An owner's job, and an administrator's when
|
||||||
// a team has nobody left to do it.
|
// a team has nobody left to do it.
|
||||||
func handleRenameTeam(db *sql.DB) http.HandlerFunc {
|
func handleRenameTeam(db *sql.DB) http.HandlerFunc {
|
||||||
@@ -247,7 +376,15 @@ func handleRenameTeam(db *sql.DB) http.HandlerFunc {
|
|||||||
var req struct {
|
var req struct {
|
||||||
Name string `json:"name"`
|
Name string `json:"name"`
|
||||||
}
|
}
|
||||||
if err := decodeJSON(r, &req); err != nil || req.Name == "" {
|
// Trimmed, as handleCreateTeam trims: without it " " is a team name
|
||||||
|
// here but not at creation, which is one rule stated twice and only
|
||||||
|
// half applied.
|
||||||
|
if err := decodeJSON(r, &req); err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("name is required"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
req.Name = strings.TrimSpace(req.Name)
|
||||||
|
if req.Name == "" {
|
||||||
respond(w, http.StatusBadRequest, errResp("name is required"))
|
respond(w, http.StatusBadRequest, errResp("name is required"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
@@ -259,7 +396,7 @@ func handleRenameTeam(db *sql.DB) http.HandlerFunc {
|
|||||||
respond(w, http.StatusConflict, errResp("a team with that name already exists"))
|
respond(w, http.StatusConflict, errResp("a team with that name already exists"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -298,7 +435,7 @@ func handleSetUserDisabled(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
last, err := isLastAdmin(r.Context(), db, id)
|
last, err := isLastAdmin(r.Context(), db, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if last {
|
if last {
|
||||||
@@ -316,7 +453,7 @@ func handleSetUserDisabled(db *sql.DB) http.HandlerFunc {
|
|||||||
"UPDATE users SET disabled_at = NULL WHERE id = $1", id)
|
"UPDATE users SET disabled_at = NULL WHERE id = $1", id)
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -338,7 +475,7 @@ func handleSetUserDisabled(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
user, err := fetchUser(r.Context(), db, id)
|
user, err := fetchUser(r.Context(), db, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, user)
|
respond(w, http.StatusOK, user)
|
||||||
|
|||||||
@@ -0,0 +1,509 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"database/sql"
|
||||||
|
"errors"
|
||||||
|
"net/http"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||||
|
"github.com/go-chi/chi/v5"
|
||||||
|
)
|
||||||
|
|
||||||
|
// SettingSignupMode says who may create an account. It lives in the settings
|
||||||
|
// table with the other behaviour settings, so an administrator changes it in
|
||||||
|
// the admin page rather than in a chart.
|
||||||
|
//
|
||||||
|
// Two modes, not three. A domain-restricted mode was considered and dropped:
|
||||||
|
// with no email in this server there is nothing to verify an address against,
|
||||||
|
// so it would check the domain of a string somebody typed — a speed bump
|
||||||
|
// dressed as a control.
|
||||||
|
const (
|
||||||
|
SettingSignupMode = "signup_mode"
|
||||||
|
|
||||||
|
SignupInviteOnly = "invite_only"
|
||||||
|
SignupOpen = "open"
|
||||||
|
)
|
||||||
|
|
||||||
|
// defaultSignupMode is invite-only. An install that gets a public hostname
|
||||||
|
// before anybody has thought about sign-up should not be collecting accounts
|
||||||
|
// from the internet by default.
|
||||||
|
const defaultSignupMode = SignupInviteOnly
|
||||||
|
|
||||||
|
// inviteTTL is how long a new invite link lives. Long enough to send it and be
|
||||||
|
// read tomorrow, short enough that a link in an old chat log stops working.
|
||||||
|
const inviteTTL = 7 * 24 * time.Hour
|
||||||
|
|
||||||
|
// signupMode reads the current mode, falling back to invite-only for a missing
|
||||||
|
// or unrecognised value: the failure mode of a typo in this setting should be
|
||||||
|
// the closed door, not the open one.
|
||||||
|
func signupMode(ctx context.Context, db *sql.DB) string {
|
||||||
|
var raw string
|
||||||
|
if err := db.QueryRowContext(ctx,
|
||||||
|
"SELECT value FROM settings WHERE key = $1", SettingSignupMode).Scan(&raw); err != nil {
|
||||||
|
return defaultSignupMode
|
||||||
|
}
|
||||||
|
if raw != SignupOpen && raw != SignupInviteOnly {
|
||||||
|
return defaultSignupMode
|
||||||
|
}
|
||||||
|
return raw
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleSignupInfo tells the sign-up page what it may offer, without requiring
|
||||||
|
// a session: whether open sign-up is on, and whether the invite in the URL is
|
||||||
|
// any good. A bad invite is better reported before somebody picks a password.
|
||||||
|
func handleSignupInfo(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
out := map[string]any{"mode": signupMode(r.Context(), db)}
|
||||||
|
|
||||||
|
if token := r.URL.Query().Get("invite"); token != "" {
|
||||||
|
inv, err := loadInvite(r.Context(), db, token)
|
||||||
|
switch {
|
||||||
|
case err == nil:
|
||||||
|
out["invite_valid"] = true
|
||||||
|
out["invite_team"] = inv.teamName
|
||||||
|
default:
|
||||||
|
// Deliberately one answer for expired, revoked, used up and
|
||||||
|
// never existed. Telling a stranger which it was tells them
|
||||||
|
// something about links they do not hold.
|
||||||
|
out["invite_valid"] = false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
respond(w, http.StatusOK, out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
type invite struct {
|
||||||
|
id int64
|
||||||
|
teamID int64
|
||||||
|
teamName string
|
||||||
|
role string
|
||||||
|
}
|
||||||
|
|
||||||
|
// loadInvite resolves a raw token to a usable invite, or an error. Usable means
|
||||||
|
// it exists, has not been revoked, has not expired and has uses left.
|
||||||
|
func loadInvite(ctx context.Context, q querier, token string) (invite, error) {
|
||||||
|
var inv invite
|
||||||
|
err := q.QueryRowContext(ctx, `
|
||||||
|
SELECT i.id, i.team_id, t.name, i.role
|
||||||
|
FROM invites i
|
||||||
|
JOIN teams t ON t.id = i.team_id
|
||||||
|
WHERE i.token_hash = $1
|
||||||
|
AND i.revoked_at IS NULL
|
||||||
|
AND i.expires_at > `+nowEpoch+`
|
||||||
|
AND i.uses < i.max_uses`, hashToken(token)).
|
||||||
|
Scan(&inv.id, &inv.teamID, &inv.teamName, &inv.role)
|
||||||
|
if errors.Is(err, sql.ErrNoRows) {
|
||||||
|
return invite{}, errInviteUnusable
|
||||||
|
}
|
||||||
|
return inv, err
|
||||||
|
}
|
||||||
|
|
||||||
|
var errInviteUnusable = errors.New("invite is not usable")
|
||||||
|
|
||||||
|
// handleSignup creates an account, and puts it somewhere.
|
||||||
|
//
|
||||||
|
// Rate-limited on the same limiter as login, by address: sign-up is the other
|
||||||
|
// unauthenticated endpoint that writes, and an open install without this is a
|
||||||
|
// way to fill somebody's user table.
|
||||||
|
func handleSignup(db *sql.DB, limiter *loginLimiter, publicURL string) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
addr := clientAddr(r)
|
||||||
|
if limiter.blocked(r.Context(), "signup:"+addr, maxSignupsPerAddr) {
|
||||||
|
respond(w, http.StatusTooManyRequests, errResp("too many sign-ups from this address"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
var req struct {
|
||||||
|
Username string `json:"username"`
|
||||||
|
Email string `json:"email"`
|
||||||
|
Password string `json:"password"`
|
||||||
|
Invite string `json:"invite"`
|
||||||
|
TeamName string `json:"team_name"`
|
||||||
|
}
|
||||||
|
if err := decodeJSON(r, &req); err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
req.Username = strings.TrimSpace(req.Username)
|
||||||
|
req.Email = strings.TrimSpace(req.Email)
|
||||||
|
req.TeamName = strings.TrimSpace(req.TeamName)
|
||||||
|
|
||||||
|
if req.Username == "" || req.Email == "" {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("username and email are required"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if msg := validatePassword(req.Password); msg != "" {
|
||||||
|
respond(w, http.StatusBadRequest, errResp(msg))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
mode := signupMode(r.Context(), db)
|
||||||
|
var inv invite
|
||||||
|
hasInvite := false
|
||||||
|
if req.Invite != "" {
|
||||||
|
var err error
|
||||||
|
inv, err = loadInvite(r.Context(), db, req.Invite)
|
||||||
|
if err != nil {
|
||||||
|
limiter.fail(r.Context(), "signup:"+addr)
|
||||||
|
respond(w, http.StatusForbidden, errResp("this invite link is not usable"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
hasInvite = true
|
||||||
|
}
|
||||||
|
if !hasInvite && mode != SignupOpen {
|
||||||
|
// No invite and the door is shut. Not 404: the endpoint exists and
|
||||||
|
// saying so is how somebody knows to ask for a link.
|
||||||
|
respond(w, http.StatusForbidden,
|
||||||
|
errResp("sign-up is invite-only on this server"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !hasInvite && req.TeamName == "" {
|
||||||
|
// Open sign-up with no team would create an account that sees an
|
||||||
|
// empty queue and can be paged by nobody.
|
||||||
|
respond(w, http.StatusBadRequest, errResp("team_name is required"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
hash, err := hashPassword(req.Password)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
tx, err := db.BeginTx(r.Context(), nil)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer tx.Rollback() //nolint:errcheck
|
||||||
|
|
||||||
|
var userID int64
|
||||||
|
var invitedVia *int64
|
||||||
|
if hasInvite {
|
||||||
|
invitedVia = &inv.id
|
||||||
|
}
|
||||||
|
if err := tx.QueryRowContext(r.Context(), `
|
||||||
|
INSERT INTO users (username, email, password_hash, invited_via)
|
||||||
|
VALUES ($1, $2, $3, $4) RETURNING id`,
|
||||||
|
req.Username, req.Email, hash, invitedVia).Scan(&userID); err != nil {
|
||||||
|
if isUniqueViolation(err) {
|
||||||
|
respond(w, http.StatusConflict, errResp("username or email already exists"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
teamID, role := inv.teamID, inv.role
|
||||||
|
if !hasInvite {
|
||||||
|
// Open sign-up makes a team, and its creator owns it.
|
||||||
|
if err := tx.QueryRowContext(r.Context(),
|
||||||
|
"INSERT INTO teams (name) VALUES ($1) RETURNING id", req.TeamName).Scan(&teamID); err != nil {
|
||||||
|
if isUniqueViolation(err) {
|
||||||
|
respond(w, http.StatusConflict, errResp("a team with that name already exists"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
role = models.RoleOwner
|
||||||
|
}
|
||||||
|
|
||||||
|
if _, err := tx.ExecContext(r.Context(),
|
||||||
|
"INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3)",
|
||||||
|
teamID, userID, role); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
if hasInvite {
|
||||||
|
// Counted inside the transaction, so two people redeeming the last
|
||||||
|
// use of a link at once cannot both get in.
|
||||||
|
res, err := tx.ExecContext(r.Context(),
|
||||||
|
"UPDATE invites SET uses = uses + 1 WHERE id = $1 AND uses < max_uses", inv.id)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
|
respond(w, http.StatusForbidden, errResp("this invite link is not usable"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := tx.Commit(); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// Signed in immediately: the alternative is a form that says "now go
|
||||||
|
// and log in", which is the same credential typed twice.
|
||||||
|
if err := startSession(w, r, db, userID, publicURL); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
user, _ := fetchUser(r.Context(), db, userID)
|
||||||
|
respond(w, http.StatusCreated, meResponse{User: user, HasPassword: true})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// maxSignupsPerAddr is looser than the login limit: several people joining from
|
||||||
|
// one office share an address, and the thing being limited is account creation
|
||||||
|
// rather than password guessing.
|
||||||
|
const maxSignupsPerAddr = 10
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Invites
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
type inviteJSON struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
TeamID int64 `json:"team_id"`
|
||||||
|
Role string `json:"role"`
|
||||||
|
CreatedAt time.Time `json:"created_at"`
|
||||||
|
ExpiresAt time.Time `json:"expires_at"`
|
||||||
|
MaxUses int64 `json:"max_uses"`
|
||||||
|
Uses int64 `json:"uses"`
|
||||||
|
Revoked bool `json:"revoked"`
|
||||||
|
|
||||||
|
// URL is the whole link, returned once when the invite is created. Like an
|
||||||
|
// integration key, only its hash is stored.
|
||||||
|
URL string `json:"url,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
|
func handleListInvites(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
teamID, ok := teamParam(w, r)
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !requireTeamOwner(w, r, teamID) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
rows, err := db.QueryContext(r.Context(), `
|
||||||
|
SELECT id, team_id, role, created_at, expires_at, max_uses, uses, revoked_at
|
||||||
|
FROM invites
|
||||||
|
WHERE team_id = $1
|
||||||
|
ORDER BY id DESC`, teamID)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer rows.Close()
|
||||||
|
|
||||||
|
out := []inviteJSON{}
|
||||||
|
for rows.Next() {
|
||||||
|
var i inviteJSON
|
||||||
|
var created, expires int64
|
||||||
|
var revoked *int64
|
||||||
|
if err := rows.Scan(&i.ID, &i.TeamID, &i.Role, &created, &expires,
|
||||||
|
&i.MaxUses, &i.Uses, &revoked); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
i.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
|
i.ExpiresAt = time.Unix(expires, 0).UTC()
|
||||||
|
i.Revoked = revoked != nil
|
||||||
|
out = append(out, i)
|
||||||
|
}
|
||||||
|
if err := rows.Err(); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusOK, out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleCreateInvite mints a link into this team. Owner-only, like the rest of
|
||||||
|
// a team's configuration: deciding who joins is configuring the team.
|
||||||
|
func handleCreateInvite(db *sql.DB, publicURL string) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
teamID, ok := teamParam(w, r)
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !requireTeamOwner(w, r, teamID) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
var req struct {
|
||||||
|
Role string `json:"role"`
|
||||||
|
MaxUses int64 `json:"max_uses"`
|
||||||
|
}
|
||||||
|
if err := decodeJSON(r, &req); err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if req.Role == "" {
|
||||||
|
req.Role = models.RoleMember
|
||||||
|
}
|
||||||
|
if req.Role != models.RoleOwner && req.Role != models.RoleMember {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("role must be owner or member"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if req.MaxUses == 0 {
|
||||||
|
req.MaxUses = 1
|
||||||
|
}
|
||||||
|
if req.MaxUses < 1 || req.MaxUses > 100 {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("max_uses must be between 1 and 100"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
raw, hash, err := randomToken()
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
// created_by is nullable (ON DELETE SET NULL) for exactly this
|
||||||
|
// reason: the caller minting an invite is not always a human with a
|
||||||
|
// real users row. A team-scoped service account is owner-equivalent
|
||||||
|
// here (requireTeamOwner above already let it through), and this
|
||||||
|
// must leave created_by NULL for one the same way
|
||||||
|
// handleCreateServiceAccount already does for the analogous case —
|
||||||
|
// an unchecked zero value would violate the users(id) foreign key
|
||||||
|
// instead of recording "nobody" cleanly.
|
||||||
|
var createdBy *int64
|
||||||
|
if u, ok := userFromContext(r.Context()); ok {
|
||||||
|
id := u.ID
|
||||||
|
createdBy = &id
|
||||||
|
}
|
||||||
|
expires := time.Now().Add(inviteTTL)
|
||||||
|
|
||||||
|
var out inviteJSON
|
||||||
|
var created, expiresAt int64
|
||||||
|
if err := db.QueryRowContext(r.Context(), `
|
||||||
|
INSERT INTO invites (token_hash, team_id, role, created_by, expires_at, max_uses)
|
||||||
|
VALUES ($1, $2, $3, $4, $5, $6)
|
||||||
|
RETURNING id, team_id, role, created_at, expires_at, max_uses, uses`,
|
||||||
|
hash, teamID, req.Role, createdBy, expires.Unix(), req.MaxUses).
|
||||||
|
Scan(&out.ID, &out.TeamID, &out.Role, &created, &expiresAt, &out.MaxUses, &out.Uses); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
out.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
|
out.ExpiresAt = time.Unix(expiresAt, 0).UTC()
|
||||||
|
out.URL = strings.TrimSuffix(publicURL, "/") + "/signup?invite=" + raw
|
||||||
|
respond(w, http.StatusCreated, out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleRevokeInvite stops a link working without waiting for it to expire.
|
||||||
|
func handleRevokeInvite(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
teamID, ok := teamParam(w, r)
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !requireTeamOwner(w, r, teamID) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
id, err := strconv.ParseInt(chi.URLParam(r, "inviteID"), 10, 64)
|
||||||
|
if err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid invite id"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
res, err := db.ExecContext(r.Context(),
|
||||||
|
"UPDATE invites SET revoked_at = "+nowEpoch+
|
||||||
|
" WHERE id = $1 AND team_id = $2 AND revoked_at IS NULL", id, teamID)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
|
respond(w, http.StatusNotFound, errResp("not found"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
w.WriteHeader(http.StatusNoContent)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Onboarding
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
// handleTestNotification publishes one push to the caller's own topic.
|
||||||
|
//
|
||||||
|
// The point of the first-run checklist's notification step is not that a topic
|
||||||
|
// string has been typed but that a phone buzzes, and only the person holding it
|
||||||
|
// can tell whether it did. Published directly rather than through the outbox:
|
||||||
|
// the outbox row requires an incident, and this deliberately belongs to no
|
||||||
|
// incident.
|
||||||
|
func handleTestNotification(cfg NotifyConfig, db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
if cfg.BaseURL == "" {
|
||||||
|
respond(w, http.StatusServiceUnavailable,
|
||||||
|
errResp("this server has no ntfy configured, so it can send nothing"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
caller, ok := userFromContext(r.Context())
|
||||||
|
if !ok {
|
||||||
|
respond(w, http.StatusForbidden, errResp("this endpoint is for human accounts only"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
var topic *string
|
||||||
|
if err := db.QueryRowContext(r.Context(),
|
||||||
|
"SELECT ntfy_topic FROM users WHERE id = $1", caller.ID).Scan(&topic); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if topic == nil || *topic == "" {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("set a notification topic first"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := publish(r.Context(), cfg, ntfyMessage{
|
||||||
|
Topic: *topic,
|
||||||
|
Title: "terdut test",
|
||||||
|
Message: "If this arrived, your notifications work.",
|
||||||
|
Tags: []string{"white_check_mark"},
|
||||||
|
}); err != nil {
|
||||||
|
// The failure is the useful part here: a wrong topic, a token the
|
||||||
|
// ntfy server rejects, or an ntfy that is down all look the same
|
||||||
|
// from the phone, which is silence.
|
||||||
|
respond(w, http.StatusBadGateway, errResp("ntfy rejected the test: "+err.Error()))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
w.WriteHeader(http.StatusNoContent)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleDismissOnboarding hides the first-run checklist, or brings it back.
|
||||||
|
// Stored per user rather than in the browser: somebody who finishes setting up
|
||||||
|
// on a laptop should not be nagged again on their phone.
|
||||||
|
func handleDismissOnboarding(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
var req struct {
|
||||||
|
Dismissed *bool `json:"dismissed"`
|
||||||
|
}
|
||||||
|
if err := decodeJSON(r, &req); err != nil || req.Dismissed == nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("dismissed is required"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
caller, ok := userFromContext(r.Context())
|
||||||
|
if !ok {
|
||||||
|
respond(w, http.StatusForbidden, errResp("this endpoint is for human accounts only"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
var err error
|
||||||
|
if *req.Dismissed {
|
||||||
|
_, err = db.ExecContext(r.Context(),
|
||||||
|
"UPDATE users SET onboarding_dismissed_at = "+nowEpoch+" WHERE id = $1", caller.ID)
|
||||||
|
} else {
|
||||||
|
_, err = db.ExecContext(r.Context(),
|
||||||
|
"UPDATE users SET onboarding_dismissed_at = NULL WHERE id = $1", caller.ID)
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
w.WriteHeader(http.StatusNoContent)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,326 @@
|
|||||||
|
package api_test
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"database/sql"
|
||||||
|
"encoding/json"
|
||||||
|
"net/http"
|
||||||
|
"net/http/cookiejar"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/api"
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||||
|
)
|
||||||
|
|
||||||
|
// signup posts to the unauthenticated sign-up endpoint, the way the form does,
|
||||||
|
// and returns the response and a client holding whatever cookie came back.
|
||||||
|
func signup(t *testing.T, s *ts, body map[string]any) (*http.Response, *http.Client) {
|
||||||
|
t.Helper()
|
||||||
|
data, _ := json.Marshal(body)
|
||||||
|
jar, _ := cookiejar.New(nil)
|
||||||
|
client := &http.Client{Jar: jar}
|
||||||
|
req, _ := http.NewRequest(http.MethodPost, s.URL+"/api/signup", bytes.NewReader(data))
|
||||||
|
req.Header.Set("Content-Type", "application/json")
|
||||||
|
resp, err := client.Do(req)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("signup: %v", err)
|
||||||
|
}
|
||||||
|
return resp, client
|
||||||
|
}
|
||||||
|
|
||||||
|
// invite mints a link into the default team and returns its raw token.
|
||||||
|
func invite(t *testing.T, s *ts, role string, maxUses int64) string {
|
||||||
|
t.Helper()
|
||||||
|
var out struct {
|
||||||
|
URL string `json:"url"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/invites",
|
||||||
|
map[string]any{"role": role, "max_uses": maxUses}), &out)
|
||||||
|
if out.URL == "" {
|
||||||
|
t.Fatal("no invite URL returned")
|
||||||
|
}
|
||||||
|
// ...?invite=<token>
|
||||||
|
i := len(out.URL) - 1
|
||||||
|
for ; i >= 0 && out.URL[i] != '='; i-- {
|
||||||
|
}
|
||||||
|
return out.URL[i+1:]
|
||||||
|
}
|
||||||
|
|
||||||
|
func setSignupMode(t *testing.T, s *ts, mode string) {
|
||||||
|
t.Helper()
|
||||||
|
resp := s.req(t, http.MethodPut, "/api/admin/settings", map[string]any{"signup_mode": mode})
|
||||||
|
defer resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNoContent {
|
||||||
|
t.Fatalf("set signup mode: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The default is the closed door. An install that gets a public hostname before
|
||||||
|
// anybody has thought about sign-up should not be collecting accounts.
|
||||||
|
func TestSignup_InviteOnlyByDefault(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
|
||||||
|
var info map[string]any
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/signup", nil), &info)
|
||||||
|
if info["mode"] != "invite_only" {
|
||||||
|
t.Errorf("default sign-up mode is %v, want invite_only", info["mode"])
|
||||||
|
}
|
||||||
|
|
||||||
|
resp, _ := signup(t, s, map[string]any{
|
||||||
|
"username": "stranger", "email": "s@test.com", "password": "correct-horse-battery",
|
||||||
|
})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("sign-up without an invite: expected 403, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// An invite carries the team and the role, so redeeming one lands somewhere
|
||||||
|
// usable rather than in an account that sees an empty queue.
|
||||||
|
func TestSignup_InviteCreatesAMemberOfThatTeam(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
token := invite(t, s, "member", 1)
|
||||||
|
|
||||||
|
// The form checks the link before asking for a password.
|
||||||
|
var info map[string]any
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/signup?invite="+token, nil), &info)
|
||||||
|
if info["invite_valid"] != true {
|
||||||
|
t.Fatalf("a fresh invite should be valid: %v", info)
|
||||||
|
}
|
||||||
|
if info["invite_team"] != "Default" {
|
||||||
|
t.Errorf("the form should name the team: %v", info["invite_team"])
|
||||||
|
}
|
||||||
|
|
||||||
|
resp, client := signup(t, s, map[string]any{
|
||||||
|
"username": "newcomer", "email": "n@test.com",
|
||||||
|
"password": "correct-horse-battery", "invite": token,
|
||||||
|
})
|
||||||
|
if resp.StatusCode != http.StatusCreated {
|
||||||
|
t.Fatalf("redeeming an invite: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
var me struct {
|
||||||
|
User struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
IsAdmin bool `json:"is_admin"`
|
||||||
|
} `json:"user"`
|
||||||
|
}
|
||||||
|
decode(t, resp, &me)
|
||||||
|
if me.User.IsAdmin {
|
||||||
|
t.Error("somebody who signs up must not be an administrator")
|
||||||
|
}
|
||||||
|
|
||||||
|
// Signed in already: the cookie came back with the response.
|
||||||
|
got, err := client.Get(s.URL + "/api/teams")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
teams := list(t, got)
|
||||||
|
if len(teams) != 1 || teams[0]["name"] != "Default" || teams[0]["role"] != "member" {
|
||||||
|
t.Errorf("expected membership of Default as member, got %v", teams)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A single-use link is single-use, and the check is inside the transaction so
|
||||||
|
// two people redeeming the last use at once cannot both get in.
|
||||||
|
func TestSignup_InviteCannotBeUsedTwice(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
token := invite(t, s, "member", 1)
|
||||||
|
|
||||||
|
first, _ := signup(t, s, map[string]any{
|
||||||
|
"username": "first", "email": "f@test.com",
|
||||||
|
"password": "correct-horse-battery", "invite": token,
|
||||||
|
})
|
||||||
|
first.Body.Close()
|
||||||
|
if first.StatusCode != http.StatusCreated {
|
||||||
|
t.Fatalf("first redemption: %d", first.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
second, _ := signup(t, s, map[string]any{
|
||||||
|
"username": "second", "email": "s@test.com",
|
||||||
|
"password": "correct-horse-battery", "invite": token,
|
||||||
|
})
|
||||||
|
second.Body.Close()
|
||||||
|
if second.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("second redemption: expected 403, got %d", second.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// And the link reports itself unusable before anybody types a password.
|
||||||
|
var info map[string]any
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/signup?invite="+token, nil), &info)
|
||||||
|
if info["invite_valid"] != false {
|
||||||
|
t.Error("a used-up invite should report itself invalid")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Revoking stops a link without waiting for it to expire.
|
||||||
|
func TestSignup_RevokedInviteStopsWorking(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
token := invite(t, s, "member", 5)
|
||||||
|
|
||||||
|
invites := list(t, s.req(t, http.MethodGet, "/api/teams/"+defaultTeam+"/invites", nil))
|
||||||
|
if len(invites) != 1 {
|
||||||
|
t.Fatalf("expected one invite, got %d", len(invites))
|
||||||
|
}
|
||||||
|
id := int64(invites[0]["id"].(float64))
|
||||||
|
|
||||||
|
resp := s.req(t, http.MethodDelete, "/api/teams/"+defaultTeam+"/invites/"+id64(id), nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNoContent {
|
||||||
|
t.Fatalf("revoke: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
used, _ := signup(t, s, map[string]any{
|
||||||
|
"username": "late", "email": "l@test.com",
|
||||||
|
"password": "correct-horse-battery", "invite": token,
|
||||||
|
})
|
||||||
|
used.Body.Close()
|
||||||
|
if used.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("a revoked invite: expected 403, got %d", used.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Open sign-up makes a team, because an account in no team sees an empty queue
|
||||||
|
// and can be paged by nobody.
|
||||||
|
func TestSignup_OpenModeMakesATeam(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
setSignupMode(t, s, "open")
|
||||||
|
|
||||||
|
missing, _ := signup(t, s, map[string]any{
|
||||||
|
"username": "solo", "email": "s@test.com", "password": "correct-horse-battery",
|
||||||
|
})
|
||||||
|
missing.Body.Close()
|
||||||
|
if missing.StatusCode != http.StatusBadRequest {
|
||||||
|
t.Errorf("open sign-up with no team name: expected 400, got %d", missing.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
resp, client := signup(t, s, map[string]any{
|
||||||
|
"username": "solo", "email": "s@test.com",
|
||||||
|
"password": "correct-horse-battery", "team_name": "Solo",
|
||||||
|
})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusCreated {
|
||||||
|
t.Fatalf("open sign-up: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
got, err := client.Get(s.URL + "/api/teams")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
teams := list(t, got)
|
||||||
|
if len(teams) != 1 || teams[0]["name"] != "Solo" || teams[0]["role"] != "owner" {
|
||||||
|
t.Errorf("the creator should own their new team, got %v", teams)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Switching the mode is an administrator's decision, and it takes effect at
|
||||||
|
// once rather than at the next restart.
|
||||||
|
func TestSignup_ModeIsAnAdminSetting(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
_, call := member(t, s, "plain")
|
||||||
|
|
||||||
|
resp := call(http.MethodPut, "/api/admin/settings", map[string]any{"signup_mode": "open"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("a member changing the mode: expected 403, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
bad := s.req(t, http.MethodPut, "/api/admin/settings", map[string]any{"signup_mode": "everybody"})
|
||||||
|
bad.Body.Close()
|
||||||
|
if bad.StatusCode != http.StatusBadRequest {
|
||||||
|
t.Errorf("an unknown mode: expected 400, got %d", bad.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
setSignupMode(t, s, "open")
|
||||||
|
var info map[string]any
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/signup", nil), &info)
|
||||||
|
if info["mode"] != "open" {
|
||||||
|
t.Errorf("the change should be visible at once, got %v", info["mode"])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Minting a link is configuring the team, so it is an owner's job.
|
||||||
|
func TestSignup_InvitesAreOwnerOnly(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
_, call := member(t, s, "plain")
|
||||||
|
|
||||||
|
resp := call(http.MethodPost, "/api/teams/"+defaultTeam+"/invites", map[string]any{"role": "member"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("a member minting an invite: expected 403, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A password still has to be a password, and a taken username is still taken.
|
||||||
|
func TestSignup_ValidatesLikeTheRestOfTheServer(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
token := invite(t, s, "member", 5)
|
||||||
|
|
||||||
|
short, _ := signup(t, s, map[string]any{
|
||||||
|
"username": "shorty", "email": "sh@test.com", "password": "abc", "invite": token,
|
||||||
|
})
|
||||||
|
short.Body.Close()
|
||||||
|
if short.StatusCode != http.StatusBadRequest {
|
||||||
|
t.Errorf("a short password: expected 400, got %d", short.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
taken, _ := signup(t, s, map[string]any{
|
||||||
|
"username": "admin", "email": "other@test.com",
|
||||||
|
"password": "correct-horse-battery", "invite": token,
|
||||||
|
})
|
||||||
|
taken.Body.Close()
|
||||||
|
if taken.StatusCode != http.StatusConflict {
|
||||||
|
t.Errorf("an existing username: expected 409, got %d", taken.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A team-scoped service account has no users row to attribute created_by to.
|
||||||
|
// Before this fix, handleCreateInvite wrote its zero-value caller.ID straight
|
||||||
|
// into that (nullable, ON DELETE SET NULL) foreign key instead of leaving it
|
||||||
|
// NULL the way handleCreateServiceAccount already does for the same
|
||||||
|
// situation — a 500, not the 201 TestServiceAccount_TeamScopeManagesItsOwnInvites
|
||||||
|
// now confirms. This pins the column itself ends up NULL, not just "some
|
||||||
|
// response came back".
|
||||||
|
func TestSignup_InviteCreatedByAServiceAccountLeavesCreatedByNull(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||||
|
teamA := createTeamAs(t, s, instanceKey, "team-a")
|
||||||
|
keyA := createServiceAccount(t, s, instanceKey, "team-a-sa", models.ServiceAccountScopeTeam, teamA)
|
||||||
|
|
||||||
|
var created struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
}
|
||||||
|
decode(t, s.reqAs(t, keyA, http.MethodPost, "/api/teams/"+id64(teamA)+"/invites", map[string]any{}), &created)
|
||||||
|
|
||||||
|
var createdBy sql.NullInt64
|
||||||
|
if err := s.db.QueryRow("SELECT created_by FROM invites WHERE id = $1", created.ID).Scan(&createdBy); err != nil {
|
||||||
|
t.Fatalf("read back invites.created_by: %v", err)
|
||||||
|
}
|
||||||
|
if createdBy.Valid {
|
||||||
|
t.Errorf("expected created_by to be NULL for a service-account-minted invite, got %d", createdBy.Int64)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Neither of these has a real user_id to act on behalf of; both must 403 a
|
||||||
|
// service account explicitly rather than 500 (handleTestNotification, which
|
||||||
|
// used to query ntfy_topic for user id 0) or silently no-op (handleDismissOnboarding,
|
||||||
|
// which used to UPDATE ... WHERE id = 0, affecting nothing and still
|
||||||
|
// returning 204).
|
||||||
|
func TestServiceAccount_HumanOnlyEndpointsRefuseExplicitly(t *testing.T) {
|
||||||
|
// BaseURL set (even to a fake, unreachable address) so handleTestNotification
|
||||||
|
// reaches its AsHuman() check instead of short-circuiting on "ntfy not
|
||||||
|
// configured" first — this test is about the human-only check, not ntfy.
|
||||||
|
s := newTS(t, api.NotifyConfig{BaseURL: "http://ntfy.invalid"})
|
||||||
|
instanceKey := createServiceAccount(t, s, s.key, "terdut-operator", models.ServiceAccountScopeInstance, 0)
|
||||||
|
|
||||||
|
if resp := s.reqAs(t, instanceKey, http.MethodPost, "/api/me/notify/test", nil); resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("expected 403 for a service account testing notifications, got %d", resp.StatusCode)
|
||||||
|
} else {
|
||||||
|
resp.Body.Close()
|
||||||
|
}
|
||||||
|
if resp := s.reqAs(t, instanceKey, http.MethodPut, "/api/me/onboarding",
|
||||||
|
map[string]bool{"dismissed": true}); resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("expected 403 for a service account dismissing onboarding, got %d", resp.StatusCode)
|
||||||
|
} else {
|
||||||
|
resp.Body.Close()
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,113 @@
|
|||||||
|
package api
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"database/sql"
|
||||||
|
"net/http"
|
||||||
|
"strconv"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/models"
|
||||||
|
)
|
||||||
|
|
||||||
|
const (
|
||||||
|
similarDefaultLimit = 5
|
||||||
|
similarMaxLimit = 20
|
||||||
|
)
|
||||||
|
|
||||||
|
// handleIncidentSimilar lists earlier, resolved incidents in the same team with
|
||||||
|
// the same signature that someone left notes on, incidents with a resolution
|
||||||
|
// note first. This is the "have we seen this before" answer for a responder
|
||||||
|
// looking at a fresh incident; the plain notes are one timeline fetch away.
|
||||||
|
func handleIncidentSimilar(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
id, ok := incidentIDParam(w, r, db)
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
limit := similarDefaultLimit
|
||||||
|
if v := r.URL.Query().Get("limit"); v != "" {
|
||||||
|
n, err := strconv.Atoi(v)
|
||||||
|
if err != nil || n < 1 {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid limit"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
limit = min(n, similarMaxLimit)
|
||||||
|
}
|
||||||
|
|
||||||
|
out, err := similarIncidents(r.Context(), db, id, limit)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusOK, out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func similarIncidents(ctx context.Context, q querier, id int64, limit int) ([]models.SimilarIncident, error) {
|
||||||
|
rows, err := q.QueryContext(ctx, `
|
||||||
|
SELECT o.id, o.title, o.triggered_at, o.resolved_at,
|
||||||
|
(SELECT COUNT(*) FROM incident_events e
|
||||||
|
WHERE e.incident_id = o.id AND e.type = $3)
|
||||||
|
FROM incidents i
|
||||||
|
JOIN incidents o ON o.team_id = i.team_id AND o.signature = i.signature
|
||||||
|
WHERE i.id = $1 AND o.id <> i.id AND o.resolved_at IS NOT NULL
|
||||||
|
AND EXISTS (SELECT 1 FROM incident_events e
|
||||||
|
WHERE e.incident_id = o.id AND e.type IN ($3, $4))
|
||||||
|
ORDER BY EXISTS (SELECT 1 FROM incident_events e
|
||||||
|
WHERE e.incident_id = o.id AND e.type = $4) DESC,
|
||||||
|
o.triggered_at DESC
|
||||||
|
LIMIT $2`, id, limit, evNote, evResolutionNote)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
defer rows.Close()
|
||||||
|
|
||||||
|
out := []models.SimilarIncident{}
|
||||||
|
ids := []int64{}
|
||||||
|
for rows.Next() {
|
||||||
|
var s models.SimilarIncident
|
||||||
|
var triggered, resolved int64
|
||||||
|
if err := rows.Scan(&s.ID, &s.Title, &triggered, &resolved, &s.NoteCount); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
s.TriggeredAt = time.Unix(triggered, 0).UTC()
|
||||||
|
s.ResolvedAt = time.Unix(resolved, 0).UTC()
|
||||||
|
s.ResolutionNotes = []models.IncidentEvent{}
|
||||||
|
out = append(out, s)
|
||||||
|
ids = append(ids, s.ID)
|
||||||
|
}
|
||||||
|
if err := rows.Err(); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if len(out) == 0 {
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
nrows, err := q.QueryContext(ctx, `
|
||||||
|
SELECT e.id, e.incident_id, e.type, e.user_id, u.username, e.detail, e.created_at
|
||||||
|
FROM incident_events e
|
||||||
|
LEFT JOIN users u ON u.id = e.user_id
|
||||||
|
WHERE e.incident_id = ANY($1) AND e.type = $2
|
||||||
|
ORDER BY e.created_at, e.id`, ids, evResolutionNote)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
defer nrows.Close()
|
||||||
|
|
||||||
|
byID := make(map[int64]*models.SimilarIncident, len(out))
|
||||||
|
for i := range out {
|
||||||
|
byID[out[i].ID] = &out[i]
|
||||||
|
}
|
||||||
|
for nrows.Next() {
|
||||||
|
var e models.IncidentEvent
|
||||||
|
var ts int64
|
||||||
|
if err := nrows.Scan(&e.ID, &e.IncidentID, &e.Type, &e.UserID, &e.Username, &e.Detail, &ts); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
e.CreatedAt = time.Unix(ts, 0).UTC()
|
||||||
|
s := byID[e.IncidentID]
|
||||||
|
s.ResolutionNotes = append(s.ResolutionNotes, e)
|
||||||
|
}
|
||||||
|
return out, nrows.Err()
|
||||||
|
}
|
||||||
@@ -0,0 +1,99 @@
|
|||||||
|
package api_test
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"encoding/json"
|
||||||
|
"net/http"
|
||||||
|
"strconv"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// postGrouped posts a firing webhook whose group labels are exactly the given
|
||||||
|
// map, unlike postWebhook, which only ever groups by alertname.
|
||||||
|
func postGrouped(t *testing.T, s *ts, fingerprint, startsAt string, groupLabels map[string]string) {
|
||||||
|
t.Helper()
|
||||||
|
labels := map[string]string{}
|
||||||
|
for k, v := range groupLabels {
|
||||||
|
labels[k] = v
|
||||||
|
}
|
||||||
|
payload := map[string]any{
|
||||||
|
"version": "4", "status": "firing",
|
||||||
|
"groupKey": fingerprint,
|
||||||
|
"groupLabels": groupLabels,
|
||||||
|
"alerts": []map[string]any{
|
||||||
|
amAlert(fingerprint, groupLabels["alertname"], "firing", startsAt, zeroTime, labels),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
data, _ := json.Marshal(payload)
|
||||||
|
resp, err := http.Post(s.URL+"/api/integrations/"+s.ingestKey+"/alertmanager",
|
||||||
|
"application/json", bytes.NewReader(data))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("post webhook: %v", err)
|
||||||
|
}
|
||||||
|
resp.Body.Close()
|
||||||
|
}
|
||||||
|
|
||||||
|
func similar(t *testing.T, s *ts, id int) []map[string]any {
|
||||||
|
t.Helper()
|
||||||
|
var out []map[string]any
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/incidents/"+strconv.Itoa(id)+"/similar", nil), &out)
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// Same alert on another instance is the same problem; a resolution note left on
|
||||||
|
// the first one is what the second one should be shown.
|
||||||
|
func TestSimilar_IgnoresVolatileLabelsAndLeadsWithResolutionNote(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
postGrouped(t, s, "fp-a", "2026-05-20T10:00:00Z",
|
||||||
|
map[string]string{"alertname": "DiskFull", "instance": "web-1", "job": "node"})
|
||||||
|
s.req(t, http.MethodPost, "/api/incidents/1/resolve",
|
||||||
|
map[string]string{"resolution": "rotated the logs"}).Body.Close()
|
||||||
|
|
||||||
|
postGrouped(t, s, "fp-b", "2026-05-21T10:00:00Z",
|
||||||
|
map[string]string{"alertname": "DiskFull", "instance": "web-2", "job": "node"})
|
||||||
|
|
||||||
|
got := similar(t, s, 2)
|
||||||
|
if len(got) != 1 || int(got[0]["id"].(float64)) != 1 {
|
||||||
|
t.Fatalf("expected incident 1 as the only similar one, got %v", got)
|
||||||
|
}
|
||||||
|
notes := got[0]["resolution_notes"].([]any)
|
||||||
|
if len(notes) != 1 || notes[0].(map[string]any)["detail"] != "rotated the logs" {
|
||||||
|
t.Fatalf("expected the resolution note, got %v", notes)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A different stable label (job) is a different problem, and an incident nobody
|
||||||
|
// wrote a note on has nothing to show.
|
||||||
|
func TestSimilar_DifferentSignatureOrNoNotesIsExcluded(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
postGrouped(t, s, "fp-1", "2026-05-20T10:00:00Z",
|
||||||
|
map[string]string{"alertname": "DiskFull", "job": "node"})
|
||||||
|
s.req(t, http.MethodPost, "/api/incidents/1/notes", map[string]string{"content": "checked"}).Body.Close()
|
||||||
|
s.req(t, http.MethodPost, "/api/incidents/1/resolve", nil).Body.Close()
|
||||||
|
|
||||||
|
postGrouped(t, s, "fp-2", "2026-05-20T11:00:00Z",
|
||||||
|
map[string]string{"alertname": "DiskFull", "job": "db"})
|
||||||
|
s.req(t, http.MethodPost, "/api/incidents/2/resolve", nil).Body.Close()
|
||||||
|
|
||||||
|
postGrouped(t, s, "fp-3", "2026-05-21T10:00:00Z",
|
||||||
|
map[string]string{"alertname": "DiskFull", "job": "db"})
|
||||||
|
|
||||||
|
// Incident 3 matches 2 by signature, but 2 has no notes.
|
||||||
|
if got := similar(t, s, 3); len(got) != 0 {
|
||||||
|
t.Fatalf("expected nothing similar to incident 3, got %v", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// An open incident is not "earlier experience" yet, and the incident itself is
|
||||||
|
// never its own match.
|
||||||
|
func TestSimilar_OpenIncidentsAreNotListed(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
postGrouped(t, s, "fp-o1", "2026-05-20T10:00:00Z", map[string]string{"alertname": "Flap"})
|
||||||
|
s.req(t, http.MethodPost, "/api/incidents/1/notes",
|
||||||
|
map[string]any{"content": "still open", "pinned": true}).Body.Close()
|
||||||
|
postGrouped(t, s, "fp-o2", "2026-05-21T10:00:00Z", map[string]string{"alertname": "Flap"})
|
||||||
|
|
||||||
|
if got := similar(t, s, 2); len(got) != 0 {
|
||||||
|
t.Fatalf("expected an open incident not to be listed, got %v", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,171 @@
|
|||||||
|
package api_test
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"net/http"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// listSources reads a team's alert sources as the Sources page does.
|
||||||
|
func listSources(t *testing.T, tm teamFixture) []map[string]any {
|
||||||
|
t.Helper()
|
||||||
|
return list(t, tm.call(http.MethodGet, "/api/teams/"+id64(tm.id)+"/integrations", nil))
|
||||||
|
}
|
||||||
|
|
||||||
|
// addSource mints a second source in a team and returns its key.
|
||||||
|
func addSource(t *testing.T, tm teamFixture, name string) string {
|
||||||
|
t.Helper()
|
||||||
|
var out struct {
|
||||||
|
Key string `json:"key"`
|
||||||
|
}
|
||||||
|
decode(t, tm.call(http.MethodPost, "/api/teams/"+id64(tm.id)+"/integrations",
|
||||||
|
map[string]string{"name": name}), &out)
|
||||||
|
return out.Key
|
||||||
|
}
|
||||||
|
|
||||||
|
// A source that has never posted is "never", with nothing to say about alerts.
|
||||||
|
func TestSources_NeverUsedIsBlank(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
tm := newTeam(t, s, "red")
|
||||||
|
|
||||||
|
got := listSources(t, tm)
|
||||||
|
if len(got) != 1 {
|
||||||
|
t.Fatalf("expected 1 source, got %d", len(got))
|
||||||
|
}
|
||||||
|
src := got[0]
|
||||||
|
if src["status"] != "never" || src["last_used_at"] != nil || src["last_alert_at"] != nil {
|
||||||
|
t.Errorf("a source nobody has posted on should be blank, got %v", src)
|
||||||
|
}
|
||||||
|
if src["alerts_24h"].(float64) != 0 {
|
||||||
|
t.Errorf("alerts_24h = %v, want 0", src["alerts_24h"])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Each source is credited with what arrived on its own key, and only that.
|
||||||
|
func TestSources_AlertsAreAttributedToTheirSource(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
tm := newTeam(t, s, "red")
|
||||||
|
second := addSource(t, tm, "staging")
|
||||||
|
|
||||||
|
postToIntegration(t, s, tm.key, "fp-1", "DiskFull")
|
||||||
|
postToIntegration(t, s, tm.key, "fp-2", "CPUHot")
|
||||||
|
|
||||||
|
got := listSources(t, tm)
|
||||||
|
first, other := got[0], got[1]
|
||||||
|
if first["status"] != "active" || first["last_used_at"] == nil || first["last_alert_at"] == nil {
|
||||||
|
t.Errorf("the source that posted should be active with timestamps, got %v", first)
|
||||||
|
}
|
||||||
|
if first["alerts_24h"].(float64) != 2 {
|
||||||
|
t.Errorf("alerts_24h = %v, want 2", first["alerts_24h"])
|
||||||
|
}
|
||||||
|
if other["status"] != "never" || other["alerts_24h"].(float64) != 0 {
|
||||||
|
t.Errorf("the other source should be untouched, got %v", other)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Re-sending the same alert on the other key moves it: last sender wins.
|
||||||
|
postToIntegration(t, s, second, "fp-1", "DiskFull")
|
||||||
|
got = listSources(t, tm)
|
||||||
|
if got[0]["alerts_24h"].(float64) != 1 || got[1]["alerts_24h"].(float64) != 1 {
|
||||||
|
t.Errorf("fp-1 should have moved to the second source, got %v and %v",
|
||||||
|
got[0]["alerts_24h"], got[1]["alerts_24h"])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A payload with no alerts in it is a webhook, not an alert: the source was
|
||||||
|
// heard from, and nothing arrived.
|
||||||
|
func TestSources_EmptyPayloadStampsUseButNotAlert(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
tm := newTeam(t, s, "red")
|
||||||
|
|
||||||
|
resp, err := http.Post(s.URL+"/api/integrations/"+tm.key+"/alertmanager",
|
||||||
|
"application/json", bytes.NewReader([]byte(`{"version":"4","status":"firing","alerts":[]}`)))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("post: %v", err)
|
||||||
|
}
|
||||||
|
resp.Body.Close()
|
||||||
|
|
||||||
|
src := listSources(t, tm)[0]
|
||||||
|
if src["status"] != "active" || src["last_alert_at"] != nil {
|
||||||
|
t.Errorf("want active with no alert yet, got %v", src)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Quiet is "has posted, not lately"; the alert counter forgets after a day but
|
||||||
|
// the last alert's timestamp is kept.
|
||||||
|
func TestSources_QuietAfterADay(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
tm := newTeam(t, s, "red")
|
||||||
|
postToIntegration(t, s, tm.key, "fp-1", "DiskFull")
|
||||||
|
|
||||||
|
old := time.Now().Add(-48 * time.Hour).Unix()
|
||||||
|
s.exec(t, "UPDATE integrations SET last_used_at = $1", old)
|
||||||
|
s.exec(t, "UPDATE alerts SET received_at = $1 WHERE fingerprint = 'fp-1'", old)
|
||||||
|
|
||||||
|
src := listSources(t, tm)[0]
|
||||||
|
if src["status"] != "quiet" {
|
||||||
|
t.Errorf("status = %v, want quiet", src["status"])
|
||||||
|
}
|
||||||
|
if src["alerts_24h"].(float64) != 0 {
|
||||||
|
t.Errorf("alerts_24h = %v, want 0", src["alerts_24h"])
|
||||||
|
}
|
||||||
|
if src["last_alert_at"] == nil {
|
||||||
|
t.Error("last_alert_at should survive the day")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Revoking a source does not take its alerts with it.
|
||||||
|
func TestSources_RevokeKeepsTheAlerts(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
tm := newTeam(t, s, "red")
|
||||||
|
postToIntegration(t, s, tm.key, "fp-1", "DiskFull")
|
||||||
|
|
||||||
|
id := int64(listSources(t, tm)[0]["id"].(float64))
|
||||||
|
resp := tm.call(http.MethodDelete, "/api/teams/"+id64(tm.id)+"/integrations/"+id64(id), nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNoContent {
|
||||||
|
t.Fatalf("revoke: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
if got := len(list(t, tm.call(http.MethodGet, "/api/alerts", nil))); got != 1 {
|
||||||
|
t.Errorf("the alert should outlive its source, got %d alerts", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Renaming is an owner's, scoped to the team, and does not touch the key.
|
||||||
|
func TestSources_Rename(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
tm := newTeam(t, s, "red")
|
||||||
|
other := newTeam(t, s, "blue")
|
||||||
|
id := int64(listSources(t, tm)[0]["id"].(float64))
|
||||||
|
path := "/api/teams/" + id64(tm.id) + "/integrations/" + id64(id)
|
||||||
|
|
||||||
|
resp := tm.call(http.MethodPatch, path, map[string]string{"name": " prod "})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNoContent {
|
||||||
|
t.Fatalf("rename: %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
if name := listSources(t, tm)[0]["name"]; name != "prod" {
|
||||||
|
t.Errorf("name = %q, want it trimmed to prod", name)
|
||||||
|
}
|
||||||
|
postToIntegration(t, s, tm.key, "fp-1", "DiskFull") // the old key still works
|
||||||
|
|
||||||
|
for name, body := range map[string]map[string]string{
|
||||||
|
"empty": {"name": " "},
|
||||||
|
"too long": {"name": strings.Repeat("x", 101)},
|
||||||
|
} {
|
||||||
|
resp := tm.call(http.MethodPatch, path, body)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusBadRequest {
|
||||||
|
t.Errorf("%s name: expected 400, got %d", name, resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Another team's owner cannot reach it.
|
||||||
|
resp = other.call(http.MethodPatch, "/api/teams/"+id64(other.id)+"/integrations/"+id64(id),
|
||||||
|
map[string]string{"name": "mine now"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNotFound {
|
||||||
|
t.Errorf("renaming another team's source: expected 404, got %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -24,7 +24,7 @@ func handleStatsAlerts(db *sql.DB) http.HandlerFunc {
|
|||||||
FROM alerts WHERE %s`, where), args.all()...,
|
FROM alerts WHERE %s`, where), args.all()...,
|
||||||
).Scan(&total, &firing, &resolved)
|
).Scan(&total, &firing, &resolved)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, map[string]int64{
|
respond(w, http.StatusOK, map[string]int64{
|
||||||
@@ -55,7 +55,7 @@ func handleStatsTop(db *sql.DB) http.HandlerFunc {
|
|||||||
ORDER BY cnt DESC
|
ORDER BY cnt DESC
|
||||||
LIMIT %s`, where, args.add(limit)), args.all()...)
|
LIMIT %s`, where, args.add(limit)), args.all()...)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -68,11 +68,15 @@ func handleStatsTop(db *sql.DB) http.HandlerFunc {
|
|||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
var e entry
|
var e entry
|
||||||
if err := rows.Scan(&e.Name, &e.Count); err != nil {
|
if err := rows.Scan(&e.Name, &e.Count); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
result = append(result, e)
|
result = append(result, e)
|
||||||
}
|
}
|
||||||
|
if err := rows.Err(); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
respond(w, http.StatusOK, result)
|
respond(w, http.StatusOK, result)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -89,7 +93,7 @@ func handleStatsByHour(db *sql.DB) http.HandlerFunc {
|
|||||||
GROUP BY hr
|
GROUP BY hr
|
||||||
ORDER BY hr ASC`, where), args.all()...)
|
ORDER BY hr ASC`, where), args.all()...)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -99,11 +103,15 @@ func handleStatsByHour(db *sql.DB) http.HandlerFunc {
|
|||||||
var hr int
|
var hr int
|
||||||
var cnt int64
|
var cnt int64
|
||||||
if err := rows.Scan(&hr, &cnt); err != nil {
|
if err := rows.Scan(&hr, &cnt); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
counts[hr] = cnt
|
counts[hr] = cnt
|
||||||
}
|
}
|
||||||
|
if err := rows.Err(); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
type entry struct {
|
type entry struct {
|
||||||
Hour int `json:"hour"`
|
Hour int `json:"hour"`
|
||||||
@@ -121,8 +129,7 @@ func handleStatsByDay(db *sql.DB) http.HandlerFunc {
|
|||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
where, args := statsFilter(r.URL.Query(), "received_at", callerTeamIDs(r.Context()))
|
where, args := statsFilter(r.URL.Query(), "received_at", callerTeamIDs(r.Context()))
|
||||||
|
|
||||||
// Postgres EXTRACT(DOW …) → 0=Sunday … 6=Saturday, the same numbering
|
// Postgres EXTRACT(DOW …) → 0=Sunday … 6=Saturday.
|
||||||
// SQLite's strftime('%w') returned, so the frontend needs no change.
|
|
||||||
rows, err := db.QueryContext(r.Context(), fmt.Sprintf(`
|
rows, err := db.QueryContext(r.Context(), fmt.Sprintf(`
|
||||||
SELECT EXTRACT(DOW FROM to_timestamp(received_at) AT TIME ZONE 'UTC')::int AS dow,
|
SELECT EXTRACT(DOW FROM to_timestamp(received_at) AT TIME ZONE 'UTC')::int AS dow,
|
||||||
COUNT(*) AS cnt
|
COUNT(*) AS cnt
|
||||||
@@ -131,7 +138,7 @@ func handleStatsByDay(db *sql.DB) http.HandlerFunc {
|
|||||||
GROUP BY dow
|
GROUP BY dow
|
||||||
ORDER BY dow ASC`, where), args.all()...)
|
ORDER BY dow ASC`, where), args.all()...)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -141,11 +148,15 @@ func handleStatsByDay(db *sql.DB) http.HandlerFunc {
|
|||||||
var dow int
|
var dow int
|
||||||
var cnt int64
|
var cnt int64
|
||||||
if err := rows.Scan(&dow, &cnt); err != nil {
|
if err := rows.Scan(&dow, &cnt); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
counts[dow] = cnt
|
counts[dow] = cnt
|
||||||
}
|
}
|
||||||
|
if err := rows.Err(); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
dayNames := [7]string{"Sunday", "Monday", "Tuesday", "Wednesday", "Thursday", "Friday", "Saturday"}
|
dayNames := [7]string{"Sunday", "Monday", "Tuesday", "Wednesday", "Thursday", "Friday", "Saturday"}
|
||||||
type entry struct {
|
type entry struct {
|
||||||
@@ -186,7 +197,7 @@ func handleStatsIncidents(db *sql.DB) http.HandlerFunc {
|
|||||||
FROM incidents WHERE %s`, where), args.all()...,
|
FROM incidents WHERE %s`, where), args.all()...,
|
||||||
).Scan(&total, &triggered, &acknowledged, &resolved, &mtta, &mttr)
|
).Scan(&total, &triggered, &acknowledged, &resolved, &mtta, &mttr)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -13,21 +13,21 @@ import (
|
|||||||
"github.com/go-chi/chi/v5"
|
"github.com/go-chi/chi/v5"
|
||||||
)
|
)
|
||||||
|
|
||||||
// handleListTeams lists the caller's own teams, each with their role in it. An
|
// handleListTeams lists the caller's own teams, each with their role in it.
|
||||||
// administrator listing every team goes through the admin endpoint instead:
|
// An administrator listing every team goes through the admin endpoint instead:
|
||||||
// this one answers "what am I part of", which is what the UI's team filter and
|
// this answers "what am I part of", which is what the UI's team filter and the
|
||||||
// the combined queue are built from.
|
// combined queue are built from.
|
||||||
func handleListTeams(db *sql.DB) http.HandlerFunc {
|
func handleListTeams(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
caller, _ := userFromContext(r.Context())
|
caller, _ := userFromContext(r.Context())
|
||||||
rows, err := db.QueryContext(r.Context(), `
|
rows, err := db.QueryContext(r.Context(), `
|
||||||
SELECT t.id, t.name, t.created_at, m.role
|
SELECT t.id, t.name, t.created_at, m.role, m.source
|
||||||
FROM teams t
|
FROM teams t
|
||||||
JOIN team_members m ON m.team_id = t.id
|
JOIN team_members m ON m.team_id = t.id
|
||||||
WHERE m.user_id = $1
|
WHERE m.user_id = $1
|
||||||
ORDER BY t.name`, caller.ID)
|
ORDER BY t.name`, caller.ID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -36,15 +36,77 @@ func handleListTeams(db *sql.DB) http.HandlerFunc {
|
|||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
var t models.Team
|
var t models.Team
|
||||||
var created int64
|
var created int64
|
||||||
if err := rows.Scan(&t.ID, &t.Name, &created, &t.Role); err != nil {
|
if err := rows.Scan(&t.ID, &t.Name, &created, &t.Role, &t.Source); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
t.CreatedAt = time.Unix(created, 0).UTC()
|
t.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
teams = append(teams, t)
|
teams = append(teams, t)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusOK, teams)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleUserTeams lists one user's teams, for the admin page's per-user view:
|
||||||
|
// "what is this person in", which /api/teams cannot answer because it is always
|
||||||
|
// about the caller.
|
||||||
|
//
|
||||||
|
// Self or admin, matching the other per-user endpoints. It says which teams
|
||||||
|
// somebody belongs to and in what role — not anything those teams own, so it
|
||||||
|
// stays on the accounts side of the line the administrator flag draws.
|
||||||
|
func handleUserTeams(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
id, err := strconv.ParseInt(chi.URLParam(r, "id"), 10, 64)
|
||||||
|
if err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid user id"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !requireSelfOrAdmin(w, r, id) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// A user with no teams and a user who does not exist both list nothing,
|
||||||
|
// so the existence check is what tells them apart.
|
||||||
|
var exists bool
|
||||||
|
if err := db.QueryRowContext(r.Context(),
|
||||||
|
"SELECT EXISTS (SELECT 1 FROM users WHERE id = $1)", id).Scan(&exists); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !exists {
|
||||||
|
respond(w, http.StatusNotFound, errResp("user not found"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
rows, err := db.QueryContext(r.Context(), `
|
||||||
|
SELECT t.id, t.name, t.created_at, m.role, m.source
|
||||||
|
FROM teams t
|
||||||
|
JOIN team_members m ON m.team_id = t.id
|
||||||
|
WHERE m.user_id = $1
|
||||||
|
ORDER BY t.name`, id)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer rows.Close()
|
||||||
|
|
||||||
|
teams := []models.Team{}
|
||||||
|
for rows.Next() {
|
||||||
|
var t models.Team
|
||||||
|
var created int64
|
||||||
|
if err := rows.Scan(&t.ID, &t.Name, &created, &t.Role, &t.Source); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
t.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
|
teams = append(teams, t)
|
||||||
|
}
|
||||||
|
if err := rows.Err(); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, teams)
|
respond(w, http.StatusOK, teams)
|
||||||
@@ -54,10 +116,25 @@ func handleListTeams(db *sql.DB) http.HandlerFunc {
|
|||||||
// handleCreateTeam creates a team and makes its creator the first owner. A team
|
// handleCreateTeam creates a team and makes its creator the first owner. A team
|
||||||
// with no owner would need an administrator to repair before anybody could use
|
// with no owner would need an administrator to repair before anybody could use
|
||||||
// it, so the two happen in one transaction.
|
// it, so the two happen in one transaction.
|
||||||
|
//
|
||||||
|
// An instance-scoped service account may also create a team (SERVICE-ACCOUNTS.md:
|
||||||
|
// it acts with the same reach system administration has over teams), but it
|
||||||
|
// is not a users row and cannot become an owner the way a person does. The
|
||||||
|
// team it creates starts with no human owner at all — not a bug, the expected
|
||||||
|
// shape for one terdut-operator is about to provision: a system administrator
|
||||||
|
// can always act as owner to repair or hand it off (requireTeamOwner), and
|
||||||
|
// the account that created it mints itself a team-scoped credential for it
|
||||||
|
// next, via POST /api/service-accounts.
|
||||||
func handleCreateTeam(db *sql.DB) http.HandlerFunc {
|
func handleCreateTeam(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
var req struct {
|
var req struct {
|
||||||
Name string `json:"name"`
|
Name string `json:"name"`
|
||||||
|
// ExternalID makes the call idempotent for automation: a team
|
||||||
|
// already carrying it is returned as-is (200) instead of created, so
|
||||||
|
// a client that crashed between the POST and recording the id finds
|
||||||
|
// its own team again. Instance-scoped service accounts only; a name
|
||||||
|
// that belongs to a different team is still a 409.
|
||||||
|
ExternalID string `json:"external_id"`
|
||||||
}
|
}
|
||||||
if err := decodeJSON(r, &req); err != nil {
|
if err := decodeJSON(r, &req); err != nil {
|
||||||
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||||
@@ -69,40 +146,75 @@ func handleCreateTeam(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
caller, _ := userFromContext(r.Context())
|
caller, isUser := userFromContext(r.Context())
|
||||||
|
if req.ExternalID != "" && !isInstanceServiceAccount(r.Context()) {
|
||||||
|
respond(w, http.StatusForbidden, errResp("external_id is for instance-scoped service accounts"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !isUser && !isInstanceServiceAccount(r.Context()) {
|
||||||
|
// A team-scoped service account authenticates as owner of exactly
|
||||||
|
// one team already (see serveAsServiceAccount); letting it create
|
||||||
|
// another would reach outside that boundary.
|
||||||
|
respond(w, http.StatusForbidden, errResp("instance-scoped service account or user access required"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
tx, err := db.BeginTx(r.Context(), nil)
|
tx, err := db.BeginTx(r.Context(), nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer tx.Rollback() //nolint:errcheck
|
defer tx.Rollback() //nolint:errcheck
|
||||||
|
|
||||||
var team models.Team
|
var team models.Team
|
||||||
var created int64
|
var created int64
|
||||||
|
if req.ExternalID != "" {
|
||||||
|
err := tx.QueryRowContext(r.Context(),
|
||||||
|
"SELECT id, name, created_at FROM teams WHERE external_id = $1", req.ExternalID).
|
||||||
|
Scan(&team.ID, &team.Name, &created)
|
||||||
|
if err == nil {
|
||||||
|
team.ExternalID = &req.ExternalID
|
||||||
|
team.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
|
respond(w, http.StatusOK, team)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !errors.Is(err, sql.ErrNoRows) {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
}
|
||||||
|
var externalID *string
|
||||||
|
if req.ExternalID != "" {
|
||||||
|
externalID = &req.ExternalID
|
||||||
|
}
|
||||||
if err := tx.QueryRowContext(r.Context(),
|
if err := tx.QueryRowContext(r.Context(),
|
||||||
"INSERT INTO teams (name) VALUES ($1) RETURNING id, name, created_at",
|
"INSERT INTO teams (name, external_id) VALUES ($1, $2) RETURNING id, name, created_at",
|
||||||
req.Name).Scan(&team.ID, &team.Name, &created); err != nil {
|
req.Name, externalID).Scan(&team.ID, &team.Name, &created); err != nil {
|
||||||
if isUniqueViolation(err) {
|
if isUniqueViolation(err) {
|
||||||
respond(w, http.StatusConflict, errResp("a team with that name already exists"))
|
respond(w, http.StatusConflict, errResp("a team with that name already exists"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if _, err := tx.ExecContext(r.Context(),
|
if isUser {
|
||||||
"INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3)",
|
if _, err := tx.ExecContext(r.Context(),
|
||||||
team.ID, caller.ID, models.RoleOwner); err != nil {
|
"INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3)",
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
team.ID, caller.ID, models.RoleOwner); err != nil {
|
||||||
return
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if err := tx.Commit(); err != nil {
|
if err := tx.Commit(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
team.CreatedAt = time.Unix(created, 0).UTC()
|
team.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
team.Role = models.RoleOwner
|
team.ExternalID = externalID
|
||||||
|
if isUser {
|
||||||
|
team.Role = models.RoleOwner
|
||||||
|
}
|
||||||
respond(w, http.StatusCreated, team)
|
respond(w, http.StatusCreated, team)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -127,7 +239,7 @@ func handleDeleteTeam(db *sql.DB) http.HandlerFunc {
|
|||||||
if err := db.QueryRowContext(r.Context(),
|
if err := db.QueryRowContext(r.Context(),
|
||||||
"SELECT COUNT(*) FROM incidents WHERE team_id = $1 AND resolved_at IS NULL", teamID).
|
"SELECT COUNT(*) FROM incidents WHERE team_id = $1 AND resolved_at IS NULL", teamID).
|
||||||
Scan(&open); err != nil {
|
Scan(&open); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if open > 0 {
|
if open > 0 {
|
||||||
@@ -137,7 +249,7 @@ func handleDeleteTeam(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
res, err := db.ExecContext(r.Context(), "DELETE FROM teams WHERE id = $1", teamID)
|
res, err := db.ExecContext(r.Context(), "DELETE FROM teams WHERE id = $1", teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -148,8 +260,40 @@ func handleDeleteTeam(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// handleListTeamMembers names everybody in a team. Visible to any member: you
|
// Member statuses, as the Members page colours them.
|
||||||
// can see who else is on the rota you are on.
|
const (
|
||||||
|
memberOnCall = "oncall"
|
||||||
|
memberReachable = "reachable"
|
||||||
|
memberUnpageable = "unpageable"
|
||||||
|
)
|
||||||
|
|
||||||
|
// memberStatus is a team member with what matters about them at 03:00: whether
|
||||||
|
// they are on call, whether a page to them would go anywhere, and whether they
|
||||||
|
// have been around. The extra fields are output only.
|
||||||
|
type memberStatus struct {
|
||||||
|
models.TeamMember
|
||||||
|
|
||||||
|
// Status is unpageable when a page to them would go nowhere — even when
|
||||||
|
// they are on call, since that is the case that matters most — on_call when
|
||||||
|
// the rota has them today, reachable otherwise.
|
||||||
|
Status string `json:"status"`
|
||||||
|
|
||||||
|
OnCall bool `json:"on_call"`
|
||||||
|
|
||||||
|
// NextShift is the first day after today the rota has them (YYYY-MM-DD).
|
||||||
|
NextShift *string `json:"next_shift,omitempty"`
|
||||||
|
|
||||||
|
// Pageable is whether they have an ntfy topic and an enabled account — the
|
||||||
|
// conditions pageLevel and the notifier skip on. Never the topic itself.
|
||||||
|
Pageable bool `json:"pageable"`
|
||||||
|
Problem string `json:"problem,omitempty"`
|
||||||
|
|
||||||
|
// LastActiveAt is the last time they used a session or an API key.
|
||||||
|
LastActiveAt *time.Time `json:"last_active_at,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleListTeamMembers names everybody in a team, with their status. Visible to
|
||||||
|
// any member: you can see who else is on the rota you are on.
|
||||||
func handleListTeamMembers(db *sql.DB) http.HandlerFunc {
|
func handleListTeamMembers(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
teamID, ok := teamParam(w, r)
|
teamID, ok := teamParam(w, r)
|
||||||
@@ -161,30 +305,60 @@ func handleListTeamMembers(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
|
|
||||||
rows, err := db.QueryContext(r.Context(), `
|
rows, err := db.QueryContext(r.Context(), `
|
||||||
SELECT m.team_id, m.user_id, u.username, m.role, m.joined_at
|
SELECT m.team_id, m.user_id, u.username, m.role, m.joined_at, m.source,
|
||||||
|
u.ntfy_topic IS NOT NULL AND u.ntfy_topic <> '',
|
||||||
|
u.disabled_at IS NOT NULL,
|
||||||
|
GREATEST(
|
||||||
|
COALESCE((SELECT MAX(last_seen_at) FROM sessions WHERE user_id = u.id), 0),
|
||||||
|
COALESCE((SELECT MAX(last_used_at) FROM api_keys WHERE user_id = u.id), 0)),
|
||||||
|
EXISTS (SELECT 1 FROM schedule_entries s
|
||||||
|
WHERE s.team_id = m.team_id AND s.user_id = u.id AND s.date = $2),
|
||||||
|
(SELECT MIN(date) FROM schedule_entries s
|
||||||
|
WHERE s.team_id = m.team_id AND s.user_id = u.id AND s.date > $2)
|
||||||
FROM team_members m
|
FROM team_members m
|
||||||
JOIN users u ON u.id = m.user_id
|
JOIN users u ON u.id = m.user_id
|
||||||
WHERE m.team_id = $1
|
WHERE m.team_id = $1
|
||||||
ORDER BY u.username`, teamID)
|
ORDER BY u.username`, teamID, todayUTC())
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
|
|
||||||
members := []models.TeamMember{}
|
members := []memberStatus{}
|
||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
var m models.TeamMember
|
var m memberStatus
|
||||||
var joined int64
|
var joined, lastActive int64
|
||||||
if err := rows.Scan(&m.TeamID, &m.UserID, &m.Username, &m.Role, &joined); err != nil {
|
var hasTopic, disabled bool
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
if err := rows.Scan(&m.TeamID, &m.UserID, &m.Username, &m.Role, &joined, &m.Source,
|
||||||
|
&hasTopic, &disabled, &lastActive, &m.OnCall, &m.NextShift); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
m.JoinedAt = time.Unix(joined, 0).UTC()
|
m.JoinedAt = time.Unix(joined, 0).UTC()
|
||||||
|
if lastActive > 0 {
|
||||||
|
t := time.Unix(lastActive, 0).UTC()
|
||||||
|
m.LastActiveAt = &t
|
||||||
|
}
|
||||||
|
switch {
|
||||||
|
case disabled:
|
||||||
|
m.Problem = "account is disabled"
|
||||||
|
case !hasTopic:
|
||||||
|
m.Problem = "has no ntfy topic"
|
||||||
|
}
|
||||||
|
m.Pageable = m.Problem == ""
|
||||||
|
switch {
|
||||||
|
case !m.Pageable:
|
||||||
|
m.Status = memberUnpageable
|
||||||
|
case m.OnCall:
|
||||||
|
m.Status = memberOnCall
|
||||||
|
default:
|
||||||
|
m.Status = memberReachable
|
||||||
|
}
|
||||||
members = append(members, m)
|
members = append(members, m)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, members)
|
respond(w, http.StatusOK, members)
|
||||||
@@ -219,6 +393,28 @@ func handleAddTeamMember(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if managed, err := isSSOManagedMember(r.Context(), db, teamID, req.UserID); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
} else if managed {
|
||||||
|
respond(w, http.StatusConflict, errResp(ssoManagedMsg))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// Demoting the last owner is removing them by another route: the team
|
||||||
|
// would have nobody who can edit it.
|
||||||
|
if req.Role == models.RoleMember {
|
||||||
|
last, err := isLastTeamOwner(r.Context(), db, teamID, req.UserID)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if last {
|
||||||
|
respond(w, http.StatusConflict, errResp("cannot demote the last owner of a team"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
_, err := db.ExecContext(r.Context(), `
|
_, err := db.ExecContext(r.Context(), `
|
||||||
INSERT INTO team_members (team_id, user_id, role)
|
INSERT INTO team_members (team_id, user_id, role)
|
||||||
VALUES ($1, $2, $3)
|
VALUES ($1, $2, $3)
|
||||||
@@ -254,9 +450,17 @@ func handleRemoveTeamMember(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if managed, err := isSSOManagedMember(r.Context(), db, teamID, userID); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
} else if managed {
|
||||||
|
respond(w, http.StatusConflict, errResp(ssoManagedMsg))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
last, err := isLastTeamOwner(r.Context(), db, teamID, userID)
|
last, err := isLastTeamOwner(r.Context(), db, teamID, userID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if last {
|
if last {
|
||||||
@@ -267,7 +471,7 @@ func handleRemoveTeamMember(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"DELETE FROM team_members WHERE team_id = $1 AND user_id = $2", teamID, userID)
|
"DELETE FROM team_members WHERE team_id = $1 AND user_id = $2", teamID, userID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -278,6 +482,20 @@ func handleRemoveTeamMember(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ssoManagedMsg is the refusal for editing access that single sign-on owns.
|
||||||
|
const ssoManagedMsg = "this membership is managed by single sign-on; change the user's groups in the identity provider"
|
||||||
|
|
||||||
|
// isSSOManagedMember reports whether the membership comes from the group sync.
|
||||||
|
// Editing it here would be undone at the person's next sign-in, so it is refused
|
||||||
|
// instead of appearing to work.
|
||||||
|
func isSSOManagedMember(ctx context.Context, db *sql.DB, teamID, userID int64) (bool, error) {
|
||||||
|
var managed bool
|
||||||
|
err := db.QueryRowContext(ctx,
|
||||||
|
"SELECT EXISTS (SELECT 1 FROM team_members WHERE team_id = $1 AND user_id = $2 AND source = 'oidc')",
|
||||||
|
teamID, userID).Scan(&managed)
|
||||||
|
return managed, err
|
||||||
|
}
|
||||||
|
|
||||||
func isLastTeamOwner(ctx context.Context, db *sql.DB, teamID, userID int64) (bool, error) {
|
func isLastTeamOwner(ctx context.Context, db *sql.DB, teamID, userID int64) (bool, error) {
|
||||||
var last bool
|
var last bool
|
||||||
err := db.QueryRowContext(ctx, `
|
err := db.QueryRowContext(ctx, `
|
||||||
@@ -293,8 +511,42 @@ func isLastTeamOwner(ctx context.Context, db *sql.DB, teamID, userID int64) (boo
|
|||||||
// Integrations
|
// Integrations
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
// handleListIntegrations lists a team's integrations. Never the keys: those
|
// sourceQuietAfter is how long a source may go without posting before the
|
||||||
// exist in plaintext only in the response that created them.
|
// Sources page calls it quiet rather than active. A day is longer than any
|
||||||
|
// repeat_interval worth having, so an Alertmanager that is up and has anything
|
||||||
|
// firing never crosses it; a source with nothing firing may, and that is a
|
||||||
|
// reason to look, not proof of a fault — which is why this is a colour and not
|
||||||
|
// an alarm. Dead man's switches are where silence pages.
|
||||||
|
const sourceQuietAfter = 24 * time.Hour
|
||||||
|
|
||||||
|
const (
|
||||||
|
sourceActive = "active"
|
||||||
|
sourceQuiet = "quiet"
|
||||||
|
sourceNever = "never"
|
||||||
|
)
|
||||||
|
|
||||||
|
// integrationStatus is an integration as the Sources page shows it.
|
||||||
|
type integrationStatus struct {
|
||||||
|
models.Integration
|
||||||
|
|
||||||
|
// Status is active when the key posted within sourceQuietAfter, quiet when
|
||||||
|
// it has posted but not lately, never when it has not posted at all.
|
||||||
|
Status string `json:"status"`
|
||||||
|
|
||||||
|
// LastAlertAt is when an alert last arrived on this source, which is not the
|
||||||
|
// same as when it last posted: a payload with nothing usable in it stamps
|
||||||
|
// last_used_at and not this. Absent until an alert has arrived since
|
||||||
|
// migration 010 started recording it.
|
||||||
|
LastAlertAt *time.Time `json:"last_alert_at,omitempty"`
|
||||||
|
|
||||||
|
// Alerts24h counts the distinct alerts this source refreshed in the last
|
||||||
|
// day. An alert re-sent every few hours counts once, not once per re-send.
|
||||||
|
Alerts24h int64 `json:"alerts_24h"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleListIntegrations lists a team's integrations with what each has been
|
||||||
|
// delivering. Never the keys: those exist in plaintext only in the response that
|
||||||
|
// created them.
|
||||||
func handleListIntegrations(db *sql.DB) http.HandlerFunc {
|
func handleListIntegrations(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
teamID, ok := teamParam(w, r)
|
teamID, ok := teamParam(w, r)
|
||||||
@@ -305,32 +557,49 @@ func handleListIntegrations(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
now := time.Now()
|
||||||
rows, err := db.QueryContext(r.Context(), `
|
rows, err := db.QueryContext(r.Context(), `
|
||||||
SELECT id, team_id, kind, name, created_at, last_used_at
|
SELECT i.id, i.team_id, i.kind, i.name, i.created_at, i.last_used_at,
|
||||||
FROM integrations
|
-- Scalar subqueries, not a join and GROUP BY: each is a
|
||||||
WHERE team_id = $1
|
-- single range over alerts_integration_idx, where the join
|
||||||
ORDER BY id`, teamID)
|
-- would read every alert a source ever delivered.
|
||||||
|
(SELECT MAX(received_at) FROM alerts WHERE integration_id = i.id),
|
||||||
|
(SELECT COUNT(*) FROM alerts
|
||||||
|
WHERE integration_id = i.id AND received_at >= $2)
|
||||||
|
FROM integrations i
|
||||||
|
WHERE i.team_id = $1
|
||||||
|
ORDER BY i.id`, teamID, now.Add(-sourceQuietAfter).Unix())
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
|
|
||||||
integrations := []models.Integration{}
|
integrations := []integrationStatus{}
|
||||||
for rows.Next() {
|
for rows.Next() {
|
||||||
var i models.Integration
|
var i integrationStatus
|
||||||
var created int64
|
var created int64
|
||||||
var lastUsed *int64
|
var lastUsed, lastAlert *int64
|
||||||
if err := rows.Scan(&i.ID, &i.TeamID, &i.Kind, &i.Name, &created, &lastUsed); err != nil {
|
if err := rows.Scan(&i.ID, &i.TeamID, &i.Kind, &i.Name, &created, &lastUsed,
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
&lastAlert, &i.Alerts24h); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
i.CreatedAt = time.Unix(created, 0).UTC()
|
i.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
i.LastUsedAt = unixPtr(lastUsed)
|
i.LastUsedAt = unixPtr(lastUsed)
|
||||||
|
i.LastAlertAt = unixPtr(lastAlert)
|
||||||
|
switch {
|
||||||
|
case i.LastUsedAt == nil:
|
||||||
|
i.Status = sourceNever
|
||||||
|
case now.Sub(*i.LastUsedAt) > sourceQuietAfter:
|
||||||
|
i.Status = sourceQuiet
|
||||||
|
default:
|
||||||
|
i.Status = sourceActive
|
||||||
|
}
|
||||||
integrations = append(integrations, i)
|
integrations = append(integrations, i)
|
||||||
}
|
}
|
||||||
if err := rows.Err(); err != nil {
|
if err := rows.Err(); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, integrations)
|
respond(w, http.StatusOK, integrations)
|
||||||
@@ -373,7 +642,7 @@ func handleCreateIntegration(db *sql.DB, publicURL string) http.HandlerFunc {
|
|||||||
|
|
||||||
raw, hash, err := randomToken()
|
raw, hash, err := randomToken()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -385,7 +654,11 @@ func handleCreateIntegration(db *sql.DB, publicURL string) http.HandlerFunc {
|
|||||||
RETURNING id, team_id, kind, name, created_at`,
|
RETURNING id, team_id, kind, name, created_at`,
|
||||||
teamID, req.Kind, req.Name, hash).
|
teamID, req.Kind, req.Name, hash).
|
||||||
Scan(&i.ID, &i.TeamID, &i.Kind, &i.Name, &created); err != nil {
|
Scan(&i.ID, &i.TeamID, &i.Kind, &i.Name, &created); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
if isUniqueViolation(err) {
|
||||||
|
respond(w, http.StatusConflict, errResp("an integration with that name already exists in this team"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
i.CreatedAt = time.Unix(created, 0).UTC()
|
i.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
@@ -395,6 +668,58 @@ func handleCreateIntegration(db *sql.DB, publicURL string) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// handleRenameIntegration renames a source. The key is untouched, so nothing
|
||||||
|
// posting with it notices.
|
||||||
|
func handleRenameIntegration(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
teamID, ok := teamParam(w, r)
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !requireTeamOwner(w, r, teamID) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
id, err := strconv.ParseInt(chi.URLParam(r, "integrationID"), 10, 64)
|
||||||
|
if err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid integration id"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
var req struct {
|
||||||
|
Name string `json:"name"`
|
||||||
|
}
|
||||||
|
if err := decodeJSON(r, &req); err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
req.Name = strings.TrimSpace(req.Name)
|
||||||
|
if req.Name == "" {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("name is required"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if len(req.Name) > 100 {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("name is too long"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
res, err := db.ExecContext(r.Context(),
|
||||||
|
"UPDATE integrations SET name = $1 WHERE id = $2 AND team_id = $3", req.Name, id, teamID)
|
||||||
|
if err != nil {
|
||||||
|
if isUniqueViolation(err) {
|
||||||
|
respond(w, http.StatusConflict, errResp("an integration with that name already exists in this team"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
|
respond(w, http.StatusNotFound, errResp("not found"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
w.WriteHeader(http.StatusNoContent)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func handleDeleteIntegration(db *sql.DB) http.HandlerFunc {
|
func handleDeleteIntegration(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
teamID, ok := teamParam(w, r)
|
teamID, ok := teamParam(w, r)
|
||||||
@@ -413,7 +738,7 @@ func handleDeleteIntegration(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"DELETE FROM integrations WHERE id = $1 AND team_id = $2", id, teamID)
|
"DELETE FROM integrations WHERE id = $1 AND team_id = $2", id, teamID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -431,24 +756,32 @@ func integrationPath(key, kind string) string {
|
|||||||
return "/api/integrations/" + key + "/" + kind
|
return "/api/integrations/" + key + "/" + kind
|
||||||
}
|
}
|
||||||
|
|
||||||
// teamIDForKey resolves an integration key to its team, and stamps the key's
|
// alertSource is who an arriving webhook is from: the integration whose key it
|
||||||
|
// used, and the team that integration puts its alerts in.
|
||||||
|
type alertSource struct {
|
||||||
|
integrationID int64
|
||||||
|
teamID int64
|
||||||
|
}
|
||||||
|
|
||||||
|
// sourceForKey resolves an integration key to its source, and stamps the key's
|
||||||
// last use. An unknown key is not an error worth distinguishing: the caller is
|
// last use. An unknown key is not an error worth distinguishing: the caller is
|
||||||
// told nothing beyond "no".
|
// told nothing beyond "no".
|
||||||
func teamIDForKey(ctx context.Context, db *sql.DB, key string) (int64, error) {
|
func sourceForKey(ctx context.Context, db *sql.DB, key string) (alertSource, error) {
|
||||||
var teamID int64
|
var src alertSource
|
||||||
err := db.QueryRowContext(ctx,
|
err := db.QueryRowContext(ctx,
|
||||||
"SELECT team_id FROM integrations WHERE key_hash = $1", hashToken(key)).Scan(&teamID)
|
"SELECT id, team_id FROM integrations WHERE key_hash = $1", hashToken(key)).
|
||||||
|
Scan(&src.integrationID, &src.teamID)
|
||||||
if errors.Is(err, sql.ErrNoRows) {
|
if errors.Is(err, sql.ErrNoRows) {
|
||||||
return 0, errUnknownIntegration
|
return alertSource{}, errUnknownIntegration
|
||||||
}
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return 0, err
|
return alertSource{}, err
|
||||||
}
|
}
|
||||||
// Best effort, like an API key's: a failed stamp must not reject an alert.
|
// Best effort, like an API key's: a failed stamp must not reject an alert.
|
||||||
db.ExecContext(ctx, //nolint:errcheck
|
db.ExecContext(ctx, //nolint:errcheck
|
||||||
"UPDATE integrations SET last_used_at = $1 WHERE key_hash = $2",
|
"UPDATE integrations SET last_used_at = $1 WHERE id = $2",
|
||||||
time.Now().Unix(), hashToken(key))
|
time.Now().Unix(), src.integrationID)
|
||||||
return teamID, nil
|
return src, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
var errUnknownIntegration = errors.New("unknown integration key")
|
var errUnknownIntegration = errors.New("unknown integration key")
|
||||||
@@ -463,31 +796,28 @@ func teamParam(w http.ResponseWriter, r *http.Request) (int64, bool) {
|
|||||||
return id, true
|
return id, true
|
||||||
}
|
}
|
||||||
|
|
||||||
// defaultTeamID is the oldest team, which on an upgraded install is the
|
|
||||||
// "Default" team every pre-teams row was moved into and on a fresh one is the
|
|
||||||
// team migration 003 creates. Bootstrap puts the first user in it, so somebody
|
|
||||||
// signing in to a new server lands somewhere rather than in no team at all.
|
|
||||||
func defaultTeamID(ctx context.Context, db *sql.DB) (int64, error) {
|
|
||||||
var id int64
|
|
||||||
err := db.QueryRowContext(ctx, "SELECT id FROM teams ORDER BY id LIMIT 1").Scan(&id)
|
|
||||||
return id, err
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// A team's dead man's switches
|
// A team's dead man's switches
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
// deadmanResponse is the wire shape of a team's switch configuration. The
|
// deadmanSwitchRequest is what creating a switch takes. The timeout is seconds,
|
||||||
// timeout is seconds rather than a duration string, because that is what the
|
// because that is what the column holds and what arithmetic is done on; a client
|
||||||
// column holds and what arithmetic is done on; a client renders it.
|
// renders it.
|
||||||
type deadmanResponse struct {
|
type deadmanSwitchRequest struct {
|
||||||
TeamID int64 `json:"team_id"`
|
Name string `json:"name"`
|
||||||
Matchers string `json:"matchers"`
|
Matcher string `json:"matcher"`
|
||||||
TimeoutSeconds int64 `json:"timeout_seconds"`
|
TimeoutSeconds int64 `json:"timeout_seconds"`
|
||||||
Severity string `json:"severity"`
|
Severity string `json:"severity"`
|
||||||
}
|
}
|
||||||
|
|
||||||
func handleGetTeamDeadman(db *sql.DB) http.HandlerFunc {
|
// deadmanSeverities are the severities an incident can open at.
|
||||||
|
var deadmanSeverities = map[string]bool{"critical": true, "error": true, "warning": true, "info": true}
|
||||||
|
|
||||||
|
// handleListTeamDeadman lists a team's switches with what each one's heartbeats
|
||||||
|
// are doing. A team with none gets an empty list, which is a configuration and
|
||||||
|
// not an absence: answering 404 would make "off" indistinguishable from "this
|
||||||
|
// server does not do this".
|
||||||
|
func handleListTeamDeadman(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
teamID, ok := teamParam(w, r)
|
teamID, ok := teamParam(w, r)
|
||||||
if !ok {
|
if !ok {
|
||||||
@@ -497,27 +827,26 @@ func handleGetTeamDeadman(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
out := deadmanResponse{TeamID: teamID, Severity: "critical"}
|
set, err := deadmanSetForTeam(r.Context(), db, teamID)
|
||||||
err := db.QueryRowContext(r.Context(),
|
if err != nil {
|
||||||
"SELECT matchers, timeout_seconds, severity FROM deadman_configs WHERE team_id = $1",
|
serverError(w, r, err)
|
||||||
teamID).Scan(&out.Matchers, &out.TimeoutSeconds, &out.Severity)
|
return
|
||||||
if err != nil && !errors.Is(err, sql.ErrNoRows) {
|
}
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
out, err := deadmanStatuses(r.Context(), db, teamID, set, time.Now())
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
// A team with no row watches nothing, which is a configuration and not
|
|
||||||
// an absence: answering 404 would make "off" indistinguishable from
|
|
||||||
// "this server does not do this".
|
|
||||||
respond(w, http.StatusOK, out)
|
respond(w, http.StatusOK, out)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// handleSetTeamDeadman replaces a team's switch configuration.
|
// handleCreateTeamDeadman adds one switch.
|
||||||
//
|
//
|
||||||
// Validated by parsing: a matcher string that survives ParseDeadmanConfig with
|
// Validated by parsing: a matcher with no alertname is rejected rather than
|
||||||
// nothing usable in it is rejected rather than stored, because a switch that
|
// stored, because a switch that silently watches nothing is the failure this
|
||||||
// silently watches nothing is the failure this feature exists to prevent.
|
// feature exists to prevent.
|
||||||
func handleSetTeamDeadman(db *sql.DB) http.HandlerFunc {
|
func handleCreateTeamDeadman(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
teamID, ok := teamParam(w, r)
|
teamID, ok := teamParam(w, r)
|
||||||
if !ok {
|
if !ok {
|
||||||
@@ -527,50 +856,190 @@ func handleSetTeamDeadman(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
var req struct {
|
var req deadmanSwitchRequest
|
||||||
Matchers string `json:"matchers"`
|
|
||||||
TimeoutSeconds int64 `json:"timeout_seconds"`
|
|
||||||
Severity string `json:"severity"`
|
|
||||||
}
|
|
||||||
if err := decodeJSON(r, &req); err != nil {
|
if err := decodeJSON(r, &req); err != nil {
|
||||||
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
req.Matchers = strings.TrimSpace(req.Matchers)
|
req.Matcher = strings.TrimSpace(req.Matcher)
|
||||||
|
req.Name = strings.TrimSpace(req.Name)
|
||||||
if req.Severity == "" {
|
if req.Severity == "" {
|
||||||
req.Severity = "critical"
|
req.Severity = "critical"
|
||||||
}
|
}
|
||||||
if req.TimeoutSeconds < 0 {
|
if !deadmanSeverities[req.Severity] {
|
||||||
respond(w, http.StatusBadRequest, errResp("timeout_seconds must not be negative"))
|
respond(w, http.StatusBadRequest, errResp("severity must be critical, error, warning or info"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if req.Matchers != "" {
|
if req.TimeoutSeconds <= 0 {
|
||||||
parsed := parseDeadmanQuietly(req.Matchers, time.Duration(req.TimeoutSeconds)*time.Second, req.Severity)
|
respond(w, http.StatusBadRequest, errResp("timeout_seconds must be positive"))
|
||||||
if len(parsed.Matchers) == 0 {
|
return
|
||||||
respond(w, http.StatusBadRequest, errResp(
|
}
|
||||||
"no usable matchers: each must name an alertname, as in alertname=Watchdog,cluster=prod"))
|
if strings.Contains(req.Matcher, ";") {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("one matcher per switch: add another switch instead of separating with ;"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
m, err := parseDeadmanMatcher(req.Matcher)
|
||||||
|
if err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp(
|
||||||
|
"unusable matcher ("+err.Error()+"): each must name an alertname, as in alertname=Watchdog,cluster=prod"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if req.Name == "" {
|
||||||
|
req.Name = m.config()
|
||||||
|
}
|
||||||
|
if len(req.Name) > 100 {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("name is too long"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
var id int64
|
||||||
|
if err := db.QueryRowContext(r.Context(), `
|
||||||
|
INSERT INTO deadman_switches (team_id, name, matcher, timeout_seconds, severity)
|
||||||
|
VALUES ($1, $2, $3, $4, $5) RETURNING id`,
|
||||||
|
teamID, req.Name, m.config(), req.TimeoutSeconds, req.Severity).Scan(&id); err != nil {
|
||||||
|
if isUniqueViolation(err) {
|
||||||
|
respond(w, http.StatusConflict, errResp("a switch with that name already exists in this team"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
}
|
serverError(w, r, err)
|
||||||
|
|
||||||
if _, err := db.ExecContext(r.Context(), `
|
|
||||||
INSERT INTO deadman_configs (team_id, matchers, timeout_seconds, severity, updated_at)
|
|
||||||
VALUES ($1, $2, $3, $4, `+nowEpoch+`)
|
|
||||||
ON CONFLICT (team_id) DO UPDATE SET
|
|
||||||
matchers = excluded.matchers,
|
|
||||||
timeout_seconds = excluded.timeout_seconds,
|
|
||||||
severity = excluded.severity,
|
|
||||||
updated_at = excluded.updated_at`,
|
|
||||||
teamID, req.Matchers, req.TimeoutSeconds, req.Severity); err != nil {
|
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
respond(w, http.StatusOK, deadmanResponse{
|
respond(w, http.StatusCreated, deadmanSwitchStatus{
|
||||||
TeamID: teamID,
|
ID: id, Name: req.Name, Matcher: m.config(),
|
||||||
Matchers: req.Matchers,
|
TimeoutSeconds: req.TimeoutSeconds, Severity: req.Severity,
|
||||||
TimeoutSeconds: req.TimeoutSeconds,
|
Status: switchDormant, Sources: []deadmanSource{},
|
||||||
Severity: req.Severity,
|
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// handleUpdateTeamDeadman replaces one switch's configuration in place.
|
||||||
|
// Added alongside create/delete so an automated caller (terdut-operator) can
|
||||||
|
// reconcile a spec change without deleting and recreating the switch, which
|
||||||
|
// would otherwise be the only option and would needlessly rotate its id for
|
||||||
|
// no reason a reconciler's diff should ever manufacture.
|
||||||
|
func handleUpdateTeamDeadman(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
teamID, ok := teamParam(w, r)
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !requireTeamOwner(w, r, teamID) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
switchID, err := strconv.ParseInt(chi.URLParam(r, "switchID"), 10, 64)
|
||||||
|
if err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid switch id"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
var req deadmanSwitchRequest
|
||||||
|
if err := decodeJSON(r, &req); err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
req.Matcher = strings.TrimSpace(req.Matcher)
|
||||||
|
req.Name = strings.TrimSpace(req.Name)
|
||||||
|
if req.Severity == "" {
|
||||||
|
req.Severity = "critical"
|
||||||
|
}
|
||||||
|
if !deadmanSeverities[req.Severity] {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("severity must be critical, error, warning or info"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if req.TimeoutSeconds <= 0 {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("timeout_seconds must be positive"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if strings.Contains(req.Matcher, ";") {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("one matcher per switch: add another switch instead of separating with ;"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
m, err := parseDeadmanMatcher(req.Matcher)
|
||||||
|
if err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp(
|
||||||
|
"unusable matcher ("+err.Error()+"): each must name an alertname, as in alertname=Watchdog,cluster=prod"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if req.Name == "" {
|
||||||
|
req.Name = m.config()
|
||||||
|
}
|
||||||
|
if len(req.Name) > 100 {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("name is too long"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
res, err := db.ExecContext(r.Context(), `
|
||||||
|
UPDATE deadman_switches
|
||||||
|
SET name = $1, matcher = $2, timeout_seconds = $3, severity = $4
|
||||||
|
WHERE id = $5 AND team_id = $6`,
|
||||||
|
req.Name, m.config(), req.TimeoutSeconds, req.Severity, switchID, teamID)
|
||||||
|
if err != nil {
|
||||||
|
if isUniqueViolation(err) {
|
||||||
|
respond(w, http.StatusConflict, errResp("a switch with that name already exists in this team"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
|
respond(w, http.StatusNotFound, errResp("switch not found"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// The full status, not the bare request echoed back: an update can
|
||||||
|
// change whether the switch is dormant, alive or dead (a longer
|
||||||
|
// timeout can revive one that just went dead), and a caller
|
||||||
|
// reconciling against status deserves the same view
|
||||||
|
// handleListTeamDeadman would give it.
|
||||||
|
set, err := deadmanSetForTeam(r.Context(), db, teamID)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
statuses, err := deadmanStatuses(r.Context(), db, teamID, set, time.Now())
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
for _, s := range statuses {
|
||||||
|
if s.ID == switchID {
|
||||||
|
respond(w, http.StatusOK, s)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
}
|
||||||
|
serverError(w, r, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleDeleteTeamDeadman removes a switch. An incident it already opened stays
|
||||||
|
// open until somebody resolves it: deleting the switch says "stop watching", not
|
||||||
|
// "the problem is gone".
|
||||||
|
func handleDeleteTeamDeadman(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
teamID, ok := teamParam(w, r)
|
||||||
|
if !ok {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !requireTeamOwner(w, r, teamID) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
switchID, err := strconv.ParseInt(chi.URLParam(r, "switchID"), 10, 64)
|
||||||
|
if err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid switch id"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
res, err := db.ExecContext(r.Context(),
|
||||||
|
"DELETE FROM deadman_switches WHERE id = $1 AND team_id = $2", switchID, teamID)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
|
respond(w, http.StatusNotFound, errResp("switch not found"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
w.WriteHeader(http.StatusNoContent)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -144,7 +144,6 @@ func TestTeams_IncidentsAreScopedToTheReceivingTeam(t *testing.T) {
|
|||||||
otherID := int64(blueIncidents[0]["id"].(float64))
|
otherID := int64(blueIncidents[0]["id"].(float64))
|
||||||
for _, path := range []string{
|
for _, path := range []string{
|
||||||
"/api/incidents/" + id64(otherID),
|
"/api/incidents/" + id64(otherID),
|
||||||
"/api/incidents/" + id64(otherID) + "/alerts",
|
|
||||||
"/api/incidents/" + id64(otherID) + "/timeline",
|
"/api/incidents/" + id64(otherID) + "/timeline",
|
||||||
} {
|
} {
|
||||||
resp := red.call(http.MethodGet, path, nil)
|
resp := red.call(http.MethodGet, path, nil)
|
||||||
@@ -327,6 +326,112 @@ func TestTeams_MemberCannotConfigureTheTeam(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A team's own OIDC group binding follows the same rule as its schedule and
|
||||||
|
// its integrations: an owner sets it, a member may only read it, an outsider
|
||||||
|
// learns nothing, and an administrator can still reach it to repair a team
|
||||||
|
// whose owner has left.
|
||||||
|
func TestTeamOIDCGroups_OwnerOnlyToEdit(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
team := newTeam(t, s, "sre") // team.call authenticates as its owner
|
||||||
|
|
||||||
|
// A plain member of the same team.
|
||||||
|
var plain struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodPost, "/api/users",
|
||||||
|
map[string]string{"username": "plain", "email": "plain@test.com"}), &plain)
|
||||||
|
resp := s.req(t, http.MethodPost, "/api/teams/"+id64(team.id)+"/members",
|
||||||
|
map[string]any{"user_id": plain.ID, "role": "member"})
|
||||||
|
resp.Body.Close()
|
||||||
|
var key struct {
|
||||||
|
Key string `json:"key"`
|
||||||
|
}
|
||||||
|
decode(t, s.req(t, http.MethodPost, "/api/users/"+id64(plain.ID)+"/api-keys",
|
||||||
|
map[string]string{"name": "test"}), &key)
|
||||||
|
memberCall := func(method, path string, body any) *http.Response {
|
||||||
|
t.Helper()
|
||||||
|
var r io.Reader
|
||||||
|
if body != nil {
|
||||||
|
data, _ := json.Marshal(body)
|
||||||
|
r = bytes.NewReader(data)
|
||||||
|
}
|
||||||
|
req, _ := http.NewRequest(method, s.URL+path, r)
|
||||||
|
req.Header.Set("Authorization", "Bearer "+key.Key)
|
||||||
|
if body != nil {
|
||||||
|
req.Header.Set("Content-Type", "application/json")
|
||||||
|
}
|
||||||
|
resp, err := http.DefaultClient.Do(req)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("%s %s: %v", method, path, err)
|
||||||
|
}
|
||||||
|
return resp
|
||||||
|
}
|
||||||
|
|
||||||
|
// A member of a different team altogether.
|
||||||
|
_, outsiderCall := member(t, s, "outsider")
|
||||||
|
|
||||||
|
path := "/api/teams/" + id64(team.id) + "/oidc-groups"
|
||||||
|
|
||||||
|
resp = team.call(http.MethodPut, path, map[string]string{"member_group": "sre", "owner_group": "sre-leads"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNoContent {
|
||||||
|
t.Errorf("owner PUT: %d, want 204", resp.StatusCode)
|
||||||
|
}
|
||||||
|
var got struct {
|
||||||
|
MemberGroup string `json:"member_group"`
|
||||||
|
OwnerGroup string `json:"owner_group"`
|
||||||
|
}
|
||||||
|
decode(t, team.call(http.MethodGet, path, nil), &got)
|
||||||
|
if got.MemberGroup != "sre" || got.OwnerGroup != "sre-leads" {
|
||||||
|
t.Errorf("owner GET after PUT: %+v", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
resp = memberCall(http.MethodGet, path, nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
t.Errorf("member GET: %d, want 200", resp.StatusCode)
|
||||||
|
}
|
||||||
|
resp = memberCall(http.MethodPut, path, map[string]string{"member_group": "anything"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusForbidden {
|
||||||
|
t.Errorf("member PUT: %d, want 403", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// 404, not 403: whether the team exists is itself something only its
|
||||||
|
// members should learn.
|
||||||
|
resp = outsiderCall(http.MethodGet, path, nil)
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNotFound {
|
||||||
|
t.Errorf("outsider GET: %d, want 404", resp.StatusCode)
|
||||||
|
}
|
||||||
|
resp = outsiderCall(http.MethodPut, path, map[string]string{"member_group": "anything"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNotFound {
|
||||||
|
t.Errorf("outsider PUT: %d, want 404", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// An administrator who is not a member may still set it, the same bypass
|
||||||
|
// that lets one repair a team whose owner has left.
|
||||||
|
resp = s.req(t, http.MethodPut, path, map[string]string{"member_group": "sre2"})
|
||||||
|
resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusNoContent {
|
||||||
|
t.Errorf("admin PUT: %d, want 204", resp.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// An empty string clears a binding, stored as NULL rather than the literal
|
||||||
|
// empty string, so an empty group claim can never accidentally match it.
|
||||||
|
resp = team.call(http.MethodPut, path, map[string]string{"member_group": "", "owner_group": ""})
|
||||||
|
resp.Body.Close()
|
||||||
|
var cleared struct {
|
||||||
|
MemberGroup string `json:"member_group"`
|
||||||
|
OwnerGroup string `json:"owner_group"`
|
||||||
|
}
|
||||||
|
decode(t, team.call(http.MethodGet, path, nil), &cleared)
|
||||||
|
if cleared.MemberGroup != "" || cleared.OwnerGroup != "" {
|
||||||
|
t.Errorf("cleared: %+v", cleared)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// A team is not somewhere an outsider can look, whatever they know about it.
|
// A team is not somewhere an outsider can look, whatever they know about it.
|
||||||
func TestTeams_OutsiderSeesNothing(t *testing.T) {
|
func TestTeams_OutsiderSeesNothing(t *testing.T) {
|
||||||
s := newTS(t)
|
s := newTS(t)
|
||||||
@@ -348,3 +453,63 @@ func TestTeams_OutsiderSeesNothing(t *testing.T) {
|
|||||||
t.Errorf("blue's team list: %v", teams)
|
t.Errorf("blue's team list: %v", teams)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// GET /api/teams?name= (TEAM-LOOKUP.md)
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
// Everybody signed in can list users to name them, but only an admin (or the
|
||||||
|
// row's owner) sees an email or an ntfy topic, which is a publish secret.
|
||||||
|
func TestListUsers_RedactsEmailAndTopicForOthers(t *testing.T) {
|
||||||
|
s := newTS(t)
|
||||||
|
red := newTeam(t, s, "red")
|
||||||
|
_ = newTeam(t, s, "blue")
|
||||||
|
s.exec(t, "UPDATE users SET ntfy_topic = 'secret-topic'")
|
||||||
|
|
||||||
|
var asAdmin []map[string]any
|
||||||
|
decode(t, s.req(t, http.MethodGet, "/api/users", nil), &asAdmin)
|
||||||
|
for _, u := range asAdmin {
|
||||||
|
if u["email"] == "" || u["ntfy_topic"] != "secret-topic" {
|
||||||
|
t.Errorf("admin should see everything, got %v", u)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
asMember := list(t, red.call(http.MethodGet, "/api/users", nil))
|
||||||
|
if len(asMember) < 3 {
|
||||||
|
t.Fatalf("expected the whole user list, got %v", asMember)
|
||||||
|
}
|
||||||
|
for _, u := range asMember {
|
||||||
|
own := u["username"] == "red-user"
|
||||||
|
if own != (u["email"] != "") || own != (u["ntfy_topic"] != nil) {
|
||||||
|
t.Errorf("only red-user's own row should keep email and topic, got %v", u)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Names identify integrations and switches within a team.
|
||||||
|
func TestTeamNames_AreUniquePerTeam(t *testing.T) {
|
||||||
|
s := newTS(t) // creates one integration named "test" in the default team
|
||||||
|
|
||||||
|
dup := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/integrations", map[string]string{"name": "test"})
|
||||||
|
dup.Body.Close()
|
||||||
|
if dup.StatusCode != http.StatusConflict {
|
||||||
|
t.Errorf("duplicate integration name: expected 409, got %d", dup.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
body := map[string]any{"matcher": "alertname=Watchdog", "timeout_seconds": 60, "severity": "critical"}
|
||||||
|
first := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches", body)
|
||||||
|
first.Body.Close()
|
||||||
|
second := s.req(t, http.MethodPost, "/api/teams/"+defaultTeam+"/deadman/switches", body)
|
||||||
|
second.Body.Close()
|
||||||
|
if first.StatusCode != http.StatusCreated || second.StatusCode != http.StatusConflict {
|
||||||
|
t.Errorf("duplicate switch name: expected 201 then 409, got %d then %d", first.StatusCode, second.StatusCode)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The same name in another team is fine.
|
||||||
|
other := newTeam(t, s, "elsewhere")
|
||||||
|
ok := s.req(t, http.MethodPost, "/api/teams/"+id64(other.id)+"/integrations", map[string]string{"name": "test"})
|
||||||
|
ok.Body.Close()
|
||||||
|
if ok.StatusCode != http.StatusCreated {
|
||||||
|
t.Errorf("same name in another team: expected 201, got %d", ok.StatusCode)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -13,9 +13,8 @@ import (
|
|||||||
"git.ryuvia.com/niklas/terdut-server/internal/db"
|
"git.ryuvia.com/niklas/terdut-server/internal/db"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Tests run against a real Postgres, because the server does. SQLite's
|
// Tests run against a real Postgres, because the server does. Isolation is
|
||||||
// ":memory:" gave every test a private database for free; Postgres has no
|
// bought with a schema per test.
|
||||||
// equivalent, so isolation is bought with a schema per test.
|
|
||||||
//
|
//
|
||||||
// A schema rather than a database: CREATE DATABASE copies a template on disk and
|
// A schema rather than a database: CREATE DATABASE copies a template on disk and
|
||||||
// costs a hundred milliseconds or so each time, while CREATE SCHEMA plus the one
|
// costs a hundred milliseconds or so each time, while CREATE SCHEMA plus the one
|
||||||
|
|||||||
@@ -15,6 +15,10 @@ import (
|
|||||||
"github.com/go-chi/chi/v5"
|
"github.com/go-chi/chi/v5"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// bootstrapLockKey is the transaction-scoped advisory lock handleBootstrap
|
||||||
|
// holds; distinct from the migration and notifier keys.
|
||||||
|
const bootstrapLockKey = 0x7465726475744254
|
||||||
|
|
||||||
func handleBootstrap(db *sql.DB) http.HandlerFunc {
|
func handleBootstrap(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
var req struct {
|
var req struct {
|
||||||
@@ -40,15 +44,30 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
h, err := hashPassword(req.Password)
|
h, err := hashPassword(req.Password)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
passwordHash = &h
|
passwordHash = &h
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Check-then-insert has to be one atomic step: two concurrent calls on
|
||||||
|
// an empty install would otherwise both see zero users and both create
|
||||||
|
// an admin. The transaction-scoped lock serialises them, and the loser
|
||||||
|
// sees the winner's row.
|
||||||
|
tx, err := db.BeginTx(r.Context(), nil)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer tx.Rollback() //nolint:errcheck
|
||||||
|
if _, err := tx.ExecContext(r.Context(), "SELECT pg_advisory_xact_lock($1)", bootstrapLockKey); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
var count int
|
var count int
|
||||||
if err := db.QueryRowContext(r.Context(), "SELECT COUNT(*) FROM users").Scan(&count); err != nil {
|
if err := tx.QueryRowContext(r.Context(), "SELECT COUNT(*) FROM users").Scan(&count); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if count > 0 {
|
if count > 0 {
|
||||||
@@ -57,34 +76,29 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
|
|
||||||
var userID int64
|
var userID int64
|
||||||
if err := db.QueryRowContext(r.Context(),
|
if err := tx.QueryRowContext(r.Context(),
|
||||||
"INSERT INTO users (username, email, password_hash, is_admin) VALUES ($1, $2, $3, true) RETURNING id",
|
"INSERT INTO users (username, email, password_hash, is_admin) VALUES ($1, $2, $3, true) RETURNING id",
|
||||||
req.Username, req.Email, passwordHash).Scan(&userID); err != nil {
|
req.Username, req.Email, passwordHash).Scan(&userID); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
raw, hash, err := randomToken()
|
raw, hash, err := randomToken()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
var keyID int64
|
var keyID int64
|
||||||
if err := db.QueryRowContext(r.Context(),
|
if err := tx.QueryRowContext(r.Context(),
|
||||||
"INSERT INTO api_keys (user_id, key_hash, name) VALUES ($1, $2, $3) RETURNING id",
|
"INSERT INTO api_keys (user_id, key_hash, name) VALUES ($1, $2, $3) RETURNING id",
|
||||||
userID, hash, "bootstrap").Scan(&keyID); err != nil {
|
userID, hash, "bootstrap").Scan(&keyID); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
// The default team exists from migration 003, on a fresh install too.
|
if err := tx.Commit(); err != nil {
|
||||||
// Without a membership the first user signs in to a working server with
|
serverError(w, r, err)
|
||||||
// no queue, no schedule and nowhere for an integration to hang off.
|
return
|
||||||
if teamID, err := defaultTeamID(r.Context(), db); err == nil {
|
|
||||||
db.ExecContext(r.Context(), //nolint:errcheck
|
|
||||||
"INSERT INTO team_members (team_id, user_id, role) VALUES ($1, $2, $3) "+
|
|
||||||
"ON CONFLICT (team_id, user_id) DO NOTHING",
|
|
||||||
teamID, userID, models.RoleOwner)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
user, _ := fetchUser(r.Context(), db, userID)
|
user, _ := fetchUser(r.Context(), db, userID)
|
||||||
@@ -93,12 +107,19 @@ func handleBootstrap(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// handleListUsers is readable by anyone signed in, because the assignment
|
||||||
|
// control and the schedule need to name people. What it returns about other
|
||||||
|
// people is therefore only what naming them takes: email and ntfy_topic are
|
||||||
|
// blanked unless the caller is an admin or the row is their own. The topic in
|
||||||
|
// particular is a publish secret.
|
||||||
func handleListUsers(db *sql.DB) http.HandlerFunc {
|
func handleListUsers(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
caller, _ := userFromContext(r.Context())
|
||||||
|
seeAll := caller.IsAdmin
|
||||||
rows, err := db.QueryContext(r.Context(),
|
rows, err := db.QueryContext(r.Context(),
|
||||||
"SELECT id, username, email, created_at, ntfy_topic, is_admin, disabled_at FROM users ORDER BY id")
|
"SELECT id, username, email, created_at, ntfy_topic, is_admin, admin_source, disabled_at FROM users ORDER BY id")
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
defer rows.Close()
|
defer rows.Close()
|
||||||
@@ -108,14 +129,22 @@ func handleListUsers(db *sql.DB) http.HandlerFunc {
|
|||||||
var u models.User
|
var u models.User
|
||||||
var ts int64
|
var ts int64
|
||||||
var disabled *int64
|
var disabled *int64
|
||||||
if err := rows.Scan(&u.ID, &u.Username, &u.Email, &ts, &u.NtfyTopic, &u.IsAdmin, &disabled); err != nil {
|
if err := rows.Scan(&u.ID, &u.Username, &u.Email, &ts, &u.NtfyTopic, &u.IsAdmin, &u.AdminSource, &disabled); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
u.CreatedAt = time.Unix(ts, 0).UTC()
|
u.CreatedAt = time.Unix(ts, 0).UTC()
|
||||||
u.DisabledAt = unixPtr(disabled)
|
u.DisabledAt = unixPtr(disabled)
|
||||||
|
if !seeAll && u.ID != caller.ID {
|
||||||
|
u.Email = ""
|
||||||
|
u.NtfyTopic = nil
|
||||||
|
}
|
||||||
users = append(users, u)
|
users = append(users, u)
|
||||||
}
|
}
|
||||||
|
if err := rows.Err(); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
respond(w, http.StatusOK, users)
|
respond(w, http.StatusOK, users)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -143,7 +172,7 @@ func handleCreateUser(db *sql.DB) http.HandlerFunc {
|
|||||||
respond(w, http.StatusConflict, errResp("username or email already exists"))
|
respond(w, http.StatusConflict, errResp("username or email already exists"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
user, _ := fetchUser(r.Context(), db, id)
|
user, _ := fetchUser(r.Context(), db, id)
|
||||||
@@ -181,7 +210,7 @@ func handleSetNotifyTarget(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"UPDATE users SET ntfy_topic = $1 WHERE id = $2", topic, id)
|
"UPDATE users SET ntfy_topic = $1 WHERE id = $2", topic, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -191,7 +220,7 @@ func handleSetNotifyTarget(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
user, err := fetchUser(r.Context(), db, id)
|
user, err := fetchUser(r.Context(), db, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, user)
|
respond(w, http.StatusOK, user)
|
||||||
@@ -213,7 +242,7 @@ func handleDeleteUser(db *sql.DB) http.HandlerFunc {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
if last, err := isLastAdmin(r.Context(), db, id); err != nil {
|
if last, err := isLastAdmin(r.Context(), db, id); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
} else if last {
|
} else if last {
|
||||||
respond(w, http.StatusConflict, errResp("cannot delete the last administrator"))
|
respond(w, http.StatusConflict, errResp("cannot delete the last administrator"))
|
||||||
@@ -222,7 +251,7 @@ func handleDeleteUser(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
res, err := db.ExecContext(r.Context(), "DELETE FROM users WHERE id = $1", id)
|
res, err := db.ExecContext(r.Context(), "DELETE FROM users WHERE id = $1", id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
n, _ := res.RowsAffected()
|
n, _ := res.RowsAffected()
|
||||||
@@ -234,6 +263,11 @@ func handleDeleteUser(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// maxAPIKeyExpiryDays bounds expires_in_days: generous enough for any real
|
||||||
|
// rotation policy, tight enough to reject a typo (a year in hours, say) that
|
||||||
|
// would otherwise mint a key that outlives the server by decades.
|
||||||
|
const maxAPIKeyExpiryDays = 3650 // ~10 years
|
||||||
|
|
||||||
func handleCreateAPIKey(db *sql.DB) http.HandlerFunc {
|
func handleCreateAPIKey(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
userID, err := strconv.ParseInt(chi.URLParam(r, "id"), 10, 64)
|
userID, err := strconv.ParseInt(chi.URLParam(r, "id"), 10, 64)
|
||||||
@@ -247,6 +281,11 @@ func handleCreateAPIKey(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
var req struct {
|
var req struct {
|
||||||
Name string `json:"name"`
|
Name string `json:"name"`
|
||||||
|
// ExpiresInDays is optional and, left zero, means the key never
|
||||||
|
// expires — the only behavior any key had before this field
|
||||||
|
// existed, so an existing integration that does not send it is
|
||||||
|
// unaffected.
|
||||||
|
ExpiresInDays int64 `json:"expires_in_days,omitempty"`
|
||||||
}
|
}
|
||||||
if err := decodeJSON(r, &req); err != nil {
|
if err := decodeJSON(r, &req); err != nil {
|
||||||
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
respond(w, http.StatusBadRequest, errResp("invalid request body"))
|
||||||
@@ -256,6 +295,10 @@ func handleCreateAPIKey(db *sql.DB) http.HandlerFunc {
|
|||||||
respond(w, http.StatusBadRequest, errResp("name is required"))
|
respond(w, http.StatusBadRequest, errResp("name is required"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
if req.ExpiresInDays < 0 || req.ExpiresInDays > maxAPIKeyExpiryDays {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("expires_in_days must be 0 (never expires) or up to "+strconv.Itoa(maxAPIKeyExpiryDays)))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
var exists int
|
var exists int
|
||||||
if err := db.QueryRowContext(r.Context(), "SELECT 1 FROM users WHERE id = $1", userID).Scan(&exists); err != nil {
|
if err := db.QueryRowContext(r.Context(), "SELECT 1 FROM users WHERE id = $1", userID).Scan(&exists); err != nil {
|
||||||
@@ -265,21 +308,80 @@ func handleCreateAPIKey(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
raw, hash, err := randomToken()
|
raw, hash, err := randomToken()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
var expiresAt *int64
|
||||||
|
var expiresAtTime *time.Time
|
||||||
|
if req.ExpiresInDays > 0 {
|
||||||
|
t := time.Now().AddDate(0, 0, int(req.ExpiresInDays)).UTC()
|
||||||
|
u := t.Unix()
|
||||||
|
expiresAt = &u
|
||||||
|
expiresAtTime = &t
|
||||||
|
}
|
||||||
var keyID int64
|
var keyID int64
|
||||||
if err := db.QueryRowContext(r.Context(),
|
if err := db.QueryRowContext(r.Context(),
|
||||||
"INSERT INTO api_keys (user_id, key_hash, name) VALUES ($1, $2, $3) RETURNING id",
|
"INSERT INTO api_keys (user_id, key_hash, name, expires_at) VALUES ($1, $2, $3, $4) RETURNING id",
|
||||||
userID, hash, req.Name).Scan(&keyID); err != nil {
|
userID, hash, req.Name, expiresAt).Scan(&keyID); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
key := models.APIKey{ID: keyID, UserID: userID, Name: req.Name, Key: raw, CreatedAt: time.Now().UTC()}
|
key := models.APIKey{
|
||||||
|
ID: keyID, UserID: userID, Name: req.Name, Key: raw,
|
||||||
|
CreatedAt: time.Now().UTC(), ExpiresAt: expiresAtTime,
|
||||||
|
}
|
||||||
respond(w, http.StatusCreated, key)
|
respond(w, http.StatusCreated, key)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// handleListAPIKeys lists a user's own API keys: never the raw key itself
|
||||||
|
// (only ever returned once, at creation), just enough to tell them apart,
|
||||||
|
// see which are stale (last_used_at) and which are about to stop working
|
||||||
|
// (expires_at) — the data handleCreateAPIKey and apiKeyUser's last-use stamp
|
||||||
|
// already produce, with no endpoint to read it back until now.
|
||||||
|
func handleListAPIKeys(db *sql.DB) http.HandlerFunc {
|
||||||
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
userID, err := strconv.ParseInt(chi.URLParam(r, "id"), 10, 64)
|
||||||
|
if err != nil {
|
||||||
|
respond(w, http.StatusBadRequest, errResp("invalid user id"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !requireSelfOrAdmin(w, r, userID) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
rows, err := db.QueryContext(r.Context(),
|
||||||
|
`SELECT id, name, created_at, last_used_at, expires_at
|
||||||
|
FROM api_keys WHERE user_id = $1 ORDER BY created_at DESC`, userID)
|
||||||
|
if err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer rows.Close()
|
||||||
|
|
||||||
|
keys := []models.APIKey{}
|
||||||
|
for rows.Next() {
|
||||||
|
var k models.APIKey
|
||||||
|
var created int64
|
||||||
|
var lastUsed, expires *int64
|
||||||
|
if err := rows.Scan(&k.ID, &k.Name, &created, &lastUsed, &expires); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
k.UserID = userID
|
||||||
|
k.CreatedAt = time.Unix(created, 0).UTC()
|
||||||
|
k.LastUsedAt = unixPtr(lastUsed)
|
||||||
|
k.ExpiresAt = unixPtr(expires)
|
||||||
|
keys = append(keys, k)
|
||||||
|
}
|
||||||
|
if err := rows.Err(); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
respond(w, http.StatusOK, keys)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func handleDeleteAPIKey(db *sql.DB) http.HandlerFunc {
|
func handleDeleteAPIKey(db *sql.DB) http.HandlerFunc {
|
||||||
return func(w http.ResponseWriter, r *http.Request) {
|
return func(w http.ResponseWriter, r *http.Request) {
|
||||||
userID, err := strconv.ParseInt(chi.URLParam(r, "id"), 10, 64)
|
userID, err := strconv.ParseInt(chi.URLParam(r, "id"), 10, 64)
|
||||||
@@ -299,7 +401,7 @@ func handleDeleteAPIKey(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"DELETE FROM api_keys WHERE id = $1 AND user_id = $2", keyID, userID)
|
"DELETE FROM api_keys WHERE id = $1 AND user_id = $2", keyID, userID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
n, _ := res.RowsAffected()
|
n, _ := res.RowsAffected()
|
||||||
@@ -329,8 +431,8 @@ func fetchUser(ctx context.Context, db *sql.DB, id int64) (models.User, error) {
|
|||||||
var ts int64
|
var ts int64
|
||||||
var disabled *int64
|
var disabled *int64
|
||||||
err := db.QueryRowContext(ctx,
|
err := db.QueryRowContext(ctx,
|
||||||
"SELECT id, username, email, created_at, ntfy_topic, is_admin, disabled_at FROM users WHERE id = $1", id).
|
"SELECT id, username, email, created_at, ntfy_topic, is_admin, admin_source, disabled_at FROM users WHERE id = $1", id).
|
||||||
Scan(&u.ID, &u.Username, &u.Email, &ts, &u.NtfyTopic, &u.IsAdmin, &disabled)
|
Scan(&u.ID, &u.Username, &u.Email, &ts, &u.NtfyTopic, &u.IsAdmin, &u.AdminSource, &disabled)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return u, err
|
return u, err
|
||||||
}
|
}
|
||||||
@@ -361,13 +463,25 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc {
|
|||||||
}
|
}
|
||||||
|
|
||||||
if !*req.IsAdmin {
|
if !*req.IsAdmin {
|
||||||
|
var managed bool
|
||||||
|
if err := db.QueryRowContext(r.Context(),
|
||||||
|
"SELECT EXISTS (SELECT 1 FROM users WHERE id = $1 AND is_admin AND admin_source = 'oidc')",
|
||||||
|
id).Scan(&managed); err != nil {
|
||||||
|
serverError(w, r, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if managed {
|
||||||
|
respond(w, http.StatusConflict, errResp("administrator access is managed by single sign-on; change the user's groups in the identity provider"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
caller, _ := userFromContext(r.Context())
|
caller, _ := userFromContext(r.Context())
|
||||||
if caller.ID == id {
|
if caller.ID == id {
|
||||||
respond(w, http.StatusConflict, errResp("cannot revoke your own administrator access"))
|
respond(w, http.StatusConflict, errResp("cannot revoke your own administrator access"))
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if last, err := isLastAdmin(r.Context(), db, id); err != nil {
|
if last, err := isLastAdmin(r.Context(), db, id); err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
} else if last {
|
} else if last {
|
||||||
respond(w, http.StatusConflict, errResp("cannot revoke the last administrator"))
|
respond(w, http.StatusConflict, errResp("cannot revoke the last administrator"))
|
||||||
@@ -378,7 +492,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc {
|
|||||||
res, err := db.ExecContext(r.Context(),
|
res, err := db.ExecContext(r.Context(),
|
||||||
"UPDATE users SET is_admin = $1 WHERE id = $2", *req.IsAdmin, id)
|
"UPDATE users SET is_admin = $1 WHERE id = $2", *req.IsAdmin, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
if n, _ := res.RowsAffected(); n == 0 {
|
if n, _ := res.RowsAffected(); n == 0 {
|
||||||
@@ -388,7 +502,7 @@ func handleSetAdmin(db *sql.DB) http.HandlerFunc {
|
|||||||
|
|
||||||
user, err := fetchUser(r.Context(), db, id)
|
user, err := fetchUser(r.Context(), db, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
respond(w, http.StatusInternalServerError, errResp("internal error"))
|
serverError(w, r, err)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
respond(w, http.StatusOK, user)
|
respond(w, http.StatusOK, user)
|
||||||
|
|||||||
@@ -1,17 +1,25 @@
|
|||||||
package config
|
package config
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"net/url"
|
||||||
"os"
|
"os"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
"time"
|
"time"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// MinOperatorKeyLength is the shortest TERDUT_OPERATOR_KEY accepted: it is a
|
||||||
|
// bearer credential with instance reach, so a short one is refused outright.
|
||||||
|
const MinOperatorKeyLength = 32
|
||||||
|
|
||||||
type Config struct {
|
type Config struct {
|
||||||
Addr string
|
Addr string
|
||||||
|
|
||||||
// DSN is the Postgres connection string, e.g.
|
// DSN is the Postgres connection string, e.g.
|
||||||
// postgres://terdut:secret@host:5432/terdut?sslmode=require. Required:
|
// postgres://terdut:secret@host:5432/terdut?sslmode=require. Required:
|
||||||
// unlike the SQLite path it replaced there is no sensible default, and a
|
// there is no sensible default, and a server that silently came up against the wrong database would be worse
|
||||||
// server that silently came up against the wrong database would be worse
|
|
||||||
// than one that refuses to start.
|
// than one that refuses to start.
|
||||||
DSN string
|
DSN string
|
||||||
|
|
||||||
@@ -22,25 +30,6 @@ type Config struct {
|
|||||||
// repeat_interval (default 4h), which is what refreshes the alert.
|
// repeat_interval (default 4h), which is what refreshes the alert.
|
||||||
StaleAfter time.Duration
|
StaleAfter time.Duration
|
||||||
|
|
||||||
// DeadmanMatchers selects the alerts that are heartbeats rather than
|
|
||||||
// problems: receiving one opens no incident, and the absence of one does.
|
|
||||||
//
|
|
||||||
// ";" separates matchers, "," the label conditions within one, "=" is exact
|
|
||||||
// equality — `alertname=Watchdog,cluster=prod; alertname=Heartbeat`. Every
|
|
||||||
// matcher must name an alertname. See api.ParseDeadmanConfig.
|
|
||||||
DeadmanMatchers string
|
|
||||||
|
|
||||||
// DeadmanTimeout is how long a heartbeat may go unheard before its switch is
|
|
||||||
// declared dead. It must be *shorter* than the Alertmanager repeat_interval
|
|
||||||
// of the route carrying the heartbeat — the opposite of StaleAfter, and the
|
|
||||||
// reason a dead man's switch usually wants a route of its own. Zero disables
|
|
||||||
// dead man's switch handling entirely.
|
|
||||||
DeadmanTimeout time.Duration
|
|
||||||
|
|
||||||
// DeadmanSeverity is the severity a dead man's switch incident opens at.
|
|
||||||
// These incidents have no member alerts to derive one from.
|
|
||||||
DeadmanSeverity string
|
|
||||||
|
|
||||||
// NtfyURL is the ntfy server push notifications are published to. Empty
|
// NtfyURL is the ntfy server push notifications are published to. Empty
|
||||||
// disables notifications entirely.
|
// disables notifications entirely.
|
||||||
NtfyURL string
|
NtfyURL string
|
||||||
@@ -59,39 +48,209 @@ type Config struct {
|
|||||||
// NotifyRepeat is how long an incident may sit unacknowledged before it is
|
// NotifyRepeat is how long an incident may sit unacknowledged before it is
|
||||||
// notified again. Zero disables reminders.
|
// notified again. Zero disables reminders.
|
||||||
NotifyRepeat time.Duration
|
NotifyRepeat time.Duration
|
||||||
|
|
||||||
|
// DisablePasswordLogin refuses signing in, or signing up, with a password.
|
||||||
|
// It is how an install moves to SSO only, and turning it back off is the way
|
||||||
|
// in when the identity provider is down. Stated negatively so that the zero
|
||||||
|
// Config, which is what a test or a new caller builds, keeps passwords working.
|
||||||
|
DisablePasswordLogin bool
|
||||||
|
|
||||||
|
// OperatorKey, when set, is the credential of the instance-scoped service
|
||||||
|
// account "terdut-operator", created or re-keyed at every start. It is how
|
||||||
|
// terdut-operator gets in without a bootstrap handshake: the operator
|
||||||
|
// generates the key, hands it to the server here, and uses it as its bearer
|
||||||
|
// token. Empty means no such account is managed.
|
||||||
|
OperatorKey string
|
||||||
|
|
||||||
|
// TrustedProxies is how many reverse proxies sit in front of the server and
|
||||||
|
// append to X-Forwarded-For. The per-address rate limits take the client
|
||||||
|
// address that many entries from the right, because everything further left
|
||||||
|
// is whatever the client chose to send. 0 ignores the header and uses the
|
||||||
|
// connection's own address.
|
||||||
|
TrustedProxies int
|
||||||
|
|
||||||
|
// OIDC configures single sign-on. The zero value, with no Issuer, is off.
|
||||||
|
OIDC OIDC
|
||||||
|
|
||||||
|
// OperatorMode declares this install gitops-managed: writes to teams,
|
||||||
|
// escalation policies, dead man's switches and integrations from a human
|
||||||
|
// (a session or a user's own API key) are refused, while a service
|
||||||
|
// account's are not. Deploy-time and restart-required, like the rest of
|
||||||
|
// "where this server is plugged in" — it is a statement about who owns
|
||||||
|
// this install's configuration, not a per-request toggle.
|
||||||
|
OperatorMode bool
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// OIDC is the single sign-on configuration. Groups from the provider decide
|
||||||
|
// who may sign in, which teams they belong to, and whether they administer the
|
||||||
|
// install, in the manner of Grafana's org and role mapping.
|
||||||
|
type OIDC struct {
|
||||||
|
// Issuer is the provider's issuer URL. Discovery is fetched from
|
||||||
|
// <Issuer>/.well-known/openid-configuration. For Authentik this is the
|
||||||
|
// application's issuer, e.g. https://auth.example.com/application/o/terdut/.
|
||||||
|
// Empty turns single sign-on off.
|
||||||
|
Issuer string
|
||||||
|
ClientID string
|
||||||
|
ClientSecret string
|
||||||
|
|
||||||
|
// Name is what the sign-in button calls the provider.
|
||||||
|
Name string
|
||||||
|
|
||||||
|
// Scopes to request. The groups claim normally needs "profile" on Authentik.
|
||||||
|
Scopes []string
|
||||||
|
|
||||||
|
// UsernameClaim, EmailClaim and GroupsClaim name the ID token claims read.
|
||||||
|
UsernameClaim string
|
||||||
|
EmailClaim string
|
||||||
|
GroupsClaim string
|
||||||
|
|
||||||
|
// TrustEmail links a sign-in to an existing local user by email even when the
|
||||||
|
// provider does not vouch that the address is verified. Authentik reports
|
||||||
|
// email_verified false unless told otherwise, and an install that runs its
|
||||||
|
// own provider has already decided that its addresses can be trusted.
|
||||||
|
TrustEmail bool
|
||||||
|
|
||||||
|
// AllowedGroups gates sign-in: somebody in none of them is refused, however
|
||||||
|
// well the provider authenticated them. Empty admits everybody the provider
|
||||||
|
// authenticates, and access control is left to the provider.
|
||||||
|
AllowedGroups []string
|
||||||
|
|
||||||
|
// AdminGroup grants the system administrator flag while the user is in it.
|
||||||
|
AdminGroup string
|
||||||
|
|
||||||
|
// SessionMaxAge is the hard ceiling on a session made by an SSO login. The
|
||||||
|
// login is the only moment groups are re-read, so this is how long a change
|
||||||
|
// in the provider may take to reach terdut.
|
||||||
|
SessionMaxAge time.Duration
|
||||||
|
}
|
||||||
|
|
||||||
|
// Enabled reports whether single sign-on is configured.
|
||||||
|
func (o OIDC) Enabled() bool { return o.Issuer != "" }
|
||||||
|
|
||||||
func Load() Config {
|
func Load() Config {
|
||||||
addr := os.Getenv("TERDUT_ADDR")
|
addr := os.Getenv("TERDUT_ADDR")
|
||||||
if addr == "" {
|
if addr == "" {
|
||||||
addr = ":8080"
|
addr = ":8080"
|
||||||
}
|
}
|
||||||
deadmanMatchers := os.Getenv("TERDUT_DEADMAN_MATCHERS")
|
|
||||||
if deadmanMatchers == "" {
|
|
||||||
deadmanMatchers = "alertname=Watchdog"
|
|
||||||
}
|
|
||||||
deadmanSeverity := os.Getenv("TERDUT_DEADMAN_SEVERITY")
|
|
||||||
if deadmanSeverity == "" {
|
|
||||||
deadmanSeverity = "critical"
|
|
||||||
}
|
|
||||||
return Config{
|
return Config{
|
||||||
Addr: addr,
|
Addr: addr,
|
||||||
DSN: os.Getenv("TERDUT_DB_DSN"),
|
DSN: os.Getenv("TERDUT_DB_DSN"),
|
||||||
ArchiveAfter: duration("TERDUT_ARCHIVE_AFTER", 7*24*time.Hour),
|
ArchiveAfter: duration("TERDUT_ARCHIVE_AFTER", 7*24*time.Hour),
|
||||||
StaleAfter: duration("TERDUT_STALE_AFTER", 6*time.Hour),
|
StaleAfter: duration("TERDUT_STALE_AFTER", 6*time.Hour),
|
||||||
|
|
||||||
DeadmanMatchers: deadmanMatchers,
|
|
||||||
DeadmanTimeout: duration("TERDUT_DEADMAN_TIMEOUT", 15*time.Minute),
|
|
||||||
DeadmanSeverity: deadmanSeverity,
|
|
||||||
|
|
||||||
NtfyURL: os.Getenv("TERDUT_NTFY_URL"),
|
NtfyURL: os.Getenv("TERDUT_NTFY_URL"),
|
||||||
NtfyToken: os.Getenv("TERDUT_NTFY_TOKEN"),
|
NtfyToken: os.Getenv("TERDUT_NTFY_TOKEN"),
|
||||||
NtfyFallbackTopic: os.Getenv("TERDUT_NTFY_FALLBACK_TOPIC"),
|
NtfyFallbackTopic: os.Getenv("TERDUT_NTFY_FALLBACK_TOPIC"),
|
||||||
PublicURL: os.Getenv("TERDUT_PUBLIC_URL"),
|
PublicURL: os.Getenv("TERDUT_PUBLIC_URL"),
|
||||||
NotifyRepeat: duration("TERDUT_NOTIFY_REPEAT", 15*time.Minute),
|
NotifyRepeat: duration("TERDUT_NOTIFY_REPEAT", 15*time.Minute),
|
||||||
|
|
||||||
|
DisablePasswordLogin: !boolean("TERDUT_PASSWORD_LOGIN", true),
|
||||||
|
|
||||||
|
OperatorKey: strings.TrimSpace(os.Getenv("TERDUT_OPERATOR_KEY")),
|
||||||
|
TrustedProxies: integer("TERDUT_TRUSTED_PROXIES", 1),
|
||||||
|
OIDC: loadOIDC(),
|
||||||
|
|
||||||
|
OperatorMode: boolean("TERDUT_OPERATOR_MODE", false),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func loadOIDC() OIDC {
|
||||||
|
o := OIDC{
|
||||||
|
Issuer: strings.TrimSpace(os.Getenv("TERDUT_OIDC_ISSUER")),
|
||||||
|
ClientID: os.Getenv("TERDUT_OIDC_CLIENT_ID"),
|
||||||
|
ClientSecret: os.Getenv("TERDUT_OIDC_CLIENT_SECRET"),
|
||||||
|
Name: str("TERDUT_OIDC_NAME", "SSO"),
|
||||||
|
Scopes: list("TERDUT_OIDC_SCOPES", "openid profile email"),
|
||||||
|
UsernameClaim: str("TERDUT_OIDC_USERNAME_CLAIM", "preferred_username"),
|
||||||
|
EmailClaim: str("TERDUT_OIDC_EMAIL_CLAIM", "email"),
|
||||||
|
GroupsClaim: str("TERDUT_OIDC_GROUPS_CLAIM", "groups"),
|
||||||
|
TrustEmail: boolean("TERDUT_OIDC_TRUST_EMAIL", false),
|
||||||
|
AllowedGroups: list("TERDUT_OIDC_ALLOWED_GROUPS", ""),
|
||||||
|
AdminGroup: os.Getenv("TERDUT_OIDC_ADMIN_GROUP"),
|
||||||
|
SessionMaxAge: duration("TERDUT_OIDC_SESSION_MAX_AGE", 12*time.Hour),
|
||||||
|
}
|
||||||
|
return o
|
||||||
|
}
|
||||||
|
|
||||||
|
// Validate reports a configuration the server should refuse to start with.
|
||||||
|
// Single sign-on is the only part that can be inconsistent: a half-configured
|
||||||
|
// provider would come up and then fail every login, which is harder to notice
|
||||||
|
// than not starting.
|
||||||
|
func (c Config) Validate() error {
|
||||||
|
if c.OperatorKey != "" && len(c.OperatorKey) < MinOperatorKeyLength {
|
||||||
|
return fmt.Errorf("TERDUT_OPERATOR_KEY must be at least %d characters", MinOperatorKeyLength)
|
||||||
|
}
|
||||||
|
o := c.OIDC
|
||||||
|
if !o.Enabled() {
|
||||||
|
if c.DisablePasswordLogin {
|
||||||
|
return errors.New("TERDUT_PASSWORD_LOGIN=false without TERDUT_OIDC_ISSUER leaves no way to sign in")
|
||||||
|
}
|
||||||
|
if o.AdminGroup != "" || len(o.AllowedGroups) > 0 {
|
||||||
|
return errors.New("TERDUT_OIDC_* group settings are set but TERDUT_OIDC_ISSUER is not")
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if u, err := url.Parse(o.Issuer); err != nil || u.Scheme == "" || u.Host == "" {
|
||||||
|
return fmt.Errorf("TERDUT_OIDC_ISSUER %q is not a URL", o.Issuer)
|
||||||
|
}
|
||||||
|
if o.ClientID == "" || o.ClientSecret == "" {
|
||||||
|
return errors.New("TERDUT_OIDC_CLIENT_ID and TERDUT_OIDC_CLIENT_SECRET are required with TERDUT_OIDC_ISSUER")
|
||||||
|
}
|
||||||
|
if c.PublicURL == "" {
|
||||||
|
return errors.New("TERDUT_PUBLIC_URL is required with TERDUT_OIDC_ISSUER: it is the base of the redirect URI")
|
||||||
|
}
|
||||||
|
if o.SessionMaxAge <= 0 {
|
||||||
|
return errors.New("TERDUT_OIDC_SESSION_MAX_AGE must be positive")
|
||||||
|
}
|
||||||
|
// Team grants are no longer visible here: they live on each team's own
|
||||||
|
// oidc_member_group/oidc_owner_group columns, set by that team's owner, not
|
||||||
|
// in config Validate can see at startup. The one thing left to guard against
|
||||||
|
// is an install nobody can administer at all.
|
||||||
|
if c.DisablePasswordLogin && o.AdminGroup == "" {
|
||||||
|
return errors.New("TERDUT_PASSWORD_LOGIN=false with no TERDUT_OIDC_ADMIN_GROUP leaves nobody able to administer the install")
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func str(env, def string) string {
|
||||||
|
if s := strings.TrimSpace(os.Getenv(env)); s != "" {
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
return def
|
||||||
|
}
|
||||||
|
|
||||||
|
// integer reads a non-negative int env var; anything else takes the default.
|
||||||
|
func integer(env string, def int) int {
|
||||||
|
if s := os.Getenv(env); s != "" {
|
||||||
|
if n, err := strconv.Atoi(strings.TrimSpace(s)); err == nil && n >= 0 {
|
||||||
|
return n
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return def
|
||||||
|
}
|
||||||
|
|
||||||
|
// list reads a comma- or space-separated env var.
|
||||||
|
func list(env, def string) []string {
|
||||||
|
s := os.Getenv(env)
|
||||||
|
if strings.TrimSpace(s) == "" {
|
||||||
|
s = def
|
||||||
|
}
|
||||||
|
return strings.FieldsFunc(s, func(r rune) bool { return r == ',' || r == ' ' })
|
||||||
|
}
|
||||||
|
|
||||||
|
// boolean reads a true/false env var. An unrecognised value takes the default,
|
||||||
|
// so the two flags read this way (password login on, trusting email off) both
|
||||||
|
// fail towards the cautious setting.
|
||||||
|
func boolean(env string, def bool) bool {
|
||||||
|
switch strings.ToLower(strings.TrimSpace(os.Getenv(env))) {
|
||||||
|
case "true", "1", "yes":
|
||||||
|
return true
|
||||||
|
case "false", "0", "no":
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
return def
|
||||||
|
}
|
||||||
|
|
||||||
// duration reads a time.ParseDuration-formatted env var. An unset or
|
// duration reads a time.ParseDuration-formatted env var. An unset or
|
||||||
// unparseable value falls back to def rather than failing startup: a typo in one
|
// unparseable value falls back to def rather than failing startup: a typo in one
|
||||||
// tuning knob should not take the server down.
|
// tuning knob should not take the server down.
|
||||||
|
|||||||
@@ -0,0 +1,79 @@
|
|||||||
|
package config
|
||||||
|
|
||||||
|
import (
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestValidate(t *testing.T) {
|
||||||
|
base := func() map[string]string {
|
||||||
|
return map[string]string{
|
||||||
|
"TERDUT_PUBLIC_URL": "https://terdut.example.com",
|
||||||
|
"TERDUT_OIDC_ISSUER": "https://auth.example.com/application/o/terdut/",
|
||||||
|
"TERDUT_OIDC_CLIENT_ID": "id",
|
||||||
|
"TERDUT_OIDC_CLIENT_SECRET": "secret",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
env func(map[string]string)
|
||||||
|
wantErr string // substring; empty means valid
|
||||||
|
}{
|
||||||
|
{"off by default", func(m map[string]string) { clear(m) }, ""},
|
||||||
|
{"minimal sso", func(m map[string]string) {}, ""},
|
||||||
|
{"groups without issuer", func(m map[string]string) {
|
||||||
|
clear(m)
|
||||||
|
m["TERDUT_OIDC_ADMIN_GROUP"] = "admins"
|
||||||
|
}, "ISSUER is not"},
|
||||||
|
{"missing secret", func(m map[string]string) { delete(m, "TERDUT_OIDC_CLIENT_SECRET") }, "CLIENT_SECRET"},
|
||||||
|
{"missing public url", func(m map[string]string) { delete(m, "TERDUT_PUBLIC_URL") }, "PUBLIC_URL"},
|
||||||
|
{"bad issuer", func(m map[string]string) { m["TERDUT_OIDC_ISSUER"] = "not a url" }, "not a URL"},
|
||||||
|
{"password off without sso", func(m map[string]string) {
|
||||||
|
clear(m)
|
||||||
|
m["TERDUT_PASSWORD_LOGIN"] = "false"
|
||||||
|
}, "no way to sign in"},
|
||||||
|
{"password off with sso but no grants", func(m map[string]string) {
|
||||||
|
m["TERDUT_PASSWORD_LOGIN"] = "false"
|
||||||
|
}, "nobody able"},
|
||||||
|
{"password off with admin group", func(m map[string]string) {
|
||||||
|
m["TERDUT_PASSWORD_LOGIN"] = "false"
|
||||||
|
m["TERDUT_OIDC_ADMIN_GROUP"] = "admins"
|
||||||
|
}, ""},
|
||||||
|
}
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
env := base()
|
||||||
|
tt.env(env)
|
||||||
|
for _, k := range []string{
|
||||||
|
"TERDUT_PUBLIC_URL", "TERDUT_PASSWORD_LOGIN", "TERDUT_OIDC_ISSUER", "TERDUT_OIDC_CLIENT_ID",
|
||||||
|
"TERDUT_OIDC_CLIENT_SECRET", "TERDUT_OIDC_ADMIN_GROUP",
|
||||||
|
} {
|
||||||
|
t.Setenv(k, env[k])
|
||||||
|
}
|
||||||
|
err := Load().Validate()
|
||||||
|
switch {
|
||||||
|
case tt.wantErr == "" && err != nil:
|
||||||
|
t.Errorf("unexpected error: %v", err)
|
||||||
|
case tt.wantErr != "" && (err == nil || !strings.Contains(err.Error(), tt.wantErr)):
|
||||||
|
t.Errorf("error %v, want one containing %q", err, tt.wantErr)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLoad_OIDCDefaults(t *testing.T) {
|
||||||
|
t.Setenv("TERDUT_OIDC_ISSUER", "https://auth.example.com/")
|
||||||
|
o := Load().OIDC
|
||||||
|
if o.UsernameClaim != "preferred_username" || o.EmailClaim != "email" || o.GroupsClaim != "groups" {
|
||||||
|
t.Errorf("claim defaults: %+v", o)
|
||||||
|
}
|
||||||
|
if strings.Join(o.Scopes, " ") != "openid profile email" {
|
||||||
|
t.Errorf("scopes: %v", o.Scopes)
|
||||||
|
}
|
||||||
|
if o.SessionMaxAge.Hours() != 12 {
|
||||||
|
t.Errorf("max age: %v", o.SessionMaxAge)
|
||||||
|
}
|
||||||
|
if Load().DisablePasswordLogin {
|
||||||
|
t.Error("password login should be on by default")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,10 +1,12 @@
|
|||||||
package db
|
package db
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"context"
|
||||||
"database/sql"
|
"database/sql"
|
||||||
"embed"
|
"embed"
|
||||||
"fmt"
|
"fmt"
|
||||||
"io/fs"
|
"io/fs"
|
||||||
|
"log"
|
||||||
"sort"
|
"sort"
|
||||||
"strings"
|
"strings"
|
||||||
"time"
|
"time"
|
||||||
@@ -15,13 +17,25 @@ import (
|
|||||||
//go:embed migrations
|
//go:embed migrations
|
||||||
var migrationsFS embed.FS
|
var migrationsFS embed.FS
|
||||||
|
|
||||||
|
// pingAttempts and pingRetryDelay bound the retry on the first connection.
|
||||||
|
// This pod's own IP can reach the Postgres pod's node before that node's
|
||||||
|
// NetworkPolicy enforcement (kube-router, reacting to the pod's creation
|
||||||
|
// event) has added it to the allowed-source set, which fails the ping with
|
||||||
|
// "connection refused" rather than a timeout. That race resolves within
|
||||||
|
// several seconds in practice; five attempts two seconds apart give it
|
||||||
|
// comfortable room without turning a genuinely absent database into a long
|
||||||
|
// hang.
|
||||||
|
const (
|
||||||
|
pingAttempts = 5
|
||||||
|
pingRetryDelay = 2 * time.Second
|
||||||
|
)
|
||||||
|
|
||||||
// Open connects to Postgres. dsn is a libpq connection string or URL, e.g.
|
// Open connects to Postgres. dsn is a libpq connection string or URL, e.g.
|
||||||
// postgres://terdut:secret@localhost:5432/terdut?sslmode=disable.
|
// postgres://terdut:secret@localhost:5432/terdut?sslmode=disable.
|
||||||
//
|
//
|
||||||
// The pool is modest on purpose: this server's concurrency comes from a handful
|
// The pool is modest on purpose: this server's concurrency comes from a handful
|
||||||
// of HTTP handlers plus two background loops, and a cloud-native-pg instance
|
// of HTTP handlers plus two background loops, and a cloud-native-pg instance
|
||||||
// sized for it has a low max_connections. It is still a pool, unlike the single
|
// sized for it has a low max_connections.
|
||||||
// connection SQLite forced, so the notifier no longer blocks a webhook.
|
|
||||||
func Open(dsn string) (*sql.DB, error) {
|
func Open(dsn string) (*sql.DB, error) {
|
||||||
if dsn == "" {
|
if dsn == "" {
|
||||||
return nil, fmt.Errorf("empty DSN: set TERDUT_DB_DSN")
|
return nil, fmt.Errorf("empty DSN: set TERDUT_DB_DSN")
|
||||||
@@ -33,20 +47,53 @@ func Open(dsn string) (*sql.DB, error) {
|
|||||||
db.SetMaxOpenConns(10)
|
db.SetMaxOpenConns(10)
|
||||||
db.SetMaxIdleConns(5)
|
db.SetMaxIdleConns(5)
|
||||||
db.SetConnMaxLifetime(time.Hour)
|
db.SetConnMaxLifetime(time.Hour)
|
||||||
if err := db.Ping(); err != nil {
|
|
||||||
db.Close()
|
for attempt := 1; ; attempt++ {
|
||||||
return nil, fmt.Errorf("ping: %w", err)
|
err = db.Ping()
|
||||||
|
if err == nil {
|
||||||
|
return db, nil
|
||||||
|
}
|
||||||
|
if attempt == pingAttempts {
|
||||||
|
db.Close()
|
||||||
|
return nil, fmt.Errorf("ping: %w", err)
|
||||||
|
}
|
||||||
|
log.Printf("open db: ping attempt %d/%d failed, retrying in %s: %v", attempt, pingAttempts, pingRetryDelay, err)
|
||||||
|
time.Sleep(pingRetryDelay)
|
||||||
}
|
}
|
||||||
return db, nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// migrationLockKey is the Postgres advisory lock Migrate holds for its whole
|
||||||
|
// run. Two replicas starting at once would otherwise race the check-then-apply
|
||||||
|
// loop below against schema_migrations: the loser could crash on a
|
||||||
|
// duplicate-key insert, or contend with the winner's uncommitted DDL. Blocking
|
||||||
|
// (pg_advisory_lock, not pg_try_advisory_lock as the archiver and notifier
|
||||||
|
// use): on boot there is no later tick to defer to, so the right behaviour is
|
||||||
|
// to wait for the other replica to finish migrating, not to skip ahead and
|
||||||
|
// start serving against an unmigrated schema.
|
||||||
|
const migrationLockKey int64 = 7265_0003
|
||||||
|
|
||||||
// Migrate applies every embedded migration that has not been applied yet, in
|
// Migrate applies every embedded migration that has not been applied yet, in
|
||||||
// filename order, recording each in schema_migrations.
|
// filename order, recording each in schema_migrations.
|
||||||
//
|
//
|
||||||
// Each file runs inside a transaction, which SQLite's version did not do: a
|
// Each file runs inside a transaction, so a migration that fails half way
|
||||||
// migration that failed half way used to leave the schema in whatever state it
|
// leaves the schema as it was: Postgres has transactional DDL.
|
||||||
// had reached. Postgres has transactional DDL, so the rollback is real.
|
|
||||||
func Migrate(db *sql.DB) error {
|
func Migrate(db *sql.DB) error {
|
||||||
|
ctx := context.Background()
|
||||||
|
conn, err := db.Conn(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("migrate: acquire connection: %w", err)
|
||||||
|
}
|
||||||
|
defer conn.Close()
|
||||||
|
|
||||||
|
if _, err := conn.ExecContext(ctx, "SELECT pg_advisory_lock($1)", migrationLockKey); err != nil {
|
||||||
|
return fmt.Errorf("migrate: acquire advisory lock: %w", err)
|
||||||
|
}
|
||||||
|
defer func() {
|
||||||
|
if _, err := conn.ExecContext(ctx, "SELECT pg_advisory_unlock($1)", migrationLockKey); err != nil {
|
||||||
|
log.Printf("migrate: release advisory lock: %v", err)
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
|
||||||
if _, err := db.Exec(`CREATE TABLE IF NOT EXISTS schema_migrations (
|
if _, err := db.Exec(`CREATE TABLE IF NOT EXISTS schema_migrations (
|
||||||
version TEXT PRIMARY KEY,
|
version TEXT PRIMARY KEY,
|
||||||
applied_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
applied_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
||||||
|
|||||||
@@ -0,0 +1,125 @@
|
|||||||
|
package db_test
|
||||||
|
|
||||||
|
import (
|
||||||
|
"database/sql"
|
||||||
|
"fmt"
|
||||||
|
"net/url"
|
||||||
|
"os"
|
||||||
|
"strings"
|
||||||
|
"sync"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"git.ryuvia.com/niklas/terdut-server/internal/db"
|
||||||
|
|
||||||
|
_ "github.com/jackc/pgx/v5/stdlib"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TERDUT_TEST_DSN must point at a database the test role may create schemas
|
||||||
|
// in; see internal/api/testdb_test.go for the fuller rationale this mirrors.
|
||||||
|
// An unset DSN fails rather than skips, deliberately.
|
||||||
|
const testDSNEnv = "TERDUT_TEST_DSN"
|
||||||
|
|
||||||
|
// TestMigrate_ConcurrentCallersDoNotRace reproduces two replicas starting at
|
||||||
|
// once against a brand-new, unmigrated schema: both call db.Migrate at the
|
||||||
|
// same time. Before migrationLockKey, the loser could crash on a
|
||||||
|
// duplicate-key insert into schema_migrations, or contend with the winner's
|
||||||
|
// uncommitted DDL; with the advisory lock, one blocks until the other
|
||||||
|
// finishes and both return cleanly.
|
||||||
|
func TestMigrate_ConcurrentCallersDoNotRace(t *testing.T) {
|
||||||
|
dsn := os.Getenv(testDSNEnv)
|
||||||
|
if dsn == "" {
|
||||||
|
t.Fatalf("%s is not set: these tests need Postgres.\n"+
|
||||||
|
"Run `make test-db` for a local one, then\n"+
|
||||||
|
" export %s=postgres://terdut:terdut@localhost:5432/terdut_test?sslmode=disable",
|
||||||
|
testDSNEnv, testDSNEnv)
|
||||||
|
}
|
||||||
|
|
||||||
|
schema := fmt.Sprintf("migrate_race_%d", os.Getpid())
|
||||||
|
admin, err := sql.Open("pgx", dsn)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("connect to %s: %v", testDSNEnv, err)
|
||||||
|
}
|
||||||
|
defer admin.Close()
|
||||||
|
if _, err := admin.Exec("CREATE SCHEMA " + schema); err != nil {
|
||||||
|
t.Fatalf("create schema %s: %v", schema, err)
|
||||||
|
}
|
||||||
|
t.Cleanup(func() {
|
||||||
|
cleanup, err := sql.Open("pgx", dsn)
|
||||||
|
if err != nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer cleanup.Close()
|
||||||
|
if _, err := cleanup.Exec("DROP SCHEMA " + schema + " CASCADE"); err != nil {
|
||||||
|
t.Logf("drop schema %s: %v", schema, err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
scoped := withSearchPath(dsn, schema)
|
||||||
|
|
||||||
|
const callers = 2
|
||||||
|
errs := make([]error, callers)
|
||||||
|
var wg sync.WaitGroup
|
||||||
|
for i := range callers {
|
||||||
|
wg.Add(1)
|
||||||
|
go func(i int) {
|
||||||
|
defer wg.Done()
|
||||||
|
database, err := db.Open(scoped)
|
||||||
|
if err != nil {
|
||||||
|
errs[i] = fmt.Errorf("open: %w", err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
defer database.Close()
|
||||||
|
errs[i] = db.Migrate(database)
|
||||||
|
}(i)
|
||||||
|
}
|
||||||
|
wg.Wait()
|
||||||
|
|
||||||
|
for i, err := range errs {
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Migrate #%d: %v", i, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
entries, err := os.ReadDir("migrations")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read migrations dir: %v", err)
|
||||||
|
}
|
||||||
|
var want int
|
||||||
|
for _, e := range entries {
|
||||||
|
if !e.IsDir() && strings.HasSuffix(e.Name(), ".sql") {
|
||||||
|
want++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
check, err := sql.Open("pgx", scoped)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("connect for verification: %v", err)
|
||||||
|
}
|
||||||
|
defer check.Close()
|
||||||
|
|
||||||
|
var got int
|
||||||
|
if err := check.QueryRow("SELECT COUNT(*) FROM schema_migrations").Scan(&got); err != nil {
|
||||||
|
t.Fatalf("count schema_migrations: %v", err)
|
||||||
|
}
|
||||||
|
if got != want {
|
||||||
|
t.Fatalf("schema_migrations has %d row(s) after two concurrent Migrate calls, want %d (one per migration file, no duplicates)", got, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// withSearchPath pins a DSN to one schema. Copied from
|
||||||
|
// internal/api/testdb_test.go rather than shared: that helper lives in the
|
||||||
|
// api_test package, a separate compiled package this one cannot import.
|
||||||
|
func withSearchPath(dsn, schema string) string {
|
||||||
|
opt := "-csearch_path=" + schema
|
||||||
|
|
||||||
|
if strings.HasPrefix(dsn, "postgres://") || strings.HasPrefix(dsn, "postgresql://") {
|
||||||
|
u, err := url.Parse(dsn)
|
||||||
|
if err == nil {
|
||||||
|
q := u.Query()
|
||||||
|
q.Set("options", opt)
|
||||||
|
u.RawQuery = q.Encode()
|
||||||
|
return u.String()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return dsn + " options='" + opt + "'"
|
||||||
|
}
|
||||||
@@ -1,182 +0,0 @@
|
|||||||
-- The Postgres baseline: the schema as it stood at the end of the SQLite line,
|
|
||||||
-- in one file rather than ten.
|
|
||||||
--
|
|
||||||
-- The ten SQLite migrations are in git history up to the commit that introduced
|
|
||||||
-- this one, and they replay against nothing here: their shape was incremental
|
|
||||||
-- (columns added, then dropped again in 008) and 008's backfill rewrote data
|
|
||||||
-- that a Postgres install never had. An existing SQLite database is carried over
|
|
||||||
-- by scripts/sqlite-to-postgres.go, which copies rows into this schema.
|
|
||||||
--
|
|
||||||
-- Two conventions inherited deliberately:
|
|
||||||
--
|
|
||||||
-- * Timestamps are BIGINT unix seconds, not timestamptz. Everything in Go
|
|
||||||
-- already speaks epochs, and converting was a second change riding along
|
|
||||||
-- with the port. Worth revisiting on its own.
|
|
||||||
--
|
|
||||||
-- * Ids are GENERATED BY DEFAULT, not ALWAYS, so the migration script can
|
|
||||||
-- insert rows with their original ids and keep every foreign key intact.
|
|
||||||
-- setval at the end of the copy puts the sequences past them.
|
|
||||||
|
|
||||||
CREATE TABLE users (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
username TEXT NOT NULL UNIQUE,
|
|
||||||
email TEXT NOT NULL UNIQUE,
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
-- Where this user's notifications go. NULL means they get none; incidents
|
|
||||||
-- assigned to them fall back to the configured fallback topic.
|
|
||||||
ntfy_topic TEXT,
|
|
||||||
-- NULL means the user has no password and can only use API keys.
|
|
||||||
password_hash TEXT
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE TABLE api_keys (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
|
||||||
key_hash TEXT NOT NULL UNIQUE,
|
|
||||||
name TEXT NOT NULL,
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
last_used_at BIGINT
|
|
||||||
);
|
|
||||||
|
|
||||||
-- A session is a browser's credential, the cookie counterpart of an API key:
|
|
||||||
-- only the hash of the token is stored. expires_at slides forward while the
|
|
||||||
-- session is in use, so an on-call phone stays signed in.
|
|
||||||
CREATE TABLE sessions (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
token_hash TEXT NOT NULL UNIQUE,
|
|
||||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
|
||||||
created_at BIGINT NOT NULL,
|
|
||||||
last_seen_at BIGINT NOT NULL,
|
|
||||||
expires_at BIGINT NOT NULL,
|
|
||||||
user_agent TEXT
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX idx_sessions_user ON sessions(user_id);
|
|
||||||
|
|
||||||
-- The machine-owned signal record: what Alertmanager says is true right now.
|
|
||||||
-- Workflow state lives on incidents, never here, because the webhook upsert owns
|
|
||||||
-- these rows and would overwrite it.
|
|
||||||
CREATE TABLE alerts (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
fingerprint TEXT NOT NULL UNIQUE,
|
|
||||||
name TEXT NOT NULL,
|
|
||||||
status TEXT NOT NULL CHECK (status IN ('firing', 'resolved')),
|
|
||||||
labels JSONB NOT NULL DEFAULT '{}'::jsonb,
|
|
||||||
annotations JSONB NOT NULL DEFAULT '{}'::jsonb,
|
|
||||||
starts_at BIGINT NOT NULL,
|
|
||||||
ends_at BIGINT,
|
|
||||||
generator_url TEXT NOT NULL DEFAULT '',
|
|
||||||
received_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
archived_at BIGINT,
|
|
||||||
-- Why the alert left the firing state: 'alertmanager' when a resolved
|
|
||||||
-- webhook set it, 'expiry' when the sweeper inferred it from staleness.
|
|
||||||
resolution_source TEXT
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX alerts_status_idx ON alerts(status);
|
|
||||||
CREATE INDEX alerts_name_idx ON alerts(name);
|
|
||||||
CREATE INDEX alerts_received_at_idx ON alerts(received_at DESC);
|
|
||||||
CREATE INDEX alerts_archived_at_idx ON alerts(archived_at);
|
|
||||||
|
|
||||||
CREATE TABLE schedule_entries (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
|
||||||
date TEXT NOT NULL UNIQUE, -- YYYY-MM-DD; one person per day
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX schedule_entries_date_idx ON schedule_entries(date);
|
|
||||||
|
|
||||||
-- The human work item: what people acknowledge, assign, snooze, discuss and
|
|
||||||
-- resolve. Correlation uses Alertmanager's own groupKey, so incidents follow the
|
|
||||||
-- group_by routing tree the operator already tuned.
|
|
||||||
CREATE TABLE incidents (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
group_key TEXT NOT NULL, -- Alertmanager groupKey, opaque
|
|
||||||
title TEXT NOT NULL, -- rendered from group_labels
|
|
||||||
group_labels JSONB NOT NULL DEFAULT '{}'::jsonb,
|
|
||||||
status TEXT NOT NULL CHECK (status IN ('triggered', 'acknowledged', 'resolved')),
|
|
||||||
severity TEXT, -- highest `severity` label across firing members
|
|
||||||
triggered_at BIGINT NOT NULL,
|
|
||||||
acknowledged_by BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
|
||||||
acknowledged_at BIGINT,
|
|
||||||
assigned_to BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
|
||||||
snoozed_until BIGINT,
|
|
||||||
resolved_at BIGINT,
|
|
||||||
resolution_source TEXT, -- 'alerts' | 'manual'
|
|
||||||
archived_at BIGINT
|
|
||||||
);
|
|
||||||
|
|
||||||
-- Load-bearing: at most one OPEN incident per group_key. This is what makes
|
|
||||||
-- "resolved incident + a new alert occurrence = a new incident" work, and it is
|
|
||||||
-- the constraint the webhook's find-or-open lookup relies on.
|
|
||||||
CREATE UNIQUE INDEX incidents_open_group_key_idx ON incidents(group_key) WHERE resolved_at IS NULL;
|
|
||||||
CREATE INDEX incidents_status_idx ON incidents(status);
|
|
||||||
CREATE INDEX incidents_triggered_at_idx ON incidents(triggered_at DESC);
|
|
||||||
CREATE INDEX incidents_archived_at_idx ON incidents(archived_at);
|
|
||||||
|
|
||||||
-- Membership is historical, not a pointer on alerts: one alert row (one
|
|
||||||
-- fingerprint) resolves and re-fires over time and belongs to a different
|
|
||||||
-- incident each occurrence.
|
|
||||||
CREATE TABLE incident_alerts (
|
|
||||||
incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
|
|
||||||
alert_id BIGINT NOT NULL REFERENCES alerts(id) ON DELETE CASCADE,
|
|
||||||
added_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
PRIMARY KEY (incident_id, alert_id)
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX incident_alerts_alert_id_idx ON incident_alerts(alert_id);
|
|
||||||
|
|
||||||
-- The timeline. Append-only, and the only history this server keeps: alert rows
|
|
||||||
-- are mutated in place, so without this there is no record that anything
|
|
||||||
-- happened. Notes are events too, so one query renders the whole story.
|
|
||||||
CREATE TABLE incident_events (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
|
|
||||||
-- triggered | alert_added | alert_resolved | acknowledged | unacknowledged
|
|
||||||
-- | assigned | snoozed | unsnoozed | resolved | note | notified | notify_failed
|
|
||||||
type TEXT NOT NULL,
|
|
||||||
user_id BIGINT REFERENCES users(id) ON DELETE SET NULL, -- NULL = the server acted
|
|
||||||
alert_id BIGINT REFERENCES alerts(id) ON DELETE SET NULL,
|
|
||||||
detail TEXT,
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX incident_events_incident_idx ON incident_events(incident_id, created_at);
|
|
||||||
|
|
||||||
-- Delivery is an outbox rather than an inline HTTP call: a POST made while
|
|
||||||
-- holding the webhook's transaction would hold a connection open across a
|
|
||||||
-- network round trip. The webhook inserts a row; the notifier goroutine
|
|
||||||
-- delivers it.
|
|
||||||
CREATE TABLE notifications (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
|
|
||||||
-- Nullable: a notification sent to the fallback topic belongs to nobody,
|
|
||||||
-- because nobody was on call when the incident opened.
|
|
||||||
user_id BIGINT REFERENCES users(id) ON DELETE SET NULL,
|
|
||||||
topic TEXT NOT NULL, -- resolved at enqueue: who was on call then
|
|
||||||
kind TEXT NOT NULL CHECK (kind IN ('triggered', 'reminder', 'resolved')),
|
|
||||||
created_at BIGINT NOT NULL,
|
|
||||||
send_after BIGINT NOT NULL, -- retry backoff watermark
|
|
||||||
attempts BIGINT NOT NULL DEFAULT 0,
|
|
||||||
sent_at BIGINT,
|
|
||||||
last_error TEXT -- kept after the last attempt, for debugging
|
|
||||||
);
|
|
||||||
|
|
||||||
-- The delivery loop's only query: what is due and still unsent.
|
|
||||||
CREATE INDEX notifications_pending_idx ON notifications(send_after) WHERE sent_at IS NULL;
|
|
||||||
-- Reminders and resolved notices both look up an incident's newest row.
|
|
||||||
CREATE INDEX notifications_incident_idx ON notifications(incident_id, id DESC);
|
|
||||||
|
|
||||||
-- A notification body is stored on the ntfy server and cached on the device, so
|
|
||||||
-- a real API key must never appear in one. Each delivery mints its own token
|
|
||||||
-- instead: one incident, one action, one day.
|
|
||||||
CREATE TABLE incident_ack_tokens (
|
|
||||||
token_hash TEXT PRIMARY KEY, -- SHA-256 of the raw token, as with api_keys
|
|
||||||
incident_id BIGINT NOT NULL REFERENCES incidents(id) ON DELETE CASCADE,
|
|
||||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
|
||||||
created_at BIGINT NOT NULL,
|
|
||||||
expires_at BIGINT NOT NULL
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX incident_ack_tokens_expires_idx ON incident_ack_tokens(expires_at);
|
|
||||||
@@ -0,0 +1,587 @@
|
|||||||
|
-- Terdut Server schema. One baseline: the project has not shipped, so the
|
||||||
|
-- migration history that led here (SQLite import, a Default team, per-team
|
||||||
|
-- deadman configs later replaced by switches) is not carried. Later changes are
|
||||||
|
-- new numbered files after this one.
|
||||||
|
--
|
||||||
|
-- Timestamps are Unix epoch seconds in BIGINT columns throughout.
|
||||||
|
|
||||||
|
CREATE TABLE alerts (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
fingerprint text NOT NULL,
|
||||||
|
name text NOT NULL,
|
||||||
|
status text NOT NULL,
|
||||||
|
labels jsonb DEFAULT '{}'::jsonb NOT NULL,
|
||||||
|
annotations jsonb DEFAULT '{}'::jsonb NOT NULL,
|
||||||
|
starts_at bigint NOT NULL,
|
||||||
|
ends_at bigint,
|
||||||
|
generator_url text DEFAULT ''::text NOT NULL,
|
||||||
|
received_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
archived_at bigint,
|
||||||
|
resolution_source text,
|
||||||
|
team_id bigint NOT NULL,
|
||||||
|
integration_id bigint,
|
||||||
|
CONSTRAINT alerts_status_check CHECK ((status = ANY (ARRAY['firing'::text, 'resolved'::text])))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE api_keys (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
user_id bigint NOT NULL,
|
||||||
|
key_hash text NOT NULL,
|
||||||
|
name text NOT NULL,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
last_used_at bigint,
|
||||||
|
expires_at bigint
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE deadman_switches (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
team_id bigint NOT NULL,
|
||||||
|
name text NOT NULL,
|
||||||
|
matcher text NOT NULL,
|
||||||
|
timeout_seconds bigint NOT NULL,
|
||||||
|
severity text DEFAULT 'critical'::text NOT NULL,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
CONSTRAINT deadman_switches_timeout_seconds_check CHECK ((timeout_seconds > 0))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE device_logins (
|
||||||
|
device_hash text NOT NULL,
|
||||||
|
user_code text NOT NULL,
|
||||||
|
status text DEFAULT 'pending'::text NOT NULL,
|
||||||
|
user_id bigint,
|
||||||
|
expires_at bigint NOT NULL,
|
||||||
|
last_polled_at bigint DEFAULT 0 NOT NULL,
|
||||||
|
CONSTRAINT device_logins_status_check CHECK ((status = ANY (ARRAY['pending'::text, 'approved'::text, 'denied'::text])))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE escalation_levels (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
team_id bigint NOT NULL,
|
||||||
|
"position" bigint NOT NULL,
|
||||||
|
timeout_seconds bigint NOT NULL,
|
||||||
|
CONSTRAINT escalation_levels_timeout_seconds_check CHECK ((timeout_seconds > 0))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE escalation_policies (
|
||||||
|
team_id bigint NOT NULL,
|
||||||
|
repeat_count bigint DEFAULT 0 NOT NULL,
|
||||||
|
fallback_topic text DEFAULT ''::text NOT NULL,
|
||||||
|
updated_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
CONSTRAINT escalation_policies_repeat_count_check CHECK (((repeat_count >= 0) AND (repeat_count <= 10)))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE escalation_targets (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
level_id bigint NOT NULL,
|
||||||
|
kind text NOT NULL,
|
||||||
|
user_id bigint,
|
||||||
|
CONSTRAINT escalation_targets_check CHECK ((((kind = 'user'::text) AND (user_id IS NOT NULL)) OR ((kind = 'oncall'::text) AND (user_id IS NULL)))),
|
||||||
|
CONSTRAINT escalation_targets_kind_check CHECK ((kind = ANY (ARRAY['user'::text, 'oncall'::text])))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE incident_ack_tokens (
|
||||||
|
token_hash text NOT NULL,
|
||||||
|
incident_id bigint NOT NULL,
|
||||||
|
user_id bigint NOT NULL,
|
||||||
|
created_at bigint NOT NULL,
|
||||||
|
expires_at bigint NOT NULL
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE incident_alerts (
|
||||||
|
incident_id bigint NOT NULL,
|
||||||
|
alert_id bigint NOT NULL,
|
||||||
|
added_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE incident_events (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
incident_id bigint NOT NULL,
|
||||||
|
type text NOT NULL,
|
||||||
|
user_id bigint,
|
||||||
|
alert_id bigint,
|
||||||
|
detail text,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
service_account_id bigint,
|
||||||
|
actor_user_id bigint,
|
||||||
|
actor_service_account_id bigint,
|
||||||
|
CONSTRAINT incident_events_actor_xor_chk CHECK (((user_id IS NULL) OR (service_account_id IS NULL))),
|
||||||
|
CONSTRAINT incident_events_assign_actor_xor_chk CHECK (((actor_user_id IS NULL) OR (actor_service_account_id IS NULL)))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE incidents (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
group_key text NOT NULL,
|
||||||
|
title text NOT NULL,
|
||||||
|
group_labels jsonb DEFAULT '{}'::jsonb NOT NULL,
|
||||||
|
status text NOT NULL,
|
||||||
|
severity text,
|
||||||
|
triggered_at bigint NOT NULL,
|
||||||
|
acknowledged_by bigint,
|
||||||
|
acknowledged_at bigint,
|
||||||
|
assigned_to bigint,
|
||||||
|
snoozed_until bigint,
|
||||||
|
resolved_at bigint,
|
||||||
|
resolution_source text,
|
||||||
|
archived_at bigint,
|
||||||
|
team_id bigint NOT NULL,
|
||||||
|
escalation_level bigint DEFAULT 0 NOT NULL,
|
||||||
|
escalation_level_at bigint,
|
||||||
|
escalation_round bigint DEFAULT 0 NOT NULL,
|
||||||
|
signature text DEFAULT ''::text NOT NULL,
|
||||||
|
acknowledged_by_service_account_id bigint,
|
||||||
|
CONSTRAINT incidents_ack_actor_xor_chk CHECK (((acknowledged_by IS NULL) OR (acknowledged_by_service_account_id IS NULL))),
|
||||||
|
CONSTRAINT incidents_status_check CHECK ((status = ANY (ARRAY['triggered'::text, 'acknowledged'::text, 'resolved'::text])))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE integrations (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
team_id bigint NOT NULL,
|
||||||
|
kind text NOT NULL,
|
||||||
|
name text NOT NULL,
|
||||||
|
key_hash text NOT NULL,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
last_used_at bigint,
|
||||||
|
CONSTRAINT integrations_kind_check CHECK ((kind = 'alertmanager'::text))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE invites (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
token_hash text NOT NULL,
|
||||||
|
team_id bigint NOT NULL,
|
||||||
|
role text NOT NULL,
|
||||||
|
created_by bigint,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
expires_at bigint NOT NULL,
|
||||||
|
max_uses bigint DEFAULT 1 NOT NULL,
|
||||||
|
uses bigint DEFAULT 0 NOT NULL,
|
||||||
|
revoked_at bigint,
|
||||||
|
CONSTRAINT invites_max_uses_check CHECK (((max_uses > 0) AND (max_uses <= 100))),
|
||||||
|
CONSTRAINT invites_role_check CHECK ((role = ANY (ARRAY['owner'::text, 'member'::text])))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE notifications (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
incident_id bigint NOT NULL,
|
||||||
|
user_id bigint,
|
||||||
|
topic text NOT NULL,
|
||||||
|
kind text NOT NULL,
|
||||||
|
created_at bigint NOT NULL,
|
||||||
|
send_after bigint NOT NULL,
|
||||||
|
attempts bigint DEFAULT 0 NOT NULL,
|
||||||
|
sent_at bigint,
|
||||||
|
last_error text,
|
||||||
|
CONSTRAINT notifications_kind_check CHECK ((kind = ANY (ARRAY['triggered'::text, 'reminder'::text, 'resolved'::text, 'escalated'::text])))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE oidc_logins (
|
||||||
|
state_hash text NOT NULL,
|
||||||
|
nonce text NOT NULL,
|
||||||
|
pkce_verifier text NOT NULL,
|
||||||
|
expires_at bigint NOT NULL,
|
||||||
|
next text DEFAULT '/'::text NOT NULL
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE rate_limit_counters (
|
||||||
|
key text NOT NULL,
|
||||||
|
window_start bigint NOT NULL,
|
||||||
|
count integer NOT NULL
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE schedule_entries (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
user_id bigint NOT NULL,
|
||||||
|
date text NOT NULL,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
team_id bigint NOT NULL
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE service_account_keys (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
service_account_id bigint NOT NULL,
|
||||||
|
key_hash text NOT NULL,
|
||||||
|
name text NOT NULL,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
last_used_at bigint
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE service_accounts (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
name text NOT NULL,
|
||||||
|
scope text NOT NULL,
|
||||||
|
team_id bigint,
|
||||||
|
created_by bigint,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
CONSTRAINT service_accounts_scope_check CHECK ((scope = ANY (ARRAY['instance'::text, 'team'::text]))),
|
||||||
|
CONSTRAINT service_accounts_scope_team_id_chk CHECK ((((scope = 'team'::text) AND (team_id IS NOT NULL)) OR ((scope = 'instance'::text) AND (team_id IS NULL))))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE sessions (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
token_hash text NOT NULL,
|
||||||
|
user_id bigint NOT NULL,
|
||||||
|
created_at bigint NOT NULL,
|
||||||
|
last_seen_at bigint NOT NULL,
|
||||||
|
expires_at bigint NOT NULL,
|
||||||
|
user_agent text,
|
||||||
|
max_expires_at bigint
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE settings (
|
||||||
|
key text NOT NULL,
|
||||||
|
value text NOT NULL,
|
||||||
|
updated_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE team_members (
|
||||||
|
team_id bigint NOT NULL,
|
||||||
|
user_id bigint NOT NULL,
|
||||||
|
role text NOT NULL,
|
||||||
|
joined_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
source text DEFAULT 'manual'::text NOT NULL,
|
||||||
|
CONSTRAINT team_members_role_check CHECK ((role = ANY (ARRAY['owner'::text, 'member'::text]))),
|
||||||
|
CONSTRAINT team_members_source_check CHECK ((source = ANY (ARRAY['manual'::text, 'oidc'::text])))
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE teams (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
name text NOT NULL,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
oidc_member_group text,
|
||||||
|
oidc_owner_group text,
|
||||||
|
-- A stable identity for a team managed by automation (terdut-operator:
|
||||||
|
-- "<namespace>/<name>" of its TerdutTeam), so it can find or recreate its own
|
||||||
|
-- team without trusting a display name. NULL for a team a person made.
|
||||||
|
external_id text
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE user_identities (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
user_id bigint NOT NULL,
|
||||||
|
issuer text NOT NULL,
|
||||||
|
subject text NOT NULL,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
last_login_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE users (
|
||||||
|
id bigint GENERATED BY DEFAULT AS IDENTITY NOT NULL,
|
||||||
|
username text NOT NULL,
|
||||||
|
email text NOT NULL,
|
||||||
|
created_at bigint DEFAULT (floor(EXTRACT(epoch FROM now())))::bigint NOT NULL,
|
||||||
|
ntfy_topic text,
|
||||||
|
password_hash text,
|
||||||
|
is_admin boolean DEFAULT false NOT NULL,
|
||||||
|
disabled_at bigint,
|
||||||
|
invited_via bigint,
|
||||||
|
onboarding_dismissed_at bigint,
|
||||||
|
admin_source text DEFAULT 'manual'::text NOT NULL,
|
||||||
|
CONSTRAINT users_admin_source_check CHECK ((admin_source = ANY (ARRAY['manual'::text, 'oidc'::text])))
|
||||||
|
);
|
||||||
|
|
||||||
|
ALTER TABLE alerts
|
||||||
|
ADD CONSTRAINT alerts_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE api_keys
|
||||||
|
ADD CONSTRAINT api_keys_key_hash_key UNIQUE (key_hash);
|
||||||
|
|
||||||
|
ALTER TABLE api_keys
|
||||||
|
ADD CONSTRAINT api_keys_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE deadman_switches
|
||||||
|
ADD CONSTRAINT deadman_switches_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE device_logins
|
||||||
|
ADD CONSTRAINT device_logins_pkey PRIMARY KEY (device_hash);
|
||||||
|
|
||||||
|
ALTER TABLE device_logins
|
||||||
|
ADD CONSTRAINT device_logins_user_code_key UNIQUE (user_code);
|
||||||
|
|
||||||
|
ALTER TABLE escalation_levels
|
||||||
|
ADD CONSTRAINT escalation_levels_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE escalation_levels
|
||||||
|
ADD CONSTRAINT escalation_levels_team_id_position_key UNIQUE (team_id, "position");
|
||||||
|
|
||||||
|
ALTER TABLE escalation_policies
|
||||||
|
ADD CONSTRAINT escalation_policies_pkey PRIMARY KEY (team_id);
|
||||||
|
|
||||||
|
ALTER TABLE escalation_targets
|
||||||
|
ADD CONSTRAINT escalation_targets_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE incident_ack_tokens
|
||||||
|
ADD CONSTRAINT incident_ack_tokens_pkey PRIMARY KEY (token_hash);
|
||||||
|
|
||||||
|
ALTER TABLE incident_alerts
|
||||||
|
ADD CONSTRAINT incident_alerts_pkey PRIMARY KEY (incident_id, alert_id);
|
||||||
|
|
||||||
|
ALTER TABLE incident_events
|
||||||
|
ADD CONSTRAINT incident_events_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE incidents
|
||||||
|
ADD CONSTRAINT incidents_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE integrations
|
||||||
|
ADD CONSTRAINT integrations_key_hash_key UNIQUE (key_hash);
|
||||||
|
|
||||||
|
ALTER TABLE integrations
|
||||||
|
ADD CONSTRAINT integrations_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE invites
|
||||||
|
ADD CONSTRAINT invites_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE invites
|
||||||
|
ADD CONSTRAINT invites_token_hash_key UNIQUE (token_hash);
|
||||||
|
|
||||||
|
ALTER TABLE notifications
|
||||||
|
ADD CONSTRAINT notifications_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE oidc_logins
|
||||||
|
ADD CONSTRAINT oidc_logins_pkey PRIMARY KEY (state_hash);
|
||||||
|
|
||||||
|
ALTER TABLE rate_limit_counters
|
||||||
|
ADD CONSTRAINT rate_limit_counters_pkey PRIMARY KEY (key);
|
||||||
|
|
||||||
|
ALTER TABLE schedule_entries
|
||||||
|
ADD CONSTRAINT schedule_entries_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE service_account_keys
|
||||||
|
ADD CONSTRAINT service_account_keys_key_hash_key UNIQUE (key_hash);
|
||||||
|
|
||||||
|
ALTER TABLE service_account_keys
|
||||||
|
ADD CONSTRAINT service_account_keys_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE service_accounts
|
||||||
|
ADD CONSTRAINT service_accounts_name_key UNIQUE (name);
|
||||||
|
|
||||||
|
ALTER TABLE service_accounts
|
||||||
|
ADD CONSTRAINT service_accounts_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE sessions
|
||||||
|
ADD CONSTRAINT sessions_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE sessions
|
||||||
|
ADD CONSTRAINT sessions_token_hash_key UNIQUE (token_hash);
|
||||||
|
|
||||||
|
ALTER TABLE settings
|
||||||
|
ADD CONSTRAINT settings_pkey PRIMARY KEY (key);
|
||||||
|
|
||||||
|
ALTER TABLE team_members
|
||||||
|
ADD CONSTRAINT team_members_pkey PRIMARY KEY (team_id, user_id);
|
||||||
|
|
||||||
|
ALTER TABLE teams
|
||||||
|
ADD CONSTRAINT teams_name_key UNIQUE (name);
|
||||||
|
|
||||||
|
ALTER TABLE teams
|
||||||
|
ADD CONSTRAINT teams_external_id_key UNIQUE (external_id);
|
||||||
|
|
||||||
|
ALTER TABLE teams
|
||||||
|
ADD CONSTRAINT teams_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE user_identities
|
||||||
|
ADD CONSTRAINT user_identities_issuer_subject_key UNIQUE (issuer, subject);
|
||||||
|
|
||||||
|
ALTER TABLE user_identities
|
||||||
|
ADD CONSTRAINT user_identities_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE users
|
||||||
|
ADD CONSTRAINT users_email_key UNIQUE (email);
|
||||||
|
|
||||||
|
ALTER TABLE users
|
||||||
|
ADD CONSTRAINT users_pkey PRIMARY KEY (id);
|
||||||
|
|
||||||
|
ALTER TABLE users
|
||||||
|
ADD CONSTRAINT users_username_key UNIQUE (username);
|
||||||
|
|
||||||
|
CREATE INDEX alerts_archived_at_idx ON alerts USING btree (archived_at);
|
||||||
|
|
||||||
|
CREATE INDEX alerts_integration_idx ON alerts USING btree (integration_id, received_at) WHERE (integration_id IS NOT NULL);
|
||||||
|
|
||||||
|
CREATE INDEX alerts_name_idx ON alerts USING btree (name);
|
||||||
|
|
||||||
|
CREATE INDEX alerts_received_at_idx ON alerts USING btree (received_at DESC);
|
||||||
|
|
||||||
|
CREATE INDEX alerts_status_idx ON alerts USING btree (status);
|
||||||
|
|
||||||
|
CREATE UNIQUE INDEX alerts_team_fingerprint_idx ON alerts USING btree (team_id, fingerprint);
|
||||||
|
|
||||||
|
CREATE INDEX alerts_team_received_idx ON alerts USING btree (team_id, received_at DESC);
|
||||||
|
|
||||||
|
CREATE INDEX deadman_switches_team_idx ON deadman_switches USING btree (team_id);
|
||||||
|
|
||||||
|
CREATE INDEX device_logins_expires_idx ON device_logins USING btree (expires_at);
|
||||||
|
|
||||||
|
CREATE INDEX escalation_targets_level_idx ON escalation_targets USING btree (level_id);
|
||||||
|
|
||||||
|
CREATE INDEX idx_sessions_user ON sessions USING btree (user_id);
|
||||||
|
|
||||||
|
CREATE INDEX incident_ack_tokens_expires_idx ON incident_ack_tokens USING btree (expires_at);
|
||||||
|
|
||||||
|
CREATE INDEX incident_alerts_alert_id_idx ON incident_alerts USING btree (alert_id);
|
||||||
|
|
||||||
|
CREATE INDEX incident_events_actor_service_account_id_idx ON incident_events USING btree (actor_service_account_id);
|
||||||
|
|
||||||
|
CREATE INDEX incident_events_actor_user_id_idx ON incident_events USING btree (actor_user_id);
|
||||||
|
|
||||||
|
CREATE INDEX incident_events_incident_idx ON incident_events USING btree (incident_id, created_at);
|
||||||
|
|
||||||
|
CREATE INDEX incident_events_service_account_id_idx ON incident_events USING btree (service_account_id);
|
||||||
|
|
||||||
|
CREATE INDEX incidents_acknowledged_by_service_account_id_idx ON incidents USING btree (acknowledged_by_service_account_id);
|
||||||
|
|
||||||
|
CREATE INDEX incidents_archived_at_idx ON incidents USING btree (archived_at);
|
||||||
|
|
||||||
|
CREATE INDEX incidents_escalation_idx ON incidents USING btree (escalation_level_at) WHERE ((resolved_at IS NULL) AND (status = 'triggered'::text));
|
||||||
|
|
||||||
|
CREATE UNIQUE INDEX incidents_open_group_key_idx ON incidents USING btree (team_id, group_key) WHERE (resolved_at IS NULL);
|
||||||
|
|
||||||
|
CREATE INDEX incidents_signature_idx ON incidents USING btree (team_id, signature, triggered_at DESC);
|
||||||
|
|
||||||
|
CREATE INDEX incidents_status_idx ON incidents USING btree (status);
|
||||||
|
|
||||||
|
CREATE INDEX incidents_team_triggered_idx ON incidents USING btree (team_id, triggered_at DESC);
|
||||||
|
|
||||||
|
CREATE INDEX incidents_triggered_at_idx ON incidents USING btree (triggered_at DESC);
|
||||||
|
|
||||||
|
CREATE INDEX integrations_team_idx ON integrations USING btree (team_id);
|
||||||
|
|
||||||
|
-- A name identifies an integration (and a switch) within its team, so a client
|
||||||
|
-- that manages them declaratively can look one up by name instead of listing
|
||||||
|
-- and matching.
|
||||||
|
CREATE UNIQUE INDEX integrations_team_name_key ON integrations (team_id, name);
|
||||||
|
CREATE UNIQUE INDEX deadman_switches_team_name_key ON deadman_switches (team_id, name);
|
||||||
|
|
||||||
|
CREATE INDEX invites_team_idx ON invites USING btree (team_id);
|
||||||
|
|
||||||
|
CREATE INDEX notifications_incident_idx ON notifications USING btree (incident_id, id DESC);
|
||||||
|
|
||||||
|
CREATE INDEX notifications_pending_idx ON notifications USING btree (send_after) WHERE (sent_at IS NULL);
|
||||||
|
|
||||||
|
CREATE INDEX oidc_logins_expires_idx ON oidc_logins USING btree (expires_at);
|
||||||
|
|
||||||
|
CREATE INDEX schedule_entries_date_idx ON schedule_entries USING btree (date);
|
||||||
|
|
||||||
|
CREATE UNIQUE INDEX schedule_entries_team_date_idx ON schedule_entries USING btree (team_id, date);
|
||||||
|
|
||||||
|
CREATE INDEX service_account_keys_service_account_id_idx ON service_account_keys USING btree (service_account_id);
|
||||||
|
|
||||||
|
CREATE INDEX service_accounts_team_id_idx ON service_accounts USING btree (team_id);
|
||||||
|
|
||||||
|
CREATE INDEX team_members_user_idx ON team_members USING btree (user_id);
|
||||||
|
|
||||||
|
CREATE INDEX user_identities_user_idx ON user_identities USING btree (user_id);
|
||||||
|
|
||||||
|
CREATE INDEX users_is_admin_idx ON users USING btree (is_admin) WHERE is_admin;
|
||||||
|
|
||||||
|
ALTER TABLE alerts
|
||||||
|
ADD CONSTRAINT alerts_integration_id_fkey FOREIGN KEY (integration_id) REFERENCES integrations(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE alerts
|
||||||
|
ADD CONSTRAINT alerts_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE api_keys
|
||||||
|
ADD CONSTRAINT api_keys_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE deadman_switches
|
||||||
|
ADD CONSTRAINT deadman_switches_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE device_logins
|
||||||
|
ADD CONSTRAINT device_logins_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE escalation_levels
|
||||||
|
ADD CONSTRAINT escalation_levels_team_id_fkey FOREIGN KEY (team_id) REFERENCES escalation_policies(team_id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE escalation_policies
|
||||||
|
ADD CONSTRAINT escalation_policies_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE escalation_targets
|
||||||
|
ADD CONSTRAINT escalation_targets_level_id_fkey FOREIGN KEY (level_id) REFERENCES escalation_levels(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE escalation_targets
|
||||||
|
ADD CONSTRAINT escalation_targets_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE incident_ack_tokens
|
||||||
|
ADD CONSTRAINT incident_ack_tokens_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE incident_ack_tokens
|
||||||
|
ADD CONSTRAINT incident_ack_tokens_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE incident_alerts
|
||||||
|
ADD CONSTRAINT incident_alerts_alert_id_fkey FOREIGN KEY (alert_id) REFERENCES alerts(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE incident_alerts
|
||||||
|
ADD CONSTRAINT incident_alerts_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE incident_events
|
||||||
|
ADD CONSTRAINT incident_events_actor_service_account_id_fkey FOREIGN KEY (actor_service_account_id) REFERENCES service_accounts(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE incident_events
|
||||||
|
ADD CONSTRAINT incident_events_actor_user_id_fkey FOREIGN KEY (actor_user_id) REFERENCES users(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE incident_events
|
||||||
|
ADD CONSTRAINT incident_events_alert_id_fkey FOREIGN KEY (alert_id) REFERENCES alerts(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE incident_events
|
||||||
|
ADD CONSTRAINT incident_events_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE incident_events
|
||||||
|
ADD CONSTRAINT incident_events_service_account_id_fkey FOREIGN KEY (service_account_id) REFERENCES service_accounts(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE incident_events
|
||||||
|
ADD CONSTRAINT incident_events_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE incidents
|
||||||
|
ADD CONSTRAINT incidents_acknowledged_by_fkey FOREIGN KEY (acknowledged_by) REFERENCES users(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE incidents
|
||||||
|
ADD CONSTRAINT incidents_acknowledged_by_service_account_id_fkey FOREIGN KEY (acknowledged_by_service_account_id) REFERENCES service_accounts(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE incidents
|
||||||
|
ADD CONSTRAINT incidents_assigned_to_fkey FOREIGN KEY (assigned_to) REFERENCES users(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE incidents
|
||||||
|
ADD CONSTRAINT incidents_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE integrations
|
||||||
|
ADD CONSTRAINT integrations_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE invites
|
||||||
|
ADD CONSTRAINT invites_created_by_fkey FOREIGN KEY (created_by) REFERENCES users(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE invites
|
||||||
|
ADD CONSTRAINT invites_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE notifications
|
||||||
|
ADD CONSTRAINT notifications_incident_id_fkey FOREIGN KEY (incident_id) REFERENCES incidents(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE notifications
|
||||||
|
ADD CONSTRAINT notifications_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE schedule_entries
|
||||||
|
ADD CONSTRAINT schedule_entries_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE schedule_entries
|
||||||
|
ADD CONSTRAINT schedule_entries_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE service_account_keys
|
||||||
|
ADD CONSTRAINT service_account_keys_service_account_id_fkey FOREIGN KEY (service_account_id) REFERENCES service_accounts(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE service_accounts
|
||||||
|
ADD CONSTRAINT service_accounts_created_by_fkey FOREIGN KEY (created_by) REFERENCES users(id) ON DELETE SET NULL;
|
||||||
|
|
||||||
|
ALTER TABLE service_accounts
|
||||||
|
ADD CONSTRAINT service_accounts_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE sessions
|
||||||
|
ADD CONSTRAINT sessions_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE team_members
|
||||||
|
ADD CONSTRAINT team_members_team_id_fkey FOREIGN KEY (team_id) REFERENCES teams(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE team_members
|
||||||
|
ADD CONSTRAINT team_members_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE user_identities
|
||||||
|
ADD CONSTRAINT user_identities_user_id_fkey FOREIGN KEY (user_id) REFERENCES users(id) ON DELETE CASCADE;
|
||||||
|
|
||||||
|
ALTER TABLE users
|
||||||
|
ADD CONSTRAINT users_invited_via_fkey FOREIGN KEY (invited_via) REFERENCES invites(id) ON DELETE SET NULL;
|
||||||
@@ -1,25 +0,0 @@
|
|||||||
-- A system administrator role, and the first thing in this server that one user
|
|
||||||
-- can do and another cannot.
|
|
||||||
--
|
|
||||||
-- Until now every authenticated caller could create and delete users, set
|
|
||||||
-- anybody's password and mint API keys for anybody — auth.go said so in a
|
|
||||||
-- comment. That was defensible with one operator and a hand-made account; it is
|
|
||||||
-- not once people sign themselves up (see #7).
|
|
||||||
--
|
|
||||||
-- EVERY EXISTING USER BECOMES AN ADMIN. They already hold these powers, so
|
|
||||||
-- this migration changes nobody's access: it names what is already true, and
|
|
||||||
-- leaves demotion as a deliberate act somebody performs afterwards. The
|
|
||||||
-- alternative — promoting only user 1 — would silently strip the others, and
|
|
||||||
-- could leave an install whose only admin is an account nobody has a password
|
|
||||||
-- for.
|
|
||||||
--
|
|
||||||
-- New users are not admins: the column defaults to false, and the only ways to
|
|
||||||
-- become one are this backfill, the bootstrap endpoint, or an existing admin
|
|
||||||
-- granting it.
|
|
||||||
ALTER TABLE users ADD COLUMN is_admin BOOLEAN NOT NULL DEFAULT false;
|
|
||||||
|
|
||||||
UPDATE users SET is_admin = true;
|
|
||||||
|
|
||||||
-- The queue's assignment dropdown and the on-call schedule read every user, and
|
|
||||||
-- the admin screens in #5 will filter on this.
|
|
||||||
CREATE INDEX users_is_admin_idx ON users(is_admin) WHERE is_admin;
|
|
||||||
@@ -1,103 +0,0 @@
|
|||||||
-- Teams: the unit of tenancy. Everything a person works on now belongs to one.
|
|
||||||
--
|
|
||||||
-- Until this migration the install was one shared space — every user saw every
|
|
||||||
-- alert and every incident, and the Alertmanager webhook was unauthenticated, so
|
|
||||||
-- anything that could reach the port could open an incident for everybody.
|
|
||||||
--
|
|
||||||
-- The shape, in one paragraph: a team owns its incidents, alerts, schedule and
|
|
||||||
-- integrations. A user belongs to as many teams as they like, with a role in
|
|
||||||
-- each: an `owner` configures the team, a `member` works its incidents. An
|
|
||||||
-- integration key is what an alert arrives on, and the key is what says which
|
|
||||||
-- team the alert belongs to.
|
|
||||||
--
|
|
||||||
-- EVERYTHING EXISTING MOVES INTO ONE DEFAULT TEAM, and every existing user
|
|
||||||
-- becomes an owner of it. That keeps an upgrade a no-op for the people using it:
|
|
||||||
-- the same queue, the same schedule, the same incidents, with a name on them.
|
|
||||||
|
|
||||||
CREATE TABLE teams (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
name TEXT NOT NULL UNIQUE,
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
|
||||||
);
|
|
||||||
|
|
||||||
-- role is free text with a CHECK rather than an enum, so adding a third role
|
|
||||||
-- later is a migration and not a type rewrite.
|
|
||||||
CREATE TABLE team_members (
|
|
||||||
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
|
|
||||||
user_id BIGINT NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
|
||||||
role TEXT NOT NULL CHECK (role IN ('owner', 'member')),
|
|
||||||
joined_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
PRIMARY KEY (team_id, user_id)
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX team_members_user_idx ON team_members(user_id);
|
|
||||||
|
|
||||||
-- How alerts get in, and the only thing that says which team they belong to.
|
|
||||||
-- The key is stored as a SHA-256 hash, like api_keys and the ack tokens: a
|
|
||||||
-- leaked database gives nobody the ability to post alerts.
|
|
||||||
CREATE TABLE integrations (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
team_id BIGINT NOT NULL REFERENCES teams(id) ON DELETE CASCADE,
|
|
||||||
kind TEXT NOT NULL CHECK (kind IN ('alertmanager')),
|
|
||||||
name TEXT NOT NULL,
|
|
||||||
key_hash TEXT NOT NULL UNIQUE,
|
|
||||||
created_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint,
|
|
||||||
last_used_at BIGINT
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX integrations_team_idx ON integrations(team_id);
|
|
||||||
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
-- The default team, and everything that already exists moving into it.
|
|
||||||
--
|
|
||||||
-- Created unconditionally, even on an empty install, so there is always a team
|
|
||||||
-- for the bootstrap user to land in and for the first integration to hang off.
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
INSERT INTO teams (name) VALUES ('Default');
|
|
||||||
|
|
||||||
INSERT INTO team_members (team_id, user_id, role)
|
|
||||||
SELECT (SELECT id FROM teams WHERE name = 'Default'), id, 'owner' FROM users;
|
|
||||||
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
-- team_id on everything a team owns.
|
|
||||||
--
|
|
||||||
-- Added nullable, backfilled, then made NOT NULL: adding a NOT NULL column with
|
|
||||||
-- no default to a table with rows is rejected, and a DEFAULT pointing at the
|
|
||||||
-- default team would quietly keep working after the default team is gone.
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
ALTER TABLE alerts ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
|
|
||||||
ALTER TABLE incidents ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
|
|
||||||
ALTER TABLE schedule_entries ADD COLUMN team_id BIGINT REFERENCES teams(id) ON DELETE CASCADE;
|
|
||||||
|
|
||||||
UPDATE alerts SET team_id = (SELECT id FROM teams WHERE name = 'Default');
|
|
||||||
UPDATE incidents SET team_id = (SELECT id FROM teams WHERE name = 'Default');
|
|
||||||
UPDATE schedule_entries SET team_id = (SELECT id FROM teams WHERE name = 'Default');
|
|
||||||
|
|
||||||
ALTER TABLE alerts ALTER COLUMN team_id SET NOT NULL;
|
|
||||||
ALTER TABLE incidents ALTER COLUMN team_id SET NOT NULL;
|
|
||||||
ALTER TABLE schedule_entries ALTER COLUMN team_id SET NOT NULL;
|
|
||||||
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
-- The uniqueness rules were all written for one tenant, and every one of them
|
|
||||||
-- is wrong now: two teams monitoring two clusters legitimately see the same
|
|
||||||
-- fingerprint, the same groupKey, and want somebody on call on the same day.
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
ALTER TABLE alerts DROP CONSTRAINT alerts_fingerprint_key;
|
|
||||||
CREATE UNIQUE INDEX alerts_team_fingerprint_idx ON alerts(team_id, fingerprint);
|
|
||||||
|
|
||||||
DROP INDEX incidents_open_group_key_idx;
|
|
||||||
-- Still load-bearing, now per team: at most one OPEN incident per group_key
|
|
||||||
-- within a team. This is what makes "resolved incident + a new alert occurrence
|
|
||||||
-- = a new incident" work, and what the webhook's find-or-open lookup relies on.
|
|
||||||
CREATE UNIQUE INDEX incidents_open_group_key_idx
|
|
||||||
ON incidents(team_id, group_key) WHERE resolved_at IS NULL;
|
|
||||||
|
|
||||||
ALTER TABLE schedule_entries DROP CONSTRAINT schedule_entries_date_key;
|
|
||||||
CREATE UNIQUE INDEX schedule_entries_team_date_idx ON schedule_entries(team_id, date);
|
|
||||||
|
|
||||||
-- The list views all filter by team first.
|
|
||||||
CREATE INDEX alerts_team_received_idx ON alerts(team_id, received_at DESC);
|
|
||||||
CREATE INDEX incidents_team_triggered_idx ON incidents(team_id, triggered_at DESC);
|
|
||||||
@@ -1,39 +0,0 @@
|
|||||||
-- Dead man's switches become a team's own configuration.
|
|
||||||
--
|
|
||||||
-- They were three environment variables — TERDUT_DEADMAN_MATCHERS, _TIMEOUT and
|
|
||||||
-- _SEVERITY — which made them one setting for the whole install. That was the
|
|
||||||
-- last piece of the alerting path a team could not control: a team could take
|
|
||||||
-- its own alerts on its own key and still not say which of them were
|
|
||||||
-- heartbeats, or how long a silence had to last before somebody was paged.
|
|
||||||
--
|
|
||||||
-- One row per team rather than one row per switch. The unit of monitoring is
|
|
||||||
-- still the fingerprint, as it always was — two clusters sending the same
|
|
||||||
-- heartbeat alertname are two independent switches — and the matcher string
|
|
||||||
-- keeps the format the environment variable used, so a value can be moved from
|
|
||||||
-- one to the other unchanged.
|
|
||||||
--
|
|
||||||
-- No rows are seeded here: a migration cannot read the environment. The server
|
|
||||||
-- inserts a row per team at startup from its own configuration, and the same
|
|
||||||
-- values therefore carry forward into the first team's row without anybody
|
|
||||||
-- retyping them. See seedDeadmanConfigs.
|
|
||||||
CREATE TABLE deadman_configs (
|
|
||||||
team_id BIGINT PRIMARY KEY REFERENCES teams(id) ON DELETE CASCADE,
|
|
||||||
|
|
||||||
-- ";" separates matchers, "," the label conditions within one, "=" is exact
|
|
||||||
-- equality: `alertname=Watchdog,cluster=prod; alertname=EdgeHeartbeat`.
|
|
||||||
-- Every matcher must name an alertname. Empty watches nothing.
|
|
||||||
matchers TEXT NOT NULL DEFAULT '',
|
|
||||||
|
|
||||||
-- Seconds rather than a Go duration string: the column is compared and
|
|
||||||
-- arithmetic is done on it, and a value that has to be parsed before it can
|
|
||||||
-- be believed is a value that can be stored unparseable. Zero disables the
|
|
||||||
-- team's switches entirely.
|
|
||||||
timeout_seconds BIGINT NOT NULL DEFAULT 0,
|
|
||||||
|
|
||||||
-- The severity these incidents open at. They have no member alerts to
|
|
||||||
-- derive one from, and a heartbeat's own severity label is meaningless —
|
|
||||||
-- Watchdog ships as "none".
|
|
||||||
severity TEXT NOT NULL DEFAULT 'critical',
|
|
||||||
|
|
||||||
updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
|
||||||
);
|
|
||||||
@@ -1,35 +0,0 @@
|
|||||||
-- Settings that an administrator can change without a redeploy, and the flag
|
|
||||||
-- that takes an account out of use without deleting it.
|
|
||||||
--
|
|
||||||
-- Three of the server's tunables were environment variables, which meant
|
|
||||||
-- changing how long an incident waits before it is paged again required editing
|
|
||||||
-- a chart, merging it, and waiting for a reconcile. They are behaviour, not
|
|
||||||
-- infrastructure, and the difference is who needs to change them and how often.
|
|
||||||
--
|
|
||||||
-- What stays in the environment: the ntfy URL and token, the database DSN, the
|
|
||||||
-- listen address and the public URL. Those are where the server is plugged in
|
|
||||||
-- rather than how it behaves, they are needed before the database is open, and
|
|
||||||
-- two of them are credentials.
|
|
||||||
--
|
|
||||||
-- Key/value rather than a column per setting. A settings table with one row and
|
|
||||||
-- a column per knob needs a migration for every new knob, and #6 and #7 will
|
|
||||||
-- both add some. The cost is that values are text and the accessor has to say
|
|
||||||
-- what type it wanted; settings.go does that in one place.
|
|
||||||
--
|
|
||||||
-- No rows are seeded here: a migration cannot read the environment. The server
|
|
||||||
-- inserts each key from its own configuration at startup, once, so an install
|
|
||||||
-- that upgrades keeps exactly the behaviour it had. See SeedSettings.
|
|
||||||
CREATE TABLE settings (
|
|
||||||
key TEXT PRIMARY KEY,
|
|
||||||
value TEXT NOT NULL,
|
|
||||||
updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
|
||||||
);
|
|
||||||
|
|
||||||
-- Disabling an account rather than deleting it: the person has left, or the
|
|
||||||
-- credential is suspect, and their incidents, acknowledgements and timeline
|
|
||||||
-- entries must stay exactly where they are. Deleting a user nulls their
|
|
||||||
-- acknowledged_by and assigned_to, which quietly rewrites history.
|
|
||||||
--
|
|
||||||
-- A disabled user cannot sign in and their API keys stop working, but they are
|
|
||||||
-- still a name the timeline can show and still a member of their teams.
|
|
||||||
ALTER TABLE users ADD COLUMN disabled_at BIGINT;
|
|
||||||
@@ -1,95 +0,0 @@
|
|||||||
-- Escalation: page somebody else when the first person does not answer.
|
|
||||||
--
|
|
||||||
-- This is the gap the whole multi-tenancy line of work was opened to close.
|
|
||||||
-- Until now an unacknowledged incident re-paged the same topic every
|
|
||||||
-- notify_repeat forever, which is a louder version of the same silence: if the
|
|
||||||
-- person on call is asleep, has no signal, or has left, nothing else happens.
|
|
||||||
--
|
|
||||||
-- Shape: one policy per team, an ordered list of levels, each level with a
|
|
||||||
-- timeout and a set of targets. When a level's timeout passes and the incident
|
|
||||||
-- is still triggered, the next level is paged. When the last level passes, the
|
|
||||||
-- chain repeats repeat_count times, and then the team's fallback topic is paged
|
|
||||||
-- once as the end of the line.
|
|
||||||
--
|
|
||||||
-- A team WITHOUT a policy keeps exactly today's behaviour: page the assignee,
|
|
||||||
-- then remind on the same topic. Escalation is opt-in per team, and the two
|
|
||||||
-- never both run for one incident -- see enqueueReminders.
|
|
||||||
CREATE TABLE escalation_policies (
|
|
||||||
-- One per team for now, hence the team as the key rather than an id with a
|
|
||||||
-- unique index: routing different alerts to different chains needs the
|
|
||||||
-- alert to carry something to route ON, which is a separate question.
|
|
||||||
team_id BIGINT PRIMARY KEY REFERENCES teams(id) ON DELETE CASCADE,
|
|
||||||
|
|
||||||
-- How many extra times to run the whole chain after it has been walked
|
|
||||||
-- once. 0 means walk it once and stop at the fallback.
|
|
||||||
repeat_count BIGINT NOT NULL DEFAULT 0 CHECK (repeat_count >= 0 AND repeat_count <= 10),
|
|
||||||
|
|
||||||
-- Where the last page goes when every level has been tried. Per team now:
|
|
||||||
-- TERDUT_NTFY_FALLBACK_TOPIC was one topic for the whole install, which in
|
|
||||||
-- a multi-team server pages the wrong people. Empty means the chain simply
|
|
||||||
-- ends.
|
|
||||||
fallback_topic TEXT NOT NULL DEFAULT '',
|
|
||||||
|
|
||||||
updated_at BIGINT NOT NULL DEFAULT FLOOR(EXTRACT(EPOCH FROM now()))::bigint
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE TABLE escalation_levels (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
team_id BIGINT NOT NULL REFERENCES escalation_policies(team_id) ON DELETE CASCADE,
|
|
||||||
-- 1-based, dense. The API rewrites the whole ladder on every edit rather
|
|
||||||
-- than patching one rung, so there is no way to leave a gap.
|
|
||||||
position BIGINT NOT NULL,
|
|
||||||
-- How long this level has to produce an acknowledgement before the next one
|
|
||||||
-- is paged. Seconds, like every other duration in this schema.
|
|
||||||
timeout_seconds BIGINT NOT NULL CHECK (timeout_seconds > 0),
|
|
||||||
|
|
||||||
UNIQUE (team_id, position)
|
|
||||||
);
|
|
||||||
|
|
||||||
-- Who a level pages. Either a named person, or whoever the team's rota says is
|
|
||||||
-- on call today -- which is the target that keeps working when the rota
|
|
||||||
-- changes and nobody remembers to edit the policy.
|
|
||||||
CREATE TABLE escalation_targets (
|
|
||||||
id BIGINT GENERATED BY DEFAULT AS IDENTITY PRIMARY KEY,
|
|
||||||
level_id BIGINT NOT NULL REFERENCES escalation_levels(id) ON DELETE CASCADE,
|
|
||||||
kind TEXT NOT NULL CHECK (kind IN ('user', 'oncall')),
|
|
||||||
-- Set for kind='user', NULL for kind='oncall'.
|
|
||||||
user_id BIGINT REFERENCES users(id) ON DELETE CASCADE,
|
|
||||||
|
|
||||||
CHECK ((kind = 'user' AND user_id IS NOT NULL) OR (kind = 'oncall' AND user_id IS NULL))
|
|
||||||
);
|
|
||||||
|
|
||||||
CREATE INDEX escalation_targets_level_idx ON escalation_targets(level_id);
|
|
||||||
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
-- Where an incident is in its chain.
|
|
||||||
--
|
|
||||||
-- On the incident rather than in a side table: it is read on every notifier
|
|
||||||
-- tick alongside the incident's status, and one row per incident is exactly
|
|
||||||
-- what the state is.
|
|
||||||
-- ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
-- 0 means no level has been paged yet, which is the state of every incident
|
|
||||||
-- that existed before escalation and of every incident in a team with no
|
|
||||||
-- policy. 1 is the first level.
|
|
||||||
ALTER TABLE incidents ADD COLUMN escalation_level BIGINT NOT NULL DEFAULT 0;
|
|
||||||
|
|
||||||
-- When the current level was entered, and therefore what its timeout is
|
|
||||||
-- measured from. NULL while escalation_level is 0.
|
|
||||||
ALTER TABLE incidents ADD COLUMN escalation_level_at BIGINT;
|
|
||||||
|
|
||||||
-- How many times the chain has been walked in full. Compared against the
|
|
||||||
-- policy's repeat_count.
|
|
||||||
ALTER TABLE incidents ADD COLUMN escalation_round BIGINT NOT NULL DEFAULT 0;
|
|
||||||
|
|
||||||
-- The notifier's escalation query: incidents still waiting, oldest level first.
|
|
||||||
CREATE INDEX incidents_escalation_idx
|
|
||||||
ON incidents(escalation_level_at)
|
|
||||||
WHERE resolved_at IS NULL AND status = 'triggered';
|
|
||||||
|
|
||||||
-- 'escalated' joins the outbox kinds: a page that went out because nobody
|
|
||||||
-- answered the last one, which is worth telling apart from the first page and
|
|
||||||
-- from a reminder when reading the timeline or debugging a delivery.
|
|
||||||
ALTER TABLE notifications DROP CONSTRAINT notifications_kind_check;
|
|
||||||
ALTER TABLE notifications ADD CONSTRAINT notifications_kind_check
|
|
||||||
CHECK (kind IN ('triggered', 'reminder', 'resolved', 'escalated'));
|
|
||||||
@@ -33,7 +33,7 @@ type Alert struct {
|
|||||||
// refreshed. The sweeper stale-dates against it (see expireStale), API
|
// refreshed. The sweeper stale-dates against it (see expireStale), API
|
||||||
// clients render it, and GET /api/alerts is ordered by it. Anything that
|
// clients render it, and GET /api/alerts is ordered by it. Anything that
|
||||||
// stops the webhook handler from advancing it on a re-send is a breaking
|
// stops the webhook handler from advancing it on a re-send is a breaking
|
||||||
// change — see "received_at is a liveness heartbeat" in the README and
|
// change — see "received_at is a liveness heartbeat" in docs/api.md and
|
||||||
// TestWebhook_ResendBumpsReceivedAt.
|
// TestWebhook_ResendBumpsReceivedAt.
|
||||||
ReceivedAt time.Time `json:"received_at"`
|
ReceivedAt time.Time `json:"received_at"`
|
||||||
|
|
||||||
@@ -52,7 +52,7 @@ type Alert struct {
|
|||||||
// inferred. Under "expiry" nothing ever reported an end, so EndsAt is only
|
// inferred. Under "expiry" nothing ever reported an end, so EndsAt is only
|
||||||
// an upper bound (see expireStale) and ReceivedAt is the more truthful
|
// an upper bound (see expireStale) and ReceivedAt is the more truthful
|
||||||
// signal. Treat the value set as open — see "resolution_source says how much
|
// signal. Treat the value set as open — see "resolution_source says how much
|
||||||
// to trust ends_at" in the README, and TestWebhook_ResolvedSetsSource /
|
// to trust ends_at" in docs/api.md, and TestWebhook_ResolvedSetsSource /
|
||||||
// TestExpiry_StaleFiringAlert.
|
// TestExpiry_StaleFiringAlert.
|
||||||
ResolutionSource *string `json:"resolution_source,omitempty"`
|
ResolutionSource *string `json:"resolution_source,omitempty"`
|
||||||
|
|
||||||
|
|||||||
@@ -11,6 +11,13 @@ import "time"
|
|||||||
// the webhook and the sweeper may flip to "resolved" once every member alert has
|
// the webhook and the sweeper may flip to "resolved" once every member alert has
|
||||||
// stopped firing.
|
// stopped firing.
|
||||||
type Incident struct {
|
type Incident struct {
|
||||||
|
// EscalationLevel is which rung of its team's ladder this incident is on,
|
||||||
|
// 0 for none — either the team has no ladder, or somebody has answered.
|
||||||
|
// EscalationDueAt is when the current level runs out, so a client can say
|
||||||
|
// how long is left rather than only what already happened.
|
||||||
|
EscalationLevel int64 `json:"escalation_level"`
|
||||||
|
EscalationDueAt *time.Time `json:"escalation_due_at,omitempty"`
|
||||||
|
|
||||||
// TeamID is the team that owns this incident, fixed when it opens: an
|
// TeamID is the team that owns this incident, fixed when it opens: an
|
||||||
// incident never moves between teams. TeamName rides along so the combined
|
// incident never moves between teams. TeamName rides along so the combined
|
||||||
// queue can badge each row without a second request.
|
// queue can badge each row without a second request.
|
||||||
@@ -36,6 +43,13 @@ type Incident struct {
|
|||||||
AcknowledgedByUser *string `json:"acknowledged_by,omitempty"`
|
AcknowledgedByUser *string `json:"acknowledged_by,omitempty"`
|
||||||
AcknowledgedAt *time.Time `json:"acknowledged_at,omitempty"`
|
AcknowledgedAt *time.Time `json:"acknowledged_at,omitempty"`
|
||||||
|
|
||||||
|
// AcknowledgedByServiceAccountID/Name are the service-account-shaped
|
||||||
|
// parallel to AcknowledgedByID/User above: mutually exclusive with it,
|
||||||
|
// populated when a service account (not a human) acknowledged this
|
||||||
|
// incident. See migration 015 and terdut-server#25.
|
||||||
|
AcknowledgedByServiceAccountID *int64 `json:"acknowledged_by_service_account_id,omitempty"`
|
||||||
|
AcknowledgedByServiceAccountName *string `json:"acknowledged_by_service_account,omitempty"`
|
||||||
|
|
||||||
AssignedToID *int64 `json:"assigned_to_id,omitempty"`
|
AssignedToID *int64 `json:"assigned_to_id,omitempty"`
|
||||||
AssignedToUser *string `json:"assigned_to,omitempty"`
|
AssignedToUser *string `json:"assigned_to,omitempty"`
|
||||||
|
|
||||||
@@ -60,15 +74,46 @@ type Incident struct {
|
|||||||
// and is the only history this server keeps — alert rows are mutated in place.
|
// and is the only history this server keeps — alert rows are mutated in place.
|
||||||
//
|
//
|
||||||
// Type is one of: triggered, alert_added, alert_resolved, acknowledged,
|
// Type is one of: triggered, alert_added, alert_resolved, acknowledged,
|
||||||
// unacknowledged, assigned, snoozed, unsnoozed, resolved, note. A nil UserID
|
// unacknowledged, assigned, archived, unarchived, snoozed, unsnoozed,
|
||||||
// means the server acted rather than a person.
|
// resolved, note. UserID and
|
||||||
|
// ServiceAccountID are mutually exclusive; both nil means the server acted
|
||||||
|
// rather than any caller.
|
||||||
type IncidentEvent struct {
|
type IncidentEvent struct {
|
||||||
ID int64 `json:"id"`
|
ID int64 `json:"id"`
|
||||||
IncidentID int64 `json:"incident_id"`
|
IncidentID int64 `json:"incident_id"`
|
||||||
Type string `json:"type"`
|
Type string `json:"type"`
|
||||||
UserID *int64 `json:"user_id,omitempty"`
|
UserID *int64 `json:"user_id,omitempty"`
|
||||||
Username *string `json:"username,omitempty"`
|
Username *string `json:"username,omitempty"`
|
||||||
AlertID *int64 `json:"alert_id,omitempty"`
|
|
||||||
Detail *string `json:"detail,omitempty"`
|
// ServiceAccountID/Name are the service-account-shaped parallel to
|
||||||
CreatedAt time.Time `json:"created_at"`
|
// UserID/Username above: mutually exclusive with it, populated when a
|
||||||
|
// service account (not a human, and not nil-meaning-the-server-acted)
|
||||||
|
// performed this event. Named Name, not Username — a ServiceAccount has
|
||||||
|
// a Name field, not a Username. See migration 015 and terdut-server#25.
|
||||||
|
ServiceAccountID *int64 `json:"service_account_id,omitempty"`
|
||||||
|
ServiceAccountName *string `json:"service_account_name,omitempty"`
|
||||||
|
|
||||||
|
// Actor* name who performed an 'assigned' event, whose UserID is the
|
||||||
|
// assignee. Mutually exclusive; unset on every other event type and on
|
||||||
|
// assignments made before migration 018. See terdut-server#35.
|
||||||
|
ActorUserID *int64 `json:"actor_user_id,omitempty"`
|
||||||
|
ActorUsername *string `json:"actor_username,omitempty"`
|
||||||
|
ActorServiceAccountID *int64 `json:"actor_service_account_id,omitempty"`
|
||||||
|
ActorServiceAccountName *string `json:"actor_service_account_name,omitempty"`
|
||||||
|
|
||||||
|
AlertID *int64 `json:"alert_id,omitempty"`
|
||||||
|
Detail *string `json:"detail,omitempty"`
|
||||||
|
CreatedAt time.Time `json:"created_at"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// SimilarIncident is an earlier, resolved incident with the same signature as
|
||||||
|
// the one being looked at. ResolutionNotes are the "what fixed it" notes;
|
||||||
|
// NoteCount counts the plain working notes, which live on the timeline.
|
||||||
|
type SimilarIncident struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
Title string `json:"title"`
|
||||||
|
TriggeredAt time.Time `json:"triggered_at"`
|
||||||
|
ResolvedAt time.Time `json:"resolved_at"`
|
||||||
|
NoteCount int `json:"note_count"`
|
||||||
|
ResolutionNotes []IncidentEvent `json:"resolution_notes"`
|
||||||
}
|
}
|
||||||
|
|||||||